nvcurve with some fixes and better limits
This commit is contained in:
commit
024dcbceb0
70 files changed
+18890
No files matched your search
@@ -0,0 +1,13 @@
|
||||
from .gpu import get_gpu, discover_gpus
|
||||
from .vfcurve import read_curve, read_clock_offsets, write_offsets, reset_offsets
|
||||
from .monitoring import poll, read_voltage
|
||||
from .ranges import get_clock_ranges
|
||||
from .snapshot import save as snapshot_save, restore as snapshot_restore, list_snapshots
|
||||
|
||||
__all__ = [
|
||||
"get_gpu", "discover_gpus",
|
||||
"read_curve", "read_clock_offsets", "write_offsets", "reset_offsets",
|
||||
"poll", "read_voltage",
|
||||
"get_clock_ranges",
|
||||
"snapshot_save", "snapshot_restore", "list_snapshots",
|
||||
]
|
||||
@@ -0,0 +1,93 @@
|
||||
"""GPU discovery and initialization."""
|
||||
|
||||
import ctypes
|
||||
import sys
|
||||
|
||||
from ..nvapi.bootstrap import query_interface
|
||||
from ..nvapi.constants import FUNC
|
||||
from ..nvapi.types import GpuInfo
|
||||
|
||||
|
||||
def init_nvapi() -> None:
|
||||
"""Initialize NvAPI. Must be called before any GPU operations."""
|
||||
init_fn = query_interface(FUNC["Initialize"], nargs=0)
|
||||
if not init_fn or init_fn() != 0:
|
||||
print("NvAPI_Initialize failed")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def enumerate_gpus() -> tuple[ctypes.Array, int]:
|
||||
"""Return (gpu_handles_array, count). Exits if no GPUs found."""
|
||||
gpus = (ctypes.c_void_p * 64)()
|
||||
ngpu = ctypes.c_int32()
|
||||
query_interface(FUNC["EnumPhysicalGPUs"])(ctypes.byref(gpus), ctypes.byref(ngpu))
|
||||
if ngpu.value == 0:
|
||||
print("No NVIDIA GPUs found")
|
||||
sys.exit(1)
|
||||
return gpus, ngpu.value
|
||||
|
||||
|
||||
def get_gpu_name(gpu) -> str:
|
||||
"""Return the full name string for a GPU handle."""
|
||||
name_buf = ctypes.create_string_buffer(256)
|
||||
query_interface(FUNC["GetFullName"])(gpu, name_buf)
|
||||
return name_buf.value.decode(errors="replace")
|
||||
|
||||
|
||||
def discover_gpus() -> list[GpuInfo]:
|
||||
"""Initialize NvAPI and NVML, and return a list of GpuInfo for all physical GPUs."""
|
||||
init_nvapi()
|
||||
gpus, count = enumerate_gpus()
|
||||
infos = []
|
||||
|
||||
try:
|
||||
import pynvml
|
||||
pynvml.nvmlInit()
|
||||
has_nvml = True
|
||||
except Exception:
|
||||
has_nvml = False
|
||||
|
||||
for i in range(count):
|
||||
name = get_gpu_name(gpus[i])
|
||||
uuid = None
|
||||
pci_bus_id = None
|
||||
if has_nvml:
|
||||
try:
|
||||
handle = pynvml.nvmlDeviceGetHandleByIndex(i)
|
||||
uuid = pynvml.nvmlDeviceGetUUID(handle)
|
||||
# NVML might return bytes
|
||||
if isinstance(uuid, bytes):
|
||||
uuid = uuid.decode('utf-8', errors='ignore')
|
||||
pci_info = pynvml.nvmlDeviceGetPciInfo(handle)
|
||||
# Parse something like "00000000:01:00.0" -> bus is 1
|
||||
if isinstance(pci_info.bus, bytes):
|
||||
pci_bus_id = int(pci_info.bus.decode('utf-8', errors='ignore'), 16)
|
||||
else:
|
||||
pci_bus_id = pci_info.bus
|
||||
except Exception:
|
||||
pass
|
||||
infos.append(GpuInfo(name=name, index=i, uuid=uuid, pci_bus_id=pci_bus_id))
|
||||
|
||||
if has_nvml:
|
||||
try:
|
||||
pynvml.nvmlShutdown()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return infos
|
||||
|
||||
|
||||
def get_gpu(index: int = 0):
|
||||
"""Initialize NvAPI, enumerate GPUs, and return the handle for `index`.
|
||||
|
||||
Also returns the GPU name as a convenience.
|
||||
Returns (handle, name).
|
||||
"""
|
||||
init_nvapi()
|
||||
gpus, count = enumerate_gpus()
|
||||
if index >= count:
|
||||
print(f"GPU index {index} out of range (found {count} GPU(s))")
|
||||
sys.exit(1)
|
||||
gpu = gpus[index]
|
||||
name = get_gpu_name(gpu)
|
||||
return gpu, name
|
||||
@@ -0,0 +1,321 @@
|
||||
"""Hardware Abstraction Layer for Global Limits (Power, Clock Offsets).
|
||||
|
||||
Uses NVML (via pynvml) for all operations.
|
||||
|
||||
Clock offsets use nvmlDeviceSetClockOffsets / nvmlDeviceGetClockOffsets
|
||||
(introduced in driver 555.85). The older per-domain functions
|
||||
(nvmlDeviceSet/GetGpcClkVfOffset, nvmlDeviceSet/GetMemClkVfOffset) are used
|
||||
as a fallback when the new API is unavailable or returns an error.
|
||||
set_clock_offsets accepts Optional values and only touches the domains
|
||||
that are explicitly specified, leaving others unchanged on hardware.
|
||||
"""
|
||||
|
||||
import ctypes
|
||||
import subprocess
|
||||
import logging
|
||||
from typing import Optional
|
||||
|
||||
try:
|
||||
import pynvml
|
||||
_NVML_AVAILABLE = True
|
||||
except ImportError:
|
||||
_NVML_AVAILABLE = False
|
||||
|
||||
log = logging.getLogger("nvcurve.hal.limits")
|
||||
|
||||
# ── NVML library / handle helpers ─────────────────────────────────────────────
|
||||
|
||||
_nvml_lib: Optional[ctypes.CDLL] = None
|
||||
|
||||
|
||||
def _nvml_cdll() -> ctypes.CDLL:
|
||||
"""Return a ctypes handle to libnvidia-ml, reusing pynvml's load if possible."""
|
||||
global _nvml_lib
|
||||
if _nvml_lib is not None:
|
||||
return _nvml_lib
|
||||
# Prefer to reuse the library already loaded by pynvml to avoid dlopen races.
|
||||
for attr in ("nvml", "_nvml"): # attribute name varies by pynvml version
|
||||
mod = getattr(pynvml, attr, None)
|
||||
lib = getattr(mod, "_lib", None) or getattr(mod, "_nvmlLib", None)
|
||||
if lib is not None:
|
||||
_nvml_lib = lib
|
||||
return _nvml_lib
|
||||
_nvml_lib = ctypes.CDLL("libnvidia-ml.so.1")
|
||||
return _nvml_lib
|
||||
|
||||
|
||||
def _get_handle(gpu_index: int):
|
||||
"""Return an NVML device handle, initialising pynvml if needed."""
|
||||
if not _NVML_AVAILABLE:
|
||||
raise RuntimeError("NVML not available (install nvidia-ml-py)")
|
||||
return pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
|
||||
|
||||
|
||||
# ── Power limit ───────────────────────────────────────────────────────────────
|
||||
|
||||
def get_power_limit(gpu_index: int = 0) -> dict:
|
||||
"""Return dict with power_limit_w, default_power_limit_w, min_power_limit_w, max_power_limit_w."""
|
||||
out = {
|
||||
"power_limit_w": None,
|
||||
"default_power_limit_w": None,
|
||||
"min_power_limit_w": None,
|
||||
"max_power_limit_w": None,
|
||||
}
|
||||
try:
|
||||
handle = _get_handle(gpu_index)
|
||||
limit = pynvml.nvmlDeviceGetPowerManagementLimit(handle)
|
||||
constrs = pynvml.nvmlDeviceGetPowerManagementLimitConstraints(handle)
|
||||
out["power_limit_w"] = limit // 1000
|
||||
out["min_power_limit_w"] = constrs[0] // 1000
|
||||
out["max_power_limit_w"] = constrs[1] // 1000
|
||||
try:
|
||||
default = pynvml.nvmlDeviceGetPowerManagementDefaultLimit(handle)
|
||||
out["default_power_limit_w"] = default // 1000
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
log.warning("get_power_limit: %s", exc)
|
||||
return out
|
||||
|
||||
|
||||
def set_power_limit(limit_w: int, gpu_index: int = 0) -> tuple[bool, str]:
|
||||
"""Set the board power limit (Watts)."""
|
||||
try:
|
||||
handle = _get_handle(gpu_index)
|
||||
pynvml.nvmlDeviceSetPowerManagementLimit(handle, limit_w * 1000)
|
||||
return True, "OK"
|
||||
except Exception as exc:
|
||||
log.debug("NVML set_power_limit failed: %s — falling back to nvidia-smi", exc)
|
||||
|
||||
ret = subprocess.run(
|
||||
["nvidia-smi", "-i", str(gpu_index), "-pl", str(limit_w)],
|
||||
capture_output=True, text=True,
|
||||
)
|
||||
if ret.returncode == 0:
|
||||
return True, "OK"
|
||||
return False, ret.stderr.strip() or ret.stdout.strip()
|
||||
|
||||
|
||||
# ── Clock offsets (GPC + memory) ──────────────────────────────────────────────
|
||||
|
||||
# The correct struct layout (per NVML docs and driver 590.x headers):
|
||||
#
|
||||
# typedef struct {
|
||||
# unsigned int version; // nvmlClockOffset_v1
|
||||
# nvmlClockType_t type; // NVML_CLOCK_GRAPHICS (0) or NVML_CLOCK_MEM (2)
|
||||
# nvmlPstates_t pstate; // NVML_PSTATE_0 (0)
|
||||
# int clockOffsetMHz;
|
||||
# } nvmlClockOffset_t;
|
||||
#
|
||||
# nvmlDeviceSet/GetClockOffsets are called ONCE PER CLOCK DOMAIN.
|
||||
# pynvml (nvidia-ml-py ≥ 12) exposes c_nvmlClockOffset_t and nvmlClockOffset_v1
|
||||
# as ctypes objects; we use them when available and fall back to our own definition.
|
||||
|
||||
class _ClockOffset(ctypes.Structure):
|
||||
_fields_ = [
|
||||
("version", ctypes.c_uint),
|
||||
("type", ctypes.c_uint), # nvmlClockType_t
|
||||
("pstate", ctypes.c_uint), # nvmlPstates_t
|
||||
("clockOffsetMHz", ctypes.c_int),
|
||||
]
|
||||
|
||||
_CLOCK_OFFSET_VER = (1 << 24) | ctypes.sizeof(_ClockOffset) # = 0x01000010 (16 bytes)
|
||||
|
||||
# NVML clock-type constants (same values as pynvml).
|
||||
_NVML_CLOCK_GRAPHICS = 0
|
||||
_NVML_CLOCK_MEM = 2
|
||||
|
||||
|
||||
def _make_clock_offset(clock_type: int, pstate: int = 0, offset_mhz: int = 0) -> ctypes.Structure:
|
||||
"""Return a populated nvmlClockOffset_t struct, using pynvml's type when available."""
|
||||
if hasattr(pynvml, "c_nvmlClockOffset_t") and hasattr(pynvml, "nvmlClockOffset_v1"):
|
||||
info = pynvml.c_nvmlClockOffset_t()
|
||||
info.version = pynvml.nvmlClockOffset_v1
|
||||
info.type = clock_type
|
||||
info.pstate = pstate
|
||||
info.clockOffsetMHz = offset_mhz
|
||||
return info
|
||||
info = _ClockOffset()
|
||||
info.version = _CLOCK_OFFSET_VER
|
||||
info.type = clock_type
|
||||
info.pstate = pstate
|
||||
info.clockOffsetMHz = offset_mhz
|
||||
return info
|
||||
|
||||
|
||||
def _try_nvml_fn(name: str):
|
||||
"""Return a ctypes-callable for an NVML function, or None if not found."""
|
||||
lib = _nvml_cdll()
|
||||
try:
|
||||
return getattr(lib, name)
|
||||
except AttributeError:
|
||||
return None
|
||||
|
||||
|
||||
def get_clock_offsets(gpu_index: int = 0) -> dict:
|
||||
"""Return GPC and memory clock offsets (MHz).
|
||||
|
||||
Keys: gpc_offset_mhz, mem_offset_mhz (both int or None on failure).
|
||||
Calls nvmlDeviceGetClockOffsets once per clock domain (GRAPHICS, MEM).
|
||||
"""
|
||||
out = {"gpc_offset_mhz": None, "mem_offset_mhz": None}
|
||||
if not _NVML_AVAILABLE:
|
||||
return out
|
||||
try:
|
||||
handle = _get_handle(gpu_index)
|
||||
|
||||
# Try pynvml wrapper first (nvidia-ml-py ≥ 12 exposes it correctly).
|
||||
# Fall back to ctypes-direct if pynvml doesn't have it.
|
||||
_pynvml_get = getattr(pynvml, "nvmlDeviceGetClockOffsets", None)
|
||||
fn_get = _try_nvml_fn("nvmlDeviceGetClockOffsets") if _pynvml_get is None else None
|
||||
|
||||
used_new_api = False
|
||||
for clock_type, key in ((_NVML_CLOCK_GRAPHICS, "gpc_offset_mhz"),
|
||||
(_NVML_CLOCK_MEM, "mem_offset_mhz")):
|
||||
info = _make_clock_offset(clock_type, pstate=0)
|
||||
try:
|
||||
if _pynvml_get is not None:
|
||||
rc = _pynvml_get(handle, ctypes.byref(info))
|
||||
elif fn_get is not None:
|
||||
rc = fn_get(handle, ctypes.byref(info))
|
||||
else:
|
||||
break
|
||||
if rc == 0:
|
||||
out[key] = int(info.clockOffsetMHz)
|
||||
used_new_api = True
|
||||
else:
|
||||
log.debug("nvmlDeviceGetClockOffsets(type=%d) returned %d", clock_type, rc)
|
||||
except Exception as exc:
|
||||
log.debug("nvmlDeviceGetClockOffsets(type=%d): %s", clock_type, exc)
|
||||
|
||||
if used_new_api:
|
||||
return out
|
||||
|
||||
# Deprecated per-domain fallback.
|
||||
if hasattr(pynvml, "nvmlDeviceGetGpcClkVfOffset"):
|
||||
try:
|
||||
out["gpc_offset_mhz"] = int(pynvml.nvmlDeviceGetGpcClkVfOffset(handle))
|
||||
except Exception as exc:
|
||||
log.debug("nvmlDeviceGetGpcClkVfOffset: %s", exc)
|
||||
if hasattr(pynvml, "nvmlDeviceGetMemClkVfOffset"):
|
||||
try:
|
||||
res = pynvml.nvmlDeviceGetMemClkVfOffset(handle)
|
||||
out["mem_offset_mhz"] = int(res[0] if isinstance(res, (list, tuple)) else res)
|
||||
except Exception as exc:
|
||||
log.debug("nvmlDeviceGetMemClkVfOffset: %s", exc)
|
||||
|
||||
except Exception as exc:
|
||||
log.warning("get_clock_offsets: %s", exc)
|
||||
return out
|
||||
|
||||
|
||||
def set_clock_offsets(
|
||||
gpc_offset_mhz: Optional[int] = None,
|
||||
mem_offset_mhz: Optional[int] = None,
|
||||
gpu_index: int = 0,
|
||||
) -> tuple[bool, str]:
|
||||
"""Set clock offsets (MHz) for the specified domains only.
|
||||
|
||||
Pass None for a domain to leave it untouched on hardware.
|
||||
Calls nvmlDeviceSetClockOffsets once per requested domain (GRAPHICS, MEM).
|
||||
Falls back to deprecated per-domain functions when the new API returns an
|
||||
error (e.g. NVML_ERROR_DEPRECATED=25 on Blackwell with driver 590.x).
|
||||
"""
|
||||
if gpc_offset_mhz is None and mem_offset_mhz is None:
|
||||
return True, "OK"
|
||||
if not _NVML_AVAILABLE:
|
||||
return False, "NVML not available (install nvidia-ml-py)"
|
||||
try:
|
||||
handle = _get_handle(gpu_index)
|
||||
|
||||
domains = []
|
||||
if gpc_offset_mhz is not None:
|
||||
domains.append((_NVML_CLOCK_GRAPHICS, gpc_offset_mhz))
|
||||
if mem_offset_mhz is not None:
|
||||
domains.append((_NVML_CLOCK_MEM, mem_offset_mhz))
|
||||
|
||||
_pynvml_set = getattr(pynvml, "nvmlDeviceSetClockOffsets", None)
|
||||
fn_set = _try_nvml_fn("nvmlDeviceSetClockOffsets") if _pynvml_set is None else None
|
||||
|
||||
if _pynvml_set is not None or fn_set is not None:
|
||||
all_ok = True
|
||||
for clock_type, offset in domains:
|
||||
info = _make_clock_offset(clock_type, pstate=0, offset_mhz=offset)
|
||||
try:
|
||||
rc = _pynvml_set(handle, ctypes.byref(info)) if _pynvml_set else fn_set(handle, ctypes.byref(info))
|
||||
if rc != 0:
|
||||
log.debug("nvmlDeviceSetClockOffsets(type=%d) returned %d — trying fallback", clock_type, rc)
|
||||
all_ok = False
|
||||
break
|
||||
except Exception as exc:
|
||||
log.debug("nvmlDeviceSetClockOffsets(type=%d): %s — trying fallback", clock_type, exc)
|
||||
all_ok = False
|
||||
break
|
||||
if all_ok:
|
||||
return True, "OK"
|
||||
# Non-zero rc (e.g. 25=DEPRECATED on Blackwell) — fall through to deprecated path.
|
||||
|
||||
# Deprecated per-domain fallback (works on Blackwell/driver 590.x).
|
||||
errs = []
|
||||
if gpc_offset_mhz is not None and hasattr(pynvml, "nvmlDeviceSetGpcClkVfOffset"):
|
||||
try:
|
||||
pynvml.nvmlDeviceSetGpcClkVfOffset(handle, gpc_offset_mhz)
|
||||
except Exception as exc:
|
||||
errs.append(f"GPC: {exc}")
|
||||
if mem_offset_mhz is not None and hasattr(pynvml, "nvmlDeviceSetMemClkVfOffset"):
|
||||
try:
|
||||
pynvml.nvmlDeviceSetMemClkVfOffset(handle, mem_offset_mhz)
|
||||
except Exception as exc:
|
||||
errs.append(f"MEM: {exc}")
|
||||
if errs:
|
||||
return False, "; ".join(errs)
|
||||
return True, "OK"
|
||||
|
||||
except Exception as exc:
|
||||
log.warning("set_clock_offsets: %s", exc)
|
||||
return False, str(exc)
|
||||
|
||||
|
||||
# ── Range queries ─────────────────────────────────────────────────────────────
|
||||
|
||||
def get_mem_offset_range(gpu_index: int = 0) -> dict:
|
||||
"""Return the min/max allowed memory clock offset (MHz).
|
||||
|
||||
Keys: min_mem_offset_mhz, max_mem_offset_mhz.
|
||||
Uses nvmlDeviceGetMemClkMinMaxVfOffset; falls back to observed RTX values.
|
||||
"""
|
||||
# Observed RTX 5090 defaults (NvAPI GetClockBoostRanges says -1000/+3000).
|
||||
out = {"min_mem_offset_mhz": -2000, "max_mem_offset_mhz": 3000}
|
||||
if not _NVML_AVAILABLE:
|
||||
return out
|
||||
try:
|
||||
handle = _get_handle(gpu_index)
|
||||
|
||||
if hasattr(pynvml, "nvmlDeviceGetMemClkMinMaxVfOffset"):
|
||||
result = pynvml.nvmlDeviceGetMemClkMinMaxVfOffset(handle)
|
||||
if isinstance(result, (list, tuple)) and len(result) >= 2:
|
||||
out["min_mem_offset_mhz"] = int(result[0])
|
||||
out["max_mem_offset_mhz"] = int(result[1])
|
||||
else:
|
||||
min_v = getattr(result, "minOffset", None)
|
||||
max_v = getattr(result, "maxOffset", None)
|
||||
if min_v is not None:
|
||||
out["min_mem_offset_mhz"] = int(min_v)
|
||||
if max_v is not None:
|
||||
out["max_mem_offset_mhz"] = int(max_v)
|
||||
return out
|
||||
|
||||
fn = _try_nvml_fn("nvmlDeviceGetMemClkMinMaxVfOffset")
|
||||
if fn is not None:
|
||||
min_v = ctypes.c_int(0)
|
||||
max_v = ctypes.c_int(0)
|
||||
rc = fn(handle, ctypes.byref(min_v), ctypes.byref(max_v))
|
||||
if rc == 0:
|
||||
out["min_mem_offset_mhz"] = int(min_v.value)
|
||||
out["max_mem_offset_mhz"] = int(max_v.value)
|
||||
|
||||
except Exception as exc:
|
||||
log.debug("get_mem_offset_range: %s", exc)
|
||||
return out
|
||||
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
"""Live GPU monitoring.
|
||||
|
||||
Voltage is read via NvAPI GetCurrentVoltage.
|
||||
Clock, temperature, power draw, and fan speed are read via NVML (nvidia-ml-py).
|
||||
"""
|
||||
|
||||
import struct
|
||||
import time
|
||||
from typing import Optional
|
||||
|
||||
from ..nvapi.bootstrap import nvcall
|
||||
from ..nvapi.constants import FUNC, VOLT_SIZE
|
||||
from ..nvapi.types import MonitoringSample
|
||||
|
||||
try:
|
||||
import pynvml as _pynvml
|
||||
_NVML_AVAILABLE = True
|
||||
except ImportError:
|
||||
_pynvml = None
|
||||
_NVML_AVAILABLE = False
|
||||
|
||||
_nvml_initialized = False
|
||||
|
||||
|
||||
def init_nvml() -> bool:
|
||||
"""Initialize NVML. Call once at startup. Returns True on success."""
|
||||
global _nvml_initialized
|
||||
if not _NVML_AVAILABLE:
|
||||
return False
|
||||
try:
|
||||
_pynvml.nvmlInit()
|
||||
_nvml_initialized = True
|
||||
return True
|
||||
except _pynvml.NVMLError:
|
||||
return False
|
||||
|
||||
|
||||
def shutdown_nvml() -> None:
|
||||
"""Shut down NVML. Call at process exit."""
|
||||
global _nvml_initialized
|
||||
if _NVML_AVAILABLE and _nvml_initialized:
|
||||
try:
|
||||
_pynvml.nvmlShutdown()
|
||||
except _pynvml.NVMLError:
|
||||
pass
|
||||
_nvml_initialized = False
|
||||
|
||||
|
||||
def get_driver_version() -> Optional[str]:
|
||||
"""Return the NVIDIA driver version string, or None if unavailable."""
|
||||
if not (_NVML_AVAILABLE and _nvml_initialized):
|
||||
return None
|
||||
try:
|
||||
return _pynvml.nvmlSystemGetDriverVersion()
|
||||
except _pynvml.NVMLError:
|
||||
return None
|
||||
|
||||
|
||||
def get_vram_total(gpu_index: int = 0) -> Optional[int]:
|
||||
"""Return total VRAM in bytes, or None if unavailable."""
|
||||
if not (_NVML_AVAILABLE and _nvml_initialized):
|
||||
return None
|
||||
try:
|
||||
handle = _pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
|
||||
return _pynvml.nvmlDeviceGetMemoryInfo(handle).total
|
||||
except _pynvml.NVMLError:
|
||||
return None
|
||||
|
||||
|
||||
def read_voltage(gpu) -> tuple[Optional[int], str]:
|
||||
"""Read current GPU core voltage in µV via NvAPI GetCurrentVoltage.
|
||||
|
||||
Returns (voltage_uV, "OK") or (None, error).
|
||||
"""
|
||||
d, err = nvcall(FUNC["GetCurrentVoltage"], gpu, VOLT_SIZE, ver=1)
|
||||
if not d:
|
||||
return None, err
|
||||
return struct.unpack_from("<I", d, 0x28)[0], "OK"
|
||||
|
||||
|
||||
def _nvml_read(gpu_index: int) -> dict:
|
||||
"""Read all NVML fields. Returns a dict with keys matching MonitoringSample fields."""
|
||||
out = {
|
||||
"clock_mhz": None, "temp_c": None, "power_w": None, "fan_pct": None,
|
||||
"pstate": None, "mem_used_bytes": None, "mem_total_bytes": None,
|
||||
"gpu_util_pct": None, "mem_util_pct": None, "mem_clock_mhz": None,
|
||||
}
|
||||
if not (_NVML_AVAILABLE and _nvml_initialized):
|
||||
return out
|
||||
try:
|
||||
handle = _pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
|
||||
|
||||
out["clock_mhz"] = float(_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_GRAPHICS))
|
||||
out["mem_clock_mhz"] = float(_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_MEM))
|
||||
out["temp_c"] = float(_pynvml.nvmlDeviceGetTemperature(handle, _pynvml.NVML_TEMPERATURE_GPU))
|
||||
out["power_w"] = _pynvml.nvmlDeviceGetPowerUsage(handle) / 1000.0 # mW → W
|
||||
out["pstate"] = int(_pynvml.nvmlDeviceGetPerformanceState(handle))
|
||||
|
||||
mem = _pynvml.nvmlDeviceGetMemoryInfo(handle)
|
||||
out["mem_used_bytes"] = mem.used
|
||||
out["mem_total_bytes"] = mem.total
|
||||
|
||||
util = _pynvml.nvmlDeviceGetUtilizationRates(handle)
|
||||
out["gpu_util_pct"] = float(util.gpu)
|
||||
out["mem_util_pct"] = float(util.memory)
|
||||
|
||||
try:
|
||||
out["fan_pct"] = float(_pynvml.nvmlDeviceGetFanSpeed(handle))
|
||||
except _pynvml.NVMLError:
|
||||
pass
|
||||
except _pynvml.NVMLError:
|
||||
pass
|
||||
return out
|
||||
|
||||
|
||||
def poll(gpu, gpu_index: int = 0) -> MonitoringSample:
|
||||
"""Read all available monitoring data and return a MonitoringSample."""
|
||||
voltage_uv, _ = read_voltage(gpu)
|
||||
nvml = _nvml_read(gpu_index)
|
||||
return MonitoringSample(
|
||||
timestamp=time.time(),
|
||||
voltage_uv=voltage_uv,
|
||||
**nvml,
|
||||
)
|
||||
@@ -0,0 +1,31 @@
|
||||
"""Clock boost range queries."""
|
||||
|
||||
import struct
|
||||
from typing import Optional
|
||||
|
||||
from ..nvapi.bootstrap import nvcall
|
||||
from ..nvapi.constants import FUNC, RANGES_SIZE
|
||||
|
||||
|
||||
def get_clock_ranges(gpu) -> tuple[Optional[dict], str]:
|
||||
"""Read clock domain min/max offset ranges via GetClockBoostRanges.
|
||||
|
||||
Returns ({"num_domains": int, "domains": [[int, ...], ...]}, "OK")
|
||||
or (None, error).
|
||||
|
||||
On RTX 5090: GPU core ±3000 MHz, memory -3000/+3000 MHz.
|
||||
"""
|
||||
d, err = nvcall(FUNC["GetClockBoostRanges"], gpu, RANGES_SIZE, ver=1)
|
||||
if not d:
|
||||
return None, err
|
||||
|
||||
num = struct.unpack_from("<I", d, 4)[0]
|
||||
domains = []
|
||||
for i in range(min(num, 32)):
|
||||
base = 0x08 + i * 0x48
|
||||
if base + 0x48 > len(d):
|
||||
break
|
||||
words = [struct.unpack_from("<i", d, base + j)[0] for j in range(0, 0x48, 4)]
|
||||
domains.append(words)
|
||||
|
||||
return {"num_domains": num, "domains": domains}, "OK"
|
||||
@@ -0,0 +1,156 @@
|
||||
"""Save and restore ClockBoostTable snapshots to/from disk."""
|
||||
|
||||
import ctypes
|
||||
import json
|
||||
import os
|
||||
import struct
|
||||
from datetime import datetime
|
||||
from typing import Optional
|
||||
|
||||
from ..nvapi.bootstrap import nvcall_raw
|
||||
from ..nvapi.constants import FUNC, CT_SIZE, CT_BASE, CT_STRIDE, CT_DELTA_OFF, CT_POINTS
|
||||
from ..nvapi.types import SnapshotInfo
|
||||
from .vfcurve import read_clock_table_raw, get_boost_mask
|
||||
|
||||
|
||||
def save(gpu, gpu_name: str, snapshot_dir: str, max_snapshots: int = 0) -> Optional[str]:
|
||||
"""Save the current ClockBoostTable to disk.
|
||||
|
||||
Writes both a binary .bin file and a human-readable .json metadata file.
|
||||
If max_snapshots > 0, deletes the oldest snapshots to stay within the limit.
|
||||
Returns the binary filepath on success, or None on failure.
|
||||
"""
|
||||
raw, err = read_clock_table_raw(gpu)
|
||||
if not raw:
|
||||
print(f"Failed to read ClockBoostTable: {err}")
|
||||
return None
|
||||
|
||||
os.makedirs(snapshot_dir, exist_ok=True)
|
||||
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
|
||||
bin_path = os.path.join(snapshot_dir, f"clock_boost_table_{ts}.bin")
|
||||
meta_path = os.path.join(snapshot_dir, f"clock_boost_table_{ts}.json")
|
||||
|
||||
with open(bin_path, "wb") as f:
|
||||
f.write(raw)
|
||||
|
||||
offsets = []
|
||||
max_entries = (len(raw) - CT_BASE) // CT_STRIDE
|
||||
for i in range(max_entries):
|
||||
off = CT_BASE + i * CT_STRIDE + CT_DELTA_OFF
|
||||
delta = struct.unpack_from("<i", raw, off)[0]
|
||||
offsets.append(delta)
|
||||
|
||||
meta = {
|
||||
"gpu": gpu_name,
|
||||
"timestamp": datetime.now().isoformat(),
|
||||
"file": bin_path,
|
||||
"size": len(raw),
|
||||
"offsets_kHz": offsets,
|
||||
"nonzero_offsets": sum(1 for o in offsets if o != 0),
|
||||
}
|
||||
with open(meta_path, "w") as f:
|
||||
json.dump(meta, f, indent=2)
|
||||
|
||||
print(f"Snapshot saved:")
|
||||
print(f" Binary: {bin_path}")
|
||||
print(f" Metadata: {meta_path}")
|
||||
print(f" Size: {len(raw)} bytes")
|
||||
print(f" Non-zero offsets: {meta['nonzero_offsets']}")
|
||||
|
||||
if max_snapshots > 0:
|
||||
_prune_snapshots(snapshot_dir, max_snapshots)
|
||||
|
||||
return bin_path
|
||||
|
||||
|
||||
def _prune_snapshots(snapshot_dir: str, max_snapshots: int) -> None:
|
||||
"""Delete oldest snapshots (both .bin and .json) to stay within max_snapshots."""
|
||||
bins = sorted(
|
||||
f for f in os.listdir(snapshot_dir) if f.endswith(".bin")
|
||||
) # oldest first (lexicographic = chronological for our timestamp format)
|
||||
excess = len(bins) - max_snapshots
|
||||
for fname in bins[:excess]:
|
||||
stem = fname[:-4] # strip .bin
|
||||
for ext in (".bin", ".json"):
|
||||
try:
|
||||
os.remove(os.path.join(snapshot_dir, stem + ext))
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def restore(gpu, snapshot_dir: str, filepath: str = None) -> bool:
|
||||
"""Restore a ClockBoostTable snapshot from disk.
|
||||
|
||||
If no filepath is given, uses the most recent snapshot in snapshot_dir.
|
||||
Returns True on success.
|
||||
"""
|
||||
if filepath is None:
|
||||
if not os.path.isdir(snapshot_dir):
|
||||
print(f"No snapshots found in {snapshot_dir}")
|
||||
return False
|
||||
bins = sorted(
|
||||
[f for f in os.listdir(snapshot_dir) if f.endswith(".bin")],
|
||||
reverse=True,
|
||||
)
|
||||
if not bins:
|
||||
print(f"No snapshot .bin files in {snapshot_dir}")
|
||||
return False
|
||||
filepath = os.path.join(snapshot_dir, bins[0])
|
||||
|
||||
if not os.path.isfile(filepath):
|
||||
print(f"Snapshot file not found: {filepath}")
|
||||
return False
|
||||
|
||||
with open(filepath, "rb") as f:
|
||||
raw = f.read()
|
||||
|
||||
if len(raw) != CT_SIZE:
|
||||
print(f"Snapshot size mismatch: expected {CT_SIZE}, got {len(raw)}")
|
||||
return False
|
||||
|
||||
vw = struct.unpack_from("<I", raw, 0)[0]
|
||||
expected_vw = (1 << 16) | CT_SIZE
|
||||
if vw != expected_vw:
|
||||
print(f"Version word mismatch: 0x{vw:08X} (expected 0x{expected_vw:08X})")
|
||||
return False
|
||||
|
||||
buf = ctypes.create_string_buffer(CT_SIZE)
|
||||
ctypes.memmove(buf, raw, CT_SIZE)
|
||||
|
||||
# Full mask for restore — write all points
|
||||
mask, _ = get_boost_mask(gpu)
|
||||
if mask:
|
||||
for i in range(32):
|
||||
buf[4 + i] = mask[i]
|
||||
|
||||
print(f"Restoring from: {filepath}")
|
||||
ret, desc = nvcall_raw(FUNC["SetClockBoostTable"], gpu, buf)
|
||||
print(f"SetClockBoostTable returned: {ret} ({desc})")
|
||||
return ret == 0
|
||||
|
||||
|
||||
def list_snapshots(snapshot_dir: str) -> list[SnapshotInfo]:
|
||||
"""Return metadata for all snapshots in snapshot_dir, newest first."""
|
||||
if not os.path.isdir(snapshot_dir):
|
||||
return []
|
||||
|
||||
results = []
|
||||
for fname in sorted(os.listdir(snapshot_dir), reverse=True):
|
||||
if not fname.endswith(".json"):
|
||||
continue
|
||||
meta_path = os.path.join(snapshot_dir, fname)
|
||||
try:
|
||||
with open(meta_path) as f:
|
||||
meta = json.load(f)
|
||||
bin_path = meta.get("file", meta_path.replace(".json", ".bin"))
|
||||
results.append(SnapshotInfo(
|
||||
filepath=bin_path,
|
||||
timestamp=meta.get("timestamp", ""),
|
||||
gpu=meta.get("gpu", ""),
|
||||
nonzero_offsets=meta.get("nonzero_offsets", 0),
|
||||
size=meta.get("size", 0),
|
||||
))
|
||||
except (json.JSONDecodeError, KeyError):
|
||||
continue
|
||||
|
||||
return results
|
||||
@@ -0,0 +1,286 @@
|
||||
"""Read and write the GPU V/F curve via NvAPI."""
|
||||
|
||||
import ctypes
|
||||
import struct
|
||||
from typing import Optional
|
||||
|
||||
from ..nvapi.bootstrap import nvcall, nvcall_raw
|
||||
from ..nvapi.constants import (
|
||||
FUNC,
|
||||
VFP_SIZE, VFP_BASE, VFP_STRIDE,
|
||||
CT_SIZE, CT_BASE, CT_STRIDE, CT_DELTA_OFF, CT_POINTS,
|
||||
)
|
||||
from ..nvapi.types import VFPoint, CurveState
|
||||
|
||||
|
||||
# ── Mask helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
# The boost mask is a static property of the GPU/driver — it does not change
|
||||
# at runtime. Cache it per GPU handle to avoid redundant GetClockBoostMask
|
||||
# calls on every HAL operation.
|
||||
_boost_mask_cache: dict[int, bytes] = {}
|
||||
|
||||
|
||||
def get_boost_mask(gpu) -> tuple[Optional[bytes], str]:
|
||||
"""Read the canonical 32-byte clock boost mask from the driver.
|
||||
|
||||
The result is cached per GPU handle: subsequent calls return the cached
|
||||
value without hitting the driver again.
|
||||
|
||||
Returns (mask_bytes, "OK") or (None, error).
|
||||
"""
|
||||
if gpu in _boost_mask_cache:
|
||||
return _boost_mask_cache[gpu], "OK"
|
||||
|
||||
from ..nvapi.constants import MASK_SIZE
|
||||
def fill(b):
|
||||
for i in range(4, 4 + 32):
|
||||
b[i] = 0xFF
|
||||
d, err = nvcall(FUNC["GetClockBoostMask"], gpu, MASK_SIZE, ver=1, pre_fill=fill)
|
||||
if d and len(d) >= 36:
|
||||
mask = d[4:36]
|
||||
_boost_mask_cache[gpu] = mask
|
||||
return mask, "OK"
|
||||
return None, err
|
||||
|
||||
|
||||
def set_mask_bit(buf, point: int, offset: int = 4) -> None:
|
||||
"""Set a single bit in the 256-bit mask for one point."""
|
||||
byte_idx = offset + (point // 8)
|
||||
bit_idx = point % 8
|
||||
buf[byte_idx] = int.from_bytes(buf[byte_idx:byte_idx + 1], "little") | (1 << bit_idx)
|
||||
|
||||
|
||||
def set_mask_bits(buf, points: set[int], offset: int = 4) -> None:
|
||||
"""Set mask bits for a set of points."""
|
||||
for p in points:
|
||||
set_mask_bit(buf, p, offset)
|
||||
|
||||
|
||||
# ── Readers ───────────────────────────────────────────────────────────────────
|
||||
|
||||
def read_vfp_curve(gpu) -> tuple[Optional[list[tuple[int, int]]], str]:
|
||||
"""Read the base V/F curve (frequency + voltage pairs).
|
||||
|
||||
Returns ([(freq_kHz, volt_uV), ...], "OK") or (None, error).
|
||||
"""
|
||||
mask, mask_err = get_boost_mask(gpu)
|
||||
if not mask:
|
||||
return None, f"GetClockBoostMask failed: {mask_err}"
|
||||
|
||||
def fill(buf):
|
||||
for i in range(32):
|
||||
buf[4 + i] = mask[i]
|
||||
|
||||
d, err = nvcall(FUNC["GetVFPCurve"], gpu, VFP_SIZE, ver=1, pre_fill=fill)
|
||||
if not d:
|
||||
return None, err
|
||||
|
||||
points = []
|
||||
max_entries = (len(d) - VFP_BASE) // VFP_STRIDE
|
||||
for i in range(max_entries):
|
||||
off = VFP_BASE + i * VFP_STRIDE
|
||||
freq = struct.unpack_from("<I", d, off)[0]
|
||||
volt = struct.unpack_from("<I", d, off + 4)[0]
|
||||
points.append((freq, volt))
|
||||
return points, "OK"
|
||||
|
||||
|
||||
def read_clock_table_raw(gpu) -> tuple[Optional[bytes], str]:
|
||||
"""Read the raw ClockBoostTable buffer.
|
||||
|
||||
Used for snapshots, inspection, and as the baseline for writes.
|
||||
Returns (bytes, "OK") or (None, error).
|
||||
"""
|
||||
mask, mask_err = get_boost_mask(gpu)
|
||||
if not mask:
|
||||
return None, f"GetClockBoostMask failed: {mask_err}"
|
||||
|
||||
def fill(buf):
|
||||
for i in range(32):
|
||||
buf[4 + i] = mask[i]
|
||||
|
||||
return nvcall(FUNC["GetClockBoostTable"], gpu, CT_SIZE, ver=1, pre_fill=fill)
|
||||
|
||||
|
||||
def read_clock_table_parsed(gpu) -> tuple[Optional[list[tuple[int, int]]], str]:
|
||||
"""Read per-point offsets and flags from the ClockBoostTable.
|
||||
|
||||
Returns a list of (delta_kHz, flags) tuples, or (None, error).
|
||||
"""
|
||||
d, err = read_clock_table_raw(gpu)
|
||||
if not d:
|
||||
return None, err
|
||||
|
||||
entries = []
|
||||
max_entries = (len(d) - CT_BASE) // CT_STRIDE
|
||||
for i in range(max_entries):
|
||||
base_off = CT_BASE + i * CT_STRIDE
|
||||
flags = struct.unpack_from("<I", d, base_off)[0]
|
||||
delta = struct.unpack_from("<i", d, base_off + CT_DELTA_OFF)[0]
|
||||
entries.append((delta, flags))
|
||||
|
||||
return entries, "OK"
|
||||
|
||||
|
||||
def read_clock_offsets(gpu) -> tuple[Optional[list[int]], str]:
|
||||
"""Read per-point frequency offsets (kHz, signed) from the ClockBoostTable.
|
||||
|
||||
Returns a list of integers, or (None, error).
|
||||
"""
|
||||
parsed, err = read_clock_table_parsed(gpu)
|
||||
if not parsed:
|
||||
return None, err
|
||||
|
||||
offsets = [delta for delta, flags in parsed]
|
||||
return offsets, "OK"
|
||||
|
||||
|
||||
def read_clock_entry_full(data: bytes, point: int) -> dict:
|
||||
"""Extract all 9 raw fields from a single ClockBoostTable entry.
|
||||
|
||||
Useful for diagnostics and verifying unknown fields.
|
||||
"""
|
||||
base = CT_BASE + point * CT_STRIDE
|
||||
fields = {}
|
||||
for j in range(9):
|
||||
off = base + j * 4
|
||||
if j == 5: # freqDelta is signed
|
||||
fields[f"field_{j:02d}_0x{j * 4:02X}"] = struct.unpack_from("<i", data, off)[0]
|
||||
else:
|
||||
fields[f"field_{j:02d}_0x{j * 4:02X}"] = struct.unpack_from("<I", data, off)[0]
|
||||
fields["freqDelta_kHz"] = fields["field_05_0x14"]
|
||||
return fields
|
||||
|
||||
|
||||
def read_curve(gpu, gpu_name: str = "") -> tuple[Optional[CurveState], str]:
|
||||
"""Read both the VFP curve and ClockBoostTable and merge into CurveState.
|
||||
|
||||
Returns (CurveState, "OK") or (None, error).
|
||||
"""
|
||||
import time
|
||||
vfp_points, vfp_err = read_vfp_curve(gpu)
|
||||
if not vfp_points:
|
||||
return None, vfp_err
|
||||
|
||||
ct_entries, ct_err = read_clock_table_parsed(gpu)
|
||||
if not ct_entries:
|
||||
return None, ct_err
|
||||
|
||||
points = []
|
||||
in_memory = False
|
||||
for i, (freq_khz, volt_uv) in enumerate(vfp_points):
|
||||
if freq_khz == 0 and volt_uv == 0:
|
||||
break # end of populated entries
|
||||
|
||||
delta_khz = ct_entries[i][0] if i < len(ct_entries) else 0
|
||||
flags = ct_entries[i][1] if i < len(ct_entries) else 0
|
||||
|
||||
if flags == 1:
|
||||
in_memory = True
|
||||
|
||||
points.append(VFPoint(
|
||||
index=i,
|
||||
freq_khz=freq_khz,
|
||||
volt_uv=volt_uv,
|
||||
delta_khz=delta_khz,
|
||||
domain="memory" if in_memory else "gpu",
|
||||
))
|
||||
|
||||
return CurveState(points=points, timestamp=time.time(), gpu_name=gpu_name), "OK"
|
||||
|
||||
|
||||
# ── Writers ───────────────────────────────────────────────────────────────────
|
||||
|
||||
def build_write_buffer(
|
||||
gpu,
|
||||
point_deltas: dict[int, int],
|
||||
full_mask: bool = False,
|
||||
) -> tuple[Optional[ctypes.Array], str]:
|
||||
"""Build a SetClockBoostTable buffer with specified per-point deltas.
|
||||
|
||||
Strategy: read the current ClockBoostTable, modify only the targeted
|
||||
entries' freqDelta fields, set only the targeted mask bits (single-bit per
|
||||
write to avoid touching neighbouring points).
|
||||
|
||||
Args:
|
||||
full_mask: if True, copy the complete GetClockBoostMask into the write
|
||||
buffer instead of the default sparse (per-point) mask.
|
||||
Older GPUs (e.g. Pascal) may require this.
|
||||
|
||||
Returns (mutable_buffer, "OK") or (None, error).
|
||||
"""
|
||||
current_raw, err = read_clock_table_raw(gpu)
|
||||
if not current_raw:
|
||||
return None, f"Cannot read current ClockBoostTable: {err}"
|
||||
|
||||
buf = ctypes.create_string_buffer(CT_SIZE)
|
||||
ctypes.memmove(buf, current_raw, CT_SIZE)
|
||||
|
||||
# Rewrite version word explicitly
|
||||
struct.pack_into("<I", buf, 0, (1 << 16) | CT_SIZE)
|
||||
|
||||
if full_mask:
|
||||
# Copy the complete boost mask — required by some older drivers/GPUs
|
||||
mask, mask_err = get_boost_mask(gpu)
|
||||
if not mask:
|
||||
return None, f"Cannot read boost mask: {mask_err}"
|
||||
for i, b in enumerate(mask):
|
||||
buf[4 + i] = b
|
||||
else:
|
||||
# Sparse mask — set only bits for points we're writing
|
||||
for i in range(4, 4 + 32):
|
||||
buf[i] = 0x00
|
||||
set_mask_bits(buf, set(point_deltas.keys()))
|
||||
|
||||
for point, delta_khz in point_deltas.items():
|
||||
off = CT_BASE + point * CT_STRIDE + CT_DELTA_OFF
|
||||
struct.pack_into("<i", buf, off, delta_khz)
|
||||
|
||||
return buf, "OK"
|
||||
|
||||
|
||||
def write_offsets(
|
||||
gpu,
|
||||
point_deltas: dict[int, int],
|
||||
dry_run: bool = False,
|
||||
full_mask: bool = False,
|
||||
) -> tuple[int, str]:
|
||||
"""Write per-point frequency offsets via SetClockBoostTable.
|
||||
|
||||
Args:
|
||||
gpu: NvAPI GPU handle
|
||||
point_deltas: {point_index: delta_kHz} — only these points are written
|
||||
dry_run: if True, build the buffer but don't call the driver
|
||||
full_mask: if True, use the full GetClockBoostMask (for older GPUs)
|
||||
|
||||
Returns (return_code, description).
|
||||
"""
|
||||
buf, err = build_write_buffer(gpu, point_deltas, full_mask=full_mask)
|
||||
if buf is None:
|
||||
return -999, err
|
||||
|
||||
if dry_run:
|
||||
return 0, "DRY RUN — buffer built but not sent to driver"
|
||||
|
||||
return nvcall_raw(FUNC["SetClockBoostTable"], gpu, buf)
|
||||
|
||||
|
||||
def write_global_offset(gpu, delta_khz: int, dry_run: bool = False) -> tuple[int, str]:
|
||||
"""Apply a uniform frequency offset to all GPU core points."""
|
||||
curve, err = read_curve(gpu)
|
||||
if not curve:
|
||||
return -999, f"Failed to read curve: {err}"
|
||||
|
||||
point_deltas = {p.index: delta_khz for p in curve.points if p.domain == "gpu"}
|
||||
return write_offsets(gpu, point_deltas, dry_run=dry_run)
|
||||
|
||||
|
||||
def reset_offsets(gpu, dry_run: bool = False) -> tuple[int, str]:
|
||||
"""Zero all GPU core frequency offsets."""
|
||||
curve, err = read_curve(gpu)
|
||||
if not curve:
|
||||
return -999, f"Failed to read curve: {err}"
|
||||
|
||||
point_deltas = {p.index: 0 for p in curve.points if p.domain == "gpu"}
|
||||
return write_offsets(gpu, point_deltas, dry_run=dry_run)
|
||||
Reference in new issue
Block a user