nvcurve with some fixes and better limits

This commit is contained in:
ARIA committed 2026-05-09 15:05:29 +02:00
commit 024dcbceb0
70 files changed
+18890

No files matched your search

+13
View File
@@ -0,0 +1,13 @@
from .gpu import get_gpu, discover_gpus
from .vfcurve import read_curve, read_clock_offsets, write_offsets, reset_offsets
from .monitoring import poll, read_voltage
from .ranges import get_clock_ranges
from .snapshot import save as snapshot_save, restore as snapshot_restore, list_snapshots
__all__ = [
"get_gpu", "discover_gpus",
"read_curve", "read_clock_offsets", "write_offsets", "reset_offsets",
"poll", "read_voltage",
"get_clock_ranges",
"snapshot_save", "snapshot_restore", "list_snapshots",
]
+93
View File
@@ -0,0 +1,93 @@
"""GPU discovery and initialization."""
import ctypes
import sys
from ..nvapi.bootstrap import query_interface
from ..nvapi.constants import FUNC
from ..nvapi.types import GpuInfo
def init_nvapi() -> None:
"""Initialize NvAPI. Must be called before any GPU operations."""
init_fn = query_interface(FUNC["Initialize"], nargs=0)
if not init_fn or init_fn() != 0:
print("NvAPI_Initialize failed")
sys.exit(1)
def enumerate_gpus() -> tuple[ctypes.Array, int]:
"""Return (gpu_handles_array, count). Exits if no GPUs found."""
gpus = (ctypes.c_void_p * 64)()
ngpu = ctypes.c_int32()
query_interface(FUNC["EnumPhysicalGPUs"])(ctypes.byref(gpus), ctypes.byref(ngpu))
if ngpu.value == 0:
print("No NVIDIA GPUs found")
sys.exit(1)
return gpus, ngpu.value
def get_gpu_name(gpu) -> str:
"""Return the full name string for a GPU handle."""
name_buf = ctypes.create_string_buffer(256)
query_interface(FUNC["GetFullName"])(gpu, name_buf)
return name_buf.value.decode(errors="replace")
def discover_gpus() -> list[GpuInfo]:
"""Initialize NvAPI and NVML, and return a list of GpuInfo for all physical GPUs."""
init_nvapi()
gpus, count = enumerate_gpus()
infos = []
try:
import pynvml
pynvml.nvmlInit()
has_nvml = True
except Exception:
has_nvml = False
for i in range(count):
name = get_gpu_name(gpus[i])
uuid = None
pci_bus_id = None
if has_nvml:
try:
handle = pynvml.nvmlDeviceGetHandleByIndex(i)
uuid = pynvml.nvmlDeviceGetUUID(handle)
# NVML might return bytes
if isinstance(uuid, bytes):
uuid = uuid.decode('utf-8', errors='ignore')
pci_info = pynvml.nvmlDeviceGetPciInfo(handle)
# Parse something like "00000000:01:00.0" -> bus is 1
if isinstance(pci_info.bus, bytes):
pci_bus_id = int(pci_info.bus.decode('utf-8', errors='ignore'), 16)
else:
pci_bus_id = pci_info.bus
except Exception:
pass
infos.append(GpuInfo(name=name, index=i, uuid=uuid, pci_bus_id=pci_bus_id))
if has_nvml:
try:
pynvml.nvmlShutdown()
except Exception:
pass
return infos
def get_gpu(index: int = 0):
"""Initialize NvAPI, enumerate GPUs, and return the handle for `index`.
Also returns the GPU name as a convenience.
Returns (handle, name).
"""
init_nvapi()
gpus, count = enumerate_gpus()
if index >= count:
print(f"GPU index {index} out of range (found {count} GPU(s))")
sys.exit(1)
gpu = gpus[index]
name = get_gpu_name(gpu)
return gpu, name
+321
View File
@@ -0,0 +1,321 @@
"""Hardware Abstraction Layer for Global Limits (Power, Clock Offsets).
Uses NVML (via pynvml) for all operations.
Clock offsets use nvmlDeviceSetClockOffsets / nvmlDeviceGetClockOffsets
(introduced in driver 555.85). The older per-domain functions
(nvmlDeviceSet/GetGpcClkVfOffset, nvmlDeviceSet/GetMemClkVfOffset) are used
as a fallback when the new API is unavailable or returns an error.
set_clock_offsets accepts Optional values and only touches the domains
that are explicitly specified, leaving others unchanged on hardware.
"""
import ctypes
import subprocess
import logging
from typing import Optional
try:
import pynvml
_NVML_AVAILABLE = True
except ImportError:
_NVML_AVAILABLE = False
log = logging.getLogger("nvcurve.hal.limits")
# ── NVML library / handle helpers ─────────────────────────────────────────────
_nvml_lib: Optional[ctypes.CDLL] = None
def _nvml_cdll() -> ctypes.CDLL:
"""Return a ctypes handle to libnvidia-ml, reusing pynvml's load if possible."""
global _nvml_lib
if _nvml_lib is not None:
return _nvml_lib
# Prefer to reuse the library already loaded by pynvml to avoid dlopen races.
for attr in ("nvml", "_nvml"): # attribute name varies by pynvml version
mod = getattr(pynvml, attr, None)
lib = getattr(mod, "_lib", None) or getattr(mod, "_nvmlLib", None)
if lib is not None:
_nvml_lib = lib
return _nvml_lib
_nvml_lib = ctypes.CDLL("libnvidia-ml.so.1")
return _nvml_lib
def _get_handle(gpu_index: int):
"""Return an NVML device handle, initialising pynvml if needed."""
if not _NVML_AVAILABLE:
raise RuntimeError("NVML not available (install nvidia-ml-py)")
return pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
# ── Power limit ───────────────────────────────────────────────────────────────
def get_power_limit(gpu_index: int = 0) -> dict:
"""Return dict with power_limit_w, default_power_limit_w, min_power_limit_w, max_power_limit_w."""
out = {
"power_limit_w": None,
"default_power_limit_w": None,
"min_power_limit_w": None,
"max_power_limit_w": None,
}
try:
handle = _get_handle(gpu_index)
limit = pynvml.nvmlDeviceGetPowerManagementLimit(handle)
constrs = pynvml.nvmlDeviceGetPowerManagementLimitConstraints(handle)
out["power_limit_w"] = limit // 1000
out["min_power_limit_w"] = constrs[0] // 1000
out["max_power_limit_w"] = constrs[1] // 1000
try:
default = pynvml.nvmlDeviceGetPowerManagementDefaultLimit(handle)
out["default_power_limit_w"] = default // 1000
except Exception:
pass
except Exception as exc:
log.warning("get_power_limit: %s", exc)
return out
def set_power_limit(limit_w: int, gpu_index: int = 0) -> tuple[bool, str]:
"""Set the board power limit (Watts)."""
try:
handle = _get_handle(gpu_index)
pynvml.nvmlDeviceSetPowerManagementLimit(handle, limit_w * 1000)
return True, "OK"
except Exception as exc:
log.debug("NVML set_power_limit failed: %s — falling back to nvidia-smi", exc)
ret = subprocess.run(
["nvidia-smi", "-i", str(gpu_index), "-pl", str(limit_w)],
capture_output=True, text=True,
)
if ret.returncode == 0:
return True, "OK"
return False, ret.stderr.strip() or ret.stdout.strip()
# ── Clock offsets (GPC + memory) ──────────────────────────────────────────────
# The correct struct layout (per NVML docs and driver 590.x headers):
#
# typedef struct {
# unsigned int version; // nvmlClockOffset_v1
# nvmlClockType_t type; // NVML_CLOCK_GRAPHICS (0) or NVML_CLOCK_MEM (2)
# nvmlPstates_t pstate; // NVML_PSTATE_0 (0)
# int clockOffsetMHz;
# } nvmlClockOffset_t;
#
# nvmlDeviceSet/GetClockOffsets are called ONCE PER CLOCK DOMAIN.
# pynvml (nvidia-ml-py ≥ 12) exposes c_nvmlClockOffset_t and nvmlClockOffset_v1
# as ctypes objects; we use them when available and fall back to our own definition.
class _ClockOffset(ctypes.Structure):
_fields_ = [
("version", ctypes.c_uint),
("type", ctypes.c_uint), # nvmlClockType_t
("pstate", ctypes.c_uint), # nvmlPstates_t
("clockOffsetMHz", ctypes.c_int),
]
_CLOCK_OFFSET_VER = (1 << 24) | ctypes.sizeof(_ClockOffset) # = 0x01000010 (16 bytes)
# NVML clock-type constants (same values as pynvml).
_NVML_CLOCK_GRAPHICS = 0
_NVML_CLOCK_MEM = 2
def _make_clock_offset(clock_type: int, pstate: int = 0, offset_mhz: int = 0) -> ctypes.Structure:
"""Return a populated nvmlClockOffset_t struct, using pynvml's type when available."""
if hasattr(pynvml, "c_nvmlClockOffset_t") and hasattr(pynvml, "nvmlClockOffset_v1"):
info = pynvml.c_nvmlClockOffset_t()
info.version = pynvml.nvmlClockOffset_v1
info.type = clock_type
info.pstate = pstate
info.clockOffsetMHz = offset_mhz
return info
info = _ClockOffset()
info.version = _CLOCK_OFFSET_VER
info.type = clock_type
info.pstate = pstate
info.clockOffsetMHz = offset_mhz
return info
def _try_nvml_fn(name: str):
"""Return a ctypes-callable for an NVML function, or None if not found."""
lib = _nvml_cdll()
try:
return getattr(lib, name)
except AttributeError:
return None
def get_clock_offsets(gpu_index: int = 0) -> dict:
"""Return GPC and memory clock offsets (MHz).
Keys: gpc_offset_mhz, mem_offset_mhz (both int or None on failure).
Calls nvmlDeviceGetClockOffsets once per clock domain (GRAPHICS, MEM).
"""
out = {"gpc_offset_mhz": None, "mem_offset_mhz": None}
if not _NVML_AVAILABLE:
return out
try:
handle = _get_handle(gpu_index)
# Try pynvml wrapper first (nvidia-ml-py ≥ 12 exposes it correctly).
# Fall back to ctypes-direct if pynvml doesn't have it.
_pynvml_get = getattr(pynvml, "nvmlDeviceGetClockOffsets", None)
fn_get = _try_nvml_fn("nvmlDeviceGetClockOffsets") if _pynvml_get is None else None
used_new_api = False
for clock_type, key in ((_NVML_CLOCK_GRAPHICS, "gpc_offset_mhz"),
(_NVML_CLOCK_MEM, "mem_offset_mhz")):
info = _make_clock_offset(clock_type, pstate=0)
try:
if _pynvml_get is not None:
rc = _pynvml_get(handle, ctypes.byref(info))
elif fn_get is not None:
rc = fn_get(handle, ctypes.byref(info))
else:
break
if rc == 0:
out[key] = int(info.clockOffsetMHz)
used_new_api = True
else:
log.debug("nvmlDeviceGetClockOffsets(type=%d) returned %d", clock_type, rc)
except Exception as exc:
log.debug("nvmlDeviceGetClockOffsets(type=%d): %s", clock_type, exc)
if used_new_api:
return out
# Deprecated per-domain fallback.
if hasattr(pynvml, "nvmlDeviceGetGpcClkVfOffset"):
try:
out["gpc_offset_mhz"] = int(pynvml.nvmlDeviceGetGpcClkVfOffset(handle))
except Exception as exc:
log.debug("nvmlDeviceGetGpcClkVfOffset: %s", exc)
if hasattr(pynvml, "nvmlDeviceGetMemClkVfOffset"):
try:
res = pynvml.nvmlDeviceGetMemClkVfOffset(handle)
out["mem_offset_mhz"] = int(res[0] if isinstance(res, (list, tuple)) else res)
except Exception as exc:
log.debug("nvmlDeviceGetMemClkVfOffset: %s", exc)
except Exception as exc:
log.warning("get_clock_offsets: %s", exc)
return out
def set_clock_offsets(
gpc_offset_mhz: Optional[int] = None,
mem_offset_mhz: Optional[int] = None,
gpu_index: int = 0,
) -> tuple[bool, str]:
"""Set clock offsets (MHz) for the specified domains only.
Pass None for a domain to leave it untouched on hardware.
Calls nvmlDeviceSetClockOffsets once per requested domain (GRAPHICS, MEM).
Falls back to deprecated per-domain functions when the new API returns an
error (e.g. NVML_ERROR_DEPRECATED=25 on Blackwell with driver 590.x).
"""
if gpc_offset_mhz is None and mem_offset_mhz is None:
return True, "OK"
if not _NVML_AVAILABLE:
return False, "NVML not available (install nvidia-ml-py)"
try:
handle = _get_handle(gpu_index)
domains = []
if gpc_offset_mhz is not None:
domains.append((_NVML_CLOCK_GRAPHICS, gpc_offset_mhz))
if mem_offset_mhz is not None:
domains.append((_NVML_CLOCK_MEM, mem_offset_mhz))
_pynvml_set = getattr(pynvml, "nvmlDeviceSetClockOffsets", None)
fn_set = _try_nvml_fn("nvmlDeviceSetClockOffsets") if _pynvml_set is None else None
if _pynvml_set is not None or fn_set is not None:
all_ok = True
for clock_type, offset in domains:
info = _make_clock_offset(clock_type, pstate=0, offset_mhz=offset)
try:
rc = _pynvml_set(handle, ctypes.byref(info)) if _pynvml_set else fn_set(handle, ctypes.byref(info))
if rc != 0:
log.debug("nvmlDeviceSetClockOffsets(type=%d) returned %d — trying fallback", clock_type, rc)
all_ok = False
break
except Exception as exc:
log.debug("nvmlDeviceSetClockOffsets(type=%d): %s — trying fallback", clock_type, exc)
all_ok = False
break
if all_ok:
return True, "OK"
# Non-zero rc (e.g. 25=DEPRECATED on Blackwell) — fall through to deprecated path.
# Deprecated per-domain fallback (works on Blackwell/driver 590.x).
errs = []
if gpc_offset_mhz is not None and hasattr(pynvml, "nvmlDeviceSetGpcClkVfOffset"):
try:
pynvml.nvmlDeviceSetGpcClkVfOffset(handle, gpc_offset_mhz)
except Exception as exc:
errs.append(f"GPC: {exc}")
if mem_offset_mhz is not None and hasattr(pynvml, "nvmlDeviceSetMemClkVfOffset"):
try:
pynvml.nvmlDeviceSetMemClkVfOffset(handle, mem_offset_mhz)
except Exception as exc:
errs.append(f"MEM: {exc}")
if errs:
return False, "; ".join(errs)
return True, "OK"
except Exception as exc:
log.warning("set_clock_offsets: %s", exc)
return False, str(exc)
# ── Range queries ─────────────────────────────────────────────────────────────
def get_mem_offset_range(gpu_index: int = 0) -> dict:
"""Return the min/max allowed memory clock offset (MHz).
Keys: min_mem_offset_mhz, max_mem_offset_mhz.
Uses nvmlDeviceGetMemClkMinMaxVfOffset; falls back to observed RTX values.
"""
# Observed RTX 5090 defaults (NvAPI GetClockBoostRanges says -1000/+3000).
out = {"min_mem_offset_mhz": -2000, "max_mem_offset_mhz": 3000}
if not _NVML_AVAILABLE:
return out
try:
handle = _get_handle(gpu_index)
if hasattr(pynvml, "nvmlDeviceGetMemClkMinMaxVfOffset"):
result = pynvml.nvmlDeviceGetMemClkMinMaxVfOffset(handle)
if isinstance(result, (list, tuple)) and len(result) >= 2:
out["min_mem_offset_mhz"] = int(result[0])
out["max_mem_offset_mhz"] = int(result[1])
else:
min_v = getattr(result, "minOffset", None)
max_v = getattr(result, "maxOffset", None)
if min_v is not None:
out["min_mem_offset_mhz"] = int(min_v)
if max_v is not None:
out["max_mem_offset_mhz"] = int(max_v)
return out
fn = _try_nvml_fn("nvmlDeviceGetMemClkMinMaxVfOffset")
if fn is not None:
min_v = ctypes.c_int(0)
max_v = ctypes.c_int(0)
rc = fn(handle, ctypes.byref(min_v), ctypes.byref(max_v))
if rc == 0:
out["min_mem_offset_mhz"] = int(min_v.value)
out["max_mem_offset_mhz"] = int(max_v.value)
except Exception as exc:
log.debug("get_mem_offset_range: %s", exc)
return out
+124
View File
@@ -0,0 +1,124 @@
"""Live GPU monitoring.
Voltage is read via NvAPI GetCurrentVoltage.
Clock, temperature, power draw, and fan speed are read via NVML (nvidia-ml-py).
"""
import struct
import time
from typing import Optional
from ..nvapi.bootstrap import nvcall
from ..nvapi.constants import FUNC, VOLT_SIZE
from ..nvapi.types import MonitoringSample
try:
import pynvml as _pynvml
_NVML_AVAILABLE = True
except ImportError:
_pynvml = None
_NVML_AVAILABLE = False
_nvml_initialized = False
def init_nvml() -> bool:
"""Initialize NVML. Call once at startup. Returns True on success."""
global _nvml_initialized
if not _NVML_AVAILABLE:
return False
try:
_pynvml.nvmlInit()
_nvml_initialized = True
return True
except _pynvml.NVMLError:
return False
def shutdown_nvml() -> None:
"""Shut down NVML. Call at process exit."""
global _nvml_initialized
if _NVML_AVAILABLE and _nvml_initialized:
try:
_pynvml.nvmlShutdown()
except _pynvml.NVMLError:
pass
_nvml_initialized = False
def get_driver_version() -> Optional[str]:
"""Return the NVIDIA driver version string, or None if unavailable."""
if not (_NVML_AVAILABLE and _nvml_initialized):
return None
try:
return _pynvml.nvmlSystemGetDriverVersion()
except _pynvml.NVMLError:
return None
def get_vram_total(gpu_index: int = 0) -> Optional[int]:
"""Return total VRAM in bytes, or None if unavailable."""
if not (_NVML_AVAILABLE and _nvml_initialized):
return None
try:
handle = _pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
return _pynvml.nvmlDeviceGetMemoryInfo(handle).total
except _pynvml.NVMLError:
return None
def read_voltage(gpu) -> tuple[Optional[int], str]:
"""Read current GPU core voltage in µV via NvAPI GetCurrentVoltage.
Returns (voltage_uV, "OK") or (None, error).
"""
d, err = nvcall(FUNC["GetCurrentVoltage"], gpu, VOLT_SIZE, ver=1)
if not d:
return None, err
return struct.unpack_from("<I", d, 0x28)[0], "OK"
def _nvml_read(gpu_index: int) -> dict:
"""Read all NVML fields. Returns a dict with keys matching MonitoringSample fields."""
out = {
"clock_mhz": None, "temp_c": None, "power_w": None, "fan_pct": None,
"pstate": None, "mem_used_bytes": None, "mem_total_bytes": None,
"gpu_util_pct": None, "mem_util_pct": None, "mem_clock_mhz": None,
}
if not (_NVML_AVAILABLE and _nvml_initialized):
return out
try:
handle = _pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
out["clock_mhz"] = float(_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_GRAPHICS))
out["mem_clock_mhz"] = float(_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_MEM))
out["temp_c"] = float(_pynvml.nvmlDeviceGetTemperature(handle, _pynvml.NVML_TEMPERATURE_GPU))
out["power_w"] = _pynvml.nvmlDeviceGetPowerUsage(handle) / 1000.0 # mW → W
out["pstate"] = int(_pynvml.nvmlDeviceGetPerformanceState(handle))
mem = _pynvml.nvmlDeviceGetMemoryInfo(handle)
out["mem_used_bytes"] = mem.used
out["mem_total_bytes"] = mem.total
util = _pynvml.nvmlDeviceGetUtilizationRates(handle)
out["gpu_util_pct"] = float(util.gpu)
out["mem_util_pct"] = float(util.memory)
try:
out["fan_pct"] = float(_pynvml.nvmlDeviceGetFanSpeed(handle))
except _pynvml.NVMLError:
pass
except _pynvml.NVMLError:
pass
return out
def poll(gpu, gpu_index: int = 0) -> MonitoringSample:
"""Read all available monitoring data and return a MonitoringSample."""
voltage_uv, _ = read_voltage(gpu)
nvml = _nvml_read(gpu_index)
return MonitoringSample(
timestamp=time.time(),
voltage_uv=voltage_uv,
**nvml,
)
+31
View File
@@ -0,0 +1,31 @@
"""Clock boost range queries."""
import struct
from typing import Optional
from ..nvapi.bootstrap import nvcall
from ..nvapi.constants import FUNC, RANGES_SIZE
def get_clock_ranges(gpu) -> tuple[Optional[dict], str]:
"""Read clock domain min/max offset ranges via GetClockBoostRanges.
Returns ({"num_domains": int, "domains": [[int, ...], ...]}, "OK")
or (None, error).
On RTX 5090: GPU core ±3000 MHz, memory -3000/+3000 MHz.
"""
d, err = nvcall(FUNC["GetClockBoostRanges"], gpu, RANGES_SIZE, ver=1)
if not d:
return None, err
num = struct.unpack_from("<I", d, 4)[0]
domains = []
for i in range(min(num, 32)):
base = 0x08 + i * 0x48
if base + 0x48 > len(d):
break
words = [struct.unpack_from("<i", d, base + j)[0] for j in range(0, 0x48, 4)]
domains.append(words)
return {"num_domains": num, "domains": domains}, "OK"
+156
View File
@@ -0,0 +1,156 @@
"""Save and restore ClockBoostTable snapshots to/from disk."""
import ctypes
import json
import os
import struct
from datetime import datetime
from typing import Optional
from ..nvapi.bootstrap import nvcall_raw
from ..nvapi.constants import FUNC, CT_SIZE, CT_BASE, CT_STRIDE, CT_DELTA_OFF, CT_POINTS
from ..nvapi.types import SnapshotInfo
from .vfcurve import read_clock_table_raw, get_boost_mask
def save(gpu, gpu_name: str, snapshot_dir: str, max_snapshots: int = 0) -> Optional[str]:
"""Save the current ClockBoostTable to disk.
Writes both a binary .bin file and a human-readable .json metadata file.
If max_snapshots > 0, deletes the oldest snapshots to stay within the limit.
Returns the binary filepath on success, or None on failure.
"""
raw, err = read_clock_table_raw(gpu)
if not raw:
print(f"Failed to read ClockBoostTable: {err}")
return None
os.makedirs(snapshot_dir, exist_ok=True)
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
bin_path = os.path.join(snapshot_dir, f"clock_boost_table_{ts}.bin")
meta_path = os.path.join(snapshot_dir, f"clock_boost_table_{ts}.json")
with open(bin_path, "wb") as f:
f.write(raw)
offsets = []
max_entries = (len(raw) - CT_BASE) // CT_STRIDE
for i in range(max_entries):
off = CT_BASE + i * CT_STRIDE + CT_DELTA_OFF
delta = struct.unpack_from("<i", raw, off)[0]
offsets.append(delta)
meta = {
"gpu": gpu_name,
"timestamp": datetime.now().isoformat(),
"file": bin_path,
"size": len(raw),
"offsets_kHz": offsets,
"nonzero_offsets": sum(1 for o in offsets if o != 0),
}
with open(meta_path, "w") as f:
json.dump(meta, f, indent=2)
print(f"Snapshot saved:")
print(f" Binary: {bin_path}")
print(f" Metadata: {meta_path}")
print(f" Size: {len(raw)} bytes")
print(f" Non-zero offsets: {meta['nonzero_offsets']}")
if max_snapshots > 0:
_prune_snapshots(snapshot_dir, max_snapshots)
return bin_path
def _prune_snapshots(snapshot_dir: str, max_snapshots: int) -> None:
"""Delete oldest snapshots (both .bin and .json) to stay within max_snapshots."""
bins = sorted(
f for f in os.listdir(snapshot_dir) if f.endswith(".bin")
) # oldest first (lexicographic = chronological for our timestamp format)
excess = len(bins) - max_snapshots
for fname in bins[:excess]:
stem = fname[:-4] # strip .bin
for ext in (".bin", ".json"):
try:
os.remove(os.path.join(snapshot_dir, stem + ext))
except OSError:
pass
def restore(gpu, snapshot_dir: str, filepath: str = None) -> bool:
"""Restore a ClockBoostTable snapshot from disk.
If no filepath is given, uses the most recent snapshot in snapshot_dir.
Returns True on success.
"""
if filepath is None:
if not os.path.isdir(snapshot_dir):
print(f"No snapshots found in {snapshot_dir}")
return False
bins = sorted(
[f for f in os.listdir(snapshot_dir) if f.endswith(".bin")],
reverse=True,
)
if not bins:
print(f"No snapshot .bin files in {snapshot_dir}")
return False
filepath = os.path.join(snapshot_dir, bins[0])
if not os.path.isfile(filepath):
print(f"Snapshot file not found: {filepath}")
return False
with open(filepath, "rb") as f:
raw = f.read()
if len(raw) != CT_SIZE:
print(f"Snapshot size mismatch: expected {CT_SIZE}, got {len(raw)}")
return False
vw = struct.unpack_from("<I", raw, 0)[0]
expected_vw = (1 << 16) | CT_SIZE
if vw != expected_vw:
print(f"Version word mismatch: 0x{vw:08X} (expected 0x{expected_vw:08X})")
return False
buf = ctypes.create_string_buffer(CT_SIZE)
ctypes.memmove(buf, raw, CT_SIZE)
# Full mask for restore — write all points
mask, _ = get_boost_mask(gpu)
if mask:
for i in range(32):
buf[4 + i] = mask[i]
print(f"Restoring from: {filepath}")
ret, desc = nvcall_raw(FUNC["SetClockBoostTable"], gpu, buf)
print(f"SetClockBoostTable returned: {ret} ({desc})")
return ret == 0
def list_snapshots(snapshot_dir: str) -> list[SnapshotInfo]:
"""Return metadata for all snapshots in snapshot_dir, newest first."""
if not os.path.isdir(snapshot_dir):
return []
results = []
for fname in sorted(os.listdir(snapshot_dir), reverse=True):
if not fname.endswith(".json"):
continue
meta_path = os.path.join(snapshot_dir, fname)
try:
with open(meta_path) as f:
meta = json.load(f)
bin_path = meta.get("file", meta_path.replace(".json", ".bin"))
results.append(SnapshotInfo(
filepath=bin_path,
timestamp=meta.get("timestamp", ""),
gpu=meta.get("gpu", ""),
nonzero_offsets=meta.get("nonzero_offsets", 0),
size=meta.get("size", 0),
))
except (json.JSONDecodeError, KeyError):
continue
return results
+286
View File
@@ -0,0 +1,286 @@
"""Read and write the GPU V/F curve via NvAPI."""
import ctypes
import struct
from typing import Optional
from ..nvapi.bootstrap import nvcall, nvcall_raw
from ..nvapi.constants import (
FUNC,
VFP_SIZE, VFP_BASE, VFP_STRIDE,
CT_SIZE, CT_BASE, CT_STRIDE, CT_DELTA_OFF, CT_POINTS,
)
from ..nvapi.types import VFPoint, CurveState
# ── Mask helpers ─────────────────────────────────────────────────────────────
# The boost mask is a static property of the GPU/driver — it does not change
# at runtime. Cache it per GPU handle to avoid redundant GetClockBoostMask
# calls on every HAL operation.
_boost_mask_cache: dict[int, bytes] = {}
def get_boost_mask(gpu) -> tuple[Optional[bytes], str]:
"""Read the canonical 32-byte clock boost mask from the driver.
The result is cached per GPU handle: subsequent calls return the cached
value without hitting the driver again.
Returns (mask_bytes, "OK") or (None, error).
"""
if gpu in _boost_mask_cache:
return _boost_mask_cache[gpu], "OK"
from ..nvapi.constants import MASK_SIZE
def fill(b):
for i in range(4, 4 + 32):
b[i] = 0xFF
d, err = nvcall(FUNC["GetClockBoostMask"], gpu, MASK_SIZE, ver=1, pre_fill=fill)
if d and len(d) >= 36:
mask = d[4:36]
_boost_mask_cache[gpu] = mask
return mask, "OK"
return None, err
def set_mask_bit(buf, point: int, offset: int = 4) -> None:
"""Set a single bit in the 256-bit mask for one point."""
byte_idx = offset + (point // 8)
bit_idx = point % 8
buf[byte_idx] = int.from_bytes(buf[byte_idx:byte_idx + 1], "little") | (1 << bit_idx)
def set_mask_bits(buf, points: set[int], offset: int = 4) -> None:
"""Set mask bits for a set of points."""
for p in points:
set_mask_bit(buf, p, offset)
# ── Readers ───────────────────────────────────────────────────────────────────
def read_vfp_curve(gpu) -> tuple[Optional[list[tuple[int, int]]], str]:
"""Read the base V/F curve (frequency + voltage pairs).
Returns ([(freq_kHz, volt_uV), ...], "OK") or (None, error).
"""
mask, mask_err = get_boost_mask(gpu)
if not mask:
return None, f"GetClockBoostMask failed: {mask_err}"
def fill(buf):
for i in range(32):
buf[4 + i] = mask[i]
d, err = nvcall(FUNC["GetVFPCurve"], gpu, VFP_SIZE, ver=1, pre_fill=fill)
if not d:
return None, err
points = []
max_entries = (len(d) - VFP_BASE) // VFP_STRIDE
for i in range(max_entries):
off = VFP_BASE + i * VFP_STRIDE
freq = struct.unpack_from("<I", d, off)[0]
volt = struct.unpack_from("<I", d, off + 4)[0]
points.append((freq, volt))
return points, "OK"
def read_clock_table_raw(gpu) -> tuple[Optional[bytes], str]:
"""Read the raw ClockBoostTable buffer.
Used for snapshots, inspection, and as the baseline for writes.
Returns (bytes, "OK") or (None, error).
"""
mask, mask_err = get_boost_mask(gpu)
if not mask:
return None, f"GetClockBoostMask failed: {mask_err}"
def fill(buf):
for i in range(32):
buf[4 + i] = mask[i]
return nvcall(FUNC["GetClockBoostTable"], gpu, CT_SIZE, ver=1, pre_fill=fill)
def read_clock_table_parsed(gpu) -> tuple[Optional[list[tuple[int, int]]], str]:
"""Read per-point offsets and flags from the ClockBoostTable.
Returns a list of (delta_kHz, flags) tuples, or (None, error).
"""
d, err = read_clock_table_raw(gpu)
if not d:
return None, err
entries = []
max_entries = (len(d) - CT_BASE) // CT_STRIDE
for i in range(max_entries):
base_off = CT_BASE + i * CT_STRIDE
flags = struct.unpack_from("<I", d, base_off)[0]
delta = struct.unpack_from("<i", d, base_off + CT_DELTA_OFF)[0]
entries.append((delta, flags))
return entries, "OK"
def read_clock_offsets(gpu) -> tuple[Optional[list[int]], str]:
"""Read per-point frequency offsets (kHz, signed) from the ClockBoostTable.
Returns a list of integers, or (None, error).
"""
parsed, err = read_clock_table_parsed(gpu)
if not parsed:
return None, err
offsets = [delta for delta, flags in parsed]
return offsets, "OK"
def read_clock_entry_full(data: bytes, point: int) -> dict:
"""Extract all 9 raw fields from a single ClockBoostTable entry.
Useful for diagnostics and verifying unknown fields.
"""
base = CT_BASE + point * CT_STRIDE
fields = {}
for j in range(9):
off = base + j * 4
if j == 5: # freqDelta is signed
fields[f"field_{j:02d}_0x{j * 4:02X}"] = struct.unpack_from("<i", data, off)[0]
else:
fields[f"field_{j:02d}_0x{j * 4:02X}"] = struct.unpack_from("<I", data, off)[0]
fields["freqDelta_kHz"] = fields["field_05_0x14"]
return fields
def read_curve(gpu, gpu_name: str = "") -> tuple[Optional[CurveState], str]:
"""Read both the VFP curve and ClockBoostTable and merge into CurveState.
Returns (CurveState, "OK") or (None, error).
"""
import time
vfp_points, vfp_err = read_vfp_curve(gpu)
if not vfp_points:
return None, vfp_err
ct_entries, ct_err = read_clock_table_parsed(gpu)
if not ct_entries:
return None, ct_err
points = []
in_memory = False
for i, (freq_khz, volt_uv) in enumerate(vfp_points):
if freq_khz == 0 and volt_uv == 0:
break # end of populated entries
delta_khz = ct_entries[i][0] if i < len(ct_entries) else 0
flags = ct_entries[i][1] if i < len(ct_entries) else 0
if flags == 1:
in_memory = True
points.append(VFPoint(
index=i,
freq_khz=freq_khz,
volt_uv=volt_uv,
delta_khz=delta_khz,
domain="memory" if in_memory else "gpu",
))
return CurveState(points=points, timestamp=time.time(), gpu_name=gpu_name), "OK"
# ── Writers ───────────────────────────────────────────────────────────────────
def build_write_buffer(
gpu,
point_deltas: dict[int, int],
full_mask: bool = False,
) -> tuple[Optional[ctypes.Array], str]:
"""Build a SetClockBoostTable buffer with specified per-point deltas.
Strategy: read the current ClockBoostTable, modify only the targeted
entries' freqDelta fields, set only the targeted mask bits (single-bit per
write to avoid touching neighbouring points).
Args:
full_mask: if True, copy the complete GetClockBoostMask into the write
buffer instead of the default sparse (per-point) mask.
Older GPUs (e.g. Pascal) may require this.
Returns (mutable_buffer, "OK") or (None, error).
"""
current_raw, err = read_clock_table_raw(gpu)
if not current_raw:
return None, f"Cannot read current ClockBoostTable: {err}"
buf = ctypes.create_string_buffer(CT_SIZE)
ctypes.memmove(buf, current_raw, CT_SIZE)
# Rewrite version word explicitly
struct.pack_into("<I", buf, 0, (1 << 16) | CT_SIZE)
if full_mask:
# Copy the complete boost mask — required by some older drivers/GPUs
mask, mask_err = get_boost_mask(gpu)
if not mask:
return None, f"Cannot read boost mask: {mask_err}"
for i, b in enumerate(mask):
buf[4 + i] = b
else:
# Sparse mask — set only bits for points we're writing
for i in range(4, 4 + 32):
buf[i] = 0x00
set_mask_bits(buf, set(point_deltas.keys()))
for point, delta_khz in point_deltas.items():
off = CT_BASE + point * CT_STRIDE + CT_DELTA_OFF
struct.pack_into("<i", buf, off, delta_khz)
return buf, "OK"
def write_offsets(
gpu,
point_deltas: dict[int, int],
dry_run: bool = False,
full_mask: bool = False,
) -> tuple[int, str]:
"""Write per-point frequency offsets via SetClockBoostTable.
Args:
gpu: NvAPI GPU handle
point_deltas: {point_index: delta_kHz} — only these points are written
dry_run: if True, build the buffer but don't call the driver
full_mask: if True, use the full GetClockBoostMask (for older GPUs)
Returns (return_code, description).
"""
buf, err = build_write_buffer(gpu, point_deltas, full_mask=full_mask)
if buf is None:
return -999, err
if dry_run:
return 0, "DRY RUN — buffer built but not sent to driver"
return nvcall_raw(FUNC["SetClockBoostTable"], gpu, buf)
def write_global_offset(gpu, delta_khz: int, dry_run: bool = False) -> tuple[int, str]:
"""Apply a uniform frequency offset to all GPU core points."""
curve, err = read_curve(gpu)
if not curve:
return -999, f"Failed to read curve: {err}"
point_deltas = {p.index: delta_khz for p in curve.points if p.domain == "gpu"}
return write_offsets(gpu, point_deltas, dry_run=dry_run)
def reset_offsets(gpu, dry_run: bool = False) -> tuple[int, str]:
"""Zero all GPU core frequency offsets."""
curve, err = read_curve(gpu)
if not curve:
return -999, f"Failed to read curve: {err}"
point_deltas = {p.index: 0 for p in curve.points if p.domain == "gpu"}
return write_offsets(gpu, point_deltas, dry_run=dry_run)