Files
nvcurve/nvcurve/hal/monitoring.py
T
ARIA 6cb33187d3 feat: full fan control — all fans or individual fans
The fan curve previously only controlled fan index 0; secondary fans
stayed on driver control. The curve can now target all fans (new
default) or any individual fan(s).

Backend:
- hal/fans.py: get_num_fans() via nvmlDeviceGetNumFans; get_fan_info()
  returns per-fan speeds; set_fan_speed() accepts a fan index list
  (None = all fans; all-fans mode is lenient toward driver-locked
  fans, explicit lists are strict); reset_fan() restores all fans.
- server.py: per-GPU fan_targets state; the poller applies the curve to
  all target fans and logs write failures (once per distinct error);
  activation validates targets against the hardware (stale indices fall
  back to all fans); POST /api/fans accepts fans, POST /api/fans/speed
  accepts a fan index, GET /api/fans returns num_fans/fans/fan_targets.
- Persistence format is now {"curve": ..., "fans": ...}; legacy
  bare-curve entries migrate to "all fans" at startup.
- Profiles save/apply fan_targets alongside fan_curve.
- MonitoringSample carries per-fan speeds for live gauges.

Frontend:
- Fans tab: All / Fan 1 / Fan 2 / ... selector with live per-fan %;
  the selection is applied together with the curve.
- Live monitor: per-fan gauges with sparklines for multi-fan GPUs.
2026-09-10 15:48:28 +02:00

217 lines
6.8 KiB
Python

"""Live GPU monitoring.
Voltage is read via NvAPI GetCurrentVoltage.
Clock, temperature, power draw, and fan speed are read via NVML (nvidia-ml-py).
"""
import contextlib
import struct
import time
from typing import Any
from ..nvapi.bootstrap import nvcall
from ..nvapi.constants import FUNC, VOLT_SIZE
from ..nvapi.types import MonitoringSample
try:
import pynvml as _pynvml_import
_NVML_AVAILABLE = True
except ImportError:
_pynvml_import = None
_NVML_AVAILABLE = False
# Aliased as Any so attribute access is not flagged when the import failed.
_pynvml: Any = _pynvml_import
_nvml_initialized = False
def init_nvml() -> bool:
"""Initialize NVML. Call once at startup. Returns True on success."""
global _nvml_initialized
if not _NVML_AVAILABLE:
return False
try:
_pynvml.nvmlInit()
_nvml_initialized = True
return True
except _pynvml.NVMLError:
return False
def shutdown_nvml() -> None:
"""Shut down NVML. Call at process exit."""
global _nvml_initialized
if _NVML_AVAILABLE and _nvml_initialized:
with contextlib.suppress(_pynvml.NVMLError):
_pynvml.nvmlShutdown()
_nvml_initialized = False
def get_driver_version() -> str | None:
"""Return the NVIDIA driver version string, or None if unavailable."""
if not (_NVML_AVAILABLE and _nvml_initialized):
return None
try:
return _pynvml.nvmlSystemGetDriverVersion()
except _pynvml.NVMLError:
return None
# NVML clock-throttle reason bits (from nvmlClocksThrottleReason* constants).
# Mapped to short human-readable labels for the dashboard.
_THROTTLE_REASONS: dict[int, str] = {
0x1: "GPU Idle",
0x2: "Application Clocks",
0x4: "SW Power Cap",
0x8: "HW Slowdown",
0x10: "Sync Boost",
0x20: "SW Thermal Slowdown",
0x40: "HW Thermal Slowdown",
0x80: "HW Power Brake",
0x100: "Display Clock Setting",
}
def throttle_reasons_label(mask: int | None) -> str | None:
"""Convert a throttle-reasons bitmask to a comma-joined label.
Returns None when the mask is unknown, or "No" when nothing is active.
"""
if mask is None:
return None
if mask == 0:
return "No"
active = [label for bit, label in _THROTTLE_REASONS.items() if mask & bit]
# Include any unknown high bits so nothing is silently dropped.
known = 0
for bit in _THROTTLE_REASONS:
known |= bit
extra = mask & ~known
if extra:
active.append(f"0x{extra:x}")
return ", ".join(active) if active else "No"
def get_vram_total(gpu_index: int = 0) -> int | None:
"""Return total VRAM in bytes, or None if unavailable."""
if not (_NVML_AVAILABLE and _nvml_initialized):
return None
try:
handle = _pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
return _pynvml.nvmlDeviceGetMemoryInfo(handle).total
except _pynvml.NVMLError:
return None
def read_voltage(gpu) -> tuple[int | None, str]:
"""Read current GPU core voltage in µV via NvAPI GetCurrentVoltage.
Returns (voltage_uV, "OK") or (None, error).
"""
d, err = nvcall(FUNC["GetCurrentVoltage"], gpu, VOLT_SIZE, ver=1)
if not d:
return None, err
return struct.unpack_from("<I", d, 0x28)[0], "OK"
def _nvml_read(gpu_index: int) -> dict:
"""Read all NVML fields. Returns a dict with keys matching MonitoringSample fields."""
out: dict[str, Any] = {
"clock_mhz": None,
"temp_c": None,
"power_w": None,
"fan_pct": None,
"pstate": None,
"mem_used_bytes": None,
"mem_total_bytes": None,
"gpu_util_pct": None,
"mem_util_pct": None,
"mem_clock_mhz": None,
"throttle_reasons": None,
"pcie_link_width": None,
"pcie_link_generation": None,
"mem_temp_c": None,
}
if not (_NVML_AVAILABLE and _nvml_initialized):
return out
try:
handle = _pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
out["clock_mhz"] = float(
_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_GRAPHICS)
)
out["mem_clock_mhz"] = float(
_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_MEM)
)
out["temp_c"] = float(
_pynvml.nvmlDeviceGetTemperature(handle, _pynvml.NVML_TEMPERATURE_GPU)
)
out["power_w"] = _pynvml.nvmlDeviceGetPowerUsage(handle) / 1000.0 # mW → W
out["pstate"] = int(_pynvml.nvmlDeviceGetPerformanceState(handle))
mem = _pynvml.nvmlDeviceGetMemoryInfo(handle)
out["mem_used_bytes"] = mem.used
out["mem_total_bytes"] = mem.total
util = _pynvml.nvmlDeviceGetUtilizationRates(handle)
out["gpu_util_pct"] = float(util.gpu)
out["mem_util_pct"] = float(util.memory)
with contextlib.suppress(_pynvml.NVMLError):
out["fan_pct"] = float(_pynvml.nvmlDeviceGetFanSpeed(handle))
# Per-fan speeds via the v2 API (fan_pct above stays fan 0 for legacy clients).
try:
num_fans = int(_pynvml.nvmlDeviceGetNumFans(handle))
fan_list: list[float | None] = []
for i in range(num_fans):
try:
fan_list.append(float(_pynvml.nvmlDeviceGetFanSpeed_v2(handle, i)))
except _pynvml.NVMLError:
fan_list.append(None)
if fan_list:
out["fans"] = fan_list
except _pynvml.NVMLError:
pass
with contextlib.suppress(_pynvml.NVMLError):
out["throttle_reasons"] = int(
_pynvml.nvmlDeviceGetCurrentClocksThrottleReasons(handle)
)
# PCIe link downclocks when idle, so read it live each poll.
with contextlib.suppress(_pynvml.NVMLError):
out["pcie_link_width"] = int(_pynvml.nvmlDeviceGetCurrPcieLinkWidth(handle))
with contextlib.suppress(_pynvml.NVMLError):
out["pcie_link_generation"] = int(
_pynvml.nvmlDeviceGetCurrPcieLinkGeneration(handle)
)
# VRAM temp (if the GPU exposes a MEMORY thermal sensor).
with contextlib.suppress(_pynvml.NVMLError):
sensors = _pynvml.nvmlDeviceGetThermalSettings(handle, 0)
for s in sensors:
if (
s.target == _pynvml.NVML_THERMAL_TARGET_MEMORY
and s.currentTemp is not None
and s.currentTemp > 0
):
out["mem_temp_c"] = float(s.currentTemp)
break
except _pynvml.NVMLError:
pass
return out
def poll(gpu, gpu_index: int = 0) -> MonitoringSample:
"""Read all available monitoring data and return a MonitoringSample."""
voltage_uv, _ = read_voltage(gpu)
nvml = _nvml_read(gpu_index)
return MonitoringSample(
timestamp=time.time(),
voltage_uv=voltage_uv,
**nvml,
)