The fan curve previously only controlled fan index 0; secondary fans
stayed on driver control. The curve can now target all fans (new
default) or any individual fan(s).
Backend:
- hal/fans.py: get_num_fans() via nvmlDeviceGetNumFans; get_fan_info()
returns per-fan speeds; set_fan_speed() accepts a fan index list
(None = all fans; all-fans mode is lenient toward driver-locked
fans, explicit lists are strict); reset_fan() restores all fans.
- server.py: per-GPU fan_targets state; the poller applies the curve to
all target fans and logs write failures (once per distinct error);
activation validates targets against the hardware (stale indices fall
back to all fans); POST /api/fans accepts fans, POST /api/fans/speed
accepts a fan index, GET /api/fans returns num_fans/fans/fan_targets.
- Persistence format is now {"curve": ..., "fans": ...}; legacy
bare-curve entries migrate to "all fans" at startup.
- Profiles save/apply fan_targets alongside fan_curve.
- MonitoringSample carries per-fan speeds for live gauges.
Frontend:
- Fans tab: All / Fan 1 / Fan 2 / ... selector with live per-fan %;
the selection is applied together with the curve.
- Live monitor: per-fan gauges with sparklines for multi-fan GPUs.
217 lines
6.8 KiB
Python
217 lines
6.8 KiB
Python
"""Live GPU monitoring.
|
|
|
|
Voltage is read via NvAPI GetCurrentVoltage.
|
|
Clock, temperature, power draw, and fan speed are read via NVML (nvidia-ml-py).
|
|
"""
|
|
|
|
import contextlib
|
|
import struct
|
|
import time
|
|
from typing import Any
|
|
|
|
from ..nvapi.bootstrap import nvcall
|
|
from ..nvapi.constants import FUNC, VOLT_SIZE
|
|
from ..nvapi.types import MonitoringSample
|
|
|
|
try:
|
|
import pynvml as _pynvml_import
|
|
|
|
_NVML_AVAILABLE = True
|
|
except ImportError:
|
|
_pynvml_import = None
|
|
_NVML_AVAILABLE = False
|
|
|
|
# Aliased as Any so attribute access is not flagged when the import failed.
|
|
_pynvml: Any = _pynvml_import
|
|
|
|
_nvml_initialized = False
|
|
|
|
|
|
def init_nvml() -> bool:
|
|
"""Initialize NVML. Call once at startup. Returns True on success."""
|
|
global _nvml_initialized
|
|
if not _NVML_AVAILABLE:
|
|
return False
|
|
try:
|
|
_pynvml.nvmlInit()
|
|
_nvml_initialized = True
|
|
return True
|
|
except _pynvml.NVMLError:
|
|
return False
|
|
|
|
|
|
def shutdown_nvml() -> None:
|
|
"""Shut down NVML. Call at process exit."""
|
|
global _nvml_initialized
|
|
if _NVML_AVAILABLE and _nvml_initialized:
|
|
with contextlib.suppress(_pynvml.NVMLError):
|
|
_pynvml.nvmlShutdown()
|
|
_nvml_initialized = False
|
|
|
|
|
|
def get_driver_version() -> str | None:
|
|
"""Return the NVIDIA driver version string, or None if unavailable."""
|
|
if not (_NVML_AVAILABLE and _nvml_initialized):
|
|
return None
|
|
try:
|
|
return _pynvml.nvmlSystemGetDriverVersion()
|
|
except _pynvml.NVMLError:
|
|
return None
|
|
|
|
|
|
# NVML clock-throttle reason bits (from nvmlClocksThrottleReason* constants).
|
|
# Mapped to short human-readable labels for the dashboard.
|
|
_THROTTLE_REASONS: dict[int, str] = {
|
|
0x1: "GPU Idle",
|
|
0x2: "Application Clocks",
|
|
0x4: "SW Power Cap",
|
|
0x8: "HW Slowdown",
|
|
0x10: "Sync Boost",
|
|
0x20: "SW Thermal Slowdown",
|
|
0x40: "HW Thermal Slowdown",
|
|
0x80: "HW Power Brake",
|
|
0x100: "Display Clock Setting",
|
|
}
|
|
|
|
|
|
def throttle_reasons_label(mask: int | None) -> str | None:
|
|
"""Convert a throttle-reasons bitmask to a comma-joined label.
|
|
|
|
Returns None when the mask is unknown, or "No" when nothing is active.
|
|
"""
|
|
if mask is None:
|
|
return None
|
|
if mask == 0:
|
|
return "No"
|
|
active = [label for bit, label in _THROTTLE_REASONS.items() if mask & bit]
|
|
# Include any unknown high bits so nothing is silently dropped.
|
|
known = 0
|
|
for bit in _THROTTLE_REASONS:
|
|
known |= bit
|
|
extra = mask & ~known
|
|
if extra:
|
|
active.append(f"0x{extra:x}")
|
|
return ", ".join(active) if active else "No"
|
|
|
|
|
|
def get_vram_total(gpu_index: int = 0) -> int | None:
|
|
"""Return total VRAM in bytes, or None if unavailable."""
|
|
if not (_NVML_AVAILABLE and _nvml_initialized):
|
|
return None
|
|
try:
|
|
handle = _pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
|
|
return _pynvml.nvmlDeviceGetMemoryInfo(handle).total
|
|
except _pynvml.NVMLError:
|
|
return None
|
|
|
|
|
|
def read_voltage(gpu) -> tuple[int | None, str]:
|
|
"""Read current GPU core voltage in µV via NvAPI GetCurrentVoltage.
|
|
|
|
Returns (voltage_uV, "OK") or (None, error).
|
|
"""
|
|
d, err = nvcall(FUNC["GetCurrentVoltage"], gpu, VOLT_SIZE, ver=1)
|
|
if not d:
|
|
return None, err
|
|
return struct.unpack_from("<I", d, 0x28)[0], "OK"
|
|
|
|
|
|
def _nvml_read(gpu_index: int) -> dict:
|
|
"""Read all NVML fields. Returns a dict with keys matching MonitoringSample fields."""
|
|
out: dict[str, Any] = {
|
|
"clock_mhz": None,
|
|
"temp_c": None,
|
|
"power_w": None,
|
|
"fan_pct": None,
|
|
"pstate": None,
|
|
"mem_used_bytes": None,
|
|
"mem_total_bytes": None,
|
|
"gpu_util_pct": None,
|
|
"mem_util_pct": None,
|
|
"mem_clock_mhz": None,
|
|
"throttle_reasons": None,
|
|
"pcie_link_width": None,
|
|
"pcie_link_generation": None,
|
|
"mem_temp_c": None,
|
|
}
|
|
if not (_NVML_AVAILABLE and _nvml_initialized):
|
|
return out
|
|
try:
|
|
handle = _pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
|
|
|
|
out["clock_mhz"] = float(
|
|
_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_GRAPHICS)
|
|
)
|
|
out["mem_clock_mhz"] = float(
|
|
_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_MEM)
|
|
)
|
|
out["temp_c"] = float(
|
|
_pynvml.nvmlDeviceGetTemperature(handle, _pynvml.NVML_TEMPERATURE_GPU)
|
|
)
|
|
out["power_w"] = _pynvml.nvmlDeviceGetPowerUsage(handle) / 1000.0 # mW → W
|
|
out["pstate"] = int(_pynvml.nvmlDeviceGetPerformanceState(handle))
|
|
|
|
mem = _pynvml.nvmlDeviceGetMemoryInfo(handle)
|
|
out["mem_used_bytes"] = mem.used
|
|
out["mem_total_bytes"] = mem.total
|
|
|
|
util = _pynvml.nvmlDeviceGetUtilizationRates(handle)
|
|
out["gpu_util_pct"] = float(util.gpu)
|
|
out["mem_util_pct"] = float(util.memory)
|
|
|
|
with contextlib.suppress(_pynvml.NVMLError):
|
|
out["fan_pct"] = float(_pynvml.nvmlDeviceGetFanSpeed(handle))
|
|
|
|
# Per-fan speeds via the v2 API (fan_pct above stays fan 0 for legacy clients).
|
|
try:
|
|
num_fans = int(_pynvml.nvmlDeviceGetNumFans(handle))
|
|
fan_list: list[float | None] = []
|
|
for i in range(num_fans):
|
|
try:
|
|
fan_list.append(float(_pynvml.nvmlDeviceGetFanSpeed_v2(handle, i)))
|
|
except _pynvml.NVMLError:
|
|
fan_list.append(None)
|
|
if fan_list:
|
|
out["fans"] = fan_list
|
|
except _pynvml.NVMLError:
|
|
pass
|
|
|
|
with contextlib.suppress(_pynvml.NVMLError):
|
|
out["throttle_reasons"] = int(
|
|
_pynvml.nvmlDeviceGetCurrentClocksThrottleReasons(handle)
|
|
)
|
|
|
|
# PCIe link downclocks when idle, so read it live each poll.
|
|
with contextlib.suppress(_pynvml.NVMLError):
|
|
out["pcie_link_width"] = int(_pynvml.nvmlDeviceGetCurrPcieLinkWidth(handle))
|
|
with contextlib.suppress(_pynvml.NVMLError):
|
|
out["pcie_link_generation"] = int(
|
|
_pynvml.nvmlDeviceGetCurrPcieLinkGeneration(handle)
|
|
)
|
|
|
|
# VRAM temp (if the GPU exposes a MEMORY thermal sensor).
|
|
with contextlib.suppress(_pynvml.NVMLError):
|
|
sensors = _pynvml.nvmlDeviceGetThermalSettings(handle, 0)
|
|
for s in sensors:
|
|
if (
|
|
s.target == _pynvml.NVML_THERMAL_TARGET_MEMORY
|
|
and s.currentTemp is not None
|
|
and s.currentTemp > 0
|
|
):
|
|
out["mem_temp_c"] = float(s.currentTemp)
|
|
break
|
|
except _pynvml.NVMLError:
|
|
pass
|
|
return out
|
|
|
|
|
|
def poll(gpu, gpu_index: int = 0) -> MonitoringSample:
|
|
"""Read all available monitoring data and return a MonitoringSample."""
|
|
voltage_uv, _ = read_voltage(gpu)
|
|
nvml = _nvml_read(gpu_index)
|
|
return MonitoringSample(
|
|
timestamp=time.time(),
|
|
voltage_uv=voltage_uv,
|
|
**nvml,
|
|
)
|