feat: full fan control — all fans or individual fans

The fan curve previously only controlled fan index 0; secondary fans
stayed on driver control. The curve can now target all fans (new
default) or any individual fan(s).

Backend:
- hal/fans.py: get_num_fans() via nvmlDeviceGetNumFans; get_fan_info()
  returns per-fan speeds; set_fan_speed() accepts a fan index list
  (None = all fans; all-fans mode is lenient toward driver-locked
  fans, explicit lists are strict); reset_fan() restores all fans.
- server.py: per-GPU fan_targets state; the poller applies the curve to
  all target fans and logs write failures (once per distinct error);
  activation validates targets against the hardware (stale indices fall
  back to all fans); POST /api/fans accepts fans, POST /api/fans/speed
  accepts a fan index, GET /api/fans returns num_fans/fans/fan_targets.
- Persistence format is now {"curve": ..., "fans": ...}; legacy
  bare-curve entries migrate to "all fans" at startup.
- Profiles save/apply fan_targets alongside fan_curve.
- MonitoringSample carries per-fan speeds for live gauges.

Frontend:
- Fans tab: All / Fan 1 / Fan 2 / ... selector with live per-fan %;
  the selection is applied together with the curve.
- Live monitor: per-fan gauges with sparklines for multi-fan GPUs.
This commit is contained in:
ARIA committed 2026-09-10 15:48:28 +02:00
1 parent 34a9bc6d6e
commit 6cb33187d3
12 files changed
+382 -77

No files matched your search

+133 -39
View File
@@ -1,10 +1,15 @@
"""Hardware Abstraction Layer for Fan Control.
Uses NVML (via pynvml) for all operations:
- nvmlDeviceGetFanSpeed_v2 : read current fan speed % for a fan index
- nvmlDeviceSetFanSpeed_v2 : set fan speed % for a fan index
- nvmlDeviceGetMinMaxFanSpeed: get min/max fan speed constraints
- nvmlDeviceGetTemperature : read GPU temp for curve interpolation
- nvmlDeviceGetNumFans : number of fans on the device
- nvmlDeviceGetFanSpeed_v2 : read current fan speed % for a fan index
- nvmlDeviceSetFanSpeed_v2 : set fan speed % for a fan index
- nvmlDeviceSetDefaultFanSpeed_v2 : restore automatic control for a fan index
- nvmlDeviceGetMinMaxFanSpeed : get min/max fan speed constraints
- nvmlDeviceGetTemperature : read GPU temp for curve interpolation
Fans are addressed by 0-based index. Passing ``fans=None`` to the set/reset
helpers means "all fans on the device".
"""
import ctypes
@@ -24,9 +29,6 @@ pynvml: Any = _pynvml_import
log = logging.getLogger("nvcurve.hal.fans")
# We use fan index 0 (first/primary fan) for all operations.
_FAN_INDEX = 0
def _get_handle(gpu_index: int):
"""Return an NVML device handle."""
@@ -35,13 +37,42 @@ def _get_handle(gpu_index: int):
return pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
def _num_fans(handle) -> int:
"""Return the number of fans on the device (>= 1 on query failure)."""
try:
return max(0, int(pynvml.nvmlDeviceGetNumFans(handle)))
except pynvml.NVMLError:
# GetNumFans unsupported: assume at least the primary fan exists.
return 1
def get_num_fans(gpu_index: int = 0) -> int:
"""Return the number of fans on the GPU (0 if NVML is unavailable)."""
if not _NVML_AVAILABLE:
return 0
try:
return _num_fans(_get_handle(gpu_index))
except pynvml.NVMLError as exc:
log.warning("get_num_fans: %s", exc)
return 0
def get_fan_info(gpu_index: int = 0) -> dict:
"""Return current fan state: fan_pct, fan_mode, min_fan_pct, max_fan_pct.
"""Return current fan state for all fans.
Returns a dict with:
fan_pct : current speed % of fan 0 (legacy, None on failure)
fans : [{"index": i, "fan_pct": pct | None}, ...] per fan
num_fans : number of fans on the device
fan_mode : None (the server derives "auto"/"curve")
min_fan_pct / max_fan_pct : device-wide speed constraints
Returns None values on failure.
"""
out: dict[str, float | None] = {
out: dict[str, Any] = {
"fan_pct": None,
"fans": [],
"num_fans": 0,
"fan_mode": None,
"min_fan_pct": None,
"max_fan_pct": None,
@@ -50,36 +81,54 @@ def get_fan_info(gpu_index: int = 0) -> dict:
return out
try:
handle = _get_handle(gpu_index)
# Get current fan speed using v2 API (fan index 0)
try:
out["fan_pct"] = float(pynvml.nvmlDeviceGetFanSpeed_v2(handle, _FAN_INDEX))
except pynvml.NVMLError:
# Fallback to legacy v1 API
try:
out["fan_pct"] = float(pynvml.nvmlDeviceGetFanSpeed(handle))
except pynvml.NVMLError:
pass
# Get min/max fan speed constraints
try:
min_s = ctypes.c_uint(0)
max_s = ctypes.c_uint(0)
pynvml.nvmlDeviceGetMinMaxFanSpeed(handle, min_s, max_s)
out["min_fan_pct"] = int(min_s.value)
out["max_fan_pct"] = int(max_s.value)
except pynvml.NVMLError:
out["min_fan_pct"] = 0
out["max_fan_pct"] = 100
except pynvml.NVMLError as exc:
log.warning("get_fan_info: %s", exc)
return out
out["num_fans"] = _num_fans(handle)
# Per-fan speeds via the v2 API; fan 0 falls back to the legacy v1 API.
for i in range(out["num_fans"]):
pct: float | None = None
try:
pct = float(pynvml.nvmlDeviceGetFanSpeed_v2(handle, i))
except pynvml.NVMLError:
if i == 0:
try:
pct = float(pynvml.nvmlDeviceGetFanSpeed(handle))
except pynvml.NVMLError:
pct = None
out["fans"].append({"index": i, "fan_pct": pct})
out["fan_pct"] = out["fans"][0]["fan_pct"] if out["fans"] else None
# Get min/max fan speed constraints
try:
min_s = ctypes.c_uint(0)
max_s = ctypes.c_uint(0)
pynvml.nvmlDeviceGetMinMaxFanSpeed(handle, min_s, max_s)
out["min_fan_pct"] = int(min_s.value)
out["max_fan_pct"] = int(max_s.value)
except pynvml.NVMLError:
out["min_fan_pct"] = 0
out["max_fan_pct"] = 100
return out
def set_fan_speed(gpu_index: int, pct: int) -> tuple[bool, str]:
"""Set fan speed to a percentage (0-100) on the primary fan."""
def set_fan_speed(
gpu_index: int,
pct: int,
fans: list[int] | None = None,
) -> tuple[bool, str]:
"""Set fan speed to a percentage (0-100).
fans=None targets every fan on the device; fans=[0, 1] targets the
listed fan indices. In all-fans mode, fans the driver does not allow
manual control of (e.g. driver-mirrored secondary fans) are skipped
with a warning instead of failing the whole operation; explicit fan
lists are strict and fail if any selected fan cannot be set.
"""
try:
pct = max(0, min(100, int(pct)))
except (TypeError, ValueError):
@@ -88,7 +137,47 @@ def set_fan_speed(gpu_index: int, pct: int) -> tuple[bool, str]:
return False, "NVML not available"
try:
handle = _get_handle(gpu_index)
pynvml.nvmlDeviceSetFanSpeed_v2(handle, _FAN_INDEX, pct)
num_fans = _num_fans(handle)
if fans is None:
targets = list(range(num_fans))
strict = False
else:
targets: list[int] = []
for f in fans:
try:
f = int(f)
except (TypeError, ValueError):
return False, f"Invalid fan index: {f!r}"
if f < 0 or f >= num_fans:
return False, f"Fan index {f} out of range (0-{num_fans - 1})"
targets.append(f)
strict = True
if not targets:
return False, "No fans selected"
if not targets:
return False, "No fans available on this GPU"
skipped: list[str] = []
for i in targets:
try:
pynvml.nvmlDeviceSetFanSpeed_v2(handle, i, pct)
except pynvml.NVMLError as exc:
if strict:
log.warning(
"set_fan_speed(%d, fan %d, %d): %s", gpu_index, i, pct, exc
)
return False, str(exc)
# All-fans mode: secondary fans may be driver-controlled and
# reject manual writes; skip them and report in the message.
log.debug("set_fan_speed: fan %d not settable: %s", i, exc)
skipped.append(f"fan {i + 1}")
if skipped:
return (
True,
f"OK ({len(skipped)} fan(s) not manually controllable: {', '.join(skipped)})",
)
return True, "OK"
except pynvml.NVMLError as exc:
log.warning("set_fan_speed(%d, %d): %s", gpu_index, pct, exc)
@@ -96,10 +185,11 @@ def set_fan_speed(gpu_index: int, pct: int) -> tuple[bool, str]:
def reset_fan(gpu_index: int = 0) -> tuple[bool, str]:
"""Restore automatic fan control.
"""Restore automatic fan control for all fans.
Tries nvidia-smi --fan=default first (most reliable), then falls back to
NVML nvmlDeviceSetDefaultFanSpeed_v2.
Tries nvidia-smi --fan=default first (most reliable, resets all fans on
the device), then falls back to NVML nvmlDeviceSetDefaultFanSpeed_v2
per fan index.
"""
if not _NVML_AVAILABLE:
return False, "NVML not available"
@@ -124,10 +214,14 @@ def reset_fan(gpu_index: int = 0) -> tuple[bool, str]:
except Exception as exc:
log.debug("nvidia-smi -fan default error: %s", exc)
# Fallback: use NVML to reset to default fan speed
# Fallback: use NVML to reset every fan to default speed
try:
handle = _get_handle(gpu_index)
pynvml.nvmlDeviceSetDefaultFanSpeed_v2(handle, _FAN_INDEX)
for i in range(_num_fans(handle)):
try:
pynvml.nvmlDeviceSetDefaultFanSpeed_v2(handle, i)
except pynvml.NVMLError as exc:
log.debug("reset_fan: fan %d: %s", i, exc)
return True, "OK"
except pynvml.NVMLError as exc:
return False, f"Failed to reset fan: {exc}"
+14
View File
@@ -162,6 +162,20 @@ def _nvml_read(gpu_index: int) -> dict:
with contextlib.suppress(_pynvml.NVMLError):
out["fan_pct"] = float(_pynvml.nvmlDeviceGetFanSpeed(handle))
# Per-fan speeds via the v2 API (fan_pct above stays fan 0 for legacy clients).
try:
num_fans = int(_pynvml.nvmlDeviceGetNumFans(handle))
fan_list: list[float | None] = []
for i in range(num_fans):
try:
fan_list.append(float(_pynvml.nvmlDeviceGetFanSpeed_v2(handle, i)))
except _pynvml.NVMLError:
fan_list.append(None)
if fan_list:
out["fans"] = fan_list
except _pynvml.NVMLError:
pass
with contextlib.suppress(_pynvml.NVMLError):
out["throttle_reasons"] = int(
_pynvml.nvmlDeviceGetCurrentClocksThrottleReasons(handle)