feat: add fan curve control via temperature-based fan speed curves

- Backend HAL (hal/fans.py): NVML v2 fan read/set/reset with min/max queries
- Server endpoints: GET/POST /api/fans, POST /api/fans/reset, POST /api/fans/speed
- Background fan poller: reads GPU temp every 2s, interpolates fan speed from curve
- Profile integration: fan_curve field saved/applied, auto-restore on shutdown
- Frontend: FanCurveEditor (SVG chart with drag/add/delete points), FanMonitor sidebar
- App.tsx: three-tab layout (Curve, Performance, Fans)
- GaugeCard: optional history sparkline, Fan Mode card without sparkline
- fan_mode field populated as 'curve' or 'auto' in GET /api/fans
This commit is contained in:
ARIA committed 2026-08-01 15:59:39 +02:00
1 parent 7605d21e27
commit 477c4becae
11 files changed
+1288 -10

No files matched your search

+198
View File
@@ -0,0 +1,198 @@
"""Hardware Abstraction Layer for Fan Control.
Uses NVML (via pynvml) for all operations:
- nvmlDeviceGetFanSpeed_v2 : read current fan speed % for a fan index
- nvmlDeviceSetFanSpeed_v2 : set fan speed % for a fan index
- nvmlDeviceGetMinMaxFanSpeed: get min/max fan speed constraints
- nvmlDeviceGetTemperature : read GPU temp for curve interpolation
"""
import ctypes
import logging
from typing import List, Optional
try:
import pynvml
_NVML_AVAILABLE = True
except ImportError:
_NVML_AVAILABLE = False
log = logging.getLogger("nvcurve.hal.fans")
# We use fan index 0 (first/primary fan) for all operations.
_FAN_INDEX = 0
def _get_handle(gpu_index: int):
"""Return an NVML device handle."""
if not _NVML_AVAILABLE:
raise RuntimeError("NVML not available (install nvidia-ml-py)")
return pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
def get_fan_info(gpu_index: int = 0) -> dict:
"""Return current fan state: fan_pct, fan_mode, min_fan_pct, max_fan_pct.
Returns None values on failure.
"""
out = {
"fan_pct": None,
"fan_mode": None,
"min_fan_pct": None,
"max_fan_pct": None,
}
if not _NVML_AVAILABLE:
return out
try:
handle = _get_handle(gpu_index)
# Get current fan speed using v2 API (fan index 0)
try:
out["fan_pct"] = float(pynvml.nvmlDeviceGetFanSpeed_v2(handle, _FAN_INDEX))
except pynvml.NVMLError:
# Fallback to legacy v1 API
try:
out["fan_pct"] = float(pynvml.nvmlDeviceGetFanSpeed(handle))
except pynvml.NVMLError:
pass
# Get min/max fan speed constraints
try:
min_s = ctypes.c_uint(0)
max_s = ctypes.c_uint(0)
pynvml.nvmlDeviceGetMinMaxFanSpeed(handle, min_s, max_s)
out["min_fan_pct"] = int(min_s.value)
out["max_fan_pct"] = int(max_s.value)
except pynvml.NVMLError:
out["min_fan_pct"] = 0
out["max_fan_pct"] = 100
except pynvml.NVMLError as exc:
log.warning("get_fan_info: %s", exc)
return out
def set_fan_speed(gpu_index: int, pct: int) -> tuple[bool, str]:
"""Set fan speed to a percentage (0-100) on the primary fan."""
pct = max(0, min(100, int(pct)))
if not _NVML_AVAILABLE:
return False, "NVML not available"
try:
handle = _get_handle(gpu_index)
pynvml.nvmlDeviceSetFanSpeed_v2(handle, _FAN_INDEX, pct)
return True, "OK"
except pynvml.NVMLError as exc:
log.warning("set_fan_speed(%d, %d): %s", gpu_index, pct, exc)
return False, str(exc)
def reset_fan(gpu_index: int = 0) -> tuple[bool, str]:
"""Restore automatic fan control.
Tries nvidia-smi --fan=default first (most reliable), then falls back to
NVML nvmlDeviceSetDefaultFanSpeed_v2.
"""
if not _NVML_AVAILABLE:
return False, "NVML not available"
import subprocess
# Try nvidia-smi approach first (most reliable for restoring auto)
try:
ret = subprocess.run(
["nvidia-smi", "-i", str(gpu_index), "-fan", "default"],
capture_output=True, text=True, timeout=10,
)
if ret.returncode == 0:
return True, "OK"
log.debug("nvidia-smi -fan default failed: %s", ret.stderr.strip())
except FileNotFoundError:
log.debug("nvidia-smi not found, falling back to NVML")
except subprocess.TimeoutExpired:
log.warning("nvidia-smi -fan default timed out")
except Exception as exc:
log.debug("nvidia-smi -fan default error: %s", exc)
# Fallback: use NVML to reset to default fan speed
try:
handle = _get_handle(gpu_index)
pynvml.nvmlDeviceSetDefaultFanSpeed_v2(handle, _FAN_INDEX)
return True, "OK"
except pynvml.NVMLError as exc:
return False, f"Failed to reset fan: {exc}"
def get_temp(gpu_index: int = 0) -> Optional[float]:
"""Read current GPU temperature in °C."""
if not _NVML_AVAILABLE:
return None
try:
handle = _get_handle(gpu_index)
return float(pynvml.nvmlDeviceGetTemperature(handle, pynvml.NVML_TEMPERATURE_GPU))
except pynvml.NVMLError as exc:
log.debug("get_temp: %s", exc)
return None
def interpolate_fan_speed(curve: List[dict], temp_c: float) -> Optional[int]:
"""Interpolate target fan speed from a curve at a given temperature.
curve: list of {temp_c: int, fan_pct: int} sorted by temp_c
Returns fan_pct clamped to 0-100, or None if curve is empty.
"""
if not curve or len(curve) < 2:
return None
temp = float(temp_c)
# Find the two surrounding points
for i in range(len(curve) - 1):
t0, f0 = curve[i]["temp_c"], curve[i]["fan_pct"]
t1, f1 = curve[i + 1]["temp_c"], curve[i + 1]["fan_pct"]
if t0 == t1:
continue
if t0 <= temp <= t1:
fraction = (temp - t0) / (t1 - t0)
result = f0 + fraction * (f1 - f0)
return max(0, min(100, int(round(result))))
# Outside range: clamp to first or last point
if temp <= curve[0]["temp_c"]:
return max(0, min(100, curve[0]["fan_pct"]))
return max(0, min(100, curve[-1]["fan_pct"]))
def validate_curve(curve: List[dict]) -> tuple[bool, str]:
"""Validate a fan curve.
Returns (True, "OK") or (False, error_message).
"""
if not curve or len(curve) < 2:
return False, "Fan curve requires at least 2 points"
temps = [p["temp_c"] for p in curve]
speeds = [p["fan_pct"] for p in curve]
# Check for duplicate temperatures
if len(temps) != len(set(temps)):
return False, "Fan curve has duplicate temperature values"
# Check temperature range
for t in temps:
if t < 0 or t > 120:
return False, f"Temperature {t}°C out of range (0-120)"
# Check fan speed range
for s in speeds:
if s < 0 or s > 100:
return False, f"Fan speed {s}% out of range (0-100)"
# Check sorted by temperature
for i in range(len(temps) - 1):
if temps[i] >= temps[i + 1]:
return False, "Fan curve points must be sorted by ascending temperature"
return True, "OK"
+1
View File
@@ -14,6 +14,7 @@ class ProfileData:
curve_deltas: Dict[str, int] # { "index": delta_khz }
mem_offset_mhz: Optional[int] = None
power_limit_w: Optional[int] = None
fan_curve: Optional[List[Dict[str, int]]] = None
def save_profile(profile_dir: str, data: ProfileData) -> str:
+180
View File
@@ -38,6 +38,14 @@ from .hal.limits import (
set_clock_offsets,
get_mem_offset_range,
)
from .hal.fans import (
get_fan_info,
set_fan_speed,
reset_fan,
get_temp,
interpolate_fan_speed,
validate_curve,
)
from .profiles.native import (
ProfileData,
save_profile,
@@ -169,6 +177,25 @@ async def _monitor_poller(gpu_index: int) -> None:
await asyncio.sleep(cfg.poll_interval_s)
async def _fan_poller(gpu_index: int) -> None:
"""Continuously read GPU temp, interpolate fan speed from active curve, and apply."""
while True:
try:
g_state = _state["gpus"].get(gpu_index)
if g_state and g_state.get("fan_curve_active") and g_state.get("fan_curve"):
temp = await _run(get_temp, gpu_index)
if temp is not None:
curve = g_state["fan_curve"]
target = interpolate_fan_speed(curve, temp)
if target is not None:
await _run(set_fan_speed, gpu_index, target)
except asyncio.CancelledError:
return
except Exception as exc:
log.warning("Fan poller error for GPU %d: %s", gpu_index, exc)
await asyncio.sleep(2.0)
# ── Lifespan ──────────────────────────────────────────────────────────────────
@asynccontextmanager
@@ -196,6 +223,9 @@ async def lifespan(app: FastAPI):
"active_profile": None,
"monitor_clients": set(),
"curve_clients": set(),
"fan_curve": None,
"fan_curve_active": False,
"fan_poller_task": None,
}
_state["gpus"][idx] = g_state
@@ -252,6 +282,23 @@ async def lifespan(app: FastAPI):
except asyncio.CancelledError:
pass
for gpu_index, g_state in _state["gpus"].items():
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
if g_state.get("fan_curve_active"):
g_state["fan_curve_active"] = False
g_state["fan_curve"] = None
try:
await loop.run_in_executor(None, reset_fan, gpu_index)
log.info("GPU %d: restored automatic fan control on shutdown", gpu_index)
except Exception as exc:
log.warning("GPU %d: failed to restore automatic fan control on shutdown: %s",
gpu_index, exc)
await loop.run_in_executor(None, shutdown_nvml)
@@ -305,6 +352,19 @@ class ConfigUpdateRequest(BaseModel):
gpu_index: int = 0
class FanCurvePoint(BaseModel):
temp_c: int
fan_pct: int
class FanCurveRequest(BaseModel):
curve: list[FanCurvePoint]
class FanSpeedRequest(BaseModel):
fan_pct: int
# ── Helper: run blocking HAL call in thread pool ──────────────────────────────
async def _run(fn, *args):
@@ -513,6 +573,7 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
curve_deltas=curve_deltas,
mem_offset_mhz=mem_offset_mhz,
power_limit_w=power_limit_w,
fan_curve=g_state.get("fan_curve") if g_state.get("fan_curve_active") else None,
)
filepath = await _run(save_profile, cfg.profile_dir, data)
g_state["active_profile"] = req.name
@@ -625,6 +686,36 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
await _update_offsets_and_broadcast(gpu_index)
# Apply fan curve if present in profile
if profile.fan_curve:
ok, msg = validate_curve(profile.fan_curve)
if not ok:
errs.append(f"Fan curve: {msg}")
else:
# Stop existing fan poller if running
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_curve"] = profile.fan_curve
g_state["fan_curve_active"] = True
g_state["fan_poller_task"] = asyncio.create_task(_fan_poller(gpu_index))
elif g_state.get("fan_curve_active"):
# Profile has no fan curve, deactivate any active fan curve
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_poller_task"] = None
g_state["fan_curve"] = None
g_state["fan_curve_active"] = False
await _run(reset_fan, gpu_index)
if not errs:
g_state["active_profile"] = name
return errs
@@ -812,6 +903,95 @@ async def api_limits_reset(gpu_index: int = 0):
return {"ok": True}
# ── Fan endpoints ──────────────────────────────────────────────────────────────
@app.get("/api/fans")
async def api_fans(gpu_index: int = 0):
"""Current fan state: fan %, curve, and whether curve control is active."""
_get_gpu_state(gpu_index)
g_state = _state["gpus"][gpu_index]
info = await _run(get_fan_info, gpu_index)
curve_active = g_state.get("fan_curve_active", False)
return {
**info,
"fan_mode": "curve" if curve_active else "auto",
"curve": g_state.get("fan_curve"),
"curve_active": curve_active,
}
@app.post("/api/fans")
async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
"""Set or update the fan curve. Starts the fan control poller."""
g_state = _get_gpu_state(gpu_index)
curve_data = [{"temp_c": p.temp_c, "fan_pct": p.fan_pct} for p in req.curve]
ok, msg = validate_curve(curve_data)
if not ok:
raise HTTPException(status_code=400, detail=msg)
# Test that fan control is available on this GPU, probing with the
# target for the *current* temperature so the fan is never briefly
# set to an inappropriate speed.
if curve_data:
sample = await _run(poll, g_state["gpu"], gpu_index)
test_temp = sample.temp_c if sample and sample.temp_c is not None else curve_data[0]["temp_c"]
target = interpolate_fan_speed(curve_data, test_temp)
if target is not None:
fan_ok, fan_msg = await _run(set_fan_speed, gpu_index, target)
if not fan_ok:
raise HTTPException(status_code=500, detail=f"Fan control not available: {fan_msg}")
# Stop existing poller if running
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_curve"] = curve_data
g_state["fan_curve_active"] = True
g_state["fan_poller_task"] = asyncio.create_task(_fan_poller(gpu_index))
return {"ok": True}
@app.post("/api/fans/reset")
async def api_fans_reset(gpu_index: int = 0):
"""Deactivate fan curve control and restore automatic fan mode."""
g_state = _get_gpu_state(gpu_index)
# Stop poller
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_poller_task"] = None
g_state["fan_curve"] = None
g_state["fan_curve_active"] = False
ok, msg = await _run(reset_fan, gpu_index)
if not ok:
log.warning("Fan reset warning: %s", msg)
return {"ok": True}
@app.post("/api/fans/speed")
async def api_fans_speed(req: FanSpeedRequest, gpu_index: int = 0):
"""One-shot set fan to an exact percentage (bypasses curve)."""
_get_gpu_state(gpu_index)
pct = max(0, min(100, req.fan_pct))
ok, msg = await _run(set_fan_speed, gpu_index, pct)
if not ok:
raise HTTPException(status_code=500, detail=msg)
return {"ok": True}
# ── Write endpoints ────────────────────────────────────────────────────────────
async def _reconcile_check(gpu_index: int) -> dict | None: