feat: add fan curve control via temperature-based fan speed curves
- Backend HAL (hal/fans.py): NVML v2 fan read/set/reset with min/max queries - Server endpoints: GET/POST /api/fans, POST /api/fans/reset, POST /api/fans/speed - Background fan poller: reads GPU temp every 2s, interpolates fan speed from curve - Profile integration: fan_curve field saved/applied, auto-restore on shutdown - Frontend: FanCurveEditor (SVG chart with drag/add/delete points), FanMonitor sidebar - App.tsx: three-tab layout (Curve, Performance, Fans) - GaugeCard: optional history sparkline, Fan Mode card without sparkline - fan_mode field populated as 'curve' or 'auto' in GET /api/fans
This commit is contained in:
1 parent
7605d21e27
commit
477c4becae
11 files changed
+1288
-10
No files matched your search
@@ -0,0 +1,198 @@
|
||||
"""Hardware Abstraction Layer for Fan Control.
|
||||
|
||||
Uses NVML (via pynvml) for all operations:
|
||||
- nvmlDeviceGetFanSpeed_v2 : read current fan speed % for a fan index
|
||||
- nvmlDeviceSetFanSpeed_v2 : set fan speed % for a fan index
|
||||
- nvmlDeviceGetMinMaxFanSpeed: get min/max fan speed constraints
|
||||
- nvmlDeviceGetTemperature : read GPU temp for curve interpolation
|
||||
"""
|
||||
|
||||
import ctypes
|
||||
import logging
|
||||
from typing import List, Optional
|
||||
|
||||
try:
|
||||
import pynvml
|
||||
_NVML_AVAILABLE = True
|
||||
except ImportError:
|
||||
_NVML_AVAILABLE = False
|
||||
|
||||
log = logging.getLogger("nvcurve.hal.fans")
|
||||
|
||||
# We use fan index 0 (first/primary fan) for all operations.
|
||||
_FAN_INDEX = 0
|
||||
|
||||
|
||||
def _get_handle(gpu_index: int):
|
||||
"""Return an NVML device handle."""
|
||||
if not _NVML_AVAILABLE:
|
||||
raise RuntimeError("NVML not available (install nvidia-ml-py)")
|
||||
return pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
|
||||
|
||||
|
||||
def get_fan_info(gpu_index: int = 0) -> dict:
|
||||
"""Return current fan state: fan_pct, fan_mode, min_fan_pct, max_fan_pct.
|
||||
|
||||
Returns None values on failure.
|
||||
"""
|
||||
out = {
|
||||
"fan_pct": None,
|
||||
"fan_mode": None,
|
||||
"min_fan_pct": None,
|
||||
"max_fan_pct": None,
|
||||
}
|
||||
if not _NVML_AVAILABLE:
|
||||
return out
|
||||
try:
|
||||
handle = _get_handle(gpu_index)
|
||||
|
||||
# Get current fan speed using v2 API (fan index 0)
|
||||
try:
|
||||
out["fan_pct"] = float(pynvml.nvmlDeviceGetFanSpeed_v2(handle, _FAN_INDEX))
|
||||
except pynvml.NVMLError:
|
||||
# Fallback to legacy v1 API
|
||||
try:
|
||||
out["fan_pct"] = float(pynvml.nvmlDeviceGetFanSpeed(handle))
|
||||
except pynvml.NVMLError:
|
||||
pass
|
||||
|
||||
# Get min/max fan speed constraints
|
||||
try:
|
||||
min_s = ctypes.c_uint(0)
|
||||
max_s = ctypes.c_uint(0)
|
||||
pynvml.nvmlDeviceGetMinMaxFanSpeed(handle, min_s, max_s)
|
||||
out["min_fan_pct"] = int(min_s.value)
|
||||
out["max_fan_pct"] = int(max_s.value)
|
||||
except pynvml.NVMLError:
|
||||
out["min_fan_pct"] = 0
|
||||
out["max_fan_pct"] = 100
|
||||
|
||||
except pynvml.NVMLError as exc:
|
||||
log.warning("get_fan_info: %s", exc)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
def set_fan_speed(gpu_index: int, pct: int) -> tuple[bool, str]:
|
||||
"""Set fan speed to a percentage (0-100) on the primary fan."""
|
||||
pct = max(0, min(100, int(pct)))
|
||||
if not _NVML_AVAILABLE:
|
||||
return False, "NVML not available"
|
||||
try:
|
||||
handle = _get_handle(gpu_index)
|
||||
pynvml.nvmlDeviceSetFanSpeed_v2(handle, _FAN_INDEX, pct)
|
||||
return True, "OK"
|
||||
except pynvml.NVMLError as exc:
|
||||
log.warning("set_fan_speed(%d, %d): %s", gpu_index, pct, exc)
|
||||
return False, str(exc)
|
||||
|
||||
|
||||
def reset_fan(gpu_index: int = 0) -> tuple[bool, str]:
|
||||
"""Restore automatic fan control.
|
||||
|
||||
Tries nvidia-smi --fan=default first (most reliable), then falls back to
|
||||
NVML nvmlDeviceSetDefaultFanSpeed_v2.
|
||||
"""
|
||||
if not _NVML_AVAILABLE:
|
||||
return False, "NVML not available"
|
||||
|
||||
import subprocess
|
||||
|
||||
# Try nvidia-smi approach first (most reliable for restoring auto)
|
||||
try:
|
||||
ret = subprocess.run(
|
||||
["nvidia-smi", "-i", str(gpu_index), "-fan", "default"],
|
||||
capture_output=True, text=True, timeout=10,
|
||||
)
|
||||
if ret.returncode == 0:
|
||||
return True, "OK"
|
||||
log.debug("nvidia-smi -fan default failed: %s", ret.stderr.strip())
|
||||
except FileNotFoundError:
|
||||
log.debug("nvidia-smi not found, falling back to NVML")
|
||||
except subprocess.TimeoutExpired:
|
||||
log.warning("nvidia-smi -fan default timed out")
|
||||
except Exception as exc:
|
||||
log.debug("nvidia-smi -fan default error: %s", exc)
|
||||
|
||||
# Fallback: use NVML to reset to default fan speed
|
||||
try:
|
||||
handle = _get_handle(gpu_index)
|
||||
pynvml.nvmlDeviceSetDefaultFanSpeed_v2(handle, _FAN_INDEX)
|
||||
return True, "OK"
|
||||
except pynvml.NVMLError as exc:
|
||||
return False, f"Failed to reset fan: {exc}"
|
||||
|
||||
|
||||
def get_temp(gpu_index: int = 0) -> Optional[float]:
|
||||
"""Read current GPU temperature in °C."""
|
||||
if not _NVML_AVAILABLE:
|
||||
return None
|
||||
try:
|
||||
handle = _get_handle(gpu_index)
|
||||
return float(pynvml.nvmlDeviceGetTemperature(handle, pynvml.NVML_TEMPERATURE_GPU))
|
||||
except pynvml.NVMLError as exc:
|
||||
log.debug("get_temp: %s", exc)
|
||||
return None
|
||||
|
||||
|
||||
def interpolate_fan_speed(curve: List[dict], temp_c: float) -> Optional[int]:
|
||||
"""Interpolate target fan speed from a curve at a given temperature.
|
||||
|
||||
curve: list of {temp_c: int, fan_pct: int} sorted by temp_c
|
||||
Returns fan_pct clamped to 0-100, or None if curve is empty.
|
||||
"""
|
||||
if not curve or len(curve) < 2:
|
||||
return None
|
||||
|
||||
temp = float(temp_c)
|
||||
|
||||
# Find the two surrounding points
|
||||
for i in range(len(curve) - 1):
|
||||
t0, f0 = curve[i]["temp_c"], curve[i]["fan_pct"]
|
||||
t1, f1 = curve[i + 1]["temp_c"], curve[i + 1]["fan_pct"]
|
||||
|
||||
if t0 == t1:
|
||||
continue
|
||||
|
||||
if t0 <= temp <= t1:
|
||||
fraction = (temp - t0) / (t1 - t0)
|
||||
result = f0 + fraction * (f1 - f0)
|
||||
return max(0, min(100, int(round(result))))
|
||||
|
||||
# Outside range: clamp to first or last point
|
||||
if temp <= curve[0]["temp_c"]:
|
||||
return max(0, min(100, curve[0]["fan_pct"]))
|
||||
return max(0, min(100, curve[-1]["fan_pct"]))
|
||||
|
||||
|
||||
def validate_curve(curve: List[dict]) -> tuple[bool, str]:
|
||||
"""Validate a fan curve.
|
||||
|
||||
Returns (True, "OK") or (False, error_message).
|
||||
"""
|
||||
if not curve or len(curve) < 2:
|
||||
return False, "Fan curve requires at least 2 points"
|
||||
|
||||
temps = [p["temp_c"] for p in curve]
|
||||
speeds = [p["fan_pct"] for p in curve]
|
||||
|
||||
# Check for duplicate temperatures
|
||||
if len(temps) != len(set(temps)):
|
||||
return False, "Fan curve has duplicate temperature values"
|
||||
|
||||
# Check temperature range
|
||||
for t in temps:
|
||||
if t < 0 or t > 120:
|
||||
return False, f"Temperature {t}°C out of range (0-120)"
|
||||
|
||||
# Check fan speed range
|
||||
for s in speeds:
|
||||
if s < 0 or s > 100:
|
||||
return False, f"Fan speed {s}% out of range (0-100)"
|
||||
|
||||
# Check sorted by temperature
|
||||
for i in range(len(temps) - 1):
|
||||
if temps[i] >= temps[i + 1]:
|
||||
return False, "Fan curve points must be sorted by ascending temperature"
|
||||
|
||||
return True, "OK"
|
||||
@@ -14,6 +14,7 @@ class ProfileData:
|
||||
curve_deltas: Dict[str, int] # { "index": delta_khz }
|
||||
mem_offset_mhz: Optional[int] = None
|
||||
power_limit_w: Optional[int] = None
|
||||
fan_curve: Optional[List[Dict[str, int]]] = None
|
||||
|
||||
|
||||
def save_profile(profile_dir: str, data: ProfileData) -> str:
|
||||
|
||||
@@ -38,6 +38,14 @@ from .hal.limits import (
|
||||
set_clock_offsets,
|
||||
get_mem_offset_range,
|
||||
)
|
||||
from .hal.fans import (
|
||||
get_fan_info,
|
||||
set_fan_speed,
|
||||
reset_fan,
|
||||
get_temp,
|
||||
interpolate_fan_speed,
|
||||
validate_curve,
|
||||
)
|
||||
from .profiles.native import (
|
||||
ProfileData,
|
||||
save_profile,
|
||||
@@ -169,6 +177,25 @@ async def _monitor_poller(gpu_index: int) -> None:
|
||||
await asyncio.sleep(cfg.poll_interval_s)
|
||||
|
||||
|
||||
async def _fan_poller(gpu_index: int) -> None:
|
||||
"""Continuously read GPU temp, interpolate fan speed from active curve, and apply."""
|
||||
while True:
|
||||
try:
|
||||
g_state = _state["gpus"].get(gpu_index)
|
||||
if g_state and g_state.get("fan_curve_active") and g_state.get("fan_curve"):
|
||||
temp = await _run(get_temp, gpu_index)
|
||||
if temp is not None:
|
||||
curve = g_state["fan_curve"]
|
||||
target = interpolate_fan_speed(curve, temp)
|
||||
if target is not None:
|
||||
await _run(set_fan_speed, gpu_index, target)
|
||||
except asyncio.CancelledError:
|
||||
return
|
||||
except Exception as exc:
|
||||
log.warning("Fan poller error for GPU %d: %s", gpu_index, exc)
|
||||
await asyncio.sleep(2.0)
|
||||
|
||||
|
||||
# ── Lifespan ──────────────────────────────────────────────────────────────────
|
||||
|
||||
@asynccontextmanager
|
||||
@@ -196,6 +223,9 @@ async def lifespan(app: FastAPI):
|
||||
"active_profile": None,
|
||||
"monitor_clients": set(),
|
||||
"curve_clients": set(),
|
||||
"fan_curve": None,
|
||||
"fan_curve_active": False,
|
||||
"fan_poller_task": None,
|
||||
}
|
||||
_state["gpus"][idx] = g_state
|
||||
|
||||
@@ -252,6 +282,23 @@ async def lifespan(app: FastAPI):
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
|
||||
for gpu_index, g_state in _state["gpus"].items():
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
if g_state.get("fan_curve_active"):
|
||||
g_state["fan_curve_active"] = False
|
||||
g_state["fan_curve"] = None
|
||||
try:
|
||||
await loop.run_in_executor(None, reset_fan, gpu_index)
|
||||
log.info("GPU %d: restored automatic fan control on shutdown", gpu_index)
|
||||
except Exception as exc:
|
||||
log.warning("GPU %d: failed to restore automatic fan control on shutdown: %s",
|
||||
gpu_index, exc)
|
||||
|
||||
await loop.run_in_executor(None, shutdown_nvml)
|
||||
|
||||
|
||||
@@ -305,6 +352,19 @@ class ConfigUpdateRequest(BaseModel):
|
||||
gpu_index: int = 0
|
||||
|
||||
|
||||
class FanCurvePoint(BaseModel):
|
||||
temp_c: int
|
||||
fan_pct: int
|
||||
|
||||
|
||||
class FanCurveRequest(BaseModel):
|
||||
curve: list[FanCurvePoint]
|
||||
|
||||
|
||||
class FanSpeedRequest(BaseModel):
|
||||
fan_pct: int
|
||||
|
||||
|
||||
# ── Helper: run blocking HAL call in thread pool ──────────────────────────────
|
||||
|
||||
async def _run(fn, *args):
|
||||
@@ -513,6 +573,7 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
|
||||
curve_deltas=curve_deltas,
|
||||
mem_offset_mhz=mem_offset_mhz,
|
||||
power_limit_w=power_limit_w,
|
||||
fan_curve=g_state.get("fan_curve") if g_state.get("fan_curve_active") else None,
|
||||
)
|
||||
filepath = await _run(save_profile, cfg.profile_dir, data)
|
||||
g_state["active_profile"] = req.name
|
||||
@@ -625,6 +686,36 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
|
||||
|
||||
await _update_offsets_and_broadcast(gpu_index)
|
||||
|
||||
# Apply fan curve if present in profile
|
||||
if profile.fan_curve:
|
||||
ok, msg = validate_curve(profile.fan_curve)
|
||||
if not ok:
|
||||
errs.append(f"Fan curve: {msg}")
|
||||
else:
|
||||
# Stop existing fan poller if running
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
|
||||
g_state["fan_curve"] = profile.fan_curve
|
||||
g_state["fan_curve_active"] = True
|
||||
g_state["fan_poller_task"] = asyncio.create_task(_fan_poller(gpu_index))
|
||||
elif g_state.get("fan_curve_active"):
|
||||
# Profile has no fan curve, deactivate any active fan curve
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
g_state["fan_poller_task"] = None
|
||||
g_state["fan_curve"] = None
|
||||
g_state["fan_curve_active"] = False
|
||||
await _run(reset_fan, gpu_index)
|
||||
|
||||
if not errs:
|
||||
g_state["active_profile"] = name
|
||||
return errs
|
||||
@@ -812,6 +903,95 @@ async def api_limits_reset(gpu_index: int = 0):
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
# ── Fan endpoints ──────────────────────────────────────────────────────────────
|
||||
|
||||
@app.get("/api/fans")
|
||||
async def api_fans(gpu_index: int = 0):
|
||||
"""Current fan state: fan %, curve, and whether curve control is active."""
|
||||
_get_gpu_state(gpu_index)
|
||||
g_state = _state["gpus"][gpu_index]
|
||||
info = await _run(get_fan_info, gpu_index)
|
||||
curve_active = g_state.get("fan_curve_active", False)
|
||||
return {
|
||||
**info,
|
||||
"fan_mode": "curve" if curve_active else "auto",
|
||||
"curve": g_state.get("fan_curve"),
|
||||
"curve_active": curve_active,
|
||||
}
|
||||
|
||||
|
||||
@app.post("/api/fans")
|
||||
async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
|
||||
"""Set or update the fan curve. Starts the fan control poller."""
|
||||
g_state = _get_gpu_state(gpu_index)
|
||||
|
||||
curve_data = [{"temp_c": p.temp_c, "fan_pct": p.fan_pct} for p in req.curve]
|
||||
ok, msg = validate_curve(curve_data)
|
||||
if not ok:
|
||||
raise HTTPException(status_code=400, detail=msg)
|
||||
|
||||
# Test that fan control is available on this GPU, probing with the
|
||||
# target for the *current* temperature so the fan is never briefly
|
||||
# set to an inappropriate speed.
|
||||
if curve_data:
|
||||
sample = await _run(poll, g_state["gpu"], gpu_index)
|
||||
test_temp = sample.temp_c if sample and sample.temp_c is not None else curve_data[0]["temp_c"]
|
||||
target = interpolate_fan_speed(curve_data, test_temp)
|
||||
if target is not None:
|
||||
fan_ok, fan_msg = await _run(set_fan_speed, gpu_index, target)
|
||||
if not fan_ok:
|
||||
raise HTTPException(status_code=500, detail=f"Fan control not available: {fan_msg}")
|
||||
|
||||
# Stop existing poller if running
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
|
||||
g_state["fan_curve"] = curve_data
|
||||
g_state["fan_curve_active"] = True
|
||||
g_state["fan_poller_task"] = asyncio.create_task(_fan_poller(gpu_index))
|
||||
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
@app.post("/api/fans/reset")
|
||||
async def api_fans_reset(gpu_index: int = 0):
|
||||
"""Deactivate fan curve control and restore automatic fan mode."""
|
||||
g_state = _get_gpu_state(gpu_index)
|
||||
|
||||
# Stop poller
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
g_state["fan_poller_task"] = None
|
||||
|
||||
g_state["fan_curve"] = None
|
||||
g_state["fan_curve_active"] = False
|
||||
|
||||
ok, msg = await _run(reset_fan, gpu_index)
|
||||
if not ok:
|
||||
log.warning("Fan reset warning: %s", msg)
|
||||
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
@app.post("/api/fans/speed")
|
||||
async def api_fans_speed(req: FanSpeedRequest, gpu_index: int = 0):
|
||||
"""One-shot set fan to an exact percentage (bypasses curve)."""
|
||||
_get_gpu_state(gpu_index)
|
||||
pct = max(0, min(100, req.fan_pct))
|
||||
ok, msg = await _run(set_fan_speed, gpu_index, pct)
|
||||
if not ok:
|
||||
raise HTTPException(status_code=500, detail=msg)
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
# ── Write endpoints ────────────────────────────────────────────────────────────
|
||||
|
||||
async def _reconcile_check(gpu_index: int) -> dict | None:
|
||||
|
||||
Reference in new issue
Block a user