feat: add fan curve control via temperature-based fan speed curves

- Backend HAL (hal/fans.py): NVML v2 fan read/set/reset with min/max queries
- Server endpoints: GET/POST /api/fans, POST /api/fans/reset, POST /api/fans/speed
- Background fan poller: reads GPU temp every 2s, interpolates fan speed from curve
- Profile integration: fan_curve field saved/applied, auto-restore on shutdown
- Frontend: FanCurveEditor (SVG chart with drag/add/delete points), FanMonitor sidebar
- App.tsx: three-tab layout (Curve, Performance, Fans)
- GaugeCard: optional history sparkline, Fan Mode card without sparkline
- fan_mode field populated as 'curve' or 'auto' in GET /api/fans
This commit is contained in:
ARIA committed 2026-08-01 15:59:39 +02:00
1 parent 7605d21e27
commit 477c4becae
11 files changed
+1288 -10

No files matched your search

+180
View File
@@ -38,6 +38,14 @@ from .hal.limits import (
set_clock_offsets,
get_mem_offset_range,
)
from .hal.fans import (
get_fan_info,
set_fan_speed,
reset_fan,
get_temp,
interpolate_fan_speed,
validate_curve,
)
from .profiles.native import (
ProfileData,
save_profile,
@@ -169,6 +177,25 @@ async def _monitor_poller(gpu_index: int) -> None:
await asyncio.sleep(cfg.poll_interval_s)
async def _fan_poller(gpu_index: int) -> None:
"""Continuously read GPU temp, interpolate fan speed from active curve, and apply."""
while True:
try:
g_state = _state["gpus"].get(gpu_index)
if g_state and g_state.get("fan_curve_active") and g_state.get("fan_curve"):
temp = await _run(get_temp, gpu_index)
if temp is not None:
curve = g_state["fan_curve"]
target = interpolate_fan_speed(curve, temp)
if target is not None:
await _run(set_fan_speed, gpu_index, target)
except asyncio.CancelledError:
return
except Exception as exc:
log.warning("Fan poller error for GPU %d: %s", gpu_index, exc)
await asyncio.sleep(2.0)
# ── Lifespan ──────────────────────────────────────────────────────────────────
@asynccontextmanager
@@ -196,6 +223,9 @@ async def lifespan(app: FastAPI):
"active_profile": None,
"monitor_clients": set(),
"curve_clients": set(),
"fan_curve": None,
"fan_curve_active": False,
"fan_poller_task": None,
}
_state["gpus"][idx] = g_state
@@ -252,6 +282,23 @@ async def lifespan(app: FastAPI):
except asyncio.CancelledError:
pass
for gpu_index, g_state in _state["gpus"].items():
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
if g_state.get("fan_curve_active"):
g_state["fan_curve_active"] = False
g_state["fan_curve"] = None
try:
await loop.run_in_executor(None, reset_fan, gpu_index)
log.info("GPU %d: restored automatic fan control on shutdown", gpu_index)
except Exception as exc:
log.warning("GPU %d: failed to restore automatic fan control on shutdown: %s",
gpu_index, exc)
await loop.run_in_executor(None, shutdown_nvml)
@@ -305,6 +352,19 @@ class ConfigUpdateRequest(BaseModel):
gpu_index: int = 0
class FanCurvePoint(BaseModel):
temp_c: int
fan_pct: int
class FanCurveRequest(BaseModel):
curve: list[FanCurvePoint]
class FanSpeedRequest(BaseModel):
fan_pct: int
# ── Helper: run blocking HAL call in thread pool ──────────────────────────────
async def _run(fn, *args):
@@ -513,6 +573,7 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
curve_deltas=curve_deltas,
mem_offset_mhz=mem_offset_mhz,
power_limit_w=power_limit_w,
fan_curve=g_state.get("fan_curve") if g_state.get("fan_curve_active") else None,
)
filepath = await _run(save_profile, cfg.profile_dir, data)
g_state["active_profile"] = req.name
@@ -625,6 +686,36 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
await _update_offsets_and_broadcast(gpu_index)
# Apply fan curve if present in profile
if profile.fan_curve:
ok, msg = validate_curve(profile.fan_curve)
if not ok:
errs.append(f"Fan curve: {msg}")
else:
# Stop existing fan poller if running
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_curve"] = profile.fan_curve
g_state["fan_curve_active"] = True
g_state["fan_poller_task"] = asyncio.create_task(_fan_poller(gpu_index))
elif g_state.get("fan_curve_active"):
# Profile has no fan curve, deactivate any active fan curve
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_poller_task"] = None
g_state["fan_curve"] = None
g_state["fan_curve_active"] = False
await _run(reset_fan, gpu_index)
if not errs:
g_state["active_profile"] = name
return errs
@@ -812,6 +903,95 @@ async def api_limits_reset(gpu_index: int = 0):
return {"ok": True}
# ── Fan endpoints ──────────────────────────────────────────────────────────────
@app.get("/api/fans")
async def api_fans(gpu_index: int = 0):
"""Current fan state: fan %, curve, and whether curve control is active."""
_get_gpu_state(gpu_index)
g_state = _state["gpus"][gpu_index]
info = await _run(get_fan_info, gpu_index)
curve_active = g_state.get("fan_curve_active", False)
return {
**info,
"fan_mode": "curve" if curve_active else "auto",
"curve": g_state.get("fan_curve"),
"curve_active": curve_active,
}
@app.post("/api/fans")
async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
"""Set or update the fan curve. Starts the fan control poller."""
g_state = _get_gpu_state(gpu_index)
curve_data = [{"temp_c": p.temp_c, "fan_pct": p.fan_pct} for p in req.curve]
ok, msg = validate_curve(curve_data)
if not ok:
raise HTTPException(status_code=400, detail=msg)
# Test that fan control is available on this GPU, probing with the
# target for the *current* temperature so the fan is never briefly
# set to an inappropriate speed.
if curve_data:
sample = await _run(poll, g_state["gpu"], gpu_index)
test_temp = sample.temp_c if sample and sample.temp_c is not None else curve_data[0]["temp_c"]
target = interpolate_fan_speed(curve_data, test_temp)
if target is not None:
fan_ok, fan_msg = await _run(set_fan_speed, gpu_index, target)
if not fan_ok:
raise HTTPException(status_code=500, detail=f"Fan control not available: {fan_msg}")
# Stop existing poller if running
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_curve"] = curve_data
g_state["fan_curve_active"] = True
g_state["fan_poller_task"] = asyncio.create_task(_fan_poller(gpu_index))
return {"ok": True}
@app.post("/api/fans/reset")
async def api_fans_reset(gpu_index: int = 0):
"""Deactivate fan curve control and restore automatic fan mode."""
g_state = _get_gpu_state(gpu_index)
# Stop poller
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_poller_task"] = None
g_state["fan_curve"] = None
g_state["fan_curve_active"] = False
ok, msg = await _run(reset_fan, gpu_index)
if not ok:
log.warning("Fan reset warning: %s", msg)
return {"ok": True}
@app.post("/api/fans/speed")
async def api_fans_speed(req: FanSpeedRequest, gpu_index: int = 0):
"""One-shot set fan to an exact percentage (bypasses curve)."""
_get_gpu_state(gpu_index)
pct = max(0, min(100, req.fan_pct))
ok, msg = await _run(set_fan_speed, gpu_index, pct)
if not ok:
raise HTTPException(status_code=500, detail=msg)
return {"ok": True}
# ── Write endpoints ────────────────────────────────────────────────────────────
async def _reconcile_check(gpu_index: int) -> dict | None: