feat: add fan curve control via temperature-based fan speed curves
- Backend HAL (hal/fans.py): NVML v2 fan read/set/reset with min/max queries - Server endpoints: GET/POST /api/fans, POST /api/fans/reset, POST /api/fans/speed - Background fan poller: reads GPU temp every 2s, interpolates fan speed from curve - Profile integration: fan_curve field saved/applied, auto-restore on shutdown - Frontend: FanCurveEditor (SVG chart with drag/add/delete points), FanMonitor sidebar - App.tsx: three-tab layout (Curve, Performance, Fans) - GaugeCard: optional history sparkline, Fan Mode card without sparkline - fan_mode field populated as 'curve' or 'auto' in GET /api/fans
This commit is contained in:
1 parent
7605d21e27
commit
477c4becae
11 files changed
+1288
-10
No files matched your search
@@ -38,6 +38,14 @@ from .hal.limits import (
|
||||
set_clock_offsets,
|
||||
get_mem_offset_range,
|
||||
)
|
||||
from .hal.fans import (
|
||||
get_fan_info,
|
||||
set_fan_speed,
|
||||
reset_fan,
|
||||
get_temp,
|
||||
interpolate_fan_speed,
|
||||
validate_curve,
|
||||
)
|
||||
from .profiles.native import (
|
||||
ProfileData,
|
||||
save_profile,
|
||||
@@ -169,6 +177,25 @@ async def _monitor_poller(gpu_index: int) -> None:
|
||||
await asyncio.sleep(cfg.poll_interval_s)
|
||||
|
||||
|
||||
async def _fan_poller(gpu_index: int) -> None:
|
||||
"""Continuously read GPU temp, interpolate fan speed from active curve, and apply."""
|
||||
while True:
|
||||
try:
|
||||
g_state = _state["gpus"].get(gpu_index)
|
||||
if g_state and g_state.get("fan_curve_active") and g_state.get("fan_curve"):
|
||||
temp = await _run(get_temp, gpu_index)
|
||||
if temp is not None:
|
||||
curve = g_state["fan_curve"]
|
||||
target = interpolate_fan_speed(curve, temp)
|
||||
if target is not None:
|
||||
await _run(set_fan_speed, gpu_index, target)
|
||||
except asyncio.CancelledError:
|
||||
return
|
||||
except Exception as exc:
|
||||
log.warning("Fan poller error for GPU %d: %s", gpu_index, exc)
|
||||
await asyncio.sleep(2.0)
|
||||
|
||||
|
||||
# ── Lifespan ──────────────────────────────────────────────────────────────────
|
||||
|
||||
@asynccontextmanager
|
||||
@@ -196,6 +223,9 @@ async def lifespan(app: FastAPI):
|
||||
"active_profile": None,
|
||||
"monitor_clients": set(),
|
||||
"curve_clients": set(),
|
||||
"fan_curve": None,
|
||||
"fan_curve_active": False,
|
||||
"fan_poller_task": None,
|
||||
}
|
||||
_state["gpus"][idx] = g_state
|
||||
|
||||
@@ -252,6 +282,23 @@ async def lifespan(app: FastAPI):
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
|
||||
for gpu_index, g_state in _state["gpus"].items():
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
if g_state.get("fan_curve_active"):
|
||||
g_state["fan_curve_active"] = False
|
||||
g_state["fan_curve"] = None
|
||||
try:
|
||||
await loop.run_in_executor(None, reset_fan, gpu_index)
|
||||
log.info("GPU %d: restored automatic fan control on shutdown", gpu_index)
|
||||
except Exception as exc:
|
||||
log.warning("GPU %d: failed to restore automatic fan control on shutdown: %s",
|
||||
gpu_index, exc)
|
||||
|
||||
await loop.run_in_executor(None, shutdown_nvml)
|
||||
|
||||
|
||||
@@ -305,6 +352,19 @@ class ConfigUpdateRequest(BaseModel):
|
||||
gpu_index: int = 0
|
||||
|
||||
|
||||
class FanCurvePoint(BaseModel):
|
||||
temp_c: int
|
||||
fan_pct: int
|
||||
|
||||
|
||||
class FanCurveRequest(BaseModel):
|
||||
curve: list[FanCurvePoint]
|
||||
|
||||
|
||||
class FanSpeedRequest(BaseModel):
|
||||
fan_pct: int
|
||||
|
||||
|
||||
# ── Helper: run blocking HAL call in thread pool ──────────────────────────────
|
||||
|
||||
async def _run(fn, *args):
|
||||
@@ -513,6 +573,7 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
|
||||
curve_deltas=curve_deltas,
|
||||
mem_offset_mhz=mem_offset_mhz,
|
||||
power_limit_w=power_limit_w,
|
||||
fan_curve=g_state.get("fan_curve") if g_state.get("fan_curve_active") else None,
|
||||
)
|
||||
filepath = await _run(save_profile, cfg.profile_dir, data)
|
||||
g_state["active_profile"] = req.name
|
||||
@@ -625,6 +686,36 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
|
||||
|
||||
await _update_offsets_and_broadcast(gpu_index)
|
||||
|
||||
# Apply fan curve if present in profile
|
||||
if profile.fan_curve:
|
||||
ok, msg = validate_curve(profile.fan_curve)
|
||||
if not ok:
|
||||
errs.append(f"Fan curve: {msg}")
|
||||
else:
|
||||
# Stop existing fan poller if running
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
|
||||
g_state["fan_curve"] = profile.fan_curve
|
||||
g_state["fan_curve_active"] = True
|
||||
g_state["fan_poller_task"] = asyncio.create_task(_fan_poller(gpu_index))
|
||||
elif g_state.get("fan_curve_active"):
|
||||
# Profile has no fan curve, deactivate any active fan curve
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
g_state["fan_poller_task"] = None
|
||||
g_state["fan_curve"] = None
|
||||
g_state["fan_curve_active"] = False
|
||||
await _run(reset_fan, gpu_index)
|
||||
|
||||
if not errs:
|
||||
g_state["active_profile"] = name
|
||||
return errs
|
||||
@@ -812,6 +903,95 @@ async def api_limits_reset(gpu_index: int = 0):
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
# ── Fan endpoints ──────────────────────────────────────────────────────────────
|
||||
|
||||
@app.get("/api/fans")
|
||||
async def api_fans(gpu_index: int = 0):
|
||||
"""Current fan state: fan %, curve, and whether curve control is active."""
|
||||
_get_gpu_state(gpu_index)
|
||||
g_state = _state["gpus"][gpu_index]
|
||||
info = await _run(get_fan_info, gpu_index)
|
||||
curve_active = g_state.get("fan_curve_active", False)
|
||||
return {
|
||||
**info,
|
||||
"fan_mode": "curve" if curve_active else "auto",
|
||||
"curve": g_state.get("fan_curve"),
|
||||
"curve_active": curve_active,
|
||||
}
|
||||
|
||||
|
||||
@app.post("/api/fans")
|
||||
async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
|
||||
"""Set or update the fan curve. Starts the fan control poller."""
|
||||
g_state = _get_gpu_state(gpu_index)
|
||||
|
||||
curve_data = [{"temp_c": p.temp_c, "fan_pct": p.fan_pct} for p in req.curve]
|
||||
ok, msg = validate_curve(curve_data)
|
||||
if not ok:
|
||||
raise HTTPException(status_code=400, detail=msg)
|
||||
|
||||
# Test that fan control is available on this GPU, probing with the
|
||||
# target for the *current* temperature so the fan is never briefly
|
||||
# set to an inappropriate speed.
|
||||
if curve_data:
|
||||
sample = await _run(poll, g_state["gpu"], gpu_index)
|
||||
test_temp = sample.temp_c if sample and sample.temp_c is not None else curve_data[0]["temp_c"]
|
||||
target = interpolate_fan_speed(curve_data, test_temp)
|
||||
if target is not None:
|
||||
fan_ok, fan_msg = await _run(set_fan_speed, gpu_index, target)
|
||||
if not fan_ok:
|
||||
raise HTTPException(status_code=500, detail=f"Fan control not available: {fan_msg}")
|
||||
|
||||
# Stop existing poller if running
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
|
||||
g_state["fan_curve"] = curve_data
|
||||
g_state["fan_curve_active"] = True
|
||||
g_state["fan_poller_task"] = asyncio.create_task(_fan_poller(gpu_index))
|
||||
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
@app.post("/api/fans/reset")
|
||||
async def api_fans_reset(gpu_index: int = 0):
|
||||
"""Deactivate fan curve control and restore automatic fan mode."""
|
||||
g_state = _get_gpu_state(gpu_index)
|
||||
|
||||
# Stop poller
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
g_state["fan_poller_task"] = None
|
||||
|
||||
g_state["fan_curve"] = None
|
||||
g_state["fan_curve_active"] = False
|
||||
|
||||
ok, msg = await _run(reset_fan, gpu_index)
|
||||
if not ok:
|
||||
log.warning("Fan reset warning: %s", msg)
|
||||
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
@app.post("/api/fans/speed")
|
||||
async def api_fans_speed(req: FanSpeedRequest, gpu_index: int = 0):
|
||||
"""One-shot set fan to an exact percentage (bypasses curve)."""
|
||||
_get_gpu_state(gpu_index)
|
||||
pct = max(0, min(100, req.fan_pct))
|
||||
ok, msg = await _run(set_fan_speed, gpu_index, pct)
|
||||
if not ok:
|
||||
raise HTTPException(status_code=500, detail=msg)
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
# ── Write endpoints ────────────────────────────────────────────────────────────
|
||||
|
||||
async def _reconcile_check(gpu_index: int) -> dict | None:
|
||||
|
||||
Reference in new issue
Block a user