feat: full fan control — all fans or individual fans

The fan curve previously only controlled fan index 0; secondary fans
stayed on driver control. The curve can now target all fans (new
default) or any individual fan(s).

Backend:
- hal/fans.py: get_num_fans() via nvmlDeviceGetNumFans; get_fan_info()
  returns per-fan speeds; set_fan_speed() accepts a fan index list
  (None = all fans; all-fans mode is lenient toward driver-locked
  fans, explicit lists are strict); reset_fan() restores all fans.
- server.py: per-GPU fan_targets state; the poller applies the curve to
  all target fans and logs write failures (once per distinct error);
  activation validates targets against the hardware (stale indices fall
  back to all fans); POST /api/fans accepts fans, POST /api/fans/speed
  accepts a fan index, GET /api/fans returns num_fans/fans/fan_targets.
- Persistence format is now {"curve": ..., "fans": ...}; legacy
  bare-curve entries migrate to "all fans" at startup.
- Profiles save/apply fan_targets alongside fan_curve.
- MonitoringSample carries per-fan speeds for live gauges.

Frontend:
- Fans tab: All / Fan 1 / Fan 2 / ... selector with live per-fan %;
  the selection is applied together with the curve.
- Live monitor: per-fan gauges with sparklines for multi-fan GPUs.
This commit is contained in:
ARIA committed 2026-09-10 15:48:28 +02:00
1 parent 34a9bc6d6e
commit 6cb33187d3
12 files changed
+382 -77

No files matched your search

+83 -16
View File
@@ -24,6 +24,7 @@ from .config import Config, default_config
from .hal.dashboard import get_dashboard_info
from .hal.fans import (
get_fan_info,
get_num_fans,
get_temp,
interpolate_fan_speed,
reset_fan,
@@ -149,6 +150,7 @@ def _sample_dict(s) -> dict:
"temp_c": s.temp_c,
"power_w": s.power_w,
"fan_pct": s.fan_pct,
"fans": s.fans,
"pstate": s.pstate,
"pstate_label": f"P{s.pstate}" if s.pstate is not None else None,
"mem_used_bytes": s.mem_used_bytes,
@@ -219,7 +221,8 @@ async def _monitor_poller(gpu_index: int) -> None:
async def _fan_poller(gpu_index: int) -> None:
"""Continuously read GPU temp, interpolate fan speed from active curve, and apply."""
"""Continuously read GPU temp, interpolate fan speed from active curve, and apply it to the target fans."""
last_write_error: str | None = None
while True:
try:
g_state = _state["gpus"].get(gpu_index)
@@ -229,7 +232,25 @@ async def _fan_poller(gpu_index: int) -> None:
curve = g_state["fan_curve"]
target = interpolate_fan_speed(curve, temp)
if target is not None:
await _run(set_fan_speed, gpu_index, target)
ok, msg = await _run(
set_fan_speed,
gpu_index,
target,
g_state.get("fan_targets"),
)
if not ok:
# Log once per distinct failure so a stuck target
# list doesn't spam a warning every 2 s tick.
if msg != last_write_error:
log.warning(
"Fan write failed for GPU %d (targets=%s): %s",
gpu_index,
g_state.get("fan_targets"),
msg,
)
last_write_error = msg
else:
last_write_error = None
except asyncio.CancelledError:
return
except Exception as exc:
@@ -237,15 +258,34 @@ async def _fan_poller(gpu_index: int) -> None:
await asyncio.sleep(2.0)
async def _activate_fan_curve(gpu_index: int, curve: list) -> None:
async def _activate_fan_curve(
gpu_index: int, curve: list, fans: list[int] | None = None
) -> None:
"""Set the active fan curve, (re)start the poller, and persist it.
Persistence (config.json) is what makes the curve survive server restarts:
fan control is volatile, so the driver reverts to automatic mode on reboot
and the saved curve is re-applied at the next server start.
fans=None targets all fans on the device; a list targets the given
fan indices. Persistence (config.json) is what makes the curve survive
server restarts: fan control is volatile, so the driver reverts to
automatic mode on reboot and the saved curve is re-applied at the next
server start.
"""
g_state = _get_gpu_state(gpu_index)
# Validate explicit fan targets against the hardware. Stale indices
# (e.g. a profile saved on a 2-fan GPU applied to a 1-fan GPU, or a
# persisted entry restored after a hardware change) would otherwise
# make the poller fail silently on every tick.
if fans is not None:
num_fans = await _run(get_num_fans, gpu_index)
if not fans or (num_fans > 0 and any(f < 0 or f >= num_fans for f in fans)):
log.warning(
"Fan targets %s invalid for GPU %d (%d fan(s)); falling back to all fans",
fans,
gpu_index,
num_fans,
)
fans = None
# Stop existing poller if running
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
@@ -254,10 +294,11 @@ async def _activate_fan_curve(gpu_index: int, curve: list) -> None:
g_state["fan_curve"] = curve
g_state["fan_curve_active"] = True
g_state["fan_targets"] = fans
g_state["fan_poller_task"] = asyncio.create_task(_fan_poller(gpu_index))
cfg: Config = _state["config"]
cfg.fan_curves[_gpu_stable_key(gpu_index)] = curve
cfg.fan_curves[_gpu_stable_key(gpu_index)] = {"curve": curve, "fans": fans}
_persist_fan_curves(cfg.fan_curves)
@@ -276,6 +317,7 @@ async def _deactivate_fan_curve(gpu_index: int, reset_hardware: bool = True) ->
g_state["fan_curve"] = None
g_state["fan_curve_active"] = False
g_state["fan_targets"] = None
if reset_hardware:
ok, msg = await _run(reset_fan, gpu_index)
@@ -389,7 +431,15 @@ async def lifespan(app: FastAPI):
if g_state.get("fan_curve_active"):
continue # already activated by the auto-load profile path
key = _gpu_stable_key(gpu_index)
curve = cfg.fan_curves.get(key)
entry = cfg.fan_curves.get(key)
if not entry:
continue
# Migrate the legacy format (bare curve list) to the current
# {"curve": ..., "fans": ...} shape; legacy entries targeted all fans.
if isinstance(entry, list):
entry = {"curve": entry, "fans": None}
curve = entry.get("curve") if isinstance(entry, dict) else None
fans = entry.get("fans") if isinstance(entry, dict) else None
if not curve:
continue
ok, msg = validate_curve(curve)
@@ -398,7 +448,7 @@ async def lifespan(app: FastAPI):
continue
log.info("Restoring persisted fan curve on GPU %d (%s)", gpu_index, key)
try:
await _activate_fan_curve(gpu_index, curve)
await _activate_fan_curve(gpu_index, curve, fans)
except Exception as exc:
log.warning(
"Failed to restore persisted fan curve on GPU %d: %s",
@@ -543,10 +593,14 @@ class FanCurvePoint(BaseModel):
class FanCurveRequest(BaseModel):
curve: list[FanCurvePoint]
# Fan indices to control (0-based); None = all fans on the device.
fans: list[int] | None = None
class FanSpeedRequest(BaseModel):
fan_pct: int
# Specific fan index to set (0-based); None = all fans on the device.
fan: int | None = None
class LoginRequest(BaseModel):
@@ -873,6 +927,9 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
mem_offset_mhz=mem_offset_mhz,
power_limit_w=power_limit_w,
fan_curve=g_state.get("fan_curve") if g_state.get("fan_curve_active") else None,
fan_targets=g_state.get("fan_targets")
if g_state.get("fan_curve_active")
else None,
)
filepath = await _run(save_profile, cfg.profile_dir, data)
g_state["active_profile"] = req.name
@@ -1024,7 +1081,7 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
if not ok:
errs.append(f"Fan curve: {msg}")
else:
await _activate_fan_curve(gpu_index, profile.fan_curve)
await _activate_fan_curve(gpu_index, profile.fan_curve, profile.fan_targets)
elif g_state.get("fan_curve_active"):
await _deactivate_fan_curve(gpu_index, reset_hardware=True)
@@ -1226,7 +1283,7 @@ async def api_limits_reset(gpu_index: int = 0):
@app.get("/api/fans")
async def api_fans(gpu_index: int = 0):
"""Current fan state: fan %, curve, and whether curve control is active."""
"""Current fan state: per-fan %, curve, and whether curve control is active."""
_get_gpu_state(gpu_index)
g_state = _state["gpus"][gpu_index]
info = await _run(get_fan_info, gpu_index)
@@ -1236,12 +1293,16 @@ async def api_fans(gpu_index: int = 0):
"fan_mode": "curve" if curve_active else "auto",
"curve": g_state.get("fan_curve"),
"curve_active": curve_active,
"fan_targets": g_state.get("fan_targets"),
}
@app.post("/api/fans")
async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
"""Set or update the fan curve. Starts the fan control poller."""
"""Set or update the fan curve. Starts the fan control poller.
req.fans selects which fans the curve drives (None = all fans).
"""
g_state = _get_gpu_state(gpu_index)
curve_data = [{"temp_c": p.temp_c, "fan_pct": p.fan_pct} for p in req.curve]
@@ -1249,6 +1310,8 @@ async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
if not ok:
raise HTTPException(status_code=400, detail=msg)
fans = sorted(set(req.fans)) if req.fans is not None else None
# Test that fan control is available on this GPU, probing with the
# target for the *current* temperature so the fan is never briefly
# set to an inappropriate speed.
@@ -1261,14 +1324,14 @@ async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
)
target = interpolate_fan_speed(curve_data, test_temp)
if target is not None:
fan_ok, fan_msg = await _run(set_fan_speed, gpu_index, target)
fan_ok, fan_msg = await _run(set_fan_speed, gpu_index, target, fans)
if not fan_ok:
raise HTTPException(
status_code=500, detail=f"Fan control not available: {fan_msg}"
)
# Activate the curve (starts the poller) and persist it so it survives restarts.
await _activate_fan_curve(gpu_index, curve_data)
await _activate_fan_curve(gpu_index, curve_data, fans)
return {"ok": True}
@@ -1287,10 +1350,14 @@ async def api_fans_reset(gpu_index: int = 0):
@app.post("/api/fans/speed")
async def api_fans_speed(req: FanSpeedRequest, gpu_index: int = 0):
"""One-shot set fan to an exact percentage (bypasses curve)."""
"""One-shot set fan(s) to an exact percentage (bypasses curve).
req.fan selects a single fan index; None sets all fans.
"""
_get_gpu_state(gpu_index)
pct = max(0, min(100, req.fan_pct))
ok, msg = await _run(set_fan_speed, gpu_index, pct)
fans = [req.fan] if req.fan is not None else None
ok, msg = await _run(set_fan_speed, gpu_index, pct, fans)
if not ok:
raise HTTPException(status_code=500, detail=msg)
return {"ok": True}