Compare commits

..
2 Commits
Author SHA1 Message Date
Pakobbix 8956fc9d7b Merge pull request 'feat: full fan control — all fans or individual fans' (#8) from feat/multi-fan-control into main
Reviewed-on: #8
2026-09-10 13:50:55 +00:00
ARIA 6cb33187d3 feat: full fan control — all fans or individual fans
The fan curve previously only controlled fan index 0; secondary fans
stayed on driver control. The curve can now target all fans (new
default) or any individual fan(s).

Backend:
- hal/fans.py: get_num_fans() via nvmlDeviceGetNumFans; get_fan_info()
  returns per-fan speeds; set_fan_speed() accepts a fan index list
  (None = all fans; all-fans mode is lenient toward driver-locked
  fans, explicit lists are strict); reset_fan() restores all fans.
- server.py: per-GPU fan_targets state; the poller applies the curve to
  all target fans and logs write failures (once per distinct error);
  activation validates targets against the hardware (stale indices fall
  back to all fans); POST /api/fans accepts fans, POST /api/fans/speed
  accepts a fan index, GET /api/fans returns num_fans/fans/fan_targets.
- Persistence format is now {"curve": ..., "fans": ...}; legacy
  bare-curve entries migrate to "all fans" at startup.
- Profiles save/apply fan_targets alongside fan_curve.
- MonitoringSample carries per-fan speeds for live gauges.

Frontend:
- Fans tab: All / Fan 1 / Fan 2 / ... selector with live per-fan %;
  the selection is applied together with the curve.
- Live monitor: per-fan gauges with sparklines for multi-fan GPUs.
2026-09-10 15:48:28 +02:00
12 changed files with 358 additions and 53 deletions

No files matched your search

+1 -1
View File
@@ -13,7 +13,7 @@ NVCurve brings MSI Afterburner-style per-point voltage-frequency curve control t
> [!IMPORTANT]
> **Blackwell GPU memory** — This is a specialized fork with extended memory offset support (up to +3000 MHz) for Blackwell GPUs (RTX 50-series). \
> **Fan Controls** — There is an additional "Fans" tab to setup a customized fan curve. \
> **Fan Controls** — There is an additional "Fans" tab to setup a customized fan curve, controlling all fans or individual fans. \
> **Dashboard** — The default tab is an Dashboard with additional information (PCIe link speed, VBIOS information, Max Core Clock, Throttle Reason and much much more.) \
> **Authentification** — For production deplyoment, I added authentification with bcrypt hashing to allow only one or multiple people to have access. \
> Installing the pre-built PyPI package will NOT include these features. You must build from source.
+2 -2
View File
@@ -17,7 +17,7 @@ NVCurve provides two ways to interact with your GPU:
- **Per-Point Curve Editing** — Adjust the frequency offset for any individual voltage point on the V/F curve.
- **Extended Memory Offset** — Memory clock offset up to +3000 MHz (Blackwell GPUs). Standard NVCurve caps at +1000 MHz.
- **Fan Curve Control** — Custom temperature-to-fan-speed curves via NVML, adjustable through the web UI and savable in profiles.
- **Fan Curve Control** — Custom temperature-to-fan-speed curves via NVML (all fans or individual fans), adjustable through the web UI and savable in profiles.
- **Curve Flattening** — Select multiple points and flatten them to a common frequency using anchor-point targeting.
- **Live Monitoring** — Track GPU voltage, clock speed, temperature, and power draw in real time via NvAPI and NVML.
- **Profile Management** — Save, load, and switch between named profiles. Set a default profile that auto-applies on startup.
@@ -29,7 +29,7 @@ NVCurve provides two ways to interact with your GPU:
NVCurve consists of two components:
| Component | Description |
|---|---|
| --- | --- |
| **Python Backend** | Talks directly to `libnvidia-api.so` (via ctypes) and `libnvidia-ml.so` to read and write GPU hardware state. Exposes functionality through a FastAPI REST + WebSocket server. |
| **React Frontend** | Runs in the browser and communicates with the backend over HTTP and WebSockets. Handles curve visualization, point editing, live monitoring, and profile management. |
+4 -4
View File
@@ -168,9 +168,9 @@ export const api = {
/** Fan control */
fans: (gpuIndex: number) => get<FanState>("/fans", gpuIndex),
updateFans: (curve: FanPoint[], gpuIndex: number) =>
post("/fans", { curve }, gpuIndex),
updateFans: (curve: FanPoint[], gpuIndex: number, fans?: number[] | null) =>
post("/fans", { curve, fans: fans ?? null }, gpuIndex),
resetFans: (gpuIndex: number) => post("/fans/reset", undefined, gpuIndex),
setFanSpeed: (fanPct: number, gpuIndex: number) =>
post("/fans/speed", { fan_pct: fanPct }, gpuIndex),
setFanSpeed: (fanPct: number, gpuIndex: number, fan?: number | null) =>
post("/fans/speed", { fan_pct: fanPct, fan: fan ?? null }, gpuIndex),
};
+105 -6
View File
@@ -2,7 +2,7 @@ import { useState, useEffect, useRef, useCallback } from "react";
import { Check, X, RotateCcw, Plus } from "lucide-react";
import { api } from "../../api/client.js";
import { useCurveStore } from "../../store/curveStore.js";
import type { FanPoint, FanState } from "../../types.js";
import type { FanInfo, FanPoint, FanState } from "../../types.js";
import { toast } from "sonner";
import { ConfirmDialog } from "../common/ConfirmDialog.js";
@@ -44,6 +44,29 @@ function yToFan(y: number) {
return Math.round(FAN_MAX - ((y - PAD.top) / PLOT_H) * (FAN_MAX - FAN_MIN));
}
function sameTargets(
a: number[] | null | undefined,
b: number[] | null | undefined,
) {
if (a === null || a === undefined) return b === null || b === undefined;
if (b === null || b === undefined) return false;
if (a.length !== b.length) return false;
return a.every((v, i) => v === b[i]);
}
function fanLabel(targets: number[] | null): string {
return targets === null
? "all fans"
: targets.map((i) => `Fan ${i + 1}`).join(", ");
}
const chipCls = (selected: boolean) =>
`px-2 py-0.5 rounded-full border text-xs font-medium transition-colors ${
selected
? "bg-cyan-500/15 border-cyan-500/40 text-cyan-300"
: "bg-zinc-800 border-zinc-700 text-zinc-500 hover:text-zinc-300"
}`;
export function FanCurveEditor({ onChanged }: { onChanged?: () => void }) {
const { selectedGpuIndex } = useCurveStore();
const [fanState, setFanState] = useState<FanState | null>(null);
@@ -54,6 +77,9 @@ export function FanCurveEditor({ onChanged }: { onChanged?: () => void }) {
const [confirmApply, setConfirmApply] = useState(false);
const [confirmReset, setConfirmReset] = useState(false);
const [dragIdx, setDragIdx] = useState<number | null>(null);
// Fan selection: undefined = no pending change (follow server state),
// null = all fans, list = specific fan indices.
const [fanSel, setFanSel] = useState<number[] | null | undefined>(undefined);
const svgRef = useRef<SVGSVGElement>(null);
async function fetchFans() {
@@ -61,6 +87,7 @@ export function FanCurveEditor({ onChanged }: { onChanged?: () => void }) {
setLoading(true);
const data = await api.fans(selectedGpuIndex);
setFanState(data);
setFanSel(undefined);
if (data.curve && data.curve.length > 0) {
setPending(null);
}
@@ -83,14 +110,41 @@ export function FanCurveEditor({ onChanged }: { onChanged?: () => void }) {
const curveActive = fanState?.curve_active ?? false;
const isDefaults = !pending && !fanState?.curve;
const fanTargets =
fanSel === undefined ? (fanState?.fan_targets ?? null) : fanSel;
const fansChanged =
fanSel !== undefined && !sameTargets(fanSel, fanState?.fan_targets ?? null);
const fanList: FanInfo[] =
fanState?.fans && fanState.fans.length > 0
? fanState.fans
: [{ index: 0, fan_pct: null }];
function toggleFan(idx: number) {
const numFans = fanState?.num_fans ?? 1;
let list: number[];
if (fanTargets === null) {
// Start from all fans, then drop the toggled one
list = Array.from({ length: numFans }, (_, i) => i).filter(
(i) => i !== idx,
);
} else {
list = fanTargets.includes(idx)
? fanTargets.filter((i) => i !== idx)
: [...fanTargets, idx].sort((a, b) => a - b);
}
if (list.length === 0) return; // keep at least one fan selected
setFanSel(list.length === numFans ? null : list);
}
async function handleApply() {
const curveToApply = pending ?? fanState?.curve ?? defaultCurve();
if (curveToApply.length < 2) return;
setBusy(true);
setError(null);
try {
await api.updateFans(curveToApply, selectedGpuIndex);
await api.updateFans(curveToApply, selectedGpuIndex, fanTargets);
setPending(null);
setFanSel(undefined);
setConfirmApply(false);
await fetchFans();
onChanged?.();
@@ -109,6 +163,7 @@ export function FanCurveEditor({ onChanged }: { onChanged?: () => void }) {
try {
await api.resetFans(selectedGpuIndex);
setPending(null);
setFanSel(undefined);
setConfirmReset(false);
await fetchFans();
onChanged?.();
@@ -251,9 +306,10 @@ export function FanCurveEditor({ onChanged }: { onChanged?: () => void }) {
<button
onClick={() => {
setPending(null);
setFanSel(undefined);
setError(null);
}}
disabled={!hasPending || busy}
disabled={(!hasPending && !fansChanged) || busy}
className="flex items-center gap-1.5 px-2 py-1 rounded bg-zinc-800 hover:bg-zinc-700 text-zinc-300 text-xs transition-colors disabled:opacity-40 disabled:cursor-not-allowed"
>
<X size={12} />
@@ -270,6 +326,48 @@ export function FanCurveEditor({ onChanged }: { onChanged?: () => void }) {
</div>
</div>
{/* Fan selection */}
<div className="px-4 py-2 border-b border-zinc-800/60 flex items-center gap-2 flex-wrap">
<span className="text-xs text-zinc-500 uppercase tracking-wider font-semibold">
Fans
</span>
{fanState?.num_fans === 0 ? (
<span className="text-xs text-zinc-600">
No fans available on this GPU
</span>
) : (
<>
<button
onClick={() => setFanSel(null)}
className={chipCls(fanTargets === null)}
title="Control all fans"
>
All
</button>
{fanList.map((f) => (
<button
key={f.index}
onClick={() => toggleFan(f.index)}
className={chipCls(
fanTargets === null || fanTargets.includes(f.index),
)}
title={`Control Fan ${f.index + 1} with the curve`}
>
Fan {f.index + 1}
<span className="ml-1.5 font-mono text-[10px] opacity-80">
{f.fan_pct !== null ? `${Math.round(f.fan_pct)}%` : "—"}
</span>
</button>
))}
</>
)}
{fansChanged && (
<span className="inline-flex items-center gap-1 px-2 py-0.5 rounded-full bg-cyan-500/15 border border-cyan-500/30 text-cyan-400 text-xs">
Fans: {fanLabel(fanTargets)}
</span>
)}
</div>
{/* Error banner */}
{error && (
<div className="px-3 py-1.5 bg-red-900/40 border-b border-red-700 text-red-300 text-xs flex items-center justify-between">
@@ -557,13 +655,14 @@ export function FanCurveEditor({ onChanged }: { onChanged?: () => void }) {
{/* Info */}
<div className="px-4 pb-3 text-[10px] text-zinc-600">
{curveActive
? "Fan curve is active. Server adjusts fan speed based on GPU temperature."
? `Fan curve is active — controlling ${fanLabel(fanTargets)}. Server adjusts fan speed based on GPU temperature.`
: isDefaults
? "These are default values. Click Apply to enable curve control, or edit points first."
: "Apply a curve to enable automatic fan control based on temperature."}
<span className="block mt-1 text-zinc-700">
Drag points to adjust · click the chart to add a point · click the ✕
(chart or table) to remove one.
(chart or table) to remove one · pick which fans the curve controls
above.
</span>
</div>
</div>
@@ -571,7 +670,7 @@ export function FanCurveEditor({ onChanged }: { onChanged?: () => void }) {
{confirmApply && (
<ConfirmDialog
message="Apply fan curve?"
detail="Fan control will switch to curve-based mode. The server will adjust fan speed based on GPU temperature."
detail={`Fan control will switch to curve-based mode for ${fanLabel(fanTargets)}. The server will adjust fan speed based on GPU temperature.`}
confirmLabel="Apply"
onConfirm={handleApply}
onCancel={() => setConfirmApply(false)}
@@ -47,6 +47,8 @@ export function FanMonitor({
}: Props) {
const fanHistory = pluck(history, "fan_pct");
const tempHistory = pluck(history, "temp_c");
const fans = monitor?.fans ?? null;
const multiFan = fans !== null && fans.length > 1;
const currentTemp = monitor?.temp_c ?? null;
const targetFan = computeTargetFan(fanCurve, currentTemp);
@@ -68,6 +70,18 @@ export function FanMonitor({
)}
</div>
<div className="flex flex-col gap-2 mt-1">
{multiFan ? (
fans!.map((_, i) => (
<GaugeCard
key={i}
label={`Fan ${i + 1}`}
value={fmt.pct(fans![i])}
history={history.map((s) => s.fans?.[i] ?? s.fan_pct ?? 0)}
color="#fb923c"
max={100}
/>
))
) : (
<GaugeCard
label="Fan Speed"
value={fmt.pct(monitor?.fan_pct)}
@@ -75,6 +89,7 @@ export function FanMonitor({
color="#fb923c"
max={100}
/>
)}
<GaugeCard
label="GPU Temp"
value={fmt.celsius(monitor?.temp_c)}
+11
View File
@@ -25,6 +25,7 @@ export interface MonitoringSample {
temp_c: number | null;
power_w: number | null;
fan_pct: number | null;
fans: (number | null)[] | null;
pstate: number | null;
pstate_label: string | null;
mem_used_bytes: number | null;
@@ -111,13 +112,22 @@ export interface FanPoint {
fan_pct: number;
}
export interface FanInfo {
index: number;
fan_pct: number | null;
}
export interface FanState {
fan_pct: number | null;
fans: FanInfo[] | null;
num_fans: number | null;
fan_mode: "auto" | "curve" | null;
min_fan_pct: number | null;
max_fan_pct: number | null;
curve: FanPoint[] | null;
curve_active: boolean;
// Fan indices controlled by the active curve; null = all fans.
fan_targets: number[] | null;
}
export interface ProfileData {
@@ -127,4 +137,5 @@ export interface ProfileData {
mem_offset_mhz: number | null;
power_limit_w: number | null;
fan_curve: FanPoint[] | null;
fan_targets: number[] | null;
}
+4 -2
View File
@@ -34,8 +34,10 @@ class Config:
# curve applied via the UI survives server restarts (fan control itself is
# volatile — the driver reverts to automatic mode on reboot).
# Key = stable GPU identifier (same as auto_load_profiles).
# Value = list of {"temp_c": int, "fan_pct": int} sorted by temp_c.
fan_curves: dict[str, list] = field(default_factory=dict)
# Value = {"curve": [{"temp_c": int, "fan_pct": int}, ...] sorted by temp_c,
# "fans": [fan indices] | None (None = all fans)}.
# Legacy entries (bare curve list) are migrated at load time.
fan_curves: dict[str, object] = field(default_factory=dict)
# Module-level default config instance.
+116 -22
View File
@@ -1,10 +1,15 @@
"""Hardware Abstraction Layer for Fan Control.
Uses NVML (via pynvml) for all operations:
- nvmlDeviceGetNumFans : number of fans on the device
- nvmlDeviceGetFanSpeed_v2 : read current fan speed % for a fan index
- nvmlDeviceSetFanSpeed_v2 : set fan speed % for a fan index
- nvmlDeviceGetMinMaxFanSpeed: get min/max fan speed constraints
- nvmlDeviceSetDefaultFanSpeed_v2 : restore automatic control for a fan index
- nvmlDeviceGetMinMaxFanSpeed : get min/max fan speed constraints
- nvmlDeviceGetTemperature : read GPU temp for curve interpolation
Fans are addressed by 0-based index. Passing ``fans=None`` to the set/reset
helpers means "all fans on the device".
"""
import ctypes
@@ -24,9 +29,6 @@ pynvml: Any = _pynvml_import
log = logging.getLogger("nvcurve.hal.fans")
# We use fan index 0 (first/primary fan) for all operations.
_FAN_INDEX = 0
def _get_handle(gpu_index: int):
"""Return an NVML device handle."""
@@ -35,13 +37,42 @@ def _get_handle(gpu_index: int):
return pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
def _num_fans(handle) -> int:
"""Return the number of fans on the device (>= 1 on query failure)."""
try:
return max(0, int(pynvml.nvmlDeviceGetNumFans(handle)))
except pynvml.NVMLError:
# GetNumFans unsupported: assume at least the primary fan exists.
return 1
def get_num_fans(gpu_index: int = 0) -> int:
"""Return the number of fans on the GPU (0 if NVML is unavailable)."""
if not _NVML_AVAILABLE:
return 0
try:
return _num_fans(_get_handle(gpu_index))
except pynvml.NVMLError as exc:
log.warning("get_num_fans: %s", exc)
return 0
def get_fan_info(gpu_index: int = 0) -> dict:
"""Return current fan state: fan_pct, fan_mode, min_fan_pct, max_fan_pct.
"""Return current fan state for all fans.
Returns a dict with:
fan_pct : current speed % of fan 0 (legacy, None on failure)
fans : [{"index": i, "fan_pct": pct | None}, ...] per fan
num_fans : number of fans on the device
fan_mode : None (the server derives "auto"/"curve")
min_fan_pct / max_fan_pct : device-wide speed constraints
Returns None values on failure.
"""
out: dict[str, float | None] = {
out: dict[str, Any] = {
"fan_pct": None,
"fans": [],
"num_fans": 0,
"fan_mode": None,
"min_fan_pct": None,
"max_fan_pct": None,
@@ -50,16 +81,26 @@ def get_fan_info(gpu_index: int = 0) -> dict:
return out
try:
handle = _get_handle(gpu_index)
except pynvml.NVMLError as exc:
log.warning("get_fan_info: %s", exc)
return out
# Get current fan speed using v2 API (fan index 0)
out["num_fans"] = _num_fans(handle)
# Per-fan speeds via the v2 API; fan 0 falls back to the legacy v1 API.
for i in range(out["num_fans"]):
pct: float | None = None
try:
out["fan_pct"] = float(pynvml.nvmlDeviceGetFanSpeed_v2(handle, _FAN_INDEX))
pct = float(pynvml.nvmlDeviceGetFanSpeed_v2(handle, i))
except pynvml.NVMLError:
# Fallback to legacy v1 API
if i == 0:
try:
out["fan_pct"] = float(pynvml.nvmlDeviceGetFanSpeed(handle))
pct = float(pynvml.nvmlDeviceGetFanSpeed(handle))
except pynvml.NVMLError:
pass
pct = None
out["fans"].append({"index": i, "fan_pct": pct})
out["fan_pct"] = out["fans"][0]["fan_pct"] if out["fans"] else None
# Get min/max fan speed constraints
try:
@@ -72,14 +113,22 @@ def get_fan_info(gpu_index: int = 0) -> dict:
out["min_fan_pct"] = 0
out["max_fan_pct"] = 100
except pynvml.NVMLError as exc:
log.warning("get_fan_info: %s", exc)
return out
def set_fan_speed(gpu_index: int, pct: int) -> tuple[bool, str]:
"""Set fan speed to a percentage (0-100) on the primary fan."""
def set_fan_speed(
gpu_index: int,
pct: int,
fans: list[int] | None = None,
) -> tuple[bool, str]:
"""Set fan speed to a percentage (0-100).
fans=None targets every fan on the device; fans=[0, 1] targets the
listed fan indices. In all-fans mode, fans the driver does not allow
manual control of (e.g. driver-mirrored secondary fans) are skipped
with a warning instead of failing the whole operation; explicit fan
lists are strict and fail if any selected fan cannot be set.
"""
try:
pct = max(0, min(100, int(pct)))
except (TypeError, ValueError):
@@ -88,7 +137,47 @@ def set_fan_speed(gpu_index: int, pct: int) -> tuple[bool, str]:
return False, "NVML not available"
try:
handle = _get_handle(gpu_index)
pynvml.nvmlDeviceSetFanSpeed_v2(handle, _FAN_INDEX, pct)
num_fans = _num_fans(handle)
if fans is None:
targets = list(range(num_fans))
strict = False
else:
targets: list[int] = []
for f in fans:
try:
f = int(f)
except (TypeError, ValueError):
return False, f"Invalid fan index: {f!r}"
if f < 0 or f >= num_fans:
return False, f"Fan index {f} out of range (0-{num_fans - 1})"
targets.append(f)
strict = True
if not targets:
return False, "No fans selected"
if not targets:
return False, "No fans available on this GPU"
skipped: list[str] = []
for i in targets:
try:
pynvml.nvmlDeviceSetFanSpeed_v2(handle, i, pct)
except pynvml.NVMLError as exc:
if strict:
log.warning(
"set_fan_speed(%d, fan %d, %d): %s", gpu_index, i, pct, exc
)
return False, str(exc)
# All-fans mode: secondary fans may be driver-controlled and
# reject manual writes; skip them and report in the message.
log.debug("set_fan_speed: fan %d not settable: %s", i, exc)
skipped.append(f"fan {i + 1}")
if skipped:
return (
True,
f"OK ({len(skipped)} fan(s) not manually controllable: {', '.join(skipped)})",
)
return True, "OK"
except pynvml.NVMLError as exc:
log.warning("set_fan_speed(%d, %d): %s", gpu_index, pct, exc)
@@ -96,10 +185,11 @@ def set_fan_speed(gpu_index: int, pct: int) -> tuple[bool, str]:
def reset_fan(gpu_index: int = 0) -> tuple[bool, str]:
"""Restore automatic fan control.
"""Restore automatic fan control for all fans.
Tries nvidia-smi --fan=default first (most reliable), then falls back to
NVML nvmlDeviceSetDefaultFanSpeed_v2.
Tries nvidia-smi --fan=default first (most reliable, resets all fans on
the device), then falls back to NVML nvmlDeviceSetDefaultFanSpeed_v2
per fan index.
"""
if not _NVML_AVAILABLE:
return False, "NVML not available"
@@ -124,10 +214,14 @@ def reset_fan(gpu_index: int = 0) -> tuple[bool, str]:
except Exception as exc:
log.debug("nvidia-smi -fan default error: %s", exc)
# Fallback: use NVML to reset to default fan speed
# Fallback: use NVML to reset every fan to default speed
try:
handle = _get_handle(gpu_index)
pynvml.nvmlDeviceSetDefaultFanSpeed_v2(handle, _FAN_INDEX)
for i in range(_num_fans(handle)):
try:
pynvml.nvmlDeviceSetDefaultFanSpeed_v2(handle, i)
except pynvml.NVMLError as exc:
log.debug("reset_fan: fan %d: %s", i, exc)
return True, "OK"
except pynvml.NVMLError as exc:
return False, f"Failed to reset fan: {exc}"
+14
View File
@@ -162,6 +162,20 @@ def _nvml_read(gpu_index: int) -> dict:
with contextlib.suppress(_pynvml.NVMLError):
out["fan_pct"] = float(_pynvml.nvmlDeviceGetFanSpeed(handle))
# Per-fan speeds via the v2 API (fan_pct above stays fan 0 for legacy clients).
try:
num_fans = int(_pynvml.nvmlDeviceGetNumFans(handle))
fan_list: list[float | None] = []
for i in range(num_fans):
try:
fan_list.append(float(_pynvml.nvmlDeviceGetFanSpeed_v2(handle, i)))
except _pynvml.NVMLError:
fan_list.append(None)
if fan_list:
out["fans"] = fan_list
except _pynvml.NVMLError:
pass
with contextlib.suppress(_pynvml.NVMLError):
out["throttle_reasons"] = int(
_pynvml.nvmlDeviceGetCurrentClocksThrottleReasons(handle)
+1
View File
@@ -61,6 +61,7 @@ class MonitoringSample:
pcie_link_width: int | None = None # Current PCIe link width (x1..x16)
pcie_link_generation: int | None = None # Current PCIe link generation (1..5)
mem_temp_c: float | None = None # VRAM temperature (if the GPU exposes it)
fans: list[float | None] | None = None # Per-fan speed % (index-aligned)
@dataclass
+2
View File
@@ -17,6 +17,8 @@ class ProfileData:
mem_offset_mhz: int | None = None
power_limit_w: int | None = None
fan_curve: list[dict[str, int]] | None = None
# Fan indices controlled by fan_curve (0-based); None = all fans.
fan_targets: list[int] | None = None
def save_profile(profile_dir: str, data: ProfileData) -> str:
+83 -16
View File
@@ -24,6 +24,7 @@ from .config import Config, default_config
from .hal.dashboard import get_dashboard_info
from .hal.fans import (
get_fan_info,
get_num_fans,
get_temp,
interpolate_fan_speed,
reset_fan,
@@ -149,6 +150,7 @@ def _sample_dict(s) -> dict:
"temp_c": s.temp_c,
"power_w": s.power_w,
"fan_pct": s.fan_pct,
"fans": s.fans,
"pstate": s.pstate,
"pstate_label": f"P{s.pstate}" if s.pstate is not None else None,
"mem_used_bytes": s.mem_used_bytes,
@@ -219,7 +221,8 @@ async def _monitor_poller(gpu_index: int) -> None:
async def _fan_poller(gpu_index: int) -> None:
"""Continuously read GPU temp, interpolate fan speed from active curve, and apply."""
"""Continuously read GPU temp, interpolate fan speed from active curve, and apply it to the target fans."""
last_write_error: str | None = None
while True:
try:
g_state = _state["gpus"].get(gpu_index)
@@ -229,7 +232,25 @@ async def _fan_poller(gpu_index: int) -> None:
curve = g_state["fan_curve"]
target = interpolate_fan_speed(curve, temp)
if target is not None:
await _run(set_fan_speed, gpu_index, target)
ok, msg = await _run(
set_fan_speed,
gpu_index,
target,
g_state.get("fan_targets"),
)
if not ok:
# Log once per distinct failure so a stuck target
# list doesn't spam a warning every 2 s tick.
if msg != last_write_error:
log.warning(
"Fan write failed for GPU %d (targets=%s): %s",
gpu_index,
g_state.get("fan_targets"),
msg,
)
last_write_error = msg
else:
last_write_error = None
except asyncio.CancelledError:
return
except Exception as exc:
@@ -237,15 +258,34 @@ async def _fan_poller(gpu_index: int) -> None:
await asyncio.sleep(2.0)
async def _activate_fan_curve(gpu_index: int, curve: list) -> None:
async def _activate_fan_curve(
gpu_index: int, curve: list, fans: list[int] | None = None
) -> None:
"""Set the active fan curve, (re)start the poller, and persist it.
Persistence (config.json) is what makes the curve survive server restarts:
fan control is volatile, so the driver reverts to automatic mode on reboot
and the saved curve is re-applied at the next server start.
fans=None targets all fans on the device; a list targets the given
fan indices. Persistence (config.json) is what makes the curve survive
server restarts: fan control is volatile, so the driver reverts to
automatic mode on reboot and the saved curve is re-applied at the next
server start.
"""
g_state = _get_gpu_state(gpu_index)
# Validate explicit fan targets against the hardware. Stale indices
# (e.g. a profile saved on a 2-fan GPU applied to a 1-fan GPU, or a
# persisted entry restored after a hardware change) would otherwise
# make the poller fail silently on every tick.
if fans is not None:
num_fans = await _run(get_num_fans, gpu_index)
if not fans or (num_fans > 0 and any(f < 0 or f >= num_fans for f in fans)):
log.warning(
"Fan targets %s invalid for GPU %d (%d fan(s)); falling back to all fans",
fans,
gpu_index,
num_fans,
)
fans = None
# Stop existing poller if running
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
@@ -254,10 +294,11 @@ async def _activate_fan_curve(gpu_index: int, curve: list) -> None:
g_state["fan_curve"] = curve
g_state["fan_curve_active"] = True
g_state["fan_targets"] = fans
g_state["fan_poller_task"] = asyncio.create_task(_fan_poller(gpu_index))
cfg: Config = _state["config"]
cfg.fan_curves[_gpu_stable_key(gpu_index)] = curve
cfg.fan_curves[_gpu_stable_key(gpu_index)] = {"curve": curve, "fans": fans}
_persist_fan_curves(cfg.fan_curves)
@@ -276,6 +317,7 @@ async def _deactivate_fan_curve(gpu_index: int, reset_hardware: bool = True) ->
g_state["fan_curve"] = None
g_state["fan_curve_active"] = False
g_state["fan_targets"] = None
if reset_hardware:
ok, msg = await _run(reset_fan, gpu_index)
@@ -389,7 +431,15 @@ async def lifespan(app: FastAPI):
if g_state.get("fan_curve_active"):
continue # already activated by the auto-load profile path
key = _gpu_stable_key(gpu_index)
curve = cfg.fan_curves.get(key)
entry = cfg.fan_curves.get(key)
if not entry:
continue
# Migrate the legacy format (bare curve list) to the current
# {"curve": ..., "fans": ...} shape; legacy entries targeted all fans.
if isinstance(entry, list):
entry = {"curve": entry, "fans": None}
curve = entry.get("curve") if isinstance(entry, dict) else None
fans = entry.get("fans") if isinstance(entry, dict) else None
if not curve:
continue
ok, msg = validate_curve(curve)
@@ -398,7 +448,7 @@ async def lifespan(app: FastAPI):
continue
log.info("Restoring persisted fan curve on GPU %d (%s)", gpu_index, key)
try:
await _activate_fan_curve(gpu_index, curve)
await _activate_fan_curve(gpu_index, curve, fans)
except Exception as exc:
log.warning(
"Failed to restore persisted fan curve on GPU %d: %s",
@@ -543,10 +593,14 @@ class FanCurvePoint(BaseModel):
class FanCurveRequest(BaseModel):
curve: list[FanCurvePoint]
# Fan indices to control (0-based); None = all fans on the device.
fans: list[int] | None = None
class FanSpeedRequest(BaseModel):
fan_pct: int
# Specific fan index to set (0-based); None = all fans on the device.
fan: int | None = None
class LoginRequest(BaseModel):
@@ -873,6 +927,9 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
mem_offset_mhz=mem_offset_mhz,
power_limit_w=power_limit_w,
fan_curve=g_state.get("fan_curve") if g_state.get("fan_curve_active") else None,
fan_targets=g_state.get("fan_targets")
if g_state.get("fan_curve_active")
else None,
)
filepath = await _run(save_profile, cfg.profile_dir, data)
g_state["active_profile"] = req.name
@@ -1024,7 +1081,7 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
if not ok:
errs.append(f"Fan curve: {msg}")
else:
await _activate_fan_curve(gpu_index, profile.fan_curve)
await _activate_fan_curve(gpu_index, profile.fan_curve, profile.fan_targets)
elif g_state.get("fan_curve_active"):
await _deactivate_fan_curve(gpu_index, reset_hardware=True)
@@ -1226,7 +1283,7 @@ async def api_limits_reset(gpu_index: int = 0):
@app.get("/api/fans")
async def api_fans(gpu_index: int = 0):
"""Current fan state: fan %, curve, and whether curve control is active."""
"""Current fan state: per-fan %, curve, and whether curve control is active."""
_get_gpu_state(gpu_index)
g_state = _state["gpus"][gpu_index]
info = await _run(get_fan_info, gpu_index)
@@ -1236,12 +1293,16 @@ async def api_fans(gpu_index: int = 0):
"fan_mode": "curve" if curve_active else "auto",
"curve": g_state.get("fan_curve"),
"curve_active": curve_active,
"fan_targets": g_state.get("fan_targets"),
}
@app.post("/api/fans")
async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
"""Set or update the fan curve. Starts the fan control poller."""
"""Set or update the fan curve. Starts the fan control poller.
req.fans selects which fans the curve drives (None = all fans).
"""
g_state = _get_gpu_state(gpu_index)
curve_data = [{"temp_c": p.temp_c, "fan_pct": p.fan_pct} for p in req.curve]
@@ -1249,6 +1310,8 @@ async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
if not ok:
raise HTTPException(status_code=400, detail=msg)
fans = sorted(set(req.fans)) if req.fans is not None else None
# Test that fan control is available on this GPU, probing with the
# target for the *current* temperature so the fan is never briefly
# set to an inappropriate speed.
@@ -1261,14 +1324,14 @@ async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
)
target = interpolate_fan_speed(curve_data, test_temp)
if target is not None:
fan_ok, fan_msg = await _run(set_fan_speed, gpu_index, target)
fan_ok, fan_msg = await _run(set_fan_speed, gpu_index, target, fans)
if not fan_ok:
raise HTTPException(
status_code=500, detail=f"Fan control not available: {fan_msg}"
)
# Activate the curve (starts the poller) and persist it so it survives restarts.
await _activate_fan_curve(gpu_index, curve_data)
await _activate_fan_curve(gpu_index, curve_data, fans)
return {"ok": True}
@@ -1287,10 +1350,14 @@ async def api_fans_reset(gpu_index: int = 0):
@app.post("/api/fans/speed")
async def api_fans_speed(req: FanSpeedRequest, gpu_index: int = 0):
"""One-shot set fan to an exact percentage (bypasses curve)."""
"""One-shot set fan(s) to an exact percentage (bypasses curve).
req.fan selects a single fan index; None sets all fans.
"""
_get_gpu_state(gpu_index)
pct = max(0, min(100, req.fan_pct))
ok, msg = await _run(set_fan_speed, gpu_index, pct)
fans = [req.fan] if req.fan is not None else None
ok, msg = await _run(set_fan_speed, gpu_index, pct, fans)
if not ok:
raise HTTPException(status_code=500, detail=msg)
return {"ok": True}