feat: add Dashboard tab with live GPU overview #5

Merged
Pakobbix merged 1 commits from feat/dashboard-tab into main 2026-09-02 14:57:54 +00:00
11 changed files with 902 additions and 67 deletions

No files matched your search

+20 -4
View File
@@ -1,8 +1,10 @@
import { useGpu } from "./hooks/useGpu"; import { useGpu } from "./hooks/useGpu";
import { useCurve } from "./hooks/useCurve"; import { useCurve } from "./hooks/useCurve";
import { useMonitor } from "./hooks/useMonitor"; import { useMonitor } from "./hooks/useMonitor";
import { useDashboard } from "./hooks/useDashboard";
import { StatusBar } from "./components/Monitor/StatusBar"; import { StatusBar } from "./components/Monitor/StatusBar";
import { LiveMonitor } from "./components/Monitor/LiveMonitor"; import { LiveMonitor } from "./components/Monitor/LiveMonitor";
import { Dashboard } from "./components/Dashboard/Dashboard";
import { CurveEditor } from "./components/CurveEditor/CurveEditor"; import { CurveEditor } from "./components/CurveEditor/CurveEditor";
import { PointTable } from "./components/PointTable/PointTable"; import { PointTable } from "./components/PointTable/PointTable";
import { PerformancePanel } from "./components/Limits/PerformancePanel"; import { PerformancePanel } from "./components/Limits/PerformancePanel";
@@ -89,12 +91,13 @@ function MainApp({
const gpuInfo = useGpu(); const gpuInfo = useGpu();
const { curve, wsStatus: curveWsStatus } = useCurve(); const { curve, wsStatus: curveWsStatus } = useCurve();
const { monitor, monitorHistory, wsStatus: monitorWsStatus } = useMonitor(); const { monitor, monitorHistory, wsStatus: monitorWsStatus } = useMonitor();
const { dashboard, loading: dashboardLoading } = useDashboard();
const { setCurve, activeProfile, setActiveProfile, selectedGpuIndex } = const { setCurve, activeProfile, setActiveProfile, selectedGpuIndex } =
useCurveStore(); useCurveStore();
const [activeTab, setActiveTab] = useState<"curve" | "performance" | "fans">( const [activeTab, setActiveTab] = useState<
"curve", "dashboard" | "curve" | "performance" | "fans"
); >("dashboard");
const [fanState, setFanState] = useState<FanState | null>(null); const [fanState, setFanState] = useState<FanState | null>(null);
const [activeDomain, setActiveDomain] = useState<"gpu" | "memory">("gpu"); const [activeDomain, setActiveDomain] = useState<"gpu" | "memory">("gpu");
const [isProfileOpen, setIsProfileOpen] = useState(false); const [isProfileOpen, setIsProfileOpen] = useState(false);
@@ -168,6 +171,12 @@ function MainApp({
{/* Tab Header and Profile Selector */} {/* Tab Header and Profile Selector */}
<div className="flex justify-between items-end border-b border-zinc-800 pb-2"> <div className="flex justify-between items-end border-b border-zinc-800 pb-2">
<div className="flex gap-6"> <div className="flex gap-6">
<button
onClick={() => setActiveTab("dashboard")}
className={`text-lg font-medium pb-2 -mb-[9px] border-b-2 transition-colors ${activeTab === "dashboard" ? "border-pink-500 text-zinc-100" : "border-transparent text-zinc-500 hover:text-zinc-300"}`}
>
Dashboard
</button>
<button <button
onClick={() => setActiveTab("curve")} onClick={() => setActiveTab("curve")}
className={`text-lg font-medium pb-2 -mb-[9px] border-b-2 transition-colors ${activeTab === "curve" ? "border-pink-500 text-zinc-100" : "border-transparent text-zinc-500 hover:text-zinc-300"}`} className={`text-lg font-medium pb-2 -mb-[9px] border-b-2 transition-colors ${activeTab === "curve" ? "border-pink-500 text-zinc-100" : "border-transparent text-zinc-500 hover:text-zinc-300"}`}
@@ -222,7 +231,14 @@ function MainApp({
{/* Main content area */} {/* Main content area */}
<div className="flex flex-col gap-4 w-full"> <div className="flex flex-col gap-4 w-full">
{activeTab === "curve" ? ( {activeTab === "dashboard" ? (
<Dashboard
monitor={monitor}
history={monitorHistory}
dashboard={dashboard}
loading={dashboardLoading}
/>
) : activeTab === "curve" ? (
<> <>
<div className="flex gap-4 items-stretch w-full"> <div className="flex gap-4 items-stretch w-full">
<div className="flex-1 min-w-0 flex flex-col"> <div className="flex-1 min-w-0 flex flex-col">
+2
View File
@@ -7,6 +7,7 @@ import type {
ProfileData, ProfileData,
FanState, FanState,
FanPoint, FanPoint,
DashboardInfo,
} from "../types"; } from "../types";
export class ApiError extends Error { export class ApiError extends Error {
@@ -108,6 +109,7 @@ export const api = {
gpus: () => get<GpuInfo[]>("/gpus"), gpus: () => get<GpuInfo[]>("/gpus"),
gpu: (gpuIndex: number) => get<GpuInfo>("/gpu", gpuIndex), gpu: (gpuIndex: number) => get<GpuInfo>("/gpu", gpuIndex),
dashboard: (gpuIndex: number) => get<DashboardInfo>("/dashboard", gpuIndex),
curve: (gpuIndex: number) => get<CurveState>("/curve", gpuIndex), curve: (gpuIndex: number) => get<CurveState>("/curve", gpuIndex),
ranges: (gpuIndex: number) => ranges: (gpuIndex: number) =>
get<Record<string, { min_khz: number; max_khz: number }>>( get<Record<string, { min_khz: number; max_khz: number }>>(
@@ -0,0 +1,318 @@
import { Loader } from "lucide-react";
import { GaugeCard } from "../Monitor/GaugeCard";
import { fmt } from "../../utils/units";
import type { MonitoringSample, DashboardInfo } from "../../types";
interface Props {
monitor: MonitoringSample | null;
history: MonitoringSample[];
dashboard: DashboardInfo | null;
loading: boolean;
}
function pluck<K extends keyof MonitoringSample>(
history: MonitoringSample[],
key: K,
): number[] {
return history.map((s) => (s[key] as number | null) ?? 0);
}
function pcieLink(
width: number | null | undefined,
gen: number | null | undefined,
): string | null {
if (width == null && gen == null) return null;
return `x${width ?? "?"} Gen${gen ?? "?"}`;
}
/** A static label/value pair. Renders nothing when the value is unavailable. */
function InfoItem({
label,
value,
accent,
}: {
label: string;
value: string | null | undefined;
accent?: string;
}) {
if (value == null || value === "") return null;
return (
<div className="flex flex-col gap-0.5 min-w-0">
<span className="text-xs text-zinc-500">{label}</span>
<span
className={`text-sm font-mono truncate ${accent ?? "text-zinc-200"}`}
title={value}
>
{value}
</span>
</div>
);
}
function TempRow({ label, value }: { label: string; value: number | null }) {
if (value == null) return null;
return (
<div className="flex items-baseline justify-between gap-2">
<span className="text-xs text-zinc-500">{label}</span>
<span className="text-lg font-mono font-semibold text-orange-400">
{fmt.celsius(value)}
</span>
</div>
);
}
export function Dashboard({ monitor, history, dashboard: d, loading }: Props) {
if (loading && !d) {
return (
<div className="bg-zinc-900 rounded-lg p-10 text-center text-zinc-500 flex items-center justify-center gap-2">
<Loader size={18} className="animate-spin" /> Loading dashboard…
</div>
);
}
const throttleLabel = monitor?.throttle_reasons_label ?? null;
const isThrottled = throttleLabel != null && throttleLabel !== "No";
const memUsed = monitor?.mem_used_mib ?? null;
const memTotal = monitor?.mem_total_mib ?? null;
const memLabel =
memUsed != null && memTotal != null
? `${memUsed.toFixed(0)} / ${memTotal.toFixed(0)} MiB`
: "—";
const powerLimit = d?.power_limit_w ?? null;
const powerRange =
d?.power_min_limit_w != null && d?.power_max_limit_w != null
? `${d.power_min_limit_w}–${d.power_max_limit_w} W`
: null;
return (
<div className="flex flex-col gap-4 w-full">
{/* ── Critical row ─────────────────────────────────────────────── */}
<div className="grid grid-cols-1 md:grid-cols-3 gap-4">
{/* Throttling */}
<div
className={`rounded-lg p-4 border ${
isThrottled
? "bg-amber-500/10 border-amber-500/30"
: "bg-emerald-500/10 border-emerald-500/30"
}`}
>
<div className="text-xs text-zinc-500 uppercase tracking-wider font-semibold">
Throttling
</div>
<div
className={`text-2xl font-semibold mt-1 break-words ${
isThrottled ? "text-amber-400" : "text-emerald-400"
}`}
>
{throttleLabel ?? "—"}
</div>
</div>
{/* GPU Voltage */}
<div className="bg-zinc-900 rounded-lg p-4 border border-zinc-800">
<div className="text-xs text-zinc-500 uppercase tracking-wider font-semibold">
GPU Voltage
</div>
<div className="text-2xl font-mono font-semibold mt-1 text-violet-400">
{fmt.mv(monitor?.voltage_mv)}
</div>
</div>
{/* Temperature */}
<div className="bg-zinc-900 rounded-lg p-4 border border-zinc-800">
<div className="text-xs text-zinc-500 uppercase tracking-wider font-semibold">
Temperature
</div>
<div className="flex flex-col gap-1 mt-1">
<TempRow label="GPU" value={monitor?.temp_c ?? null} />
<TempRow label="VRAM" value={monitor?.mem_temp_c ?? null} />
</div>
</div>
</div>
{/* ── Live Monitor ─────────────────────────────────────────────── */}
<div className="bg-zinc-900 rounded-lg p-4">
<div className="text-xs text-zinc-500 uppercase tracking-wider font-semibold pb-3 border-b border-zinc-800">
Live Monitor
</div>
<div className="grid grid-cols-2 md:grid-cols-3 lg:grid-cols-4 gap-3 mt-3">
<GaugeCard
label="Core Clock"
value={fmt.mhz(monitor?.clock_mhz)}
history={pluck(history, "clock_mhz")}
color="#34d399"
max={d?.max_graphics_clock_mhz ?? 3000}
/>
<GaugeCard
label="Mem Clock"
value={fmt.mhz(monitor?.mem_clock_mhz)}
history={pluck(history, "mem_clock_mhz")}
color="#67e8f9"
max={d?.max_memory_clock_mhz ?? 20000}
/>
<GaugeCard
label="Voltage"
value={fmt.mv(monitor?.voltage_mv)}
history={pluck(history, "voltage_mv")}
color="#a78bfa"
max={1100}
/>
<GaugeCard
label="Power Draw"
value={fmt.watts(monitor?.power_w)}
history={pluck(history, "power_w")}
color="#f472b6"
max={powerLimit != null && powerLimit > 0 ? powerLimit : 600}
/>
<GaugeCard
label="GPU Util"
value={fmt.pct(monitor?.gpu_util_pct)}
history={pluck(history, "gpu_util_pct")}
color="#facc15"
max={100}
/>
<GaugeCard
label="Mem Util"
value={fmt.pct(monitor?.mem_util_pct)}
history={pluck(history, "mem_util_pct")}
color="#38bdf8"
max={100}
/>
<GaugeCard
label="Fan Speed"
value={fmt.pct(monitor?.fan_pct)}
history={pluck(history, "fan_pct")}
color="#fb923c"
max={100}
/>
<GaugeCard
label="P-State"
value={monitor?.pstate_label ?? "Unknown"}
history={pluck(history, "pstate")}
color="#a8a29e"
max={15}
/>
<GaugeCard
label="VRAM Used"
value={memLabel}
history={pluck(history, "mem_used_mib")}
color="#a78bfa"
max={memTotal ?? 32768}
/>
</div>
</div>
{/* ── GPU Information ──────────────────────────────────────────── */}
<div className="bg-zinc-900 rounded-lg p-4">
<div className="text-xs text-zinc-500 uppercase tracking-wider font-semibold pb-3 border-b border-zinc-800">
GPU Information
</div>
<div className="grid grid-cols-2 md:grid-cols-3 lg:grid-cols-4 gap-x-6 gap-y-3 mt-3">
<InfoItem label="Driver" value={d?.driver_version} />
<InfoItem label="VBIOS" value={d?.vbios_version} />
<InfoItem
label="CUDA Cores"
value={d?.cuda_cores != null ? String(d.cuda_cores) : null}
/>
<InfoItem label="CUDA Compute" value={d?.cuda_compute_capability} />
<InfoItem
label="ROP Count"
value={d?.rop_count != null ? String(d.rop_count) : null}
/>
<InfoItem
label="VRAM Type"
value={d?.vram_type}
accent="text-cyan-400"
/>
<InfoItem label="VRAM Total" value={fmt.bytes(d?.vram_total_bytes)} />
<InfoItem
label="Bus Width"
value={
d?.memory_bus_width != null ? `${d.memory_bus_width}-bit` : null
}
/>
<InfoItem
label="CPU Accessible VRAM"
value={fmt.bytes(d?.bar1_total_bytes)}
/>
<InfoItem
label="Resizable BAR"
value={
d?.resizable_bar == null
? null
: d.resizable_bar
? "Enabled"
: "Disabled"
}
accent={
d?.resizable_bar == null
? undefined
: d.resizable_bar
? "text-emerald-400"
: "text-zinc-400"
}
/>
<InfoItem
label="PCIe Link"
value={pcieLink(
monitor?.pcie_link_width,
monitor?.pcie_link_generation,
)}
accent="text-cyan-400"
/>
<InfoItem
label="PCIe Max"
value={pcieLink(
d?.pcie_max_link_width,
d?.pcie_max_link_generation,
)}
/>
<InfoItem
label="Max Core Clock"
value={fmt.mhz(d?.max_graphics_clock_mhz)}
/>
<InfoItem
label="Max Mem Clock"
value={fmt.mhz(d?.max_memory_clock_mhz)}
/>
<InfoItem
label="Power Limit"
value={powerLimit != null ? `${powerLimit} W` : null}
/>
<InfoItem label="Power Range" value={powerRange} />
<InfoItem
label="Temp Slowdown"
value={
d?.temp_slowdown_c != null ? `${d.temp_slowdown_c} °C` : null
}
/>
<InfoItem
label="Temp Shutdown"
value={
d?.temp_shutdown_c != null ? `${d.temp_shutdown_c} °C` : null
}
/>
<InfoItem
label="Persistence Mode"
value={
d?.persistence_mode == null
? null
: d.persistence_mode
? "Enabled"
: "Disabled"
}
/>
<InfoItem
label="Fans"
value={d?.num_fans != null ? String(d.num_fans) : null}
/>
<InfoItem label="Serial" value={d?.serial} />
<InfoItem label="Board Part No." value={d?.board_part_number} />
<InfoItem label="UUID" value={d?.uuid} />
</div>
</div>
</div>
);
}
+47
View File
@@ -0,0 +1,47 @@
import { useEffect, useState } from "react";
import { api } from "../api/client";
import { useCurveStore } from "../store/curveStore";
import type { DashboardInfo } from "../types";
interface DashboardState {
gpuIndex: number;
data: DashboardInfo | null;
done: boolean;
}
/**
* Fetch static GPU info for the Dashboard tab.
* Re-fetches when the selected GPU changes.
*/
export function useDashboard() {
const { selectedGpuIndex } = useCurveStore();
const [state, setState] = useState<DashboardState>({
gpuIndex: -1,
data: null,
done: false,
});
useEffect(() => {
let cancelled = false;
api
.dashboard(selectedGpuIndex)
.then((data) => {
if (!cancelled)
setState({ gpuIndex: selectedGpuIndex, data, done: true });
})
.catch((err) => {
console.error("Failed to load dashboard info:", err);
if (!cancelled)
setState({ gpuIndex: selectedGpuIndex, data: null, done: true });
});
return () => {
cancelled = true;
};
}, [selectedGpuIndex]);
const isCurrent = state.gpuIndex === selectedGpuIndex;
return {
dashboard: isCurrent ? state.data : null,
loading: !isCurrent || !state.done,
};
}
+42 -2
View File
@@ -8,7 +8,7 @@ export interface VFPoint {
delta_mhz: number; delta_mhz: number;
effective_freq_khz: number; effective_freq_khz: number;
effective_freq_mhz: number; effective_freq_mhz: number;
domain: 'gpu' | 'memory'; domain: "gpu" | "memory";
} }
export interface CurveState { export interface CurveState {
@@ -34,6 +34,46 @@ export interface MonitoringSample {
gpu_util_pct: number | null; gpu_util_pct: number | null;
mem_util_pct: number | null; mem_util_pct: number | null;
mem_clock_mhz: number | null; mem_clock_mhz: number | null;
throttle_reasons: number | null;
throttle_reasons_label: string | null;
pcie_link_width: number | null;
pcie_link_generation: number | null;
mem_temp_c: number | null;
}
export interface DashboardInfo {
name: string | null;
index: number;
driver_version: string | null;
vbios_version: string | null;
serial: string | null;
board_part_number: string | null;
uuid: string | null;
cuda_cores: number | null;
cuda_compute_capability: string | null;
rop_count: number | null;
vram_total_bytes: number | null;
vram_type: string | null;
memory_bus_width: number | null;
bar1_total_bytes: number | null;
bar1_used_bytes: number | null;
resizable_bar: boolean | null;
pcie_link_width: number | null;
pcie_link_generation: number | null;
pcie_max_link_width: number | null;
pcie_max_link_generation: number | null;
max_graphics_clock_mhz: number | null;
max_memory_clock_mhz: number | null;
max_video_clock_mhz: number | null;
power_limit_w: number | null;
power_min_limit_w: number | null;
power_max_limit_w: number | null;
persistence_mode: boolean | null;
num_fans: number | null;
supported_throttle_reasons: number | null;
temp_slowdown_c: number | null;
temp_shutdown_c: number | null;
temp_gpu_max_c: number | null;
} }
export interface GpuInfo { export interface GpuInfo {
@@ -73,7 +113,7 @@ export interface FanPoint {
export interface FanState { export interface FanState {
fan_pct: number | null; fan_pct: number | null;
fan_mode: 'auto' | 'curve' | null; fan_mode: "auto" | "curve" | null;
min_fan_pct: number | null; min_fan_pct: number | null;
max_fan_pct: number | null; max_fan_pct: number | null;
curve: FanPoint[] | null; curve: FanPoint[] | null;
+14 -6
View File
@@ -1,16 +1,24 @@
export const fmt = { export const fmt = {
mhz: (v: number | null | undefined, decimals = 0) => mhz: (v: number | null | undefined, decimals = 0) =>
v == null ? '—' : `${v.toFixed(decimals)} MHz`, v == null ? "—" : `${v.toFixed(decimals)} MHz`,
mv: (v: number | null | undefined, decimals = 0) => mv: (v: number | null | undefined, decimals = 0) =>
v == null ? '—' : `${v.toFixed(decimals)} mV`, v == null ? "—" : `${v.toFixed(decimals)} mV`,
celsius: (v: number | null | undefined, decimals = 0) => celsius: (v: number | null | undefined, decimals = 0) =>
v == null ? '—' : `${v.toFixed(decimals)} °C`, v == null ? "—" : `${v.toFixed(decimals)} °C`,
watts: (v: number | null | undefined, decimals = 0) => watts: (v: number | null | undefined, decimals = 0) =>
v == null ? '—' : `${v.toFixed(decimals)} W`, v == null ? "—" : `${v.toFixed(decimals)} W`,
pct: (v: number | null | undefined, decimals = 0) => pct: (v: number | null | undefined, decimals = 0) =>
v == null ? '—' : `${v.toFixed(decimals)}%`, v == null ? "—" : `${v.toFixed(decimals)}%`,
mib: (v: number | null | undefined) => mib: (v: number | null | undefined) =>
v == null ? '—' : `${v.toFixed(0)} MiB`, v == null ? "—" : `${v.toFixed(0)} MiB`,
// Human-readable byte size (auto-scales to MiB / GiB).
bytes: (v: number | null | undefined) => {
if (v == null) return "—";
const gib = v / 1024 ** 3;
if (gib >= 1) return `${gib.toFixed(2)} GiB`;
const mib = v / 1024 ** 2;
return `${mib.toFixed(0)} MiB`;
},
}; };
export const conv = { export const conv = {
+18 -7
View File
@@ -1,13 +1,24 @@
from .gpu import get_gpu, discover_gpus from .dashboard import get_dashboard_info
from .vfcurve import read_curve, read_clock_offsets, write_offsets, reset_offsets from .gpu import discover_gpus, get_gpu
from .monitoring import poll, read_voltage from .monitoring import poll, read_voltage
from .ranges import get_clock_ranges from .ranges import get_clock_ranges
from .snapshot import save as snapshot_save, restore as snapshot_restore, list_snapshots from .snapshot import list_snapshots
from .snapshot import restore as snapshot_restore
from .snapshot import save as snapshot_save
from .vfcurve import read_clock_offsets, read_curve, reset_offsets, write_offsets
__all__ = [ __all__ = [
"get_gpu", "discover_gpus", "get_gpu",
"read_curve", "read_clock_offsets", "write_offsets", "reset_offsets", "discover_gpus",
"poll", "read_voltage", "get_dashboard_info",
"read_curve",
"read_clock_offsets",
"write_offsets",
"reset_offsets",
"poll",
"read_voltage",
"get_clock_ranges", "get_clock_ranges",
"snapshot_save", "snapshot_restore", "list_snapshots", "snapshot_save",
"snapshot_restore",
"list_snapshots",
] ]
+285
View File
@@ -0,0 +1,285 @@
"""Static GPU information for the Dashboard tab.
Collects hardware-identifying and capability data via NVML (pynvml).
Everything here is a one-shot read (no polling) — live values (clocks, temps,
power, throttle) come from the monitoring WebSocket instead.
Fields that NVML cannot provide (e.g. ROP count, VRAM type) are filled from a
best-effort per-model table keyed on the GPU name. When a value cannot be
determined, the field is ``None`` and the frontend omits it.
"""
import logging
from typing import Any
try:
import pynvml as _pynvml_import
_NVML_AVAILABLE = True
except ImportError:
_pynvml_import = None
_NVML_AVAILABLE = False
# Aliased as Any so attribute access is not flagged when the import failed.
pynvml: Any = _pynvml_import
log = logging.getLogger("nvcurve.hal.dashboard")
# ── Best-effort per-model specs ───────────────────────────────────────────────
# NVML does not expose the ROP count or the VRAM type, so we infer them from
# the GPU model name. Keyed on the model token (e.g. "RTX 5090"). Values are
# (vram_type, rop_count). Only well-known cards are listed; unknown models
# simply yield None for both.
_GPU_SPECS: dict[str, tuple[str, int]] = {
# Blackwell (RTX 50) — GDDR7
"RTX 5090": ("GDDR7", 192),
"RTX 5080": ("GDDR7", 112),
"RTX 5070 Ti": ("GDDR7", 96),
"RTX 5070": ("GDDR7", 64),
"RTX 5060 Ti": ("GDDR7", 48),
"RTX 5060": ("GDDR7", 48),
# Ada (RTX 40)
"RTX 4090": ("GDDR6X", 128),
"RTX 4080 Super": ("GDDR6X", 112),
"RTX 4080": ("GDDR6X", 112),
"RTX 4070 Ti Super": ("GDDR6X", 96),
"RTX 4070 Super": ("GDDR6X", 64),
"RTX 4070 Ti": ("GDDR6X", 80),
"RTX 4070": ("GDDR6", 64),
"RTX 4060 Ti": ("GDDR6", 48),
"RTX 4060": ("GDDR6", 32),
# Ampere (RTX 30)
"RTX 3090 Ti": ("GDDR6X", 88),
"RTX 3090": ("GDDR6X", 88),
"RTX 3080 Ti": ("GDDR6X", 88),
"RTX 3080": ("GDDR6X", 88),
"RTX 3070 Ti": ("GDDR6", 80),
"RTX 3070": ("GDDR6", 80),
"RTX 3060 Ti": ("GDDR6", 64),
"RTX 3060": ("GDDR6", 48),
# Turing (RTX 20)
"RTX 2080 Ti": ("GDDR6", 64),
"RTX 2080 Super": ("GDDR6", 48),
"RTX 2080": ("GDDR6", 48),
"RTX 2070 Super": ("GDDR6", 48),
"RTX 2070": ("GDDR6", 48),
"RTX 2060 Super": ("GDDR6", 48),
"RTX 2060": ("GDDR6", 36),
# Pascal / Maxwell (GTX 10 / GTX 9)
"GTX 1080 Ti": ("GDDR5X", 64),
"GTX 1080": ("GDDR5X", 64),
"GTX 1070": ("GDDR5", 64),
"GTX 1060": ("GDDR5", 96),
"GTX 980 Ti": ("GDDR5", 64),
"GTX 980": ("GDDR5", 64),
}
def _lookup_spec(gpu_name: str) -> tuple[str | None, int | None]:
"""Return (vram_type, rop_count) inferred from the GPU name, or (None, None)."""
if not gpu_name:
return None, None
name = gpu_name.upper()
# Longest keys first so "RTX 4070 Ti Super" wins over "RTX 4070 Ti".
for key in sorted(_GPU_SPECS, key=len, reverse=True):
if key.upper() in name:
vram_type, rop = _GPU_SPECS[key]
return vram_type, rop
return None, None
def _safe(fn, *args, **kwargs):
"""Call an NVML function, returning None on any error."""
try:
return fn(*args, **kwargs)
except Exception:
return None
def _to_int(val) -> int | None:
try:
return int(val)
except (TypeError, ValueError):
return None
def _to_float(val) -> float | None:
try:
return float(val)
except (TypeError, ValueError):
return None
def _get_handle(gpu_index: int):
if not _NVML_AVAILABLE:
return None
try:
return pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
except Exception:
return None
def _read_temp_thresholds(handle) -> dict:
out: dict[str, int | None] = {
"temp_slowdown_c": None,
"temp_shutdown_c": None,
"temp_gpu_max_c": None,
}
if handle is None:
return out
mapping = {
"temp_slowdown_c": "NVML_TEMPERATURE_THRESHOLD_SLOWDOWN",
"temp_shutdown_c": "NVML_TEMPERATURE_THRESHOLD_SHUTDOWN",
"temp_gpu_max_c": "NVML_TEMPERATURE_THRESHOLD_GPU_MAX",
}
for key, const in mapping.items():
val = _safe(
pynvml.nvmlDeviceGetTemperatureThreshold,
handle,
getattr(pynvml, const, None),
)
if isinstance(val, (int, float)):
out[key] = _to_int(val)
return out
def get_dashboard_info(gpu_index: int = 0, gpu_name: str = "") -> dict:
"""Collect static GPU info for the dashboard.
``gpu_name`` is passed in (already known by the server) so the spec lookup
works even if NVML name retrieval fails.
"""
handle = _get_handle(gpu_index)
out: dict[str, Any] = {
"name": gpu_name or None,
"index": gpu_index,
"driver_version": None,
"vbios_version": None,
"serial": None,
"board_part_number": None,
"uuid": None,
"cuda_cores": None,
"cuda_compute_capability": None,
"rop_count": None,
"vram_total_bytes": None,
"vram_type": None,
"memory_bus_width": None,
"bar1_total_bytes": None,
"bar1_used_bytes": None,
"resizable_bar": None,
"pcie_link_width": None,
"pcie_link_generation": None,
"pcie_max_link_width": None,
"pcie_max_link_generation": None,
"max_graphics_clock_mhz": None,
"max_memory_clock_mhz": None,
"max_video_clock_mhz": None,
"power_limit_w": None,
"power_min_limit_w": None,
"power_max_limit_w": None,
"persistence_mode": None,
"num_fans": None,
"supported_throttle_reasons": None,
}
if not _NVML_AVAILABLE:
return out
# Driver version (system-wide)
out["driver_version"] = _safe(pynvml.nvmlSystemGetDriverVersion)
if handle is None:
return out
# Identity
out["vbios_version"] = _safe(pynvml.nvmlDeviceGetVbiosVersion, handle)
serial = _safe(pynvml.nvmlDeviceGetSerial, handle)
if isinstance(serial, bytes):
serial = serial.decode(errors="replace")
out["serial"] = serial or None
bpn = _safe(pynvml.nvmlDeviceGetBoardPartNumber, handle)
if isinstance(bpn, bytes):
bpn = bpn.decode(errors="replace")
out["board_part_number"] = bpn or None
uuid = _safe(pynvml.nvmlDeviceGetUUID, handle)
if isinstance(uuid, bytes):
uuid = uuid.decode(errors="replace")
out["uuid"] = uuid or None
# Compute
cores = _safe(pynvml.nvmlDeviceGetNumGpuCores, handle)
if isinstance(cores, (int, float)):
out["cuda_cores"] = _to_int(cores)
cc = _safe(pynvml.nvmlDeviceGetCudaComputeCapability, handle)
if isinstance(cc, (list, tuple)) and len(cc) >= 2:
out["cuda_compute_capability"] = f"{cc[0]}.{cc[1]}"
# Memory
mem = _safe(pynvml.nvmlDeviceGetMemoryInfo, handle)
if mem is not None:
out["vram_total_bytes"] = _to_int(mem.total)
bus_width = _safe(pynvml.nvmlDeviceGetMemoryBusWidth, handle)
if isinstance(bus_width, (int, float)):
out["memory_bus_width"] = _to_int(bus_width)
bar1 = _safe(pynvml.nvmlDeviceGetBAR1MemoryInfo, handle)
if bar1 is not None:
out["bar1_total_bytes"] = _to_int(bar1.bar1Total)
out["bar1_used_bytes"] = _to_int(bar1.bar1Used)
# Best-effort specs (VRAM type + ROP count) from the GPU name.
vram_type, rop = _lookup_spec(gpu_name or (out.get("name") or ""))
out["vram_type"] = vram_type
out["rop_count"] = rop
# Resizable BAR: enabled when the BAR1 aperture is a substantial fraction
# of total VRAM (legacy BAR1 is a fixed 256 MB).
if out["bar1_total_bytes"] is not None and out["vram_total_bytes"]:
out["resizable_bar"] = out["bar1_total_bytes"] >= out["vram_total_bytes"] * 0.5
# PCIe
out["pcie_link_width"] = _safe(pynvml.nvmlDeviceGetCurrPcieLinkWidth, handle)
out["pcie_link_generation"] = _safe(
pynvml.nvmlDeviceGetCurrPcieLinkGeneration, handle
)
out["pcie_max_link_width"] = _safe(pynvml.nvmlDeviceGetMaxPcieLinkWidth, handle)
out["pcie_max_link_generation"] = _safe(
pynvml.nvmlDeviceGetMaxPcieLinkGeneration, handle
)
# Max clocks
out["max_graphics_clock_mhz"] = _safe(
pynvml.nvmlDeviceGetMaxClockInfo, handle, pynvml.NVML_CLOCK_GRAPHICS
)
out["max_memory_clock_mhz"] = _safe(
pynvml.nvmlDeviceGetMaxClockInfo, handle, pynvml.NVML_CLOCK_MEM
)
out["max_video_clock_mhz"] = _safe(
pynvml.nvmlDeviceGetMaxClockInfo, handle, pynvml.NVML_CLOCK_VIDEO
)
# Power limits (mW → W)
limit = _safe(pynvml.nvmlDeviceGetPowerManagementLimit, handle)
if isinstance(limit, (int, float)):
out["power_limit_w"] = (_to_int(limit) or 0) // 1000
constrs = _safe(pynvml.nvmlDeviceGetPowerManagementLimitConstraints, handle)
if isinstance(constrs, (list, tuple)) and len(constrs) >= 2:
out["power_min_limit_w"] = (_to_int(constrs[0]) or 0) // 1000
out["power_max_limit_w"] = (_to_int(constrs[1]) or 0) // 1000
# Misc
persist = _safe(pynvml.nvmlDeviceGetPersistenceMode, handle)
if isinstance(persist, (bool, int)):
out["persistence_mode"] = bool(persist)
fans = _safe(pynvml.nvmlDeviceGetNumFans, handle)
if isinstance(fans, (int, float)):
out["num_fans"] = _to_int(fans)
out["supported_throttle_reasons"] = _safe(
pynvml.nvmlDeviceGetSupportedClocksThrottleReasons, handle
)
# Temperature thresholds (static). Current temps come from the monitor
# WebSocket (live), not this one-shot read.
out.update(_read_temp_thresholds(handle))
return out
+97 -19
View File
@@ -4,21 +4,26 @@ Voltage is read via NvAPI GetCurrentVoltage.
Clock, temperature, power draw, and fan speed are read via NVML (nvidia-ml-py). Clock, temperature, power draw, and fan speed are read via NVML (nvidia-ml-py).
""" """
import contextlib
import struct import struct
import time import time
from typing import Optional from typing import Any
from ..nvapi.bootstrap import nvcall from ..nvapi.bootstrap import nvcall
from ..nvapi.constants import FUNC, VOLT_SIZE from ..nvapi.constants import FUNC, VOLT_SIZE
from ..nvapi.types import MonitoringSample from ..nvapi.types import MonitoringSample
try: try:
import pynvml as _pynvml import pynvml as _pynvml_import
_NVML_AVAILABLE = True _NVML_AVAILABLE = True
except ImportError: except ImportError:
_pynvml = None _pynvml_import = None
_NVML_AVAILABLE = False _NVML_AVAILABLE = False
# Aliased as Any so attribute access is not flagged when the import failed.
_pynvml: Any = _pynvml_import
_nvml_initialized = False _nvml_initialized = False
@@ -39,14 +44,12 @@ def shutdown_nvml() -> None:
"""Shut down NVML. Call at process exit.""" """Shut down NVML. Call at process exit."""
global _nvml_initialized global _nvml_initialized
if _NVML_AVAILABLE and _nvml_initialized: if _NVML_AVAILABLE and _nvml_initialized:
try: with contextlib.suppress(_pynvml.NVMLError):
_pynvml.nvmlShutdown() _pynvml.nvmlShutdown()
except _pynvml.NVMLError:
pass
_nvml_initialized = False _nvml_initialized = False
def get_driver_version() -> Optional[str]: def get_driver_version() -> str | None:
"""Return the NVIDIA driver version string, or None if unavailable.""" """Return the NVIDIA driver version string, or None if unavailable."""
if not (_NVML_AVAILABLE and _nvml_initialized): if not (_NVML_AVAILABLE and _nvml_initialized):
return None return None
@@ -56,7 +59,42 @@ def get_driver_version() -> Optional[str]:
return None return None
def get_vram_total(gpu_index: int = 0) -> Optional[int]: # NVML clock-throttle reason bits (from nvmlClocksThrottleReason* constants).
# Mapped to short human-readable labels for the dashboard.
_THROTTLE_REASONS: dict[int, str] = {
0x1: "GPU Idle",
0x2: "Application Clocks",
0x4: "SW Power Cap",
0x8: "HW Slowdown",
0x10: "Sync Boost",
0x20: "SW Thermal Slowdown",
0x40: "HW Thermal Slowdown",
0x80: "HW Power Brake",
0x100: "Display Clock Setting",
}
def throttle_reasons_label(mask: int | None) -> str | None:
"""Convert a throttle-reasons bitmask to a comma-joined label.
Returns None when the mask is unknown, or "No" when nothing is active.
"""
if mask is None:
return None
if mask == 0:
return "No"
active = [label for bit, label in _THROTTLE_REASONS.items() if mask & bit]
# Include any unknown high bits so nothing is silently dropped.
known = 0
for bit in _THROTTLE_REASONS:
known |= bit
extra = mask & ~known
if extra:
active.append(f"0x{extra:x}")
return ", ".join(active) if active else "No"
def get_vram_total(gpu_index: int = 0) -> int | None:
"""Return total VRAM in bytes, or None if unavailable.""" """Return total VRAM in bytes, or None if unavailable."""
if not (_NVML_AVAILABLE and _nvml_initialized): if not (_NVML_AVAILABLE and _nvml_initialized):
return None return None
@@ -67,7 +105,7 @@ def get_vram_total(gpu_index: int = 0) -> Optional[int]:
return None return None
def read_voltage(gpu) -> tuple[Optional[int], str]: def read_voltage(gpu) -> tuple[int | None, str]:
"""Read current GPU core voltage in µV via NvAPI GetCurrentVoltage. """Read current GPU core voltage in µV via NvAPI GetCurrentVoltage.
Returns (voltage_uV, "OK") or (None, error). Returns (voltage_uV, "OK") or (None, error).
@@ -80,19 +118,36 @@ def read_voltage(gpu) -> tuple[Optional[int], str]:
def _nvml_read(gpu_index: int) -> dict: def _nvml_read(gpu_index: int) -> dict:
"""Read all NVML fields. Returns a dict with keys matching MonitoringSample fields.""" """Read all NVML fields. Returns a dict with keys matching MonitoringSample fields."""
out = { out: dict[str, Any] = {
"clock_mhz": None, "temp_c": None, "power_w": None, "fan_pct": None, "clock_mhz": None,
"pstate": None, "mem_used_bytes": None, "mem_total_bytes": None, "temp_c": None,
"gpu_util_pct": None, "mem_util_pct": None, "mem_clock_mhz": None, "power_w": None,
"fan_pct": None,
"pstate": None,
"mem_used_bytes": None,
"mem_total_bytes": None,
"gpu_util_pct": None,
"mem_util_pct": None,
"mem_clock_mhz": None,
"throttle_reasons": None,
"pcie_link_width": None,
"pcie_link_generation": None,
"mem_temp_c": None,
} }
if not (_NVML_AVAILABLE and _nvml_initialized): if not (_NVML_AVAILABLE and _nvml_initialized):
return out return out
try: try:
handle = _pynvml.nvmlDeviceGetHandleByIndex(gpu_index) handle = _pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
out["clock_mhz"] = float(_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_GRAPHICS)) out["clock_mhz"] = float(
out["mem_clock_mhz"] = float(_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_MEM)) _pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_GRAPHICS)
out["temp_c"] = float(_pynvml.nvmlDeviceGetTemperature(handle, _pynvml.NVML_TEMPERATURE_GPU)) )
out["mem_clock_mhz"] = float(
_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_MEM)
)
out["temp_c"] = float(
_pynvml.nvmlDeviceGetTemperature(handle, _pynvml.NVML_TEMPERATURE_GPU)
)
out["power_w"] = _pynvml.nvmlDeviceGetPowerUsage(handle) / 1000.0 # mW → W out["power_w"] = _pynvml.nvmlDeviceGetPowerUsage(handle) / 1000.0 # mW → W
out["pstate"] = int(_pynvml.nvmlDeviceGetPerformanceState(handle)) out["pstate"] = int(_pynvml.nvmlDeviceGetPerformanceState(handle))
@@ -104,10 +159,33 @@ def _nvml_read(gpu_index: int) -> dict:
out["gpu_util_pct"] = float(util.gpu) out["gpu_util_pct"] = float(util.gpu)
out["mem_util_pct"] = float(util.memory) out["mem_util_pct"] = float(util.memory)
try: with contextlib.suppress(_pynvml.NVMLError):
out["fan_pct"] = float(_pynvml.nvmlDeviceGetFanSpeed(handle)) out["fan_pct"] = float(_pynvml.nvmlDeviceGetFanSpeed(handle))
except _pynvml.NVMLError:
pass with contextlib.suppress(_pynvml.NVMLError):
out["throttle_reasons"] = int(
_pynvml.nvmlDeviceGetCurrentClocksThrottleReasons(handle)
)
# PCIe link downclocks when idle, so read it live each poll.
with contextlib.suppress(_pynvml.NVMLError):
out["pcie_link_width"] = int(_pynvml.nvmlDeviceGetCurrPcieLinkWidth(handle))
with contextlib.suppress(_pynvml.NVMLError):
out["pcie_link_generation"] = int(
_pynvml.nvmlDeviceGetCurrPcieLinkGeneration(handle)
)
# VRAM temp (if the GPU exposes a MEMORY thermal sensor).
with contextlib.suppress(_pynvml.NVMLError):
sensors = _pynvml.nvmlDeviceGetThermalSettings(handle, 0)
for s in sensors:
if (
s.target == _pynvml.NVML_THERMAL_TARGET_MEMORY
and s.currentTemp is not None
and s.currentTemp > 0
):
out["mem_temp_c"] = float(s.currentTemp)
break
except _pynvml.NVMLError: except _pynvml.NVMLError:
pass pass
return out return out
+5 -1
View File
@@ -4,7 +4,7 @@ Raw struct manipulation stays in the call layer (bootstrap.py + hal/).
Everything above HAL works with these types. Everything above HAL works with these types.
""" """
from dataclasses import dataclass, field from dataclasses import dataclass
@dataclass @dataclass
@@ -57,6 +57,10 @@ class MonitoringSample:
gpu_util_pct: float | None # GPU core utilization (0–100) gpu_util_pct: float | None # GPU core utilization (0–100)
mem_util_pct: float | None # Memory bus utilization (0–100) mem_util_pct: float | None # Memory bus utilization (0–100)
mem_clock_mhz: float | None = None # Current memory clock (NVML_CLOCK_MEM) mem_clock_mhz: float | None = None # Current memory clock (NVML_CLOCK_MEM)
throttle_reasons: int | None = None # Active clock-throttle bitmask (NVML)
pcie_link_width: int | None = None # Current PCIe link width (x1..x16)
pcie_link_generation: int | None = None # Current PCIe link generation (1..5)
mem_temp_c: float | None = None # VRAM temperature (if the GPU exposes it)
@dataclass @dataclass
+54 -28
View File
@@ -9,7 +9,7 @@ Requires root (NvAPI needs it).
import asyncio import asyncio
import logging import logging
import os import os
from contextlib import asynccontextmanager from contextlib import asynccontextmanager, suppress
from pathlib import Path from pathlib import Path
from typing import Any from typing import Any
@@ -21,6 +21,7 @@ from pydantic import BaseModel
from . import auth from . import auth
from .config import Config, default_config from .config import Config, default_config
from .hal.dashboard import get_dashboard_info
from .hal.fans import ( from .hal.fans import (
get_fan_info, get_fan_info,
get_temp, get_temp,
@@ -43,6 +44,7 @@ from .hal.monitoring import (
init_nvml, init_nvml,
poll, poll,
shutdown_nvml, shutdown_nvml,
throttle_reasons_label,
) )
from .hal.ranges import get_clock_ranges from .hal.ranges import get_clock_ranges
from .hal.snapshot import ( from .hal.snapshot import (
@@ -159,9 +161,29 @@ def _sample_dict(s) -> dict:
else None, else None,
"gpu_util_pct": s.gpu_util_pct, "gpu_util_pct": s.gpu_util_pct,
"mem_util_pct": s.mem_util_pct, "mem_util_pct": s.mem_util_pct,
"throttle_reasons": s.throttle_reasons,
"throttle_reasons_label": throttle_reasons_label(s.throttle_reasons),
"pcie_link_width": s.pcie_link_width,
"pcie_link_generation": s.pcie_link_generation,
"mem_temp_c": s.mem_temp_c,
} }
def _int_key_deltas(deltas: dict) -> dict[int, int]:
"""Convert string-keyed deltas (from JSON) to int-keyed.
Raises ValueError if any key is not a valid integer (corrupted profile),
so a bad profile fails closed rather than partially applying to hardware.
"""
out: dict[int, int] = {}
for k, v in deltas.items():
try:
out[int(k)] = v
except (TypeError, ValueError) as exc:
raise ValueError(f"Invalid curve point index in profile: {k!r}") from exc
return out
# ── WebSocket broadcast ─────────────────────────────────────────────────────── # ── WebSocket broadcast ───────────────────────────────────────────────────────
@@ -309,18 +331,14 @@ async def lifespan(app: FastAPI):
for task in poller_tasks: for task in poller_tasks:
task.cancel() task.cancel()
for task in poller_tasks: for task in poller_tasks:
try: with suppress(asyncio.CancelledError):
await task await task
except asyncio.CancelledError:
pass
for gpu_index, g_state in _state["gpus"].items(): for gpu_index, g_state in _state["gpus"].items():
if g_state.get("fan_poller_task"): if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel() g_state["fan_poller_task"].cancel()
try: with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"] await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
if g_state.get("fan_curve_active"): if g_state.get("fan_curve_active"):
g_state["fan_curve_active"] = False g_state["fan_curve_active"] = False
g_state["fan_curve"] = None g_state["fan_curve"] = None
@@ -345,7 +363,10 @@ app = FastAPI(title="nvcurve", version="0.5.0", lifespan=lifespan)
app.add_middleware( app.add_middleware(
CORSMiddleware, CORSMiddleware,
allow_origins=["*"], # The SPA is served same-origin by this server, so CORS only matters for
# local development (e.g. the Vite dev server). Restrict to localhost
# origins rather than a wildcard.
allow_origin_regex=r"https?://(localhost|127\.0\.0\.1)(:\d+)?$",
allow_methods=["*"], allow_methods=["*"],
allow_headers=["*"], allow_headers=["*"],
) )
@@ -589,6 +610,17 @@ async def api_gpu(gpu_index: int = 0):
} }
@app.get("/api/dashboard")
async def api_dashboard(gpu_index: int = 0):
"""Static GPU info for the Dashboard tab (VBIOS, CUDA cores, PCIe, BAR1, etc.).
Live values (clocks, temps, power, throttle) come from the monitor WebSocket.
"""
_, g_state = _require_gpu(gpu_index)
info = await _run(get_dashboard_info, gpu_index, g_state["gpu_name"])
return info
@app.get("/api/curve") @app.get("/api/curve")
async def api_curve(gpu_index: int = 0): async def api_curve(gpu_index: int = 0):
"""Full CurveState: all V/F points with base freq, voltage, delta, effective freq.""" """Full CurveState: all V/F points with base freq, voltage, delta, effective freq."""
@@ -780,9 +812,7 @@ async def _auto_apply_profile_with_retry(
profile = await _run(load_profile, filepath) # raises FileNotFoundError if missing profile = await _run(load_profile, filepath) # raises FileNotFoundError if missing
expected: dict[int, int] = ( expected: dict[int, int] = (
{int(k): v for k, v in profile.curve_deltas.items()} _int_key_deltas(profile.curve_deltas) if profile.curve_deltas else {}
if profile.curve_deltas
else {}
) )
for attempt in range(max_retries): for attempt in range(max_retries):
@@ -880,7 +910,7 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
# Apply curve deltas (after mem offset which may have wiped them). # Apply curve deltas (after mem offset which may have wiped them).
async with g_state["write_lock"]: async with g_state["write_lock"]:
if profile.curve_deltas: if profile.curve_deltas:
deltas = {int(k): v for k, v in profile.curve_deltas.items()} deltas = _int_key_deltas(profile.curve_deltas)
errors = validate_write(deltas, cfg.max_delta_khz) errors = validate_write(deltas, cfg.max_delta_khz)
if errors: if errors:
errs.append("Curve: " + "; ".join(errors)) errs.append("Curve: " + "; ".join(errors))
@@ -910,10 +940,8 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
# Stop existing fan poller if running # Stop existing fan poller if running
if g_state.get("fan_poller_task"): if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel() g_state["fan_poller_task"].cancel()
try: with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"] await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_curve"] = profile.fan_curve g_state["fan_curve"] = profile.fan_curve
g_state["fan_curve_active"] = True g_state["fan_curve_active"] = True
@@ -922,10 +950,8 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
# Profile has no fan curve, deactivate any active fan curve # Profile has no fan curve, deactivate any active fan curve
if g_state.get("fan_poller_task"): if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel() g_state["fan_poller_task"].cancel()
try: with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"] await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_poller_task"] = None g_state["fan_poller_task"] = None
g_state["fan_curve"] = None g_state["fan_curve"] = None
g_state["fan_curve_active"] = False g_state["fan_curve_active"] = False
@@ -942,10 +968,14 @@ async def api_profile_apply(name: str, gpu_index: int = 0):
_require_gpu(gpu_index) _require_gpu(gpu_index)
try: try:
errs = await _apply_profile(name, gpu_index) errs = await _apply_profile(name, gpu_index)
except FileNotFoundError: except FileNotFoundError as err:
raise HTTPException(status_code=404, detail=f"Profile '{name}' not found") raise HTTPException(
status_code=404, detail=f"Profile '{name}' not found"
) from err
except Exception as e: except Exception as e:
raise HTTPException(status_code=500, detail=f"Failed to load profile: {e}") raise HTTPException(
status_code=500, detail=f"Failed to load profile: {e}"
) from e
if errs: if errs:
raise HTTPException(status_code=500, detail="; ".join(errs)) raise HTTPException(status_code=500, detail="; ".join(errs))
return {"ok": True} return {"ok": True}
@@ -1169,10 +1199,8 @@ async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
# Stop existing poller if running # Stop existing poller if running
if g_state.get("fan_poller_task"): if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel() g_state["fan_poller_task"].cancel()
try: with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"] await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_curve"] = curve_data g_state["fan_curve"] = curve_data
g_state["fan_curve_active"] = True g_state["fan_curve_active"] = True
@@ -1189,10 +1217,8 @@ async def api_fans_reset(gpu_index: int = 0):
# Stop poller # Stop poller
if g_state.get("fan_poller_task"): if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel() g_state["fan_poller_task"].cancel()
try: with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"] await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_poller_task"] = None g_state["fan_poller_task"] = None
g_state["fan_curve"] = None g_state["fan_curve"] = None
@@ -1234,7 +1260,7 @@ async def _reconcile_check(gpu_index: int) -> dict | None:
if current is None: if current is None:
return None # Can't read — let the write attempt proceed return None # Can't read — let the write attempt proceed
changed = [i for i, (a, b) in enumerate(zip(last, current)) if a != b] changed = [i for i, (a, b) in enumerate(zip(last, current, strict=False)) if a != b]
if not changed: if not changed:
return None return None