feat: add Dashboard tab with live GPU overview

Add a new Dashboard tab as the default/first tab (Dashboard - Curve -
Performance - Fans) showing a full GPU overview.

Backend:
- New hal/dashboard.py: one-shot static GPU info via NVML (VBIOS, CUDA
  cores, compute capability, bus width, BAR1/CPU-accessible VRAM,
  Resizable BAR, PCIe max, max clocks, power limits, temp thresholds,
  persistence mode, fan count, serial, board part number, UUID).
  ROP count and VRAM type are best-effort from a per-model table since
  NVML does not expose them.
- New /api/dashboard endpoint.
- MonitoringSample gains live throttle_reasons (+label), PCIe link
  width/generation (downclocks when idle, so read per poll), and VRAM
  temp (from the MEMORY thermal sensor, if exposed).

Frontend:
- New Dashboard component: critical top row (throttling, voltage,
  GPU/VRAM temps), full live-monitor grid with sparklines, and a static
  GPU-information grid. Unavailable fields are omitted.
- useDashboard hook + DashboardInfo type + api.client dashboard().

Also: restrict CORS to localhost origins (end-anchored), make
_int_key_deltas fail closed on bad keys, and clean up lint blockers in
server.py/monitoring.py.
This commit is contained in:
ARIA committed 2026-09-02 16:56:47 +02:00
1 parent 3c55263250
commit da507d55bb
11 files changed
+907 -72

No files matched your search

+18 -7
View File
@@ -1,13 +1,24 @@
from .gpu import get_gpu, discover_gpus
from .vfcurve import read_curve, read_clock_offsets, write_offsets, reset_offsets
from .dashboard import get_dashboard_info
from .gpu import discover_gpus, get_gpu
from .monitoring import poll, read_voltage
from .ranges import get_clock_ranges
from .snapshot import save as snapshot_save, restore as snapshot_restore, list_snapshots
from .snapshot import list_snapshots
from .snapshot import restore as snapshot_restore
from .snapshot import save as snapshot_save
from .vfcurve import read_clock_offsets, read_curve, reset_offsets, write_offsets
__all__ = [
"get_gpu", "discover_gpus",
"read_curve", "read_clock_offsets", "write_offsets", "reset_offsets",
"poll", "read_voltage",
"get_gpu",
"discover_gpus",
"get_dashboard_info",
"read_curve",
"read_clock_offsets",
"write_offsets",
"reset_offsets",
"poll",
"read_voltage",
"get_clock_ranges",
"snapshot_save", "snapshot_restore", "list_snapshots",
"snapshot_save",
"snapshot_restore",
"list_snapshots",
]
+285
View File
@@ -0,0 +1,285 @@
"""Static GPU information for the Dashboard tab.
Collects hardware-identifying and capability data via NVML (pynvml).
Everything here is a one-shot read (no polling) — live values (clocks, temps,
power, throttle) come from the monitoring WebSocket instead.
Fields that NVML cannot provide (e.g. ROP count, VRAM type) are filled from a
best-effort per-model table keyed on the GPU name. When a value cannot be
determined, the field is ``None`` and the frontend omits it.
"""
import logging
from typing import Any
try:
import pynvml as _pynvml_import
_NVML_AVAILABLE = True
except ImportError:
_pynvml_import = None
_NVML_AVAILABLE = False
# Aliased as Any so attribute access is not flagged when the import failed.
pynvml: Any = _pynvml_import
log = logging.getLogger("nvcurve.hal.dashboard")
# ── Best-effort per-model specs ───────────────────────────────────────────────
# NVML does not expose the ROP count or the VRAM type, so we infer them from
# the GPU model name. Keyed on the model token (e.g. "RTX 5090"). Values are
# (vram_type, rop_count). Only well-known cards are listed; unknown models
# simply yield None for both.
_GPU_SPECS: dict[str, tuple[str, int]] = {
# Blackwell (RTX 50) — GDDR7
"RTX 5090": ("GDDR7", 192),
"RTX 5080": ("GDDR7", 112),
"RTX 5070 Ti": ("GDDR7", 96),
"RTX 5070": ("GDDR7", 64),
"RTX 5060 Ti": ("GDDR7", 48),
"RTX 5060": ("GDDR7", 48),
# Ada (RTX 40)
"RTX 4090": ("GDDR6X", 128),
"RTX 4080 Super": ("GDDR6X", 112),
"RTX 4080": ("GDDR6X", 112),
"RTX 4070 Ti Super": ("GDDR6X", 96),
"RTX 4070 Super": ("GDDR6X", 64),
"RTX 4070 Ti": ("GDDR6X", 80),
"RTX 4070": ("GDDR6", 64),
"RTX 4060 Ti": ("GDDR6", 48),
"RTX 4060": ("GDDR6", 32),
# Ampere (RTX 30)
"RTX 3090 Ti": ("GDDR6X", 88),
"RTX 3090": ("GDDR6X", 88),
"RTX 3080 Ti": ("GDDR6X", 88),
"RTX 3080": ("GDDR6X", 88),
"RTX 3070 Ti": ("GDDR6", 80),
"RTX 3070": ("GDDR6", 80),
"RTX 3060 Ti": ("GDDR6", 64),
"RTX 3060": ("GDDR6", 48),
# Turing (RTX 20)
"RTX 2080 Ti": ("GDDR6", 64),
"RTX 2080 Super": ("GDDR6", 48),
"RTX 2080": ("GDDR6", 48),
"RTX 2070 Super": ("GDDR6", 48),
"RTX 2070": ("GDDR6", 48),
"RTX 2060 Super": ("GDDR6", 48),
"RTX 2060": ("GDDR6", 36),
# Pascal / Maxwell (GTX 10 / GTX 9)
"GTX 1080 Ti": ("GDDR5X", 64),
"GTX 1080": ("GDDR5X", 64),
"GTX 1070": ("GDDR5", 64),
"GTX 1060": ("GDDR5", 96),
"GTX 980 Ti": ("GDDR5", 64),
"GTX 980": ("GDDR5", 64),
}
def _lookup_spec(gpu_name: str) -> tuple[str | None, int | None]:
"""Return (vram_type, rop_count) inferred from the GPU name, or (None, None)."""
if not gpu_name:
return None, None
name = gpu_name.upper()
# Longest keys first so "RTX 4070 Ti Super" wins over "RTX 4070 Ti".
for key in sorted(_GPU_SPECS, key=len, reverse=True):
if key.upper() in name:
vram_type, rop = _GPU_SPECS[key]
return vram_type, rop
return None, None
def _safe(fn, *args, **kwargs):
"""Call an NVML function, returning None on any error."""
try:
return fn(*args, **kwargs)
except Exception:
return None
def _to_int(val) -> int | None:
try:
return int(val)
except (TypeError, ValueError):
return None
def _to_float(val) -> float | None:
try:
return float(val)
except (TypeError, ValueError):
return None
def _get_handle(gpu_index: int):
if not _NVML_AVAILABLE:
return None
try:
return pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
except Exception:
return None
def _read_temp_thresholds(handle) -> dict:
out: dict[str, int | None] = {
"temp_slowdown_c": None,
"temp_shutdown_c": None,
"temp_gpu_max_c": None,
}
if handle is None:
return out
mapping = {
"temp_slowdown_c": "NVML_TEMPERATURE_THRESHOLD_SLOWDOWN",
"temp_shutdown_c": "NVML_TEMPERATURE_THRESHOLD_SHUTDOWN",
"temp_gpu_max_c": "NVML_TEMPERATURE_THRESHOLD_GPU_MAX",
}
for key, const in mapping.items():
val = _safe(
pynvml.nvmlDeviceGetTemperatureThreshold,
handle,
getattr(pynvml, const, None),
)
if isinstance(val, (int, float)):
out[key] = _to_int(val)
return out
def get_dashboard_info(gpu_index: int = 0, gpu_name: str = "") -> dict:
"""Collect static GPU info for the dashboard.
``gpu_name`` is passed in (already known by the server) so the spec lookup
works even if NVML name retrieval fails.
"""
handle = _get_handle(gpu_index)
out: dict[str, Any] = {
"name": gpu_name or None,
"index": gpu_index,
"driver_version": None,
"vbios_version": None,
"serial": None,
"board_part_number": None,
"uuid": None,
"cuda_cores": None,
"cuda_compute_capability": None,
"rop_count": None,
"vram_total_bytes": None,
"vram_type": None,
"memory_bus_width": None,
"bar1_total_bytes": None,
"bar1_used_bytes": None,
"resizable_bar": None,
"pcie_link_width": None,
"pcie_link_generation": None,
"pcie_max_link_width": None,
"pcie_max_link_generation": None,
"max_graphics_clock_mhz": None,
"max_memory_clock_mhz": None,
"max_video_clock_mhz": None,
"power_limit_w": None,
"power_min_limit_w": None,
"power_max_limit_w": None,
"persistence_mode": None,
"num_fans": None,
"supported_throttle_reasons": None,
}
if not _NVML_AVAILABLE:
return out
# Driver version (system-wide)
out["driver_version"] = _safe(pynvml.nvmlSystemGetDriverVersion)
if handle is None:
return out
# Identity
out["vbios_version"] = _safe(pynvml.nvmlDeviceGetVbiosVersion, handle)
serial = _safe(pynvml.nvmlDeviceGetSerial, handle)
if isinstance(serial, bytes):
serial = serial.decode(errors="replace")
out["serial"] = serial or None
bpn = _safe(pynvml.nvmlDeviceGetBoardPartNumber, handle)
if isinstance(bpn, bytes):
bpn = bpn.decode(errors="replace")
out["board_part_number"] = bpn or None
uuid = _safe(pynvml.nvmlDeviceGetUUID, handle)
if isinstance(uuid, bytes):
uuid = uuid.decode(errors="replace")
out["uuid"] = uuid or None
# Compute
cores = _safe(pynvml.nvmlDeviceGetNumGpuCores, handle)
if isinstance(cores, (int, float)):
out["cuda_cores"] = _to_int(cores)
cc = _safe(pynvml.nvmlDeviceGetCudaComputeCapability, handle)
if isinstance(cc, (list, tuple)) and len(cc) >= 2:
out["cuda_compute_capability"] = f"{cc[0]}.{cc[1]}"
# Memory
mem = _safe(pynvml.nvmlDeviceGetMemoryInfo, handle)
if mem is not None:
out["vram_total_bytes"] = _to_int(mem.total)
bus_width = _safe(pynvml.nvmlDeviceGetMemoryBusWidth, handle)
if isinstance(bus_width, (int, float)):
out["memory_bus_width"] = _to_int(bus_width)
bar1 = _safe(pynvml.nvmlDeviceGetBAR1MemoryInfo, handle)
if bar1 is not None:
out["bar1_total_bytes"] = _to_int(bar1.bar1Total)
out["bar1_used_bytes"] = _to_int(bar1.bar1Used)
# Best-effort specs (VRAM type + ROP count) from the GPU name.
vram_type, rop = _lookup_spec(gpu_name or (out.get("name") or ""))
out["vram_type"] = vram_type
out["rop_count"] = rop
# Resizable BAR: enabled when the BAR1 aperture is a substantial fraction
# of total VRAM (legacy BAR1 is a fixed 256 MB).
if out["bar1_total_bytes"] is not None and out["vram_total_bytes"]:
out["resizable_bar"] = out["bar1_total_bytes"] >= out["vram_total_bytes"] * 0.5
# PCIe
out["pcie_link_width"] = _safe(pynvml.nvmlDeviceGetCurrPcieLinkWidth, handle)
out["pcie_link_generation"] = _safe(
pynvml.nvmlDeviceGetCurrPcieLinkGeneration, handle
)
out["pcie_max_link_width"] = _safe(pynvml.nvmlDeviceGetMaxPcieLinkWidth, handle)
out["pcie_max_link_generation"] = _safe(
pynvml.nvmlDeviceGetMaxPcieLinkGeneration, handle
)
# Max clocks
out["max_graphics_clock_mhz"] = _safe(
pynvml.nvmlDeviceGetMaxClockInfo, handle, pynvml.NVML_CLOCK_GRAPHICS
)
out["max_memory_clock_mhz"] = _safe(
pynvml.nvmlDeviceGetMaxClockInfo, handle, pynvml.NVML_CLOCK_MEM
)
out["max_video_clock_mhz"] = _safe(
pynvml.nvmlDeviceGetMaxClockInfo, handle, pynvml.NVML_CLOCK_VIDEO
)
# Power limits (mW → W)
limit = _safe(pynvml.nvmlDeviceGetPowerManagementLimit, handle)
if isinstance(limit, (int, float)):
out["power_limit_w"] = (_to_int(limit) or 0) // 1000
constrs = _safe(pynvml.nvmlDeviceGetPowerManagementLimitConstraints, handle)
if isinstance(constrs, (list, tuple)) and len(constrs) >= 2:
out["power_min_limit_w"] = (_to_int(constrs[0]) or 0) // 1000
out["power_max_limit_w"] = (_to_int(constrs[1]) or 0) // 1000
# Misc
persist = _safe(pynvml.nvmlDeviceGetPersistenceMode, handle)
if isinstance(persist, (bool, int)):
out["persistence_mode"] = bool(persist)
fans = _safe(pynvml.nvmlDeviceGetNumFans, handle)
if isinstance(fans, (int, float)):
out["num_fans"] = _to_int(fans)
out["supported_throttle_reasons"] = _safe(
pynvml.nvmlDeviceGetSupportedClocksThrottleReasons, handle
)
# Temperature thresholds (static). Current temps come from the monitor
# WebSocket (live), not this one-shot read.
out.update(_read_temp_thresholds(handle))
return out
+97 -19
View File
@@ -4,21 +4,26 @@ Voltage is read via NvAPI GetCurrentVoltage.
Clock, temperature, power draw, and fan speed are read via NVML (nvidia-ml-py).
"""
import contextlib
import struct
import time
from typing import Optional
from typing import Any
from ..nvapi.bootstrap import nvcall
from ..nvapi.constants import FUNC, VOLT_SIZE
from ..nvapi.types import MonitoringSample
try:
import pynvml as _pynvml
import pynvml as _pynvml_import
_NVML_AVAILABLE = True
except ImportError:
_pynvml = None
_pynvml_import = None
_NVML_AVAILABLE = False
# Aliased as Any so attribute access is not flagged when the import failed.
_pynvml: Any = _pynvml_import
_nvml_initialized = False
@@ -39,14 +44,12 @@ def shutdown_nvml() -> None:
"""Shut down NVML. Call at process exit."""
global _nvml_initialized
if _NVML_AVAILABLE and _nvml_initialized:
try:
with contextlib.suppress(_pynvml.NVMLError):
_pynvml.nvmlShutdown()
except _pynvml.NVMLError:
pass
_nvml_initialized = False
def get_driver_version() -> Optional[str]:
def get_driver_version() -> str | None:
"""Return the NVIDIA driver version string, or None if unavailable."""
if not (_NVML_AVAILABLE and _nvml_initialized):
return None
@@ -56,7 +59,42 @@ def get_driver_version() -> Optional[str]:
return None
def get_vram_total(gpu_index: int = 0) -> Optional[int]:
# NVML clock-throttle reason bits (from nvmlClocksThrottleReason* constants).
# Mapped to short human-readable labels for the dashboard.
_THROTTLE_REASONS: dict[int, str] = {
0x1: "GPU Idle",
0x2: "Application Clocks",
0x4: "SW Power Cap",
0x8: "HW Slowdown",
0x10: "Sync Boost",
0x20: "SW Thermal Slowdown",
0x40: "HW Thermal Slowdown",
0x80: "HW Power Brake",
0x100: "Display Clock Setting",
}
def throttle_reasons_label(mask: int | None) -> str | None:
"""Convert a throttle-reasons bitmask to a comma-joined label.
Returns None when the mask is unknown, or "No" when nothing is active.
"""
if mask is None:
return None
if mask == 0:
return "No"
active = [label for bit, label in _THROTTLE_REASONS.items() if mask & bit]
# Include any unknown high bits so nothing is silently dropped.
known = 0
for bit in _THROTTLE_REASONS:
known |= bit
extra = mask & ~known
if extra:
active.append(f"0x{extra:x}")
return ", ".join(active) if active else "No"
def get_vram_total(gpu_index: int = 0) -> int | None:
"""Return total VRAM in bytes, or None if unavailable."""
if not (_NVML_AVAILABLE and _nvml_initialized):
return None
@@ -67,7 +105,7 @@ def get_vram_total(gpu_index: int = 0) -> Optional[int]:
return None
def read_voltage(gpu) -> tuple[Optional[int], str]:
def read_voltage(gpu) -> tuple[int | None, str]:
"""Read current GPU core voltage in µV via NvAPI GetCurrentVoltage.
Returns (voltage_uV, "OK") or (None, error).
@@ -80,19 +118,36 @@ def read_voltage(gpu) -> tuple[Optional[int], str]:
def _nvml_read(gpu_index: int) -> dict:
"""Read all NVML fields. Returns a dict with keys matching MonitoringSample fields."""
out = {
"clock_mhz": None, "temp_c": None, "power_w": None, "fan_pct": None,
"pstate": None, "mem_used_bytes": None, "mem_total_bytes": None,
"gpu_util_pct": None, "mem_util_pct": None, "mem_clock_mhz": None,
out: dict[str, Any] = {
"clock_mhz": None,
"temp_c": None,
"power_w": None,
"fan_pct": None,
"pstate": None,
"mem_used_bytes": None,
"mem_total_bytes": None,
"gpu_util_pct": None,
"mem_util_pct": None,
"mem_clock_mhz": None,
"throttle_reasons": None,
"pcie_link_width": None,
"pcie_link_generation": None,
"mem_temp_c": None,
}
if not (_NVML_AVAILABLE and _nvml_initialized):
return out
try:
handle = _pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
out["clock_mhz"] = float(_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_GRAPHICS))
out["mem_clock_mhz"] = float(_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_MEM))
out["temp_c"] = float(_pynvml.nvmlDeviceGetTemperature(handle, _pynvml.NVML_TEMPERATURE_GPU))
out["clock_mhz"] = float(
_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_GRAPHICS)
)
out["mem_clock_mhz"] = float(
_pynvml.nvmlDeviceGetClockInfo(handle, _pynvml.NVML_CLOCK_MEM)
)
out["temp_c"] = float(
_pynvml.nvmlDeviceGetTemperature(handle, _pynvml.NVML_TEMPERATURE_GPU)
)
out["power_w"] = _pynvml.nvmlDeviceGetPowerUsage(handle) / 1000.0 # mW → W
out["pstate"] = int(_pynvml.nvmlDeviceGetPerformanceState(handle))
@@ -104,10 +159,33 @@ def _nvml_read(gpu_index: int) -> dict:
out["gpu_util_pct"] = float(util.gpu)
out["mem_util_pct"] = float(util.memory)
try:
with contextlib.suppress(_pynvml.NVMLError):
out["fan_pct"] = float(_pynvml.nvmlDeviceGetFanSpeed(handle))
except _pynvml.NVMLError:
pass
with contextlib.suppress(_pynvml.NVMLError):
out["throttle_reasons"] = int(
_pynvml.nvmlDeviceGetCurrentClocksThrottleReasons(handle)
)
# PCIe link downclocks when idle, so read it live each poll.
with contextlib.suppress(_pynvml.NVMLError):
out["pcie_link_width"] = int(_pynvml.nvmlDeviceGetCurrPcieLinkWidth(handle))
with contextlib.suppress(_pynvml.NVMLError):
out["pcie_link_generation"] = int(
_pynvml.nvmlDeviceGetCurrPcieLinkGeneration(handle)
)
# VRAM temp (if the GPU exposes a MEMORY thermal sensor).
with contextlib.suppress(_pynvml.NVMLError):
sensors = _pynvml.nvmlDeviceGetThermalSettings(handle, 0)
for s in sensors:
if (
s.target == _pynvml.NVML_THERMAL_TARGET_MEMORY
and s.currentTemp is not None
and s.currentTemp > 0
):
out["mem_temp_c"] = float(s.currentTemp)
break
except _pynvml.NVMLError:
pass
return out
+10 -6
View File
@@ -4,15 +4,15 @@ Raw struct manipulation stays in the call layer (bootstrap.py + hal/).
Everything above HAL works with these types.
"""
from dataclasses import dataclass, field
from dataclasses import dataclass
@dataclass
class VFPoint:
index: int
freq_khz: int # Base frequency from VFP curve
volt_uv: int # Voltage from VFP curve
delta_khz: int # Offset from ClockBoostTable (signed)
freq_khz: int # Base frequency from VFP curve
volt_uv: int # Voltage from VFP curve
delta_khz: int # Offset from ClockBoostTable (signed)
domain: str = "gpu" # "gpu" or "memory"
@property
@@ -51,12 +51,16 @@ class MonitoringSample:
temp_c: float | None
power_w: float | None
fan_pct: float | None
pstate: int | None # Performance state: 0 (P0, max) – 15 (P15, min)
pstate: int | None # Performance state: 0 (P0, max) – 15 (P15, min)
mem_used_bytes: int | None # VRAM used (bytes)
mem_total_bytes: int | None # VRAM total (bytes)
mem_total_bytes: int | None # VRAM total (bytes)
gpu_util_pct: float | None # GPU core utilization (0–100)
mem_util_pct: float | None # Memory bus utilization (0–100)
mem_clock_mhz: float | None = None # Current memory clock (NVML_CLOCK_MEM)
throttle_reasons: int | None = None # Active clock-throttle bitmask (NVML)
pcie_link_width: int | None = None # Current PCIe link width (x1..x16)
pcie_link_generation: int | None = None # Current PCIe link generation (1..5)
mem_temp_c: float | None = None # VRAM temperature (if the GPU exposes it)
@dataclass
+54 -28
View File
@@ -9,7 +9,7 @@ Requires root (NvAPI needs it).
import asyncio
import logging
import os
from contextlib import asynccontextmanager
from contextlib import asynccontextmanager, suppress
from pathlib import Path
from typing import Any
@@ -21,6 +21,7 @@ from pydantic import BaseModel
from . import auth
from .config import Config, default_config
from .hal.dashboard import get_dashboard_info
from .hal.fans import (
get_fan_info,
get_temp,
@@ -43,6 +44,7 @@ from .hal.monitoring import (
init_nvml,
poll,
shutdown_nvml,
throttle_reasons_label,
)
from .hal.ranges import get_clock_ranges
from .hal.snapshot import (
@@ -159,9 +161,29 @@ def _sample_dict(s) -> dict:
else None,
"gpu_util_pct": s.gpu_util_pct,
"mem_util_pct": s.mem_util_pct,
"throttle_reasons": s.throttle_reasons,
"throttle_reasons_label": throttle_reasons_label(s.throttle_reasons),
"pcie_link_width": s.pcie_link_width,
"pcie_link_generation": s.pcie_link_generation,
"mem_temp_c": s.mem_temp_c,
}
def _int_key_deltas(deltas: dict) -> dict[int, int]:
"""Convert string-keyed deltas (from JSON) to int-keyed.
Raises ValueError if any key is not a valid integer (corrupted profile),
so a bad profile fails closed rather than partially applying to hardware.
"""
out: dict[int, int] = {}
for k, v in deltas.items():
try:
out[int(k)] = v
except (TypeError, ValueError) as exc:
raise ValueError(f"Invalid curve point index in profile: {k!r}") from exc
return out
# ── WebSocket broadcast ───────────────────────────────────────────────────────
@@ -309,18 +331,14 @@ async def lifespan(app: FastAPI):
for task in poller_tasks:
task.cancel()
for task in poller_tasks:
try:
with suppress(asyncio.CancelledError):
await task
except asyncio.CancelledError:
pass
for gpu_index, g_state in _state["gpus"].items():
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
if g_state.get("fan_curve_active"):
g_state["fan_curve_active"] = False
g_state["fan_curve"] = None
@@ -345,7 +363,10 @@ app = FastAPI(title="nvcurve", version="0.5.0", lifespan=lifespan)
app.add_middleware(
CORSMiddleware,
allow_origins=["*"],
# The SPA is served same-origin by this server, so CORS only matters for
# local development (e.g. the Vite dev server). Restrict to localhost
# origins rather than a wildcard.
allow_origin_regex=r"https?://(localhost|127\.0\.0\.1)(:\d+)?$",
allow_methods=["*"],
allow_headers=["*"],
)
@@ -589,6 +610,17 @@ async def api_gpu(gpu_index: int = 0):
}
@app.get("/api/dashboard")
async def api_dashboard(gpu_index: int = 0):
"""Static GPU info for the Dashboard tab (VBIOS, CUDA cores, PCIe, BAR1, etc.).
Live values (clocks, temps, power, throttle) come from the monitor WebSocket.
"""
_, g_state = _require_gpu(gpu_index)
info = await _run(get_dashboard_info, gpu_index, g_state["gpu_name"])
return info
@app.get("/api/curve")
async def api_curve(gpu_index: int = 0):
"""Full CurveState: all V/F points with base freq, voltage, delta, effective freq."""
@@ -780,9 +812,7 @@ async def _auto_apply_profile_with_retry(
profile = await _run(load_profile, filepath) # raises FileNotFoundError if missing
expected: dict[int, int] = (
{int(k): v for k, v in profile.curve_deltas.items()}
if profile.curve_deltas
else {}
_int_key_deltas(profile.curve_deltas) if profile.curve_deltas else {}
)
for attempt in range(max_retries):
@@ -880,7 +910,7 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
# Apply curve deltas (after mem offset which may have wiped them).
async with g_state["write_lock"]:
if profile.curve_deltas:
deltas = {int(k): v for k, v in profile.curve_deltas.items()}
deltas = _int_key_deltas(profile.curve_deltas)
errors = validate_write(deltas, cfg.max_delta_khz)
if errors:
errs.append("Curve: " + "; ".join(errors))
@@ -910,10 +940,8 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
# Stop existing fan poller if running
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_curve"] = profile.fan_curve
g_state["fan_curve_active"] = True
@@ -922,10 +950,8 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
# Profile has no fan curve, deactivate any active fan curve
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_poller_task"] = None
g_state["fan_curve"] = None
g_state["fan_curve_active"] = False
@@ -942,10 +968,14 @@ async def api_profile_apply(name: str, gpu_index: int = 0):
_require_gpu(gpu_index)
try:
errs = await _apply_profile(name, gpu_index)
except FileNotFoundError:
raise HTTPException(status_code=404, detail=f"Profile '{name}' not found")
except FileNotFoundError as err:
raise HTTPException(
status_code=404, detail=f"Profile '{name}' not found"
) from err
except Exception as e:
raise HTTPException(status_code=500, detail=f"Failed to load profile: {e}")
raise HTTPException(
status_code=500, detail=f"Failed to load profile: {e}"
) from e
if errs:
raise HTTPException(status_code=500, detail="; ".join(errs))
return {"ok": True}
@@ -1169,10 +1199,8 @@ async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
# Stop existing poller if running
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_curve"] = curve_data
g_state["fan_curve_active"] = True
@@ -1189,10 +1217,8 @@ async def api_fans_reset(gpu_index: int = 0):
# Stop poller
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_poller_task"] = None
g_state["fan_curve"] = None
@@ -1234,7 +1260,7 @@ async def _reconcile_check(gpu_index: int) -> dict | None:
if current is None:
return None # Can't read — let the write attempt proceed
changed = [i for i, (a, b) in enumerate(zip(last, current)) if a != b]
changed = [i for i, (a, b) in enumerate(zip(last, current, strict=False)) if a != b]
if not changed:
return None