feat: add Dashboard tab with live GPU overview

Add a new Dashboard tab as the default/first tab (Dashboard - Curve -
Performance - Fans) showing a full GPU overview.

Backend:
- New hal/dashboard.py: one-shot static GPU info via NVML (VBIOS, CUDA
  cores, compute capability, bus width, BAR1/CPU-accessible VRAM,
  Resizable BAR, PCIe max, max clocks, power limits, temp thresholds,
  persistence mode, fan count, serial, board part number, UUID).
  ROP count and VRAM type are best-effort from a per-model table since
  NVML does not expose them.
- New /api/dashboard endpoint.
- MonitoringSample gains live throttle_reasons (+label), PCIe link
  width/generation (downclocks when idle, so read per poll), and VRAM
  temp (from the MEMORY thermal sensor, if exposed).

Frontend:
- New Dashboard component: critical top row (throttling, voltage,
  GPU/VRAM temps), full live-monitor grid with sparklines, and a static
  GPU-information grid. Unavailable fields are omitted.
- useDashboard hook + DashboardInfo type + api.client dashboard().

Also: restrict CORS to localhost origins (end-anchored), make
_int_key_deltas fail closed on bad keys, and clean up lint blockers in
server.py/monitoring.py.
This commit is contained in:
ARIA committed 2026-09-02 16:56:47 +02:00
1 parent 3c55263250
commit da507d55bb
11 files changed
+907 -72

No files matched your search

+54 -28
View File
@@ -9,7 +9,7 @@ Requires root (NvAPI needs it).
import asyncio
import logging
import os
from contextlib import asynccontextmanager
from contextlib import asynccontextmanager, suppress
from pathlib import Path
from typing import Any
@@ -21,6 +21,7 @@ from pydantic import BaseModel
from . import auth
from .config import Config, default_config
from .hal.dashboard import get_dashboard_info
from .hal.fans import (
get_fan_info,
get_temp,
@@ -43,6 +44,7 @@ from .hal.monitoring import (
init_nvml,
poll,
shutdown_nvml,
throttle_reasons_label,
)
from .hal.ranges import get_clock_ranges
from .hal.snapshot import (
@@ -159,9 +161,29 @@ def _sample_dict(s) -> dict:
else None,
"gpu_util_pct": s.gpu_util_pct,
"mem_util_pct": s.mem_util_pct,
"throttle_reasons": s.throttle_reasons,
"throttle_reasons_label": throttle_reasons_label(s.throttle_reasons),
"pcie_link_width": s.pcie_link_width,
"pcie_link_generation": s.pcie_link_generation,
"mem_temp_c": s.mem_temp_c,
}
def _int_key_deltas(deltas: dict) -> dict[int, int]:
"""Convert string-keyed deltas (from JSON) to int-keyed.
Raises ValueError if any key is not a valid integer (corrupted profile),
so a bad profile fails closed rather than partially applying to hardware.
"""
out: dict[int, int] = {}
for k, v in deltas.items():
try:
out[int(k)] = v
except (TypeError, ValueError) as exc:
raise ValueError(f"Invalid curve point index in profile: {k!r}") from exc
return out
# ── WebSocket broadcast ───────────────────────────────────────────────────────
@@ -309,18 +331,14 @@ async def lifespan(app: FastAPI):
for task in poller_tasks:
task.cancel()
for task in poller_tasks:
try:
with suppress(asyncio.CancelledError):
await task
except asyncio.CancelledError:
pass
for gpu_index, g_state in _state["gpus"].items():
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
if g_state.get("fan_curve_active"):
g_state["fan_curve_active"] = False
g_state["fan_curve"] = None
@@ -345,7 +363,10 @@ app = FastAPI(title="nvcurve", version="0.5.0", lifespan=lifespan)
app.add_middleware(
CORSMiddleware,
allow_origins=["*"],
# The SPA is served same-origin by this server, so CORS only matters for
# local development (e.g. the Vite dev server). Restrict to localhost
# origins rather than a wildcard.
allow_origin_regex=r"https?://(localhost|127\.0\.0\.1)(:\d+)?$",
allow_methods=["*"],
allow_headers=["*"],
)
@@ -589,6 +610,17 @@ async def api_gpu(gpu_index: int = 0):
}
@app.get("/api/dashboard")
async def api_dashboard(gpu_index: int = 0):
"""Static GPU info for the Dashboard tab (VBIOS, CUDA cores, PCIe, BAR1, etc.).
Live values (clocks, temps, power, throttle) come from the monitor WebSocket.
"""
_, g_state = _require_gpu(gpu_index)
info = await _run(get_dashboard_info, gpu_index, g_state["gpu_name"])
return info
@app.get("/api/curve")
async def api_curve(gpu_index: int = 0):
"""Full CurveState: all V/F points with base freq, voltage, delta, effective freq."""
@@ -780,9 +812,7 @@ async def _auto_apply_profile_with_retry(
profile = await _run(load_profile, filepath) # raises FileNotFoundError if missing
expected: dict[int, int] = (
{int(k): v for k, v in profile.curve_deltas.items()}
if profile.curve_deltas
else {}
_int_key_deltas(profile.curve_deltas) if profile.curve_deltas else {}
)
for attempt in range(max_retries):
@@ -880,7 +910,7 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
# Apply curve deltas (after mem offset which may have wiped them).
async with g_state["write_lock"]:
if profile.curve_deltas:
deltas = {int(k): v for k, v in profile.curve_deltas.items()}
deltas = _int_key_deltas(profile.curve_deltas)
errors = validate_write(deltas, cfg.max_delta_khz)
if errors:
errs.append("Curve: " + "; ".join(errors))
@@ -910,10 +940,8 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
# Stop existing fan poller if running
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_curve"] = profile.fan_curve
g_state["fan_curve_active"] = True
@@ -922,10 +950,8 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
# Profile has no fan curve, deactivate any active fan curve
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_poller_task"] = None
g_state["fan_curve"] = None
g_state["fan_curve_active"] = False
@@ -942,10 +968,14 @@ async def api_profile_apply(name: str, gpu_index: int = 0):
_require_gpu(gpu_index)
try:
errs = await _apply_profile(name, gpu_index)
except FileNotFoundError:
raise HTTPException(status_code=404, detail=f"Profile '{name}' not found")
except FileNotFoundError as err:
raise HTTPException(
status_code=404, detail=f"Profile '{name}' not found"
) from err
except Exception as e:
raise HTTPException(status_code=500, detail=f"Failed to load profile: {e}")
raise HTTPException(
status_code=500, detail=f"Failed to load profile: {e}"
) from e
if errs:
raise HTTPException(status_code=500, detail="; ".join(errs))
return {"ok": True}
@@ -1169,10 +1199,8 @@ async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
# Stop existing poller if running
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_curve"] = curve_data
g_state["fan_curve_active"] = True
@@ -1189,10 +1217,8 @@ async def api_fans_reset(gpu_index: int = 0):
# Stop poller
if g_state.get("fan_poller_task"):
g_state["fan_poller_task"].cancel()
try:
with suppress(asyncio.CancelledError):
await g_state["fan_poller_task"]
except asyncio.CancelledError:
pass
g_state["fan_poller_task"] = None
g_state["fan_curve"] = None
@@ -1234,7 +1260,7 @@ async def _reconcile_check(gpu_index: int) -> dict | None:
if current is None:
return None # Can't read — let the write attempt proceed
changed = [i for i, (a, b) in enumerate(zip(last, current)) if a != b]
changed = [i for i, (a, b) in enumerate(zip(last, current, strict=False)) if a != b]
if not changed:
return None