feat: add Dashboard tab with live GPU overview
Add a new Dashboard tab as the default/first tab (Dashboard - Curve - Performance - Fans) showing a full GPU overview. Backend: - New hal/dashboard.py: one-shot static GPU info via NVML (VBIOS, CUDA cores, compute capability, bus width, BAR1/CPU-accessible VRAM, Resizable BAR, PCIe max, max clocks, power limits, temp thresholds, persistence mode, fan count, serial, board part number, UUID). ROP count and VRAM type are best-effort from a per-model table since NVML does not expose them. - New /api/dashboard endpoint. - MonitoringSample gains live throttle_reasons (+label), PCIe link width/generation (downclocks when idle, so read per poll), and VRAM temp (from the MEMORY thermal sensor, if exposed). Frontend: - New Dashboard component: critical top row (throttling, voltage, GPU/VRAM temps), full live-monitor grid with sparklines, and a static GPU-information grid. Unavailable fields are omitted. - useDashboard hook + DashboardInfo type + api.client dashboard(). Also: restrict CORS to localhost origins (end-anchored), make _int_key_deltas fail closed on bad keys, and clean up lint blockers in server.py/monitoring.py.
This commit is contained in:
1 parent
3c55263250
commit
da507d55bb
11 files changed
+907
-72
No files matched your search
+54
-28
@@ -9,7 +9,7 @@ Requires root (NvAPI needs it).
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
from contextlib import asynccontextmanager
|
||||
from contextlib import asynccontextmanager, suppress
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
@@ -21,6 +21,7 @@ from pydantic import BaseModel
|
||||
|
||||
from . import auth
|
||||
from .config import Config, default_config
|
||||
from .hal.dashboard import get_dashboard_info
|
||||
from .hal.fans import (
|
||||
get_fan_info,
|
||||
get_temp,
|
||||
@@ -43,6 +44,7 @@ from .hal.monitoring import (
|
||||
init_nvml,
|
||||
poll,
|
||||
shutdown_nvml,
|
||||
throttle_reasons_label,
|
||||
)
|
||||
from .hal.ranges import get_clock_ranges
|
||||
from .hal.snapshot import (
|
||||
@@ -159,9 +161,29 @@ def _sample_dict(s) -> dict:
|
||||
else None,
|
||||
"gpu_util_pct": s.gpu_util_pct,
|
||||
"mem_util_pct": s.mem_util_pct,
|
||||
"throttle_reasons": s.throttle_reasons,
|
||||
"throttle_reasons_label": throttle_reasons_label(s.throttle_reasons),
|
||||
"pcie_link_width": s.pcie_link_width,
|
||||
"pcie_link_generation": s.pcie_link_generation,
|
||||
"mem_temp_c": s.mem_temp_c,
|
||||
}
|
||||
|
||||
|
||||
def _int_key_deltas(deltas: dict) -> dict[int, int]:
|
||||
"""Convert string-keyed deltas (from JSON) to int-keyed.
|
||||
|
||||
Raises ValueError if any key is not a valid integer (corrupted profile),
|
||||
so a bad profile fails closed rather than partially applying to hardware.
|
||||
"""
|
||||
out: dict[int, int] = {}
|
||||
for k, v in deltas.items():
|
||||
try:
|
||||
out[int(k)] = v
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise ValueError(f"Invalid curve point index in profile: {k!r}") from exc
|
||||
return out
|
||||
|
||||
|
||||
# ── WebSocket broadcast ───────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@@ -309,18 +331,14 @@ async def lifespan(app: FastAPI):
|
||||
for task in poller_tasks:
|
||||
task.cancel()
|
||||
for task in poller_tasks:
|
||||
try:
|
||||
with suppress(asyncio.CancelledError):
|
||||
await task
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
|
||||
for gpu_index, g_state in _state["gpus"].items():
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
with suppress(asyncio.CancelledError):
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
if g_state.get("fan_curve_active"):
|
||||
g_state["fan_curve_active"] = False
|
||||
g_state["fan_curve"] = None
|
||||
@@ -345,7 +363,10 @@ app = FastAPI(title="nvcurve", version="0.5.0", lifespan=lifespan)
|
||||
|
||||
app.add_middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=["*"],
|
||||
# The SPA is served same-origin by this server, so CORS only matters for
|
||||
# local development (e.g. the Vite dev server). Restrict to localhost
|
||||
# origins rather than a wildcard.
|
||||
allow_origin_regex=r"https?://(localhost|127\.0\.0\.1)(:\d+)?$",
|
||||
allow_methods=["*"],
|
||||
allow_headers=["*"],
|
||||
)
|
||||
@@ -589,6 +610,17 @@ async def api_gpu(gpu_index: int = 0):
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/dashboard")
|
||||
async def api_dashboard(gpu_index: int = 0):
|
||||
"""Static GPU info for the Dashboard tab (VBIOS, CUDA cores, PCIe, BAR1, etc.).
|
||||
|
||||
Live values (clocks, temps, power, throttle) come from the monitor WebSocket.
|
||||
"""
|
||||
_, g_state = _require_gpu(gpu_index)
|
||||
info = await _run(get_dashboard_info, gpu_index, g_state["gpu_name"])
|
||||
return info
|
||||
|
||||
|
||||
@app.get("/api/curve")
|
||||
async def api_curve(gpu_index: int = 0):
|
||||
"""Full CurveState: all V/F points with base freq, voltage, delta, effective freq."""
|
||||
@@ -780,9 +812,7 @@ async def _auto_apply_profile_with_retry(
|
||||
profile = await _run(load_profile, filepath) # raises FileNotFoundError if missing
|
||||
|
||||
expected: dict[int, int] = (
|
||||
{int(k): v for k, v in profile.curve_deltas.items()}
|
||||
if profile.curve_deltas
|
||||
else {}
|
||||
_int_key_deltas(profile.curve_deltas) if profile.curve_deltas else {}
|
||||
)
|
||||
|
||||
for attempt in range(max_retries):
|
||||
@@ -880,7 +910,7 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
|
||||
# Apply curve deltas (after mem offset which may have wiped them).
|
||||
async with g_state["write_lock"]:
|
||||
if profile.curve_deltas:
|
||||
deltas = {int(k): v for k, v in profile.curve_deltas.items()}
|
||||
deltas = _int_key_deltas(profile.curve_deltas)
|
||||
errors = validate_write(deltas, cfg.max_delta_khz)
|
||||
if errors:
|
||||
errs.append("Curve: " + "; ".join(errors))
|
||||
@@ -910,10 +940,8 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
|
||||
# Stop existing fan poller if running
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
with suppress(asyncio.CancelledError):
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
|
||||
g_state["fan_curve"] = profile.fan_curve
|
||||
g_state["fan_curve_active"] = True
|
||||
@@ -922,10 +950,8 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
|
||||
# Profile has no fan curve, deactivate any active fan curve
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
with suppress(asyncio.CancelledError):
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
g_state["fan_poller_task"] = None
|
||||
g_state["fan_curve"] = None
|
||||
g_state["fan_curve_active"] = False
|
||||
@@ -942,10 +968,14 @@ async def api_profile_apply(name: str, gpu_index: int = 0):
|
||||
_require_gpu(gpu_index)
|
||||
try:
|
||||
errs = await _apply_profile(name, gpu_index)
|
||||
except FileNotFoundError:
|
||||
raise HTTPException(status_code=404, detail=f"Profile '{name}' not found")
|
||||
except FileNotFoundError as err:
|
||||
raise HTTPException(
|
||||
status_code=404, detail=f"Profile '{name}' not found"
|
||||
) from err
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500, detail=f"Failed to load profile: {e}")
|
||||
raise HTTPException(
|
||||
status_code=500, detail=f"Failed to load profile: {e}"
|
||||
) from e
|
||||
if errs:
|
||||
raise HTTPException(status_code=500, detail="; ".join(errs))
|
||||
return {"ok": True}
|
||||
@@ -1169,10 +1199,8 @@ async def api_fans_update(req: FanCurveRequest, gpu_index: int = 0):
|
||||
# Stop existing poller if running
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
with suppress(asyncio.CancelledError):
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
|
||||
g_state["fan_curve"] = curve_data
|
||||
g_state["fan_curve_active"] = True
|
||||
@@ -1189,10 +1217,8 @@ async def api_fans_reset(gpu_index: int = 0):
|
||||
# Stop poller
|
||||
if g_state.get("fan_poller_task"):
|
||||
g_state["fan_poller_task"].cancel()
|
||||
try:
|
||||
with suppress(asyncio.CancelledError):
|
||||
await g_state["fan_poller_task"]
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
g_state["fan_poller_task"] = None
|
||||
|
||||
g_state["fan_curve"] = None
|
||||
@@ -1234,7 +1260,7 @@ async def _reconcile_check(gpu_index: int) -> dict | None:
|
||||
if current is None:
|
||||
return None # Can't read — let the write attempt proceed
|
||||
|
||||
changed = [i for i, (a, b) in enumerate(zip(last, current)) if a != b]
|
||||
changed = [i for i, (a, b) in enumerate(zip(last, current, strict=False)) if a != b]
|
||||
if not changed:
|
||||
return None
|
||||
|
||||
|
||||
Reference in new issue
Block a user