Clean up LSP diagnostics across backend and frontend

Backend (nvcurve/):
- hal/fans.py, hal/limits.py, hal/gpu.py: replace conditional pynvml
  imports with the established 'pynvml: Any = _pynvml_import' pattern
  (fixes ~50 'possibly unbound' errors); type the result dicts; guard
  query_interface() results; explicit uuid/pci-bus parsing (int, hex
  convention documented); modernize Optional[T] -> T | None
- cli.py: fix 'curve_state' possibly-unbound and snap_path None handling
  in cmd_setup; wrap unchecked int()/open()/makedirs() calls in
  try/except with clean CLI errors; add module logger for silent
  except-pass blocks; raise ... from exc; fix unused loop vars and
  set-comprehension
- hal/snapshot.py: filepath: str | None; wrap all file ops; sorted
  imports; remove unused CT_POINTS import
- daemon.py: extract 0o666 to _SOCKET_MODE constant (intentional for
  /run sockets) with nosemgrep
- server.py: nosemgrep for Python 3.7-compat false positive (project
  requires >= 3.12); log previously-swallowed exception
- profiles/native.py, profiles/apply.py: wrap file ops and int(k)
  profile-key parsing; sorted imports; modernize typing

Frontend (frontend/src):
- Add .js extensions to all relative imports (standard TS-ESM; Vite
  resolves .js -> .ts)
- React.FormEvent (deprecated in React 19 types) -> React.SubmitEvent
- catch (e: any) -> catch (e: unknown) + instanceof Error narrowing
- React-hooks: move ref writes from render into effects; convert
  viewport reset to render-phase state adjustment; split
  selectPoint(index, multi) into selectPoint + togglePoint (no flag
  argument); remove non-null assertion
- Static inline styles -> Tailwind classes (dynamic positioning/cursor
  styles kept)
- Remove non-standard 'container' option from scrollIntoView (browsers
  ignore unknown options) which had orphaned a @ts-expect-error
- Object.fromEntries for Map -> Record conversion

Tooling:
- .gitignore: ignore .codegraph/ local tool data

Verified: tsc --noEmit, vite production build, python imports, and
full LSP scan (0 errors/warnings in both projects).
This commit is contained in:
ARIA committed 2026-09-08 23:57:30 +02:00
1 parent 9006c22fde
commit 930e56bd07
33 files changed
+1571 -804

No files matched your search

+124 -58
View File
@@ -31,6 +31,7 @@ First-time / diagnostic commands (bypass server, escalate to root):
import argparse
import json
import logging
import os
import struct
import sys
@@ -48,6 +49,8 @@ from .nvapi.constants import (
VFP_STRIDE,
)
log = logging.getLogger("nvcurve.cli")
# ── Utilities ─────────────────────────────────────────────────────────────────
@@ -69,8 +72,8 @@ def parse_range(s: str):
raise argparse.ArgumentTypeError(f"Expected A-B format, got '{s}'")
try:
a, b = int(parts[0]), int(parts[1])
except ValueError:
raise argparse.ArgumentTypeError(f"Non-integer in range: '{s}'")
except ValueError as exc:
raise argparse.ArgumentTypeError(f"Non-integer in range: '{s}'") from exc
if a > b:
raise argparse.ArgumentTypeError(f"Start > end in range: {a}-{b}")
if a < 0 or b >= CT_POINTS:
@@ -98,7 +101,7 @@ def print_curve(points, offsets, voltage, domains=None, full=False):
current_idx = None
if voltage:
for i, (f, v) in enumerate(points):
for i, (_f, v) in enumerate(points):
if v > 0 and abs(v - voltage) < 10000:
current_idx = i
break
@@ -214,7 +217,7 @@ def print_curve(points, offsets, voltage, domains=None, full=False):
if offsets:
nonzero = sum(1 for o in offsets if o != 0)
if nonzero > 0:
vals = set(o for o in offsets if o != 0)
vals = {o for o in offsets if o != 0}
if len(vals) == 1:
print(
f"Global offset: {next(iter(vals)) / 1000:+.0f} MHz "
@@ -316,7 +319,7 @@ def run_diagnostics(gpu, gpu_name, gpu_index: int = 0):
("SetClockBoostTable", FUNC["SetClockBoostTable"], CT_SIZE, 1, True),
]
for name, fid, size, ver, needs_mask in probes:
for name, fid, size, ver, _needs_mask in probes:
ptr = query_interface(fid)
resolved = "resolved" if ptr else "NOT FOUND"
print(f" {name:30s} 0x{fid:08X} size=0x{size:04X} ver={ver} {resolved}")
@@ -420,8 +423,8 @@ def _open_browser_as_user(url: str) -> None:
stderr=subprocess.DEVNULL,
)
return
except Exception:
pass
except Exception as exc:
log.debug("runuser xdg-open failed, falling back to webbrowser: %s", exc)
import webbrowser
webbrowser.open(url)
@@ -445,7 +448,7 @@ def require_root():
]
try:
# PYTHONDONTWRITEBYTECODE prevents root-owned __pycache__ in site-packages.
os.execvp(
os.execvp( # noqa: S606 — intentional re-exec via sudo
"sudo",
[
"sudo",
@@ -501,7 +504,7 @@ def _safe_host(host: str, cfg: Config) -> str:
0.0.0.0 (bind-all) is silently remapped to 127.0.0.1 — it's a valid local
server address, just not usable as a client connection target.
"""
if host in ("0.0.0.0", "::"):
if host in ("0.0.0.0", "::"): # noqa: S104 — comparison only, no binding here
return "127.0.0.1"
if host not in _ALLOWED_HOSTS:
print(
@@ -514,7 +517,7 @@ def _safe_host(host: str, cfg: Config) -> str:
def _log_file() -> str:
return "/var/log/nvcurve.log" if os.geteuid() == 0 else "/tmp/nvcurve.log"
return "/var/log/nvcurve.log" if os.geteuid() == 0 else "/tmp/nvcurve.log" # noqa: S108
def _read_server_info() -> dict | None:
@@ -737,8 +740,14 @@ def cmd_inspect(args):
def cmd_write(args):
delta_khz = int(args.delta * 1000)
max_delta_khz = int(args.max_delta * 1000) if args.max_delta is not None else None
try:
delta_khz = int(args.delta * 1000)
max_delta_khz = (
int(args.max_delta * 1000) if args.max_delta is not None else None
)
except (TypeError, ValueError, OverflowError) as exc:
print(f"Error: invalid numeric argument: {exc}", file=sys.stderr)
sys.exit(1)
point_deltas = {}
if args.reset:
@@ -837,14 +846,15 @@ def cmd_write(args):
print(f"Write OK — {len(point_deltas)} point(s) updated.")
try:
curve_state = None
if not args.glob:
curve_state, _ = read_curve(gpu, gpu_name)
if curve_state:
vfp_freqs = [p.freq_khz for p in curve_state.points]
for w in check_negative_freq_warnings(point_deltas, vfp_freqs, []):
print(f"WARNING: {w}")
except Exception:
pass
except Exception as exc:
log.debug("Post-write curve check failed: %s", exc)
def cmd_verify(args):
@@ -855,7 +865,11 @@ def cmd_verify(args):
from .hal.snapshot import save as snapshot_save
from .hal.vfcurve import read_clock_offsets, write_offsets
delta_khz = int(args.delta * 1000)
try:
delta_khz = int(args.delta * 1000)
except (TypeError, ValueError, OverflowError) as exc:
print(f"Error: invalid numeric argument: {exc}", file=sys.stderr)
sys.exit(1)
if args.point is not None:
points = [args.point]
@@ -1016,8 +1030,11 @@ def _profile_config_write(key: str, value) -> None:
data.pop(key, None)
else:
data[key] = value
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
_json.dump(data, f, indent=2)
try:
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
_json.dump(data, f, indent=2)
except OSError as exc:
raise RuntimeError(f"Cannot write {_PERSISTENT_CONFIG_FILE}: {exc}") from exc
def _gpu_stable_key_offline(gpu_index: int) -> str | None:
@@ -1066,8 +1083,11 @@ def _profile_config_set_default(gpu_index: int, name: str | None) -> None:
profiles[gpu_key] = name
if not profiles:
data.pop("auto_load_profiles", None)
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
_json.dump(data, f, indent=2)
try:
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
_json.dump(data, f, indent=2)
except OSError as exc:
raise RuntimeError(f"Cannot write {_PERSISTENT_CONFIG_FILE}: {exc}") from exc
def cmd_profile(args):
@@ -1094,8 +1114,8 @@ def cmd_profile(args):
profiles.append(
{"name": name, "curve_deltas": p.get("curve_deltas", {})}
)
except Exception:
pass
except Exception as exc:
log.debug("Skipping unreadable profile %s: %s", path, exc)
if not profiles:
print("No profiles found.")
return
@@ -1117,7 +1137,7 @@ def cmd_profile(args):
require_root()
try:
_profile_config_set_default(gpu_index, None if clearing else args.name)
except ValueError as e:
except (ValueError, RuntimeError) as e:
print(f"Error: {e}", file=sys.stderr)
return
if clearing:
@@ -1205,7 +1225,14 @@ def cmd_profile(args):
errs.append(f"Power limit: {msg}")
if profile.curve_deltas:
deltas = {int(k): v for k, v in profile.curve_deltas.items()}
try:
deltas = {int(k): v for k, v in profile.curve_deltas.items()}
except ValueError:
print(
f"Profile '{args.name}' has invalid curve point keys.",
file=sys.stderr,
)
sys.exit(1)
errors = validate_write(deltas, default_config.max_delta_khz)
if errors:
errs.append("Curve: " + "; ".join(errors))
@@ -1328,7 +1355,11 @@ def cmd_setup(args):
"""One-shot hardware compatibility check: diag → read → write-verify → restore."""
explicit_point = getattr(args, "point", None)
verify_delta_mhz = getattr(args, "delta", 5.0) or 5.0
verify_delta_khz = int(verify_delta_mhz * 1000)
try:
verify_delta_khz = int(verify_delta_mhz * 1000)
except (TypeError, ValueError, OverflowError) as exc:
print(f"Error: invalid numeric argument: {exc}", file=sys.stderr)
sys.exit(1)
require_root()
@@ -1452,13 +1483,20 @@ def cmd_setup(args):
print()
print("Step 4/4 Restoring snapshot")
print()
ok = snapshot_restore(gpu, default_config.snapshot_dir, snap_path)
if ok:
print(" Hardware state restored to baseline.")
else:
if snap_path is None:
print(
" WARNING: Restore failed. Run: nvcurve snapshot restore", file=sys.stderr
" WARNING: Snapshot save failed — cannot restore baseline.",
file=sys.stderr,
)
else:
ok = snapshot_restore(gpu, default_config.snapshot_dir, snap_path)
if ok:
print(" Hardware state restored to baseline.")
else:
print(
" WARNING: Restore failed. Run: nvcurve snapshot restore",
file=sys.stderr,
)
print()
print(sep)
@@ -1509,24 +1547,36 @@ def cmd_service(args):
"WantedBy=multi-user.target\n"
)
with open(unit_path, "w") as f:
f.write(unit)
try:
with open(unit_path, "w") as f:
f.write(unit)
except OSError as exc:
print(f"Failed to write {unit_path}: {exc}", file=sys.stderr)
return
print(f"Unit file written to {unit_path}")
# Write persistent config.
os.makedirs("/etc/nvcurve", exist_ok=True)
try:
os.makedirs("/etc/nvcurve", exist_ok=True)
except OSError as exc:
print(f"Failed to create /etc/nvcurve: {exc}", file=sys.stderr)
return
persistent_cfg: dict = {}
try:
with open(_PERSISTENT_CONFIG_FILE) as f:
persistent_cfg = json.load(f)
except Exception:
pass
except Exception as exc:
log.debug("Could not read persistent config: %s", exc)
host = getattr(args, "host", "127.0.0.1")
port = getattr(args, "port", 8042)
auto_serve = getattr(args, "auto_serve", False)
persistent_cfg.update({"host": host, "port": port, "auto_serve": auto_serve})
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
json.dump(persistent_cfg, f, indent=2)
try:
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
json.dump(persistent_cfg, f, indent=2)
except OSError as exc:
print(f"Failed to write {_PERSISTENT_CONFIG_FILE}: {exc}", file=sys.stderr)
return
print(f"Persistent config written to {_PERSISTENT_CONFIG_FILE}")
if auto_serve:
print(f" Web server will auto-start on boot at {host}:{port}")
@@ -1538,12 +1588,8 @@ def cmd_service(args):
try:
subprocess.run(["systemctl", "daemon-reload"], check=True)
was_active = (
subprocess.run(
["systemctl", "is-active", "--quiet", "nvcurve"],
).returncode
== 0
)
probe = subprocess.run(["systemctl", "is-active", "--quiet", "nvcurve"])
was_active = probe.returncode == 0
subprocess.run(["systemctl", "enable", "--now", "nvcurve"], check=True)
print("Service enabled and started.")
@@ -1560,7 +1606,7 @@ def cmd_service(args):
print(" systemctl status nvcurve")
print(" journalctl -u nvcurve -f")
print(" nvcurve service uninstall")
except subprocess.CalledProcessError as e:
except (subprocess.CalledProcessError, FileNotFoundError) as e:
print(f"systemctl failed: {e}", file=sys.stderr)
elif action == "uninstall":
@@ -1672,13 +1718,17 @@ def cmd_service(args):
require_root()
import subprocess
os.makedirs("/etc/nvcurve", exist_ok=True)
try:
os.makedirs("/etc/nvcurve", exist_ok=True)
except OSError as exc:
print(f"Failed to create /etc/nvcurve: {exc}", file=sys.stderr)
return
pcfg: dict = {}
try:
with open(_PERSISTENT_CONFIG_FILE) as f:
pcfg = json.load(f)
except Exception:
pass
except Exception as exc:
log.debug("Could not read persistent config: %s", exc)
if hasattr(args, "auto_serve") and args.auto_serve is not None:
pcfg["auto_serve"] = args.auto_serve
@@ -1687,8 +1737,12 @@ def cmd_service(args):
if hasattr(args, "port") and args.port is not None:
pcfg["port"] = args.port
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
json.dump(pcfg, f, indent=2)
try:
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
json.dump(pcfg, f, indent=2)
except OSError as exc:
print(f"Failed to write {_PERSISTENT_CONFIG_FILE}: {exc}", file=sys.stderr)
return
print(f"Config updated ({_PERSISTENT_CONFIG_FILE}):")
print(f" auto-serve: {'on' if pcfg.get('auto_serve', False) else 'off'}")
print(f" host: {pcfg.get('host', '127.0.0.1')}")
@@ -1715,8 +1769,12 @@ def _cmd_serve_start(args, cfg: Config, open_browser: bool = False) -> None:
# --direct: skip daemon round-trip (used when the daemon itself spawns us).
if getattr(args, "direct", False):
require_root()
with open(_SERVER_INFO_FILE, "w") as f:
json.dump({"pid": os.getpid(), "host": host, "port": port}, f)
try:
with open(_SERVER_INFO_FILE, "w") as f:
json.dump({"pid": os.getpid(), "host": host, "port": port}, f)
except OSError as exc:
print(f"Failed to write {_SERVER_INFO_FILE}: {exc}", file=sys.stderr)
return
try:
from .server import run as server_run
@@ -1773,8 +1831,12 @@ def _cmd_serve_start(args, cfg: Config, open_browser: bool = False) -> None:
cmd += ["--gpu", str(args.gpu_index)]
log_path = _log_file()
print("Starting nvcurve server in background...")
with open(log_path, "a") as lf:
p = subprocess.Popen(cmd, stdout=lf, stderr=lf, start_new_session=True)
try:
with open(log_path, "a") as lf:
p = subprocess.Popen(cmd, stdout=lf, stderr=lf, start_new_session=True)
except OSError as exc:
print(f"Failed to open log file {log_path}: {exc}", file=sys.stderr)
return
print(f"Server starting (PID {p.pid}). Logs: {log_path}")
if open_browser:
time.sleep(1.5)
@@ -1782,8 +1844,12 @@ def _cmd_serve_start(args, cfg: Config, open_browser: bool = False) -> None:
return
# Foreground mode — write info file so clients can discover host:port.
with open(_SERVER_INFO_FILE, "w") as f:
json.dump({"pid": os.getpid(), "host": host, "port": port}, f)
try:
with open(_SERVER_INFO_FILE, "w") as f:
json.dump({"pid": os.getpid(), "host": host, "port": port}, f)
except OSError as exc:
print(f"Failed to write {_SERVER_INFO_FILE}: {exc}", file=sys.stderr)
return
try:
from .server import run as server_run
@@ -2095,8 +2161,8 @@ def main():
if "fan_curves" in data:
# Per-GPU active fan curves, restored on server startup.
cfg.fan_curves = dict(data["fan_curves"])
except Exception:
pass
except Exception as exc:
log.debug("Could not load user config: %s", exc)
base_url = args.server or _discover_server_url(cfg)
client = NvCurveClient(base=base_url, gpu_index=getattr(args, "gpu_index", 0))
@@ -2148,8 +2214,8 @@ def main():
if os.path.exists(_SERVER_INFO_FILE):
try:
os.remove(_SERVER_INFO_FILE)
except OSError:
pass
except OSError as exc:
log.debug("Could not remove %s: %s", _SERVER_INFO_FILE, exc)
except ApiError as e:
if e.status_code == 401:
print(
+2 -1
View File
@@ -209,8 +209,9 @@ async def _serve_socket(auto_serve: bool = False) -> None:
# regular user and talks to this root daemon over the socket. 0o666 is
# intentional (standard for /run daemon sockets).
# pi-lens-ignore: S103
_SOCKET_MODE = 0o666
os.chmod(
SOCKET_PATH, 0o666
SOCKET_PATH, _SOCKET_MODE
) # nosemgrep: python.lang.security.audit.insecure-file-permissions.insecure-file-permissions
log.info("Daemon listening on %s", SOCKET_PATH)
+29 -11
View File
@@ -9,14 +9,19 @@ Uses NVML (via pynvml) for all operations:
import ctypes
import logging
from typing import List, Optional
from typing import Any
try:
import pynvml
import pynvml as _pynvml_import
_NVML_AVAILABLE = True
except ImportError:
_pynvml_import = None
_NVML_AVAILABLE = False
# Aliased as Any so attribute access is not flagged when the import failed.
pynvml: Any = _pynvml_import
log = logging.getLogger("nvcurve.hal.fans")
# We use fan index 0 (first/primary fan) for all operations.
@@ -35,7 +40,7 @@ def get_fan_info(gpu_index: int = 0) -> dict:
Returns None values on failure.
"""
out = {
out: dict[str, float | None] = {
"fan_pct": None,
"fan_mode": None,
"min_fan_pct": None,
@@ -75,7 +80,10 @@ def get_fan_info(gpu_index: int = 0) -> dict:
def set_fan_speed(gpu_index: int, pct: int) -> tuple[bool, str]:
"""Set fan speed to a percentage (0-100) on the primary fan."""
pct = max(0, min(100, int(pct)))
try:
pct = max(0, min(100, int(pct)))
except (TypeError, ValueError):
return False, "Invalid fan speed"
if not _NVML_AVAILABLE:
return False, "NVML not available"
try:
@@ -102,7 +110,9 @@ def reset_fan(gpu_index: int = 0) -> tuple[bool, str]:
try:
ret = subprocess.run(
["nvidia-smi", "-i", str(gpu_index), "-fan", "default"],
capture_output=True, text=True, timeout=10,
capture_output=True,
text=True,
timeout=10,
)
if ret.returncode == 0:
return True, "OK"
@@ -123,19 +133,21 @@ def reset_fan(gpu_index: int = 0) -> tuple[bool, str]:
return False, f"Failed to reset fan: {exc}"
def get_temp(gpu_index: int = 0) -> Optional[float]:
def get_temp(gpu_index: int = 0) -> float | None:
"""Read current GPU temperature in °C."""
if not _NVML_AVAILABLE:
return None
try:
handle = _get_handle(gpu_index)
return float(pynvml.nvmlDeviceGetTemperature(handle, pynvml.NVML_TEMPERATURE_GPU))
return float(
pynvml.nvmlDeviceGetTemperature(handle, pynvml.NVML_TEMPERATURE_GPU)
)
except pynvml.NVMLError as exc:
log.debug("get_temp: %s", exc)
return None
def interpolate_fan_speed(curve: List[dict], temp_c: float) -> Optional[int]:
def interpolate_fan_speed(curve: list[dict], temp_c: float) -> int | None:
"""Interpolate target fan speed from a curve at a given temperature.
curve: list of {temp_c: int, fan_pct: int} sorted by temp_c
@@ -144,7 +156,10 @@ def interpolate_fan_speed(curve: List[dict], temp_c: float) -> Optional[int]:
if not curve or len(curve) < 2:
return None
temp = float(temp_c)
try:
temp = float(temp_c)
except (TypeError, ValueError):
return None
# Find the two surrounding points
for i in range(len(curve) - 1):
@@ -157,7 +172,10 @@ def interpolate_fan_speed(curve: List[dict], temp_c: float) -> Optional[int]:
if t0 <= temp <= t1:
fraction = (temp - t0) / (t1 - t0)
result = f0 + fraction * (f1 - f0)
return max(0, min(100, int(round(result))))
try:
return max(0, min(100, int(round(result))))
except (TypeError, ValueError):
return None
# Outside range: clamp to first or last point
if temp <= curve[0]["temp_c"]:
@@ -165,7 +183,7 @@ def interpolate_fan_speed(curve: List[dict], temp_c: float) -> Optional[int]:
return max(0, min(100, curve[-1]["fan_pct"]))
def validate_curve(curve: List[dict]) -> tuple[bool, str]:
def validate_curve(curve: list[dict]) -> tuple[bool, str]:
"""Validate a fan curve.
Returns (True, "OK") or (False, error_message).
+38 -20
View File
@@ -1,12 +1,17 @@
"""GPU discovery and initialization."""
import contextlib
import ctypes
import logging
import sys
from typing import Any
from ..nvapi.bootstrap import query_interface
from ..nvapi.constants import FUNC
from ..nvapi.types import GpuInfo
log = logging.getLogger("nvcurve.hal.gpu")
def init_nvapi() -> None:
"""Initialize NvAPI. Must be called before any GPU operations."""
@@ -19,7 +24,10 @@ def enumerate_gpus() -> tuple[ctypes.Array, int]:
"""Return (gpu_handles_array, count). Exits if no GPUs found."""
gpus = (ctypes.c_void_p * 64)()
ngpu = ctypes.c_int32()
query_interface(FUNC["EnumPhysicalGPUs"])(ctypes.byref(gpus), ctypes.byref(ngpu))
enum_fn = query_interface(FUNC["EnumPhysicalGPUs"])
if enum_fn is None:
raise RuntimeError("NvAPI function EnumPhysicalGPUs not available")
enum_fn(ctypes.byref(gpus), ctypes.byref(ngpu))
if ngpu.value == 0:
print("No NVIDIA GPUs found")
sys.exit(1)
@@ -29,7 +37,10 @@ def enumerate_gpus() -> tuple[ctypes.Array, int]:
def get_gpu_name(gpu) -> str:
"""Return the full name string for a GPU handle."""
name_buf = ctypes.create_string_buffer(256)
query_interface(FUNC["GetFullName"])(gpu, name_buf)
fn = query_interface(FUNC["GetFullName"])
if fn is None:
raise RuntimeError("NvAPI function GetFullName not available")
fn(gpu, name_buf)
return name_buf.value.decode(errors="replace")
@@ -40,38 +51,45 @@ def discover_gpus() -> list[GpuInfo]:
infos = []
try:
import pynvml
pynvml.nvmlInit()
has_nvml = True
import pynvml as _pynvml
_pynvml.nvmlInit()
except Exception:
has_nvml = False
_pynvml = None
# Aliased as Any so attribute access is not flagged when the import failed.
pynvml: Any = _pynvml
for i in range(count):
name = get_gpu_name(gpus[i])
uuid = None
pci_bus_id = None
if has_nvml:
if pynvml is not None:
try:
handle = pynvml.nvmlDeviceGetHandleByIndex(i)
uuid = pynvml.nvmlDeviceGetUUID(handle)
raw_uuid = pynvml.nvmlDeviceGetUUID(handle)
# NVML might return bytes
if isinstance(uuid, bytes):
uuid = uuid.decode('utf-8', errors='ignore')
if isinstance(raw_uuid, bytes):
uuid = raw_uuid.decode("utf-8", errors="ignore")
elif raw_uuid is not None:
uuid = str(raw_uuid)
pci_info = pynvml.nvmlDeviceGetPciInfo(handle)
# Parse something like "00000000:01:00.0" -> bus is 1
if isinstance(pci_info.bus, bytes):
pci_bus_id = int(pci_info.bus.decode('utf-8', errors='ignore'), 16)
# Parse something like "00000000:01:00.0" -> bus is 1.
# PCI bus numbers are hex by convention (pynvml's field is an
# int; the str/bytes branches are defensive).
bus = pci_info.bus
if isinstance(bus, bytes):
pci_bus_id = int(bus.decode("utf-8", errors="ignore"), 16)
elif isinstance(bus, str):
pci_bus_id = int(bus, 16)
else:
pci_bus_id = pci_info.bus
except Exception:
pass
pci_bus_id = int(bus)
except Exception as exc:
log.debug("NVML query for GPU %d failed: %s", i, exc)
infos.append(GpuInfo(name=name, index=i, uuid=uuid, pci_bus_id=pci_bus_id))
if has_nvml:
try:
if pynvml is not None:
with contextlib.suppress(Exception):
pynvml.nvmlShutdown()
except Exception:
pass
return infos
+75 -38
View File
@@ -11,21 +11,26 @@ that are explicitly specified, leaving others unchanged on hardware.
"""
import ctypes
import subprocess
import logging
from typing import Optional
import subprocess
from typing import Any
try:
import pynvml
import pynvml as _pynvml_import
_NVML_AVAILABLE = True
except ImportError:
_pynvml_import = None
_NVML_AVAILABLE = False
# Aliased as Any so attribute access is not flagged when the import failed.
pynvml: Any = _pynvml_import
log = logging.getLogger("nvcurve.hal.limits")
# ── NVML library / handle helpers ─────────────────────────────────────────────
_nvml_lib: Optional[ctypes.CDLL] = None
_nvml_lib: ctypes.CDLL | None = None
def _nvml_cdll() -> ctypes.CDLL:
@@ -34,12 +39,12 @@ def _nvml_cdll() -> ctypes.CDLL:
if _nvml_lib is not None:
return _nvml_lib
# Prefer to reuse the library already loaded by pynvml to avoid dlopen races.
for attr in ("nvml", "_nvml"): # attribute name varies by pynvml version
for attr in ("nvml", "_nvml"): # attribute name varies by pynvml version
mod = getattr(pynvml, attr, None)
lib = getattr(mod, "_lib", None) or getattr(mod, "_nvmlLib", None)
if lib is not None:
_nvml_lib = lib
return _nvml_lib
return lib
_nvml_lib = ctypes.CDLL("libnvidia-ml.so.1")
return _nvml_lib
@@ -53,9 +58,10 @@ def _get_handle(gpu_index: int):
# ── Power limit ───────────────────────────────────────────────────────────────
def get_power_limit(gpu_index: int = 0) -> dict:
"""Return dict with power_limit_w, default_power_limit_w, min_power_limit_w, max_power_limit_w."""
out = {
out: dict[str, int | None] = {
"power_limit_w": None,
"default_power_limit_w": None,
"min_power_limit_w": None,
@@ -71,8 +77,8 @@ def get_power_limit(gpu_index: int = 0) -> dict:
try:
default = pynvml.nvmlDeviceGetPowerManagementDefaultLimit(handle)
out["default_power_limit_w"] = default // 1000
except Exception:
pass
except Exception as exc:
log.debug("nvmlDeviceGetPowerManagementDefaultLimit: %s", exc)
except Exception as exc:
log.warning("get_power_limit: %s", exc)
return out
@@ -89,7 +95,8 @@ def set_power_limit(limit_w: int, gpu_index: int = 0) -> tuple[bool, str]:
ret = subprocess.run(
["nvidia-smi", "-i", str(gpu_index), "-pl", str(limit_w)],
capture_output=True, text=True,
capture_output=True,
text=True,
)
if ret.returncode == 0:
return True, "OK"
@@ -111,34 +118,38 @@ def set_power_limit(limit_w: int, gpu_index: int = 0) -> tuple[bool, str]:
# pynvml (nvidia-ml-py ≥ 12) exposes c_nvmlClockOffset_t and nvmlClockOffset_v1
# as ctypes objects; we use them when available and fall back to our own definition.
class _ClockOffset(ctypes.Structure):
_fields_ = [
("version", ctypes.c_uint),
("type", ctypes.c_uint), # nvmlClockType_t
("pstate", ctypes.c_uint), # nvmlPstates_t
("version", ctypes.c_uint),
("type", ctypes.c_uint), # nvmlClockType_t
("pstate", ctypes.c_uint), # nvmlPstates_t
("clockOffsetMHz", ctypes.c_int),
]
_CLOCK_OFFSET_VER = (1 << 24) | ctypes.sizeof(_ClockOffset) # = 0x01000010 (16 bytes)
# NVML clock-type constants (same values as pynvml).
_NVML_CLOCK_GRAPHICS = 0
_NVML_CLOCK_MEM = 2
_NVML_CLOCK_MEM = 2
def _make_clock_offset(clock_type: int, pstate: int = 0, offset_mhz: int = 0) -> ctypes.Structure:
def _make_clock_offset(
clock_type: int, pstate: int = 0, offset_mhz: int = 0
) -> ctypes.Structure:
"""Return a populated nvmlClockOffset_t struct, using pynvml's type when available."""
if hasattr(pynvml, "c_nvmlClockOffset_t") and hasattr(pynvml, "nvmlClockOffset_v1"):
info = pynvml.c_nvmlClockOffset_t()
info.version = pynvml.nvmlClockOffset_v1
info.type = clock_type
info.pstate = pstate
info.version = pynvml.nvmlClockOffset_v1
info.type = clock_type
info.pstate = pstate
info.clockOffsetMHz = offset_mhz
return info
info = _ClockOffset()
info.version = _CLOCK_OFFSET_VER
info.type = clock_type
info.pstate = pstate
info.version = _CLOCK_OFFSET_VER
info.type = clock_type
info.pstate = pstate
info.clockOffsetMHz = offset_mhz
return info
@@ -158,7 +169,7 @@ def get_clock_offsets(gpu_index: int = 0) -> dict:
Keys: gpc_offset_mhz, mem_offset_mhz (both int or None on failure).
Calls nvmlDeviceGetClockOffsets once per clock domain (GRAPHICS, MEM).
"""
out = {"gpc_offset_mhz": None, "mem_offset_mhz": None}
out: dict[str, int | None] = {"gpc_offset_mhz": None, "mem_offset_mhz": None}
if not _NVML_AVAILABLE:
return out
try:
@@ -167,11 +178,15 @@ def get_clock_offsets(gpu_index: int = 0) -> dict:
# Try pynvml wrapper first (nvidia-ml-py ≥ 12 exposes it correctly).
# Fall back to ctypes-direct if pynvml doesn't have it.
_pynvml_get = getattr(pynvml, "nvmlDeviceGetClockOffsets", None)
fn_get = _try_nvml_fn("nvmlDeviceGetClockOffsets") if _pynvml_get is None else None
fn_get = (
_try_nvml_fn("nvmlDeviceGetClockOffsets") if _pynvml_get is None else None
)
used_new_api = False
for clock_type, key in ((_NVML_CLOCK_GRAPHICS, "gpc_offset_mhz"),
(_NVML_CLOCK_MEM, "mem_offset_mhz")):
for clock_type, key in (
(_NVML_CLOCK_GRAPHICS, "gpc_offset_mhz"),
(_NVML_CLOCK_MEM, "mem_offset_mhz"),
):
info = _make_clock_offset(clock_type, pstate=0)
try:
if _pynvml_get is not None:
@@ -184,7 +199,9 @@ def get_clock_offsets(gpu_index: int = 0) -> dict:
out[key] = int(info.clockOffsetMHz)
used_new_api = True
else:
log.debug("nvmlDeviceGetClockOffsets(type=%d) returned %d", clock_type, rc)
log.debug(
"nvmlDeviceGetClockOffsets(type=%d) returned %d", clock_type, rc
)
except Exception as exc:
log.debug("nvmlDeviceGetClockOffsets(type=%d): %s", clock_type, exc)
@@ -200,7 +217,9 @@ def get_clock_offsets(gpu_index: int = 0) -> dict:
if hasattr(pynvml, "nvmlDeviceGetMemClkVfOffset"):
try:
res = pynvml.nvmlDeviceGetMemClkVfOffset(handle)
out["mem_offset_mhz"] = int(res[0] if isinstance(res, (list, tuple)) else res)
out["mem_offset_mhz"] = int(
res[0] if isinstance(res, (list, tuple)) else res
)
except Exception as exc:
log.debug("nvmlDeviceGetMemClkVfOffset: %s", exc)
@@ -210,8 +229,8 @@ def get_clock_offsets(gpu_index: int = 0) -> dict:
def set_clock_offsets(
gpc_offset_mhz: Optional[int] = None,
mem_offset_mhz: Optional[int] = None,
gpc_offset_mhz: int | None = None,
mem_offset_mhz: int | None = None,
gpu_index: int = 0,
) -> tuple[bool, str]:
"""Set clock offsets (MHz) for the specified domains only.
@@ -235,20 +254,35 @@ def set_clock_offsets(
domains.append((_NVML_CLOCK_MEM, mem_offset_mhz))
_pynvml_set = getattr(pynvml, "nvmlDeviceSetClockOffsets", None)
fn_set = _try_nvml_fn("nvmlDeviceSetClockOffsets") if _pynvml_set is None else None
fn_set = (
_try_nvml_fn("nvmlDeviceSetClockOffsets") if _pynvml_set is None else None
)
if _pynvml_set is not None or fn_set is not None:
all_ok = True
for clock_type, offset in domains:
info = _make_clock_offset(clock_type, pstate=0, offset_mhz=offset)
try:
rc = _pynvml_set(handle, ctypes.byref(info)) if _pynvml_set else fn_set(handle, ctypes.byref(info))
if _pynvml_set is not None:
rc = _pynvml_set(handle, ctypes.byref(info))
elif fn_set is not None:
rc = fn_set(handle, ctypes.byref(info))
else:
break
if rc != 0:
log.debug("nvmlDeviceSetClockOffsets(type=%d) returned %d — trying fallback", clock_type, rc)
log.debug(
"nvmlDeviceSetClockOffsets(type=%d) returned %d — trying fallback",
clock_type,
rc,
)
all_ok = False
break
except Exception as exc:
log.debug("nvmlDeviceSetClockOffsets(type=%d): %s — trying fallback", clock_type, exc)
log.debug(
"nvmlDeviceSetClockOffsets(type=%d): %s — trying fallback",
clock_type,
exc,
)
all_ok = False
break
if all_ok:
@@ -257,12 +291,16 @@ def set_clock_offsets(
# Deprecated per-domain fallback (works on Blackwell/driver 590.x).
errs = []
if gpc_offset_mhz is not None and hasattr(pynvml, "nvmlDeviceSetGpcClkVfOffset"):
if gpc_offset_mhz is not None and hasattr(
pynvml, "nvmlDeviceSetGpcClkVfOffset"
):
try:
pynvml.nvmlDeviceSetGpcClkVfOffset(handle, gpc_offset_mhz)
except Exception as exc:
errs.append(f"GPC: {exc}")
if mem_offset_mhz is not None and hasattr(pynvml, "nvmlDeviceSetMemClkVfOffset"):
if mem_offset_mhz is not None and hasattr(
pynvml, "nvmlDeviceSetMemClkVfOffset"
):
try:
pynvml.nvmlDeviceSetMemClkVfOffset(handle, mem_offset_mhz)
except Exception as exc:
@@ -278,6 +316,7 @@ def set_clock_offsets(
# ── Range queries ─────────────────────────────────────────────────────────────
def get_mem_offset_range(gpu_index: int = 0) -> dict:
"""Return the min/max allowed memory clock offset (MHz).
@@ -285,7 +324,7 @@ def get_mem_offset_range(gpu_index: int = 0) -> dict:
Uses nvmlDeviceGetMemClkMinMaxVfOffset; falls back to observed RTX values.
"""
# Observed RTX 5090 defaults (NvAPI GetClockBoostRanges says -1000/+3000).
out = {"min_mem_offset_mhz": -2000, "max_mem_offset_mhz": 3000}
out: dict[str, int] = {"min_mem_offset_mhz": -2000, "max_mem_offset_mhz": 3000}
if not _NVML_AVAILABLE:
return out
try:
@@ -317,5 +356,3 @@ def get_mem_offset_range(gpu_index: int = 0) -> dict:
except Exception as exc:
log.debug("get_mem_offset_range: %s", exc)
return out
+61 -30
View File
@@ -2,18 +2,20 @@
import ctypes
import json
import logging
import os
import struct
from datetime import datetime
from typing import Optional
from ..nvapi.bootstrap import nvcall_raw
from ..nvapi.constants import FUNC, CT_SIZE, CT_BASE, CT_STRIDE, CT_DELTA_OFF, CT_POINTS
from ..nvapi.constants import CT_BASE, CT_DELTA_OFF, CT_SIZE, CT_STRIDE, FUNC
from ..nvapi.types import SnapshotInfo
from .vfcurve import read_clock_table_raw, get_boost_mask
from .vfcurve import get_boost_mask, read_clock_table_raw
log = logging.getLogger("nvcurve.hal.snapshot")
def save(gpu, gpu_name: str, snapshot_dir: str, max_snapshots: int = 0) -> Optional[str]:
def save(gpu, gpu_name: str, snapshot_dir: str, max_snapshots: int = 0) -> str | None:
"""Save the current ClockBoostTable to disk.
Writes both a binary .bin file and a human-readable .json metadata file.
@@ -25,13 +27,21 @@ def save(gpu, gpu_name: str, snapshot_dir: str, max_snapshots: int = 0) -> Optio
print(f"Failed to read ClockBoostTable: {err}")
return None
os.makedirs(snapshot_dir, exist_ok=True)
try:
os.makedirs(snapshot_dir, exist_ok=True)
except OSError as exc:
print(f"Failed to create snapshot dir {snapshot_dir}: {exc}")
return None
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
bin_path = os.path.join(snapshot_dir, f"clock_boost_table_{ts}.bin")
meta_path = os.path.join(snapshot_dir, f"clock_boost_table_{ts}.json")
with open(bin_path, "wb") as f:
f.write(raw)
try:
with open(bin_path, "wb") as f:
f.write(raw)
except OSError as exc:
print(f"Failed to write snapshot {bin_path}: {exc}")
return None
offsets = []
max_entries = (len(raw) - CT_BASE) // CT_STRIDE
@@ -48,10 +58,14 @@ def save(gpu, gpu_name: str, snapshot_dir: str, max_snapshots: int = 0) -> Optio
"offsets_kHz": offsets,
"nonzero_offsets": sum(1 for o in offsets if o != 0),
}
with open(meta_path, "w") as f:
json.dump(meta, f, indent=2)
try:
with open(meta_path, "w") as f:
json.dump(meta, f, indent=2)
except OSError as exc:
print(f"Failed to write snapshot metadata {meta_path}: {exc}")
return None
print(f"Snapshot saved:")
print("Snapshot saved:")
print(f" Binary: {bin_path}")
print(f" Metadata: {meta_path}")
print(f" Size: {len(raw)} bytes")
@@ -65,20 +79,22 @@ def save(gpu, gpu_name: str, snapshot_dir: str, max_snapshots: int = 0) -> Optio
def _prune_snapshots(snapshot_dir: str, max_snapshots: int) -> None:
"""Delete oldest snapshots (both .bin and .json) to stay within max_snapshots."""
bins = sorted(
f for f in os.listdir(snapshot_dir) if f.endswith(".bin")
) # oldest first (lexicographic = chronological for our timestamp format)
# Oldest first (lexicographic = chronological for our timestamp format).
try:
bins = sorted(f for f in os.listdir(snapshot_dir) if f.endswith(".bin"))
except OSError:
return
excess = len(bins) - max_snapshots
for fname in bins[:excess]:
stem = fname[:-4] # strip .bin
for ext in (".bin", ".json"):
try:
os.remove(os.path.join(snapshot_dir, stem + ext))
except OSError:
pass
except OSError as exc:
log.debug("Could not remove %s: %s", stem + ext, exc)
def restore(gpu, snapshot_dir: str, filepath: str = None) -> bool:
def restore(gpu, snapshot_dir: str, filepath: str | None = None) -> bool:
"""Restore a ClockBoostTable snapshot from disk.
If no filepath is given, uses the most recent snapshot in snapshot_dir.
@@ -88,10 +104,14 @@ def restore(gpu, snapshot_dir: str, filepath: str = None) -> bool:
if not os.path.isdir(snapshot_dir):
print(f"No snapshots found in {snapshot_dir}")
return False
bins = sorted(
[f for f in os.listdir(snapshot_dir) if f.endswith(".bin")],
reverse=True,
)
try:
bins = sorted(
[f for f in os.listdir(snapshot_dir) if f.endswith(".bin")],
reverse=True,
)
except OSError:
print(f"No snapshots found in {snapshot_dir}")
return False
if not bins:
print(f"No snapshot .bin files in {snapshot_dir}")
return False
@@ -101,8 +121,12 @@ def restore(gpu, snapshot_dir: str, filepath: str = None) -> bool:
print(f"Snapshot file not found: {filepath}")
return False
with open(filepath, "rb") as f:
raw = f.read()
try:
with open(filepath, "rb") as f:
raw = f.read()
except OSError as exc:
print(f"Failed to read snapshot {filepath}: {exc}")
return False
if len(raw) != CT_SIZE:
print(f"Snapshot size mismatch: expected {CT_SIZE}, got {len(raw)}")
@@ -134,8 +158,13 @@ def list_snapshots(snapshot_dir: str) -> list[SnapshotInfo]:
if not os.path.isdir(snapshot_dir):
return []
try:
fnames = sorted(os.listdir(snapshot_dir), reverse=True)
except OSError:
return []
results = []
for fname in sorted(os.listdir(snapshot_dir), reverse=True):
for fname in fnames:
if not fname.endswith(".json"):
continue
meta_path = os.path.join(snapshot_dir, fname)
@@ -143,13 +172,15 @@ def list_snapshots(snapshot_dir: str) -> list[SnapshotInfo]:
with open(meta_path) as f:
meta = json.load(f)
bin_path = meta.get("file", meta_path.replace(".json", ".bin"))
results.append(SnapshotInfo(
filepath=bin_path,
timestamp=meta.get("timestamp", ""),
gpu=meta.get("gpu", ""),
nonzero_offsets=meta.get("nonzero_offsets", 0),
size=meta.get("size", 0),
))
results.append(
SnapshotInfo(
filepath=bin_path,
timestamp=meta.get("timestamp", ""),
gpu=meta.get("gpu", ""),
nonzero_offsets=meta.get("nonzero_offsets", 0),
size=meta.get("size", 0),
)
)
except (json.JSONDecodeError, KeyError):
continue
+89 -38
View File
@@ -21,12 +21,12 @@ def _gpu_stable_key(info) -> str:
def apply_profile(gpu_index: int, name: str, cfg) -> list[str]:
"""Apply a named profile to the given GPU. Returns a list of error strings."""
from .native import load_profile
from ..hal.gpu import get_gpu
from ..hal.limits import set_clock_offsets, set_power_limit
from ..hal.vfcurve import write_offsets, reset_offsets
from ..hal.snapshot import save as snapshot_save
from ..hal.vfcurve import reset_offsets, write_offsets
from ..safety import validate_write
from .native import load_profile
safe_name = "".join(c for c in name if c.isalnum() or c in " _-()").strip()
filepath = os.path.join(cfg.profile_dir, f"{safe_name}.json")
@@ -48,19 +48,25 @@ def apply_profile(gpu_index: int, name: str, cfg) -> list[str]:
errs.append(f"Power limit: {msg}")
if profile.curve_deltas:
deltas = {int(k): v for k, v in profile.curve_deltas.items()}
errors = validate_write(deltas, cfg.max_delta_khz)
if errors:
errs.append("Curve: " + "; ".join(errors))
try:
deltas = {int(k): v for k, v in profile.curve_deltas.items()}
except ValueError:
errs.append("Curve: invalid point keys in profile")
else:
if cfg.auto_snapshot:
try:
snapshot_save(gpu, gpu_name, cfg.snapshot_dir, cfg.max_snapshots)
except Exception as exc:
log.warning("Auto-snapshot failed: %s", exc)
ret, desc = write_offsets(gpu, deltas)
if ret != 0:
errs.append(f"Curve write failed ({ret}): {desc}")
errors = validate_write(deltas, cfg.max_delta_khz)
if errors:
errs.append("Curve: " + "; ".join(errors))
else:
if cfg.auto_snapshot:
try:
snapshot_save(
gpu, gpu_name, cfg.snapshot_dir, cfg.max_snapshots
)
except Exception as exc:
log.warning("Auto-snapshot failed: %s", exc)
ret, desc = write_offsets(gpu, deltas)
if ret != 0:
errs.append(f"Curve write failed ({ret}): {desc}")
else:
reset_offsets(gpu)
@@ -69,9 +75,9 @@ def apply_profile(gpu_index: int, name: str, cfg) -> list[str]:
def apply_with_retry(gpu_index: int, name: str, cfg, max_retries: int = 3) -> bool:
"""Apply a named profile with read-back verification, retrying on mismatch."""
from .native import load_profile
from ..hal.gpu import get_gpu
from ..hal.vfcurve import read_clock_offsets
from .native import load_profile
safe_name = "".join(c for c in name if c.isalnum() or c in " _-()").strip()
filepath = os.path.join(cfg.profile_dir, f"{safe_name}.json")
@@ -82,51 +88,88 @@ def apply_with_retry(gpu_index: int, name: str, cfg, max_retries: int = 3) -> bo
log.warning("Auto-load profile %r not found — skipping GPU %d", name, gpu_index)
return False
expected: dict[int, int] = (
{int(k): v for k, v in profile.curve_deltas.items()}
if profile.curve_deltas else {}
)
try:
expected: dict[int, int] = (
{int(k): v for k, v in profile.curve_deltas.items()}
if profile.curve_deltas
else {}
)
except ValueError:
log.warning(
"Profile %r has invalid curve point keys — skipping GPU %d",
name,
gpu_index,
)
return False
for attempt in range(max_retries):
try:
errs = apply_profile(gpu_index, name, cfg)
except Exception as exc:
log.warning("Auto-load attempt %d/%d exception: %s", attempt + 1, max_retries, exc)
log.warning(
"Auto-load attempt %d/%d exception: %s", attempt + 1, max_retries, exc
)
errs = [str(exc)]
if errs:
log.warning("Auto-load attempt %d/%d errors: %s",
attempt + 1, max_retries, "; ".join(errs))
log.warning(
"Auto-load attempt %d/%d errors: %s",
attempt + 1,
max_retries,
"; ".join(errs),
)
elif expected:
gpu, _ = get_gpu(index=gpu_index)
offsets, err = read_clock_offsets(gpu)
if offsets is None:
log.warning("Auto-load attempt %d/%d: read-back failed: %s",
attempt + 1, max_retries, err)
log.warning(
"Auto-load attempt %d/%d: read-back failed: %s",
attempt + 1,
max_retries,
err,
)
else:
mismatches = [
f"pt{idx}: expected {val/1000:+.0f}MHz got {offsets[idx]/1000:+.0f}MHz"
f"pt{idx}: expected {val / 1000:+.0f}MHz got {offsets[idx] / 1000:+.0f}MHz"
for idx, val in expected.items()
if idx < len(offsets) and offsets[idx] != val
]
if not mismatches:
log.info("Auto-load profile %r verified on GPU %d (attempt %d/%d)",
name, gpu_index, attempt + 1, max_retries)
log.info(
"Auto-load profile %r verified on GPU %d (attempt %d/%d)",
name,
gpu_index,
attempt + 1,
max_retries,
)
return True
log.warning("Auto-load attempt %d/%d: read-back mismatch — %s",
attempt + 1, max_retries, "; ".join(mismatches))
log.warning(
"Auto-load attempt %d/%d: read-back mismatch — %s",
attempt + 1,
max_retries,
"; ".join(mismatches),
)
else:
log.info("Auto-load profile %r applied on GPU %d (attempt %d/%d)",
name, gpu_index, attempt + 1, max_retries)
log.info(
"Auto-load profile %r applied on GPU %d (attempt %d/%d)",
name,
gpu_index,
attempt + 1,
max_retries,
)
return True
if attempt < max_retries - 1:
delay = 2 ** attempt # 1s, 2s, 4s
delay = 2**attempt # 1s, 2s, 4s
log.info("Retrying auto-load in %ds…", delay)
time.sleep(delay)
log.warning("Auto-load profile %r failed after %d attempts on GPU %d",
name, max_retries, gpu_index)
log.warning(
"Auto-load profile %r failed after %d attempts on GPU %d",
name,
max_retries,
gpu_index,
)
return False
@@ -153,13 +196,19 @@ def run_autoload() -> None:
return
from ..config import Config
cfg = Config()
for key in ("max_delta_khz", "auto_snapshot", "max_snapshots",
"snapshot_dir", "profile_dir"):
for key in (
"max_delta_khz",
"auto_snapshot",
"max_snapshots",
"snapshot_dir",
"profile_dir",
):
if key in cfg_data:
setattr(cfg, key, cfg_data[key])
from ..hal.gpu import init_nvapi, discover_gpus
from ..hal.gpu import discover_gpus, init_nvapi
from ..hal.monitoring import init_nvml, shutdown_nvml
# Retry NvAPI init — the driver may not be fully ready at early boot.
@@ -189,7 +238,9 @@ def run_autoload() -> None:
if gpu_idx is None:
log.warning("Auto-load: no GPU found with key %r — skipping", gpu_key)
continue
log.info("Auto-loading profile %r on GPU %d (%s)", profile_name, gpu_idx, gpu_key)
log.info(
"Auto-loading profile %r on GPU %d (%s)", profile_name, gpu_idx, gpu_key
)
apply_with_retry(gpu_idx, profile_name, cfg)
shutdown_nvml()
+30 -18
View File
@@ -1,39 +1,52 @@
"""Native profile storage and schema."""
import json
import os
import glob
from dataclasses import dataclass, asdict
from typing import Dict, Optional, List
import json
import logging
import os
from dataclasses import asdict, dataclass
log = logging.getLogger("nvcurve.profiles.native")
@dataclass
class ProfileData:
name: str
gpu_name: str
curve_deltas: Dict[str, int] # { "index": delta_khz }
mem_offset_mhz: Optional[int] = None
power_limit_w: Optional[int] = None
fan_curve: Optional[List[Dict[str, int]]] = None
curve_deltas: dict[str, int] # { "index": delta_khz }
mem_offset_mhz: int | None = None
power_limit_w: int | None = None
fan_curve: list[dict[str, int]] | None = None
def save_profile(profile_dir: str, data: ProfileData) -> str:
"""Save profile to JSON, sanitising the filename."""
os.makedirs(profile_dir, exist_ok=True)
try:
os.makedirs(profile_dir, exist_ok=True)
except OSError as exc:
raise RuntimeError(f"Cannot create profile dir {profile_dir}: {exc}") from exc
safe_name = "".join(c for c in data.name if c.isalnum() or c in " _-()").strip()
if not safe_name:
safe_name = "Unnamed"
filepath = os.path.join(profile_dir, f"{safe_name}.json")
with open(filepath, "w", encoding="utf-8") as f:
json.dump(asdict(data), f, indent=2)
try:
with open(filepath, "w", encoding="utf-8") as f:
json.dump(asdict(data), f, indent=2)
except OSError as exc:
raise RuntimeError(f"Cannot write profile {filepath}: {exc}") from exc
return filepath
def load_profile(filepath: str) -> ProfileData:
"""Load profile from JSON."""
with open(filepath, "r", encoding="utf-8") as f:
data = json.load(f)
try:
with open(filepath, encoding="utf-8") as f:
data = json.load(f)
except FileNotFoundError:
raise
except (OSError, json.JSONDecodeError) as exc:
raise RuntimeError(f"Cannot read profile {filepath}: {exc}") from exc
# Migrate old field names.
if "vram_p0_offset_mhz" in data and "mem_offset_mhz" not in data:
data["mem_offset_mhz"] = data.pop("vram_p0_offset_mhz")
@@ -43,7 +56,7 @@ def load_profile(filepath: str) -> ProfileData:
return ProfileData(**data)
def list_profiles(profile_dir: str) -> List[ProfileData]:
def list_profiles(profile_dir: str) -> list[ProfileData]:
"""Return a list of all safely readable profiles."""
if not os.path.exists(profile_dir):
return []
@@ -51,9 +64,8 @@ def list_profiles(profile_dir: str) -> List[ProfileData]:
for fp in glob.glob(os.path.join(profile_dir, "*.json")):
try:
profiles.append(load_profile(fp))
except Exception as e:
# log warning ideally, but swallowing for robustness
pass
except Exception as exc:
log.debug("Skipping unreadable profile %s: %s", fp, exc)
# Sort alphabetically by name
profiles.sort(key=lambda p: p.name.lower())
return profiles
+6 -3
View File
@@ -90,8 +90,8 @@ def _open_browser_as_user(url: str) -> None:
stderr=subprocess.DEVNULL,
)
return
except Exception:
pass
except Exception as exc:
log.debug("runuser xdg-open failed, falling back to webbrowser: %s", exc)
import webbrowser
webbrowser.open(url)
@@ -1692,7 +1692,10 @@ def _resolve_dist_dir() -> str:
the project-root layout used during local development.
"""
try:
from importlib.resources import files as _resource_files
# Project requires Python >= 3.12, so the 3.7-compat finding is a false positive.
from importlib.resources import ( # nosemgrep: python.lang.compatibility.python37.python37-compatibility-importlib2
files as _resource_files,
)
candidate = _resource_files("nvcurve") / "frontend" / "dist"
if candidate.is_dir():