Clean up LSP diagnostics across backend and frontend
Backend (nvcurve/): - hal/fans.py, hal/limits.py, hal/gpu.py: replace conditional pynvml imports with the established 'pynvml: Any = _pynvml_import' pattern (fixes ~50 'possibly unbound' errors); type the result dicts; guard query_interface() results; explicit uuid/pci-bus parsing (int, hex convention documented); modernize Optional[T] -> T | None - cli.py: fix 'curve_state' possibly-unbound and snap_path None handling in cmd_setup; wrap unchecked int()/open()/makedirs() calls in try/except with clean CLI errors; add module logger for silent except-pass blocks; raise ... from exc; fix unused loop vars and set-comprehension - hal/snapshot.py: filepath: str | None; wrap all file ops; sorted imports; remove unused CT_POINTS import - daemon.py: extract 0o666 to _SOCKET_MODE constant (intentional for /run sockets) with nosemgrep - server.py: nosemgrep for Python 3.7-compat false positive (project requires >= 3.12); log previously-swallowed exception - profiles/native.py, profiles/apply.py: wrap file ops and int(k) profile-key parsing; sorted imports; modernize typing Frontend (frontend/src): - Add .js extensions to all relative imports (standard TS-ESM; Vite resolves .js -> .ts) - React.FormEvent (deprecated in React 19 types) -> React.SubmitEvent - catch (e: any) -> catch (e: unknown) + instanceof Error narrowing - React-hooks: move ref writes from render into effects; convert viewport reset to render-phase state adjustment; split selectPoint(index, multi) into selectPoint + togglePoint (no flag argument); remove non-null assertion - Static inline styles -> Tailwind classes (dynamic positioning/cursor styles kept) - Remove non-standard 'container' option from scrollIntoView (browsers ignore unknown options) which had orphaned a @ts-expect-error - Object.fromEntries for Map -> Record conversion Tooling: - .gitignore: ignore .codegraph/ local tool data Verified: tsc --noEmit, vite production build, python imports, and full LSP scan (0 errors/warnings in both projects).
This commit is contained in:
1 parent
9006c22fde
commit
930e56bd07
33 files changed
+1571
-804
No files matched your search
+124
-58
@@ -31,6 +31,7 @@ First-time / diagnostic commands (bypass server, escalate to root):
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import struct
|
||||
import sys
|
||||
@@ -48,6 +49,8 @@ from .nvapi.constants import (
|
||||
VFP_STRIDE,
|
||||
)
|
||||
|
||||
log = logging.getLogger("nvcurve.cli")
|
||||
|
||||
# ── Utilities ─────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@@ -69,8 +72,8 @@ def parse_range(s: str):
|
||||
raise argparse.ArgumentTypeError(f"Expected A-B format, got '{s}'")
|
||||
try:
|
||||
a, b = int(parts[0]), int(parts[1])
|
||||
except ValueError:
|
||||
raise argparse.ArgumentTypeError(f"Non-integer in range: '{s}'")
|
||||
except ValueError as exc:
|
||||
raise argparse.ArgumentTypeError(f"Non-integer in range: '{s}'") from exc
|
||||
if a > b:
|
||||
raise argparse.ArgumentTypeError(f"Start > end in range: {a}-{b}")
|
||||
if a < 0 or b >= CT_POINTS:
|
||||
@@ -98,7 +101,7 @@ def print_curve(points, offsets, voltage, domains=None, full=False):
|
||||
|
||||
current_idx = None
|
||||
if voltage:
|
||||
for i, (f, v) in enumerate(points):
|
||||
for i, (_f, v) in enumerate(points):
|
||||
if v > 0 and abs(v - voltage) < 10000:
|
||||
current_idx = i
|
||||
break
|
||||
@@ -214,7 +217,7 @@ def print_curve(points, offsets, voltage, domains=None, full=False):
|
||||
if offsets:
|
||||
nonzero = sum(1 for o in offsets if o != 0)
|
||||
if nonzero > 0:
|
||||
vals = set(o for o in offsets if o != 0)
|
||||
vals = {o for o in offsets if o != 0}
|
||||
if len(vals) == 1:
|
||||
print(
|
||||
f"Global offset: {next(iter(vals)) / 1000:+.0f} MHz "
|
||||
@@ -316,7 +319,7 @@ def run_diagnostics(gpu, gpu_name, gpu_index: int = 0):
|
||||
("SetClockBoostTable", FUNC["SetClockBoostTable"], CT_SIZE, 1, True),
|
||||
]
|
||||
|
||||
for name, fid, size, ver, needs_mask in probes:
|
||||
for name, fid, size, ver, _needs_mask in probes:
|
||||
ptr = query_interface(fid)
|
||||
resolved = "resolved" if ptr else "NOT FOUND"
|
||||
print(f" {name:30s} 0x{fid:08X} size=0x{size:04X} ver={ver} {resolved}")
|
||||
@@ -420,8 +423,8 @@ def _open_browser_as_user(url: str) -> None:
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
return
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
log.debug("runuser xdg-open failed, falling back to webbrowser: %s", exc)
|
||||
import webbrowser
|
||||
|
||||
webbrowser.open(url)
|
||||
@@ -445,7 +448,7 @@ def require_root():
|
||||
]
|
||||
try:
|
||||
# PYTHONDONTWRITEBYTECODE prevents root-owned __pycache__ in site-packages.
|
||||
os.execvp(
|
||||
os.execvp( # noqa: S606 — intentional re-exec via sudo
|
||||
"sudo",
|
||||
[
|
||||
"sudo",
|
||||
@@ -501,7 +504,7 @@ def _safe_host(host: str, cfg: Config) -> str:
|
||||
0.0.0.0 (bind-all) is silently remapped to 127.0.0.1 — it's a valid local
|
||||
server address, just not usable as a client connection target.
|
||||
"""
|
||||
if host in ("0.0.0.0", "::"):
|
||||
if host in ("0.0.0.0", "::"): # noqa: S104 — comparison only, no binding here
|
||||
return "127.0.0.1"
|
||||
if host not in _ALLOWED_HOSTS:
|
||||
print(
|
||||
@@ -514,7 +517,7 @@ def _safe_host(host: str, cfg: Config) -> str:
|
||||
|
||||
|
||||
def _log_file() -> str:
|
||||
return "/var/log/nvcurve.log" if os.geteuid() == 0 else "/tmp/nvcurve.log"
|
||||
return "/var/log/nvcurve.log" if os.geteuid() == 0 else "/tmp/nvcurve.log" # noqa: S108
|
||||
|
||||
|
||||
def _read_server_info() -> dict | None:
|
||||
@@ -737,8 +740,14 @@ def cmd_inspect(args):
|
||||
|
||||
|
||||
def cmd_write(args):
|
||||
delta_khz = int(args.delta * 1000)
|
||||
max_delta_khz = int(args.max_delta * 1000) if args.max_delta is not None else None
|
||||
try:
|
||||
delta_khz = int(args.delta * 1000)
|
||||
max_delta_khz = (
|
||||
int(args.max_delta * 1000) if args.max_delta is not None else None
|
||||
)
|
||||
except (TypeError, ValueError, OverflowError) as exc:
|
||||
print(f"Error: invalid numeric argument: {exc}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
point_deltas = {}
|
||||
|
||||
if args.reset:
|
||||
@@ -837,14 +846,15 @@ def cmd_write(args):
|
||||
print(f"Write OK — {len(point_deltas)} point(s) updated.")
|
||||
|
||||
try:
|
||||
curve_state = None
|
||||
if not args.glob:
|
||||
curve_state, _ = read_curve(gpu, gpu_name)
|
||||
if curve_state:
|
||||
vfp_freqs = [p.freq_khz for p in curve_state.points]
|
||||
for w in check_negative_freq_warnings(point_deltas, vfp_freqs, []):
|
||||
print(f"WARNING: {w}")
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
log.debug("Post-write curve check failed: %s", exc)
|
||||
|
||||
|
||||
def cmd_verify(args):
|
||||
@@ -855,7 +865,11 @@ def cmd_verify(args):
|
||||
from .hal.snapshot import save as snapshot_save
|
||||
from .hal.vfcurve import read_clock_offsets, write_offsets
|
||||
|
||||
delta_khz = int(args.delta * 1000)
|
||||
try:
|
||||
delta_khz = int(args.delta * 1000)
|
||||
except (TypeError, ValueError, OverflowError) as exc:
|
||||
print(f"Error: invalid numeric argument: {exc}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
if args.point is not None:
|
||||
points = [args.point]
|
||||
@@ -1016,8 +1030,11 @@ def _profile_config_write(key: str, value) -> None:
|
||||
data.pop(key, None)
|
||||
else:
|
||||
data[key] = value
|
||||
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
|
||||
_json.dump(data, f, indent=2)
|
||||
try:
|
||||
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
|
||||
_json.dump(data, f, indent=2)
|
||||
except OSError as exc:
|
||||
raise RuntimeError(f"Cannot write {_PERSISTENT_CONFIG_FILE}: {exc}") from exc
|
||||
|
||||
|
||||
def _gpu_stable_key_offline(gpu_index: int) -> str | None:
|
||||
@@ -1066,8 +1083,11 @@ def _profile_config_set_default(gpu_index: int, name: str | None) -> None:
|
||||
profiles[gpu_key] = name
|
||||
if not profiles:
|
||||
data.pop("auto_load_profiles", None)
|
||||
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
|
||||
_json.dump(data, f, indent=2)
|
||||
try:
|
||||
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
|
||||
_json.dump(data, f, indent=2)
|
||||
except OSError as exc:
|
||||
raise RuntimeError(f"Cannot write {_PERSISTENT_CONFIG_FILE}: {exc}") from exc
|
||||
|
||||
|
||||
def cmd_profile(args):
|
||||
@@ -1094,8 +1114,8 @@ def cmd_profile(args):
|
||||
profiles.append(
|
||||
{"name": name, "curve_deltas": p.get("curve_deltas", {})}
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
log.debug("Skipping unreadable profile %s: %s", path, exc)
|
||||
if not profiles:
|
||||
print("No profiles found.")
|
||||
return
|
||||
@@ -1117,7 +1137,7 @@ def cmd_profile(args):
|
||||
require_root()
|
||||
try:
|
||||
_profile_config_set_default(gpu_index, None if clearing else args.name)
|
||||
except ValueError as e:
|
||||
except (ValueError, RuntimeError) as e:
|
||||
print(f"Error: {e}", file=sys.stderr)
|
||||
return
|
||||
if clearing:
|
||||
@@ -1205,7 +1225,14 @@ def cmd_profile(args):
|
||||
errs.append(f"Power limit: {msg}")
|
||||
|
||||
if profile.curve_deltas:
|
||||
deltas = {int(k): v for k, v in profile.curve_deltas.items()}
|
||||
try:
|
||||
deltas = {int(k): v for k, v in profile.curve_deltas.items()}
|
||||
except ValueError:
|
||||
print(
|
||||
f"Profile '{args.name}' has invalid curve point keys.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
errors = validate_write(deltas, default_config.max_delta_khz)
|
||||
if errors:
|
||||
errs.append("Curve: " + "; ".join(errors))
|
||||
@@ -1328,7 +1355,11 @@ def cmd_setup(args):
|
||||
"""One-shot hardware compatibility check: diag → read → write-verify → restore."""
|
||||
explicit_point = getattr(args, "point", None)
|
||||
verify_delta_mhz = getattr(args, "delta", 5.0) or 5.0
|
||||
verify_delta_khz = int(verify_delta_mhz * 1000)
|
||||
try:
|
||||
verify_delta_khz = int(verify_delta_mhz * 1000)
|
||||
except (TypeError, ValueError, OverflowError) as exc:
|
||||
print(f"Error: invalid numeric argument: {exc}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
require_root()
|
||||
|
||||
@@ -1452,13 +1483,20 @@ def cmd_setup(args):
|
||||
print()
|
||||
print("Step 4/4 Restoring snapshot")
|
||||
print()
|
||||
ok = snapshot_restore(gpu, default_config.snapshot_dir, snap_path)
|
||||
if ok:
|
||||
print(" Hardware state restored to baseline.")
|
||||
else:
|
||||
if snap_path is None:
|
||||
print(
|
||||
" WARNING: Restore failed. Run: nvcurve snapshot restore", file=sys.stderr
|
||||
" WARNING: Snapshot save failed — cannot restore baseline.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
else:
|
||||
ok = snapshot_restore(gpu, default_config.snapshot_dir, snap_path)
|
||||
if ok:
|
||||
print(" Hardware state restored to baseline.")
|
||||
else:
|
||||
print(
|
||||
" WARNING: Restore failed. Run: nvcurve snapshot restore",
|
||||
file=sys.stderr,
|
||||
)
|
||||
|
||||
print()
|
||||
print(sep)
|
||||
@@ -1509,24 +1547,36 @@ def cmd_service(args):
|
||||
"WantedBy=multi-user.target\n"
|
||||
)
|
||||
|
||||
with open(unit_path, "w") as f:
|
||||
f.write(unit)
|
||||
try:
|
||||
with open(unit_path, "w") as f:
|
||||
f.write(unit)
|
||||
except OSError as exc:
|
||||
print(f"Failed to write {unit_path}: {exc}", file=sys.stderr)
|
||||
return
|
||||
print(f"Unit file written to {unit_path}")
|
||||
|
||||
# Write persistent config.
|
||||
os.makedirs("/etc/nvcurve", exist_ok=True)
|
||||
try:
|
||||
os.makedirs("/etc/nvcurve", exist_ok=True)
|
||||
except OSError as exc:
|
||||
print(f"Failed to create /etc/nvcurve: {exc}", file=sys.stderr)
|
||||
return
|
||||
persistent_cfg: dict = {}
|
||||
try:
|
||||
with open(_PERSISTENT_CONFIG_FILE) as f:
|
||||
persistent_cfg = json.load(f)
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
log.debug("Could not read persistent config: %s", exc)
|
||||
host = getattr(args, "host", "127.0.0.1")
|
||||
port = getattr(args, "port", 8042)
|
||||
auto_serve = getattr(args, "auto_serve", False)
|
||||
persistent_cfg.update({"host": host, "port": port, "auto_serve": auto_serve})
|
||||
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
|
||||
json.dump(persistent_cfg, f, indent=2)
|
||||
try:
|
||||
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
|
||||
json.dump(persistent_cfg, f, indent=2)
|
||||
except OSError as exc:
|
||||
print(f"Failed to write {_PERSISTENT_CONFIG_FILE}: {exc}", file=sys.stderr)
|
||||
return
|
||||
print(f"Persistent config written to {_PERSISTENT_CONFIG_FILE}")
|
||||
if auto_serve:
|
||||
print(f" Web server will auto-start on boot at {host}:{port}")
|
||||
@@ -1538,12 +1588,8 @@ def cmd_service(args):
|
||||
try:
|
||||
subprocess.run(["systemctl", "daemon-reload"], check=True)
|
||||
|
||||
was_active = (
|
||||
subprocess.run(
|
||||
["systemctl", "is-active", "--quiet", "nvcurve"],
|
||||
).returncode
|
||||
== 0
|
||||
)
|
||||
probe = subprocess.run(["systemctl", "is-active", "--quiet", "nvcurve"])
|
||||
was_active = probe.returncode == 0
|
||||
|
||||
subprocess.run(["systemctl", "enable", "--now", "nvcurve"], check=True)
|
||||
print("Service enabled and started.")
|
||||
@@ -1560,7 +1606,7 @@ def cmd_service(args):
|
||||
print(" systemctl status nvcurve")
|
||||
print(" journalctl -u nvcurve -f")
|
||||
print(" nvcurve service uninstall")
|
||||
except subprocess.CalledProcessError as e:
|
||||
except (subprocess.CalledProcessError, FileNotFoundError) as e:
|
||||
print(f"systemctl failed: {e}", file=sys.stderr)
|
||||
|
||||
elif action == "uninstall":
|
||||
@@ -1672,13 +1718,17 @@ def cmd_service(args):
|
||||
require_root()
|
||||
import subprocess
|
||||
|
||||
os.makedirs("/etc/nvcurve", exist_ok=True)
|
||||
try:
|
||||
os.makedirs("/etc/nvcurve", exist_ok=True)
|
||||
except OSError as exc:
|
||||
print(f"Failed to create /etc/nvcurve: {exc}", file=sys.stderr)
|
||||
return
|
||||
pcfg: dict = {}
|
||||
try:
|
||||
with open(_PERSISTENT_CONFIG_FILE) as f:
|
||||
pcfg = json.load(f)
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
log.debug("Could not read persistent config: %s", exc)
|
||||
|
||||
if hasattr(args, "auto_serve") and args.auto_serve is not None:
|
||||
pcfg["auto_serve"] = args.auto_serve
|
||||
@@ -1687,8 +1737,12 @@ def cmd_service(args):
|
||||
if hasattr(args, "port") and args.port is not None:
|
||||
pcfg["port"] = args.port
|
||||
|
||||
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
|
||||
json.dump(pcfg, f, indent=2)
|
||||
try:
|
||||
with open(_PERSISTENT_CONFIG_FILE, "w") as f:
|
||||
json.dump(pcfg, f, indent=2)
|
||||
except OSError as exc:
|
||||
print(f"Failed to write {_PERSISTENT_CONFIG_FILE}: {exc}", file=sys.stderr)
|
||||
return
|
||||
print(f"Config updated ({_PERSISTENT_CONFIG_FILE}):")
|
||||
print(f" auto-serve: {'on' if pcfg.get('auto_serve', False) else 'off'}")
|
||||
print(f" host: {pcfg.get('host', '127.0.0.1')}")
|
||||
@@ -1715,8 +1769,12 @@ def _cmd_serve_start(args, cfg: Config, open_browser: bool = False) -> None:
|
||||
# --direct: skip daemon round-trip (used when the daemon itself spawns us).
|
||||
if getattr(args, "direct", False):
|
||||
require_root()
|
||||
with open(_SERVER_INFO_FILE, "w") as f:
|
||||
json.dump({"pid": os.getpid(), "host": host, "port": port}, f)
|
||||
try:
|
||||
with open(_SERVER_INFO_FILE, "w") as f:
|
||||
json.dump({"pid": os.getpid(), "host": host, "port": port}, f)
|
||||
except OSError as exc:
|
||||
print(f"Failed to write {_SERVER_INFO_FILE}: {exc}", file=sys.stderr)
|
||||
return
|
||||
try:
|
||||
from .server import run as server_run
|
||||
|
||||
@@ -1773,8 +1831,12 @@ def _cmd_serve_start(args, cfg: Config, open_browser: bool = False) -> None:
|
||||
cmd += ["--gpu", str(args.gpu_index)]
|
||||
log_path = _log_file()
|
||||
print("Starting nvcurve server in background...")
|
||||
with open(log_path, "a") as lf:
|
||||
p = subprocess.Popen(cmd, stdout=lf, stderr=lf, start_new_session=True)
|
||||
try:
|
||||
with open(log_path, "a") as lf:
|
||||
p = subprocess.Popen(cmd, stdout=lf, stderr=lf, start_new_session=True)
|
||||
except OSError as exc:
|
||||
print(f"Failed to open log file {log_path}: {exc}", file=sys.stderr)
|
||||
return
|
||||
print(f"Server starting (PID {p.pid}). Logs: {log_path}")
|
||||
if open_browser:
|
||||
time.sleep(1.5)
|
||||
@@ -1782,8 +1844,12 @@ def _cmd_serve_start(args, cfg: Config, open_browser: bool = False) -> None:
|
||||
return
|
||||
|
||||
# Foreground mode — write info file so clients can discover host:port.
|
||||
with open(_SERVER_INFO_FILE, "w") as f:
|
||||
json.dump({"pid": os.getpid(), "host": host, "port": port}, f)
|
||||
try:
|
||||
with open(_SERVER_INFO_FILE, "w") as f:
|
||||
json.dump({"pid": os.getpid(), "host": host, "port": port}, f)
|
||||
except OSError as exc:
|
||||
print(f"Failed to write {_SERVER_INFO_FILE}: {exc}", file=sys.stderr)
|
||||
return
|
||||
try:
|
||||
from .server import run as server_run
|
||||
|
||||
@@ -2095,8 +2161,8 @@ def main():
|
||||
if "fan_curves" in data:
|
||||
# Per-GPU active fan curves, restored on server startup.
|
||||
cfg.fan_curves = dict(data["fan_curves"])
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
log.debug("Could not load user config: %s", exc)
|
||||
|
||||
base_url = args.server or _discover_server_url(cfg)
|
||||
client = NvCurveClient(base=base_url, gpu_index=getattr(args, "gpu_index", 0))
|
||||
@@ -2148,8 +2214,8 @@ def main():
|
||||
if os.path.exists(_SERVER_INFO_FILE):
|
||||
try:
|
||||
os.remove(_SERVER_INFO_FILE)
|
||||
except OSError:
|
||||
pass
|
||||
except OSError as exc:
|
||||
log.debug("Could not remove %s: %s", _SERVER_INFO_FILE, exc)
|
||||
except ApiError as e:
|
||||
if e.status_code == 401:
|
||||
print(
|
||||
|
||||
+2
-1
@@ -209,8 +209,9 @@ async def _serve_socket(auto_serve: bool = False) -> None:
|
||||
# regular user and talks to this root daemon over the socket. 0o666 is
|
||||
# intentional (standard for /run daemon sockets).
|
||||
# pi-lens-ignore: S103
|
||||
_SOCKET_MODE = 0o666
|
||||
os.chmod(
|
||||
SOCKET_PATH, 0o666
|
||||
SOCKET_PATH, _SOCKET_MODE
|
||||
) # nosemgrep: python.lang.security.audit.insecure-file-permissions.insecure-file-permissions
|
||||
log.info("Daemon listening on %s", SOCKET_PATH)
|
||||
|
||||
|
||||
+29
-11
@@ -9,14 +9,19 @@ Uses NVML (via pynvml) for all operations:
|
||||
|
||||
import ctypes
|
||||
import logging
|
||||
from typing import List, Optional
|
||||
from typing import Any
|
||||
|
||||
try:
|
||||
import pynvml
|
||||
import pynvml as _pynvml_import
|
||||
|
||||
_NVML_AVAILABLE = True
|
||||
except ImportError:
|
||||
_pynvml_import = None
|
||||
_NVML_AVAILABLE = False
|
||||
|
||||
# Aliased as Any so attribute access is not flagged when the import failed.
|
||||
pynvml: Any = _pynvml_import
|
||||
|
||||
log = logging.getLogger("nvcurve.hal.fans")
|
||||
|
||||
# We use fan index 0 (first/primary fan) for all operations.
|
||||
@@ -35,7 +40,7 @@ def get_fan_info(gpu_index: int = 0) -> dict:
|
||||
|
||||
Returns None values on failure.
|
||||
"""
|
||||
out = {
|
||||
out: dict[str, float | None] = {
|
||||
"fan_pct": None,
|
||||
"fan_mode": None,
|
||||
"min_fan_pct": None,
|
||||
@@ -75,7 +80,10 @@ def get_fan_info(gpu_index: int = 0) -> dict:
|
||||
|
||||
def set_fan_speed(gpu_index: int, pct: int) -> tuple[bool, str]:
|
||||
"""Set fan speed to a percentage (0-100) on the primary fan."""
|
||||
pct = max(0, min(100, int(pct)))
|
||||
try:
|
||||
pct = max(0, min(100, int(pct)))
|
||||
except (TypeError, ValueError):
|
||||
return False, "Invalid fan speed"
|
||||
if not _NVML_AVAILABLE:
|
||||
return False, "NVML not available"
|
||||
try:
|
||||
@@ -102,7 +110,9 @@ def reset_fan(gpu_index: int = 0) -> tuple[bool, str]:
|
||||
try:
|
||||
ret = subprocess.run(
|
||||
["nvidia-smi", "-i", str(gpu_index), "-fan", "default"],
|
||||
capture_output=True, text=True, timeout=10,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=10,
|
||||
)
|
||||
if ret.returncode == 0:
|
||||
return True, "OK"
|
||||
@@ -123,19 +133,21 @@ def reset_fan(gpu_index: int = 0) -> tuple[bool, str]:
|
||||
return False, f"Failed to reset fan: {exc}"
|
||||
|
||||
|
||||
def get_temp(gpu_index: int = 0) -> Optional[float]:
|
||||
def get_temp(gpu_index: int = 0) -> float | None:
|
||||
"""Read current GPU temperature in °C."""
|
||||
if not _NVML_AVAILABLE:
|
||||
return None
|
||||
try:
|
||||
handle = _get_handle(gpu_index)
|
||||
return float(pynvml.nvmlDeviceGetTemperature(handle, pynvml.NVML_TEMPERATURE_GPU))
|
||||
return float(
|
||||
pynvml.nvmlDeviceGetTemperature(handle, pynvml.NVML_TEMPERATURE_GPU)
|
||||
)
|
||||
except pynvml.NVMLError as exc:
|
||||
log.debug("get_temp: %s", exc)
|
||||
return None
|
||||
|
||||
|
||||
def interpolate_fan_speed(curve: List[dict], temp_c: float) -> Optional[int]:
|
||||
def interpolate_fan_speed(curve: list[dict], temp_c: float) -> int | None:
|
||||
"""Interpolate target fan speed from a curve at a given temperature.
|
||||
|
||||
curve: list of {temp_c: int, fan_pct: int} sorted by temp_c
|
||||
@@ -144,7 +156,10 @@ def interpolate_fan_speed(curve: List[dict], temp_c: float) -> Optional[int]:
|
||||
if not curve or len(curve) < 2:
|
||||
return None
|
||||
|
||||
temp = float(temp_c)
|
||||
try:
|
||||
temp = float(temp_c)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
# Find the two surrounding points
|
||||
for i in range(len(curve) - 1):
|
||||
@@ -157,7 +172,10 @@ def interpolate_fan_speed(curve: List[dict], temp_c: float) -> Optional[int]:
|
||||
if t0 <= temp <= t1:
|
||||
fraction = (temp - t0) / (t1 - t0)
|
||||
result = f0 + fraction * (f1 - f0)
|
||||
return max(0, min(100, int(round(result))))
|
||||
try:
|
||||
return max(0, min(100, int(round(result))))
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
# Outside range: clamp to first or last point
|
||||
if temp <= curve[0]["temp_c"]:
|
||||
@@ -165,7 +183,7 @@ def interpolate_fan_speed(curve: List[dict], temp_c: float) -> Optional[int]:
|
||||
return max(0, min(100, curve[-1]["fan_pct"]))
|
||||
|
||||
|
||||
def validate_curve(curve: List[dict]) -> tuple[bool, str]:
|
||||
def validate_curve(curve: list[dict]) -> tuple[bool, str]:
|
||||
"""Validate a fan curve.
|
||||
|
||||
Returns (True, "OK") or (False, error_message).
|
||||
|
||||
+38
-20
@@ -1,12 +1,17 @@
|
||||
"""GPU discovery and initialization."""
|
||||
|
||||
import contextlib
|
||||
import ctypes
|
||||
import logging
|
||||
import sys
|
||||
from typing import Any
|
||||
|
||||
from ..nvapi.bootstrap import query_interface
|
||||
from ..nvapi.constants import FUNC
|
||||
from ..nvapi.types import GpuInfo
|
||||
|
||||
log = logging.getLogger("nvcurve.hal.gpu")
|
||||
|
||||
|
||||
def init_nvapi() -> None:
|
||||
"""Initialize NvAPI. Must be called before any GPU operations."""
|
||||
@@ -19,7 +24,10 @@ def enumerate_gpus() -> tuple[ctypes.Array, int]:
|
||||
"""Return (gpu_handles_array, count). Exits if no GPUs found."""
|
||||
gpus = (ctypes.c_void_p * 64)()
|
||||
ngpu = ctypes.c_int32()
|
||||
query_interface(FUNC["EnumPhysicalGPUs"])(ctypes.byref(gpus), ctypes.byref(ngpu))
|
||||
enum_fn = query_interface(FUNC["EnumPhysicalGPUs"])
|
||||
if enum_fn is None:
|
||||
raise RuntimeError("NvAPI function EnumPhysicalGPUs not available")
|
||||
enum_fn(ctypes.byref(gpus), ctypes.byref(ngpu))
|
||||
if ngpu.value == 0:
|
||||
print("No NVIDIA GPUs found")
|
||||
sys.exit(1)
|
||||
@@ -29,7 +37,10 @@ def enumerate_gpus() -> tuple[ctypes.Array, int]:
|
||||
def get_gpu_name(gpu) -> str:
|
||||
"""Return the full name string for a GPU handle."""
|
||||
name_buf = ctypes.create_string_buffer(256)
|
||||
query_interface(FUNC["GetFullName"])(gpu, name_buf)
|
||||
fn = query_interface(FUNC["GetFullName"])
|
||||
if fn is None:
|
||||
raise RuntimeError("NvAPI function GetFullName not available")
|
||||
fn(gpu, name_buf)
|
||||
return name_buf.value.decode(errors="replace")
|
||||
|
||||
|
||||
@@ -40,38 +51,45 @@ def discover_gpus() -> list[GpuInfo]:
|
||||
infos = []
|
||||
|
||||
try:
|
||||
import pynvml
|
||||
pynvml.nvmlInit()
|
||||
has_nvml = True
|
||||
import pynvml as _pynvml
|
||||
|
||||
_pynvml.nvmlInit()
|
||||
except Exception:
|
||||
has_nvml = False
|
||||
_pynvml = None
|
||||
# Aliased as Any so attribute access is not flagged when the import failed.
|
||||
pynvml: Any = _pynvml
|
||||
|
||||
for i in range(count):
|
||||
name = get_gpu_name(gpus[i])
|
||||
uuid = None
|
||||
pci_bus_id = None
|
||||
if has_nvml:
|
||||
if pynvml is not None:
|
||||
try:
|
||||
handle = pynvml.nvmlDeviceGetHandleByIndex(i)
|
||||
uuid = pynvml.nvmlDeviceGetUUID(handle)
|
||||
raw_uuid = pynvml.nvmlDeviceGetUUID(handle)
|
||||
# NVML might return bytes
|
||||
if isinstance(uuid, bytes):
|
||||
uuid = uuid.decode('utf-8', errors='ignore')
|
||||
if isinstance(raw_uuid, bytes):
|
||||
uuid = raw_uuid.decode("utf-8", errors="ignore")
|
||||
elif raw_uuid is not None:
|
||||
uuid = str(raw_uuid)
|
||||
pci_info = pynvml.nvmlDeviceGetPciInfo(handle)
|
||||
# Parse something like "00000000:01:00.0" -> bus is 1
|
||||
if isinstance(pci_info.bus, bytes):
|
||||
pci_bus_id = int(pci_info.bus.decode('utf-8', errors='ignore'), 16)
|
||||
# Parse something like "00000000:01:00.0" -> bus is 1.
|
||||
# PCI bus numbers are hex by convention (pynvml's field is an
|
||||
# int; the str/bytes branches are defensive).
|
||||
bus = pci_info.bus
|
||||
if isinstance(bus, bytes):
|
||||
pci_bus_id = int(bus.decode("utf-8", errors="ignore"), 16)
|
||||
elif isinstance(bus, str):
|
||||
pci_bus_id = int(bus, 16)
|
||||
else:
|
||||
pci_bus_id = pci_info.bus
|
||||
except Exception:
|
||||
pass
|
||||
pci_bus_id = int(bus)
|
||||
except Exception as exc:
|
||||
log.debug("NVML query for GPU %d failed: %s", i, exc)
|
||||
infos.append(GpuInfo(name=name, index=i, uuid=uuid, pci_bus_id=pci_bus_id))
|
||||
|
||||
if has_nvml:
|
||||
try:
|
||||
if pynvml is not None:
|
||||
with contextlib.suppress(Exception):
|
||||
pynvml.nvmlShutdown()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return infos
|
||||
|
||||
|
||||
+75
-38
@@ -11,21 +11,26 @@ that are explicitly specified, leaving others unchanged on hardware.
|
||||
"""
|
||||
|
||||
import ctypes
|
||||
import subprocess
|
||||
import logging
|
||||
from typing import Optional
|
||||
import subprocess
|
||||
from typing import Any
|
||||
|
||||
try:
|
||||
import pynvml
|
||||
import pynvml as _pynvml_import
|
||||
|
||||
_NVML_AVAILABLE = True
|
||||
except ImportError:
|
||||
_pynvml_import = None
|
||||
_NVML_AVAILABLE = False
|
||||
|
||||
# Aliased as Any so attribute access is not flagged when the import failed.
|
||||
pynvml: Any = _pynvml_import
|
||||
|
||||
log = logging.getLogger("nvcurve.hal.limits")
|
||||
|
||||
# ── NVML library / handle helpers ─────────────────────────────────────────────
|
||||
|
||||
_nvml_lib: Optional[ctypes.CDLL] = None
|
||||
_nvml_lib: ctypes.CDLL | None = None
|
||||
|
||||
|
||||
def _nvml_cdll() -> ctypes.CDLL:
|
||||
@@ -34,12 +39,12 @@ def _nvml_cdll() -> ctypes.CDLL:
|
||||
if _nvml_lib is not None:
|
||||
return _nvml_lib
|
||||
# Prefer to reuse the library already loaded by pynvml to avoid dlopen races.
|
||||
for attr in ("nvml", "_nvml"): # attribute name varies by pynvml version
|
||||
for attr in ("nvml", "_nvml"): # attribute name varies by pynvml version
|
||||
mod = getattr(pynvml, attr, None)
|
||||
lib = getattr(mod, "_lib", None) or getattr(mod, "_nvmlLib", None)
|
||||
if lib is not None:
|
||||
_nvml_lib = lib
|
||||
return _nvml_lib
|
||||
return lib
|
||||
_nvml_lib = ctypes.CDLL("libnvidia-ml.so.1")
|
||||
return _nvml_lib
|
||||
|
||||
@@ -53,9 +58,10 @@ def _get_handle(gpu_index: int):
|
||||
|
||||
# ── Power limit ───────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def get_power_limit(gpu_index: int = 0) -> dict:
|
||||
"""Return dict with power_limit_w, default_power_limit_w, min_power_limit_w, max_power_limit_w."""
|
||||
out = {
|
||||
out: dict[str, int | None] = {
|
||||
"power_limit_w": None,
|
||||
"default_power_limit_w": None,
|
||||
"min_power_limit_w": None,
|
||||
@@ -71,8 +77,8 @@ def get_power_limit(gpu_index: int = 0) -> dict:
|
||||
try:
|
||||
default = pynvml.nvmlDeviceGetPowerManagementDefaultLimit(handle)
|
||||
out["default_power_limit_w"] = default // 1000
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
log.debug("nvmlDeviceGetPowerManagementDefaultLimit: %s", exc)
|
||||
except Exception as exc:
|
||||
log.warning("get_power_limit: %s", exc)
|
||||
return out
|
||||
@@ -89,7 +95,8 @@ def set_power_limit(limit_w: int, gpu_index: int = 0) -> tuple[bool, str]:
|
||||
|
||||
ret = subprocess.run(
|
||||
["nvidia-smi", "-i", str(gpu_index), "-pl", str(limit_w)],
|
||||
capture_output=True, text=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
if ret.returncode == 0:
|
||||
return True, "OK"
|
||||
@@ -111,34 +118,38 @@ def set_power_limit(limit_w: int, gpu_index: int = 0) -> tuple[bool, str]:
|
||||
# pynvml (nvidia-ml-py ≥ 12) exposes c_nvmlClockOffset_t and nvmlClockOffset_v1
|
||||
# as ctypes objects; we use them when available and fall back to our own definition.
|
||||
|
||||
|
||||
class _ClockOffset(ctypes.Structure):
|
||||
_fields_ = [
|
||||
("version", ctypes.c_uint),
|
||||
("type", ctypes.c_uint), # nvmlClockType_t
|
||||
("pstate", ctypes.c_uint), # nvmlPstates_t
|
||||
("version", ctypes.c_uint),
|
||||
("type", ctypes.c_uint), # nvmlClockType_t
|
||||
("pstate", ctypes.c_uint), # nvmlPstates_t
|
||||
("clockOffsetMHz", ctypes.c_int),
|
||||
]
|
||||
|
||||
|
||||
_CLOCK_OFFSET_VER = (1 << 24) | ctypes.sizeof(_ClockOffset) # = 0x01000010 (16 bytes)
|
||||
|
||||
# NVML clock-type constants (same values as pynvml).
|
||||
_NVML_CLOCK_GRAPHICS = 0
|
||||
_NVML_CLOCK_MEM = 2
|
||||
_NVML_CLOCK_MEM = 2
|
||||
|
||||
|
||||
def _make_clock_offset(clock_type: int, pstate: int = 0, offset_mhz: int = 0) -> ctypes.Structure:
|
||||
def _make_clock_offset(
|
||||
clock_type: int, pstate: int = 0, offset_mhz: int = 0
|
||||
) -> ctypes.Structure:
|
||||
"""Return a populated nvmlClockOffset_t struct, using pynvml's type when available."""
|
||||
if hasattr(pynvml, "c_nvmlClockOffset_t") and hasattr(pynvml, "nvmlClockOffset_v1"):
|
||||
info = pynvml.c_nvmlClockOffset_t()
|
||||
info.version = pynvml.nvmlClockOffset_v1
|
||||
info.type = clock_type
|
||||
info.pstate = pstate
|
||||
info.version = pynvml.nvmlClockOffset_v1
|
||||
info.type = clock_type
|
||||
info.pstate = pstate
|
||||
info.clockOffsetMHz = offset_mhz
|
||||
return info
|
||||
info = _ClockOffset()
|
||||
info.version = _CLOCK_OFFSET_VER
|
||||
info.type = clock_type
|
||||
info.pstate = pstate
|
||||
info.version = _CLOCK_OFFSET_VER
|
||||
info.type = clock_type
|
||||
info.pstate = pstate
|
||||
info.clockOffsetMHz = offset_mhz
|
||||
return info
|
||||
|
||||
@@ -158,7 +169,7 @@ def get_clock_offsets(gpu_index: int = 0) -> dict:
|
||||
Keys: gpc_offset_mhz, mem_offset_mhz (both int or None on failure).
|
||||
Calls nvmlDeviceGetClockOffsets once per clock domain (GRAPHICS, MEM).
|
||||
"""
|
||||
out = {"gpc_offset_mhz": None, "mem_offset_mhz": None}
|
||||
out: dict[str, int | None] = {"gpc_offset_mhz": None, "mem_offset_mhz": None}
|
||||
if not _NVML_AVAILABLE:
|
||||
return out
|
||||
try:
|
||||
@@ -167,11 +178,15 @@ def get_clock_offsets(gpu_index: int = 0) -> dict:
|
||||
# Try pynvml wrapper first (nvidia-ml-py ≥ 12 exposes it correctly).
|
||||
# Fall back to ctypes-direct if pynvml doesn't have it.
|
||||
_pynvml_get = getattr(pynvml, "nvmlDeviceGetClockOffsets", None)
|
||||
fn_get = _try_nvml_fn("nvmlDeviceGetClockOffsets") if _pynvml_get is None else None
|
||||
fn_get = (
|
||||
_try_nvml_fn("nvmlDeviceGetClockOffsets") if _pynvml_get is None else None
|
||||
)
|
||||
|
||||
used_new_api = False
|
||||
for clock_type, key in ((_NVML_CLOCK_GRAPHICS, "gpc_offset_mhz"),
|
||||
(_NVML_CLOCK_MEM, "mem_offset_mhz")):
|
||||
for clock_type, key in (
|
||||
(_NVML_CLOCK_GRAPHICS, "gpc_offset_mhz"),
|
||||
(_NVML_CLOCK_MEM, "mem_offset_mhz"),
|
||||
):
|
||||
info = _make_clock_offset(clock_type, pstate=0)
|
||||
try:
|
||||
if _pynvml_get is not None:
|
||||
@@ -184,7 +199,9 @@ def get_clock_offsets(gpu_index: int = 0) -> dict:
|
||||
out[key] = int(info.clockOffsetMHz)
|
||||
used_new_api = True
|
||||
else:
|
||||
log.debug("nvmlDeviceGetClockOffsets(type=%d) returned %d", clock_type, rc)
|
||||
log.debug(
|
||||
"nvmlDeviceGetClockOffsets(type=%d) returned %d", clock_type, rc
|
||||
)
|
||||
except Exception as exc:
|
||||
log.debug("nvmlDeviceGetClockOffsets(type=%d): %s", clock_type, exc)
|
||||
|
||||
@@ -200,7 +217,9 @@ def get_clock_offsets(gpu_index: int = 0) -> dict:
|
||||
if hasattr(pynvml, "nvmlDeviceGetMemClkVfOffset"):
|
||||
try:
|
||||
res = pynvml.nvmlDeviceGetMemClkVfOffset(handle)
|
||||
out["mem_offset_mhz"] = int(res[0] if isinstance(res, (list, tuple)) else res)
|
||||
out["mem_offset_mhz"] = int(
|
||||
res[0] if isinstance(res, (list, tuple)) else res
|
||||
)
|
||||
except Exception as exc:
|
||||
log.debug("nvmlDeviceGetMemClkVfOffset: %s", exc)
|
||||
|
||||
@@ -210,8 +229,8 @@ def get_clock_offsets(gpu_index: int = 0) -> dict:
|
||||
|
||||
|
||||
def set_clock_offsets(
|
||||
gpc_offset_mhz: Optional[int] = None,
|
||||
mem_offset_mhz: Optional[int] = None,
|
||||
gpc_offset_mhz: int | None = None,
|
||||
mem_offset_mhz: int | None = None,
|
||||
gpu_index: int = 0,
|
||||
) -> tuple[bool, str]:
|
||||
"""Set clock offsets (MHz) for the specified domains only.
|
||||
@@ -235,20 +254,35 @@ def set_clock_offsets(
|
||||
domains.append((_NVML_CLOCK_MEM, mem_offset_mhz))
|
||||
|
||||
_pynvml_set = getattr(pynvml, "nvmlDeviceSetClockOffsets", None)
|
||||
fn_set = _try_nvml_fn("nvmlDeviceSetClockOffsets") if _pynvml_set is None else None
|
||||
fn_set = (
|
||||
_try_nvml_fn("nvmlDeviceSetClockOffsets") if _pynvml_set is None else None
|
||||
)
|
||||
|
||||
if _pynvml_set is not None or fn_set is not None:
|
||||
all_ok = True
|
||||
for clock_type, offset in domains:
|
||||
info = _make_clock_offset(clock_type, pstate=0, offset_mhz=offset)
|
||||
try:
|
||||
rc = _pynvml_set(handle, ctypes.byref(info)) if _pynvml_set else fn_set(handle, ctypes.byref(info))
|
||||
if _pynvml_set is not None:
|
||||
rc = _pynvml_set(handle, ctypes.byref(info))
|
||||
elif fn_set is not None:
|
||||
rc = fn_set(handle, ctypes.byref(info))
|
||||
else:
|
||||
break
|
||||
if rc != 0:
|
||||
log.debug("nvmlDeviceSetClockOffsets(type=%d) returned %d — trying fallback", clock_type, rc)
|
||||
log.debug(
|
||||
"nvmlDeviceSetClockOffsets(type=%d) returned %d — trying fallback",
|
||||
clock_type,
|
||||
rc,
|
||||
)
|
||||
all_ok = False
|
||||
break
|
||||
except Exception as exc:
|
||||
log.debug("nvmlDeviceSetClockOffsets(type=%d): %s — trying fallback", clock_type, exc)
|
||||
log.debug(
|
||||
"nvmlDeviceSetClockOffsets(type=%d): %s — trying fallback",
|
||||
clock_type,
|
||||
exc,
|
||||
)
|
||||
all_ok = False
|
||||
break
|
||||
if all_ok:
|
||||
@@ -257,12 +291,16 @@ def set_clock_offsets(
|
||||
|
||||
# Deprecated per-domain fallback (works on Blackwell/driver 590.x).
|
||||
errs = []
|
||||
if gpc_offset_mhz is not None and hasattr(pynvml, "nvmlDeviceSetGpcClkVfOffset"):
|
||||
if gpc_offset_mhz is not None and hasattr(
|
||||
pynvml, "nvmlDeviceSetGpcClkVfOffset"
|
||||
):
|
||||
try:
|
||||
pynvml.nvmlDeviceSetGpcClkVfOffset(handle, gpc_offset_mhz)
|
||||
except Exception as exc:
|
||||
errs.append(f"GPC: {exc}")
|
||||
if mem_offset_mhz is not None and hasattr(pynvml, "nvmlDeviceSetMemClkVfOffset"):
|
||||
if mem_offset_mhz is not None and hasattr(
|
||||
pynvml, "nvmlDeviceSetMemClkVfOffset"
|
||||
):
|
||||
try:
|
||||
pynvml.nvmlDeviceSetMemClkVfOffset(handle, mem_offset_mhz)
|
||||
except Exception as exc:
|
||||
@@ -278,6 +316,7 @@ def set_clock_offsets(
|
||||
|
||||
# ── Range queries ─────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def get_mem_offset_range(gpu_index: int = 0) -> dict:
|
||||
"""Return the min/max allowed memory clock offset (MHz).
|
||||
|
||||
@@ -285,7 +324,7 @@ def get_mem_offset_range(gpu_index: int = 0) -> dict:
|
||||
Uses nvmlDeviceGetMemClkMinMaxVfOffset; falls back to observed RTX values.
|
||||
"""
|
||||
# Observed RTX 5090 defaults (NvAPI GetClockBoostRanges says -1000/+3000).
|
||||
out = {"min_mem_offset_mhz": -2000, "max_mem_offset_mhz": 3000}
|
||||
out: dict[str, int] = {"min_mem_offset_mhz": -2000, "max_mem_offset_mhz": 3000}
|
||||
if not _NVML_AVAILABLE:
|
||||
return out
|
||||
try:
|
||||
@@ -317,5 +356,3 @@ def get_mem_offset_range(gpu_index: int = 0) -> dict:
|
||||
except Exception as exc:
|
||||
log.debug("get_mem_offset_range: %s", exc)
|
||||
return out
|
||||
|
||||
|
||||
+61
-30
@@ -2,18 +2,20 @@
|
||||
|
||||
import ctypes
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import struct
|
||||
from datetime import datetime
|
||||
from typing import Optional
|
||||
|
||||
from ..nvapi.bootstrap import nvcall_raw
|
||||
from ..nvapi.constants import FUNC, CT_SIZE, CT_BASE, CT_STRIDE, CT_DELTA_OFF, CT_POINTS
|
||||
from ..nvapi.constants import CT_BASE, CT_DELTA_OFF, CT_SIZE, CT_STRIDE, FUNC
|
||||
from ..nvapi.types import SnapshotInfo
|
||||
from .vfcurve import read_clock_table_raw, get_boost_mask
|
||||
from .vfcurve import get_boost_mask, read_clock_table_raw
|
||||
|
||||
log = logging.getLogger("nvcurve.hal.snapshot")
|
||||
|
||||
|
||||
def save(gpu, gpu_name: str, snapshot_dir: str, max_snapshots: int = 0) -> Optional[str]:
|
||||
def save(gpu, gpu_name: str, snapshot_dir: str, max_snapshots: int = 0) -> str | None:
|
||||
"""Save the current ClockBoostTable to disk.
|
||||
|
||||
Writes both a binary .bin file and a human-readable .json metadata file.
|
||||
@@ -25,13 +27,21 @@ def save(gpu, gpu_name: str, snapshot_dir: str, max_snapshots: int = 0) -> Optio
|
||||
print(f"Failed to read ClockBoostTable: {err}")
|
||||
return None
|
||||
|
||||
os.makedirs(snapshot_dir, exist_ok=True)
|
||||
try:
|
||||
os.makedirs(snapshot_dir, exist_ok=True)
|
||||
except OSError as exc:
|
||||
print(f"Failed to create snapshot dir {snapshot_dir}: {exc}")
|
||||
return None
|
||||
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
|
||||
bin_path = os.path.join(snapshot_dir, f"clock_boost_table_{ts}.bin")
|
||||
meta_path = os.path.join(snapshot_dir, f"clock_boost_table_{ts}.json")
|
||||
|
||||
with open(bin_path, "wb") as f:
|
||||
f.write(raw)
|
||||
try:
|
||||
with open(bin_path, "wb") as f:
|
||||
f.write(raw)
|
||||
except OSError as exc:
|
||||
print(f"Failed to write snapshot {bin_path}: {exc}")
|
||||
return None
|
||||
|
||||
offsets = []
|
||||
max_entries = (len(raw) - CT_BASE) // CT_STRIDE
|
||||
@@ -48,10 +58,14 @@ def save(gpu, gpu_name: str, snapshot_dir: str, max_snapshots: int = 0) -> Optio
|
||||
"offsets_kHz": offsets,
|
||||
"nonzero_offsets": sum(1 for o in offsets if o != 0),
|
||||
}
|
||||
with open(meta_path, "w") as f:
|
||||
json.dump(meta, f, indent=2)
|
||||
try:
|
||||
with open(meta_path, "w") as f:
|
||||
json.dump(meta, f, indent=2)
|
||||
except OSError as exc:
|
||||
print(f"Failed to write snapshot metadata {meta_path}: {exc}")
|
||||
return None
|
||||
|
||||
print(f"Snapshot saved:")
|
||||
print("Snapshot saved:")
|
||||
print(f" Binary: {bin_path}")
|
||||
print(f" Metadata: {meta_path}")
|
||||
print(f" Size: {len(raw)} bytes")
|
||||
@@ -65,20 +79,22 @@ def save(gpu, gpu_name: str, snapshot_dir: str, max_snapshots: int = 0) -> Optio
|
||||
|
||||
def _prune_snapshots(snapshot_dir: str, max_snapshots: int) -> None:
|
||||
"""Delete oldest snapshots (both .bin and .json) to stay within max_snapshots."""
|
||||
bins = sorted(
|
||||
f for f in os.listdir(snapshot_dir) if f.endswith(".bin")
|
||||
) # oldest first (lexicographic = chronological for our timestamp format)
|
||||
# Oldest first (lexicographic = chronological for our timestamp format).
|
||||
try:
|
||||
bins = sorted(f for f in os.listdir(snapshot_dir) if f.endswith(".bin"))
|
||||
except OSError:
|
||||
return
|
||||
excess = len(bins) - max_snapshots
|
||||
for fname in bins[:excess]:
|
||||
stem = fname[:-4] # strip .bin
|
||||
for ext in (".bin", ".json"):
|
||||
try:
|
||||
os.remove(os.path.join(snapshot_dir, stem + ext))
|
||||
except OSError:
|
||||
pass
|
||||
except OSError as exc:
|
||||
log.debug("Could not remove %s: %s", stem + ext, exc)
|
||||
|
||||
|
||||
def restore(gpu, snapshot_dir: str, filepath: str = None) -> bool:
|
||||
def restore(gpu, snapshot_dir: str, filepath: str | None = None) -> bool:
|
||||
"""Restore a ClockBoostTable snapshot from disk.
|
||||
|
||||
If no filepath is given, uses the most recent snapshot in snapshot_dir.
|
||||
@@ -88,10 +104,14 @@ def restore(gpu, snapshot_dir: str, filepath: str = None) -> bool:
|
||||
if not os.path.isdir(snapshot_dir):
|
||||
print(f"No snapshots found in {snapshot_dir}")
|
||||
return False
|
||||
bins = sorted(
|
||||
[f for f in os.listdir(snapshot_dir) if f.endswith(".bin")],
|
||||
reverse=True,
|
||||
)
|
||||
try:
|
||||
bins = sorted(
|
||||
[f for f in os.listdir(snapshot_dir) if f.endswith(".bin")],
|
||||
reverse=True,
|
||||
)
|
||||
except OSError:
|
||||
print(f"No snapshots found in {snapshot_dir}")
|
||||
return False
|
||||
if not bins:
|
||||
print(f"No snapshot .bin files in {snapshot_dir}")
|
||||
return False
|
||||
@@ -101,8 +121,12 @@ def restore(gpu, snapshot_dir: str, filepath: str = None) -> bool:
|
||||
print(f"Snapshot file not found: {filepath}")
|
||||
return False
|
||||
|
||||
with open(filepath, "rb") as f:
|
||||
raw = f.read()
|
||||
try:
|
||||
with open(filepath, "rb") as f:
|
||||
raw = f.read()
|
||||
except OSError as exc:
|
||||
print(f"Failed to read snapshot {filepath}: {exc}")
|
||||
return False
|
||||
|
||||
if len(raw) != CT_SIZE:
|
||||
print(f"Snapshot size mismatch: expected {CT_SIZE}, got {len(raw)}")
|
||||
@@ -134,8 +158,13 @@ def list_snapshots(snapshot_dir: str) -> list[SnapshotInfo]:
|
||||
if not os.path.isdir(snapshot_dir):
|
||||
return []
|
||||
|
||||
try:
|
||||
fnames = sorted(os.listdir(snapshot_dir), reverse=True)
|
||||
except OSError:
|
||||
return []
|
||||
|
||||
results = []
|
||||
for fname in sorted(os.listdir(snapshot_dir), reverse=True):
|
||||
for fname in fnames:
|
||||
if not fname.endswith(".json"):
|
||||
continue
|
||||
meta_path = os.path.join(snapshot_dir, fname)
|
||||
@@ -143,13 +172,15 @@ def list_snapshots(snapshot_dir: str) -> list[SnapshotInfo]:
|
||||
with open(meta_path) as f:
|
||||
meta = json.load(f)
|
||||
bin_path = meta.get("file", meta_path.replace(".json", ".bin"))
|
||||
results.append(SnapshotInfo(
|
||||
filepath=bin_path,
|
||||
timestamp=meta.get("timestamp", ""),
|
||||
gpu=meta.get("gpu", ""),
|
||||
nonzero_offsets=meta.get("nonzero_offsets", 0),
|
||||
size=meta.get("size", 0),
|
||||
))
|
||||
results.append(
|
||||
SnapshotInfo(
|
||||
filepath=bin_path,
|
||||
timestamp=meta.get("timestamp", ""),
|
||||
gpu=meta.get("gpu", ""),
|
||||
nonzero_offsets=meta.get("nonzero_offsets", 0),
|
||||
size=meta.get("size", 0),
|
||||
)
|
||||
)
|
||||
except (json.JSONDecodeError, KeyError):
|
||||
continue
|
||||
|
||||
|
||||
+89
-38
@@ -21,12 +21,12 @@ def _gpu_stable_key(info) -> str:
|
||||
|
||||
def apply_profile(gpu_index: int, name: str, cfg) -> list[str]:
|
||||
"""Apply a named profile to the given GPU. Returns a list of error strings."""
|
||||
from .native import load_profile
|
||||
from ..hal.gpu import get_gpu
|
||||
from ..hal.limits import set_clock_offsets, set_power_limit
|
||||
from ..hal.vfcurve import write_offsets, reset_offsets
|
||||
from ..hal.snapshot import save as snapshot_save
|
||||
from ..hal.vfcurve import reset_offsets, write_offsets
|
||||
from ..safety import validate_write
|
||||
from .native import load_profile
|
||||
|
||||
safe_name = "".join(c for c in name if c.isalnum() or c in " _-()").strip()
|
||||
filepath = os.path.join(cfg.profile_dir, f"{safe_name}.json")
|
||||
@@ -48,19 +48,25 @@ def apply_profile(gpu_index: int, name: str, cfg) -> list[str]:
|
||||
errs.append(f"Power limit: {msg}")
|
||||
|
||||
if profile.curve_deltas:
|
||||
deltas = {int(k): v for k, v in profile.curve_deltas.items()}
|
||||
errors = validate_write(deltas, cfg.max_delta_khz)
|
||||
if errors:
|
||||
errs.append("Curve: " + "; ".join(errors))
|
||||
try:
|
||||
deltas = {int(k): v for k, v in profile.curve_deltas.items()}
|
||||
except ValueError:
|
||||
errs.append("Curve: invalid point keys in profile")
|
||||
else:
|
||||
if cfg.auto_snapshot:
|
||||
try:
|
||||
snapshot_save(gpu, gpu_name, cfg.snapshot_dir, cfg.max_snapshots)
|
||||
except Exception as exc:
|
||||
log.warning("Auto-snapshot failed: %s", exc)
|
||||
ret, desc = write_offsets(gpu, deltas)
|
||||
if ret != 0:
|
||||
errs.append(f"Curve write failed ({ret}): {desc}")
|
||||
errors = validate_write(deltas, cfg.max_delta_khz)
|
||||
if errors:
|
||||
errs.append("Curve: " + "; ".join(errors))
|
||||
else:
|
||||
if cfg.auto_snapshot:
|
||||
try:
|
||||
snapshot_save(
|
||||
gpu, gpu_name, cfg.snapshot_dir, cfg.max_snapshots
|
||||
)
|
||||
except Exception as exc:
|
||||
log.warning("Auto-snapshot failed: %s", exc)
|
||||
ret, desc = write_offsets(gpu, deltas)
|
||||
if ret != 0:
|
||||
errs.append(f"Curve write failed ({ret}): {desc}")
|
||||
else:
|
||||
reset_offsets(gpu)
|
||||
|
||||
@@ -69,9 +75,9 @@ def apply_profile(gpu_index: int, name: str, cfg) -> list[str]:
|
||||
|
||||
def apply_with_retry(gpu_index: int, name: str, cfg, max_retries: int = 3) -> bool:
|
||||
"""Apply a named profile with read-back verification, retrying on mismatch."""
|
||||
from .native import load_profile
|
||||
from ..hal.gpu import get_gpu
|
||||
from ..hal.vfcurve import read_clock_offsets
|
||||
from .native import load_profile
|
||||
|
||||
safe_name = "".join(c for c in name if c.isalnum() or c in " _-()").strip()
|
||||
filepath = os.path.join(cfg.profile_dir, f"{safe_name}.json")
|
||||
@@ -82,51 +88,88 @@ def apply_with_retry(gpu_index: int, name: str, cfg, max_retries: int = 3) -> bo
|
||||
log.warning("Auto-load profile %r not found — skipping GPU %d", name, gpu_index)
|
||||
return False
|
||||
|
||||
expected: dict[int, int] = (
|
||||
{int(k): v for k, v in profile.curve_deltas.items()}
|
||||
if profile.curve_deltas else {}
|
||||
)
|
||||
try:
|
||||
expected: dict[int, int] = (
|
||||
{int(k): v for k, v in profile.curve_deltas.items()}
|
||||
if profile.curve_deltas
|
||||
else {}
|
||||
)
|
||||
except ValueError:
|
||||
log.warning(
|
||||
"Profile %r has invalid curve point keys — skipping GPU %d",
|
||||
name,
|
||||
gpu_index,
|
||||
)
|
||||
return False
|
||||
|
||||
for attempt in range(max_retries):
|
||||
try:
|
||||
errs = apply_profile(gpu_index, name, cfg)
|
||||
except Exception as exc:
|
||||
log.warning("Auto-load attempt %d/%d exception: %s", attempt + 1, max_retries, exc)
|
||||
log.warning(
|
||||
"Auto-load attempt %d/%d exception: %s", attempt + 1, max_retries, exc
|
||||
)
|
||||
errs = [str(exc)]
|
||||
|
||||
if errs:
|
||||
log.warning("Auto-load attempt %d/%d errors: %s",
|
||||
attempt + 1, max_retries, "; ".join(errs))
|
||||
log.warning(
|
||||
"Auto-load attempt %d/%d errors: %s",
|
||||
attempt + 1,
|
||||
max_retries,
|
||||
"; ".join(errs),
|
||||
)
|
||||
elif expected:
|
||||
gpu, _ = get_gpu(index=gpu_index)
|
||||
offsets, err = read_clock_offsets(gpu)
|
||||
if offsets is None:
|
||||
log.warning("Auto-load attempt %d/%d: read-back failed: %s",
|
||||
attempt + 1, max_retries, err)
|
||||
log.warning(
|
||||
"Auto-load attempt %d/%d: read-back failed: %s",
|
||||
attempt + 1,
|
||||
max_retries,
|
||||
err,
|
||||
)
|
||||
else:
|
||||
mismatches = [
|
||||
f"pt{idx}: expected {val/1000:+.0f}MHz got {offsets[idx]/1000:+.0f}MHz"
|
||||
f"pt{idx}: expected {val / 1000:+.0f}MHz got {offsets[idx] / 1000:+.0f}MHz"
|
||||
for idx, val in expected.items()
|
||||
if idx < len(offsets) and offsets[idx] != val
|
||||
]
|
||||
if not mismatches:
|
||||
log.info("Auto-load profile %r verified on GPU %d (attempt %d/%d)",
|
||||
name, gpu_index, attempt + 1, max_retries)
|
||||
log.info(
|
||||
"Auto-load profile %r verified on GPU %d (attempt %d/%d)",
|
||||
name,
|
||||
gpu_index,
|
||||
attempt + 1,
|
||||
max_retries,
|
||||
)
|
||||
return True
|
||||
log.warning("Auto-load attempt %d/%d: read-back mismatch — %s",
|
||||
attempt + 1, max_retries, "; ".join(mismatches))
|
||||
log.warning(
|
||||
"Auto-load attempt %d/%d: read-back mismatch — %s",
|
||||
attempt + 1,
|
||||
max_retries,
|
||||
"; ".join(mismatches),
|
||||
)
|
||||
else:
|
||||
log.info("Auto-load profile %r applied on GPU %d (attempt %d/%d)",
|
||||
name, gpu_index, attempt + 1, max_retries)
|
||||
log.info(
|
||||
"Auto-load profile %r applied on GPU %d (attempt %d/%d)",
|
||||
name,
|
||||
gpu_index,
|
||||
attempt + 1,
|
||||
max_retries,
|
||||
)
|
||||
return True
|
||||
|
||||
if attempt < max_retries - 1:
|
||||
delay = 2 ** attempt # 1s, 2s, 4s
|
||||
delay = 2**attempt # 1s, 2s, 4s
|
||||
log.info("Retrying auto-load in %ds…", delay)
|
||||
time.sleep(delay)
|
||||
|
||||
log.warning("Auto-load profile %r failed after %d attempts on GPU %d",
|
||||
name, max_retries, gpu_index)
|
||||
log.warning(
|
||||
"Auto-load profile %r failed after %d attempts on GPU %d",
|
||||
name,
|
||||
max_retries,
|
||||
gpu_index,
|
||||
)
|
||||
return False
|
||||
|
||||
|
||||
@@ -153,13 +196,19 @@ def run_autoload() -> None:
|
||||
return
|
||||
|
||||
from ..config import Config
|
||||
|
||||
cfg = Config()
|
||||
for key in ("max_delta_khz", "auto_snapshot", "max_snapshots",
|
||||
"snapshot_dir", "profile_dir"):
|
||||
for key in (
|
||||
"max_delta_khz",
|
||||
"auto_snapshot",
|
||||
"max_snapshots",
|
||||
"snapshot_dir",
|
||||
"profile_dir",
|
||||
):
|
||||
if key in cfg_data:
|
||||
setattr(cfg, key, cfg_data[key])
|
||||
|
||||
from ..hal.gpu import init_nvapi, discover_gpus
|
||||
from ..hal.gpu import discover_gpus, init_nvapi
|
||||
from ..hal.monitoring import init_nvml, shutdown_nvml
|
||||
|
||||
# Retry NvAPI init — the driver may not be fully ready at early boot.
|
||||
@@ -189,7 +238,9 @@ def run_autoload() -> None:
|
||||
if gpu_idx is None:
|
||||
log.warning("Auto-load: no GPU found with key %r — skipping", gpu_key)
|
||||
continue
|
||||
log.info("Auto-loading profile %r on GPU %d (%s)", profile_name, gpu_idx, gpu_key)
|
||||
log.info(
|
||||
"Auto-loading profile %r on GPU %d (%s)", profile_name, gpu_idx, gpu_key
|
||||
)
|
||||
apply_with_retry(gpu_idx, profile_name, cfg)
|
||||
|
||||
shutdown_nvml()
|
||||
+30
-18
@@ -1,39 +1,52 @@
|
||||
"""Native profile storage and schema."""
|
||||
|
||||
import json
|
||||
import os
|
||||
import glob
|
||||
from dataclasses import dataclass, asdict
|
||||
from typing import Dict, Optional, List
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
from dataclasses import asdict, dataclass
|
||||
|
||||
log = logging.getLogger("nvcurve.profiles.native")
|
||||
|
||||
|
||||
@dataclass
|
||||
class ProfileData:
|
||||
name: str
|
||||
gpu_name: str
|
||||
curve_deltas: Dict[str, int] # { "index": delta_khz }
|
||||
mem_offset_mhz: Optional[int] = None
|
||||
power_limit_w: Optional[int] = None
|
||||
fan_curve: Optional[List[Dict[str, int]]] = None
|
||||
curve_deltas: dict[str, int] # { "index": delta_khz }
|
||||
mem_offset_mhz: int | None = None
|
||||
power_limit_w: int | None = None
|
||||
fan_curve: list[dict[str, int]] | None = None
|
||||
|
||||
|
||||
def save_profile(profile_dir: str, data: ProfileData) -> str:
|
||||
"""Save profile to JSON, sanitising the filename."""
|
||||
os.makedirs(profile_dir, exist_ok=True)
|
||||
try:
|
||||
os.makedirs(profile_dir, exist_ok=True)
|
||||
except OSError as exc:
|
||||
raise RuntimeError(f"Cannot create profile dir {profile_dir}: {exc}") from exc
|
||||
safe_name = "".join(c for c in data.name if c.isalnum() or c in " _-()").strip()
|
||||
if not safe_name:
|
||||
safe_name = "Unnamed"
|
||||
|
||||
|
||||
filepath = os.path.join(profile_dir, f"{safe_name}.json")
|
||||
with open(filepath, "w", encoding="utf-8") as f:
|
||||
json.dump(asdict(data), f, indent=2)
|
||||
try:
|
||||
with open(filepath, "w", encoding="utf-8") as f:
|
||||
json.dump(asdict(data), f, indent=2)
|
||||
except OSError as exc:
|
||||
raise RuntimeError(f"Cannot write profile {filepath}: {exc}") from exc
|
||||
return filepath
|
||||
|
||||
|
||||
def load_profile(filepath: str) -> ProfileData:
|
||||
"""Load profile from JSON."""
|
||||
with open(filepath, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
try:
|
||||
with open(filepath, encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
except FileNotFoundError:
|
||||
raise
|
||||
except (OSError, json.JSONDecodeError) as exc:
|
||||
raise RuntimeError(f"Cannot read profile {filepath}: {exc}") from exc
|
||||
# Migrate old field names.
|
||||
if "vram_p0_offset_mhz" in data and "mem_offset_mhz" not in data:
|
||||
data["mem_offset_mhz"] = data.pop("vram_p0_offset_mhz")
|
||||
@@ -43,7 +56,7 @@ def load_profile(filepath: str) -> ProfileData:
|
||||
return ProfileData(**data)
|
||||
|
||||
|
||||
def list_profiles(profile_dir: str) -> List[ProfileData]:
|
||||
def list_profiles(profile_dir: str) -> list[ProfileData]:
|
||||
"""Return a list of all safely readable profiles."""
|
||||
if not os.path.exists(profile_dir):
|
||||
return []
|
||||
@@ -51,9 +64,8 @@ def list_profiles(profile_dir: str) -> List[ProfileData]:
|
||||
for fp in glob.glob(os.path.join(profile_dir, "*.json")):
|
||||
try:
|
||||
profiles.append(load_profile(fp))
|
||||
except Exception as e:
|
||||
# log warning ideally, but swallowing for robustness
|
||||
pass
|
||||
except Exception as exc:
|
||||
log.debug("Skipping unreadable profile %s: %s", fp, exc)
|
||||
# Sort alphabetically by name
|
||||
profiles.sort(key=lambda p: p.name.lower())
|
||||
return profiles
|
||||
|
||||
+6
-3
@@ -90,8 +90,8 @@ def _open_browser_as_user(url: str) -> None:
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
return
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
log.debug("runuser xdg-open failed, falling back to webbrowser: %s", exc)
|
||||
import webbrowser
|
||||
|
||||
webbrowser.open(url)
|
||||
@@ -1692,7 +1692,10 @@ def _resolve_dist_dir() -> str:
|
||||
the project-root layout used during local development.
|
||||
"""
|
||||
try:
|
||||
from importlib.resources import files as _resource_files
|
||||
# Project requires Python >= 3.12, so the 3.7-compat finding is a false positive.
|
||||
from importlib.resources import ( # nosemgrep: python.lang.compatibility.python37.python37-compatibility-importlib2
|
||||
files as _resource_files,
|
||||
)
|
||||
|
||||
candidate = _resource_files("nvcurve") / "frontend" / "dist"
|
||||
if candidate.is_dir():
|
||||
|
||||
Reference in new issue
Block a user