Files
nvcurve/nvcurve/profiles/apply.py
T
ARIA 930e56bd07 Clean up LSP diagnostics across backend and frontend
Backend (nvcurve/):
- hal/fans.py, hal/limits.py, hal/gpu.py: replace conditional pynvml
  imports with the established 'pynvml: Any = _pynvml_import' pattern
  (fixes ~50 'possibly unbound' errors); type the result dicts; guard
  query_interface() results; explicit uuid/pci-bus parsing (int, hex
  convention documented); modernize Optional[T] -> T | None
- cli.py: fix 'curve_state' possibly-unbound and snap_path None handling
  in cmd_setup; wrap unchecked int()/open()/makedirs() calls in
  try/except with clean CLI errors; add module logger for silent
  except-pass blocks; raise ... from exc; fix unused loop vars and
  set-comprehension
- hal/snapshot.py: filepath: str | None; wrap all file ops; sorted
  imports; remove unused CT_POINTS import
- daemon.py: extract 0o666 to _SOCKET_MODE constant (intentional for
  /run sockets) with nosemgrep
- server.py: nosemgrep for Python 3.7-compat false positive (project
  requires >= 3.12); log previously-swallowed exception
- profiles/native.py, profiles/apply.py: wrap file ops and int(k)
  profile-key parsing; sorted imports; modernize typing

Frontend (frontend/src):
- Add .js extensions to all relative imports (standard TS-ESM; Vite
  resolves .js -> .ts)
- React.FormEvent (deprecated in React 19 types) -> React.SubmitEvent
- catch (e: any) -> catch (e: unknown) + instanceof Error narrowing
- React-hooks: move ref writes from render into effects; convert
  viewport reset to render-phase state adjustment; split
  selectPoint(index, multi) into selectPoint + togglePoint (no flag
  argument); remove non-null assertion
- Static inline styles -> Tailwind classes (dynamic positioning/cursor
  styles kept)
- Remove non-standard 'container' option from scrollIntoView (browsers
  ignore unknown options) which had orphaned a @ts-expect-error
- Object.fromEntries for Map -> Record conversion

Tooling:
- .gitignore: ignore .codegraph/ local tool data

Verified: tsc --noEmit, vite production build, python imports, and
full LSP scan (0 errors/warnings in both projects).
2026-09-08 23:57:30 +02:00

247 lines
7.8 KiB
Python

"""Apply saved profiles to hardware, with optional read-back verification."""
import json
import logging
import os
import sys
import time
log = logging.getLogger("nvcurve.profiles.apply")
_PERSISTENT_CONFIG_FILE = "/etc/nvcurve/config.json"
def _gpu_stable_key(info) -> str:
if info.uuid:
return info.uuid
if info.pci_bus_id is not None:
return f"pci:{info.pci_bus_id:04x}"
return f"idx:{info.index}"
def apply_profile(gpu_index: int, name: str, cfg) -> list[str]:
"""Apply a named profile to the given GPU. Returns a list of error strings."""
from ..hal.gpu import get_gpu
from ..hal.limits import set_clock_offsets, set_power_limit
from ..hal.snapshot import save as snapshot_save
from ..hal.vfcurve import reset_offsets, write_offsets
from ..safety import validate_write
from .native import load_profile
safe_name = "".join(c for c in name if c.isalnum() or c in " _-()").strip()
filepath = os.path.join(cfg.profile_dir, f"{safe_name}.json")
profile = load_profile(filepath) # raises FileNotFoundError if missing
gpu, gpu_name = get_gpu(index=gpu_index)
errs: list[str] = []
# Apply mem offset first — driver may reset curve table as a side-effect.
if profile.mem_offset_mhz is not None:
ok, msg = set_clock_offsets(None, profile.mem_offset_mhz, gpu_index)
if not ok:
errs.append(f"Mem offset: {msg}")
if profile.power_limit_w is not None:
ok, msg = set_power_limit(profile.power_limit_w, gpu_index)
if not ok:
errs.append(f"Power limit: {msg}")
if profile.curve_deltas:
try:
deltas = {int(k): v for k, v in profile.curve_deltas.items()}
except ValueError:
errs.append("Curve: invalid point keys in profile")
else:
errors = validate_write(deltas, cfg.max_delta_khz)
if errors:
errs.append("Curve: " + "; ".join(errors))
else:
if cfg.auto_snapshot:
try:
snapshot_save(
gpu, gpu_name, cfg.snapshot_dir, cfg.max_snapshots
)
except Exception as exc:
log.warning("Auto-snapshot failed: %s", exc)
ret, desc = write_offsets(gpu, deltas)
if ret != 0:
errs.append(f"Curve write failed ({ret}): {desc}")
else:
reset_offsets(gpu)
return errs
def apply_with_retry(gpu_index: int, name: str, cfg, max_retries: int = 3) -> bool:
"""Apply a named profile with read-back verification, retrying on mismatch."""
from ..hal.gpu import get_gpu
from ..hal.vfcurve import read_clock_offsets
from .native import load_profile
safe_name = "".join(c for c in name if c.isalnum() or c in " _-()").strip()
filepath = os.path.join(cfg.profile_dir, f"{safe_name}.json")
try:
profile = load_profile(filepath)
except FileNotFoundError:
log.warning("Auto-load profile %r not found — skipping GPU %d", name, gpu_index)
return False
try:
expected: dict[int, int] = (
{int(k): v for k, v in profile.curve_deltas.items()}
if profile.curve_deltas
else {}
)
except ValueError:
log.warning(
"Profile %r has invalid curve point keys — skipping GPU %d",
name,
gpu_index,
)
return False
for attempt in range(max_retries):
try:
errs = apply_profile(gpu_index, name, cfg)
except Exception as exc:
log.warning(
"Auto-load attempt %d/%d exception: %s", attempt + 1, max_retries, exc
)
errs = [str(exc)]
if errs:
log.warning(
"Auto-load attempt %d/%d errors: %s",
attempt + 1,
max_retries,
"; ".join(errs),
)
elif expected:
gpu, _ = get_gpu(index=gpu_index)
offsets, err = read_clock_offsets(gpu)
if offsets is None:
log.warning(
"Auto-load attempt %d/%d: read-back failed: %s",
attempt + 1,
max_retries,
err,
)
else:
mismatches = [
f"pt{idx}: expected {val / 1000:+.0f}MHz got {offsets[idx] / 1000:+.0f}MHz"
for idx, val in expected.items()
if idx < len(offsets) and offsets[idx] != val
]
if not mismatches:
log.info(
"Auto-load profile %r verified on GPU %d (attempt %d/%d)",
name,
gpu_index,
attempt + 1,
max_retries,
)
return True
log.warning(
"Auto-load attempt %d/%d: read-back mismatch — %s",
attempt + 1,
max_retries,
"; ".join(mismatches),
)
else:
log.info(
"Auto-load profile %r applied on GPU %d (attempt %d/%d)",
name,
gpu_index,
attempt + 1,
max_retries,
)
return True
if attempt < max_retries - 1:
delay = 2**attempt # 1s, 2s, 4s
log.info("Retrying auto-load in %ds…", delay)
time.sleep(delay)
log.warning(
"Auto-load profile %r failed after %d attempts on GPU %d",
name,
max_retries,
gpu_index,
)
return False
def run_autoload() -> None:
"""Read config and apply all configured auto-load profiles. Requires root."""
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s %(levelname)s %(name)s: %(message)s",
)
if os.geteuid() != 0:
print("nvcurve autoload: must run as root", file=sys.stderr)
sys.exit(1)
try:
with open(_PERSISTENT_CONFIG_FILE) as f:
cfg_data = json.load(f)
except Exception:
cfg_data = {}
auto_load_profiles: dict = cfg_data.get("auto_load_profiles", {})
if not auto_load_profiles:
log.info("No auto-load profiles configured.")
return
from ..config import Config
cfg = Config()
for key in (
"max_delta_khz",
"auto_snapshot",
"max_snapshots",
"snapshot_dir",
"profile_dir",
):
if key in cfg_data:
setattr(cfg, key, cfg_data[key])
from ..hal.gpu import discover_gpus, init_nvapi
from ..hal.monitoring import init_nvml, shutdown_nvml
# Retry NvAPI init — the driver may not be fully ready at early boot.
nvapi_ready = False
for attempt in range(1, 17):
try:
init_nvapi()
nvapi_ready = True
break
except Exception as exc:
log.warning("NvAPI init attempt %d/16 failed: %s", attempt, exc)
time.sleep(2)
if not nvapi_ready:
log.error("Failed to initialize NvAPI after 16 attempts.")
sys.exit(1)
init_nvml() # best-effort
gpus = discover_gpus()
if not gpus:
log.warning("No GPUs discovered.")
key_to_idx = {_gpu_stable_key(info): info.index for info in gpus}
for gpu_key, profile_name in auto_load_profiles.items():
if not profile_name:
continue
gpu_idx = key_to_idx.get(gpu_key)
if gpu_idx is None:
log.warning("Auto-load: no GPU found with key %r — skipping", gpu_key)
continue
log.info(
"Auto-loading profile %r on GPU %d (%s)", profile_name, gpu_idx, gpu_key
)
apply_with_retry(gpu_idx, profile_name, cfg)
shutdown_nvml()