Backend (nvcurve/): - hal/fans.py, hal/limits.py, hal/gpu.py: replace conditional pynvml imports with the established 'pynvml: Any = _pynvml_import' pattern (fixes ~50 'possibly unbound' errors); type the result dicts; guard query_interface() results; explicit uuid/pci-bus parsing (int, hex convention documented); modernize Optional[T] -> T | None - cli.py: fix 'curve_state' possibly-unbound and snap_path None handling in cmd_setup; wrap unchecked int()/open()/makedirs() calls in try/except with clean CLI errors; add module logger for silent except-pass blocks; raise ... from exc; fix unused loop vars and set-comprehension - hal/snapshot.py: filepath: str | None; wrap all file ops; sorted imports; remove unused CT_POINTS import - daemon.py: extract 0o666 to _SOCKET_MODE constant (intentional for /run sockets) with nosemgrep - server.py: nosemgrep for Python 3.7-compat false positive (project requires >= 3.12); log previously-swallowed exception - profiles/native.py, profiles/apply.py: wrap file ops and int(k) profile-key parsing; sorted imports; modernize typing Frontend (frontend/src): - Add .js extensions to all relative imports (standard TS-ESM; Vite resolves .js -> .ts) - React.FormEvent (deprecated in React 19 types) -> React.SubmitEvent - catch (e: any) -> catch (e: unknown) + instanceof Error narrowing - React-hooks: move ref writes from render into effects; convert viewport reset to render-phase state adjustment; split selectPoint(index, multi) into selectPoint + togglePoint (no flag argument); remove non-null assertion - Static inline styles -> Tailwind classes (dynamic positioning/cursor styles kept) - Remove non-standard 'container' option from scrollIntoView (browsers ignore unknown options) which had orphaned a @ts-expect-error - Object.fromEntries for Map -> Record conversion Tooling: - .gitignore: ignore .codegraph/ local tool data Verified: tsc --noEmit, vite production build, python imports, and full LSP scan (0 errors/warnings in both projects).
247 lines
7.8 KiB
Python
247 lines
7.8 KiB
Python
"""Apply saved profiles to hardware, with optional read-back verification."""
|
|
|
|
import json
|
|
import logging
|
|
import os
|
|
import sys
|
|
import time
|
|
|
|
log = logging.getLogger("nvcurve.profiles.apply")
|
|
|
|
_PERSISTENT_CONFIG_FILE = "/etc/nvcurve/config.json"
|
|
|
|
|
|
def _gpu_stable_key(info) -> str:
|
|
if info.uuid:
|
|
return info.uuid
|
|
if info.pci_bus_id is not None:
|
|
return f"pci:{info.pci_bus_id:04x}"
|
|
return f"idx:{info.index}"
|
|
|
|
|
|
def apply_profile(gpu_index: int, name: str, cfg) -> list[str]:
|
|
"""Apply a named profile to the given GPU. Returns a list of error strings."""
|
|
from ..hal.gpu import get_gpu
|
|
from ..hal.limits import set_clock_offsets, set_power_limit
|
|
from ..hal.snapshot import save as snapshot_save
|
|
from ..hal.vfcurve import reset_offsets, write_offsets
|
|
from ..safety import validate_write
|
|
from .native import load_profile
|
|
|
|
safe_name = "".join(c for c in name if c.isalnum() or c in " _-()").strip()
|
|
filepath = os.path.join(cfg.profile_dir, f"{safe_name}.json")
|
|
|
|
profile = load_profile(filepath) # raises FileNotFoundError if missing
|
|
gpu, gpu_name = get_gpu(index=gpu_index)
|
|
|
|
errs: list[str] = []
|
|
|
|
# Apply mem offset first — driver may reset curve table as a side-effect.
|
|
if profile.mem_offset_mhz is not None:
|
|
ok, msg = set_clock_offsets(None, profile.mem_offset_mhz, gpu_index)
|
|
if not ok:
|
|
errs.append(f"Mem offset: {msg}")
|
|
|
|
if profile.power_limit_w is not None:
|
|
ok, msg = set_power_limit(profile.power_limit_w, gpu_index)
|
|
if not ok:
|
|
errs.append(f"Power limit: {msg}")
|
|
|
|
if profile.curve_deltas:
|
|
try:
|
|
deltas = {int(k): v for k, v in profile.curve_deltas.items()}
|
|
except ValueError:
|
|
errs.append("Curve: invalid point keys in profile")
|
|
else:
|
|
errors = validate_write(deltas, cfg.max_delta_khz)
|
|
if errors:
|
|
errs.append("Curve: " + "; ".join(errors))
|
|
else:
|
|
if cfg.auto_snapshot:
|
|
try:
|
|
snapshot_save(
|
|
gpu, gpu_name, cfg.snapshot_dir, cfg.max_snapshots
|
|
)
|
|
except Exception as exc:
|
|
log.warning("Auto-snapshot failed: %s", exc)
|
|
ret, desc = write_offsets(gpu, deltas)
|
|
if ret != 0:
|
|
errs.append(f"Curve write failed ({ret}): {desc}")
|
|
else:
|
|
reset_offsets(gpu)
|
|
|
|
return errs
|
|
|
|
|
|
def apply_with_retry(gpu_index: int, name: str, cfg, max_retries: int = 3) -> bool:
|
|
"""Apply a named profile with read-back verification, retrying on mismatch."""
|
|
from ..hal.gpu import get_gpu
|
|
from ..hal.vfcurve import read_clock_offsets
|
|
from .native import load_profile
|
|
|
|
safe_name = "".join(c for c in name if c.isalnum() or c in " _-()").strip()
|
|
filepath = os.path.join(cfg.profile_dir, f"{safe_name}.json")
|
|
|
|
try:
|
|
profile = load_profile(filepath)
|
|
except FileNotFoundError:
|
|
log.warning("Auto-load profile %r not found — skipping GPU %d", name, gpu_index)
|
|
return False
|
|
|
|
try:
|
|
expected: dict[int, int] = (
|
|
{int(k): v for k, v in profile.curve_deltas.items()}
|
|
if profile.curve_deltas
|
|
else {}
|
|
)
|
|
except ValueError:
|
|
log.warning(
|
|
"Profile %r has invalid curve point keys — skipping GPU %d",
|
|
name,
|
|
gpu_index,
|
|
)
|
|
return False
|
|
|
|
for attempt in range(max_retries):
|
|
try:
|
|
errs = apply_profile(gpu_index, name, cfg)
|
|
except Exception as exc:
|
|
log.warning(
|
|
"Auto-load attempt %d/%d exception: %s", attempt + 1, max_retries, exc
|
|
)
|
|
errs = [str(exc)]
|
|
|
|
if errs:
|
|
log.warning(
|
|
"Auto-load attempt %d/%d errors: %s",
|
|
attempt + 1,
|
|
max_retries,
|
|
"; ".join(errs),
|
|
)
|
|
elif expected:
|
|
gpu, _ = get_gpu(index=gpu_index)
|
|
offsets, err = read_clock_offsets(gpu)
|
|
if offsets is None:
|
|
log.warning(
|
|
"Auto-load attempt %d/%d: read-back failed: %s",
|
|
attempt + 1,
|
|
max_retries,
|
|
err,
|
|
)
|
|
else:
|
|
mismatches = [
|
|
f"pt{idx}: expected {val / 1000:+.0f}MHz got {offsets[idx] / 1000:+.0f}MHz"
|
|
for idx, val in expected.items()
|
|
if idx < len(offsets) and offsets[idx] != val
|
|
]
|
|
if not mismatches:
|
|
log.info(
|
|
"Auto-load profile %r verified on GPU %d (attempt %d/%d)",
|
|
name,
|
|
gpu_index,
|
|
attempt + 1,
|
|
max_retries,
|
|
)
|
|
return True
|
|
log.warning(
|
|
"Auto-load attempt %d/%d: read-back mismatch — %s",
|
|
attempt + 1,
|
|
max_retries,
|
|
"; ".join(mismatches),
|
|
)
|
|
else:
|
|
log.info(
|
|
"Auto-load profile %r applied on GPU %d (attempt %d/%d)",
|
|
name,
|
|
gpu_index,
|
|
attempt + 1,
|
|
max_retries,
|
|
)
|
|
return True
|
|
|
|
if attempt < max_retries - 1:
|
|
delay = 2**attempt # 1s, 2s, 4s
|
|
log.info("Retrying auto-load in %ds…", delay)
|
|
time.sleep(delay)
|
|
|
|
log.warning(
|
|
"Auto-load profile %r failed after %d attempts on GPU %d",
|
|
name,
|
|
max_retries,
|
|
gpu_index,
|
|
)
|
|
return False
|
|
|
|
|
|
def run_autoload() -> None:
|
|
"""Read config and apply all configured auto-load profiles. Requires root."""
|
|
logging.basicConfig(
|
|
level=logging.INFO,
|
|
format="%(asctime)s %(levelname)s %(name)s: %(message)s",
|
|
)
|
|
|
|
if os.geteuid() != 0:
|
|
print("nvcurve autoload: must run as root", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
try:
|
|
with open(_PERSISTENT_CONFIG_FILE) as f:
|
|
cfg_data = json.load(f)
|
|
except Exception:
|
|
cfg_data = {}
|
|
|
|
auto_load_profiles: dict = cfg_data.get("auto_load_profiles", {})
|
|
if not auto_load_profiles:
|
|
log.info("No auto-load profiles configured.")
|
|
return
|
|
|
|
from ..config import Config
|
|
|
|
cfg = Config()
|
|
for key in (
|
|
"max_delta_khz",
|
|
"auto_snapshot",
|
|
"max_snapshots",
|
|
"snapshot_dir",
|
|
"profile_dir",
|
|
):
|
|
if key in cfg_data:
|
|
setattr(cfg, key, cfg_data[key])
|
|
|
|
from ..hal.gpu import discover_gpus, init_nvapi
|
|
from ..hal.monitoring import init_nvml, shutdown_nvml
|
|
|
|
# Retry NvAPI init — the driver may not be fully ready at early boot.
|
|
nvapi_ready = False
|
|
for attempt in range(1, 17):
|
|
try:
|
|
init_nvapi()
|
|
nvapi_ready = True
|
|
break
|
|
except Exception as exc:
|
|
log.warning("NvAPI init attempt %d/16 failed: %s", attempt, exc)
|
|
time.sleep(2)
|
|
if not nvapi_ready:
|
|
log.error("Failed to initialize NvAPI after 16 attempts.")
|
|
sys.exit(1)
|
|
|
|
init_nvml() # best-effort
|
|
gpus = discover_gpus()
|
|
if not gpus:
|
|
log.warning("No GPUs discovered.")
|
|
|
|
key_to_idx = {_gpu_stable_key(info): info.index for info in gpus}
|
|
for gpu_key, profile_name in auto_load_profiles.items():
|
|
if not profile_name:
|
|
continue
|
|
gpu_idx = key_to_idx.get(gpu_key)
|
|
if gpu_idx is None:
|
|
log.warning("Auto-load: no GPU found with key %r — skipping", gpu_key)
|
|
continue
|
|
log.info(
|
|
"Auto-loading profile %r on GPU %d (%s)", profile_name, gpu_idx, gpu_key
|
|
)
|
|
apply_with_retry(gpu_idx, profile_name, cfg)
|
|
|
|
shutdown_nvml()
|