fix: retry NvAPI init on autoload to handle early boot race

init_nvapi() previously called sys.exit(1) directly, which raised
SystemExit (uncaught by except Exception). The autoload subprocess
would die instantly if the NVIDIA driver wasn't fully loaded at boot.

- Raise RuntimeError from init_nvapi() so callers can handle failures
- Add 16-attempt retry loop (2s interval, 32s total) in run_autoload()
This commit is contained in:
hhofmann committed 2026-08-09 08:28:59 +02:00
1 parent 0c1f853a5c
commit f417297713
2 files changed
+10 -3

No files matched your search

+1 -2
View File
@@ -12,8 +12,7 @@ def init_nvapi() -> None:
"""Initialize NvAPI. Must be called before any GPU operations."""
init_fn = query_interface(FUNC["Initialize"], nargs=0)
if not init_fn or init_fn() != 0:
print("NvAPI_Initialize failed")
sys.exit(1)
raise RuntimeError("NvAPI_Initialize failed")
def enumerate_gpus() -> tuple[ctypes.Array, int]:
+9 -1
View File
@@ -162,10 +162,18 @@ def run_autoload() -> None:
from ..hal.gpu import init_nvapi, discover_gpus
from ..hal.monitoring import init_nvml, shutdown_nvml
# Retry NvAPI init — the driver may not be fully ready at early boot.
nvapi_ready = False
for attempt in range(1, 17):
try:
init_nvapi()
nvapi_ready = True
break
except Exception as exc:
log.error("Failed to initialize NvAPI: %s", exc)
log.warning("NvAPI init attempt %d/16 failed: %s", attempt, exc)
time.sleep(2)
if not nvapi_ready:
log.error("Failed to initialize NvAPI after 16 attempts.")
sys.exit(1)
init_nvml() # best-effort