Merge pull request 'fix: retry NvAPI init on autoload to handle early boot race' (#3) from fix/autoload-nvapi-retry into main
Reviewed-on: #3
This commit was merged in pull request #3.
This commit is contained in:
commit
af23a10f25
2 files changed
+10
-3
No files matched your search
+1
-2
@@ -12,8 +12,7 @@ def init_nvapi() -> None:
|
|||||||
"""Initialize NvAPI. Must be called before any GPU operations."""
|
"""Initialize NvAPI. Must be called before any GPU operations."""
|
||||||
init_fn = query_interface(FUNC["Initialize"], nargs=0)
|
init_fn = query_interface(FUNC["Initialize"], nargs=0)
|
||||||
if not init_fn or init_fn() != 0:
|
if not init_fn or init_fn() != 0:
|
||||||
print("NvAPI_Initialize failed")
|
raise RuntimeError("NvAPI_Initialize failed")
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
|
|
||||||
def enumerate_gpus() -> tuple[ctypes.Array, int]:
|
def enumerate_gpus() -> tuple[ctypes.Array, int]:
|
||||||
|
|||||||
@@ -162,10 +162,18 @@ def run_autoload() -> None:
|
|||||||
from ..hal.gpu import init_nvapi, discover_gpus
|
from ..hal.gpu import init_nvapi, discover_gpus
|
||||||
from ..hal.monitoring import init_nvml, shutdown_nvml
|
from ..hal.monitoring import init_nvml, shutdown_nvml
|
||||||
|
|
||||||
|
# Retry NvAPI init — the driver may not be fully ready at early boot.
|
||||||
|
nvapi_ready = False
|
||||||
|
for attempt in range(1, 17):
|
||||||
try:
|
try:
|
||||||
init_nvapi()
|
init_nvapi()
|
||||||
|
nvapi_ready = True
|
||||||
|
break
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
log.error("Failed to initialize NvAPI: %s", exc)
|
log.warning("NvAPI init attempt %d/16 failed: %s", attempt, exc)
|
||||||
|
time.sleep(2)
|
||||||
|
if not nvapi_ready:
|
||||||
|
log.error("Failed to initialize NvAPI after 16 attempts.")
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
|
|
||||||
init_nvml() # best-effort
|
init_nvml() # best-effort
|
||||||
|
|||||||
Reference in new issue
Block a user