Backend (nvcurve/): - hal/fans.py, hal/limits.py, hal/gpu.py: replace conditional pynvml imports with the established 'pynvml: Any = _pynvml_import' pattern (fixes ~50 'possibly unbound' errors); type the result dicts; guard query_interface() results; explicit uuid/pci-bus parsing (int, hex convention documented); modernize Optional[T] -> T | None - cli.py: fix 'curve_state' possibly-unbound and snap_path None handling in cmd_setup; wrap unchecked int()/open()/makedirs() calls in try/except with clean CLI errors; add module logger for silent except-pass blocks; raise ... from exc; fix unused loop vars and set-comprehension - hal/snapshot.py: filepath: str | None; wrap all file ops; sorted imports; remove unused CT_POINTS import - daemon.py: extract 0o666 to _SOCKET_MODE constant (intentional for /run sockets) with nosemgrep - server.py: nosemgrep for Python 3.7-compat false positive (project requires >= 3.12); log previously-swallowed exception - profiles/native.py, profiles/apply.py: wrap file ops and int(k) profile-key parsing; sorted imports; modernize typing Frontend (frontend/src): - Add .js extensions to all relative imports (standard TS-ESM; Vite resolves .js -> .ts) - React.FormEvent (deprecated in React 19 types) -> React.SubmitEvent - catch (e: any) -> catch (e: unknown) + instanceof Error narrowing - React-hooks: move ref writes from render into effects; convert viewport reset to render-phase state adjustment; split selectPoint(index, multi) into selectPoint + togglePoint (no flag argument); remove non-null assertion - Static inline styles -> Tailwind classes (dynamic positioning/cursor styles kept) - Remove non-standard 'container' option from scrollIntoView (browsers ignore unknown options) which had orphaned a @ts-expect-error - Object.fromEntries for Map -> Record conversion Tooling: - .gitignore: ignore .codegraph/ local tool data Verified: tsc --noEmit, vite production build, python imports, and full LSP scan (0 errors/warnings in both projects).
250 lines
7.9 KiB
Python
250 lines
7.9 KiB
Python
"""Minimal daemon: apply auto-load profiles on boot, manage web server via Unix socket.
|
|
|
|
Run via: nvcurve daemon
|
|
Or via systemd: nvcurve service install
|
|
|
|
Protocol: newline-delimited JSON, one request → one response, connection closed.
|
|
|
|
Commands:
|
|
{"cmd": "ping"}
|
|
{"cmd": "serve_start", "host": "127.0.0.1", "port": 8042}
|
|
{"cmd": "serve_stop"}
|
|
{"cmd": "serve_status"}
|
|
|
|
Requires root.
|
|
"""
|
|
|
|
import asyncio
|
|
import contextlib
|
|
import json
|
|
import logging
|
|
import os
|
|
import signal
|
|
import subprocess
|
|
import sys
|
|
|
|
from .config import Config
|
|
|
|
log = logging.getLogger("nvcurve.daemon")
|
|
|
|
SOCKET_PATH = "/run/nvcurve-daemon.sock"
|
|
_PERSISTENT_CONFIG_FILE = "/etc/nvcurve/config.json"
|
|
|
|
# Global server subprocess — only touched from the asyncio event loop.
|
|
_server_proc: subprocess.Popen | None = None
|
|
_cfg: Config | None = None # Config instance, set in run()
|
|
|
|
|
|
# ── Socket command handlers ────────────────────────────────────────────────────
|
|
|
|
|
|
async def _handle_serve_start(host: str, port: int) -> dict:
|
|
global _server_proc
|
|
if _server_proc is not None and _server_proc.poll() is None:
|
|
return {
|
|
"ok": False,
|
|
"error": "web server already running",
|
|
"pid": _server_proc.pid,
|
|
}
|
|
|
|
cmd = [
|
|
sys.executable,
|
|
"-m",
|
|
"nvcurve",
|
|
"serve",
|
|
"start",
|
|
"--host",
|
|
host,
|
|
"--port",
|
|
str(port),
|
|
"--direct",
|
|
]
|
|
log_path = "/var/log/nvcurve-server.log"
|
|
log.info("Starting web server on %s:%d (log: %s)", host, port, log_path)
|
|
try:
|
|
with open(log_path, "a") as lf:
|
|
_server_proc = subprocess.Popen(
|
|
cmd,
|
|
stdout=lf,
|
|
stderr=lf,
|
|
env={**os.environ, "PYTHONDONTWRITEBYTECODE": "1"},
|
|
)
|
|
except OSError as exc:
|
|
return {"ok": False, "error": f"cannot open log file {log_path}: {exc}"}
|
|
log.info("Web server started (PID %d)", _server_proc.pid)
|
|
return {"ok": True, "pid": _server_proc.pid}
|
|
|
|
|
|
async def _handle_serve_stop() -> dict:
|
|
global _server_proc
|
|
if _server_proc is None or _server_proc.poll() is not None:
|
|
_server_proc = None
|
|
return {"ok": False, "error": "web server is not running"}
|
|
log.info("Stopping web server (PID %d)…", _server_proc.pid)
|
|
_server_proc.terminate()
|
|
try:
|
|
await asyncio.get_running_loop().run_in_executor(None, _server_proc.wait, 5)
|
|
except Exception:
|
|
_server_proc.kill()
|
|
_server_proc = None
|
|
return {"ok": True}
|
|
|
|
|
|
async def _handle_serve_status() -> dict:
|
|
global _server_proc
|
|
if _server_proc is not None and _server_proc.poll() is None:
|
|
return {"ok": True, "running": True, "pid": _server_proc.pid}
|
|
_server_proc = None
|
|
return {"ok": True, "running": False}
|
|
|
|
|
|
async def _dispatch(req: dict) -> dict:
|
|
cmd = req.get("cmd")
|
|
if cmd == "ping":
|
|
return {"ok": True}
|
|
elif cmd == "serve_start":
|
|
if _cfg is None:
|
|
return {"ok": False, "error": "config not initialized"}
|
|
host = req.get("host", _cfg.host)
|
|
port = req.get("port", _cfg.port)
|
|
return await _handle_serve_start(host, port)
|
|
elif cmd == "serve_stop":
|
|
return await _handle_serve_stop()
|
|
elif cmd == "serve_status":
|
|
return await _handle_serve_status()
|
|
else:
|
|
return {"ok": False, "error": f"unknown command: {cmd!r}"}
|
|
|
|
|
|
async def _handle_client(
|
|
reader: asyncio.StreamReader, writer: asyncio.StreamWriter
|
|
) -> None:
|
|
resp: dict = {"ok": False, "error": "internal error"}
|
|
try:
|
|
data = await asyncio.wait_for(reader.readline(), timeout=5.0)
|
|
req = json.loads(data)
|
|
resp = await _dispatch(req)
|
|
except asyncio.TimeoutError:
|
|
resp = {"ok": False, "error": "timeout reading request"}
|
|
except json.JSONDecodeError as exc:
|
|
resp = {"ok": False, "error": f"invalid JSON: {exc}"}
|
|
except Exception as exc:
|
|
resp = {"ok": False, "error": str(exc)}
|
|
finally:
|
|
with contextlib.suppress(Exception):
|
|
writer.write(json.dumps(resp).encode() + b"\n")
|
|
await writer.drain()
|
|
writer.close()
|
|
with contextlib.suppress(Exception):
|
|
await writer.wait_closed()
|
|
|
|
|
|
# ── Entrypoint ─────────────────────────────────────────────────────────────────
|
|
|
|
|
|
def _load_persistent_config() -> dict:
|
|
try:
|
|
with open(_PERSISTENT_CONFIG_FILE) as f:
|
|
return json.load(f)
|
|
except Exception:
|
|
return {}
|
|
|
|
|
|
def run() -> None:
|
|
"""Run the nvcurve daemon: apply auto-load profiles, then serve the Unix socket."""
|
|
global _cfg
|
|
|
|
logging.basicConfig(
|
|
level=logging.INFO,
|
|
format="%(asctime)s %(levelname)s %(name)s: %(message)s",
|
|
)
|
|
|
|
if os.geteuid() != 0:
|
|
print("nvcurve daemon: must run as root", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
cfg_data = _load_persistent_config()
|
|
|
|
_cfg = Config()
|
|
for key in (
|
|
"max_delta_khz",
|
|
"auto_snapshot",
|
|
"max_snapshots",
|
|
"snapshot_dir",
|
|
"profile_dir",
|
|
"host",
|
|
"port",
|
|
):
|
|
if key in cfg_data:
|
|
setattr(_cfg, key, cfg_data[key])
|
|
|
|
# Apply auto-load profiles in a subprocess so the daemon process itself
|
|
# never loads NvAPI/NVML/HAL modules — keeps steady-state RSS low.
|
|
auto_load_profiles: dict = cfg_data.get("auto_load_profiles", {})
|
|
if auto_load_profiles:
|
|
log.info("Applying auto-load profiles…")
|
|
subprocess.run(
|
|
[sys.executable, "-m", "nvcurve", "autoload"],
|
|
check=False,
|
|
)
|
|
|
|
auto_serve: bool = cfg_data.get("auto_serve", False)
|
|
asyncio.run(_serve_socket(auto_serve=auto_serve))
|
|
|
|
log.info("Daemon stopped.")
|
|
|
|
|
|
async def _serve_socket(auto_serve: bool = False) -> None:
|
|
global _server_proc
|
|
|
|
# Clean up stale socket from a previous (unclean) run.
|
|
if os.path.exists(SOCKET_PATH):
|
|
try:
|
|
os.unlink(SOCKET_PATH)
|
|
except OSError as exc:
|
|
log.warning("Could not remove stale socket %s: %s", SOCKET_PATH, exc)
|
|
|
|
server = await asyncio.start_unix_server(_handle_client, path=SOCKET_PATH)
|
|
# The socket must be connectable by unprivileged users: the CLI runs as the
|
|
# regular user and talks to this root daemon over the socket. 0o666 is
|
|
# intentional (standard for /run daemon sockets).
|
|
# pi-lens-ignore: S103
|
|
_SOCKET_MODE = 0o666
|
|
os.chmod(
|
|
SOCKET_PATH, _SOCKET_MODE
|
|
) # nosemgrep: python.lang.security.audit.insecure-file-permissions.insecure-file-permissions
|
|
log.info("Daemon listening on %s", SOCKET_PATH)
|
|
|
|
if auto_serve:
|
|
if _cfg is None:
|
|
log.warning("auto_serve requested but config not initialized")
|
|
else:
|
|
log.info("auto_serve enabled — starting web server on boot")
|
|
await _handle_serve_start(_cfg.host, _cfg.port)
|
|
|
|
stop_event = asyncio.Event()
|
|
loop = asyncio.get_running_loop()
|
|
loop.add_signal_handler(signal.SIGTERM, stop_event.set)
|
|
loop.add_signal_handler(signal.SIGINT, stop_event.set)
|
|
|
|
async with server:
|
|
await stop_event.wait()
|
|
|
|
# Gracefully stop the web server if it's running.
|
|
if _server_proc is not None and _server_proc.poll() is None:
|
|
log.info("Stopping web server (PID %d)…", _server_proc.pid)
|
|
_server_proc.terminate()
|
|
try:
|
|
await asyncio.wait_for(
|
|
asyncio.get_running_loop().run_in_executor(None, _server_proc.wait),
|
|
timeout=5,
|
|
)
|
|
except asyncio.TimeoutError:
|
|
_server_proc.kill()
|
|
|
|
if os.path.exists(SOCKET_PATH):
|
|
try:
|
|
os.unlink(SOCKET_PATH)
|
|
except OSError as exc:
|
|
log.warning("Could not remove socket %s on shutdown: %s", SOCKET_PATH, exc)
|