Files
nvcurve/nvcurve/daemon.py
T
ARIA 930e56bd07 Clean up LSP diagnostics across backend and frontend
Backend (nvcurve/):
- hal/fans.py, hal/limits.py, hal/gpu.py: replace conditional pynvml
  imports with the established 'pynvml: Any = _pynvml_import' pattern
  (fixes ~50 'possibly unbound' errors); type the result dicts; guard
  query_interface() results; explicit uuid/pci-bus parsing (int, hex
  convention documented); modernize Optional[T] -> T | None
- cli.py: fix 'curve_state' possibly-unbound and snap_path None handling
  in cmd_setup; wrap unchecked int()/open()/makedirs() calls in
  try/except with clean CLI errors; add module logger for silent
  except-pass blocks; raise ... from exc; fix unused loop vars and
  set-comprehension
- hal/snapshot.py: filepath: str | None; wrap all file ops; sorted
  imports; remove unused CT_POINTS import
- daemon.py: extract 0o666 to _SOCKET_MODE constant (intentional for
  /run sockets) with nosemgrep
- server.py: nosemgrep for Python 3.7-compat false positive (project
  requires >= 3.12); log previously-swallowed exception
- profiles/native.py, profiles/apply.py: wrap file ops and int(k)
  profile-key parsing; sorted imports; modernize typing

Frontend (frontend/src):
- Add .js extensions to all relative imports (standard TS-ESM; Vite
  resolves .js -> .ts)
- React.FormEvent (deprecated in React 19 types) -> React.SubmitEvent
- catch (e: any) -> catch (e: unknown) + instanceof Error narrowing
- React-hooks: move ref writes from render into effects; convert
  viewport reset to render-phase state adjustment; split
  selectPoint(index, multi) into selectPoint + togglePoint (no flag
  argument); remove non-null assertion
- Static inline styles -> Tailwind classes (dynamic positioning/cursor
  styles kept)
- Remove non-standard 'container' option from scrollIntoView (browsers
  ignore unknown options) which had orphaned a @ts-expect-error
- Object.fromEntries for Map -> Record conversion

Tooling:
- .gitignore: ignore .codegraph/ local tool data

Verified: tsc --noEmit, vite production build, python imports, and
full LSP scan (0 errors/warnings in both projects).
2026-09-08 23:57:30 +02:00

250 lines
7.9 KiB
Python

"""Minimal daemon: apply auto-load profiles on boot, manage web server via Unix socket.
Run via: nvcurve daemon
Or via systemd: nvcurve service install
Protocol: newline-delimited JSON, one request → one response, connection closed.
Commands:
{"cmd": "ping"}
{"cmd": "serve_start", "host": "127.0.0.1", "port": 8042}
{"cmd": "serve_stop"}
{"cmd": "serve_status"}
Requires root.
"""
import asyncio
import contextlib
import json
import logging
import os
import signal
import subprocess
import sys
from .config import Config
log = logging.getLogger("nvcurve.daemon")
SOCKET_PATH = "/run/nvcurve-daemon.sock"
_PERSISTENT_CONFIG_FILE = "/etc/nvcurve/config.json"
# Global server subprocess — only touched from the asyncio event loop.
_server_proc: subprocess.Popen | None = None
_cfg: Config | None = None # Config instance, set in run()
# ── Socket command handlers ────────────────────────────────────────────────────
async def _handle_serve_start(host: str, port: int) -> dict:
global _server_proc
if _server_proc is not None and _server_proc.poll() is None:
return {
"ok": False,
"error": "web server already running",
"pid": _server_proc.pid,
}
cmd = [
sys.executable,
"-m",
"nvcurve",
"serve",
"start",
"--host",
host,
"--port",
str(port),
"--direct",
]
log_path = "/var/log/nvcurve-server.log"
log.info("Starting web server on %s:%d (log: %s)", host, port, log_path)
try:
with open(log_path, "a") as lf:
_server_proc = subprocess.Popen(
cmd,
stdout=lf,
stderr=lf,
env={**os.environ, "PYTHONDONTWRITEBYTECODE": "1"},
)
except OSError as exc:
return {"ok": False, "error": f"cannot open log file {log_path}: {exc}"}
log.info("Web server started (PID %d)", _server_proc.pid)
return {"ok": True, "pid": _server_proc.pid}
async def _handle_serve_stop() -> dict:
global _server_proc
if _server_proc is None or _server_proc.poll() is not None:
_server_proc = None
return {"ok": False, "error": "web server is not running"}
log.info("Stopping web server (PID %d)…", _server_proc.pid)
_server_proc.terminate()
try:
await asyncio.get_running_loop().run_in_executor(None, _server_proc.wait, 5)
except Exception:
_server_proc.kill()
_server_proc = None
return {"ok": True}
async def _handle_serve_status() -> dict:
global _server_proc
if _server_proc is not None and _server_proc.poll() is None:
return {"ok": True, "running": True, "pid": _server_proc.pid}
_server_proc = None
return {"ok": True, "running": False}
async def _dispatch(req: dict) -> dict:
cmd = req.get("cmd")
if cmd == "ping":
return {"ok": True}
elif cmd == "serve_start":
if _cfg is None:
return {"ok": False, "error": "config not initialized"}
host = req.get("host", _cfg.host)
port = req.get("port", _cfg.port)
return await _handle_serve_start(host, port)
elif cmd == "serve_stop":
return await _handle_serve_stop()
elif cmd == "serve_status":
return await _handle_serve_status()
else:
return {"ok": False, "error": f"unknown command: {cmd!r}"}
async def _handle_client(
reader: asyncio.StreamReader, writer: asyncio.StreamWriter
) -> None:
resp: dict = {"ok": False, "error": "internal error"}
try:
data = await asyncio.wait_for(reader.readline(), timeout=5.0)
req = json.loads(data)
resp = await _dispatch(req)
except asyncio.TimeoutError:
resp = {"ok": False, "error": "timeout reading request"}
except json.JSONDecodeError as exc:
resp = {"ok": False, "error": f"invalid JSON: {exc}"}
except Exception as exc:
resp = {"ok": False, "error": str(exc)}
finally:
with contextlib.suppress(Exception):
writer.write(json.dumps(resp).encode() + b"\n")
await writer.drain()
writer.close()
with contextlib.suppress(Exception):
await writer.wait_closed()
# ── Entrypoint ─────────────────────────────────────────────────────────────────
def _load_persistent_config() -> dict:
try:
with open(_PERSISTENT_CONFIG_FILE) as f:
return json.load(f)
except Exception:
return {}
def run() -> None:
"""Run the nvcurve daemon: apply auto-load profiles, then serve the Unix socket."""
global _cfg
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s %(levelname)s %(name)s: %(message)s",
)
if os.geteuid() != 0:
print("nvcurve daemon: must run as root", file=sys.stderr)
sys.exit(1)
cfg_data = _load_persistent_config()
_cfg = Config()
for key in (
"max_delta_khz",
"auto_snapshot",
"max_snapshots",
"snapshot_dir",
"profile_dir",
"host",
"port",
):
if key in cfg_data:
setattr(_cfg, key, cfg_data[key])
# Apply auto-load profiles in a subprocess so the daemon process itself
# never loads NvAPI/NVML/HAL modules — keeps steady-state RSS low.
auto_load_profiles: dict = cfg_data.get("auto_load_profiles", {})
if auto_load_profiles:
log.info("Applying auto-load profiles…")
subprocess.run(
[sys.executable, "-m", "nvcurve", "autoload"],
check=False,
)
auto_serve: bool = cfg_data.get("auto_serve", False)
asyncio.run(_serve_socket(auto_serve=auto_serve))
log.info("Daemon stopped.")
async def _serve_socket(auto_serve: bool = False) -> None:
global _server_proc
# Clean up stale socket from a previous (unclean) run.
if os.path.exists(SOCKET_PATH):
try:
os.unlink(SOCKET_PATH)
except OSError as exc:
log.warning("Could not remove stale socket %s: %s", SOCKET_PATH, exc)
server = await asyncio.start_unix_server(_handle_client, path=SOCKET_PATH)
# The socket must be connectable by unprivileged users: the CLI runs as the
# regular user and talks to this root daemon over the socket. 0o666 is
# intentional (standard for /run daemon sockets).
# pi-lens-ignore: S103
_SOCKET_MODE = 0o666
os.chmod(
SOCKET_PATH, _SOCKET_MODE
) # nosemgrep: python.lang.security.audit.insecure-file-permissions.insecure-file-permissions
log.info("Daemon listening on %s", SOCKET_PATH)
if auto_serve:
if _cfg is None:
log.warning("auto_serve requested but config not initialized")
else:
log.info("auto_serve enabled — starting web server on boot")
await _handle_serve_start(_cfg.host, _cfg.port)
stop_event = asyncio.Event()
loop = asyncio.get_running_loop()
loop.add_signal_handler(signal.SIGTERM, stop_event.set)
loop.add_signal_handler(signal.SIGINT, stop_event.set)
async with server:
await stop_event.wait()
# Gracefully stop the web server if it's running.
if _server_proc is not None and _server_proc.poll() is None:
log.info("Stopping web server (PID %d)…", _server_proc.pid)
_server_proc.terminate()
try:
await asyncio.wait_for(
asyncio.get_running_loop().run_in_executor(None, _server_proc.wait),
timeout=5,
)
except asyncio.TimeoutError:
_server_proc.kill()
if os.path.exists(SOCKET_PATH):
try:
os.unlink(SOCKET_PATH)
except OSError as exc:
log.warning("Could not remove socket %s on shutdown: %s", SOCKET_PATH, exc)