diff --git a/Makefile b/Makefile
index 43b5e7e..eca3bef 100644
--- a/Makefile
+++ b/Makefile
@@ -24,6 +24,7 @@ dev: ## Create/refresh the dev environment (uv sync)
test: ## Run the test suite
$(UV) run python tests/test_security.py
+ $(UV) run python tests/test_rm_power.py
clean: ## Remove build artifacts
rm -rf frontend/dist frontend/node_modules
\ No newline at end of file
diff --git a/docs/Installation.md b/docs/Installation.md
index d765481..567af52 100644
--- a/docs/Installation.md
+++ b/docs/Installation.md
@@ -74,6 +74,7 @@ nvcurve setup
```
This performs four checks:
+
1. **NvAPI function probe** — verifies all required functions resolve in your driver
2. **Curve read** — reads and displays your current V/F curve as a baseline
3. **Write-verify** — writes `+5 MHz` to a safe point, reads it back, and confirms the change
diff --git a/docs/Usage-Guide.md b/docs/Usage-Guide.md
index 8d4f998..0af17c1 100644
--- a/docs/Usage-Guide.md
+++ b/docs/Usage-Guide.md
@@ -84,6 +84,23 @@ The monitoring panel shows real-time GPU metrics:
Data is streamed via WebSocket from the backend at a configurable poll interval (default: 1 second).
+### Performance Limits
+
+The Performance panel controls the board power limit and the memory clock offset. The power limit slider is bounded by the GPU's VBIOS minimum and maximum (shown at the slider ends); changes are applied on **Apply** and reset to the hardware default on **Reset**.
+
+#### Experimental NVIDIA power control
+
+On compatible drivers, an **Experimental NVIDIA power control** checkbox appears in the Performance panel. Enabling it switches power-limit application from the standard NVML call to an undocumented driver (RM) interface, which **permits caps below the VBIOS minimum, down to 30 W**. The native maximum still applies.
+
+> **Warning.** This uses an undocumented driver interface for *all* power limits, including resets. It may cause instability or stop working after a driver update. Enable it at your own risk. The option is clearly labelled with a red warning in the UI, and profiles saved while it is enabled are marked accordingly.
+
+Notes:
+
+- The checkbox only appears when the driver exposes a compatible RM power layout (detected with a read-only probe — no writes).
+- In this mode there is **no automatic fallback** to NVML: if the RM route fails, the error is reported rather than silently switching backends.
+- The mode is per-GPU and persisted across server restarts. Reset restores the default through the same route, so it can also clear a previously-set below-minimum cap.
+- The CLI reports availability via `nvcurve read --diag` ("Experimental RM power: available").
+
### Multi-GPU
When multiple NVIDIA GPUs are detected, a GPU selector dropdown appears in the status bar. Switching GPUs resets pending edits, selection state, and monitoring for the new target.
diff --git a/frontend/src/components/CurveEditor/CurveTooltip.tsx b/frontend/src/components/CurveEditor/CurveTooltip.tsx
index e46ec58..520f823 100644
--- a/frontend/src/components/CurveEditor/CurveTooltip.tsx
+++ b/frontend/src/components/CurveEditor/CurveTooltip.tsx
@@ -1,12 +1,12 @@
-import { fmt } from '../../utils/units.js';
-import type { VFPoint } from '../../types.js';
+import { fmt } from "../../utils/units.js";
+import type { VFPoint } from "../../types.js";
interface Props {
- point: VFPoint;
- /** Pending delta in kHz (from store), if any */
- pendingDeltaKhz?: number;
- /** True if this point is being held up by monotonicity enforcement */
- isClamped?: boolean;
+ point: VFPoint;
+ /** Pending delta in kHz (from store), if any */
+ pendingDeltaKhz?: number;
+ /** True if this point is being held up by monotonicity enforcement */
+ isClamped?: boolean;
}
/**
@@ -14,49 +14,70 @@ interface Props {
* SVG so it's never clipped by the SVG viewport.
*/
export function CurveTooltip({ point, pendingDeltaKhz, isClamped }: Props) {
- const hasPending = pendingDeltaKhz !== undefined;
- const pendingMhz = hasPending ? pendingDeltaKhz! / 1000 : 0;
- const deltaChange = hasPending ? pendingDeltaKhz! - point.delta_khz : 0;
- const pendingEffMhz = hasPending ? point.freq_mhz + deltaChange / 1000 : null;
+ const hasPending = pendingDeltaKhz !== undefined;
+ const pendingMhz = hasPending ? pendingDeltaKhz! / 1000 : 0;
+ const deltaChange = hasPending ? pendingDeltaKhz! - point.delta_khz : 0;
+ const pendingEffMhz = hasPending
+ ? point.freq_mhz + deltaChange / 1000
+ : null;
- return (
-
-
Point {point.index}
-
- Volt: {fmt.mv(point.volt_mv, 1)}
-
-
- Offset:
- 0 ? 'text-emerald-400' : point.delta_khz < 0 ? 'text-red-400' : 'text-zinc-400'}>
- {point.delta_khz > 0 ? '+' : ''}{fmt.mhz(point.delta_mhz, 1)}
-
-
-
- Eff.: {fmt.mhz(point.freq_mhz, 0)}
- {isClamped && ⇡ }
-
- {isClamped && (
-
- Clamped by lower-voltage point
-
- )}
- {hasPending && (
- <>
-
+ return (
+
+
Point {point.index}
- Pending:
- 0 ? 'text-cyan-400' : pendingMhz < 0 ? 'text-orange-400' : 'text-zinc-400'}>
- {pendingMhz > 0 ? '+' : ''}{pendingMhz.toFixed(1)} MHz
-
+ Volt:
+ {fmt.mv(point.volt_mv, 1)}
-
-
→ Eff.: {fmt.mhz(pendingEffMhz, 0)}
+
+ Offset:
+ 0
+ ? "text-emerald-400"
+ : point.delta_khz < 0
+ ? "text-red-400"
+ : "text-zinc-400"
+ }
+ >
+ {point.delta_khz > 0 ? "+" : ""}
+ {fmt.mhz(point.delta_mhz, 1)}
+
-
- >
- )}
-
- );
+
+ Eff.:
+ {fmt.mhz(point.freq_mhz, 0)}
+ {isClamped && ⇡ }
+
+ {isClamped && (
+
+ Clamped by lower-voltage point
+
+ )}
+ {hasPending && (
+ <>
+
+
+ Pending:
+ 0
+ ? "text-cyan-400"
+ : pendingMhz < 0
+ ? "text-orange-400"
+ : "text-zinc-400"
+ }
+ >
+ {pendingMhz > 0 ? "+" : ""}
+ {pendingMhz.toFixed(1)} MHz
+
+
+
+ → Eff.:
+ {fmt.mhz(pendingEffMhz, 0)}
+
+
+ >
+ )}
+
+ );
}
diff --git a/frontend/src/components/Limits/PerformancePanel.tsx b/frontend/src/components/Limits/PerformancePanel.tsx
index 0628f22..82b69d9 100644
--- a/frontend/src/components/Limits/PerformancePanel.tsx
+++ b/frontend/src/components/Limits/PerformancePanel.tsx
@@ -72,6 +72,21 @@ export function PerformancePanel() {
}
}
+ async function handleModeChange(enabled: boolean) {
+ setBusy(true);
+ try {
+ await api.updateLimits(
+ { power_cap_mode: enabled ? "ioctl" : "nvml" },
+ selectedGpuIndex,
+ );
+ await fetchLimits();
+ } catch (e: unknown) {
+ toast.error(e instanceof Error ? e.message : String(e));
+ } finally {
+ setBusy(false);
+ }
+ }
+
if (loading && !limits) {
return (
@@ -158,12 +173,80 @@ export function PerformancePanel() {
)}
+ {/* ── Experimental NVIDIA power control ─────────────────────────── */}
+ {(limits.rm_power_supported || limits.power_cap_mode === "ioctl") && (
+
+
+ handleModeChange(e.target.checked)}
+ className="accent-red-500"
+ />
+
+ Experimental NVIDIA power control
+
+
+ {limits.rm_power_supported ? (
+
+
⚠ WARNING: uses an
+ undocumented driver interface for ALL power limits, including
+ resets. Allows values below the VBIOS minimum (down to 30 W)
+ and may cause instability or stop working after driver
+ updates. Enable at your own risk. Based on the work of{" "}
+
+ panchovix
+ {" "}
+ (LACT PR #1205).
+
+ ) : (
+
+
+ ⚠ Interface not currently available.
+
+ The driver no longer exposes the RM power interface (it may
+ have been updated). Experimental mode is still enabled, so
+ power-limit changes will fail. Uncheck to switch back to the
+ standard NVML mode.
+
+ )}
+
+ )}
+
{/* ── Board Power Limit ─────────────────────────────────────────── */}
-
- Board Power Limit
-
+
+
+ Board Power Limit
+
+ {limits.power_cap_mode === "ioctl" &&
+ limits.min_power_limit_w_native != null && (
+
+ VBIOS min {limits.min_power_limit_w_native} W
+
+ )}
+
{badges}
)}
+ {p.power_cap_mode === "ioctl" && (
+
+ ⚠ experimental power (below VBIOS min)
+
+ )}
diff --git a/frontend/src/types.ts b/frontend/src/types.ts
index 60b4252..2e8b8bd 100644
--- a/frontend/src/types.ts
+++ b/frontend/src/types.ts
@@ -98,7 +98,11 @@ export interface LimitsState {
power_limit_w: number | null;
default_power_limit_w: number | null;
min_power_limit_w: number | null;
+ min_power_limit_w_native: number | null;
max_power_limit_w: number | null;
+ // "nvml" (default) or "ioctl" (experimental RM power control)
+ power_cap_mode: "nvml" | "ioctl";
+ rm_power_supported: boolean;
// Clock offsets — current values
gpc_offset_mhz: number | null;
mem_offset_mhz: number | null;
@@ -136,6 +140,8 @@ export interface ProfileData {
curve_deltas: Record
;
mem_offset_mhz: number | null;
power_limit_w: number | null;
+ // "nvml" (default) or "ioctl" (experimental RM power control)
+ power_cap_mode: "nvml" | "ioctl" | null;
fan_curve: FanPoint[] | null;
fan_targets: number[] | null;
}
diff --git a/frontend/src/utils/curveHelpers.ts b/frontend/src/utils/curveHelpers.ts
index 817992d..a745306 100644
--- a/frontend/src/utils/curveHelpers.ts
+++ b/frontend/src/utils/curveHelpers.ts
@@ -1,4 +1,4 @@
-import type { VFPoint } from '../types.js';
+import type { VFPoint } from "../types.js";
/**
* Approximate reference frequency (MHz) for a point: effective − delta.
@@ -14,35 +14,37 @@ import type { VFPoint } from '../types.js';
* useful as a faint reference line ("where this point would sit with no boost").
*/
export function refBaseMhz(p: VFPoint): number {
- return p.freq_mhz - p.delta_mhz;
+ return p.freq_mhz - p.delta_mhz;
}
/** Find which VF point the GPU is currently near based on voltage reading */
export function findCurrentPoint(
- points: VFPoint[],
- voltage_mv: number | null,
+ points: VFPoint[],
+ voltage_mv: number | null,
): VFPoint | null {
- if (voltage_mv == null || points.length === 0) return null;
- return points.reduce((best, p) =>
- Math.abs(p.volt_mv - voltage_mv) < Math.abs(best.volt_mv - voltage_mv) ? p : best,
- );
+ if (voltage_mv == null || points.length === 0) return null;
+ return points.reduce((best, p) =>
+ Math.abs(p.volt_mv - voltage_mv) < Math.abs(best.volt_mv - voltage_mv)
+ ? p
+ : best,
+ );
}
/** Voltage domain extent, with padding */
export function voltExtent(points: VFPoint[], padMv = 20): [number, number] {
- if (points.length === 0) return [600, 1100];
- const min = Math.min(...points.map((p) => p.volt_mv));
- const max = Math.max(...points.map((p) => p.volt_mv));
- return [min - padMv, max + padMv];
+ if (points.length === 0) return [600, 1100];
+ const min = Math.min(...points.map((p) => p.volt_mv));
+ const max = Math.max(...points.map((p) => p.volt_mv));
+ return [min - padMv, max + padMv];
}
/** Frequency domain extent for the effective (boosted) curve, with padding */
export function freqExtent(points: VFPoint[], padMhz = 50): [number, number] {
- if (points.length === 0) return [1000, 3000];
- const allFreqs = points.flatMap((p) => [p.freq_mhz, refBaseMhz(p)]);
- const min = Math.min(...allFreqs);
- const max = Math.max(...points.map((p) => p.freq_mhz));
- return [min - padMhz, max + padMhz];
+ if (points.length === 0) return [1000, 3000];
+ const allFreqs = points.flatMap((p) => [p.freq_mhz, refBaseMhz(p)]);
+ const min = Math.min(...allFreqs);
+ const max = Math.max(...points.map((p) => p.freq_mhz));
+ return [min - padMhz, max + padMhz];
}
/**
@@ -59,19 +61,19 @@ export function freqExtent(points: VFPoint[], padMhz = 50): [number, number] {
* point 104 has +315 MHz → effective also 3907 MHz (clamped).
*/
export function detectClampedPoints(points: VFPoint[]): Set {
- const clamped = new Set();
- let ceiling = -Infinity;
- let ceilingOffset = -Infinity;
+ const clamped = new Set();
+ let ceiling = -Infinity;
+ let ceilingOffset = -Infinity;
- for (const p of points) {
- if (p.freq_mhz <= ceiling && p.delta_khz < ceilingOffset) {
- clamped.add(p.index);
- }
- if (p.freq_mhz > ceiling) {
- ceiling = p.freq_mhz;
- ceilingOffset = p.delta_khz;
- }
+ for (const p of points) {
+ if (p.freq_mhz <= ceiling && p.delta_khz < ceilingOffset) {
+ clamped.add(p.index);
}
+ if (p.freq_mhz > ceiling) {
+ ceiling = p.freq_mhz;
+ ceilingOffset = p.delta_khz;
+ }
+ }
- return clamped;
+ return clamped;
}
diff --git a/hatch_build.py b/hatch_build.py
index 3f3fbd7..52e3c4f 100644
--- a/hatch_build.py
+++ b/hatch_build.py
@@ -20,7 +20,13 @@ import sys
from hatchling.builders.hooks.plugin.interface import BuildHookInterface
# Frontend inputs that must be newer than dist/index.html to trigger a rebuild.
-_FRONTEND_INPUTS = ("src", "index.html", "vite.config.ts", "package.json", "tsconfig.json")
+_FRONTEND_INPUTS = (
+ "src",
+ "index.html",
+ "vite.config.ts",
+ "package.json",
+ "tsconfig.json",
+)
class FrontendBuildHook(BuildHookInterface):
diff --git a/install.sh b/install.sh
index 65e40bd..480b8fc 100755
--- a/install.sh
+++ b/install.sh
@@ -25,19 +25,19 @@ fail() {
}
# --- prerequisites -----------------------------------------------------------
-command -v git >/dev/null 2>&1 \
- || fail "git is required. Install it first."
-command -v node >/dev/null 2>&1 \
- || fail "Node.js 18+ is required (e.g. 'sudo pacman -S nodejs npm' or 'sudo apt install nodejs npm')."
-command -v npm >/dev/null 2>&1 \
- || fail "npm is required (usually installed together with Node.js)."
+command -v git >/dev/null 2>&1 ||
+ fail "git is required. Install it first."
+command -v node >/dev/null 2>&1 ||
+ fail "Node.js 18+ is required (e.g. 'sudo pacman -S nodejs npm' or 'sudo apt install nodejs npm')."
+command -v npm >/dev/null 2>&1 ||
+ fail "npm is required (usually installed together with Node.js)."
if ! command -v uv >/dev/null 2>&1; then
echo "uv not found — installing it from https://astral.sh/uv ..."
curl -LsSf https://astral.sh/uv/install.sh | sh
export PATH="$HOME/.local/bin:$PATH"
- command -v uv >/dev/null 2>&1 \
- || fail "uv installation failed. Install uv manually: https://docs.astral.sh/uv/"
+ command -v uv >/dev/null 2>&1 ||
+ fail "uv installation failed. Install uv manually: https://docs.astral.sh/uv/"
fi
# --- locate the source tree ---------------------------------------------------
@@ -59,4 +59,4 @@ uv tool install --force "$src"
echo
echo "NVCurve installed."
echo " Verify your GPU: nvcurve setup"
-echo " Start the web UI: nvcurve"
\ No newline at end of file
+echo " Start the web UI: nvcurve"
diff --git a/nvcurve/cli.py b/nvcurve/cli.py
index 35b2aa7..a40c371 100644
--- a/nvcurve/cli.py
+++ b/nvcurve/cli.py
@@ -405,6 +405,11 @@ def run_diagnostics(gpu, gpu_name, gpu_index: int = 0):
print(f" Default: {fmt_w(def_w)}")
if min_w is not None and max_w is not None:
print(f" Range: {min_w} – {max_w} W")
+ if pwr.get("rm_power_supported"):
+ print(
+ " Experimental RM power: available (opt-in via web UI or profile;"
+ " extends range to 30 W)"
+ )
# ── Privilege / browser helpers ───────────────────────────────────────────────
@@ -1207,12 +1212,33 @@ def cmd_profile(args):
power_limit_w = None
mem_offset_mhz = None
+ # Capture the GPU's power-cap mode: prefer the running server (most
+ # current), else fall back to the persisted per-GPU mode from config
+ # (so a profile saved while the server is down or auth is enabled
+ # still records the GPU's actual mode rather than assuming nvml).
+ power_cap_mode = "nvml"
+ try:
+ from .client import NvCurveClient
+
+ base = getattr(args, "server", None) or _discover_server_url(default_config)
+ limits = NvCurveClient(base=base, gpu_index=gpu_index).limits()
+ if limits.get("power_cap_mode") in ("nvml", "ioctl"):
+ power_cap_mode = limits["power_cap_mode"]
+ except Exception as exc:
+ log.debug("Could not read power-cap mode from server: %s", exc)
+ gpu_key = _gpu_stable_key_offline(gpu_index)
+ if gpu_key is not None:
+ persisted = default_config.power_cap_modes.get(gpu_key)
+ if persisted in ("nvml", "ioctl"):
+ power_cap_mode = persisted
+
data = ProfileData(
name=args.name,
gpu_name=gpu_name,
curve_deltas=curve_deltas,
mem_offset_mhz=mem_offset_mhz,
power_limit_w=power_limit_w,
+ power_cap_mode=power_cap_mode,
)
filepath = save_profile(default_config.profile_dir, data)
print(f"Saved profile '{args.name}' to {filepath}")
@@ -1249,7 +1275,8 @@ def cmd_profile(args):
errs.append(f"Mem offset: {msg}")
if profile.power_limit_w is not None:
- ok, msg = set_power_limit(profile.power_limit_w, gpu_index)
+ mode = profile.power_cap_mode or "nvml"
+ ok, msg = set_power_limit(profile.power_limit_w, gpu_index, mode)
if not ok:
errs.append(f"Power limit: {msg}")
@@ -2296,6 +2323,14 @@ def main():
if "fan_curves" in data:
# Per-GPU active fan curves, restored on server startup.
cfg.fan_curves = dict(data["fan_curves"])
+ if "power_cap_modes" in data:
+ # Per-GPU experimental power-cap mode. "nvml" is the default
+ # (the server treats it as unset); keep only valid values.
+ cfg.power_cap_modes = {
+ str(k): str(v)
+ for k, v in dict(data["power_cap_modes"]).items()
+ if str(v) in ("nvml", "ioctl")
+ }
except Exception as exc:
log.debug("Could not load user config: %s", exc)
diff --git a/nvcurve/client.py b/nvcurve/client.py
index 1a3a3cf..4abc789 100644
--- a/nvcurve/client.py
+++ b/nvcurve/client.py
@@ -145,6 +145,9 @@ class NvCurveClient:
def snapshots(self) -> list:
return self._get("/api/snapshots")
+ def limits(self) -> dict:
+ return self._get("/api/limits")
+
# ── Profiles ─────────────────────────────────────────────────────────────
def profiles(self) -> dict:
diff --git a/nvcurve/config.py b/nvcurve/config.py
index 733afbf..0f7d32c 100644
--- a/nvcurve/config.py
+++ b/nvcurve/config.py
@@ -54,6 +54,11 @@ class Config:
# Legacy entries (bare curve list) are migrated at load time.
fan_curves: dict[str, object] = field(default_factory=dict)
+ # Per-GPU power-cap mode: "nvml" (default, never stored) or "ioctl"
+ # (experimental RM power control — permits caps below the VBIOS minimum).
+ # Key = stable GPU identifier (same as auto_load_profiles).
+ power_cap_modes: dict[str, str] = field(default_factory=dict)
+
# Module-level default config instance.
default_config = Config()
diff --git a/nvcurve/hal/limits.py b/nvcurve/hal/limits.py
index f681d95..24c90f9 100644
--- a/nvcurve/hal/limits.py
+++ b/nvcurve/hal/limits.py
@@ -59,20 +59,37 @@ def _get_handle(gpu_index: int):
# ── Power limit ───────────────────────────────────────────────────────────────
-def get_power_limit(gpu_index: int = 0) -> dict:
- """Return dict with power_limit_w, default_power_limit_w, min_power_limit_w, max_power_limit_w."""
- out: dict[str, int | None] = {
+def get_power_limit(gpu_index: int = 0, mode: str = "nvml") -> dict:
+ """Return dict with power limit info.
+
+ Keys: power_limit_w, default_power_limit_w, min_power_limit_w,
+ min_power_limit_w_native, max_power_limit_w, rm_power_supported,
+ power_cap_mode.
+
+ mode: "nvml" (default) or "ioctl" (experimental RM power control).
+ min_power_limit_w is the effective minimum: in ioctl mode it is
+ extended to the experimental floor (30 W) when the RM interface is
+ present and validated; min_power_limit_w_native is always the VBIOS
+ minimum. The RM probe is GET-only (no writes) and safe to run on
+ every call.
+ """
+ out: dict[str, int | bool | str | None] = {
"power_limit_w": None,
"default_power_limit_w": None,
"min_power_limit_w": None,
+ "min_power_limit_w_native": None,
"max_power_limit_w": None,
+ "rm_power_supported": False,
+ "power_cap_mode": mode,
}
try:
handle = _get_handle(gpu_index)
limit = pynvml.nvmlDeviceGetPowerManagementLimit(handle)
constrs = pynvml.nvmlDeviceGetPowerManagementLimitConstraints(handle)
out["power_limit_w"] = limit // 1000
- out["min_power_limit_w"] = constrs[0] // 1000
+ native_min = constrs[0] // 1000
+ out["min_power_limit_w"] = native_min
+ out["min_power_limit_w_native"] = native_min
out["max_power_limit_w"] = constrs[1] // 1000
try:
default = pynvml.nvmlDeviceGetPowerManagementDefaultLimit(handle)
@@ -81,11 +98,39 @@ def get_power_limit(gpu_index: int = 0) -> dict:
log.debug("nvmlDeviceGetPowerManagementDefaultLimit: %s", exc)
except Exception as exc:
log.warning("get_power_limit: %s", exc)
+ return out
+
+ # GET-only RM discovery — reported so the UI can offer the experimental
+ # mode; the effective minimum only changes in ioctl mode.
+ try:
+ from . import rm_power
+
+ bounds = rm_power.probe_gpu(gpu_index)
+ out["rm_power_supported"] = bounds is not None
+ if bounds is not None and mode == "ioctl":
+ out["min_power_limit_w"] = bounds.lower_min_mw() // 1000
+ except Exception as exc:
+ log.debug("RM power probe failed: %s", exc)
return out
-def set_power_limit(limit_w: int, gpu_index: int = 0) -> tuple[bool, str]:
- """Set the board power limit (Watts)."""
+def set_power_limit(
+ limit_w: int, gpu_index: int = 0, mode: str = "nvml"
+) -> tuple[bool, str]:
+ """Set the board power limit (Watts).
+
+ mode "ioctl" (experimental) applies the limit through the undocumented
+ RM interface, which permits values below the VBIOS minimum. It has no
+ fallback: failures are reported, never silently switched to NVML.
+ """
+ if mode == "ioctl":
+ from . import rm_power
+
+ try:
+ rm_power.set_power_limit_w(gpu_index, limit_w)
+ return True, "OK"
+ except rm_power.RmPowerError as exc:
+ return False, str(exc)
try:
handle = _get_handle(gpu_index)
pynvml.nvmlDeviceSetPowerManagementLimit(handle, limit_w * 1000)
diff --git a/nvcurve/hal/rm_power.py b/nvcurve/hal/rm_power.py
new file mode 100644
index 0000000..3a8f9d5
--- /dev/null
+++ b/nvcurve/hal/rm_power.py
@@ -0,0 +1,627 @@
+"""Undocumented NVIDIA RM power-limit interface (EXPERIMENTAL).
+
+Port of the approach from LACT PR #1205 (ilya-zlobintsev/LACT): applies board
+power limits through the private NV2080 power-limit "ordinary client"
+interface on /dev/nvidiactl, which permits caps below the VBIOS minimum
+(down to 30 W). The native maximum still applies.
+
+EXPERIMENTAL — uses an undocumented driver interface. It may break after
+driver updates. Discovery is GET-only and validates the RM payload against
+NVML before any write is issued; a failed write restores the previous
+request (even if it was below the VBIOS minimum).
+"""
+
+from __future__ import annotations
+
+import contextlib
+import ctypes
+import fcntl
+import logging
+import os
+import struct
+import sys
+from collections.abc import Callable
+from dataclasses import dataclass
+
+log = logging.getLogger("nvcurve.hal.rm_power")
+
+# ── ioctl constants (nv-ioctl.h / nv-ioctl-numbers.h) ─────────────────────────
+
+NV_IOCTL_MAGIC = ord("N") # 0x4E — user-space RM interface
+NV_ESC_RM_ALLOC = 0x2B
+NV_ESC_RM_CONTROL = 0x2A
+
+# 'F' magic interface (kernel-open/common/inc/nv-ioctl-numbers.h) —
+# NV_ESC_REGISTER_FD lives here, not in the 'N' RM interface.
+NV_IOCTL_MAGIC_F = ord("F") # 0x46
+NV_IOCTL_BASE_F = 200
+NV_ESC_REGISTER_FD = NV_IOCTL_BASE_F + 1 # 201
+
+# RM class IDs (nv0080.h / nv2080.h)
+NV01_DEVICE_0 = 0x0080
+NV20_SUBDEVICE_0 = 0x2080
+
+# NV01_ROOT GPU queries (ctrl0000gpu.h) — resolve PCI identity to the RM
+# device/subdevice instance numbers used by NV0080 and NV2080 allocations;
+# neither number is a Linux device minor.
+_CTRL_GPU_GET_ATTACHED_IDS = 0x201
+_CTRL_GPU_GET_ID_INFO_V2 = 0x205
+_CTRL_GPU_GET_PCI_INFO = 0x21B
+_MAX_GPUS = 32
+_INVALID_GPU_ID = 0xFFFFFFFF
+
+# Private NV2080 power-limit client commands. Payloads compared against
+# NvAPI and GSP from R595, R610 and R615 (native RM payloads, without
+# NvAPI's 0x10-byte transport prefix).
+_PWR_GET_INFO = 0x2080_A630
+_PWR_GET_CONTROL = 0x2080_A632
+_PWR_SET_CONTROL = 0x2080_E633
+_ORDINARY_CLIENT = 0xFE
+_LOWER_LIMIT_MW = 30_000 # experimental floor: 30 W
+
+
+def _ioctl_rw(size: int, nr: int, magic: int = NV_IOCTL_MAGIC) -> int:
+ """Linux ioctl request code: dir=RW, given size/type/nr."""
+ return (2 << 30) | (size << 16) | (magic << 8) | nr
+
+
+def _ioctl_call(fd: int, code: int, arg) -> None:
+ """Issue an ioctl, converting errno failures to RmPowerError.
+
+ The driver normally reports failures as an RM status in the parameter
+ struct, but an experimental interface can also fail at the kernel level
+ (ENOTTY/EBADF/EPERM across driver versions). Converting to RmPowerError
+ keeps the module's error contract uniform and lets callers clean up fds.
+ """
+ try:
+ fcntl.ioctl(fd, code, arg)
+ except OSError as exc:
+ raise RmPowerError(f"ioctl 0x{code:x} failed: {exc}") from exc
+
+
+# ── NVOS parameter structs (nvos.h) ──────────────────────────────────────────
+
+
+class _NVOS21(ctypes.Structure):
+ _fields_ = [
+ ("hRoot", ctypes.c_uint32),
+ ("hObjectParent", ctypes.c_uint32),
+ ("hObjectNew", ctypes.c_uint32),
+ ("hClass", ctypes.c_uint32),
+ ("pAllocParms", ctypes.c_uint64),
+ ("paramsSize", ctypes.c_uint32),
+ ("status", ctypes.c_uint32),
+ ]
+
+
+class _NVOS64(ctypes.Structure):
+ _fields_ = [
+ ("hRoot", ctypes.c_uint32),
+ ("hObjectParent", ctypes.c_uint32),
+ ("hObjectNew", ctypes.c_uint32),
+ ("hClass", ctypes.c_uint32),
+ ("pAllocParms", ctypes.c_uint64),
+ ("pRightsRequested", ctypes.c_uint64),
+ ("paramsSize", ctypes.c_uint32),
+ ("flags", ctypes.c_uint32),
+ ("status", ctypes.c_uint32),
+ ]
+
+
+class _NVOS54(ctypes.Structure):
+ _fields_ = [
+ ("hClient", ctypes.c_uint32),
+ ("hObject", ctypes.c_uint32),
+ ("cmd", ctypes.c_uint32),
+ ("flags", ctypes.c_uint32),
+ ("params", ctypes.c_uint64),
+ ("paramsSize", ctypes.c_uint32),
+ ("status", ctypes.c_uint32),
+ ]
+
+
+class _NV0080_ALLOC(ctypes.Structure):
+ _fields_ = [
+ ("deviceId", ctypes.c_uint32),
+ ("deviceFlags", ctypes.c_uint32),
+ ("vgpuInstance", ctypes.c_uint32),
+ ("pad", ctypes.c_uint32),
+ ]
+
+
+class _NV2080_ALLOC(ctypes.Structure):
+ _fields_ = [
+ ("subDeviceId", ctypes.c_uint32),
+ ("clientShare", ctypes.c_uint32),
+ ("flags", ctypes.c_uint32),
+ ("pad", ctypes.c_uint32),
+ ]
+
+
+# ── Errors ───────────────────────────────────────────────────────────────────
+
+
+class RmPowerError(RuntimeError):
+ """Raised when the RM power-limit interface is unavailable or fails."""
+
+
+# ── Power-limit layouts and bounds ───────────────────────────────────────────
+
+
+@dataclass(frozen=True)
+class PowerLimitLayout:
+ """Byte offsets of the private power-limit payloads for one wire format."""
+
+ name: str
+ info_size: int
+ control_size: int
+ info_min_at: int
+ request_at: int
+ client_at: int
+ mask_end: int
+
+
+EXTENDED_LAYOUT = PowerLimitLayout(
+ name="extended",
+ info_size=0x924,
+ control_size=0x328,
+ info_min_at=0x28,
+ request_at=0x2C,
+ client_at=0x30,
+ mask_end=0x24,
+)
+LEGACY_LAYOUT = PowerLimitLayout(
+ name="legacy",
+ info_size=0x488,
+ control_size=0x188,
+ info_min_at=0xC,
+ request_at=0xC,
+ client_at=0x10,
+ mask_end=0x8,
+)
+
+
+@dataclass(frozen=True)
+class PowerLimitBounds:
+ """Power limit bounds in milliwatts (NVML/RM units)."""
+
+ min_mw: int
+ default_mw: int
+ max_mw: int
+
+ def lower_min_mw(self) -> int:
+ """Effective minimum when the experimental route is active."""
+ return min(self.min_mw, _LOWER_LIMIT_MW)
+
+
+@dataclass(frozen=True)
+class LowerPowerLimit:
+ """A validated RM power-limit layout that can be written."""
+
+ bounds: PowerLimitBounds
+ layout: PowerLimitLayout
+
+ def lower_min_mw(self) -> int:
+ return self.bounds.lower_min_mw()
+
+
+# ── PCI identity → RM instance resolution ────────────────────────────────────
+
+
+@dataclass(frozen=True)
+class PciLocation:
+ domain: int
+ bus: int
+ dev: int
+ func: int = 0
+
+
+def resolve_gpu_instance(
+ pci: PciLocation,
+ query: Callable[[int, bytearray], None],
+) -> tuple[int, int]:
+ """Resolve (device_instance, subdevice_instance) by PCI identity.
+
+ /dev/nvidiaN minors and RM device instances can have different orders;
+ the RM object must be matched by PCI domain/bus/slot, not by index.
+ The RM query exposes domain/bus/slot but no PCI function, so only
+ function-zero devices can be matched (never another function of a
+ multifunction device).
+ """
+ if pci.func != 0:
+ raise RmPowerError("RM GPU lookup requires PCI function zero")
+
+ attached = bytearray(_MAX_GPUS * 4)
+ query(_CTRL_GPU_GET_ATTACHED_IDS, attached)
+
+ for i in range(_MAX_GPUS):
+ gpu_id = struct.unpack_from(" None:
+ """Issue an NVOS54 RM control whose parameter block is a byte buffer."""
+ arr = (ctypes.c_uint8 * len(buf)).from_buffer(buf)
+ req = _NVOS54(
+ hClient=client,
+ hObject=obj,
+ cmd=cmd,
+ flags=0,
+ params=ctypes.addressof(arr),
+ paramsSize=len(buf),
+ status=0,
+ )
+ _ioctl_call(fd, _ioctl_rw(ctypes.sizeof(_NVOS54), NV_ESC_RM_CONTROL), req)
+ if req.status != 0:
+ raise RmPowerError(
+ f"RM control 0x{cmd:08x} failed with status 0x{req.status:x}"
+ )
+
+
+def _alloc_client(fd: int) -> int:
+ """Allocate an RM client (NVOS21, all-zero parameters)."""
+ req = _NVOS21()
+ _ioctl_call(fd, _ioctl_rw(ctypes.sizeof(_NVOS21), NV_ESC_RM_ALLOC), req)
+ if req.status != 0:
+ raise RmPowerError(f"could not allocate RM client (status 0x{req.status:x})")
+ return req.hObjectNew
+
+
+def _alloc_object(
+ fd: int, client: int, parent: int, class_id: int, alloc_params: ctypes.Structure
+) -> int:
+ """Allocate an RM object (NVOS64) and return its handle."""
+ req = _NVOS64(
+ hRoot=client,
+ hObjectParent=parent,
+ hObjectNew=0,
+ hClass=class_id,
+ pAllocParms=ctypes.addressof(alloc_params),
+ pRightsRequested=0,
+ paramsSize=ctypes.sizeof(alloc_params),
+ flags=0,
+ status=0,
+ )
+ _ioctl_call(fd, _ioctl_rw(ctypes.sizeof(_NVOS64), NV_ESC_RM_ALLOC), req)
+ if req.status != 0:
+ raise RmPowerError(
+ f"RM class 0x{class_id:x} allocation failed (status 0x{req.status:x})"
+ )
+ return req.hObjectNew
+
+
+def _register_fd(device_fd: int, nvidiactl_fd: int) -> None:
+ """Register the nvidiactl client with the device fd (NV_ESC_REGISTER_FD).
+
+ The ioctl is issued on the /dev/nvidiaN fd; the argument is the
+ nvidiactl fd to associate with it.
+ """
+ _ioctl_call(
+ device_fd,
+ _ioctl_rw(4, NV_ESC_REGISTER_FD, NV_IOCTL_MAGIC_F),
+ struct.pack("i", nvidiactl_fd),
+ )
+
+
+class RmHandle:
+ """An NVIDIA RM client with device + subdevice objects for one GPU."""
+
+ def __init__(
+ self,
+ nvidiactl_fd: int,
+ device_fd: int,
+ client_handle: int,
+ device_handle: int,
+ subdevice_handle: int,
+ ) -> None:
+ self._nvidiactl_fd = nvidiactl_fd
+ self._device_fd = device_fd
+ self.client_handle = client_handle
+ self.device_handle = device_handle
+ self.subdevice_handle = subdevice_handle
+
+ @classmethod
+ def open(cls, gpu_index: int) -> RmHandle:
+ """Open an RM handle for the GPU at the given NVML index.
+
+ The RM device/subdevice instances are resolved by PCI identity
+ (minors and RM instances can have different orders).
+ """
+ pynvml = _ensure_nvml()
+
+ try:
+ handle = pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
+ minor = int(pynvml.nvmlDeviceGetMinorNumber(handle))
+ pci_info = pynvml.nvmlDeviceGetPciInfo(handle)
+ pci = PciLocation(
+ domain=int(pci_info.domain),
+ bus=int(pci_info.bus),
+ dev=int(pci_info.device),
+ )
+ except Exception as exc:
+ raise RmPowerError(f"NVML query for GPU {gpu_index} failed: {exc}") from exc
+
+ try:
+ nvidiactl_fd = os.open("/dev/nvidiactl", os.O_RDWR)
+ except OSError as exc:
+ raise RmPowerError(f"could not open /dev/nvidiactl: {exc}") from exc
+
+ try:
+ client_handle = _alloc_client(nvidiactl_fd)
+ device_instance, subdevice_instance = resolve_gpu_instance(
+ pci,
+ lambda cmd, buf: _rm_control(
+ nvidiactl_fd, client_handle, client_handle, cmd, buf
+ ),
+ )
+ except RmPowerError:
+ os.close(nvidiactl_fd)
+ raise
+
+ try:
+ device_fd = os.open(f"/dev/nvidia{minor}", os.O_RDWR)
+ except OSError as exc:
+ os.close(nvidiactl_fd)
+ raise RmPowerError(f"could not open /dev/nvidia{minor}: {exc}") from exc
+
+ try:
+ _register_fd(device_fd, nvidiactl_fd)
+ device_handle = _alloc_object(
+ nvidiactl_fd,
+ client_handle,
+ client_handle,
+ NV01_DEVICE_0,
+ _NV0080_ALLOC(deviceId=device_instance),
+ )
+ subdevice_handle = _alloc_object(
+ nvidiactl_fd,
+ client_handle,
+ device_handle,
+ NV20_SUBDEVICE_0,
+ _NV2080_ALLOC(subDeviceId=subdevice_instance),
+ )
+ except RmPowerError:
+ os.close(device_fd)
+ os.close(nvidiactl_fd)
+ raise
+
+ return cls(
+ nvidiactl_fd, device_fd, client_handle, device_handle, subdevice_handle
+ )
+
+ def control(self, cmd: int, buf: bytearray) -> None:
+ """Issue an NVOS54 RM control on the subdevice with a byte buffer."""
+ _rm_control(
+ self._nvidiactl_fd, self.client_handle, self.subdevice_handle, cmd, buf
+ )
+
+ def close(self) -> None:
+ """Close the fds; the driver reclaims the RM client objects."""
+ with contextlib.suppress(OSError):
+ os.close(self._device_fd)
+ with contextlib.suppress(OSError):
+ os.close(self._nvidiactl_fd)
+
+
+# ── Power-limit probe / set (pure logic, testable with a fake query) ─────────
+
+
+def _u32(data: bytearray | bytes, offset: int) -> int:
+ return struct.unpack_from(" None:
+ if _u32(data, 0) != 0xFF or _u32(data, 4) != 1:
+ raise RmPowerError("unrecognized RM power client layout")
+ # The extended layout has additional mask words; accepting only its low
+ # word would allow an unexpected client to be included in a later SET.
+ if any(byte != 0 for byte in data[8 : layout.mask_end]):
+ raise RmPowerError("unrecognized RM power client layout")
+
+
+def _read_bounds(
+ layout: PowerLimitLayout, query: Callable[[int, bytearray], None]
+) -> PowerLimitBounds:
+ info = bytearray(layout.info_size)
+ query(_PWR_GET_INFO, info)
+ _validate_header(layout, info)
+ bounds = PowerLimitBounds(
+ min_mw=_u32(info, layout.info_min_at),
+ default_mw=_u32(info, layout.info_min_at + 4),
+ max_mw=_u32(info, layout.info_min_at + 8),
+ )
+ if not (
+ bounds.min_mw > 0
+ and bounds.min_mw <= bounds.default_mw
+ and bounds.default_mw <= bounds.max_mw
+ ):
+ raise RmPowerError("invalid RM power limit bounds")
+ return bounds
+
+
+def _read_control(
+ layout: PowerLimitLayout, query: Callable[[int, bytearray], None]
+) -> bytearray:
+ control = bytearray(layout.control_size)
+ control[4:8] = struct.pack(" LowerPowerLimit:
+ """GET-only discovery of the RM power-limit layout.
+
+ Probes the two known wire formats using GETs only. A driver version
+ number is not evidence that the payload still has the same layout or
+ units, so the bounds and the current request are validated against
+ NVML. Discovery never issues a SET.
+ """
+ if sys.byteorder != "little":
+ raise RmPowerError("little-endian host required")
+ errors: list[str] = []
+ for layout in (EXTENDED_LAYOUT, LEGACY_LAYOUT):
+ try:
+ bounds = _read_bounds(layout, query)
+ if bounds != nvml_bounds:
+ raise RmPowerError("RM power bounds differ from NVML")
+ control = _read_control(layout, query)
+ if _u32(control, layout.request_at) != nvml_current_mw:
+ raise RmPowerError("RM ordinary power request differs from NVML")
+ return LowerPowerLimit(bounds=bounds, layout=layout)
+ except RmPowerError as exc:
+ errors.append(f"{layout.name}: {exc}")
+ raise RmPowerError("no compatible RM power layout: " + "; ".join(errors))
+
+
+def set_limit(
+ limit_mw: int,
+ support: LowerPowerLimit,
+ query: Callable[[int, bytearray], None],
+) -> None:
+ """Set the ordinary-client power request with readback verification.
+
+ Keeps the entire current payload, changing only entry 0's request.
+ Mask 1 and selector 0xFE prevent modifying any other entry or the
+ additional F8 client. A failed SET can have side effects, so the
+ previous request is restored even on transport failure — and the
+ restore uses 0xFE so a previous limit below the VBIOS minimum can
+ also be restored.
+ """
+ layout = support.layout
+ bounds = _read_bounds(layout, query)
+ if bounds != support.bounds:
+ raise RmPowerError("RM power bounds changed since discovery")
+ lower = bounds.lower_min_mw()
+ if not (lower <= limit_mw <= bounds.max_mw):
+ raise RmPowerError(
+ f"power limit {limit_mw} mW outside supported range "
+ f"{lower}..{bounds.max_mw} mW"
+ )
+
+ before = _read_control(layout, query)
+ if _u32(before, layout.request_at) == limit_mw:
+ return
+
+ expected = bytearray(before)
+ expected[layout.request_at : layout.request_at + 4] = struct.pack(" tuple[PowerLimitBounds, int]:
+ """Return (bounds, current_mw) from NVML for the given GPU."""
+ pynvml = _ensure_nvml()
+
+ try:
+ handle = pynvml.nvmlDeviceGetHandleByIndex(gpu_index)
+ min_mw, max_mw = pynvml.nvmlDeviceGetPowerManagementLimitConstraints(handle)
+ default_mw = pynvml.nvmlDeviceGetPowerManagementDefaultLimit(handle)
+ current_mw = pynvml.nvmlDeviceGetPowerManagementLimit(handle)
+ bounds = PowerLimitBounds(int(min_mw), int(default_mw), int(max_mw))
+ current = int(current_mw)
+ except Exception as exc:
+ raise RmPowerError(
+ f"NVML power state for GPU {gpu_index} unavailable: {exc}"
+ ) from exc
+ return bounds, current
+
+
+def probe_gpu(gpu_index: int = 0) -> PowerLimitBounds | None:
+ """GET-only discovery of the RM power-limit interface for a GPU.
+
+ Returns the validated power bounds (milliwatts) when a compatible RM
+ layout is present, else None. Never issues a write.
+ """
+ try:
+ bounds, current = _nvml_power_state(gpu_index)
+ except Exception as exc:
+ log.debug("RM probe: NVML state unavailable: %s", exc)
+ return None
+ try:
+ handle = RmHandle.open(gpu_index)
+ except RmPowerError as exc:
+ log.debug("RM probe: handle open failed: %s", exc)
+ return None
+ try:
+ probe(bounds, current, handle.control)
+ return bounds
+ except RmPowerError as exc:
+ log.debug("RM probe: %s", exc)
+ return None
+ finally:
+ handle.close()
+
+
+def set_power_limit_w(gpu_index: int, limit_w: int) -> None:
+ """Set the board power limit (watts) via the RM interface.
+
+ Raises RmPowerError on any failure (probe, range, write, readback).
+ A failed write restores the previous request.
+ """
+ bounds, current = _nvml_power_state(gpu_index)
+ handle = RmHandle.open(gpu_index)
+ try:
+ support = probe(bounds, current, handle.control)
+ set_limit(int(limit_w) * 1000, support, handle.control)
+ finally:
+ handle.close()
diff --git a/nvcurve/nvapi/bootstrap.py b/nvcurve/nvapi/bootstrap.py
index f56174c..6660fa7 100644
--- a/nvcurve/nvapi/bootstrap.py
+++ b/nvcurve/nvapi/bootstrap.py
@@ -9,7 +9,7 @@ from .errors import NVAPI_ERRORS
def load_nvapi() -> ctypes.CDLL:
"""Load libnvidia-api.so from the NVIDIA driver."""
- for name in ("libnvidia-api.so", "libnvidia-api.so.1"):
+ for name in ("libnvidia-api.so", "libnvidia-api.so.1"): # gitleaks:allow
try:
return ctypes.CDLL(name)
except OSError:
diff --git a/nvcurve/profiles/apply.py b/nvcurve/profiles/apply.py
index d2e324e..2dc8155 100644
--- a/nvcurve/profiles/apply.py
+++ b/nvcurve/profiles/apply.py
@@ -43,7 +43,8 @@ def apply_profile(gpu_index: int, name: str, cfg) -> list[str]:
errs.append(f"Mem offset: {msg}")
if profile.power_limit_w is not None:
- ok, msg = set_power_limit(profile.power_limit_w, gpu_index)
+ mode = profile.power_cap_mode or "nvml"
+ ok, msg = set_power_limit(profile.power_limit_w, gpu_index, mode)
if not ok:
errs.append(f"Power limit: {msg}")
diff --git a/nvcurve/profiles/native.py b/nvcurve/profiles/native.py
index 454112a..bfb83e6 100644
--- a/nvcurve/profiles/native.py
+++ b/nvcurve/profiles/native.py
@@ -16,6 +16,9 @@ class ProfileData:
curve_deltas: dict[str, int] # { "index": delta_khz }
mem_offset_mhz: int | None = None
power_limit_w: int | None = None
+ # How power_limit_w is applied: "nvml" (default) or "ioctl" (experimental
+ # RM power control, permits values below the VBIOS minimum).
+ power_cap_mode: str | None = None
fan_curve: list[dict[str, int]] | None = None
# Fan indices controlled by fan_curve (0-based); None = all fans.
fan_targets: list[int] | None = None
@@ -55,6 +58,9 @@ def load_profile(filepath: str) -> ProfileData:
# Drop removed fields so old profiles don't cause TypeError.
for obsolete in ("gpu_locked_min_mhz", "gpu_locked_max_mhz", "vram_p0_offset_mhz"):
data.pop(obsolete, None)
+ # Normalize the experimental power-cap mode; unknown values fall back to NVML.
+ if data.get("power_cap_mode") not in (None, "nvml", "ioctl"):
+ data["power_cap_mode"] = None
return ProfileData(**data)
diff --git a/nvcurve/server.py b/nvcurve/server.py
index 28cb99a..25c6028 100644
--- a/nvcurve/server.py
+++ b/nvcurve/server.py
@@ -569,6 +569,8 @@ class SnapshotRestoreRequest(BaseModel):
class LimitsRequest(BaseModel):
power_limit_w: int | None = None
mem_offset_mhz: int | None = None
+ # "nvml" (default) or "ioctl" (experimental RM power control).
+ power_cap_mode: str | None = None
class ProfileSaveRequest(BaseModel):
@@ -914,6 +916,17 @@ def _persist_fan_curves(fan_curves: dict) -> None:
_persist_config_field("fan_curves", fan_curves if fan_curves else None)
+def _persist_power_cap_modes(modes: dict[str, str]) -> None:
+ """Persist the per-GPU experimental power-cap mode dict to config.json."""
+ _persist_config_field("power_cap_modes", modes if modes else None)
+
+
+def _power_cap_mode(cfg: Config, gpu_index: int) -> str:
+ """Return the effective power-cap mode for a GPU ("nvml" or "ioctl")."""
+ mode = cfg.power_cap_modes.get(_gpu_stable_key(gpu_index), "nvml")
+ return mode if mode in ("nvml", "ioctl") else "nvml"
+
+
@app.get("/api/profiles")
async def api_profiles(gpu_index: int = 0):
"""List saved native profiles, the active profile name, and the auto-load profile name."""
@@ -941,13 +954,15 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
curve_deltas = {str(p.index): p.delta_khz for p in state.points if p.delta_khz != 0}
try:
- power_info = await _run(get_power_limit, gpu_index)
+ mode = _power_cap_mode(cfg, gpu_index)
+ power_info = await _run(get_power_limit, gpu_index, mode)
offsets = await _run(get_clock_offsets, gpu_index)
power_limit_w = power_info.get("power_limit_w")
mem_offset_mhz = offsets.get("mem_offset_mhz")
except Exception:
power_limit_w = None
mem_offset_mhz = None
+ mode = "nvml"
data = ProfileData(
name=req.name,
@@ -955,6 +970,7 @@ async def api_profile_save(req: ProfileSaveRequest, gpu_index: int = 0):
curve_deltas=curve_deltas,
mem_offset_mhz=mem_offset_mhz,
power_limit_w=power_limit_w,
+ power_cap_mode=mode,
fan_curve=g_state.get("fan_curve") if g_state.get("fan_curve_active") else None,
fan_targets=g_state.get("fan_targets")
if g_state.get("fan_curve_active")
@@ -1075,7 +1091,8 @@ async def _apply_profile(name: str, gpu_index: int = 0) -> list[str]:
errs.append(f"Mem offset: {msg}")
if profile.power_limit_w is not None:
- ok, msg = await _run(set_power_limit, profile.power_limit_w, gpu_index)
+ mode = profile.power_cap_mode or "nvml"
+ ok, msg = await _run(set_power_limit, profile.power_limit_w, gpu_index, mode)
if not ok:
errs.append(f"Power limit: {msg}")
@@ -1204,7 +1221,9 @@ async def api_config_update(req: ConfigUpdateRequest):
@app.get("/api/limits")
async def api_limits(gpu_index: int = 0):
"""Current performance limits: power and clock offsets."""
- power = await _run(get_power_limit, gpu_index)
+ cfg: Config = _state["config"]
+ mode = _power_cap_mode(cfg, gpu_index)
+ power = await _run(get_power_limit, gpu_index, mode)
offsets = await _run(get_clock_offsets, gpu_index)
mem_off_range = await _run(get_mem_offset_range, gpu_index)
return {
@@ -1218,10 +1237,36 @@ async def api_limits(gpu_index: int = 0):
async def api_limits_update(req: LimitsRequest, gpu_index: int = 0):
"""Update performance limits."""
g_state = _get_gpu_state(gpu_index)
+ cfg: Config = _state["config"]
errs = []
+ if req.power_cap_mode is not None:
+ if req.power_cap_mode not in ("nvml", "ioctl"):
+ raise HTTPException(
+ status_code=400, detail="power_cap_mode must be 'nvml' or 'ioctl'"
+ )
+ if req.power_cap_mode == "ioctl":
+ # Verify the GPU actually exposes the RM interface before enabling,
+ # so a client can't lock a GPU into a mode where every power
+ # operation fails (ioctl mode has no NVML fallback by design).
+ info = await _run(get_power_limit, gpu_index, "ioctl")
+ if not info.get("rm_power_supported"):
+ raise HTTPException(
+ status_code=409,
+ detail="Experimental RM power control is not supported "
+ "on this GPU/driver",
+ )
+ key = _gpu_stable_key(gpu_index)
+ if req.power_cap_mode == "nvml":
+ cfg.power_cap_modes.pop(key, None)
+ else:
+ cfg.power_cap_modes[key] = "ioctl"
+ _persist_power_cap_modes(cfg.power_cap_modes)
+
+ mode = _power_cap_mode(cfg, gpu_index)
+
if req.power_limit_w is not None:
- ok, msg = await _run(set_power_limit, req.power_limit_w, gpu_index)
+ ok, msg = await _run(set_power_limit, req.power_limit_w, gpu_index, mode)
if not ok:
errs.append(f"Power Limit: {msg}")
@@ -1284,12 +1329,17 @@ async def _update_offsets_and_broadcast(gpu_index: int) -> None:
async def api_limits_reset(gpu_index: int = 0):
"""Reset power limit to hardware default and memory clock offset to 0."""
g_state = _get_gpu_state(gpu_index)
+ cfg: Config = _state["config"]
errs = []
- power = await _run(get_power_limit, gpu_index)
+ # Reset uses the GPU's current mode: in ioctl mode the default is
+ # restored through the RM route (which can also restore a previous
+ # below-VBIOS-minimum cap).
+ mode = _power_cap_mode(cfg, gpu_index)
+ power = await _run(get_power_limit, gpu_index, mode)
default_w = power.get("default_power_limit_w")
if default_w is not None:
- ok, msg = await _run(set_power_limit, default_w, gpu_index)
+ ok, msg = await _run(set_power_limit, default_w, gpu_index, mode)
if not ok:
errs.append(f"Power Limit: {msg}")
diff --git a/scripts/nv_vfcurve_rw.py b/scripts/nv_vfcurve_rw.py
index bec42e1..1628ce7 100644
--- a/scripts/nv_vfcurve_rw.py
+++ b/scripts/nv_vfcurve_rw.py
@@ -55,20 +55,20 @@ Key findings:
See NvAPI_VF_Curve_Documentation.md for full technical details.
"""
+import argparse
import ctypes
-import struct
-import sys
import json
import os
+import struct
+import sys
import time
-import argparse
from datetime import datetime
-from typing import Optional, List, Tuple, Set, Dict
# ═══════════════════════════════════════════════════════════════════════════
# NvAPI bootstrap
# ═══════════════════════════════════════════════════════════════════════════
+
def load_nvapi():
"""Load libnvidia-api.so from the NVIDIA driver."""
for name in ("libnvidia-api.so", "libnvidia-api.so.1"):
@@ -145,23 +145,20 @@ def nvcall_raw(fid: int, gpu, buf: ctypes.Array):
FUNC = {
# Bootstrap
- "Initialize": 0x0150E828,
- "EnumPhysicalGPUs": 0xE5AC921F,
- "GetFullName": 0xCEEE8E9F,
-
+ "Initialize": 0x0150E828,
+ "EnumPhysicalGPUs": 0xE5AC921F,
+ "GetFullName": 0xCEEE8E9F,
# V/F curve (read)
- "GetVFPCurve": 0x21537AD4, # ClkVfPointsGetStatus
- "GetClockBoostMask": 0x507B4B59, # ClkVfPointsGetInfo
- "GetClockBoostTable": 0x23F1B133, # ClkVfPointsGetControl
- "GetCurrentVoltage": 0x465F9BCF, # ClientVoltRailsGetStatus
+ "GetVFPCurve": 0x21537AD4, # ClkVfPointsGetStatus
+ "GetClockBoostMask": 0x507B4B59, # ClkVfPointsGetInfo
+ "GetClockBoostTable": 0x23F1B133, # ClkVfPointsGetControl
+ "GetCurrentVoltage": 0x465F9BCF, # ClientVoltRailsGetStatus
"GetClockBoostRanges": 0x64B43A6A, # ClkDomainsGetInfo
-
# Additional read
- "GetPerfLimits": 0xE440B867, # PerfClientLimitsGetStatus
+ "GetPerfLimits": 0xE440B867, # PerfClientLimitsGetStatus
"GetVoltBoostPercent": 0x9DF23CA1, # ClientVoltRailsGetControl
-
# Write
- "SetClockBoostTable": 0x0733E009, # ClkVfPointsSetControl
+ "SetClockBoostTable": 0x0733E009, # ClkVfPointsSetControl
}
# ═══════════════════════════════════════════════════════════════════════════
@@ -170,33 +167,33 @@ FUNC = {
# ═══════════════════════════════════════════════════════════════════════════
# GetVFPCurve (0x21537AD4)
-VFP_SIZE = 0x1C28
-VFP_BASE = 0x48
-VFP_STRIDE = 0x1C # 28 bytes
+VFP_SIZE = 0x1C28
+VFP_BASE = 0x48
+VFP_STRIDE = 0x1C # 28 bytes
VFP_MAX_ENTRIES = (VFP_SIZE - VFP_BASE) // VFP_STRIDE # 255
# Get/SetClockBoostTable (0x23F1B133 / 0x0733E009)
-CT_SIZE = 0x2420
-CT_BASE = 0x44
-CT_STRIDE = 0x24 # 36 bytes
+CT_SIZE = 0x2420
+CT_BASE = 0x44
+CT_STRIDE = 0x24 # 36 bytes
CT_DELTA_OFF = 0x14 # freqDelta offset within entry
CT_MAX_ENTRIES = (CT_SIZE - CT_BASE) // CT_STRIDE # 255
# GetClockBoostMask (0x507B4B59)
-MASK_SIZE = 0x182C
+MASK_SIZE = 0x182C
# Other structs
-VOLT_SIZE = 0x004C
-RANGES_SIZE = 0x0928
-PERF_SIZE = 0x030C
-VBOOST_SIZE = 0x0028
+VOLT_SIZE = 0x004C
+RANGES_SIZE = 0x0928
+PERF_SIZE = 0x030C
+VBOOST_SIZE = 0x0028
# Mask location within VFP/CT structs
-MASK_OFFSET = 0x04
-MASK_BYTES = 32 # 256 bits — covers up to 256 points
+MASK_OFFSET = 0x04
+MASK_BYTES = 32 # 256 bits — covers up to 256 points
# Safety constants
-MAX_DELTA_KHZ = 300_000 # ±300 MHz hard cap for safety
+MAX_DELTA_KHZ = 300_000 # ±300 MHz hard cap for safety
SNAPSHOT_DIR = os.path.expanduser("~/.cache/nv_vfcurve")
@@ -205,6 +202,7 @@ SNAPSHOT_DIR = os.path.expanduser("~/.cache/nv_vfcurve")
# GPU initialization
# ═══════════════════════════════════════════════════════════════════════════
+
def init_gpu() -> tuple:
"""Initialize NvAPI, enumerate GPUs, return (handle, name)."""
init_fn = nvfunc(FUNC["Initialize"], 0)
@@ -236,18 +234,20 @@ def init_gpu() -> tuple:
# also distinguishes GPU core vs memory clock domains.
# ═══════════════════════════════════════════════════════════════════════════
+
class BoostMask:
"""Parsed GetClockBoostMask data.
Provides the raw mask bytes for copying into other calls, plus
parsed per-entry enabled info for filtering.
"""
+
def __init__(self, raw: bytes):
self.raw = raw
self.size = len(raw)
# The mask field at offset 0x04, 16 bytes — same position as in VFP/CT structs
- self.mask_bytes = raw[MASK_OFFSET:MASK_OFFSET + MASK_BYTES]
+ self.mask_bytes = raw[MASK_OFFSET : MASK_OFFSET + MASK_BYTES]
self.entries = []
self._parse_entries()
@@ -260,7 +260,7 @@ class BoostMask:
enabled = bool(self.mask_bytes[byte_idx] & (1 << bit_idx))
self.entries.append({"index": i, "enabled": enabled})
- def get_enabled_indices(self) -> List[int]:
+ def get_enabled_indices(self) -> list[int]:
"""Return list of point indices that are enabled in the mask."""
return [e["index"] for e in self.entries if e["enabled"]]
@@ -273,12 +273,13 @@ class BoostMask:
buf[offset + i] = self.mask_bytes[i]
-def read_boost_mask(gpu) -> Tuple[Optional[BoostMask], str]:
+def read_boost_mask(gpu) -> tuple[BoostMask | None, str]:
"""Read the clock boost mask — the canonical source of active point info.
Per nvapioc, this mask must be copied into VFP and ClockBoostTable calls.
Using all-0xFF works on some GPUs (Blackwell) but fails on others (Pascal).
"""
+
def fill(buf):
for i in range(MASK_OFFSET, MASK_OFFSET + MASK_BYTES):
buf[i] = 0xFF
@@ -294,20 +295,22 @@ def read_boost_mask(gpu) -> Tuple[Optional[BoostMask], str]:
# Point classification — GPU core vs memory
# ═══════════════════════════════════════════════════════════════════════════
+
class CurveInfo:
"""Holds classified point information for the GPU's V/F curve.
Combines data from GetClockBoostMask, GetVFPCurve, and GetClockBoostTable
to determine which points are GPU core and which are memory.
"""
+
def __init__(self):
- self.gpu_points: List[int] = [] # GPU core V/F point indices
- self.mem_points: List[int] = [] # Memory V/F point indices
- self.total_points: int = 0 # Total populated entries
- self.mask: Optional[BoostMask] = None
+ self.gpu_points: list[int] = [] # GPU core V/F point indices
+ self.mem_points: list[int] = [] # Memory V/F point indices
+ self.total_points: int = 0 # Total populated entries
+ self.mask: BoostMask | None = None
@staticmethod
- def build(gpu, mask: Optional[BoostMask] = None) -> 'CurveInfo':
+ def build(gpu, mask: BoostMask | None = None) -> "CurveInfo":
"""Classify all points by reading CT field_00 and VFP data.
field_00 == 0: GPU core data point
@@ -340,7 +343,7 @@ class CurveInfo:
has_vfp_data = False
if vfp_points and i < len(vfp_points):
f, v = vfp_points[i]
- has_vfp_data = (f > 0 or v > 0)
+ has_vfp_data = f > 0 or v > 0
has_ct_data = False
for j in range(9):
@@ -378,13 +381,15 @@ class CurveInfo:
# Data readers (mask-aware)
# ═══════════════════════════════════════════════════════════════════════════
+
def _fill_mask_from_boost(buf, mask: BoostMask):
"""Copy boost mask into buffer."""
mask.copy_mask_into(buf)
-def _read_vfp_with_mask(gpu, mask: Optional[BoostMask]) -> Optional[List[Tuple[int, int]]]:
+def _read_vfp_with_mask(gpu, mask: BoostMask | None) -> list[tuple[int, int]] | None:
"""Read VFP curve using the canonical boost mask."""
+
def fill(buf):
_fill_mask_from_boost(buf, mask)
@@ -403,8 +408,9 @@ def _read_vfp_with_mask(gpu, mask: Optional[BoostMask]) -> Optional[List[Tuple[i
return points
-def _read_clock_table_raw_with_mask(gpu, mask: Optional[BoostMask]) -> Optional[bytes]:
+def _read_clock_table_raw_with_mask(gpu, mask: BoostMask | None) -> bytes | None:
"""Read raw ClockBoostTable using the canonical boost mask."""
+
def fill(buf):
_fill_mask_from_boost(buf, mask)
@@ -412,13 +418,14 @@ def _read_clock_table_raw_with_mask(gpu, mask: Optional[BoostMask]) -> Optional[
return d if d else None
-def read_vfp_curve(gpu, mask: Optional[BoostMask] = None,
- curve_info: Optional[CurveInfo] = None
- ) -> Tuple[Optional[List[Tuple[int, int]]], str]:
+def read_vfp_curve(
+ gpu, mask: BoostMask | None = None, curve_info: CurveInfo | None = None
+) -> tuple[list[tuple[int, int]] | None, str]:
"""Read V/F curve (frequency + voltage pairs).
Returns up to 255 entries. Use curve_info to determine which are GPU/mem.
"""
+
def fill(buf):
_fill_mask_from_boost(buf, mask)
@@ -442,18 +449,20 @@ def read_vfp_curve(gpu, mask: Optional[BoostMask] = None,
return points, "OK"
-def read_clock_table_raw(gpu, mask: Optional[BoostMask] = None
- ) -> Tuple[Optional[bytes], str]:
+def read_clock_table_raw(
+ gpu, mask: BoostMask | None = None
+) -> tuple[bytes | None, str]:
"""Read the raw ClockBoostTable buffer."""
+
def fill(buf):
_fill_mask_from_boost(buf, mask)
return nvcall(FUNC["GetClockBoostTable"], gpu, CT_SIZE, ver=1, pre_fill=fill)
-def read_clock_offsets(gpu, mask: Optional[BoostMask] = None,
- curve_info: Optional[CurveInfo] = None
- ) -> Tuple[Optional[List[int]], str]:
+def read_clock_offsets(
+ gpu, mask: BoostMask | None = None, curve_info: CurveInfo | None = None
+) -> tuple[list[int] | None, str]:
"""Read per-point frequency offsets from the ClockBoostTable."""
d, err = read_clock_table_raw(gpu, mask)
if not d:
@@ -482,14 +491,18 @@ def read_clock_entry_full(data: bytes, point: int) -> dict:
for j in range(9):
off = base + j * 4
if j == 5:
- fields[f"field_{j:02d}_0x{j*4:02X}"] = struct.unpack_from(" Tuple[Optional[int], str]:
+def read_voltage(gpu) -> tuple[int | None, str]:
"""Read current GPU core voltage in µV."""
d, err = nvcall(FUNC["GetCurrentVoltage"], gpu, VOLT_SIZE, ver=1)
if not d:
@@ -497,7 +510,7 @@ def read_voltage(gpu) -> Tuple[Optional[int], str]:
return struct.unpack_from(" Tuple[Optional[dict], str]:
+def read_clock_ranges(gpu) -> tuple[dict | None, str]:
"""Read clock domain min/max offset ranges."""
d, err = nvcall(FUNC["GetClockBoostRanges"], gpu, RANGES_SIZE, ver=1)
if not d:
@@ -508,8 +521,7 @@ def read_clock_ranges(gpu) -> Tuple[Optional[dict], str]:
base = 0x08 + i * 0x48
if base + 0x48 > len(d):
break
- words = [struct.unpack_from(" Tuple[Optional[dict], str]:
# Mask bit helpers
# ═══════════════════════════════════════════════════════════════════════════
+
def set_mask_bit(buf, point: int, offset=MASK_OFFSET):
"""Set a single bit in the mask field."""
byte_idx = offset + (point // 8)
bit_idx = point % 8
- buf[byte_idx] = int.from_bytes(buf[byte_idx:byte_idx+1], 'little') | (1 << bit_idx)
+ buf[byte_idx] = int.from_bytes(buf[byte_idx : byte_idx + 1], "little") | (
+ 1 << bit_idx
+ )
-def set_mask_bits(buf, points: Set[int], offset=MASK_OFFSET):
+def set_mask_bits(buf, points: set[int], offset=MASK_OFFSET):
"""Set mask bits for a set of points."""
for p in points:
set_mask_bit(buf, p, offset)
@@ -535,11 +550,12 @@ def set_mask_bits(buf, points: Set[int], offset=MASK_OFFSET):
# Write operations
# ═══════════════════════════════════════════════════════════════════════════
+
def build_write_buffer(
gpu,
point_deltas: dict,
- mask: Optional[BoostMask] = None,
-) -> Tuple[Optional[ctypes.Array], str]:
+ mask: BoostMask | None = None,
+) -> tuple[ctypes.Array | None, str]:
"""Build a SetClockBoostTable buffer with specified per-point deltas.
Strategy: read the current ClockBoostTable (using canonical mask),
@@ -576,9 +592,9 @@ def build_write_buffer(
def write_clock_offsets(
gpu,
point_deltas: dict,
- mask: Optional[BoostMask] = None,
+ mask: BoostMask | None = None,
dry_run: bool = False,
-) -> Tuple[int, str]:
+) -> tuple[int, str]:
"""Write per-point frequency offsets via SetClockBoostTable."""
buf, err = build_write_buffer(gpu, point_deltas, mask)
if buf is None:
@@ -595,9 +611,10 @@ def write_clock_offsets(
# Safety checks
# ═══════════════════════════════════════════════════════════════════════════
-def validate_write_request(point_deltas: dict,
- curve_info: Optional[CurveInfo] = None
- ) -> Optional[str]:
+
+def validate_write_request(
+ point_deltas: dict, curve_info: CurveInfo | None = None
+) -> str | None:
"""Return an error message if the write request is unsafe, else None."""
mem_points = set()
if curve_info:
@@ -608,14 +625,18 @@ def validate_write_request(point_deltas: dict,
return f"Point {point} out of range (0–{CT_MAX_ENTRIES - 1})"
if point in mem_points:
- return (f"Point {point} is a memory clock entry. "
- "Memory offsets use a different mechanism (NVML). "
- "Use --force if you really mean it.")
+ return (
+ f"Point {point} is a memory clock entry. "
+ "Memory offsets use a different mechanism (NVML). "
+ "Use --force if you really mean it."
+ )
if abs(delta_khz) > MAX_DELTA_KHZ:
- return (f"Delta {delta_khz/1000:+.0f} MHz for point {point} exceeds "
- f"safety limit of ±{MAX_DELTA_KHZ/1000:.0f} MHz. "
- "Use --max-delta to raise the limit if needed.")
+ return (
+ f"Delta {delta_khz / 1000:+.0f} MHz for point {point} exceeds "
+ f"safety limit of ±{MAX_DELTA_KHZ / 1000:.0f} MHz. "
+ "Use --max-delta to raise the limit if needed."
+ )
return None
@@ -624,11 +645,12 @@ def validate_write_request(point_deltas: dict,
# Hex dump utility
# ═══════════════════════════════════════════════════════════════════════════
+
def hexdump(data: bytes, start: int, length: int, cols: int = 16) -> str:
lines = []
end = min(start + length, len(data))
for off in range(start, end, cols):
- chunk = data[off:off + cols]
+ chunk = data[off : off + cols]
hx = " ".join(f"{b:02x}" for b in chunk)
asc = "".join(chr(b) if 32 <= b < 127 else "." for b in chunk)
lines.append(f" {off:04x}: {hx:<{cols * 3}} {asc}")
@@ -639,7 +661,8 @@ def hexdump(data: bytes, start: int, length: int, cols: int = 16) -> str:
# Snapshot save/restore
# ═══════════════════════════════════════════════════════════════════════════
-def snapshot_save(gpu, gpu_name: str, mask: Optional[BoostMask] = None):
+
+def snapshot_save(gpu, gpu_name: str, mask: BoostMask | None = None):
"""Save the current ClockBoostTable to disk."""
raw, err = read_clock_table_raw(gpu, mask)
if not raw:
@@ -674,7 +697,7 @@ def snapshot_save(gpu, gpu_name: str, mask: Optional[BoostMask] = None):
with open(meta_fname, "w") as f:
json.dump(meta, f, indent=2)
- print(f"Snapshot saved:")
+ print("Snapshot saved:")
print(f" Binary: {fname}")
print(f" Metadata: {meta_fname}")
print(f" Size: {len(raw)} bytes")
@@ -682,7 +705,7 @@ def snapshot_save(gpu, gpu_name: str, mask: Optional[BoostMask] = None):
return True
-def snapshot_restore(gpu, mask: Optional[BoostMask] = None, filepath: str = None):
+def snapshot_restore(gpu, mask: BoostMask | None = None, filepath: str = None):
"""Restore a ClockBoostTable snapshot from disk."""
if filepath is None:
if not os.path.isdir(SNAPSHOT_DIR):
@@ -730,7 +753,8 @@ def snapshot_restore(gpu, mask: Optional[BoostMask] = None, filepath: str = None
# Diagnostics
# ═══════════════════════════════════════════════════════════════════════════
-def run_diagnostics(gpu, gpu_name, mask: Optional[BoostMask] = None):
+
+def run_diagnostics(gpu, gpu_name, mask: BoostMask | None = None):
"""Probe all known functions and report results."""
print(f"GPU: {gpu_name}")
print()
@@ -739,14 +763,14 @@ def run_diagnostics(gpu, gpu_name, mask: Optional[BoostMask] = None):
print("=== Function probe ===")
print()
probes = [
- ("GetVFPCurve", FUNC["GetVFPCurve"], VFP_SIZE, 1),
- ("GetClockBoostMask", FUNC["GetClockBoostMask"], MASK_SIZE, 1),
- ("GetClockBoostTable", FUNC["GetClockBoostTable"], CT_SIZE, 1),
- ("GetCurrentVoltage", FUNC["GetCurrentVoltage"], VOLT_SIZE, 1),
+ ("GetVFPCurve", FUNC["GetVFPCurve"], VFP_SIZE, 1),
+ ("GetClockBoostMask", FUNC["GetClockBoostMask"], MASK_SIZE, 1),
+ ("GetClockBoostTable", FUNC["GetClockBoostTable"], CT_SIZE, 1),
+ ("GetCurrentVoltage", FUNC["GetCurrentVoltage"], VOLT_SIZE, 1),
("GetClockBoostRanges", FUNC["GetClockBoostRanges"], RANGES_SIZE, 1),
- ("GetPerfLimits", FUNC["GetPerfLimits"], PERF_SIZE, 2),
- ("GetVoltBoostPercent", FUNC["GetVoltBoostPercent"], VBOOST_SIZE, 1),
- ("SetClockBoostTable", FUNC["SetClockBoostTable"], CT_SIZE, 1),
+ ("GetPerfLimits", FUNC["GetPerfLimits"], PERF_SIZE, 2),
+ ("GetVoltBoostPercent", FUNC["GetVoltBoostPercent"], VBOOST_SIZE, 1),
+ ("SetClockBoostTable", FUNC["SetClockBoostTable"], CT_SIZE, 1),
]
for name, fid, size, ver in probes:
ptr = QI(fid)
@@ -768,7 +792,9 @@ def run_diagnostics(gpu, gpu_name, mask: Optional[BoostMask] = None):
# Step 3: test reads with the proper mask
needs_mask_fns = {
- FUNC["GetVFPCurve"], FUNC["GetClockBoostMask"], FUNC["GetClockBoostTable"]
+ FUNC["GetVFPCurve"],
+ FUNC["GetClockBoostMask"],
+ FUNC["GetClockBoostTable"],
}
print()
@@ -817,8 +843,10 @@ def run_diagnostics(gpu, gpu_name, mask: Optional[BoostMask] = None):
# Output formatting
# ═══════════════════════════════════════════════════════════════════════════
-def print_curve(points, offsets, voltage, curve_info: Optional[CurveInfo] = None,
- full=False):
+
+def print_curve(
+ points, offsets, voltage, curve_info: CurveInfo | None = None, full=False
+):
"""Print formatted V/F curve table."""
if voltage:
print(f"Current voltage: {voltage / 1000:.1f} mV")
@@ -845,9 +873,7 @@ def print_curve(points, offsets, voltage, curve_info: Optional[CurveInfo] = None
for i, (f, v) in enumerate(points):
if f == 0 and v == 0:
continue
- if i in mem_set:
- show.append(i)
- elif f != prev_freq or i == len(points) - 1:
+ if i in mem_set or f != prev_freq or i == len(points) - 1:
show.append(i)
prev_freq = f
@@ -882,52 +908,78 @@ def print_curve(points, offsets, voltage, curve_info: Optional[CurveInfo] = None
# Summary
if curve_info and curve_info.gpu_points:
- gpu_data = [(points[i][0], points[i][1]) for i in curve_info.gpu_points
- if i < len(points) and points[i][0] > 0]
+ gpu_data = [
+ (points[i][0], points[i][1])
+ for i in curve_info.gpu_points
+ if i < len(points) and points[i][0] > 0
+ ]
if gpu_data:
freqs = [f for f, v in gpu_data]
volts = [v for f, v in gpu_data]
print()
- print(f"GPU core: {min(freqs)/1000:.0f} – {max(freqs)/1000:.0f} MHz, "
- f"{min(volts)/1000:.0f} – {max(volts)/1000:.0f} mV "
- f"({len(gpu_data)} points)")
+ print(
+ f"GPU core: {min(freqs) / 1000:.0f} – {max(freqs) / 1000:.0f} MHz, "
+ f"{min(volts) / 1000:.0f} – {max(volts) / 1000:.0f} mV "
+ f"({len(gpu_data)} points)"
+ )
if curve_info and curve_info.mem_points:
- mem_data = [(points[i][0], points[i][1]) for i in curve_info.mem_points
- if i < len(points) and points[i][0] > 0]
+ mem_data = [
+ (points[i][0], points[i][1])
+ for i in curve_info.mem_points
+ if i < len(points) and points[i][0] > 0
+ ]
if mem_data:
freqs = [f for f, v in mem_data]
volts = [v for f, v in mem_data]
- print(f"Memory: {min(freqs)/1000:.0f} – {max(freqs)/1000:.0f} MHz, "
- f"{min(volts)/1000:.0f} – {max(volts)/1000:.0f} mV "
- f"({len(mem_data)} points)")
+ print(
+ f"Memory: {min(freqs) / 1000:.0f} – {max(freqs) / 1000:.0f} MHz, "
+ f"{min(volts) / 1000:.0f} – {max(volts) / 1000:.0f} mV "
+ f"({len(mem_data)} points)"
+ )
if offsets:
- gpu_indices = set(curve_info.gpu_points) if curve_info else set(range(len(offsets)))
- gpu_offsets = [offsets[i] for i in gpu_indices
- if i < len(offsets) and offsets[i] != 0]
+ gpu_indices = (
+ set(curve_info.gpu_points) if curve_info else set(range(len(offsets)))
+ )
+ gpu_offsets = [
+ offsets[i] for i in gpu_indices if i < len(offsets) and offsets[i] != 0
+ ]
if gpu_offsets:
vals = set(gpu_offsets)
if len(vals) == 1:
- print(f"GPU offset: {next(iter(vals))/1000:+.0f} MHz "
- f"(uniform across {len(gpu_offsets)} points)")
+ print(
+ f"GPU offset: {next(iter(vals)) / 1000:+.0f} MHz "
+ f"(uniform across {len(gpu_offsets)} points)"
+ )
else:
- print(f"GPU offsets: {len(gpu_offsets)} points active "
- f"(range: {min(vals)/1000:+.0f} to {max(vals)/1000:+.0f} MHz)")
+ print(
+ f"GPU offsets: {len(gpu_offsets)} points active "
+ f"(range: {min(vals) / 1000:+.0f} to {max(vals) / 1000:+.0f} MHz)"
+ )
-def output_json(gpu_name, points, offsets, voltage,
- curve_info: Optional[CurveInfo] = None):
+def output_json(
+ gpu_name, points, offsets, voltage, curve_info: CurveInfo | None = None
+):
"""Output JSON format."""
data = {
"gpu": gpu_name,
"current_voltage_uV": voltage,
"layout": {
- "vfp_curve": {"size": VFP_SIZE, "base": VFP_BASE,
- "stride": VFP_STRIDE, "max_entries": VFP_MAX_ENTRIES},
- "clock_table": {"size": CT_SIZE, "base": CT_BASE,
- "stride": CT_STRIDE, "delta_offset": CT_DELTA_OFF,
- "max_entries": CT_MAX_ENTRIES},
+ "vfp_curve": {
+ "size": VFP_SIZE,
+ "base": VFP_BASE,
+ "stride": VFP_STRIDE,
+ "max_entries": VFP_MAX_ENTRIES,
+ },
+ "clock_table": {
+ "size": CT_SIZE,
+ "base": CT_BASE,
+ "stride": CT_STRIDE,
+ "delta_offset": CT_DELTA_OFF,
+ "max_entries": CT_MAX_ENTRIES,
+ },
},
"curve_info": {
"gpu_points": curve_info.gpu_points if curve_info else [],
@@ -958,6 +1010,7 @@ def output_json(gpu_name, points, offsets, voltage,
# Write command handler
# ═══════════════════════════════════════════════════════════════════════════
+
def cmd_write(gpu, gpu_name, args, mask, curve_info):
"""Handle write subcommand."""
delta_khz = int(args.delta * 1000)
@@ -977,21 +1030,27 @@ def cmd_write(gpu, gpu_name, args, mask, curve_info):
elif args.point is not None:
point_deltas[args.point] = delta_khz
- print(f"Target: point {args.point}, delta {args.delta:+.0f} MHz "
- f"({delta_khz:+d} kHz)")
+ print(
+ f"Target: point {args.point}, delta {args.delta:+.0f} MHz "
+ f"({delta_khz:+d} kHz)"
+ )
elif args.range:
start, end = args.range
for i in range(start, end + 1):
point_deltas[i] = delta_khz
- print(f"Target: points {start}–{end} ({len(point_deltas)} points), "
- f"delta {args.delta:+.0f} MHz")
+ print(
+ f"Target: points {start}–{end} ({len(point_deltas)} points), "
+ f"delta {args.delta:+.0f} MHz"
+ )
elif args.glob:
for i in gpu_points:
point_deltas[i] = delta_khz
- print(f"Target: all {len(point_deltas)} GPU core points, "
- f"delta {args.delta:+.0f} MHz")
+ print(
+ f"Target: all {len(point_deltas)} GPU core points, "
+ f"delta {args.delta:+.0f} MHz"
+ )
else:
print("Error: specify --point N, --range A-B, --global, or --reset")
@@ -1013,11 +1072,17 @@ def cmd_write(gpu, gpu_name, args, mask, curve_info):
changed = 0
for point in sorted(point_deltas.keys()):
new = point_deltas[point]
- old = current_offsets[point] if current_offsets and point < len(current_offsets) else 0
+ old = (
+ current_offsets[point]
+ if current_offsets and point < len(current_offsets)
+ else 0
+ )
if old != new:
changed += 1
if changed <= 20:
- print(f" Point {point:3d}: {old/1000:+8.0f} MHz → {new/1000:+8.0f} MHz")
+ print(
+ f" Point {point:3d}: {old / 1000:+8.0f} MHz → {new / 1000:+8.0f} MHz"
+ )
if changed > 20:
print(f" ... and {changed - 20} more points")
if changed == 0:
@@ -1036,8 +1101,10 @@ def cmd_write(gpu, gpu_name, args, mask, curve_info):
first_pt = min(point_deltas.keys())
entry_off = CT_BASE + first_pt * CT_STRIDE
- print(f"\nEntry for point {first_pt} (offset 0x{entry_off:04X}, "
- f"stride 0x{CT_STRIDE:02X}):")
+ print(
+ f"\nEntry for point {first_pt} (offset 0x{entry_off:04X}, "
+ f"stride 0x{CT_STRIDE:02X}):"
+ )
print(hexdump(bytes(buf), entry_off, CT_STRIDE))
return
@@ -1070,8 +1137,10 @@ def cmd_write(gpu, gpu_name, args, mask, curve_info):
actual = new_offsets[point] if point < len(new_offsets) else 0
if actual != expected:
mismatches += 1
- print(f" MISMATCH point {point}: expected {expected/1000:+.0f} MHz, "
- f"got {actual/1000:+.0f} MHz")
+ print(
+ f" MISMATCH point {point}: expected {expected / 1000:+.0f} MHz, "
+ f"got {actual / 1000:+.0f} MHz"
+ )
if mismatches == 0:
print(f"Verified: all {len(point_deltas)} points match expected values.")
@@ -1083,6 +1152,7 @@ def cmd_write(gpu, gpu_name, args, mask, curve_info):
# Verify command handler
# ═══════════════════════════════════════════════════════════════════════════
+
def cmd_verify(gpu, gpu_name, args, mask, curve_info):
"""Write-verify-read cycle for a single point or range."""
delta_khz = int(args.delta * 1000)
@@ -1095,14 +1165,14 @@ def cmd_verify(gpu, gpu_name, args, mask, curve_info):
print("Error: --point or --range required for verify mode")
return
- point_deltas = {p: delta_khz for p in points}
+ point_deltas = dict.fromkeys(points, delta_khz)
err = validate_write_request(point_deltas, curve_info)
if err:
print(f"Safety check FAILED: {err}")
return
- print(f"=== Write-Verify Cycle ===")
+ print("=== Write-Verify Cycle ===")
print(f"GPU: {gpu_name}")
if curve_info:
print(f"Curve: {curve_info.describe()}")
@@ -1121,7 +1191,7 @@ def cmd_verify(gpu, gpu_name, args, mask, curve_info):
for p in points[:5]:
entry = read_clock_entry_full(before_raw, p) if before_raw else {}
off_val = before_offsets[p] if p < len(before_offsets) else 0
- print(f" Point {p:3d}: freqDelta = {off_val/1000:+8.0f} MHz")
+ print(f" Point {p:3d}: freqDelta = {off_val / 1000:+8.0f} MHz")
if entry:
print(f" All fields: {entry}")
@@ -1156,8 +1226,10 @@ def cmd_verify(gpu, gpu_name, args, mask, curve_info):
match = "OK" if actual == expected else "MISMATCH"
if actual != expected:
all_ok = False
- print(f" Point {p:3d}: expected {expected/1000:+8.0f} MHz, "
- f"got {actual/1000:+8.0f} MHz [{match}]")
+ print(
+ f" Point {p:3d}: expected {expected / 1000:+8.0f} MHz, "
+ f"got {actual / 1000:+8.0f} MHz [{match}]"
+ )
# Step 5: Check for collateral damage
print()
@@ -1169,8 +1241,10 @@ def cmd_verify(gpu, gpu_name, args, mask, curve_info):
continue
if before_offsets[i] != after_offsets[i]:
collateral += 1
- print(f" WARNING: Point {i} changed unexpectedly: "
- f"{before_offsets[i]/1000:+.0f} → {after_offsets[i]/1000:+.0f} MHz")
+ print(
+ f" WARNING: Point {i} changed unexpectedly: "
+ f"{before_offsets[i] / 1000:+.0f} → {after_offsets[i] / 1000:+.0f} MHz"
+ )
if collateral == 0:
print(" No unintended changes detected.")
@@ -1187,14 +1261,16 @@ def cmd_verify(gpu, gpu_name, args, mask, curve_info):
continue
if before_entry[key] != after_entry[key]:
field_changes += 1
- print(f" Point {p}, {key}: {before_entry[key]} → {after_entry[key]}")
+ print(
+ f" Point {p}, {key}: {before_entry[key]} → {after_entry[key]}"
+ )
if field_changes == 0:
print(" No unknown fields changed.")
# Step 7: Read voltage
voltage, _ = read_voltage(gpu)
if voltage:
- print(f"\nCurrent voltage after write: {voltage/1000:.1f} mV")
+ print(f"\nCurrent voltage after write: {voltage / 1000:.1f} mV")
# Summary
print()
@@ -1215,6 +1291,7 @@ def cmd_verify(gpu, gpu_name, args, mask, curve_info):
# Inspect command
# ═══════════════════════════════════════════════════════════════════════════
+
def cmd_inspect(gpu, gpu_name, args, mask, curve_info):
"""Show detailed field-level data for specific points."""
raw, err = read_clock_table_raw(gpu, mask)
@@ -1246,8 +1323,9 @@ def cmd_inspect(gpu, gpu_name, args, mask, curve_info):
print(f"GPU: {gpu_name}")
if curve_info:
print(f"Curve: {curve_info.describe()}")
- print(f"ClockBoostTable entry detail (stride=0x{CT_STRIDE:02X}, "
- f"9 fields × 4 bytes)")
+ print(
+ f"ClockBoostTable entry detail (stride=0x{CT_STRIDE:02X}, 9 fields × 4 bytes)"
+ )
print()
for p in indices:
@@ -1267,7 +1345,7 @@ def cmd_inspect(gpu, gpu_name, args, mask, curve_info):
freq_str = ""
if vfp_points and p < len(vfp_points):
f, v = vfp_points[p]
- freq_str = f" (VFP: {f/1000:.0f} MHz @ {v/1000:.0f} mV)"
+ freq_str = f" (VFP: {f / 1000:.0f} MHz @ {v / 1000:.0f} mV)"
print(f"Point {p:3d} — buffer offset 0x{off:04X}{domain}{freq_str}")
for key, val in entry.items():
@@ -1275,8 +1353,10 @@ def cmd_inspect(gpu, gpu_name, args, mask, curve_info):
continue
marker = " ← freqDelta" if "0x14" in key else ""
if "0x14" in key:
- print(f" {key}: {val:12d} (0x{val & 0xFFFFFFFF:08X})"
- f" = {val/1000:+.0f} MHz{marker}")
+ print(
+ f" {key}: {val:12d} (0x{val & 0xFFFFFFFF:08X})"
+ f" = {val / 1000:+.0f} MHz{marker}"
+ )
else:
print(f" {key}: {val:12d} (0x{val:08X})")
print()
@@ -1286,6 +1366,7 @@ def cmd_inspect(gpu, gpu_name, args, mask, curve_info):
# Read command handler
# ═══════════════════════════════════════════════════════════════════════════
+
def cmd_read(gpu, gpu_name, args, mask, curve_info):
"""Handle read subcommand."""
if args.diag:
@@ -1309,10 +1390,13 @@ def cmd_read(gpu, gpu_name, args, mask, curve_info):
print(f"GPU: {gpu_name}")
if args.raw:
+
def fill_vfp(buf):
_fill_mask_from_boost(buf, mask)
- vfp_raw, _ = nvcall(FUNC["GetVFPCurve"], gpu, VFP_SIZE,
- ver=1, pre_fill=fill_vfp)
+
+ vfp_raw, _ = nvcall(
+ FUNC["GetVFPCurve"], gpu, VFP_SIZE, ver=1, pre_fill=fill_vfp
+ )
ct_raw, _ = read_clock_table_raw(gpu, mask)
if vfp_raw:
@@ -1348,7 +1432,8 @@ def cmd_read(gpu, gpu_name, args, mask, curve_info):
# Argument parsing
# ═══════════════════════════════════════════════════════════════════════════
-def parse_range(s: str) -> Tuple[int, int]:
+
+def parse_range(s: str) -> tuple[int, int]:
"""Parse 'A-B' into (A, B) tuple."""
parts = s.split("-")
if len(parts) != 2:
@@ -1360,7 +1445,9 @@ def parse_range(s: str) -> Tuple[int, int]:
if a > b:
raise argparse.ArgumentTypeError(f"Start > end in range: {a}-{b}")
if a < 0 or b >= CT_MAX_ENTRIES:
- raise argparse.ArgumentTypeError(f"Range {a}-{b} outside 0–{CT_MAX_ENTRIES - 1}")
+ raise argparse.ArgumentTypeError(
+ f"Range {a}-{b} outside 0–{CT_MAX_ENTRIES - 1}"
+ )
return (a, b)
@@ -1390,14 +1477,16 @@ Examples:
# --- read ---
p_read = sub.add_parser("read", help="Read V/F curve (default)")
- p_read.add_argument("--full", action="store_true",
- help="Show all points including empty slots")
- p_read.add_argument("--json", action="store_true",
- help="JSON output with domain classification")
- p_read.add_argument("--raw", action="store_true",
- help="Include hex dumps")
- p_read.add_argument("--diag", action="store_true",
- help="Probe all functions with mask comparison")
+ p_read.add_argument(
+ "--full", action="store_true", help="Show all points including empty slots"
+ )
+ p_read.add_argument(
+ "--json", action="store_true", help="JSON output with domain classification"
+ )
+ p_read.add_argument("--raw", action="store_true", help="Include hex dumps")
+ p_read.add_argument(
+ "--diag", action="store_true", help="Probe all functions with mask comparison"
+ )
# --- inspect ---
p_insp = sub.add_parser("inspect", help="Show detailed entry fields")
@@ -1409,30 +1498,42 @@ Examples:
tgt = p_write.add_mutually_exclusive_group()
tgt.add_argument("--point", type=int, help="Single point index")
tgt.add_argument("--range", type=parse_range, help="Point range A-B")
- tgt.add_argument("--global", dest="glob", action="store_true",
- help="All GPU core points")
- tgt.add_argument("--reset", action="store_true",
- help="Reset all GPU core offsets to 0")
- p_write.add_argument("--delta", type=float, default=0.0,
- help="Frequency offset in MHz (e.g. 15, -30)")
- p_write.add_argument("--dry-run", action="store_true",
- help="Preview changes without applying")
- p_write.add_argument("--force", action="store_true",
- help="Allow modifying memory points")
- p_write.add_argument("--max-delta", type=float, default=300.0,
- help="Override safety limit (MHz, default 300)")
+ tgt.add_argument(
+ "--global", dest="glob", action="store_true", help="All GPU core points"
+ )
+ tgt.add_argument(
+ "--reset", action="store_true", help="Reset all GPU core offsets to 0"
+ )
+ p_write.add_argument(
+ "--delta",
+ type=float,
+ default=0.0,
+ help="Frequency offset in MHz (e.g. 15, -30)",
+ )
+ p_write.add_argument(
+ "--dry-run", action="store_true", help="Preview changes without applying"
+ )
+ p_write.add_argument(
+ "--force", action="store_true", help="Allow modifying memory points"
+ )
+ p_write.add_argument(
+ "--max-delta",
+ type=float,
+ default=300.0,
+ help="Override safety limit (MHz, default 300)",
+ )
# --- verify ---
p_ver = sub.add_parser("verify", help="Write-verify-read cycle")
p_ver.add_argument("--point", type=int, help="Single point index")
p_ver.add_argument("--range", type=parse_range, help="Point range A-B")
- p_ver.add_argument("--delta", type=float, required=True,
- help="Frequency offset in MHz")
+ p_ver.add_argument(
+ "--delta", type=float, required=True, help="Frequency offset in MHz"
+ )
# --- snapshot ---
p_snap = sub.add_parser("snapshot", help="Save/restore ClockBoostTable")
- p_snap.add_argument("action", choices=["save", "restore"],
- help="save or restore")
+ p_snap.add_argument("action", choices=["save", "restore"], help="save or restore")
p_snap.add_argument("--file", help="Snapshot file path (for restore)")
args = parser.parse_args()
diff --git a/tests/test_rm_power.py b/tests/test_rm_power.py
new file mode 100644
index 0000000..2541665
--- /dev/null
+++ b/tests/test_rm_power.py
@@ -0,0 +1,355 @@
+"""Unit tests for the RM power-limit interface (fake RM, no hardware).
+
+Standalone (no pytest required):
+
+ python tests/test_rm_power.py
+
+Also works under pytest if available. Ports the test battery from LACT PR
+#1205 (ilya-zlobintsev/LACT): layout discovery, NVML cross-validation,
+write minimality, readback verification, and failure restoration.
+"""
+
+import os
+import sys
+
+sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
+
+from nvcurve.hal.rm_power import ( # noqa: E402
+ _CTRL_GPU_GET_ATTACHED_IDS,
+ _CTRL_GPU_GET_ID_INFO_V2,
+ _CTRL_GPU_GET_PCI_INFO,
+ _PWR_GET_CONTROL,
+ _PWR_GET_INFO,
+ _PWR_SET_CONTROL,
+ EXTENDED_LAYOUT,
+ LEGACY_LAYOUT,
+ PciLocation,
+ PowerLimitBounds,
+ RmPowerError,
+ _u32,
+ probe,
+ resolve_gpu_instance,
+ set_limit,
+)
+
+PASS = 0
+FAIL = 0
+
+
+def check(name: str, cond: bool) -> None:
+ global PASS, FAIL
+ if cond:
+ PASS += 1
+ print(f" PASS {name}")
+ else:
+ FAIL += 1
+ print(f" FAIL {name}")
+
+
+BOUNDS = PowerLimitBounds(min_mw=250_000, default_mw=300_000, max_mw=325_000)
+
+
+class FakeRm:
+ """In-memory fake of the RM power-limit client (both wire layouts)."""
+
+ def __init__(self, layout, current: int) -> None:
+ self.layout = layout
+ self.control = bytearray(layout.control_size)
+ self.control[0:8] = bytes([0xFF, 0, 0, 0, 1, 0, 0, 0])
+ self.control[layout.request_at - 4 : layout.request_at] = bytes(
+ [0x67, 0x67, 0, 0]
+ )
+ self.control[layout.request_at : layout.request_at + 4] = current.to_bytes(
+ 4, "little"
+ )
+ self.control[layout.client_at] = 0xFE
+ self.reads: list[tuple[int, int]] = []
+ self.writes: list[bytes] = []
+ self.fail_first_write = False
+ self.fail_readback = False
+ self.fail_restore = False
+
+ def query(self, cmd: int, data: bytearray) -> None:
+ if cmd == _PWR_GET_INFO:
+ self.reads.append((cmd, len(data)))
+ if len(data) != self.layout.info_size:
+ raise RmPowerError("Unsupported INFO size")
+ data[0:8] = bytes([0xFF, 0, 0, 0, 1, 0, 0, 0])
+ for index, value in enumerate([250_000, 300_000, 325_000]):
+ offset = self.layout.info_min_at + 4 * index
+ data[offset : offset + 4] = value.to_bytes(4, "little")
+ elif cmd == _PWR_GET_CONTROL:
+ self.reads.append((cmd, len(data)))
+ if len(data) != self.layout.control_size:
+ raise RmPowerError("Unsupported CONTROL size")
+ if data[self.layout.client_at] != 0xFE:
+ raise AssertionError("unexpected client selector on GET")
+ if self.fail_readback and len(self.writes) == 1:
+ raise RmPowerError("readback unavailable")
+ data[:] = self.control
+ elif cmd == _PWR_SET_CONTROL:
+ if len(data) != self.layout.control_size:
+ raise AssertionError("bad SET size")
+ if data[4:8] != (1).to_bytes(4, "little"):
+ raise AssertionError("bad SET mask")
+ if data[self.layout.client_at] != 0xFE:
+ raise AssertionError("bad SET client selector")
+ # Only the request field may differ from the current state.
+ for i, (a, b) in enumerate(zip(data, self.control, strict=True)):
+ if self.layout.request_at <= i < self.layout.request_at + 4:
+ continue
+ if a != b:
+ raise AssertionError(f"SET modified byte {i:#x}")
+ self.writes.append(bytes(data))
+ if self.fail_restore and len(self.writes) > 1:
+ raise RmPowerError("restore unavailable")
+ self.control[:] = data
+ if self.fail_first_write and len(self.writes) == 1:
+ raise RmPowerError("SET failed after modifying hardware")
+ else:
+ raise AssertionError(f"unexpected command {cmd:#x}")
+
+
+def test_detects_both_layouts_with_gets_without_a_driver_version() -> None:
+ for layout in (EXTENDED_LAYOUT, LEGACY_LAYOUT):
+ rm = FakeRm(layout, 250_000)
+ support = probe(BOUNDS, 250_000, rm.query)
+ check(
+ f"{layout.name}: detected",
+ support.bounds == BOUNDS and support.layout == layout,
+ )
+ expected = (
+ [(_PWR_GET_INFO, 0x924), (_PWR_GET_CONTROL, 0x328)]
+ if layout == EXTENDED_LAYOUT
+ else [
+ (_PWR_GET_INFO, 0x924),
+ (_PWR_GET_INFO, 0x488),
+ (_PWR_GET_CONTROL, 0x188),
+ ]
+ )
+ check(f"{layout.name}: GETs only, expected sequence", rm.reads == expected)
+ check(f"{layout.name}: no writes during discovery", rm.writes == [])
+
+
+def test_unknown_layout_and_nvml_mismatches_never_write() -> None:
+ calls: list[tuple[int, int]] = []
+
+ def failing(cmd: int, data: bytearray) -> None:
+ calls.append((cmd, len(data)))
+ raise RmPowerError("Unsupported payload")
+
+ try:
+ probe(BOUNDS, 250_000, failing)
+ check("unknown layout rejected", False)
+ except RmPowerError:
+ check("unknown layout rejected", True)
+ check(
+ "unknown layout: only GETs attempted",
+ calls == [(_PWR_GET_INFO, 0x924), (_PWR_GET_INFO, 0x488)],
+ )
+
+ for layout in (EXTENDED_LAYOUT, LEGACY_LAYOUT):
+ rm = FakeRm(layout, 250_000)
+ try:
+ probe(BOUNDS, 300_000, rm.query)
+ check(f"{layout.name}: current mismatch rejected", False)
+ except RmPowerError:
+ check(f"{layout.name}: current mismatch rejected", True)
+ other_bounds = PowerLimitBounds(
+ min_mw=BOUNDS.min_mw, default_mw=BOUNDS.default_mw, max_mw=350_000
+ )
+ try:
+ probe(other_bounds, 250_000, rm.query)
+ check(f"{layout.name}: bounds mismatch rejected", False)
+ except RmPowerError:
+ check(f"{layout.name}: bounds mismatch rejected", True)
+ check(f"{layout.name}: no writes on mismatch", rm.writes == [])
+
+
+def test_rejects_unrecognized_headers_masks_and_client_values() -> None:
+ for layout in (EXTENDED_LAYOUT, LEGACY_LAYOUT):
+ for at, value in [(0, 0), (4, 3), (layout.client_at, 0xF8)]:
+ rm = FakeRm(layout, 250_000)
+ rm.control[at] = value
+ try:
+ probe(BOUNDS, 250_000, rm.query)
+ check(f"{layout.name}: bad header/client rejected", False)
+ except RmPowerError:
+ check(f"{layout.name}: bad header/client rejected", True)
+ check(f"{layout.name}: no writes on bad header", rm.writes == [])
+ for current in (0, 0xFFFFFFFF):
+ rm = FakeRm(layout, current)
+ try:
+ probe(BOUNDS, current, rm.query)
+ check(f"{layout.name}: empty request rejected", False)
+ except RmPowerError:
+ check(f"{layout.name}: empty request rejected", True)
+ # The extended layout has additional mask words. Accepting only its low
+ # word would allow an unexpected client to be included in a later SET.
+ rm = FakeRm(EXTENDED_LAYOUT, 250_000)
+ rm.control[8] = 1
+ try:
+ probe(BOUNDS, 250_000, rm.query)
+ check("extended: nonzero mask word rejected", False)
+ except RmPowerError:
+ check("extended: nonzero mask word rejected", True)
+ check("extended: no writes on mask violation", rm.writes == [])
+
+
+def test_changes_only_fe_request_and_keeps_vbios_maximum() -> None:
+ for layout in (EXTENDED_LAYOUT, LEGACY_LAYOUT):
+ rm = FakeRm(layout, 250_000)
+ support = probe(BOUNDS, 250_000, rm.query)
+ for cap in (150_000, 30_000, 250_000):
+ set_limit(cap, support, rm.query)
+ check(
+ f"{layout.name}: set {cap} mW",
+ _u32(rm.control, layout.request_at) == cap,
+ )
+ writes = len(rm.writes)
+ for cap in (0, 29_999, 325_001, 350_000, 0xFFFFFFFF):
+ try:
+ set_limit(cap, support, rm.query)
+ check(f"{layout.name}: out-of-range {cap} rejected", False)
+ except RmPowerError:
+ check(f"{layout.name}: out-of-range {cap} rejected", True)
+ check(
+ f"{layout.name}: no writes for out-of-range caps",
+ len(rm.writes) == writes,
+ )
+
+
+def test_restores_previous_below_minimum_request_after_set_or_readback_failure() -> (
+ None
+):
+ for layout in (EXTENDED_LAYOUT, LEGACY_LAYOUT):
+ for fail_set in (False, True):
+ rm = FakeRm(layout, 100_000)
+ original = bytes(rm.control)
+ rm.fail_first_write = fail_set
+ rm.fail_readback = not fail_set
+ support = probe(BOUNDS, 100_000, rm.query)
+ try:
+ set_limit(150_000, support, rm.query)
+ check(f"{layout.name}: failure reported", False)
+ except RmPowerError:
+ check(f"{layout.name}: failure reported", True)
+ check(f"{layout.name}: restore issued", len(rm.writes) == 2)
+ check(
+ f"{layout.name}: previous request restored",
+ bytes(rm.control) == original,
+ )
+
+
+def test_reports_restore_failure_and_rejects_wrong_client_before_writing() -> None:
+ for layout in (EXTENDED_LAYOUT, LEGACY_LAYOUT):
+ rm = FakeRm(layout, 100_000)
+ rm.fail_first_write = True
+ rm.fail_restore = True
+ support = probe(BOUNDS, 100_000, rm.query)
+ try:
+ set_limit(150_000, support, rm.query)
+ check(f"{layout.name}: restore failure reported", False)
+ except RmPowerError as exc:
+ check(
+ f"{layout.name}: restore failure reported",
+ "restoration also failed" in str(exc),
+ )
+ rm = FakeRm(layout, 250_000)
+ rm.control[layout.client_at] = 0xF8
+ try:
+ set_limit(150_000, support, rm.query)
+ check(f"{layout.name}: wrong client rejected", False)
+ except RmPowerError:
+ check(f"{layout.name}: wrong client rejected", True)
+ check(f"{layout.name}: no writes for wrong client", rm.writes == [])
+
+
+# ── PCI identity → RM instance resolution ────────────────────────────────────
+
+
+def test_resolves_pci_identity_when_minor_and_rm_orders_differ() -> None:
+ # This host has Ada at minor 5/RM 4 and the 5090 at minor 4/RM 5.
+ # IDs are opaque and enumeration order must not select the device.
+ pci = PciLocation(domain=0, bus=0x0D, dev=0, func=0)
+ instances = resolve_gpu_instance(pci, lambda cmd, data: _fake_root(cmd, data))
+ check("resolves by PCI identity", instances == (5, 2))
+
+
+def _fake_root(cmd: int, data: bytearray) -> None:
+ if cmd == _CTRL_GPU_GET_ATTACHED_IDS:
+ data[0:4] = (0x2E00).to_bytes(4, "little")
+ data[4:8] = (0x0D00).to_bytes(4, "little")
+ elif cmd == _CTRL_GPU_GET_PCI_INFO:
+ gpu_id = _u32(data, 0)
+ bus = 0x2E if gpu_id == 0x2E00 else 0x0D
+ data[8:10] = bus.to_bytes(2, "little")
+ elif cmd == _CTRL_GPU_GET_ID_INFO_V2:
+ if _u32(data, 0) != 0x0D00:
+ raise AssertionError("unexpected gpu id in ID_INFO_V2")
+ data[8:12] = (5).to_bytes(4, "little")
+ data[12:16] = (2).to_bytes(4, "little")
+ else:
+ raise AssertionError(f"unexpected command {cmd:#x}")
+
+
+def test_does_not_fall_back_to_another_gpu_when_pci_is_missing() -> None:
+ pci = PciLocation(domain=1, bus=0x0D, dev=0, func=0)
+
+ def query(cmd: int, data: bytearray) -> None:
+ if cmd == _CTRL_GPU_GET_ATTACHED_IDS:
+ data[0:4] = (0x0D00).to_bytes(4, "little")
+ elif cmd == _CTRL_GPU_GET_PCI_INFO:
+ data[8:10] = (0x0D).to_bytes(2, "little")
+ else:
+ raise AssertionError("must not allocate a GPU from another PCI domain")
+
+ try:
+ resolve_gpu_instance(pci, query)
+ check("foreign PCI domain rejected", False)
+ except RmPowerError:
+ check("foreign PCI domain rejected", True)
+
+
+def test_rejects_nonzero_pci_function() -> None:
+ pci = PciLocation(domain=0, bus=0x0D, dev=0, func=1)
+ try:
+ resolve_gpu_instance(pci, lambda cmd, data: None)
+ check("nonzero function rejected", False)
+ except RmPowerError:
+ check("nonzero function rejected", True)
+
+
+def test_propagates_rm_query_failure() -> None:
+ pci = PciLocation(domain=0, bus=0x0D, dev=0, func=0)
+ try:
+ resolve_gpu_instance(
+ pci, lambda cmd, data: (_ for _ in ()).throw(RmPowerError("RM unavailable"))
+ )
+ check("RM query failure propagated", False)
+ except RmPowerError as exc:
+ check("RM query failure propagated", "RM unavailable" in str(exc))
+
+
+def main() -> int:
+ tests = [
+ test_detects_both_layouts_with_gets_without_a_driver_version,
+ test_unknown_layout_and_nvml_mismatches_never_write,
+ test_rejects_unrecognized_headers_masks_and_client_values,
+ test_changes_only_fe_request_and_keeps_vbios_maximum,
+ test_restores_previous_below_minimum_request_after_set_or_readback_failure,
+ test_reports_restore_failure_and_rejects_wrong_client_before_writing,
+ test_resolves_pci_identity_when_minor_and_rm_orders_differ,
+ test_does_not_fall_back_to_another_gpu_when_pci_is_missing,
+ test_rejects_nonzero_pci_function,
+ test_propagates_rm_query_failure,
+ ]
+ for t in tests:
+ print(f"== {t.__name__} ==")
+ t()
+ print(f"\n{PASS} passed, {FAIL} failed")
+ return 1 if FAIL else 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())