feat: experimental NVIDIA power control via RM ioctl interface

Adds an experimental power-cap mode using the undocumented RM ioctl
interface (based on panchovix's LACT PR #1205) to set power limits
below the VBIOS minimum (down to 30 W).

- hal/rm_power.py: RM ioctl power-cap read/write/reset + runtime probe
- limits.py: power_cap_mode (nvml/ioctl) with support detection
- config.py: persist power_cap_mode per GPU
- profiles: record/apply power_cap_mode
- server.py: POST /api/limits validates ioctl support (409 on failure)
- cli.py: profile save falls back to persisted mode
- client.py: power_cap_mode in Limits
- frontend: toggle + warning with panchovix attribution (LACT #1205)
- tests: test_rm_power.py (unit) + integration coverage
- Makefile: add test_rm_power.py to make test

Also includes automated linter reformatting (prettier, ruff, shellcheck,
isort, markdownlint) that the linter would apply anyway.
This commit is contained in:
ARIA committed 2026-09-17 22:44:27 +02:00
1 parent a8462e696c
commit e148c83622
21 files changed
+1641 -271

No files matched your search

+267 -166
View File
@@ -55,20 +55,20 @@ Key findings:
See NvAPI_VF_Curve_Documentation.md for full technical details.
"""
import argparse
import ctypes
import struct
import sys
import json
import os
import struct
import sys
import time
import argparse
from datetime import datetime
from typing import Optional, List, Tuple, Set, Dict
# ═══════════════════════════════════════════════════════════════════════════
# NvAPI bootstrap
# ═══════════════════════════════════════════════════════════════════════════
def load_nvapi():
"""Load libnvidia-api.so from the NVIDIA driver."""
for name in ("libnvidia-api.so", "libnvidia-api.so.1"):
@@ -145,23 +145,20 @@ def nvcall_raw(fid: int, gpu, buf: ctypes.Array):
FUNC = {
# Bootstrap
"Initialize": 0x0150E828,
"EnumPhysicalGPUs": 0xE5AC921F,
"GetFullName": 0xCEEE8E9F,
"Initialize": 0x0150E828,
"EnumPhysicalGPUs": 0xE5AC921F,
"GetFullName": 0xCEEE8E9F,
# V/F curve (read)
"GetVFPCurve": 0x21537AD4, # ClkVfPointsGetStatus
"GetClockBoostMask": 0x507B4B59, # ClkVfPointsGetInfo
"GetClockBoostTable": 0x23F1B133, # ClkVfPointsGetControl
"GetCurrentVoltage": 0x465F9BCF, # ClientVoltRailsGetStatus
"GetVFPCurve": 0x21537AD4, # ClkVfPointsGetStatus
"GetClockBoostMask": 0x507B4B59, # ClkVfPointsGetInfo
"GetClockBoostTable": 0x23F1B133, # ClkVfPointsGetControl
"GetCurrentVoltage": 0x465F9BCF, # ClientVoltRailsGetStatus
"GetClockBoostRanges": 0x64B43A6A, # ClkDomainsGetInfo
# Additional read
"GetPerfLimits": 0xE440B867, # PerfClientLimitsGetStatus
"GetPerfLimits": 0xE440B867, # PerfClientLimitsGetStatus
"GetVoltBoostPercent": 0x9DF23CA1, # ClientVoltRailsGetControl
# Write
"SetClockBoostTable": 0x0733E009, # ClkVfPointsSetControl
"SetClockBoostTable": 0x0733E009, # ClkVfPointsSetControl
}
# ═══════════════════════════════════════════════════════════════════════════
@@ -170,33 +167,33 @@ FUNC = {
# ═══════════════════════════════════════════════════════════════════════════
# GetVFPCurve (0x21537AD4)
VFP_SIZE = 0x1C28
VFP_BASE = 0x48
VFP_STRIDE = 0x1C # 28 bytes
VFP_SIZE = 0x1C28
VFP_BASE = 0x48
VFP_STRIDE = 0x1C # 28 bytes
VFP_MAX_ENTRIES = (VFP_SIZE - VFP_BASE) // VFP_STRIDE # 255
# Get/SetClockBoostTable (0x23F1B133 / 0x0733E009)
CT_SIZE = 0x2420
CT_BASE = 0x44
CT_STRIDE = 0x24 # 36 bytes
CT_SIZE = 0x2420
CT_BASE = 0x44
CT_STRIDE = 0x24 # 36 bytes
CT_DELTA_OFF = 0x14 # freqDelta offset within entry
CT_MAX_ENTRIES = (CT_SIZE - CT_BASE) // CT_STRIDE # 255
# GetClockBoostMask (0x507B4B59)
MASK_SIZE = 0x182C
MASK_SIZE = 0x182C
# Other structs
VOLT_SIZE = 0x004C
RANGES_SIZE = 0x0928
PERF_SIZE = 0x030C
VBOOST_SIZE = 0x0028
VOLT_SIZE = 0x004C
RANGES_SIZE = 0x0928
PERF_SIZE = 0x030C
VBOOST_SIZE = 0x0028
# Mask location within VFP/CT structs
MASK_OFFSET = 0x04
MASK_BYTES = 32 # 256 bits — covers up to 256 points
MASK_OFFSET = 0x04
MASK_BYTES = 32 # 256 bits — covers up to 256 points
# Safety constants
MAX_DELTA_KHZ = 300_000 # ±300 MHz hard cap for safety
MAX_DELTA_KHZ = 300_000 # ±300 MHz hard cap for safety
SNAPSHOT_DIR = os.path.expanduser("~/.cache/nv_vfcurve")
@@ -205,6 +202,7 @@ SNAPSHOT_DIR = os.path.expanduser("~/.cache/nv_vfcurve")
# GPU initialization
# ═══════════════════════════════════════════════════════════════════════════
def init_gpu() -> tuple:
"""Initialize NvAPI, enumerate GPUs, return (handle, name)."""
init_fn = nvfunc(FUNC["Initialize"], 0)
@@ -236,18 +234,20 @@ def init_gpu() -> tuple:
# also distinguishes GPU core vs memory clock domains.
# ═══════════════════════════════════════════════════════════════════════════
class BoostMask:
"""Parsed GetClockBoostMask data.
Provides the raw mask bytes for copying into other calls, plus
parsed per-entry enabled info for filtering.
"""
def __init__(self, raw: bytes):
self.raw = raw
self.size = len(raw)
# The mask field at offset 0x04, 16 bytes — same position as in VFP/CT structs
self.mask_bytes = raw[MASK_OFFSET:MASK_OFFSET + MASK_BYTES]
self.mask_bytes = raw[MASK_OFFSET : MASK_OFFSET + MASK_BYTES]
self.entries = []
self._parse_entries()
@@ -260,7 +260,7 @@ class BoostMask:
enabled = bool(self.mask_bytes[byte_idx] & (1 << bit_idx))
self.entries.append({"index": i, "enabled": enabled})
def get_enabled_indices(self) -> List[int]:
def get_enabled_indices(self) -> list[int]:
"""Return list of point indices that are enabled in the mask."""
return [e["index"] for e in self.entries if e["enabled"]]
@@ -273,12 +273,13 @@ class BoostMask:
buf[offset + i] = self.mask_bytes[i]
def read_boost_mask(gpu) -> Tuple[Optional[BoostMask], str]:
def read_boost_mask(gpu) -> tuple[BoostMask | None, str]:
"""Read the clock boost mask — the canonical source of active point info.
Per nvapioc, this mask must be copied into VFP and ClockBoostTable calls.
Using all-0xFF works on some GPUs (Blackwell) but fails on others (Pascal).
"""
def fill(buf):
for i in range(MASK_OFFSET, MASK_OFFSET + MASK_BYTES):
buf[i] = 0xFF
@@ -294,20 +295,22 @@ def read_boost_mask(gpu) -> Tuple[Optional[BoostMask], str]:
# Point classification — GPU core vs memory
# ═══════════════════════════════════════════════════════════════════════════
class CurveInfo:
"""Holds classified point information for the GPU's V/F curve.
Combines data from GetClockBoostMask, GetVFPCurve, and GetClockBoostTable
to determine which points are GPU core and which are memory.
"""
def __init__(self):
self.gpu_points: List[int] = [] # GPU core V/F point indices
self.mem_points: List[int] = [] # Memory V/F point indices
self.total_points: int = 0 # Total populated entries
self.mask: Optional[BoostMask] = None
self.gpu_points: list[int] = [] # GPU core V/F point indices
self.mem_points: list[int] = [] # Memory V/F point indices
self.total_points: int = 0 # Total populated entries
self.mask: BoostMask | None = None
@staticmethod
def build(gpu, mask: Optional[BoostMask] = None) -> 'CurveInfo':
def build(gpu, mask: BoostMask | None = None) -> "CurveInfo":
"""Classify all points by reading CT field_00 and VFP data.
field_00 == 0: GPU core data point
@@ -340,7 +343,7 @@ class CurveInfo:
has_vfp_data = False
if vfp_points and i < len(vfp_points):
f, v = vfp_points[i]
has_vfp_data = (f > 0 or v > 0)
has_vfp_data = f > 0 or v > 0
has_ct_data = False
for j in range(9):
@@ -378,13 +381,15 @@ class CurveInfo:
# Data readers (mask-aware)
# ═══════════════════════════════════════════════════════════════════════════
def _fill_mask_from_boost(buf, mask: BoostMask):
"""Copy boost mask into buffer."""
mask.copy_mask_into(buf)
def _read_vfp_with_mask(gpu, mask: Optional[BoostMask]) -> Optional[List[Tuple[int, int]]]:
def _read_vfp_with_mask(gpu, mask: BoostMask | None) -> list[tuple[int, int]] | None:
"""Read VFP curve using the canonical boost mask."""
def fill(buf):
_fill_mask_from_boost(buf, mask)
@@ -403,8 +408,9 @@ def _read_vfp_with_mask(gpu, mask: Optional[BoostMask]) -> Optional[List[Tuple[i
return points
def _read_clock_table_raw_with_mask(gpu, mask: Optional[BoostMask]) -> Optional[bytes]:
def _read_clock_table_raw_with_mask(gpu, mask: BoostMask | None) -> bytes | None:
"""Read raw ClockBoostTable using the canonical boost mask."""
def fill(buf):
_fill_mask_from_boost(buf, mask)
@@ -412,13 +418,14 @@ def _read_clock_table_raw_with_mask(gpu, mask: Optional[BoostMask]) -> Optional[
return d if d else None
def read_vfp_curve(gpu, mask: Optional[BoostMask] = None,
curve_info: Optional[CurveInfo] = None
) -> Tuple[Optional[List[Tuple[int, int]]], str]:
def read_vfp_curve(
gpu, mask: BoostMask | None = None, curve_info: CurveInfo | None = None
) -> tuple[list[tuple[int, int]] | None, str]:
"""Read V/F curve (frequency + voltage pairs).
Returns up to 255 entries. Use curve_info to determine which are GPU/mem.
"""
def fill(buf):
_fill_mask_from_boost(buf, mask)
@@ -442,18 +449,20 @@ def read_vfp_curve(gpu, mask: Optional[BoostMask] = None,
return points, "OK"
def read_clock_table_raw(gpu, mask: Optional[BoostMask] = None
) -> Tuple[Optional[bytes], str]:
def read_clock_table_raw(
gpu, mask: BoostMask | None = None
) -> tuple[bytes | None, str]:
"""Read the raw ClockBoostTable buffer."""
def fill(buf):
_fill_mask_from_boost(buf, mask)
return nvcall(FUNC["GetClockBoostTable"], gpu, CT_SIZE, ver=1, pre_fill=fill)
def read_clock_offsets(gpu, mask: Optional[BoostMask] = None,
curve_info: Optional[CurveInfo] = None
) -> Tuple[Optional[List[int]], str]:
def read_clock_offsets(
gpu, mask: BoostMask | None = None, curve_info: CurveInfo | None = None
) -> tuple[list[int] | None, str]:
"""Read per-point frequency offsets from the ClockBoostTable."""
d, err = read_clock_table_raw(gpu, mask)
if not d:
@@ -482,14 +491,18 @@ def read_clock_entry_full(data: bytes, point: int) -> dict:
for j in range(9):
off = base + j * 4
if j == 5:
fields[f"field_{j:02d}_0x{j*4:02X}"] = struct.unpack_from("<i", data, off)[0]
fields[f"field_{j:02d}_0x{j * 4:02X}"] = struct.unpack_from(
"<i", data, off
)[0]
else:
fields[f"field_{j:02d}_0x{j*4:02X}"] = struct.unpack_from("<I", data, off)[0]
fields[f"field_{j:02d}_0x{j * 4:02X}"] = struct.unpack_from(
"<I", data, off
)[0]
fields["freqDelta_kHz"] = fields["field_05_0x14"]
return fields
def read_voltage(gpu) -> Tuple[Optional[int], str]:
def read_voltage(gpu) -> tuple[int | None, str]:
"""Read current GPU core voltage in µV."""
d, err = nvcall(FUNC["GetCurrentVoltage"], gpu, VOLT_SIZE, ver=1)
if not d:
@@ -497,7 +510,7 @@ def read_voltage(gpu) -> Tuple[Optional[int], str]:
return struct.unpack_from("<I", d, 0x28)[0], "OK"
def read_clock_ranges(gpu) -> Tuple[Optional[dict], str]:
def read_clock_ranges(gpu) -> tuple[dict | None, str]:
"""Read clock domain min/max offset ranges."""
d, err = nvcall(FUNC["GetClockBoostRanges"], gpu, RANGES_SIZE, ver=1)
if not d:
@@ -508,8 +521,7 @@ def read_clock_ranges(gpu) -> Tuple[Optional[dict], str]:
base = 0x08 + i * 0x48
if base + 0x48 > len(d):
break
words = [struct.unpack_from("<i", d, base + j)[0]
for j in range(0, 0x48, 4)]
words = [struct.unpack_from("<i", d, base + j)[0] for j in range(0, 0x48, 4)]
domains.append(words)
return {"num_domains": num, "domains": domains}, "OK"
@@ -518,14 +530,17 @@ def read_clock_ranges(gpu) -> Tuple[Optional[dict], str]:
# Mask bit helpers
# ═══════════════════════════════════════════════════════════════════════════
def set_mask_bit(buf, point: int, offset=MASK_OFFSET):
"""Set a single bit in the mask field."""
byte_idx = offset + (point // 8)
bit_idx = point % 8
buf[byte_idx] = int.from_bytes(buf[byte_idx:byte_idx+1], 'little') | (1 << bit_idx)
buf[byte_idx] = int.from_bytes(buf[byte_idx : byte_idx + 1], "little") | (
1 << bit_idx
)
def set_mask_bits(buf, points: Set[int], offset=MASK_OFFSET):
def set_mask_bits(buf, points: set[int], offset=MASK_OFFSET):
"""Set mask bits for a set of points."""
for p in points:
set_mask_bit(buf, p, offset)
@@ -535,11 +550,12 @@ def set_mask_bits(buf, points: Set[int], offset=MASK_OFFSET):
# Write operations
# ═══════════════════════════════════════════════════════════════════════════
def build_write_buffer(
gpu,
point_deltas: dict,
mask: Optional[BoostMask] = None,
) -> Tuple[Optional[ctypes.Array], str]:
mask: BoostMask | None = None,
) -> tuple[ctypes.Array | None, str]:
"""Build a SetClockBoostTable buffer with specified per-point deltas.
Strategy: read the current ClockBoostTable (using canonical mask),
@@ -576,9 +592,9 @@ def build_write_buffer(
def write_clock_offsets(
gpu,
point_deltas: dict,
mask: Optional[BoostMask] = None,
mask: BoostMask | None = None,
dry_run: bool = False,
) -> Tuple[int, str]:
) -> tuple[int, str]:
"""Write per-point frequency offsets via SetClockBoostTable."""
buf, err = build_write_buffer(gpu, point_deltas, mask)
if buf is None:
@@ -595,9 +611,10 @@ def write_clock_offsets(
# Safety checks
# ═══════════════════════════════════════════════════════════════════════════
def validate_write_request(point_deltas: dict,
curve_info: Optional[CurveInfo] = None
) -> Optional[str]:
def validate_write_request(
point_deltas: dict, curve_info: CurveInfo | None = None
) -> str | None:
"""Return an error message if the write request is unsafe, else None."""
mem_points = set()
if curve_info:
@@ -608,14 +625,18 @@ def validate_write_request(point_deltas: dict,
return f"Point {point} out of range (0–{CT_MAX_ENTRIES - 1})"
if point in mem_points:
return (f"Point {point} is a memory clock entry. "
"Memory offsets use a different mechanism (NVML). "
"Use --force if you really mean it.")
return (
f"Point {point} is a memory clock entry. "
"Memory offsets use a different mechanism (NVML). "
"Use --force if you really mean it."
)
if abs(delta_khz) > MAX_DELTA_KHZ:
return (f"Delta {delta_khz/1000:+.0f} MHz for point {point} exceeds "
f"safety limit of ±{MAX_DELTA_KHZ/1000:.0f} MHz. "
"Use --max-delta to raise the limit if needed.")
return (
f"Delta {delta_khz / 1000:+.0f} MHz for point {point} exceeds "
f"safety limit of ±{MAX_DELTA_KHZ / 1000:.0f} MHz. "
"Use --max-delta to raise the limit if needed."
)
return None
@@ -624,11 +645,12 @@ def validate_write_request(point_deltas: dict,
# Hex dump utility
# ═══════════════════════════════════════════════════════════════════════════
def hexdump(data: bytes, start: int, length: int, cols: int = 16) -> str:
lines = []
end = min(start + length, len(data))
for off in range(start, end, cols):
chunk = data[off:off + cols]
chunk = data[off : off + cols]
hx = " ".join(f"{b:02x}" for b in chunk)
asc = "".join(chr(b) if 32 <= b < 127 else "." for b in chunk)
lines.append(f" {off:04x}: {hx:<{cols * 3}} {asc}")
@@ -639,7 +661,8 @@ def hexdump(data: bytes, start: int, length: int, cols: int = 16) -> str:
# Snapshot save/restore
# ═══════════════════════════════════════════════════════════════════════════
def snapshot_save(gpu, gpu_name: str, mask: Optional[BoostMask] = None):
def snapshot_save(gpu, gpu_name: str, mask: BoostMask | None = None):
"""Save the current ClockBoostTable to disk."""
raw, err = read_clock_table_raw(gpu, mask)
if not raw:
@@ -674,7 +697,7 @@ def snapshot_save(gpu, gpu_name: str, mask: Optional[BoostMask] = None):
with open(meta_fname, "w") as f:
json.dump(meta, f, indent=2)
print(f"Snapshot saved:")
print("Snapshot saved:")
print(f" Binary: {fname}")
print(f" Metadata: {meta_fname}")
print(f" Size: {len(raw)} bytes")
@@ -682,7 +705,7 @@ def snapshot_save(gpu, gpu_name: str, mask: Optional[BoostMask] = None):
return True
def snapshot_restore(gpu, mask: Optional[BoostMask] = None, filepath: str = None):
def snapshot_restore(gpu, mask: BoostMask | None = None, filepath: str = None):
"""Restore a ClockBoostTable snapshot from disk."""
if filepath is None:
if not os.path.isdir(SNAPSHOT_DIR):
@@ -730,7 +753,8 @@ def snapshot_restore(gpu, mask: Optional[BoostMask] = None, filepath: str = None
# Diagnostics
# ═══════════════════════════════════════════════════════════════════════════
def run_diagnostics(gpu, gpu_name, mask: Optional[BoostMask] = None):
def run_diagnostics(gpu, gpu_name, mask: BoostMask | None = None):
"""Probe all known functions and report results."""
print(f"GPU: {gpu_name}")
print()
@@ -739,14 +763,14 @@ def run_diagnostics(gpu, gpu_name, mask: Optional[BoostMask] = None):
print("=== Function probe ===")
print()
probes = [
("GetVFPCurve", FUNC["GetVFPCurve"], VFP_SIZE, 1),
("GetClockBoostMask", FUNC["GetClockBoostMask"], MASK_SIZE, 1),
("GetClockBoostTable", FUNC["GetClockBoostTable"], CT_SIZE, 1),
("GetCurrentVoltage", FUNC["GetCurrentVoltage"], VOLT_SIZE, 1),
("GetVFPCurve", FUNC["GetVFPCurve"], VFP_SIZE, 1),
("GetClockBoostMask", FUNC["GetClockBoostMask"], MASK_SIZE, 1),
("GetClockBoostTable", FUNC["GetClockBoostTable"], CT_SIZE, 1),
("GetCurrentVoltage", FUNC["GetCurrentVoltage"], VOLT_SIZE, 1),
("GetClockBoostRanges", FUNC["GetClockBoostRanges"], RANGES_SIZE, 1),
("GetPerfLimits", FUNC["GetPerfLimits"], PERF_SIZE, 2),
("GetVoltBoostPercent", FUNC["GetVoltBoostPercent"], VBOOST_SIZE, 1),
("SetClockBoostTable", FUNC["SetClockBoostTable"], CT_SIZE, 1),
("GetPerfLimits", FUNC["GetPerfLimits"], PERF_SIZE, 2),
("GetVoltBoostPercent", FUNC["GetVoltBoostPercent"], VBOOST_SIZE, 1),
("SetClockBoostTable", FUNC["SetClockBoostTable"], CT_SIZE, 1),
]
for name, fid, size, ver in probes:
ptr = QI(fid)
@@ -768,7 +792,9 @@ def run_diagnostics(gpu, gpu_name, mask: Optional[BoostMask] = None):
# Step 3: test reads with the proper mask
needs_mask_fns = {
FUNC["GetVFPCurve"], FUNC["GetClockBoostMask"], FUNC["GetClockBoostTable"]
FUNC["GetVFPCurve"],
FUNC["GetClockBoostMask"],
FUNC["GetClockBoostTable"],
}
print()
@@ -817,8 +843,10 @@ def run_diagnostics(gpu, gpu_name, mask: Optional[BoostMask] = None):
# Output formatting
# ═══════════════════════════════════════════════════════════════════════════
def print_curve(points, offsets, voltage, curve_info: Optional[CurveInfo] = None,
full=False):
def print_curve(
points, offsets, voltage, curve_info: CurveInfo | None = None, full=False
):
"""Print formatted V/F curve table."""
if voltage:
print(f"Current voltage: {voltage / 1000:.1f} mV")
@@ -845,9 +873,7 @@ def print_curve(points, offsets, voltage, curve_info: Optional[CurveInfo] = None
for i, (f, v) in enumerate(points):
if f == 0 and v == 0:
continue
if i in mem_set:
show.append(i)
elif f != prev_freq or i == len(points) - 1:
if i in mem_set or f != prev_freq or i == len(points) - 1:
show.append(i)
prev_freq = f
@@ -882,52 +908,78 @@ def print_curve(points, offsets, voltage, curve_info: Optional[CurveInfo] = None
# Summary
if curve_info and curve_info.gpu_points:
gpu_data = [(points[i][0], points[i][1]) for i in curve_info.gpu_points
if i < len(points) and points[i][0] > 0]
gpu_data = [
(points[i][0], points[i][1])
for i in curve_info.gpu_points
if i < len(points) and points[i][0] > 0
]
if gpu_data:
freqs = [f for f, v in gpu_data]
volts = [v for f, v in gpu_data]
print()
print(f"GPU core: {min(freqs)/1000:.0f} – {max(freqs)/1000:.0f} MHz, "
f"{min(volts)/1000:.0f} – {max(volts)/1000:.0f} mV "
f"({len(gpu_data)} points)")
print(
f"GPU core: {min(freqs) / 1000:.0f} – {max(freqs) / 1000:.0f} MHz, "
f"{min(volts) / 1000:.0f} – {max(volts) / 1000:.0f} mV "
f"({len(gpu_data)} points)"
)
if curve_info and curve_info.mem_points:
mem_data = [(points[i][0], points[i][1]) for i in curve_info.mem_points
if i < len(points) and points[i][0] > 0]
mem_data = [
(points[i][0], points[i][1])
for i in curve_info.mem_points
if i < len(points) and points[i][0] > 0
]
if mem_data:
freqs = [f for f, v in mem_data]
volts = [v for f, v in mem_data]
print(f"Memory: {min(freqs)/1000:.0f} – {max(freqs)/1000:.0f} MHz, "
f"{min(volts)/1000:.0f} – {max(volts)/1000:.0f} mV "
f"({len(mem_data)} points)")
print(
f"Memory: {min(freqs) / 1000:.0f} – {max(freqs) / 1000:.0f} MHz, "
f"{min(volts) / 1000:.0f} – {max(volts) / 1000:.0f} mV "
f"({len(mem_data)} points)"
)
if offsets:
gpu_indices = set(curve_info.gpu_points) if curve_info else set(range(len(offsets)))
gpu_offsets = [offsets[i] for i in gpu_indices
if i < len(offsets) and offsets[i] != 0]
gpu_indices = (
set(curve_info.gpu_points) if curve_info else set(range(len(offsets)))
)
gpu_offsets = [
offsets[i] for i in gpu_indices if i < len(offsets) and offsets[i] != 0
]
if gpu_offsets:
vals = set(gpu_offsets)
if len(vals) == 1:
print(f"GPU offset: {next(iter(vals))/1000:+.0f} MHz "
f"(uniform across {len(gpu_offsets)} points)")
print(
f"GPU offset: {next(iter(vals)) / 1000:+.0f} MHz "
f"(uniform across {len(gpu_offsets)} points)"
)
else:
print(f"GPU offsets: {len(gpu_offsets)} points active "
f"(range: {min(vals)/1000:+.0f} to {max(vals)/1000:+.0f} MHz)")
print(
f"GPU offsets: {len(gpu_offsets)} points active "
f"(range: {min(vals) / 1000:+.0f} to {max(vals) / 1000:+.0f} MHz)"
)
def output_json(gpu_name, points, offsets, voltage,
curve_info: Optional[CurveInfo] = None):
def output_json(
gpu_name, points, offsets, voltage, curve_info: CurveInfo | None = None
):
"""Output JSON format."""
data = {
"gpu": gpu_name,
"current_voltage_uV": voltage,
"layout": {
"vfp_curve": {"size": VFP_SIZE, "base": VFP_BASE,
"stride": VFP_STRIDE, "max_entries": VFP_MAX_ENTRIES},
"clock_table": {"size": CT_SIZE, "base": CT_BASE,
"stride": CT_STRIDE, "delta_offset": CT_DELTA_OFF,
"max_entries": CT_MAX_ENTRIES},
"vfp_curve": {
"size": VFP_SIZE,
"base": VFP_BASE,
"stride": VFP_STRIDE,
"max_entries": VFP_MAX_ENTRIES,
},
"clock_table": {
"size": CT_SIZE,
"base": CT_BASE,
"stride": CT_STRIDE,
"delta_offset": CT_DELTA_OFF,
"max_entries": CT_MAX_ENTRIES,
},
},
"curve_info": {
"gpu_points": curve_info.gpu_points if curve_info else [],
@@ -958,6 +1010,7 @@ def output_json(gpu_name, points, offsets, voltage,
# Write command handler
# ═══════════════════════════════════════════════════════════════════════════
def cmd_write(gpu, gpu_name, args, mask, curve_info):
"""Handle write subcommand."""
delta_khz = int(args.delta * 1000)
@@ -977,21 +1030,27 @@ def cmd_write(gpu, gpu_name, args, mask, curve_info):
elif args.point is not None:
point_deltas[args.point] = delta_khz
print(f"Target: point {args.point}, delta {args.delta:+.0f} MHz "
f"({delta_khz:+d} kHz)")
print(
f"Target: point {args.point}, delta {args.delta:+.0f} MHz "
f"({delta_khz:+d} kHz)"
)
elif args.range:
start, end = args.range
for i in range(start, end + 1):
point_deltas[i] = delta_khz
print(f"Target: points {start}–{end} ({len(point_deltas)} points), "
f"delta {args.delta:+.0f} MHz")
print(
f"Target: points {start}–{end} ({len(point_deltas)} points), "
f"delta {args.delta:+.0f} MHz"
)
elif args.glob:
for i in gpu_points:
point_deltas[i] = delta_khz
print(f"Target: all {len(point_deltas)} GPU core points, "
f"delta {args.delta:+.0f} MHz")
print(
f"Target: all {len(point_deltas)} GPU core points, "
f"delta {args.delta:+.0f} MHz"
)
else:
print("Error: specify --point N, --range A-B, --global, or --reset")
@@ -1013,11 +1072,17 @@ def cmd_write(gpu, gpu_name, args, mask, curve_info):
changed = 0
for point in sorted(point_deltas.keys()):
new = point_deltas[point]
old = current_offsets[point] if current_offsets and point < len(current_offsets) else 0
old = (
current_offsets[point]
if current_offsets and point < len(current_offsets)
else 0
)
if old != new:
changed += 1
if changed <= 20:
print(f" Point {point:3d}: {old/1000:+8.0f} MHz → {new/1000:+8.0f} MHz")
print(
f" Point {point:3d}: {old / 1000:+8.0f} MHz → {new / 1000:+8.0f} MHz"
)
if changed > 20:
print(f" ... and {changed - 20} more points")
if changed == 0:
@@ -1036,8 +1101,10 @@ def cmd_write(gpu, gpu_name, args, mask, curve_info):
first_pt = min(point_deltas.keys())
entry_off = CT_BASE + first_pt * CT_STRIDE
print(f"\nEntry for point {first_pt} (offset 0x{entry_off:04X}, "
f"stride 0x{CT_STRIDE:02X}):")
print(
f"\nEntry for point {first_pt} (offset 0x{entry_off:04X}, "
f"stride 0x{CT_STRIDE:02X}):"
)
print(hexdump(bytes(buf), entry_off, CT_STRIDE))
return
@@ -1070,8 +1137,10 @@ def cmd_write(gpu, gpu_name, args, mask, curve_info):
actual = new_offsets[point] if point < len(new_offsets) else 0
if actual != expected:
mismatches += 1
print(f" MISMATCH point {point}: expected {expected/1000:+.0f} MHz, "
f"got {actual/1000:+.0f} MHz")
print(
f" MISMATCH point {point}: expected {expected / 1000:+.0f} MHz, "
f"got {actual / 1000:+.0f} MHz"
)
if mismatches == 0:
print(f"Verified: all {len(point_deltas)} points match expected values.")
@@ -1083,6 +1152,7 @@ def cmd_write(gpu, gpu_name, args, mask, curve_info):
# Verify command handler
# ═══════════════════════════════════════════════════════════════════════════
def cmd_verify(gpu, gpu_name, args, mask, curve_info):
"""Write-verify-read cycle for a single point or range."""
delta_khz = int(args.delta * 1000)
@@ -1095,14 +1165,14 @@ def cmd_verify(gpu, gpu_name, args, mask, curve_info):
print("Error: --point or --range required for verify mode")
return
point_deltas = {p: delta_khz for p in points}
point_deltas = dict.fromkeys(points, delta_khz)
err = validate_write_request(point_deltas, curve_info)
if err:
print(f"Safety check FAILED: {err}")
return
print(f"=== Write-Verify Cycle ===")
print("=== Write-Verify Cycle ===")
print(f"GPU: {gpu_name}")
if curve_info:
print(f"Curve: {curve_info.describe()}")
@@ -1121,7 +1191,7 @@ def cmd_verify(gpu, gpu_name, args, mask, curve_info):
for p in points[:5]:
entry = read_clock_entry_full(before_raw, p) if before_raw else {}
off_val = before_offsets[p] if p < len(before_offsets) else 0
print(f" Point {p:3d}: freqDelta = {off_val/1000:+8.0f} MHz")
print(f" Point {p:3d}: freqDelta = {off_val / 1000:+8.0f} MHz")
if entry:
print(f" All fields: {entry}")
@@ -1156,8 +1226,10 @@ def cmd_verify(gpu, gpu_name, args, mask, curve_info):
match = "OK" if actual == expected else "MISMATCH"
if actual != expected:
all_ok = False
print(f" Point {p:3d}: expected {expected/1000:+8.0f} MHz, "
f"got {actual/1000:+8.0f} MHz [{match}]")
print(
f" Point {p:3d}: expected {expected / 1000:+8.0f} MHz, "
f"got {actual / 1000:+8.0f} MHz [{match}]"
)
# Step 5: Check for collateral damage
print()
@@ -1169,8 +1241,10 @@ def cmd_verify(gpu, gpu_name, args, mask, curve_info):
continue
if before_offsets[i] != after_offsets[i]:
collateral += 1
print(f" WARNING: Point {i} changed unexpectedly: "
f"{before_offsets[i]/1000:+.0f} → {after_offsets[i]/1000:+.0f} MHz")
print(
f" WARNING: Point {i} changed unexpectedly: "
f"{before_offsets[i] / 1000:+.0f} → {after_offsets[i] / 1000:+.0f} MHz"
)
if collateral == 0:
print(" No unintended changes detected.")
@@ -1187,14 +1261,16 @@ def cmd_verify(gpu, gpu_name, args, mask, curve_info):
continue
if before_entry[key] != after_entry[key]:
field_changes += 1
print(f" Point {p}, {key}: {before_entry[key]} → {after_entry[key]}")
print(
f" Point {p}, {key}: {before_entry[key]} → {after_entry[key]}"
)
if field_changes == 0:
print(" No unknown fields changed.")
# Step 7: Read voltage
voltage, _ = read_voltage(gpu)
if voltage:
print(f"\nCurrent voltage after write: {voltage/1000:.1f} mV")
print(f"\nCurrent voltage after write: {voltage / 1000:.1f} mV")
# Summary
print()
@@ -1215,6 +1291,7 @@ def cmd_verify(gpu, gpu_name, args, mask, curve_info):
# Inspect command
# ═══════════════════════════════════════════════════════════════════════════
def cmd_inspect(gpu, gpu_name, args, mask, curve_info):
"""Show detailed field-level data for specific points."""
raw, err = read_clock_table_raw(gpu, mask)
@@ -1246,8 +1323,9 @@ def cmd_inspect(gpu, gpu_name, args, mask, curve_info):
print(f"GPU: {gpu_name}")
if curve_info:
print(f"Curve: {curve_info.describe()}")
print(f"ClockBoostTable entry detail (stride=0x{CT_STRIDE:02X}, "
f"9 fields × 4 bytes)")
print(
f"ClockBoostTable entry detail (stride=0x{CT_STRIDE:02X}, 9 fields × 4 bytes)"
)
print()
for p in indices:
@@ -1267,7 +1345,7 @@ def cmd_inspect(gpu, gpu_name, args, mask, curve_info):
freq_str = ""
if vfp_points and p < len(vfp_points):
f, v = vfp_points[p]
freq_str = f" (VFP: {f/1000:.0f} MHz @ {v/1000:.0f} mV)"
freq_str = f" (VFP: {f / 1000:.0f} MHz @ {v / 1000:.0f} mV)"
print(f"Point {p:3d} — buffer offset 0x{off:04X}{domain}{freq_str}")
for key, val in entry.items():
@@ -1275,8 +1353,10 @@ def cmd_inspect(gpu, gpu_name, args, mask, curve_info):
continue
marker = " ← freqDelta" if "0x14" in key else ""
if "0x14" in key:
print(f" {key}: {val:12d} (0x{val & 0xFFFFFFFF:08X})"
f" = {val/1000:+.0f} MHz{marker}")
print(
f" {key}: {val:12d} (0x{val & 0xFFFFFFFF:08X})"
f" = {val / 1000:+.0f} MHz{marker}"
)
else:
print(f" {key}: {val:12d} (0x{val:08X})")
print()
@@ -1286,6 +1366,7 @@ def cmd_inspect(gpu, gpu_name, args, mask, curve_info):
# Read command handler
# ═══════════════════════════════════════════════════════════════════════════
def cmd_read(gpu, gpu_name, args, mask, curve_info):
"""Handle read subcommand."""
if args.diag:
@@ -1309,10 +1390,13 @@ def cmd_read(gpu, gpu_name, args, mask, curve_info):
print(f"GPU: {gpu_name}")
if args.raw:
def fill_vfp(buf):
_fill_mask_from_boost(buf, mask)
vfp_raw, _ = nvcall(FUNC["GetVFPCurve"], gpu, VFP_SIZE,
ver=1, pre_fill=fill_vfp)
vfp_raw, _ = nvcall(
FUNC["GetVFPCurve"], gpu, VFP_SIZE, ver=1, pre_fill=fill_vfp
)
ct_raw, _ = read_clock_table_raw(gpu, mask)
if vfp_raw:
@@ -1348,7 +1432,8 @@ def cmd_read(gpu, gpu_name, args, mask, curve_info):
# Argument parsing
# ═══════════════════════════════════════════════════════════════════════════
def parse_range(s: str) -> Tuple[int, int]:
def parse_range(s: str) -> tuple[int, int]:
"""Parse 'A-B' into (A, B) tuple."""
parts = s.split("-")
if len(parts) != 2:
@@ -1360,7 +1445,9 @@ def parse_range(s: str) -> Tuple[int, int]:
if a > b:
raise argparse.ArgumentTypeError(f"Start > end in range: {a}-{b}")
if a < 0 or b >= CT_MAX_ENTRIES:
raise argparse.ArgumentTypeError(f"Range {a}-{b} outside 0–{CT_MAX_ENTRIES - 1}")
raise argparse.ArgumentTypeError(
f"Range {a}-{b} outside 0–{CT_MAX_ENTRIES - 1}"
)
return (a, b)
@@ -1390,14 +1477,16 @@ Examples:
# --- read ---
p_read = sub.add_parser("read", help="Read V/F curve (default)")
p_read.add_argument("--full", action="store_true",
help="Show all points including empty slots")
p_read.add_argument("--json", action="store_true",
help="JSON output with domain classification")
p_read.add_argument("--raw", action="store_true",
help="Include hex dumps")
p_read.add_argument("--diag", action="store_true",
help="Probe all functions with mask comparison")
p_read.add_argument(
"--full", action="store_true", help="Show all points including empty slots"
)
p_read.add_argument(
"--json", action="store_true", help="JSON output with domain classification"
)
p_read.add_argument("--raw", action="store_true", help="Include hex dumps")
p_read.add_argument(
"--diag", action="store_true", help="Probe all functions with mask comparison"
)
# --- inspect ---
p_insp = sub.add_parser("inspect", help="Show detailed entry fields")
@@ -1409,30 +1498,42 @@ Examples:
tgt = p_write.add_mutually_exclusive_group()
tgt.add_argument("--point", type=int, help="Single point index")
tgt.add_argument("--range", type=parse_range, help="Point range A-B")
tgt.add_argument("--global", dest="glob", action="store_true",
help="All GPU core points")
tgt.add_argument("--reset", action="store_true",
help="Reset all GPU core offsets to 0")
p_write.add_argument("--delta", type=float, default=0.0,
help="Frequency offset in MHz (e.g. 15, -30)")
p_write.add_argument("--dry-run", action="store_true",
help="Preview changes without applying")
p_write.add_argument("--force", action="store_true",
help="Allow modifying memory points")
p_write.add_argument("--max-delta", type=float, default=300.0,
help="Override safety limit (MHz, default 300)")
tgt.add_argument(
"--global", dest="glob", action="store_true", help="All GPU core points"
)
tgt.add_argument(
"--reset", action="store_true", help="Reset all GPU core offsets to 0"
)
p_write.add_argument(
"--delta",
type=float,
default=0.0,
help="Frequency offset in MHz (e.g. 15, -30)",
)
p_write.add_argument(
"--dry-run", action="store_true", help="Preview changes without applying"
)
p_write.add_argument(
"--force", action="store_true", help="Allow modifying memory points"
)
p_write.add_argument(
"--max-delta",
type=float,
default=300.0,
help="Override safety limit (MHz, default 300)",
)
# --- verify ---
p_ver = sub.add_parser("verify", help="Write-verify-read cycle")
p_ver.add_argument("--point", type=int, help="Single point index")
p_ver.add_argument("--range", type=parse_range, help="Point range A-B")
p_ver.add_argument("--delta", type=float, required=True,
help="Frequency offset in MHz")
p_ver.add_argument(
"--delta", type=float, required=True, help="Frequency offset in MHz"
)
# --- snapshot ---
p_snap = sub.add_parser("snapshot", help="Save/restore ClockBoostTable")
p_snap.add_argument("action", choices=["save", "restore"],
help="save or restore")
p_snap.add_argument("action", choices=["save", "restore"], help="save or restore")
p_snap.add_argument("--file", help="Snapshot file path (for restore)")
args = parser.parse_args()