diff --git a/lact-daemon/src/server/gpu_controller/nvidia.rs b/lact-daemon/src/server/gpu_controller/nvidia.rs index 3ede2491..7a4b7ff5 100644 --- a/lact-daemon/src/server/gpu_controller/nvidia.rs +++ b/lact-daemon/src/server/gpu_controller/nvidia.rs @@ -821,7 +821,7 @@ impl GpuController for NvidiaGpuController { if let Some(mask) = self.nvapi_thermals_mask && let Ok(thermals) = nvapi.get_thermals(*handle, mask) { - if let Some(hotspot) = thermals.hotspot(arch.as_ref()) { + if let Some(hotspot) = nvapi.read_hotspot(&thermals, *handle, arch.as_ref()) { temps.insert( "GPU Hotspot".to_owned(), TemperatureEntry { diff --git a/lact-daemon/src/server/gpu_controller/nvidia/nvapi.rs b/lact-daemon/src/server/gpu_controller/nvidia/nvapi.rs index a337f401..c0f4ba62 100644 --- a/lact-daemon/src/server/gpu_controller/nvidia/nvapi.rs +++ b/lact-daemon/src/server/gpu_controller/nvidia/nvapi.rs @@ -7,12 +7,12 @@ use crate::bindings::nvidia::{ NVAPI_MAX_PHYSICAL_GPUS, NVAPI_SHORT_STRING_MAX, NvAPI_Status, NvPhysicalGpuHandle, NvS32, - NvU8, NvU32, + NvU8, NvU16, NvU32, NvU64, }; use anyhow::{Context, bail}; use nvml_wrapper::enums::device::DeviceArchitecture; use std::{ - ffi::{CStr, c_char}, + ffi::{CStr, c_char, c_uint}, mem::{self, transmute}, ptr, }; @@ -32,6 +32,9 @@ const QUERY_NVAPI_GPU_CLOCK_CLIENT_CLK_VF_POINTS_GET_STATUS: u32 = 0x21537ad4; const QUERY_NVAPI_GPU_CLOCK_CLIENT_CLK_VF_POINTS_GET_INFO: u32 = 0x507b4b59; const QUERY_NVAPI_GPU_CLOCK_CLIENT_CLK_VF_POINTS_SET_CONTROL: u32 = 0x733e009; const QUERY_NVAPI_GPU_CLOCK_CLIENT_CLK_VF_POINTS_GET_CONTROL: u32 = 0x23f1b133; +const QUERY_NVAPI_GPU_REGISTER_OP: u32 = 0x2eb3c140; + +const REG_OFFSET_BLACKWELL_HOTSPOT_AGGREGATED: u32 = 0xad0aa0; pub const CLOCK_CLIENT_CLK_VF_POINT_TYPE_PROG: NvU32 = 0; @@ -236,6 +239,58 @@ impl NvApi { Ok(handles.into_iter().take(count as usize).collect()) } + pub unsafe fn read_blackwell_hotspot( + &self, + handle: NvPhysicalGpuHandle, + ) -> anyhow::Result { + self.read_temp_register(handle, REG_OFFSET_BLACKWELL_HOTSPOT_AGGREGATED) + } + + unsafe fn read_temp_register( + &self, + handle: NvPhysicalGpuHandle, + offset: u32, + ) -> anyhow::Result { + let raw = self.read_u32_register(handle, offset)?; + let value = (raw & 0xFFFF) / 256; + if value == 0 || value >= u64::from(u8::MAX) { + bail!("Invalid temperature reported: {value}"); + } + Ok(value) + } + + unsafe fn read_u32_register( + &self, + handle: NvPhysicalGpuHandle, + offset: u32, + ) -> anyhow::Result { + let f = self.query_interface(QUERY_NVAPI_GPU_REGISTER_OP)?; + let f: unsafe extern "C" fn( + handle: NvPhysicalGpuHandle, + params: &mut NvGpuRegisterOpDataV1, + ) -> NvAPI_Status = transmute(f); + + let mut op = [NvGpuRegisterOp::default(); 256]; + op[0] = NvGpuRegisterOp { + offset, + flags: (REG_OP_FLAG_READ | REG_OP_FLAG_32BIT | REG_OP_FLAG_TYPE_GLOBAL) + .try_into() + .unwrap(), + ..Default::default() + }; + + let mut params = NvGpuRegisterOpDataV1 { + op_count: 1, + op, + ..Default::default() + }; + + let status = f(handle, &mut params); + self.handle_status(status)?; + + Ok(params.op[0].value) + } + unsafe fn query_interface(&self, id: u32) -> anyhow::Result<*const ()> { let query_interface = self .lib @@ -291,6 +346,24 @@ impl NvApi { Ok(()) } + + /// Gets the value from `NvApiThermals` if possible, otherwise reads from register + pub fn read_hotspot( + &self, + thermals: &NvApiThermals, + handle: NvPhysicalGpuHandle, + arch: Option<&DeviceArchitecture>, + ) -> Option { + if arch.is_some_and(|arch| arch.as_c() >= DeviceArchitecture::Blackwell.as_c()) { + unsafe { + self.read_blackwell_hotspot(handle) + .ok() + .and_then(|value| value.try_into().ok()) + } + } else { + thermals.get_value(9) + } + } } impl Drop for NvApi { @@ -319,14 +392,6 @@ impl NvApiThermals { .filter(|&value| value > 0 && value < 255) } - pub fn hotspot(&self, arch: Option<&DeviceArchitecture>) -> Option { - if arch.is_some_and(|arch| arch.as_c() >= DeviceArchitecture::Blackwell.as_c()) { - None - } else { - self.get_value(9) - } - } - pub fn vram(&self, vram_type: Option<&str>) -> Option { match vram_type { Some("GDDR7") => self.get_value(10), @@ -474,6 +539,38 @@ pub struct ClockClientClkVfPointControlProgV1 { pub freq_offset_khz: NvS32, } +const REG_OP_FLAG_READ: c_uint = 1; +const REG_OP_FLAG_32BIT: c_uint = 4; +const REG_OP_FLAG_TYPE_GLOBAL: c_uint = 16; + +#[repr(C)] +#[derive(Debug, Copy, Clone, Default)] +struct NvGpuRegisterOp { + pub flags: NvU16, + pub status: NvU16, + pub offset: NvU32, + pub write_mask: NvU64, + pub value: NvU64, +} + +#[repr(C)] +#[derive(Debug, Copy, Clone)] +struct NvGpuRegisterOpDataV1 { + pub version: NvU32, + pub op_count: NvU32, + pub op: [NvGpuRegisterOp; 256usize], +} + +impl Default for NvGpuRegisterOpDataV1 { + fn default() -> Self { + Self { + version: make_version::(1), + op_count: 0, + op: [NvGpuRegisterOp::default(); 256], + } + } +} + #[allow(clippy::cast_possible_truncation)] const fn make_version(version: usize) -> u32 { (mem::size_of::() | (version << 16)) as u32