From: "Poosa, Karthik" <karthik.poosa@intel.com>
To: Rodrigo Vivi <rodrigo.vivi@intel.com>
Cc: <intel-xe@lists.freedesktop.org>, <anshuman.gupta@intel.com>,
<badal.nilawar@intel.com>, <raag.jadav@intel.com>,
<riana.tauro@intel.com>, <sk.anirban@intel.com>,
<mallesh.koujalagi@intel.com>, <soham.purkait@intel.com>
Subject: Re: [PATCH v6 10/13] drm/xe/hwmon: Add platform-aware VRAM thermal channel support
Date: Fri, 2 Oct 2026 00:32:45 +0530 [thread overview]
Message-ID: <b144384a-96e8-4cbc-9574-09752bccdeef@intel.com> (raw)
In-Reply-To: <arv9lBb85GqnSxdi@intel.com>
On 29-09-2026 23:34, Rodrigo Vivi wrote:
> On Fri, Sep 25, 2026 at 02:24:26AM +0530, Karthik Poosa wrote:
>> On CRI, derive the number of available VRAM temperature channels from the
>> CRI_MSU_VRAM_ENABLE register, where each enabled MSU contributes
>> VRAM_CHANNELS_PER_MSU channels.
>>
>> Increase the maximum supported VRAM channel count to 80 and use the
>> derived count to control VRAM sensor enumeration, channel access, and
>> VRAM label allocation. Add a has_fixed_vram_channels platform flag to
>> preserve the existing fixed-channel behavior on non-CRI platforms.
>>
>> Signed-off-by: Karthik Poosa <karthik.poosa@intel.com>
>> ---
>>
>> v2:
>> - Use the highest enabled MSU index to determine the VRAM channel bound
>> instead of counting enabled MSUs. A sparse mask could expose channels
>> beyond the allocated label array, leaving the label pointer unset and
>> potentially hanging during sysfs reads.
>> - Also return an error when no temperature label matches the requested
>> channel.
>>
>> drivers/gpu/drm/xe/regs/xe_pcode_regs.h | 1 +
>> drivers/gpu/drm/xe/xe_device_types.h | 2 +
>> drivers/gpu/drm/xe/xe_hwmon.c | 164 +++++++++++++++++++++---
>> drivers/gpu/drm/xe/xe_pci.c | 2 +
>> drivers/gpu/drm/xe/xe_pci_types.h | 1 +
>> 5 files changed, 154 insertions(+), 16 deletions(-)
>>
>> diff --git a/drivers/gpu/drm/xe/regs/xe_pcode_regs.h b/drivers/gpu/drm/xe/regs/xe_pcode_regs.h
>> index a969661f0c95..d9e4847dd68e 100644
>> --- a/drivers/gpu/drm/xe/regs/xe_pcode_regs.h
>> +++ b/drivers/gpu/drm/xe/regs/xe_pcode_regs.h
>> @@ -28,6 +28,7 @@
>> #define BMG_PACKAGE_TEMPERATURE XE_REG(0x138434)
>>
>> #define CRI_PACKAGE_ENERGY_STATUS XE_REG(0x138120)
>> +#define CRI_MSU_VRAM_ENABLE XE_REG(0x138340)
> can you please share again the ref to this?
shared offline
>
>> #define CRI_PACKAGE_TEMPERATURE XE_REG(0x138344)
>> #define CRI_VRAM_TEMPERATURE XE_REG(0x138348)
>> #define CRI_PLATFORM_ENERGY_STATUS XE_REG(0x138458)
>> diff --git a/drivers/gpu/drm/xe/xe_device_types.h b/drivers/gpu/drm/xe/xe_device_types.h
>> index 87554605fa41..8a1ca65f5665 100644
>> --- a/drivers/gpu/drm/xe/xe_device_types.h
>> +++ b/drivers/gpu/drm/xe/xe_device_types.h
>> @@ -184,6 +184,8 @@ struct xe_device {
>> u8 has_drm_ras:1;
>> /** @info.has_fan_control: Device supports fan control */
>> u8 has_fan_control:1;
>> + /** @info.has_fixed_vram_channels: Device has fixed VRAM temperature channels */
>> + u8 has_fixed_vram_channels:1;
>> /** @info.has_flat_ccs: Whether flat CCS metadata is used */
>> u8 has_flat_ccs:1;
>> /** @info.has_gsc_nvm: Device has gsc non-volatile memory */
>> diff --git a/drivers/gpu/drm/xe/xe_hwmon.c b/drivers/gpu/drm/xe/xe_hwmon.c
>> index 1b2e2fbe43c8..0422db647d11 100644
>> --- a/drivers/gpu/drm/xe/xe_hwmon.c
>> +++ b/drivers/gpu/drm/xe/xe_hwmon.c
>> @@ -39,7 +39,11 @@ enum xe_hwmon_reg_operation {
>> REG_READ64,
>> };
>>
>> -#define MAX_VRAM_CHANNELS (16)
>> +#define MAX_VRAM_CHANNELS 80
>> +#define FIXED_VRAM_CHANNELS 16
>> +
>> +/* Each MSU enable bit maps to one VRAM subsystem of 4 channels. */
>> +#define VRAM_CHANNELS_PER_MSU 4
>>
>> enum xe_hwmon_channel {
>> CHANNEL_CARD,
>> @@ -48,6 +52,7 @@ enum xe_hwmon_channel {
>> CHANNEL_MCTRL,
>> CHANNEL_PCIE,
>> CHANNEL_VRAM_N,
>> + /* Compile-time upper bound; actual channel count is hwmon->temp.vram_count */
>> CHANNEL_VRAM_N_MAX = CHANNEL_VRAM_N + MAX_VRAM_CHANNELS - 1,
>> CHANNEL_MAX,
>> };
>> @@ -144,18 +149,22 @@ struct xe_hwmon_thermal_info {
>> /** @data: temperature limits in dwords */
>> u32 data[DIV_ROUND_UP(TEMP_LIMIT_MAX, sizeof(u32))];
>> };
>> - /** @count: no of temperature sensors available for the platform */
>> + /** @count: temperature sensors count from READ_THERMAL_CONFIG */
>> u8 count;
>> + /** @vram_count: exclusive upper bound for VRAM temperature channel indices */
>> + u8 vram_count;
>> /** @available: temperature sensor availability cached at registration */
>> bool available[CHANNEL_MAX];
>> + /** @msu_mask: VRAM subsystem enable mask read from MMIO */
>> + u32 msu_mask;
>> union {
>> /** @value: per-sensor raw mailbox temperature; bit7=sign, bits6:0=magnitude */
>> u8 value[U8_MAX + 1];
>> /** @dword: sensor values as dwords, u32-aligned for pcode reads */
>> u32 dword[DIV_ROUND_UP(U8_MAX + 1, sizeof(u32))];
>> };
>> - /** @vram_label: vram label names */
>> - char vram_label[MAX_VRAM_CHANNELS][MAX_LABEL_SIZE];
>> + /** @vram_label: vram label names, dynamically allocated based on vram_count */
>> + char (*vram_label)[MAX_LABEL_SIZE];
>> };
>>
>> /**
>> @@ -269,6 +278,14 @@ static int xe_hwmon_pcode_rmw_power_limit(const struct xe_hwmon *hwmon, u32 attr
>> return ret;
>> }
>>
>> +static bool xe_hwmon_vram_channel_enabled(const struct xe_hwmon *hwmon, int index)
>> +{
>> + if (hwmon->xe->info.platform == XE_CRESCENTISLAND)
>> + return hwmon->temp.msu_mask & BIT(index / VRAM_CHANNELS_PER_MSU);
>> +
>> + return true;
>> +}
>> +
>> static struct xe_reg xe_hwmon_get_reg(struct xe_hwmon *hwmon, enum xe_hwmon_reg hwmon_reg,
>> int channel)
>> {
>> @@ -281,14 +298,16 @@ static struct xe_reg xe_hwmon_get_reg(struct xe_hwmon *hwmon, enum xe_hwmon_reg
>> return CRI_PACKAGE_TEMPERATURE;
>> else if (channel == CHANNEL_VRAM)
>> return CRI_VRAM_TEMPERATURE;
>> - else if (in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
>> + else if (in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count) &&
>> + xe_hwmon_vram_channel_enabled(hwmon, channel - CHANNEL_VRAM_N))
>> return CRI_VRAM_TEMPERATURE_N(channel - CHANNEL_VRAM_N);
>> } else if (xe->info.platform == XE_BATTLEMAGE) {
>> if (channel == CHANNEL_PKG)
>> return BMG_PACKAGE_TEMPERATURE;
>> else if (channel == CHANNEL_VRAM)
>> return BMG_VRAM_TEMPERATURE;
>> - else if (in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
>> + else if (in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count) &&
>> + xe_hwmon_vram_channel_enabled(hwmon, channel - CHANNEL_VRAM_N))
>> return BMG_VRAM_TEMPERATURE_N(channel - CHANNEL_VRAM_N);
>> } else if (xe->info.platform == XE_DG2) {
>> if (channel == CHANNEL_PKG)
>> @@ -798,6 +817,70 @@ static const struct hwmon_channel_info * const hwmon_info[] = {
>> HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> + HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> I was wondering if there was something better we could get here...
> and LLM suggested me a reduction from 80 to 8 lines:
>
> #define TEMP_CFG_CARD HWMON_T_LABEL
> #define TEMP_CFG_PKG (HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | \
> HWMON_T_LABEL | HWMON_T_MAX)
> #define TEMP_CFG_SENSOR (HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL)
>
> /*
> * Indexed by &enum xe_hwmon_channel and zero-terminated; hwmon derives the
> * channel count from the array contents. Sizing on CHANNEL_MAX keeps this in
> * lockstep with the enum.
> */
> static const u32 hwmon_temp_config[CHANNEL_MAX + 1] = {
> [CHANNEL_CARD] = TEMP_CFG_CARD,
> [CHANNEL_PKG] = TEMP_CFG_PKG,
> [CHANNEL_VRAM] = TEMP_CFG_SENSOR,
> [CHANNEL_MCTRL] = TEMP_CFG_SENSOR,
> [CHANNEL_PCIE] = TEMP_CFG_SENSOR,
> [CHANNEL_VRAM_N ... CHANNEL_VRAM_N_MAX] = TEMP_CFG_SENSOR,
> [CHANNEL_MAX] = 0, /* terminator */
> };
>
> static const struct hwmon_channel_info hwmon_temp_info = {
> .type = hwmon_temp,
> .config = hwmon_temp_config,
> };
>
> static const struct hwmon_channel_info * const hwmon_info[] = {
> &hwmon_temp_info,
> HWMON_CHANNEL_INFO(power, ...),
> /* unchanged */
> };
>
> can be done in a follow up though...
i think we shall make do this in separate patch series
> Everything else looks sane... I just need to double check the regs again...
shared these details offline
>
>> HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL),
>> HWMON_CHANNEL_INFO(power, HWMON_P_MAX | HWMON_P_RATED_MAX | HWMON_P_LABEL | HWMON_P_CRIT |
>> HWMON_P_CAP,
>> @@ -812,15 +895,37 @@ static const struct hwmon_channel_info * const hwmon_info[] = {
>> static int xe_hwmon_pcode_read_thermal_info(struct xe_hwmon *hwmon)
>> {
>> struct xe_tile *root_tile = xe_device_get_root_tile(hwmon->xe);
>> + struct xe_mmio *mmio = xe_root_tile_mmio(hwmon->xe);
>> u32 config = 0;
>> int ret;
>>
>> + /*
>> + * Only CRI reports dynamic VRAM channel state. Fixed-channel platforms
>> + * set the count up front so those sensors stay visible even if the thermal
>> + * mailbox reads below fail.
>> + */
>> + if (hwmon->xe->info.has_fixed_vram_channels)
>> + hwmon->temp.vram_count = FIXED_VRAM_CHANNELS;
>> +
>> + if (hwmon->xe->info.platform == XE_CRESCENTISLAND) {
>> + hwmon->temp.msu_mask = xe_mmio_read32(mmio, CRI_MSU_VRAM_ENABLE);
>> + hwmon->temp.vram_count = fls(hwmon->temp.msu_mask) * VRAM_CHANNELS_PER_MSU;
>> + drm_dbg(&hwmon->xe->drm, "MSU VRAM enable mask 0x%x, VRAM channel bound %d\n",
>> + hwmon->temp.msu_mask, hwmon->temp.vram_count);
>> + if (hwmon->temp.vram_count > MAX_VRAM_CHANNELS) {
>> + drm_warn(&hwmon->xe->drm,
>> + "VRAM channel bound %d exceeds max %d, clamping\n",
>> + hwmon->temp.vram_count, MAX_VRAM_CHANNELS);
>> + hwmon->temp.vram_count = MAX_VRAM_CHANNELS;
>> + }
>> + }
>> +
>> ret = xe_pcode_read(root_tile, PCODE_MBOX(PCODE_THERMAL_INFO, READ_THERMAL_LIMITS, 0),
>> &hwmon->temp.data[0], &hwmon->temp.data[1]);
>> if (ret)
>> return ret;
>>
>> - drm_dbg(&hwmon->xe->drm, "thermal info read val 0x%x val1 0x%x\n",
>> + drm_dbg(&hwmon->xe->drm, "thermal info read val 0x%08x val1 0x%08x\n",
>> hwmon->temp.data[0], hwmon->temp.data[1]);
>>
>> ret = xe_pcode_read(root_tile, PCODE_MBOX(PCODE_THERMAL_INFO, READ_THERMAL_CONFIG, 0),
>> @@ -1025,6 +1130,14 @@ static bool xe_hwmon_temp_probe(struct xe_hwmon *hwmon, int channel)
>> if (!xe_reg_is_valid(reg))
>> return false;
>>
>> + /*
>> + * On CRI, VRAM channel presence comes from the MSU enable mask
>> + * (applied in xe_hwmon_get_reg()); the temperature register's
>> + * 0xffffffff sentinel is not reliable for VRAM channels.
>> + */
>> + if (hwmon->xe->info.platform == XE_CRESCENTISLAND && channel >= CHANNEL_VRAM_N)
>> + return true;
>> +
>> reg_val = xe_mmio_read32(mmio, reg);
>> if (!mmio_temp_valid(hwmon, reg_val)) {
>> drm_dbg(&hwmon->xe->drm,
>> @@ -1042,20 +1155,31 @@ static bool xe_hwmon_temp_probe(struct xe_hwmon *hwmon, int channel)
>> }
>> }
>>
>> -static void xe_hwmon_init_temp_info(struct xe_hwmon *hwmon)
>> +static int xe_hwmon_init_temp_info(struct xe_hwmon *hwmon)
>> {
>> + struct device *dev = hwmon->xe->drm.dev;
>> int channel;
>>
>> if (hwmon->xe->info.has_mbx_thermal_info && xe_hwmon_pcode_read_thermal_info(hwmon))
>> drm_warn(&hwmon->xe->drm, "Thermal mailbox not supported by card firmware\n");
>>
>> + /* vram_count is known only after reading thermal info above. */
>> + if (hwmon->temp.vram_count) {
>> + hwmon->temp.vram_label = devm_kcalloc(dev, hwmon->temp.vram_count,
>> + MAX_LABEL_SIZE, GFP_KERNEL);
>> + if (!hwmon->temp.vram_label)
>> + return -ENOMEM;
>> + }
>> +
>> for (channel = 0; channel < CHANNEL_MAX; channel++) {
>> hwmon->temp.available[channel] = xe_hwmon_temp_probe(hwmon, channel);
>> - if (hwmon->temp.available[channel] &&
>> - in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
>> + if (hwmon->temp.available[channel] && hwmon->temp.vram_label &&
>> + in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count))
>> snprintf(hwmon->temp.vram_label[channel - CHANNEL_VRAM_N], MAX_LABEL_SIZE,
>> "vram_ch_%d", channel - CHANNEL_VRAM_N);
>> }
>> +
>> + return 0;
>> }
>>
>> static umode_t
>> @@ -1551,8 +1675,10 @@ static int xe_hwmon_read_label(struct device *dev,
>> *str = "mctrl";
>> else if (channel == CHANNEL_PCIE)
>> *str = "pcie";
>> - else if (in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
>> + else if (in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count))
>> *str = hwmon->temp.vram_label[channel - CHANNEL_VRAM_N];
>> + else
>> + return -EOPNOTSUPP;
>> return 0;
>> case hwmon_power:
>> case hwmon_energy:
>> @@ -1580,7 +1706,7 @@ static const struct hwmon_chip_info hwmon_chip_info = {
>> .info = hwmon_info,
>> };
>>
>> -static void
>> +static int
>> xe_hwmon_get_preregistration_info(struct xe_hwmon *hwmon)
>> {
>> struct xe_mmio *mmio = xe_root_tile_mmio(hwmon->xe);
>> @@ -1648,7 +1774,7 @@ xe_hwmon_get_preregistration_info(struct xe_hwmon *hwmon)
>> if (xe_hwmon_is_visible(hwmon, hwmon_fan, hwmon_fan_input, channel))
>> xe_hwmon_fan_input_read(hwmon, channel, &fan_speed);
>>
>> - xe_hwmon_init_temp_info(hwmon);
>> + return xe_hwmon_init_temp_info(hwmon);
>> }
>>
>> int xe_hwmon_register(struct xe_device *xe)
>> @@ -1677,7 +1803,9 @@ int xe_hwmon_register(struct xe_device *xe)
>> hwmon->xe = xe;
>> xe->hwmon = hwmon;
>>
>> - xe_hwmon_get_preregistration_info(hwmon);
>> + ret = xe_hwmon_get_preregistration_info(hwmon);
>> + if (ret)
>> + goto err_null_hwmon;
>>
>> drm_dbg(&xe->drm, "Register xe hwmon interface\n");
>>
>> @@ -1687,10 +1815,14 @@ int xe_hwmon_register(struct xe_device *xe)
>> hwmon_groups);
>> if (IS_ERR(hwmon->hwmon_dev)) {
>> drm_err(&xe->drm, "Failed to register xe hwmon (%pe)\n", hwmon->hwmon_dev);
>> - xe->hwmon = NULL;
>> - return PTR_ERR(hwmon->hwmon_dev);
>> + ret = PTR_ERR(hwmon->hwmon_dev);
>> + goto err_null_hwmon;
>> }
>>
>> return 0;
>> +
>> +err_null_hwmon:
>> + xe->hwmon = NULL;
>> + return ret;
>> }
>> MODULE_IMPORT_NS("INTEL_PMT_TELEMETRY");
>> diff --git a/drivers/gpu/drm/xe/xe_pci.c b/drivers/gpu/drm/xe/xe_pci.c
>> index e656fc012902..fa32436ecf6f 100644
>> --- a/drivers/gpu/drm/xe/xe_pci.c
>> +++ b/drivers/gpu/drm/xe/xe_pci.c
>> @@ -416,6 +416,7 @@ static const struct xe_device_desc bmg_desc = {
>> .dma_mask_size = 46,
>> .has_display = true,
>> .has_fan_control = true,
>> + .has_fixed_vram_channels = true,
>> .has_flat_ccs = 1,
>> .has_mbx_power_limits = true,
>> .has_mbx_thermal_info = true,
>> @@ -794,6 +795,7 @@ static int xe_info_init_early(struct xe_device *xe,
>> xe->info.has_cached_pt = desc->has_cached_pt;
>> xe->info.has_drm_ras = desc->has_drm_ras;
>> xe->info.has_fan_control = desc->has_fan_control;
>> + xe->info.has_fixed_vram_channels = desc->has_fixed_vram_channels;
>> /* runtime fusing may force flat_ccs to disabled later */
>> xe->info.has_flat_ccs = desc->has_flat_ccs;
>> xe->info.has_mbx_power_limits = desc->has_mbx_power_limits;
>> diff --git a/drivers/gpu/drm/xe/xe_pci_types.h b/drivers/gpu/drm/xe/xe_pci_types.h
>> index 0041ec5676d3..4d1048ca7abe 100644
>> --- a/drivers/gpu/drm/xe/xe_pci_types.h
>> +++ b/drivers/gpu/drm/xe/xe_pci_types.h
>> @@ -42,6 +42,7 @@ struct xe_device_desc {
>> u8 has_display:1;
>> u8 has_drm_ras:1;
>> u8 has_fan_control:1;
>> + u8 has_fixed_vram_channels:1;
>> u8 has_flat_ccs:1;
>> u8 has_gsc_nvm:1;
>> u8 has_heci_gscfi:1;
>> --
>> 2.25.1
>>
next prev parent reply other threads:[~2026-10-01 19:02 UTC|newest]
Thread overview: 36+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-24 20:54 [PATCH v6 00/13] drm/xe/hwmon: Update hwmon thermal mailbox Karthik Poosa
2026-09-24 20:54 ` [PATCH v6 01/13] drm/xe/hwmon: Handle pcode read failures of xe_hwmon_pcode_rmw_power_limit Karthik Poosa
2026-09-28 0:11 ` Rodrigo Vivi
2026-09-28 15:32 ` Poosa, Karthik
2026-09-24 20:54 ` [PATCH v6 02/13] drm/xe/hwmon: Decode mailbox temperature as sign-magnitude Karthik Poosa
2026-09-28 0:15 ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 03/13] drm/xe/hwmon: Fix memory controller thermal data handling Karthik Poosa
2026-09-24 20:54 ` [PATCH v6 04/13] drm/xe/hwmon: Add helpers to validate thermal sensor readings Karthik Poosa
2026-09-28 0:19 ` Rodrigo Vivi
2026-09-28 16:08 ` Poosa, Karthik
2026-09-28 17:59 ` Rodrigo Vivi
2026-09-28 19:38 ` Rodrigo Vivi
2026-09-29 16:48 ` Poosa, Karthik
2026-09-29 18:26 ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 05/13] drm/xe/hwmon: Handle unavailable memory controller sensors Karthik Poosa
2026-09-28 16:08 ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 06/13] drm/xe/hwmon: Detect unavailable PCIe thermal sensors Karthik Poosa
2026-09-28 16:10 ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 07/13] drm/xe/hwmon: Consolidate temperature sensor availability checks Karthik Poosa
2026-09-28 19:39 ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 08/13] drm/xe/hwmon: Cache temperature availability to reduce probe time Karthik Poosa
2026-09-28 21:17 ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 09/13] drm/xe/hwmon: use CRI-specific package and VRAM temperature registers Karthik Poosa
2026-09-24 20:54 ` [PATCH v6 10/13] drm/xe/hwmon: Add platform-aware VRAM thermal channel support Karthik Poosa
2026-09-29 18:04 ` Rodrigo Vivi
2026-10-01 19:02 ` Poosa, Karthik [this message]
2026-09-24 20:54 ` [PATCH v6 11/13] drm/xe/hwmon: Decode CRI temperature registers as IEEE-754 Karthik Poosa
2026-09-29 18:24 ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 12/13] drm/xe/hwmon: Update memory controller temperature offset for CRI Karthik Poosa
2026-09-29 18:27 ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 13/13] drm/xe/hwmon: Update PCIE temperature sensor " Karthik Poosa
2026-09-29 18:29 ` Rodrigo Vivi
2026-09-30 17:11 ` Poosa, Karthik
2026-09-24 21:05 ` ✓ CI.KUnit: success for drm/xe/hwmon: Update hwmon thermal mailbox (rev3) Patchwork
2026-09-24 22:34 ` ✓ Xe.CI.BAT: " Patchwork
2026-09-25 11:16 ` ✗ Xe.CI.FULL: failure " Patchwork
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=b144384a-96e8-4cbc-9574-09752bccdeef@intel.com \
--to=karthik.poosa@intel.com \
--cc=anshuman.gupta@intel.com \
--cc=badal.nilawar@intel.com \
--cc=intel-xe@lists.freedesktop.org \
--cc=mallesh.koujalagi@intel.com \
--cc=raag.jadav@intel.com \
--cc=riana.tauro@intel.com \
--cc=rodrigo.vivi@intel.com \
--cc=sk.anirban@intel.com \
--cc=soham.purkait@intel.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox