Intel-XE Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: "Poosa, Karthik" <karthik.poosa@intel.com>
To: Rodrigo Vivi <rodrigo.vivi@intel.com>
Cc: <intel-xe@lists.freedesktop.org>, <anshuman.gupta@intel.com>,
	<badal.nilawar@intel.com>, <raag.jadav@intel.com>,
	<riana.tauro@intel.com>, <sk.anirban@intel.com>,
	<mallesh.koujalagi@intel.com>, <soham.purkait@intel.com>
Subject: Re: [PATCH v6 10/13] drm/xe/hwmon: Add platform-aware VRAM thermal channel support
Date: Fri, 2 Oct 2026 00:32:45 +0530	[thread overview]
Message-ID: <b144384a-96e8-4cbc-9574-09752bccdeef@intel.com> (raw)
In-Reply-To: <arv9lBb85GqnSxdi@intel.com>


On 29-09-2026 23:34, Rodrigo Vivi wrote:
> On Fri, Sep 25, 2026 at 02:24:26AM +0530, Karthik Poosa wrote:
>> On CRI, derive the number of available VRAM temperature channels from the
>> CRI_MSU_VRAM_ENABLE register, where each enabled MSU contributes
>> VRAM_CHANNELS_PER_MSU channels.
>>   
>> Increase the maximum supported VRAM channel count to 80 and use the
>> derived count to control VRAM sensor enumeration, channel access, and
>> VRAM label allocation. Add a has_fixed_vram_channels platform flag to
>> preserve the existing fixed-channel behavior on non-CRI platforms.
>>
>> Signed-off-by: Karthik Poosa <karthik.poosa@intel.com>
>> ---
>>
>> v2:
>>   - Use the highest enabled MSU index to determine the VRAM channel bound
>>     instead of counting enabled MSUs. A sparse mask could expose channels
>>     beyond the allocated label array, leaving the label pointer unset and
>>     potentially hanging during sysfs reads.
>>   - Also return an error when no temperature label matches the requested
>>     channel.
>>
>>   drivers/gpu/drm/xe/regs/xe_pcode_regs.h |   1 +
>>   drivers/gpu/drm/xe/xe_device_types.h    |   2 +
>>   drivers/gpu/drm/xe/xe_hwmon.c           | 164 +++++++++++++++++++++---
>>   drivers/gpu/drm/xe/xe_pci.c             |   2 +
>>   drivers/gpu/drm/xe/xe_pci_types.h       |   1 +
>>   5 files changed, 154 insertions(+), 16 deletions(-)
>>
>> diff --git a/drivers/gpu/drm/xe/regs/xe_pcode_regs.h b/drivers/gpu/drm/xe/regs/xe_pcode_regs.h
>> index a969661f0c95..d9e4847dd68e 100644
>> --- a/drivers/gpu/drm/xe/regs/xe_pcode_regs.h
>> +++ b/drivers/gpu/drm/xe/regs/xe_pcode_regs.h
>> @@ -28,6 +28,7 @@
>>   #define BMG_PACKAGE_TEMPERATURE			XE_REG(0x138434)
>>   
>>   #define CRI_PACKAGE_ENERGY_STATUS		XE_REG(0x138120)
>> +#define CRI_MSU_VRAM_ENABLE			XE_REG(0x138340)
> can you please share again the ref to this?
shared offline
>
>>   #define CRI_PACKAGE_TEMPERATURE			XE_REG(0x138344)
>>   #define CRI_VRAM_TEMPERATURE			XE_REG(0x138348)
>>   #define CRI_PLATFORM_ENERGY_STATUS		XE_REG(0x138458)
>> diff --git a/drivers/gpu/drm/xe/xe_device_types.h b/drivers/gpu/drm/xe/xe_device_types.h
>> index 87554605fa41..8a1ca65f5665 100644
>> --- a/drivers/gpu/drm/xe/xe_device_types.h
>> +++ b/drivers/gpu/drm/xe/xe_device_types.h
>> @@ -184,6 +184,8 @@ struct xe_device {
>>   		u8 has_drm_ras:1;
>>   		/** @info.has_fan_control: Device supports fan control */
>>   		u8 has_fan_control:1;
>> +		/** @info.has_fixed_vram_channels: Device has fixed VRAM temperature channels */
>> +		u8 has_fixed_vram_channels:1;
>>   		/** @info.has_flat_ccs: Whether flat CCS metadata is used */
>>   		u8 has_flat_ccs:1;
>>   		/** @info.has_gsc_nvm: Device has gsc non-volatile memory */
>> diff --git a/drivers/gpu/drm/xe/xe_hwmon.c b/drivers/gpu/drm/xe/xe_hwmon.c
>> index 1b2e2fbe43c8..0422db647d11 100644
>> --- a/drivers/gpu/drm/xe/xe_hwmon.c
>> +++ b/drivers/gpu/drm/xe/xe_hwmon.c
>> @@ -39,7 +39,11 @@ enum xe_hwmon_reg_operation {
>>   	REG_READ64,
>>   };
>>   
>> -#define MAX_VRAM_CHANNELS      (16)
>> +#define MAX_VRAM_CHANNELS	80
>> +#define FIXED_VRAM_CHANNELS	16
>> +
>> +/* Each MSU enable bit maps to one VRAM subsystem of 4 channels. */
>> +#define VRAM_CHANNELS_PER_MSU	4
>>   
>>   enum xe_hwmon_channel {
>>   	CHANNEL_CARD,
>> @@ -48,6 +52,7 @@ enum xe_hwmon_channel {
>>   	CHANNEL_MCTRL,
>>   	CHANNEL_PCIE,
>>   	CHANNEL_VRAM_N,
>> +	/* Compile-time upper bound; actual channel count is hwmon->temp.vram_count */
>>   	CHANNEL_VRAM_N_MAX = CHANNEL_VRAM_N + MAX_VRAM_CHANNELS - 1,
>>   	CHANNEL_MAX,
>>   };
>> @@ -144,18 +149,22 @@ struct xe_hwmon_thermal_info {
>>   		/** @data: temperature limits in dwords */
>>   		u32 data[DIV_ROUND_UP(TEMP_LIMIT_MAX, sizeof(u32))];
>>   	};
>> -	/** @count: no of temperature sensors available for the platform */
>> +	/** @count: temperature sensors count from READ_THERMAL_CONFIG */
>>   	u8 count;
>> +	/** @vram_count: exclusive upper bound for VRAM temperature channel indices */
>> +	u8 vram_count;
>>   	/** @available: temperature sensor availability cached at registration */
>>   	bool available[CHANNEL_MAX];
>> +	/** @msu_mask: VRAM subsystem enable mask read from MMIO */
>> +	u32 msu_mask;
>>   	union {
>>   		/** @value: per-sensor raw mailbox temperature; bit7=sign, bits6:0=magnitude */
>>   		u8 value[U8_MAX + 1];
>>   		/** @dword: sensor values as dwords, u32-aligned for pcode reads */
>>   		u32 dword[DIV_ROUND_UP(U8_MAX + 1, sizeof(u32))];
>>   	};
>> -	/** @vram_label: vram label names */
>> -	char vram_label[MAX_VRAM_CHANNELS][MAX_LABEL_SIZE];
>> +	/** @vram_label: vram label names, dynamically allocated based on vram_count */
>> +	char (*vram_label)[MAX_LABEL_SIZE];
>>   };
>>   
>>   /**
>> @@ -269,6 +278,14 @@ static int xe_hwmon_pcode_rmw_power_limit(const struct xe_hwmon *hwmon, u32 attr
>>   	return ret;
>>   }
>>   
>> +static bool xe_hwmon_vram_channel_enabled(const struct xe_hwmon *hwmon, int index)
>> +{
>> +	if (hwmon->xe->info.platform == XE_CRESCENTISLAND)
>> +		return hwmon->temp.msu_mask & BIT(index / VRAM_CHANNELS_PER_MSU);
>> +
>> +	return true;
>> +}
>> +
>>   static struct xe_reg xe_hwmon_get_reg(struct xe_hwmon *hwmon, enum xe_hwmon_reg hwmon_reg,
>>   				      int channel)
>>   {
>> @@ -281,14 +298,16 @@ static struct xe_reg xe_hwmon_get_reg(struct xe_hwmon *hwmon, enum xe_hwmon_reg
>>   				return CRI_PACKAGE_TEMPERATURE;
>>   			else if (channel == CHANNEL_VRAM)
>>   				return CRI_VRAM_TEMPERATURE;
>> -			else if (in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
>> +			else if (in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count) &&
>> +				 xe_hwmon_vram_channel_enabled(hwmon, channel - CHANNEL_VRAM_N))
>>   				return CRI_VRAM_TEMPERATURE_N(channel - CHANNEL_VRAM_N);
>>   		} else if (xe->info.platform == XE_BATTLEMAGE) {
>>   			if (channel == CHANNEL_PKG)
>>   				return BMG_PACKAGE_TEMPERATURE;
>>   			else if (channel == CHANNEL_VRAM)
>>   				return BMG_VRAM_TEMPERATURE;
>> -			else if (in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
>> +			else if (in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count) &&
>> +				 xe_hwmon_vram_channel_enabled(hwmon, channel - CHANNEL_VRAM_N))
>>   				return BMG_VRAM_TEMPERATURE_N(channel - CHANNEL_VRAM_N);
>>   		} else if (xe->info.platform == XE_DG2) {
>>   			if (channel == CHANNEL_PKG)
>> @@ -798,6 +817,70 @@ static const struct hwmon_channel_info * const hwmon_info[] = {
>>   			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>>   			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>>   			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> I was wondering if there was something better we could get here...
> and LLM suggested me a reduction from 80 to 8 lines:
>
> #define TEMP_CFG_CARD HWMON_T_LABEL
> #define TEMP_CFG_PKG  (HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | \
>                         HWMON_T_LABEL | HWMON_T_MAX)
> #define TEMP_CFG_SENSOR       (HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL)
>
> /*
>   * Indexed by &enum xe_hwmon_channel and zero-terminated; hwmon derives the
>   * channel count from the array contents. Sizing on CHANNEL_MAX keeps this in
>   * lockstep with the enum.
>   */
> static const u32 hwmon_temp_config[CHANNEL_MAX + 1] = {
>        [CHANNEL_CARD]                          = TEMP_CFG_CARD,
>        [CHANNEL_PKG]                           = TEMP_CFG_PKG,
>        [CHANNEL_VRAM]                          = TEMP_CFG_SENSOR,
>        [CHANNEL_MCTRL]                         = TEMP_CFG_SENSOR,
>        [CHANNEL_PCIE]                          = TEMP_CFG_SENSOR,
>        [CHANNEL_VRAM_N ... CHANNEL_VRAM_N_MAX] = TEMP_CFG_SENSOR,
>        [CHANNEL_MAX]                           = 0,    /* terminator */
> };
>
> static const struct hwmon_channel_info hwmon_temp_info = {
>        .type = hwmon_temp,
>        .config = hwmon_temp_config,
> };
>
> static const struct hwmon_channel_info * const hwmon_info[] = {
>        &hwmon_temp_info,
>        HWMON_CHANNEL_INFO(power, ...),
>        /* unchanged */
> };
>
> can be done in a follow up though...
i think we shall make do this in separate patch series
> Everything else looks sane... I just need to double check the regs again...
shared these details offline
>
>>   			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL),
>>   	HWMON_CHANNEL_INFO(power, HWMON_P_MAX | HWMON_P_RATED_MAX | HWMON_P_LABEL | HWMON_P_CRIT |
>>   			   HWMON_P_CAP,
>> @@ -812,15 +895,37 @@ static const struct hwmon_channel_info * const hwmon_info[] = {
>>   static int xe_hwmon_pcode_read_thermal_info(struct xe_hwmon *hwmon)
>>   {
>>   	struct xe_tile *root_tile = xe_device_get_root_tile(hwmon->xe);
>> +	struct xe_mmio *mmio = xe_root_tile_mmio(hwmon->xe);
>>   	u32 config = 0;
>>   	int ret;
>>   
>> +	/*
>> +	 * Only CRI reports dynamic VRAM channel state. Fixed-channel platforms
>> +	 * set the count up front so those sensors stay visible even if the thermal
>> +	 * mailbox reads below fail.
>> +	 */
>> +	if (hwmon->xe->info.has_fixed_vram_channels)
>> +		hwmon->temp.vram_count = FIXED_VRAM_CHANNELS;
>> +
>> +	if (hwmon->xe->info.platform == XE_CRESCENTISLAND) {
>> +		hwmon->temp.msu_mask = xe_mmio_read32(mmio, CRI_MSU_VRAM_ENABLE);
>> +		hwmon->temp.vram_count = fls(hwmon->temp.msu_mask) * VRAM_CHANNELS_PER_MSU;
>> +		drm_dbg(&hwmon->xe->drm, "MSU VRAM enable mask 0x%x, VRAM channel bound %d\n",
>> +			hwmon->temp.msu_mask, hwmon->temp.vram_count);
>> +		if (hwmon->temp.vram_count > MAX_VRAM_CHANNELS) {
>> +			drm_warn(&hwmon->xe->drm,
>> +				 "VRAM channel bound %d exceeds max %d, clamping\n",
>> +				 hwmon->temp.vram_count, MAX_VRAM_CHANNELS);
>> +			hwmon->temp.vram_count = MAX_VRAM_CHANNELS;
>> +		}
>> +	}
>> +
>>   	ret = xe_pcode_read(root_tile, PCODE_MBOX(PCODE_THERMAL_INFO, READ_THERMAL_LIMITS, 0),
>>   			    &hwmon->temp.data[0], &hwmon->temp.data[1]);
>>   	if (ret)
>>   		return ret;
>>   
>> -	drm_dbg(&hwmon->xe->drm, "thermal info read val 0x%x val1 0x%x\n",
>> +	drm_dbg(&hwmon->xe->drm, "thermal info read val 0x%08x val1 0x%08x\n",
>>   		hwmon->temp.data[0], hwmon->temp.data[1]);
>>   
>>   	ret = xe_pcode_read(root_tile, PCODE_MBOX(PCODE_THERMAL_INFO, READ_THERMAL_CONFIG, 0),
>> @@ -1025,6 +1130,14 @@ static bool xe_hwmon_temp_probe(struct xe_hwmon *hwmon, int channel)
>>   		if (!xe_reg_is_valid(reg))
>>   			return false;
>>   
>> +		/*
>> +		 * On CRI, VRAM channel presence comes from the MSU enable mask
>> +		 * (applied in xe_hwmon_get_reg()); the temperature register's
>> +		 * 0xffffffff sentinel is not reliable for VRAM channels.
>> +		 */
>> +		if (hwmon->xe->info.platform == XE_CRESCENTISLAND && channel >= CHANNEL_VRAM_N)
>> +			return true;
>> +
>>   		reg_val = xe_mmio_read32(mmio, reg);
>>   		if (!mmio_temp_valid(hwmon, reg_val)) {
>>   			drm_dbg(&hwmon->xe->drm,
>> @@ -1042,20 +1155,31 @@ static bool xe_hwmon_temp_probe(struct xe_hwmon *hwmon, int channel)
>>   	}
>>   }
>>   
>> -static void xe_hwmon_init_temp_info(struct xe_hwmon *hwmon)
>> +static int xe_hwmon_init_temp_info(struct xe_hwmon *hwmon)
>>   {
>> +	struct device *dev = hwmon->xe->drm.dev;
>>   	int channel;
>>   
>>   	if (hwmon->xe->info.has_mbx_thermal_info && xe_hwmon_pcode_read_thermal_info(hwmon))
>>   		drm_warn(&hwmon->xe->drm, "Thermal mailbox not supported by card firmware\n");
>>   
>> +	/* vram_count is known only after reading thermal info above. */
>> +	if (hwmon->temp.vram_count) {
>> +		hwmon->temp.vram_label = devm_kcalloc(dev, hwmon->temp.vram_count,
>> +						      MAX_LABEL_SIZE, GFP_KERNEL);
>> +		if (!hwmon->temp.vram_label)
>> +			return -ENOMEM;
>> +	}
>> +
>>   	for (channel = 0; channel < CHANNEL_MAX; channel++) {
>>   		hwmon->temp.available[channel] = xe_hwmon_temp_probe(hwmon, channel);
>> -		if (hwmon->temp.available[channel] &&
>> -		    in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
>> +		if (hwmon->temp.available[channel] && hwmon->temp.vram_label &&
>> +		    in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count))
>>   			snprintf(hwmon->temp.vram_label[channel - CHANNEL_VRAM_N], MAX_LABEL_SIZE,
>>   				 "vram_ch_%d", channel - CHANNEL_VRAM_N);
>>   	}
>> +
>> +	return 0;
>>   }
>>   
>>   static umode_t
>> @@ -1551,8 +1675,10 @@ static int xe_hwmon_read_label(struct device *dev,
>>   			*str = "mctrl";
>>   		else if (channel == CHANNEL_PCIE)
>>   			*str = "pcie";
>> -		else if (in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
>> +		else if (in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count))
>>   			*str = hwmon->temp.vram_label[channel - CHANNEL_VRAM_N];
>> +		else
>> +			return -EOPNOTSUPP;
>>   		return 0;
>>   	case hwmon_power:
>>   	case hwmon_energy:
>> @@ -1580,7 +1706,7 @@ static const struct hwmon_chip_info hwmon_chip_info = {
>>   	.info = hwmon_info,
>>   };
>>   
>> -static void
>> +static int
>>   xe_hwmon_get_preregistration_info(struct xe_hwmon *hwmon)
>>   {
>>   	struct xe_mmio *mmio = xe_root_tile_mmio(hwmon->xe);
>> @@ -1648,7 +1774,7 @@ xe_hwmon_get_preregistration_info(struct xe_hwmon *hwmon)
>>   		if (xe_hwmon_is_visible(hwmon, hwmon_fan, hwmon_fan_input, channel))
>>   			xe_hwmon_fan_input_read(hwmon, channel, &fan_speed);
>>   
>> -	xe_hwmon_init_temp_info(hwmon);
>> +	return xe_hwmon_init_temp_info(hwmon);
>>   }
>>   
>>   int xe_hwmon_register(struct xe_device *xe)
>> @@ -1677,7 +1803,9 @@ int xe_hwmon_register(struct xe_device *xe)
>>   	hwmon->xe = xe;
>>   	xe->hwmon = hwmon;
>>   
>> -	xe_hwmon_get_preregistration_info(hwmon);
>> +	ret = xe_hwmon_get_preregistration_info(hwmon);
>> +	if (ret)
>> +		goto err_null_hwmon;
>>   
>>   	drm_dbg(&xe->drm, "Register xe hwmon interface\n");
>>   
>> @@ -1687,10 +1815,14 @@ int xe_hwmon_register(struct xe_device *xe)
>>   								hwmon_groups);
>>   	if (IS_ERR(hwmon->hwmon_dev)) {
>>   		drm_err(&xe->drm, "Failed to register xe hwmon (%pe)\n", hwmon->hwmon_dev);
>> -		xe->hwmon = NULL;
>> -		return PTR_ERR(hwmon->hwmon_dev);
>> +		ret = PTR_ERR(hwmon->hwmon_dev);
>> +		goto err_null_hwmon;
>>   	}
>>   
>>   	return 0;
>> +
>> +err_null_hwmon:
>> +	xe->hwmon = NULL;
>> +	return ret;
>>   }
>>   MODULE_IMPORT_NS("INTEL_PMT_TELEMETRY");
>> diff --git a/drivers/gpu/drm/xe/xe_pci.c b/drivers/gpu/drm/xe/xe_pci.c
>> index e656fc012902..fa32436ecf6f 100644
>> --- a/drivers/gpu/drm/xe/xe_pci.c
>> +++ b/drivers/gpu/drm/xe/xe_pci.c
>> @@ -416,6 +416,7 @@ static const struct xe_device_desc bmg_desc = {
>>   	.dma_mask_size = 46,
>>   	.has_display = true,
>>   	.has_fan_control = true,
>> +	.has_fixed_vram_channels = true,
>>   	.has_flat_ccs = 1,
>>   	.has_mbx_power_limits = true,
>>   	.has_mbx_thermal_info = true,
>> @@ -794,6 +795,7 @@ static int xe_info_init_early(struct xe_device *xe,
>>   	xe->info.has_cached_pt = desc->has_cached_pt;
>>   	xe->info.has_drm_ras = desc->has_drm_ras;
>>   	xe->info.has_fan_control = desc->has_fan_control;
>> +	xe->info.has_fixed_vram_channels = desc->has_fixed_vram_channels;
>>   	/* runtime fusing may force flat_ccs to disabled later */
>>   	xe->info.has_flat_ccs = desc->has_flat_ccs;
>>   	xe->info.has_mbx_power_limits = desc->has_mbx_power_limits;
>> diff --git a/drivers/gpu/drm/xe/xe_pci_types.h b/drivers/gpu/drm/xe/xe_pci_types.h
>> index 0041ec5676d3..4d1048ca7abe 100644
>> --- a/drivers/gpu/drm/xe/xe_pci_types.h
>> +++ b/drivers/gpu/drm/xe/xe_pci_types.h
>> @@ -42,6 +42,7 @@ struct xe_device_desc {
>>   	u8 has_display:1;
>>   	u8 has_drm_ras:1;
>>   	u8 has_fan_control:1;
>> +	u8 has_fixed_vram_channels:1;
>>   	u8 has_flat_ccs:1;
>>   	u8 has_gsc_nvm:1;
>>   	u8 has_heci_gscfi:1;
>> -- 
>> 2.25.1
>>

  reply	other threads:[~2026-10-01 19:02 UTC|newest]

Thread overview: 36+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-24 20:54 [PATCH v6 00/13] drm/xe/hwmon: Update hwmon thermal mailbox Karthik Poosa
2026-09-24 20:54 ` [PATCH v6 01/13] drm/xe/hwmon: Handle pcode read failures of xe_hwmon_pcode_rmw_power_limit Karthik Poosa
2026-09-28  0:11   ` Rodrigo Vivi
2026-09-28 15:32     ` Poosa, Karthik
2026-09-24 20:54 ` [PATCH v6 02/13] drm/xe/hwmon: Decode mailbox temperature as sign-magnitude Karthik Poosa
2026-09-28  0:15   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 03/13] drm/xe/hwmon: Fix memory controller thermal data handling Karthik Poosa
2026-09-24 20:54 ` [PATCH v6 04/13] drm/xe/hwmon: Add helpers to validate thermal sensor readings Karthik Poosa
2026-09-28  0:19   ` Rodrigo Vivi
2026-09-28 16:08     ` Poosa, Karthik
2026-09-28 17:59       ` Rodrigo Vivi
2026-09-28 19:38   ` Rodrigo Vivi
2026-09-29 16:48     ` Poosa, Karthik
2026-09-29 18:26       ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 05/13] drm/xe/hwmon: Handle unavailable memory controller sensors Karthik Poosa
2026-09-28 16:08   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 06/13] drm/xe/hwmon: Detect unavailable PCIe thermal sensors Karthik Poosa
2026-09-28 16:10   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 07/13] drm/xe/hwmon: Consolidate temperature sensor availability checks Karthik Poosa
2026-09-28 19:39   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 08/13] drm/xe/hwmon: Cache temperature availability to reduce probe time Karthik Poosa
2026-09-28 21:17   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 09/13] drm/xe/hwmon: use CRI-specific package and VRAM temperature registers Karthik Poosa
2026-09-24 20:54 ` [PATCH v6 10/13] drm/xe/hwmon: Add platform-aware VRAM thermal channel support Karthik Poosa
2026-09-29 18:04   ` Rodrigo Vivi
2026-10-01 19:02     ` Poosa, Karthik [this message]
2026-09-24 20:54 ` [PATCH v6 11/13] drm/xe/hwmon: Decode CRI temperature registers as IEEE-754 Karthik Poosa
2026-09-29 18:24   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 12/13] drm/xe/hwmon: Update memory controller temperature offset for CRI Karthik Poosa
2026-09-29 18:27   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 13/13] drm/xe/hwmon: Update PCIE temperature sensor " Karthik Poosa
2026-09-29 18:29   ` Rodrigo Vivi
2026-09-30 17:11     ` Poosa, Karthik
2026-09-24 21:05 ` ✓ CI.KUnit: success for drm/xe/hwmon: Update hwmon thermal mailbox (rev3) Patchwork
2026-09-24 22:34 ` ✓ Xe.CI.BAT: " Patchwork
2026-09-25 11:16 ` ✗ Xe.CI.FULL: failure " Patchwork

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=b144384a-96e8-4cbc-9574-09752bccdeef@intel.com \
    --to=karthik.poosa@intel.com \
    --cc=anshuman.gupta@intel.com \
    --cc=badal.nilawar@intel.com \
    --cc=intel-xe@lists.freedesktop.org \
    --cc=mallesh.koujalagi@intel.com \
    --cc=raag.jadav@intel.com \
    --cc=riana.tauro@intel.com \
    --cc=rodrigo.vivi@intel.com \
    --cc=sk.anirban@intel.com \
    --cc=soham.purkait@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox