All of lore.kernel.org
 help / color / mirror / Atom feed
From: "Poosa, Karthik" <karthik.poosa@intel.com>
To: Rodrigo Vivi <rodrigo.vivi@intel.com>
Cc: <intel-xe@lists.freedesktop.org>, <anshuman.gupta@intel.com>,
	<badal.nilawar@intel.com>, <raag.jadav@intel.com>,
	<riana.tauro@intel.com>, <sk.anirban@intel.com>,
	<mallesh.koujalagi@intel.com>, <soham.purkait@intel.com>
Subject: Re: [PATCH v6 10/13] drm/xe/hwmon: Add platform-aware VRAM thermal channel support
Date: Fri, 2 Oct 2026 00:32:45 +0530	[thread overview]
Message-ID: <b144384a-96e8-4cbc-9574-09752bccdeef@intel.com> (raw)
In-Reply-To: <arv9lBb85GqnSxdi@intel.com>


On 29-09-2026 23:34, Rodrigo Vivi wrote:
> On Fri, Sep 25, 2026 at 02:24:26AM +0530, Karthik Poosa wrote:
>> On CRI, derive the number of available VRAM temperature channels from the
>> CRI_MSU_VRAM_ENABLE register, where each enabled MSU contributes
>> VRAM_CHANNELS_PER_MSU channels.
>>   
>> Increase the maximum supported VRAM channel count to 80 and use the
>> derived count to control VRAM sensor enumeration, channel access, and
>> VRAM label allocation. Add a has_fixed_vram_channels platform flag to
>> preserve the existing fixed-channel behavior on non-CRI platforms.
>>
>> Signed-off-by: Karthik Poosa <karthik.poosa@intel.com>
>> ---
>>
>> v2:
>>   - Use the highest enabled MSU index to determine the VRAM channel bound
>>     instead of counting enabled MSUs. A sparse mask could expose channels
>>     beyond the allocated label array, leaving the label pointer unset and
>>     potentially hanging during sysfs reads.
>>   - Also return an error when no temperature label matches the requested
>>     channel.
>>
>>   drivers/gpu/drm/xe/regs/xe_pcode_regs.h |   1 +
>>   drivers/gpu/drm/xe/xe_device_types.h    |   2 +
>>   drivers/gpu/drm/xe/xe_hwmon.c           | 164 +++++++++++++++++++++---
>>   drivers/gpu/drm/xe/xe_pci.c             |   2 +
>>   drivers/gpu/drm/xe/xe_pci_types.h       |   1 +
>>   5 files changed, 154 insertions(+), 16 deletions(-)
>>
>> diff --git a/drivers/gpu/drm/xe/regs/xe_pcode_regs.h b/drivers/gpu/drm/xe/regs/xe_pcode_regs.h
>> index a969661f0c95..d9e4847dd68e 100644
>> --- a/drivers/gpu/drm/xe/regs/xe_pcode_regs.h
>> +++ b/drivers/gpu/drm/xe/regs/xe_pcode_regs.h
>> @@ -28,6 +28,7 @@
>>   #define BMG_PACKAGE_TEMPERATURE			XE_REG(0x138434)
>>   
>>   #define CRI_PACKAGE_ENERGY_STATUS		XE_REG(0x138120)
>> +#define CRI_MSU_VRAM_ENABLE			XE_REG(0x138340)
> can you please share again the ref to this?
shared offline
>
>>   #define CRI_PACKAGE_TEMPERATURE			XE_REG(0x138344)
>>   #define CRI_VRAM_TEMPERATURE			XE_REG(0x138348)
>>   #define CRI_PLATFORM_ENERGY_STATUS		XE_REG(0x138458)
>> diff --git a/drivers/gpu/drm/xe/xe_device_types.h b/drivers/gpu/drm/xe/xe_device_types.h
>> index 87554605fa41..8a1ca65f5665 100644
>> --- a/drivers/gpu/drm/xe/xe_device_types.h
>> +++ b/drivers/gpu/drm/xe/xe_device_types.h
>> @@ -184,6 +184,8 @@ struct xe_device {
>>   		u8 has_drm_ras:1;
>>   		/** @info.has_fan_control: Device supports fan control */
>>   		u8 has_fan_control:1;
>> +		/** @info.has_fixed_vram_channels: Device has fixed VRAM temperature channels */
>> +		u8 has_fixed_vram_channels:1;
>>   		/** @info.has_flat_ccs: Whether flat CCS metadata is used */
>>   		u8 has_flat_ccs:1;
>>   		/** @info.has_gsc_nvm: Device has gsc non-volatile memory */
>> diff --git a/drivers/gpu/drm/xe/xe_hwmon.c b/drivers/gpu/drm/xe/xe_hwmon.c
>> index 1b2e2fbe43c8..0422db647d11 100644
>> --- a/drivers/gpu/drm/xe/xe_hwmon.c
>> +++ b/drivers/gpu/drm/xe/xe_hwmon.c
>> @@ -39,7 +39,11 @@ enum xe_hwmon_reg_operation {
>>   	REG_READ64,
>>   };
>>   
>> -#define MAX_VRAM_CHANNELS      (16)
>> +#define MAX_VRAM_CHANNELS	80
>> +#define FIXED_VRAM_CHANNELS	16
>> +
>> +/* Each MSU enable bit maps to one VRAM subsystem of 4 channels. */
>> +#define VRAM_CHANNELS_PER_MSU	4
>>   
>>   enum xe_hwmon_channel {
>>   	CHANNEL_CARD,
>> @@ -48,6 +52,7 @@ enum xe_hwmon_channel {
>>   	CHANNEL_MCTRL,
>>   	CHANNEL_PCIE,
>>   	CHANNEL_VRAM_N,
>> +	/* Compile-time upper bound; actual channel count is hwmon->temp.vram_count */
>>   	CHANNEL_VRAM_N_MAX = CHANNEL_VRAM_N + MAX_VRAM_CHANNELS - 1,
>>   	CHANNEL_MAX,
>>   };
>> @@ -144,18 +149,22 @@ struct xe_hwmon_thermal_info {
>>   		/** @data: temperature limits in dwords */
>>   		u32 data[DIV_ROUND_UP(TEMP_LIMIT_MAX, sizeof(u32))];
>>   	};
>> -	/** @count: no of temperature sensors available for the platform */
>> +	/** @count: temperature sensors count from READ_THERMAL_CONFIG */
>>   	u8 count;
>> +	/** @vram_count: exclusive upper bound for VRAM temperature channel indices */
>> +	u8 vram_count;
>>   	/** @available: temperature sensor availability cached at registration */
>>   	bool available[CHANNEL_MAX];
>> +	/** @msu_mask: VRAM subsystem enable mask read from MMIO */
>> +	u32 msu_mask;
>>   	union {
>>   		/** @value: per-sensor raw mailbox temperature; bit7=sign, bits6:0=magnitude */
>>   		u8 value[U8_MAX + 1];
>>   		/** @dword: sensor values as dwords, u32-aligned for pcode reads */
>>   		u32 dword[DIV_ROUND_UP(U8_MAX + 1, sizeof(u32))];
>>   	};
>> -	/** @vram_label: vram label names */
>> -	char vram_label[MAX_VRAM_CHANNELS][MAX_LABEL_SIZE];
>> +	/** @vram_label: vram label names, dynamically allocated based on vram_count */
>> +	char (*vram_label)[MAX_LABEL_SIZE];
>>   };
>>   
>>   /**
>> @@ -269,6 +278,14 @@ static int xe_hwmon_pcode_rmw_power_limit(const struct xe_hwmon *hwmon, u32 attr
>>   	return ret;
>>   }
>>   
>> +static bool xe_hwmon_vram_channel_enabled(const struct xe_hwmon *hwmon, int index)
>> +{
>> +	if (hwmon->xe->info.platform == XE_CRESCENTISLAND)
>> +		return hwmon->temp.msu_mask & BIT(index / VRAM_CHANNELS_PER_MSU);
>> +
>> +	return true;
>> +}
>> +
>>   static struct xe_reg xe_hwmon_get_reg(struct xe_hwmon *hwmon, enum xe_hwmon_reg hwmon_reg,
>>   				      int channel)
>>   {
>> @@ -281,14 +298,16 @@ static struct xe_reg xe_hwmon_get_reg(struct xe_hwmon *hwmon, enum xe_hwmon_reg
>>   				return CRI_PACKAGE_TEMPERATURE;
>>   			else if (channel == CHANNEL_VRAM)
>>   				return CRI_VRAM_TEMPERATURE;
>> -			else if (in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
>> +			else if (in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count) &&
>> +				 xe_hwmon_vram_channel_enabled(hwmon, channel - CHANNEL_VRAM_N))
>>   				return CRI_VRAM_TEMPERATURE_N(channel - CHANNEL_VRAM_N);
>>   		} else if (xe->info.platform == XE_BATTLEMAGE) {
>>   			if (channel == CHANNEL_PKG)
>>   				return BMG_PACKAGE_TEMPERATURE;
>>   			else if (channel == CHANNEL_VRAM)
>>   				return BMG_VRAM_TEMPERATURE;
>> -			else if (in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
>> +			else if (in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count) &&
>> +				 xe_hwmon_vram_channel_enabled(hwmon, channel - CHANNEL_VRAM_N))
>>   				return BMG_VRAM_TEMPERATURE_N(channel - CHANNEL_VRAM_N);
>>   		} else if (xe->info.platform == XE_DG2) {
>>   			if (channel == CHANNEL_PKG)
>> @@ -798,6 +817,70 @@ static const struct hwmon_channel_info * const hwmon_info[] = {
>>   			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>>   			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>>   			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> I was wondering if there was something better we could get here...
> and LLM suggested me a reduction from 80 to 8 lines:
>
> #define TEMP_CFG_CARD HWMON_T_LABEL
> #define TEMP_CFG_PKG  (HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | \
>                         HWMON_T_LABEL | HWMON_T_MAX)
> #define TEMP_CFG_SENSOR       (HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL)
>
> /*
>   * Indexed by &enum xe_hwmon_channel and zero-terminated; hwmon derives the
>   * channel count from the array contents. Sizing on CHANNEL_MAX keeps this in
>   * lockstep with the enum.
>   */
> static const u32 hwmon_temp_config[CHANNEL_MAX + 1] = {
>        [CHANNEL_CARD]                          = TEMP_CFG_CARD,
>        [CHANNEL_PKG]                           = TEMP_CFG_PKG,
>        [CHANNEL_VRAM]                          = TEMP_CFG_SENSOR,
>        [CHANNEL_MCTRL]                         = TEMP_CFG_SENSOR,
>        [CHANNEL_PCIE]                          = TEMP_CFG_SENSOR,
>        [CHANNEL_VRAM_N ... CHANNEL_VRAM_N_MAX] = TEMP_CFG_SENSOR,
>        [CHANNEL_MAX]                           = 0,    /* terminator */
> };
>
> static const struct hwmon_channel_info hwmon_temp_info = {
>        .type = hwmon_temp,
>        .config = hwmon_temp_config,
> };
>
> static const struct hwmon_channel_info * const hwmon_info[] = {
>        &hwmon_temp_info,
>        HWMON_CHANNEL_INFO(power, ...),
>        /* unchanged */
> };
>
> can be done in a follow up though...
i think we shall make do this in separate patch series
> Everything else looks sane... I just need to double check the regs again...
shared these details offline
>
>>   			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL),
>>   	HWMON_CHANNEL_INFO(power, HWMON_P_MAX | HWMON_P_RATED_MAX | HWMON_P_LABEL | HWMON_P_CRIT |
>>   			   HWMON_P_CAP,
>> @@ -812,15 +895,37 @@ static const struct hwmon_channel_info * const hwmon_info[] = {
>>   static int xe_hwmon_pcode_read_thermal_info(struct xe_hwmon *hwmon)
>>   {
>>   	struct xe_tile *root_tile = xe_device_get_root_tile(hwmon->xe);
>> +	struct xe_mmio *mmio = xe_root_tile_mmio(hwmon->xe);
>>   	u32 config = 0;
>>   	int ret;
>>   
>> +	/*
>> +	 * Only CRI reports dynamic VRAM channel state. Fixed-channel platforms
>> +	 * set the count up front so those sensors stay visible even if the thermal
>> +	 * mailbox reads below fail.
>> +	 */
>> +	if (hwmon->xe->info.has_fixed_vram_channels)
>> +		hwmon->temp.vram_count = FIXED_VRAM_CHANNELS;
>> +
>> +	if (hwmon->xe->info.platform == XE_CRESCENTISLAND) {
>> +		hwmon->temp.msu_mask = xe_mmio_read32(mmio, CRI_MSU_VRAM_ENABLE);
>> +		hwmon->temp.vram_count = fls(hwmon->temp.msu_mask) * VRAM_CHANNELS_PER_MSU;
>> +		drm_dbg(&hwmon->xe->drm, "MSU VRAM enable mask 0x%x, VRAM channel bound %d\n",
>> +			hwmon->temp.msu_mask, hwmon->temp.vram_count);
>> +		if (hwmon->temp.vram_count > MAX_VRAM_CHANNELS) {
>> +			drm_warn(&hwmon->xe->drm,
>> +				 "VRAM channel bound %d exceeds max %d, clamping\n",
>> +				 hwmon->temp.vram_count, MAX_VRAM_CHANNELS);
>> +			hwmon->temp.vram_count = MAX_VRAM_CHANNELS;
>> +		}
>> +	}
>> +
>>   	ret = xe_pcode_read(root_tile, PCODE_MBOX(PCODE_THERMAL_INFO, READ_THERMAL_LIMITS, 0),
>>   			    &hwmon->temp.data[0], &hwmon->temp.data[1]);
>>   	if (ret)
>>   		return ret;
>>   
>> -	drm_dbg(&hwmon->xe->drm, "thermal info read val 0x%x val1 0x%x\n",
>> +	drm_dbg(&hwmon->xe->drm, "thermal info read val 0x%08x val1 0x%08x\n",
>>   		hwmon->temp.data[0], hwmon->temp.data[1]);
>>   
>>   	ret = xe_pcode_read(root_tile, PCODE_MBOX(PCODE_THERMAL_INFO, READ_THERMAL_CONFIG, 0),
>> @@ -1025,6 +1130,14 @@ static bool xe_hwmon_temp_probe(struct xe_hwmon *hwmon, int channel)
>>   		if (!xe_reg_is_valid(reg))
>>   			return false;
>>   
>> +		/*
>> +		 * On CRI, VRAM channel presence comes from the MSU enable mask
>> +		 * (applied in xe_hwmon_get_reg()); the temperature register's
>> +		 * 0xffffffff sentinel is not reliable for VRAM channels.
>> +		 */
>> +		if (hwmon->xe->info.platform == XE_CRESCENTISLAND && channel >= CHANNEL_VRAM_N)
>> +			return true;
>> +
>>   		reg_val = xe_mmio_read32(mmio, reg);
>>   		if (!mmio_temp_valid(hwmon, reg_val)) {
>>   			drm_dbg(&hwmon->xe->drm,
>> @@ -1042,20 +1155,31 @@ static bool xe_hwmon_temp_probe(struct xe_hwmon *hwmon, int channel)
>>   	}
>>   }
>>   
>> -static void xe_hwmon_init_temp_info(struct xe_hwmon *hwmon)
>> +static int xe_hwmon_init_temp_info(struct xe_hwmon *hwmon)
>>   {
>> +	struct device *dev = hwmon->xe->drm.dev;
>>   	int channel;
>>   
>>   	if (hwmon->xe->info.has_mbx_thermal_info && xe_hwmon_pcode_read_thermal_info(hwmon))
>>   		drm_warn(&hwmon->xe->drm, "Thermal mailbox not supported by card firmware\n");
>>   
>> +	/* vram_count is known only after reading thermal info above. */
>> +	if (hwmon->temp.vram_count) {
>> +		hwmon->temp.vram_label = devm_kcalloc(dev, hwmon->temp.vram_count,
>> +						      MAX_LABEL_SIZE, GFP_KERNEL);
>> +		if (!hwmon->temp.vram_label)
>> +			return -ENOMEM;
>> +	}
>> +
>>   	for (channel = 0; channel < CHANNEL_MAX; channel++) {
>>   		hwmon->temp.available[channel] = xe_hwmon_temp_probe(hwmon, channel);
>> -		if (hwmon->temp.available[channel] &&
>> -		    in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
>> +		if (hwmon->temp.available[channel] && hwmon->temp.vram_label &&
>> +		    in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count))
>>   			snprintf(hwmon->temp.vram_label[channel - CHANNEL_VRAM_N], MAX_LABEL_SIZE,
>>   				 "vram_ch_%d", channel - CHANNEL_VRAM_N);
>>   	}
>> +
>> +	return 0;
>>   }
>>   
>>   static umode_t
>> @@ -1551,8 +1675,10 @@ static int xe_hwmon_read_label(struct device *dev,
>>   			*str = "mctrl";
>>   		else if (channel == CHANNEL_PCIE)
>>   			*str = "pcie";
>> -		else if (in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
>> +		else if (in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count))
>>   			*str = hwmon->temp.vram_label[channel - CHANNEL_VRAM_N];
>> +		else
>> +			return -EOPNOTSUPP;
>>   		return 0;
>>   	case hwmon_power:
>>   	case hwmon_energy:
>> @@ -1580,7 +1706,7 @@ static const struct hwmon_chip_info hwmon_chip_info = {
>>   	.info = hwmon_info,
>>   };
>>   
>> -static void
>> +static int
>>   xe_hwmon_get_preregistration_info(struct xe_hwmon *hwmon)
>>   {
>>   	struct xe_mmio *mmio = xe_root_tile_mmio(hwmon->xe);
>> @@ -1648,7 +1774,7 @@ xe_hwmon_get_preregistration_info(struct xe_hwmon *hwmon)
>>   		if (xe_hwmon_is_visible(hwmon, hwmon_fan, hwmon_fan_input, channel))
>>   			xe_hwmon_fan_input_read(hwmon, channel, &fan_speed);
>>   
>> -	xe_hwmon_init_temp_info(hwmon);
>> +	return xe_hwmon_init_temp_info(hwmon);
>>   }
>>   
>>   int xe_hwmon_register(struct xe_device *xe)
>> @@ -1677,7 +1803,9 @@ int xe_hwmon_register(struct xe_device *xe)
>>   	hwmon->xe = xe;
>>   	xe->hwmon = hwmon;
>>   
>> -	xe_hwmon_get_preregistration_info(hwmon);
>> +	ret = xe_hwmon_get_preregistration_info(hwmon);
>> +	if (ret)
>> +		goto err_null_hwmon;
>>   
>>   	drm_dbg(&xe->drm, "Register xe hwmon interface\n");
>>   
>> @@ -1687,10 +1815,14 @@ int xe_hwmon_register(struct xe_device *xe)
>>   								hwmon_groups);
>>   	if (IS_ERR(hwmon->hwmon_dev)) {
>>   		drm_err(&xe->drm, "Failed to register xe hwmon (%pe)\n", hwmon->hwmon_dev);
>> -		xe->hwmon = NULL;
>> -		return PTR_ERR(hwmon->hwmon_dev);
>> +		ret = PTR_ERR(hwmon->hwmon_dev);
>> +		goto err_null_hwmon;
>>   	}
>>   
>>   	return 0;
>> +
>> +err_null_hwmon:
>> +	xe->hwmon = NULL;
>> +	return ret;
>>   }
>>   MODULE_IMPORT_NS("INTEL_PMT_TELEMETRY");
>> diff --git a/drivers/gpu/drm/xe/xe_pci.c b/drivers/gpu/drm/xe/xe_pci.c
>> index e656fc012902..fa32436ecf6f 100644
>> --- a/drivers/gpu/drm/xe/xe_pci.c
>> +++ b/drivers/gpu/drm/xe/xe_pci.c
>> @@ -416,6 +416,7 @@ static const struct xe_device_desc bmg_desc = {
>>   	.dma_mask_size = 46,
>>   	.has_display = true,
>>   	.has_fan_control = true,
>> +	.has_fixed_vram_channels = true,
>>   	.has_flat_ccs = 1,
>>   	.has_mbx_power_limits = true,
>>   	.has_mbx_thermal_info = true,
>> @@ -794,6 +795,7 @@ static int xe_info_init_early(struct xe_device *xe,
>>   	xe->info.has_cached_pt = desc->has_cached_pt;
>>   	xe->info.has_drm_ras = desc->has_drm_ras;
>>   	xe->info.has_fan_control = desc->has_fan_control;
>> +	xe->info.has_fixed_vram_channels = desc->has_fixed_vram_channels;
>>   	/* runtime fusing may force flat_ccs to disabled later */
>>   	xe->info.has_flat_ccs = desc->has_flat_ccs;
>>   	xe->info.has_mbx_power_limits = desc->has_mbx_power_limits;
>> diff --git a/drivers/gpu/drm/xe/xe_pci_types.h b/drivers/gpu/drm/xe/xe_pci_types.h
>> index 0041ec5676d3..4d1048ca7abe 100644
>> --- a/drivers/gpu/drm/xe/xe_pci_types.h
>> +++ b/drivers/gpu/drm/xe/xe_pci_types.h
>> @@ -42,6 +42,7 @@ struct xe_device_desc {
>>   	u8 has_display:1;
>>   	u8 has_drm_ras:1;
>>   	u8 has_fan_control:1;
>> +	u8 has_fixed_vram_channels:1;
>>   	u8 has_flat_ccs:1;
>>   	u8 has_gsc_nvm:1;
>>   	u8 has_heci_gscfi:1;
>> -- 
>> 2.25.1
>>

  reply	other threads:[~2026-10-01 19:02 UTC|newest]

Thread overview: 36+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-24 20:54 [PATCH v6 00/13] drm/xe/hwmon: Update hwmon thermal mailbox Karthik Poosa
2026-09-24 20:54 ` [PATCH v6 01/13] drm/xe/hwmon: Handle pcode read failures of xe_hwmon_pcode_rmw_power_limit Karthik Poosa
2026-09-28  0:11   ` Rodrigo Vivi
2026-09-28 15:32     ` Poosa, Karthik
2026-09-24 20:54 ` [PATCH v6 02/13] drm/xe/hwmon: Decode mailbox temperature as sign-magnitude Karthik Poosa
2026-09-28  0:15   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 03/13] drm/xe/hwmon: Fix memory controller thermal data handling Karthik Poosa
2026-09-24 20:54 ` [PATCH v6 04/13] drm/xe/hwmon: Add helpers to validate thermal sensor readings Karthik Poosa
2026-09-28  0:19   ` Rodrigo Vivi
2026-09-28 16:08     ` Poosa, Karthik
2026-09-28 17:59       ` Rodrigo Vivi
2026-09-28 19:38   ` Rodrigo Vivi
2026-09-29 16:48     ` Poosa, Karthik
2026-09-29 18:26       ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 05/13] drm/xe/hwmon: Handle unavailable memory controller sensors Karthik Poosa
2026-09-28 16:08   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 06/13] drm/xe/hwmon: Detect unavailable PCIe thermal sensors Karthik Poosa
2026-09-28 16:10   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 07/13] drm/xe/hwmon: Consolidate temperature sensor availability checks Karthik Poosa
2026-09-28 19:39   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 08/13] drm/xe/hwmon: Cache temperature availability to reduce probe time Karthik Poosa
2026-09-28 21:17   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 09/13] drm/xe/hwmon: use CRI-specific package and VRAM temperature registers Karthik Poosa
2026-09-24 20:54 ` [PATCH v6 10/13] drm/xe/hwmon: Add platform-aware VRAM thermal channel support Karthik Poosa
2026-09-29 18:04   ` Rodrigo Vivi
2026-10-01 19:02     ` Poosa, Karthik [this message]
2026-09-24 20:54 ` [PATCH v6 11/13] drm/xe/hwmon: Decode CRI temperature registers as IEEE-754 Karthik Poosa
2026-09-29 18:24   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 12/13] drm/xe/hwmon: Update memory controller temperature offset for CRI Karthik Poosa
2026-09-29 18:27   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 13/13] drm/xe/hwmon: Update PCIE temperature sensor " Karthik Poosa
2026-09-29 18:29   ` Rodrigo Vivi
2026-09-30 17:11     ` Poosa, Karthik
2026-09-24 21:05 ` ✓ CI.KUnit: success for drm/xe/hwmon: Update hwmon thermal mailbox (rev3) Patchwork
2026-09-24 22:34 ` ✓ Xe.CI.BAT: " Patchwork
2026-09-25 11:16 ` ✗ Xe.CI.FULL: failure " Patchwork

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=b144384a-96e8-4cbc-9574-09752bccdeef@intel.com \
    --to=karthik.poosa@intel.com \
    --cc=anshuman.gupta@intel.com \
    --cc=badal.nilawar@intel.com \
    --cc=intel-xe@lists.freedesktop.org \
    --cc=mallesh.koujalagi@intel.com \
    --cc=raag.jadav@intel.com \
    --cc=riana.tauro@intel.com \
    --cc=rodrigo.vivi@intel.com \
    --cc=sk.anirban@intel.com \
    --cc=soham.purkait@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.