Intel-XE Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: Rodrigo Vivi <rodrigo.vivi@intel.com>
To: Karthik Poosa <karthik.poosa@intel.com>
Cc: <intel-xe@lists.freedesktop.org>, <anshuman.gupta@intel.com>,
	<badal.nilawar@intel.com>, <raag.jadav@intel.com>,
	<riana.tauro@intel.com>, <sk.anirban@intel.com>,
	<mallesh.koujalagi@intel.com>, <soham.purkait@intel.com>
Subject: Re: [PATCH v6 10/13] drm/xe/hwmon: Add platform-aware VRAM thermal channel support
Date: Tue, 29 Sep 2026 14:04:04 -0400	[thread overview]
Message-ID: <arv9lBb85GqnSxdi@intel.com> (raw)
In-Reply-To: <20260924205429.2846256-11-karthik.poosa@intel.com>

On Fri, Sep 25, 2026 at 02:24:26AM +0530, Karthik Poosa wrote:
> On CRI, derive the number of available VRAM temperature channels from the
> CRI_MSU_VRAM_ENABLE register, where each enabled MSU contributes
> VRAM_CHANNELS_PER_MSU channels.
>  
> Increase the maximum supported VRAM channel count to 80 and use the
> derived count to control VRAM sensor enumeration, channel access, and
> VRAM label allocation. Add a has_fixed_vram_channels platform flag to
> preserve the existing fixed-channel behavior on non-CRI platforms.
> 
> Signed-off-by: Karthik Poosa <karthik.poosa@intel.com>
> ---
> 
> v2:
>  - Use the highest enabled MSU index to determine the VRAM channel bound
>    instead of counting enabled MSUs. A sparse mask could expose channels
>    beyond the allocated label array, leaving the label pointer unset and
>    potentially hanging during sysfs reads.
>  - Also return an error when no temperature label matches the requested
>    channel.
> 
>  drivers/gpu/drm/xe/regs/xe_pcode_regs.h |   1 +
>  drivers/gpu/drm/xe/xe_device_types.h    |   2 +
>  drivers/gpu/drm/xe/xe_hwmon.c           | 164 +++++++++++++++++++++---
>  drivers/gpu/drm/xe/xe_pci.c             |   2 +
>  drivers/gpu/drm/xe/xe_pci_types.h       |   1 +
>  5 files changed, 154 insertions(+), 16 deletions(-)
> 
> diff --git a/drivers/gpu/drm/xe/regs/xe_pcode_regs.h b/drivers/gpu/drm/xe/regs/xe_pcode_regs.h
> index a969661f0c95..d9e4847dd68e 100644
> --- a/drivers/gpu/drm/xe/regs/xe_pcode_regs.h
> +++ b/drivers/gpu/drm/xe/regs/xe_pcode_regs.h
> @@ -28,6 +28,7 @@
>  #define BMG_PACKAGE_TEMPERATURE			XE_REG(0x138434)
>  
>  #define CRI_PACKAGE_ENERGY_STATUS		XE_REG(0x138120)
> +#define CRI_MSU_VRAM_ENABLE			XE_REG(0x138340)

can you please share again the ref to this?

>  #define CRI_PACKAGE_TEMPERATURE			XE_REG(0x138344)
>  #define CRI_VRAM_TEMPERATURE			XE_REG(0x138348)
>  #define CRI_PLATFORM_ENERGY_STATUS		XE_REG(0x138458)
> diff --git a/drivers/gpu/drm/xe/xe_device_types.h b/drivers/gpu/drm/xe/xe_device_types.h
> index 87554605fa41..8a1ca65f5665 100644
> --- a/drivers/gpu/drm/xe/xe_device_types.h
> +++ b/drivers/gpu/drm/xe/xe_device_types.h
> @@ -184,6 +184,8 @@ struct xe_device {
>  		u8 has_drm_ras:1;
>  		/** @info.has_fan_control: Device supports fan control */
>  		u8 has_fan_control:1;
> +		/** @info.has_fixed_vram_channels: Device has fixed VRAM temperature channels */
> +		u8 has_fixed_vram_channels:1;
>  		/** @info.has_flat_ccs: Whether flat CCS metadata is used */
>  		u8 has_flat_ccs:1;
>  		/** @info.has_gsc_nvm: Device has gsc non-volatile memory */
> diff --git a/drivers/gpu/drm/xe/xe_hwmon.c b/drivers/gpu/drm/xe/xe_hwmon.c
> index 1b2e2fbe43c8..0422db647d11 100644
> --- a/drivers/gpu/drm/xe/xe_hwmon.c
> +++ b/drivers/gpu/drm/xe/xe_hwmon.c
> @@ -39,7 +39,11 @@ enum xe_hwmon_reg_operation {
>  	REG_READ64,
>  };
>  
> -#define MAX_VRAM_CHANNELS      (16)
> +#define MAX_VRAM_CHANNELS	80
> +#define FIXED_VRAM_CHANNELS	16
> +
> +/* Each MSU enable bit maps to one VRAM subsystem of 4 channels. */
> +#define VRAM_CHANNELS_PER_MSU	4
>  
>  enum xe_hwmon_channel {
>  	CHANNEL_CARD,
> @@ -48,6 +52,7 @@ enum xe_hwmon_channel {
>  	CHANNEL_MCTRL,
>  	CHANNEL_PCIE,
>  	CHANNEL_VRAM_N,
> +	/* Compile-time upper bound; actual channel count is hwmon->temp.vram_count */
>  	CHANNEL_VRAM_N_MAX = CHANNEL_VRAM_N + MAX_VRAM_CHANNELS - 1,
>  	CHANNEL_MAX,
>  };
> @@ -144,18 +149,22 @@ struct xe_hwmon_thermal_info {
>  		/** @data: temperature limits in dwords */
>  		u32 data[DIV_ROUND_UP(TEMP_LIMIT_MAX, sizeof(u32))];
>  	};
> -	/** @count: no of temperature sensors available for the platform */
> +	/** @count: temperature sensors count from READ_THERMAL_CONFIG */
>  	u8 count;
> +	/** @vram_count: exclusive upper bound for VRAM temperature channel indices */
> +	u8 vram_count;
>  	/** @available: temperature sensor availability cached at registration */
>  	bool available[CHANNEL_MAX];
> +	/** @msu_mask: VRAM subsystem enable mask read from MMIO */
> +	u32 msu_mask;
>  	union {
>  		/** @value: per-sensor raw mailbox temperature; bit7=sign, bits6:0=magnitude */
>  		u8 value[U8_MAX + 1];
>  		/** @dword: sensor values as dwords, u32-aligned for pcode reads */
>  		u32 dword[DIV_ROUND_UP(U8_MAX + 1, sizeof(u32))];
>  	};
> -	/** @vram_label: vram label names */
> -	char vram_label[MAX_VRAM_CHANNELS][MAX_LABEL_SIZE];
> +	/** @vram_label: vram label names, dynamically allocated based on vram_count */
> +	char (*vram_label)[MAX_LABEL_SIZE];
>  };
>  
>  /**
> @@ -269,6 +278,14 @@ static int xe_hwmon_pcode_rmw_power_limit(const struct xe_hwmon *hwmon, u32 attr
>  	return ret;
>  }
>  
> +static bool xe_hwmon_vram_channel_enabled(const struct xe_hwmon *hwmon, int index)
> +{
> +	if (hwmon->xe->info.platform == XE_CRESCENTISLAND)
> +		return hwmon->temp.msu_mask & BIT(index / VRAM_CHANNELS_PER_MSU);
> +
> +	return true;
> +}
> +
>  static struct xe_reg xe_hwmon_get_reg(struct xe_hwmon *hwmon, enum xe_hwmon_reg hwmon_reg,
>  				      int channel)
>  {
> @@ -281,14 +298,16 @@ static struct xe_reg xe_hwmon_get_reg(struct xe_hwmon *hwmon, enum xe_hwmon_reg
>  				return CRI_PACKAGE_TEMPERATURE;
>  			else if (channel == CHANNEL_VRAM)
>  				return CRI_VRAM_TEMPERATURE;
> -			else if (in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
> +			else if (in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count) &&
> +				 xe_hwmon_vram_channel_enabled(hwmon, channel - CHANNEL_VRAM_N))
>  				return CRI_VRAM_TEMPERATURE_N(channel - CHANNEL_VRAM_N);
>  		} else if (xe->info.platform == XE_BATTLEMAGE) {
>  			if (channel == CHANNEL_PKG)
>  				return BMG_PACKAGE_TEMPERATURE;
>  			else if (channel == CHANNEL_VRAM)
>  				return BMG_VRAM_TEMPERATURE;
> -			else if (in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
> +			else if (in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count) &&
> +				 xe_hwmon_vram_channel_enabled(hwmon, channel - CHANNEL_VRAM_N))
>  				return BMG_VRAM_TEMPERATURE_N(channel - CHANNEL_VRAM_N);
>  		} else if (xe->info.platform == XE_DG2) {
>  			if (channel == CHANNEL_PKG)
> @@ -798,6 +817,70 @@ static const struct hwmon_channel_info * const hwmon_info[] = {
>  			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>  			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
>  			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,
> +			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL,

I was wondering if there was something better we could get here...
and LLM suggested me a reduction from 80 to 8 lines:

#define TEMP_CFG_CARD HWMON_T_LABEL
#define TEMP_CFG_PKG  (HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | \
                       HWMON_T_LABEL | HWMON_T_MAX)
#define TEMP_CFG_SENSOR       (HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL)

/*
 * Indexed by &enum xe_hwmon_channel and zero-terminated; hwmon derives the
 * channel count from the array contents. Sizing on CHANNEL_MAX keeps this in
 * lockstep with the enum.
 */
static const u32 hwmon_temp_config[CHANNEL_MAX + 1] = {
      [CHANNEL_CARD]                          = TEMP_CFG_CARD,
      [CHANNEL_PKG]                           = TEMP_CFG_PKG,
      [CHANNEL_VRAM]                          = TEMP_CFG_SENSOR,
      [CHANNEL_MCTRL]                         = TEMP_CFG_SENSOR,
      [CHANNEL_PCIE]                          = TEMP_CFG_SENSOR,
      [CHANNEL_VRAM_N ... CHANNEL_VRAM_N_MAX] = TEMP_CFG_SENSOR,
      [CHANNEL_MAX]                           = 0,    /* terminator */
};

static const struct hwmon_channel_info hwmon_temp_info = {
      .type = hwmon_temp,
      .config = hwmon_temp_config,
};

static const struct hwmon_channel_info * const hwmon_info[] = {
      &hwmon_temp_info,
      HWMON_CHANNEL_INFO(power, ...),
      /* unchanged */
};

can be done in a follow up though...

Everything else looks sane... I just need to double check the regs again...

>  			   HWMON_T_CRIT | HWMON_T_EMERGENCY | HWMON_T_INPUT | HWMON_T_LABEL),
>  	HWMON_CHANNEL_INFO(power, HWMON_P_MAX | HWMON_P_RATED_MAX | HWMON_P_LABEL | HWMON_P_CRIT |
>  			   HWMON_P_CAP,
> @@ -812,15 +895,37 @@ static const struct hwmon_channel_info * const hwmon_info[] = {
>  static int xe_hwmon_pcode_read_thermal_info(struct xe_hwmon *hwmon)
>  {
>  	struct xe_tile *root_tile = xe_device_get_root_tile(hwmon->xe);
> +	struct xe_mmio *mmio = xe_root_tile_mmio(hwmon->xe);
>  	u32 config = 0;
>  	int ret;
>  
> +	/*
> +	 * Only CRI reports dynamic VRAM channel state. Fixed-channel platforms
> +	 * set the count up front so those sensors stay visible even if the thermal
> +	 * mailbox reads below fail.
> +	 */
> +	if (hwmon->xe->info.has_fixed_vram_channels)
> +		hwmon->temp.vram_count = FIXED_VRAM_CHANNELS;
> +
> +	if (hwmon->xe->info.platform == XE_CRESCENTISLAND) {
> +		hwmon->temp.msu_mask = xe_mmio_read32(mmio, CRI_MSU_VRAM_ENABLE);
> +		hwmon->temp.vram_count = fls(hwmon->temp.msu_mask) * VRAM_CHANNELS_PER_MSU;
> +		drm_dbg(&hwmon->xe->drm, "MSU VRAM enable mask 0x%x, VRAM channel bound %d\n",
> +			hwmon->temp.msu_mask, hwmon->temp.vram_count);
> +		if (hwmon->temp.vram_count > MAX_VRAM_CHANNELS) {
> +			drm_warn(&hwmon->xe->drm,
> +				 "VRAM channel bound %d exceeds max %d, clamping\n",
> +				 hwmon->temp.vram_count, MAX_VRAM_CHANNELS);
> +			hwmon->temp.vram_count = MAX_VRAM_CHANNELS;
> +		}
> +	}
> +
>  	ret = xe_pcode_read(root_tile, PCODE_MBOX(PCODE_THERMAL_INFO, READ_THERMAL_LIMITS, 0),
>  			    &hwmon->temp.data[0], &hwmon->temp.data[1]);
>  	if (ret)
>  		return ret;
>  
> -	drm_dbg(&hwmon->xe->drm, "thermal info read val 0x%x val1 0x%x\n",
> +	drm_dbg(&hwmon->xe->drm, "thermal info read val 0x%08x val1 0x%08x\n",
>  		hwmon->temp.data[0], hwmon->temp.data[1]);
>  
>  	ret = xe_pcode_read(root_tile, PCODE_MBOX(PCODE_THERMAL_INFO, READ_THERMAL_CONFIG, 0),
> @@ -1025,6 +1130,14 @@ static bool xe_hwmon_temp_probe(struct xe_hwmon *hwmon, int channel)
>  		if (!xe_reg_is_valid(reg))
>  			return false;
>  
> +		/*
> +		 * On CRI, VRAM channel presence comes from the MSU enable mask
> +		 * (applied in xe_hwmon_get_reg()); the temperature register's
> +		 * 0xffffffff sentinel is not reliable for VRAM channels.
> +		 */
> +		if (hwmon->xe->info.platform == XE_CRESCENTISLAND && channel >= CHANNEL_VRAM_N)
> +			return true;
> +
>  		reg_val = xe_mmio_read32(mmio, reg);
>  		if (!mmio_temp_valid(hwmon, reg_val)) {
>  			drm_dbg(&hwmon->xe->drm,
> @@ -1042,20 +1155,31 @@ static bool xe_hwmon_temp_probe(struct xe_hwmon *hwmon, int channel)
>  	}
>  }
>  
> -static void xe_hwmon_init_temp_info(struct xe_hwmon *hwmon)
> +static int xe_hwmon_init_temp_info(struct xe_hwmon *hwmon)
>  {
> +	struct device *dev = hwmon->xe->drm.dev;
>  	int channel;
>  
>  	if (hwmon->xe->info.has_mbx_thermal_info && xe_hwmon_pcode_read_thermal_info(hwmon))
>  		drm_warn(&hwmon->xe->drm, "Thermal mailbox not supported by card firmware\n");
>  
> +	/* vram_count is known only after reading thermal info above. */
> +	if (hwmon->temp.vram_count) {
> +		hwmon->temp.vram_label = devm_kcalloc(dev, hwmon->temp.vram_count,
> +						      MAX_LABEL_SIZE, GFP_KERNEL);
> +		if (!hwmon->temp.vram_label)
> +			return -ENOMEM;
> +	}
> +
>  	for (channel = 0; channel < CHANNEL_MAX; channel++) {
>  		hwmon->temp.available[channel] = xe_hwmon_temp_probe(hwmon, channel);
> -		if (hwmon->temp.available[channel] &&
> -		    in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
> +		if (hwmon->temp.available[channel] && hwmon->temp.vram_label &&
> +		    in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count))
>  			snprintf(hwmon->temp.vram_label[channel - CHANNEL_VRAM_N], MAX_LABEL_SIZE,
>  				 "vram_ch_%d", channel - CHANNEL_VRAM_N);
>  	}
> +
> +	return 0;
>  }
>  
>  static umode_t
> @@ -1551,8 +1675,10 @@ static int xe_hwmon_read_label(struct device *dev,
>  			*str = "mctrl";
>  		else if (channel == CHANNEL_PCIE)
>  			*str = "pcie";
> -		else if (in_range(channel, CHANNEL_VRAM_N, MAX_VRAM_CHANNELS))
> +		else if (in_range(channel, CHANNEL_VRAM_N, hwmon->temp.vram_count))
>  			*str = hwmon->temp.vram_label[channel - CHANNEL_VRAM_N];
> +		else
> +			return -EOPNOTSUPP;
>  		return 0;
>  	case hwmon_power:
>  	case hwmon_energy:
> @@ -1580,7 +1706,7 @@ static const struct hwmon_chip_info hwmon_chip_info = {
>  	.info = hwmon_info,
>  };
>  
> -static void
> +static int
>  xe_hwmon_get_preregistration_info(struct xe_hwmon *hwmon)
>  {
>  	struct xe_mmio *mmio = xe_root_tile_mmio(hwmon->xe);
> @@ -1648,7 +1774,7 @@ xe_hwmon_get_preregistration_info(struct xe_hwmon *hwmon)
>  		if (xe_hwmon_is_visible(hwmon, hwmon_fan, hwmon_fan_input, channel))
>  			xe_hwmon_fan_input_read(hwmon, channel, &fan_speed);
>  
> -	xe_hwmon_init_temp_info(hwmon);
> +	return xe_hwmon_init_temp_info(hwmon);
>  }
>  
>  int xe_hwmon_register(struct xe_device *xe)
> @@ -1677,7 +1803,9 @@ int xe_hwmon_register(struct xe_device *xe)
>  	hwmon->xe = xe;
>  	xe->hwmon = hwmon;
>  
> -	xe_hwmon_get_preregistration_info(hwmon);
> +	ret = xe_hwmon_get_preregistration_info(hwmon);
> +	if (ret)
> +		goto err_null_hwmon;
>  
>  	drm_dbg(&xe->drm, "Register xe hwmon interface\n");
>  
> @@ -1687,10 +1815,14 @@ int xe_hwmon_register(struct xe_device *xe)
>  								hwmon_groups);
>  	if (IS_ERR(hwmon->hwmon_dev)) {
>  		drm_err(&xe->drm, "Failed to register xe hwmon (%pe)\n", hwmon->hwmon_dev);
> -		xe->hwmon = NULL;
> -		return PTR_ERR(hwmon->hwmon_dev);
> +		ret = PTR_ERR(hwmon->hwmon_dev);
> +		goto err_null_hwmon;
>  	}
>  
>  	return 0;
> +
> +err_null_hwmon:
> +	xe->hwmon = NULL;
> +	return ret;
>  }
>  MODULE_IMPORT_NS("INTEL_PMT_TELEMETRY");
> diff --git a/drivers/gpu/drm/xe/xe_pci.c b/drivers/gpu/drm/xe/xe_pci.c
> index e656fc012902..fa32436ecf6f 100644
> --- a/drivers/gpu/drm/xe/xe_pci.c
> +++ b/drivers/gpu/drm/xe/xe_pci.c
> @@ -416,6 +416,7 @@ static const struct xe_device_desc bmg_desc = {
>  	.dma_mask_size = 46,
>  	.has_display = true,
>  	.has_fan_control = true,
> +	.has_fixed_vram_channels = true,
>  	.has_flat_ccs = 1,
>  	.has_mbx_power_limits = true,
>  	.has_mbx_thermal_info = true,
> @@ -794,6 +795,7 @@ static int xe_info_init_early(struct xe_device *xe,
>  	xe->info.has_cached_pt = desc->has_cached_pt;
>  	xe->info.has_drm_ras = desc->has_drm_ras;
>  	xe->info.has_fan_control = desc->has_fan_control;
> +	xe->info.has_fixed_vram_channels = desc->has_fixed_vram_channels;
>  	/* runtime fusing may force flat_ccs to disabled later */
>  	xe->info.has_flat_ccs = desc->has_flat_ccs;
>  	xe->info.has_mbx_power_limits = desc->has_mbx_power_limits;
> diff --git a/drivers/gpu/drm/xe/xe_pci_types.h b/drivers/gpu/drm/xe/xe_pci_types.h
> index 0041ec5676d3..4d1048ca7abe 100644
> --- a/drivers/gpu/drm/xe/xe_pci_types.h
> +++ b/drivers/gpu/drm/xe/xe_pci_types.h
> @@ -42,6 +42,7 @@ struct xe_device_desc {
>  	u8 has_display:1;
>  	u8 has_drm_ras:1;
>  	u8 has_fan_control:1;
> +	u8 has_fixed_vram_channels:1;
>  	u8 has_flat_ccs:1;
>  	u8 has_gsc_nvm:1;
>  	u8 has_heci_gscfi:1;
> -- 
> 2.25.1
> 

  reply	other threads:[~2026-09-29 18:04 UTC|newest]

Thread overview: 36+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-24 20:54 [PATCH v6 00/13] drm/xe/hwmon: Update hwmon thermal mailbox Karthik Poosa
2026-09-24 20:54 ` [PATCH v6 01/13] drm/xe/hwmon: Handle pcode read failures of xe_hwmon_pcode_rmw_power_limit Karthik Poosa
2026-09-28  0:11   ` Rodrigo Vivi
2026-09-28 15:32     ` Poosa, Karthik
2026-09-24 20:54 ` [PATCH v6 02/13] drm/xe/hwmon: Decode mailbox temperature as sign-magnitude Karthik Poosa
2026-09-28  0:15   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 03/13] drm/xe/hwmon: Fix memory controller thermal data handling Karthik Poosa
2026-09-24 20:54 ` [PATCH v6 04/13] drm/xe/hwmon: Add helpers to validate thermal sensor readings Karthik Poosa
2026-09-28  0:19   ` Rodrigo Vivi
2026-09-28 16:08     ` Poosa, Karthik
2026-09-28 17:59       ` Rodrigo Vivi
2026-09-28 19:38   ` Rodrigo Vivi
2026-09-29 16:48     ` Poosa, Karthik
2026-09-29 18:26       ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 05/13] drm/xe/hwmon: Handle unavailable memory controller sensors Karthik Poosa
2026-09-28 16:08   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 06/13] drm/xe/hwmon: Detect unavailable PCIe thermal sensors Karthik Poosa
2026-09-28 16:10   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 07/13] drm/xe/hwmon: Consolidate temperature sensor availability checks Karthik Poosa
2026-09-28 19:39   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 08/13] drm/xe/hwmon: Cache temperature availability to reduce probe time Karthik Poosa
2026-09-28 21:17   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 09/13] drm/xe/hwmon: use CRI-specific package and VRAM temperature registers Karthik Poosa
2026-09-24 20:54 ` [PATCH v6 10/13] drm/xe/hwmon: Add platform-aware VRAM thermal channel support Karthik Poosa
2026-09-29 18:04   ` Rodrigo Vivi [this message]
2026-10-01 19:02     ` Poosa, Karthik
2026-09-24 20:54 ` [PATCH v6 11/13] drm/xe/hwmon: Decode CRI temperature registers as IEEE-754 Karthik Poosa
2026-09-29 18:24   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 12/13] drm/xe/hwmon: Update memory controller temperature offset for CRI Karthik Poosa
2026-09-29 18:27   ` Rodrigo Vivi
2026-09-24 20:54 ` [PATCH v6 13/13] drm/xe/hwmon: Update PCIE temperature sensor " Karthik Poosa
2026-09-29 18:29   ` Rodrigo Vivi
2026-09-30 17:11     ` Poosa, Karthik
2026-09-24 21:05 ` ✓ CI.KUnit: success for drm/xe/hwmon: Update hwmon thermal mailbox (rev3) Patchwork
2026-09-24 22:34 ` ✓ Xe.CI.BAT: " Patchwork
2026-09-25 11:16 ` ✗ Xe.CI.FULL: failure " Patchwork

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=arv9lBb85GqnSxdi@intel.com \
    --to=rodrigo.vivi@intel.com \
    --cc=anshuman.gupta@intel.com \
    --cc=badal.nilawar@intel.com \
    --cc=intel-xe@lists.freedesktop.org \
    --cc=karthik.poosa@intel.com \
    --cc=mallesh.koujalagi@intel.com \
    --cc=raag.jadav@intel.com \
    --cc=riana.tauro@intel.com \
    --cc=sk.anirban@intel.com \
    --cc=soham.purkait@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox