Intel-XE Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: Rodrigo Vivi <rodrigo.vivi@intel.com>
To: Karthik Poosa <karthik.poosa@intel.com>
Cc: <intel-xe@lists.freedesktop.org>, <anshuman.gupta@intel.com>,
	<badal.nilawar@intel.com>, <raag.jadav@intel.com>,
	<riana.tauro@intel.com>, <sk.anirban@intel.com>,
	<mallesh.koujalagi@intel.com>, <soham.purkait@intel.com>
Subject: Re: [PATCH v3 1/5] drm/xe/hwmon: Detect unavailable temperature sensors
Date: Fri, 18 Sep 2026 05:40:44 -0400	[thread overview]
Message-ID: <aq0HHDqoXxXFlsam@intel.com> (raw)
In-Reply-To: <20260910182756.638830-2-karthik.poosa@intel.com>

On Thu, Sep 10, 2026 at 11:57:52PM +0530, Karthik Poosa wrote:
> Add is_temp_available() to validate sensor presence.
> A temperature reading of 0xFFFFFFF on MMIO and 0xFF from mailbox on CRI
> platforms indicates that the corresponding sensor is not present and
> should be treated as unavailable.
> 
> Use this check from xe_hwmon_temp_is_visible() callback so that attributes
> for unavailable sensors are not exposed during hwmon device registration.

I'm afraid this patch is doing much more than what stated here.
Please split this patch into smaller sections.
It is likely regressing older platforms becuase it is hard to spot the
differences.

> 
> Signed-off-by: Karthik Poosa <karthik.poosa@intel.com>
> ---
> v2:
>  - Address review comments from sashiko-bot@kernel.org.
>  - Address review comments for Badal, Raag.
>  - Update is_temp_valid() to check values based on mailbox or MMIO.
>  - Add is_temp_is_visible().
>  - Update is_vram_ch_available() to is_temp_available() to check if sensor
>    is present.
> 
> v3:
>  - Rephrase a debug log.
> 
>  drivers/gpu/drm/xe/xe_hwmon.c | 100 ++++++++++++++++++++++++----------
>  1 file changed, 71 insertions(+), 29 deletions(-)
> 
> diff --git a/drivers/gpu/drm/xe/xe_hwmon.c b/drivers/gpu/drm/xe/xe_hwmon.c
> index 5284cab6703d..faed2f5da394 100644
> --- a/drivers/gpu/drm/xe/xe_hwmon.c
> +++ b/drivers/gpu/drm/xe/xe_hwmon.c
> @@ -813,12 +813,20 @@ static int xe_hwmon_pcode_read_thermal_info(struct xe_hwmon *hwmon)
>  	return ret;
>  }
>  
> +static inline bool is_temp_valid(const struct xe_hwmon *hwmon, u32 value, bool is_mmio)
> +{
> +	if (hwmon->xe->info.platform >= XE_CRESCENTISLAND)
> +		return value != (is_mmio ? U32_MAX : U8_MAX);
> +
> +	return value != 0;
> +}

first of all, please use 2 separate functions for this.
second, you have a bug here and the mailbox logic is never protected s8 is passed,
so U8_MAX will never work.

LLM suggestion:

static bool mmio_temp_valid(const struct xe_hwmon *hwmon, u32 v)
{
        return hwmon->xe->info.has_mbx_temp_sentinel ? v != U32_MAX
                                                     : v != 0;
}

static bool mbx_temp_valid(const struct xe_hwmon *hwmon, s8 v)
{
        return hwmon->xe->info.has_mbx_temp_sentinel ? v != (s8)0xff
                                                     : v != 0;
}


> +
>  static int get_mc_temp(struct xe_hwmon *hwmon, long *val)
>  {
>  	struct xe_tile *root_tile = xe_device_get_root_tile(hwmon->xe);
>  	u32 *dword = (u32 *)hwmon->temp.value;
> +	int ret, i, count = 0;
>  	s32 average = 0;
> -	int ret, i;
>  
>  	for (i = 0; i < DIV_ROUND_UP(TEMP_LIMIT_MAX, sizeof(u32)); i++) {
>  		ret = xe_pcode_read(root_tile, PCODE_MBOX(PCODE_THERMAL_INFO, READ_THERMAL_DATA, i),
> @@ -828,11 +836,23 @@ static int get_mc_temp(struct xe_hwmon *hwmon, long *val)
>  		drm_dbg(&hwmon->xe->drm, "thermal data for group %d val 0x%x\n", i, dword[i]);
>  	}
>  
> -	for (i = TEMP_INDEX_MCTRL; i < hwmon->temp.count - 1; i++)
> +	for (i = TEMP_INDEX_MCTRL; i < hwmon->temp.count - 1; i++) {
> +		if (!is_temp_valid(hwmon, hwmon->temp.value[i], false))
> +			continue;
>  		average += hwmon->temp.value[i];
> +		count++;
> +	}
> +
> +	if (!count) {
> +		drm_dbg(&hwmon->xe->drm, "Memory controller temperature unavailable!\n");
> +		return -ENXIO;
> +	}
> +
> +	average /= count;
> +
> +	if (val)
> +		*val = average * MILLIDEGREE_PER_DEGREE;
>  
> -	average /= (hwmon->temp.count - TEMP_INDEX_MCTRL - 1);
> -	*val = average * MILLIDEGREE_PER_DEGREE;
>  	return 0;
>  }
>  
> @@ -852,7 +872,13 @@ static int get_pcie_temp(struct xe_hwmon *hwmon, long *val)
>  		data = REG_FIELD_GET(PCIE_SENSOR_MASK, data);
>  
>  	data = REG_FIELD_GET(TEMP_MASK, data);
> -	*val = (s8)data * MILLIDEGREE_PER_DEGREE;
> +	if (!is_temp_valid(hwmon, data, false)) {
> +		drm_dbg(&hwmon->xe->drm, "PCIe temperature not available\n");
> +		return -ENXIO;
> +	}
> +
> +	if (val)
> +		*val = (s8)data * MILLIDEGREE_PER_DEGREE;
>  
>  	return 0;
>  }
> @@ -951,19 +977,39 @@ static void xe_hwmon_get_voltage(struct xe_hwmon *hwmon, int channel, long *valu
>  	*value = DIV_ROUND_CLOSEST(REG_FIELD_GET(VOLTAGE_MASK, reg_val) * 2500, SF_VOLTAGE);
>  }
>  
> -static inline bool is_vram_ch_available(struct xe_hwmon *hwmon, int channel)
> +static bool is_temp_available(struct xe_hwmon *hwmon, int channel)
>  {
>  	struct xe_mmio *mmio = xe_root_tile_mmio(hwmon->xe);
> -	int vram_id = channel - CHANNEL_VRAM_N;
> -	struct xe_reg vram_reg;
> +	struct xe_reg reg;
> +	u32 reg_val;
>  
> -	vram_reg = xe_hwmon_get_reg(hwmon, REG_TEMP, channel);
> -	if (!xe_reg_is_valid(vram_reg) || !xe_mmio_read32(mmio, vram_reg))
> -		return false;
> +	switch (channel) {
> +	case CHANNEL_PKG:
> +	case CHANNEL_VRAM:
> +	case CHANNEL_VRAM_N...CHANNEL_VRAM_N_MAX:
> +		reg = xe_hwmon_get_reg(hwmon, REG_TEMP, channel);
> +		if (!xe_reg_is_valid(reg))
> +			return false;
> +
> +		reg_val = xe_mmio_read32(mmio, reg);
> +		if (!is_temp_valid(hwmon, reg_val, true)) {
> +			drm_dbg(&hwmon->xe->drm,
> +				"channel %d temperautre unavailable, val 0x%x\n", channel, reg_val);

typo

> +			return false;
> +		}
> +
> +		if (channel >= CHANNEL_VRAM_N)
> +			sprintf(hwmon->temp.vram_label[channel - CHANNEL_VRAM_N],
> +				"vram_ch_%d", channel - CHANNEL_VRAM_N);
>  
> -	/* Create label only for available vram channel */
> -	sprintf(hwmon->temp.vram_label[vram_id], "vram_ch_%d", vram_id);
> -	return true;
> +		return true;
> +	case CHANNEL_MCTRL:
> +		return hwmon->temp.count && !get_mc_temp(hwmon, NULL);
> +	case CHANNEL_PCIE:
> +		return hwmon->temp.count && !get_pcie_temp(hwmon, NULL);
> +	default:
> +		return false;
> +	}
>  }
>  
>  static umode_t
> @@ -973,14 +1019,13 @@ xe_hwmon_temp_is_visible(struct xe_hwmon *hwmon, u32 attr, int channel)
>  	case hwmon_temp_emergency:
>  		switch (channel) {
>  		case CHANNEL_PKG:
> -			return hwmon->temp.limit[TEMP_LIMIT_PKG_SHUTDOWN] ? 0444 : 0;
> -		case CHANNEL_VRAM:
> -			return hwmon->temp.limit[TEMP_LIMIT_MEM_SHUTDOWN] ? 0444 : 0;
>  		case CHANNEL_MCTRL:
>  		case CHANNEL_PCIE:
> -			return hwmon->temp.count ? 0444 : 0;
> +			return (is_temp_available(hwmon, channel) &&
> +				hwmon->temp.limit[TEMP_LIMIT_PKG_SHUTDOWN]) ? 0444 : 0;
> +		case CHANNEL_VRAM:
>  		case CHANNEL_VRAM_N...CHANNEL_VRAM_N_MAX:
> -			return (is_vram_ch_available(hwmon, channel) &&
> +			return (is_temp_available(hwmon, channel) &&
>  				hwmon->temp.limit[TEMP_LIMIT_MEM_SHUTDOWN]) ? 0444 : 0;
>  		default:
>  			return 0;
> @@ -988,14 +1033,13 @@ xe_hwmon_temp_is_visible(struct xe_hwmon *hwmon, u32 attr, int channel)
>  	case hwmon_temp_crit:
>  		switch (channel) {
>  		case CHANNEL_PKG:
> -			return hwmon->temp.limit[TEMP_LIMIT_PKG_CRIT] ? 0444 : 0;
> -		case CHANNEL_VRAM:
> -			return hwmon->temp.limit[TEMP_LIMIT_MEM_CRIT] ? 0444 : 0;
>  		case CHANNEL_MCTRL:
>  		case CHANNEL_PCIE:
> -			return hwmon->temp.count ? 0444 : 0;
> +			return (is_temp_available(hwmon, channel) &&
> +				hwmon->temp.limit[TEMP_LIMIT_PKG_CRIT]) ? 0444 : 0;
> +		case CHANNEL_VRAM:
>  		case CHANNEL_VRAM_N...CHANNEL_VRAM_N_MAX:
> -			return (is_vram_ch_available(hwmon, channel) &&
> +			return (is_temp_available(hwmon, channel) &&
>  				hwmon->temp.limit[TEMP_LIMIT_MEM_CRIT]) ? 0444 : 0;
>  		default:
>  			return 0;
> @@ -1003,7 +1047,8 @@ xe_hwmon_temp_is_visible(struct xe_hwmon *hwmon, u32 attr, int channel)
>  	case hwmon_temp_max:
>  		switch (channel) {
>  		case CHANNEL_PKG:
> -			return hwmon->temp.limit[TEMP_LIMIT_PKG_MAX] ? 0444 : 0;
> +			return (is_temp_available(hwmon, channel) &&
> +				hwmon->temp.limit[TEMP_LIMIT_PKG_MAX]) ? 0444 : 0;
>  		default:
>  			return 0;
>  		}
> @@ -1012,13 +1057,10 @@ xe_hwmon_temp_is_visible(struct xe_hwmon *hwmon, u32 attr, int channel)
>  		switch (channel) {
>  		case CHANNEL_PKG:
>  		case CHANNEL_VRAM:
> -			return xe_reg_is_valid(xe_hwmon_get_reg(hwmon, REG_TEMP,
> -								channel)) ? 0444 : 0;
>  		case CHANNEL_MCTRL:
>  		case CHANNEL_PCIE:
> -			return hwmon->temp.count ? 0444 : 0;
>  		case CHANNEL_VRAM_N...CHANNEL_VRAM_N_MAX:
> -			return is_vram_ch_available(hwmon, channel) ? 0444 : 0;
> +			return is_temp_available(hwmon, channel) ? 0444 : 0;
>  		default:
>  			return 0;
>  		}
> -- 
> 2.25.1
> 

  parent reply	other threads:[~2026-09-18  9:41 UTC|newest]

Thread overview: 21+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-10 18:27 [PATCH v3 0/5] drm/xe/hwmon: Update hwmon thermal mailbox handling Karthik Poosa
2026-09-10 18:27 ` [PATCH v3 1/5] drm/xe/hwmon: Detect unavailable temperature sensors Karthik Poosa
2026-09-10 18:42   ` sashiko-bot
2026-09-17 11:02     ` Poosa, Karthik
2026-09-18  9:40   ` Rodrigo Vivi [this message]
2026-09-23 15:49     ` Poosa, Karthik
2026-09-10 18:27 ` [PATCH v3 2/5] drm/xe/hwmon: Use VRAM temperature sensor count from thermal config on CRI Karthik Poosa
2026-09-10 18:47   ` sashiko-bot
2026-09-18  9:50   ` Rodrigo Vivi
2026-09-10 18:27 ` [PATCH v3 3/5] drm/xe/hwmon: Fix memory controller thermal data handling Karthik Poosa
2026-09-18 10:00   ` Rodrigo Vivi
2026-09-10 18:27 ` [PATCH v3 4/5] drm/xe/hwmon: Decode CRI temperature registers as IEEE-754 Karthik Poosa
2026-09-18 10:13   ` Rodrigo Vivi
2026-09-24  6:54     ` Poosa, Karthik
2026-09-10 18:27 ` [PATCH v3 5/5] drm/xe/hwmon: Disable memory controller and PCIe temperatures on CRI Karthik Poosa
2026-09-18 10:20   ` Rodrigo Vivi
2026-09-23 18:37     ` Poosa, Karthik
2026-09-10 18:47 ` ✓ CI.KUnit: success for drm/xe/hwmon: Update hwmon thermal mailbox handling (rev3) Patchwork
2026-09-10 19:52 ` ✓ Xe.CI.BAT: " Patchwork
2026-09-11  4:49 ` ✓ Xe.CI.FULL: " Patchwork
2026-09-24 11:16 ` [PATCH v3 0/5] drm/xe/hwmon: Update hwmon thermal mailbox handling Poosa, Karthik

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=aq0HHDqoXxXFlsam@intel.com \
    --to=rodrigo.vivi@intel.com \
    --cc=anshuman.gupta@intel.com \
    --cc=badal.nilawar@intel.com \
    --cc=intel-xe@lists.freedesktop.org \
    --cc=karthik.poosa@intel.com \
    --cc=mallesh.koujalagi@intel.com \
    --cc=raag.jadav@intel.com \
    --cc=riana.tauro@intel.com \
    --cc=sk.anirban@intel.com \
    --cc=soham.purkait@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox