Intel-XE Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: Rodrigo Vivi <rodrigo.vivi@intel.com>
To: Tejas Upadhyay <tejas.upadhyay@intel.com>
Cc: intel-xe@lists.freedesktop.org, himal.prasad.ghimiray@intel.com,
	"José Roberto de Souza" <jose.souza@intel.com>,
	"Michal Mrozek" <michal.mrozek@intel.com>
Subject: Re: [PATCH V15 13/14] drm/xe/uapi: Expose ban reason in EXEC_QUEUE_GET_PROPERTY_BAN
Date: Tue, 11 Aug 2026 16:08:14 -0400	[thread overview]
Message-ID: <anuBLhouGcnw_rUw@intel.com> (raw)
In-Reply-To: <20260811124016.3614699-29-tejas.upadhyay@intel.com>

On Tue, Aug 11, 2026 at 06:10:17PM +0530, Tejas Upadhyay wrote:
> Extend DRM_XE_EXEC_QUEUE_GET_PROPERTY_BAN to return a bitmask indicating
> the reason for the ban, rather than a simple boolean. This allows
> userspace to distinguish between different ban causes:
> 
> - DRM_XE_EXEC_QUEUE_BAN_REASON_GPU_HANG (bit 0): exec queue was banned
>   due to a GPU hang or job timeout detected by the TDR.
> - DRM_XE_EXEC_QUEUE_BAN_REASON_PAGE_OFFLINE (bit 1): exec queue was
>   banned because a VRAM page backing its resources was taken offline.
> 
> The ban_reason field is added to struct xe_exec_queue and set at the
> point where the ban is triggered:
> - In guc_exec_queue_timedout_job() for GPU hang.
> - In xe_ttm_vram_purge_page() for memory page offline, before calling
>   xe_exec_queue_kill() or xe_vm_kill().
> 
> The reset_status op is updated to return u64 with the reason bitmask.
> When a queue is banned but no explicit reason was recorded (e.g., from a
> generic CAT error), it defaults to GPU_HANG for backward compatibility.
> A value of 0 means the exec queue is not banned.
> 
> v2(Sashiko):
> - Use atomic_t for ban_reason to fix concurrent updates from TDR and
>   page-offline
> - Guard GPU_HANG bit with !exec_queue_killed to avoid masking
>   page-offline reason
> - Clear ban_reason on queue recovery (clear_exec_queue_banned path)
> - Use atomic_read in guc_exec_queue_reset_status for lockless read
> 
> Assisted-by: Copilot:claude-opus-4.6
> Acked-by: José Roberto de Souza <jose.souza@intel.com>
> Acked-by: Michal Mrozek <michal.mrozek@intel.com>
> Signed-off-by: Tejas Upadhyay <tejas.upadhyay@intel.com>
> ---
>  drivers/gpu/drm/xe/xe_exec_queue_types.h |  7 ++++--
>  drivers/gpu/drm/xe/xe_execlist.c         |  4 +--
>  drivers/gpu/drm/xe/xe_guc_submit.c       | 32 ++++++++++++++++++++----
>  drivers/gpu/drm/xe/xe_ttm_vram_mgr.c     | 10 +++++++-
>  include/uapi/drm/xe_drm.h                | 12 ++++++++-
>  5 files changed, 54 insertions(+), 11 deletions(-)
> 
> diff --git a/drivers/gpu/drm/xe/xe_exec_queue_types.h b/drivers/gpu/drm/xe/xe_exec_queue_types.h
> index b2276559c2f6..a21916359e2f 100644
> --- a/drivers/gpu/drm/xe/xe_exec_queue_types.h
> +++ b/drivers/gpu/drm/xe/xe_exec_queue_types.h
> @@ -156,6 +156,9 @@ struct xe_exec_queue {
>  	 */
>  	unsigned long flags;
>  
> +	/** @ban_reason: Bitmask of ban reasons (DRM_XE_EXEC_QUEUE_BAN_REASON_*) */
> +	atomic_t ban_reason;
> +
>  	union {
>  		/** @multi_gt_list: list head for VM bind engines if multi-GT */
>  		struct list_head multi_gt_list;
> @@ -350,8 +353,8 @@ struct xe_exec_queue_ops {
>  	 * signalled when this function is called.
>  	 */
>  	void (*resume)(struct xe_exec_queue *q);
> -	/** @reset_status: check exec queue reset status */
> -	bool (*reset_status)(struct xe_exec_queue *q);
> +	/** @reset_status: check exec queue ban status, returns ban reason bitmask */
> +	u64 (*reset_status)(struct xe_exec_queue *q);
>  };
>  
>  #endif
> diff --git a/drivers/gpu/drm/xe/xe_execlist.c b/drivers/gpu/drm/xe/xe_execlist.c
> index cc33ae80e8cf..534e7c4e0099 100644
> --- a/drivers/gpu/drm/xe/xe_execlist.c
> +++ b/drivers/gpu/drm/xe/xe_execlist.c
> @@ -452,10 +452,10 @@ static void execlist_exec_queue_resume(struct xe_exec_queue *q)
>  	/* NIY */
>  }
>  
> -static bool execlist_exec_queue_reset_status(struct xe_exec_queue *q)
> +static u64 execlist_exec_queue_reset_status(struct xe_exec_queue *q)
>  {
>  	/* NIY */
> -	return false;
> +	return 0;
>  }
>  
>  static const struct xe_exec_queue_ops execlist_exec_queue_ops = {
> diff --git a/drivers/gpu/drm/xe/xe_guc_submit.c b/drivers/gpu/drm/xe/xe_guc_submit.c
> index 9036f89dff7d..f52577de1ac2 100644
> --- a/drivers/gpu/drm/xe/xe_guc_submit.c
> +++ b/drivers/gpu/drm/xe/xe_guc_submit.c
> @@ -6,6 +6,7 @@
>  #include "xe_guc_submit.h"
>  
>  #include <linux/bitfield.h>
> +#include <uapi/drm/xe_drm.h>
>  #include <linux/bitmap.h>
>  #include <linux/circ_buf.h>
>  #include <linux/dma-fence-array.h>
> @@ -1593,6 +1594,12 @@ guc_exec_queue_timedout_job(struct drm_sched_job *drm_job)
>  	if (!exec_queue_killed(q))
>  		wedged = guc_submit_hint_wedged(exec_queue_to_guc(q));
>  
> +	/*
> +	 * Only tag as GPU hang if this is the original timeout, not a
> +	 * consequence of a prior kill (e.g., page-offline).
> +	 */
> +	if (!exec_queue_killed(q))
> +		atomic_or(DRM_XE_EXEC_QUEUE_BAN_REASON_GPU_HANG, &q->ban_reason);
>  	set_exec_queue_banned(q);
>  
>  	/* Kick job / queue off hardware */
> @@ -1676,6 +1683,9 @@ guc_exec_queue_timedout_job(struct drm_sched_job *drm_job)
>  		if (timeout_needs_gt_reset(q, job, skip_timeout_check)) {
>  			if (!xe_sched_invalidate_job(job, 2)) {
>  				clear_exec_queue_banned(q);
> +				/* protect concurrent page offline reasons */
> +				atomic_andnot(DRM_XE_EXEC_QUEUE_BAN_REASON_GPU_HANG,
> +					      &q->ban_reason);
>  				xe_gt_reset_async(q->gt);
>  				goto rearm;
>  			}
> @@ -2570,13 +2580,25 @@ static void guc_exec_queue_multi_queue_drop_suspend(struct xe_exec_queue *q)
>  	}
>  }
>  
> -static bool guc_exec_queue_reset_status(struct xe_exec_queue *q)
> +static u64 guc_exec_queue_reset_status(struct xe_exec_queue *q)
>  {
> -	if (xe_exec_queue_is_multi_queue_secondary(q) &&
> -	    guc_exec_queue_reset_status(xe_exec_queue_multi_queue_primary(q)))
> -		return true;
> +	if (xe_exec_queue_is_multi_queue_secondary(q)) {
> +		u64 status = guc_exec_queue_reset_status(xe_exec_queue_multi_queue_primary(q));
>  
> -	return exec_queue_reset(q) || exec_queue_killed_or_banned_or_wedged(q);
> +		if (status)
> +			return status;
> +	}
> +
> +	if (exec_queue_reset(q) || exec_queue_killed_or_banned_or_wedged(q)) {
> +		u64 reason = atomic_read_acquire(&q->ban_reason);
> +
> +		/* If no specific reason was recorded, default to GPU hang */
> +		if (!reason)
> +			reason = DRM_XE_EXEC_QUEUE_BAN_REASON_GPU_HANG;
> +		return reason;
> +	}
> +
> +	return 0;
>  }
>  
>  /*
> diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
> index c22669955147..5bb66c7b5505 100644
> --- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
> +++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
> @@ -7,6 +7,7 @@
>  #include <drm/drm_managed.h>
>  #include <drm/drm_drv.h>
>  #include <drm/drm_buddy.h>
> +#include <uapi/drm/xe_drm.h>
>  
>  #include <drm/ttm/ttm_placement.h>
>  #include <drm/ttm/ttm_range_manager.h>
> @@ -541,7 +542,12 @@ static int xe_ttm_vram_purge_page(struct xe_device *xe, struct xe_bo *bo)
>  	xe_bo_unlock(bo);
>  	/*  Ban VM if BO is PPGTT */
>  	if (vm && (flags & XE_BO_FLAG_PAGETABLE)) {
> +		struct xe_exec_queue *eq;
> +
>  		down_write(&vm->lock);
> +		list_for_each_entry(eq, &vm->preempt.exec_queues, lr.link)
> +			atomic_or(DRM_XE_EXEC_QUEUE_BAN_REASON_PAGE_OFFLINE, &eq->ban_reason);
> +		smp_wmb(); /* Force all queue bits to be visible before killing the VM */
>  		xe_vm_kill(vm, true);
>  		up_write(&vm->lock);
>  	}
> @@ -553,7 +559,9 @@ static int xe_ttm_vram_purge_page(struct xe_device *xe, struct xe_bo *bo)
>  	/*  Ban exec queue if BO is lrc */
>  	if (q && xe_exec_queue_get_unless_zero(q)) {
>  		/* ban queue */
> -		q_to_put = q;
> +                atomic_or(DRM_XE_EXEC_QUEUE_BAN_REASON_PAGE_OFFLINE, &q->ban_reason);
> +                smp_wmb(); /* Force bit change to finish before state change triggers */
> +                q_to_put = q;
>  	}
>  
>  	if (bo->purgeable.state == XE_MADV_PURGEABLE_PURGED) {
> diff --git a/include/uapi/drm/xe_drm.h b/include/uapi/drm/xe_drm.h
> index 509202a7b13e..1600e8f0885a 100644
> --- a/include/uapi/drm/xe_drm.h
> +++ b/include/uapi/drm/xe_drm.h
> @@ -1503,7 +1503,17 @@ struct drm_xe_exec_queue_get_property {
>  	/** @property: property to get */
>  	__u32 property;
>  
> -	/** @value: property value */
> +	/**
> +	 * @value: property value
> +	 *
> +	 * For %DRM_XE_EXEC_QUEUE_GET_PROPERTY_BAN, this is a bitmask of:
> +	 *  - %DRM_XE_EXEC_QUEUE_BAN_REASON_GPU_HANG - banned due to GPU hang/timeout
> +	 *  - %DRM_XE_EXEC_QUEUE_BAN_REASON_PAGE_OFFLINE - banned due to memory page offline
> +	 *
> +	 * Value of 0 means the exec queue is not banned.
> +	 */

I have the feeling that the documentation result will be better
with this block documented above, in the struct doc, along with
the property value.

Could you please check that?

but the patch looks good, so one way or another:

Reviewed-by: Rodrigo Vivi <rodrigo.vivi@intel.com>

> +#define DRM_XE_EXEC_QUEUE_BAN_REASON_GPU_HANG		(1 << 0)
> +#define DRM_XE_EXEC_QUEUE_BAN_REASON_PAGE_OFFLINE	(1 << 1)
>  	__u64 value;
>  
>  	/** @reserved: Reserved */
> -- 
> 2.52.0
> 

  reply	other threads:[~2026-08-11 20:08 UTC|newest]

Thread overview: 19+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-11 12:40 [PATCH V15 00/14] Add memory page offlining support Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 01/14] drm/xe: Link VRAM object with gpu buddy Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 02/14] [DO_NOT_MERGE]drm/gpu: Add gpu_buddy_allocated_addr_to_block helper Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 03/14] drm/xe: Link LRC BO and its execution Queue Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 04/14] drm/xe: Extend BO purge to handle vram pages as well Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 05/14] drm/xe/bo: Make xe_bo_is_user() public Tejas Upadhyay
2026-08-11 15:38   ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 06/14] drm/xe: Guard teardown paths against purged BOs Tejas Upadhyay
2026-08-12  3:24   ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 07/14] drm/xe/vram: Extract buddy alloc and free helpers Tejas Upadhyay
2026-08-12  3:25   ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 08/14] drm/xe/vram: Add page offline data structures and lifecycle Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 09/14] drm/xe/vram: Add VRAM page offline fault handler Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 10/14] drm/xe/configfs: Add vram bad page reservation policy Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 11/14] drm/xe/vram: Use RCU for lock-free sysfs reads of bad page lists Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 12/14] drm/xe: Add sysfs interface for bad gpu vram pages Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 13/14] drm/xe/uapi: Expose ban reason in EXEC_QUEUE_GET_PROPERTY_BAN Tejas Upadhyay
2026-08-11 20:08   ` Rodrigo Vivi [this message]
2026-08-11 12:40 ` [PATCH V15 14/14] drm/xe: Add fault-inject based VRAM page offline injection Tejas Upadhyay

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=anuBLhouGcnw_rUw@intel.com \
    --to=rodrigo.vivi@intel.com \
    --cc=himal.prasad.ghimiray@intel.com \
    --cc=intel-xe@lists.freedesktop.org \
    --cc=jose.souza@intel.com \
    --cc=michal.mrozek@intel.com \
    --cc=tejas.upadhyay@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox