Intel-XE Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: "Ghimiray, Himal Prasad" <himal.prasad.ghimiray@intel.com>
To: Matthew Brost <matthew.brost@intel.com>,
	<intel-xe@lists.freedesktop.org>
Subject: Re: [PATCH v3] drm/xe: Add mnemonic error reason to page fault diagnostics
Date: Fri, 18 Sep 2026 09:17:46 +0530	[thread overview]
Message-ID: <cc53d16b-6542-45ea-8793-73376aac6cf8@intel.com> (raw)
In-Reply-To: <20260917203850.321385-1-matthew.brost@intel.com>



On 18-09-2026 02:08, Matthew Brost wrote:
> xe_pagefault_print() was firing without indicating why servicing of a
> fault failed, making it hard to distinguish benign races (e.g. a VM's
> file descriptor closing mid-fault) from real bugs.
> 
> Add enum xe_pagefault_error, uniquely identifying the high-level point
> at which xe_pagefault_service(), xe_pagefault_handle_vma(), and
> xe_svm_handle_pagefault() fail. Since consumer.page_addr is always
> page (4K) aligned, steal the reserved low byte to carry this code
> without growing struct xe_pagefault. Add xe_pagefault_set_error() /
> xe_pagefault_get_error() to encode/decode it, and xe_pagefault_addr()
> as the single accessor for the real (masked) faulted address; update
> all real address consumers (xe_pagefault_match(), cache alignment,
> xe_vm_add_fault_entry_pf()) to go through it instead of reading
> consumer.page_addr directly. xe_pagefault_set_start_addr() now
> preserves any already-recorded error bits rather than clobbering them.
> 
> xe_pagefault_print() decodes the error via a new
> xe_pagefault_error_to_str() and reports it in an added "Error:" line.
> 
> xe_pagefault_asid_to_vm() now distinguishes an ASID with no VM at all
> (XE_PAGEFAULT_ERROR_VM_NOT_FOUND, e.g. the owning file descriptor was
> already closed) from a VM found but not in fault mode
> (XE_PAGEFAULT_ERROR_VM_NOT_IN_FAULT_MODE).
> 
> Signed-off-by: Matthew Brost <matthew.brost@intel.com>
> Assisted-by: GitHub_Copilot:claude-sonnet-5
> 
> ---
> v2:
>   - Add more XE_PAGEFAULT_ERROR_* types for SVM
> v3:
>   - Reset to XE_PAGEFAULT_ERROR_NONE on retries (Sashiko)
> ---
>   drivers/gpu/drm/xe/xe_pagefault.c       | 102 +++++++++++++++++++-----
>   drivers/gpu/drm/xe/xe_pagefault.h       |  67 +++++++++++++++-
>   drivers/gpu/drm/xe/xe_pagefault_types.h |  82 +++++++++++++++++++
>   drivers/gpu/drm/xe/xe_svm.c             |  21 ++++-
>   drivers/gpu/drm/xe/xe_vm.c              |   3 +-
>   5 files changed, 250 insertions(+), 25 deletions(-)
> 
> diff --git a/drivers/gpu/drm/xe/xe_pagefault.c b/drivers/gpu/drm/xe/xe_pagefault.c
> index aeb56ff5d58e..c4aafdbdb518 100644
> --- a/drivers/gpu/drm/xe/xe_pagefault.c
> +++ b/drivers/gpu/drm/xe/xe_pagefault.c
> @@ -158,8 +158,14 @@ static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma,
>   	lockdep_assert_held(&vm->lock);
>   
>   	needs_vram = xe_vma_need_vram_for_atomic(vm->xe, vma, atomic);
> -	if (needs_vram < 0 || (needs_vram && xe_vma_is_userptr(vma)))
> -		return needs_vram < 0 ? needs_vram : -EACCES;
> +	if (needs_vram < 0) {
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VMA_NEEDS_VRAM_CHECK);
> +		return needs_vram;
> +	}
> +	if (needs_vram && xe_vma_is_userptr(vma)) {
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VMA_ATOMIC_USERPTR);
> +		return -EACCES;
> +	}
>   
>   	xe_gt_stats_incr(gt, XE_GT_STATS_ID_VMA_PAGEFAULT_COUNT, 1);
>   	xe_gt_stats_incr(gt, XE_GT_STATS_ID_VMA_PAGEFAULT_KB,
> @@ -178,13 +184,17 @@ static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma,
>   	}
>   
>   	do {
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_NONE);
> +
>   		if (xe_vma_is_userptr(vma) &&
>   		    xe_vma_userptr_check_repin(to_userptr_vma(vma))) {
>   			struct xe_userptr_vma *uvma = to_userptr_vma(vma);
>   
>   			err = xe_vma_userptr_pin_pages(uvma);
> -			if (err)
> +			if (err) {
> +				xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VMA_USERPTR_PIN);
>   				return err;
> +			}
>   		}
>   
>   		/* Lock VM and BOs dma-resv */
> @@ -195,8 +205,10 @@ static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma,
>   						 needs_vram == 1);
>   			drm_exec_retry_on_contention(&exec);
>   			xe_validation_retry_on_oom(&ctx, &err);
> -			if (err)
> +			if (err) {
> +				xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VMA_VALIDATE);
>   				break;
> +			}
>   
>   			/* Bind VMA only to the GT that has faulted */
>   			trace_xe_vma_pf_bind(vma);
> @@ -206,6 +218,7 @@ static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma,
>   			if (IS_ERR(fence)) {
>   				err = PTR_ERR(fence);
>   				xe_validation_retry_on_oom(&ctx, &err);
> +				xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VMA_REBIND);
>   				break;
>   			}
>   		}
> @@ -230,16 +243,22 @@ xe_pagefault_access_is_atomic(enum xe_pagefault_access_type access_type)
>   	return (access_type & XE_PAGEFAULT_ACCESS_TYPE_MASK) == XE_PAGEFAULT_ACCESS_TYPE_ATOMIC;
>   }
>   
> -static struct xe_vm *xe_pagefault_asid_to_vm(struct xe_device *xe, u32 asid)
> +static struct xe_vm *xe_pagefault_asid_to_vm(struct xe_pagefault *pf, u32 asid)
>   {
> +	struct xe_device *xe = gt_to_xe(pf->gt);
>   	struct xe_vm *vm;
>   
>   	down_read(&xe->usm.lock);
>   	vm = xa_load(&xe->usm.asid_to_vm, asid);
> -	if (vm && xe_vm_in_fault_mode(vm))
> -		xe_vm_get(vm);
> -	else
> +	if (!vm) {
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VM_NOT_FOUND);
> +		vm = ERR_PTR(-EINVAL);
> +	} else if (!xe_vm_in_fault_mode(vm)) {
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VM_NOT_IN_FAULT_MODE);
>   		vm = ERR_PTR(-EINVAL);
> +	} else {
> +		xe_vm_get(vm);
> +	}
>   	up_read(&xe->usm.lock);
>   
>   	return vm;
> @@ -248,7 +267,6 @@ static struct xe_vm *xe_pagefault_asid_to_vm(struct xe_device *xe, u32 asid)
>   static int xe_pagefault_service(struct xe_pagefault *pf)
>   {
>   	struct xe_gt *gt = pf->gt;
> -	struct xe_device *xe = gt_to_xe(gt);
>   	struct xe_vm *vm;
>   	struct xe_vma *vma = NULL;
>   	int err;
> @@ -259,7 +277,7 @@ static int xe_pagefault_service(struct xe_pagefault *pf)
>   	if (pf->consumer.fault_type_level == XE_PAGEFAULT_TYPE_LEVEL_NACK)
>   		return -EFAULT;
>   
> -	vm = xe_pagefault_asid_to_vm(xe, asid);
> +	vm = xe_pagefault_asid_to_vm(pf, asid);
>   	if (IS_ERR(vm))
>   		return PTR_ERR(vm);
>   
> @@ -267,18 +285,21 @@ static int xe_pagefault_service(struct xe_pagefault *pf)
>   
>   	if (xe_vm_is_closed(vm)) {
>   		err = -ENOENT;
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VM_CLOSED);
>   		goto unlock_vm;
>   	}
>   
> -	vma = xe_vm_find_vma_by_addr(vm, pf->consumer.page_addr);
> +	vma = xe_vm_find_vma_by_addr(vm, xe_pagefault_addr(pf));
>   	if (!vma) {
>   		err = -EINVAL;
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VMA_NOT_FOUND);
>   		goto unlock_vm;
>   	}
>   
>   	if (xe_vma_read_only(vma) &&
>   	    pf->consumer.access_type != XE_PAGEFAULT_ACCESS_TYPE_READ) {
>   		err = -EPERM;
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_READ_ONLY_VIOLATION);
>   		goto unlock_vm;
>   	}
>   
> @@ -286,7 +307,7 @@ static int xe_pagefault_service(struct xe_pagefault *pf)
>   
>   	if (xe_vma_is_cpu_addr_mirror(vma))
>   		err = xe_svm_handle_pagefault(vm, vma, pf, gt,
> -					      pf->consumer.page_addr, atomic);
> +					      xe_pagefault_addr(pf), atomic);
>   	else
>   		err = xe_pagefault_handle_vma(gt, vma, pf, atomic);
>   
> @@ -375,7 +396,7 @@ static bool xe_pagefault_match(struct xe_pagefault *pf, u64 start,
>   			       u64 end, u64 cache_asid)
>   {
>   	struct xe_device *xe = gt_to_xe(pf->gt);
> -	u64 page_addr = pf->consumer.page_addr;
> +	u64 page_addr = xe_pagefault_addr(pf);
>   	u32 pf_asid = FIELD_GET(XE_PAGEFAULT_ASID_MASK, pf->consumer.id);
>   
>   	xe_assert(xe, pf->consumer.alloc_state !=
> @@ -499,7 +520,7 @@ static bool xe_pagefault_queue_pop(struct xe_pagefault_queue *pf_queue,
>   	if (FIELD_GET(XE_PAGEFAULT_REQUEUE_MASK,
>   		      lpf->consumer.fault_type_level))
>   		align = SZ_4K;
> -	pf_work->cache.start = ALIGN_DOWN(lpf->consumer.page_addr, align);
> +	pf_work->cache.start = ALIGN_DOWN(xe_pagefault_addr(lpf), align);
>   	pf_work->cache.end = pf_work->cache.start + align;
>   	pf_work->cache.asid = FIELD_GET(XE_PAGEFAULT_ASID_MASK, lpf->consumer.id);
>   	pf_work->cache.pf = lpf;
> @@ -542,10 +563,53 @@ static bool xe_pagefault_queue_pop(struct xe_pagefault_queue *pf_queue,
>   	return true;
>   }
>   
> +static const char *xe_pagefault_error_to_str(enum xe_pagefault_error error)
> +{
> +	switch (error) {
> +	case XE_PAGEFAULT_ERROR_NONE:
> +		return "NONE";
> +	case XE_PAGEFAULT_ERROR_VM_NOT_FOUND:
> +		return "VM_NOT_FOUND";
> +	case XE_PAGEFAULT_ERROR_VM_NOT_IN_FAULT_MODE:
> +		return "VM_NOT_IN_FAULT_MODE";
> +	case XE_PAGEFAULT_ERROR_VM_CLOSED:
> +		return "VM_CLOSED";
> +	case XE_PAGEFAULT_ERROR_VMA_NOT_FOUND:
> +		return "VMA_NOT_FOUND";
> +	case XE_PAGEFAULT_ERROR_READ_ONLY_VIOLATION:
> +		return "READ_ONLY_VIOLATION";
> +	case XE_PAGEFAULT_ERROR_VMA_NEEDS_VRAM_CHECK:
> +		return "VMA_NEEDS_VRAM_CHECK";
> +	case XE_PAGEFAULT_ERROR_VMA_ATOMIC_USERPTR:
> +		return "VMA_ATOMIC_USERPTR";
> +	case XE_PAGEFAULT_ERROR_VMA_USERPTR_PIN:
> +		return "VMA_USERPTR_PIN";
> +	case XE_PAGEFAULT_ERROR_VMA_VALIDATE:
> +		return "VMA_VALIDATE";
> +	case XE_PAGEFAULT_ERROR_VMA_REBIND:
> +		return "VMA_REBIND";
> +	case XE_PAGEFAULT_ERROR_SVM_GARBAGE_COLLECTOR:
> +		return "SVM_GARBAGE_COLLECTOR";
> +	case XE_PAGEFAULT_ERROR_SVM_RANGE_NOT_FOUND:
> +		return "SVM_RANGE_NOT_FOUND";
> +	case XE_PAGEFAULT_ERROR_SVM_REBIND:
> +		return "SVM_REBIND";
> +	case XE_PAGEFAULT_ERROR_SVM_NEEDS_VRAM_CHECK:
> +		return "SVM_NEEDS_VRAM_CHECK";
> +	case XE_PAGEFAULT_ERROR_SVM_VMA_NOT_FOUND:
> +		return "SVM_VMA_NOT_FOUND";
> +	case XE_PAGEFAULT_ERROR_SVM_SERVICE_FAILED:
> +		return "SVM_SERVICE_FAILED";
> +	default:
> +		return "UNKNOWN";
> +	}
> +}
> +
>   static void xe_pagefault_print(struct xe_pagefault *pf)
>   {
>   	u8 engine_class = FIELD_GET(XE_PAGEFAULT_ENGINE_CLASS_MASK,
>   				    pf->consumer.engine_class_instance);
> +	u64 addr = xe_pagefault_addr(pf);
>   
>   	xe_gt_info(pf->gt, "\n\tASID: %lu\n"
>   		   "\tFaulted Address: 0x%08x%08x\n"
> @@ -554,11 +618,12 @@ static void xe_pagefault_print(struct xe_pagefault *pf)
>   		   "\tFaultLevel: %lu\n"
>   		   "\tEngineClass: %d %s\n"
>   		   "\tEngineInstance: %lu\n"
> -		   "\tSRCID: 0x%02lx\n",
> +		   "\tSRCID: 0x%02lx\n"
> +		   "\tError: %s\n",
>   		   FIELD_GET(XE_PAGEFAULT_ASID_MASK,
>   			     pf->consumer.id),
> -		   upper_32_bits(pf->consumer.page_addr),
> -		   lower_32_bits(pf->consumer.page_addr),
> +		   upper_32_bits(addr),
> +		   lower_32_bits(addr),
>   		   FIELD_GET(XE_PAGEFAULT_TYPE_MASK,
>   			     pf->consumer.fault_type_level),
>   		   FIELD_GET(XE_PAGEFAULT_ACCESS_TYPE_MASK,
> @@ -570,7 +635,8 @@ static void xe_pagefault_print(struct xe_pagefault *pf)
>   		   FIELD_GET(XE_PAGEFAULT_ENGINE_INSTANCE_MASK,
>   			     pf->consumer.engine_class_instance),
>   		   FIELD_GET(XE_PAGEFAULT_SRCID_MASK,
> -			     pf->consumer.id));
> +			     pf->consumer.id),
> +		   xe_pagefault_error_to_str(xe_pagefault_get_error(pf)));
>   }
>   
>   static void xe_pagefault_save_to_vm(struct xe_device *xe, struct xe_pagefault *pf)
> diff --git a/drivers/gpu/drm/xe/xe_pagefault.h b/drivers/gpu/drm/xe/xe_pagefault.h
> index e9c5d1f03760..799c984dee84 100644
> --- a/drivers/gpu/drm/xe/xe_pagefault.h
> +++ b/drivers/gpu/drm/xe/xe_pagefault.h
> @@ -6,6 +6,8 @@
>   #ifndef _XE_PAGEFAULT_H_
>   #define _XE_PAGEFAULT_H_
>   
> +#include <linux/bitfield.h>
> +
>   #include "xe_pagefault_types.h"
>   
>   struct drm_printer;
> @@ -21,6 +23,61 @@ int xe_pagefault_handler(struct xe_device *xe, struct xe_pagefault *pf);
>   
>   void xe_pagefault_print_info(struct xe_device *xe, struct drm_printer *p);
>   
> +/*
> + * consumer.page_addr is always page (4K) aligned, so the low bits are
> + * reserved and unused by the real address. Steal a byte of those bits to
> + * record an &enum xe_pagefault_error describing the high-level point at
> + * which servicing of the fault failed, so it can be reported by
> + * xe_pagefault_print(). All real address consumers of page_addr must go
> + * through xe_pagefault_addr() to mask off these reserved bits.
> + */
> +#define XE_PAGEFAULT_ERROR_MASK		GENMASK_ULL(7, 0)
> +
> +/**
> + * xe_pagefault_set_error() - record the failure reason for a pagefault
> + * @pf: Pagefault entry
> + * @error: Failure reason
> + *
> + * Encodes @error into the reserved low bits of consumer.page_addr. Should be
> + * called at the high-level point a pagefault fails to service so
> + * xe_pagefault_print() can later report a mnemonic failure reason.
> + */
> +static inline void
> +xe_pagefault_set_error(struct xe_pagefault *pf, enum xe_pagefault_error error)
> +{
> +	pf->consumer.page_addr &= ~XE_PAGEFAULT_ERROR_MASK;
> +	pf->consumer.page_addr |= FIELD_PREP(XE_PAGEFAULT_ERROR_MASK, error);
> +}
> +
> +/**
> + * xe_pagefault_get_error() - read the failure reason for a pagefault
> + * @pf: Pagefault entry
> + *
> + * Return: The &enum xe_pagefault_error previously recorded via
> + * xe_pagefault_set_error(), or %XE_PAGEFAULT_ERROR_NONE if none was recorded.
> + */
> +static inline enum xe_pagefault_error
> +xe_pagefault_get_error(struct xe_pagefault *pf)
> +{
> +	return FIELD_GET(XE_PAGEFAULT_ERROR_MASK, pf->consumer.page_addr);
> +}
> +
> +/**
> + * xe_pagefault_addr() - read the real faulted address for a pagefault
> + * @pf: Pagefault entry
> + *
> + * consumer.page_addr may have failure reason bits encoded into its reserved
> + * low bits by xe_pagefault_set_error(). This masks those bits off, returning
> + * the real page address. All accesses to the faulted address must go through
> + * this helper rather than reading consumer.page_addr directly.
> + *
> + * Return: The real (page aligned) faulted address.
> + */
> +static inline u64 xe_pagefault_addr(struct xe_pagefault *pf)
> +{
> +	return pf->consumer.page_addr & ~XE_PAGEFAULT_ERROR_MASK;
> +}
> +
>   #define XE_PAGEFAULT_END_ADDR_MASK	(~0xfffull)
>   
>   /**
> @@ -69,11 +126,17 @@ static inline u64 xe_pagefault_end_addr(struct xe_pagefault *pf)
>    * The pagefault consumer stores the resolved fault range so subsequent faults
>    * hitting the same range can be immediately acknowledged without re-running
>    * the full fault handling path.
> + *
> + * The start address shares storage with the failure reason recorded by
> + * xe_pagefault_set_error() and therefore must be masked with
> + * %XE_PAGEFAULT_ERROR_MASK before storing so any previously recorded error is
> + * preserved.
>    */
>   static inline void
>   xe_pagefault_set_start_addr(struct xe_pagefault *pf, u64 start_addr)
>   {
> -	pf->consumer.page_addr = start_addr;
> +	pf->consumer.page_addr &= XE_PAGEFAULT_ERROR_MASK;
> +	pf->consumer.page_addr |= (start_addr & ~XE_PAGEFAULT_ERROR_MASK);
>   }
>   
>   /**
> @@ -87,7 +150,7 @@ xe_pagefault_set_start_addr(struct xe_pagefault *pf, u64 start_addr)
>    */
>   static inline u64 xe_pagefault_start_addr(struct xe_pagefault *pf)
>   {
> -	return pf->consumer.page_addr;
> +	return xe_pagefault_addr(pf);
>   }
>   
>   #endif
> diff --git a/drivers/gpu/drm/xe/xe_pagefault_types.h b/drivers/gpu/drm/xe/xe_pagefault_types.h
> index 907189b73286..8ae6b9848bdc 100644
> --- a/drivers/gpu/drm/xe/xe_pagefault_types.h
> +++ b/drivers/gpu/drm/xe/xe_pagefault_types.h
> @@ -32,6 +32,88 @@ enum xe_pagefault_type {
>   	XE_PAGEFAULT_TYPE_ATOMIC_ACCESS_VIOLATION	= 2,
>   };
>   
> +/**
> + * enum xe_pagefault_error - Xe page fault servicing error
> + *
> + * Uniquely identifies the high-level point at which servicing of a page
> + * fault failed. Encoded into the reserved low bits of
> + * &xe_pagefault.consumer.page_addr (which is always page aligned) so the
> + * failure reason can be threaded back up to xe_pagefault_print() without
> + * growing the size of struct xe_pagefault. See xe_pagefault_set_error() and
> + * xe_pagefault_error_to_str().
> + */
> +enum xe_pagefault_error {
> +	/** @XE_PAGEFAULT_ERROR_NONE: No error recorded */
> +	XE_PAGEFAULT_ERROR_NONE = 0,
> +	/**
> +	 * @XE_PAGEFAULT_ERROR_VM_NOT_FOUND: VM lookup by ASID failed, e.g.
> +	 * the VM's file descriptor was already closed and the ASID has been
> +	 * torn down
> +	 */
> +	XE_PAGEFAULT_ERROR_VM_NOT_FOUND,
> +	/**
> +	 * @XE_PAGEFAULT_ERROR_VM_NOT_IN_FAULT_MODE: VM found by ASID lookup
> +	 * but is not in fault mode
> +	 */
> +	XE_PAGEFAULT_ERROR_VM_NOT_IN_FAULT_MODE,
> +	/** @XE_PAGEFAULT_ERROR_VM_CLOSED: VM found but already closed */
> +	XE_PAGEFAULT_ERROR_VM_CLOSED,
> +	/** @XE_PAGEFAULT_ERROR_VMA_NOT_FOUND: No VMA covers the faulted address */
> +	XE_PAGEFAULT_ERROR_VMA_NOT_FOUND,
> +	/**
> +	 * @XE_PAGEFAULT_ERROR_READ_ONLY_VIOLATION: Write/atomic fault on a
> +	 * read-only VMA
> +	 */
> +	XE_PAGEFAULT_ERROR_READ_ONLY_VIOLATION,
> +	/**
> +	 * @XE_PAGEFAULT_ERROR_VMA_NEEDS_VRAM_CHECK: Failed determining if VMA
> +	 * requires VRAM for an atomic access
> +	 */
> +	XE_PAGEFAULT_ERROR_VMA_NEEDS_VRAM_CHECK,
> +	/**
> +	 * @XE_PAGEFAULT_ERROR_VMA_ATOMIC_USERPTR: Atomic access requires VRAM
> +	 * but VMA is a userptr, which is unsupported
> +	 */
> +	XE_PAGEFAULT_ERROR_VMA_ATOMIC_USERPTR,
> +	/** @XE_PAGEFAULT_ERROR_VMA_USERPTR_PIN: Userptr page pin/repin failed */
> +	XE_PAGEFAULT_ERROR_VMA_USERPTR_PIN,
> +	/**
> +	 * @XE_PAGEFAULT_ERROR_VMA_VALIDATE: Failed to lock/validate VMA's BO
> +	 * or migrate it to VRAM
> +	 */
> +	XE_PAGEFAULT_ERROR_VMA_VALIDATE,
> +	/** @XE_PAGEFAULT_ERROR_VMA_REBIND: Failed to rebind VMA into page tables */
> +	XE_PAGEFAULT_ERROR_VMA_REBIND,
> +	/**
> +	 * @XE_PAGEFAULT_ERROR_SVM_GARBAGE_COLLECTOR: Failed processing
> +	 * pending SVM garbage collection (unmaps) prior to servicing the
> +	 * fault
> +	 */
> +	XE_PAGEFAULT_ERROR_SVM_GARBAGE_COLLECTOR,
> +	/**
> +	 * @XE_PAGEFAULT_ERROR_SVM_RANGE_NOT_FOUND: Failed to find or insert
> +	 * an SVM range covering the faulted address
> +	 */
> +	XE_PAGEFAULT_ERROR_SVM_RANGE_NOT_FOUND,
> +	/** @XE_PAGEFAULT_ERROR_SVM_REBIND: Failed to rebind an SVM range into page tables */
> +	XE_PAGEFAULT_ERROR_SVM_REBIND,
> +	/**
> +	 * @XE_PAGEFAULT_ERROR_SVM_NEEDS_VRAM_CHECK: Failed determining if SVM
> +	 * VMA requires VRAM for an atomic access
> +	 */
> +	XE_PAGEFAULT_ERROR_SVM_NEEDS_VRAM_CHECK,
> +	/**
> +	 * @XE_PAGEFAULT_ERROR_SVM_VMA_NOT_FOUND: SVM VMA re-lookup after a
> +	 * range split failed to find a covering VMA
> +	 */
> +	XE_PAGEFAULT_ERROR_SVM_VMA_NOT_FOUND,
> +	/**
> +	 * @XE_PAGEFAULT_ERROR_SVM_SERVICE_FAILED: SVM range population,
> +	 * migration, or bind failed
> +	 */
> +	XE_PAGEFAULT_ERROR_SVM_SERVICE_FAILED,
> +};
> +
>   /** struct xe_pagefault_ops - Xe pagefault ops (producer) */
>   struct xe_pagefault_ops {
>   	/**
> diff --git a/drivers/gpu/drm/xe/xe_svm.c b/drivers/gpu/drm/xe/xe_svm.c
> index 6c3033fc4db7..f39e647512ad 100644
> --- a/drivers/gpu/drm/xe/xe_svm.c
> +++ b/drivers/gpu/drm/xe/xe_svm.c
> @@ -1304,16 +1304,20 @@ static int __xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
>   
>   	/* Always process UNMAPs first so view SVM ranges is current */
>   	err = xe_svm_garbage_collector(vm);
> -	if (err)
> +	if (err) {
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_SVM_GARBAGE_COLLECTOR);
>   		return err;
> +	}
>   
>   	dpagemap = ctx.devmem_only ? xe_tile_local_pagemap(tile) :
>   		xe_vma_resolve_pagemap(vma, tile);
>   	ctx.device_private_page_owner = xe_svm_private_page_owner(vm, !dpagemap);
>   	range = xe_svm_range_find_or_insert(vm, fault_addr, vma, &ctx);
>   
> -	if (IS_ERR(range))
> +	if (IS_ERR(range)) {
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_SVM_RANGE_NOT_FOUND);
>   		return PTR_ERR(range);
> +	}
>   
>   	xe_svm_range_fault_count_stats_incr(gt, range);
>   
> @@ -1415,6 +1419,7 @@ static int __xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
>   			err = PTR_ERR(fence);
>   			xe_validation_retry_on_oom(&vctx, &err);
>   			xe_svm_range_bind_us_stats_incr(gt, range, bind_start);
> +			xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_SVM_REBIND);
>   			break;
>   		}
>   	}
> @@ -1437,6 +1442,7 @@ static int __xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
>   
>   err_out:
>   	if (err == -EAGAIN) {
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_NONE);
>   		ctx.timeslice_ms <<= 1;	/* Double timeslice if we have to retry */
>   		range_debug(range, "PAGE FAULT - RETRY BIND");
>   		goto retry;
> @@ -1469,8 +1475,10 @@ int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
>   	int need_vram, ret;
>   retry:
>   	need_vram = xe_vma_need_vram_for_atomic(vm->xe, vma, atomic);
> -	if (need_vram < 0)
> +	if (need_vram < 0) {
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_SVM_NEEDS_VRAM_CHECK);
>   		return need_vram;
> +	}
>   
>   	ret =  __xe_svm_handle_pagefault(vm, vma, pf, gt, fault_addr,
>   					 need_vram ? true : false);
> @@ -1480,11 +1488,16 @@ int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
>   		 * may have been split by xe_svm_range_set_default_attr.
>   		 */
>   		vma = xe_vm_find_vma_by_addr(vm, fault_addr);
> -		if (!vma)
> +		if (!vma) {
> +			xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_SVM_VMA_NOT_FOUND);
>   			return -EINVAL;
> +		}
>   
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_NONE);
>   		goto retry;
>   	}
> +	if (ret && xe_pagefault_get_error(pf) == XE_PAGEFAULT_ERROR_NONE)
> +		xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_SVM_SERVICE_FAILED);
>   	return ret;
>   }

LGTM
Reviewed-by: Himal Prasad Ghimiray <himal.prasad.ghimiray@intel.com>

>   
> diff --git a/drivers/gpu/drm/xe/xe_vm.c b/drivers/gpu/drm/xe/xe_vm.c
> index efa5ff6cc823..17dc4debe7c1 100644
> --- a/drivers/gpu/drm/xe/xe_vm.c
> +++ b/drivers/gpu/drm/xe/xe_vm.c
> @@ -29,6 +29,7 @@
>   #include "xe_exec_queue.h"
>   #include "xe_gt.h"
>   #include "xe_migrate.h"
> +#include "xe_pagefault.h"
>   #include "xe_pat.h"
>   #include "xe_pm.h"
>   #include "xe_preempt_fence.h"
> @@ -643,7 +644,7 @@ void xe_vm_add_fault_entry_pf(struct xe_vm *vm, struct xe_pagefault *pf)
>   		return;
>   	}
>   
> -	e->address = pf->consumer.page_addr;
> +	e->address = xe_pagefault_addr(pf);
>   	/*
>   	 * TODO:
>   	 * Address precision is currently always SZ_4K, but this may change


  parent reply	other threads:[~2026-09-18  3:48 UTC|newest]

Thread overview: 5+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-17 20:38 [PATCH v3] drm/xe: Add mnemonic error reason to page fault diagnostics Matthew Brost
2026-09-17 22:52 ` ✓ CI.KUnit: success for drm/xe: Add mnemonic error reason to page fault diagnostics (rev3) Patchwork
2026-09-18  0:02 ` ✓ Xe.CI.BAT: " Patchwork
2026-09-18  3:47 ` Ghimiray, Himal Prasad [this message]
2026-09-18  5:15 ` ✓ Xe.CI.FULL: " Patchwork

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=cc53d16b-6542-45ea-8793-73376aac6cf8@intel.com \
    --to=himal.prasad.ghimiray@intel.com \
    --cc=intel-xe@lists.freedesktop.org \
    --cc=matthew.brost@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox