From: Matthew Brost <matthew.brost@intel.com>
To: intel-xe@lists.freedesktop.org
Subject: [PATCH v3] drm/xe: Add mnemonic error reason to page fault diagnostics
Date: Thu, 17 Sep 2026 13:38:50 -0700 [thread overview]
Message-ID: <20260917203850.321385-1-matthew.brost@intel.com> (raw)
xe_pagefault_print() was firing without indicating why servicing of a
fault failed, making it hard to distinguish benign races (e.g. a VM's
file descriptor closing mid-fault) from real bugs.
Add enum xe_pagefault_error, uniquely identifying the high-level point
at which xe_pagefault_service(), xe_pagefault_handle_vma(), and
xe_svm_handle_pagefault() fail. Since consumer.page_addr is always
page (4K) aligned, steal the reserved low byte to carry this code
without growing struct xe_pagefault. Add xe_pagefault_set_error() /
xe_pagefault_get_error() to encode/decode it, and xe_pagefault_addr()
as the single accessor for the real (masked) faulted address; update
all real address consumers (xe_pagefault_match(), cache alignment,
xe_vm_add_fault_entry_pf()) to go through it instead of reading
consumer.page_addr directly. xe_pagefault_set_start_addr() now
preserves any already-recorded error bits rather than clobbering them.
xe_pagefault_print() decodes the error via a new
xe_pagefault_error_to_str() and reports it in an added "Error:" line.
xe_pagefault_asid_to_vm() now distinguishes an ASID with no VM at all
(XE_PAGEFAULT_ERROR_VM_NOT_FOUND, e.g. the owning file descriptor was
already closed) from a VM found but not in fault mode
(XE_PAGEFAULT_ERROR_VM_NOT_IN_FAULT_MODE).
Signed-off-by: Matthew Brost <matthew.brost@intel.com>
Assisted-by: GitHub_Copilot:claude-sonnet-5
---
v2:
- Add more XE_PAGEFAULT_ERROR_* types for SVM
v3:
- Reset to XE_PAGEFAULT_ERROR_NONE on retries (Sashiko)
---
drivers/gpu/drm/xe/xe_pagefault.c | 102 +++++++++++++++++++-----
drivers/gpu/drm/xe/xe_pagefault.h | 67 +++++++++++++++-
drivers/gpu/drm/xe/xe_pagefault_types.h | 82 +++++++++++++++++++
drivers/gpu/drm/xe/xe_svm.c | 21 ++++-
drivers/gpu/drm/xe/xe_vm.c | 3 +-
5 files changed, 250 insertions(+), 25 deletions(-)
diff --git a/drivers/gpu/drm/xe/xe_pagefault.c b/drivers/gpu/drm/xe/xe_pagefault.c
index aeb56ff5d58e..c4aafdbdb518 100644
--- a/drivers/gpu/drm/xe/xe_pagefault.c
+++ b/drivers/gpu/drm/xe/xe_pagefault.c
@@ -158,8 +158,14 @@ static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma,
lockdep_assert_held(&vm->lock);
needs_vram = xe_vma_need_vram_for_atomic(vm->xe, vma, atomic);
- if (needs_vram < 0 || (needs_vram && xe_vma_is_userptr(vma)))
- return needs_vram < 0 ? needs_vram : -EACCES;
+ if (needs_vram < 0) {
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VMA_NEEDS_VRAM_CHECK);
+ return needs_vram;
+ }
+ if (needs_vram && xe_vma_is_userptr(vma)) {
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VMA_ATOMIC_USERPTR);
+ return -EACCES;
+ }
xe_gt_stats_incr(gt, XE_GT_STATS_ID_VMA_PAGEFAULT_COUNT, 1);
xe_gt_stats_incr(gt, XE_GT_STATS_ID_VMA_PAGEFAULT_KB,
@@ -178,13 +184,17 @@ static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma,
}
do {
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_NONE);
+
if (xe_vma_is_userptr(vma) &&
xe_vma_userptr_check_repin(to_userptr_vma(vma))) {
struct xe_userptr_vma *uvma = to_userptr_vma(vma);
err = xe_vma_userptr_pin_pages(uvma);
- if (err)
+ if (err) {
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VMA_USERPTR_PIN);
return err;
+ }
}
/* Lock VM and BOs dma-resv */
@@ -195,8 +205,10 @@ static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma,
needs_vram == 1);
drm_exec_retry_on_contention(&exec);
xe_validation_retry_on_oom(&ctx, &err);
- if (err)
+ if (err) {
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VMA_VALIDATE);
break;
+ }
/* Bind VMA only to the GT that has faulted */
trace_xe_vma_pf_bind(vma);
@@ -206,6 +218,7 @@ static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma,
if (IS_ERR(fence)) {
err = PTR_ERR(fence);
xe_validation_retry_on_oom(&ctx, &err);
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VMA_REBIND);
break;
}
}
@@ -230,16 +243,22 @@ xe_pagefault_access_is_atomic(enum xe_pagefault_access_type access_type)
return (access_type & XE_PAGEFAULT_ACCESS_TYPE_MASK) == XE_PAGEFAULT_ACCESS_TYPE_ATOMIC;
}
-static struct xe_vm *xe_pagefault_asid_to_vm(struct xe_device *xe, u32 asid)
+static struct xe_vm *xe_pagefault_asid_to_vm(struct xe_pagefault *pf, u32 asid)
{
+ struct xe_device *xe = gt_to_xe(pf->gt);
struct xe_vm *vm;
down_read(&xe->usm.lock);
vm = xa_load(&xe->usm.asid_to_vm, asid);
- if (vm && xe_vm_in_fault_mode(vm))
- xe_vm_get(vm);
- else
+ if (!vm) {
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VM_NOT_FOUND);
+ vm = ERR_PTR(-EINVAL);
+ } else if (!xe_vm_in_fault_mode(vm)) {
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VM_NOT_IN_FAULT_MODE);
vm = ERR_PTR(-EINVAL);
+ } else {
+ xe_vm_get(vm);
+ }
up_read(&xe->usm.lock);
return vm;
@@ -248,7 +267,6 @@ static struct xe_vm *xe_pagefault_asid_to_vm(struct xe_device *xe, u32 asid)
static int xe_pagefault_service(struct xe_pagefault *pf)
{
struct xe_gt *gt = pf->gt;
- struct xe_device *xe = gt_to_xe(gt);
struct xe_vm *vm;
struct xe_vma *vma = NULL;
int err;
@@ -259,7 +277,7 @@ static int xe_pagefault_service(struct xe_pagefault *pf)
if (pf->consumer.fault_type_level == XE_PAGEFAULT_TYPE_LEVEL_NACK)
return -EFAULT;
- vm = xe_pagefault_asid_to_vm(xe, asid);
+ vm = xe_pagefault_asid_to_vm(pf, asid);
if (IS_ERR(vm))
return PTR_ERR(vm);
@@ -267,18 +285,21 @@ static int xe_pagefault_service(struct xe_pagefault *pf)
if (xe_vm_is_closed(vm)) {
err = -ENOENT;
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VM_CLOSED);
goto unlock_vm;
}
- vma = xe_vm_find_vma_by_addr(vm, pf->consumer.page_addr);
+ vma = xe_vm_find_vma_by_addr(vm, xe_pagefault_addr(pf));
if (!vma) {
err = -EINVAL;
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_VMA_NOT_FOUND);
goto unlock_vm;
}
if (xe_vma_read_only(vma) &&
pf->consumer.access_type != XE_PAGEFAULT_ACCESS_TYPE_READ) {
err = -EPERM;
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_READ_ONLY_VIOLATION);
goto unlock_vm;
}
@@ -286,7 +307,7 @@ static int xe_pagefault_service(struct xe_pagefault *pf)
if (xe_vma_is_cpu_addr_mirror(vma))
err = xe_svm_handle_pagefault(vm, vma, pf, gt,
- pf->consumer.page_addr, atomic);
+ xe_pagefault_addr(pf), atomic);
else
err = xe_pagefault_handle_vma(gt, vma, pf, atomic);
@@ -375,7 +396,7 @@ static bool xe_pagefault_match(struct xe_pagefault *pf, u64 start,
u64 end, u64 cache_asid)
{
struct xe_device *xe = gt_to_xe(pf->gt);
- u64 page_addr = pf->consumer.page_addr;
+ u64 page_addr = xe_pagefault_addr(pf);
u32 pf_asid = FIELD_GET(XE_PAGEFAULT_ASID_MASK, pf->consumer.id);
xe_assert(xe, pf->consumer.alloc_state !=
@@ -499,7 +520,7 @@ static bool xe_pagefault_queue_pop(struct xe_pagefault_queue *pf_queue,
if (FIELD_GET(XE_PAGEFAULT_REQUEUE_MASK,
lpf->consumer.fault_type_level))
align = SZ_4K;
- pf_work->cache.start = ALIGN_DOWN(lpf->consumer.page_addr, align);
+ pf_work->cache.start = ALIGN_DOWN(xe_pagefault_addr(lpf), align);
pf_work->cache.end = pf_work->cache.start + align;
pf_work->cache.asid = FIELD_GET(XE_PAGEFAULT_ASID_MASK, lpf->consumer.id);
pf_work->cache.pf = lpf;
@@ -542,10 +563,53 @@ static bool xe_pagefault_queue_pop(struct xe_pagefault_queue *pf_queue,
return true;
}
+static const char *xe_pagefault_error_to_str(enum xe_pagefault_error error)
+{
+ switch (error) {
+ case XE_PAGEFAULT_ERROR_NONE:
+ return "NONE";
+ case XE_PAGEFAULT_ERROR_VM_NOT_FOUND:
+ return "VM_NOT_FOUND";
+ case XE_PAGEFAULT_ERROR_VM_NOT_IN_FAULT_MODE:
+ return "VM_NOT_IN_FAULT_MODE";
+ case XE_PAGEFAULT_ERROR_VM_CLOSED:
+ return "VM_CLOSED";
+ case XE_PAGEFAULT_ERROR_VMA_NOT_FOUND:
+ return "VMA_NOT_FOUND";
+ case XE_PAGEFAULT_ERROR_READ_ONLY_VIOLATION:
+ return "READ_ONLY_VIOLATION";
+ case XE_PAGEFAULT_ERROR_VMA_NEEDS_VRAM_CHECK:
+ return "VMA_NEEDS_VRAM_CHECK";
+ case XE_PAGEFAULT_ERROR_VMA_ATOMIC_USERPTR:
+ return "VMA_ATOMIC_USERPTR";
+ case XE_PAGEFAULT_ERROR_VMA_USERPTR_PIN:
+ return "VMA_USERPTR_PIN";
+ case XE_PAGEFAULT_ERROR_VMA_VALIDATE:
+ return "VMA_VALIDATE";
+ case XE_PAGEFAULT_ERROR_VMA_REBIND:
+ return "VMA_REBIND";
+ case XE_PAGEFAULT_ERROR_SVM_GARBAGE_COLLECTOR:
+ return "SVM_GARBAGE_COLLECTOR";
+ case XE_PAGEFAULT_ERROR_SVM_RANGE_NOT_FOUND:
+ return "SVM_RANGE_NOT_FOUND";
+ case XE_PAGEFAULT_ERROR_SVM_REBIND:
+ return "SVM_REBIND";
+ case XE_PAGEFAULT_ERROR_SVM_NEEDS_VRAM_CHECK:
+ return "SVM_NEEDS_VRAM_CHECK";
+ case XE_PAGEFAULT_ERROR_SVM_VMA_NOT_FOUND:
+ return "SVM_VMA_NOT_FOUND";
+ case XE_PAGEFAULT_ERROR_SVM_SERVICE_FAILED:
+ return "SVM_SERVICE_FAILED";
+ default:
+ return "UNKNOWN";
+ }
+}
+
static void xe_pagefault_print(struct xe_pagefault *pf)
{
u8 engine_class = FIELD_GET(XE_PAGEFAULT_ENGINE_CLASS_MASK,
pf->consumer.engine_class_instance);
+ u64 addr = xe_pagefault_addr(pf);
xe_gt_info(pf->gt, "\n\tASID: %lu\n"
"\tFaulted Address: 0x%08x%08x\n"
@@ -554,11 +618,12 @@ static void xe_pagefault_print(struct xe_pagefault *pf)
"\tFaultLevel: %lu\n"
"\tEngineClass: %d %s\n"
"\tEngineInstance: %lu\n"
- "\tSRCID: 0x%02lx\n",
+ "\tSRCID: 0x%02lx\n"
+ "\tError: %s\n",
FIELD_GET(XE_PAGEFAULT_ASID_MASK,
pf->consumer.id),
- upper_32_bits(pf->consumer.page_addr),
- lower_32_bits(pf->consumer.page_addr),
+ upper_32_bits(addr),
+ lower_32_bits(addr),
FIELD_GET(XE_PAGEFAULT_TYPE_MASK,
pf->consumer.fault_type_level),
FIELD_GET(XE_PAGEFAULT_ACCESS_TYPE_MASK,
@@ -570,7 +635,8 @@ static void xe_pagefault_print(struct xe_pagefault *pf)
FIELD_GET(XE_PAGEFAULT_ENGINE_INSTANCE_MASK,
pf->consumer.engine_class_instance),
FIELD_GET(XE_PAGEFAULT_SRCID_MASK,
- pf->consumer.id));
+ pf->consumer.id),
+ xe_pagefault_error_to_str(xe_pagefault_get_error(pf)));
}
static void xe_pagefault_save_to_vm(struct xe_device *xe, struct xe_pagefault *pf)
diff --git a/drivers/gpu/drm/xe/xe_pagefault.h b/drivers/gpu/drm/xe/xe_pagefault.h
index e9c5d1f03760..799c984dee84 100644
--- a/drivers/gpu/drm/xe/xe_pagefault.h
+++ b/drivers/gpu/drm/xe/xe_pagefault.h
@@ -6,6 +6,8 @@
#ifndef _XE_PAGEFAULT_H_
#define _XE_PAGEFAULT_H_
+#include <linux/bitfield.h>
+
#include "xe_pagefault_types.h"
struct drm_printer;
@@ -21,6 +23,61 @@ int xe_pagefault_handler(struct xe_device *xe, struct xe_pagefault *pf);
void xe_pagefault_print_info(struct xe_device *xe, struct drm_printer *p);
+/*
+ * consumer.page_addr is always page (4K) aligned, so the low bits are
+ * reserved and unused by the real address. Steal a byte of those bits to
+ * record an &enum xe_pagefault_error describing the high-level point at
+ * which servicing of the fault failed, so it can be reported by
+ * xe_pagefault_print(). All real address consumers of page_addr must go
+ * through xe_pagefault_addr() to mask off these reserved bits.
+ */
+#define XE_PAGEFAULT_ERROR_MASK GENMASK_ULL(7, 0)
+
+/**
+ * xe_pagefault_set_error() - record the failure reason for a pagefault
+ * @pf: Pagefault entry
+ * @error: Failure reason
+ *
+ * Encodes @error into the reserved low bits of consumer.page_addr. Should be
+ * called at the high-level point a pagefault fails to service so
+ * xe_pagefault_print() can later report a mnemonic failure reason.
+ */
+static inline void
+xe_pagefault_set_error(struct xe_pagefault *pf, enum xe_pagefault_error error)
+{
+ pf->consumer.page_addr &= ~XE_PAGEFAULT_ERROR_MASK;
+ pf->consumer.page_addr |= FIELD_PREP(XE_PAGEFAULT_ERROR_MASK, error);
+}
+
+/**
+ * xe_pagefault_get_error() - read the failure reason for a pagefault
+ * @pf: Pagefault entry
+ *
+ * Return: The &enum xe_pagefault_error previously recorded via
+ * xe_pagefault_set_error(), or %XE_PAGEFAULT_ERROR_NONE if none was recorded.
+ */
+static inline enum xe_pagefault_error
+xe_pagefault_get_error(struct xe_pagefault *pf)
+{
+ return FIELD_GET(XE_PAGEFAULT_ERROR_MASK, pf->consumer.page_addr);
+}
+
+/**
+ * xe_pagefault_addr() - read the real faulted address for a pagefault
+ * @pf: Pagefault entry
+ *
+ * consumer.page_addr may have failure reason bits encoded into its reserved
+ * low bits by xe_pagefault_set_error(). This masks those bits off, returning
+ * the real page address. All accesses to the faulted address must go through
+ * this helper rather than reading consumer.page_addr directly.
+ *
+ * Return: The real (page aligned) faulted address.
+ */
+static inline u64 xe_pagefault_addr(struct xe_pagefault *pf)
+{
+ return pf->consumer.page_addr & ~XE_PAGEFAULT_ERROR_MASK;
+}
+
#define XE_PAGEFAULT_END_ADDR_MASK (~0xfffull)
/**
@@ -69,11 +126,17 @@ static inline u64 xe_pagefault_end_addr(struct xe_pagefault *pf)
* The pagefault consumer stores the resolved fault range so subsequent faults
* hitting the same range can be immediately acknowledged without re-running
* the full fault handling path.
+ *
+ * The start address shares storage with the failure reason recorded by
+ * xe_pagefault_set_error() and therefore must be masked with
+ * %XE_PAGEFAULT_ERROR_MASK before storing so any previously recorded error is
+ * preserved.
*/
static inline void
xe_pagefault_set_start_addr(struct xe_pagefault *pf, u64 start_addr)
{
- pf->consumer.page_addr = start_addr;
+ pf->consumer.page_addr &= XE_PAGEFAULT_ERROR_MASK;
+ pf->consumer.page_addr |= (start_addr & ~XE_PAGEFAULT_ERROR_MASK);
}
/**
@@ -87,7 +150,7 @@ xe_pagefault_set_start_addr(struct xe_pagefault *pf, u64 start_addr)
*/
static inline u64 xe_pagefault_start_addr(struct xe_pagefault *pf)
{
- return pf->consumer.page_addr;
+ return xe_pagefault_addr(pf);
}
#endif
diff --git a/drivers/gpu/drm/xe/xe_pagefault_types.h b/drivers/gpu/drm/xe/xe_pagefault_types.h
index 907189b73286..8ae6b9848bdc 100644
--- a/drivers/gpu/drm/xe/xe_pagefault_types.h
+++ b/drivers/gpu/drm/xe/xe_pagefault_types.h
@@ -32,6 +32,88 @@ enum xe_pagefault_type {
XE_PAGEFAULT_TYPE_ATOMIC_ACCESS_VIOLATION = 2,
};
+/**
+ * enum xe_pagefault_error - Xe page fault servicing error
+ *
+ * Uniquely identifies the high-level point at which servicing of a page
+ * fault failed. Encoded into the reserved low bits of
+ * &xe_pagefault.consumer.page_addr (which is always page aligned) so the
+ * failure reason can be threaded back up to xe_pagefault_print() without
+ * growing the size of struct xe_pagefault. See xe_pagefault_set_error() and
+ * xe_pagefault_error_to_str().
+ */
+enum xe_pagefault_error {
+ /** @XE_PAGEFAULT_ERROR_NONE: No error recorded */
+ XE_PAGEFAULT_ERROR_NONE = 0,
+ /**
+ * @XE_PAGEFAULT_ERROR_VM_NOT_FOUND: VM lookup by ASID failed, e.g.
+ * the VM's file descriptor was already closed and the ASID has been
+ * torn down
+ */
+ XE_PAGEFAULT_ERROR_VM_NOT_FOUND,
+ /**
+ * @XE_PAGEFAULT_ERROR_VM_NOT_IN_FAULT_MODE: VM found by ASID lookup
+ * but is not in fault mode
+ */
+ XE_PAGEFAULT_ERROR_VM_NOT_IN_FAULT_MODE,
+ /** @XE_PAGEFAULT_ERROR_VM_CLOSED: VM found but already closed */
+ XE_PAGEFAULT_ERROR_VM_CLOSED,
+ /** @XE_PAGEFAULT_ERROR_VMA_NOT_FOUND: No VMA covers the faulted address */
+ XE_PAGEFAULT_ERROR_VMA_NOT_FOUND,
+ /**
+ * @XE_PAGEFAULT_ERROR_READ_ONLY_VIOLATION: Write/atomic fault on a
+ * read-only VMA
+ */
+ XE_PAGEFAULT_ERROR_READ_ONLY_VIOLATION,
+ /**
+ * @XE_PAGEFAULT_ERROR_VMA_NEEDS_VRAM_CHECK: Failed determining if VMA
+ * requires VRAM for an atomic access
+ */
+ XE_PAGEFAULT_ERROR_VMA_NEEDS_VRAM_CHECK,
+ /**
+ * @XE_PAGEFAULT_ERROR_VMA_ATOMIC_USERPTR: Atomic access requires VRAM
+ * but VMA is a userptr, which is unsupported
+ */
+ XE_PAGEFAULT_ERROR_VMA_ATOMIC_USERPTR,
+ /** @XE_PAGEFAULT_ERROR_VMA_USERPTR_PIN: Userptr page pin/repin failed */
+ XE_PAGEFAULT_ERROR_VMA_USERPTR_PIN,
+ /**
+ * @XE_PAGEFAULT_ERROR_VMA_VALIDATE: Failed to lock/validate VMA's BO
+ * or migrate it to VRAM
+ */
+ XE_PAGEFAULT_ERROR_VMA_VALIDATE,
+ /** @XE_PAGEFAULT_ERROR_VMA_REBIND: Failed to rebind VMA into page tables */
+ XE_PAGEFAULT_ERROR_VMA_REBIND,
+ /**
+ * @XE_PAGEFAULT_ERROR_SVM_GARBAGE_COLLECTOR: Failed processing
+ * pending SVM garbage collection (unmaps) prior to servicing the
+ * fault
+ */
+ XE_PAGEFAULT_ERROR_SVM_GARBAGE_COLLECTOR,
+ /**
+ * @XE_PAGEFAULT_ERROR_SVM_RANGE_NOT_FOUND: Failed to find or insert
+ * an SVM range covering the faulted address
+ */
+ XE_PAGEFAULT_ERROR_SVM_RANGE_NOT_FOUND,
+ /** @XE_PAGEFAULT_ERROR_SVM_REBIND: Failed to rebind an SVM range into page tables */
+ XE_PAGEFAULT_ERROR_SVM_REBIND,
+ /**
+ * @XE_PAGEFAULT_ERROR_SVM_NEEDS_VRAM_CHECK: Failed determining if SVM
+ * VMA requires VRAM for an atomic access
+ */
+ XE_PAGEFAULT_ERROR_SVM_NEEDS_VRAM_CHECK,
+ /**
+ * @XE_PAGEFAULT_ERROR_SVM_VMA_NOT_FOUND: SVM VMA re-lookup after a
+ * range split failed to find a covering VMA
+ */
+ XE_PAGEFAULT_ERROR_SVM_VMA_NOT_FOUND,
+ /**
+ * @XE_PAGEFAULT_ERROR_SVM_SERVICE_FAILED: SVM range population,
+ * migration, or bind failed
+ */
+ XE_PAGEFAULT_ERROR_SVM_SERVICE_FAILED,
+};
+
/** struct xe_pagefault_ops - Xe pagefault ops (producer) */
struct xe_pagefault_ops {
/**
diff --git a/drivers/gpu/drm/xe/xe_svm.c b/drivers/gpu/drm/xe/xe_svm.c
index 6c3033fc4db7..f39e647512ad 100644
--- a/drivers/gpu/drm/xe/xe_svm.c
+++ b/drivers/gpu/drm/xe/xe_svm.c
@@ -1304,16 +1304,20 @@ static int __xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
/* Always process UNMAPs first so view SVM ranges is current */
err = xe_svm_garbage_collector(vm);
- if (err)
+ if (err) {
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_SVM_GARBAGE_COLLECTOR);
return err;
+ }
dpagemap = ctx.devmem_only ? xe_tile_local_pagemap(tile) :
xe_vma_resolve_pagemap(vma, tile);
ctx.device_private_page_owner = xe_svm_private_page_owner(vm, !dpagemap);
range = xe_svm_range_find_or_insert(vm, fault_addr, vma, &ctx);
- if (IS_ERR(range))
+ if (IS_ERR(range)) {
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_SVM_RANGE_NOT_FOUND);
return PTR_ERR(range);
+ }
xe_svm_range_fault_count_stats_incr(gt, range);
@@ -1415,6 +1419,7 @@ static int __xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
err = PTR_ERR(fence);
xe_validation_retry_on_oom(&vctx, &err);
xe_svm_range_bind_us_stats_incr(gt, range, bind_start);
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_SVM_REBIND);
break;
}
}
@@ -1437,6 +1442,7 @@ static int __xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
err_out:
if (err == -EAGAIN) {
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_NONE);
ctx.timeslice_ms <<= 1; /* Double timeslice if we have to retry */
range_debug(range, "PAGE FAULT - RETRY BIND");
goto retry;
@@ -1469,8 +1475,10 @@ int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
int need_vram, ret;
retry:
need_vram = xe_vma_need_vram_for_atomic(vm->xe, vma, atomic);
- if (need_vram < 0)
+ if (need_vram < 0) {
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_SVM_NEEDS_VRAM_CHECK);
return need_vram;
+ }
ret = __xe_svm_handle_pagefault(vm, vma, pf, gt, fault_addr,
need_vram ? true : false);
@@ -1480,11 +1488,16 @@ int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
* may have been split by xe_svm_range_set_default_attr.
*/
vma = xe_vm_find_vma_by_addr(vm, fault_addr);
- if (!vma)
+ if (!vma) {
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_SVM_VMA_NOT_FOUND);
return -EINVAL;
+ }
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_NONE);
goto retry;
}
+ if (ret && xe_pagefault_get_error(pf) == XE_PAGEFAULT_ERROR_NONE)
+ xe_pagefault_set_error(pf, XE_PAGEFAULT_ERROR_SVM_SERVICE_FAILED);
return ret;
}
diff --git a/drivers/gpu/drm/xe/xe_vm.c b/drivers/gpu/drm/xe/xe_vm.c
index efa5ff6cc823..17dc4debe7c1 100644
--- a/drivers/gpu/drm/xe/xe_vm.c
+++ b/drivers/gpu/drm/xe/xe_vm.c
@@ -29,6 +29,7 @@
#include "xe_exec_queue.h"
#include "xe_gt.h"
#include "xe_migrate.h"
+#include "xe_pagefault.h"
#include "xe_pat.h"
#include "xe_pm.h"
#include "xe_preempt_fence.h"
@@ -643,7 +644,7 @@ void xe_vm_add_fault_entry_pf(struct xe_vm *vm, struct xe_pagefault *pf)
return;
}
- e->address = pf->consumer.page_addr;
+ e->address = xe_pagefault_addr(pf);
/*
* TODO:
* Address precision is currently always SZ_4K, but this may change
--
2.34.1
next reply other threads:[~2026-09-17 20:38 UTC|newest]
Thread overview: 5+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-17 20:38 Matthew Brost [this message]
2026-09-17 22:52 ` ✓ CI.KUnit: success for drm/xe: Add mnemonic error reason to page fault diagnostics (rev3) Patchwork
2026-09-18 0:02 ` ✓ Xe.CI.BAT: " Patchwork
2026-09-18 3:47 ` [PATCH v3] drm/xe: Add mnemonic error reason to page fault diagnostics Ghimiray, Himal Prasad
2026-09-18 5:15 ` ✓ Xe.CI.FULL: success for drm/xe: Add mnemonic error reason to page fault diagnostics (rev3) Patchwork
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260917203850.321385-1-matthew.brost@intel.com \
--to=matthew.brost@intel.com \
--cc=intel-xe@lists.freedesktop.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox