From: "Ghimiray, Himal Prasad" <himal.prasad.ghimiray@intel.com>
To: Tejas Upadhyay <tejas.upadhyay@intel.com>,
<intel-xe@lists.freedesktop.org>
Subject: Re: [PATCH V15 14/14] drm/xe: Add fault-inject based VRAM page offline injection
Date: Sun, 16 Aug 2026 20:05:20 +0530 [thread overview]
Message-ID: <d8e4019f-238a-4bd6-b6f1-6e2b3bfea612@intel.com> (raw)
In-Reply-To: <20260811124016.3614699-30-tejas.upadhyay@intel.com>
On 11-08-2026 18:10, Tejas Upadhyay wrote:
> Add a fault-inject based debugfs interface for testing VRAM page
> offlining. This replaces the previous standalone debugfs approach
> with the standard kernel fault-inject infrastructure.
>
> Two debugfs entries are created under the xe debugfs root for
> CRI platforms:
> - inject_mempage_offline/: Standard fault-inject knobs (probability,
> times, interval, etc.) — only available with
> CONFIG_FAULT_INJECTION_DEBUG_FS
> - inject_mempage_offline_trigger: Write a PFN value to inject a
> specific page, or write "0" to auto-pick the last unallocated
> VRAM page
>
> The trigger accepts:
> - "0" : auto-pick last unallocated page
> - "0xPFN" : inject fault at a specific PFN address
>
> On kernels with CONFIG_FAULT_INJECTION_DEBUG_FS, injection is gated
> by should_fail() (probability/times must be configured first).
> On kernels without it, the trigger always injects directly.
>
> The injection reports the page as faulted via
> xe_ttm_vram_handle_addr_fault(), exercising the full page offlining
> path.
>
> Usage (with CONFIG_FAULT_INJECTION_DEBUG_FS):
> echo 100 > inject_mempage_offline/probability
> echo 1 > inject_mempage_offline/times
> echo 0 > inject_mempage_offline_trigger
>
> Usage (without CONFIG_FAULT_INJECTION_DEBUG_FS):
> echo 0 > inject_mempage_offline_trigger
>
> v2(sashiko):
> - use cond_resched()
> - validate input first and fix addr < 0 case
> - validate vr, move block, found var as local to scope_guard
>
> Signed-off-by: Tejas Upadhyay <tejas.upadhyay@intel.com>
> ---
> drivers/gpu/drm/xe/xe_debugfs.c | 50 ++++++++++++++++++++++++++
> drivers/gpu/drm/xe/xe_debugfs.h | 2 ++
> drivers/gpu/drm/xe/xe_ttm_vram_mgr.c | 52 ++++++++++++++++++++++++++++
> drivers/gpu/drm/xe/xe_ttm_vram_mgr.h | 1 +
> 4 files changed, 105 insertions(+)
>
> diff --git a/drivers/gpu/drm/xe/xe_debugfs.c b/drivers/gpu/drm/xe/xe_debugfs.c
> index eeceab4a9901..2fd1ff806864 100644
> --- a/drivers/gpu/drm/xe/xe_debugfs.c
> +++ b/drivers/gpu/drm/xe/xe_debugfs.c
> @@ -32,6 +32,7 @@
> #include "xe_sriov_vf.h"
> #include "xe_step.h"
> #include "xe_tile_debugfs.h"
> +#include "xe_ttm_vram_mgr.h"
> #include "xe_vsec.h"
> #include "xe_wa.h"
>
> @@ -43,12 +44,18 @@
>
> DECLARE_FAULT_ATTR(gt_reset_failure);
> DECLARE_FAULT_ATTR(inject_csc_hw_error);
> +DECLARE_FAULT_ATTR(inject_mempage_offline);
>
> static bool csc_hw_error_available(struct xe_device *xe)
> {
> return !IS_SRIOV_VF(xe) && xe->info.platform == XE_BATTLEMAGE;
> }
>
> +static bool is_crescent_island(struct xe_device *xe)
> +{
> + return xe->info.platform == XE_CRESCENTISLAND;
> +}
> +
> /*
> * Fault injection table. Each entry registers a debugfs attribute; add a
> * matching FAULT_ACTION() below for every entry added here.
> @@ -63,6 +70,9 @@ static struct {
> { .name = "inject_csc_hw_error",
> .attr = &inject_csc_hw_error,
> .is_visible = csc_hw_error_available },
> + { .name = "inject_mempage_offline",
> + .attr = &inject_mempage_offline,
> + .is_visible = is_crescent_island },
> };
>
> /*
> @@ -77,6 +87,41 @@ bool xe_fault_##name(void) \
>
> FAULT_ACTION(gt_reset, gt_reset_failure)
> FAULT_ACTION(csc_hw_error, inject_csc_hw_error)
> +FAULT_ACTION(mempage_offline, inject_mempage_offline)
> +
> +static ssize_t inject_mempage_offline_trigger(struct file *f,
> + const char __user *ubuf,
> + size_t size, loff_t *pos)
> +{
> + struct xe_device *xe = file_inode(f)->i_private;
> + struct xe_tile *tile = xe_device_get_root_tile(xe);
> + struct xe_vram_region *vr = tile->mem.vram;
> + u64 pfn;
> + int ret;
> +
> + if (!vr)
> + return -ENODEV;
> +
> + ret = kstrtou64_from_user(ubuf, size, 0, &pfn);
> + if (ret)
> + return ret;
> +
> + if (IS_ENABLED(CONFIG_FAULT_INJECTION_DEBUG_FS) &&
> + !should_fail(&inject_mempage_offline, 1))
> + return size;
Use the FAULT_ACTION xe_fault_mempage_offline generated above.
> +
> + if (pfn == 0)
> + return xe_ttm_vram_inject_fault(xe) ?: size;
> +
> + /* User provided PFN — convert to DPA and inject */
> + return xe_ttm_vram_handle_addr_fault(xe,
> + (pfn << PAGE_SHIFT) + vr->dpa_base) ?: size;
> +}
> +
> +static const struct file_operations inject_mempage_offline_fops = {
> + .owner = THIS_MODULE,
> + .write = inject_mempage_offline_trigger,
> +};
>
> static void xe_fault_inject_debugfs_register(struct xe_device *xe,
> struct dentry *root)
> @@ -91,6 +136,11 @@ static void xe_fault_inject_debugfs_register(struct xe_device *xe,
> fault_create_debugfs_attr(xe_fault_inject_entry[i].name, root,
> xe_fault_inject_entry[i].attr);
> }
> +
> + if (is_crescent_island(xe)) {
> + debugfs_create_file("inject_mempage_offline_trigger", 0200,
> + root, xe, &inject_mempage_offline_fops);
> + }
> }
>
> static void read_residency_counter(struct xe_device *xe, struct xe_mmio *mmio,
> diff --git a/drivers/gpu/drm/xe/xe_debugfs.h b/drivers/gpu/drm/xe/xe_debugfs.h
> index cd56f7442b99..727747f13101 100644
> --- a/drivers/gpu/drm/xe/xe_debugfs.h
> +++ b/drivers/gpu/drm/xe/xe_debugfs.h
> @@ -13,10 +13,12 @@ struct xe_device;
> #ifdef CONFIG_DEBUG_FS
> bool xe_fault_gt_reset(void);
> bool xe_fault_csc_hw_error(void);
> +bool xe_fault_mempage_offline(void);
> void xe_debugfs_register(struct xe_device *xe);
> #else
> static inline bool xe_fault_gt_reset(void) { return false; }
> static inline bool xe_fault_csc_hw_error(void) { return false; }
> +static inline bool xe_fault_mempage_offline(void) { return false; }
> static inline void xe_debugfs_register(struct xe_device *xe) { }
> #endif
>
> diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
> index 5bb66c7b5505..c985d633c10b 100644
> --- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
> +++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
> @@ -834,6 +834,58 @@ int xe_ttm_vram_handle_addr_fault(struct xe_device *xe, u64 addr)
> }
> EXPORT_SYMBOL(xe_ttm_vram_handle_addr_fault);
>
> +/**
> + * xe_ttm_vram_inject_fault - Inject a VRAM page fault for testing
> + * @xe: xe device instance
> + *
> + * Picks the last unallocated VRAM page and reports it as faulted
> + * via xe_ttm_vram_handle_addr_fault(). Used by the fault-inject
> + * debugfs interface for testing page offlining.
> + *
> + * Return: 0 on success, negative error code on failure.
> + */
> +int xe_ttm_vram_inject_fault(struct xe_device *xe)
> +{
> + struct xe_tile *tile = xe_device_get_root_tile(xe);
> + struct xe_vram_region *vr = tile->mem.vram;
> + struct xe_ttm_vram_mgr *vram_mgr = &vr->ttm;
> + struct gpu_buddy *mm = &vram_mgr->mm;
> + u64 addr;
> +
> + if (vr->actual_physical_size < SZ_4K)
> + return -ENOSPC;
> +
> + addr = vr->actual_physical_size - SZ_4K;
> + while (addr < vr->actual_physical_size) {
> + struct gpu_buddy_block *block;
> + bool found = false;
> +
> + scoped_guard(mutex, &vram_mgr->lock) {
> + block = gpu_buddy_allocated_addr_to_block(mm, addr);
> + if (!block)
> + found = true;
> + }
> +
> + /*
> + * Intentional race window: xe_ttm_vram_handle_addr_fault()
> + * re-acquires vram_mgr->lock internally, so we cannot hold
> + * it here. A concurrent allocation claiming this page between
> + * the two calls is an acceptable false negative for this
> + * test-only path.
> + */
> + if (found)
> + return xe_ttm_vram_handle_addr_fault(xe, addr + vr->dpa_base);
> +
> + cond_resched();
> + if (addr == 0)
> + break;
> + addr -= SZ_4K;
> + }
> +
> + return -ENOSPC;
> +}
> +EXPORT_SYMBOL(xe_ttm_vram_inject_fault);
> +
> static size_t serialize_bad_pages(struct xe_ttm_vram_mgr *mgr, char *buf, size_t max_len)
> {
> struct xe_ttm_vram_offline_resource *pos;
> diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.h b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.h
> index eb55b0f74ef3..9feb999a1f8d 100644
> --- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.h
> +++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.h
> @@ -32,6 +32,7 @@ void xe_ttm_vram_get_used(struct ttm_resource_manager *man,
> u64 *used, u64 *used_visible);
>
> int xe_ttm_vram_handle_addr_fault(struct xe_device *xe, u64 addr);
> +int xe_ttm_vram_inject_fault(struct xe_device *xe);
> int xe_ttm_vram_sysfs_init(struct xe_device *xe);
> static inline struct xe_ttm_vram_mgr_resource *
> to_xe_ttm_vram_mgr_resource(struct ttm_resource *res)
prev parent reply other threads:[~2026-08-16 14:35 UTC|newest]
Thread overview: 30+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-11 12:40 [PATCH V15 00/14] Add memory page offlining support Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 01/14] drm/xe: Link VRAM object with gpu buddy Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 02/14] [DO_NOT_MERGE]drm/gpu: Add gpu_buddy_allocated_addr_to_block helper Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 03/14] drm/xe: Link LRC BO and its execution Queue Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 04/14] drm/xe: Extend BO purge to handle vram pages as well Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 05/14] drm/xe/bo: Make xe_bo_is_user() public Tejas Upadhyay
2026-08-11 15:38 ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 06/14] drm/xe: Guard teardown paths against purged BOs Tejas Upadhyay
2026-08-12 3:24 ` Ghimiray, Himal Prasad
2026-08-12 9:40 ` Upadhyay, Tejas
2026-08-11 12:40 ` [PATCH V15 07/14] drm/xe/vram: Extract buddy alloc and free helpers Tejas Upadhyay
2026-08-12 3:25 ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 08/14] drm/xe/vram: Add page offline data structures and lifecycle Tejas Upadhyay
2026-08-12 12:01 ` Ghimiray, Himal Prasad
2026-08-14 6:38 ` Upadhyay, Tejas
2026-08-11 12:40 ` [PATCH V15 09/14] drm/xe/vram: Add VRAM page offline fault handler Tejas Upadhyay
2026-08-13 12:44 ` Ghimiray, Himal Prasad
2026-08-14 5:19 ` Upadhyay, Tejas
2026-08-14 10:16 ` Upadhyay, Tejas
2026-08-11 12:40 ` [PATCH V15 10/14] drm/xe/configfs: Add vram bad page reservation policy Tejas Upadhyay
2026-08-12 12:21 ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 11/14] drm/xe/vram: Use RCU for lock-free sysfs reads of bad page lists Tejas Upadhyay
2026-08-12 12:14 ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 12/14] drm/xe: Add sysfs interface for bad gpu vram pages Tejas Upadhyay
2026-08-12 12:27 ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 13/14] drm/xe/uapi: Expose ban reason in EXEC_QUEUE_GET_PROPERTY_BAN Tejas Upadhyay
2026-08-11 20:08 ` Rodrigo Vivi
2026-08-12 12:24 ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 14/14] drm/xe: Add fault-inject based VRAM page offline injection Tejas Upadhyay
2026-08-16 14:35 ` Ghimiray, Himal Prasad [this message]
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=d8e4019f-238a-4bd6-b6f1-6e2b3bfea612@intel.com \
--to=himal.prasad.ghimiray@intel.com \
--cc=intel-xe@lists.freedesktop.org \
--cc=tejas.upadhyay@intel.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.