From: "Ghimiray, Himal Prasad" <himal.prasad.ghimiray@intel.com>
To: Tejas Upadhyay <tejas.upadhyay@intel.com>,
<intel-xe@lists.freedesktop.org>
Subject: Re: [PATCH V15 10/14] drm/xe/configfs: Add vram bad page reservation policy
Date: Wed, 12 Aug 2026 17:51:35 +0530 [thread overview]
Message-ID: <68ad0fe4-bb9b-403f-8dc3-551f22536b11@intel.com> (raw)
In-Reply-To: <20260811124016.3614699-26-tejas.upadhyay@intel.com>
On 11-08-2026 18:10, Tejas Upadhyay wrote:
> The interface enables setting the policy for how bad pages are
> handled in VRAM. This is crucial for maintaining system
> stability in scenarios where VRAM degradation occurs.
>
> By default policy will be "reserve", which can be changed to
> "logging" only.
>
> v3:
> - All FW communication moved under RAS
> v2:
> - Add CRI check and rebase
>
> Signed-off-by: Tejas Upadhyay <tejas.upadhyay@intel.com>
> ---
> drivers/gpu/drm/xe/xe_configfs.c | 64 +++++++++++++++++++++++++++-
> drivers/gpu/drm/xe/xe_configfs.h | 2 +
> drivers/gpu/drm/xe/xe_ttm_vram_mgr.c | 10 +++++
> 3 files changed, 75 insertions(+), 1 deletion(-)
>
> diff --git a/drivers/gpu/drm/xe/xe_configfs.c b/drivers/gpu/drm/xe/xe_configfs.c
> index 052cce962161..c4f386d4bf09 100644
> --- a/drivers/gpu/drm/xe/xe_configfs.c
> +++ b/drivers/gpu/drm/xe/xe_configfs.c
> @@ -61,7 +61,8 @@
> * ├── survivability_mode
> * ├── gt_types_allowed
> * ├── engines_allowed
> - * └── enable_psmi
> + * ├── enable_psmi
> + * └── bad_page_reservation
> *
> * After configuring the attributes as per next section, the device can be
> * probed with::
> @@ -159,6 +160,16 @@
> *
> * This attribute can only be set before binding to the device.
> *
> + * Bad pages reservation:
> + * ---------------------
> + *
> + * Disable vram bad pages reservation, instead just report it in dmesg.
> + * Example to disable it::
> + *
> + * # echo 0 > /sys/kernel/config/xe/0000:03:00.0/bad_page_reservation
> + *
> + * This attribute can only be set before binding to the device.
> + *
> * Context restore BB
> * ------------------
> *
> @@ -275,6 +286,7 @@ struct xe_config_group_device {
> bool survivability_mode;
> bool enable_psmi;
> bool enable_multi_queue;
> + bool bad_page_reservation;
> struct {
> unsigned int max_vfs;
> bool admin_only_pf;
> @@ -295,6 +307,7 @@ static const struct xe_config_device device_defaults = {
> .survivability_mode = false,
> .enable_psmi = false,
> .enable_multi_queue = true,
> + .bad_page_reservation = true,
> .sriov = {
> .max_vfs = XE_DEFAULT_MAX_VFS,
> .admin_only_pf = XE_DEFAULT_ADMIN_ONLY_PF,
> @@ -616,6 +629,32 @@ static ssize_t enable_multi_queue_store(struct config_item *item, const char *pa
> return len;
> }
>
> +static ssize_t bad_page_reservation_show(struct config_item *item, char *page)
> +{
> + struct xe_config_device *dev = to_xe_config_device(item);
> +
> + return sprintf(page, "%d\n", dev->bad_page_reservation);
> +}
> +
> +static ssize_t bad_page_reservation_store(struct config_item *item, const char *page, size_t len)
> +{
> + struct xe_config_group_device *dev = to_xe_config_group_device(item);
> + bool val;
> + int ret;
> +
> + ret = kstrtobool(page, &val);
> + if (ret)
> + return ret;
> +
> + guard(mutex)(&dev->lock);
> + if (is_bound(dev))
> + return -EBUSY;
> +
> + dev->config.bad_page_reservation = val;
> +
> + return len;
> +}
> +
> static bool wa_bb_read_advance(bool dereference, char **p,
> const char *append, size_t len,
> size_t *max_size)
> @@ -855,6 +894,7 @@ CONFIGFS_ATTR(, ctx_restore_mid_bb);
> CONFIGFS_ATTR(, ctx_restore_post_bb);
> CONFIGFS_ATTR(, enable_multi_queue);
> CONFIGFS_ATTR(, enable_psmi);
> +CONFIGFS_ATTR(, bad_page_reservation);
> CONFIGFS_ATTR(, engines_allowed);
> CONFIGFS_ATTR(, gt_types_allowed);
> CONFIGFS_ATTR(, survivability_mode);
> @@ -864,6 +904,7 @@ static struct configfs_attribute *xe_config_device_attrs[] = {
> &attr_ctx_restore_post_bb,
> &attr_enable_multi_queue,
> &attr_enable_psmi,
> + &attr_bad_page_reservation,
> &attr_engines_allowed,
> &attr_gt_types_allowed,
> &attr_survivability_mode,
> @@ -1142,6 +1183,7 @@ static void dump_custom_dev_config(struct pci_dev *pdev,
> PRI_CUSTOM_ATTR("%llx", engines_allowed);
> PRI_CUSTOM_ATTR("%d", enable_multi_queue);
> PRI_CUSTOM_ATTR("%d", enable_psmi);
> + PRI_CUSTOM_ATTR("%d", bad_page_reservation);
> PRI_CUSTOM_ATTR("%d", survivability_mode);
> PRI_CUSTOM_ATTR("%u", sriov.admin_only_pf);
>
> @@ -1290,6 +1332,26 @@ bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev)
> return ret;
> }
>
> +/**
> + * xe_configfs_get_bad_page_reservation - get configfs bad_page_reservation setting
> + * @pdev: pci device
> + *
> + * Return: bad_page_reservation setting in configfs
Nit: Better to document configfs val 0 means Logging only
1 means logging and offlining
With that
Reviewed-by: Himal Prasad Ghimiray <himal.prasad.ghimiray@intel.com>
> + */
> +bool xe_configfs_get_bad_page_reservation(struct pci_dev *pdev)
> +{
> + struct xe_config_group_device *dev = find_xe_config_group_device(pdev);
> + bool ret;
> +
> + if (!dev)
> + return device_defaults.bad_page_reservation;
> +
> + ret = dev->config.bad_page_reservation;
> + config_group_put(&dev->group);
> +
> + return ret;
> +}
> +
> /**
> * xe_configfs_get_ctx_restore_mid_bb - get configfs ctx_restore_mid_bb setting
> * @pdev: pci device
> diff --git a/drivers/gpu/drm/xe/xe_configfs.h b/drivers/gpu/drm/xe/xe_configfs.h
> index 4fbbeafba473..7405cc5f3207 100644
> --- a/drivers/gpu/drm/xe/xe_configfs.h
> +++ b/drivers/gpu/drm/xe/xe_configfs.h
> @@ -24,6 +24,7 @@ bool xe_configfs_media_gt_allowed(struct pci_dev *pdev);
> u64 xe_configfs_get_engines_allowed(struct pci_dev *pdev);
> bool xe_configfs_get_psmi_enabled(struct pci_dev *pdev);
> bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev);
> +bool xe_configfs_get_bad_page_reservation(struct pci_dev *pdev);
> u32 xe_configfs_get_ctx_restore_mid_bb(struct pci_dev *pdev,
> enum xe_engine_class class,
> const u32 **cs);
> @@ -44,6 +45,7 @@ static inline bool xe_configfs_media_gt_allowed(struct pci_dev *pdev) { return t
> static inline u64 xe_configfs_get_engines_allowed(struct pci_dev *pdev) { return U64_MAX; }
> static inline bool xe_configfs_get_psmi_enabled(struct pci_dev *pdev) { return false; }
> static inline bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev) { return true; }
> +static inline bool xe_configfs_get_bad_page_reservation(struct pci_dev *pdev) { return true; }
> static inline u32 xe_configfs_get_ctx_restore_mid_bb(struct pci_dev *pdev,
> enum xe_engine_class class,
> const u32 **cs) { return 0; }
> diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
> index 370bcf50c7f7..6280886e2ebb 100644
> --- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
> +++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
> @@ -13,6 +13,7 @@
>
> #include "regs/xe_regs.h"
> #include "xe_bo.h"
> +#include "xe_configfs.h"
> #include "xe_device.h"
> #include "xe_exec_queue.h"
> #include "xe_lrc.h"
> @@ -792,6 +793,7 @@ int xe_ttm_vram_handle_addr_fault(struct xe_device *xe, u64 addr)
> struct xe_ttm_vram_mgr *vram_mgr;
> struct xe_vram_region *vr;
> struct gpu_buddy *mm;
> + bool policy;
>
> vr = xe_ttm_vram_addr_to_region(xe, addr);
> if (IS_ERR(vr)) {
> @@ -811,6 +813,14 @@ int xe_ttm_vram_handle_addr_fault(struct xe_device *xe, u64 addr)
> vram_mgr = &vr->ttm;
> mm = &vram_mgr->mm;
>
> + policy = xe_configfs_get_bad_page_reservation(to_pci_dev(xe->drm.dev));
> + if (!policy) {
> + drm_err(&xe->drm, "0x%llx is reported as corrupted address by HW\n",
> + addr);
> + /* Let RAS report to FW to drop addr from SRAM queue */
> + return -EOPNOTSUPP;
> + }
> +
> /* Reserve page at address */
> return xe_ttm_vram_reserve_page_at_addr(xe, addr, vram_mgr, mm);
> }
next prev parent reply other threads:[~2026-08-12 12:21 UTC|newest]
Thread overview: 30+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-11 12:40 [PATCH V15 00/14] Add memory page offlining support Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 01/14] drm/xe: Link VRAM object with gpu buddy Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 02/14] [DO_NOT_MERGE]drm/gpu: Add gpu_buddy_allocated_addr_to_block helper Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 03/14] drm/xe: Link LRC BO and its execution Queue Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 04/14] drm/xe: Extend BO purge to handle vram pages as well Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 05/14] drm/xe/bo: Make xe_bo_is_user() public Tejas Upadhyay
2026-08-11 15:38 ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 06/14] drm/xe: Guard teardown paths against purged BOs Tejas Upadhyay
2026-08-12 3:24 ` Ghimiray, Himal Prasad
2026-08-12 9:40 ` Upadhyay, Tejas
2026-08-11 12:40 ` [PATCH V15 07/14] drm/xe/vram: Extract buddy alloc and free helpers Tejas Upadhyay
2026-08-12 3:25 ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 08/14] drm/xe/vram: Add page offline data structures and lifecycle Tejas Upadhyay
2026-08-12 12:01 ` Ghimiray, Himal Prasad
2026-08-14 6:38 ` Upadhyay, Tejas
2026-08-11 12:40 ` [PATCH V15 09/14] drm/xe/vram: Add VRAM page offline fault handler Tejas Upadhyay
2026-08-13 12:44 ` Ghimiray, Himal Prasad
2026-08-14 5:19 ` Upadhyay, Tejas
2026-08-14 10:16 ` Upadhyay, Tejas
2026-08-11 12:40 ` [PATCH V15 10/14] drm/xe/configfs: Add vram bad page reservation policy Tejas Upadhyay
2026-08-12 12:21 ` Ghimiray, Himal Prasad [this message]
2026-08-11 12:40 ` [PATCH V15 11/14] drm/xe/vram: Use RCU for lock-free sysfs reads of bad page lists Tejas Upadhyay
2026-08-12 12:14 ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 12/14] drm/xe: Add sysfs interface for bad gpu vram pages Tejas Upadhyay
2026-08-12 12:27 ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 13/14] drm/xe/uapi: Expose ban reason in EXEC_QUEUE_GET_PROPERTY_BAN Tejas Upadhyay
2026-08-11 20:08 ` Rodrigo Vivi
2026-08-12 12:24 ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 14/14] drm/xe: Add fault-inject based VRAM page offline injection Tejas Upadhyay
2026-08-16 14:35 ` Ghimiray, Himal Prasad
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=68ad0fe4-bb9b-403f-8dc3-551f22536b11@intel.com \
--to=himal.prasad.ghimiray@intel.com \
--cc=intel-xe@lists.freedesktop.org \
--cc=tejas.upadhyay@intel.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox