All of lore.kernel.org
 help / color / mirror / Atom feed
From: "Ghimiray, Himal Prasad" <himal.prasad.ghimiray@intel.com>
To: Tejas Upadhyay <tejas.upadhyay@intel.com>,
	<intel-xe@lists.freedesktop.org>
Subject: Re: [PATCH V15 10/14] drm/xe/configfs: Add vram bad page reservation policy
Date: Wed, 12 Aug 2026 17:51:35 +0530	[thread overview]
Message-ID: <68ad0fe4-bb9b-403f-8dc3-551f22536b11@intel.com> (raw)
In-Reply-To: <20260811124016.3614699-26-tejas.upadhyay@intel.com>



On 11-08-2026 18:10, Tejas Upadhyay wrote:
> The interface enables setting the policy for how bad pages are
> handled in VRAM. This is crucial for maintaining system
> stability in scenarios where VRAM degradation occurs.
> 
> By default policy will be "reserve", which can be changed to
> "logging" only.
> 
> v3:
> - All FW communication moved under RAS
> v2:
> - Add CRI check and rebase
> 
> Signed-off-by: Tejas Upadhyay <tejas.upadhyay@intel.com>
> ---
>   drivers/gpu/drm/xe/xe_configfs.c     | 64 +++++++++++++++++++++++++++-
>   drivers/gpu/drm/xe/xe_configfs.h     |  2 +
>   drivers/gpu/drm/xe/xe_ttm_vram_mgr.c | 10 +++++
>   3 files changed, 75 insertions(+), 1 deletion(-)
> 
> diff --git a/drivers/gpu/drm/xe/xe_configfs.c b/drivers/gpu/drm/xe/xe_configfs.c
> index 052cce962161..c4f386d4bf09 100644
> --- a/drivers/gpu/drm/xe/xe_configfs.c
> +++ b/drivers/gpu/drm/xe/xe_configfs.c
> @@ -61,7 +61,8 @@
>    *	    ├── survivability_mode
>    *	    ├── gt_types_allowed
>    *	    ├── engines_allowed
> - *	    └── enable_psmi
> + *          ├── enable_psmi
> + *          └── bad_page_reservation
>    *
>    * After configuring the attributes as per next section, the device can be
>    * probed with::
> @@ -159,6 +160,16 @@
>    *
>    * This attribute can only be set before binding to the device.
>    *
> + * Bad pages reservation:
> + * ---------------------
> + *
> + * Disable vram bad pages reservation, instead just report it in dmesg.
> + *  Example to disable it::
> + *
> + *      # echo 0 > /sys/kernel/config/xe/0000:03:00.0/bad_page_reservation
> + *
> + * This attribute can only be set before binding to the device.
> + *
>    * Context restore BB
>    * ------------------
>    *
> @@ -275,6 +286,7 @@ struct xe_config_group_device {
>   		bool survivability_mode;
>   		bool enable_psmi;
>   		bool enable_multi_queue;
> +		bool bad_page_reservation;
>   		struct {
>   			unsigned int max_vfs;
>   			bool admin_only_pf;
> @@ -295,6 +307,7 @@ static const struct xe_config_device device_defaults = {
>   	.survivability_mode = false,
>   	.enable_psmi = false,
>   	.enable_multi_queue = true,
> +	.bad_page_reservation = true,
>   	.sriov = {
>   		.max_vfs = XE_DEFAULT_MAX_VFS,
>   		.admin_only_pf = XE_DEFAULT_ADMIN_ONLY_PF,
> @@ -616,6 +629,32 @@ static ssize_t enable_multi_queue_store(struct config_item *item, const char *pa
>   	return len;
>   }
>   
> +static ssize_t bad_page_reservation_show(struct config_item *item, char *page)
> +{
> +	struct xe_config_device *dev = to_xe_config_device(item);
> +
> +	return sprintf(page, "%d\n", dev->bad_page_reservation);
> +}
> +
> +static ssize_t bad_page_reservation_store(struct config_item *item, const char *page, size_t len)
> +{
> +	struct xe_config_group_device *dev = to_xe_config_group_device(item);
> +	bool val;
> +	int ret;
> +
> +	ret = kstrtobool(page, &val);
> +	if (ret)
> +		return ret;
> +
> +	guard(mutex)(&dev->lock);
> +	if (is_bound(dev))
> +		return -EBUSY;
> +
> +	dev->config.bad_page_reservation = val;
> +
> +	return len;
> +}
> +
>   static bool wa_bb_read_advance(bool dereference, char **p,
>   			       const char *append, size_t len,
>   			       size_t *max_size)
> @@ -855,6 +894,7 @@ CONFIGFS_ATTR(, ctx_restore_mid_bb);
>   CONFIGFS_ATTR(, ctx_restore_post_bb);
>   CONFIGFS_ATTR(, enable_multi_queue);
>   CONFIGFS_ATTR(, enable_psmi);
> +CONFIGFS_ATTR(, bad_page_reservation);
>   CONFIGFS_ATTR(, engines_allowed);
>   CONFIGFS_ATTR(, gt_types_allowed);
>   CONFIGFS_ATTR(, survivability_mode);
> @@ -864,6 +904,7 @@ static struct configfs_attribute *xe_config_device_attrs[] = {
>   	&attr_ctx_restore_post_bb,
>   	&attr_enable_multi_queue,
>   	&attr_enable_psmi,
> +	&attr_bad_page_reservation,
>   	&attr_engines_allowed,
>   	&attr_gt_types_allowed,
>   	&attr_survivability_mode,
> @@ -1142,6 +1183,7 @@ static void dump_custom_dev_config(struct pci_dev *pdev,
>   	PRI_CUSTOM_ATTR("%llx", engines_allowed);
>   	PRI_CUSTOM_ATTR("%d", enable_multi_queue);
>   	PRI_CUSTOM_ATTR("%d", enable_psmi);
> +	PRI_CUSTOM_ATTR("%d", bad_page_reservation);
>   	PRI_CUSTOM_ATTR("%d", survivability_mode);
>   	PRI_CUSTOM_ATTR("%u", sriov.admin_only_pf);
>   
> @@ -1290,6 +1332,26 @@ bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev)
>   	return ret;
>   }
>   
> +/**
> + * xe_configfs_get_bad_page_reservation - get configfs bad_page_reservation setting
> + * @pdev: pci device
> + *
> + * Return: bad_page_reservation setting in configfs

Nit: Better to document configfs val 0 means Logging only
                                      1 means logging and offlining

With that

Reviewed-by: Himal Prasad Ghimiray <himal.prasad.ghimiray@intel.com>

> + */
> +bool xe_configfs_get_bad_page_reservation(struct pci_dev *pdev)
> +{
> +	struct xe_config_group_device *dev = find_xe_config_group_device(pdev);
> +	bool ret;
> +
> +	if (!dev)
> +		return device_defaults.bad_page_reservation;
> +
> +	ret = dev->config.bad_page_reservation;
> +	config_group_put(&dev->group);
> +
> +	return ret;
> +}
> +
>   /**
>    * xe_configfs_get_ctx_restore_mid_bb - get configfs ctx_restore_mid_bb setting
>    * @pdev: pci device
> diff --git a/drivers/gpu/drm/xe/xe_configfs.h b/drivers/gpu/drm/xe/xe_configfs.h
> index 4fbbeafba473..7405cc5f3207 100644
> --- a/drivers/gpu/drm/xe/xe_configfs.h
> +++ b/drivers/gpu/drm/xe/xe_configfs.h
> @@ -24,6 +24,7 @@ bool xe_configfs_media_gt_allowed(struct pci_dev *pdev);
>   u64 xe_configfs_get_engines_allowed(struct pci_dev *pdev);
>   bool xe_configfs_get_psmi_enabled(struct pci_dev *pdev);
>   bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev);
> +bool xe_configfs_get_bad_page_reservation(struct pci_dev *pdev);
>   u32 xe_configfs_get_ctx_restore_mid_bb(struct pci_dev *pdev,
>   				       enum xe_engine_class class,
>   				       const u32 **cs);
> @@ -44,6 +45,7 @@ static inline bool xe_configfs_media_gt_allowed(struct pci_dev *pdev) { return t
>   static inline u64 xe_configfs_get_engines_allowed(struct pci_dev *pdev) { return U64_MAX; }
>   static inline bool xe_configfs_get_psmi_enabled(struct pci_dev *pdev) { return false; }
>   static inline bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev) { return true; }
> +static inline bool xe_configfs_get_bad_page_reservation(struct pci_dev *pdev) { return true; }
>   static inline u32 xe_configfs_get_ctx_restore_mid_bb(struct pci_dev *pdev,
>   						     enum xe_engine_class class,
>   						     const u32 **cs) { return 0; }
> diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
> index 370bcf50c7f7..6280886e2ebb 100644
> --- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
> +++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
> @@ -13,6 +13,7 @@
>   
>   #include "regs/xe_regs.h"
>   #include "xe_bo.h"
> +#include "xe_configfs.h"
>   #include "xe_device.h"
>   #include "xe_exec_queue.h"
>   #include "xe_lrc.h"
> @@ -792,6 +793,7 @@ int xe_ttm_vram_handle_addr_fault(struct xe_device *xe, u64 addr)
>   	struct xe_ttm_vram_mgr *vram_mgr;
>   	struct xe_vram_region *vr;
>   	struct gpu_buddy *mm;
> +	bool policy;
>   
>   	vr = xe_ttm_vram_addr_to_region(xe, addr);
>   	if (IS_ERR(vr)) {
> @@ -811,6 +813,14 @@ int xe_ttm_vram_handle_addr_fault(struct xe_device *xe, u64 addr)
>   	vram_mgr = &vr->ttm;
>   	mm = &vram_mgr->mm;
>   
> +	policy = xe_configfs_get_bad_page_reservation(to_pci_dev(xe->drm.dev));
> +	if (!policy) {
> +		drm_err(&xe->drm, "0x%llx is reported as corrupted address by HW\n",
> +			addr);
> +		/* Let RAS report to FW to drop addr from SRAM queue */
> +		return -EOPNOTSUPP;
> +	}
> +
>   	/* Reserve page at address */
>   	return xe_ttm_vram_reserve_page_at_addr(xe, addr, vram_mgr, mm);
>   }


  reply	other threads:[~2026-08-12 12:21 UTC|newest]

Thread overview: 30+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-11 12:40 [PATCH V15 00/14] Add memory page offlining support Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 01/14] drm/xe: Link VRAM object with gpu buddy Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 02/14] [DO_NOT_MERGE]drm/gpu: Add gpu_buddy_allocated_addr_to_block helper Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 03/14] drm/xe: Link LRC BO and its execution Queue Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 04/14] drm/xe: Extend BO purge to handle vram pages as well Tejas Upadhyay
2026-08-11 12:40 ` [PATCH V15 05/14] drm/xe/bo: Make xe_bo_is_user() public Tejas Upadhyay
2026-08-11 15:38   ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 06/14] drm/xe: Guard teardown paths against purged BOs Tejas Upadhyay
2026-08-12  3:24   ` Ghimiray, Himal Prasad
2026-08-12  9:40     ` Upadhyay, Tejas
2026-08-11 12:40 ` [PATCH V15 07/14] drm/xe/vram: Extract buddy alloc and free helpers Tejas Upadhyay
2026-08-12  3:25   ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 08/14] drm/xe/vram: Add page offline data structures and lifecycle Tejas Upadhyay
2026-08-12 12:01   ` Ghimiray, Himal Prasad
2026-08-14  6:38     ` Upadhyay, Tejas
2026-08-11 12:40 ` [PATCH V15 09/14] drm/xe/vram: Add VRAM page offline fault handler Tejas Upadhyay
2026-08-13 12:44   ` Ghimiray, Himal Prasad
2026-08-14  5:19     ` Upadhyay, Tejas
2026-08-14 10:16       ` Upadhyay, Tejas
2026-08-11 12:40 ` [PATCH V15 10/14] drm/xe/configfs: Add vram bad page reservation policy Tejas Upadhyay
2026-08-12 12:21   ` Ghimiray, Himal Prasad [this message]
2026-08-11 12:40 ` [PATCH V15 11/14] drm/xe/vram: Use RCU for lock-free sysfs reads of bad page lists Tejas Upadhyay
2026-08-12 12:14   ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 12/14] drm/xe: Add sysfs interface for bad gpu vram pages Tejas Upadhyay
2026-08-12 12:27   ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 13/14] drm/xe/uapi: Expose ban reason in EXEC_QUEUE_GET_PROPERTY_BAN Tejas Upadhyay
2026-08-11 20:08   ` Rodrigo Vivi
2026-08-12 12:24   ` Ghimiray, Himal Prasad
2026-08-11 12:40 ` [PATCH V15 14/14] drm/xe: Add fault-inject based VRAM page offline injection Tejas Upadhyay
2026-08-16 14:35   ` Ghimiray, Himal Prasad

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=68ad0fe4-bb9b-403f-8dc3-551f22536b11@intel.com \
    --to=himal.prasad.ghimiray@intel.com \
    --cc=intel-xe@lists.freedesktop.org \
    --cc=tejas.upadhyay@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.