Linux Documentation
 help / color / mirror / Atom feed
From: <mhonap@nvidia.com>
To: <alex@shazbot.org>, <jgg@ziepe.ca>, <ankita@nvidia.com>,
	<jic23@kernel.org>, <dave.jiang@intel.com>,
	<alejandro.lucero-palau@amd.com>, <smadhavan@nvidia.com>,
	<corbet@lwn.net>, <skhan@linuxfoundation.org>,
	<dave@stgolabs.net>, <alison.schofield@intel.com>,
	<vishal.l.verma@intel.com>, <iweiny@kernel.org>,
	<ming.li@zohomail.com>, <yishaih@nvidia.com>,
	<skolothumtho@nvidia.com>, <kevin.tian@intel.com>,
	<bhelgaas@google.com>, <dmatlack@google.com>, <kees@kernel.org>,
	<gustavoars@kernel.org>
Cc: <cjia@nvidia.com>, <kjaju@nvidia.com>, <vsethi@nvidia.com>,
	<zhiw@nvidia.com>, <mhonap@nvidia.com>,
	<linux-doc@vger.kernel.org>, <linux-kernel@vger.kernel.org>,
	<kvm@vger.kernel.org>, <linux-cxl@vger.kernel.org>,
	<linux-pci@vger.kernel.org>, <linux-kselftest@vger.kernel.org>,
	<linux-hardening@vger.kernel.org>
Subject: [PATCH v4 12/27] vfio/pci: Let a provider exclude a BAR sub-range from mmap
Date: Thu, 13 Aug 2026 15:06:16 +0530	[thread overview]
Message-ID: <20260813093631.2288172-13-mhonap@nvidia.com> (raw)
In-Reply-To: <20260813093631.2288172-1-mhonap@nvidia.com>

From: Manish Honap <mhonap@nvidia.com>

Some devices expose registers in a BAR that must be reached only through
a trap, not a direct guest mapping. A CXL Type-2 device's HDM decoder
block is one: mapping it would let userspace reprogram the physical
decoder that governs host memory decode. Give a provider a way to mark a
BAR sub-range off-limits to mmap; it is advertised as a sparse-mmap
region and refused in the mmap path, while the provider's own region
still serves it.

Signed-off-by: Manish Honap <mhonap@nvidia.com>
---
 drivers/vfio/pci/vfio_pci_core.c   | 72 ++++++++++++++++++++++++++++++
 drivers/vfio/pci/vfio_pci_dmabuf.c | 13 ++++++
 drivers/vfio/pci/vfio_pci_priv.h   | 13 ++++++
 include/linux/vfio_pci_core.h      |  6 +++
 4 files changed, 104 insertions(+)

diff --git a/drivers/vfio/pci/vfio_pci_core.c b/drivers/vfio/pci/vfio_pci_core.c
index 0f9b5dfeea66..49dfbdaf3f05 100644
--- a/drivers/vfio/pci/vfio_pci_core.c
+++ b/drivers/vfio/pci/vfio_pci_core.c
@@ -1009,6 +1009,67 @@ static int msix_mmappable_cap(struct vfio_pci_core_device *vdev,
 	return vfio_info_add_capability(caps, &header, sizeof(header));
 }
 
+/*
+ * A provider can keep a BAR sub-range off mmap (for example a CXL device's
+ * trapped HDM decoder block). Callers hold the resource so /dev/mem is already
+ * blocked; this only governs the vfio mmap path.
+ */
+void vfio_pci_core_set_mmap_exclude(struct vfio_pci_core_device *vdev, int bar,
+				    u64 start, u64 len)
+{
+	vdev->mmap_exclude_bar = bar;
+	vdev->mmap_exclude_start = start;
+	vdev->mmap_exclude_len = len;
+}
+EXPORT_SYMBOL_GPL(vfio_pci_core_set_mmap_exclude);
+
+/* Advertise the BAR as mmappable minus the excluded sub-range. */
+static int vfio_pci_mmap_exclude_cap(struct vfio_pci_core_device *vdev,
+				     int index, struct vfio_info_cap *caps)
+{
+	u64 bar_len = pci_resource_len(vdev->pdev, index);
+	u64 excl_start = ALIGN_DOWN(vdev->mmap_exclude_start, PAGE_SIZE);
+	u64 excl_end = ALIGN(vdev->mmap_exclude_start + vdev->mmap_exclude_len,
+			     PAGE_SIZE);
+	struct vfio_region_info_cap_sparse_mmap *sparse;
+	int nr_areas = 0, i = 0, ret;
+	size_t size;
+
+	/*
+	 * mmap is page granular, so the mmappable areas must stop at the page
+	 * boundaries enclosing the excluded sub-range. The byte-granular
+	 * exclusion still governs the fault and read/write paths; only the
+	 * advertised mmap areas round out to whole pages.
+	 */
+	if (excl_start > 0)
+		nr_areas++;
+	if (excl_end < bar_len)
+		nr_areas++;
+
+	size = struct_size(sparse, areas, nr_areas);
+	sparse = kzalloc(size, GFP_KERNEL);
+	if (!sparse)
+		return -ENOMEM;
+
+	sparse->header.id = VFIO_REGION_INFO_CAP_SPARSE_MMAP;
+	sparse->header.version = 1;
+	sparse->nr_areas = nr_areas;
+
+	if (excl_start > 0) {
+		sparse->areas[i].offset = 0;
+		sparse->areas[i].size = excl_start;
+		i++;
+	}
+	if (excl_end < bar_len) {
+		sparse->areas[i].offset = excl_end;
+		sparse->areas[i].size = bar_len - excl_end;
+	}
+
+	ret = vfio_info_add_capability(caps, &sparse->header, size);
+	kfree(sparse);
+	return ret;
+}
+
 int vfio_pci_core_register_dev_region(struct vfio_pci_core_device *vdev,
 				      unsigned int type, unsigned int subtype,
 				      const struct vfio_pci_regops *ops,
@@ -1157,6 +1218,13 @@ int vfio_pci_ioctl_get_region_info(struct vfio_device *core_vdev,
 				if (ret)
 					return ret;
 			}
+			if (vdev->mmap_exclude_len &&
+			    info->index == vdev->mmap_exclude_bar) {
+				ret = vfio_pci_mmap_exclude_cap(vdev, info->index,
+								caps);
+				if (ret)
+					return ret;
+			}
 		}
 
 		break;
@@ -1851,6 +1919,10 @@ int vfio_pci_core_mmap(struct vfio_device *core_vdev, struct vm_area_struct *vma
 	if (req_start + req_len > phys_len)
 		return -EINVAL;
 
+	/* An excluded sub-range is reachable only through its trap, not mmap. */
+	if (vfio_pci_bar_is_excluded(vdev, index, req_start, req_len))
+		return -EINVAL;
+
 	/*
 	 * Ensure the BAR resource region is reserved for use.
 	 */
diff --git a/drivers/vfio/pci/vfio_pci_dmabuf.c b/drivers/vfio/pci/vfio_pci_dmabuf.c
index c16f460c01d6..51983105d38b 100644
--- a/drivers/vfio/pci/vfio_pci_dmabuf.c
+++ b/drivers/vfio/pci/vfio_pci_dmabuf.c
@@ -177,11 +177,24 @@ int vfio_pci_core_get_dmabuf_phys(struct vfio_pci_core_device *vdev,
 				  size_t nr_ranges)
 {
 	struct pci_dev *pdev = vdev->pdev;
+	unsigned int i;
 
 	*provider = pcim_p2pdma_provider(pdev, region_index);
 	if (!*provider)
 		return -EINVAL;
 
+	/*
+	 * A provider (e.g. vfio-cxl) can exclude a BAR sub-range that must be
+	 * reached only through its trap. The mmap and read/write paths already
+	 * refuse it; reject a DMA-BUF export overlapping it too, so a device fd
+	 * holder cannot map the excluded registers to a peer and bypass the trap.
+	 */
+	for (i = 0; i < nr_ranges; i++)
+		if (vfio_pci_bar_is_excluded(vdev, region_index,
+					     dma_ranges[i].offset,
+					     dma_ranges[i].length))
+			return -EINVAL;
+
 	return vfio_pci_core_fill_phys_vec(
 		phys_vec, dma_ranges, nr_ranges,
 		pci_resource_start(pdev, region_index),
diff --git a/drivers/vfio/pci/vfio_pci_priv.h b/drivers/vfio/pci/vfio_pci_priv.h
index fca9d0dfac90..902d17815ab6 100644
--- a/drivers/vfio/pci/vfio_pci_priv.h
+++ b/drivers/vfio/pci/vfio_pci_priv.h
@@ -44,6 +44,19 @@ ssize_t vfio_pci_config_rw_single(struct vfio_pci_core_device *vdev,
 ssize_t vfio_pci_bar_rw(struct vfio_pci_core_device *vdev, char __user *buf,
 			size_t count, loff_t *ppos, bool iswrite);
 
+/*
+ * A provider (e.g. vfio-cxl) can carve a sub-range out of a BAR that must be
+ * reached only through its trap, never the direct BAR.  Returns true when
+ * [start, start + len) on this BAR overlaps that excluded range.
+ */
+static inline bool vfio_pci_bar_is_excluded(struct vfio_pci_core_device *vdev,
+					    int bar, u64 start, u64 len)
+{
+	return vdev->mmap_exclude_len && bar == vdev->mmap_exclude_bar &&
+	       start < vdev->mmap_exclude_start + vdev->mmap_exclude_len &&
+	       start + len > vdev->mmap_exclude_start;
+}
+
 #ifdef CONFIG_VFIO_PCI_VGA
 ssize_t vfio_pci_vga_rw(struct vfio_pci_core_device *vdev, char __user *buf,
 			size_t count, loff_t *ppos, bool iswrite);
diff --git a/include/linux/vfio_pci_core.h b/include/linux/vfio_pci_core.h
index 117cd67995d8..43755b91880f 100644
--- a/include/linux/vfio_pci_core.h
+++ b/include/linux/vfio_pci_core.h
@@ -162,6 +162,10 @@ struct vfio_pci_core_device {
 	struct notifier_block	nb;
 	struct rw_semaphore	memory_lock;
 	struct list_head	dmabufs;
+	/* BAR sub-range a provider keeps off mmap, reached only through a trap */
+	int			mmap_exclude_bar;
+	u64			mmap_exclude_start;
+	u64			mmap_exclude_len;
 };
 
 enum vfio_pci_io_width {
@@ -176,6 +180,8 @@ int vfio_pci_core_register_dev_region(struct vfio_pci_core_device *vdev,
 				      unsigned int type, unsigned int subtype,
 				      const struct vfio_pci_regops *ops,
 				      size_t size, u32 flags, void *data);
+void vfio_pci_core_set_mmap_exclude(struct vfio_pci_core_device *vdev, int bar,
+				    u64 start, u64 len);
 void vfio_pci_core_close_device(struct vfio_device *core_vdev);
 int vfio_pci_core_init_dev(struct vfio_device *core_vdev);
 void vfio_pci_core_release_dev(struct vfio_device *core_vdev);
-- 
2.25.1


  parent reply	other threads:[~2026-08-13  9:39 UTC|newest]

Thread overview: 28+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-13  9:36 [PATCH v4 00/27] vfio/pci: Add CXL Type-2 device passthrough support mhonap
2026-08-13  9:36 ` [PATCH v4 01/27] cxl: Fix resource.c include path and export cxl_restore_hdm_after_pci_reset mhonap
2026-08-13  9:36 ` [PATCH v4 02/27] cxl/regs: Skip sub-block region request for BAR-owning drivers mhonap
2026-08-13  9:36 ` [PATCH v4 03/27] cxl: Move component register defines to uapi/cxl/cxl_regs.h mhonap
2026-08-13  9:36 ` [PATCH v4 04/27] cxl: Establish media readiness in cxl_mem_probe() mhonap
2026-08-13  9:36 ` [PATCH v4 05/27] cxl: Add a function-scoped reset entry for vfio-pci mhonap
2026-08-13  9:36 ` [PATCH v4 06/27] vfio/pci: Add CXL ops registration interface mhonap
2026-08-13  9:36 ` [PATCH v4 07/27] vfio/pci: Detect CXL devices and load vfio-cxl on demand mhonap
2026-08-13  9:36 ` [PATCH v4 08/27] vfio/cxl: Add the vfio-cxl module skeleton mhonap
2026-08-13  9:36 ` [PATCH v4 09/27] vfio/cxl: Create the CXL memory device at bind mhonap
2026-08-13  9:36 ` [PATCH v4 10/27] vfio/cxl: Reject unsupported decoder topologies " mhonap
2026-08-13  9:36 ` [PATCH v4 11/27] vfio/cxl: Own the whole component register BAR mhonap
2026-08-13  9:36 ` mhonap [this message]
2026-08-13  9:36 ` [PATCH v4 13/27] vfio/pci: Refuse read/write to an excluded BAR sub-range mhonap
2026-08-13  9:36 ` [PATCH v4 14/27] vfio: Add CXL region type for the HDM region mhonap
2026-08-13  9:36 ` [PATCH v4 15/27] vfio/pci: Call CXL open and close hooks around device use mhonap
2026-08-13  9:36 ` [PATCH v4 16/27] vfio/cxl: Shadow the CXL DVSEC body at open mhonap
2026-08-13  9:36 ` [PATCH v4 17/27] vfio/cxl: Virtualize the CXL DVSEC mhonap
2026-08-13  9:36 ` [PATCH v4 18/27] vfio/cxl: Expose the HDM memory and trap the decoder registers mhonap
2026-08-13  9:36 ` [PATCH v4 19/27] vfio/cxl: Keep the HDM decoder block off the direct BAR mapping mhonap
2026-08-13  9:36 ` [PATCH v4 20/27] vfio/cxl: Emulate the HDM decoder commit handshake mhonap
2026-08-13  9:36 ` [PATCH v4 21/27] vfio/cxl: Describe the CXL device and decoder geometry to userspace mhonap
2026-08-13  9:36 ` [PATCH v4 22/27] vfio/cxl: Revoke the HDM mapping on reset and power transitions mhonap
2026-08-13  9:36 ` [PATCH v4 23/27] vfio/cxl: Refresh the decoder snapshot after a device reset mhonap
2026-08-13  9:36 ` [PATCH v4 24/27] vfio/cxl: Service a guest-triggered CXL reset mhonap
2026-08-13  9:36 ` [PATCH v4 25/27] vfio/pci: Provide an opt-out for the CXL Type-2 extensions mhonap
2026-08-13  9:36 ` [PATCH v4 26/27] Documentation: vfio-pci: Document CXL Type-2 device passthrough mhonap
2026-08-13  9:36 ` [PATCH v4 27/27] selftests/vfio: Add CXL Type-2 passthrough corner-case tests mhonap

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260813093631.2288172-13-mhonap@nvidia.com \
    --to=mhonap@nvidia.com \
    --cc=alejandro.lucero-palau@amd.com \
    --cc=alex@shazbot.org \
    --cc=alison.schofield@intel.com \
    --cc=ankita@nvidia.com \
    --cc=bhelgaas@google.com \
    --cc=cjia@nvidia.com \
    --cc=corbet@lwn.net \
    --cc=dave.jiang@intel.com \
    --cc=dave@stgolabs.net \
    --cc=dmatlack@google.com \
    --cc=gustavoars@kernel.org \
    --cc=iweiny@kernel.org \
    --cc=jgg@ziepe.ca \
    --cc=jic23@kernel.org \
    --cc=kees@kernel.org \
    --cc=kevin.tian@intel.com \
    --cc=kjaju@nvidia.com \
    --cc=kvm@vger.kernel.org \
    --cc=linux-cxl@vger.kernel.org \
    --cc=linux-doc@vger.kernel.org \
    --cc=linux-hardening@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-kselftest@vger.kernel.org \
    --cc=linux-pci@vger.kernel.org \
    --cc=ming.li@zohomail.com \
    --cc=skhan@linuxfoundation.org \
    --cc=skolothumtho@nvidia.com \
    --cc=smadhavan@nvidia.com \
    --cc=vishal.l.verma@intel.com \
    --cc=vsethi@nvidia.com \
    --cc=yishaih@nvidia.com \
    --cc=zhiw@nvidia.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox