* [RFC PATCH 1/6] KVM: guest_memfd: Add a writable result to get_pfn()
2026-10-05 9:55 ` [RFC PATCH 0/6] KVM: guest_memfd: back guest_memfd with an imported dma-buf Fred Griffoul
@ 2026-10-05 9:55 ` Fred Griffoul
2026-10-05 9:55 ` [RFC PATCH 2/6] dma-buf: Add get_phys() to describe a physical run Fred Griffoul
` (3 subsequent siblings)
4 siblings, 0 replies; 10+ messages in thread
From: Fred Griffoul @ 2026-10-05 9:55 UTC (permalink / raw)
To: Paolo Bonzini, Sean Christopherson, Marc Zyngier, Oliver Upton,
Sumit Semwal, Christian König, Jason Gunthorpe, Kevin Tian
Cc: David Woodhouse, Ackerley Tng, Joey Gouly, Suzuki K Poulose,
Zenghui Yu, Steffen Eiden, Catalin Marinas, Will Deacon,
Thomas Gleixner, Ingo Molnar, Borislav Petkov, Dave Hansen,
H . Peter Anvin, Joerg Roedel, Robin Murphy, Alex Williamson,
Shuah Khan, Steven Rostedt, Masami Hiramatsu, Mathieu Desnoyers,
linux-kernel, kvm, kvmarm, linux-arm-kernel, iommu, linux-media,
dri-devel, linaro-mm-sig, linux-kselftest, linux-trace-kernel,
x86
From: Fred Griffoul <fgriffo@amazon.co.uk>
A guest_memfd backing has no way to tell KVM that a page must stay
mapped but must not be written by the guest. get_pfn() returns only a
frame, and KVM takes write access from the memslot flags.
KVM_MEM_READONLY is not an option: KVM refuses it on guest_memfd slots,
because it emulates a write to a read-only slot as MMIO, which private
memory cannot support.
Add a writable output to get_pfn(). KVM sets it to true before the
call, and the backing may clear it. When it is false, the x86 and arm64
stage-2 fault paths map the page without write permission. A guest
write to that page then exits to userspace as a memory fault; KVM first
releases any page that the backing returned. Callers that do not map
the page pass NULL.
The result applies to stage-2 mappings only. When KVM writes guest
memory through the memslot's host address, the permissions of the host
mapping apply.
Signed-off-by: Fred Griffoul <fgriffo@amazon.co.uk>
---
arch/arm64/kvm/mmu.c | 13 ++++++++-----
arch/arm64/kvm/nested.c | 10 +++++++---
arch/x86/kvm/mmu/mmu.c | 18 ++++++++++++++++--
arch/x86/kvm/svm/sev.c | 4 ++--
include/linux/kvm_host.h | 11 ++++++++---
samples/kvm/gmem_provider.c | 3 ++-
virt/kvm/guest_memfd.c | 13 ++++++++-----
7 files changed, 51 insertions(+), 21 deletions(-)
diff --git a/arch/arm64/kvm/mmu.c b/arch/arm64/kvm/mmu.c
index 6c941aaa10c6..32e591edc69d 100644
--- a/arch/arm64/kvm/mmu.c
+++ b/arch/arm64/kvm/mmu.c
@@ -1607,7 +1607,7 @@ struct kvm_s2_fault_desc {
static int gmem_abort(const struct kvm_s2_fault_desc *s2fd)
{
- bool write_fault, exec_fault;
+ bool write_fault, exec_fault, writable;
bool perm_fault = kvm_vcpu_trap_is_permission_fault(s2fd->vcpu);
enum kvm_pgtable_walk_flags flags = KVM_PGTABLE_WALK_SHARED;
enum kvm_pgtable_prot prot = KVM_PGTABLE_PROT_R;
@@ -1641,14 +1641,17 @@ static int gmem_abort(const struct kvm_s2_fault_desc *s2fd)
/* Pairs with the smp_wmb() in kvm_mmu_invalidate_end(). */
smp_rmb();
- ret = kvm_gmem_get_pfn(kvm, s2fd->memslot, gfn, &pfn, &page, NULL);
- if (ret) {
+ ret = kvm_gmem_get_pfn(kvm, s2fd->memslot, gfn, &pfn, &page, NULL,
+ &writable);
+ if (ret || (write_fault && !writable)) {
kvm_prepare_memory_fault_exit(s2fd->vcpu, s2fd->fault_ipa, PAGE_SIZE,
write_fault, exec_fault, false);
- return ret;
+ if (!ret)
+ kvm_release_faultin_page(kvm, page, true, false);
+ return ret ?: -EFAULT;
}
- if (!(s2fd->memslot->flags & KVM_MEM_READONLY))
+ if (!(s2fd->memslot->flags & KVM_MEM_READONLY) && writable)
prot |= KVM_PGTABLE_PROT_W;
if (s2fd->nested)
diff --git a/arch/arm64/kvm/nested.c b/arch/arm64/kvm/nested.c
index fb54f6dad995..9ad1fa029835 100644
--- a/arch/arm64/kvm/nested.c
+++ b/arch/arm64/kvm/nested.c
@@ -1411,12 +1411,16 @@ static int kvm_translate_vncr(struct kvm_vcpu *vcpu, bool *is_gmem)
if (is_error_noslot_pfn(pfn) || (write_fault && !writable))
return -EFAULT;
} else {
- ret = kvm_gmem_get_pfn(vcpu->kvm, memslot, gfn, &pfn, &page, NULL);
- if (ret) {
+ ret = kvm_gmem_get_pfn(vcpu->kvm, memslot, gfn, &pfn, &page, NULL,
+ &writable);
+ if (ret || (write_fault && !writable)) {
kvm_prepare_memory_fault_exit(vcpu, vt->wr.pa, PAGE_SIZE,
write_fault, false, false);
- return ret;
+ if (!ret)
+ kvm_release_faultin_page(vcpu->kvm, page, true, false);
+ return ret ?: -EFAULT;
}
+ vt->wr.pw &= writable;
}
scoped_guard(write_lock, &vcpu->kvm->mmu_lock) {
diff --git a/arch/x86/kvm/mmu/mmu.c b/arch/x86/kvm/mmu/mmu.c
index 234d0a95abf5..17ac2bfdc164 100644
--- a/arch/x86/kvm/mmu/mmu.c
+++ b/arch/x86/kvm/mmu/mmu.c
@@ -4612,6 +4612,7 @@ static void kvm_mmu_finish_page_fault(struct kvm_vcpu *vcpu,
static int kvm_mmu_faultin_pfn_gmem(struct kvm_vcpu *vcpu,
struct kvm_page_fault *fault)
{
+ bool writable;
int max_order, r;
if (!kvm_slot_has_gmem(fault->slot)) {
@@ -4620,13 +4621,26 @@ static int kvm_mmu_faultin_pfn_gmem(struct kvm_vcpu *vcpu,
}
r = kvm_gmem_get_pfn(vcpu->kvm, fault->slot, fault->gfn, &fault->pfn,
- &fault->refcounted_page, &max_order);
+ &fault->refcounted_page, &max_order, &writable);
if (r) {
kvm_mmu_prepare_memory_fault_exit(vcpu, fault);
return r;
}
- fault->map_writable = !(fault->slot->flags & KVM_MEM_READONLY);
+ /*
+ * The memory's owner has the final say on writability, on top of the
+ * memslot flag: a page it reports read-only is mapped read-only, and a
+ * guest write to it exits to userspace rather than being installed.
+ */
+ fault->map_writable = !(fault->slot->flags & KVM_MEM_READONLY) &&
+ writable;
+ if (fault->write && !fault->map_writable) {
+ kvm_mmu_prepare_memory_fault_exit(vcpu, fault);
+ kvm_release_faultin_page(vcpu->kvm, fault->refcounted_page,
+ true, false);
+ fault->refcounted_page = NULL;
+ return -EFAULT;
+ }
fault->max_level = kvm_max_level_for_order(max_order);
return RET_PF_CONTINUE;
diff --git a/arch/x86/kvm/svm/sev.c b/arch/x86/kvm/svm/sev.c
index 125779c82bc4..983e19f7dbd2 100644
--- a/arch/x86/kvm/svm/sev.c
+++ b/arch/x86/kvm/svm/sev.c
@@ -4063,7 +4063,7 @@ static void sev_snp_init_protected_guest_state(struct kvm_vcpu *vcpu)
* The new VMSA will be private memory guest memory, so retrieve the
* PFN from the gmem backend.
*/
- if (kvm_gmem_get_pfn(vcpu->kvm, slot, gfn, &pfn, &page, NULL))
+ if (kvm_gmem_get_pfn(vcpu->kvm, slot, gfn, &pfn, &page, NULL, NULL))
return;
/*
@@ -4996,7 +4996,7 @@ void sev_handle_rmp_fault(struct kvm_vcpu *vcpu, gpa_t gpa, u64 error_code)
return;
}
- ret = kvm_gmem_get_pfn(kvm, slot, gfn, &pfn, &page, &order);
+ ret = kvm_gmem_get_pfn(kvm, slot, gfn, &pfn, &page, &order, NULL);
if (ret) {
pr_warn_ratelimited("SEV: Unexpected RMP fault, no backing page for private GPA 0x%llx\n",
gpa);
diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h
index 04fa0cb126f6..f84f3ab44acf 100644
--- a/include/linux/kvm_host.h
+++ b/include/linux/kvm_host.h
@@ -660,9 +660,14 @@ struct kvm_gmem_ops {
struct kvm_memory_slot *slot, loff_t offset);
void (*unbind)(struct file *file, struct kvm *kvm,
struct kvm_memory_slot *slot);
+ /*
+ * @writable: [out] clear to have KVM map the page read-only; a guest
+ * write then exits as a memory fault. NULL if the caller does not care.
+ */
int (*get_pfn)(struct file *file, struct kvm *kvm,
struct kvm_memory_slot *slot, gfn_t gfn,
- kvm_pfn_t *pfn, struct page **page, int *max_order);
+ kvm_pfn_t *pfn, struct page **page, int *max_order,
+ bool *writable);
int (*populate)(struct file *file, struct kvm *kvm,
struct kvm_memory_slot *slot, gfn_t gfn,
kvm_pfn_t *pfn, struct page *src_page, int order);
@@ -2651,12 +2656,12 @@ static inline bool kvm_mem_is_private(struct kvm *kvm, gfn_t gfn)
#ifdef CONFIG_KVM_GUEST_MEMFD
int kvm_gmem_get_pfn(struct kvm *kvm, struct kvm_memory_slot *slot,
gfn_t gfn, kvm_pfn_t *pfn, struct page **page,
- int *max_order);
+ int *max_order, bool *writable);
#else
static inline int kvm_gmem_get_pfn(struct kvm *kvm,
struct kvm_memory_slot *slot, gfn_t gfn,
kvm_pfn_t *pfn, struct page **page,
- int *max_order)
+ int *max_order, bool *writable)
{
KVM_BUG_ON(1, kvm);
return -EIO;
diff --git a/samples/kvm/gmem_provider.c b/samples/kvm/gmem_provider.c
index 9728f5a8029b..75197c088762 100644
--- a/samples/kvm/gmem_provider.c
+++ b/samples/kvm/gmem_provider.c
@@ -157,7 +157,8 @@ static int gmem_max_order(struct gmem_info *info, gfn_t gfn, unsigned long index
static int gmem_get_pfn(struct file *file, struct kvm *kvm,
struct kvm_memory_slot *slot, gfn_t gfn,
- kvm_pfn_t *pfn, struct page **page, int *max_order)
+ kvm_pfn_t *pfn, struct page **page, int *max_order,
+ bool *writable)
{
struct gmem_info *info = to_gmem_info(file);
pgoff_t index = gfn - slot->base_gfn + slot->gmem.pgoff;
diff --git a/virt/kvm/guest_memfd.c b/virt/kvm/guest_memfd.c
index d284bb70fe05..0f4bf2cc5e8e 100644
--- a/virt/kvm/guest_memfd.c
+++ b/virt/kvm/guest_memfd.c
@@ -624,7 +624,7 @@ static void kvm_gmem_native_unbind(struct file *slot_file, struct kvm *kvm,
static int kvm_gmem_native_get_pfn(struct file *file, struct kvm *kvm,
struct kvm_memory_slot *slot, gfn_t gfn,
kvm_pfn_t *pfn, struct page **page,
- int *max_order);
+ int *max_order, bool *writable);
static void kvm_gmem_native_release(struct file *file);
static int kvm_gmem_native_mmap(struct file *file,
struct vm_area_struct *vma);
@@ -988,7 +988,7 @@ static struct folio *__kvm_gmem_get_pfn(struct file *file,
static int kvm_gmem_native_get_pfn(struct file *file, struct kvm *kvm,
struct kvm_memory_slot *slot, gfn_t gfn,
kvm_pfn_t *pfn, struct page **page,
- int *max_order)
+ int *max_order, bool *writable)
{
pgoff_t index = kvm_gmem_get_index(slot, gfn);
struct folio *folio;
@@ -1016,7 +1016,7 @@ static int kvm_gmem_native_get_pfn(struct file *file, struct kvm *kvm,
int kvm_gmem_get_pfn(struct kvm *kvm, struct kvm_memory_slot *slot,
gfn_t gfn, kvm_pfn_t *pfn, struct page **page,
- int *max_order)
+ int *max_order, bool *writable)
{
const struct kvm_gmem_ops *ops;
@@ -1029,7 +1029,10 @@ int kvm_gmem_get_pfn(struct kvm *kvm, struct kvm_memory_slot *slot,
return -EFAULT;
*page = NULL;
- return ops->get_pfn(file, kvm, slot, gfn, pfn, page, max_order);
+ if (writable)
+ *writable = true;
+ return ops->get_pfn(file, kvm, slot, gfn, pfn, page, max_order,
+ writable);
}
EXPORT_SYMBOL_FOR_KVM_INTERNAL(kvm_gmem_get_pfn);
@@ -1093,7 +1096,7 @@ static int kvm_gmem_populate_one(const struct kvm_gmem_ops *ops,
src_page, 0);
else
ret = ops->get_pfn(file, kvm, slot, gfn, &pfn,
- &ignored_page, NULL);
+ &ignored_page, NULL, NULL);
if (ret)
return ret;
--
2.47.3
^ permalink raw reply related [flat|nested] 10+ messages in thread* [RFC PATCH 2/6] dma-buf: Add get_phys() to describe a physical run
2026-10-05 9:55 ` [RFC PATCH 0/6] KVM: guest_memfd: back guest_memfd with an imported dma-buf Fred Griffoul
2026-10-05 9:55 ` [RFC PATCH 1/6] KVM: guest_memfd: Add a writable result to get_pfn() Fred Griffoul
@ 2026-10-05 9:55 ` Fred Griffoul
2026-10-05 10:07 ` Christian König
2026-10-05 9:55 ` [RFC PATCH 3/6] dma-buf: Add ranged mapping invalidation Fred Griffoul
` (2 subsequent siblings)
4 siblings, 1 reply; 10+ messages in thread
From: Fred Griffoul @ 2026-10-05 9:55 UTC (permalink / raw)
To: Paolo Bonzini, Sean Christopherson, Marc Zyngier, Oliver Upton,
Sumit Semwal, Christian König, Jason Gunthorpe, Kevin Tian
Cc: David Woodhouse, Ackerley Tng, Joey Gouly, Suzuki K Poulose,
Zenghui Yu, Steffen Eiden, Catalin Marinas, Will Deacon,
Thomas Gleixner, Ingo Molnar, Borislav Petkov, Dave Hansen,
H . Peter Anvin, Joerg Roedel, Robin Murphy, Alex Williamson,
Shuah Khan, Steven Rostedt, Masami Hiramatsu, Mathieu Desnoyers,
linux-kernel, kvm, kvmarm, linux-arm-kernel, iommu, linux-media,
dri-devel, linaro-mm-sig, linux-kselftest, linux-trace-kernel,
x86
From: Fred Griffoul <fgriffo@amazon.co.uk>
iommufd and KVM write physical addresses into their own page tables.
To do so, they must ask the exporter which frames back an offset of a
dma-buf, whether that memory is RAM or MMIO, and whether it may be
written.
Add a get_phys() operation. The importer passes an offset and a maximum
length. The exporter reports one run: the frames that start at the
offset, are backed and physically contiguous, and share one attribute
word. The run never exceeds the length. The exporter may end it early,
so importers must not assume that it is the longest possible run.
get_phys() returns -ENOENT when the byte at the offset is not backed,
and -ENODEV when the buffer is revoked. The caller holds the
reservation, and either pins the attachment or handles revocation. A
reported frame stays valid until an invalidation that covers it
returns.
The attribute word holds the memory type and a READONLY flag. Zero
means writable RAM. Importers refuse unknown types, reserved bits and
unknown flags, so attributes added later fail safely. Two flag bits are
reserved: one for holes and one for confidential memory.
Convert vfio-pci, the iommufd selftest exporter and the KVM sample.
iommufd behaves as before: it maps a buffer only when one writable run
covers all of it.
Signed-off-by: Fred Griffoul <fgriffo@amazon.co.uk>
---
drivers/dma-buf/dma-buf.c | 47 +++++++++++
drivers/iommu/iommufd/iommufd_private.h | 8 --
drivers/iommu/iommufd/iommufd_test.h | 16 ++++
drivers/iommu/iommufd/pages.c | 74 ++++-------------
drivers/iommu/iommufd/selftest.c | 106 ++++++++++++++++++++----
drivers/vfio/pci/vfio_pci_dmabuf.c | 58 ++++++-------
include/linux/dma-buf.h | 54 ++++++++++++
include/linux/vfio_pci_core.h | 3 -
samples/kvm/gmem_provider.c | 39 ++++-----
9 files changed, 267 insertions(+), 138 deletions(-)
diff --git a/drivers/dma-buf/dma-buf.c b/drivers/dma-buf/dma-buf.c
index d504c636dc29..66b85d53ed22 100644
--- a/drivers/dma-buf/dma-buf.c
+++ b/drivers/dma-buf/dma-buf.c
@@ -1389,6 +1389,53 @@ void dma_buf_invalidate_mappings(struct dma_buf *dmabuf)
}
EXPORT_SYMBOL_NS_GPL(dma_buf_invalidate_mappings, "DMA_BUF");
+/**
+ * dma_buf_get_phys - describe the run that starts at an offset
+ * @attach: attachment to query
+ * @offset: first buffer byte to describe
+ * @len: maximum number of bytes to describe
+ * @phys: physical address and length of the run
+ * @attr: DMA_BUF_PHYS_ATTR_* word of the run
+ *
+ * A run is the longest stretch of backed bytes starting at @offset whose
+ * frames are physically contiguous and share one attribute word. On success
+ * *@phys starts at the byte at @offset and covers at most @len bytes; it may
+ * be shorter than the run.
+ *
+ * The dma-buf reservation must be held. The attachment must be pinned or have
+ * revocable importer operations. A frame remains valid until a covering
+ * invalidation callback returns; the exporter must invalidate a changed range
+ * before reusing its old frames.
+ *
+ * Returns:
+ *
+ * 0 on success, -ENOENT if the byte at @offset is not backed, -ENODEV if the
+ * buffer is revoked, -EOPNOTSUPP if the exporter cannot describe itself this
+ * way, or another negative error code.
+ */
+int dma_buf_get_phys(struct dma_buf_attachment *attach, u64 offset, u64 len,
+ struct phys_vec *phys, u32 *attr)
+{
+ u64 end;
+ int ret;
+
+ if (WARN_ON_ONCE(!attach || !attach->dmabuf || !phys || !attr))
+ return -EINVAL;
+ if (!len || check_add_overflow(offset, len, &end) ||
+ end > attach->dmabuf->size)
+ return -EINVAL;
+
+ dma_resv_assert_held(attach->dmabuf->resv);
+ if (!attach->dmabuf->ops->get_phys)
+ return -EOPNOTSUPP;
+
+ ret = attach->dmabuf->ops->get_phys(attach, offset, len, phys, attr);
+ if (!ret && WARN_ON_ONCE(!phys->len || phys->len > len))
+ return -EIO;
+ return ret;
+}
+EXPORT_SYMBOL_NS_GPL(dma_buf_get_phys, "DMA_BUF");
+
/**
* DOC: cpu access
*
diff --git a/drivers/iommu/iommufd/iommufd_private.h b/drivers/iommu/iommufd/iommufd_private.h
index 43fbc5bed8de..5cded585c227 100644
--- a/drivers/iommu/iommufd/iommufd_private.h
+++ b/drivers/iommu/iommufd/iommufd_private.h
@@ -716,8 +716,6 @@ bool iommufd_should_fail(void);
int __init iommufd_test_init(void);
void iommufd_test_exit(void);
bool iommufd_selftest_is_mock_dev(struct device *dev);
-int iommufd_test_dma_buf_iommufd_map(struct dma_buf_attachment *attachment,
- struct phys_vec *phys);
#else
static inline void iommufd_test_syz_conv_iova_id(struct iommufd_ucmd *ucmd,
unsigned int ioas_id,
@@ -739,11 +737,5 @@ static inline bool iommufd_selftest_is_mock_dev(struct device *dev)
{
return false;
}
-static inline int
-iommufd_test_dma_buf_iommufd_map(struct dma_buf_attachment *attachment,
- struct phys_vec *phys)
-{
- return -EOPNOTSUPP;
-}
#endif
#endif
diff --git a/drivers/iommu/iommufd/iommufd_test.h b/drivers/iommu/iommufd/iommufd_test.h
index 52b78cbcc920..28fd9c43edc4 100644
--- a/drivers/iommu/iommufd/iommufd_test.h
+++ b/drivers/iommu/iommufd/iommufd_test.h
@@ -31,6 +31,8 @@ enum {
IOMMU_TEST_OP_PASID_CHECK_HWPT,
IOMMU_TEST_OP_DMABUF_GET,
IOMMU_TEST_OP_DMABUF_REVOKE,
+ IOMMU_TEST_OP_MD_CHECK_MAPPED,
+ IOMMU_TEST_OP_MD_IOVA_TO_PHYS,
};
enum {
@@ -193,6 +195,20 @@ struct iommu_test_cmd {
__s32 dmabuf_fd;
__u32 revoked;
} dmabuf_revoke;
+ struct {
+ /*
+ * 1: every page in [iova, iova+length) must be mapped;
+ * 0: none of them may be. Mixed is an error.
+ */
+ __u32 mapped;
+ __u32 __reserved;
+ __aligned_u64 iova;
+ __aligned_u64 length;
+ } check_mapped;
+ struct {
+ __aligned_u64 iova;
+ __aligned_u64 out_phys; /* 0 if unmapped */
+ } iova_to_phys;
};
__u32 last;
};
diff --git a/drivers/iommu/iommufd/pages.c b/drivers/iommu/iommufd/pages.c
index f9b2ae6d7e96..196d1bb330c2 100644
--- a/drivers/iommu/iommufd/pages.c
+++ b/drivers/iommu/iommufd/pages.c
@@ -1463,68 +1463,12 @@ static const struct dma_buf_attach_ops iopt_dmabuf_attach_revoke_ops = {
.invalidate_mappings = iopt_revoke_notify,
};
-/*
- * iommufd and vfio have a circular dependency. Future work for a phys
- * based private interconnect will remove this.
- */
-/*
- * Look up the exporter's phys accessor for iommufd's private-interconnect
- * path. Also fills *is_cpu_ram: true if the exporter's memory is normal
- * cache-coherent RAM (needs BATCH_CPU_MEMORY / IOMMU_CACHE), false for MMIO
- * (needs BATCH_MMIO / IOMMU_MMIO). This will be replaced by a formal
- * exporter op that returns phys + memory type together.
- */
-static int
-sym_vfio_pci_dma_buf_iommufd_map(struct dma_buf_attachment *attachment,
- struct phys_vec *phys, bool *is_cpu_ram)
-{
- typeof(&vfio_pci_dma_buf_iommufd_map) fn;
- int rc;
-
- rc = iommufd_test_dma_buf_iommufd_map(attachment, phys);
- if (rc != -EOPNOTSUPP) {
- *is_cpu_ram = false; /* test hook mimics VFIO MMIO */
- return rc;
- }
-
- /*
- * Prototype: try the sample gmem provider's dma-buf exporter. This
- * mirrors the vfio-pci private-interconnect hook, and (like it) is
- * meant to be replaced by a formal negotiated exporter op returning
- * phys + memory type. The provider serves RAM, so mark it CPU_RAM.
- */
- {
- extern int gmem_provider_dma_buf_iommufd_map(
- struct dma_buf_attachment *, struct phys_vec *);
- typeof(&gmem_provider_dma_buf_iommufd_map) gfn;
-
- gfn = symbol_get(gmem_provider_dma_buf_iommufd_map);
- if (gfn) {
- rc = gfn(attachment, phys);
- symbol_put(gmem_provider_dma_buf_iommufd_map);
- if (rc != -EOPNOTSUPP) {
- *is_cpu_ram = true;
- return rc;
- }
- }
- }
-
- if (!IS_ENABLED(CONFIG_VFIO_PCI_DMABUF))
- return -EOPNOTSUPP;
-
- fn = symbol_get(vfio_pci_dma_buf_iommufd_map);
- if (!fn)
- return -EOPNOTSUPP;
- rc = fn(attachment, phys);
- symbol_put(vfio_pci_dma_buf_iommufd_map);
- *is_cpu_ram = false; /* VFIO PCI dma-buf carries BAR (MMIO) memory */
- return rc;
-}
-
static int iopt_map_dmabuf(struct iommufd_ctx *ictx, struct iopt_pages *pages,
struct dma_buf *dmabuf)
{
struct dma_buf_attachment *attach;
+ struct phys_vec pv;
+ u32 attr;
int rc;
attach = dma_buf_dynamic_attach(dmabuf, iommufd_global_device(),
@@ -1546,10 +1490,20 @@ static int iopt_map_dmabuf(struct iommufd_ctx *ictx, struct iopt_pages *pages,
if (rc)
goto err_detach;
- rc = sym_vfio_pci_dma_buf_iommufd_map(attach, &pages->dmabuf.phys,
- &pages->dmabuf.is_cpu_ram);
+ /* One backed, writable run covering the buffer: refuse the rest. */
+ rc = dma_buf_get_phys(attach, 0, dmabuf->size, &pv, &attr);
+ if (rc == -ENOENT)
+ rc = -EOPNOTSUPP;
if (rc)
goto err_unpin;
+ if (pv.len != dmabuf->size || !dma_buf_phys_attr_known(attr) ||
+ (attr & DMA_BUF_PHYS_ATTR_FLAGS_MASK)) {
+ rc = -EOPNOTSUPP;
+ goto err_unpin;
+ }
+ pages->dmabuf.phys = pv;
+ pages->dmabuf.is_cpu_ram =
+ dma_buf_phys_attr_type(attr) == DMA_BUF_PHYS_ATTR_RAM;
dma_resv_unlock(dmabuf->resv);
diff --git a/drivers/iommu/iommufd/selftest.c b/drivers/iommu/iommufd/selftest.c
index af07c642a526..0899272d1e66 100644
--- a/drivers/iommu/iommufd/selftest.c
+++ b/drivers/iommu/iommufd/selftest.c
@@ -1962,32 +1962,31 @@ static void iommufd_test_dma_buf_release(struct dma_buf *dmabuf)
kfree(priv);
}
-static const struct dma_buf_ops iommufd_test_dmabuf_ops = {
- .attach = iommufd_test_dma_buf_attach,
- .detach = iommufd_test_dma_buf_detach,
- .map_dma_buf = iommufd_test_dma_buf_map,
- .release = iommufd_test_dma_buf_release,
- .unmap_dma_buf = iommufd_test_dma_buf_unmap,
-};
-
-int iommufd_test_dma_buf_iommufd_map(struct dma_buf_attachment *attachment,
- struct phys_vec *phys)
+static int iommufd_test_dma_buf_get_phys(struct dma_buf_attachment *attachment,
+ u64 offset, u64 len,
+ struct phys_vec *phys, u32 *attr)
{
struct iommufd_test_dma_buf *priv = attachment->dmabuf->priv;
dma_resv_assert_held(attachment->dmabuf->resv);
-
- if (attachment->dmabuf->ops != &iommufd_test_dmabuf_ops)
- return -EOPNOTSUPP;
-
if (priv->revoked)
return -ENODEV;
- phys->paddr = virt_to_phys(priv->memory);
- phys->len = priv->length;
+ phys->paddr = virt_to_phys(priv->memory) + offset;
+ phys->len = len;
+ *attr = DMA_BUF_PHYS_ATTR_MMIO;
return 0;
}
+static const struct dma_buf_ops iommufd_test_dmabuf_ops = {
+ .attach = iommufd_test_dma_buf_attach,
+ .detach = iommufd_test_dma_buf_detach,
+ .map_dma_buf = iommufd_test_dma_buf_map,
+ .release = iommufd_test_dma_buf_release,
+ .unmap_dma_buf = iommufd_test_dma_buf_unmap,
+ .get_phys = iommufd_test_dma_buf_get_phys,
+};
+
static int iommufd_test_dmabuf_get(struct iommufd_ucmd *ucmd,
unsigned int open_flags,
size_t len)
@@ -2031,6 +2030,73 @@ static int iommufd_test_dmabuf_get(struct iommufd_ucmd *ucmd,
return rc;
}
+/*
+ * Report the physical address the mock domain resolves @iova to, or 0 if
+ * it is unmapped. Lets a test check that two IOVAs share one frame (a
+ * scratch substitution) without knowing the frame in advance.
+ */
+static int iommufd_test_md_iova_to_phys(struct iommufd_ucmd *ucmd,
+ unsigned int mockpt_id,
+ unsigned long iova)
+{
+ struct iommu_test_cmd *cmd = ucmd->cmd;
+ struct iommufd_hw_pagetable *hwpt;
+ struct mock_iommu_domain *mock;
+ unsigned int page_size;
+ int rc;
+
+ hwpt = get_md_pagetable(ucmd, mockpt_id, &mock);
+ if (IS_ERR(hwpt))
+ return PTR_ERR(hwpt);
+
+ page_size = 1 << __ffs(mock->domain.pgsize_bitmap);
+ if (iova % page_size) {
+ rc = -EINVAL;
+ goto out_put;
+ }
+ cmd->iova_to_phys.out_phys =
+ mock->domain.ops->iova_to_phys(&mock->domain, iova);
+ rc = iommufd_ucmd_respond(ucmd, sizeof(*cmd));
+out_put:
+ iommufd_put_object(ucmd->ictx, &hwpt->obj);
+ return rc;
+}
+
+static int iommufd_test_md_check_mapped(struct iommufd_ucmd *ucmd,
+ unsigned int mockpt_id,
+ unsigned long iova, size_t length,
+ bool mapped)
+{
+ struct iommufd_hw_pagetable *hwpt;
+ struct mock_iommu_domain *mock;
+ unsigned int page_size;
+ int rc = 0;
+
+ hwpt = get_md_pagetable(ucmd, mockpt_id, &mock);
+ if (IS_ERR(hwpt))
+ return PTR_ERR(hwpt);
+
+ page_size = 1 << __ffs(mock->domain.pgsize_bitmap);
+ if (iova % page_size || length % page_size || !length) {
+ rc = -EINVAL;
+ goto out_put;
+ }
+
+ for (; length; length -= page_size, iova += page_size) {
+ bool is_mapped =
+ mock->domain.ops->iova_to_phys(&mock->domain, iova) != 0;
+
+ if (is_mapped != mapped) {
+ rc = -ENOENT;
+ goto out_put;
+ }
+ }
+
+out_put:
+ iommufd_put_object(ucmd->ictx, &hwpt->obj);
+ return rc;
+}
+
static int iommufd_test_dmabuf_revoke(struct iommufd_ucmd *ucmd, int fd,
bool revoked)
{
@@ -2143,6 +2209,14 @@ int iommufd_test(struct iommufd_ucmd *ucmd)
return iommufd_test_dmabuf_revoke(ucmd,
cmd->dmabuf_revoke.dmabuf_fd,
cmd->dmabuf_revoke.revoked);
+ case IOMMU_TEST_OP_MD_CHECK_MAPPED:
+ return iommufd_test_md_check_mapped(ucmd, cmd->id,
+ cmd->check_mapped.iova,
+ cmd->check_mapped.length,
+ cmd->check_mapped.mapped);
+ case IOMMU_TEST_OP_MD_IOVA_TO_PHYS:
+ return iommufd_test_md_iova_to_phys(ucmd, cmd->id,
+ cmd->iova_to_phys.iova);
default:
return -EOPNOTSUPP;
}
diff --git a/drivers/vfio/pci/vfio_pci_dmabuf.c b/drivers/vfio/pci/vfio_pci_dmabuf.c
index c16f460c01d6..381c3d338e8c 100644
--- a/drivers/vfio/pci/vfio_pci_dmabuf.c
+++ b/drivers/vfio/pci/vfio_pci_dmabuf.c
@@ -99,46 +99,46 @@ static void vfio_pci_dma_buf_release(struct dma_buf *dmabuf)
kfree(priv);
}
-static const struct dma_buf_ops vfio_pci_dmabuf_ops = {
- .attach = vfio_pci_dma_buf_attach,
- .map_dma_buf = vfio_pci_dma_buf_map,
- .unmap_dma_buf = vfio_pci_dma_buf_unmap,
- .release = vfio_pci_dma_buf_release,
-};
-
/*
- * This is a temporary "private interconnect" between VFIO DMABUF and iommufd.
- * It allows the two co-operating drivers to exchange the physical address of
- * the BAR. This is to be replaced with a formal DMABUF system for negotiated
- * interconnect types.
+ * Report the BAR's physical range for importers which program it into their own
+ * translation tables, such as iommufd. A BAR is MMIO, never cache-coherent RAM.
*
- * If this function succeeds the following are true:
- * - There is one physical range and it is pointing to MMIO
- * - When move_notify is called it means revoke, not move, vfio_dma_buf_map
- * will fail if it is currently revoked
+ * When move_notify is called it means revoke, not move, so this fails while
+ * revoked and vfio_dma_buf_map() does the same.
*/
-int vfio_pci_dma_buf_iommufd_map(struct dma_buf_attachment *attachment,
- struct phys_vec *phys)
+static int vfio_pci_dma_buf_get_phys(struct dma_buf_attachment *attachment,
+ u64 offset, u64 len,
+ struct phys_vec *phys, u32 *attr)
{
- struct vfio_pci_dma_buf *priv;
+ struct vfio_pci_dma_buf *priv = attachment->dmabuf->priv;
+ u32 i;
dma_resv_assert_held(attachment->dmabuf->resv);
-
- if (attachment->dmabuf->ops != &vfio_pci_dmabuf_ops)
- return -EOPNOTSUPP;
-
- priv = attachment->dmabuf->priv;
if (priv->revoked)
return -ENODEV;
- /* More than one range to iommufd will require proper DMABUF support */
- if (priv->nr_ranges != 1)
- return -EOPNOTSUPP;
-
- *phys = priv->phys_vec[0];
+ /* Report from @offset to the end of the BAR range containing it. */
+ for (i = 0; i < priv->nr_ranges; i++) {
+ if (offset < priv->phys_vec[i].len)
+ break;
+ offset -= priv->phys_vec[i].len;
+ }
+ if (i == priv->nr_ranges)
+ return -EINVAL;
+ phys->paddr = priv->phys_vec[i].paddr + offset;
+ phys->len = min_t(u64, priv->phys_vec[i].len - offset, len);
+ *attr = DMA_BUF_PHYS_ATTR_MMIO;
return 0;
}
-EXPORT_SYMBOL_FOR_MODULES(vfio_pci_dma_buf_iommufd_map, "iommufd");
+
+static const struct dma_buf_ops vfio_pci_dmabuf_ops = {
+ .attach = vfio_pci_dma_buf_attach,
+ .map_dma_buf = vfio_pci_dma_buf_map,
+ .unmap_dma_buf = vfio_pci_dma_buf_unmap,
+ .release = vfio_pci_dma_buf_release,
+ .get_phys = vfio_pci_dma_buf_get_phys,
+};
+
int vfio_pci_core_fill_phys_vec(struct phys_vec *phys_vec,
struct vfio_region_dma_range *dma_ranges,
diff --git a/include/linux/dma-buf.h b/include/linux/dma-buf.h
index d1203da56fc5..b223962e20c2 100644
--- a/include/linux/dma-buf.h
+++ b/include/linux/dma-buf.h
@@ -13,6 +13,7 @@
#ifndef __DMA_BUF_H__
#define __DMA_BUF_H__
+#include <linux/bitfield.h>
#include <linux/iosys-map.h>
#include <linux/file.h>
#include <linux/err.h>
@@ -23,6 +24,7 @@
#include <linux/dma-fence.h>
#include <linux/wait.h>
#include <linux/pci-p2pdma.h>
+#include <linux/types.h>
struct device;
struct dma_buf;
@@ -182,6 +184,29 @@ struct dma_buf_ops {
struct sg_table *,
enum dma_data_direction);
+ /**
+ * @get_phys:
+ *
+ * Describe the run that starts at @offset, for an importer that
+ * programs its own translation tables. A run is the longest stretch
+ * of backed bytes whose frames are physically contiguous and share
+ * one attribute word. Report it in *@phys, starting at the byte at
+ * @offset and clipped at @offset + @len, with its DMA_BUF_PHYS_ATTR_*
+ * word in *@attr. The exporter may stop before the end of the run;
+ * importers must not assume the reported run is maximal.
+ *
+ * Return 0 on success, -ENOENT if the byte at @offset is not backed,
+ * -ENODEV if the buffer is revoked, or another negative error. Do not
+ * wait for memory to become available.
+ *
+ * The dma-buf reservation is held. The attachment must be pinned or
+ * have revocable importer operations. A reported frame remains valid
+ * until a covering invalidation callback returns; an exporter must
+ * invalidate every change before reusing an old frame.
+ */
+ int (*get_phys)(struct dma_buf_attachment *attach, u64 offset, u64 len,
+ struct phys_vec *phys, u32 *attr);
+
/* TODO: Add try_map_dma_buf version, to return immed with -EBUSY
* if the call would block.
*/
@@ -576,6 +601,35 @@ void dma_buf_unmap_attachment(struct dma_buf_attachment *, struct sg_table *,
enum dma_data_direction);
void dma_buf_invalidate_mappings(struct dma_buf *dma_buf);
bool dma_buf_attach_revocable(struct dma_buf_attachment *attach);
+/* bits 0-7: memory type (a value, not flags) */
+#define DMA_BUF_PHYS_ATTR_TYPE_MASK GENMASK(7, 0)
+#define DMA_BUF_PHYS_ATTR_RAM 0x00 /* cache-coherent system RAM */
+#define DMA_BUF_PHYS_ATTR_MMIO 0x01 /* device MMIO, uncached */
+/* bits 8-15: reserved for a second value field; must be zero */
+#define DMA_BUF_PHYS_ATTR_RSVD_MASK GENMASK(15, 8)
+/* bits 16-31: flags; undefined bits must be zero */
+#define DMA_BUF_PHYS_ATTR_READONLY BIT(16)
+/* BIT(17): reserved (hole, for importers that walk across gaps) */
+/* BIT(18): reserved (private, for confidential computing) */
+#define DMA_BUF_PHYS_ATTR_FLAGS_MASK (DMA_BUF_PHYS_ATTR_READONLY)
+
+static inline u32 dma_buf_phys_attr_type(u32 attrs)
+{
+ return FIELD_GET(DMA_BUF_PHYS_ATTR_TYPE_MASK, attrs);
+}
+
+static inline bool dma_buf_phys_attr_known(u32 attrs)
+{
+ return dma_buf_phys_attr_type(attrs) <= DMA_BUF_PHYS_ATTR_MMIO &&
+ !(attrs & DMA_BUF_PHYS_ATTR_RSVD_MASK) &&
+ !(attrs & ~(DMA_BUF_PHYS_ATTR_TYPE_MASK |
+ DMA_BUF_PHYS_ATTR_RSVD_MASK |
+ DMA_BUF_PHYS_ATTR_FLAGS_MASK));
+}
+
+int dma_buf_get_phys(struct dma_buf_attachment *attach, u64 offset, u64 len,
+ struct phys_vec *phys, u32 *attr);
+
int dma_buf_begin_cpu_access(struct dma_buf *dma_buf,
enum dma_data_direction dir);
int dma_buf_end_cpu_access(struct dma_buf *dma_buf,
diff --git a/include/linux/vfio_pci_core.h b/include/linux/vfio_pci_core.h
index 9a1674c152aa..2a1d13abdb0c 100644
--- a/include/linux/vfio_pci_core.h
+++ b/include/linux/vfio_pci_core.h
@@ -257,7 +257,4 @@ vfio_pci_core_get_iomap(struct vfio_pci_core_device *vdev, unsigned int bar)
return vdev->barmap[bar];
}
-int vfio_pci_dma_buf_iommufd_map(struct dma_buf_attachment *attachment,
- struct phys_vec *phys);
-
#endif /* VFIO_PCI_CORE_H */
diff --git a/samples/kvm/gmem_provider.c b/samples/kvm/gmem_provider.c
index 75197c088762..b6824fe5d228 100644
--- a/samples/kvm/gmem_provider.c
+++ b/samples/kvm/gmem_provider.c
@@ -487,35 +487,30 @@ static void gmem_dma_buf_release(struct dma_buf *dmabuf)
kfree(priv);
}
-static const struct dma_buf_ops gmem_dma_buf_ops = {
- .attach = gmem_dma_buf_attach,
- .map_dma_buf = gmem_dma_buf_map,
- .unmap_dma_buf = gmem_dma_buf_unmap,
- .release = gmem_dma_buf_release,
-};
-
-/*
- * Private interconnect for iommufd (mirrors vfio_pci_dma_buf_iommufd_map).
- * Returns the single contiguous phys range for the exported region so iommufd
- * can program the IOMMU directly, bypassing the DMA API.
- */
-int gmem_provider_dma_buf_iommufd_map(struct dma_buf_attachment *attach,
- struct phys_vec *phys);
-int gmem_provider_dma_buf_iommufd_map(struct dma_buf_attachment *attach,
- struct phys_vec *phys)
+/* Report this flat sample region through the generic dma-buf operation. */
+static int gmem_dma_buf_get_phys(struct dma_buf_attachment *attach,
+ u64 offset, u64 len,
+ struct phys_vec *phys, u32 *attr)
{
- struct gmem_dmabuf *priv;
+ struct gmem_dmabuf *priv = attach->dmabuf->priv;
dma_resv_assert_held(attach->dmabuf->resv);
- if (attach->dmabuf->ops != &gmem_dma_buf_ops)
- return -EOPNOTSUPP;
- priv = attach->dmabuf->priv;
if (priv->revoked)
return -ENODEV;
- *phys = priv->phys;
+
+ phys->paddr = priv->phys.paddr + offset;
+ phys->len = len;
+ *attr = DMA_BUF_PHYS_ATTR_RAM;
return 0;
}
-EXPORT_SYMBOL_FOR_MODULES(gmem_provider_dma_buf_iommufd_map, "iommufd");
+
+static const struct dma_buf_ops gmem_dma_buf_ops = {
+ .attach = gmem_dma_buf_attach,
+ .map_dma_buf = gmem_dma_buf_map,
+ .unmap_dma_buf = gmem_dma_buf_unmap,
+ .release = gmem_dma_buf_release,
+ .get_phys = gmem_dma_buf_get_phys,
+};
/* Called with info->dmabufs_lock held on the revoke path. */
static void gmem_dma_buf_revoke_all(struct gmem_info *info)
--
2.47.3
^ permalink raw reply related [flat|nested] 10+ messages in thread* Re: [RFC PATCH 2/6] dma-buf: Add get_phys() to describe a physical run
2026-10-05 9:55 ` [RFC PATCH 2/6] dma-buf: Add get_phys() to describe a physical run Fred Griffoul
@ 2026-10-05 10:07 ` Christian König
2026-10-05 13:20 ` Fred Griffoul
0 siblings, 1 reply; 10+ messages in thread
From: Christian König @ 2026-10-05 10:07 UTC (permalink / raw)
To: Fred Griffoul, Paolo Bonzini, Sean Christopherson, Marc Zyngier,
Oliver Upton, Sumit Semwal, Jason Gunthorpe, Kevin Tian
Cc: David Woodhouse, Ackerley Tng, Joey Gouly, Suzuki K Poulose,
Zenghui Yu, Steffen Eiden, Catalin Marinas, Will Deacon,
Thomas Gleixner, Ingo Molnar, Borislav Petkov, Dave Hansen,
H . Peter Anvin, Joerg Roedel, Robin Murphy, Alex Williamson,
Shuah Khan, Steven Rostedt, Masami Hiramatsu, Mathieu Desnoyers,
linux-kernel, kvm, kvmarm, linux-arm-kernel, iommu, linux-media,
dri-devel, linaro-mm-sig, linux-kselftest, linux-trace-kernel,
x86
On 10/5/26 11:55, Fred Griffoul wrote:
> From: Fred Griffoul <fgriffo@amazon.co.uk>
>
> iommufd and KVM write physical addresses into their own page tables.
> To do so, they must ask the exporter which frames back an offset of a
> dma-buf, whether that memory is RAM or MMIO, and whether it may be
> written.
Well filling page tables by the importer is an absolutely clear NO-GO for the DMA-buf design, we have gotten down that path already and it took us years to remove this functionality again.
The problem you are facing here is that dma_buf_mmap() doesn't work because you don't have a VMA.
I can understand the reasoning that you don't want to have a VMA, but I don't think that this is a valid justification to add complexity to DMA-buf and especially bring an approach back which we have already deprecated.
A possible solution might be Jasons patch set to directly negotiate exposing PCI BAR regions as DMA-buf, but that certainly needs more discussion.
But the approach outlined here is an absolutely clear NAK from my side.
Regards,
Christian.
>
> Add a get_phys() operation. The importer passes an offset and a maximum
> length. The exporter reports one run: the frames that start at the
> offset, are backed and physically contiguous, and share one attribute
> word. The run never exceeds the length. The exporter may end it early,
> so importers must not assume that it is the longest possible run.
>
> get_phys() returns -ENOENT when the byte at the offset is not backed,
> and -ENODEV when the buffer is revoked. The caller holds the
> reservation, and either pins the attachment or handles revocation. A
> reported frame stays valid until an invalidation that covers it
> returns.
>
> The attribute word holds the memory type and a READONLY flag. Zero
> means writable RAM. Importers refuse unknown types, reserved bits and
> unknown flags, so attributes added later fail safely. Two flag bits are
> reserved: one for holes and one for confidential memory.
>
> Convert vfio-pci, the iommufd selftest exporter and the KVM sample.
> iommufd behaves as before: it maps a buffer only when one writable run
> covers all of it.
>
> Signed-off-by: Fred Griffoul <fgriffo@amazon.co.uk>
> ---
> drivers/dma-buf/dma-buf.c | 47 +++++++++++
> drivers/iommu/iommufd/iommufd_private.h | 8 --
> drivers/iommu/iommufd/iommufd_test.h | 16 ++++
> drivers/iommu/iommufd/pages.c | 74 ++++-------------
> drivers/iommu/iommufd/selftest.c | 106 ++++++++++++++++++++----
> drivers/vfio/pci/vfio_pci_dmabuf.c | 58 ++++++-------
> include/linux/dma-buf.h | 54 ++++++++++++
> include/linux/vfio_pci_core.h | 3 -
> samples/kvm/gmem_provider.c | 39 ++++-----
> 9 files changed, 267 insertions(+), 138 deletions(-)
>
> diff --git a/drivers/dma-buf/dma-buf.c b/drivers/dma-buf/dma-buf.c
> index d504c636dc29..66b85d53ed22 100644
> --- a/drivers/dma-buf/dma-buf.c
> +++ b/drivers/dma-buf/dma-buf.c
> @@ -1389,6 +1389,53 @@ void dma_buf_invalidate_mappings(struct dma_buf *dmabuf)
> }
> EXPORT_SYMBOL_NS_GPL(dma_buf_invalidate_mappings, "DMA_BUF");
>
> +/**
> + * dma_buf_get_phys - describe the run that starts at an offset
> + * @attach: attachment to query
> + * @offset: first buffer byte to describe
> + * @len: maximum number of bytes to describe
> + * @phys: physical address and length of the run
> + * @attr: DMA_BUF_PHYS_ATTR_* word of the run
> + *
> + * A run is the longest stretch of backed bytes starting at @offset whose
> + * frames are physically contiguous and share one attribute word. On success
> + * *@phys starts at the byte at @offset and covers at most @len bytes; it may
> + * be shorter than the run.
> + *
> + * The dma-buf reservation must be held. The attachment must be pinned or have
> + * revocable importer operations. A frame remains valid until a covering
> + * invalidation callback returns; the exporter must invalidate a changed range
> + * before reusing its old frames.
> + *
> + * Returns:
> + *
> + * 0 on success, -ENOENT if the byte at @offset is not backed, -ENODEV if the
> + * buffer is revoked, -EOPNOTSUPP if the exporter cannot describe itself this
> + * way, or another negative error code.
> + */
> +int dma_buf_get_phys(struct dma_buf_attachment *attach, u64 offset, u64 len,
> + struct phys_vec *phys, u32 *attr)
> +{
> + u64 end;
> + int ret;
> +
> + if (WARN_ON_ONCE(!attach || !attach->dmabuf || !phys || !attr))
> + return -EINVAL;
> + if (!len || check_add_overflow(offset, len, &end) ||
> + end > attach->dmabuf->size)
> + return -EINVAL;
> +
> + dma_resv_assert_held(attach->dmabuf->resv);
> + if (!attach->dmabuf->ops->get_phys)
> + return -EOPNOTSUPP;
> +
> + ret = attach->dmabuf->ops->get_phys(attach, offset, len, phys, attr);
> + if (!ret && WARN_ON_ONCE(!phys->len || phys->len > len))
> + return -EIO;
> + return ret;
> +}
> +EXPORT_SYMBOL_NS_GPL(dma_buf_get_phys, "DMA_BUF");
> +
> /**
> * DOC: cpu access
> *
> diff --git a/drivers/iommu/iommufd/iommufd_private.h b/drivers/iommu/iommufd/iommufd_private.h
> index 43fbc5bed8de..5cded585c227 100644
> --- a/drivers/iommu/iommufd/iommufd_private.h
> +++ b/drivers/iommu/iommufd/iommufd_private.h
> @@ -716,8 +716,6 @@ bool iommufd_should_fail(void);
> int __init iommufd_test_init(void);
> void iommufd_test_exit(void);
> bool iommufd_selftest_is_mock_dev(struct device *dev);
> -int iommufd_test_dma_buf_iommufd_map(struct dma_buf_attachment *attachment,
> - struct phys_vec *phys);
> #else
> static inline void iommufd_test_syz_conv_iova_id(struct iommufd_ucmd *ucmd,
> unsigned int ioas_id,
> @@ -739,11 +737,5 @@ static inline bool iommufd_selftest_is_mock_dev(struct device *dev)
> {
> return false;
> }
> -static inline int
> -iommufd_test_dma_buf_iommufd_map(struct dma_buf_attachment *attachment,
> - struct phys_vec *phys)
> -{
> - return -EOPNOTSUPP;
> -}
> #endif
> #endif
> diff --git a/drivers/iommu/iommufd/iommufd_test.h b/drivers/iommu/iommufd/iommufd_test.h
> index 52b78cbcc920..28fd9c43edc4 100644
> --- a/drivers/iommu/iommufd/iommufd_test.h
> +++ b/drivers/iommu/iommufd/iommufd_test.h
> @@ -31,6 +31,8 @@ enum {
> IOMMU_TEST_OP_PASID_CHECK_HWPT,
> IOMMU_TEST_OP_DMABUF_GET,
> IOMMU_TEST_OP_DMABUF_REVOKE,
> + IOMMU_TEST_OP_MD_CHECK_MAPPED,
> + IOMMU_TEST_OP_MD_IOVA_TO_PHYS,
> };
>
> enum {
> @@ -193,6 +195,20 @@ struct iommu_test_cmd {
> __s32 dmabuf_fd;
> __u32 revoked;
> } dmabuf_revoke;
> + struct {
> + /*
> + * 1: every page in [iova, iova+length) must be mapped;
> + * 0: none of them may be. Mixed is an error.
> + */
> + __u32 mapped;
> + __u32 __reserved;
> + __aligned_u64 iova;
> + __aligned_u64 length;
> + } check_mapped;
> + struct {
> + __aligned_u64 iova;
> + __aligned_u64 out_phys; /* 0 if unmapped */
> + } iova_to_phys;
> };
> __u32 last;
> };
> diff --git a/drivers/iommu/iommufd/pages.c b/drivers/iommu/iommufd/pages.c
> index f9b2ae6d7e96..196d1bb330c2 100644
> --- a/drivers/iommu/iommufd/pages.c
> +++ b/drivers/iommu/iommufd/pages.c
> @@ -1463,68 +1463,12 @@ static const struct dma_buf_attach_ops iopt_dmabuf_attach_revoke_ops = {
> .invalidate_mappings = iopt_revoke_notify,
> };
>
> -/*
> - * iommufd and vfio have a circular dependency. Future work for a phys
> - * based private interconnect will remove this.
> - */
> -/*
> - * Look up the exporter's phys accessor for iommufd's private-interconnect
> - * path. Also fills *is_cpu_ram: true if the exporter's memory is normal
> - * cache-coherent RAM (needs BATCH_CPU_MEMORY / IOMMU_CACHE), false for MMIO
> - * (needs BATCH_MMIO / IOMMU_MMIO). This will be replaced by a formal
> - * exporter op that returns phys + memory type together.
> - */
> -static int
> -sym_vfio_pci_dma_buf_iommufd_map(struct dma_buf_attachment *attachment,
> - struct phys_vec *phys, bool *is_cpu_ram)
> -{
> - typeof(&vfio_pci_dma_buf_iommufd_map) fn;
> - int rc;
> -
> - rc = iommufd_test_dma_buf_iommufd_map(attachment, phys);
> - if (rc != -EOPNOTSUPP) {
> - *is_cpu_ram = false; /* test hook mimics VFIO MMIO */
> - return rc;
> - }
> -
> - /*
> - * Prototype: try the sample gmem provider's dma-buf exporter. This
> - * mirrors the vfio-pci private-interconnect hook, and (like it) is
> - * meant to be replaced by a formal negotiated exporter op returning
> - * phys + memory type. The provider serves RAM, so mark it CPU_RAM.
> - */
> - {
> - extern int gmem_provider_dma_buf_iommufd_map(
> - struct dma_buf_attachment *, struct phys_vec *);
> - typeof(&gmem_provider_dma_buf_iommufd_map) gfn;
> -
> - gfn = symbol_get(gmem_provider_dma_buf_iommufd_map);
> - if (gfn) {
> - rc = gfn(attachment, phys);
> - symbol_put(gmem_provider_dma_buf_iommufd_map);
> - if (rc != -EOPNOTSUPP) {
> - *is_cpu_ram = true;
> - return rc;
> - }
> - }
> - }
> -
> - if (!IS_ENABLED(CONFIG_VFIO_PCI_DMABUF))
> - return -EOPNOTSUPP;
> -
> - fn = symbol_get(vfio_pci_dma_buf_iommufd_map);
> - if (!fn)
> - return -EOPNOTSUPP;
> - rc = fn(attachment, phys);
> - symbol_put(vfio_pci_dma_buf_iommufd_map);
> - *is_cpu_ram = false; /* VFIO PCI dma-buf carries BAR (MMIO) memory */
> - return rc;
> -}
> -
> static int iopt_map_dmabuf(struct iommufd_ctx *ictx, struct iopt_pages *pages,
> struct dma_buf *dmabuf)
> {
> struct dma_buf_attachment *attach;
> + struct phys_vec pv;
> + u32 attr;
> int rc;
>
> attach = dma_buf_dynamic_attach(dmabuf, iommufd_global_device(),
> @@ -1546,10 +1490,20 @@ static int iopt_map_dmabuf(struct iommufd_ctx *ictx, struct iopt_pages *pages,
> if (rc)
> goto err_detach;
>
> - rc = sym_vfio_pci_dma_buf_iommufd_map(attach, &pages->dmabuf.phys,
> - &pages->dmabuf.is_cpu_ram);
> + /* One backed, writable run covering the buffer: refuse the rest. */
> + rc = dma_buf_get_phys(attach, 0, dmabuf->size, &pv, &attr);
> + if (rc == -ENOENT)
> + rc = -EOPNOTSUPP;
> if (rc)
> goto err_unpin;
> + if (pv.len != dmabuf->size || !dma_buf_phys_attr_known(attr) ||
> + (attr & DMA_BUF_PHYS_ATTR_FLAGS_MASK)) {
> + rc = -EOPNOTSUPP;
> + goto err_unpin;
> + }
> + pages->dmabuf.phys = pv;
> + pages->dmabuf.is_cpu_ram =
> + dma_buf_phys_attr_type(attr) == DMA_BUF_PHYS_ATTR_RAM;
>
> dma_resv_unlock(dmabuf->resv);
>
> diff --git a/drivers/iommu/iommufd/selftest.c b/drivers/iommu/iommufd/selftest.c
> index af07c642a526..0899272d1e66 100644
> --- a/drivers/iommu/iommufd/selftest.c
> +++ b/drivers/iommu/iommufd/selftest.c
> @@ -1962,32 +1962,31 @@ static void iommufd_test_dma_buf_release(struct dma_buf *dmabuf)
> kfree(priv);
> }
>
> -static const struct dma_buf_ops iommufd_test_dmabuf_ops = {
> - .attach = iommufd_test_dma_buf_attach,
> - .detach = iommufd_test_dma_buf_detach,
> - .map_dma_buf = iommufd_test_dma_buf_map,
> - .release = iommufd_test_dma_buf_release,
> - .unmap_dma_buf = iommufd_test_dma_buf_unmap,
> -};
> -
> -int iommufd_test_dma_buf_iommufd_map(struct dma_buf_attachment *attachment,
> - struct phys_vec *phys)
> +static int iommufd_test_dma_buf_get_phys(struct dma_buf_attachment *attachment,
> + u64 offset, u64 len,
> + struct phys_vec *phys, u32 *attr)
> {
> struct iommufd_test_dma_buf *priv = attachment->dmabuf->priv;
>
> dma_resv_assert_held(attachment->dmabuf->resv);
> -
> - if (attachment->dmabuf->ops != &iommufd_test_dmabuf_ops)
> - return -EOPNOTSUPP;
> -
> if (priv->revoked)
> return -ENODEV;
>
> - phys->paddr = virt_to_phys(priv->memory);
> - phys->len = priv->length;
> + phys->paddr = virt_to_phys(priv->memory) + offset;
> + phys->len = len;
> + *attr = DMA_BUF_PHYS_ATTR_MMIO;
> return 0;
> }
>
> +static const struct dma_buf_ops iommufd_test_dmabuf_ops = {
> + .attach = iommufd_test_dma_buf_attach,
> + .detach = iommufd_test_dma_buf_detach,
> + .map_dma_buf = iommufd_test_dma_buf_map,
> + .release = iommufd_test_dma_buf_release,
> + .unmap_dma_buf = iommufd_test_dma_buf_unmap,
> + .get_phys = iommufd_test_dma_buf_get_phys,
> +};
> +
> static int iommufd_test_dmabuf_get(struct iommufd_ucmd *ucmd,
> unsigned int open_flags,
> size_t len)
> @@ -2031,6 +2030,73 @@ static int iommufd_test_dmabuf_get(struct iommufd_ucmd *ucmd,
> return rc;
> }
>
> +/*
> + * Report the physical address the mock domain resolves @iova to, or 0 if
> + * it is unmapped. Lets a test check that two IOVAs share one frame (a
> + * scratch substitution) without knowing the frame in advance.
> + */
> +static int iommufd_test_md_iova_to_phys(struct iommufd_ucmd *ucmd,
> + unsigned int mockpt_id,
> + unsigned long iova)
> +{
> + struct iommu_test_cmd *cmd = ucmd->cmd;
> + struct iommufd_hw_pagetable *hwpt;
> + struct mock_iommu_domain *mock;
> + unsigned int page_size;
> + int rc;
> +
> + hwpt = get_md_pagetable(ucmd, mockpt_id, &mock);
> + if (IS_ERR(hwpt))
> + return PTR_ERR(hwpt);
> +
> + page_size = 1 << __ffs(mock->domain.pgsize_bitmap);
> + if (iova % page_size) {
> + rc = -EINVAL;
> + goto out_put;
> + }
> + cmd->iova_to_phys.out_phys =
> + mock->domain.ops->iova_to_phys(&mock->domain, iova);
> + rc = iommufd_ucmd_respond(ucmd, sizeof(*cmd));
> +out_put:
> + iommufd_put_object(ucmd->ictx, &hwpt->obj);
> + return rc;
> +}
> +
> +static int iommufd_test_md_check_mapped(struct iommufd_ucmd *ucmd,
> + unsigned int mockpt_id,
> + unsigned long iova, size_t length,
> + bool mapped)
> +{
> + struct iommufd_hw_pagetable *hwpt;
> + struct mock_iommu_domain *mock;
> + unsigned int page_size;
> + int rc = 0;
> +
> + hwpt = get_md_pagetable(ucmd, mockpt_id, &mock);
> + if (IS_ERR(hwpt))
> + return PTR_ERR(hwpt);
> +
> + page_size = 1 << __ffs(mock->domain.pgsize_bitmap);
> + if (iova % page_size || length % page_size || !length) {
> + rc = -EINVAL;
> + goto out_put;
> + }
> +
> + for (; length; length -= page_size, iova += page_size) {
> + bool is_mapped =
> + mock->domain.ops->iova_to_phys(&mock->domain, iova) != 0;
> +
> + if (is_mapped != mapped) {
> + rc = -ENOENT;
> + goto out_put;
> + }
> + }
> +
> +out_put:
> + iommufd_put_object(ucmd->ictx, &hwpt->obj);
> + return rc;
> +}
> +
> static int iommufd_test_dmabuf_revoke(struct iommufd_ucmd *ucmd, int fd,
> bool revoked)
> {
> @@ -2143,6 +2209,14 @@ int iommufd_test(struct iommufd_ucmd *ucmd)
> return iommufd_test_dmabuf_revoke(ucmd,
> cmd->dmabuf_revoke.dmabuf_fd,
> cmd->dmabuf_revoke.revoked);
> + case IOMMU_TEST_OP_MD_CHECK_MAPPED:
> + return iommufd_test_md_check_mapped(ucmd, cmd->id,
> + cmd->check_mapped.iova,
> + cmd->check_mapped.length,
> + cmd->check_mapped.mapped);
> + case IOMMU_TEST_OP_MD_IOVA_TO_PHYS:
> + return iommufd_test_md_iova_to_phys(ucmd, cmd->id,
> + cmd->iova_to_phys.iova);
> default:
> return -EOPNOTSUPP;
> }
> diff --git a/drivers/vfio/pci/vfio_pci_dmabuf.c b/drivers/vfio/pci/vfio_pci_dmabuf.c
> index c16f460c01d6..381c3d338e8c 100644
> --- a/drivers/vfio/pci/vfio_pci_dmabuf.c
> +++ b/drivers/vfio/pci/vfio_pci_dmabuf.c
> @@ -99,46 +99,46 @@ static void vfio_pci_dma_buf_release(struct dma_buf *dmabuf)
> kfree(priv);
> }
>
> -static const struct dma_buf_ops vfio_pci_dmabuf_ops = {
> - .attach = vfio_pci_dma_buf_attach,
> - .map_dma_buf = vfio_pci_dma_buf_map,
> - .unmap_dma_buf = vfio_pci_dma_buf_unmap,
> - .release = vfio_pci_dma_buf_release,
> -};
> -
> /*
> - * This is a temporary "private interconnect" between VFIO DMABUF and iommufd.
> - * It allows the two co-operating drivers to exchange the physical address of
> - * the BAR. This is to be replaced with a formal DMABUF system for negotiated
> - * interconnect types.
> + * Report the BAR's physical range for importers which program it into their own
> + * translation tables, such as iommufd. A BAR is MMIO, never cache-coherent RAM.
> *
> - * If this function succeeds the following are true:
> - * - There is one physical range and it is pointing to MMIO
> - * - When move_notify is called it means revoke, not move, vfio_dma_buf_map
> - * will fail if it is currently revoked
> + * When move_notify is called it means revoke, not move, so this fails while
> + * revoked and vfio_dma_buf_map() does the same.
> */
> -int vfio_pci_dma_buf_iommufd_map(struct dma_buf_attachment *attachment,
> - struct phys_vec *phys)
> +static int vfio_pci_dma_buf_get_phys(struct dma_buf_attachment *attachment,
> + u64 offset, u64 len,
> + struct phys_vec *phys, u32 *attr)
> {
> - struct vfio_pci_dma_buf *priv;
> + struct vfio_pci_dma_buf *priv = attachment->dmabuf->priv;
> + u32 i;
>
> dma_resv_assert_held(attachment->dmabuf->resv);
> -
> - if (attachment->dmabuf->ops != &vfio_pci_dmabuf_ops)
> - return -EOPNOTSUPP;
> -
> - priv = attachment->dmabuf->priv;
> if (priv->revoked)
> return -ENODEV;
>
> - /* More than one range to iommufd will require proper DMABUF support */
> - if (priv->nr_ranges != 1)
> - return -EOPNOTSUPP;
> -
> - *phys = priv->phys_vec[0];
> + /* Report from @offset to the end of the BAR range containing it. */
> + for (i = 0; i < priv->nr_ranges; i++) {
> + if (offset < priv->phys_vec[i].len)
> + break;
> + offset -= priv->phys_vec[i].len;
> + }
> + if (i == priv->nr_ranges)
> + return -EINVAL;
> + phys->paddr = priv->phys_vec[i].paddr + offset;
> + phys->len = min_t(u64, priv->phys_vec[i].len - offset, len);
> + *attr = DMA_BUF_PHYS_ATTR_MMIO;
> return 0;
> }
> -EXPORT_SYMBOL_FOR_MODULES(vfio_pci_dma_buf_iommufd_map, "iommufd");
> +
> +static const struct dma_buf_ops vfio_pci_dmabuf_ops = {
> + .attach = vfio_pci_dma_buf_attach,
> + .map_dma_buf = vfio_pci_dma_buf_map,
> + .unmap_dma_buf = vfio_pci_dma_buf_unmap,
> + .release = vfio_pci_dma_buf_release,
> + .get_phys = vfio_pci_dma_buf_get_phys,
> +};
> +
>
> int vfio_pci_core_fill_phys_vec(struct phys_vec *phys_vec,
> struct vfio_region_dma_range *dma_ranges,
> diff --git a/include/linux/dma-buf.h b/include/linux/dma-buf.h
> index d1203da56fc5..b223962e20c2 100644
> --- a/include/linux/dma-buf.h
> +++ b/include/linux/dma-buf.h
> @@ -13,6 +13,7 @@
> #ifndef __DMA_BUF_H__
> #define __DMA_BUF_H__
>
> +#include <linux/bitfield.h>
> #include <linux/iosys-map.h>
> #include <linux/file.h>
> #include <linux/err.h>
> @@ -23,6 +24,7 @@
> #include <linux/dma-fence.h>
> #include <linux/wait.h>
> #include <linux/pci-p2pdma.h>
> +#include <linux/types.h>
>
> struct device;
> struct dma_buf;
> @@ -182,6 +184,29 @@ struct dma_buf_ops {
> struct sg_table *,
> enum dma_data_direction);
>
> + /**
> + * @get_phys:
> + *
> + * Describe the run that starts at @offset, for an importer that
> + * programs its own translation tables. A run is the longest stretch
> + * of backed bytes whose frames are physically contiguous and share
> + * one attribute word. Report it in *@phys, starting at the byte at
> + * @offset and clipped at @offset + @len, with its DMA_BUF_PHYS_ATTR_*
> + * word in *@attr. The exporter may stop before the end of the run;
> + * importers must not assume the reported run is maximal.
> + *
> + * Return 0 on success, -ENOENT if the byte at @offset is not backed,
> + * -ENODEV if the buffer is revoked, or another negative error. Do not
> + * wait for memory to become available.
> + *
> + * The dma-buf reservation is held. The attachment must be pinned or
> + * have revocable importer operations. A reported frame remains valid
> + * until a covering invalidation callback returns; an exporter must
> + * invalidate every change before reusing an old frame.
> + */
> + int (*get_phys)(struct dma_buf_attachment *attach, u64 offset, u64 len,
> + struct phys_vec *phys, u32 *attr);
> +
> /* TODO: Add try_map_dma_buf version, to return immed with -EBUSY
> * if the call would block.
> */
> @@ -576,6 +601,35 @@ void dma_buf_unmap_attachment(struct dma_buf_attachment *, struct sg_table *,
> enum dma_data_direction);
> void dma_buf_invalidate_mappings(struct dma_buf *dma_buf);
> bool dma_buf_attach_revocable(struct dma_buf_attachment *attach);
> +/* bits 0-7: memory type (a value, not flags) */
> +#define DMA_BUF_PHYS_ATTR_TYPE_MASK GENMASK(7, 0)
> +#define DMA_BUF_PHYS_ATTR_RAM 0x00 /* cache-coherent system RAM */
> +#define DMA_BUF_PHYS_ATTR_MMIO 0x01 /* device MMIO, uncached */
> +/* bits 8-15: reserved for a second value field; must be zero */
> +#define DMA_BUF_PHYS_ATTR_RSVD_MASK GENMASK(15, 8)
> +/* bits 16-31: flags; undefined bits must be zero */
> +#define DMA_BUF_PHYS_ATTR_READONLY BIT(16)
> +/* BIT(17): reserved (hole, for importers that walk across gaps) */
> +/* BIT(18): reserved (private, for confidential computing) */
> +#define DMA_BUF_PHYS_ATTR_FLAGS_MASK (DMA_BUF_PHYS_ATTR_READONLY)
> +
> +static inline u32 dma_buf_phys_attr_type(u32 attrs)
> +{
> + return FIELD_GET(DMA_BUF_PHYS_ATTR_TYPE_MASK, attrs);
> +}
> +
> +static inline bool dma_buf_phys_attr_known(u32 attrs)
> +{
> + return dma_buf_phys_attr_type(attrs) <= DMA_BUF_PHYS_ATTR_MMIO &&
> + !(attrs & DMA_BUF_PHYS_ATTR_RSVD_MASK) &&
> + !(attrs & ~(DMA_BUF_PHYS_ATTR_TYPE_MASK |
> + DMA_BUF_PHYS_ATTR_RSVD_MASK |
> + DMA_BUF_PHYS_ATTR_FLAGS_MASK));
> +}
> +
> +int dma_buf_get_phys(struct dma_buf_attachment *attach, u64 offset, u64 len,
> + struct phys_vec *phys, u32 *attr);
> +
> int dma_buf_begin_cpu_access(struct dma_buf *dma_buf,
> enum dma_data_direction dir);
> int dma_buf_end_cpu_access(struct dma_buf *dma_buf,
> diff --git a/include/linux/vfio_pci_core.h b/include/linux/vfio_pci_core.h
> index 9a1674c152aa..2a1d13abdb0c 100644
> --- a/include/linux/vfio_pci_core.h
> +++ b/include/linux/vfio_pci_core.h
> @@ -257,7 +257,4 @@ vfio_pci_core_get_iomap(struct vfio_pci_core_device *vdev, unsigned int bar)
> return vdev->barmap[bar];
> }
>
> -int vfio_pci_dma_buf_iommufd_map(struct dma_buf_attachment *attachment,
> - struct phys_vec *phys);
> -
> #endif /* VFIO_PCI_CORE_H */
> diff --git a/samples/kvm/gmem_provider.c b/samples/kvm/gmem_provider.c
> index 75197c088762..b6824fe5d228 100644
> --- a/samples/kvm/gmem_provider.c
> +++ b/samples/kvm/gmem_provider.c
> @@ -487,35 +487,30 @@ static void gmem_dma_buf_release(struct dma_buf *dmabuf)
> kfree(priv);
> }
>
> -static const struct dma_buf_ops gmem_dma_buf_ops = {
> - .attach = gmem_dma_buf_attach,
> - .map_dma_buf = gmem_dma_buf_map,
> - .unmap_dma_buf = gmem_dma_buf_unmap,
> - .release = gmem_dma_buf_release,
> -};
> -
> -/*
> - * Private interconnect for iommufd (mirrors vfio_pci_dma_buf_iommufd_map).
> - * Returns the single contiguous phys range for the exported region so iommufd
> - * can program the IOMMU directly, bypassing the DMA API.
> - */
> -int gmem_provider_dma_buf_iommufd_map(struct dma_buf_attachment *attach,
> - struct phys_vec *phys);
> -int gmem_provider_dma_buf_iommufd_map(struct dma_buf_attachment *attach,
> - struct phys_vec *phys)
> +/* Report this flat sample region through the generic dma-buf operation. */
> +static int gmem_dma_buf_get_phys(struct dma_buf_attachment *attach,
> + u64 offset, u64 len,
> + struct phys_vec *phys, u32 *attr)
> {
> - struct gmem_dmabuf *priv;
> + struct gmem_dmabuf *priv = attach->dmabuf->priv;
>
> dma_resv_assert_held(attach->dmabuf->resv);
> - if (attach->dmabuf->ops != &gmem_dma_buf_ops)
> - return -EOPNOTSUPP;
> - priv = attach->dmabuf->priv;
> if (priv->revoked)
> return -ENODEV;
> - *phys = priv->phys;
> +
> + phys->paddr = priv->phys.paddr + offset;
> + phys->len = len;
> + *attr = DMA_BUF_PHYS_ATTR_RAM;
> return 0;
> }
> -EXPORT_SYMBOL_FOR_MODULES(gmem_provider_dma_buf_iommufd_map, "iommufd");
> +
> +static const struct dma_buf_ops gmem_dma_buf_ops = {
> + .attach = gmem_dma_buf_attach,
> + .map_dma_buf = gmem_dma_buf_map,
> + .unmap_dma_buf = gmem_dma_buf_unmap,
> + .release = gmem_dma_buf_release,
> + .get_phys = gmem_dma_buf_get_phys,
> +};
>
> /* Called with info->dmabufs_lock held on the revoke path. */
> static void gmem_dma_buf_revoke_all(struct gmem_info *info)
> --
> 2.47.3
>
^ permalink raw reply [flat|nested] 10+ messages in thread* Re: [RFC PATCH 2/6] dma-buf: Add get_phys() to describe a physical run
2026-10-05 10:07 ` Christian König
@ 2026-10-05 13:20 ` Fred Griffoul
2026-10-05 14:53 ` Christian König
0 siblings, 1 reply; 10+ messages in thread
From: Fred Griffoul @ 2026-10-05 13:20 UTC (permalink / raw)
To: Paolo Bonzini, Sean Christopherson, Marc Zyngier, Oliver Upton,
Sumit Semwal, Christian König, Jason Gunthorpe, Kevin Tian
Cc: David Woodhouse, Ackerley Tng, Joey Gouly, Suzuki K Poulose,
Zenghui Yu, Steffen Eiden, Catalin Marinas, Will Deacon,
Thomas Gleixner, Ingo Molnar, Borislav Petkov, Dave Hansen,
H . Peter Anvin, Joerg Roedel, Robin Murphy, Alex Williamson,
Shuah Khan, Steven Rostedt, Masami Hiramatsu, Mathieu Desnoyers,
linux-kernel, kvm, kvmarm, linux-arm-kernel, iommu, linux-media,
dri-devel, linaro-mm-sig, linux-kselftest, linux-trace-kernel,
x86
On 10/5/26 Christian König wrote:
> Well filling page tables by the importer is an absolutely clear NO-GO
> for the DMA-buf design, we have gotten down that path already and it
> took us years to remove this functionality again.
Understood, thanks for the quick and clear answers on this patch and on
patch 3.
> The problem you are facing here is that dma_buf_mmap() doesn't work
> because you don't have a VMA.
Right: the memory has no struct page and is deliberately not mapped in
the host, so there is no VMA to hand to KVM.
I'll post a new version of this series. It drops get_phys(), the ranged
invalidation and the guest_memfd dma-buf import (patches 2-5), and it
does not touch drivers/dma-buf, as you suggested in the PAL discussion:
https://lore.kernel.org/all/c413710b-4c28-4ed8-88ec-aeb8c4482011@amd.com/
KVM gets its frames through the KVM-private guest_memfd provider
interface from David's series. iommufd gets them through a private
interface with the exporter, as it already does for vfio-pci. The new
version adds a small registry in iommufd for that, which replaces the
symbol_get() of the sample in David's series. The dma-buf is then only
the handle and the existing revoke.
Thanks,
Fred
^ permalink raw reply [flat|nested] 10+ messages in thread
* Re: [RFC PATCH 2/6] dma-buf: Add get_phys() to describe a physical run
2026-10-05 13:20 ` Fred Griffoul
@ 2026-10-05 14:53 ` Christian König
0 siblings, 0 replies; 10+ messages in thread
From: Christian König @ 2026-10-05 14:53 UTC (permalink / raw)
To: Fred Griffoul, Paolo Bonzini, Sean Christopherson, Marc Zyngier,
Oliver Upton, Sumit Semwal, Jason Gunthorpe, Kevin Tian
Cc: David Woodhouse, Ackerley Tng, Joey Gouly, Suzuki K Poulose,
Zenghui Yu, Steffen Eiden, Catalin Marinas, Will Deacon,
Thomas Gleixner, Ingo Molnar, Borislav Petkov, Dave Hansen,
H . Peter Anvin, Joerg Roedel, Robin Murphy, Alex Williamson,
Shuah Khan, Steven Rostedt, Masami Hiramatsu, Mathieu Desnoyers,
linux-kernel, kvm, kvmarm, linux-arm-kernel, iommu, linux-media,
dri-devel, linaro-mm-sig, linux-kselftest, linux-trace-kernel,
x86
On 10/5/26 15:20, Fred Griffoul wrote:
> On 10/5/26 Christian König wrote:
>> Well filling page tables by the importer is an absolutely clear NO-GO
>> for the DMA-buf design, we have gotten down that path already and it
>> took us years to remove this functionality again.
>
> Understood, thanks for the quick and clear answers on this patch and on
> patch 3.
>
>> The problem you are facing here is that dma_buf_mmap() doesn't work
>> because you don't have a VMA.
>
> Right: the memory has no struct page and is deliberately not mapped in
> the host, so there is no VMA to hand to KVM.
>
> I'll post a new version of this series. It drops get_phys(), the ranged
> invalidation and the guest_memfd dma-buf import (patches 2-5), and it
> does not touch drivers/dma-buf, as you suggested in the PAL discussion:
>
> https://lore.kernel.org/all/c413710b-4c28-4ed8-88ec-aeb8c4482011@amd.com/
>
> KVM gets its frames through the KVM-private guest_memfd provider
> interface from David's series. iommufd gets them through a private
> interface with the exporter, as it already does for vfio-pci. The new
> version adds a small registry in iommufd for that, which replaces the
> symbol_get() of the sample in David's series. The dma-buf is then only
> the handle and the existing revoke.
Yeah that approach sounds totally sane to me.
We have drivers which stuff all kind of resources, not just memory, into a DMA-buf. That ranges from MMIO BARs to trigger FW operations (doorbells) over full HW blocks (ordered appends, global wave sync etc...) where two or more applications need to share which one is used.
That you then have a exporter private interface is for accessing this is perfectly ok.
The problem you are facing here is more that this is not limited to one driver/module but multiple ones and you don't want module inter dependencies because of the symbols.
So symbol_get() is probably the right thing to do as long as we don't have something like weak symbols or similar.
Regards,
Christian.
>
> Thanks,
> Fred
^ permalink raw reply [flat|nested] 10+ messages in thread
* [RFC PATCH 3/6] dma-buf: Add ranged mapping invalidation
2026-10-05 9:55 ` [RFC PATCH 0/6] KVM: guest_memfd: back guest_memfd with an imported dma-buf Fred Griffoul
2026-10-05 9:55 ` [RFC PATCH 1/6] KVM: guest_memfd: Add a writable result to get_pfn() Fred Griffoul
2026-10-05 9:55 ` [RFC PATCH 2/6] dma-buf: Add get_phys() to describe a physical run Fred Griffoul
@ 2026-10-05 9:55 ` Fred Griffoul
2026-10-05 10:08 ` Christian König
2026-10-05 9:55 ` [RFC PATCH 4/6] dma-buf: Allow dynamic attach without a device Fred Griffoul
2026-10-05 9:55 ` [RFC PATCH 5/6] KVM: guest_memfd: Add dma-buf backing Fred Griffoul
4 siblings, 1 reply; 10+ messages in thread
From: Fred Griffoul @ 2026-10-05 9:55 UTC (permalink / raw)
To: Paolo Bonzini, Sean Christopherson, Marc Zyngier, Oliver Upton,
Sumit Semwal, Christian König, Jason Gunthorpe, Kevin Tian
Cc: David Woodhouse, Ackerley Tng, Joey Gouly, Suzuki K Poulose,
Zenghui Yu, Steffen Eiden, Catalin Marinas, Will Deacon,
Thomas Gleixner, Ingo Molnar, Borislav Petkov, Dave Hansen,
H . Peter Anvin, Joerg Roedel, Robin Murphy, Alex Williamson,
Shuah Khan, Steven Rostedt, Masami Hiramatsu, Mathieu Desnoyers,
linux-kernel, kvm, kvmarm, linux-arm-kernel, iommu, linux-media,
dri-devel, linaro-mm-sig, linux-kselftest, linux-trace-kernel,
x86
From: Fred Griffoul <fgriffo@amazon.co.uk>
dma_buf_invalidate_mappings() tells every importer that the whole
buffer changed. An exporter that changes one part of its memory cannot
say which bytes changed, so importers throw away mappings that are
still valid.
Add an exporter helper that invalidates a byte range, and an importer
callback that receives it. The callback means that the address, the
attributes or the backing of the range changed. If part of the range is
no longer backed, get_phys() returns -ENOENT for it. Importers must stop
using their old answer before the callback returns. Importers that do
not implement the callback still receive a whole-buffer invalidation.
Signed-off-by: Fred Griffoul <fgriffo@amazon.co.uk>
---
drivers/dma-buf/dma-buf.c | 30 ++++++++++++++++++++++++++++++
include/linux/dma-buf.h | 16 ++++++++++++++++
2 files changed, 46 insertions(+)
diff --git a/drivers/dma-buf/dma-buf.c b/drivers/dma-buf/dma-buf.c
index 66b85d53ed22..e7010163eb2f 100644
--- a/drivers/dma-buf/dma-buf.c
+++ b/drivers/dma-buf/dma-buf.c
@@ -1389,6 +1389,36 @@ void dma_buf_invalidate_mappings(struct dma_buf *dmabuf)
}
EXPORT_SYMBOL_NS_GPL(dma_buf_invalidate_mappings, "DMA_BUF");
+/**
+ * dma_buf_invalidate_mappings_range - notify attachments that a range changed
+ * @dmabuf: buffer whose layout changed
+ * @offset: first changed byte
+ * @length: number of changed bytes
+ *
+ * Importers with a ranged callback stop using their old mappings of the range
+ * before returning. Other importers receive the existing whole-buffer
+ * callback, which is correct but coarser. The reservation lock must be held.
+ */
+void dma_buf_invalidate_mappings_range(struct dma_buf *dmabuf,
+ unsigned long offset,
+ unsigned long length)
+{
+ struct dma_buf_attachment *attach;
+
+ dma_resv_assert_held(dmabuf->resv);
+ list_for_each_entry(attach, &dmabuf->attachments, node) {
+ const struct dma_buf_attach_ops *ops = attach->importer_ops;
+
+ if (!ops)
+ continue;
+ if (ops->invalidate_mappings_range)
+ ops->invalidate_mappings_range(attach, offset, length);
+ else if (ops->invalidate_mappings)
+ ops->invalidate_mappings(attach);
+ }
+}
+EXPORT_SYMBOL_NS_GPL(dma_buf_invalidate_mappings_range, "DMA_BUF");
+
/**
* dma_buf_get_phys - describe the run that starts at an offset
* @attach: attachment to query
diff --git a/include/linux/dma-buf.h b/include/linux/dma-buf.h
index b223962e20c2..55c3fe60a0ba 100644
--- a/include/linux/dma-buf.h
+++ b/include/linux/dma-buf.h
@@ -485,6 +485,19 @@ struct dma_buf_attach_ops {
* required behavior.
*/
void (*invalidate_mappings)(struct dma_buf_attachment *attach);
+
+ /**
+ * @invalidate_mappings_range: [optional] a byte range changed
+ *
+ * The exporter changed the address, attributes or backing of
+ * [@offset, @offset + @length). The importer must stop using its old
+ * answer for that range before returning.
+ * Importers without this callback receive @invalidate_mappings for
+ * the whole buffer instead.
+ */
+ void (*invalidate_mappings_range)(struct dma_buf_attachment *attach,
+ unsigned long offset,
+ unsigned long length);
};
/**
@@ -600,6 +613,9 @@ struct sg_table *dma_buf_map_attachment(struct dma_buf_attachment *,
void dma_buf_unmap_attachment(struct dma_buf_attachment *, struct sg_table *,
enum dma_data_direction);
void dma_buf_invalidate_mappings(struct dma_buf *dma_buf);
+void dma_buf_invalidate_mappings_range(struct dma_buf *dma_buf,
+ unsigned long offset,
+ unsigned long length);
bool dma_buf_attach_revocable(struct dma_buf_attachment *attach);
/* bits 0-7: memory type (a value, not flags) */
#define DMA_BUF_PHYS_ATTR_TYPE_MASK GENMASK(7, 0)
--
2.47.3
^ permalink raw reply related [flat|nested] 10+ messages in thread* Re: [RFC PATCH 3/6] dma-buf: Add ranged mapping invalidation
2026-10-05 9:55 ` [RFC PATCH 3/6] dma-buf: Add ranged mapping invalidation Fred Griffoul
@ 2026-10-05 10:08 ` Christian König
0 siblings, 0 replies; 10+ messages in thread
From: Christian König @ 2026-10-05 10:08 UTC (permalink / raw)
To: Fred Griffoul, Paolo Bonzini, Sean Christopherson, Marc Zyngier,
Oliver Upton, Sumit Semwal, Jason Gunthorpe, Kevin Tian
Cc: David Woodhouse, Ackerley Tng, Joey Gouly, Suzuki K Poulose,
Zenghui Yu, Steffen Eiden, Catalin Marinas, Will Deacon,
Thomas Gleixner, Ingo Molnar, Borislav Petkov, Dave Hansen,
H . Peter Anvin, Joerg Roedel, Robin Murphy, Alex Williamson,
Shuah Khan, Steven Rostedt, Masami Hiramatsu, Mathieu Desnoyers,
linux-kernel, kvm, kvmarm, linux-arm-kernel, iommu, linux-media,
dri-devel, linaro-mm-sig, linux-kselftest, linux-trace-kernel,
x86
On 10/5/26 11:55, Fred Griffoul wrote:
> From: Fred Griffoul <fgriffo@amazon.co.uk>
>
> dma_buf_invalidate_mappings() tells every importer that the whole
> buffer changed. An exporter that changes one part of its memory cannot
> say which bytes changed, so importers throw away mappings that are
> still valid.
>
> Add an exporter helper that invalidates a byte range, and an importer
> callback that receives it. The callback means that the address, the
> attributes or the backing of the range changed. If part of the range is
> no longer backed, get_phys() returns -ENOENT for it. Importers must stop
> using their old answer before the callback returns. Importers that do
> not implement the callback still receive a whole-buffer invalidation.
Yeah that is exactly one of the reasons why we don't allow that.
Clear NAK to the whole approach. See the reply to patch #2 for a detailed description.
Regards,
Christian.
>
> Signed-off-by: Fred Griffoul <fgriffo@amazon.co.uk>
> ---
> drivers/dma-buf/dma-buf.c | 30 ++++++++++++++++++++++++++++++
> include/linux/dma-buf.h | 16 ++++++++++++++++
> 2 files changed, 46 insertions(+)
>
> diff --git a/drivers/dma-buf/dma-buf.c b/drivers/dma-buf/dma-buf.c
> index 66b85d53ed22..e7010163eb2f 100644
> --- a/drivers/dma-buf/dma-buf.c
> +++ b/drivers/dma-buf/dma-buf.c
> @@ -1389,6 +1389,36 @@ void dma_buf_invalidate_mappings(struct dma_buf *dmabuf)
> }
> EXPORT_SYMBOL_NS_GPL(dma_buf_invalidate_mappings, "DMA_BUF");
>
> +/**
> + * dma_buf_invalidate_mappings_range - notify attachments that a range changed
> + * @dmabuf: buffer whose layout changed
> + * @offset: first changed byte
> + * @length: number of changed bytes
> + *
> + * Importers with a ranged callback stop using their old mappings of the range
> + * before returning. Other importers receive the existing whole-buffer
> + * callback, which is correct but coarser. The reservation lock must be held.
> + */
> +void dma_buf_invalidate_mappings_range(struct dma_buf *dmabuf,
> + unsigned long offset,
> + unsigned long length)
> +{
> + struct dma_buf_attachment *attach;
> +
> + dma_resv_assert_held(dmabuf->resv);
> + list_for_each_entry(attach, &dmabuf->attachments, node) {
> + const struct dma_buf_attach_ops *ops = attach->importer_ops;
> +
> + if (!ops)
> + continue;
> + if (ops->invalidate_mappings_range)
> + ops->invalidate_mappings_range(attach, offset, length);
> + else if (ops->invalidate_mappings)
> + ops->invalidate_mappings(attach);
> + }
> +}
> +EXPORT_SYMBOL_NS_GPL(dma_buf_invalidate_mappings_range, "DMA_BUF");
> +
> /**
> * dma_buf_get_phys - describe the run that starts at an offset
> * @attach: attachment to query
> diff --git a/include/linux/dma-buf.h b/include/linux/dma-buf.h
> index b223962e20c2..55c3fe60a0ba 100644
> --- a/include/linux/dma-buf.h
> +++ b/include/linux/dma-buf.h
> @@ -485,6 +485,19 @@ struct dma_buf_attach_ops {
> * required behavior.
> */
> void (*invalidate_mappings)(struct dma_buf_attachment *attach);
> +
> + /**
> + * @invalidate_mappings_range: [optional] a byte range changed
> + *
> + * The exporter changed the address, attributes or backing of
> + * [@offset, @offset + @length). The importer must stop using its old
> + * answer for that range before returning.
> + * Importers without this callback receive @invalidate_mappings for
> + * the whole buffer instead.
> + */
> + void (*invalidate_mappings_range)(struct dma_buf_attachment *attach,
> + unsigned long offset,
> + unsigned long length);
> };
>
> /**
> @@ -600,6 +613,9 @@ struct sg_table *dma_buf_map_attachment(struct dma_buf_attachment *,
> void dma_buf_unmap_attachment(struct dma_buf_attachment *, struct sg_table *,
> enum dma_data_direction);
> void dma_buf_invalidate_mappings(struct dma_buf *dma_buf);
> +void dma_buf_invalidate_mappings_range(struct dma_buf *dma_buf,
> + unsigned long offset,
> + unsigned long length);
> bool dma_buf_attach_revocable(struct dma_buf_attachment *attach);
> /* bits 0-7: memory type (a value, not flags) */
> #define DMA_BUF_PHYS_ATTR_TYPE_MASK GENMASK(7, 0)
> --
> 2.47.3
>
^ permalink raw reply [flat|nested] 10+ messages in thread
* [RFC PATCH 4/6] dma-buf: Allow dynamic attach without a device
2026-10-05 9:55 ` [RFC PATCH 0/6] KVM: guest_memfd: back guest_memfd with an imported dma-buf Fred Griffoul
` (2 preceding siblings ...)
2026-10-05 9:55 ` [RFC PATCH 3/6] dma-buf: Add ranged mapping invalidation Fred Griffoul
@ 2026-10-05 9:55 ` Fred Griffoul
2026-10-05 9:55 ` [RFC PATCH 5/6] KVM: guest_memfd: Add dma-buf backing Fred Griffoul
4 siblings, 0 replies; 10+ messages in thread
From: Fred Griffoul @ 2026-10-05 9:55 UTC (permalink / raw)
To: Paolo Bonzini, Sean Christopherson, Marc Zyngier, Oliver Upton,
Sumit Semwal, Christian König, Jason Gunthorpe, Kevin Tian
Cc: David Woodhouse, Ackerley Tng, Joey Gouly, Suzuki K Poulose,
Zenghui Yu, Steffen Eiden, Catalin Marinas, Will Deacon,
Thomas Gleixner, Ingo Molnar, Borislav Petkov, Dave Hansen,
H . Peter Anvin, Joerg Roedel, Robin Murphy, Alex Williamson,
Shuah Khan, Steven Rostedt, Masami Hiramatsu, Mathieu Desnoyers,
linux-kernel, kvm, kvmarm, linux-arm-kernel, iommu, linux-media,
dri-devel, linaro-mm-sig, linux-kselftest, linux-trace-kernel,
x86
From: Fred Griffoul <fgriffo@amazon.co.uk>
dma_buf_dynamic_attach() requires a struct device: the device for which
the buffer is mapped for DMA. Exporters may also check it before they
accept the attachment. An importer that does not perform DMA has no
such device. iommufd already attaches with a global placeholder device
for this reason. KVM would need one too, although it only needs to know
where the memory is so that it can map it into a guest.
Allow @dev to be NULL for a dynamic importer. Such an importer learns
where the memory is from get_phys() and builds its own mappings. It
must not map the attachment for DMA, and dma_buf_map_attachment()
refuses an attachment without a device. An exporter that needs a device
to answer can refuse the attachment in its attach op, as it can for any
other reason.
Signed-off-by: Fred Griffoul <fgriffo@amazon.co.uk>
---
drivers/dma-buf/dma-buf.c | 15 +++++++++++++--
include/trace/events/dma_buf.h | 2 +-
2 files changed, 14 insertions(+), 3 deletions(-)
diff --git a/drivers/dma-buf/dma-buf.c b/drivers/dma-buf/dma-buf.c
index e7010163eb2f..3d43b529a523 100644
--- a/drivers/dma-buf/dma-buf.c
+++ b/drivers/dma-buf/dma-buf.c
@@ -1007,6 +1007,13 @@ dma_buf_pin_on_map(struct dma_buf_attachment *attach)
* Note that this can fail if the backing storage of @dmabuf is in a place not
* accessible to @dev, and cannot be moved to a more suitable place. This is
* indicated with the error code -EBUSY.
+ *
+ * @dev may be NULL for a dynamic importer that is not a DMA master: one that
+ * only asks the exporter where the memory is, through &dma_buf_ops.get_phys,
+ * and builds its own mappings from the answer. Such an importer must never
+ * call dma_buf_map_attachment(), and an exporter that needs a device to
+ * answer (a peer-to-peer check, say) refuses the attachment in its
+ * &dma_buf_ops.attach.
*/
struct dma_buf_attachment *
dma_buf_dynamic_attach(struct dma_buf *dmabuf, struct device *dev,
@@ -1016,7 +1023,7 @@ dma_buf_dynamic_attach(struct dma_buf *dmabuf, struct device *dev,
struct dma_buf_attachment *attach;
int ret;
- if (WARN_ON(!dmabuf || !dev))
+ if (WARN_ON(!dmabuf || (!dev && !importer_ops)))
return ERR_PTR(-EINVAL);
attach = kzalloc_obj(*attach);
@@ -1175,6 +1182,9 @@ struct sg_table *dma_buf_map_attachment(struct dma_buf_attachment *attach,
if (WARN_ON(!attach || !attach->dmabuf))
return ERR_PTR(-EINVAL);
+ /* An importer without a device cannot be a DMA master. */
+ if (WARN_ON(!attach->dev))
+ return ERR_PTR(-EINVAL);
dma_resv_assert_held(attach->dmabuf->resv);
@@ -1847,7 +1857,8 @@ static int dma_buf_debug_show(struct seq_file *s, void *unused)
attach_count = 0;
list_for_each_entry(attach_obj, &buf_obj->attachments, node) {
- seq_printf(s, "\t%s\n", dev_name(attach_obj->dev));
+ seq_printf(s, "\t%s\n", attach_obj->dev ?
+ dev_name(attach_obj->dev) : "none");
attach_count++;
}
dma_resv_unlock(buf_obj->resv);
diff --git a/include/trace/events/dma_buf.h b/include/trace/events/dma_buf.h
index 3bb88d05bcc8..7cabca53b798 100644
--- a/include/trace/events/dma_buf.h
+++ b/include/trace/events/dma_buf.h
@@ -40,7 +40,7 @@ DECLARE_EVENT_CLASS(dma_buf_attach_dev,
TP_ARGS(dmabuf, attach, is_dynamic, dev),
TP_STRUCT__entry(
- __string( dev_name, dev_name(dev))
+ __string(dev_name, dev ? dev_name(dev) : "none")
__string( exp_name, dmabuf->exp_name)
__field( size_t, size)
__field( ino_t, ino)
--
2.47.3
^ permalink raw reply related [flat|nested] 10+ messages in thread* [RFC PATCH 5/6] KVM: guest_memfd: Add dma-buf backing
2026-10-05 9:55 ` [RFC PATCH 0/6] KVM: guest_memfd: back guest_memfd with an imported dma-buf Fred Griffoul
` (3 preceding siblings ...)
2026-10-05 9:55 ` [RFC PATCH 4/6] dma-buf: Allow dynamic attach without a device Fred Griffoul
@ 2026-10-05 9:55 ` Fred Griffoul
4 siblings, 0 replies; 10+ messages in thread
From: Fred Griffoul @ 2026-10-05 9:55 UTC (permalink / raw)
To: Paolo Bonzini, Sean Christopherson, Marc Zyngier, Oliver Upton,
Sumit Semwal, Christian König, Jason Gunthorpe, Kevin Tian
Cc: David Woodhouse, Ackerley Tng, Joey Gouly, Suzuki K Poulose,
Zenghui Yu, Steffen Eiden, Catalin Marinas, Will Deacon,
Thomas Gleixner, Ingo Molnar, Borislav Petkov, Dave Hansen,
H . Peter Anvin, Joerg Roedel, Robin Murphy, Alex Williamson,
Shuah Khan, Steven Rostedt, Masami Hiramatsu, Mathieu Desnoyers,
linux-kernel, kvm, kvmarm, linux-arm-kernel, iommu, linux-media,
dri-devel, linaro-mm-sig, linux-kselftest, linux-trace-kernel,
x86
From: Fred Griffoul <fgriffo@amazon.co.uk>
A memory owner that exports a dma-buf should be able to use that one
object for both guest stage-2 mappings and device IOMMU mappings. A
second, KVM-specific backing interface would duplicate the description
of the memory layout and the rules for invalidating it.
Add GUEST_MEMFD_FLAG_USE_DMABUF and a dmabuf_fd field to
KVM_CREATE_GUEST_MEMFD. This backing has its own kvm_gmem_ops.
guest_memfd attaches to the buffer as a revocable importer without a
device. At creation, it asks for the run at offset 0 and refuses MMIO
or unknown attributes, so a buffer that KVM can never map fails early.
Offset 0 may be unbacked.
On a fault, guest_memfd takes the reservation once. It asks get_phys()
about the PUD-sized, PMD-sized and page-sized blocks around the faulting
page, in that order, and maps the first block that one aligned RAM run
covers. If the page is not backed or the run is not supported, the
fault returns -EFAULT. A READONLY run is mapped without stage-2 write
permission.
The invalidation callback only runs KVM's invalidation over the
affected range. It does not allocate memory or call the exporter. A
fault that read the layout during the change sees mmu_invalidate_seq
move and retries.
Callbacks can start as soon as guest_memfd attaches. A dma-buf-backed
file is therefore added to the inode's list of guest_memfd files under
filemap_invalidate_lock.
mmap() uses the exporter's dma-buf mmap. fallocate() and populate are
not supported. Import the DMA_BUF symbol namespace so that a
CONFIG_KVM=m build passes modpost.
Signed-off-by: Fred Griffoul <fgriffo@amazon.co.uk>
---
include/linux/kvm_host.h | 1 +
include/uapi/linux/kvm.h | 6 +-
tools/include/uapi/linux/kvm.h | 6 +-
virt/kvm/Kconfig | 1 +
virt/kvm/guest_memfd.c | 239 +++++++++++++++++++++++++++++++--
5 files changed, 243 insertions(+), 10 deletions(-)
diff --git a/include/linux/kvm_host.h b/include/linux/kvm_host.h
index f84f3ab44acf..0158dc46580b 100644
--- a/include/linux/kvm_host.h
+++ b/include/linux/kvm_host.h
@@ -834,6 +834,7 @@ static inline u64 kvm_gmem_get_supported_flags(struct kvm *kvm)
if (!kvm || kvm_arch_supports_gmem_init_shared(kvm))
flags |= GUEST_MEMFD_FLAG_INIT_SHARED;
+ flags |= GUEST_MEMFD_FLAG_USE_DMABUF;
return flags;
}
diff --git a/include/uapi/linux/kvm.h b/include/uapi/linux/kvm.h
index 419011097fa8..f8c83013773d 100644
--- a/include/uapi/linux/kvm.h
+++ b/include/uapi/linux/kvm.h
@@ -1654,11 +1654,15 @@ struct kvm_memory_attributes {
#define KVM_CREATE_GUEST_MEMFD _IOWR(KVMIO, 0xd4, struct kvm_create_guest_memfd)
#define GUEST_MEMFD_FLAG_MMAP (1ULL << 0)
#define GUEST_MEMFD_FLAG_INIT_SHARED (1ULL << 1)
+#define GUEST_MEMFD_FLAG_USE_DMABUF (1ULL << 2)
struct kvm_create_guest_memfd {
__u64 size;
__u64 flags;
- __u64 reserved[6];
+ /* With USE_DMABUF, memory comes from this dma-buf's leading bytes. */
+ __s32 dmabuf_fd;
+ __u32 pad;
+ __u64 reserved[5];
};
#define KVM_PRE_FAULT_MEMORY _IOWR(KVMIO, 0xd5, struct kvm_pre_fault_memory)
diff --git a/tools/include/uapi/linux/kvm.h b/tools/include/uapi/linux/kvm.h
index d0c0c8605976..ebeb2bc353ed 100644
--- a/tools/include/uapi/linux/kvm.h
+++ b/tools/include/uapi/linux/kvm.h
@@ -1644,11 +1644,15 @@ struct kvm_memory_attributes {
#define KVM_CREATE_GUEST_MEMFD _IOWR(KVMIO, 0xd4, struct kvm_create_guest_memfd)
#define GUEST_MEMFD_FLAG_MMAP (1ULL << 0)
#define GUEST_MEMFD_FLAG_INIT_SHARED (1ULL << 1)
+#define GUEST_MEMFD_FLAG_USE_DMABUF (1ULL << 2)
struct kvm_create_guest_memfd {
__u64 size;
__u64 flags;
- __u64 reserved[6];
+ /* With USE_DMABUF, memory comes from this dma-buf's leading bytes. */
+ __s32 dmabuf_fd;
+ __u32 pad;
+ __u64 reserved[5];
};
#define KVM_PRE_FAULT_MEMORY _IOWR(KVMIO, 0xd5, struct kvm_pre_fault_memory)
diff --git a/virt/kvm/Kconfig b/virt/kvm/Kconfig
index 794976b88c6f..7ac4c8a371eb 100644
--- a/virt/kvm/Kconfig
+++ b/virt/kvm/Kconfig
@@ -105,6 +105,7 @@ config KVM_GENERIC_MEMORY_ATTRIBUTES
config KVM_GUEST_MEMFD
select XARRAY_MULTI
+ select DMA_SHARED_BUFFER
bool
config HAVE_KVM_ARCH_GMEM_PREPARE
diff --git a/virt/kvm/guest_memfd.c b/virt/kvm/guest_memfd.c
index 0f4bf2cc5e8e..f401cc560eb4 100644
--- a/virt/kvm/guest_memfd.c
+++ b/virt/kvm/guest_memfd.c
@@ -1,4 +1,7 @@
// SPDX-License-Identifier: GPL-2.0
+#include <linux/dma-buf.h>
+#include <linux/dma-resv.h>
+#include <linux/module.h>
#include <linux/anon_inodes.h>
#include <linux/backing-dev.h>
#include <linux/falloc.h>
@@ -44,6 +47,8 @@ struct gmem_inode {
struct list_head gmem_file_list;
u64 flags;
+ struct dma_buf *dmabuf;
+ struct dma_buf_attachment *attach;
};
static __always_inline struct gmem_inode *GMEM_I(struct inode *inode)
@@ -477,21 +482,37 @@ static const struct vm_operations_struct kvm_gmem_vm_ops = {
#endif
};
-static int kvm_gmem_native_mmap(struct file *file, struct vm_area_struct *vma)
+static int kvm_gmem_validate_mmap(struct file *file,
+ struct vm_area_struct *vma)
{
if (!kvm_gmem_supports_mmap(file_inode(file)))
return -ENODEV;
-
if ((vma->vm_flags & (VM_SHARED | VM_MAYSHARE)) !=
- (VM_SHARED | VM_MAYSHARE)) {
+ (VM_SHARED | VM_MAYSHARE))
return -EINVAL;
- }
+ return 0;
+}
- vma->vm_ops = &kvm_gmem_vm_ops;
+static int kvm_gmem_native_mmap(struct file *file, struct vm_area_struct *vma)
+{
+ int ret = kvm_gmem_validate_mmap(file, vma);
+ if (ret)
+ return ret;
+ vma->vm_ops = &kvm_gmem_vm_ops;
return 0;
}
+static int kvm_gmem_dmabuf_mmap(struct file *file, struct vm_area_struct *vma)
+{
+ struct dma_buf *dmabuf = GMEM_I(file_inode(file))->dmabuf;
+ int ret = kvm_gmem_validate_mmap(file, vma);
+
+ if (ret)
+ return ret;
+ return dma_buf_mmap(dmabuf, vma, vma->vm_pgoff);
+}
+
/* File-op dispatchers — thin: they all go through the backing's ops table. */
static int kvm_gmem_fops_release(struct inode *inode, struct file *file)
@@ -617,6 +638,172 @@ bool __weak kvm_arch_supports_gmem_init_shared(struct kvm *kvm)
return true;
}
+/* ---- a dma-buf behind the file ----------------------------------------- */
+
+/*
+ * Refuse a buffer whose first run guest_memfd can never map, such as MMIO,
+ * at creation rather than on every fault. A gap at offset 0 is legal. This
+ * is advisory: each fault checks the run it uses.
+ */
+static int gmem_dmabuf_check(struct dma_buf_attachment *attach, u64 size)
+{
+ struct phys_vec pv;
+ u32 attr;
+ int ret;
+
+ dma_resv_lock(attach->dmabuf->resv, NULL);
+ ret = dma_buf_get_phys(attach, 0, size, &pv, &attr);
+ dma_resv_unlock(attach->dmabuf->resv);
+ if (ret == -ENOENT)
+ return 0;
+ if (ret)
+ return ret;
+ if (!dma_buf_phys_attr_known(attr) ||
+ dma_buf_phys_attr_type(attr) != DMA_BUF_PHYS_ATTR_RAM)
+ return -EOPNOTSUPP;
+ return 0;
+}
+
+static void gmem_dmabuf_invalidate_range(struct dma_buf_attachment *attach,
+ unsigned long offset,
+ unsigned long len)
+{
+ struct inode *inode = attach->importer_priv;
+ pgoff_t npages = i_size_read(inode) >> PAGE_SHIFT;
+ pgoff_t start = offset >> PAGE_SHIFT;
+ pgoff_t end = min_t(pgoff_t, DIV_ROUND_UP(offset + len, PAGE_SIZE),
+ npages);
+
+ if (start >= end)
+ return;
+ /* Lock order: dma-buf reservation, file invalidation, then mmu_lock. */
+ filemap_invalidate_lock(inode->i_mapping);
+ kvm_gmem_invalidate_start(inode, start, end);
+ kvm_gmem_invalidate_end(inode, start, end);
+ filemap_invalidate_unlock(inode->i_mapping);
+}
+
+static void gmem_dmabuf_invalidate(struct dma_buf_attachment *attach)
+{
+ struct inode *inode = attach->importer_priv;
+
+ gmem_dmabuf_invalidate_range(attach, 0, i_size_read(inode));
+}
+
+static const struct dma_buf_attach_ops gmem_dmabuf_attach_ops = {
+ .allow_peer2peer = true,
+ .invalidate_mappings = gmem_dmabuf_invalidate,
+ .invalidate_mappings_range = gmem_dmabuf_invalidate_range,
+};
+
+static int kvm_gmem_dmabuf_attach(struct inode *inode, int dmabuf_fd)
+{
+ struct gmem_inode *gi = GMEM_I(inode);
+ struct dma_buf_attachment *attach;
+ struct dma_buf *dmabuf;
+ int ret;
+
+ dmabuf = dma_buf_get(dmabuf_fd);
+ if (IS_ERR(dmabuf))
+ return PTR_ERR(dmabuf);
+ if (dmabuf->size < i_size_read(inode)) {
+ ret = -EINVAL;
+ goto put;
+ }
+
+ gi->dmabuf = dmabuf;
+ attach = dma_buf_dynamic_attach(dmabuf, NULL, &gmem_dmabuf_attach_ops,
+ inode);
+ if (IS_ERR(attach)) {
+ ret = PTR_ERR(attach);
+ goto clear_put;
+ }
+ gi->attach = attach;
+ ret = gmem_dmabuf_check(attach, i_size_read(inode));
+ if (ret)
+ goto detach;
+ return 0;
+
+detach:
+ dma_buf_detach(dmabuf, attach);
+ gi->attach = NULL;
+clear_put:
+ gi->dmabuf = NULL;
+put:
+ dma_buf_put(dmabuf);
+ return ret;
+}
+
+static void kvm_gmem_dmabuf_release(struct inode *inode)
+{
+ struct gmem_inode *gi = GMEM_I(inode);
+
+ if (!gi->dmabuf)
+ return;
+ dma_buf_detach(gi->dmabuf, gi->attach);
+ dma_buf_put(gi->dmabuf);
+ gi->attach = NULL;
+ gi->dmabuf = NULL;
+}
+
+static int gmem_dmabuf_get_pfn(struct file *file, struct kvm *kvm,
+ struct kvm_memory_slot *slot, gfn_t gfn,
+ kvm_pfn_t *pfn, struct page **page,
+ int *max_order, bool *writable)
+{
+ static const int orders[] = { PUD_ORDER, PMD_ORDER, 0 };
+ struct gmem_inode *gi = GMEM_I(file_inode(file));
+ struct gmem_file *f = gmem_file_of(file);
+ pgoff_t index = kvm_gmem_get_index(slot, gfn);
+ pgoff_t npages = i_size_read(file_inode(file)) >> PAGE_SHIFT;
+ unsigned int oi;
+ int ret = -EFAULT;
+
+ if (file != READ_ONCE(slot->gmem.file))
+ return -EFAULT;
+ if (xa_load(&f->bindings, index) != slot)
+ return -EIO;
+
+ /* A block maps at an order only if one aligned RAM run covers it. */
+ dma_resv_lock(gi->dmabuf->resv, NULL);
+ for (oi = max_order ? 0 : ARRAY_SIZE(orders) - 1;
+ oi < ARRAY_SIZE(orders); oi++) {
+ int order = orders[oi];
+ pgoff_t bs = ALIGN_DOWN(index, 1UL << order);
+ u64 blen = (u64)PAGE_SIZE << order;
+ struct phys_vec pv;
+ u32 attr;
+
+ if (bs + (1UL << order) > npages)
+ continue;
+ if (dma_buf_get_phys(gi->attach, (u64)bs << PAGE_SHIFT, blen,
+ &pv, &attr) ||
+ pv.len != blen || !IS_ALIGNED(pv.paddr, blen) ||
+ !dma_buf_phys_attr_known(attr) ||
+ dma_buf_phys_attr_type(attr) != DMA_BUF_PHYS_ATTR_RAM)
+ continue;
+
+ *pfn = PHYS_PFN(pv.paddr) + (index - bs);
+ *page = NULL;
+ if (max_order)
+ *max_order = order;
+ if (writable && (attr & DMA_BUF_PHYS_ATTR_READONLY))
+ *writable = false;
+ ret = 0;
+ break;
+ }
+ dma_resv_unlock(gi->dmabuf->resv);
+ return ret;
+}
+
+static int kvm_gmem_dmabuf_populate(struct file *file, struct kvm *kvm,
+ struct kvm_memory_slot *slot, gfn_t gfn,
+ kvm_pfn_t *pfn, struct page *src_page,
+ int order)
+{
+ return -EOPNOTSUPP;
+}
+
static int kvm_gmem_native_bind(struct file *file, struct kvm *kvm,
struct kvm_memory_slot *slot, loff_t offset);
static void kvm_gmem_native_unbind(struct file *slot_file, struct kvm *kvm,
@@ -654,7 +841,17 @@ static const struct kvm_gmem_ops kvm_gmem_native_ops = {
.fallocate = kvm_gmem_native_fallocate,
};
-static int __kvm_gmem_create(struct kvm *kvm, loff_t size, u64 flags)
+static const struct kvm_gmem_ops kvm_gmem_dmabuf_ops = {
+ .bind = kvm_gmem_native_bind,
+ .unbind = kvm_gmem_native_unbind,
+ .get_pfn = gmem_dmabuf_get_pfn,
+ .populate = kvm_gmem_dmabuf_populate,
+ .release = kvm_gmem_native_release,
+ .mmap = kvm_gmem_dmabuf_mmap,
+};
+
+static int __kvm_gmem_create(struct kvm *kvm, loff_t size, u64 flags,
+ int dmabuf_fd)
{
static const char *name = "[kvm-gmem]";
struct gmem_file *f;
@@ -695,6 +892,12 @@ static int __kvm_gmem_create(struct kvm *kvm, loff_t size, u64 flags)
GMEM_I(inode)->flags = flags;
+ if (flags & GUEST_MEMFD_FLAG_USE_DMABUF) {
+ err = kvm_gmem_dmabuf_attach(inode, dmabuf_fd);
+ if (err)
+ goto err_inode;
+ }
+
file = alloc_file_pseudo(inode, kvm_gmem_mnt, name, O_RDWR, &kvm_gmem_fops);
if (IS_ERR(file)) {
err = PTR_ERR(file);
@@ -703,12 +906,17 @@ static int __kvm_gmem_create(struct kvm *kvm, loff_t size, u64 flags)
file->f_flags |= O_LARGEFILE;
file->private_data = &f->backing;
- f->backing.ops = &kvm_gmem_native_ops;
+ f->backing.ops = flags & GUEST_MEMFD_FLAG_USE_DMABUF ?
+ &kvm_gmem_dmabuf_ops : &kvm_gmem_native_ops;
kvm_get_kvm(kvm);
f->kvm = kvm;
xa_init(&f->bindings);
+ if (flags & GUEST_MEMFD_FLAG_USE_DMABUF)
+ filemap_invalidate_lock(inode->i_mapping);
list_add(&f->entry, &GMEM_I(inode)->gmem_file_list);
+ if (flags & GUEST_MEMFD_FLAG_USE_DMABUF)
+ filemap_invalidate_unlock(inode->i_mapping);
fd_install(fd, file);
return fd;
@@ -734,8 +942,11 @@ int kvm_gmem_create(struct kvm *kvm, struct kvm_create_guest_memfd *args)
if (size <= 0 || !PAGE_ALIGNED(size))
return -EINVAL;
+ if (args->pad ||
+ (!(flags & GUEST_MEMFD_FLAG_USE_DMABUF) && args->dmabuf_fd))
+ return -EINVAL;
- return __kvm_gmem_create(kvm, size, flags);
+ return __kvm_gmem_create(kvm, size, flags, args->dmabuf_fd);
}
/*
@@ -1211,6 +1422,8 @@ static struct inode *kvm_gmem_alloc_inode(struct super_block *sb)
mpol_shared_policy_init(&gi->policy, NULL);
gi->flags = 0;
+ gi->dmabuf = NULL;
+ gi->attach = NULL;
INIT_LIST_HEAD(&gi->gmem_file_list);
return &gi->vfs_inode;
}
@@ -1225,11 +1438,19 @@ static void kvm_gmem_free_inode(struct inode *inode)
kmem_cache_free(kvm_gmem_inode_cachep, GMEM_I(inode));
}
+static void kvm_gmem_evict_inode(struct inode *inode)
+{
+ kvm_gmem_dmabuf_release(inode);
+ truncate_inode_pages_final(&inode->i_data);
+ clear_inode(inode);
+}
+
static const struct super_operations kvm_gmem_super_operations = {
.statfs = simple_statfs,
.alloc_inode = kvm_gmem_alloc_inode,
.destroy_inode = kvm_gmem_destroy_inode,
.free_inode = kvm_gmem_free_inode,
+ .evict_inode = kvm_gmem_evict_inode,
};
static int kvm_gmem_init_fs_context(struct fs_context *fc)
@@ -1292,3 +1513,5 @@ void kvm_gmem_exit(void)
rcu_barrier();
kmem_cache_destroy(kvm_gmem_inode_cachep);
}
+
+MODULE_IMPORT_NS("DMA_BUF");
--
2.47.3
^ permalink raw reply related [flat|nested] 10+ messages in thread