* [PATCH v2 1/7] vduse: add v1 API definition
2025-09-16 13:08 [PATCH v2 0/7] Add multiple address spaces support to VDUSE Eugenio Pérez
@ 2025-09-16 13:08 ` Eugenio Pérez
2025-09-16 13:08 ` [PATCH v2 2/7] vduse: make domain_lock an rwlock Eugenio Pérez
` (5 subsequent siblings)
6 siblings, 0 replies; 27+ messages in thread
From: Eugenio Pérez @ 2025-09-16 13:08 UTC (permalink / raw)
To: Michael S . Tsirkin
Cc: Stefano Garzarella, jasowang, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization, Eugenio Pérez
This allows the kernel to detect whether the userspace VDUSE device
supports the VQ group and ASID features. VDUSE devices that don't set
the V1 API will not receive the new messages, and vdpa device will be
created with only one vq group and asid.
The next patches implement the new feature incrementally, only enabling
the VDUSE device to set the V1 API version by the end of the series.
Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
---
include/uapi/linux/vduse.h | 4 ++++
1 file changed, 4 insertions(+)
diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h
index 10ad71aa00d6..ccb92a1efce0 100644
--- a/include/uapi/linux/vduse.h
+++ b/include/uapi/linux/vduse.h
@@ -10,6 +10,10 @@
#define VDUSE_API_VERSION 0
+/* VQ groups and ASID support */
+
+#define VDUSE_API_VERSION_1 1
+
/*
* Get the version of VDUSE API that kernel supported (VDUSE_API_VERSION).
* This is used for future extension.
--
2.51.0
^ permalink raw reply related [flat|nested] 27+ messages in thread* [PATCH v2 2/7] vduse: make domain_lock an rwlock
2025-09-16 13:08 [PATCH v2 0/7] Add multiple address spaces support to VDUSE Eugenio Pérez
2025-09-16 13:08 ` [PATCH v2 1/7] vduse: add v1 API definition Eugenio Pérez
@ 2025-09-16 13:08 ` Eugenio Pérez
2025-09-17 8:36 ` Jason Wang
2025-09-17 10:01 ` Yongji Xie
2025-09-16 13:08 ` [PATCH v2 3/7] vduse: add vq group support Eugenio Pérez
` (4 subsequent siblings)
6 siblings, 2 replies; 27+ messages in thread
From: Eugenio Pérez @ 2025-09-16 13:08 UTC (permalink / raw)
To: Michael S . Tsirkin
Cc: Stefano Garzarella, jasowang, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization, Eugenio Pérez
It will be used in a few more scenarios read-only so make it more
scalable.
Suggested-by: Xie Yongji <xieyongji@bytedance.com>
Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
---
v2: New in v2
---
drivers/vdpa/vdpa_user/vduse_dev.c | 41 +++++++++++++++---------------
1 file changed, 21 insertions(+), 20 deletions(-)
diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
index e7bced0b5542..2b6a8958ffe0 100644
--- a/drivers/vdpa/vdpa_user/vduse_dev.c
+++ b/drivers/vdpa/vdpa_user/vduse_dev.c
@@ -14,6 +14,7 @@
#include <linux/cdev.h>
#include <linux/device.h>
#include <linux/eventfd.h>
+#include <linux/rwlock.h>
#include <linux/slab.h>
#include <linux/wait.h>
#include <linux/dma-map-ops.h>
@@ -117,7 +118,7 @@ struct vduse_dev {
struct vduse_umem *umem;
struct mutex mem_lock;
unsigned int bounce_size;
- struct mutex domain_lock;
+ rwlock_t domain_lock;
};
struct vduse_dev_msg {
@@ -1176,9 +1177,9 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
if (entry.start > entry.last)
break;
- mutex_lock(&dev->domain_lock);
+ read_lock(&dev->domain_lock);
if (!dev->domain) {
- mutex_unlock(&dev->domain_lock);
+ read_unlock(&dev->domain_lock);
break;
}
spin_lock(&dev->domain->iotlb_lock);
@@ -1193,7 +1194,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
entry.perm = map->perm;
}
spin_unlock(&dev->domain->iotlb_lock);
- mutex_unlock(&dev->domain_lock);
+ read_unlock(&dev->domain_lock);
ret = -EINVAL;
if (!f)
break;
@@ -1346,10 +1347,10 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
sizeof(umem.reserved)))
break;
- mutex_lock(&dev->domain_lock);
+ write_lock(&dev->domain_lock);
ret = vduse_dev_reg_umem(dev, umem.iova,
umem.uaddr, umem.size);
- mutex_unlock(&dev->domain_lock);
+ write_unlock(&dev->domain_lock);
break;
}
case VDUSE_IOTLB_DEREG_UMEM: {
@@ -1363,10 +1364,10 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
if (!is_mem_zero((const char *)umem.reserved,
sizeof(umem.reserved)))
break;
- mutex_lock(&dev->domain_lock);
+ write_lock(&dev->domain_lock);
ret = vduse_dev_dereg_umem(dev, umem.iova,
umem.size);
- mutex_unlock(&dev->domain_lock);
+ write_unlock(&dev->domain_lock);
break;
}
case VDUSE_IOTLB_GET_INFO: {
@@ -1385,9 +1386,9 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
sizeof(info.reserved)))
break;
- mutex_lock(&dev->domain_lock);
+ read_lock(&dev->domain_lock);
if (!dev->domain) {
- mutex_unlock(&dev->domain_lock);
+ read_unlock(&dev->domain_lock);
break;
}
spin_lock(&dev->domain->iotlb_lock);
@@ -1402,7 +1403,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
info.capability |= VDUSE_IOVA_CAP_UMEM;
}
spin_unlock(&dev->domain->iotlb_lock);
- mutex_unlock(&dev->domain_lock);
+ read_unlock(&dev->domain_lock);
if (!map)
break;
@@ -1425,10 +1426,10 @@ static int vduse_dev_release(struct inode *inode, struct file *file)
{
struct vduse_dev *dev = file->private_data;
- mutex_lock(&dev->domain_lock);
+ write_lock(&dev->domain_lock);
if (dev->domain)
vduse_dev_dereg_umem(dev, 0, dev->domain->bounce_size);
- mutex_unlock(&dev->domain_lock);
+ write_unlock(&dev->domain_lock);
spin_lock(&dev->msg_lock);
/* Make sure the inflight messages can processed after reconncection */
list_splice_init(&dev->recv_list, &dev->send_list);
@@ -1647,7 +1648,7 @@ static struct vduse_dev *vduse_dev_create(void)
mutex_init(&dev->lock);
mutex_init(&dev->mem_lock);
- mutex_init(&dev->domain_lock);
+ rwlock_init(&dev->domain_lock);
spin_lock_init(&dev->msg_lock);
INIT_LIST_HEAD(&dev->send_list);
INIT_LIST_HEAD(&dev->recv_list);
@@ -1805,7 +1806,7 @@ static ssize_t bounce_size_store(struct device *device,
int ret;
ret = -EPERM;
- mutex_lock(&dev->domain_lock);
+ write_lock(&dev->domain_lock);
if (dev->domain)
goto unlock;
@@ -1821,7 +1822,7 @@ static ssize_t bounce_size_store(struct device *device,
dev->bounce_size = bounce_size & PAGE_MASK;
ret = count;
unlock:
- mutex_unlock(&dev->domain_lock);
+ write_unlock(&dev->domain_lock);
return ret;
}
@@ -2045,11 +2046,11 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
if (ret)
return ret;
- mutex_lock(&dev->domain_lock);
+ write_lock(&dev->domain_lock);
if (!dev->domain)
dev->domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
dev->bounce_size);
- mutex_unlock(&dev->domain_lock);
+ write_unlock(&dev->domain_lock);
if (!dev->domain) {
put_device(&dev->vdev->vdpa.dev);
return -ENOMEM;
@@ -2059,10 +2060,10 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
ret = _vdpa_register_device(&dev->vdev->vdpa, dev->vq_num);
if (ret) {
put_device(&dev->vdev->vdpa.dev);
- mutex_lock(&dev->domain_lock);
+ write_lock(&dev->domain_lock);
vduse_domain_destroy(dev->domain);
dev->domain = NULL;
- mutex_unlock(&dev->domain_lock);
+ write_unlock(&dev->domain_lock);
return ret;
}
--
2.51.0
^ permalink raw reply related [flat|nested] 27+ messages in thread* Re: [PATCH v2 2/7] vduse: make domain_lock an rwlock
2025-09-16 13:08 ` [PATCH v2 2/7] vduse: make domain_lock an rwlock Eugenio Pérez
@ 2025-09-17 8:36 ` Jason Wang
2025-09-17 10:01 ` Yongji Xie
1 sibling, 0 replies; 27+ messages in thread
From: Jason Wang @ 2025-09-17 8:36 UTC (permalink / raw)
To: Eugenio Pérez
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Tue, Sep 16, 2025 at 9:08 PM Eugenio Pérez <eperezma@redhat.com> wrote:
>
> It will be used in a few more scenarios read-only so make it more
> scalable.
>
> Suggested-by: Xie Yongji <xieyongji@bytedance.com>
> Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
Acked-by: Jason Wang <jasowang@redhat.com>
Thanks
> ---
> v2: New in v2
> ---
> drivers/vdpa/vdpa_user/vduse_dev.c | 41 +++++++++++++++---------------
> 1 file changed, 21 insertions(+), 20 deletions(-)
>
> diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> index e7bced0b5542..2b6a8958ffe0 100644
> --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> @@ -14,6 +14,7 @@
> #include <linux/cdev.h>
> #include <linux/device.h>
> #include <linux/eventfd.h>
> +#include <linux/rwlock.h>
> #include <linux/slab.h>
> #include <linux/wait.h>
> #include <linux/dma-map-ops.h>
> @@ -117,7 +118,7 @@ struct vduse_dev {
> struct vduse_umem *umem;
> struct mutex mem_lock;
> unsigned int bounce_size;
> - struct mutex domain_lock;
> + rwlock_t domain_lock;
> };
>
> struct vduse_dev_msg {
> @@ -1176,9 +1177,9 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> if (entry.start > entry.last)
> break;
>
> - mutex_lock(&dev->domain_lock);
> + read_lock(&dev->domain_lock);
> if (!dev->domain) {
> - mutex_unlock(&dev->domain_lock);
> + read_unlock(&dev->domain_lock);
> break;
> }
> spin_lock(&dev->domain->iotlb_lock);
> @@ -1193,7 +1194,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> entry.perm = map->perm;
> }
> spin_unlock(&dev->domain->iotlb_lock);
> - mutex_unlock(&dev->domain_lock);
> + read_unlock(&dev->domain_lock);
> ret = -EINVAL;
> if (!f)
> break;
> @@ -1346,10 +1347,10 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> sizeof(umem.reserved)))
> break;
>
> - mutex_lock(&dev->domain_lock);
> + write_lock(&dev->domain_lock);
> ret = vduse_dev_reg_umem(dev, umem.iova,
> umem.uaddr, umem.size);
> - mutex_unlock(&dev->domain_lock);
> + write_unlock(&dev->domain_lock);
> break;
> }
> case VDUSE_IOTLB_DEREG_UMEM: {
> @@ -1363,10 +1364,10 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> if (!is_mem_zero((const char *)umem.reserved,
> sizeof(umem.reserved)))
> break;
> - mutex_lock(&dev->domain_lock);
> + write_lock(&dev->domain_lock);
> ret = vduse_dev_dereg_umem(dev, umem.iova,
> umem.size);
> - mutex_unlock(&dev->domain_lock);
> + write_unlock(&dev->domain_lock);
> break;
> }
> case VDUSE_IOTLB_GET_INFO: {
> @@ -1385,9 +1386,9 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> sizeof(info.reserved)))
> break;
>
> - mutex_lock(&dev->domain_lock);
> + read_lock(&dev->domain_lock);
> if (!dev->domain) {
> - mutex_unlock(&dev->domain_lock);
> + read_unlock(&dev->domain_lock);
> break;
> }
> spin_lock(&dev->domain->iotlb_lock);
> @@ -1402,7 +1403,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> info.capability |= VDUSE_IOVA_CAP_UMEM;
> }
> spin_unlock(&dev->domain->iotlb_lock);
> - mutex_unlock(&dev->domain_lock);
> + read_unlock(&dev->domain_lock);
> if (!map)
> break;
>
> @@ -1425,10 +1426,10 @@ static int vduse_dev_release(struct inode *inode, struct file *file)
> {
> struct vduse_dev *dev = file->private_data;
>
> - mutex_lock(&dev->domain_lock);
> + write_lock(&dev->domain_lock);
> if (dev->domain)
> vduse_dev_dereg_umem(dev, 0, dev->domain->bounce_size);
> - mutex_unlock(&dev->domain_lock);
> + write_unlock(&dev->domain_lock);
> spin_lock(&dev->msg_lock);
> /* Make sure the inflight messages can processed after reconncection */
> list_splice_init(&dev->recv_list, &dev->send_list);
> @@ -1647,7 +1648,7 @@ static struct vduse_dev *vduse_dev_create(void)
>
> mutex_init(&dev->lock);
> mutex_init(&dev->mem_lock);
> - mutex_init(&dev->domain_lock);
> + rwlock_init(&dev->domain_lock);
> spin_lock_init(&dev->msg_lock);
> INIT_LIST_HEAD(&dev->send_list);
> INIT_LIST_HEAD(&dev->recv_list);
> @@ -1805,7 +1806,7 @@ static ssize_t bounce_size_store(struct device *device,
> int ret;
>
> ret = -EPERM;
> - mutex_lock(&dev->domain_lock);
> + write_lock(&dev->domain_lock);
> if (dev->domain)
> goto unlock;
>
> @@ -1821,7 +1822,7 @@ static ssize_t bounce_size_store(struct device *device,
> dev->bounce_size = bounce_size & PAGE_MASK;
> ret = count;
> unlock:
> - mutex_unlock(&dev->domain_lock);
> + write_unlock(&dev->domain_lock);
> return ret;
> }
>
> @@ -2045,11 +2046,11 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> if (ret)
> return ret;
>
> - mutex_lock(&dev->domain_lock);
> + write_lock(&dev->domain_lock);
> if (!dev->domain)
> dev->domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
> dev->bounce_size);
> - mutex_unlock(&dev->domain_lock);
> + write_unlock(&dev->domain_lock);
> if (!dev->domain) {
> put_device(&dev->vdev->vdpa.dev);
> return -ENOMEM;
> @@ -2059,10 +2060,10 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> ret = _vdpa_register_device(&dev->vdev->vdpa, dev->vq_num);
> if (ret) {
> put_device(&dev->vdev->vdpa.dev);
> - mutex_lock(&dev->domain_lock);
> + write_lock(&dev->domain_lock);
> vduse_domain_destroy(dev->domain);
> dev->domain = NULL;
> - mutex_unlock(&dev->domain_lock);
> + write_unlock(&dev->domain_lock);
> return ret;
> }
>
> --
> 2.51.0
>
^ permalink raw reply [flat|nested] 27+ messages in thread* Re: [PATCH v2 2/7] vduse: make domain_lock an rwlock
2025-09-16 13:08 ` [PATCH v2 2/7] vduse: make domain_lock an rwlock Eugenio Pérez
2025-09-17 8:36 ` Jason Wang
@ 2025-09-17 10:01 ` Yongji Xie
1 sibling, 0 replies; 27+ messages in thread
From: Yongji Xie @ 2025-09-17 10:01 UTC (permalink / raw)
To: Eugenio Pérez
Cc: Michael S . Tsirkin, Stefano Garzarella, Jason Wang, Xuan Zhuo,
linux-kernel, Maxime Coquelin, Cindy Lu, Laurent Vivier,
virtualization
On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
>
> It will be used in a few more scenarios read-only so make it more
> scalable.
>
> Suggested-by: Xie Yongji <xieyongji@bytedance.com>
> Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
Reviewed-by: Xie Yongji <xieyongji@bytedance.com>
Thanks,
Yongji
^ permalink raw reply [flat|nested] 27+ messages in thread
* [PATCH v2 3/7] vduse: add vq group support
2025-09-16 13:08 [PATCH v2 0/7] Add multiple address spaces support to VDUSE Eugenio Pérez
2025-09-16 13:08 ` [PATCH v2 1/7] vduse: add v1 API definition Eugenio Pérez
2025-09-16 13:08 ` [PATCH v2 2/7] vduse: make domain_lock an rwlock Eugenio Pérez
@ 2025-09-16 13:08 ` Eugenio Pérez
2025-09-17 8:36 ` Jason Wang
2025-09-16 13:08 ` [PATCH v2 4/7] vduse: return internal vq group struct as map token Eugenio Pérez
` (3 subsequent siblings)
6 siblings, 1 reply; 27+ messages in thread
From: Eugenio Pérez @ 2025-09-16 13:08 UTC (permalink / raw)
To: Michael S . Tsirkin
Cc: Stefano Garzarella, jasowang, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization, Eugenio Pérez
This allows sepparate the different virtqueues in groups that shares the
same address space. Asking the VDUSE device for the groups of the vq at
the beginning as they're needed for the DMA API.
Allocating 3 vq groups as net is the device that need the most groups:
* Dataplane (guest passthrough)
* CVQ
* Shadowed vrings.
Future versions of the series can include dynamic allocation of the
groups array so VDUSE can declare more groups.
Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
---
v2:
* Now the vq group is in vduse_vq_config struct instead of issuing one
VDUSE message per vq.
v1:
* Fix: Remove BIT_ULL(VIRTIO_S_*), as _S_ is already the bit (Maxime)
RFC v3:
* Increase VDUSE_MAX_VQ_GROUPS to 0xffff (Jason). It was set to a lower
value to reduce memory consumption, but vqs are already limited to
that value and userspace VDUSE is able to allocate that many vqs.
* Remove the descs vq group capability as it will not be used and we can
add it on top.
* Do not ask for vq groups in number of vq groups < 2.
* Move the valid vq groups range check to vduse_validate_config.
RFC v2:
* Cache group information in kernel, as we need to provide the vq map
tokens properly.
* Add descs vq group to optimize SVQ forwarding and support indirect
descriptors out of the box.
---
drivers/vdpa/vdpa_user/vduse_dev.c | 37 ++++++++++++++++++++++++++----
include/uapi/linux/vduse.h | 12 +++++++---
2 files changed, 41 insertions(+), 8 deletions(-)
diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
index 2b6a8958ffe0..42f8807911d4 100644
--- a/drivers/vdpa/vdpa_user/vduse_dev.c
+++ b/drivers/vdpa/vdpa_user/vduse_dev.c
@@ -59,6 +59,7 @@ struct vduse_virtqueue {
struct vdpa_vq_state state;
bool ready;
bool kicked;
+ u32 vq_group;
spinlock_t kick_lock;
spinlock_t irq_lock;
struct eventfd_ctx *kickfd;
@@ -115,6 +116,7 @@ struct vduse_dev {
u8 status;
u32 vq_num;
u32 vq_align;
+ u32 ngroups;
struct vduse_umem *umem;
struct mutex mem_lock;
unsigned int bounce_size;
@@ -593,6 +595,13 @@ static int vduse_vdpa_set_vq_state(struct vdpa_device *vdpa, u16 idx,
return 0;
}
+static u32 vduse_get_vq_group(struct vdpa_device *vdpa, u16 idx)
+{
+ struct vduse_dev *dev = vdpa_to_vduse(vdpa);
+
+ return dev->vqs[idx]->vq_group;
+}
+
static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
struct vdpa_vq_state *state)
{
@@ -790,6 +799,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
.set_vq_cb = vduse_vdpa_set_vq_cb,
.set_vq_num = vduse_vdpa_set_vq_num,
.get_vq_size = vduse_vdpa_get_vq_size,
+ .get_vq_group = vduse_get_vq_group,
.set_vq_ready = vduse_vdpa_set_vq_ready,
.get_vq_ready = vduse_vdpa_get_vq_ready,
.set_vq_state = vduse_vdpa_set_vq_state,
@@ -1253,12 +1263,21 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
if (config.index >= dev->vq_num)
break;
- if (!is_mem_zero((const char *)config.reserved,
- sizeof(config.reserved)))
+ if (dev->api_version < VDUSE_API_VERSION_1 && config.group)
+ break;
+
+ if (dev->api_version >= VDUSE_API_VERSION_1 &&
+ config.group > dev->ngroups)
+ break;
+
+ if (config.reserved1 ||
+ !is_mem_zero((const char *)config.reserved2,
+ sizeof(config.reserved2)))
break;
index = array_index_nospec(config.index, dev->vq_num);
dev->vqs[index]->num_max = config.max_size;
+ dev->vqs[index]->vq_group = config.group;
ret = 0;
break;
}
@@ -1738,12 +1757,19 @@ static bool features_is_valid(struct vduse_dev_config *config)
return true;
}
-static bool vduse_validate_config(struct vduse_dev_config *config)
+static bool vduse_validate_config(struct vduse_dev_config *config,
+ u64 api_version)
{
if (!is_mem_zero((const char *)config->reserved,
sizeof(config->reserved)))
return false;
+ if (api_version < VDUSE_API_VERSION_1 && config->ngroups)
+ return false;
+
+ if (api_version >= VDUSE_API_VERSION_1 && config->ngroups > 0xffff)
+ return false;
+
if (config->vq_align > PAGE_SIZE)
return false;
@@ -1859,6 +1885,7 @@ static int vduse_create_dev(struct vduse_dev_config *config,
dev->device_features = config->features;
dev->device_id = config->device_id;
dev->vendor_id = config->vendor_id;
+ dev->ngroups = (dev->api_version < 1) ? 1 : (config->ngroups ?: 1);
dev->name = kstrdup(config->name, GFP_KERNEL);
if (!dev->name)
goto err_str;
@@ -1937,7 +1964,7 @@ static long vduse_ioctl(struct file *file, unsigned int cmd,
break;
ret = -EINVAL;
- if (vduse_validate_config(&config) == false)
+ if (!vduse_validate_config(&config, control->api_version))
break;
buf = vmemdup_user(argp + size, config.config_size);
@@ -2018,7 +2045,7 @@ static int vduse_dev_init_vdpa(struct vduse_dev *dev, const char *name)
vdev = vdpa_alloc_device(struct vduse_vdpa, vdpa, dev->dev,
&vduse_vdpa_config_ops, &vduse_map_ops,
- 1, 1, name, true);
+ dev->ngroups, 1, name, true);
if (IS_ERR(vdev))
return PTR_ERR(vdev);
diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h
index ccb92a1efce0..a3d51cf6df3a 100644
--- a/include/uapi/linux/vduse.h
+++ b/include/uapi/linux/vduse.h
@@ -31,6 +31,7 @@
* @features: virtio features
* @vq_num: the number of virtqueues
* @vq_align: the allocation alignment of virtqueue's metadata
+ * @ngroups: number of vq groups that VDUSE device declares
* @reserved: for future use, needs to be initialized to zero
* @config_size: the size of the configuration space
* @config: the buffer of the configuration space
@@ -45,7 +46,8 @@ struct vduse_dev_config {
__u64 features;
__u32 vq_num;
__u32 vq_align;
- __u32 reserved[13];
+ __u32 ngroups; /* if VDUSE_API_VERSION >= 1 */
+ __u32 reserved[12];
__u32 config_size;
__u8 config[];
};
@@ -122,14 +124,18 @@ struct vduse_config_data {
* struct vduse_vq_config - basic configuration of a virtqueue
* @index: virtqueue index
* @max_size: the max size of virtqueue
- * @reserved: for future use, needs to be initialized to zero
+ * @reserved1: for future use, needs to be initialized to zero
+ * @group: virtqueue group
+ * @reserved2: for future use, needs to be initialized to zero
*
* Structure used by VDUSE_VQ_SETUP ioctl to setup a virtqueue.
*/
struct vduse_vq_config {
__u32 index;
__u16 max_size;
- __u16 reserved[13];
+ __u16 reserved1;
+ __u32 group;
+ __u16 reserved2[10];
};
/*
--
2.51.0
^ permalink raw reply related [flat|nested] 27+ messages in thread* Re: [PATCH v2 3/7] vduse: add vq group support
2025-09-16 13:08 ` [PATCH v2 3/7] vduse: add vq group support Eugenio Pérez
@ 2025-09-17 8:36 ` Jason Wang
2025-09-17 15:41 ` Eugenio Perez Martin
0 siblings, 1 reply; 27+ messages in thread
From: Jason Wang @ 2025-09-17 8:36 UTC (permalink / raw)
To: Eugenio Pérez
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Tue, Sep 16, 2025 at 9:08 PM Eugenio Pérez <eperezma@redhat.com> wrote:
>
> This allows sepparate the different virtqueues in groups that shares the
> same address space. Asking the VDUSE device for the groups of the vq at
> the beginning as they're needed for the DMA API.
>
> Allocating 3 vq groups as net is the device that need the most groups:
> * Dataplane (guest passthrough)
> * CVQ
> * Shadowed vrings.
>
> Future versions of the series can include dynamic allocation of the
> groups array so VDUSE can declare more groups.
>
> Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> ---
> v2:
> * Now the vq group is in vduse_vq_config struct instead of issuing one
> VDUSE message per vq.
>
> v1:
> * Fix: Remove BIT_ULL(VIRTIO_S_*), as _S_ is already the bit (Maxime)
>
> RFC v3:
> * Increase VDUSE_MAX_VQ_GROUPS to 0xffff (Jason). It was set to a lower
> value to reduce memory consumption, but vqs are already limited to
> that value and userspace VDUSE is able to allocate that many vqs.
> * Remove the descs vq group capability as it will not be used and we can
> add it on top.
> * Do not ask for vq groups in number of vq groups < 2.
> * Move the valid vq groups range check to vduse_validate_config.
>
> RFC v2:
> * Cache group information in kernel, as we need to provide the vq map
> tokens properly.
> * Add descs vq group to optimize SVQ forwarding and support indirect
> descriptors out of the box.
> ---
> drivers/vdpa/vdpa_user/vduse_dev.c | 37 ++++++++++++++++++++++++++----
> include/uapi/linux/vduse.h | 12 +++++++---
> 2 files changed, 41 insertions(+), 8 deletions(-)
>
> diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> index 2b6a8958ffe0..42f8807911d4 100644
> --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> @@ -59,6 +59,7 @@ struct vduse_virtqueue {
> struct vdpa_vq_state state;
> bool ready;
> bool kicked;
> + u32 vq_group;
> spinlock_t kick_lock;
> spinlock_t irq_lock;
> struct eventfd_ctx *kickfd;
> @@ -115,6 +116,7 @@ struct vduse_dev {
> u8 status;
> u32 vq_num;
> u32 vq_align;
> + u32 ngroups;
> struct vduse_umem *umem;
> struct mutex mem_lock;
> unsigned int bounce_size;
> @@ -593,6 +595,13 @@ static int vduse_vdpa_set_vq_state(struct vdpa_device *vdpa, u16 idx,
> return 0;
> }
>
> +static u32 vduse_get_vq_group(struct vdpa_device *vdpa, u16 idx)
> +{
> + struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> +
> + return dev->vqs[idx]->vq_group;
I wonder if we should fail if VDUSE_VQ_SETUP is not set by userspace?
> +}
> +
> static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
> struct vdpa_vq_state *state)
> {
> @@ -790,6 +799,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
> .set_vq_cb = vduse_vdpa_set_vq_cb,
> .set_vq_num = vduse_vdpa_set_vq_num,
> .get_vq_size = vduse_vdpa_get_vq_size,
> + .get_vq_group = vduse_get_vq_group,
> .set_vq_ready = vduse_vdpa_set_vq_ready,
> .get_vq_ready = vduse_vdpa_get_vq_ready,
> .set_vq_state = vduse_vdpa_set_vq_state,
> @@ -1253,12 +1263,21 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> if (config.index >= dev->vq_num)
> break;
>
> - if (!is_mem_zero((const char *)config.reserved,
> - sizeof(config.reserved)))
> + if (dev->api_version < VDUSE_API_VERSION_1 && config.group)
> + break;
> +
> + if (dev->api_version >= VDUSE_API_VERSION_1 &&
> + config.group > dev->ngroups)
> + break;
> +
> + if (config.reserved1 ||
> + !is_mem_zero((const char *)config.reserved2,
> + sizeof(config.reserved2)))
> break;
What's the reason for having this check? I mean if we don't check this
since the day0, it might be too late to do that.
>
> index = array_index_nospec(config.index, dev->vq_num);
> dev->vqs[index]->num_max = config.max_size;
> + dev->vqs[index]->vq_group = config.group;
> ret = 0;
> break;
> }
> @@ -1738,12 +1757,19 @@ static bool features_is_valid(struct vduse_dev_config *config)
> return true;
> }
>
> -static bool vduse_validate_config(struct vduse_dev_config *config)
> +static bool vduse_validate_config(struct vduse_dev_config *config,
> + u64 api_version)
> {
> if (!is_mem_zero((const char *)config->reserved,
> sizeof(config->reserved)))
> return false;
>
> + if (api_version < VDUSE_API_VERSION_1 && config->ngroups)
> + return false;
Better with ngroups > 1?
> +
> + if (api_version >= VDUSE_API_VERSION_1 && config->ngroups > 0xffff)
> + return false;
> +
> if (config->vq_align > PAGE_SIZE)
> return false;
>
> @@ -1859,6 +1885,7 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> dev->device_features = config->features;
> dev->device_id = config->device_id;
> dev->vendor_id = config->vendor_id;
> + dev->ngroups = (dev->api_version < 1) ? 1 : (config->ngroups ?: 1);
> dev->name = kstrdup(config->name, GFP_KERNEL);
> if (!dev->name)
> goto err_str;
> @@ -1937,7 +1964,7 @@ static long vduse_ioctl(struct file *file, unsigned int cmd,
> break;
>
> ret = -EINVAL;
> - if (vduse_validate_config(&config) == false)
> + if (!vduse_validate_config(&config, control->api_version))
> break;
>
> buf = vmemdup_user(argp + size, config.config_size);
> @@ -2018,7 +2045,7 @@ static int vduse_dev_init_vdpa(struct vduse_dev *dev, const char *name)
>
> vdev = vdpa_alloc_device(struct vduse_vdpa, vdpa, dev->dev,
> &vduse_vdpa_config_ops, &vduse_map_ops,
> - 1, 1, name, true);
> + dev->ngroups, 1, name, true);
> if (IS_ERR(vdev))
> return PTR_ERR(vdev);
>
> diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h
> index ccb92a1efce0..a3d51cf6df3a 100644
> --- a/include/uapi/linux/vduse.h
> +++ b/include/uapi/linux/vduse.h
> @@ -31,6 +31,7 @@
> * @features: virtio features
> * @vq_num: the number of virtqueues
> * @vq_align: the allocation alignment of virtqueue's metadata
> + * @ngroups: number of vq groups that VDUSE device declares
> * @reserved: for future use, needs to be initialized to zero
> * @config_size: the size of the configuration space
> * @config: the buffer of the configuration space
> @@ -45,7 +46,8 @@ struct vduse_dev_config {
> __u64 features;
> __u32 vq_num;
> __u32 vq_align;
> - __u32 reserved[13];
> + __u32 ngroups; /* if VDUSE_API_VERSION >= 1 */
> + __u32 reserved[12];
> __u32 config_size;
> __u8 config[];
> };
> @@ -122,14 +124,18 @@ struct vduse_config_data {
> * struct vduse_vq_config - basic configuration of a virtqueue
> * @index: virtqueue index
> * @max_size: the max size of virtqueue
> - * @reserved: for future use, needs to be initialized to zero
> + * @reserved1: for future use, needs to be initialized to zero
> + * @group: virtqueue group
> + * @reserved2: for future use, needs to be initialized to zero
> *
> * Structure used by VDUSE_VQ_SETUP ioctl to setup a virtqueue.
> */
> struct vduse_vq_config {
> __u32 index;
> __u16 max_size;
> - __u16 reserved[13];
> + __u16 reserved1;
> + __u32 group;
This makes me think if u16 is sufficient as I see:
u32 (*get_vq_group)(struct vdpa_device *vdev, u16 idx);
The index is u16 but the group is u32 ...
> + __u16 reserved2[10];
> };
>
> /*
> --
> 2.51.0
>
Thanks
^ permalink raw reply [flat|nested] 27+ messages in thread* Re: [PATCH v2 3/7] vduse: add vq group support
2025-09-17 8:36 ` Jason Wang
@ 2025-09-17 15:41 ` Eugenio Perez Martin
2025-09-18 6:00 ` Jason Wang
0 siblings, 1 reply; 27+ messages in thread
From: Eugenio Perez Martin @ 2025-09-17 15:41 UTC (permalink / raw)
To: Jason Wang
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Wed, Sep 17, 2025 at 10:37 AM Jason Wang <jasowang@redhat.com> wrote:
>
> On Tue, Sep 16, 2025 at 9:08 PM Eugenio Pérez <eperezma@redhat.com> wrote:
> >
> > This allows sepparate the different virtqueues in groups that shares the
> > same address space. Asking the VDUSE device for the groups of the vq at
> > the beginning as they're needed for the DMA API.
> >
> > Allocating 3 vq groups as net is the device that need the most groups:
> > * Dataplane (guest passthrough)
> > * CVQ
> > * Shadowed vrings.
> >
> > Future versions of the series can include dynamic allocation of the
> > groups array so VDUSE can declare more groups.
> >
> > Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> > ---
> > v2:
> > * Now the vq group is in vduse_vq_config struct instead of issuing one
> > VDUSE message per vq.
> >
> > v1:
> > * Fix: Remove BIT_ULL(VIRTIO_S_*), as _S_ is already the bit (Maxime)
> >
> > RFC v3:
> > * Increase VDUSE_MAX_VQ_GROUPS to 0xffff (Jason). It was set to a lower
> > value to reduce memory consumption, but vqs are already limited to
> > that value and userspace VDUSE is able to allocate that many vqs.
> > * Remove the descs vq group capability as it will not be used and we can
> > add it on top.
> > * Do not ask for vq groups in number of vq groups < 2.
> > * Move the valid vq groups range check to vduse_validate_config.
> >
> > RFC v2:
> > * Cache group information in kernel, as we need to provide the vq map
> > tokens properly.
> > * Add descs vq group to optimize SVQ forwarding and support indirect
> > descriptors out of the box.
> > ---
> > drivers/vdpa/vdpa_user/vduse_dev.c | 37 ++++++++++++++++++++++++++----
> > include/uapi/linux/vduse.h | 12 +++++++---
> > 2 files changed, 41 insertions(+), 8 deletions(-)
> >
> > diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> > index 2b6a8958ffe0..42f8807911d4 100644
> > --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> > +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> > @@ -59,6 +59,7 @@ struct vduse_virtqueue {
> > struct vdpa_vq_state state;
> > bool ready;
> > bool kicked;
> > + u32 vq_group;
> > spinlock_t kick_lock;
> > spinlock_t irq_lock;
> > struct eventfd_ctx *kickfd;
> > @@ -115,6 +116,7 @@ struct vduse_dev {
> > u8 status;
> > u32 vq_num;
> > u32 vq_align;
> > + u32 ngroups;
> > struct vduse_umem *umem;
> > struct mutex mem_lock;
> > unsigned int bounce_size;
> > @@ -593,6 +595,13 @@ static int vduse_vdpa_set_vq_state(struct vdpa_device *vdpa, u16 idx,
> > return 0;
> > }
> >
> > +static u32 vduse_get_vq_group(struct vdpa_device *vdpa, u16 idx)
> > +{
> > + struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > +
> > + return dev->vqs[idx]->vq_group;
>
> I wonder if we should fail if VDUSE_VQ_SETUP is not set by userspace?
>
I'm kind of ok with implementing it, but I see it as redundant as if
the VDUSE device does not call the vq max size will be 0 anyway.
> > +}
> > +
> > static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
> > struct vdpa_vq_state *state)
> > {
> > @@ -790,6 +799,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
> > .set_vq_cb = vduse_vdpa_set_vq_cb,
> > .set_vq_num = vduse_vdpa_set_vq_num,
> > .get_vq_size = vduse_vdpa_get_vq_size,
> > + .get_vq_group = vduse_get_vq_group,
> > .set_vq_ready = vduse_vdpa_set_vq_ready,
> > .get_vq_ready = vduse_vdpa_get_vq_ready,
> > .set_vq_state = vduse_vdpa_set_vq_state,
> > @@ -1253,12 +1263,21 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > if (config.index >= dev->vq_num)
> > break;
> >
> > - if (!is_mem_zero((const char *)config.reserved,
> > - sizeof(config.reserved)))
> > + if (dev->api_version < VDUSE_API_VERSION_1 && config.group)
> > + break;
> > +
> > + if (dev->api_version >= VDUSE_API_VERSION_1 &&
> > + config.group > dev->ngroups)
> > + break;
> > +
> > + if (config.reserved1 ||
> > + !is_mem_zero((const char *)config.reserved2,
> > + sizeof(config.reserved2)))
> > break;
>
> What's the reason for having this check? I mean if we don't check this
> since the day0, it might be too late to do that.
>
It's been checked from day0, both reserved1 and reserved2 were
included in the config.reserved before the patch, and the replaced
lines already checked with is_mem_zero [1].
> >
> > index = array_index_nospec(config.index, dev->vq_num);
> > dev->vqs[index]->num_max = config.max_size;
> > + dev->vqs[index]->vq_group = config.group;
> > ret = 0;
> > break;
> > }
> > @@ -1738,12 +1757,19 @@ static bool features_is_valid(struct vduse_dev_config *config)
> > return true;
> > }
> >
> > -static bool vduse_validate_config(struct vduse_dev_config *config)
> > +static bool vduse_validate_config(struct vduse_dev_config *config,
> > + u64 api_version)
> > {
> > if (!is_mem_zero((const char *)config->reserved,
> > sizeof(config->reserved)))
> > return false;
> >
> > + if (api_version < VDUSE_API_VERSION_1 && config->ngroups)
> > + return false;
>
> Better with ngroups > 1?
>
Same here. The kernel in the master branch already returns false if
that position of "reserved" is 1, and we are changing the behavior if
we don't include the 1 too.
> > +
> > + if (api_version >= VDUSE_API_VERSION_1 && config->ngroups > 0xffff)
> > + return false;
> > +
> > if (config->vq_align > PAGE_SIZE)
> > return false;
> >
> > @@ -1859,6 +1885,7 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> > dev->device_features = config->features;
> > dev->device_id = config->device_id;
> > dev->vendor_id = config->vendor_id;
> > + dev->ngroups = (dev->api_version < 1) ? 1 : (config->ngroups ?: 1);
> > dev->name = kstrdup(config->name, GFP_KERNEL);
> > if (!dev->name)
> > goto err_str;
> > @@ -1937,7 +1964,7 @@ static long vduse_ioctl(struct file *file, unsigned int cmd,
> > break;
> >
> > ret = -EINVAL;
> > - if (vduse_validate_config(&config) == false)
> > + if (!vduse_validate_config(&config, control->api_version))
> > break;
> >
> > buf = vmemdup_user(argp + size, config.config_size);
> > @@ -2018,7 +2045,7 @@ static int vduse_dev_init_vdpa(struct vduse_dev *dev, const char *name)
> >
> > vdev = vdpa_alloc_device(struct vduse_vdpa, vdpa, dev->dev,
> > &vduse_vdpa_config_ops, &vduse_map_ops,
> > - 1, 1, name, true);
> > + dev->ngroups, 1, name, true);
> > if (IS_ERR(vdev))
> > return PTR_ERR(vdev);
> >
> > diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h
> > index ccb92a1efce0..a3d51cf6df3a 100644
> > --- a/include/uapi/linux/vduse.h
> > +++ b/include/uapi/linux/vduse.h
> > @@ -31,6 +31,7 @@
> > * @features: virtio features
> > * @vq_num: the number of virtqueues
> > * @vq_align: the allocation alignment of virtqueue's metadata
> > + * @ngroups: number of vq groups that VDUSE device declares
> > * @reserved: for future use, needs to be initialized to zero
> > * @config_size: the size of the configuration space
> > * @config: the buffer of the configuration space
> > @@ -45,7 +46,8 @@ struct vduse_dev_config {
> > __u64 features;
> > __u32 vq_num;
> > __u32 vq_align;
> > - __u32 reserved[13];
> > + __u32 ngroups; /* if VDUSE_API_VERSION >= 1 */
> > + __u32 reserved[12];
> > __u32 config_size;
> > __u8 config[];
> > };
> > @@ -122,14 +124,18 @@ struct vduse_config_data {
> > * struct vduse_vq_config - basic configuration of a virtqueue
> > * @index: virtqueue index
> > * @max_size: the max size of virtqueue
> > - * @reserved: for future use, needs to be initialized to zero
> > + * @reserved1: for future use, needs to be initialized to zero
> > + * @group: virtqueue group
> > + * @reserved2: for future use, needs to be initialized to zero
> > *
> > * Structure used by VDUSE_VQ_SETUP ioctl to setup a virtqueue.
> > */
> > struct vduse_vq_config {
> > __u32 index;
> > __u16 max_size;
> > - __u16 reserved[13];
> > + __u16 reserved1;
> > + __u32 group;
>
> This makes me think if u16 is sufficient as I see:
>
> u32 (*get_vq_group)(struct vdpa_device *vdev, u16 idx);
>
> The index is u16 but the group is u32 ...
>
That's a very good point, but I got the reverse logic actually :). All
the vhost userland ioctls use an unsigned int (==u32) already, so
VDUSE should keep going with that. If any, get_vq_group (and similar)
kernel internal API should move to u32 to keep up with userland API.
[1] https://patchew.org/linux/20250606115012.1331551-1-eperezma@redhat.com/20250606115012.1331551-4-eperezma@redhat.com/
^ permalink raw reply [flat|nested] 27+ messages in thread* Re: [PATCH v2 3/7] vduse: add vq group support
2025-09-17 15:41 ` Eugenio Perez Martin
@ 2025-09-18 6:00 ` Jason Wang
2025-09-18 6:46 ` Eugenio Perez Martin
0 siblings, 1 reply; 27+ messages in thread
From: Jason Wang @ 2025-09-18 6:00 UTC (permalink / raw)
To: Eugenio Perez Martin
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Wed, Sep 17, 2025 at 11:42 PM Eugenio Perez Martin
<eperezma@redhat.com> wrote:
>
> On Wed, Sep 17, 2025 at 10:37 AM Jason Wang <jasowang@redhat.com> wrote:
> >
> > On Tue, Sep 16, 2025 at 9:08 PM Eugenio Pérez <eperezma@redhat.com> wrote:
> > >
> > > This allows sepparate the different virtqueues in groups that shares the
> > > same address space. Asking the VDUSE device for the groups of the vq at
> > > the beginning as they're needed for the DMA API.
> > >
> > > Allocating 3 vq groups as net is the device that need the most groups:
> > > * Dataplane (guest passthrough)
> > > * CVQ
> > > * Shadowed vrings.
> > >
> > > Future versions of the series can include dynamic allocation of the
> > > groups array so VDUSE can declare more groups.
> > >
> > > Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> > > ---
> > > v2:
> > > * Now the vq group is in vduse_vq_config struct instead of issuing one
> > > VDUSE message per vq.
> > >
> > > v1:
> > > * Fix: Remove BIT_ULL(VIRTIO_S_*), as _S_ is already the bit (Maxime)
> > >
> > > RFC v3:
> > > * Increase VDUSE_MAX_VQ_GROUPS to 0xffff (Jason). It was set to a lower
> > > value to reduce memory consumption, but vqs are already limited to
> > > that value and userspace VDUSE is able to allocate that many vqs.
> > > * Remove the descs vq group capability as it will not be used and we can
> > > add it on top.
> > > * Do not ask for vq groups in number of vq groups < 2.
> > > * Move the valid vq groups range check to vduse_validate_config.
> > >
> > > RFC v2:
> > > * Cache group information in kernel, as we need to provide the vq map
> > > tokens properly.
> > > * Add descs vq group to optimize SVQ forwarding and support indirect
> > > descriptors out of the box.
> > > ---
> > > drivers/vdpa/vdpa_user/vduse_dev.c | 37 ++++++++++++++++++++++++++----
> > > include/uapi/linux/vduse.h | 12 +++++++---
> > > 2 files changed, 41 insertions(+), 8 deletions(-)
> > >
> > > diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > index 2b6a8958ffe0..42f8807911d4 100644
> > > --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> > > +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > @@ -59,6 +59,7 @@ struct vduse_virtqueue {
> > > struct vdpa_vq_state state;
> > > bool ready;
> > > bool kicked;
> > > + u32 vq_group;
> > > spinlock_t kick_lock;
> > > spinlock_t irq_lock;
> > > struct eventfd_ctx *kickfd;
> > > @@ -115,6 +116,7 @@ struct vduse_dev {
> > > u8 status;
> > > u32 vq_num;
> > > u32 vq_align;
> > > + u32 ngroups;
> > > struct vduse_umem *umem;
> > > struct mutex mem_lock;
> > > unsigned int bounce_size;
> > > @@ -593,6 +595,13 @@ static int vduse_vdpa_set_vq_state(struct vdpa_device *vdpa, u16 idx,
> > > return 0;
> > > }
> > >
> > > +static u32 vduse_get_vq_group(struct vdpa_device *vdpa, u16 idx)
> > > +{
> > > + struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > > +
> > > + return dev->vqs[idx]->vq_group;
> >
> > I wonder if we should fail if VDUSE_VQ_SETUP is not set by userspace?
> >
>
> I'm kind of ok with implementing it, but I see it as redundant as if
> the VDUSE device does not call the vq max size will be 0 anyway.
Is this better to fail (I meant return int instead of a u32, probably
require changes in the vdpa core)?
>
> > > +}
> > > +
> > > static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
> > > struct vdpa_vq_state *state)
> > > {
> > > @@ -790,6 +799,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
> > > .set_vq_cb = vduse_vdpa_set_vq_cb,
> > > .set_vq_num = vduse_vdpa_set_vq_num,
> > > .get_vq_size = vduse_vdpa_get_vq_size,
> > > + .get_vq_group = vduse_get_vq_group,
> > > .set_vq_ready = vduse_vdpa_set_vq_ready,
> > > .get_vq_ready = vduse_vdpa_get_vq_ready,
> > > .set_vq_state = vduse_vdpa_set_vq_state,
> > > @@ -1253,12 +1263,21 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > > if (config.index >= dev->vq_num)
> > > break;
> > >
> > > - if (!is_mem_zero((const char *)config.reserved,
> > > - sizeof(config.reserved)))
> > > + if (dev->api_version < VDUSE_API_VERSION_1 && config.group)
> > > + break;
> > > +
> > > + if (dev->api_version >= VDUSE_API_VERSION_1 &&
> > > + config.group > dev->ngroups)
> > > + break;
> > > +
> > > + if (config.reserved1 ||
> > > + !is_mem_zero((const char *)config.reserved2,
> > > + sizeof(config.reserved2)))
> > > break;
> >
> > What's the reason for having this check? I mean if we don't check this
> > since the day0, it might be too late to do that.
> >
>
> It's been checked from day0, both reserved1 and reserved2 were
> included in the config.reserved before the patch, and the replaced
> lines already checked with is_mem_zero [1].
Ok. I see.
>
> > >
> > > index = array_index_nospec(config.index, dev->vq_num);
> > > dev->vqs[index]->num_max = config.max_size;
> > > + dev->vqs[index]->vq_group = config.group;
> > > ret = 0;
> > > break;
> > > }
> > > @@ -1738,12 +1757,19 @@ static bool features_is_valid(struct vduse_dev_config *config)
> > > return true;
> > > }
> > >
> > > -static bool vduse_validate_config(struct vduse_dev_config *config)
> > > +static bool vduse_validate_config(struct vduse_dev_config *config,
> > > + u64 api_version)
> > > {
> > > if (!is_mem_zero((const char *)config->reserved,
> > > sizeof(config->reserved)))
> > > return false;
> > >
> > > + if (api_version < VDUSE_API_VERSION_1 && config->ngroups)
> > > + return false;
> >
> > Better with ngroups > 1?
> >
>
> Same here. The kernel in the master branch already returns false if
> that position of "reserved" is 1, and we are changing the behavior if
> we don't include the 1 too.
>
> > > +
> > > + if (api_version >= VDUSE_API_VERSION_1 && config->ngroups > 0xffff)
> > > + return false;
> > > +
> > > if (config->vq_align > PAGE_SIZE)
> > > return false;
> > >
> > > @@ -1859,6 +1885,7 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> > > dev->device_features = config->features;
> > > dev->device_id = config->device_id;
> > > dev->vendor_id = config->vendor_id;
> > > + dev->ngroups = (dev->api_version < 1) ? 1 : (config->ngroups ?: 1);
> > > dev->name = kstrdup(config->name, GFP_KERNEL);
> > > if (!dev->name)
> > > goto err_str;
> > > @@ -1937,7 +1964,7 @@ static long vduse_ioctl(struct file *file, unsigned int cmd,
> > > break;
> > >
> > > ret = -EINVAL;
> > > - if (vduse_validate_config(&config) == false)
> > > + if (!vduse_validate_config(&config, control->api_version))
> > > break;
> > >
> > > buf = vmemdup_user(argp + size, config.config_size);
> > > @@ -2018,7 +2045,7 @@ static int vduse_dev_init_vdpa(struct vduse_dev *dev, const char *name)
> > >
> > > vdev = vdpa_alloc_device(struct vduse_vdpa, vdpa, dev->dev,
> > > &vduse_vdpa_config_ops, &vduse_map_ops,
> > > - 1, 1, name, true);
> > > + dev->ngroups, 1, name, true);
> > > if (IS_ERR(vdev))
> > > return PTR_ERR(vdev);
> > >
> > > diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h
> > > index ccb92a1efce0..a3d51cf6df3a 100644
> > > --- a/include/uapi/linux/vduse.h
> > > +++ b/include/uapi/linux/vduse.h
> > > @@ -31,6 +31,7 @@
> > > * @features: virtio features
> > > * @vq_num: the number of virtqueues
> > > * @vq_align: the allocation alignment of virtqueue's metadata
> > > + * @ngroups: number of vq groups that VDUSE device declares
> > > * @reserved: for future use, needs to be initialized to zero
> > > * @config_size: the size of the configuration space
> > > * @config: the buffer of the configuration space
> > > @@ -45,7 +46,8 @@ struct vduse_dev_config {
> > > __u64 features;
> > > __u32 vq_num;
> > > __u32 vq_align;
> > > - __u32 reserved[13];
> > > + __u32 ngroups; /* if VDUSE_API_VERSION >= 1 */
> > > + __u32 reserved[12];
> > > __u32 config_size;
> > > __u8 config[];
> > > };
> > > @@ -122,14 +124,18 @@ struct vduse_config_data {
> > > * struct vduse_vq_config - basic configuration of a virtqueue
> > > * @index: virtqueue index
> > > * @max_size: the max size of virtqueue
> > > - * @reserved: for future use, needs to be initialized to zero
> > > + * @reserved1: for future use, needs to be initialized to zero
> > > + * @group: virtqueue group
> > > + * @reserved2: for future use, needs to be initialized to zero
> > > *
> > > * Structure used by VDUSE_VQ_SETUP ioctl to setup a virtqueue.
> > > */
> > > struct vduse_vq_config {
> > > __u32 index;
> > > __u16 max_size;
> > > - __u16 reserved[13];
> > > + __u16 reserved1;
> > > + __u32 group;
> >
> > This makes me think if u16 is sufficient as I see:
> >
> > u32 (*get_vq_group)(struct vdpa_device *vdev, u16 idx);
> >
> > The index is u16 but the group is u32 ...
> >
>
> That's a very good point, but I got the reverse logic actually :). All
> the vhost userland ioctls use an unsigned int (==u32) already, so
> VDUSE should keep going with that. If any, get_vq_group (and similar)
> kernel internal API should move to u32 to keep up with userland API.
Ok, I see this:
struct vhost_vring_state {
unsigned int index;
unsigned int num;
};
It is just a little bit weird that we leave a hole.
Thanks
>
> [1] https://patchew.org/linux/20250606115012.1331551-1-eperezma@redhat.com/20250606115012.1331551-4-eperezma@redhat.com/
>
^ permalink raw reply [flat|nested] 27+ messages in thread* Re: [PATCH v2 3/7] vduse: add vq group support
2025-09-18 6:00 ` Jason Wang
@ 2025-09-18 6:46 ` Eugenio Perez Martin
0 siblings, 0 replies; 27+ messages in thread
From: Eugenio Perez Martin @ 2025-09-18 6:46 UTC (permalink / raw)
To: Jason Wang
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Thu, Sep 18, 2025 at 8:00 AM Jason Wang <jasowang@redhat.com> wrote:
>
> On Wed, Sep 17, 2025 at 11:42 PM Eugenio Perez Martin
> <eperezma@redhat.com> wrote:
> >
> > On Wed, Sep 17, 2025 at 10:37 AM Jason Wang <jasowang@redhat.com> wrote:
> > >
> > > On Tue, Sep 16, 2025 at 9:08 PM Eugenio Pérez <eperezma@redhat.com> wrote:
> > > >
> > > > This allows sepparate the different virtqueues in groups that shares the
> > > > same address space. Asking the VDUSE device for the groups of the vq at
> > > > the beginning as they're needed for the DMA API.
> > > >
> > > > Allocating 3 vq groups as net is the device that need the most groups:
> > > > * Dataplane (guest passthrough)
> > > > * CVQ
> > > > * Shadowed vrings.
> > > >
> > > > Future versions of the series can include dynamic allocation of the
> > > > groups array so VDUSE can declare more groups.
> > > >
> > > > Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> > > > ---
> > > > v2:
> > > > * Now the vq group is in vduse_vq_config struct instead of issuing one
> > > > VDUSE message per vq.
> > > >
> > > > v1:
> > > > * Fix: Remove BIT_ULL(VIRTIO_S_*), as _S_ is already the bit (Maxime)
> > > >
> > > > RFC v3:
> > > > * Increase VDUSE_MAX_VQ_GROUPS to 0xffff (Jason). It was set to a lower
> > > > value to reduce memory consumption, but vqs are already limited to
> > > > that value and userspace VDUSE is able to allocate that many vqs.
> > > > * Remove the descs vq group capability as it will not be used and we can
> > > > add it on top.
> > > > * Do not ask for vq groups in number of vq groups < 2.
> > > > * Move the valid vq groups range check to vduse_validate_config.
> > > >
> > > > RFC v2:
> > > > * Cache group information in kernel, as we need to provide the vq map
> > > > tokens properly.
> > > > * Add descs vq group to optimize SVQ forwarding and support indirect
> > > > descriptors out of the box.
> > > > ---
> > > > drivers/vdpa/vdpa_user/vduse_dev.c | 37 ++++++++++++++++++++++++++----
> > > > include/uapi/linux/vduse.h | 12 +++++++---
> > > > 2 files changed, 41 insertions(+), 8 deletions(-)
> > > >
> > > > diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > > index 2b6a8958ffe0..42f8807911d4 100644
> > > > --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> > > > +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > > @@ -59,6 +59,7 @@ struct vduse_virtqueue {
> > > > struct vdpa_vq_state state;
> > > > bool ready;
> > > > bool kicked;
> > > > + u32 vq_group;
> > > > spinlock_t kick_lock;
> > > > spinlock_t irq_lock;
> > > > struct eventfd_ctx *kickfd;
> > > > @@ -115,6 +116,7 @@ struct vduse_dev {
> > > > u8 status;
> > > > u32 vq_num;
> > > > u32 vq_align;
> > > > + u32 ngroups;
> > > > struct vduse_umem *umem;
> > > > struct mutex mem_lock;
> > > > unsigned int bounce_size;
> > > > @@ -593,6 +595,13 @@ static int vduse_vdpa_set_vq_state(struct vdpa_device *vdpa, u16 idx,
> > > > return 0;
> > > > }
> > > >
> > > > +static u32 vduse_get_vq_group(struct vdpa_device *vdpa, u16 idx)
> > > > +{
> > > > + struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > > > +
> > > > + return dev->vqs[idx]->vq_group;
> > >
> > > I wonder if we should fail if VDUSE_VQ_SETUP is not set by userspace?
> > >
> >
> > I'm kind of ok with implementing it, but I see it as redundant as if
> > the VDUSE device does not call the vq max size will be 0 anyway.
>
> Is this better to fail (I meant return int instead of a u32, probably
> require changes in the vdpa core)?
>
Sending a new series with this.
> >
> > > > +}
> > > > +
> > > > static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
> > > > struct vdpa_vq_state *state)
> > > > {
> > > > @@ -790,6 +799,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
> > > > .set_vq_cb = vduse_vdpa_set_vq_cb,
> > > > .set_vq_num = vduse_vdpa_set_vq_num,
> > > > .get_vq_size = vduse_vdpa_get_vq_size,
> > > > + .get_vq_group = vduse_get_vq_group,
> > > > .set_vq_ready = vduse_vdpa_set_vq_ready,
> > > > .get_vq_ready = vduse_vdpa_get_vq_ready,
> > > > .set_vq_state = vduse_vdpa_set_vq_state,
> > > > @@ -1253,12 +1263,21 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > > > if (config.index >= dev->vq_num)
> > > > break;
> > > >
> > > > - if (!is_mem_zero((const char *)config.reserved,
> > > > - sizeof(config.reserved)))
> > > > + if (dev->api_version < VDUSE_API_VERSION_1 && config.group)
> > > > + break;
> > > > +
> > > > + if (dev->api_version >= VDUSE_API_VERSION_1 &&
> > > > + config.group > dev->ngroups)
> > > > + break;
> > > > +
> > > > + if (config.reserved1 ||
> > > > + !is_mem_zero((const char *)config.reserved2,
> > > > + sizeof(config.reserved2)))
> > > > break;
> > >
> > > What's the reason for having this check? I mean if we don't check this
> > > since the day0, it might be too late to do that.
> > >
> >
> > It's been checked from day0, both reserved1 and reserved2 were
> > included in the config.reserved before the patch, and the replaced
> > lines already checked with is_mem_zero [1].
>
> Ok. I see.
>
> >
> > > >
> > > > index = array_index_nospec(config.index, dev->vq_num);
> > > > dev->vqs[index]->num_max = config.max_size;
> > > > + dev->vqs[index]->vq_group = config.group;
> > > > ret = 0;
> > > > break;
> > > > }
> > > > @@ -1738,12 +1757,19 @@ static bool features_is_valid(struct vduse_dev_config *config)
> > > > return true;
> > > > }
> > > >
> > > > -static bool vduse_validate_config(struct vduse_dev_config *config)
> > > > +static bool vduse_validate_config(struct vduse_dev_config *config,
> > > > + u64 api_version)
> > > > {
> > > > if (!is_mem_zero((const char *)config->reserved,
> > > > sizeof(config->reserved)))
> > > > return false;
> > > >
> > > > + if (api_version < VDUSE_API_VERSION_1 && config->ngroups)
> > > > + return false;
> > >
> > > Better with ngroups > 1?
> > >
> >
> > Same here. The kernel in the master branch already returns false if
> > that position of "reserved" is 1, and we are changing the behavior if
> > we don't include the 1 too.
> >
> > > > +
> > > > + if (api_version >= VDUSE_API_VERSION_1 && config->ngroups > 0xffff)
> > > > + return false;
> > > > +
> > > > if (config->vq_align > PAGE_SIZE)
> > > > return false;
> > > >
> > > > @@ -1859,6 +1885,7 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> > > > dev->device_features = config->features;
> > > > dev->device_id = config->device_id;
> > > > dev->vendor_id = config->vendor_id;
> > > > + dev->ngroups = (dev->api_version < 1) ? 1 : (config->ngroups ?: 1);
> > > > dev->name = kstrdup(config->name, GFP_KERNEL);
> > > > if (!dev->name)
> > > > goto err_str;
> > > > @@ -1937,7 +1964,7 @@ static long vduse_ioctl(struct file *file, unsigned int cmd,
> > > > break;
> > > >
> > > > ret = -EINVAL;
> > > > - if (vduse_validate_config(&config) == false)
> > > > + if (!vduse_validate_config(&config, control->api_version))
> > > > break;
> > > >
> > > > buf = vmemdup_user(argp + size, config.config_size);
> > > > @@ -2018,7 +2045,7 @@ static int vduse_dev_init_vdpa(struct vduse_dev *dev, const char *name)
> > > >
> > > > vdev = vdpa_alloc_device(struct vduse_vdpa, vdpa, dev->dev,
> > > > &vduse_vdpa_config_ops, &vduse_map_ops,
> > > > - 1, 1, name, true);
> > > > + dev->ngroups, 1, name, true);
> > > > if (IS_ERR(vdev))
> > > > return PTR_ERR(vdev);
> > > >
> > > > diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h
> > > > index ccb92a1efce0..a3d51cf6df3a 100644
> > > > --- a/include/uapi/linux/vduse.h
> > > > +++ b/include/uapi/linux/vduse.h
> > > > @@ -31,6 +31,7 @@
> > > > * @features: virtio features
> > > > * @vq_num: the number of virtqueues
> > > > * @vq_align: the allocation alignment of virtqueue's metadata
> > > > + * @ngroups: number of vq groups that VDUSE device declares
> > > > * @reserved: for future use, needs to be initialized to zero
> > > > * @config_size: the size of the configuration space
> > > > * @config: the buffer of the configuration space
> > > > @@ -45,7 +46,8 @@ struct vduse_dev_config {
> > > > __u64 features;
> > > > __u32 vq_num;
> > > > __u32 vq_align;
> > > > - __u32 reserved[13];
> > > > + __u32 ngroups; /* if VDUSE_API_VERSION >= 1 */
> > > > + __u32 reserved[12];
> > > > __u32 config_size;
> > > > __u8 config[];
> > > > };
> > > > @@ -122,14 +124,18 @@ struct vduse_config_data {
> > > > * struct vduse_vq_config - basic configuration of a virtqueue
> > > > * @index: virtqueue index
> > > > * @max_size: the max size of virtqueue
> > > > - * @reserved: for future use, needs to be initialized to zero
> > > > + * @reserved1: for future use, needs to be initialized to zero
> > > > + * @group: virtqueue group
> > > > + * @reserved2: for future use, needs to be initialized to zero
> > > > *
> > > > * Structure used by VDUSE_VQ_SETUP ioctl to setup a virtqueue.
> > > > */
> > > > struct vduse_vq_config {
> > > > __u32 index;
> > > > __u16 max_size;
> > > > - __u16 reserved[13];
> > > > + __u16 reserved1;
> > > > + __u32 group;
> > >
> > > This makes me think if u16 is sufficient as I see:
> > >
> > > u32 (*get_vq_group)(struct vdpa_device *vdev, u16 idx);
> > >
> > > The index is u16 but the group is u32 ...
> > >
> >
> > That's a very good point, but I got the reverse logic actually :). All
> > the vhost userland ioctls use an unsigned int (==u32) already, so
> > VDUSE should keep going with that. If any, get_vq_group (and similar)
> > kernel internal API should move to u32 to keep up with userland API.
>
> Ok, I see this:
>
> struct vhost_vring_state {
> unsigned int index;
> unsigned int num;
> };
>
> It is just a little bit weird that we leave a hole.
>
I don't like the hole in the struct either, but I find it cosmetic
compared to the drawbacks of the alternatives. I find it even more
cosmetic seeing that 6.17 is approaching :).
^ permalink raw reply [flat|nested] 27+ messages in thread
* [PATCH v2 4/7] vduse: return internal vq group struct as map token
2025-09-16 13:08 [PATCH v2 0/7] Add multiple address spaces support to VDUSE Eugenio Pérez
` (2 preceding siblings ...)
2025-09-16 13:08 ` [PATCH v2 3/7] vduse: add vq group support Eugenio Pérez
@ 2025-09-16 13:08 ` Eugenio Pérez
2025-09-17 8:36 ` Jason Wang
2025-09-16 13:08 ` [PATCH v2 5/7] vduse: create vduse_as to make it an array Eugenio Pérez
` (2 subsequent siblings)
6 siblings, 1 reply; 27+ messages in thread
From: Eugenio Pérez @ 2025-09-16 13:08 UTC (permalink / raw)
To: Michael S . Tsirkin
Cc: Stefano Garzarella, jasowang, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization, Eugenio Pérez
Return the internal struct that represents the vq group as virtqueue map
token, instead of the device. This allows the map functions to access
the information per group.
At this moment all the virtqueues share the same vq group, that only
can point to ASID 0. This change prepares the infrastructure for actual
per-group address space handling
Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
---
RFC v3:
* Make the vq groups a dynamic array to support an arbitrary number of
them.
---
drivers/vdpa/vdpa_user/vduse_dev.c | 52 ++++++++++++++++++++++++------
include/linux/virtio.h | 6 ++--
2 files changed, 46 insertions(+), 12 deletions(-)
diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
index 42f8807911d4..9c12ae72abc2 100644
--- a/drivers/vdpa/vdpa_user/vduse_dev.c
+++ b/drivers/vdpa/vdpa_user/vduse_dev.c
@@ -23,6 +23,7 @@
#include <linux/uio.h>
#include <linux/vdpa.h>
#include <linux/nospec.h>
+#include <linux/virtio.h>
#include <linux/vmalloc.h>
#include <linux/sched/mm.h>
#include <uapi/linux/vduse.h>
@@ -85,6 +86,10 @@ struct vduse_umem {
struct mm_struct *mm;
};
+struct vduse_vq_group_int {
+ struct vduse_dev *dev;
+};
+
struct vduse_dev {
struct vduse_vdpa *vdev;
struct device *dev;
@@ -118,6 +123,7 @@ struct vduse_dev {
u32 vq_align;
u32 ngroups;
struct vduse_umem *umem;
+ struct vduse_vq_group_int *groups;
struct mutex mem_lock;
unsigned int bounce_size;
rwlock_t domain_lock;
@@ -602,6 +608,15 @@ static u32 vduse_get_vq_group(struct vdpa_device *vdpa, u16 idx)
return dev->vqs[idx]->vq_group;
}
+static union virtio_map vduse_get_vq_map(struct vdpa_device *vdpa, u16 idx)
+{
+ struct vduse_dev *dev = vdpa_to_vduse(vdpa);
+ u32 vq_group = dev->vqs[idx]->vq_group;
+ union virtio_map ret = { .group = &dev->groups[vq_group] };
+
+ return ret;
+}
+
static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
struct vdpa_vq_state *state)
{
@@ -822,6 +837,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
.get_vq_affinity = vduse_vdpa_get_vq_affinity,
.reset = vduse_vdpa_reset,
.set_map = vduse_vdpa_set_map,
+ .get_vq_map = vduse_get_vq_map,
.free = vduse_vdpa_free,
};
@@ -829,7 +845,8 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
dma_addr_t dma_addr, size_t size,
enum dma_data_direction dir)
{
- struct vduse_iova_domain *domain = token.iova_domain;
+ struct vduse_dev *vdev = token.group->dev;
+ struct vduse_iova_domain *domain = vdev->domain;
vduse_domain_sync_single_for_device(domain, dma_addr, size, dir);
}
@@ -838,7 +855,8 @@ static void vduse_dev_sync_single_for_cpu(union virtio_map token,
dma_addr_t dma_addr, size_t size,
enum dma_data_direction dir)
{
- struct vduse_iova_domain *domain = token.iova_domain;
+ struct vduse_dev *vdev = token.group->dev;
+ struct vduse_iova_domain *domain = vdev->domain;
vduse_domain_sync_single_for_cpu(domain, dma_addr, size, dir);
}
@@ -848,7 +866,8 @@ static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
enum dma_data_direction dir,
unsigned long attrs)
{
- struct vduse_iova_domain *domain = token.iova_domain;
+ struct vduse_dev *vdev = token.group->dev;
+ struct vduse_iova_domain *domain = vdev->domain;
return vduse_domain_map_page(domain, page, offset, size, dir, attrs);
}
@@ -857,7 +876,8 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
size_t size, enum dma_data_direction dir,
unsigned long attrs)
{
- struct vduse_iova_domain *domain = token.iova_domain;
+ struct vduse_dev *vdev = token.group->dev;
+ struct vduse_iova_domain *domain = vdev->domain;
return vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
}
@@ -865,7 +885,8 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
static void *vduse_dev_alloc_coherent(union virtio_map token, size_t size,
dma_addr_t *dma_addr, gfp_t flag)
{
- struct vduse_iova_domain *domain = token.iova_domain;
+ struct vduse_dev *vdev = token.group->dev;
+ struct vduse_iova_domain *domain = vdev->domain;
unsigned long iova;
void *addr;
@@ -884,14 +905,16 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
void *vaddr, dma_addr_t dma_addr,
unsigned long attrs)
{
- struct vduse_iova_domain *domain = token.iova_domain;
+ struct vduse_dev *vdev = token.group->dev;
+ struct vduse_iova_domain *domain = vdev->domain;
vduse_domain_free_coherent(domain, size, vaddr, dma_addr, attrs);
}
static bool vduse_dev_need_sync(union virtio_map token, dma_addr_t dma_addr)
{
- struct vduse_iova_domain *domain = token.iova_domain;
+ struct vduse_dev *vdev = token.group->dev;
+ struct vduse_iova_domain *domain = vdev->domain;
return dma_addr < domain->bounce_size;
}
@@ -905,7 +928,8 @@ static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
static size_t vduse_dev_max_mapping_size(union virtio_map token)
{
- struct vduse_iova_domain *domain = token.iova_domain;
+ struct vduse_dev *vdev = token.group->dev;
+ struct vduse_iova_domain *domain = vdev->domain;
return domain->bounce_size;
}
@@ -1720,6 +1744,7 @@ static int vduse_destroy_dev(char *name)
if (dev->domain)
vduse_domain_destroy(dev->domain);
kfree(dev->name);
+ kfree(dev->groups);
vduse_dev_destroy(dev);
module_put(THIS_MODULE);
@@ -1885,7 +1910,15 @@ static int vduse_create_dev(struct vduse_dev_config *config,
dev->device_features = config->features;
dev->device_id = config->device_id;
dev->vendor_id = config->vendor_id;
+
dev->ngroups = (dev->api_version < 1) ? 1 : (config->ngroups ?: 1);
+ dev->groups = kcalloc(dev->ngroups, sizeof(dev->groups[0]),
+ GFP_KERNEL);
+ if (!dev->groups)
+ goto err_vq_groups;
+ for (u32 i = 0; i < dev->ngroups; ++i)
+ dev->groups[i].dev = dev;
+
dev->name = kstrdup(config->name, GFP_KERNEL);
if (!dev->name)
goto err_str;
@@ -1922,6 +1955,8 @@ static int vduse_create_dev(struct vduse_dev_config *config,
err_idr:
kfree(dev->name);
err_str:
+ kfree(dev->groups);
+err_vq_groups:
vduse_dev_destroy(dev);
err:
return ret;
@@ -2083,7 +2118,6 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
return -ENOMEM;
}
- dev->vdev->vdpa.vmap.iova_domain = dev->domain;
ret = _vdpa_register_device(&dev->vdev->vdpa, dev->vq_num);
if (ret) {
put_device(&dev->vdev->vdpa.dev);
diff --git a/include/linux/virtio.h b/include/linux/virtio.h
index 96c66126c074..5f8db75f7833 100644
--- a/include/linux/virtio.h
+++ b/include/linux/virtio.h
@@ -41,13 +41,13 @@ struct virtqueue {
void *priv;
};
-struct vduse_iova_domain;
+struct vduse_vq_group_int;
union virtio_map {
/* Device that performs DMA */
struct device *dma_dev;
- /* VDUSE specific mapping data */
- struct vduse_iova_domain *iova_domain;
+ /* VDUSE specific virtqueue group for doing map */
+ struct vduse_vq_group_int *group;
};
int virtqueue_add_outbuf(struct virtqueue *vq,
--
2.51.0
^ permalink raw reply related [flat|nested] 27+ messages in thread* Re: [PATCH v2 4/7] vduse: return internal vq group struct as map token
2025-09-16 13:08 ` [PATCH v2 4/7] vduse: return internal vq group struct as map token Eugenio Pérez
@ 2025-09-17 8:36 ` Jason Wang
2025-09-17 16:16 ` Eugenio Perez Martin
0 siblings, 1 reply; 27+ messages in thread
From: Jason Wang @ 2025-09-17 8:36 UTC (permalink / raw)
To: Eugenio Pérez
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
>
> Return the internal struct that represents the vq group as virtqueue map
> token, instead of the device. This allows the map functions to access
> the information per group.
>
> At this moment all the virtqueues share the same vq group, that only
> can point to ASID 0. This change prepares the infrastructure for actual
> per-group address space handling
>
> Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> ---
> RFC v3:
> * Make the vq groups a dynamic array to support an arbitrary number of
> them.
> ---
> drivers/vdpa/vdpa_user/vduse_dev.c | 52 ++++++++++++++++++++++++------
> include/linux/virtio.h | 6 ++--
> 2 files changed, 46 insertions(+), 12 deletions(-)
>
> diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> index 42f8807911d4..9c12ae72abc2 100644
> --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> @@ -23,6 +23,7 @@
> #include <linux/uio.h>
> #include <linux/vdpa.h>
> #include <linux/nospec.h>
> +#include <linux/virtio.h>
> #include <linux/vmalloc.h>
> #include <linux/sched/mm.h>
> #include <uapi/linux/vduse.h>
> @@ -85,6 +86,10 @@ struct vduse_umem {
> struct mm_struct *mm;
> };
>
> +struct vduse_vq_group_int {
> + struct vduse_dev *dev;
> +};
I remember we had some discussion over this, and the conclusion is to
have a better name.
Maybe just vduse_vq_group?
And to be conceptually correct, we need to use iova_domain here
instead of the vduse_dev. More below.
> +
> struct vduse_dev {
> struct vduse_vdpa *vdev;
> struct device *dev;
> @@ -118,6 +123,7 @@ struct vduse_dev {
> u32 vq_align;
> u32 ngroups;
> struct vduse_umem *umem;
> + struct vduse_vq_group_int *groups;
> struct mutex mem_lock;
> unsigned int bounce_size;
> rwlock_t domain_lock;
> @@ -602,6 +608,15 @@ static u32 vduse_get_vq_group(struct vdpa_device *vdpa, u16 idx)
> return dev->vqs[idx]->vq_group;
> }
>
> +static union virtio_map vduse_get_vq_map(struct vdpa_device *vdpa, u16 idx)
> +{
> + struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> + u32 vq_group = dev->vqs[idx]->vq_group;
> + union virtio_map ret = { .group = &dev->groups[vq_group] };
> +
> + return ret;
> +}
> +
> static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
> struct vdpa_vq_state *state)
> {
> @@ -822,6 +837,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
> .get_vq_affinity = vduse_vdpa_get_vq_affinity,
> .reset = vduse_vdpa_reset,
> .set_map = vduse_vdpa_set_map,
> + .get_vq_map = vduse_get_vq_map,
> .free = vduse_vdpa_free,
> };
>
> @@ -829,7 +845,8 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
> dma_addr_t dma_addr, size_t size,
> enum dma_data_direction dir)
> {
> - struct vduse_iova_domain *domain = token.iova_domain;
> + struct vduse_dev *vdev = token.group->dev;
> + struct vduse_iova_domain *domain = vdev->domain;
If we really want to do this, we need to move the iova_domian into the group.
>
> vduse_domain_sync_single_for_device(domain, dma_addr, size, dir);
> }
> @@ -838,7 +855,8 @@ static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> dma_addr_t dma_addr, size_t size,
> enum dma_data_direction dir)
> {
> - struct vduse_iova_domain *domain = token.iova_domain;
> + struct vduse_dev *vdev = token.group->dev;
> + struct vduse_iova_domain *domain = vdev->domain;
>
> vduse_domain_sync_single_for_cpu(domain, dma_addr, size, dir);
> }
> @@ -848,7 +866,8 @@ static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> enum dma_data_direction dir,
> unsigned long attrs)
> {
> - struct vduse_iova_domain *domain = token.iova_domain;
> + struct vduse_dev *vdev = token.group->dev;
> + struct vduse_iova_domain *domain = vdev->domain;
>
> return vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> }
> @@ -857,7 +876,8 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> size_t size, enum dma_data_direction dir,
> unsigned long attrs)
> {
> - struct vduse_iova_domain *domain = token.iova_domain;
> + struct vduse_dev *vdev = token.group->dev;
> + struct vduse_iova_domain *domain = vdev->domain;
>
> return vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> }
> @@ -865,7 +885,8 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> static void *vduse_dev_alloc_coherent(union virtio_map token, size_t size,
> dma_addr_t *dma_addr, gfp_t flag)
> {
> - struct vduse_iova_domain *domain = token.iova_domain;
> + struct vduse_dev *vdev = token.group->dev;
> + struct vduse_iova_domain *domain = vdev->domain;
> unsigned long iova;
> void *addr;
>
> @@ -884,14 +905,16 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
> void *vaddr, dma_addr_t dma_addr,
> unsigned long attrs)
> {
> - struct vduse_iova_domain *domain = token.iova_domain;
> + struct vduse_dev *vdev = token.group->dev;
> + struct vduse_iova_domain *domain = vdev->domain;
>
> vduse_domain_free_coherent(domain, size, vaddr, dma_addr, attrs);
> }
>
> static bool vduse_dev_need_sync(union virtio_map token, dma_addr_t dma_addr)
> {
> - struct vduse_iova_domain *domain = token.iova_domain;
> + struct vduse_dev *vdev = token.group->dev;
> + struct vduse_iova_domain *domain = vdev->domain;
>
> return dma_addr < domain->bounce_size;
> }
> @@ -905,7 +928,8 @@ static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
>
> static size_t vduse_dev_max_mapping_size(union virtio_map token)
> {
> - struct vduse_iova_domain *domain = token.iova_domain;
> + struct vduse_dev *vdev = token.group->dev;
> + struct vduse_iova_domain *domain = vdev->domain;
>
> return domain->bounce_size;
> }
> @@ -1720,6 +1744,7 @@ static int vduse_destroy_dev(char *name)
> if (dev->domain)
> vduse_domain_destroy(dev->domain);
> kfree(dev->name);
> + kfree(dev->groups);
> vduse_dev_destroy(dev);
> module_put(THIS_MODULE);
>
> @@ -1885,7 +1910,15 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> dev->device_features = config->features;
> dev->device_id = config->device_id;
> dev->vendor_id = config->vendor_id;
> +
> dev->ngroups = (dev->api_version < 1) ? 1 : (config->ngroups ?: 1);
> + dev->groups = kcalloc(dev->ngroups, sizeof(dev->groups[0]),
> + GFP_KERNEL);
> + if (!dev->groups)
> + goto err_vq_groups;
> + for (u32 i = 0; i < dev->ngroups; ++i)
> + dev->groups[i].dev = dev;
> +
> dev->name = kstrdup(config->name, GFP_KERNEL);
> if (!dev->name)
> goto err_str;
> @@ -1922,6 +1955,8 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> err_idr:
> kfree(dev->name);
> err_str:
> + kfree(dev->groups);
> +err_vq_groups:
> vduse_dev_destroy(dev);
> err:
> return ret;
> @@ -2083,7 +2118,6 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> return -ENOMEM;
> }
>
> - dev->vdev->vdpa.vmap.iova_domain = dev->domain;
> ret = _vdpa_register_device(&dev->vdev->vdpa, dev->vq_num);
> if (ret) {
> put_device(&dev->vdev->vdpa.dev);
> diff --git a/include/linux/virtio.h b/include/linux/virtio.h
> index 96c66126c074..5f8db75f7833 100644
> --- a/include/linux/virtio.h
> +++ b/include/linux/virtio.h
> @@ -41,13 +41,13 @@ struct virtqueue {
> void *priv;
> };
>
> -struct vduse_iova_domain;
> +struct vduse_vq_group_int;
>
> union virtio_map {
> /* Device that performs DMA */
> struct device *dma_dev;
> - /* VDUSE specific mapping data */
> - struct vduse_iova_domain *iova_domain;
> + /* VDUSE specific virtqueue group for doing map */
> + struct vduse_vq_group_int *group;
> };
>
> int virtqueue_add_outbuf(struct virtqueue *vq,
> --
> 2.51.0
>
Thanks
^ permalink raw reply [flat|nested] 27+ messages in thread* Re: [PATCH v2 4/7] vduse: return internal vq group struct as map token
2025-09-17 8:36 ` Jason Wang
@ 2025-09-17 16:16 ` Eugenio Perez Martin
2025-09-18 6:01 ` Jason Wang
0 siblings, 1 reply; 27+ messages in thread
From: Eugenio Perez Martin @ 2025-09-17 16:16 UTC (permalink / raw)
To: Jason Wang
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Wed, Sep 17, 2025 at 10:37 AM Jason Wang <jasowang@redhat.com> wrote:
>
> On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
> >
> > Return the internal struct that represents the vq group as virtqueue map
> > token, instead of the device. This allows the map functions to access
> > the information per group.
> >
> > At this moment all the virtqueues share the same vq group, that only
> > can point to ASID 0. This change prepares the infrastructure for actual
> > per-group address space handling
> >
> > Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> > ---
> > RFC v3:
> > * Make the vq groups a dynamic array to support an arbitrary number of
> > them.
> > ---
> > drivers/vdpa/vdpa_user/vduse_dev.c | 52 ++++++++++++++++++++++++------
> > include/linux/virtio.h | 6 ++--
> > 2 files changed, 46 insertions(+), 12 deletions(-)
> >
> > diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> > index 42f8807911d4..9c12ae72abc2 100644
> > --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> > +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> > @@ -23,6 +23,7 @@
> > #include <linux/uio.h>
> > #include <linux/vdpa.h>
> > #include <linux/nospec.h>
> > +#include <linux/virtio.h>
> > #include <linux/vmalloc.h>
> > #include <linux/sched/mm.h>
> > #include <uapi/linux/vduse.h>
> > @@ -85,6 +86,10 @@ struct vduse_umem {
> > struct mm_struct *mm;
> > };
> >
> > +struct vduse_vq_group_int {
> > + struct vduse_dev *dev;
> > +};
>
> I remember we had some discussion over this, and the conclusion is to
> have a better name.
>
> Maybe just vduse_vq_group?
>
Good catch, I also hate the _int suffix :). vduse_vq_group was used in
the vduse uapi in previous series, but now there is no reason for it.
Replacing it, thanks!
> And to be conceptually correct, we need to use iova_domain here
> instead of the vduse_dev. More below.
>
> > +
> > struct vduse_dev {
> > struct vduse_vdpa *vdev;
> > struct device *dev;
> > @@ -118,6 +123,7 @@ struct vduse_dev {
> > u32 vq_align;
> > u32 ngroups;
> > struct vduse_umem *umem;
> > + struct vduse_vq_group_int *groups;
> > struct mutex mem_lock;
> > unsigned int bounce_size;
> > rwlock_t domain_lock;
> > @@ -602,6 +608,15 @@ static u32 vduse_get_vq_group(struct vdpa_device *vdpa, u16 idx)
> > return dev->vqs[idx]->vq_group;
> > }
> >
> > +static union virtio_map vduse_get_vq_map(struct vdpa_device *vdpa, u16 idx)
> > +{
> > + struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > + u32 vq_group = dev->vqs[idx]->vq_group;
> > + union virtio_map ret = { .group = &dev->groups[vq_group] };
> > +
> > + return ret;
> > +}
> > +
> > static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
> > struct vdpa_vq_state *state)
> > {
> > @@ -822,6 +837,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
> > .get_vq_affinity = vduse_vdpa_get_vq_affinity,
> > .reset = vduse_vdpa_reset,
> > .set_map = vduse_vdpa_set_map,
> > + .get_vq_map = vduse_get_vq_map,
> > .free = vduse_vdpa_free,
> > };
> >
> > @@ -829,7 +845,8 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
> > dma_addr_t dma_addr, size_t size,
> > enum dma_data_direction dir)
> > {
> > - struct vduse_iova_domain *domain = token.iova_domain;
> > + struct vduse_dev *vdev = token.group->dev;
> > + struct vduse_iova_domain *domain = vdev->domain;
>
> If we really want to do this, we need to move the iova_domian into the group.
>
It's done in patches on top to make each patch smaller. This patch is
focused on just changing the type of the union. Would you prefer me to
reorder the patches so that part is done earlier?
> >
> > vduse_domain_sync_single_for_device(domain, dma_addr, size, dir);
> > }
> > @@ -838,7 +855,8 @@ static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> > dma_addr_t dma_addr, size_t size,
> > enum dma_data_direction dir)
> > {
> > - struct vduse_iova_domain *domain = token.iova_domain;
> > + struct vduse_dev *vdev = token.group->dev;
> > + struct vduse_iova_domain *domain = vdev->domain;
> >
> > vduse_domain_sync_single_for_cpu(domain, dma_addr, size, dir);
> > }
> > @@ -848,7 +866,8 @@ static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> > enum dma_data_direction dir,
> > unsigned long attrs)
> > {
> > - struct vduse_iova_domain *domain = token.iova_domain;
> > + struct vduse_dev *vdev = token.group->dev;
> > + struct vduse_iova_domain *domain = vdev->domain;
> >
> > return vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> > }
> > @@ -857,7 +876,8 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> > size_t size, enum dma_data_direction dir,
> > unsigned long attrs)
> > {
> > - struct vduse_iova_domain *domain = token.iova_domain;
> > + struct vduse_dev *vdev = token.group->dev;
> > + struct vduse_iova_domain *domain = vdev->domain;
> >
> > return vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> > }
> > @@ -865,7 +885,8 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> > static void *vduse_dev_alloc_coherent(union virtio_map token, size_t size,
> > dma_addr_t *dma_addr, gfp_t flag)
> > {
> > - struct vduse_iova_domain *domain = token.iova_domain;
> > + struct vduse_dev *vdev = token.group->dev;
> > + struct vduse_iova_domain *domain = vdev->domain;
> > unsigned long iova;
> > void *addr;
> >
> > @@ -884,14 +905,16 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
> > void *vaddr, dma_addr_t dma_addr,
> > unsigned long attrs)
> > {
> > - struct vduse_iova_domain *domain = token.iova_domain;
> > + struct vduse_dev *vdev = token.group->dev;
> > + struct vduse_iova_domain *domain = vdev->domain;
> >
> > vduse_domain_free_coherent(domain, size, vaddr, dma_addr, attrs);
> > }
> >
> > static bool vduse_dev_need_sync(union virtio_map token, dma_addr_t dma_addr)
> > {
> > - struct vduse_iova_domain *domain = token.iova_domain;
> > + struct vduse_dev *vdev = token.group->dev;
> > + struct vduse_iova_domain *domain = vdev->domain;
> >
> > return dma_addr < domain->bounce_size;
> > }
> > @@ -905,7 +928,8 @@ static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> >
> > static size_t vduse_dev_max_mapping_size(union virtio_map token)
> > {
> > - struct vduse_iova_domain *domain = token.iova_domain;
> > + struct vduse_dev *vdev = token.group->dev;
> > + struct vduse_iova_domain *domain = vdev->domain;
> >
> > return domain->bounce_size;
> > }
> > @@ -1720,6 +1744,7 @@ static int vduse_destroy_dev(char *name)
> > if (dev->domain)
> > vduse_domain_destroy(dev->domain);
> > kfree(dev->name);
> > + kfree(dev->groups);
> > vduse_dev_destroy(dev);
> > module_put(THIS_MODULE);
> >
> > @@ -1885,7 +1910,15 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> > dev->device_features = config->features;
> > dev->device_id = config->device_id;
> > dev->vendor_id = config->vendor_id;
> > +
> > dev->ngroups = (dev->api_version < 1) ? 1 : (config->ngroups ?: 1);
> > + dev->groups = kcalloc(dev->ngroups, sizeof(dev->groups[0]),
> > + GFP_KERNEL);
> > + if (!dev->groups)
> > + goto err_vq_groups;
> > + for (u32 i = 0; i < dev->ngroups; ++i)
> > + dev->groups[i].dev = dev;
> > +
> > dev->name = kstrdup(config->name, GFP_KERNEL);
> > if (!dev->name)
> > goto err_str;
> > @@ -1922,6 +1955,8 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> > err_idr:
> > kfree(dev->name);
> > err_str:
> > + kfree(dev->groups);
> > +err_vq_groups:
> > vduse_dev_destroy(dev);
> > err:
> > return ret;
> > @@ -2083,7 +2118,6 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> > return -ENOMEM;
> > }
> >
> > - dev->vdev->vdpa.vmap.iova_domain = dev->domain;
> > ret = _vdpa_register_device(&dev->vdev->vdpa, dev->vq_num);
> > if (ret) {
> > put_device(&dev->vdev->vdpa.dev);
> > diff --git a/include/linux/virtio.h b/include/linux/virtio.h
> > index 96c66126c074..5f8db75f7833 100644
> > --- a/include/linux/virtio.h
> > +++ b/include/linux/virtio.h
> > @@ -41,13 +41,13 @@ struct virtqueue {
> > void *priv;
> > };
> >
> > -struct vduse_iova_domain;
> > +struct vduse_vq_group_int;
> >
> > union virtio_map {
> > /* Device that performs DMA */
> > struct device *dma_dev;
> > - /* VDUSE specific mapping data */
> > - struct vduse_iova_domain *iova_domain;
> > + /* VDUSE specific virtqueue group for doing map */
> > + struct vduse_vq_group_int *group;
> > };
> >
> > int virtqueue_add_outbuf(struct virtqueue *vq,
> > --
> > 2.51.0
> >
>
> Thanks
>
^ permalink raw reply [flat|nested] 27+ messages in thread* Re: [PATCH v2 4/7] vduse: return internal vq group struct as map token
2025-09-17 16:16 ` Eugenio Perez Martin
@ 2025-09-18 6:01 ` Jason Wang
0 siblings, 0 replies; 27+ messages in thread
From: Jason Wang @ 2025-09-18 6:01 UTC (permalink / raw)
To: Eugenio Perez Martin
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Thu, Sep 18, 2025 at 12:17 AM Eugenio Perez Martin
<eperezma@redhat.com> wrote:
>
> On Wed, Sep 17, 2025 at 10:37 AM Jason Wang <jasowang@redhat.com> wrote:
> >
> > On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
> > >
> > > Return the internal struct that represents the vq group as virtqueue map
> > > token, instead of the device. This allows the map functions to access
> > > the information per group.
> > >
> > > At this moment all the virtqueues share the same vq group, that only
> > > can point to ASID 0. This change prepares the infrastructure for actual
> > > per-group address space handling
> > >
> > > Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> > > ---
> > > RFC v3:
> > > * Make the vq groups a dynamic array to support an arbitrary number of
> > > them.
> > > ---
> > > drivers/vdpa/vdpa_user/vduse_dev.c | 52 ++++++++++++++++++++++++------
> > > include/linux/virtio.h | 6 ++--
> > > 2 files changed, 46 insertions(+), 12 deletions(-)
> > >
> > > diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > index 42f8807911d4..9c12ae72abc2 100644
> > > --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> > > +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > @@ -23,6 +23,7 @@
> > > #include <linux/uio.h>
> > > #include <linux/vdpa.h>
> > > #include <linux/nospec.h>
> > > +#include <linux/virtio.h>
> > > #include <linux/vmalloc.h>
> > > #include <linux/sched/mm.h>
> > > #include <uapi/linux/vduse.h>
> > > @@ -85,6 +86,10 @@ struct vduse_umem {
> > > struct mm_struct *mm;
> > > };
> > >
> > > +struct vduse_vq_group_int {
> > > + struct vduse_dev *dev;
> > > +};
> >
> > I remember we had some discussion over this, and the conclusion is to
> > have a better name.
> >
> > Maybe just vduse_vq_group?
> >
>
> Good catch, I also hate the _int suffix :). vduse_vq_group was used in
> the vduse uapi in previous series, but now there is no reason for it.
> Replacing it, thanks!
>
> > And to be conceptually correct, we need to use iova_domain here
> > instead of the vduse_dev. More below.
> >
> > > +
> > > struct vduse_dev {
> > > struct vduse_vdpa *vdev;
> > > struct device *dev;
> > > @@ -118,6 +123,7 @@ struct vduse_dev {
> > > u32 vq_align;
> > > u32 ngroups;
> > > struct vduse_umem *umem;
> > > + struct vduse_vq_group_int *groups;
> > > struct mutex mem_lock;
> > > unsigned int bounce_size;
> > > rwlock_t domain_lock;
> > > @@ -602,6 +608,15 @@ static u32 vduse_get_vq_group(struct vdpa_device *vdpa, u16 idx)
> > > return dev->vqs[idx]->vq_group;
> > > }
> > >
> > > +static union virtio_map vduse_get_vq_map(struct vdpa_device *vdpa, u16 idx)
> > > +{
> > > + struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > > + u32 vq_group = dev->vqs[idx]->vq_group;
> > > + union virtio_map ret = { .group = &dev->groups[vq_group] };
> > > +
> > > + return ret;
> > > +}
> > > +
> > > static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
> > > struct vdpa_vq_state *state)
> > > {
> > > @@ -822,6 +837,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
> > > .get_vq_affinity = vduse_vdpa_get_vq_affinity,
> > > .reset = vduse_vdpa_reset,
> > > .set_map = vduse_vdpa_set_map,
> > > + .get_vq_map = vduse_get_vq_map,
> > > .free = vduse_vdpa_free,
> > > };
> > >
> > > @@ -829,7 +845,8 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
> > > dma_addr_t dma_addr, size_t size,
> > > enum dma_data_direction dir)
> > > {
> > > - struct vduse_iova_domain *domain = token.iova_domain;
> > > + struct vduse_dev *vdev = token.group->dev;
> > > + struct vduse_iova_domain *domain = vdev->domain;
> >
> > If we really want to do this, we need to move the iova_domian into the group.
> >
>
> It's done in patches on top to make each patch smaller. This patch is
> focused on just changing the type of the union. Would you prefer me to
> reorder the patches so that part is done earlier?
I think it would be better for logical completeness.
Thanks
^ permalink raw reply [flat|nested] 27+ messages in thread
* [PATCH v2 5/7] vduse: create vduse_as to make it an array
2025-09-16 13:08 [PATCH v2 0/7] Add multiple address spaces support to VDUSE Eugenio Pérez
` (3 preceding siblings ...)
2025-09-16 13:08 ` [PATCH v2 4/7] vduse: return internal vq group struct as map token Eugenio Pérez
@ 2025-09-16 13:08 ` Eugenio Pérez
2025-09-17 8:46 ` Jason Wang
2025-09-16 13:08 ` [PATCH v2 6/7] vduse: add vq group asid support Eugenio Pérez
2025-09-16 13:08 ` [PATCH v2 7/7] vduse: bump version number Eugenio Pérez
6 siblings, 1 reply; 27+ messages in thread
From: Eugenio Pérez @ 2025-09-16 13:08 UTC (permalink / raw)
To: Michael S . Tsirkin
Cc: Stefano Garzarella, jasowang, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization, Eugenio Pérez
This is a first step so we can make more than one different address
spaces. No change on the colde flow intended.
Acked-by: Jason Wang <jasowang@redhat.com>
Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
---
drivers/vdpa/vdpa_user/vduse_dev.c | 114 +++++++++++++++--------------
1 file changed, 59 insertions(+), 55 deletions(-)
diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
index 9c12ae72abc2..b45b1d22784f 100644
--- a/drivers/vdpa/vdpa_user/vduse_dev.c
+++ b/drivers/vdpa/vdpa_user/vduse_dev.c
@@ -86,6 +86,12 @@ struct vduse_umem {
struct mm_struct *mm;
};
+struct vduse_as {
+ struct vduse_iova_domain *domain;
+ struct vduse_umem *umem;
+ struct mutex mem_lock;
+};
+
struct vduse_vq_group_int {
struct vduse_dev *dev;
};
@@ -94,7 +100,7 @@ struct vduse_dev {
struct vduse_vdpa *vdev;
struct device *dev;
struct vduse_virtqueue **vqs;
- struct vduse_iova_domain *domain;
+ struct vduse_as as;
char *name;
struct mutex lock;
spinlock_t msg_lock;
@@ -122,9 +128,7 @@ struct vduse_dev {
u32 vq_num;
u32 vq_align;
u32 ngroups;
- struct vduse_umem *umem;
struct vduse_vq_group_int *groups;
- struct mutex mem_lock;
unsigned int bounce_size;
rwlock_t domain_lock;
};
@@ -439,7 +443,7 @@ static __poll_t vduse_dev_poll(struct file *file, poll_table *wait)
static void vduse_dev_reset(struct vduse_dev *dev)
{
int i;
- struct vduse_iova_domain *domain = dev->domain;
+ struct vduse_iova_domain *domain = dev->as.domain;
/* The coherent mappings are handled in vduse_dev_free_coherent() */
if (domain && domain->bounce_map)
@@ -788,13 +792,13 @@ static int vduse_vdpa_set_map(struct vdpa_device *vdpa,
struct vduse_dev *dev = vdpa_to_vduse(vdpa);
int ret;
- ret = vduse_domain_set_map(dev->domain, iotlb);
+ ret = vduse_domain_set_map(dev->as.domain, iotlb);
if (ret)
return ret;
ret = vduse_dev_update_iotlb(dev, 0ULL, ULLONG_MAX);
if (ret) {
- vduse_domain_clear_map(dev->domain, iotlb);
+ vduse_domain_clear_map(dev->as.domain, iotlb);
return ret;
}
@@ -846,7 +850,7 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
enum dma_data_direction dir)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->domain;
+ struct vduse_iova_domain *domain = vdev->as.domain;
vduse_domain_sync_single_for_device(domain, dma_addr, size, dir);
}
@@ -856,7 +860,7 @@ static void vduse_dev_sync_single_for_cpu(union virtio_map token,
enum dma_data_direction dir)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->domain;
+ struct vduse_iova_domain *domain = vdev->as.domain;
vduse_domain_sync_single_for_cpu(domain, dma_addr, size, dir);
}
@@ -867,7 +871,7 @@ static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
unsigned long attrs)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->domain;
+ struct vduse_iova_domain *domain = vdev->as.domain;
return vduse_domain_map_page(domain, page, offset, size, dir, attrs);
}
@@ -877,7 +881,7 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
unsigned long attrs)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->domain;
+ struct vduse_iova_domain *domain = vdev->as.domain;
return vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
}
@@ -886,7 +890,7 @@ static void *vduse_dev_alloc_coherent(union virtio_map token, size_t size,
dma_addr_t *dma_addr, gfp_t flag)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->domain;
+ struct vduse_iova_domain *domain = vdev->as.domain;
unsigned long iova;
void *addr;
@@ -906,7 +910,7 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
unsigned long attrs)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->domain;
+ struct vduse_iova_domain *domain = vdev->as.domain;
vduse_domain_free_coherent(domain, size, vaddr, dma_addr, attrs);
}
@@ -914,7 +918,7 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
static bool vduse_dev_need_sync(union virtio_map token, dma_addr_t dma_addr)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->domain;
+ struct vduse_iova_domain *domain = vdev->as.domain;
return dma_addr < domain->bounce_size;
}
@@ -929,7 +933,7 @@ static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
static size_t vduse_dev_max_mapping_size(union virtio_map token)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->domain;
+ struct vduse_iova_domain *domain = vdev->as.domain;
return domain->bounce_size;
}
@@ -1076,29 +1080,29 @@ static int vduse_dev_dereg_umem(struct vduse_dev *dev,
{
int ret;
- mutex_lock(&dev->mem_lock);
+ mutex_lock(&dev->as.mem_lock);
ret = -ENOENT;
- if (!dev->umem)
+ if (!dev->as.umem)
goto unlock;
ret = -EINVAL;
- if (!dev->domain)
+ if (!dev->as.domain)
goto unlock;
- if (dev->umem->iova != iova || size != dev->domain->bounce_size)
+ if (dev->as.umem->iova != iova || size != dev->as.domain->bounce_size)
goto unlock;
- vduse_domain_remove_user_bounce_pages(dev->domain);
- unpin_user_pages_dirty_lock(dev->umem->pages,
- dev->umem->npages, true);
- atomic64_sub(dev->umem->npages, &dev->umem->mm->pinned_vm);
- mmdrop(dev->umem->mm);
- vfree(dev->umem->pages);
- kfree(dev->umem);
- dev->umem = NULL;
+ vduse_domain_remove_user_bounce_pages(dev->as.domain);
+ unpin_user_pages_dirty_lock(dev->as.umem->pages,
+ dev->as.umem->npages, true);
+ atomic64_sub(dev->as.umem->npages, &dev->as.umem->mm->pinned_vm);
+ mmdrop(dev->as.umem->mm);
+ vfree(dev->as.umem->pages);
+ kfree(dev->as.umem);
+ dev->as.umem = NULL;
ret = 0;
unlock:
- mutex_unlock(&dev->mem_lock);
+ mutex_unlock(&dev->as.mem_lock);
return ret;
}
@@ -1111,14 +1115,14 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
unsigned long npages, lock_limit;
int ret;
- if (!dev->domain || !dev->domain->bounce_map ||
- size != dev->domain->bounce_size ||
+ if (!dev->as.domain || !dev->as.domain->bounce_map ||
+ size != dev->as.domain->bounce_size ||
iova != 0 || uaddr & ~PAGE_MASK)
return -EINVAL;
- mutex_lock(&dev->mem_lock);
+ mutex_lock(&dev->as.mem_lock);
ret = -EEXIST;
- if (dev->umem)
+ if (dev->as.umem)
goto unlock;
ret = -ENOMEM;
@@ -1142,7 +1146,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
goto out;
}
- ret = vduse_domain_add_user_bounce_pages(dev->domain,
+ ret = vduse_domain_add_user_bounce_pages(dev->as.domain,
page_list, pinned);
if (ret)
goto out;
@@ -1155,7 +1159,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
umem->mm = current->mm;
mmgrab(current->mm);
- dev->umem = umem;
+ dev->as.umem = umem;
out:
if (ret && pinned > 0)
unpin_user_pages(page_list, pinned);
@@ -1166,7 +1170,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
vfree(page_list);
kfree(umem);
}
- mutex_unlock(&dev->mem_lock);
+ mutex_unlock(&dev->as.mem_lock);
return ret;
}
@@ -1212,12 +1216,12 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
break;
read_lock(&dev->domain_lock);
- if (!dev->domain) {
+ if (!dev->as.domain) {
read_unlock(&dev->domain_lock);
break;
}
- spin_lock(&dev->domain->iotlb_lock);
- map = vhost_iotlb_itree_first(dev->domain->iotlb,
+ spin_lock(&dev->as.domain->iotlb_lock);
+ map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
entry.start, entry.last);
if (map) {
map_file = (struct vdpa_map_file *)map->opaque;
@@ -1227,7 +1231,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
entry.last = map->last;
entry.perm = map->perm;
}
- spin_unlock(&dev->domain->iotlb_lock);
+ spin_unlock(&dev->as.domain->iotlb_lock);
read_unlock(&dev->domain_lock);
ret = -EINVAL;
if (!f)
@@ -1430,22 +1434,22 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
break;
read_lock(&dev->domain_lock);
- if (!dev->domain) {
+ if (!dev->as.domain) {
read_unlock(&dev->domain_lock);
break;
}
- spin_lock(&dev->domain->iotlb_lock);
- map = vhost_iotlb_itree_first(dev->domain->iotlb,
+ spin_lock(&dev->as.domain->iotlb_lock);
+ map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
info.start, info.last);
if (map) {
info.start = map->start;
info.last = map->last;
info.capability = 0;
- if (dev->domain->bounce_map && map->start == 0 &&
- map->last == dev->domain->bounce_size - 1)
+ if (dev->as.domain->bounce_map && map->start == 0 &&
+ map->last == dev->as.domain->bounce_size - 1)
info.capability |= VDUSE_IOVA_CAP_UMEM;
}
- spin_unlock(&dev->domain->iotlb_lock);
+ spin_unlock(&dev->as.domain->iotlb_lock);
read_unlock(&dev->domain_lock);
if (!map)
break;
@@ -1470,8 +1474,8 @@ static int vduse_dev_release(struct inode *inode, struct file *file)
struct vduse_dev *dev = file->private_data;
write_lock(&dev->domain_lock);
- if (dev->domain)
- vduse_dev_dereg_umem(dev, 0, dev->domain->bounce_size);
+ if (dev->as.domain)
+ vduse_dev_dereg_umem(dev, 0, dev->as.domain->bounce_size);
write_unlock(&dev->domain_lock);
spin_lock(&dev->msg_lock);
/* Make sure the inflight messages can processed after reconncection */
@@ -1690,7 +1694,7 @@ static struct vduse_dev *vduse_dev_create(void)
return NULL;
mutex_init(&dev->lock);
- mutex_init(&dev->mem_lock);
+ mutex_init(&dev->as.mem_lock);
rwlock_init(&dev->domain_lock);
spin_lock_init(&dev->msg_lock);
INIT_LIST_HEAD(&dev->send_list);
@@ -1741,8 +1745,8 @@ static int vduse_destroy_dev(char *name)
idr_remove(&vduse_idr, dev->minor);
kvfree(dev->config);
vduse_dev_deinit_vqs(dev);
- if (dev->domain)
- vduse_domain_destroy(dev->domain);
+ if (dev->as.domain)
+ vduse_domain_destroy(dev->as.domain);
kfree(dev->name);
kfree(dev->groups);
vduse_dev_destroy(dev);
@@ -1858,7 +1862,7 @@ static ssize_t bounce_size_store(struct device *device,
ret = -EPERM;
write_lock(&dev->domain_lock);
- if (dev->domain)
+ if (dev->as.domain)
goto unlock;
ret = kstrtouint(buf, 10, &bounce_size);
@@ -2109,11 +2113,11 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
return ret;
write_lock(&dev->domain_lock);
- if (!dev->domain)
- dev->domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
+ if (!dev->as.domain)
+ dev->as.domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
dev->bounce_size);
write_unlock(&dev->domain_lock);
- if (!dev->domain) {
+ if (!dev->as.domain) {
put_device(&dev->vdev->vdpa.dev);
return -ENOMEM;
}
@@ -2122,8 +2126,8 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
if (ret) {
put_device(&dev->vdev->vdpa.dev);
write_lock(&dev->domain_lock);
- vduse_domain_destroy(dev->domain);
- dev->domain = NULL;
+ vduse_domain_destroy(dev->as.domain);
+ dev->as.domain = NULL;
write_unlock(&dev->domain_lock);
return ret;
}
--
2.51.0
^ permalink raw reply related [flat|nested] 27+ messages in thread* Re: [PATCH v2 5/7] vduse: create vduse_as to make it an array
2025-09-16 13:08 ` [PATCH v2 5/7] vduse: create vduse_as to make it an array Eugenio Pérez
@ 2025-09-17 8:46 ` Jason Wang
2025-09-17 16:34 ` Eugenio Perez Martin
0 siblings, 1 reply; 27+ messages in thread
From: Jason Wang @ 2025-09-17 8:46 UTC (permalink / raw)
To: Eugenio Pérez
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
>
> This is a first step so we can make more than one different address
> spaces. No change on the colde flow intended.
>
> Acked-by: Jason Wang <jasowang@redhat.com>
Sorry, I think I found something new.
> Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> ---
> drivers/vdpa/vdpa_user/vduse_dev.c | 114 +++++++++++++++--------------
> 1 file changed, 59 insertions(+), 55 deletions(-)
>
> diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> index 9c12ae72abc2..b45b1d22784f 100644
> --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> @@ -86,6 +86,12 @@ struct vduse_umem {
> struct mm_struct *mm;
> };
>
> +struct vduse_as {
> + struct vduse_iova_domain *domain;
> + struct vduse_umem *umem;
> + struct mutex mem_lock;
> +};
> +
> struct vduse_vq_group_int {
> struct vduse_dev *dev;
I would expect this has an indirection for as. E.g unsigned int as;
> };
> @@ -94,7 +100,7 @@ struct vduse_dev {
> struct vduse_vdpa *vdev;
> struct device *dev;
> struct vduse_virtqueue **vqs;
> - struct vduse_iova_domain *domain;
> + struct vduse_as as;
This needs to be an array per title:
"vduse: create vduse_as to make it an array"
> char *name;
> struct mutex lock;
> spinlock_t msg_lock;
> @@ -122,9 +128,7 @@ struct vduse_dev {
> u32 vq_num;
> u32 vq_align;
> u32 ngroups;
> - struct vduse_umem *umem;
> struct vduse_vq_group_int *groups;
> - struct mutex mem_lock;
> unsigned int bounce_size;
> rwlock_t domain_lock;
> };
> @@ -439,7 +443,7 @@ static __poll_t vduse_dev_poll(struct file *file, poll_table *wait)
> static void vduse_dev_reset(struct vduse_dev *dev)
> {
> int i;
> - struct vduse_iova_domain *domain = dev->domain;
> + struct vduse_iova_domain *domain = dev->as.domain;
This should be an iteration of all address spaces?
>
> /* The coherent mappings are handled in vduse_dev_free_coherent() */
> if (domain && domain->bounce_map)
> @@ -788,13 +792,13 @@ static int vduse_vdpa_set_map(struct vdpa_device *vdpa,
> struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> int ret;
>
> - ret = vduse_domain_set_map(dev->domain, iotlb);
> + ret = vduse_domain_set_map(dev->as.domain, iotlb);
The as should be indexed by asid here or I think i miss something. Or
at least fail with asid != 0?
> if (ret)
> return ret;
>
> ret = vduse_dev_update_iotlb(dev, 0ULL, ULLONG_MAX);
> if (ret) {
> - vduse_domain_clear_map(dev->domain, iotlb);
> + vduse_domain_clear_map(dev->as.domain, iotlb);
> return ret;
> }
>
> @@ -846,7 +850,7 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
> enum dma_data_direction dir)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->domain;
> + struct vduse_iova_domain *domain = vdev->as.domain;
The address space should be fetched from the virtqueue group instead
of the hard-coded one.
>
> vduse_domain_sync_single_for_device(domain, dma_addr, size, dir);
> }
> @@ -856,7 +860,7 @@ static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> enum dma_data_direction dir)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->domain;
> + struct vduse_iova_domain *domain = vdev->as.domain;
>
> vduse_domain_sync_single_for_cpu(domain, dma_addr, size, dir);
> }
> @@ -867,7 +871,7 @@ static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> unsigned long attrs)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->domain;
> + struct vduse_iova_domain *domain = vdev->as.domain;
>
> return vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> }
> @@ -877,7 +881,7 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> unsigned long attrs)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->domain;
> + struct vduse_iova_domain *domain = vdev->as.domain;
>
> return vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> }
> @@ -886,7 +890,7 @@ static void *vduse_dev_alloc_coherent(union virtio_map token, size_t size,
> dma_addr_t *dma_addr, gfp_t flag)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->domain;
> + struct vduse_iova_domain *domain = vdev->as.domain;
> unsigned long iova;
> void *addr;
>
> @@ -906,7 +910,7 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
> unsigned long attrs)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->domain;
> + struct vduse_iova_domain *domain = vdev->as.domain;
>
> vduse_domain_free_coherent(domain, size, vaddr, dma_addr, attrs);
> }
> @@ -914,7 +918,7 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
> static bool vduse_dev_need_sync(union virtio_map token, dma_addr_t dma_addr)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->domain;
> + struct vduse_iova_domain *domain = vdev->as.domain;
>
> return dma_addr < domain->bounce_size;
> }
> @@ -929,7 +933,7 @@ static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> static size_t vduse_dev_max_mapping_size(union virtio_map token)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->domain;
> + struct vduse_iova_domain *domain = vdev->as.domain;
>
> return domain->bounce_size;
> }
> @@ -1076,29 +1080,29 @@ static int vduse_dev_dereg_umem(struct vduse_dev *dev,
> {
> int ret;
>
> - mutex_lock(&dev->mem_lock);
> + mutex_lock(&dev->as.mem_lock);
> ret = -ENOENT;
> - if (!dev->umem)
> + if (!dev->as.umem)
> goto unlock;
>
> ret = -EINVAL;
> - if (!dev->domain)
> + if (!dev->as.domain)
> goto unlock;
>
> - if (dev->umem->iova != iova || size != dev->domain->bounce_size)
> + if (dev->as.umem->iova != iova || size != dev->as.domain->bounce_size)
I think it would be better to assume ASID = 0 here to ease the future extension.
> goto unlock;
>
> - vduse_domain_remove_user_bounce_pages(dev->domain);
> - unpin_user_pages_dirty_lock(dev->umem->pages,
> - dev->umem->npages, true);
> - atomic64_sub(dev->umem->npages, &dev->umem->mm->pinned_vm);
> - mmdrop(dev->umem->mm);
> - vfree(dev->umem->pages);
> - kfree(dev->umem);
> - dev->umem = NULL;
> + vduse_domain_remove_user_bounce_pages(dev->as.domain);
> + unpin_user_pages_dirty_lock(dev->as.umem->pages,
> + dev->as.umem->npages, true);
> + atomic64_sub(dev->as.umem->npages, &dev->as.umem->mm->pinned_vm);
> + mmdrop(dev->as.umem->mm);
> + vfree(dev->as.umem->pages);
> + kfree(dev->as.umem);
> + dev->as.umem = NULL;
> ret = 0;
> unlock:
> - mutex_unlock(&dev->mem_lock);
> + mutex_unlock(&dev->as.mem_lock);
> return ret;
> }
>
> @@ -1111,14 +1115,14 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> unsigned long npages, lock_limit;
> int ret;
>
> - if (!dev->domain || !dev->domain->bounce_map ||
> - size != dev->domain->bounce_size ||
> + if (!dev->as.domain || !dev->as.domain->bounce_map ||
> + size != dev->as.domain->bounce_size ||
> iova != 0 || uaddr & ~PAGE_MASK)
> return -EINVAL;
>
> - mutex_lock(&dev->mem_lock);
> + mutex_lock(&dev->as.mem_lock);
> ret = -EEXIST;
> - if (dev->umem)
> + if (dev->as.umem)
> goto unlock;
>
> ret = -ENOMEM;
> @@ -1142,7 +1146,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> goto out;
> }
>
> - ret = vduse_domain_add_user_bounce_pages(dev->domain,
> + ret = vduse_domain_add_user_bounce_pages(dev->as.domain,
> page_list, pinned);
> if (ret)
> goto out;
> @@ -1155,7 +1159,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> umem->mm = current->mm;
> mmgrab(current->mm);
>
> - dev->umem = umem;
> + dev->as.umem = umem;
> out:
> if (ret && pinned > 0)
> unpin_user_pages(page_list, pinned);
> @@ -1166,7 +1170,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> vfree(page_list);
> kfree(umem);
> }
> - mutex_unlock(&dev->mem_lock);
> + mutex_unlock(&dev->as.mem_lock);
> return ret;
> }
>
> @@ -1212,12 +1216,12 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> break;
>
> read_lock(&dev->domain_lock);
> - if (!dev->domain) {
> + if (!dev->as.domain) {
> read_unlock(&dev->domain_lock);
> break;
> }
> - spin_lock(&dev->domain->iotlb_lock);
> - map = vhost_iotlb_itree_first(dev->domain->iotlb,
> + spin_lock(&dev->as.domain->iotlb_lock);
> + map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
> entry.start, entry.last);
> if (map) {
> map_file = (struct vdpa_map_file *)map->opaque;
> @@ -1227,7 +1231,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> entry.last = map->last;
> entry.perm = map->perm;
> }
> - spin_unlock(&dev->domain->iotlb_lock);
> + spin_unlock(&dev->as.domain->iotlb_lock);
> read_unlock(&dev->domain_lock);
> ret = -EINVAL;
> if (!f)
> @@ -1430,22 +1434,22 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> break;
>
> read_lock(&dev->domain_lock);
> - if (!dev->domain) {
> + if (!dev->as.domain) {
> read_unlock(&dev->domain_lock);
> break;
> }
> - spin_lock(&dev->domain->iotlb_lock);
> - map = vhost_iotlb_itree_first(dev->domain->iotlb,
> + spin_lock(&dev->as.domain->iotlb_lock);
> + map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
> info.start, info.last);
> if (map) {
> info.start = map->start;
> info.last = map->last;
> info.capability = 0;
> - if (dev->domain->bounce_map && map->start == 0 &&
> - map->last == dev->domain->bounce_size - 1)
> + if (dev->as.domain->bounce_map && map->start == 0 &&
> + map->last == dev->as.domain->bounce_size - 1)
> info.capability |= VDUSE_IOVA_CAP_UMEM;
> }
> - spin_unlock(&dev->domain->iotlb_lock);
> + spin_unlock(&dev->as.domain->iotlb_lock);
> read_unlock(&dev->domain_lock);
> if (!map)
> break;
> @@ -1470,8 +1474,8 @@ static int vduse_dev_release(struct inode *inode, struct file *file)
> struct vduse_dev *dev = file->private_data;
>
> write_lock(&dev->domain_lock);
> - if (dev->domain)
> - vduse_dev_dereg_umem(dev, 0, dev->domain->bounce_size);
> + if (dev->as.domain)
> + vduse_dev_dereg_umem(dev, 0, dev->as.domain->bounce_size);
> write_unlock(&dev->domain_lock);
> spin_lock(&dev->msg_lock);
> /* Make sure the inflight messages can processed after reconncection */
> @@ -1690,7 +1694,7 @@ static struct vduse_dev *vduse_dev_create(void)
> return NULL;
>
> mutex_init(&dev->lock);
> - mutex_init(&dev->mem_lock);
> + mutex_init(&dev->as.mem_lock);
> rwlock_init(&dev->domain_lock);
> spin_lock_init(&dev->msg_lock);
> INIT_LIST_HEAD(&dev->send_list);
> @@ -1741,8 +1745,8 @@ static int vduse_destroy_dev(char *name)
> idr_remove(&vduse_idr, dev->minor);
> kvfree(dev->config);
> vduse_dev_deinit_vqs(dev);
> - if (dev->domain)
> - vduse_domain_destroy(dev->domain);
> + if (dev->as.domain)
> + vduse_domain_destroy(dev->as.domain);
> kfree(dev->name);
> kfree(dev->groups);
> vduse_dev_destroy(dev);
> @@ -1858,7 +1862,7 @@ static ssize_t bounce_size_store(struct device *device,
>
> ret = -EPERM;
> write_lock(&dev->domain_lock);
> - if (dev->domain)
> + if (dev->as.domain)
> goto unlock;
>
> ret = kstrtouint(buf, 10, &bounce_size);
> @@ -2109,11 +2113,11 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> return ret;
>
> write_lock(&dev->domain_lock);
> - if (!dev->domain)
> - dev->domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
> + if (!dev->as.domain)
> + dev->as.domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
> dev->bounce_size);
> write_unlock(&dev->domain_lock);
> - if (!dev->domain) {
> + if (!dev->as.domain) {
> put_device(&dev->vdev->vdpa.dev);
> return -ENOMEM;
> }
> @@ -2122,8 +2126,8 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> if (ret) {
> put_device(&dev->vdev->vdpa.dev);
> write_lock(&dev->domain_lock);
> - vduse_domain_destroy(dev->domain);
> - dev->domain = NULL;
> + vduse_domain_destroy(dev->as.domain);
> + dev->as.domain = NULL;
> write_unlock(&dev->domain_lock);
> return ret;
> }
> --
> 2.51.0
>
Thanks
^ permalink raw reply [flat|nested] 27+ messages in thread* Re: [PATCH v2 5/7] vduse: create vduse_as to make it an array
2025-09-17 8:46 ` Jason Wang
@ 2025-09-17 16:34 ` Eugenio Perez Martin
2025-09-18 6:04 ` Jason Wang
0 siblings, 1 reply; 27+ messages in thread
From: Eugenio Perez Martin @ 2025-09-17 16:34 UTC (permalink / raw)
To: Jason Wang
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Wed, Sep 17, 2025 at 10:46 AM Jason Wang <jasowang@redhat.com> wrote:
>
> On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
> >
> > This is a first step so we can make more than one different address
> > spaces. No change on the colde flow intended.
> >
> > Acked-by: Jason Wang <jasowang@redhat.com>
>
> Sorry, I think I found something new.
>
> > Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> > ---
> > drivers/vdpa/vdpa_user/vduse_dev.c | 114 +++++++++++++++--------------
> > 1 file changed, 59 insertions(+), 55 deletions(-)
> >
> > diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> > index 9c12ae72abc2..b45b1d22784f 100644
> > --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> > +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> > @@ -86,6 +86,12 @@ struct vduse_umem {
> > struct mm_struct *mm;
> > };
> >
> > +struct vduse_as {
> > + struct vduse_iova_domain *domain;
> > + struct vduse_umem *umem;
> > + struct mutex mem_lock;
> > +};
> > +
> > struct vduse_vq_group_int {
> > struct vduse_dev *dev;
>
> I would expect this has an indirection for as. E.g unsigned int as;
>
This is a change from the previous commit and *dev is needed for
taking the rwlock of the vq_group -> asid relation anyway, as function
like vduse_dev_sync_single_for_cpu could run at the same time as
VHOST_VDPA_SET_GROUP_ASID.
I'm ok with adding the "unsigned int as" indirection but it involves
memory walking that is not needed.
> > };
> > @@ -94,7 +100,7 @@ struct vduse_dev {
> > struct vduse_vdpa *vdev;
> > struct device *dev;
> > struct vduse_virtqueue **vqs;
> > - struct vduse_iova_domain *domain;
> > + struct vduse_as as;
>
> This needs to be an array per title:
>
> "vduse: create vduse_as to make it an array"
>
I meant "to make it an array in next patches". Would it work if I just
change the patch subject to:
"create vduse_as to contain per-as members"
And then specify that it will be converted to an array later in the
patch message? Or do you prefer me just to squash the patches?
> > char *name;
> > struct mutex lock;
> > spinlock_t msg_lock;
> > @@ -122,9 +128,7 @@ struct vduse_dev {
> > u32 vq_num;
> > u32 vq_align;
> > u32 ngroups;
> > - struct vduse_umem *umem;
> > struct vduse_vq_group_int *groups;
> > - struct mutex mem_lock;
> > unsigned int bounce_size;
> > rwlock_t domain_lock;
> > };
> > @@ -439,7 +443,7 @@ static __poll_t vduse_dev_poll(struct file *file, poll_table *wait)
> > static void vduse_dev_reset(struct vduse_dev *dev)
> > {
> > int i;
> > - struct vduse_iova_domain *domain = dev->domain;
> > + struct vduse_iova_domain *domain = dev->as.domain;
>
> This should be an iteration of all address spaces?
>
Yes, it will be in next patches.
The rest of the comments have the same reply actually, as this patch
just prepares the struct to make it an array in patches as small as
possible.
> >
> > /* The coherent mappings are handled in vduse_dev_free_coherent() */
> > if (domain && domain->bounce_map)
> > @@ -788,13 +792,13 @@ static int vduse_vdpa_set_map(struct vdpa_device *vdpa,
> > struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > int ret;
> >
> > - ret = vduse_domain_set_map(dev->domain, iotlb);
> > + ret = vduse_domain_set_map(dev->as.domain, iotlb);
>
> The as should be indexed by asid here or I think i miss something. Or
> at least fail with asid != 0?
>
> > if (ret)
> > return ret;
> >
> > ret = vduse_dev_update_iotlb(dev, 0ULL, ULLONG_MAX);
> > if (ret) {
> > - vduse_domain_clear_map(dev->domain, iotlb);
> > + vduse_domain_clear_map(dev->as.domain, iotlb);
> > return ret;
> > }
> >
> > @@ -846,7 +850,7 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
> > enum dma_data_direction dir)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->domain;
> > + struct vduse_iova_domain *domain = vdev->as.domain;
>
> The address space should be fetched from the virtqueue group instead
> of the hard-coded one.
>
> >
> > vduse_domain_sync_single_for_device(domain, dma_addr, size, dir);
> > }
> > @@ -856,7 +860,7 @@ static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> > enum dma_data_direction dir)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->domain;
> > + struct vduse_iova_domain *domain = vdev->as.domain;
> >
> > vduse_domain_sync_single_for_cpu(domain, dma_addr, size, dir);
> > }
> > @@ -867,7 +871,7 @@ static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> > unsigned long attrs)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->domain;
> > + struct vduse_iova_domain *domain = vdev->as.domain;
> >
> > return vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> > }
> > @@ -877,7 +881,7 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> > unsigned long attrs)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->domain;
> > + struct vduse_iova_domain *domain = vdev->as.domain;
> >
> > return vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> > }
> > @@ -886,7 +890,7 @@ static void *vduse_dev_alloc_coherent(union virtio_map token, size_t size,
> > dma_addr_t *dma_addr, gfp_t flag)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->domain;
> > + struct vduse_iova_domain *domain = vdev->as.domain;
> > unsigned long iova;
> > void *addr;
> >
> > @@ -906,7 +910,7 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
> > unsigned long attrs)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->domain;
> > + struct vduse_iova_domain *domain = vdev->as.domain;
> >
> > vduse_domain_free_coherent(domain, size, vaddr, dma_addr, attrs);
> > }
> > @@ -914,7 +918,7 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
> > static bool vduse_dev_need_sync(union virtio_map token, dma_addr_t dma_addr)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->domain;
> > + struct vduse_iova_domain *domain = vdev->as.domain;
> >
> > return dma_addr < domain->bounce_size;
> > }
> > @@ -929,7 +933,7 @@ static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> > static size_t vduse_dev_max_mapping_size(union virtio_map token)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->domain;
> > + struct vduse_iova_domain *domain = vdev->as.domain;
> >
> > return domain->bounce_size;
> > }
> > @@ -1076,29 +1080,29 @@ static int vduse_dev_dereg_umem(struct vduse_dev *dev,
> > {
> > int ret;
> >
> > - mutex_lock(&dev->mem_lock);
> > + mutex_lock(&dev->as.mem_lock);
> > ret = -ENOENT;
> > - if (!dev->umem)
> > + if (!dev->as.umem)
> > goto unlock;
> >
> > ret = -EINVAL;
> > - if (!dev->domain)
> > + if (!dev->as.domain)
> > goto unlock;
> >
> > - if (dev->umem->iova != iova || size != dev->domain->bounce_size)
> > + if (dev->as.umem->iova != iova || size != dev->as.domain->bounce_size)
>
> I think it would be better to assume ASID = 0 here to ease the future extension.
>
> > goto unlock;
> >
> > - vduse_domain_remove_user_bounce_pages(dev->domain);
> > - unpin_user_pages_dirty_lock(dev->umem->pages,
> > - dev->umem->npages, true);
> > - atomic64_sub(dev->umem->npages, &dev->umem->mm->pinned_vm);
> > - mmdrop(dev->umem->mm);
> > - vfree(dev->umem->pages);
> > - kfree(dev->umem);
> > - dev->umem = NULL;
> > + vduse_domain_remove_user_bounce_pages(dev->as.domain);
> > + unpin_user_pages_dirty_lock(dev->as.umem->pages,
> > + dev->as.umem->npages, true);
> > + atomic64_sub(dev->as.umem->npages, &dev->as.umem->mm->pinned_vm);
> > + mmdrop(dev->as.umem->mm);
> > + vfree(dev->as.umem->pages);
> > + kfree(dev->as.umem);
> > + dev->as.umem = NULL;
> > ret = 0;
> > unlock:
> > - mutex_unlock(&dev->mem_lock);
> > + mutex_unlock(&dev->as.mem_lock);
> > return ret;
> > }
> >
> > @@ -1111,14 +1115,14 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > unsigned long npages, lock_limit;
> > int ret;
> >
> > - if (!dev->domain || !dev->domain->bounce_map ||
> > - size != dev->domain->bounce_size ||
> > + if (!dev->as.domain || !dev->as.domain->bounce_map ||
> > + size != dev->as.domain->bounce_size ||
> > iova != 0 || uaddr & ~PAGE_MASK)
> > return -EINVAL;
> >
> > - mutex_lock(&dev->mem_lock);
> > + mutex_lock(&dev->as.mem_lock);
> > ret = -EEXIST;
> > - if (dev->umem)
> > + if (dev->as.umem)
> > goto unlock;
> >
> > ret = -ENOMEM;
> > @@ -1142,7 +1146,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > goto out;
> > }
> >
> > - ret = vduse_domain_add_user_bounce_pages(dev->domain,
> > + ret = vduse_domain_add_user_bounce_pages(dev->as.domain,
> > page_list, pinned);
> > if (ret)
> > goto out;
> > @@ -1155,7 +1159,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > umem->mm = current->mm;
> > mmgrab(current->mm);
> >
> > - dev->umem = umem;
> > + dev->as.umem = umem;
> > out:
> > if (ret && pinned > 0)
> > unpin_user_pages(page_list, pinned);
> > @@ -1166,7 +1170,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > vfree(page_list);
> > kfree(umem);
> > }
> > - mutex_unlock(&dev->mem_lock);
> > + mutex_unlock(&dev->as.mem_lock);
> > return ret;
> > }
> >
> > @@ -1212,12 +1216,12 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > break;
> >
> > read_lock(&dev->domain_lock);
> > - if (!dev->domain) {
> > + if (!dev->as.domain) {
> > read_unlock(&dev->domain_lock);
> > break;
> > }
> > - spin_lock(&dev->domain->iotlb_lock);
> > - map = vhost_iotlb_itree_first(dev->domain->iotlb,
> > + spin_lock(&dev->as.domain->iotlb_lock);
> > + map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
> > entry.start, entry.last);
> > if (map) {
> > map_file = (struct vdpa_map_file *)map->opaque;
> > @@ -1227,7 +1231,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > entry.last = map->last;
> > entry.perm = map->perm;
> > }
> > - spin_unlock(&dev->domain->iotlb_lock);
> > + spin_unlock(&dev->as.domain->iotlb_lock);
> > read_unlock(&dev->domain_lock);
> > ret = -EINVAL;
> > if (!f)
> > @@ -1430,22 +1434,22 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > break;
> >
> > read_lock(&dev->domain_lock);
> > - if (!dev->domain) {
> > + if (!dev->as.domain) {
> > read_unlock(&dev->domain_lock);
> > break;
> > }
> > - spin_lock(&dev->domain->iotlb_lock);
> > - map = vhost_iotlb_itree_first(dev->domain->iotlb,
> > + spin_lock(&dev->as.domain->iotlb_lock);
> > + map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
> > info.start, info.last);
> > if (map) {
> > info.start = map->start;
> > info.last = map->last;
> > info.capability = 0;
> > - if (dev->domain->bounce_map && map->start == 0 &&
> > - map->last == dev->domain->bounce_size - 1)
> > + if (dev->as.domain->bounce_map && map->start == 0 &&
> > + map->last == dev->as.domain->bounce_size - 1)
> > info.capability |= VDUSE_IOVA_CAP_UMEM;
> > }
> > - spin_unlock(&dev->domain->iotlb_lock);
> > + spin_unlock(&dev->as.domain->iotlb_lock);
> > read_unlock(&dev->domain_lock);
> > if (!map)
> > break;
> > @@ -1470,8 +1474,8 @@ static int vduse_dev_release(struct inode *inode, struct file *file)
> > struct vduse_dev *dev = file->private_data;
> >
> > write_lock(&dev->domain_lock);
> > - if (dev->domain)
> > - vduse_dev_dereg_umem(dev, 0, dev->domain->bounce_size);
> > + if (dev->as.domain)
> > + vduse_dev_dereg_umem(dev, 0, dev->as.domain->bounce_size);
> > write_unlock(&dev->domain_lock);
> > spin_lock(&dev->msg_lock);
> > /* Make sure the inflight messages can processed after reconncection */
> > @@ -1690,7 +1694,7 @@ static struct vduse_dev *vduse_dev_create(void)
> > return NULL;
> >
> > mutex_init(&dev->lock);
> > - mutex_init(&dev->mem_lock);
> > + mutex_init(&dev->as.mem_lock);
> > rwlock_init(&dev->domain_lock);
> > spin_lock_init(&dev->msg_lock);
> > INIT_LIST_HEAD(&dev->send_list);
> > @@ -1741,8 +1745,8 @@ static int vduse_destroy_dev(char *name)
> > idr_remove(&vduse_idr, dev->minor);
> > kvfree(dev->config);
> > vduse_dev_deinit_vqs(dev);
> > - if (dev->domain)
> > - vduse_domain_destroy(dev->domain);
> > + if (dev->as.domain)
> > + vduse_domain_destroy(dev->as.domain);
> > kfree(dev->name);
> > kfree(dev->groups);
> > vduse_dev_destroy(dev);
> > @@ -1858,7 +1862,7 @@ static ssize_t bounce_size_store(struct device *device,
> >
> > ret = -EPERM;
> > write_lock(&dev->domain_lock);
> > - if (dev->domain)
> > + if (dev->as.domain)
> > goto unlock;
> >
> > ret = kstrtouint(buf, 10, &bounce_size);
> > @@ -2109,11 +2113,11 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> > return ret;
> >
> > write_lock(&dev->domain_lock);
> > - if (!dev->domain)
> > - dev->domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
> > + if (!dev->as.domain)
> > + dev->as.domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
> > dev->bounce_size);
> > write_unlock(&dev->domain_lock);
> > - if (!dev->domain) {
> > + if (!dev->as.domain) {
> > put_device(&dev->vdev->vdpa.dev);
> > return -ENOMEM;
> > }
> > @@ -2122,8 +2126,8 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> > if (ret) {
> > put_device(&dev->vdev->vdpa.dev);
> > write_lock(&dev->domain_lock);
> > - vduse_domain_destroy(dev->domain);
> > - dev->domain = NULL;
> > + vduse_domain_destroy(dev->as.domain);
> > + dev->as.domain = NULL;
> > write_unlock(&dev->domain_lock);
> > return ret;
> > }
> > --
> > 2.51.0
> >
>
> Thanks
>
^ permalink raw reply [flat|nested] 27+ messages in thread* Re: [PATCH v2 5/7] vduse: create vduse_as to make it an array
2025-09-17 16:34 ` Eugenio Perez Martin
@ 2025-09-18 6:04 ` Jason Wang
0 siblings, 0 replies; 27+ messages in thread
From: Jason Wang @ 2025-09-18 6:04 UTC (permalink / raw)
To: Eugenio Perez Martin
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Thu, Sep 18, 2025 at 12:35 AM Eugenio Perez Martin
<eperezma@redhat.com> wrote:
>
> On Wed, Sep 17, 2025 at 10:46 AM Jason Wang <jasowang@redhat.com> wrote:
> >
> > On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
> > >
> > > This is a first step so we can make more than one different address
> > > spaces. No change on the colde flow intended.
> > >
> > > Acked-by: Jason Wang <jasowang@redhat.com>
> >
> > Sorry, I think I found something new.
> >
> > > Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> > > ---
> > > drivers/vdpa/vdpa_user/vduse_dev.c | 114 +++++++++++++++--------------
> > > 1 file changed, 59 insertions(+), 55 deletions(-)
> > >
> > > diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > index 9c12ae72abc2..b45b1d22784f 100644
> > > --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> > > +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > @@ -86,6 +86,12 @@ struct vduse_umem {
> > > struct mm_struct *mm;
> > > };
> > >
> > > +struct vduse_as {
> > > + struct vduse_iova_domain *domain;
> > > + struct vduse_umem *umem;
> > > + struct mutex mem_lock;
> > > +};
> > > +
> > > struct vduse_vq_group_int {
> > > struct vduse_dev *dev;
> >
> > I would expect this has an indirection for as. E.g unsigned int as;
> >
>
> This is a change from the previous commit and *dev is needed for
> taking the rwlock of the vq_group -> asid relation anyway, as function
> like vduse_dev_sync_single_for_cpu could run at the same time as
> VHOST_VDPA_SET_GROUP_ASID.
>
> I'm ok with adding the "unsigned int as" indirection but it involves
> memory walking that is not needed.
I think it should be fine unless we can notice it during the
benchmark. What's more we can refactor vduse as without caring much
about the group in the future.
>
> > > };
> > > @@ -94,7 +100,7 @@ struct vduse_dev {
> > > struct vduse_vdpa *vdev;
> > > struct device *dev;
> > > struct vduse_virtqueue **vqs;
> > > - struct vduse_iova_domain *domain;
> > > + struct vduse_as as;
> >
> > This needs to be an array per title:
> >
> > "vduse: create vduse_as to make it an array"
> >
>
> I meant "to make it an array in next patches". Would it work if I just
> change the patch subject to:
>
> "create vduse_as to contain per-as members"
>
> And then specify that it will be converted to an array later in the
> patch message? Or do you prefer me just to squash the patches?
I would prefer either:
1) make it an array in this patch
or
2) squash this patch into the next one
to avoid unnecessary changes.
>
> > > char *name;
> > > struct mutex lock;
> > > spinlock_t msg_lock;
> > > @@ -122,9 +128,7 @@ struct vduse_dev {
> > > u32 vq_num;
> > > u32 vq_align;
> > > u32 ngroups;
> > > - struct vduse_umem *umem;
> > > struct vduse_vq_group_int *groups;
> > > - struct mutex mem_lock;
> > > unsigned int bounce_size;
> > > rwlock_t domain_lock;
> > > };
> > > @@ -439,7 +443,7 @@ static __poll_t vduse_dev_poll(struct file *file, poll_table *wait)
> > > static void vduse_dev_reset(struct vduse_dev *dev)
> > > {
> > > int i;
> > > - struct vduse_iova_domain *domain = dev->domain;
> > > + struct vduse_iova_domain *domain = dev->as.domain;
> >
> > This should be an iteration of all address spaces?
> >
>
> Yes, it will be in next patches.
>
> The rest of the comments have the same reply actually, as this patch
> just prepares the struct to make it an array in patches as small as
> possible.
Right.
Thanks
^ permalink raw reply [flat|nested] 27+ messages in thread
* [PATCH v2 6/7] vduse: add vq group asid support
2025-09-16 13:08 [PATCH v2 0/7] Add multiple address spaces support to VDUSE Eugenio Pérez
` (4 preceding siblings ...)
2025-09-16 13:08 ` [PATCH v2 5/7] vduse: create vduse_as to make it an array Eugenio Pérez
@ 2025-09-16 13:08 ` Eugenio Pérez
2025-09-17 8:56 ` Jason Wang
2025-09-16 13:08 ` [PATCH v2 7/7] vduse: bump version number Eugenio Pérez
6 siblings, 1 reply; 27+ messages in thread
From: Eugenio Pérez @ 2025-09-16 13:08 UTC (permalink / raw)
To: Michael S . Tsirkin
Cc: Stefano Garzarella, jasowang, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization, Eugenio Pérez
Add support for assigning Address Space Identifiers (ASIDs) to each VQ
group. This enables mapping each group into a distinct memory space.
Now that the driver can change ASID in the middle of operation, the
domain that each vq address point is also protected by domain_lock.
Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
---
v2:
* Convert the use of mutex to rwlock.
RFC v3:
* Increase VDUSE_MAX_VQ_GROUPS to 0xffff (Jason). It was set to a lower
value to reduce memory consumption, but vqs are already limited to
that value and userspace VDUSE is able to allocate that many vqs.
* Remove TODO about merging VDUSE_IOTLB_GET_FD ioctl with
VDUSE_IOTLB_GET_INFO.
* Use of array_index_nospec in VDUSE device ioctls.
* Embed vduse_iotlb_entry into vduse_iotlb_entry_v2.
* Move the umem mutex to asid struct so there is no contention between
ASIDs.
RFC v2:
* Make iotlb entry the last one of vduse_iotlb_entry_v2 so the first
part of the struct is the same.
---
drivers/vdpa/vdpa_user/vduse_dev.c | 324 +++++++++++++++++++++--------
include/uapi/linux/vduse.h | 51 ++++-
2 files changed, 284 insertions(+), 91 deletions(-)
diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
index b45b1d22784f..06b7790380b7 100644
--- a/drivers/vdpa/vdpa_user/vduse_dev.c
+++ b/drivers/vdpa/vdpa_user/vduse_dev.c
@@ -93,6 +93,7 @@ struct vduse_as {
};
struct vduse_vq_group_int {
+ struct vduse_iova_domain *domain;
struct vduse_dev *dev;
};
@@ -100,7 +101,7 @@ struct vduse_dev {
struct vduse_vdpa *vdev;
struct device *dev;
struct vduse_virtqueue **vqs;
- struct vduse_as as;
+ struct vduse_as *as;
char *name;
struct mutex lock;
spinlock_t msg_lock;
@@ -128,6 +129,7 @@ struct vduse_dev {
u32 vq_num;
u32 vq_align;
u32 ngroups;
+ u32 nas;
struct vduse_vq_group_int *groups;
unsigned int bounce_size;
rwlock_t domain_lock;
@@ -318,7 +320,7 @@ static int vduse_dev_set_status(struct vduse_dev *dev, u8 status)
return vduse_dev_msg_sync(dev, &msg);
}
-static int vduse_dev_update_iotlb(struct vduse_dev *dev,
+static int vduse_dev_update_iotlb(struct vduse_dev *dev, u32 asid,
u64 start, u64 last)
{
struct vduse_dev_msg msg = { 0 };
@@ -327,8 +329,14 @@ static int vduse_dev_update_iotlb(struct vduse_dev *dev,
return -EINVAL;
msg.req.type = VDUSE_UPDATE_IOTLB;
- msg.req.iova.start = start;
- msg.req.iova.last = last;
+ if (dev->api_version < VDUSE_API_VERSION_1) {
+ msg.req.iova.start = start;
+ msg.req.iova.last = last;
+ } else {
+ msg.req.iova_v2.start = start;
+ msg.req.iova_v2.last = last;
+ msg.req.iova_v2.asid = asid;
+ }
return vduse_dev_msg_sync(dev, &msg);
}
@@ -440,14 +448,29 @@ static __poll_t vduse_dev_poll(struct file *file, poll_table *wait)
return mask;
}
+/* Force set the asid to a vq group without a message to the VDUSE device */
+static void vduse_set_group_asid_nomsg(struct vduse_dev *dev,
+ unsigned int group, unsigned int asid)
+{
+ write_lock(&dev->domain_lock);
+ dev->groups[group].domain = dev->as[asid].domain;
+ write_unlock(&dev->domain_lock);
+}
+
static void vduse_dev_reset(struct vduse_dev *dev)
{
int i;
- struct vduse_iova_domain *domain = dev->as.domain;
/* The coherent mappings are handled in vduse_dev_free_coherent() */
- if (domain && domain->bounce_map)
- vduse_domain_reset_bounce_map(domain);
+ for (i = 0; i < dev->nas; i++) {
+ struct vduse_iova_domain *domain = dev->as[i].domain;
+
+ if (domain && domain->bounce_map)
+ vduse_domain_reset_bounce_map(domain);
+ }
+
+ for (i = 0; i < dev->ngroups; i++)
+ vduse_set_group_asid_nomsg(dev, i, 0);
down_write(&dev->rwsem);
@@ -621,6 +644,29 @@ static union virtio_map vduse_get_vq_map(struct vdpa_device *vdpa, u16 idx)
return ret;
}
+static int vduse_set_group_asid(struct vdpa_device *vdpa, unsigned int group,
+ unsigned int asid)
+{
+ struct vduse_dev *dev = vdpa_to_vduse(vdpa);
+ struct vduse_dev_msg msg = { 0 };
+ int r;
+
+ if (dev->api_version < VDUSE_API_VERSION_1 ||
+ group >= dev->ngroups || asid >= dev->nas)
+ return -EINVAL;
+
+ msg.req.type = VDUSE_SET_VQ_GROUP_ASID;
+ msg.req.vq_group_asid.group = group;
+ msg.req.vq_group_asid.asid = asid;
+
+ r = vduse_dev_msg_sync(dev, &msg);
+ if (r < 0)
+ return r;
+
+ vduse_set_group_asid_nomsg(dev, group, asid);
+ return 0;
+}
+
static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
struct vdpa_vq_state *state)
{
@@ -792,13 +838,13 @@ static int vduse_vdpa_set_map(struct vdpa_device *vdpa,
struct vduse_dev *dev = vdpa_to_vduse(vdpa);
int ret;
- ret = vduse_domain_set_map(dev->as.domain, iotlb);
+ ret = vduse_domain_set_map(dev->as[asid].domain, iotlb);
if (ret)
return ret;
- ret = vduse_dev_update_iotlb(dev, 0ULL, ULLONG_MAX);
+ ret = vduse_dev_update_iotlb(dev, asid, 0ULL, ULLONG_MAX);
if (ret) {
- vduse_domain_clear_map(dev->as.domain, iotlb);
+ vduse_domain_clear_map(dev->as[asid].domain, iotlb);
return ret;
}
@@ -841,6 +887,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
.get_vq_affinity = vduse_vdpa_get_vq_affinity,
.reset = vduse_vdpa_reset,
.set_map = vduse_vdpa_set_map,
+ .set_group_asid = vduse_set_group_asid,
.get_vq_map = vduse_get_vq_map,
.free = vduse_vdpa_free,
};
@@ -850,9 +897,12 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
enum dma_data_direction dir)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->as.domain;
+ struct vduse_iova_domain *domain;
+ read_lock(&vdev->domain_lock);
+ domain = token.group->domain;
vduse_domain_sync_single_for_device(domain, dma_addr, size, dir);
+ read_unlock(&vdev->domain_lock);
}
static void vduse_dev_sync_single_for_cpu(union virtio_map token,
@@ -860,9 +910,12 @@ static void vduse_dev_sync_single_for_cpu(union virtio_map token,
enum dma_data_direction dir)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->as.domain;
+ struct vduse_iova_domain *domain;
+ read_lock(&vdev->domain_lock);
+ domain = token.group->domain;
vduse_domain_sync_single_for_cpu(domain, dma_addr, size, dir);
+ read_unlock(&vdev->domain_lock);
}
static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
@@ -871,9 +924,15 @@ static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
unsigned long attrs)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->as.domain;
+ struct vduse_iova_domain *domain;
+ dma_addr_t r;
- return vduse_domain_map_page(domain, page, offset, size, dir, attrs);
+ read_lock(&vdev->domain_lock);
+ domain = token.group->domain;
+ r = vduse_domain_map_page(domain, page, offset, size, dir, attrs);
+ read_unlock(&vdev->domain_lock);
+
+ return r;
}
static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
@@ -881,27 +940,31 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
unsigned long attrs)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->as.domain;
+ struct vduse_iova_domain *domain;
- return vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
+ read_lock(&vdev->domain_lock);
+ domain = token.group->domain;
+ vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
+ read_unlock(&vdev->domain_lock);
}
static void *vduse_dev_alloc_coherent(union virtio_map token, size_t size,
dma_addr_t *dma_addr, gfp_t flag)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->as.domain;
+ struct vduse_iova_domain *domain;
unsigned long iova;
- void *addr;
+ void *addr = NULL;
*dma_addr = DMA_MAPPING_ERROR;
+ read_lock(&vdev->domain_lock);
+ domain = token.group->domain;
addr = vduse_domain_alloc_coherent(domain, size,
(dma_addr_t *)&iova, flag);
- if (!addr)
- return NULL;
-
- *dma_addr = (dma_addr_t)iova;
+ if (addr)
+ *dma_addr = (dma_addr_t)iova;
+ read_unlock(&vdev->domain_lock);
return addr;
}
@@ -910,17 +973,26 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
unsigned long attrs)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->as.domain;
+ struct vduse_iova_domain *domain;
+ read_lock(&vdev->domain_lock);
+ domain = token.group->domain;
vduse_domain_free_coherent(domain, size, vaddr, dma_addr, attrs);
+ read_unlock(&vdev->domain_lock);
}
static bool vduse_dev_need_sync(union virtio_map token, dma_addr_t dma_addr)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->as.domain;
+ struct vduse_iova_domain *domain;
+ size_t bounce_size;
- return dma_addr < domain->bounce_size;
+ read_lock(&vdev->domain_lock);
+ domain = token.group->domain;
+ bounce_size = domain->bounce_size;
+ read_unlock(&vdev->domain_lock);
+
+ return dma_addr < bounce_size;
}
static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
@@ -933,9 +1005,15 @@ static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
static size_t vduse_dev_max_mapping_size(union virtio_map token)
{
struct vduse_dev *vdev = token.group->dev;
- struct vduse_iova_domain *domain = vdev->as.domain;
+ struct vduse_iova_domain *domain;
+ size_t bounce_size;
+
+ read_lock(&vdev->domain_lock);
+ domain = token.group->domain;
+ bounce_size = domain->bounce_size;
+ read_unlock(&vdev->domain_lock);
- return domain->bounce_size;
+ return bounce_size;
}
static const struct virtio_map_ops vduse_map_ops = {
@@ -1075,39 +1153,40 @@ static int vduse_dev_queue_irq_work(struct vduse_dev *dev,
return ret;
}
-static int vduse_dev_dereg_umem(struct vduse_dev *dev,
+static int vduse_dev_dereg_umem(struct vduse_dev *dev, u32 asid,
u64 iova, u64 size)
{
int ret;
- mutex_lock(&dev->as.mem_lock);
+ mutex_lock(&dev->as[asid].mem_lock);
ret = -ENOENT;
- if (!dev->as.umem)
+ if (!dev->as[asid].umem)
goto unlock;
ret = -EINVAL;
- if (!dev->as.domain)
+ if (!dev->as[asid].domain)
goto unlock;
- if (dev->as.umem->iova != iova || size != dev->as.domain->bounce_size)
+ if (dev->as[asid].umem->iova != iova ||
+ size != dev->as[asid].domain->bounce_size)
goto unlock;
- vduse_domain_remove_user_bounce_pages(dev->as.domain);
- unpin_user_pages_dirty_lock(dev->as.umem->pages,
- dev->as.umem->npages, true);
- atomic64_sub(dev->as.umem->npages, &dev->as.umem->mm->pinned_vm);
- mmdrop(dev->as.umem->mm);
- vfree(dev->as.umem->pages);
- kfree(dev->as.umem);
- dev->as.umem = NULL;
+ vduse_domain_remove_user_bounce_pages(dev->as[asid].domain);
+ unpin_user_pages_dirty_lock(dev->as[asid].umem->pages,
+ dev->as[asid].umem->npages, true);
+ atomic64_sub(dev->as[asid].umem->npages, &dev->as[asid].umem->mm->pinned_vm);
+ mmdrop(dev->as[asid].umem->mm);
+ vfree(dev->as[asid].umem->pages);
+ kfree(dev->as[asid].umem);
+ dev->as[asid].umem = NULL;
ret = 0;
unlock:
- mutex_unlock(&dev->as.mem_lock);
+ mutex_unlock(&dev->as[asid].mem_lock);
return ret;
}
static int vduse_dev_reg_umem(struct vduse_dev *dev,
- u64 iova, u64 uaddr, u64 size)
+ u32 asid, u64 iova, u64 uaddr, u64 size)
{
struct page **page_list = NULL;
struct vduse_umem *umem = NULL;
@@ -1115,14 +1194,14 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
unsigned long npages, lock_limit;
int ret;
- if (!dev->as.domain || !dev->as.domain->bounce_map ||
- size != dev->as.domain->bounce_size ||
+ if (!dev->as[asid].domain || !dev->as[asid].domain->bounce_map ||
+ size != dev->as[asid].domain->bounce_size ||
iova != 0 || uaddr & ~PAGE_MASK)
return -EINVAL;
- mutex_lock(&dev->as.mem_lock);
+ mutex_lock(&dev->as[asid].mem_lock);
ret = -EEXIST;
- if (dev->as.umem)
+ if (dev->as[asid].umem)
goto unlock;
ret = -ENOMEM;
@@ -1146,7 +1225,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
goto out;
}
- ret = vduse_domain_add_user_bounce_pages(dev->as.domain,
+ ret = vduse_domain_add_user_bounce_pages(dev->as[asid].domain,
page_list, pinned);
if (ret)
goto out;
@@ -1159,7 +1238,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
umem->mm = current->mm;
mmgrab(current->mm);
- dev->as.umem = umem;
+ dev->as[asid].umem = umem;
out:
if (ret && pinned > 0)
unpin_user_pages(page_list, pinned);
@@ -1170,7 +1249,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
vfree(page_list);
kfree(umem);
}
- mutex_unlock(&dev->as.mem_lock);
+ mutex_unlock(&dev->as[asid].mem_lock);
return ret;
}
@@ -1202,47 +1281,66 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
switch (cmd) {
case VDUSE_IOTLB_GET_FD: {
- struct vduse_iotlb_entry entry;
+ struct vduse_iotlb_entry_v2 entry;
struct vhost_iotlb_map *map;
struct vdpa_map_file *map_file;
struct file *f = NULL;
+ u32 asid;
ret = -EFAULT;
- if (copy_from_user(&entry, argp, sizeof(entry)))
- break;
+ if (dev->api_version >= VDUSE_API_VERSION_1) {
+ if (copy_from_user(&entry, argp, sizeof(entry)))
+ break;
+ } else {
+ entry.asid = 0;
+ if (copy_from_user(&entry.v1, argp,
+ sizeof(entry.v1)))
+ break;
+ }
ret = -EINVAL;
- if (entry.start > entry.last)
+ if (entry.v1.start > entry.v1.last)
+ break;
+
+ if (entry.asid >= dev->nas)
break;
read_lock(&dev->domain_lock);
- if (!dev->as.domain) {
+ asid = array_index_nospec(entry.asid, dev->nas);
+ if (!dev->as[asid].domain) {
read_unlock(&dev->domain_lock);
break;
}
- spin_lock(&dev->as.domain->iotlb_lock);
- map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
- entry.start, entry.last);
+ spin_lock(&dev->as[asid].domain->iotlb_lock);
+ map = vhost_iotlb_itree_first(dev->as[asid].domain->iotlb,
+ entry.v1.start, entry.v1.last);
if (map) {
map_file = (struct vdpa_map_file *)map->opaque;
f = get_file(map_file->file);
- entry.offset = map_file->offset;
- entry.start = map->start;
- entry.last = map->last;
- entry.perm = map->perm;
+ entry.v1.offset = map_file->offset;
+ entry.v1.start = map->start;
+ entry.v1.last = map->last;
+ entry.v1.perm = map->perm;
}
- spin_unlock(&dev->as.domain->iotlb_lock);
+ spin_unlock(&dev->as[asid].domain->iotlb_lock);
read_unlock(&dev->domain_lock);
ret = -EINVAL;
if (!f)
break;
ret = -EFAULT;
- if (copy_to_user(argp, &entry, sizeof(entry))) {
+ if (dev->api_version >= VDUSE_API_VERSION_1)
+ ret = copy_to_user(argp, &entry,
+ sizeof(entry));
+ else
+ ret = copy_to_user(argp, &entry.v1,
+ sizeof(entry.v1));
+
+ if (ret) {
fput(f);
break;
}
- ret = receive_fd(f, NULL, perm_to_file_flags(entry.perm));
+ ret = receive_fd(f, NULL, perm_to_file_flags(entry.v1.perm));
fput(f);
break;
}
@@ -1384,6 +1482,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
}
case VDUSE_IOTLB_REG_UMEM: {
struct vduse_iova_umem umem;
+ u32 asid;
ret = -EFAULT;
if (copy_from_user(&umem, argp, sizeof(umem)))
@@ -1391,17 +1490,21 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
ret = -EINVAL;
if (!is_mem_zero((const char *)umem.reserved,
- sizeof(umem.reserved)))
+ sizeof(umem.reserved)) ||
+ (dev->api_version < VDUSE_API_VERSION_1 &&
+ umem.asid != 0) || umem.asid >= dev->nas)
break;
write_lock(&dev->domain_lock);
- ret = vduse_dev_reg_umem(dev, umem.iova,
+ asid = array_index_nospec(umem.asid, dev->nas);
+ ret = vduse_dev_reg_umem(dev, asid, umem.iova,
umem.uaddr, umem.size);
write_unlock(&dev->domain_lock);
break;
}
case VDUSE_IOTLB_DEREG_UMEM: {
struct vduse_iova_umem umem;
+ u32 asid;
ret = -EFAULT;
if (copy_from_user(&umem, argp, sizeof(umem)))
@@ -1409,10 +1512,15 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
ret = -EINVAL;
if (!is_mem_zero((const char *)umem.reserved,
- sizeof(umem.reserved)))
+ sizeof(umem.reserved)) ||
+ (dev->api_version < VDUSE_API_VERSION_1 &&
+ umem.asid != 0) ||
+ umem.asid >= dev->nas)
break;
+
write_lock(&dev->domain_lock);
- ret = vduse_dev_dereg_umem(dev, umem.iova,
+ asid = array_index_nospec(umem.asid, dev->nas);
+ ret = vduse_dev_dereg_umem(dev, asid, umem.iova,
umem.size);
write_unlock(&dev->domain_lock);
break;
@@ -1420,6 +1528,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
case VDUSE_IOTLB_GET_INFO: {
struct vduse_iova_info info;
struct vhost_iotlb_map *map;
+ u32 asid;
ret = -EFAULT;
if (copy_from_user(&info, argp, sizeof(info)))
@@ -1433,23 +1542,31 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
sizeof(info.reserved)))
break;
+ if (dev->api_version < VDUSE_API_VERSION_1) {
+ if (info.asid)
+ break;
+ } else if (info.asid >= dev->nas)
+ break;
+
read_lock(&dev->domain_lock);
- if (!dev->as.domain) {
+ asid = array_index_nospec(info.asid, dev->nas);
+ if (!dev->as[asid].domain) {
read_unlock(&dev->domain_lock);
break;
}
- spin_lock(&dev->as.domain->iotlb_lock);
- map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
+ spin_lock(&dev->as[asid].domain->iotlb_lock);
+ map = vhost_iotlb_itree_first(dev->as[asid].domain->iotlb,
info.start, info.last);
if (map) {
info.start = map->start;
info.last = map->last;
info.capability = 0;
- if (dev->as.domain->bounce_map && map->start == 0 &&
- map->last == dev->as.domain->bounce_size - 1)
+ if (dev->as[asid].domain->bounce_map &&
+ map->start == 0 &&
+ map->last == dev->as[asid].domain->bounce_size - 1)
info.capability |= VDUSE_IOVA_CAP_UMEM;
}
- spin_unlock(&dev->as.domain->iotlb_lock);
+ spin_unlock(&dev->as[asid].domain->iotlb_lock);
read_unlock(&dev->domain_lock);
if (!map)
break;
@@ -1474,8 +1591,10 @@ static int vduse_dev_release(struct inode *inode, struct file *file)
struct vduse_dev *dev = file->private_data;
write_lock(&dev->domain_lock);
- if (dev->as.domain)
- vduse_dev_dereg_umem(dev, 0, dev->as.domain->bounce_size);
+ for (int i = 0; i < dev->nas; i++)
+ if (dev->as[i].domain)
+ vduse_dev_dereg_umem(dev, i, 0,
+ dev->as[i].domain->bounce_size);
write_unlock(&dev->domain_lock);
spin_lock(&dev->msg_lock);
/* Make sure the inflight messages can processed after reconncection */
@@ -1694,7 +1813,6 @@ static struct vduse_dev *vduse_dev_create(void)
return NULL;
mutex_init(&dev->lock);
- mutex_init(&dev->as.mem_lock);
rwlock_init(&dev->domain_lock);
spin_lock_init(&dev->msg_lock);
INIT_LIST_HEAD(&dev->send_list);
@@ -1745,8 +1863,11 @@ static int vduse_destroy_dev(char *name)
idr_remove(&vduse_idr, dev->minor);
kvfree(dev->config);
vduse_dev_deinit_vqs(dev);
- if (dev->as.domain)
- vduse_domain_destroy(dev->as.domain);
+ for (int i = 0; i < dev->nas; i++) {
+ if (dev->as[i].domain)
+ vduse_domain_destroy(dev->as[i].domain);
+ }
+ kfree(dev->as);
kfree(dev->name);
kfree(dev->groups);
vduse_dev_destroy(dev);
@@ -1793,12 +1914,16 @@ static bool vduse_validate_config(struct vduse_dev_config *config,
sizeof(config->reserved)))
return false;
- if (api_version < VDUSE_API_VERSION_1 && config->ngroups)
+ if (api_version < VDUSE_API_VERSION_1 &&
+ (config->ngroups || config->nas))
return false;
if (api_version >= VDUSE_API_VERSION_1 && config->ngroups > 0xffff)
return false;
+ if (api_version >= VDUSE_API_VERSION_1 && config->nas > 0xffff)
+ return false;
+
if (config->vq_align > PAGE_SIZE)
return false;
@@ -1862,7 +1987,8 @@ static ssize_t bounce_size_store(struct device *device,
ret = -EPERM;
write_lock(&dev->domain_lock);
- if (dev->as.domain)
+ /* Assuming that if the first domain is allocated, all are allocated */
+ if (dev->as[0].domain)
goto unlock;
ret = kstrtouint(buf, 10, &bounce_size);
@@ -1923,6 +2049,13 @@ static int vduse_create_dev(struct vduse_dev_config *config,
for (u32 i = 0; i < dev->ngroups; ++i)
dev->groups[i].dev = dev;
+ dev->nas = (dev->api_version < 1) ? 1 : (config->nas ?: 1);
+ dev->as = kcalloc(dev->nas, sizeof(dev->as[0]), GFP_KERNEL);
+ if (!dev->as)
+ goto err_as;
+ for (int i = 0; i < dev->nas; i++)
+ mutex_init(&dev->as[i].mem_lock);
+
dev->name = kstrdup(config->name, GFP_KERNEL);
if (!dev->name)
goto err_str;
@@ -1959,6 +2092,8 @@ static int vduse_create_dev(struct vduse_dev_config *config,
err_idr:
kfree(dev->name);
err_str:
+ kfree(dev->as);
+err_as:
kfree(dev->groups);
err_vq_groups:
vduse_dev_destroy(dev);
@@ -2084,7 +2219,7 @@ static int vduse_dev_init_vdpa(struct vduse_dev *dev, const char *name)
vdev = vdpa_alloc_device(struct vduse_vdpa, vdpa, dev->dev,
&vduse_vdpa_config_ops, &vduse_map_ops,
- dev->ngroups, 1, name, true);
+ dev->ngroups, dev->nas, name, true);
if (IS_ERR(vdev))
return PTR_ERR(vdev);
@@ -2113,11 +2248,20 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
return ret;
write_lock(&dev->domain_lock);
- if (!dev->as.domain)
- dev->as.domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
- dev->bounce_size);
+ ret = 0;
+
+ for (int i = 0; i < dev->nas; ++i) {
+ dev->as[i].domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
+ dev->bounce_size);
+ if (!dev->as[i].domain) {
+ ret = -ENOMEM;
+ for (int j = 0; j < i; ++j)
+ vduse_domain_destroy(dev->as[j].domain);
+ }
+ }
+
write_unlock(&dev->domain_lock);
- if (!dev->as.domain) {
+ if (ret == -ENOMEM) {
put_device(&dev->vdev->vdpa.dev);
return -ENOMEM;
}
@@ -2126,8 +2270,12 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
if (ret) {
put_device(&dev->vdev->vdpa.dev);
write_lock(&dev->domain_lock);
- vduse_domain_destroy(dev->as.domain);
- dev->as.domain = NULL;
+ for (int i = 0; i < dev->nas; i++) {
+ if (dev->as[i].domain) {
+ vduse_domain_destroy(dev->as[i].domain);
+ dev->as[i].domain = NULL;
+ }
+ }
write_unlock(&dev->domain_lock);
return ret;
}
diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h
index a3d51cf6df3a..da73c3f2c280 100644
--- a/include/uapi/linux/vduse.h
+++ b/include/uapi/linux/vduse.h
@@ -47,7 +47,8 @@ struct vduse_dev_config {
__u32 vq_num;
__u32 vq_align;
__u32 ngroups; /* if VDUSE_API_VERSION >= 1 */
- __u32 reserved[12];
+ __u32 nas; /* if VDUSE_API_VERSION >= 1 */
+ __u32 reserved[11];
__u32 config_size;
__u8 config[];
};
@@ -82,6 +83,18 @@ struct vduse_iotlb_entry {
__u8 perm;
};
+/**
+ * struct vduse_iotlb_entry_v2 - entry of IOTLB to describe one IOVA region in an ASID
+ * @v1: the original vduse_iotlb_entry
+ * @asid: address space ID of the IOVA region
+ *
+ * Structure used by VDUSE_IOTLB_GET_FD ioctl to find an overlapped IOVA region.
+ */
+struct vduse_iotlb_entry_v2 {
+ struct vduse_iotlb_entry v1;
+ __u32 asid;
+};
+
/*
* Find the first IOVA region that overlaps with the range [start, last]
* and return the corresponding file descriptor. Return -EINVAL means the
@@ -166,6 +179,16 @@ struct vduse_vq_state_packed {
__u16 last_used_idx;
};
+/**
+ * struct vduse_vq_group - virtqueue group
+ @ @group: Index of the virtqueue group
+ * @asid: Address space ID of the group
+ */
+struct vduse_vq_group_asid {
+ __u32 group;
+ __u32 asid;
+};
+
/**
* struct vduse_vq_info - information of a virtqueue
* @index: virtqueue index
@@ -225,6 +248,7 @@ struct vduse_vq_eventfd {
* @uaddr: start address of userspace memory, it must be aligned to page size
* @iova: start of the IOVA region
* @size: size of the IOVA region
+ * @asid: Address space ID of the IOVA region
* @reserved: for future use, needs to be initialized to zero
*
* Structure used by VDUSE_IOTLB_REG_UMEM and VDUSE_IOTLB_DEREG_UMEM
@@ -234,7 +258,8 @@ struct vduse_iova_umem {
__u64 uaddr;
__u64 iova;
__u64 size;
- __u64 reserved[3];
+ __u32 asid;
+ __u32 reserved[5];
};
/* Register userspace memory for IOVA regions */
@@ -248,6 +273,7 @@ struct vduse_iova_umem {
* @start: start of the IOVA region
* @last: last of the IOVA region
* @capability: capability of the IOVA region
+ * @asid: Address space ID of the IOVA region, only if device API version >= 1
* @reserved: for future use, needs to be initialized to zero
*
* Structure used by VDUSE_IOTLB_GET_INFO ioctl to get information of
@@ -258,7 +284,8 @@ struct vduse_iova_info {
__u64 last;
#define VDUSE_IOVA_CAP_UMEM (1 << 0)
__u64 capability;
- __u64 reserved[3];
+ __u32 asid; /* Only if device API version >= 1 */
+ __u32 reserved[5];
};
/*
@@ -280,6 +307,7 @@ enum vduse_req_type {
VDUSE_GET_VQ_STATE,
VDUSE_SET_STATUS,
VDUSE_UPDATE_IOTLB,
+ VDUSE_SET_VQ_GROUP_ASID,
};
/**
@@ -314,6 +342,18 @@ struct vduse_iova_range {
__u64 last;
};
+/**
+ * struct vduse_iova_range - IOVA range [start, last] if API_VERSION >= 1
+ * @start: start of the IOVA range
+ * @last: last of the IOVA range
+ * @asid: address space ID of the IOVA range
+ */
+struct vduse_iova_range_v2 {
+ __u64 start;
+ __u64 last;
+ __u32 asid;
+};
+
/**
* struct vduse_dev_request - control request
* @type: request type
@@ -322,6 +362,8 @@ struct vduse_iova_range {
* @vq_state: virtqueue state, only index field is available
* @s: device status
* @iova: IOVA range for updating
+ * @iova_v2: IOVA range for updating if API_VERSION >= 1
+ * @vq_group_asid: ASID of a virtqueue group
* @padding: padding
*
* Structure used by read(2) on /dev/vduse/$NAME.
@@ -334,6 +376,9 @@ struct vduse_dev_request {
struct vduse_vq_state vq_state;
struct vduse_dev_status s;
struct vduse_iova_range iova;
+ /* Following members only if vduse api version >= 1 */;
+ struct vduse_iova_range_v2 iova_v2;
+ struct vduse_vq_group_asid vq_group_asid;
__u32 padding[32];
};
};
--
2.51.0
^ permalink raw reply related [flat|nested] 27+ messages in thread* Re: [PATCH v2 6/7] vduse: add vq group asid support
2025-09-16 13:08 ` [PATCH v2 6/7] vduse: add vq group asid support Eugenio Pérez
@ 2025-09-17 8:56 ` Jason Wang
2025-09-17 16:40 ` Eugenio Perez Martin
0 siblings, 1 reply; 27+ messages in thread
From: Jason Wang @ 2025-09-17 8:56 UTC (permalink / raw)
To: Eugenio Pérez
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
>
> Add support for assigning Address Space Identifiers (ASIDs) to each VQ
> group. This enables mapping each group into a distinct memory space.
>
> Now that the driver can change ASID in the middle of operation, the
> domain that each vq address point is also protected by domain_lock.
>
> Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> ---
> v2:
> * Convert the use of mutex to rwlock.
>
> RFC v3:
> * Increase VDUSE_MAX_VQ_GROUPS to 0xffff (Jason). It was set to a lower
> value to reduce memory consumption, but vqs are already limited to
> that value and userspace VDUSE is able to allocate that many vqs.
> * Remove TODO about merging VDUSE_IOTLB_GET_FD ioctl with
> VDUSE_IOTLB_GET_INFO.
> * Use of array_index_nospec in VDUSE device ioctls.
> * Embed vduse_iotlb_entry into vduse_iotlb_entry_v2.
> * Move the umem mutex to asid struct so there is no contention between
> ASIDs.
>
> RFC v2:
> * Make iotlb entry the last one of vduse_iotlb_entry_v2 so the first
> part of the struct is the same.
> ---
> drivers/vdpa/vdpa_user/vduse_dev.c | 324 +++++++++++++++++++++--------
> include/uapi/linux/vduse.h | 51 ++++-
> 2 files changed, 284 insertions(+), 91 deletions(-)
>
> diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> index b45b1d22784f..06b7790380b7 100644
> --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> @@ -93,6 +93,7 @@ struct vduse_as {
> };
>
> struct vduse_vq_group_int {
> + struct vduse_iova_domain *domain;
> struct vduse_dev *dev;
This confuses me, I think it should be an asid. And the vduse_dev
pointer seems to be useless here.
> };
>
> @@ -100,7 +101,7 @@ struct vduse_dev {
> struct vduse_vdpa *vdev;
> struct device *dev;
> struct vduse_virtqueue **vqs;
> - struct vduse_as as;
> + struct vduse_as *as;
> char *name;
> struct mutex lock;
> spinlock_t msg_lock;
> @@ -128,6 +129,7 @@ struct vduse_dev {
> u32 vq_num;
> u32 vq_align;
> u32 ngroups;
> + u32 nas;
> struct vduse_vq_group_int *groups;
> unsigned int bounce_size;
> rwlock_t domain_lock;
> @@ -318,7 +320,7 @@ static int vduse_dev_set_status(struct vduse_dev *dev, u8 status)
> return vduse_dev_msg_sync(dev, &msg);
> }
>
> -static int vduse_dev_update_iotlb(struct vduse_dev *dev,
> +static int vduse_dev_update_iotlb(struct vduse_dev *dev, u32 asid,
> u64 start, u64 last)
> {
> struct vduse_dev_msg msg = { 0 };
> @@ -327,8 +329,14 @@ static int vduse_dev_update_iotlb(struct vduse_dev *dev,
> return -EINVAL;
>
> msg.req.type = VDUSE_UPDATE_IOTLB;
> - msg.req.iova.start = start;
> - msg.req.iova.last = last;
> + if (dev->api_version < VDUSE_API_VERSION_1) {
> + msg.req.iova.start = start;
> + msg.req.iova.last = last;
> + } else {
> + msg.req.iova_v2.start = start;
> + msg.req.iova_v2.last = last;
> + msg.req.iova_v2.asid = asid;
> + }
>
> return vduse_dev_msg_sync(dev, &msg);
> }
> @@ -440,14 +448,29 @@ static __poll_t vduse_dev_poll(struct file *file, poll_table *wait)
> return mask;
> }
>
> +/* Force set the asid to a vq group without a message to the VDUSE device */
> +static void vduse_set_group_asid_nomsg(struct vduse_dev *dev,
> + unsigned int group, unsigned int asid)
> +{
> + write_lock(&dev->domain_lock);
> + dev->groups[group].domain = dev->as[asid].domain;
I think it would be better to stick the group->as an indirection which
should be .
dev->groups.asid = asid;
Or
dev->group->as = as;
> + write_unlock(&dev->domain_lock);
> +}
> +
> static void vduse_dev_reset(struct vduse_dev *dev)
> {
> int i;
> - struct vduse_iova_domain *domain = dev->as.domain;
>
> /* The coherent mappings are handled in vduse_dev_free_coherent() */
> - if (domain && domain->bounce_map)
> - vduse_domain_reset_bounce_map(domain);
> + for (i = 0; i < dev->nas; i++) {
> + struct vduse_iova_domain *domain = dev->as[i].domain;
> +
> + if (domain && domain->bounce_map)
> + vduse_domain_reset_bounce_map(domain);
> + }
> +
> + for (i = 0; i < dev->ngroups; i++)
> + vduse_set_group_asid_nomsg(dev, i, 0);
>
> down_write(&dev->rwsem);
>
> @@ -621,6 +644,29 @@ static union virtio_map vduse_get_vq_map(struct vdpa_device *vdpa, u16 idx)
> return ret;
> }
>
> +static int vduse_set_group_asid(struct vdpa_device *vdpa, unsigned int group,
> + unsigned int asid)
> +{
> + struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> + struct vduse_dev_msg msg = { 0 };
> + int r;
> +
> + if (dev->api_version < VDUSE_API_VERSION_1 ||
> + group >= dev->ngroups || asid >= dev->nas)
> + return -EINVAL;
> +
> + msg.req.type = VDUSE_SET_VQ_GROUP_ASID;
> + msg.req.vq_group_asid.group = group;
> + msg.req.vq_group_asid.asid = asid;
> +
> + r = vduse_dev_msg_sync(dev, &msg);
> + if (r < 0)
> + return r;
> +
> + vduse_set_group_asid_nomsg(dev, group, asid);
> + return 0;
> +}
> +
> static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
> struct vdpa_vq_state *state)
> {
> @@ -792,13 +838,13 @@ static int vduse_vdpa_set_map(struct vdpa_device *vdpa,
> struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> int ret;
>
> - ret = vduse_domain_set_map(dev->as.domain, iotlb);
> + ret = vduse_domain_set_map(dev->as[asid].domain, iotlb);
> if (ret)
> return ret;
>
> - ret = vduse_dev_update_iotlb(dev, 0ULL, ULLONG_MAX);
> + ret = vduse_dev_update_iotlb(dev, asid, 0ULL, ULLONG_MAX);
> if (ret) {
> - vduse_domain_clear_map(dev->as.domain, iotlb);
> + vduse_domain_clear_map(dev->as[asid].domain, iotlb);
> return ret;
> }
>
> @@ -841,6 +887,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
> .get_vq_affinity = vduse_vdpa_get_vq_affinity,
> .reset = vduse_vdpa_reset,
> .set_map = vduse_vdpa_set_map,
> + .set_group_asid = vduse_set_group_asid,
> .get_vq_map = vduse_get_vq_map,
> .free = vduse_vdpa_free,
> };
> @@ -850,9 +897,12 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
> enum dma_data_direction dir)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->as.domain;
> + struct vduse_iova_domain *domain;
>
> + read_lock(&vdev->domain_lock);
> + domain = token.group->domain;
> vduse_domain_sync_single_for_device(domain, dma_addr, size, dir);
> + read_unlock(&vdev->domain_lock);
> }
>
> static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> @@ -860,9 +910,12 @@ static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> enum dma_data_direction dir)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->as.domain;
> + struct vduse_iova_domain *domain;
>
> + read_lock(&vdev->domain_lock);
> + domain = token.group->domain;
I think the domain is better fetched via vduse_as.
> vduse_domain_sync_single_for_cpu(domain, dma_addr, size, dir);
> + read_unlock(&vdev->domain_lock);
> }
>
> static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> @@ -871,9 +924,15 @@ static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> unsigned long attrs)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->as.domain;
> + struct vduse_iova_domain *domain;
> + dma_addr_t r;
>
> - return vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> + read_lock(&vdev->domain_lock);
> + domain = token.group->domain;
> + r = vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> + read_unlock(&vdev->domain_lock);
> +
> + return r;
> }
>
> static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> @@ -881,27 +940,31 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> unsigned long attrs)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->as.domain;
> + struct vduse_iova_domain *domain;
>
> - return vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> + read_lock(&vdev->domain_lock);
> + domain = token.group->domain;
> + vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> + read_unlock(&vdev->domain_lock);
> }
>
> static void *vduse_dev_alloc_coherent(union virtio_map token, size_t size,
> dma_addr_t *dma_addr, gfp_t flag)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->as.domain;
> + struct vduse_iova_domain *domain;
> unsigned long iova;
> - void *addr;
> + void *addr = NULL;
>
> *dma_addr = DMA_MAPPING_ERROR;
> + read_lock(&vdev->domain_lock);
> + domain = token.group->domain;
> addr = vduse_domain_alloc_coherent(domain, size,
> (dma_addr_t *)&iova, flag);
> - if (!addr)
> - return NULL;
> -
> - *dma_addr = (dma_addr_t)iova;
> + if (addr)
> + *dma_addr = (dma_addr_t)iova;
>
> + read_unlock(&vdev->domain_lock);
> return addr;
> }
>
> @@ -910,17 +973,26 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
> unsigned long attrs)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->as.domain;
> + struct vduse_iova_domain *domain;
>
> + read_lock(&vdev->domain_lock);
> + domain = token.group->domain;
> vduse_domain_free_coherent(domain, size, vaddr, dma_addr, attrs);
> + read_unlock(&vdev->domain_lock);
> }
>
> static bool vduse_dev_need_sync(union virtio_map token, dma_addr_t dma_addr)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->as.domain;
> + struct vduse_iova_domain *domain;
> + size_t bounce_size;
>
> - return dma_addr < domain->bounce_size;
> + read_lock(&vdev->domain_lock);
> + domain = token.group->domain;
> + bounce_size = domain->bounce_size;
> + read_unlock(&vdev->domain_lock);
> +
> + return dma_addr < bounce_size;
> }
>
> static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> @@ -933,9 +1005,15 @@ static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> static size_t vduse_dev_max_mapping_size(union virtio_map token)
> {
> struct vduse_dev *vdev = token.group->dev;
> - struct vduse_iova_domain *domain = vdev->as.domain;
> + struct vduse_iova_domain *domain;
> + size_t bounce_size;
> +
> + read_lock(&vdev->domain_lock);
> + domain = token.group->domain;
> + bounce_size = domain->bounce_size;
> + read_unlock(&vdev->domain_lock);
>
> - return domain->bounce_size;
> + return bounce_size;
> }
>
> static const struct virtio_map_ops vduse_map_ops = {
> @@ -1075,39 +1153,40 @@ static int vduse_dev_queue_irq_work(struct vduse_dev *dev,
> return ret;
> }
>
> -static int vduse_dev_dereg_umem(struct vduse_dev *dev,
> +static int vduse_dev_dereg_umem(struct vduse_dev *dev, u32 asid,
> u64 iova, u64 size)
> {
> int ret;
>
> - mutex_lock(&dev->as.mem_lock);
> + mutex_lock(&dev->as[asid].mem_lock);
> ret = -ENOENT;
> - if (!dev->as.umem)
> + if (!dev->as[asid].umem)
> goto unlock;
>
> ret = -EINVAL;
> - if (!dev->as.domain)
> + if (!dev->as[asid].domain)
> goto unlock;
>
> - if (dev->as.umem->iova != iova || size != dev->as.domain->bounce_size)
> + if (dev->as[asid].umem->iova != iova ||
> + size != dev->as[asid].domain->bounce_size)
> goto unlock;
>
> - vduse_domain_remove_user_bounce_pages(dev->as.domain);
> - unpin_user_pages_dirty_lock(dev->as.umem->pages,
> - dev->as.umem->npages, true);
> - atomic64_sub(dev->as.umem->npages, &dev->as.umem->mm->pinned_vm);
> - mmdrop(dev->as.umem->mm);
> - vfree(dev->as.umem->pages);
> - kfree(dev->as.umem);
> - dev->as.umem = NULL;
> + vduse_domain_remove_user_bounce_pages(dev->as[asid].domain);
> + unpin_user_pages_dirty_lock(dev->as[asid].umem->pages,
> + dev->as[asid].umem->npages, true);
> + atomic64_sub(dev->as[asid].umem->npages, &dev->as[asid].umem->mm->pinned_vm);
> + mmdrop(dev->as[asid].umem->mm);
> + vfree(dev->as[asid].umem->pages);
> + kfree(dev->as[asid].umem);
> + dev->as[asid].umem = NULL;
We can avoid those changeset if we do those in the previous correctly
as it said it would make as an array.
> ret = 0;
> unlock:
> - mutex_unlock(&dev->as.mem_lock);
> + mutex_unlock(&dev->as[asid].mem_lock);
> return ret;
> }
>
> static int vduse_dev_reg_umem(struct vduse_dev *dev,
> - u64 iova, u64 uaddr, u64 size)
> + u32 asid, u64 iova, u64 uaddr, u64 size)
> {
> struct page **page_list = NULL;
> struct vduse_umem *umem = NULL;
> @@ -1115,14 +1194,14 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> unsigned long npages, lock_limit;
> int ret;
>
> - if (!dev->as.domain || !dev->as.domain->bounce_map ||
> - size != dev->as.domain->bounce_size ||
> + if (!dev->as[asid].domain || !dev->as[asid].domain->bounce_map ||
> + size != dev->as[asid].domain->bounce_size ||
> iova != 0 || uaddr & ~PAGE_MASK)
> return -EINVAL;
>
> - mutex_lock(&dev->as.mem_lock);
> + mutex_lock(&dev->as[asid].mem_lock);
> ret = -EEXIST;
> - if (dev->as.umem)
> + if (dev->as[asid].umem)
> goto unlock;
>
> ret = -ENOMEM;
> @@ -1146,7 +1225,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> goto out;
> }
>
> - ret = vduse_domain_add_user_bounce_pages(dev->as.domain,
> + ret = vduse_domain_add_user_bounce_pages(dev->as[asid].domain,
> page_list, pinned);
> if (ret)
> goto out;
> @@ -1159,7 +1238,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> umem->mm = current->mm;
> mmgrab(current->mm);
>
> - dev->as.umem = umem;
> + dev->as[asid].umem = umem;
> out:
> if (ret && pinned > 0)
> unpin_user_pages(page_list, pinned);
> @@ -1170,7 +1249,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> vfree(page_list);
> kfree(umem);
> }
> - mutex_unlock(&dev->as.mem_lock);
> + mutex_unlock(&dev->as[asid].mem_lock);
> return ret;
> }
>
> @@ -1202,47 +1281,66 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
>
> switch (cmd) {
> case VDUSE_IOTLB_GET_FD: {
> - struct vduse_iotlb_entry entry;
> + struct vduse_iotlb_entry_v2 entry;
> struct vhost_iotlb_map *map;
> struct vdpa_map_file *map_file;
> struct file *f = NULL;
> + u32 asid;
>
> ret = -EFAULT;
> - if (copy_from_user(&entry, argp, sizeof(entry)))
> - break;
> + if (dev->api_version >= VDUSE_API_VERSION_1) {
> + if (copy_from_user(&entry, argp, sizeof(entry)))
> + break;
> + } else {
> + entry.asid = 0;
> + if (copy_from_user(&entry.v1, argp,
> + sizeof(entry.v1)))
> + break;
> + }
>
> ret = -EINVAL;
> - if (entry.start > entry.last)
> + if (entry.v1.start > entry.v1.last)
> + break;
> +
> + if (entry.asid >= dev->nas)
> break;
>
> read_lock(&dev->domain_lock);
> - if (!dev->as.domain) {
> + asid = array_index_nospec(entry.asid, dev->nas);
> + if (!dev->as[asid].domain) {
> read_unlock(&dev->domain_lock);
> break;
> }
> - spin_lock(&dev->as.domain->iotlb_lock);
> - map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
> - entry.start, entry.last);
> + spin_lock(&dev->as[asid].domain->iotlb_lock);
> + map = vhost_iotlb_itree_first(dev->as[asid].domain->iotlb,
> + entry.v1.start, entry.v1.last);
> if (map) {
> map_file = (struct vdpa_map_file *)map->opaque;
> f = get_file(map_file->file);
> - entry.offset = map_file->offset;
> - entry.start = map->start;
> - entry.last = map->last;
> - entry.perm = map->perm;
> + entry.v1.offset = map_file->offset;
> + entry.v1.start = map->start;
> + entry.v1.last = map->last;
> + entry.v1.perm = map->perm;
> }
> - spin_unlock(&dev->as.domain->iotlb_lock);
> + spin_unlock(&dev->as[asid].domain->iotlb_lock);
> read_unlock(&dev->domain_lock);
> ret = -EINVAL;
> if (!f)
> break;
>
> ret = -EFAULT;
> - if (copy_to_user(argp, &entry, sizeof(entry))) {
> + if (dev->api_version >= VDUSE_API_VERSION_1)
> + ret = copy_to_user(argp, &entry,
> + sizeof(entry));
> + else
> + ret = copy_to_user(argp, &entry.v1,
> + sizeof(entry.v1));
> +
> + if (ret) {
> fput(f);
> break;
> }
> - ret = receive_fd(f, NULL, perm_to_file_flags(entry.perm));
> + ret = receive_fd(f, NULL, perm_to_file_flags(entry.v1.perm));
Nit: if we copy_from_user() twice and stick entry for v1 format, we
can avoid a lot of lines of changes.
> fput(f);
> break;
> }
> @@ -1384,6 +1482,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> }
> case VDUSE_IOTLB_REG_UMEM: {
> struct vduse_iova_umem umem;
> + u32 asid;
>
> ret = -EFAULT;
> if (copy_from_user(&umem, argp, sizeof(umem)))
> @@ -1391,17 +1490,21 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
>
> ret = -EINVAL;
> if (!is_mem_zero((const char *)umem.reserved,
> - sizeof(umem.reserved)))
> + sizeof(umem.reserved)) ||
> + (dev->api_version < VDUSE_API_VERSION_1 &&
> + umem.asid != 0) || umem.asid >= dev->nas)
> break;
>
> write_lock(&dev->domain_lock);
> - ret = vduse_dev_reg_umem(dev, umem.iova,
> + asid = array_index_nospec(umem.asid, dev->nas);
> + ret = vduse_dev_reg_umem(dev, asid, umem.iova,
> umem.uaddr, umem.size);
> write_unlock(&dev->domain_lock);
> break;
> }
> case VDUSE_IOTLB_DEREG_UMEM: {
> struct vduse_iova_umem umem;
> + u32 asid;
>
> ret = -EFAULT;
> if (copy_from_user(&umem, argp, sizeof(umem)))
> @@ -1409,10 +1512,15 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
>
> ret = -EINVAL;
> if (!is_mem_zero((const char *)umem.reserved,
> - sizeof(umem.reserved)))
> + sizeof(umem.reserved)) ||
> + (dev->api_version < VDUSE_API_VERSION_1 &&
> + umem.asid != 0) ||
> + umem.asid >= dev->nas)
> break;
> +
> write_lock(&dev->domain_lock);
> - ret = vduse_dev_dereg_umem(dev, umem.iova,
> + asid = array_index_nospec(umem.asid, dev->nas);
> + ret = vduse_dev_dereg_umem(dev, asid, umem.iova,
> umem.size);
> write_unlock(&dev->domain_lock);
> break;
> @@ -1420,6 +1528,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> case VDUSE_IOTLB_GET_INFO: {
> struct vduse_iova_info info;
> struct vhost_iotlb_map *map;
> + u32 asid;
>
> ret = -EFAULT;
> if (copy_from_user(&info, argp, sizeof(info)))
> @@ -1433,23 +1542,31 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> sizeof(info.reserved)))
> break;
>
> + if (dev->api_version < VDUSE_API_VERSION_1) {
> + if (info.asid)
> + break;
> + } else if (info.asid >= dev->nas)
> + break;
> +
> read_lock(&dev->domain_lock);
> - if (!dev->as.domain) {
> + asid = array_index_nospec(info.asid, dev->nas);
> + if (!dev->as[asid].domain) {
> read_unlock(&dev->domain_lock);
> break;
> }
> - spin_lock(&dev->as.domain->iotlb_lock);
> - map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
> + spin_lock(&dev->as[asid].domain->iotlb_lock);
> + map = vhost_iotlb_itree_first(dev->as[asid].domain->iotlb,
> info.start, info.last);
> if (map) {
> info.start = map->start;
> info.last = map->last;
> info.capability = 0;
> - if (dev->as.domain->bounce_map && map->start == 0 &&
> - map->last == dev->as.domain->bounce_size - 1)
> + if (dev->as[asid].domain->bounce_map &&
> + map->start == 0 &&
> + map->last == dev->as[asid].domain->bounce_size - 1)
> info.capability |= VDUSE_IOVA_CAP_UMEM;
> }
> - spin_unlock(&dev->as.domain->iotlb_lock);
> + spin_unlock(&dev->as[asid].domain->iotlb_lock);
> read_unlock(&dev->domain_lock);
> if (!map)
> break;
> @@ -1474,8 +1591,10 @@ static int vduse_dev_release(struct inode *inode, struct file *file)
> struct vduse_dev *dev = file->private_data;
>
> write_lock(&dev->domain_lock);
> - if (dev->as.domain)
> - vduse_dev_dereg_umem(dev, 0, dev->as.domain->bounce_size);
> + for (int i = 0; i < dev->nas; i++)
> + if (dev->as[i].domain)
> + vduse_dev_dereg_umem(dev, i, 0,
> + dev->as[i].domain->bounce_size);
> write_unlock(&dev->domain_lock);
> spin_lock(&dev->msg_lock);
> /* Make sure the inflight messages can processed after reconncection */
> @@ -1694,7 +1813,6 @@ static struct vduse_dev *vduse_dev_create(void)
> return NULL;
>
> mutex_init(&dev->lock);
> - mutex_init(&dev->as.mem_lock);
> rwlock_init(&dev->domain_lock);
> spin_lock_init(&dev->msg_lock);
> INIT_LIST_HEAD(&dev->send_list);
> @@ -1745,8 +1863,11 @@ static int vduse_destroy_dev(char *name)
> idr_remove(&vduse_idr, dev->minor);
> kvfree(dev->config);
> vduse_dev_deinit_vqs(dev);
> - if (dev->as.domain)
> - vduse_domain_destroy(dev->as.domain);
> + for (int i = 0; i < dev->nas; i++) {
> + if (dev->as[i].domain)
> + vduse_domain_destroy(dev->as[i].domain);
> + }
> + kfree(dev->as);
> kfree(dev->name);
> kfree(dev->groups);
> vduse_dev_destroy(dev);
> @@ -1793,12 +1914,16 @@ static bool vduse_validate_config(struct vduse_dev_config *config,
> sizeof(config->reserved)))
> return false;
>
> - if (api_version < VDUSE_API_VERSION_1 && config->ngroups)
> + if (api_version < VDUSE_API_VERSION_1 &&
> + (config->ngroups || config->nas))
> return false;
>
> if (api_version >= VDUSE_API_VERSION_1 && config->ngroups > 0xffff)
> return false;
>
> + if (api_version >= VDUSE_API_VERSION_1 && config->nas > 0xffff)
> + return false;
> +
> if (config->vq_align > PAGE_SIZE)
> return false;
>
> @@ -1862,7 +1987,8 @@ static ssize_t bounce_size_store(struct device *device,
>
> ret = -EPERM;
> write_lock(&dev->domain_lock);
> - if (dev->as.domain)
> + /* Assuming that if the first domain is allocated, all are allocated */
> + if (dev->as[0].domain)
> goto unlock;
>
> ret = kstrtouint(buf, 10, &bounce_size);
> @@ -1923,6 +2049,13 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> for (u32 i = 0; i < dev->ngroups; ++i)
> dev->groups[i].dev = dev;
>
> + dev->nas = (dev->api_version < 1) ? 1 : (config->nas ?: 1);
> + dev->as = kcalloc(dev->nas, sizeof(dev->as[0]), GFP_KERNEL);
> + if (!dev->as)
> + goto err_as;
> + for (int i = 0; i < dev->nas; i++)
> + mutex_init(&dev->as[i].mem_lock);
> +
> dev->name = kstrdup(config->name, GFP_KERNEL);
> if (!dev->name)
> goto err_str;
> @@ -1959,6 +2092,8 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> err_idr:
> kfree(dev->name);
> err_str:
> + kfree(dev->as);
> +err_as:
> kfree(dev->groups);
> err_vq_groups:
> vduse_dev_destroy(dev);
> @@ -2084,7 +2219,7 @@ static int vduse_dev_init_vdpa(struct vduse_dev *dev, const char *name)
>
> vdev = vdpa_alloc_device(struct vduse_vdpa, vdpa, dev->dev,
> &vduse_vdpa_config_ops, &vduse_map_ops,
> - dev->ngroups, 1, name, true);
> + dev->ngroups, dev->nas, name, true);
> if (IS_ERR(vdev))
> return PTR_ERR(vdev);
>
> @@ -2113,11 +2248,20 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> return ret;
>
> write_lock(&dev->domain_lock);
> - if (!dev->as.domain)
> - dev->as.domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
> - dev->bounce_size);
> + ret = 0;
> +
> + for (int i = 0; i < dev->nas; ++i) {
> + dev->as[i].domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
> + dev->bounce_size);
> + if (!dev->as[i].domain) {
> + ret = -ENOMEM;
> + for (int j = 0; j < i; ++j)
> + vduse_domain_destroy(dev->as[j].domain);
> + }
> + }
> +
> write_unlock(&dev->domain_lock);
> - if (!dev->as.domain) {
> + if (ret == -ENOMEM) {
> put_device(&dev->vdev->vdpa.dev);
> return -ENOMEM;
> }
> @@ -2126,8 +2270,12 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> if (ret) {
> put_device(&dev->vdev->vdpa.dev);
> write_lock(&dev->domain_lock);
> - vduse_domain_destroy(dev->as.domain);
> - dev->as.domain = NULL;
> + for (int i = 0; i < dev->nas; i++) {
> + if (dev->as[i].domain) {
> + vduse_domain_destroy(dev->as[i].domain);
> + dev->as[i].domain = NULL;
> + }
> + }
> write_unlock(&dev->domain_lock);
> return ret;
> }
> diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h
> index a3d51cf6df3a..da73c3f2c280 100644
> --- a/include/uapi/linux/vduse.h
> +++ b/include/uapi/linux/vduse.h
> @@ -47,7 +47,8 @@ struct vduse_dev_config {
> __u32 vq_num;
> __u32 vq_align;
> __u32 ngroups; /* if VDUSE_API_VERSION >= 1 */
> - __u32 reserved[12];
> + __u32 nas; /* if VDUSE_API_VERSION >= 1 */
> + __u32 reserved[11];
> __u32 config_size;
> __u8 config[];
> };
> @@ -82,6 +83,18 @@ struct vduse_iotlb_entry {
> __u8 perm;
> };
>
> +/**
> + * struct vduse_iotlb_entry_v2 - entry of IOTLB to describe one IOVA region in an ASID
> + * @v1: the original vduse_iotlb_entry
> + * @asid: address space ID of the IOVA region
> + *
> + * Structure used by VDUSE_IOTLB_GET_FD ioctl to find an overlapped IOVA region.
> + */
> +struct vduse_iotlb_entry_v2 {
> + struct vduse_iotlb_entry v1;
> + __u32 asid;
> +};
> +
> /*
> * Find the first IOVA region that overlaps with the range [start, last]
> * and return the corresponding file descriptor. Return -EINVAL means the
> @@ -166,6 +179,16 @@ struct vduse_vq_state_packed {
> __u16 last_used_idx;
> };
>
> +/**
> + * struct vduse_vq_group - virtqueue group
> + @ @group: Index of the virtqueue group
> + * @asid: Address space ID of the group
> + */
> +struct vduse_vq_group_asid {
> + __u32 group;
> + __u32 asid;
> +};
> +
> /**
> * struct vduse_vq_info - information of a virtqueue
> * @index: virtqueue index
> @@ -225,6 +248,7 @@ struct vduse_vq_eventfd {
> * @uaddr: start address of userspace memory, it must be aligned to page size
> * @iova: start of the IOVA region
> * @size: size of the IOVA region
> + * @asid: Address space ID of the IOVA region
> * @reserved: for future use, needs to be initialized to zero
> *
> * Structure used by VDUSE_IOTLB_REG_UMEM and VDUSE_IOTLB_DEREG_UMEM
> @@ -234,7 +258,8 @@ struct vduse_iova_umem {
> __u64 uaddr;
> __u64 iova;
> __u64 size;
> - __u64 reserved[3];
> + __u32 asid;
> + __u32 reserved[5];
> };
>
> /* Register userspace memory for IOVA regions */
> @@ -248,6 +273,7 @@ struct vduse_iova_umem {
> * @start: start of the IOVA region
> * @last: last of the IOVA region
> * @capability: capability of the IOVA region
> + * @asid: Address space ID of the IOVA region, only if device API version >= 1
> * @reserved: for future use, needs to be initialized to zero
> *
> * Structure used by VDUSE_IOTLB_GET_INFO ioctl to get information of
> @@ -258,7 +284,8 @@ struct vduse_iova_info {
> __u64 last;
> #define VDUSE_IOVA_CAP_UMEM (1 << 0)
> __u64 capability;
> - __u64 reserved[3];
> + __u32 asid; /* Only if device API version >= 1 */
> + __u32 reserved[5];
> };
>
> /*
> @@ -280,6 +307,7 @@ enum vduse_req_type {
> VDUSE_GET_VQ_STATE,
> VDUSE_SET_STATUS,
> VDUSE_UPDATE_IOTLB,
> + VDUSE_SET_VQ_GROUP_ASID,
> };
>
> /**
> @@ -314,6 +342,18 @@ struct vduse_iova_range {
> __u64 last;
> };
>
> +/**
> + * struct vduse_iova_range - IOVA range [start, last] if API_VERSION >= 1
> + * @start: start of the IOVA range
> + * @last: last of the IOVA range
> + * @asid: address space ID of the IOVA range
> + */
> +struct vduse_iova_range_v2 {
> + __u64 start;
> + __u64 last;
> + __u32 asid;
> +};
> +
> /**
> * struct vduse_dev_request - control request
> * @type: request type
> @@ -322,6 +362,8 @@ struct vduse_iova_range {
> * @vq_state: virtqueue state, only index field is available
> * @s: device status
> * @iova: IOVA range for updating
> + * @iova_v2: IOVA range for updating if API_VERSION >= 1
> + * @vq_group_asid: ASID of a virtqueue group
> * @padding: padding
> *
> * Structure used by read(2) on /dev/vduse/$NAME.
> @@ -334,6 +376,9 @@ struct vduse_dev_request {
> struct vduse_vq_state vq_state;
> struct vduse_dev_status s;
> struct vduse_iova_range iova;
> + /* Following members only if vduse api version >= 1 */;
> + struct vduse_iova_range_v2 iova_v2;
> + struct vduse_vq_group_asid vq_group_asid;
> __u32 padding[32];
> };
> };
> --
> 2.51.0
>
Thanks
^ permalink raw reply [flat|nested] 27+ messages in thread* Re: [PATCH v2 6/7] vduse: add vq group asid support
2025-09-17 8:56 ` Jason Wang
@ 2025-09-17 16:40 ` Eugenio Perez Martin
2025-09-18 6:07 ` Jason Wang
2025-09-18 11:21 ` Eugenio Perez Martin
0 siblings, 2 replies; 27+ messages in thread
From: Eugenio Perez Martin @ 2025-09-17 16:40 UTC (permalink / raw)
To: Jason Wang
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Wed, Sep 17, 2025 at 10:56 AM Jason Wang <jasowang@redhat.com> wrote:
>
> On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
> >
> > Add support for assigning Address Space Identifiers (ASIDs) to each VQ
> > group. This enables mapping each group into a distinct memory space.
> >
> > Now that the driver can change ASID in the middle of operation, the
> > domain that each vq address point is also protected by domain_lock.
> >
> > Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> > ---
> > v2:
> > * Convert the use of mutex to rwlock.
> >
> > RFC v3:
> > * Increase VDUSE_MAX_VQ_GROUPS to 0xffff (Jason). It was set to a lower
> > value to reduce memory consumption, but vqs are already limited to
> > that value and userspace VDUSE is able to allocate that many vqs.
> > * Remove TODO about merging VDUSE_IOTLB_GET_FD ioctl with
> > VDUSE_IOTLB_GET_INFO.
> > * Use of array_index_nospec in VDUSE device ioctls.
> > * Embed vduse_iotlb_entry into vduse_iotlb_entry_v2.
> > * Move the umem mutex to asid struct so there is no contention between
> > ASIDs.
> >
> > RFC v2:
> > * Make iotlb entry the last one of vduse_iotlb_entry_v2 so the first
> > part of the struct is the same.
> > ---
> > drivers/vdpa/vdpa_user/vduse_dev.c | 324 +++++++++++++++++++++--------
> > include/uapi/linux/vduse.h | 51 ++++-
> > 2 files changed, 284 insertions(+), 91 deletions(-)
> >
> > diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> > index b45b1d22784f..06b7790380b7 100644
> > --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> > +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> > @@ -93,6 +93,7 @@ struct vduse_as {
> > };
> >
> > struct vduse_vq_group_int {
> > + struct vduse_iova_domain *domain;
> > struct vduse_dev *dev;
>
> This confuses me, I think it should be an asid. And the vduse_dev
> pointer seems to be useless here.
>
The *dev pointer is used to take the rwlock, in case the vhost driver
calls VHOST_VDPA_SET_GROUP_ASID (or equivalent) at the same time
vduse_dev_sync_single_for_device (or _for_cpu, or equivalent) run.
> > };
> >
> > @@ -100,7 +101,7 @@ struct vduse_dev {
> > struct vduse_vdpa *vdev;
> > struct device *dev;
> > struct vduse_virtqueue **vqs;
> > - struct vduse_as as;
> > + struct vduse_as *as;
> > char *name;
> > struct mutex lock;
> > spinlock_t msg_lock;
> > @@ -128,6 +129,7 @@ struct vduse_dev {
> > u32 vq_num;
> > u32 vq_align;
> > u32 ngroups;
> > + u32 nas;
> > struct vduse_vq_group_int *groups;
> > unsigned int bounce_size;
> > rwlock_t domain_lock;
> > @@ -318,7 +320,7 @@ static int vduse_dev_set_status(struct vduse_dev *dev, u8 status)
> > return vduse_dev_msg_sync(dev, &msg);
> > }
> >
> > -static int vduse_dev_update_iotlb(struct vduse_dev *dev,
> > +static int vduse_dev_update_iotlb(struct vduse_dev *dev, u32 asid,
> > u64 start, u64 last)
> > {
> > struct vduse_dev_msg msg = { 0 };
> > @@ -327,8 +329,14 @@ static int vduse_dev_update_iotlb(struct vduse_dev *dev,
> > return -EINVAL;
> >
> > msg.req.type = VDUSE_UPDATE_IOTLB;
> > - msg.req.iova.start = start;
> > - msg.req.iova.last = last;
> > + if (dev->api_version < VDUSE_API_VERSION_1) {
> > + msg.req.iova.start = start;
> > + msg.req.iova.last = last;
> > + } else {
> > + msg.req.iova_v2.start = start;
> > + msg.req.iova_v2.last = last;
> > + msg.req.iova_v2.asid = asid;
> > + }
> >
> > return vduse_dev_msg_sync(dev, &msg);
> > }
> > @@ -440,14 +448,29 @@ static __poll_t vduse_dev_poll(struct file *file, poll_table *wait)
> > return mask;
> > }
> >
> > +/* Force set the asid to a vq group without a message to the VDUSE device */
> > +static void vduse_set_group_asid_nomsg(struct vduse_dev *dev,
> > + unsigned int group, unsigned int asid)
> > +{
> > + write_lock(&dev->domain_lock);
> > + dev->groups[group].domain = dev->as[asid].domain;
>
> I think it would be better to stick the group->as an indirection which
> should be .
>
> dev->groups.asid = asid;
>
> Or
>
> dev->group->as = as;
>
That involves an extra memory jump for functions that may be in the
hot path. I've not profiled it, but I'm ok with changing it that way
if you prefer.
> > + write_unlock(&dev->domain_lock);
> > +}
> > +
> > static void vduse_dev_reset(struct vduse_dev *dev)
> > {
> > int i;
> > - struct vduse_iova_domain *domain = dev->as.domain;
> >
> > /* The coherent mappings are handled in vduse_dev_free_coherent() */
> > - if (domain && domain->bounce_map)
> > - vduse_domain_reset_bounce_map(domain);
> > + for (i = 0; i < dev->nas; i++) {
> > + struct vduse_iova_domain *domain = dev->as[i].domain;
> > +
> > + if (domain && domain->bounce_map)
> > + vduse_domain_reset_bounce_map(domain);
> > + }
> > +
> > + for (i = 0; i < dev->ngroups; i++)
> > + vduse_set_group_asid_nomsg(dev, i, 0);
> >
> > down_write(&dev->rwsem);
> >
> > @@ -621,6 +644,29 @@ static union virtio_map vduse_get_vq_map(struct vdpa_device *vdpa, u16 idx)
> > return ret;
> > }
> >
> > +static int vduse_set_group_asid(struct vdpa_device *vdpa, unsigned int group,
> > + unsigned int asid)
> > +{
> > + struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > + struct vduse_dev_msg msg = { 0 };
> > + int r;
> > +
> > + if (dev->api_version < VDUSE_API_VERSION_1 ||
> > + group >= dev->ngroups || asid >= dev->nas)
> > + return -EINVAL;
> > +
> > + msg.req.type = VDUSE_SET_VQ_GROUP_ASID;
> > + msg.req.vq_group_asid.group = group;
> > + msg.req.vq_group_asid.asid = asid;
> > +
> > + r = vduse_dev_msg_sync(dev, &msg);
> > + if (r < 0)
> > + return r;
> > +
> > + vduse_set_group_asid_nomsg(dev, group, asid);
> > + return 0;
> > +}
> > +
> > static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
> > struct vdpa_vq_state *state)
> > {
> > @@ -792,13 +838,13 @@ static int vduse_vdpa_set_map(struct vdpa_device *vdpa,
> > struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > int ret;
> >
> > - ret = vduse_domain_set_map(dev->as.domain, iotlb);
> > + ret = vduse_domain_set_map(dev->as[asid].domain, iotlb);
> > if (ret)
> > return ret;
> >
> > - ret = vduse_dev_update_iotlb(dev, 0ULL, ULLONG_MAX);
> > + ret = vduse_dev_update_iotlb(dev, asid, 0ULL, ULLONG_MAX);
> > if (ret) {
> > - vduse_domain_clear_map(dev->as.domain, iotlb);
> > + vduse_domain_clear_map(dev->as[asid].domain, iotlb);
> > return ret;
> > }
> >
> > @@ -841,6 +887,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
> > .get_vq_affinity = vduse_vdpa_get_vq_affinity,
> > .reset = vduse_vdpa_reset,
> > .set_map = vduse_vdpa_set_map,
> > + .set_group_asid = vduse_set_group_asid,
> > .get_vq_map = vduse_get_vq_map,
> > .free = vduse_vdpa_free,
> > };
> > @@ -850,9 +897,12 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
> > enum dma_data_direction dir)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->as.domain;
> > + struct vduse_iova_domain *domain;
> >
> > + read_lock(&vdev->domain_lock);
> > + domain = token.group->domain;
> > vduse_domain_sync_single_for_device(domain, dma_addr, size, dir);
> > + read_unlock(&vdev->domain_lock);
> > }
> >
> > static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> > @@ -860,9 +910,12 @@ static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> > enum dma_data_direction dir)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->as.domain;
> > + struct vduse_iova_domain *domain;
> >
> > + read_lock(&vdev->domain_lock);
> > + domain = token.group->domain;
>
> I think the domain is better fetched via vduse_as.
>
> > vduse_domain_sync_single_for_cpu(domain, dma_addr, size, dir);
> > + read_unlock(&vdev->domain_lock);
> > }
> >
> > static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> > @@ -871,9 +924,15 @@ static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> > unsigned long attrs)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->as.domain;
> > + struct vduse_iova_domain *domain;
> > + dma_addr_t r;
> >
> > - return vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> > + read_lock(&vdev->domain_lock);
> > + domain = token.group->domain;
> > + r = vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> > + read_unlock(&vdev->domain_lock);
> > +
> > + return r;
> > }
> >
> > static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> > @@ -881,27 +940,31 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> > unsigned long attrs)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->as.domain;
> > + struct vduse_iova_domain *domain;
> >
> > - return vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> > + read_lock(&vdev->domain_lock);
> > + domain = token.group->domain;
> > + vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> > + read_unlock(&vdev->domain_lock);
> > }
> >
> > static void *vduse_dev_alloc_coherent(union virtio_map token, size_t size,
> > dma_addr_t *dma_addr, gfp_t flag)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->as.domain;
> > + struct vduse_iova_domain *domain;
> > unsigned long iova;
> > - void *addr;
> > + void *addr = NULL;
> >
> > *dma_addr = DMA_MAPPING_ERROR;
> > + read_lock(&vdev->domain_lock);
> > + domain = token.group->domain;
> > addr = vduse_domain_alloc_coherent(domain, size,
> > (dma_addr_t *)&iova, flag);
> > - if (!addr)
> > - return NULL;
> > -
> > - *dma_addr = (dma_addr_t)iova;
> > + if (addr)
> > + *dma_addr = (dma_addr_t)iova;
> >
> > + read_unlock(&vdev->domain_lock);
> > return addr;
> > }
> >
> > @@ -910,17 +973,26 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
> > unsigned long attrs)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->as.domain;
> > + struct vduse_iova_domain *domain;
> >
> > + read_lock(&vdev->domain_lock);
> > + domain = token.group->domain;
> > vduse_domain_free_coherent(domain, size, vaddr, dma_addr, attrs);
> > + read_unlock(&vdev->domain_lock);
> > }
> >
> > static bool vduse_dev_need_sync(union virtio_map token, dma_addr_t dma_addr)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->as.domain;
> > + struct vduse_iova_domain *domain;
> > + size_t bounce_size;
> >
> > - return dma_addr < domain->bounce_size;
> > + read_lock(&vdev->domain_lock);
> > + domain = token.group->domain;
> > + bounce_size = domain->bounce_size;
> > + read_unlock(&vdev->domain_lock);
> > +
> > + return dma_addr < bounce_size;
> > }
> >
> > static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> > @@ -933,9 +1005,15 @@ static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> > static size_t vduse_dev_max_mapping_size(union virtio_map token)
> > {
> > struct vduse_dev *vdev = token.group->dev;
> > - struct vduse_iova_domain *domain = vdev->as.domain;
> > + struct vduse_iova_domain *domain;
> > + size_t bounce_size;
> > +
> > + read_lock(&vdev->domain_lock);
> > + domain = token.group->domain;
> > + bounce_size = domain->bounce_size;
> > + read_unlock(&vdev->domain_lock);
> >
> > - return domain->bounce_size;
> > + return bounce_size;
> > }
> >
> > static const struct virtio_map_ops vduse_map_ops = {
> > @@ -1075,39 +1153,40 @@ static int vduse_dev_queue_irq_work(struct vduse_dev *dev,
> > return ret;
> > }
> >
> > -static int vduse_dev_dereg_umem(struct vduse_dev *dev,
> > +static int vduse_dev_dereg_umem(struct vduse_dev *dev, u32 asid,
> > u64 iova, u64 size)
> > {
> > int ret;
> >
> > - mutex_lock(&dev->as.mem_lock);
> > + mutex_lock(&dev->as[asid].mem_lock);
> > ret = -ENOENT;
> > - if (!dev->as.umem)
> > + if (!dev->as[asid].umem)
> > goto unlock;
> >
> > ret = -EINVAL;
> > - if (!dev->as.domain)
> > + if (!dev->as[asid].domain)
> > goto unlock;
> >
> > - if (dev->as.umem->iova != iova || size != dev->as.domain->bounce_size)
> > + if (dev->as[asid].umem->iova != iova ||
> > + size != dev->as[asid].domain->bounce_size)
> > goto unlock;
> >
> > - vduse_domain_remove_user_bounce_pages(dev->as.domain);
> > - unpin_user_pages_dirty_lock(dev->as.umem->pages,
> > - dev->as.umem->npages, true);
> > - atomic64_sub(dev->as.umem->npages, &dev->as.umem->mm->pinned_vm);
> > - mmdrop(dev->as.umem->mm);
> > - vfree(dev->as.umem->pages);
> > - kfree(dev->as.umem);
> > - dev->as.umem = NULL;
> > + vduse_domain_remove_user_bounce_pages(dev->as[asid].domain);
> > + unpin_user_pages_dirty_lock(dev->as[asid].umem->pages,
> > + dev->as[asid].umem->npages, true);
> > + atomic64_sub(dev->as[asid].umem->npages, &dev->as[asid].umem->mm->pinned_vm);
> > + mmdrop(dev->as[asid].umem->mm);
> > + vfree(dev->as[asid].umem->pages);
> > + kfree(dev->as[asid].umem);
> > + dev->as[asid].umem = NULL;
>
> We can avoid those changeset if we do those in the previous correctly
> as it said it would make as an array.
>
I'm not following it. Is that different from squashing this patch with
the previous one? I'm ok with doing it, but this changes are needed
either way.
> > ret = 0;
> > unlock:
> > - mutex_unlock(&dev->as.mem_lock);
> > + mutex_unlock(&dev->as[asid].mem_lock);
> > return ret;
> > }
> >
> > static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > - u64 iova, u64 uaddr, u64 size)
> > + u32 asid, u64 iova, u64 uaddr, u64 size)
> > {
> > struct page **page_list = NULL;
> > struct vduse_umem *umem = NULL;
> > @@ -1115,14 +1194,14 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > unsigned long npages, lock_limit;
> > int ret;
> >
> > - if (!dev->as.domain || !dev->as.domain->bounce_map ||
> > - size != dev->as.domain->bounce_size ||
> > + if (!dev->as[asid].domain || !dev->as[asid].domain->bounce_map ||
> > + size != dev->as[asid].domain->bounce_size ||
> > iova != 0 || uaddr & ~PAGE_MASK)
> > return -EINVAL;
> >
> > - mutex_lock(&dev->as.mem_lock);
> > + mutex_lock(&dev->as[asid].mem_lock);
> > ret = -EEXIST;
> > - if (dev->as.umem)
> > + if (dev->as[asid].umem)
> > goto unlock;
> >
> > ret = -ENOMEM;
> > @@ -1146,7 +1225,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > goto out;
> > }
> >
> > - ret = vduse_domain_add_user_bounce_pages(dev->as.domain,
> > + ret = vduse_domain_add_user_bounce_pages(dev->as[asid].domain,
> > page_list, pinned);
> > if (ret)
> > goto out;
> > @@ -1159,7 +1238,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > umem->mm = current->mm;
> > mmgrab(current->mm);
> >
> > - dev->as.umem = umem;
> > + dev->as[asid].umem = umem;
> > out:
> > if (ret && pinned > 0)
> > unpin_user_pages(page_list, pinned);
> > @@ -1170,7 +1249,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > vfree(page_list);
> > kfree(umem);
> > }
> > - mutex_unlock(&dev->as.mem_lock);
> > + mutex_unlock(&dev->as[asid].mem_lock);
> > return ret;
> > }
> >
> > @@ -1202,47 +1281,66 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> >
> > switch (cmd) {
> > case VDUSE_IOTLB_GET_FD: {
> > - struct vduse_iotlb_entry entry;
> > + struct vduse_iotlb_entry_v2 entry;
> > struct vhost_iotlb_map *map;
> > struct vdpa_map_file *map_file;
> > struct file *f = NULL;
> > + u32 asid;
> >
> > ret = -EFAULT;
> > - if (copy_from_user(&entry, argp, sizeof(entry)))
> > - break;
> > + if (dev->api_version >= VDUSE_API_VERSION_1) {
> > + if (copy_from_user(&entry, argp, sizeof(entry)))
> > + break;
> > + } else {
> > + entry.asid = 0;
> > + if (copy_from_user(&entry.v1, argp,
> > + sizeof(entry.v1)))
> > + break;
> > + }
> >
> > ret = -EINVAL;
> > - if (entry.start > entry.last)
> > + if (entry.v1.start > entry.v1.last)
> > + break;
> > +
> > + if (entry.asid >= dev->nas)
> > break;
> >
> > read_lock(&dev->domain_lock);
> > - if (!dev->as.domain) {
> > + asid = array_index_nospec(entry.asid, dev->nas);
> > + if (!dev->as[asid].domain) {
> > read_unlock(&dev->domain_lock);
> > break;
> > }
> > - spin_lock(&dev->as.domain->iotlb_lock);
> > - map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
> > - entry.start, entry.last);
> > + spin_lock(&dev->as[asid].domain->iotlb_lock);
> > + map = vhost_iotlb_itree_first(dev->as[asid].domain->iotlb,
> > + entry.v1.start, entry.v1.last);
> > if (map) {
> > map_file = (struct vdpa_map_file *)map->opaque;
> > f = get_file(map_file->file);
> > - entry.offset = map_file->offset;
> > - entry.start = map->start;
> > - entry.last = map->last;
> > - entry.perm = map->perm;
> > + entry.v1.offset = map_file->offset;
> > + entry.v1.start = map->start;
> > + entry.v1.last = map->last;
> > + entry.v1.perm = map->perm;
> > }
> > - spin_unlock(&dev->as.domain->iotlb_lock);
> > + spin_unlock(&dev->as[asid].domain->iotlb_lock);
> > read_unlock(&dev->domain_lock);
> > ret = -EINVAL;
> > if (!f)
> > break;
> >
> > ret = -EFAULT;
> > - if (copy_to_user(argp, &entry, sizeof(entry))) {
> > + if (dev->api_version >= VDUSE_API_VERSION_1)
> > + ret = copy_to_user(argp, &entry,
> > + sizeof(entry));
> > + else
> > + ret = copy_to_user(argp, &entry.v1,
> > + sizeof(entry.v1));
> > +
> > + if (ret) {
> > fput(f);
> > break;
> > }
> > - ret = receive_fd(f, NULL, perm_to_file_flags(entry.perm));
> > + ret = receive_fd(f, NULL, perm_to_file_flags(entry.v1.perm));
>
> Nit: if we copy_from_user() twice and stick entry for v1 format, we
> can avoid a lot of lines of changes.
>
Let me draft something and put it as a reply here to check I'm
understanding your proposal.
> > fput(f);
> > break;
> > }
> > @@ -1384,6 +1482,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > }
> > case VDUSE_IOTLB_REG_UMEM: {
> > struct vduse_iova_umem umem;
> > + u32 asid;
> >
> > ret = -EFAULT;
> > if (copy_from_user(&umem, argp, sizeof(umem)))
> > @@ -1391,17 +1490,21 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> >
> > ret = -EINVAL;
> > if (!is_mem_zero((const char *)umem.reserved,
> > - sizeof(umem.reserved)))
> > + sizeof(umem.reserved)) ||
> > + (dev->api_version < VDUSE_API_VERSION_1 &&
> > + umem.asid != 0) || umem.asid >= dev->nas)
> > break;
> >
> > write_lock(&dev->domain_lock);
> > - ret = vduse_dev_reg_umem(dev, umem.iova,
> > + asid = array_index_nospec(umem.asid, dev->nas);
> > + ret = vduse_dev_reg_umem(dev, asid, umem.iova,
> > umem.uaddr, umem.size);
> > write_unlock(&dev->domain_lock);
> > break;
> > }
> > case VDUSE_IOTLB_DEREG_UMEM: {
> > struct vduse_iova_umem umem;
> > + u32 asid;
> >
> > ret = -EFAULT;
> > if (copy_from_user(&umem, argp, sizeof(umem)))
> > @@ -1409,10 +1512,15 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> >
> > ret = -EINVAL;
> > if (!is_mem_zero((const char *)umem.reserved,
> > - sizeof(umem.reserved)))
> > + sizeof(umem.reserved)) ||
> > + (dev->api_version < VDUSE_API_VERSION_1 &&
> > + umem.asid != 0) ||
> > + umem.asid >= dev->nas)
> > break;
> > +
> > write_lock(&dev->domain_lock);
> > - ret = vduse_dev_dereg_umem(dev, umem.iova,
> > + asid = array_index_nospec(umem.asid, dev->nas);
> > + ret = vduse_dev_dereg_umem(dev, asid, umem.iova,
> > umem.size);
> > write_unlock(&dev->domain_lock);
> > break;
> > @@ -1420,6 +1528,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > case VDUSE_IOTLB_GET_INFO: {
> > struct vduse_iova_info info;
> > struct vhost_iotlb_map *map;
> > + u32 asid;
> >
> > ret = -EFAULT;
> > if (copy_from_user(&info, argp, sizeof(info)))
> > @@ -1433,23 +1542,31 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > sizeof(info.reserved)))
> > break;
> >
> > + if (dev->api_version < VDUSE_API_VERSION_1) {
> > + if (info.asid)
> > + break;
> > + } else if (info.asid >= dev->nas)
> > + break;
> > +
> > read_lock(&dev->domain_lock);
> > - if (!dev->as.domain) {
> > + asid = array_index_nospec(info.asid, dev->nas);
> > + if (!dev->as[asid].domain) {
> > read_unlock(&dev->domain_lock);
> > break;
> > }
> > - spin_lock(&dev->as.domain->iotlb_lock);
> > - map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
> > + spin_lock(&dev->as[asid].domain->iotlb_lock);
> > + map = vhost_iotlb_itree_first(dev->as[asid].domain->iotlb,
> > info.start, info.last);
> > if (map) {
> > info.start = map->start;
> > info.last = map->last;
> > info.capability = 0;
> > - if (dev->as.domain->bounce_map && map->start == 0 &&
> > - map->last == dev->as.domain->bounce_size - 1)
> > + if (dev->as[asid].domain->bounce_map &&
> > + map->start == 0 &&
> > + map->last == dev->as[asid].domain->bounce_size - 1)
> > info.capability |= VDUSE_IOVA_CAP_UMEM;
> > }
> > - spin_unlock(&dev->as.domain->iotlb_lock);
> > + spin_unlock(&dev->as[asid].domain->iotlb_lock);
> > read_unlock(&dev->domain_lock);
> > if (!map)
> > break;
> > @@ -1474,8 +1591,10 @@ static int vduse_dev_release(struct inode *inode, struct file *file)
> > struct vduse_dev *dev = file->private_data;
> >
> > write_lock(&dev->domain_lock);
> > - if (dev->as.domain)
> > - vduse_dev_dereg_umem(dev, 0, dev->as.domain->bounce_size);
> > + for (int i = 0; i < dev->nas; i++)
> > + if (dev->as[i].domain)
> > + vduse_dev_dereg_umem(dev, i, 0,
> > + dev->as[i].domain->bounce_size);
> > write_unlock(&dev->domain_lock);
> > spin_lock(&dev->msg_lock);
> > /* Make sure the inflight messages can processed after reconncection */
> > @@ -1694,7 +1813,6 @@ static struct vduse_dev *vduse_dev_create(void)
> > return NULL;
> >
> > mutex_init(&dev->lock);
> > - mutex_init(&dev->as.mem_lock);
> > rwlock_init(&dev->domain_lock);
> > spin_lock_init(&dev->msg_lock);
> > INIT_LIST_HEAD(&dev->send_list);
> > @@ -1745,8 +1863,11 @@ static int vduse_destroy_dev(char *name)
> > idr_remove(&vduse_idr, dev->minor);
> > kvfree(dev->config);
> > vduse_dev_deinit_vqs(dev);
> > - if (dev->as.domain)
> > - vduse_domain_destroy(dev->as.domain);
> > + for (int i = 0; i < dev->nas; i++) {
> > + if (dev->as[i].domain)
> > + vduse_domain_destroy(dev->as[i].domain);
> > + }
> > + kfree(dev->as);
> > kfree(dev->name);
> > kfree(dev->groups);
> > vduse_dev_destroy(dev);
> > @@ -1793,12 +1914,16 @@ static bool vduse_validate_config(struct vduse_dev_config *config,
> > sizeof(config->reserved)))
> > return false;
> >
> > - if (api_version < VDUSE_API_VERSION_1 && config->ngroups)
> > + if (api_version < VDUSE_API_VERSION_1 &&
> > + (config->ngroups || config->nas))
> > return false;
> >
> > if (api_version >= VDUSE_API_VERSION_1 && config->ngroups > 0xffff)
> > return false;
> >
> > + if (api_version >= VDUSE_API_VERSION_1 && config->nas > 0xffff)
> > + return false;
> > +
> > if (config->vq_align > PAGE_SIZE)
> > return false;
> >
> > @@ -1862,7 +1987,8 @@ static ssize_t bounce_size_store(struct device *device,
> >
> > ret = -EPERM;
> > write_lock(&dev->domain_lock);
> > - if (dev->as.domain)
> > + /* Assuming that if the first domain is allocated, all are allocated */
> > + if (dev->as[0].domain)
> > goto unlock;
> >
> > ret = kstrtouint(buf, 10, &bounce_size);
> > @@ -1923,6 +2049,13 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> > for (u32 i = 0; i < dev->ngroups; ++i)
> > dev->groups[i].dev = dev;
> >
> > + dev->nas = (dev->api_version < 1) ? 1 : (config->nas ?: 1);
> > + dev->as = kcalloc(dev->nas, sizeof(dev->as[0]), GFP_KERNEL);
> > + if (!dev->as)
> > + goto err_as;
> > + for (int i = 0; i < dev->nas; i++)
> > + mutex_init(&dev->as[i].mem_lock);
> > +
> > dev->name = kstrdup(config->name, GFP_KERNEL);
> > if (!dev->name)
> > goto err_str;
> > @@ -1959,6 +2092,8 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> > err_idr:
> > kfree(dev->name);
> > err_str:
> > + kfree(dev->as);
> > +err_as:
> > kfree(dev->groups);
> > err_vq_groups:
> > vduse_dev_destroy(dev);
> > @@ -2084,7 +2219,7 @@ static int vduse_dev_init_vdpa(struct vduse_dev *dev, const char *name)
> >
> > vdev = vdpa_alloc_device(struct vduse_vdpa, vdpa, dev->dev,
> > &vduse_vdpa_config_ops, &vduse_map_ops,
> > - dev->ngroups, 1, name, true);
> > + dev->ngroups, dev->nas, name, true);
> > if (IS_ERR(vdev))
> > return PTR_ERR(vdev);
> >
> > @@ -2113,11 +2248,20 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> > return ret;
> >
> > write_lock(&dev->domain_lock);
> > - if (!dev->as.domain)
> > - dev->as.domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
> > - dev->bounce_size);
> > + ret = 0;
> > +
> > + for (int i = 0; i < dev->nas; ++i) {
> > + dev->as[i].domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
> > + dev->bounce_size);
> > + if (!dev->as[i].domain) {
> > + ret = -ENOMEM;
> > + for (int j = 0; j < i; ++j)
> > + vduse_domain_destroy(dev->as[j].domain);
> > + }
> > + }
> > +
> > write_unlock(&dev->domain_lock);
> > - if (!dev->as.domain) {
> > + if (ret == -ENOMEM) {
> > put_device(&dev->vdev->vdpa.dev);
> > return -ENOMEM;
> > }
> > @@ -2126,8 +2270,12 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> > if (ret) {
> > put_device(&dev->vdev->vdpa.dev);
> > write_lock(&dev->domain_lock);
> > - vduse_domain_destroy(dev->as.domain);
> > - dev->as.domain = NULL;
> > + for (int i = 0; i < dev->nas; i++) {
> > + if (dev->as[i].domain) {
> > + vduse_domain_destroy(dev->as[i].domain);
> > + dev->as[i].domain = NULL;
> > + }
> > + }
> > write_unlock(&dev->domain_lock);
> > return ret;
> > }
> > diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h
> > index a3d51cf6df3a..da73c3f2c280 100644
> > --- a/include/uapi/linux/vduse.h
> > +++ b/include/uapi/linux/vduse.h
> > @@ -47,7 +47,8 @@ struct vduse_dev_config {
> > __u32 vq_num;
> > __u32 vq_align;
> > __u32 ngroups; /* if VDUSE_API_VERSION >= 1 */
> > - __u32 reserved[12];
> > + __u32 nas; /* if VDUSE_API_VERSION >= 1 */
> > + __u32 reserved[11];
> > __u32 config_size;
> > __u8 config[];
> > };
> > @@ -82,6 +83,18 @@ struct vduse_iotlb_entry {
> > __u8 perm;
> > };
> >
> > +/**
> > + * struct vduse_iotlb_entry_v2 - entry of IOTLB to describe one IOVA region in an ASID
> > + * @v1: the original vduse_iotlb_entry
> > + * @asid: address space ID of the IOVA region
> > + *
> > + * Structure used by VDUSE_IOTLB_GET_FD ioctl to find an overlapped IOVA region.
> > + */
> > +struct vduse_iotlb_entry_v2 {
> > + struct vduse_iotlb_entry v1;
> > + __u32 asid;
> > +};
> > +
> > /*
> > * Find the first IOVA region that overlaps with the range [start, last]
> > * and return the corresponding file descriptor. Return -EINVAL means the
> > @@ -166,6 +179,16 @@ struct vduse_vq_state_packed {
> > __u16 last_used_idx;
> > };
> >
> > +/**
> > + * struct vduse_vq_group - virtqueue group
> > + @ @group: Index of the virtqueue group
> > + * @asid: Address space ID of the group
> > + */
> > +struct vduse_vq_group_asid {
> > + __u32 group;
> > + __u32 asid;
> > +};
> > +
> > /**
> > * struct vduse_vq_info - information of a virtqueue
> > * @index: virtqueue index
> > @@ -225,6 +248,7 @@ struct vduse_vq_eventfd {
> > * @uaddr: start address of userspace memory, it must be aligned to page size
> > * @iova: start of the IOVA region
> > * @size: size of the IOVA region
> > + * @asid: Address space ID of the IOVA region
> > * @reserved: for future use, needs to be initialized to zero
> > *
> > * Structure used by VDUSE_IOTLB_REG_UMEM and VDUSE_IOTLB_DEREG_UMEM
> > @@ -234,7 +258,8 @@ struct vduse_iova_umem {
> > __u64 uaddr;
> > __u64 iova;
> > __u64 size;
> > - __u64 reserved[3];
> > + __u32 asid;
> > + __u32 reserved[5];
> > };
> >
> > /* Register userspace memory for IOVA regions */
> > @@ -248,6 +273,7 @@ struct vduse_iova_umem {
> > * @start: start of the IOVA region
> > * @last: last of the IOVA region
> > * @capability: capability of the IOVA region
> > + * @asid: Address space ID of the IOVA region, only if device API version >= 1
> > * @reserved: for future use, needs to be initialized to zero
> > *
> > * Structure used by VDUSE_IOTLB_GET_INFO ioctl to get information of
> > @@ -258,7 +284,8 @@ struct vduse_iova_info {
> > __u64 last;
> > #define VDUSE_IOVA_CAP_UMEM (1 << 0)
> > __u64 capability;
> > - __u64 reserved[3];
> > + __u32 asid; /* Only if device API version >= 1 */
> > + __u32 reserved[5];
> > };
> >
> > /*
> > @@ -280,6 +307,7 @@ enum vduse_req_type {
> > VDUSE_GET_VQ_STATE,
> > VDUSE_SET_STATUS,
> > VDUSE_UPDATE_IOTLB,
> > + VDUSE_SET_VQ_GROUP_ASID,
> > };
> >
> > /**
> > @@ -314,6 +342,18 @@ struct vduse_iova_range {
> > __u64 last;
> > };
> >
> > +/**
> > + * struct vduse_iova_range - IOVA range [start, last] if API_VERSION >= 1
> > + * @start: start of the IOVA range
> > + * @last: last of the IOVA range
> > + * @asid: address space ID of the IOVA range
> > + */
> > +struct vduse_iova_range_v2 {
> > + __u64 start;
> > + __u64 last;
> > + __u32 asid;
> > +};
> > +
> > /**
> > * struct vduse_dev_request - control request
> > * @type: request type
> > @@ -322,6 +362,8 @@ struct vduse_iova_range {
> > * @vq_state: virtqueue state, only index field is available
> > * @s: device status
> > * @iova: IOVA range for updating
> > + * @iova_v2: IOVA range for updating if API_VERSION >= 1
> > + * @vq_group_asid: ASID of a virtqueue group
> > * @padding: padding
> > *
> > * Structure used by read(2) on /dev/vduse/$NAME.
> > @@ -334,6 +376,9 @@ struct vduse_dev_request {
> > struct vduse_vq_state vq_state;
> > struct vduse_dev_status s;
> > struct vduse_iova_range iova;
> > + /* Following members only if vduse api version >= 1 */;
> > + struct vduse_iova_range_v2 iova_v2;
> > + struct vduse_vq_group_asid vq_group_asid;
> > __u32 padding[32];
> > };
> > };
> > --
> > 2.51.0
> >
>
> Thanks
>
^ permalink raw reply [flat|nested] 27+ messages in thread* Re: [PATCH v2 6/7] vduse: add vq group asid support
2025-09-17 16:40 ` Eugenio Perez Martin
@ 2025-09-18 6:07 ` Jason Wang
2025-09-18 6:57 ` Eugenio Perez Martin
2025-09-18 11:21 ` Eugenio Perez Martin
1 sibling, 1 reply; 27+ messages in thread
From: Jason Wang @ 2025-09-18 6:07 UTC (permalink / raw)
To: Eugenio Perez Martin
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Thu, Sep 18, 2025 at 12:41 AM Eugenio Perez Martin
<eperezma@redhat.com> wrote:
>
> On Wed, Sep 17, 2025 at 10:56 AM Jason Wang <jasowang@redhat.com> wrote:
> >
> > On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
> > >
> > > Add support for assigning Address Space Identifiers (ASIDs) to each VQ
> > > group. This enables mapping each group into a distinct memory space.
> > >
> > > Now that the driver can change ASID in the middle of operation, the
> > > domain that each vq address point is also protected by domain_lock.
> > >
> > > Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> > > ---
> > > v2:
> > > * Convert the use of mutex to rwlock.
> > >
> > > RFC v3:
> > > * Increase VDUSE_MAX_VQ_GROUPS to 0xffff (Jason). It was set to a lower
> > > value to reduce memory consumption, but vqs are already limited to
> > > that value and userspace VDUSE is able to allocate that many vqs.
> > > * Remove TODO about merging VDUSE_IOTLB_GET_FD ioctl with
> > > VDUSE_IOTLB_GET_INFO.
> > > * Use of array_index_nospec in VDUSE device ioctls.
> > > * Embed vduse_iotlb_entry into vduse_iotlb_entry_v2.
> > > * Move the umem mutex to asid struct so there is no contention between
> > > ASIDs.
> > >
> > > RFC v2:
> > > * Make iotlb entry the last one of vduse_iotlb_entry_v2 so the first
> > > part of the struct is the same.
> > > ---
> > > drivers/vdpa/vdpa_user/vduse_dev.c | 324 +++++++++++++++++++++--------
> > > include/uapi/linux/vduse.h | 51 ++++-
> > > 2 files changed, 284 insertions(+), 91 deletions(-)
> > >
> > > diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > index b45b1d22784f..06b7790380b7 100644
> > > --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> > > +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > @@ -93,6 +93,7 @@ struct vduse_as {
> > > };
> > >
> > > struct vduse_vq_group_int {
> > > + struct vduse_iova_domain *domain;
> > > struct vduse_dev *dev;
> >
> > This confuses me, I think it should be an asid. And the vduse_dev
> > pointer seems to be useless here.
> >
>
> The *dev pointer is used to take the rwlock, in case the vhost driver
> calls VHOST_VDPA_SET_GROUP_ASID (or equivalent) at the same time
> vduse_dev_sync_single_for_device (or _for_cpu, or equivalent) run.
Ok, but having it in a group seems odd. A better place is to move it to the as?
>
> > > };
> > >
> > > @@ -100,7 +101,7 @@ struct vduse_dev {
> > > struct vduse_vdpa *vdev;
> > > struct device *dev;
> > > struct vduse_virtqueue **vqs;
> > > - struct vduse_as as;
> > > + struct vduse_as *as;
> > > char *name;
> > > struct mutex lock;
> > > spinlock_t msg_lock;
> > > @@ -128,6 +129,7 @@ struct vduse_dev {
> > > u32 vq_num;
> > > u32 vq_align;
> > > u32 ngroups;
> > > + u32 nas;
> > > struct vduse_vq_group_int *groups;
> > > unsigned int bounce_size;
> > > rwlock_t domain_lock;
> > > @@ -318,7 +320,7 @@ static int vduse_dev_set_status(struct vduse_dev *dev, u8 status)
> > > return vduse_dev_msg_sync(dev, &msg);
> > > }
> > >
> > > -static int vduse_dev_update_iotlb(struct vduse_dev *dev,
> > > +static int vduse_dev_update_iotlb(struct vduse_dev *dev, u32 asid,
> > > u64 start, u64 last)
> > > {
> > > struct vduse_dev_msg msg = { 0 };
> > > @@ -327,8 +329,14 @@ static int vduse_dev_update_iotlb(struct vduse_dev *dev,
> > > return -EINVAL;
> > >
> > > msg.req.type = VDUSE_UPDATE_IOTLB;
> > > - msg.req.iova.start = start;
> > > - msg.req.iova.last = last;
> > > + if (dev->api_version < VDUSE_API_VERSION_1) {
> > > + msg.req.iova.start = start;
> > > + msg.req.iova.last = last;
> > > + } else {
> > > + msg.req.iova_v2.start = start;
> > > + msg.req.iova_v2.last = last;
> > > + msg.req.iova_v2.asid = asid;
> > > + }
> > >
> > > return vduse_dev_msg_sync(dev, &msg);
> > > }
> > > @@ -440,14 +448,29 @@ static __poll_t vduse_dev_poll(struct file *file, poll_table *wait)
> > > return mask;
> > > }
> > >
> > > +/* Force set the asid to a vq group without a message to the VDUSE device */
> > > +static void vduse_set_group_asid_nomsg(struct vduse_dev *dev,
> > > + unsigned int group, unsigned int asid)
> > > +{
> > > + write_lock(&dev->domain_lock);
> > > + dev->groups[group].domain = dev->as[asid].domain;
> >
> > I think it would be better to stick the group->as an indirection which
> > should be .
> >
> > dev->groups.asid = asid;
> >
> > Or
> >
> > dev->group->as = as;
> >
>
> That involves an extra memory jump for functions that may be in the
> hot path. I've not profiled it, but I'm ok with changing it that way
> if you prefer.
I think it would be better if we can change (see my reply in previous patch).
>
> > > + write_unlock(&dev->domain_lock);
> > > +}
> > > +
> > > static void vduse_dev_reset(struct vduse_dev *dev)
> > > {
> > > int i;
> > > - struct vduse_iova_domain *domain = dev->as.domain;
> > >
> > > /* The coherent mappings are handled in vduse_dev_free_coherent() */
> > > - if (domain && domain->bounce_map)
> > > - vduse_domain_reset_bounce_map(domain);
> > > + for (i = 0; i < dev->nas; i++) {
> > > + struct vduse_iova_domain *domain = dev->as[i].domain;
> > > +
> > > + if (domain && domain->bounce_map)
> > > + vduse_domain_reset_bounce_map(domain);
> > > + }
> > > +
> > > + for (i = 0; i < dev->ngroups; i++)
> > > + vduse_set_group_asid_nomsg(dev, i, 0);
> > >
> > > down_write(&dev->rwsem);
> > >
> > > @@ -621,6 +644,29 @@ static union virtio_map vduse_get_vq_map(struct vdpa_device *vdpa, u16 idx)
> > > return ret;
> > > }
> > >
> > > +static int vduse_set_group_asid(struct vdpa_device *vdpa, unsigned int group,
> > > + unsigned int asid)
> > > +{
> > > + struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > > + struct vduse_dev_msg msg = { 0 };
> > > + int r;
> > > +
> > > + if (dev->api_version < VDUSE_API_VERSION_1 ||
> > > + group >= dev->ngroups || asid >= dev->nas)
> > > + return -EINVAL;
> > > +
> > > + msg.req.type = VDUSE_SET_VQ_GROUP_ASID;
> > > + msg.req.vq_group_asid.group = group;
> > > + msg.req.vq_group_asid.asid = asid;
> > > +
> > > + r = vduse_dev_msg_sync(dev, &msg);
> > > + if (r < 0)
> > > + return r;
> > > +
> > > + vduse_set_group_asid_nomsg(dev, group, asid);
> > > + return 0;
> > > +}
> > > +
> > > static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
> > > struct vdpa_vq_state *state)
> > > {
> > > @@ -792,13 +838,13 @@ static int vduse_vdpa_set_map(struct vdpa_device *vdpa,
> > > struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > > int ret;
> > >
> > > - ret = vduse_domain_set_map(dev->as.domain, iotlb);
> > > + ret = vduse_domain_set_map(dev->as[asid].domain, iotlb);
> > > if (ret)
> > > return ret;
> > >
> > > - ret = vduse_dev_update_iotlb(dev, 0ULL, ULLONG_MAX);
> > > + ret = vduse_dev_update_iotlb(dev, asid, 0ULL, ULLONG_MAX);
> > > if (ret) {
> > > - vduse_domain_clear_map(dev->as.domain, iotlb);
> > > + vduse_domain_clear_map(dev->as[asid].domain, iotlb);
> > > return ret;
> > > }
> > >
> > > @@ -841,6 +887,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
> > > .get_vq_affinity = vduse_vdpa_get_vq_affinity,
> > > .reset = vduse_vdpa_reset,
> > > .set_map = vduse_vdpa_set_map,
> > > + .set_group_asid = vduse_set_group_asid,
> > > .get_vq_map = vduse_get_vq_map,
> > > .free = vduse_vdpa_free,
> > > };
> > > @@ -850,9 +897,12 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
> > > enum dma_data_direction dir)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > >
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > vduse_domain_sync_single_for_device(domain, dma_addr, size, dir);
> > > + read_unlock(&vdev->domain_lock);
> > > }
> > >
> > > static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> > > @@ -860,9 +910,12 @@ static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> > > enum dma_data_direction dir)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > >
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> >
> > I think the domain is better fetched via vduse_as.
> >
> > > vduse_domain_sync_single_for_cpu(domain, dma_addr, size, dir);
> > > + read_unlock(&vdev->domain_lock);
> > > }
> > >
> > > static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> > > @@ -871,9 +924,15 @@ static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> > > unsigned long attrs)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > > + dma_addr_t r;
> > >
> > > - return vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > + r = vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> > > + read_unlock(&vdev->domain_lock);
> > > +
> > > + return r;
> > > }
> > >
> > > static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> > > @@ -881,27 +940,31 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> > > unsigned long attrs)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > >
> > > - return vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > + vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> > > + read_unlock(&vdev->domain_lock);
> > > }
> > >
> > > static void *vduse_dev_alloc_coherent(union virtio_map token, size_t size,
> > > dma_addr_t *dma_addr, gfp_t flag)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > > unsigned long iova;
> > > - void *addr;
> > > + void *addr = NULL;
> > >
> > > *dma_addr = DMA_MAPPING_ERROR;
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > addr = vduse_domain_alloc_coherent(domain, size,
> > > (dma_addr_t *)&iova, flag);
> > > - if (!addr)
> > > - return NULL;
> > > -
> > > - *dma_addr = (dma_addr_t)iova;
> > > + if (addr)
> > > + *dma_addr = (dma_addr_t)iova;
> > >
> > > + read_unlock(&vdev->domain_lock);
> > > return addr;
> > > }
> > >
> > > @@ -910,17 +973,26 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
> > > unsigned long attrs)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > >
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > vduse_domain_free_coherent(domain, size, vaddr, dma_addr, attrs);
> > > + read_unlock(&vdev->domain_lock);
> > > }
> > >
> > > static bool vduse_dev_need_sync(union virtio_map token, dma_addr_t dma_addr)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > > + size_t bounce_size;
> > >
> > > - return dma_addr < domain->bounce_size;
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > + bounce_size = domain->bounce_size;
> > > + read_unlock(&vdev->domain_lock);
> > > +
> > > + return dma_addr < bounce_size;
> > > }
> > >
> > > static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> > > @@ -933,9 +1005,15 @@ static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> > > static size_t vduse_dev_max_mapping_size(union virtio_map token)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > > + size_t bounce_size;
> > > +
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > + bounce_size = domain->bounce_size;
> > > + read_unlock(&vdev->domain_lock);
> > >
> > > - return domain->bounce_size;
> > > + return bounce_size;
> > > }
> > >
> > > static const struct virtio_map_ops vduse_map_ops = {
> > > @@ -1075,39 +1153,40 @@ static int vduse_dev_queue_irq_work(struct vduse_dev *dev,
> > > return ret;
> > > }
> > >
> > > -static int vduse_dev_dereg_umem(struct vduse_dev *dev,
> > > +static int vduse_dev_dereg_umem(struct vduse_dev *dev, u32 asid,
> > > u64 iova, u64 size)
> > > {
> > > int ret;
> > >
> > > - mutex_lock(&dev->as.mem_lock);
> > > + mutex_lock(&dev->as[asid].mem_lock);
> > > ret = -ENOENT;
> > > - if (!dev->as.umem)
> > > + if (!dev->as[asid].umem)
> > > goto unlock;
> > >
> > > ret = -EINVAL;
> > > - if (!dev->as.domain)
> > > + if (!dev->as[asid].domain)
> > > goto unlock;
> > >
> > > - if (dev->as.umem->iova != iova || size != dev->as.domain->bounce_size)
> > > + if (dev->as[asid].umem->iova != iova ||
> > > + size != dev->as[asid].domain->bounce_size)
> > > goto unlock;
> > >
> > > - vduse_domain_remove_user_bounce_pages(dev->as.domain);
> > > - unpin_user_pages_dirty_lock(dev->as.umem->pages,
> > > - dev->as.umem->npages, true);
> > > - atomic64_sub(dev->as.umem->npages, &dev->as.umem->mm->pinned_vm);
> > > - mmdrop(dev->as.umem->mm);
> > > - vfree(dev->as.umem->pages);
> > > - kfree(dev->as.umem);
> > > - dev->as.umem = NULL;
> > > + vduse_domain_remove_user_bounce_pages(dev->as[asid].domain);
> > > + unpin_user_pages_dirty_lock(dev->as[asid].umem->pages,
> > > + dev->as[asid].umem->npages, true);
> > > + atomic64_sub(dev->as[asid].umem->npages, &dev->as[asid].umem->mm->pinned_vm);
> > > + mmdrop(dev->as[asid].umem->mm);
> > > + vfree(dev->as[asid].umem->pages);
> > > + kfree(dev->as[asid].umem);
> > > + dev->as[asid].umem = NULL;
> >
> > We can avoid those changeset if we do those in the previous correctly
> > as it said it would make as an array.
> >
>
> I'm not following it. Is that different from squashing this patch with
> the previous one? I'm ok with doing it, but this changes are needed
> either way.
See my reply for whether we need to make it as an array in the last
patch. I meant basically we only need to change them once, but this
series touches this twice.
Thanks
^ permalink raw reply [flat|nested] 27+ messages in thread* Re: [PATCH v2 6/7] vduse: add vq group asid support
2025-09-18 6:07 ` Jason Wang
@ 2025-09-18 6:57 ` Eugenio Perez Martin
2025-09-19 2:15 ` Jason Wang
0 siblings, 1 reply; 27+ messages in thread
From: Eugenio Perez Martin @ 2025-09-18 6:57 UTC (permalink / raw)
To: Jason Wang
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Thu, Sep 18, 2025 at 8:07 AM Jason Wang <jasowang@redhat.com> wrote:
>
> On Thu, Sep 18, 2025 at 12:41 AM Eugenio Perez Martin
> <eperezma@redhat.com> wrote:
> >
> > On Wed, Sep 17, 2025 at 10:56 AM Jason Wang <jasowang@redhat.com> wrote:
> > >
> > > On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
> > > >
> > > > Add support for assigning Address Space Identifiers (ASIDs) to each VQ
> > > > group. This enables mapping each group into a distinct memory space.
> > > >
> > > > Now that the driver can change ASID in the middle of operation, the
> > > > domain that each vq address point is also protected by domain_lock.
> > > >
> > > > Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> > > > ---
> > > > v2:
> > > > * Convert the use of mutex to rwlock.
> > > >
> > > > RFC v3:
> > > > * Increase VDUSE_MAX_VQ_GROUPS to 0xffff (Jason). It was set to a lower
> > > > value to reduce memory consumption, but vqs are already limited to
> > > > that value and userspace VDUSE is able to allocate that many vqs.
> > > > * Remove TODO about merging VDUSE_IOTLB_GET_FD ioctl with
> > > > VDUSE_IOTLB_GET_INFO.
> > > > * Use of array_index_nospec in VDUSE device ioctls.
> > > > * Embed vduse_iotlb_entry into vduse_iotlb_entry_v2.
> > > > * Move the umem mutex to asid struct so there is no contention between
> > > > ASIDs.
> > > >
> > > > RFC v2:
> > > > * Make iotlb entry the last one of vduse_iotlb_entry_v2 so the first
> > > > part of the struct is the same.
> > > > ---
> > > > drivers/vdpa/vdpa_user/vduse_dev.c | 324 +++++++++++++++++++++--------
> > > > include/uapi/linux/vduse.h | 51 ++++-
> > > > 2 files changed, 284 insertions(+), 91 deletions(-)
> > > >
> > > > diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > > index b45b1d22784f..06b7790380b7 100644
> > > > --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> > > > +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > > @@ -93,6 +93,7 @@ struct vduse_as {
> > > > };
> > > >
> > > > struct vduse_vq_group_int {
> > > > + struct vduse_iova_domain *domain;
> > > > struct vduse_dev *dev;
> > >
> > > This confuses me, I think it should be an asid. And the vduse_dev
> > > pointer seems to be useless here.
> > >
> >
> > The *dev pointer is used to take the rwlock, in case the vhost driver
> > calls VHOST_VDPA_SET_GROUP_ASID (or equivalent) at the same time
> > vduse_dev_sync_single_for_device (or _for_cpu, or equivalent) run.
>
> Ok, but having it in a group seems odd. A better place is to move it to the as?
>
If we move to the as, we need to take the lock this way:
read_lock(vduse_vq_group->as->dev->domain_lock)
But the lock is protecting the vduse_vq_group->as change itself, as
other ioctls (SET_GROUP_ASID mainly) can modify it concurrently.
We can assume that *as will always point to a valid as, its change
will be atomic, and all the as->dev points to the exact same
vduse_dev. I can buy that, but it is a race anyway.
> >
> > > > };
> > > >
> > > > @@ -100,7 +101,7 @@ struct vduse_dev {
> > > > struct vduse_vdpa *vdev;
> > > > struct device *dev;
> > > > struct vduse_virtqueue **vqs;
> > > > - struct vduse_as as;
> > > > + struct vduse_as *as;
> > > > char *name;
> > > > struct mutex lock;
> > > > spinlock_t msg_lock;
> > > > @@ -128,6 +129,7 @@ struct vduse_dev {
> > > > u32 vq_num;
> > > > u32 vq_align;
> > > > u32 ngroups;
> > > > + u32 nas;
> > > > struct vduse_vq_group_int *groups;
> > > > unsigned int bounce_size;
> > > > rwlock_t domain_lock;
> > > > @@ -318,7 +320,7 @@ static int vduse_dev_set_status(struct vduse_dev *dev, u8 status)
> > > > return vduse_dev_msg_sync(dev, &msg);
> > > > }
> > > >
> > > > -static int vduse_dev_update_iotlb(struct vduse_dev *dev,
> > > > +static int vduse_dev_update_iotlb(struct vduse_dev *dev, u32 asid,
> > > > u64 start, u64 last)
> > > > {
> > > > struct vduse_dev_msg msg = { 0 };
> > > > @@ -327,8 +329,14 @@ static int vduse_dev_update_iotlb(struct vduse_dev *dev,
> > > > return -EINVAL;
> > > >
> > > > msg.req.type = VDUSE_UPDATE_IOTLB;
> > > > - msg.req.iova.start = start;
> > > > - msg.req.iova.last = last;
> > > > + if (dev->api_version < VDUSE_API_VERSION_1) {
> > > > + msg.req.iova.start = start;
> > > > + msg.req.iova.last = last;
> > > > + } else {
> > > > + msg.req.iova_v2.start = start;
> > > > + msg.req.iova_v2.last = last;
> > > > + msg.req.iova_v2.asid = asid;
> > > > + }
> > > >
> > > > return vduse_dev_msg_sync(dev, &msg);
> > > > }
> > > > @@ -440,14 +448,29 @@ static __poll_t vduse_dev_poll(struct file *file, poll_table *wait)
> > > > return mask;
> > > > }
> > > >
> > > > +/* Force set the asid to a vq group without a message to the VDUSE device */
> > > > +static void vduse_set_group_asid_nomsg(struct vduse_dev *dev,
> > > > + unsigned int group, unsigned int asid)
> > > > +{
> > > > + write_lock(&dev->domain_lock);
> > > > + dev->groups[group].domain = dev->as[asid].domain;
> > >
> > > I think it would be better to stick the group->as an indirection which
> > > should be .
> > >
> > > dev->groups.asid = asid;
> > >
> > > Or
> > >
> > > dev->group->as = as;
> > >
> >
> > That involves an extra memory jump for functions that may be in the
> > hot path. I've not profiled it, but I'm ok with changing it that way
> > if you prefer.
>
> I think it would be better if we can change (see my reply in previous patch).
>
> >
> > > > + write_unlock(&dev->domain_lock);
> > > > +}
> > > > +
> > > > static void vduse_dev_reset(struct vduse_dev *dev)
> > > > {
> > > > int i;
> > > > - struct vduse_iova_domain *domain = dev->as.domain;
> > > >
> > > > /* The coherent mappings are handled in vduse_dev_free_coherent() */
> > > > - if (domain && domain->bounce_map)
> > > > - vduse_domain_reset_bounce_map(domain);
> > > > + for (i = 0; i < dev->nas; i++) {
> > > > + struct vduse_iova_domain *domain = dev->as[i].domain;
> > > > +
> > > > + if (domain && domain->bounce_map)
> > > > + vduse_domain_reset_bounce_map(domain);
> > > > + }
> > > > +
> > > > + for (i = 0; i < dev->ngroups; i++)
> > > > + vduse_set_group_asid_nomsg(dev, i, 0);
> > > >
> > > > down_write(&dev->rwsem);
> > > >
> > > > @@ -621,6 +644,29 @@ static union virtio_map vduse_get_vq_map(struct vdpa_device *vdpa, u16 idx)
> > > > return ret;
> > > > }
> > > >
> > > > +static int vduse_set_group_asid(struct vdpa_device *vdpa, unsigned int group,
> > > > + unsigned int asid)
> > > > +{
> > > > + struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > > > + struct vduse_dev_msg msg = { 0 };
> > > > + int r;
> > > > +
> > > > + if (dev->api_version < VDUSE_API_VERSION_1 ||
> > > > + group >= dev->ngroups || asid >= dev->nas)
> > > > + return -EINVAL;
> > > > +
> > > > + msg.req.type = VDUSE_SET_VQ_GROUP_ASID;
> > > > + msg.req.vq_group_asid.group = group;
> > > > + msg.req.vq_group_asid.asid = asid;
> > > > +
> > > > + r = vduse_dev_msg_sync(dev, &msg);
> > > > + if (r < 0)
> > > > + return r;
> > > > +
> > > > + vduse_set_group_asid_nomsg(dev, group, asid);
> > > > + return 0;
> > > > +}
> > > > +
> > > > static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
> > > > struct vdpa_vq_state *state)
> > > > {
> > > > @@ -792,13 +838,13 @@ static int vduse_vdpa_set_map(struct vdpa_device *vdpa,
> > > > struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > > > int ret;
> > > >
> > > > - ret = vduse_domain_set_map(dev->as.domain, iotlb);
> > > > + ret = vduse_domain_set_map(dev->as[asid].domain, iotlb);
> > > > if (ret)
> > > > return ret;
> > > >
> > > > - ret = vduse_dev_update_iotlb(dev, 0ULL, ULLONG_MAX);
> > > > + ret = vduse_dev_update_iotlb(dev, asid, 0ULL, ULLONG_MAX);
> > > > if (ret) {
> > > > - vduse_domain_clear_map(dev->as.domain, iotlb);
> > > > + vduse_domain_clear_map(dev->as[asid].domain, iotlb);
> > > > return ret;
> > > > }
> > > >
> > > > @@ -841,6 +887,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
> > > > .get_vq_affinity = vduse_vdpa_get_vq_affinity,
> > > > .reset = vduse_vdpa_reset,
> > > > .set_map = vduse_vdpa_set_map,
> > > > + .set_group_asid = vduse_set_group_asid,
> > > > .get_vq_map = vduse_get_vq_map,
> > > > .free = vduse_vdpa_free,
> > > > };
> > > > @@ -850,9 +897,12 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
> > > > enum dma_data_direction dir)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > >
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > vduse_domain_sync_single_for_device(domain, dma_addr, size, dir);
> > > > + read_unlock(&vdev->domain_lock);
> > > > }
> > > >
> > > > static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> > > > @@ -860,9 +910,12 @@ static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> > > > enum dma_data_direction dir)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > >
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > >
> > > I think the domain is better fetched via vduse_as.
> > >
> > > > vduse_domain_sync_single_for_cpu(domain, dma_addr, size, dir);
> > > > + read_unlock(&vdev->domain_lock);
> > > > }
> > > >
> > > > static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> > > > @@ -871,9 +924,15 @@ static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> > > > unsigned long attrs)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > > + dma_addr_t r;
> > > >
> > > > - return vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > + r = vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> > > > + read_unlock(&vdev->domain_lock);
> > > > +
> > > > + return r;
> > > > }
> > > >
> > > > static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> > > > @@ -881,27 +940,31 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> > > > unsigned long attrs)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > >
> > > > - return vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > + vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> > > > + read_unlock(&vdev->domain_lock);
> > > > }
> > > >
> > > > static void *vduse_dev_alloc_coherent(union virtio_map token, size_t size,
> > > > dma_addr_t *dma_addr, gfp_t flag)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > > unsigned long iova;
> > > > - void *addr;
> > > > + void *addr = NULL;
> > > >
> > > > *dma_addr = DMA_MAPPING_ERROR;
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > addr = vduse_domain_alloc_coherent(domain, size,
> > > > (dma_addr_t *)&iova, flag);
> > > > - if (!addr)
> > > > - return NULL;
> > > > -
> > > > - *dma_addr = (dma_addr_t)iova;
> > > > + if (addr)
> > > > + *dma_addr = (dma_addr_t)iova;
> > > >
> > > > + read_unlock(&vdev->domain_lock);
> > > > return addr;
> > > > }
> > > >
> > > > @@ -910,17 +973,26 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
> > > > unsigned long attrs)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > >
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > vduse_domain_free_coherent(domain, size, vaddr, dma_addr, attrs);
> > > > + read_unlock(&vdev->domain_lock);
> > > > }
> > > >
> > > > static bool vduse_dev_need_sync(union virtio_map token, dma_addr_t dma_addr)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > > + size_t bounce_size;
> > > >
> > > > - return dma_addr < domain->bounce_size;
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > + bounce_size = domain->bounce_size;
> > > > + read_unlock(&vdev->domain_lock);
> > > > +
> > > > + return dma_addr < bounce_size;
> > > > }
> > > >
> > > > static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> > > > @@ -933,9 +1005,15 @@ static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> > > > static size_t vduse_dev_max_mapping_size(union virtio_map token)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > > + size_t bounce_size;
> > > > +
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > + bounce_size = domain->bounce_size;
> > > > + read_unlock(&vdev->domain_lock);
> > > >
> > > > - return domain->bounce_size;
> > > > + return bounce_size;
> > > > }
> > > >
> > > > static const struct virtio_map_ops vduse_map_ops = {
> > > > @@ -1075,39 +1153,40 @@ static int vduse_dev_queue_irq_work(struct vduse_dev *dev,
> > > > return ret;
> > > > }
> > > >
> > > > -static int vduse_dev_dereg_umem(struct vduse_dev *dev,
> > > > +static int vduse_dev_dereg_umem(struct vduse_dev *dev, u32 asid,
> > > > u64 iova, u64 size)
> > > > {
> > > > int ret;
> > > >
> > > > - mutex_lock(&dev->as.mem_lock);
> > > > + mutex_lock(&dev->as[asid].mem_lock);
> > > > ret = -ENOENT;
> > > > - if (!dev->as.umem)
> > > > + if (!dev->as[asid].umem)
> > > > goto unlock;
> > > >
> > > > ret = -EINVAL;
> > > > - if (!dev->as.domain)
> > > > + if (!dev->as[asid].domain)
> > > > goto unlock;
> > > >
> > > > - if (dev->as.umem->iova != iova || size != dev->as.domain->bounce_size)
> > > > + if (dev->as[asid].umem->iova != iova ||
> > > > + size != dev->as[asid].domain->bounce_size)
> > > > goto unlock;
> > > >
> > > > - vduse_domain_remove_user_bounce_pages(dev->as.domain);
> > > > - unpin_user_pages_dirty_lock(dev->as.umem->pages,
> > > > - dev->as.umem->npages, true);
> > > > - atomic64_sub(dev->as.umem->npages, &dev->as.umem->mm->pinned_vm);
> > > > - mmdrop(dev->as.umem->mm);
> > > > - vfree(dev->as.umem->pages);
> > > > - kfree(dev->as.umem);
> > > > - dev->as.umem = NULL;
> > > > + vduse_domain_remove_user_bounce_pages(dev->as[asid].domain);
> > > > + unpin_user_pages_dirty_lock(dev->as[asid].umem->pages,
> > > > + dev->as[asid].umem->npages, true);
> > > > + atomic64_sub(dev->as[asid].umem->npages, &dev->as[asid].umem->mm->pinned_vm);
> > > > + mmdrop(dev->as[asid].umem->mm);
> > > > + vfree(dev->as[asid].umem->pages);
> > > > + kfree(dev->as[asid].umem);
> > > > + dev->as[asid].umem = NULL;
> > >
> > > We can avoid those changeset if we do those in the previous correctly
> > > as it said it would make as an array.
> > >
> >
> > I'm not following it. Is that different from squashing this patch with
> > the previous one? I'm ok with doing it, but this changes are needed
> > either way.
>
> See my reply for whether we need to make it as an array in the last
> patch. I meant basically we only need to change them once, but this
> series touches this twice.
>
> Thanks
>
^ permalink raw reply [flat|nested] 27+ messages in thread* Re: [PATCH v2 6/7] vduse: add vq group asid support
2025-09-18 6:57 ` Eugenio Perez Martin
@ 2025-09-19 2:15 ` Jason Wang
0 siblings, 0 replies; 27+ messages in thread
From: Jason Wang @ 2025-09-19 2:15 UTC (permalink / raw)
To: Eugenio Perez Martin
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Thu, Sep 18, 2025 at 2:58 PM Eugenio Perez Martin
<eperezma@redhat.com> wrote:
>
> On Thu, Sep 18, 2025 at 8:07 AM Jason Wang <jasowang@redhat.com> wrote:
> >
> > On Thu, Sep 18, 2025 at 12:41 AM Eugenio Perez Martin
> > <eperezma@redhat.com> wrote:
> > >
> > > On Wed, Sep 17, 2025 at 10:56 AM Jason Wang <jasowang@redhat.com> wrote:
> > > >
> > > > On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
> > > > >
> > > > > Add support for assigning Address Space Identifiers (ASIDs) to each VQ
> > > > > group. This enables mapping each group into a distinct memory space.
> > > > >
> > > > > Now that the driver can change ASID in the middle of operation, the
> > > > > domain that each vq address point is also protected by domain_lock.
> > > > >
> > > > > Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> > > > > ---
> > > > > v2:
> > > > > * Convert the use of mutex to rwlock.
> > > > >
> > > > > RFC v3:
> > > > > * Increase VDUSE_MAX_VQ_GROUPS to 0xffff (Jason). It was set to a lower
> > > > > value to reduce memory consumption, but vqs are already limited to
> > > > > that value and userspace VDUSE is able to allocate that many vqs.
> > > > > * Remove TODO about merging VDUSE_IOTLB_GET_FD ioctl with
> > > > > VDUSE_IOTLB_GET_INFO.
> > > > > * Use of array_index_nospec in VDUSE device ioctls.
> > > > > * Embed vduse_iotlb_entry into vduse_iotlb_entry_v2.
> > > > > * Move the umem mutex to asid struct so there is no contention between
> > > > > ASIDs.
> > > > >
> > > > > RFC v2:
> > > > > * Make iotlb entry the last one of vduse_iotlb_entry_v2 so the first
> > > > > part of the struct is the same.
> > > > > ---
> > > > > drivers/vdpa/vdpa_user/vduse_dev.c | 324 +++++++++++++++++++++--------
> > > > > include/uapi/linux/vduse.h | 51 ++++-
> > > > > 2 files changed, 284 insertions(+), 91 deletions(-)
> > > > >
> > > > > diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > > > index b45b1d22784f..06b7790380b7 100644
> > > > > --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> > > > > +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > > > @@ -93,6 +93,7 @@ struct vduse_as {
> > > > > };
> > > > >
> > > > > struct vduse_vq_group_int {
> > > > > + struct vduse_iova_domain *domain;
> > > > > struct vduse_dev *dev;
> > > >
> > > > This confuses me, I think it should be an asid. And the vduse_dev
> > > > pointer seems to be useless here.
> > > >
> > >
> > > The *dev pointer is used to take the rwlock, in case the vhost driver
> > > calls VHOST_VDPA_SET_GROUP_ASID (or equivalent) at the same time
> > > vduse_dev_sync_single_for_device (or _for_cpu, or equivalent) run.
> >
> > Ok, but having it in a group seems odd. A better place is to move it to the as?
> >
>
> If we move to the as, we need to take the lock this way:
>
> read_lock(vduse_vq_group->as->dev->domain_lock)
>
> But the lock is protecting the vduse_vq_group->as change itself, as
> other ioctls (SET_GROUP_ASID mainly) can modify it concurrently.
>
> We can assume that *as will always point to a valid as, its change
> will be atomic, and all the as->dev points to the exact same
> vduse_dev. I can buy that, but it is a race anyway.
>
I see, let's keep the struct vduse_dev *dev pointer as is.
Thanks
^ permalink raw reply [flat|nested] 27+ messages in thread
* Re: [PATCH v2 6/7] vduse: add vq group asid support
2025-09-17 16:40 ` Eugenio Perez Martin
2025-09-18 6:07 ` Jason Wang
@ 2025-09-18 11:21 ` Eugenio Perez Martin
2025-09-19 2:14 ` Jason Wang
1 sibling, 1 reply; 27+ messages in thread
From: Eugenio Perez Martin @ 2025-09-18 11:21 UTC (permalink / raw)
To: Jason Wang
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Wed, Sep 17, 2025 at 6:40 PM Eugenio Perez Martin
<eperezma@redhat.com> wrote:
>
> On Wed, Sep 17, 2025 at 10:56 AM Jason Wang <jasowang@redhat.com> wrote:
> >
> > On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
> > >
> > > Add support for assigning Address Space Identifiers (ASIDs) to each VQ
> > > group. This enables mapping each group into a distinct memory space.
> > >
> > > Now that the driver can change ASID in the middle of operation, the
> > > domain that each vq address point is also protected by domain_lock.
> > >
> > > Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> > > ---
> > > v2:
> > > * Convert the use of mutex to rwlock.
> > >
> > > RFC v3:
> > > * Increase VDUSE_MAX_VQ_GROUPS to 0xffff (Jason). It was set to a lower
> > > value to reduce memory consumption, but vqs are already limited to
> > > that value and userspace VDUSE is able to allocate that many vqs.
> > > * Remove TODO about merging VDUSE_IOTLB_GET_FD ioctl with
> > > VDUSE_IOTLB_GET_INFO.
> > > * Use of array_index_nospec in VDUSE device ioctls.
> > > * Embed vduse_iotlb_entry into vduse_iotlb_entry_v2.
> > > * Move the umem mutex to asid struct so there is no contention between
> > > ASIDs.
> > >
> > > RFC v2:
> > > * Make iotlb entry the last one of vduse_iotlb_entry_v2 so the first
> > > part of the struct is the same.
> > > ---
> > > drivers/vdpa/vdpa_user/vduse_dev.c | 324 +++++++++++++++++++++--------
> > > include/uapi/linux/vduse.h | 51 ++++-
> > > 2 files changed, 284 insertions(+), 91 deletions(-)
> > >
> > > diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > index b45b1d22784f..06b7790380b7 100644
> > > --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> > > +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > @@ -93,6 +93,7 @@ struct vduse_as {
> > > };
> > >
> > > struct vduse_vq_group_int {
> > > + struct vduse_iova_domain *domain;
> > > struct vduse_dev *dev;
> >
> > This confuses me, I think it should be an asid. And the vduse_dev
> > pointer seems to be useless here.
> >
>
> The *dev pointer is used to take the rwlock, in case the vhost driver
> calls VHOST_VDPA_SET_GROUP_ASID (or equivalent) at the same time
> vduse_dev_sync_single_for_device (or _for_cpu, or equivalent) run.
>
> > > };
> > >
> > > @@ -100,7 +101,7 @@ struct vduse_dev {
> > > struct vduse_vdpa *vdev;
> > > struct device *dev;
> > > struct vduse_virtqueue **vqs;
> > > - struct vduse_as as;
> > > + struct vduse_as *as;
> > > char *name;
> > > struct mutex lock;
> > > spinlock_t msg_lock;
> > > @@ -128,6 +129,7 @@ struct vduse_dev {
> > > u32 vq_num;
> > > u32 vq_align;
> > > u32 ngroups;
> > > + u32 nas;
> > > struct vduse_vq_group_int *groups;
> > > unsigned int bounce_size;
> > > rwlock_t domain_lock;
> > > @@ -318,7 +320,7 @@ static int vduse_dev_set_status(struct vduse_dev *dev, u8 status)
> > > return vduse_dev_msg_sync(dev, &msg);
> > > }
> > >
> > > -static int vduse_dev_update_iotlb(struct vduse_dev *dev,
> > > +static int vduse_dev_update_iotlb(struct vduse_dev *dev, u32 asid,
> > > u64 start, u64 last)
> > > {
> > > struct vduse_dev_msg msg = { 0 };
> > > @@ -327,8 +329,14 @@ static int vduse_dev_update_iotlb(struct vduse_dev *dev,
> > > return -EINVAL;
> > >
> > > msg.req.type = VDUSE_UPDATE_IOTLB;
> > > - msg.req.iova.start = start;
> > > - msg.req.iova.last = last;
> > > + if (dev->api_version < VDUSE_API_VERSION_1) {
> > > + msg.req.iova.start = start;
> > > + msg.req.iova.last = last;
> > > + } else {
> > > + msg.req.iova_v2.start = start;
> > > + msg.req.iova_v2.last = last;
> > > + msg.req.iova_v2.asid = asid;
> > > + }
> > >
> > > return vduse_dev_msg_sync(dev, &msg);
> > > }
> > > @@ -440,14 +448,29 @@ static __poll_t vduse_dev_poll(struct file *file, poll_table *wait)
> > > return mask;
> > > }
> > >
> > > +/* Force set the asid to a vq group without a message to the VDUSE device */
> > > +static void vduse_set_group_asid_nomsg(struct vduse_dev *dev,
> > > + unsigned int group, unsigned int asid)
> > > +{
> > > + write_lock(&dev->domain_lock);
> > > + dev->groups[group].domain = dev->as[asid].domain;
> >
> > I think it would be better to stick the group->as an indirection which
> > should be .
> >
> > dev->groups.asid = asid;
> >
> > Or
> >
> > dev->group->as = as;
> >
>
> That involves an extra memory jump for functions that may be in the
> hot path. I've not profiled it, but I'm ok with changing it that way
> if you prefer.
>
> > > + write_unlock(&dev->domain_lock);
> > > +}
> > > +
> > > static void vduse_dev_reset(struct vduse_dev *dev)
> > > {
> > > int i;
> > > - struct vduse_iova_domain *domain = dev->as.domain;
> > >
> > > /* The coherent mappings are handled in vduse_dev_free_coherent() */
> > > - if (domain && domain->bounce_map)
> > > - vduse_domain_reset_bounce_map(domain);
> > > + for (i = 0; i < dev->nas; i++) {
> > > + struct vduse_iova_domain *domain = dev->as[i].domain;
> > > +
> > > + if (domain && domain->bounce_map)
> > > + vduse_domain_reset_bounce_map(domain);
> > > + }
> > > +
> > > + for (i = 0; i < dev->ngroups; i++)
> > > + vduse_set_group_asid_nomsg(dev, i, 0);
> > >
> > > down_write(&dev->rwsem);
> > >
> > > @@ -621,6 +644,29 @@ static union virtio_map vduse_get_vq_map(struct vdpa_device *vdpa, u16 idx)
> > > return ret;
> > > }
> > >
> > > +static int vduse_set_group_asid(struct vdpa_device *vdpa, unsigned int group,
> > > + unsigned int asid)
> > > +{
> > > + struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > > + struct vduse_dev_msg msg = { 0 };
> > > + int r;
> > > +
> > > + if (dev->api_version < VDUSE_API_VERSION_1 ||
> > > + group >= dev->ngroups || asid >= dev->nas)
> > > + return -EINVAL;
> > > +
> > > + msg.req.type = VDUSE_SET_VQ_GROUP_ASID;
> > > + msg.req.vq_group_asid.group = group;
> > > + msg.req.vq_group_asid.asid = asid;
> > > +
> > > + r = vduse_dev_msg_sync(dev, &msg);
> > > + if (r < 0)
> > > + return r;
> > > +
> > > + vduse_set_group_asid_nomsg(dev, group, asid);
> > > + return 0;
> > > +}
> > > +
> > > static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
> > > struct vdpa_vq_state *state)
> > > {
> > > @@ -792,13 +838,13 @@ static int vduse_vdpa_set_map(struct vdpa_device *vdpa,
> > > struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > > int ret;
> > >
> > > - ret = vduse_domain_set_map(dev->as.domain, iotlb);
> > > + ret = vduse_domain_set_map(dev->as[asid].domain, iotlb);
> > > if (ret)
> > > return ret;
> > >
> > > - ret = vduse_dev_update_iotlb(dev, 0ULL, ULLONG_MAX);
> > > + ret = vduse_dev_update_iotlb(dev, asid, 0ULL, ULLONG_MAX);
> > > if (ret) {
> > > - vduse_domain_clear_map(dev->as.domain, iotlb);
> > > + vduse_domain_clear_map(dev->as[asid].domain, iotlb);
> > > return ret;
> > > }
> > >
> > > @@ -841,6 +887,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
> > > .get_vq_affinity = vduse_vdpa_get_vq_affinity,
> > > .reset = vduse_vdpa_reset,
> > > .set_map = vduse_vdpa_set_map,
> > > + .set_group_asid = vduse_set_group_asid,
> > > .get_vq_map = vduse_get_vq_map,
> > > .free = vduse_vdpa_free,
> > > };
> > > @@ -850,9 +897,12 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
> > > enum dma_data_direction dir)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > >
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > vduse_domain_sync_single_for_device(domain, dma_addr, size, dir);
> > > + read_unlock(&vdev->domain_lock);
> > > }
> > >
> > > static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> > > @@ -860,9 +910,12 @@ static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> > > enum dma_data_direction dir)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > >
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> >
> > I think the domain is better fetched via vduse_as.
> >
> > > vduse_domain_sync_single_for_cpu(domain, dma_addr, size, dir);
> > > + read_unlock(&vdev->domain_lock);
> > > }
> > >
> > > static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> > > @@ -871,9 +924,15 @@ static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> > > unsigned long attrs)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > > + dma_addr_t r;
> > >
> > > - return vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > + r = vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> > > + read_unlock(&vdev->domain_lock);
> > > +
> > > + return r;
> > > }
> > >
> > > static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> > > @@ -881,27 +940,31 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> > > unsigned long attrs)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > >
> > > - return vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > + vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> > > + read_unlock(&vdev->domain_lock);
> > > }
> > >
> > > static void *vduse_dev_alloc_coherent(union virtio_map token, size_t size,
> > > dma_addr_t *dma_addr, gfp_t flag)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > > unsigned long iova;
> > > - void *addr;
> > > + void *addr = NULL;
> > >
> > > *dma_addr = DMA_MAPPING_ERROR;
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > addr = vduse_domain_alloc_coherent(domain, size,
> > > (dma_addr_t *)&iova, flag);
> > > - if (!addr)
> > > - return NULL;
> > > -
> > > - *dma_addr = (dma_addr_t)iova;
> > > + if (addr)
> > > + *dma_addr = (dma_addr_t)iova;
> > >
> > > + read_unlock(&vdev->domain_lock);
> > > return addr;
> > > }
> > >
> > > @@ -910,17 +973,26 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
> > > unsigned long attrs)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > >
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > vduse_domain_free_coherent(domain, size, vaddr, dma_addr, attrs);
> > > + read_unlock(&vdev->domain_lock);
> > > }
> > >
> > > static bool vduse_dev_need_sync(union virtio_map token, dma_addr_t dma_addr)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > > + size_t bounce_size;
> > >
> > > - return dma_addr < domain->bounce_size;
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > + bounce_size = domain->bounce_size;
> > > + read_unlock(&vdev->domain_lock);
> > > +
> > > + return dma_addr < bounce_size;
> > > }
> > >
> > > static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> > > @@ -933,9 +1005,15 @@ static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> > > static size_t vduse_dev_max_mapping_size(union virtio_map token)
> > > {
> > > struct vduse_dev *vdev = token.group->dev;
> > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > + struct vduse_iova_domain *domain;
> > > + size_t bounce_size;
> > > +
> > > + read_lock(&vdev->domain_lock);
> > > + domain = token.group->domain;
> > > + bounce_size = domain->bounce_size;
> > > + read_unlock(&vdev->domain_lock);
> > >
> > > - return domain->bounce_size;
> > > + return bounce_size;
> > > }
> > >
> > > static const struct virtio_map_ops vduse_map_ops = {
> > > @@ -1075,39 +1153,40 @@ static int vduse_dev_queue_irq_work(struct vduse_dev *dev,
> > > return ret;
> > > }
> > >
> > > -static int vduse_dev_dereg_umem(struct vduse_dev *dev,
> > > +static int vduse_dev_dereg_umem(struct vduse_dev *dev, u32 asid,
> > > u64 iova, u64 size)
> > > {
> > > int ret;
> > >
> > > - mutex_lock(&dev->as.mem_lock);
> > > + mutex_lock(&dev->as[asid].mem_lock);
> > > ret = -ENOENT;
> > > - if (!dev->as.umem)
> > > + if (!dev->as[asid].umem)
> > > goto unlock;
> > >
> > > ret = -EINVAL;
> > > - if (!dev->as.domain)
> > > + if (!dev->as[asid].domain)
> > > goto unlock;
> > >
> > > - if (dev->as.umem->iova != iova || size != dev->as.domain->bounce_size)
> > > + if (dev->as[asid].umem->iova != iova ||
> > > + size != dev->as[asid].domain->bounce_size)
> > > goto unlock;
> > >
> > > - vduse_domain_remove_user_bounce_pages(dev->as.domain);
> > > - unpin_user_pages_dirty_lock(dev->as.umem->pages,
> > > - dev->as.umem->npages, true);
> > > - atomic64_sub(dev->as.umem->npages, &dev->as.umem->mm->pinned_vm);
> > > - mmdrop(dev->as.umem->mm);
> > > - vfree(dev->as.umem->pages);
> > > - kfree(dev->as.umem);
> > > - dev->as.umem = NULL;
> > > + vduse_domain_remove_user_bounce_pages(dev->as[asid].domain);
> > > + unpin_user_pages_dirty_lock(dev->as[asid].umem->pages,
> > > + dev->as[asid].umem->npages, true);
> > > + atomic64_sub(dev->as[asid].umem->npages, &dev->as[asid].umem->mm->pinned_vm);
> > > + mmdrop(dev->as[asid].umem->mm);
> > > + vfree(dev->as[asid].umem->pages);
> > > + kfree(dev->as[asid].umem);
> > > + dev->as[asid].umem = NULL;
> >
> > We can avoid those changeset if we do those in the previous correctly
> > as it said it would make as an array.
> >
>
> I'm not following it. Is that different from squashing this patch with
> the previous one? I'm ok with doing it, but this changes are needed
> either way.
>
> > > ret = 0;
> > > unlock:
> > > - mutex_unlock(&dev->as.mem_lock);
> > > + mutex_unlock(&dev->as[asid].mem_lock);
> > > return ret;
> > > }
> > >
> > > static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > > - u64 iova, u64 uaddr, u64 size)
> > > + u32 asid, u64 iova, u64 uaddr, u64 size)
> > > {
> > > struct page **page_list = NULL;
> > > struct vduse_umem *umem = NULL;
> > > @@ -1115,14 +1194,14 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > > unsigned long npages, lock_limit;
> > > int ret;
> > >
> > > - if (!dev->as.domain || !dev->as.domain->bounce_map ||
> > > - size != dev->as.domain->bounce_size ||
> > > + if (!dev->as[asid].domain || !dev->as[asid].domain->bounce_map ||
> > > + size != dev->as[asid].domain->bounce_size ||
> > > iova != 0 || uaddr & ~PAGE_MASK)
> > > return -EINVAL;
> > >
> > > - mutex_lock(&dev->as.mem_lock);
> > > + mutex_lock(&dev->as[asid].mem_lock);
> > > ret = -EEXIST;
> > > - if (dev->as.umem)
> > > + if (dev->as[asid].umem)
> > > goto unlock;
> > >
> > > ret = -ENOMEM;
> > > @@ -1146,7 +1225,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > > goto out;
> > > }
> > >
> > > - ret = vduse_domain_add_user_bounce_pages(dev->as.domain,
> > > + ret = vduse_domain_add_user_bounce_pages(dev->as[asid].domain,
> > > page_list, pinned);
> > > if (ret)
> > > goto out;
> > > @@ -1159,7 +1238,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > > umem->mm = current->mm;
> > > mmgrab(current->mm);
> > >
> > > - dev->as.umem = umem;
> > > + dev->as[asid].umem = umem;
> > > out:
> > > if (ret && pinned > 0)
> > > unpin_user_pages(page_list, pinned);
> > > @@ -1170,7 +1249,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > > vfree(page_list);
> > > kfree(umem);
> > > }
> > > - mutex_unlock(&dev->as.mem_lock);
> > > + mutex_unlock(&dev->as[asid].mem_lock);
> > > return ret;
> > > }
> > >
> > > @@ -1202,47 +1281,66 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > >
> > > switch (cmd) {
> > > case VDUSE_IOTLB_GET_FD: {
> > > - struct vduse_iotlb_entry entry;
> > > + struct vduse_iotlb_entry_v2 entry;
> > > struct vhost_iotlb_map *map;
> > > struct vdpa_map_file *map_file;
> > > struct file *f = NULL;
> > > + u32 asid;
> > >
> > > ret = -EFAULT;
> > > - if (copy_from_user(&entry, argp, sizeof(entry)))
> > > - break;
> > > + if (dev->api_version >= VDUSE_API_VERSION_1) {
> > > + if (copy_from_user(&entry, argp, sizeof(entry)))
> > > + break;
> > > + } else {
> > > + entry.asid = 0;
> > > + if (copy_from_user(&entry.v1, argp,
> > > + sizeof(entry.v1)))
> > > + break;
> > > + }
> > >
> > > ret = -EINVAL;
> > > - if (entry.start > entry.last)
> > > + if (entry.v1.start > entry.v1.last)
> > > + break;
> > > +
> > > + if (entry.asid >= dev->nas)
> > > break;
> > >
> > > read_lock(&dev->domain_lock);
> > > - if (!dev->as.domain) {
> > > + asid = array_index_nospec(entry.asid, dev->nas);
> > > + if (!dev->as[asid].domain) {
> > > read_unlock(&dev->domain_lock);
> > > break;
> > > }
> > > - spin_lock(&dev->as.domain->iotlb_lock);
> > > - map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
> > > - entry.start, entry.last);
> > > + spin_lock(&dev->as[asid].domain->iotlb_lock);
> > > + map = vhost_iotlb_itree_first(dev->as[asid].domain->iotlb,
> > > + entry.v1.start, entry.v1.last);
> > > if (map) {
> > > map_file = (struct vdpa_map_file *)map->opaque;
> > > f = get_file(map_file->file);
> > > - entry.offset = map_file->offset;
> > > - entry.start = map->start;
> > > - entry.last = map->last;
> > > - entry.perm = map->perm;
> > > + entry.v1.offset = map_file->offset;
> > > + entry.v1.start = map->start;
> > > + entry.v1.last = map->last;
> > > + entry.v1.perm = map->perm;
> > > }
> > > - spin_unlock(&dev->as.domain->iotlb_lock);
> > > + spin_unlock(&dev->as[asid].domain->iotlb_lock);
> > > read_unlock(&dev->domain_lock);
> > > ret = -EINVAL;
> > > if (!f)
> > > break;
> > >
> > > ret = -EFAULT;
> > > - if (copy_to_user(argp, &entry, sizeof(entry))) {
> > > + if (dev->api_version >= VDUSE_API_VERSION_1)
> > > + ret = copy_to_user(argp, &entry,
> > > + sizeof(entry));
> > > + else
> > > + ret = copy_to_user(argp, &entry.v1,
> > > + sizeof(entry.v1));
> > > +
> > > + if (ret) {
> > > fput(f);
> > > break;
> > > }
> > > - ret = receive_fd(f, NULL, perm_to_file_flags(entry.perm));
> > > + ret = receive_fd(f, NULL, perm_to_file_flags(entry.v1.perm));
> >
> > Nit: if we copy_from_user() twice and stick entry for v1 format, we
> > can avoid a lot of lines of changes.
> >
>
> Let me draft something and put it as a reply here to check I'm
> understanding your proposal.
>
We need a "struct vduse_iotlb_entry_v2" sooner or later anyway because
we need to finish with the corresponding copy_to_user. Either that, or
duplicate the copy_to_user too with something like:
copy_to_user(argp, &entry_vq, sizeof(entry_v1);
copy_to_user(argp + sizeof(struct vduse_iotlb_entry), &asid, sizeof(asid);
Saving the struct vduse_iotlb_entry_v2, this is the only change I can
do in that direction (from this patch):
diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c
b/drivers/vdpa/vdpa_user/vduse_dev.c
index 7da248f5616c..555b0fa079de 100644
--- a/drivers/vdpa/vdpa_user/vduse_dev.c
+++ b/drivers/vdpa/vdpa_user/vduse_dev.c
@@ -1341,14 +1341,13 @@ static long vduse_dev_ioctl(struct file *file,
unsigned int cmd,
u32 asid;
ret = -EFAULT;
- if (dev->api_version >= VDUSE_API_VERSION_1) {
- if (copy_from_user(&entry, argp, sizeof(entry)))
- break;
- } else {
+ if (copy_from_user(&entry.v1, argp, sizeof(entry.v1)))
+ break;
+ if (dev->api_version < VDUSE_API_VERSION_1) {
entry.asid = 0;
- if (copy_from_user(&entry.v1, argp,
- sizeof(entry.v1)))
- break;
+ } else if (copy_from_user(&entry.asid, argp,
+ sizeof(entry.asid))) {
+ break;
}
ret = -EINVAL;
> > > fput(f);
> > > break;
> > > }
> > > @@ -1384,6 +1482,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > > }
> > > case VDUSE_IOTLB_REG_UMEM: {
> > > struct vduse_iova_umem umem;
> > > + u32 asid;
> > >
> > > ret = -EFAULT;
> > > if (copy_from_user(&umem, argp, sizeof(umem)))
> > > @@ -1391,17 +1490,21 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > >
> > > ret = -EINVAL;
> > > if (!is_mem_zero((const char *)umem.reserved,
> > > - sizeof(umem.reserved)))
> > > + sizeof(umem.reserved)) ||
> > > + (dev->api_version < VDUSE_API_VERSION_1 &&
> > > + umem.asid != 0) || umem.asid >= dev->nas)
> > > break;
> > >
> > > write_lock(&dev->domain_lock);
> > > - ret = vduse_dev_reg_umem(dev, umem.iova,
> > > + asid = array_index_nospec(umem.asid, dev->nas);
> > > + ret = vduse_dev_reg_umem(dev, asid, umem.iova,
> > > umem.uaddr, umem.size);
> > > write_unlock(&dev->domain_lock);
> > > break;
> > > }
> > > case VDUSE_IOTLB_DEREG_UMEM: {
> > > struct vduse_iova_umem umem;
> > > + u32 asid;
> > >
> > > ret = -EFAULT;
> > > if (copy_from_user(&umem, argp, sizeof(umem)))
> > > @@ -1409,10 +1512,15 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > >
> > > ret = -EINVAL;
> > > if (!is_mem_zero((const char *)umem.reserved,
> > > - sizeof(umem.reserved)))
> > > + sizeof(umem.reserved)) ||
> > > + (dev->api_version < VDUSE_API_VERSION_1 &&
> > > + umem.asid != 0) ||
> > > + umem.asid >= dev->nas)
> > > break;
> > > +
> > > write_lock(&dev->domain_lock);
> > > - ret = vduse_dev_dereg_umem(dev, umem.iova,
> > > + asid = array_index_nospec(umem.asid, dev->nas);
> > > + ret = vduse_dev_dereg_umem(dev, asid, umem.iova,
> > > umem.size);
> > > write_unlock(&dev->domain_lock);
> > > break;
> > > @@ -1420,6 +1528,7 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > > case VDUSE_IOTLB_GET_INFO: {
> > > struct vduse_iova_info info;
> > > struct vhost_iotlb_map *map;
> > > + u32 asid;
> > >
> > > ret = -EFAULT;
> > > if (copy_from_user(&info, argp, sizeof(info)))
> > > @@ -1433,23 +1542,31 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > > sizeof(info.reserved)))
> > > break;
> > >
> > > + if (dev->api_version < VDUSE_API_VERSION_1) {
> > > + if (info.asid)
> > > + break;
> > > + } else if (info.asid >= dev->nas)
> > > + break;
> > > +
> > > read_lock(&dev->domain_lock);
> > > - if (!dev->as.domain) {
> > > + asid = array_index_nospec(info.asid, dev->nas);
> > > + if (!dev->as[asid].domain) {
> > > read_unlock(&dev->domain_lock);
> > > break;
> > > }
> > > - spin_lock(&dev->as.domain->iotlb_lock);
> > > - map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
> > > + spin_lock(&dev->as[asid].domain->iotlb_lock);
> > > + map = vhost_iotlb_itree_first(dev->as[asid].domain->iotlb,
> > > info.start, info.last);
> > > if (map) {
> > > info.start = map->start;
> > > info.last = map->last;
> > > info.capability = 0;
> > > - if (dev->as.domain->bounce_map && map->start == 0 &&
> > > - map->last == dev->as.domain->bounce_size - 1)
> > > + if (dev->as[asid].domain->bounce_map &&
> > > + map->start == 0 &&
> > > + map->last == dev->as[asid].domain->bounce_size - 1)
> > > info.capability |= VDUSE_IOVA_CAP_UMEM;
> > > }
> > > - spin_unlock(&dev->as.domain->iotlb_lock);
> > > + spin_unlock(&dev->as[asid].domain->iotlb_lock);
> > > read_unlock(&dev->domain_lock);
> > > if (!map)
> > > break;
> > > @@ -1474,8 +1591,10 @@ static int vduse_dev_release(struct inode *inode, struct file *file)
> > > struct vduse_dev *dev = file->private_data;
> > >
> > > write_lock(&dev->domain_lock);
> > > - if (dev->as.domain)
> > > - vduse_dev_dereg_umem(dev, 0, dev->as.domain->bounce_size);
> > > + for (int i = 0; i < dev->nas; i++)
> > > + if (dev->as[i].domain)
> > > + vduse_dev_dereg_umem(dev, i, 0,
> > > + dev->as[i].domain->bounce_size);
> > > write_unlock(&dev->domain_lock);
> > > spin_lock(&dev->msg_lock);
> > > /* Make sure the inflight messages can processed after reconncection */
> > > @@ -1694,7 +1813,6 @@ static struct vduse_dev *vduse_dev_create(void)
> > > return NULL;
> > >
> > > mutex_init(&dev->lock);
> > > - mutex_init(&dev->as.mem_lock);
> > > rwlock_init(&dev->domain_lock);
> > > spin_lock_init(&dev->msg_lock);
> > > INIT_LIST_HEAD(&dev->send_list);
> > > @@ -1745,8 +1863,11 @@ static int vduse_destroy_dev(char *name)
> > > idr_remove(&vduse_idr, dev->minor);
> > > kvfree(dev->config);
> > > vduse_dev_deinit_vqs(dev);
> > > - if (dev->as.domain)
> > > - vduse_domain_destroy(dev->as.domain);
> > > + for (int i = 0; i < dev->nas; i++) {
> > > + if (dev->as[i].domain)
> > > + vduse_domain_destroy(dev->as[i].domain);
> > > + }
> > > + kfree(dev->as);
> > > kfree(dev->name);
> > > kfree(dev->groups);
> > > vduse_dev_destroy(dev);
> > > @@ -1793,12 +1914,16 @@ static bool vduse_validate_config(struct vduse_dev_config *config,
> > > sizeof(config->reserved)))
> > > return false;
> > >
> > > - if (api_version < VDUSE_API_VERSION_1 && config->ngroups)
> > > + if (api_version < VDUSE_API_VERSION_1 &&
> > > + (config->ngroups || config->nas))
> > > return false;
> > >
> > > if (api_version >= VDUSE_API_VERSION_1 && config->ngroups > 0xffff)
> > > return false;
> > >
> > > + if (api_version >= VDUSE_API_VERSION_1 && config->nas > 0xffff)
> > > + return false;
> > > +
> > > if (config->vq_align > PAGE_SIZE)
> > > return false;
> > >
> > > @@ -1862,7 +1987,8 @@ static ssize_t bounce_size_store(struct device *device,
> > >
> > > ret = -EPERM;
> > > write_lock(&dev->domain_lock);
> > > - if (dev->as.domain)
> > > + /* Assuming that if the first domain is allocated, all are allocated */
> > > + if (dev->as[0].domain)
> > > goto unlock;
> > >
> > > ret = kstrtouint(buf, 10, &bounce_size);
> > > @@ -1923,6 +2049,13 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> > > for (u32 i = 0; i < dev->ngroups; ++i)
> > > dev->groups[i].dev = dev;
> > >
> > > + dev->nas = (dev->api_version < 1) ? 1 : (config->nas ?: 1);
> > > + dev->as = kcalloc(dev->nas, sizeof(dev->as[0]), GFP_KERNEL);
> > > + if (!dev->as)
> > > + goto err_as;
> > > + for (int i = 0; i < dev->nas; i++)
> > > + mutex_init(&dev->as[i].mem_lock);
> > > +
> > > dev->name = kstrdup(config->name, GFP_KERNEL);
> > > if (!dev->name)
> > > goto err_str;
> > > @@ -1959,6 +2092,8 @@ static int vduse_create_dev(struct vduse_dev_config *config,
> > > err_idr:
> > > kfree(dev->name);
> > > err_str:
> > > + kfree(dev->as);
> > > +err_as:
> > > kfree(dev->groups);
> > > err_vq_groups:
> > > vduse_dev_destroy(dev);
> > > @@ -2084,7 +2219,7 @@ static int vduse_dev_init_vdpa(struct vduse_dev *dev, const char *name)
> > >
> > > vdev = vdpa_alloc_device(struct vduse_vdpa, vdpa, dev->dev,
> > > &vduse_vdpa_config_ops, &vduse_map_ops,
> > > - dev->ngroups, 1, name, true);
> > > + dev->ngroups, dev->nas, name, true);
> > > if (IS_ERR(vdev))
> > > return PTR_ERR(vdev);
> > >
> > > @@ -2113,11 +2248,20 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> > > return ret;
> > >
> > > write_lock(&dev->domain_lock);
> > > - if (!dev->as.domain)
> > > - dev->as.domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
> > > - dev->bounce_size);
> > > + ret = 0;
> > > +
> > > + for (int i = 0; i < dev->nas; ++i) {
> > > + dev->as[i].domain = vduse_domain_create(VDUSE_IOVA_SIZE - 1,
> > > + dev->bounce_size);
> > > + if (!dev->as[i].domain) {
> > > + ret = -ENOMEM;
> > > + for (int j = 0; j < i; ++j)
> > > + vduse_domain_destroy(dev->as[j].domain);
> > > + }
> > > + }
> > > +
> > > write_unlock(&dev->domain_lock);
> > > - if (!dev->as.domain) {
> > > + if (ret == -ENOMEM) {
> > > put_device(&dev->vdev->vdpa.dev);
> > > return -ENOMEM;
> > > }
> > > @@ -2126,8 +2270,12 @@ static int vdpa_dev_add(struct vdpa_mgmt_dev *mdev, const char *name,
> > > if (ret) {
> > > put_device(&dev->vdev->vdpa.dev);
> > > write_lock(&dev->domain_lock);
> > > - vduse_domain_destroy(dev->as.domain);
> > > - dev->as.domain = NULL;
> > > + for (int i = 0; i < dev->nas; i++) {
> > > + if (dev->as[i].domain) {
> > > + vduse_domain_destroy(dev->as[i].domain);
> > > + dev->as[i].domain = NULL;
> > > + }
> > > + }
> > > write_unlock(&dev->domain_lock);
> > > return ret;
> > > }
> > > diff --git a/include/uapi/linux/vduse.h b/include/uapi/linux/vduse.h
> > > index a3d51cf6df3a..da73c3f2c280 100644
> > > --- a/include/uapi/linux/vduse.h
> > > +++ b/include/uapi/linux/vduse.h
> > > @@ -47,7 +47,8 @@ struct vduse_dev_config {
> > > __u32 vq_num;
> > > __u32 vq_align;
> > > __u32 ngroups; /* if VDUSE_API_VERSION >= 1 */
> > > - __u32 reserved[12];
> > > + __u32 nas; /* if VDUSE_API_VERSION >= 1 */
> > > + __u32 reserved[11];
> > > __u32 config_size;
> > > __u8 config[];
> > > };
> > > @@ -82,6 +83,18 @@ struct vduse_iotlb_entry {
> > > __u8 perm;
> > > };
> > >
> > > +/**
> > > + * struct vduse_iotlb_entry_v2 - entry of IOTLB to describe one IOVA region in an ASID
> > > + * @v1: the original vduse_iotlb_entry
> > > + * @asid: address space ID of the IOVA region
> > > + *
> > > + * Structure used by VDUSE_IOTLB_GET_FD ioctl to find an overlapped IOVA region.
> > > + */
> > > +struct vduse_iotlb_entry_v2 {
> > > + struct vduse_iotlb_entry v1;
> > > + __u32 asid;
> > > +};
> > > +
> > > /*
> > > * Find the first IOVA region that overlaps with the range [start, last]
> > > * and return the corresponding file descriptor. Return -EINVAL means the
> > > @@ -166,6 +179,16 @@ struct vduse_vq_state_packed {
> > > __u16 last_used_idx;
> > > };
> > >
> > > +/**
> > > + * struct vduse_vq_group - virtqueue group
> > > + @ @group: Index of the virtqueue group
> > > + * @asid: Address space ID of the group
> > > + */
> > > +struct vduse_vq_group_asid {
> > > + __u32 group;
> > > + __u32 asid;
> > > +};
> > > +
> > > /**
> > > * struct vduse_vq_info - information of a virtqueue
> > > * @index: virtqueue index
> > > @@ -225,6 +248,7 @@ struct vduse_vq_eventfd {
> > > * @uaddr: start address of userspace memory, it must be aligned to page size
> > > * @iova: start of the IOVA region
> > > * @size: size of the IOVA region
> > > + * @asid: Address space ID of the IOVA region
> > > * @reserved: for future use, needs to be initialized to zero
> > > *
> > > * Structure used by VDUSE_IOTLB_REG_UMEM and VDUSE_IOTLB_DEREG_UMEM
> > > @@ -234,7 +258,8 @@ struct vduse_iova_umem {
> > > __u64 uaddr;
> > > __u64 iova;
> > > __u64 size;
> > > - __u64 reserved[3];
> > > + __u32 asid;
> > > + __u32 reserved[5];
> > > };
> > >
> > > /* Register userspace memory for IOVA regions */
> > > @@ -248,6 +273,7 @@ struct vduse_iova_umem {
> > > * @start: start of the IOVA region
> > > * @last: last of the IOVA region
> > > * @capability: capability of the IOVA region
> > > + * @asid: Address space ID of the IOVA region, only if device API version >= 1
> > > * @reserved: for future use, needs to be initialized to zero
> > > *
> > > * Structure used by VDUSE_IOTLB_GET_INFO ioctl to get information of
> > > @@ -258,7 +284,8 @@ struct vduse_iova_info {
> > > __u64 last;
> > > #define VDUSE_IOVA_CAP_UMEM (1 << 0)
> > > __u64 capability;
> > > - __u64 reserved[3];
> > > + __u32 asid; /* Only if device API version >= 1 */
> > > + __u32 reserved[5];
> > > };
> > >
> > > /*
> > > @@ -280,6 +307,7 @@ enum vduse_req_type {
> > > VDUSE_GET_VQ_STATE,
> > > VDUSE_SET_STATUS,
> > > VDUSE_UPDATE_IOTLB,
> > > + VDUSE_SET_VQ_GROUP_ASID,
> > > };
> > >
> > > /**
> > > @@ -314,6 +342,18 @@ struct vduse_iova_range {
> > > __u64 last;
> > > };
> > >
> > > +/**
> > > + * struct vduse_iova_range - IOVA range [start, last] if API_VERSION >= 1
> > > + * @start: start of the IOVA range
> > > + * @last: last of the IOVA range
> > > + * @asid: address space ID of the IOVA range
> > > + */
> > > +struct vduse_iova_range_v2 {
> > > + __u64 start;
> > > + __u64 last;
> > > + __u32 asid;
> > > +};
> > > +
> > > /**
> > > * struct vduse_dev_request - control request
> > > * @type: request type
> > > @@ -322,6 +362,8 @@ struct vduse_iova_range {
> > > * @vq_state: virtqueue state, only index field is available
> > > * @s: device status
> > > * @iova: IOVA range for updating
> > > + * @iova_v2: IOVA range for updating if API_VERSION >= 1
> > > + * @vq_group_asid: ASID of a virtqueue group
> > > * @padding: padding
> > > *
> > > * Structure used by read(2) on /dev/vduse/$NAME.
> > > @@ -334,6 +376,9 @@ struct vduse_dev_request {
> > > struct vduse_vq_state vq_state;
> > > struct vduse_dev_status s;
> > > struct vduse_iova_range iova;
> > > + /* Following members only if vduse api version >= 1 */;
> > > + struct vduse_iova_range_v2 iova_v2;
> > > + struct vduse_vq_group_asid vq_group_asid;
> > > __u32 padding[32];
> > > };
> > > };
> > > --
> > > 2.51.0
> > >
> >
> > Thanks
> >
^ permalink raw reply related [flat|nested] 27+ messages in thread* Re: [PATCH v2 6/7] vduse: add vq group asid support
2025-09-18 11:21 ` Eugenio Perez Martin
@ 2025-09-19 2:14 ` Jason Wang
0 siblings, 0 replies; 27+ messages in thread
From: Jason Wang @ 2025-09-19 2:14 UTC (permalink / raw)
To: Eugenio Perez Martin
Cc: Michael S . Tsirkin, Stefano Garzarella, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization
On Thu, Sep 18, 2025 at 7:22 PM Eugenio Perez Martin
<eperezma@redhat.com> wrote:
>
> On Wed, Sep 17, 2025 at 6:40 PM Eugenio Perez Martin
> <eperezma@redhat.com> wrote:
> >
> > On Wed, Sep 17, 2025 at 10:56 AM Jason Wang <jasowang@redhat.com> wrote:
> > >
> > > On Tue, Sep 16, 2025 at 9:09 PM Eugenio Pérez <eperezma@redhat.com> wrote:
> > > >
> > > > Add support for assigning Address Space Identifiers (ASIDs) to each VQ
> > > > group. This enables mapping each group into a distinct memory space.
> > > >
> > > > Now that the driver can change ASID in the middle of operation, the
> > > > domain that each vq address point is also protected by domain_lock.
> > > >
> > > > Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
> > > > ---
> > > > v2:
> > > > * Convert the use of mutex to rwlock.
> > > >
> > > > RFC v3:
> > > > * Increase VDUSE_MAX_VQ_GROUPS to 0xffff (Jason). It was set to a lower
> > > > value to reduce memory consumption, but vqs are already limited to
> > > > that value and userspace VDUSE is able to allocate that many vqs.
> > > > * Remove TODO about merging VDUSE_IOTLB_GET_FD ioctl with
> > > > VDUSE_IOTLB_GET_INFO.
> > > > * Use of array_index_nospec in VDUSE device ioctls.
> > > > * Embed vduse_iotlb_entry into vduse_iotlb_entry_v2.
> > > > * Move the umem mutex to asid struct so there is no contention between
> > > > ASIDs.
> > > >
> > > > RFC v2:
> > > > * Make iotlb entry the last one of vduse_iotlb_entry_v2 so the first
> > > > part of the struct is the same.
> > > > ---
> > > > drivers/vdpa/vdpa_user/vduse_dev.c | 324 +++++++++++++++++++++--------
> > > > include/uapi/linux/vduse.h | 51 ++++-
> > > > 2 files changed, 284 insertions(+), 91 deletions(-)
> > > >
> > > > diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > > index b45b1d22784f..06b7790380b7 100644
> > > > --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> > > > +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> > > > @@ -93,6 +93,7 @@ struct vduse_as {
> > > > };
> > > >
> > > > struct vduse_vq_group_int {
> > > > + struct vduse_iova_domain *domain;
> > > > struct vduse_dev *dev;
> > >
> > > This confuses me, I think it should be an asid. And the vduse_dev
> > > pointer seems to be useless here.
> > >
> >
> > The *dev pointer is used to take the rwlock, in case the vhost driver
> > calls VHOST_VDPA_SET_GROUP_ASID (or equivalent) at the same time
> > vduse_dev_sync_single_for_device (or _for_cpu, or equivalent) run.
> >
> > > > };
> > > >
> > > > @@ -100,7 +101,7 @@ struct vduse_dev {
> > > > struct vduse_vdpa *vdev;
> > > > struct device *dev;
> > > > struct vduse_virtqueue **vqs;
> > > > - struct vduse_as as;
> > > > + struct vduse_as *as;
> > > > char *name;
> > > > struct mutex lock;
> > > > spinlock_t msg_lock;
> > > > @@ -128,6 +129,7 @@ struct vduse_dev {
> > > > u32 vq_num;
> > > > u32 vq_align;
> > > > u32 ngroups;
> > > > + u32 nas;
> > > > struct vduse_vq_group_int *groups;
> > > > unsigned int bounce_size;
> > > > rwlock_t domain_lock;
> > > > @@ -318,7 +320,7 @@ static int vduse_dev_set_status(struct vduse_dev *dev, u8 status)
> > > > return vduse_dev_msg_sync(dev, &msg);
> > > > }
> > > >
> > > > -static int vduse_dev_update_iotlb(struct vduse_dev *dev,
> > > > +static int vduse_dev_update_iotlb(struct vduse_dev *dev, u32 asid,
> > > > u64 start, u64 last)
> > > > {
> > > > struct vduse_dev_msg msg = { 0 };
> > > > @@ -327,8 +329,14 @@ static int vduse_dev_update_iotlb(struct vduse_dev *dev,
> > > > return -EINVAL;
> > > >
> > > > msg.req.type = VDUSE_UPDATE_IOTLB;
> > > > - msg.req.iova.start = start;
> > > > - msg.req.iova.last = last;
> > > > + if (dev->api_version < VDUSE_API_VERSION_1) {
> > > > + msg.req.iova.start = start;
> > > > + msg.req.iova.last = last;
> > > > + } else {
> > > > + msg.req.iova_v2.start = start;
> > > > + msg.req.iova_v2.last = last;
> > > > + msg.req.iova_v2.asid = asid;
> > > > + }
> > > >
> > > > return vduse_dev_msg_sync(dev, &msg);
> > > > }
> > > > @@ -440,14 +448,29 @@ static __poll_t vduse_dev_poll(struct file *file, poll_table *wait)
> > > > return mask;
> > > > }
> > > >
> > > > +/* Force set the asid to a vq group without a message to the VDUSE device */
> > > > +static void vduse_set_group_asid_nomsg(struct vduse_dev *dev,
> > > > + unsigned int group, unsigned int asid)
> > > > +{
> > > > + write_lock(&dev->domain_lock);
> > > > + dev->groups[group].domain = dev->as[asid].domain;
> > >
> > > I think it would be better to stick the group->as an indirection which
> > > should be .
> > >
> > > dev->groups.asid = asid;
> > >
> > > Or
> > >
> > > dev->group->as = as;
> > >
> >
> > That involves an extra memory jump for functions that may be in the
> > hot path. I've not profiled it, but I'm ok with changing it that way
> > if you prefer.
> >
> > > > + write_unlock(&dev->domain_lock);
> > > > +}
> > > > +
> > > > static void vduse_dev_reset(struct vduse_dev *dev)
> > > > {
> > > > int i;
> > > > - struct vduse_iova_domain *domain = dev->as.domain;
> > > >
> > > > /* The coherent mappings are handled in vduse_dev_free_coherent() */
> > > > - if (domain && domain->bounce_map)
> > > > - vduse_domain_reset_bounce_map(domain);
> > > > + for (i = 0; i < dev->nas; i++) {
> > > > + struct vduse_iova_domain *domain = dev->as[i].domain;
> > > > +
> > > > + if (domain && domain->bounce_map)
> > > > + vduse_domain_reset_bounce_map(domain);
> > > > + }
> > > > +
> > > > + for (i = 0; i < dev->ngroups; i++)
> > > > + vduse_set_group_asid_nomsg(dev, i, 0);
> > > >
> > > > down_write(&dev->rwsem);
> > > >
> > > > @@ -621,6 +644,29 @@ static union virtio_map vduse_get_vq_map(struct vdpa_device *vdpa, u16 idx)
> > > > return ret;
> > > > }
> > > >
> > > > +static int vduse_set_group_asid(struct vdpa_device *vdpa, unsigned int group,
> > > > + unsigned int asid)
> > > > +{
> > > > + struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > > > + struct vduse_dev_msg msg = { 0 };
> > > > + int r;
> > > > +
> > > > + if (dev->api_version < VDUSE_API_VERSION_1 ||
> > > > + group >= dev->ngroups || asid >= dev->nas)
> > > > + return -EINVAL;
> > > > +
> > > > + msg.req.type = VDUSE_SET_VQ_GROUP_ASID;
> > > > + msg.req.vq_group_asid.group = group;
> > > > + msg.req.vq_group_asid.asid = asid;
> > > > +
> > > > + r = vduse_dev_msg_sync(dev, &msg);
> > > > + if (r < 0)
> > > > + return r;
> > > > +
> > > > + vduse_set_group_asid_nomsg(dev, group, asid);
> > > > + return 0;
> > > > +}
> > > > +
> > > > static int vduse_vdpa_get_vq_state(struct vdpa_device *vdpa, u16 idx,
> > > > struct vdpa_vq_state *state)
> > > > {
> > > > @@ -792,13 +838,13 @@ static int vduse_vdpa_set_map(struct vdpa_device *vdpa,
> > > > struct vduse_dev *dev = vdpa_to_vduse(vdpa);
> > > > int ret;
> > > >
> > > > - ret = vduse_domain_set_map(dev->as.domain, iotlb);
> > > > + ret = vduse_domain_set_map(dev->as[asid].domain, iotlb);
> > > > if (ret)
> > > > return ret;
> > > >
> > > > - ret = vduse_dev_update_iotlb(dev, 0ULL, ULLONG_MAX);
> > > > + ret = vduse_dev_update_iotlb(dev, asid, 0ULL, ULLONG_MAX);
> > > > if (ret) {
> > > > - vduse_domain_clear_map(dev->as.domain, iotlb);
> > > > + vduse_domain_clear_map(dev->as[asid].domain, iotlb);
> > > > return ret;
> > > > }
> > > >
> > > > @@ -841,6 +887,7 @@ static const struct vdpa_config_ops vduse_vdpa_config_ops = {
> > > > .get_vq_affinity = vduse_vdpa_get_vq_affinity,
> > > > .reset = vduse_vdpa_reset,
> > > > .set_map = vduse_vdpa_set_map,
> > > > + .set_group_asid = vduse_set_group_asid,
> > > > .get_vq_map = vduse_get_vq_map,
> > > > .free = vduse_vdpa_free,
> > > > };
> > > > @@ -850,9 +897,12 @@ static void vduse_dev_sync_single_for_device(union virtio_map token,
> > > > enum dma_data_direction dir)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > >
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > vduse_domain_sync_single_for_device(domain, dma_addr, size, dir);
> > > > + read_unlock(&vdev->domain_lock);
> > > > }
> > > >
> > > > static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> > > > @@ -860,9 +910,12 @@ static void vduse_dev_sync_single_for_cpu(union virtio_map token,
> > > > enum dma_data_direction dir)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > >
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > >
> > > I think the domain is better fetched via vduse_as.
> > >
> > > > vduse_domain_sync_single_for_cpu(domain, dma_addr, size, dir);
> > > > + read_unlock(&vdev->domain_lock);
> > > > }
> > > >
> > > > static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> > > > @@ -871,9 +924,15 @@ static dma_addr_t vduse_dev_map_page(union virtio_map token, struct page *page,
> > > > unsigned long attrs)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > > + dma_addr_t r;
> > > >
> > > > - return vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > + r = vduse_domain_map_page(domain, page, offset, size, dir, attrs);
> > > > + read_unlock(&vdev->domain_lock);
> > > > +
> > > > + return r;
> > > > }
> > > >
> > > > static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> > > > @@ -881,27 +940,31 @@ static void vduse_dev_unmap_page(union virtio_map token, dma_addr_t dma_addr,
> > > > unsigned long attrs)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > >
> > > > - return vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > + vduse_domain_unmap_page(domain, dma_addr, size, dir, attrs);
> > > > + read_unlock(&vdev->domain_lock);
> > > > }
> > > >
> > > > static void *vduse_dev_alloc_coherent(union virtio_map token, size_t size,
> > > > dma_addr_t *dma_addr, gfp_t flag)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > > unsigned long iova;
> > > > - void *addr;
> > > > + void *addr = NULL;
> > > >
> > > > *dma_addr = DMA_MAPPING_ERROR;
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > addr = vduse_domain_alloc_coherent(domain, size,
> > > > (dma_addr_t *)&iova, flag);
> > > > - if (!addr)
> > > > - return NULL;
> > > > -
> > > > - *dma_addr = (dma_addr_t)iova;
> > > > + if (addr)
> > > > + *dma_addr = (dma_addr_t)iova;
> > > >
> > > > + read_unlock(&vdev->domain_lock);
> > > > return addr;
> > > > }
> > > >
> > > > @@ -910,17 +973,26 @@ static void vduse_dev_free_coherent(union virtio_map token, size_t size,
> > > > unsigned long attrs)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > >
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > vduse_domain_free_coherent(domain, size, vaddr, dma_addr, attrs);
> > > > + read_unlock(&vdev->domain_lock);
> > > > }
> > > >
> > > > static bool vduse_dev_need_sync(union virtio_map token, dma_addr_t dma_addr)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > > + size_t bounce_size;
> > > >
> > > > - return dma_addr < domain->bounce_size;
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > + bounce_size = domain->bounce_size;
> > > > + read_unlock(&vdev->domain_lock);
> > > > +
> > > > + return dma_addr < bounce_size;
> > > > }
> > > >
> > > > static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> > > > @@ -933,9 +1005,15 @@ static int vduse_dev_mapping_error(union virtio_map token, dma_addr_t dma_addr)
> > > > static size_t vduse_dev_max_mapping_size(union virtio_map token)
> > > > {
> > > > struct vduse_dev *vdev = token.group->dev;
> > > > - struct vduse_iova_domain *domain = vdev->as.domain;
> > > > + struct vduse_iova_domain *domain;
> > > > + size_t bounce_size;
> > > > +
> > > > + read_lock(&vdev->domain_lock);
> > > > + domain = token.group->domain;
> > > > + bounce_size = domain->bounce_size;
> > > > + read_unlock(&vdev->domain_lock);
> > > >
> > > > - return domain->bounce_size;
> > > > + return bounce_size;
> > > > }
> > > >
> > > > static const struct virtio_map_ops vduse_map_ops = {
> > > > @@ -1075,39 +1153,40 @@ static int vduse_dev_queue_irq_work(struct vduse_dev *dev,
> > > > return ret;
> > > > }
> > > >
> > > > -static int vduse_dev_dereg_umem(struct vduse_dev *dev,
> > > > +static int vduse_dev_dereg_umem(struct vduse_dev *dev, u32 asid,
> > > > u64 iova, u64 size)
> > > > {
> > > > int ret;
> > > >
> > > > - mutex_lock(&dev->as.mem_lock);
> > > > + mutex_lock(&dev->as[asid].mem_lock);
> > > > ret = -ENOENT;
> > > > - if (!dev->as.umem)
> > > > + if (!dev->as[asid].umem)
> > > > goto unlock;
> > > >
> > > > ret = -EINVAL;
> > > > - if (!dev->as.domain)
> > > > + if (!dev->as[asid].domain)
> > > > goto unlock;
> > > >
> > > > - if (dev->as.umem->iova != iova || size != dev->as.domain->bounce_size)
> > > > + if (dev->as[asid].umem->iova != iova ||
> > > > + size != dev->as[asid].domain->bounce_size)
> > > > goto unlock;
> > > >
> > > > - vduse_domain_remove_user_bounce_pages(dev->as.domain);
> > > > - unpin_user_pages_dirty_lock(dev->as.umem->pages,
> > > > - dev->as.umem->npages, true);
> > > > - atomic64_sub(dev->as.umem->npages, &dev->as.umem->mm->pinned_vm);
> > > > - mmdrop(dev->as.umem->mm);
> > > > - vfree(dev->as.umem->pages);
> > > > - kfree(dev->as.umem);
> > > > - dev->as.umem = NULL;
> > > > + vduse_domain_remove_user_bounce_pages(dev->as[asid].domain);
> > > > + unpin_user_pages_dirty_lock(dev->as[asid].umem->pages,
> > > > + dev->as[asid].umem->npages, true);
> > > > + atomic64_sub(dev->as[asid].umem->npages, &dev->as[asid].umem->mm->pinned_vm);
> > > > + mmdrop(dev->as[asid].umem->mm);
> > > > + vfree(dev->as[asid].umem->pages);
> > > > + kfree(dev->as[asid].umem);
> > > > + dev->as[asid].umem = NULL;
> > >
> > > We can avoid those changeset if we do those in the previous correctly
> > > as it said it would make as an array.
> > >
> >
> > I'm not following it. Is that different from squashing this patch with
> > the previous one? I'm ok with doing it, but this changes are needed
> > either way.
> >
> > > > ret = 0;
> > > > unlock:
> > > > - mutex_unlock(&dev->as.mem_lock);
> > > > + mutex_unlock(&dev->as[asid].mem_lock);
> > > > return ret;
> > > > }
> > > >
> > > > static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > > > - u64 iova, u64 uaddr, u64 size)
> > > > + u32 asid, u64 iova, u64 uaddr, u64 size)
> > > > {
> > > > struct page **page_list = NULL;
> > > > struct vduse_umem *umem = NULL;
> > > > @@ -1115,14 +1194,14 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > > > unsigned long npages, lock_limit;
> > > > int ret;
> > > >
> > > > - if (!dev->as.domain || !dev->as.domain->bounce_map ||
> > > > - size != dev->as.domain->bounce_size ||
> > > > + if (!dev->as[asid].domain || !dev->as[asid].domain->bounce_map ||
> > > > + size != dev->as[asid].domain->bounce_size ||
> > > > iova != 0 || uaddr & ~PAGE_MASK)
> > > > return -EINVAL;
> > > >
> > > > - mutex_lock(&dev->as.mem_lock);
> > > > + mutex_lock(&dev->as[asid].mem_lock);
> > > > ret = -EEXIST;
> > > > - if (dev->as.umem)
> > > > + if (dev->as[asid].umem)
> > > > goto unlock;
> > > >
> > > > ret = -ENOMEM;
> > > > @@ -1146,7 +1225,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > > > goto out;
> > > > }
> > > >
> > > > - ret = vduse_domain_add_user_bounce_pages(dev->as.domain,
> > > > + ret = vduse_domain_add_user_bounce_pages(dev->as[asid].domain,
> > > > page_list, pinned);
> > > > if (ret)
> > > > goto out;
> > > > @@ -1159,7 +1238,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > > > umem->mm = current->mm;
> > > > mmgrab(current->mm);
> > > >
> > > > - dev->as.umem = umem;
> > > > + dev->as[asid].umem = umem;
> > > > out:
> > > > if (ret && pinned > 0)
> > > > unpin_user_pages(page_list, pinned);
> > > > @@ -1170,7 +1249,7 @@ static int vduse_dev_reg_umem(struct vduse_dev *dev,
> > > > vfree(page_list);
> > > > kfree(umem);
> > > > }
> > > > - mutex_unlock(&dev->as.mem_lock);
> > > > + mutex_unlock(&dev->as[asid].mem_lock);
> > > > return ret;
> > > > }
> > > >
> > > > @@ -1202,47 +1281,66 @@ static long vduse_dev_ioctl(struct file *file, unsigned int cmd,
> > > >
> > > > switch (cmd) {
> > > > case VDUSE_IOTLB_GET_FD: {
> > > > - struct vduse_iotlb_entry entry;
> > > > + struct vduse_iotlb_entry_v2 entry;
> > > > struct vhost_iotlb_map *map;
> > > > struct vdpa_map_file *map_file;
> > > > struct file *f = NULL;
> > > > + u32 asid;
> > > >
> > > > ret = -EFAULT;
> > > > - if (copy_from_user(&entry, argp, sizeof(entry)))
> > > > - break;
> > > > + if (dev->api_version >= VDUSE_API_VERSION_1) {
> > > > + if (copy_from_user(&entry, argp, sizeof(entry)))
> > > > + break;
> > > > + } else {
> > > > + entry.asid = 0;
> > > > + if (copy_from_user(&entry.v1, argp,
> > > > + sizeof(entry.v1)))
> > > > + break;
> > > > + }
> > > >
> > > > ret = -EINVAL;
> > > > - if (entry.start > entry.last)
> > > > + if (entry.v1.start > entry.v1.last)
> > > > + break;
> > > > +
> > > > + if (entry.asid >= dev->nas)
> > > > break;
> > > >
> > > > read_lock(&dev->domain_lock);
> > > > - if (!dev->as.domain) {
> > > > + asid = array_index_nospec(entry.asid, dev->nas);
> > > > + if (!dev->as[asid].domain) {
> > > > read_unlock(&dev->domain_lock);
> > > > break;
> > > > }
> > > > - spin_lock(&dev->as.domain->iotlb_lock);
> > > > - map = vhost_iotlb_itree_first(dev->as.domain->iotlb,
> > > > - entry.start, entry.last);
> > > > + spin_lock(&dev->as[asid].domain->iotlb_lock);
> > > > + map = vhost_iotlb_itree_first(dev->as[asid].domain->iotlb,
> > > > + entry.v1.start, entry.v1.last);
> > > > if (map) {
> > > > map_file = (struct vdpa_map_file *)map->opaque;
> > > > f = get_file(map_file->file);
> > > > - entry.offset = map_file->offset;
> > > > - entry.start = map->start;
> > > > - entry.last = map->last;
> > > > - entry.perm = map->perm;
> > > > + entry.v1.offset = map_file->offset;
> > > > + entry.v1.start = map->start;
> > > > + entry.v1.last = map->last;
> > > > + entry.v1.perm = map->perm;
> > > > }
> > > > - spin_unlock(&dev->as.domain->iotlb_lock);
> > > > + spin_unlock(&dev->as[asid].domain->iotlb_lock);
> > > > read_unlock(&dev->domain_lock);
> > > > ret = -EINVAL;
> > > > if (!f)
> > > > break;
> > > >
> > > > ret = -EFAULT;
> > > > - if (copy_to_user(argp, &entry, sizeof(entry))) {
> > > > + if (dev->api_version >= VDUSE_API_VERSION_1)
> > > > + ret = copy_to_user(argp, &entry,
> > > > + sizeof(entry));
> > > > + else
> > > > + ret = copy_to_user(argp, &entry.v1,
> > > > + sizeof(entry.v1));
> > > > +
> > > > + if (ret) {
> > > > fput(f);
> > > > break;
> > > > }
> > > > - ret = receive_fd(f, NULL, perm_to_file_flags(entry.perm));
> > > > + ret = receive_fd(f, NULL, perm_to_file_flags(entry.v1.perm));
> > >
> > > Nit: if we copy_from_user() twice and stick entry for v1 format, we
> > > can avoid a lot of lines of changes.
> > >
> >
> > Let me draft something and put it as a reply here to check I'm
> > understanding your proposal.
> >
>
> We need a "struct vduse_iotlb_entry_v2" sooner or later anyway because
> we need to finish with the corresponding copy_to_user. Either that, or
> duplicate the copy_to_user too with something like:
> copy_to_user(argp, &entry_vq, sizeof(entry_v1);
> copy_to_user(argp + sizeof(struct vduse_iotlb_entry), &asid, sizeof(asid);
>
> Saving the struct vduse_iotlb_entry_v2, this is the only change I can
> do in that direction (from this patch):
>
> diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c
> b/drivers/vdpa/vdpa_user/vduse_dev.c
> index 7da248f5616c..555b0fa079de 100644
> --- a/drivers/vdpa/vdpa_user/vduse_dev.c
> +++ b/drivers/vdpa/vdpa_user/vduse_dev.c
> @@ -1341,14 +1341,13 @@ static long vduse_dev_ioctl(struct file *file,
> unsigned int cmd,
> u32 asid;
>
> ret = -EFAULT;
> - if (dev->api_version >= VDUSE_API_VERSION_1) {
> - if (copy_from_user(&entry, argp, sizeof(entry)))
> - break;
> - } else {
> + if (copy_from_user(&entry.v1, argp, sizeof(entry.v1)))
> + break;
> + if (dev->api_version < VDUSE_API_VERSION_1) {
> entry.asid = 0;
> - if (copy_from_user(&entry.v1, argp,
> - sizeof(entry.v1)))
> - break;
> + } else if (copy_from_user(&entry.asid, argp,
> + sizeof(entry.asid))) {
> + break;
> }
>
> ret = -EINVAL;
Ok, I see, just choose the one that is easier for you.
Thanks
^ permalink raw reply [flat|nested] 27+ messages in thread
* [PATCH v2 7/7] vduse: bump version number
2025-09-16 13:08 [PATCH v2 0/7] Add multiple address spaces support to VDUSE Eugenio Pérez
` (5 preceding siblings ...)
2025-09-16 13:08 ` [PATCH v2 6/7] vduse: add vq group asid support Eugenio Pérez
@ 2025-09-16 13:08 ` Eugenio Pérez
6 siblings, 0 replies; 27+ messages in thread
From: Eugenio Pérez @ 2025-09-16 13:08 UTC (permalink / raw)
To: Michael S . Tsirkin
Cc: Stefano Garzarella, jasowang, Xuan Zhuo, linux-kernel,
Maxime Coquelin, Yongji Xie, Cindy Lu, Laurent Vivier,
virtualization, Eugenio Pérez
Finalize the series by advertising VDUSE API v1 support to userspace.
Now that all required infrastructure for v1 (ASIDs, VQ groups,
update_iotlb_v2) is in place, VDUSE devices can opt in to the new
features.
Signed-off-by: Eugenio Pérez <eperezma@redhat.com>
---
drivers/vdpa/vdpa_user/vduse_dev.c | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/drivers/vdpa/vdpa_user/vduse_dev.c b/drivers/vdpa/vdpa_user/vduse_dev.c
index 06b7790380b7..07ef309ed7f7 100644
--- a/drivers/vdpa/vdpa_user/vduse_dev.c
+++ b/drivers/vdpa/vdpa_user/vduse_dev.c
@@ -2121,7 +2121,7 @@ static long vduse_ioctl(struct file *file, unsigned int cmd,
break;
ret = -EINVAL;
- if (api_version > VDUSE_API_VERSION)
+ if (api_version > VDUSE_API_VERSION_1)
break;
ret = 0;
@@ -2188,7 +2188,7 @@ static int vduse_open(struct inode *inode, struct file *file)
if (!control)
return -ENOMEM;
- control->api_version = VDUSE_API_VERSION;
+ control->api_version = VDUSE_API_VERSION_1;
file->private_data = control;
return 0;
--
2.51.0
^ permalink raw reply related [flat|nested] 27+ messages in thread