From: Christian Brauner <brauner@kernel.org>
To: Jens Axboe <axboe@kernel.dk>
Cc: linux-fsdevel@vger.kernel.org, linux-kernel@vger.kernel.org
Subject: Re: [PATCH 2/3] userfaultfd: convert to ->read_iter()
Date: Wed, 3 Apr 2024 12:09:12 +0200 [thread overview]
Message-ID: <20240403-plant-narren-2bbfb61f19f0@brauner> (raw)
In-Reply-To: <20240402202524.1514963-3-axboe@kernel.dk>
On Tue, Apr 02, 2024 at 02:18:22PM -0600, Jens Axboe wrote:
> Rather than use the older style ->read() hook, use ->read_iter() so that
> userfaultfd can support both O_NONBLOCK and IOCB_NOWAIT for non-blocking
> read attempts.
>
> Split the fd setup into two parts, so that userfaultfd can mark the file
> mode with FMODE_NOWAIT before installing it into the process table. With
> that, we can also defer grabbing the mm until we know the rest will
> succeed, as the fd isn't visible before then.
>
> Signed-off-by: Jens Axboe <axboe@kernel.dk>
> ---
> fs/userfaultfd.c | 42 ++++++++++++++++++++++++++----------------
> 1 file changed, 26 insertions(+), 16 deletions(-)
>
> diff --git a/fs/userfaultfd.c b/fs/userfaultfd.c
> index 60dcfafdc11a..7864c2dba858 100644
> --- a/fs/userfaultfd.c
> +++ b/fs/userfaultfd.c
> @@ -282,7 +282,7 @@ static inline bool userfaultfd_huge_must_wait(struct userfaultfd_ctx *ctx,
> /*
> * Verify the pagetables are still not ok after having reigstered into
> * the fault_pending_wqh to avoid userland having to UFFDIO_WAKE any
> - * userfault that has already been resolved, if userfaultfd_read and
> + * userfault that has already been resolved, if userfaultfd_read_iter and
> * UFFDIO_COPY|ZEROPAGE are being run simultaneously on two different
> * threads.
> */
> @@ -1177,34 +1177,34 @@ static ssize_t userfaultfd_ctx_read(struct userfaultfd_ctx *ctx, int no_wait,
> return ret;
> }
>
> -static ssize_t userfaultfd_read(struct file *file, char __user *buf,
> - size_t count, loff_t *ppos)
> +static ssize_t userfaultfd_read_iter(struct kiocb *iocb, struct iov_iter *to)
> {
> + struct file *file = iocb->ki_filp;
> struct userfaultfd_ctx *ctx = file->private_data;
> ssize_t _ret, ret = 0;
> struct uffd_msg msg;
> - int no_wait = file->f_flags & O_NONBLOCK;
> struct inode *inode = file_inode(file);
> + bool no_wait;
>
> if (!userfaultfd_is_initialized(ctx))
> return -EINVAL;
>
> + no_wait = file->f_flags & O_NONBLOCK || iocb->ki_flags & IOCB_NOWAIT;
> for (;;) {
> - if (count < sizeof(msg))
> + if (iov_iter_count(to) < sizeof(msg))
> return ret ? ret : -EINVAL;
> _ret = userfaultfd_ctx_read(ctx, no_wait, &msg, inode);
> if (_ret < 0)
> return ret ? ret : _ret;
> - if (copy_to_user((__u64 __user *) buf, &msg, sizeof(msg)))
> + _ret = copy_to_iter(&msg, sizeof(msg), to);
> + if (_ret < 0)
> return ret ? ret : -EFAULT;
> ret += sizeof(msg);
> - buf += sizeof(msg);
> - count -= sizeof(msg);
> /*
> * Allow to read more than one fault at time but only
> * block if waiting for the very first one.
> */
> - no_wait = O_NONBLOCK;
> + no_wait = true;
> }
> }
>
> @@ -2172,7 +2172,7 @@ static const struct file_operations userfaultfd_fops = {
> #endif
> .release = userfaultfd_release,
> .poll = userfaultfd_poll,
> - .read = userfaultfd_read,
> + .read_iter = userfaultfd_read_iter,
> .unlocked_ioctl = userfaultfd_ioctl,
> .compat_ioctl = compat_ptr_ioctl,
> .llseek = noop_llseek,
> @@ -2192,6 +2192,7 @@ static void init_once_userfaultfd_ctx(void *mem)
> static int new_userfaultfd(int flags)
> {
> struct userfaultfd_ctx *ctx;
> + struct file *file;
> int fd;
>
> BUG_ON(!current->mm);
> @@ -2215,16 +2216,25 @@ static int new_userfaultfd(int flags)
> init_rwsem(&ctx->map_changing_lock);
> atomic_set(&ctx->mmap_changing, 0);
> ctx->mm = current->mm;
> - /* prevent the mm struct to be freed */
> - mmgrab(ctx->mm);
> +
> + fd = get_unused_fd_flags(O_RDONLY | (flags & UFFD_SHARED_FCNTL_FLAGS));
> + if (fd < 0)
> + goto err_out;
>
> /* Create a new inode so that the LSM can block the creation. */
> - fd = anon_inode_create_getfd("[userfaultfd]", &userfaultfd_fops, ctx,
> + file = anon_inode_create_getfile("[userfaultfd]", &userfaultfd_fops, ctx,
> O_RDONLY | (flags & UFFD_SHARED_FCNTL_FLAGS), NULL);
> - if (fd < 0) {
> - mmdrop(ctx->mm);
> - kmem_cache_free(userfaultfd_ctx_cachep, ctx);
> + if (IS_ERR(file)) {
> + fd = PTR_ERR(file);
> + goto err_out;
You're leaking the fd you allocated above.
> }
> + /* prevent the mm struct to be freed */
> + mmgrab(ctx->mm);
> + file->f_mode |= FMODE_NOWAIT;
> + fd_install(fd, file);
> + return fd;
> +err_out:
> + kmem_cache_free(userfaultfd_ctx_cachep, ctx);
> return fd;
> }
>
> --
> 2.43.0
>
next prev parent reply other threads:[~2024-04-03 10:09 UTC|newest]
Thread overview: 8+ messages / expand[flat|nested] mbox.gz Atom feed top
2024-04-02 20:18 [PATCHSET 0/3] Convert fs drivers to ->read_iter() Jens Axboe
2024-04-02 20:18 ` [PATCH 1/3] timerfd: convert " Jens Axboe
2024-04-02 20:18 ` [PATCH 2/3] userfaultfd: " Jens Axboe
2024-04-03 10:09 ` Christian Brauner [this message]
2024-04-03 13:44 ` Jens Axboe
2024-04-02 20:18 ` [PATCH 3/3] signalfd: " Jens Axboe
-- strict thread matches above, loose matches on Subject: below --
2024-04-03 14:02 [PATCHSET v2 0/3] Convert fs drivers " Jens Axboe
2024-04-03 14:02 ` [PATCH 2/3] userfaultfd: convert " Jens Axboe
2024-04-03 22:45 ` Al Viro
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20240403-plant-narren-2bbfb61f19f0@brauner \
--to=brauner@kernel.org \
--cc=axboe@kernel.dk \
--cc=linux-fsdevel@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.