Linux block layer
 help / color / mirror / Atom feed
From: "Darrick J. Wong" <djwong@kernel.org>
To: Christoph Hellwig <hch@lst.de>
Cc: Jens Axboe <axboe@kernel.dk>,
	Christian Brauner <brauner@kernel.org>,
	Carlos Maiolino <cem@kernel.org>,
	Tal Zussman <tz2294@columbia.edu>,
	Anuj Gupta <anuj20.g@samsung.com>,
	linux-block@vger.kernel.org, linux-xfs@vger.kernel.org,
	linux-fsdevel@vger.kernel.org
Subject: Re: [PATCH 14/17] xfs: add support for lazy direct read bounce buffering
Date: Mon, 31 Aug 2026 10:47:01 -0700	[thread overview]
Message-ID: <20260831174701.GC1933798@frogsfrogsfrogs> (raw)
In-Reply-To: <20260831064010.2574896-15-hch@lst.de>

On Mon, Aug 31, 2026 at 09:40:02AM +0300, Christoph Hellwig wrote:
> Currently direct I/O reads always bounce buffer the I/O to deal with
> the case where userspace is modifying the buffer in-flight while
> reading data into it.
> 
> This is a very expensive countermeasure for something no sane application
> should do, so try to avoid it by reading without a bounce buffer first,
> and retrying the read on a checksum failure.  This avoids the cost of
> bounce buffering for sanely behave applications.  For the rare case of
> an application regularly modifying in-flight buffers, allow forcing the
> always bounce buffer behavior through sysfs.  And now that we have that
> knob, allow disabling read-side bounce buffering entirely for those who
> live fast and dangerous.
> 
> Signed-off-by: Christoph Hellwig <hch@lst.de>

Looks good to me now.  That sysfs_match_string macro is pretty neat.
Reviewed-by: "Darrick J. Wong" <djwong@kernel.org>

--D

> ---
>  fs/xfs/xfs_file.c  |   3 +-
>  fs/xfs/xfs_ioend.c | 102 +++++++++++++++++++++++++++++++++++++++++++--
>  fs/xfs/xfs_mount.h |   8 ++++
>  fs/xfs/xfs_super.c |   1 +
>  fs/xfs/xfs_sysfs.c |  76 +++++++++++++++++++++++++++++++++
>  fs/xfs/xfs_trace.h |   1 +
>  6 files changed, 186 insertions(+), 5 deletions(-)
> 
> diff --git a/fs/xfs/xfs_file.c b/fs/xfs/xfs_file.c
> index 766c4d2055c1..daa6a854dd5f 100644
> --- a/fs/xfs/xfs_file.c
> +++ b/fs/xfs/xfs_file.c
> @@ -270,8 +270,7 @@ xfs_file_dio_read(
>  		return ret;
>  	if (mapping_stable_writes(iocb->ki_filp->f_mapping)) {
>  		ret = iomap_dio_rw(iocb, to, &xfs_read_iomap_ops,
> -				&xfs_dio_read_bounce_ops, IOMAP_DIO_BOUNCE,
> -				NULL, 0);
> +				&xfs_dio_read_bounce_ops, 0, NULL, 0);
>  	} else {
>  		ret = iomap_dio_read_simple(iocb, to, xfs_read_iomap_begin);
>  		if (ret == -ENOTBLK)
> diff --git a/fs/xfs/xfs_ioend.c b/fs/xfs/xfs_ioend.c
> index a095cf217863..c21cbd7b0a6d 100644
> --- a/fs/xfs/xfs_ioend.c
> +++ b/fs/xfs/xfs_ioend.c
> @@ -1,6 +1,6 @@
>  // SPDX-License-Identifier: GPL-2.0
>  /*
> - * Copyright (c) 2016-2025 Christoph Hellwig.
> + * Copyright (c) 2016-2026 Christoph Hellwig.
>   * All Rights Reserved.
>   */
>  #include "xfs_platform.h"
> @@ -18,15 +18,100 @@
>  #include "xfs_ioend.h"
>  #include <linux/bio-integrity.h>
>  
> +static void
> +xfs_end_bio_bounced(
> +	struct bio		*bio)
> +{
> +	iomap_finish_ioends(iomap_ioend_from_bio(bio),
> +			blk_status_to_errno(bio->bi_status));
> +}
> +
> +static void
> +xfs_dio_bounce_end_io(
> +	struct bio		*bio)
> +{
> +	struct iomap_ioend	*ioend = iomap_ioend_from_bio(bio);
> +	int			error = blk_status_to_errno(bio->bi_status);
> +	struct bio		*orig_bio = bio->bi_private;
> +
> +	if ((ioend->io_flags & IOMAP_IOEND_INTEGRITY) && !bio->bi_status)
> +		error = iomap_ioend_integrity_verify(ioend);
> +	iomap_bounce_read_end_io(ioend, orig_bio, error);
> +}
> +
> +static void
> +xfs_bounce_submit_ioend(
> +	struct iomap_ioend	*ioend)
> +{
> +	if (ioend->io_flags & IOMAP_IOEND_INTEGRITY)
> +		fs_bio_integrity_alloc(&ioend->io_bio);
> +	ioend->io_bio.bi_end_io = xfs_dio_bounce_end_io;
> +	bio_set_flag(&ioend->io_bio, BIO_COMPLETE_IN_TASK);
> +	submit_bio(&ioend->io_bio);
> +}
> +
> +static void
> +xfs_read_bounce_and_resubmit(
> +	struct iomap_ioend	*ioend)
> +{
> +	struct bio		*bio = &ioend->io_bio;
> +	unsigned short		vcnt = bio->bi_vcnt;
> +	void			*private = bio->bi_private;
> +
> +	trace_xfs_bounce_reread(XFS_I(ioend->io_inode), ioend->io_offset,
> +			ioend->io_size);
> +
> +	/*
> +	 * Free the bio integrity data for the original bio, as we'll allocate
> +	 * ons for each sub-I/O, which could deadlock if we keep the original
> +	 * one around.
> +	 */
> +	if (bio_integrity(bio))
> +		fs_bio_integrity_free(bio);
> +
> +	/*
> +	 * Reset the bio to submit the bio to the block layer again.  Switch to
> +	 * an end_io handler that simply complets the ioend, as all verification
> +	 * is done by the end_I/O handlers for the clone bio(s).
> +	 */
> +	bio_reset(bio, xfs_inode_buftarg(XFS_I(ioend->io_inode))->bt_bdev,
> +			bio->bi_opf);
> +	bio->bi_vcnt = vcnt;
> +	bio->bi_private = private;
> +	bio->bi_end_io = xfs_end_bio_bounced;
> +	bio->bi_iter = (struct bvec_iter) {
> +		.bi_sector	= ioend->io_sector,
> +		.bi_size	= ioend->io_size,
> +		.bi_offset	= ioend->io_bvec_offset,
> +	};
> +	iomap_bounce_read(ioend, bdev_logical_block_size(bio->bi_bdev),
> +			xfs_bounce_submit_ioend);
> +}
> +
>  static void
>  xfs_end_io_read(
>  	struct bio		*bio)
>  {
>  	struct iomap_ioend	*ioend = iomap_ioend_from_bio(bio);
> +	struct xfs_inode	*ip = XFS_I(ioend->io_inode);
> +	struct xfs_mount	*mp = ip->i_mount;
>  	int			error = blk_status_to_errno(bio->bi_status);
>  
> -	if (!error && (ioend->io_flags & IOMAP_IOEND_INTEGRITY))
> +	if (!error && (ioend->io_flags & IOMAP_IOEND_INTEGRITY)) {
>  		error = iomap_ioend_integrity_verify(ioend);
> +		if ((ioend->io_flags & IOMAP_IOEND_DIRECT) &&
> +		    READ_ONCE(mp->m_read_bounce) == XFS_READ_BOUNCE_LAZY) {
> +			/*
> +			 * We only really need to retry for guard tag errors,
> +			 * but right now we can't distinguish them from other
> +			 * (i.e, reftag) errors.
> +			 */
> +			if (error) {
> +				xfs_read_bounce_and_resubmit(ioend);
> +				return;
> +			}
> +		}
> +	}
>  
>  	iomap_finish_ioends(ioend, error);
>  }
> @@ -38,7 +123,18 @@ xfs_ioend_submit_read(
>  	loff_t			file_offset,
>  	u16			ioend_flags)
>  {
> -	iomap_init_ioend(inode, bio, file_offset, ioend_flags);
> +	struct xfs_inode	*ip = XFS_I(inode);
> +	struct xfs_mount	*mp = ip->i_mount;
> +	struct iomap_ioend	*ioend;
> +
> +	ioend = iomap_init_ioend(inode, bio, file_offset, ioend_flags);
> +	if ((ioend_flags & IOMAP_IOEND_DIRECT) &&
> +	    READ_ONCE(mp->m_read_bounce) == XFS_READ_BOUNCE_ALWAYS) {
> +		iomap_bounce_read(ioend, bdev_logical_block_size(bio->bi_bdev),
> +				xfs_bounce_submit_ioend);
> +		return;
> +	}
> +
>  	if (ioend_flags & IOMAP_IOEND_INTEGRITY)
>  		fs_bio_integrity_alloc(bio);
>  	bio->bi_end_io = xfs_end_io_read;
> diff --git a/fs/xfs/xfs_mount.h b/fs/xfs/xfs_mount.h
> index 216a38a354e7..894ff2f4ecbd 100644
> --- a/fs/xfs/xfs_mount.h
> +++ b/fs/xfs/xfs_mount.h
> @@ -142,6 +142,12 @@ struct xfs_freecounter {
>  	uint64_t		res_saved;
>  };
>  
> +enum xfs_read_bounce {
> +	XFS_READ_BOUNCE_NEVER,
> +	XFS_READ_BOUNCE_ALWAYS,
> +	XFS_READ_BOUNCE_LAZY,
> +};
> +
>  /*
>   * The struct xfsmount layout is optimised to separate read-mostly variables
>   * from variables that are frequently modified. We put the read-mostly variables
> @@ -177,6 +183,7 @@ typedef struct xfs_mount {
>  	struct workqueue_struct	*m_sync_workqueue;
>  	struct workqueue_struct *m_blockgc_wq;
>  	struct workqueue_struct *m_inodegc_wq;
> +	enum xfs_read_bounce	m_read_bounce;
>  
>  	int			m_bsize;	/* fs logical block size */
>  	uint8_t			m_blkbit_log;	/* blocklog + NBBY */
> @@ -291,6 +298,7 @@ typedef struct xfs_mount {
>  	struct xfs_zone_info	*m_zone_info;	/* zone allocator information */
>  	struct dentry		*m_debugfs;	/* debugfs parent */
>  	struct xfs_kobj		m_kobj;
> +	struct xfs_kobj		m_csum_kobj;
>  	struct xfs_kobj		m_error_kobj;
>  	struct xfs_kobj		m_error_meta_kobj;
>  	struct xfs_error_cfg	m_error_cfg[XFS_ERR_CLASS_MAX][XFS_ERR_ERRNO_MAX];
> diff --git a/fs/xfs/xfs_super.c b/fs/xfs/xfs_super.c
> index b24db75eaedc..fce1d2905c94 100644
> --- a/fs/xfs/xfs_super.c
> +++ b/fs/xfs/xfs_super.c
> @@ -2317,6 +2317,7 @@ xfs_init_fs_context(
>  	mp->m_logbufs = -1;
>  	mp->m_logbsize = -1;
>  	mp->m_allocsize_log = 16; /* 64k */
> +	mp->m_read_bounce = XFS_READ_BOUNCE_LAZY;
>  
>  	xfs_hooks_init(&mp->m_dir_update_hooks);
>  
> diff --git a/fs/xfs/xfs_sysfs.c b/fs/xfs/xfs_sysfs.c
> index b62712187324..2e969e8f279f 100644
> --- a/fs/xfs/xfs_sysfs.c
> +++ b/fs/xfs/xfs_sysfs.c
> @@ -392,6 +392,57 @@ const struct kobj_type xfs_stats_ktype = {
>  	.default_groups = xfs_stats_groups,
>  };
>  
> +static inline struct xfs_mount *csum_to_mp(struct kobject *kobj)
> +{
> +	return container_of(to_kobj(kobj), struct xfs_mount, m_csum_kobj);
> +}
> +
> +static const char * const bounce_modes[] = {
> +	[XFS_READ_BOUNCE_NEVER]		= "never",
> +	[XFS_READ_BOUNCE_ALWAYS]	= "always",
> +	[XFS_READ_BOUNCE_LAZY]		= "lazy",
> +};
> +
> +static ssize_t
> +read_bounce_show(
> +	struct kobject		*kobj,
> +	char			*buf)
> +{
> +	struct xfs_mount	*mp = csum_to_mp(kobj);
> +
> +	return sysfs_emit(buf, "%s\n",
> +			bounce_modes[READ_ONCE(mp->m_read_bounce)]);
> +}
> +
> +static ssize_t
> +read_bounce_store(
> +	struct kobject		*kobj,
> +	const char		*buf,
> +	size_t			count)
> +{
> +	struct xfs_mount	*mp = csum_to_mp(kobj);
> +	int			ret;
> +
> +	ret = sysfs_match_string(bounce_modes, buf);
> +	if (ret < 0)
> +		return ret;
> +	WRITE_ONCE(mp->m_read_bounce, ret);
> +	return count;
> +}
> +XFS_SYSFS_ATTR_RW(read_bounce);
> +
> +static struct attribute *xfs_csum_attrs[] = {
> +	ATTR_LIST(read_bounce),
> +	NULL,
> +};
> +ATTRIBUTE_GROUPS(xfs_csum);
> +
> +static const struct kobj_type xfs_csum_ktype = {
> +	.release = xfs_sysfs_release,
> +	.sysfs_ops = &xfs_sysfs_ops,
> +	.default_groups = xfs_csum_groups,
> +};
> +
>  /* xlog */
>  
>  static inline struct xlog *
> @@ -797,6 +848,18 @@ xfs_zoned_sysfs_del(struct xfs_mount *mp)
>  		xfs_sysfs_del(&mp->m_zoned_kobj);
>  }
>  
> +static bool
> +xfs_has_read_bounce(
> +	struct xfs_mount	*mp)
> +{
> +	if (bdev_has_integrity_csum(mp->m_ddev_targp->bt_bdev))
> +		return true;
> +	if (mp->m_rtdev_targp &&
> +	    bdev_has_integrity_csum(mp->m_rtdev_targp->bt_bdev))
> +		return true;
> +	return false;
> +}
> +
>  int
>  xfs_mount_sysfs_init(
>  	struct xfs_mount	*mp)
> @@ -837,8 +900,18 @@ xfs_mount_sysfs_init(
>  	if (error)
>  		goto out_remove_error_dir;
>  
> +	if (xfs_has_read_bounce(mp)) {
> +		/* .../xfs/<dev>/csum/ */
> +		error = xfs_sysfs_init(&mp->m_csum_kobj, &xfs_csum_ktype,
> +				       &mp->m_kobj, "csum");
> +		if (error)
> +			goto out_remove_error_metadata_dir;
> +	}
> +
>  	return 0;
>  
> +out_remove_error_metadata_dir:
> +	xfs_sysfs_del(&mp->m_error_meta_kobj);
>  out_remove_error_dir:
>  	xfs_sysfs_del(&mp->m_error_kobj);
>  out_remove_stats_dir:
> @@ -855,6 +928,9 @@ xfs_mount_sysfs_del(
>  	struct xfs_error_cfg	*cfg;
>  	int			i, j;
>  
> +	if (xfs_has_read_bounce(mp))
> +		xfs_sysfs_del(&mp->m_csum_kobj);
> +
>  	for (i = 0; i < XFS_ERR_CLASS_MAX; i++) {
>  		for (j = 0; j < XFS_ERR_ERRNO_MAX; j++) {
>  			cfg = &mp->m_error_cfg[i][j];
> diff --git a/fs/xfs/xfs_trace.h b/fs/xfs/xfs_trace.h
> index f333c938fbd9..2af9a1429ae9 100644
> --- a/fs/xfs/xfs_trace.h
> +++ b/fs/xfs/xfs_trace.h
> @@ -1896,6 +1896,7 @@ DEFINE_SIMPLE_IO_EVENT(xfs_zero_eof);
>  DEFINE_SIMPLE_IO_EVENT(xfs_end_io_direct_write);
>  DEFINE_SIMPLE_IO_EVENT(xfs_file_splice_read);
>  DEFINE_SIMPLE_IO_EVENT(xfs_zoned_map_blocks);
> +DEFINE_SIMPLE_IO_EVENT(xfs_bounce_reread);
>  
>  DECLARE_EVENT_CLASS(xfs_itrunc_class,
>  	TP_PROTO(struct xfs_inode *ip, xfs_fsize_t new_size),
> -- 
> 2.53.0
> 
> 

  reply	other threads:[~2026-08-31 17:47 UTC|newest]

Thread overview: 32+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-31  6:39 lazy bounce buffering for checksummed reads V2 Christoph Hellwig
2026-08-31  6:39 ` [PATCH 01/17] block: pass a maxlen argument to bio_iov_iter_get_pages Christoph Hellwig
2026-08-31 18:04   ` Darrick J. Wong
2026-08-31  6:39 ` [PATCH 02/17] block: warn on too larger integrity allocations Christoph Hellwig
2026-08-31 18:01   ` Darrick J. Wong
2026-09-01  8:12     ` Christoph Hellwig
2026-08-31  6:39 ` [PATCH 03/17] block: split bio_iov_iter_bounce_write Christoph Hellwig
2026-08-31 17:58   ` Darrick J. Wong
2026-08-31  6:39 ` [PATCH 04/17] block: export fs_bio_integrity_{alloc,free} Christoph Hellwig
2026-08-31 17:57   ` Darrick J. Wong
2026-08-31  6:39 ` [PATCH 05/17] iomap: respect maximum I/O size in iomap_dio_bio_iter_one Christoph Hellwig
2026-08-31 17:55   ` Darrick J. Wong
2026-08-31  6:39 ` [PATCH 06/17] iomap: add a iomap_ioend_flags helper Christoph Hellwig
2026-08-31  6:39 ` [PATCH 07/17] iomap: add a IOMAP_IOEND_INTEGRITY flag Christoph Hellwig
2026-08-31  6:39 ` [PATCH 08/17] iomap,xfs: move T10 PI handling for direct I/O into ->submit_io Christoph Hellwig
2026-08-31  6:39 ` [PATCH 09/17] xfs: move PI generation into xfs_submit_zoned_bio Christoph Hellwig
2026-08-31 17:53   ` Darrick J. Wong
2026-09-02 18:22   ` Anuj Gupta
2026-08-31  6:39 ` [PATCH 10/17] block,iomap: fix protection information verification with initial bvec offset Christoph Hellwig
2026-08-31 17:52   ` Darrick J. Wong
2026-09-01  8:12     ` Christoph Hellwig
2026-09-01 14:06       ` Darrick J. Wong
2026-08-31  6:39 ` [PATCH 11/17] iomap: better read bounce buffering support Christoph Hellwig
2026-08-31  6:40 ` [PATCH 12/17] xfs: use BIO_COMPLETE_IN_TASK for bounce buffered read I/Os Christoph Hellwig
2026-08-31  6:40 ` [PATCH 13/17] iomap,xfs: move integrity verification to the file system Christoph Hellwig
2026-08-31  6:40 ` [PATCH 14/17] xfs: add support for lazy direct read bounce buffering Christoph Hellwig
2026-08-31 17:47   ` Darrick J. Wong [this message]
2026-09-02 18:24   ` Anuj Gupta
2026-08-31  6:40 ` [PATCH 15/17] xfs: add error injection for lazy " Christoph Hellwig
2026-09-02 18:26   ` Anuj Gupta
2026-08-31  6:40 ` [PATCH 16/17] xfs: log a message at mount time when using integrity protection Christoph Hellwig
2026-08-31  6:40 ` [PATCH 17/17] block,iomap: remove the old read side bounce buffering support Christoph Hellwig

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260831174701.GC1933798@frogsfrogsfrogs \
    --to=djwong@kernel.org \
    --cc=anuj20.g@samsung.com \
    --cc=axboe@kernel.dk \
    --cc=brauner@kernel.org \
    --cc=cem@kernel.org \
    --cc=hch@lst.de \
    --cc=linux-block@vger.kernel.org \
    --cc=linux-fsdevel@vger.kernel.org \
    --cc=linux-xfs@vger.kernel.org \
    --cc=tz2294@columbia.edu \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox