All of lore.kernel.org
 help / color / mirror / Atom feed
From: Christoph Hellwig <hch@lst.de>
To: Jens Axboe <axboe@kernel.dk>,
	Christian Brauner <brauner@kernel.org>,
	"Darrick J. Wong" <djwong@kernel.org>,
	Carlos Maiolino <cem@kernel.org>
Cc: Tal Zussman <tz2294@columbia.edu>,
	Anuj Gupta <anuj20.g@samsung.com>,
	linux-block@vger.kernel.org, linux-xfs@vger.kernel.org,
	linux-fsdevel@vger.kernel.org
Subject: [PATCH 14/17] xfs: add support for lazy direct read bounce buffering
Date: Mon, 31 Aug 2026 09:40:02 +0300	[thread overview]
Message-ID: <20260831064010.2574896-15-hch@lst.de> (raw)
In-Reply-To: <20260831064010.2574896-1-hch@lst.de>

Currently direct I/O reads always bounce buffer the I/O to deal with
the case where userspace is modifying the buffer in-flight while
reading data into it.

This is a very expensive countermeasure for something no sane application
should do, so try to avoid it by reading without a bounce buffer first,
and retrying the read on a checksum failure.  This avoids the cost of
bounce buffering for sanely behave applications.  For the rare case of
an application regularly modifying in-flight buffers, allow forcing the
always bounce buffer behavior through sysfs.  And now that we have that
knob, allow disabling read-side bounce buffering entirely for those who
live fast and dangerous.

Signed-off-by: Christoph Hellwig <hch@lst.de>
---
 fs/xfs/xfs_file.c  |   3 +-
 fs/xfs/xfs_ioend.c | 102 +++++++++++++++++++++++++++++++++++++++++++--
 fs/xfs/xfs_mount.h |   8 ++++
 fs/xfs/xfs_super.c |   1 +
 fs/xfs/xfs_sysfs.c |  76 +++++++++++++++++++++++++++++++++
 fs/xfs/xfs_trace.h |   1 +
 6 files changed, 186 insertions(+), 5 deletions(-)

diff --git a/fs/xfs/xfs_file.c b/fs/xfs/xfs_file.c
index 766c4d2055c1..daa6a854dd5f 100644
--- a/fs/xfs/xfs_file.c
+++ b/fs/xfs/xfs_file.c
@@ -270,8 +270,7 @@ xfs_file_dio_read(
 		return ret;
 	if (mapping_stable_writes(iocb->ki_filp->f_mapping)) {
 		ret = iomap_dio_rw(iocb, to, &xfs_read_iomap_ops,
-				&xfs_dio_read_bounce_ops, IOMAP_DIO_BOUNCE,
-				NULL, 0);
+				&xfs_dio_read_bounce_ops, 0, NULL, 0);
 	} else {
 		ret = iomap_dio_read_simple(iocb, to, xfs_read_iomap_begin);
 		if (ret == -ENOTBLK)
diff --git a/fs/xfs/xfs_ioend.c b/fs/xfs/xfs_ioend.c
index a095cf217863..c21cbd7b0a6d 100644
--- a/fs/xfs/xfs_ioend.c
+++ b/fs/xfs/xfs_ioend.c
@@ -1,6 +1,6 @@
 // SPDX-License-Identifier: GPL-2.0
 /*
- * Copyright (c) 2016-2025 Christoph Hellwig.
+ * Copyright (c) 2016-2026 Christoph Hellwig.
  * All Rights Reserved.
  */
 #include "xfs_platform.h"
@@ -18,15 +18,100 @@
 #include "xfs_ioend.h"
 #include <linux/bio-integrity.h>
 
+static void
+xfs_end_bio_bounced(
+	struct bio		*bio)
+{
+	iomap_finish_ioends(iomap_ioend_from_bio(bio),
+			blk_status_to_errno(bio->bi_status));
+}
+
+static void
+xfs_dio_bounce_end_io(
+	struct bio		*bio)
+{
+	struct iomap_ioend	*ioend = iomap_ioend_from_bio(bio);
+	int			error = blk_status_to_errno(bio->bi_status);
+	struct bio		*orig_bio = bio->bi_private;
+
+	if ((ioend->io_flags & IOMAP_IOEND_INTEGRITY) && !bio->bi_status)
+		error = iomap_ioend_integrity_verify(ioend);
+	iomap_bounce_read_end_io(ioend, orig_bio, error);
+}
+
+static void
+xfs_bounce_submit_ioend(
+	struct iomap_ioend	*ioend)
+{
+	if (ioend->io_flags & IOMAP_IOEND_INTEGRITY)
+		fs_bio_integrity_alloc(&ioend->io_bio);
+	ioend->io_bio.bi_end_io = xfs_dio_bounce_end_io;
+	bio_set_flag(&ioend->io_bio, BIO_COMPLETE_IN_TASK);
+	submit_bio(&ioend->io_bio);
+}
+
+static void
+xfs_read_bounce_and_resubmit(
+	struct iomap_ioend	*ioend)
+{
+	struct bio		*bio = &ioend->io_bio;
+	unsigned short		vcnt = bio->bi_vcnt;
+	void			*private = bio->bi_private;
+
+	trace_xfs_bounce_reread(XFS_I(ioend->io_inode), ioend->io_offset,
+			ioend->io_size);
+
+	/*
+	 * Free the bio integrity data for the original bio, as we'll allocate
+	 * ons for each sub-I/O, which could deadlock if we keep the original
+	 * one around.
+	 */
+	if (bio_integrity(bio))
+		fs_bio_integrity_free(bio);
+
+	/*
+	 * Reset the bio to submit the bio to the block layer again.  Switch to
+	 * an end_io handler that simply complets the ioend, as all verification
+	 * is done by the end_I/O handlers for the clone bio(s).
+	 */
+	bio_reset(bio, xfs_inode_buftarg(XFS_I(ioend->io_inode))->bt_bdev,
+			bio->bi_opf);
+	bio->bi_vcnt = vcnt;
+	bio->bi_private = private;
+	bio->bi_end_io = xfs_end_bio_bounced;
+	bio->bi_iter = (struct bvec_iter) {
+		.bi_sector	= ioend->io_sector,
+		.bi_size	= ioend->io_size,
+		.bi_offset	= ioend->io_bvec_offset,
+	};
+	iomap_bounce_read(ioend, bdev_logical_block_size(bio->bi_bdev),
+			xfs_bounce_submit_ioend);
+}
+
 static void
 xfs_end_io_read(
 	struct bio		*bio)
 {
 	struct iomap_ioend	*ioend = iomap_ioend_from_bio(bio);
+	struct xfs_inode	*ip = XFS_I(ioend->io_inode);
+	struct xfs_mount	*mp = ip->i_mount;
 	int			error = blk_status_to_errno(bio->bi_status);
 
-	if (!error && (ioend->io_flags & IOMAP_IOEND_INTEGRITY))
+	if (!error && (ioend->io_flags & IOMAP_IOEND_INTEGRITY)) {
 		error = iomap_ioend_integrity_verify(ioend);
+		if ((ioend->io_flags & IOMAP_IOEND_DIRECT) &&
+		    READ_ONCE(mp->m_read_bounce) == XFS_READ_BOUNCE_LAZY) {
+			/*
+			 * We only really need to retry for guard tag errors,
+			 * but right now we can't distinguish them from other
+			 * (i.e, reftag) errors.
+			 */
+			if (error) {
+				xfs_read_bounce_and_resubmit(ioend);
+				return;
+			}
+		}
+	}
 
 	iomap_finish_ioends(ioend, error);
 }
@@ -38,7 +123,18 @@ xfs_ioend_submit_read(
 	loff_t			file_offset,
 	u16			ioend_flags)
 {
-	iomap_init_ioend(inode, bio, file_offset, ioend_flags);
+	struct xfs_inode	*ip = XFS_I(inode);
+	struct xfs_mount	*mp = ip->i_mount;
+	struct iomap_ioend	*ioend;
+
+	ioend = iomap_init_ioend(inode, bio, file_offset, ioend_flags);
+	if ((ioend_flags & IOMAP_IOEND_DIRECT) &&
+	    READ_ONCE(mp->m_read_bounce) == XFS_READ_BOUNCE_ALWAYS) {
+		iomap_bounce_read(ioend, bdev_logical_block_size(bio->bi_bdev),
+				xfs_bounce_submit_ioend);
+		return;
+	}
+
 	if (ioend_flags & IOMAP_IOEND_INTEGRITY)
 		fs_bio_integrity_alloc(bio);
 	bio->bi_end_io = xfs_end_io_read;
diff --git a/fs/xfs/xfs_mount.h b/fs/xfs/xfs_mount.h
index 216a38a354e7..894ff2f4ecbd 100644
--- a/fs/xfs/xfs_mount.h
+++ b/fs/xfs/xfs_mount.h
@@ -142,6 +142,12 @@ struct xfs_freecounter {
 	uint64_t		res_saved;
 };
 
+enum xfs_read_bounce {
+	XFS_READ_BOUNCE_NEVER,
+	XFS_READ_BOUNCE_ALWAYS,
+	XFS_READ_BOUNCE_LAZY,
+};
+
 /*
  * The struct xfsmount layout is optimised to separate read-mostly variables
  * from variables that are frequently modified. We put the read-mostly variables
@@ -177,6 +183,7 @@ typedef struct xfs_mount {
 	struct workqueue_struct	*m_sync_workqueue;
 	struct workqueue_struct *m_blockgc_wq;
 	struct workqueue_struct *m_inodegc_wq;
+	enum xfs_read_bounce	m_read_bounce;
 
 	int			m_bsize;	/* fs logical block size */
 	uint8_t			m_blkbit_log;	/* blocklog + NBBY */
@@ -291,6 +298,7 @@ typedef struct xfs_mount {
 	struct xfs_zone_info	*m_zone_info;	/* zone allocator information */
 	struct dentry		*m_debugfs;	/* debugfs parent */
 	struct xfs_kobj		m_kobj;
+	struct xfs_kobj		m_csum_kobj;
 	struct xfs_kobj		m_error_kobj;
 	struct xfs_kobj		m_error_meta_kobj;
 	struct xfs_error_cfg	m_error_cfg[XFS_ERR_CLASS_MAX][XFS_ERR_ERRNO_MAX];
diff --git a/fs/xfs/xfs_super.c b/fs/xfs/xfs_super.c
index b24db75eaedc..fce1d2905c94 100644
--- a/fs/xfs/xfs_super.c
+++ b/fs/xfs/xfs_super.c
@@ -2317,6 +2317,7 @@ xfs_init_fs_context(
 	mp->m_logbufs = -1;
 	mp->m_logbsize = -1;
 	mp->m_allocsize_log = 16; /* 64k */
+	mp->m_read_bounce = XFS_READ_BOUNCE_LAZY;
 
 	xfs_hooks_init(&mp->m_dir_update_hooks);
 
diff --git a/fs/xfs/xfs_sysfs.c b/fs/xfs/xfs_sysfs.c
index b62712187324..2e969e8f279f 100644
--- a/fs/xfs/xfs_sysfs.c
+++ b/fs/xfs/xfs_sysfs.c
@@ -392,6 +392,57 @@ const struct kobj_type xfs_stats_ktype = {
 	.default_groups = xfs_stats_groups,
 };
 
+static inline struct xfs_mount *csum_to_mp(struct kobject *kobj)
+{
+	return container_of(to_kobj(kobj), struct xfs_mount, m_csum_kobj);
+}
+
+static const char * const bounce_modes[] = {
+	[XFS_READ_BOUNCE_NEVER]		= "never",
+	[XFS_READ_BOUNCE_ALWAYS]	= "always",
+	[XFS_READ_BOUNCE_LAZY]		= "lazy",
+};
+
+static ssize_t
+read_bounce_show(
+	struct kobject		*kobj,
+	char			*buf)
+{
+	struct xfs_mount	*mp = csum_to_mp(kobj);
+
+	return sysfs_emit(buf, "%s\n",
+			bounce_modes[READ_ONCE(mp->m_read_bounce)]);
+}
+
+static ssize_t
+read_bounce_store(
+	struct kobject		*kobj,
+	const char		*buf,
+	size_t			count)
+{
+	struct xfs_mount	*mp = csum_to_mp(kobj);
+	int			ret;
+
+	ret = sysfs_match_string(bounce_modes, buf);
+	if (ret < 0)
+		return ret;
+	WRITE_ONCE(mp->m_read_bounce, ret);
+	return count;
+}
+XFS_SYSFS_ATTR_RW(read_bounce);
+
+static struct attribute *xfs_csum_attrs[] = {
+	ATTR_LIST(read_bounce),
+	NULL,
+};
+ATTRIBUTE_GROUPS(xfs_csum);
+
+static const struct kobj_type xfs_csum_ktype = {
+	.release = xfs_sysfs_release,
+	.sysfs_ops = &xfs_sysfs_ops,
+	.default_groups = xfs_csum_groups,
+};
+
 /* xlog */
 
 static inline struct xlog *
@@ -797,6 +848,18 @@ xfs_zoned_sysfs_del(struct xfs_mount *mp)
 		xfs_sysfs_del(&mp->m_zoned_kobj);
 }
 
+static bool
+xfs_has_read_bounce(
+	struct xfs_mount	*mp)
+{
+	if (bdev_has_integrity_csum(mp->m_ddev_targp->bt_bdev))
+		return true;
+	if (mp->m_rtdev_targp &&
+	    bdev_has_integrity_csum(mp->m_rtdev_targp->bt_bdev))
+		return true;
+	return false;
+}
+
 int
 xfs_mount_sysfs_init(
 	struct xfs_mount	*mp)
@@ -837,8 +900,18 @@ xfs_mount_sysfs_init(
 	if (error)
 		goto out_remove_error_dir;
 
+	if (xfs_has_read_bounce(mp)) {
+		/* .../xfs/<dev>/csum/ */
+		error = xfs_sysfs_init(&mp->m_csum_kobj, &xfs_csum_ktype,
+				       &mp->m_kobj, "csum");
+		if (error)
+			goto out_remove_error_metadata_dir;
+	}
+
 	return 0;
 
+out_remove_error_metadata_dir:
+	xfs_sysfs_del(&mp->m_error_meta_kobj);
 out_remove_error_dir:
 	xfs_sysfs_del(&mp->m_error_kobj);
 out_remove_stats_dir:
@@ -855,6 +928,9 @@ xfs_mount_sysfs_del(
 	struct xfs_error_cfg	*cfg;
 	int			i, j;
 
+	if (xfs_has_read_bounce(mp))
+		xfs_sysfs_del(&mp->m_csum_kobj);
+
 	for (i = 0; i < XFS_ERR_CLASS_MAX; i++) {
 		for (j = 0; j < XFS_ERR_ERRNO_MAX; j++) {
 			cfg = &mp->m_error_cfg[i][j];
diff --git a/fs/xfs/xfs_trace.h b/fs/xfs/xfs_trace.h
index f333c938fbd9..2af9a1429ae9 100644
--- a/fs/xfs/xfs_trace.h
+++ b/fs/xfs/xfs_trace.h
@@ -1896,6 +1896,7 @@ DEFINE_SIMPLE_IO_EVENT(xfs_zero_eof);
 DEFINE_SIMPLE_IO_EVENT(xfs_end_io_direct_write);
 DEFINE_SIMPLE_IO_EVENT(xfs_file_splice_read);
 DEFINE_SIMPLE_IO_EVENT(xfs_zoned_map_blocks);
+DEFINE_SIMPLE_IO_EVENT(xfs_bounce_reread);
 
 DECLARE_EVENT_CLASS(xfs_itrunc_class,
 	TP_PROTO(struct xfs_inode *ip, xfs_fsize_t new_size),
-- 
2.53.0


  parent reply	other threads:[~2026-08-31  6:41 UTC|newest]

Thread overview: 33+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-31  6:39 lazy bounce buffering for checksummed reads V2 Christoph Hellwig
2026-08-31  6:39 ` [PATCH 01/17] block: pass a maxlen argument to bio_iov_iter_get_pages Christoph Hellwig
2026-08-31 18:04   ` Darrick J. Wong
2026-08-31  6:39 ` [PATCH 02/17] block: warn on too larger integrity allocations Christoph Hellwig
2026-08-31 18:01   ` Darrick J. Wong
2026-09-01  8:12     ` Christoph Hellwig
2026-08-31  6:39 ` [PATCH 03/17] block: split bio_iov_iter_bounce_write Christoph Hellwig
2026-08-31 17:58   ` Darrick J. Wong
2026-08-31  6:39 ` [PATCH 04/17] block: export fs_bio_integrity_{alloc,free} Christoph Hellwig
2026-08-31 17:57   ` Darrick J. Wong
2026-08-31  6:39 ` [PATCH 05/17] iomap: respect maximum I/O size in iomap_dio_bio_iter_one Christoph Hellwig
2026-08-31 17:55   ` Darrick J. Wong
2026-08-31  6:39 ` [PATCH 06/17] iomap: add a iomap_ioend_flags helper Christoph Hellwig
2026-08-31  6:39 ` [PATCH 07/17] iomap: add a IOMAP_IOEND_INTEGRITY flag Christoph Hellwig
2026-08-31  6:39 ` [PATCH 08/17] iomap,xfs: move T10 PI handling for direct I/O into ->submit_io Christoph Hellwig
2026-08-31  6:39 ` [PATCH 09/17] xfs: move PI generation into xfs_submit_zoned_bio Christoph Hellwig
2026-08-31 17:53   ` Darrick J. Wong
2026-09-02 18:22   ` Anuj Gupta
2026-09-07  5:51     ` Christoph Hellwig
2026-08-31  6:39 ` [PATCH 10/17] block,iomap: fix protection information verification with initial bvec offset Christoph Hellwig
2026-08-31 17:52   ` Darrick J. Wong
2026-09-01  8:12     ` Christoph Hellwig
2026-09-01 14:06       ` Darrick J. Wong
2026-08-31  6:39 ` [PATCH 11/17] iomap: better read bounce buffering support Christoph Hellwig
2026-08-31  6:40 ` [PATCH 12/17] xfs: use BIO_COMPLETE_IN_TASK for bounce buffered read I/Os Christoph Hellwig
2026-08-31  6:40 ` [PATCH 13/17] iomap,xfs: move integrity verification to the file system Christoph Hellwig
2026-08-31  6:40 ` Christoph Hellwig [this message]
2026-08-31 17:47   ` [PATCH 14/17] xfs: add support for lazy direct read bounce buffering Darrick J. Wong
2026-09-02 18:24   ` Anuj Gupta
2026-08-31  6:40 ` [PATCH 15/17] xfs: add error injection for lazy " Christoph Hellwig
2026-09-02 18:26   ` Anuj Gupta
2026-08-31  6:40 ` [PATCH 16/17] xfs: log a message at mount time when using integrity protection Christoph Hellwig
2026-08-31  6:40 ` [PATCH 17/17] block,iomap: remove the old read side bounce buffering support Christoph Hellwig

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260831064010.2574896-15-hch@lst.de \
    --to=hch@lst.de \
    --cc=anuj20.g@samsung.com \
    --cc=axboe@kernel.dk \
    --cc=brauner@kernel.org \
    --cc=cem@kernel.org \
    --cc=djwong@kernel.org \
    --cc=linux-block@vger.kernel.org \
    --cc=linux-fsdevel@vger.kernel.org \
    --cc=linux-xfs@vger.kernel.org \
    --cc=tz2294@columbia.edu \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.