From: Christoph Hellwig <hch@lst.de>
To: Jens Axboe <axboe@kernel.dk>,
Christian Brauner <brauner@kernel.org>,
"Darrick J. Wong" <djwong@kernel.org>,
Carlos Maiolino <cem@kernel.org>
Cc: Tal Zussman <tz2294@columbia.edu>,
Anuj Gupta <anuj20.g@samsung.com>,
linux-block@vger.kernel.org, linux-xfs@vger.kernel.org,
linux-fsdevel@vger.kernel.org
Subject: [PATCH 14/17] xfs: add support for lazy direct read bounce buffering
Date: Mon, 31 Aug 2026 09:40:02 +0300 [thread overview]
Message-ID: <20260831064010.2574896-15-hch@lst.de> (raw)
In-Reply-To: <20260831064010.2574896-1-hch@lst.de>
Currently direct I/O reads always bounce buffer the I/O to deal with
the case where userspace is modifying the buffer in-flight while
reading data into it.
This is a very expensive countermeasure for something no sane application
should do, so try to avoid it by reading without a bounce buffer first,
and retrying the read on a checksum failure. This avoids the cost of
bounce buffering for sanely behave applications. For the rare case of
an application regularly modifying in-flight buffers, allow forcing the
always bounce buffer behavior through sysfs. And now that we have that
knob, allow disabling read-side bounce buffering entirely for those who
live fast and dangerous.
Signed-off-by: Christoph Hellwig <hch@lst.de>
---
fs/xfs/xfs_file.c | 3 +-
fs/xfs/xfs_ioend.c | 102 +++++++++++++++++++++++++++++++++++++++++++--
fs/xfs/xfs_mount.h | 8 ++++
fs/xfs/xfs_super.c | 1 +
fs/xfs/xfs_sysfs.c | 76 +++++++++++++++++++++++++++++++++
fs/xfs/xfs_trace.h | 1 +
6 files changed, 186 insertions(+), 5 deletions(-)
diff --git a/fs/xfs/xfs_file.c b/fs/xfs/xfs_file.c
index 766c4d2055c1..daa6a854dd5f 100644
--- a/fs/xfs/xfs_file.c
+++ b/fs/xfs/xfs_file.c
@@ -270,8 +270,7 @@ xfs_file_dio_read(
return ret;
if (mapping_stable_writes(iocb->ki_filp->f_mapping)) {
ret = iomap_dio_rw(iocb, to, &xfs_read_iomap_ops,
- &xfs_dio_read_bounce_ops, IOMAP_DIO_BOUNCE,
- NULL, 0);
+ &xfs_dio_read_bounce_ops, 0, NULL, 0);
} else {
ret = iomap_dio_read_simple(iocb, to, xfs_read_iomap_begin);
if (ret == -ENOTBLK)
diff --git a/fs/xfs/xfs_ioend.c b/fs/xfs/xfs_ioend.c
index a095cf217863..c21cbd7b0a6d 100644
--- a/fs/xfs/xfs_ioend.c
+++ b/fs/xfs/xfs_ioend.c
@@ -1,6 +1,6 @@
// SPDX-License-Identifier: GPL-2.0
/*
- * Copyright (c) 2016-2025 Christoph Hellwig.
+ * Copyright (c) 2016-2026 Christoph Hellwig.
* All Rights Reserved.
*/
#include "xfs_platform.h"
@@ -18,15 +18,100 @@
#include "xfs_ioend.h"
#include <linux/bio-integrity.h>
+static void
+xfs_end_bio_bounced(
+ struct bio *bio)
+{
+ iomap_finish_ioends(iomap_ioend_from_bio(bio),
+ blk_status_to_errno(bio->bi_status));
+}
+
+static void
+xfs_dio_bounce_end_io(
+ struct bio *bio)
+{
+ struct iomap_ioend *ioend = iomap_ioend_from_bio(bio);
+ int error = blk_status_to_errno(bio->bi_status);
+ struct bio *orig_bio = bio->bi_private;
+
+ if ((ioend->io_flags & IOMAP_IOEND_INTEGRITY) && !bio->bi_status)
+ error = iomap_ioend_integrity_verify(ioend);
+ iomap_bounce_read_end_io(ioend, orig_bio, error);
+}
+
+static void
+xfs_bounce_submit_ioend(
+ struct iomap_ioend *ioend)
+{
+ if (ioend->io_flags & IOMAP_IOEND_INTEGRITY)
+ fs_bio_integrity_alloc(&ioend->io_bio);
+ ioend->io_bio.bi_end_io = xfs_dio_bounce_end_io;
+ bio_set_flag(&ioend->io_bio, BIO_COMPLETE_IN_TASK);
+ submit_bio(&ioend->io_bio);
+}
+
+static void
+xfs_read_bounce_and_resubmit(
+ struct iomap_ioend *ioend)
+{
+ struct bio *bio = &ioend->io_bio;
+ unsigned short vcnt = bio->bi_vcnt;
+ void *private = bio->bi_private;
+
+ trace_xfs_bounce_reread(XFS_I(ioend->io_inode), ioend->io_offset,
+ ioend->io_size);
+
+ /*
+ * Free the bio integrity data for the original bio, as we'll allocate
+ * ons for each sub-I/O, which could deadlock if we keep the original
+ * one around.
+ */
+ if (bio_integrity(bio))
+ fs_bio_integrity_free(bio);
+
+ /*
+ * Reset the bio to submit the bio to the block layer again. Switch to
+ * an end_io handler that simply complets the ioend, as all verification
+ * is done by the end_I/O handlers for the clone bio(s).
+ */
+ bio_reset(bio, xfs_inode_buftarg(XFS_I(ioend->io_inode))->bt_bdev,
+ bio->bi_opf);
+ bio->bi_vcnt = vcnt;
+ bio->bi_private = private;
+ bio->bi_end_io = xfs_end_bio_bounced;
+ bio->bi_iter = (struct bvec_iter) {
+ .bi_sector = ioend->io_sector,
+ .bi_size = ioend->io_size,
+ .bi_offset = ioend->io_bvec_offset,
+ };
+ iomap_bounce_read(ioend, bdev_logical_block_size(bio->bi_bdev),
+ xfs_bounce_submit_ioend);
+}
+
static void
xfs_end_io_read(
struct bio *bio)
{
struct iomap_ioend *ioend = iomap_ioend_from_bio(bio);
+ struct xfs_inode *ip = XFS_I(ioend->io_inode);
+ struct xfs_mount *mp = ip->i_mount;
int error = blk_status_to_errno(bio->bi_status);
- if (!error && (ioend->io_flags & IOMAP_IOEND_INTEGRITY))
+ if (!error && (ioend->io_flags & IOMAP_IOEND_INTEGRITY)) {
error = iomap_ioend_integrity_verify(ioend);
+ if ((ioend->io_flags & IOMAP_IOEND_DIRECT) &&
+ READ_ONCE(mp->m_read_bounce) == XFS_READ_BOUNCE_LAZY) {
+ /*
+ * We only really need to retry for guard tag errors,
+ * but right now we can't distinguish them from other
+ * (i.e, reftag) errors.
+ */
+ if (error) {
+ xfs_read_bounce_and_resubmit(ioend);
+ return;
+ }
+ }
+ }
iomap_finish_ioends(ioend, error);
}
@@ -38,7 +123,18 @@ xfs_ioend_submit_read(
loff_t file_offset,
u16 ioend_flags)
{
- iomap_init_ioend(inode, bio, file_offset, ioend_flags);
+ struct xfs_inode *ip = XFS_I(inode);
+ struct xfs_mount *mp = ip->i_mount;
+ struct iomap_ioend *ioend;
+
+ ioend = iomap_init_ioend(inode, bio, file_offset, ioend_flags);
+ if ((ioend_flags & IOMAP_IOEND_DIRECT) &&
+ READ_ONCE(mp->m_read_bounce) == XFS_READ_BOUNCE_ALWAYS) {
+ iomap_bounce_read(ioend, bdev_logical_block_size(bio->bi_bdev),
+ xfs_bounce_submit_ioend);
+ return;
+ }
+
if (ioend_flags & IOMAP_IOEND_INTEGRITY)
fs_bio_integrity_alloc(bio);
bio->bi_end_io = xfs_end_io_read;
diff --git a/fs/xfs/xfs_mount.h b/fs/xfs/xfs_mount.h
index 216a38a354e7..894ff2f4ecbd 100644
--- a/fs/xfs/xfs_mount.h
+++ b/fs/xfs/xfs_mount.h
@@ -142,6 +142,12 @@ struct xfs_freecounter {
uint64_t res_saved;
};
+enum xfs_read_bounce {
+ XFS_READ_BOUNCE_NEVER,
+ XFS_READ_BOUNCE_ALWAYS,
+ XFS_READ_BOUNCE_LAZY,
+};
+
/*
* The struct xfsmount layout is optimised to separate read-mostly variables
* from variables that are frequently modified. We put the read-mostly variables
@@ -177,6 +183,7 @@ typedef struct xfs_mount {
struct workqueue_struct *m_sync_workqueue;
struct workqueue_struct *m_blockgc_wq;
struct workqueue_struct *m_inodegc_wq;
+ enum xfs_read_bounce m_read_bounce;
int m_bsize; /* fs logical block size */
uint8_t m_blkbit_log; /* blocklog + NBBY */
@@ -291,6 +298,7 @@ typedef struct xfs_mount {
struct xfs_zone_info *m_zone_info; /* zone allocator information */
struct dentry *m_debugfs; /* debugfs parent */
struct xfs_kobj m_kobj;
+ struct xfs_kobj m_csum_kobj;
struct xfs_kobj m_error_kobj;
struct xfs_kobj m_error_meta_kobj;
struct xfs_error_cfg m_error_cfg[XFS_ERR_CLASS_MAX][XFS_ERR_ERRNO_MAX];
diff --git a/fs/xfs/xfs_super.c b/fs/xfs/xfs_super.c
index b24db75eaedc..fce1d2905c94 100644
--- a/fs/xfs/xfs_super.c
+++ b/fs/xfs/xfs_super.c
@@ -2317,6 +2317,7 @@ xfs_init_fs_context(
mp->m_logbufs = -1;
mp->m_logbsize = -1;
mp->m_allocsize_log = 16; /* 64k */
+ mp->m_read_bounce = XFS_READ_BOUNCE_LAZY;
xfs_hooks_init(&mp->m_dir_update_hooks);
diff --git a/fs/xfs/xfs_sysfs.c b/fs/xfs/xfs_sysfs.c
index b62712187324..2e969e8f279f 100644
--- a/fs/xfs/xfs_sysfs.c
+++ b/fs/xfs/xfs_sysfs.c
@@ -392,6 +392,57 @@ const struct kobj_type xfs_stats_ktype = {
.default_groups = xfs_stats_groups,
};
+static inline struct xfs_mount *csum_to_mp(struct kobject *kobj)
+{
+ return container_of(to_kobj(kobj), struct xfs_mount, m_csum_kobj);
+}
+
+static const char * const bounce_modes[] = {
+ [XFS_READ_BOUNCE_NEVER] = "never",
+ [XFS_READ_BOUNCE_ALWAYS] = "always",
+ [XFS_READ_BOUNCE_LAZY] = "lazy",
+};
+
+static ssize_t
+read_bounce_show(
+ struct kobject *kobj,
+ char *buf)
+{
+ struct xfs_mount *mp = csum_to_mp(kobj);
+
+ return sysfs_emit(buf, "%s\n",
+ bounce_modes[READ_ONCE(mp->m_read_bounce)]);
+}
+
+static ssize_t
+read_bounce_store(
+ struct kobject *kobj,
+ const char *buf,
+ size_t count)
+{
+ struct xfs_mount *mp = csum_to_mp(kobj);
+ int ret;
+
+ ret = sysfs_match_string(bounce_modes, buf);
+ if (ret < 0)
+ return ret;
+ WRITE_ONCE(mp->m_read_bounce, ret);
+ return count;
+}
+XFS_SYSFS_ATTR_RW(read_bounce);
+
+static struct attribute *xfs_csum_attrs[] = {
+ ATTR_LIST(read_bounce),
+ NULL,
+};
+ATTRIBUTE_GROUPS(xfs_csum);
+
+static const struct kobj_type xfs_csum_ktype = {
+ .release = xfs_sysfs_release,
+ .sysfs_ops = &xfs_sysfs_ops,
+ .default_groups = xfs_csum_groups,
+};
+
/* xlog */
static inline struct xlog *
@@ -797,6 +848,18 @@ xfs_zoned_sysfs_del(struct xfs_mount *mp)
xfs_sysfs_del(&mp->m_zoned_kobj);
}
+static bool
+xfs_has_read_bounce(
+ struct xfs_mount *mp)
+{
+ if (bdev_has_integrity_csum(mp->m_ddev_targp->bt_bdev))
+ return true;
+ if (mp->m_rtdev_targp &&
+ bdev_has_integrity_csum(mp->m_rtdev_targp->bt_bdev))
+ return true;
+ return false;
+}
+
int
xfs_mount_sysfs_init(
struct xfs_mount *mp)
@@ -837,8 +900,18 @@ xfs_mount_sysfs_init(
if (error)
goto out_remove_error_dir;
+ if (xfs_has_read_bounce(mp)) {
+ /* .../xfs/<dev>/csum/ */
+ error = xfs_sysfs_init(&mp->m_csum_kobj, &xfs_csum_ktype,
+ &mp->m_kobj, "csum");
+ if (error)
+ goto out_remove_error_metadata_dir;
+ }
+
return 0;
+out_remove_error_metadata_dir:
+ xfs_sysfs_del(&mp->m_error_meta_kobj);
out_remove_error_dir:
xfs_sysfs_del(&mp->m_error_kobj);
out_remove_stats_dir:
@@ -855,6 +928,9 @@ xfs_mount_sysfs_del(
struct xfs_error_cfg *cfg;
int i, j;
+ if (xfs_has_read_bounce(mp))
+ xfs_sysfs_del(&mp->m_csum_kobj);
+
for (i = 0; i < XFS_ERR_CLASS_MAX; i++) {
for (j = 0; j < XFS_ERR_ERRNO_MAX; j++) {
cfg = &mp->m_error_cfg[i][j];
diff --git a/fs/xfs/xfs_trace.h b/fs/xfs/xfs_trace.h
index f333c938fbd9..2af9a1429ae9 100644
--- a/fs/xfs/xfs_trace.h
+++ b/fs/xfs/xfs_trace.h
@@ -1896,6 +1896,7 @@ DEFINE_SIMPLE_IO_EVENT(xfs_zero_eof);
DEFINE_SIMPLE_IO_EVENT(xfs_end_io_direct_write);
DEFINE_SIMPLE_IO_EVENT(xfs_file_splice_read);
DEFINE_SIMPLE_IO_EVENT(xfs_zoned_map_blocks);
+DEFINE_SIMPLE_IO_EVENT(xfs_bounce_reread);
DECLARE_EVENT_CLASS(xfs_itrunc_class,
TP_PROTO(struct xfs_inode *ip, xfs_fsize_t new_size),
--
2.53.0
next prev parent reply other threads:[~2026-08-31 6:41 UTC|newest]
Thread overview: 31+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-31 6:39 lazy bounce buffering for checksummed reads V2 Christoph Hellwig
2026-08-31 6:39 ` [PATCH 01/17] block: pass a maxlen argument to bio_iov_iter_get_pages Christoph Hellwig
2026-08-31 18:04 ` Darrick J. Wong
2026-08-31 6:39 ` [PATCH 02/17] block: warn on too larger integrity allocations Christoph Hellwig
2026-08-31 18:01 ` Darrick J. Wong
2026-09-01 8:12 ` Christoph Hellwig
2026-08-31 6:39 ` [PATCH 03/17] block: split bio_iov_iter_bounce_write Christoph Hellwig
2026-08-31 17:58 ` Darrick J. Wong
2026-08-31 6:39 ` [PATCH 04/17] block: export fs_bio_integrity_{alloc,free} Christoph Hellwig
2026-08-31 17:57 ` Darrick J. Wong
2026-08-31 6:39 ` [PATCH 05/17] iomap: respect maximum I/O size in iomap_dio_bio_iter_one Christoph Hellwig
2026-08-31 17:55 ` Darrick J. Wong
2026-08-31 6:39 ` [PATCH 06/17] iomap: add a iomap_ioend_flags helper Christoph Hellwig
2026-08-31 6:39 ` [PATCH 07/17] iomap: add a IOMAP_IOEND_INTEGRITY flag Christoph Hellwig
2026-08-31 6:39 ` [PATCH 08/17] iomap,xfs: move T10 PI handling for direct I/O into ->submit_io Christoph Hellwig
2026-08-31 6:39 ` [PATCH 09/17] xfs: move PI generation into xfs_submit_zoned_bio Christoph Hellwig
2026-08-31 17:53 ` Darrick J. Wong
2026-08-31 6:39 ` [PATCH 10/17] block,iomap: fix protection information verification with initial bvec offset Christoph Hellwig
2026-08-31 17:52 ` Darrick J. Wong
2026-09-01 8:12 ` Christoph Hellwig
2026-09-01 14:06 ` Darrick J. Wong
2026-08-31 6:39 ` [PATCH 11/17] iomap: better read bounce buffering support Christoph Hellwig
2026-08-31 6:40 ` [PATCH 12/17] xfs: use BIO_COMPLETE_IN_TASK for bounce buffered read I/Os Christoph Hellwig
2026-08-31 6:40 ` [PATCH 13/17] iomap,xfs: move integrity verification to the file system Christoph Hellwig
2026-08-31 6:40 ` Christoph Hellwig [this message]
2026-08-31 17:47 ` [PATCH 14/17] xfs: add support for lazy direct read bounce buffering Darrick J. Wong
2026-09-02 18:24 ` Anuj Gupta
2026-08-31 6:40 ` [PATCH 15/17] xfs: add error injection for lazy " Christoph Hellwig
2026-09-02 18:26 ` Anuj Gupta
2026-08-31 6:40 ` [PATCH 16/17] xfs: log a message at mount time when using integrity protection Christoph Hellwig
2026-08-31 6:40 ` [PATCH 17/17] block,iomap: remove the old read side bounce buffering support Christoph Hellwig
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260831064010.2574896-15-hch@lst.de \
--to=hch@lst.de \
--cc=anuj20.g@samsung.com \
--cc=axboe@kernel.dk \
--cc=brauner@kernel.org \
--cc=cem@kernel.org \
--cc=djwong@kernel.org \
--cc=linux-block@vger.kernel.org \
--cc=linux-fsdevel@vger.kernel.org \
--cc=linux-xfs@vger.kernel.org \
--cc=tz2294@columbia.edu \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox