From: Christoph Hellwig <hch@lst.de>
To: Carlos Maiolino <cem@kernel.org>
Cc: "Darrick J . Wong" <djwong@kernel.org>,
Jens Axboe <axboe@kernel.dk>,
Christian Brauner <brauner@kernel.org>,
linux-xfs@vger.kernel.org, linux-fsdevel@vger.kernel.org
Subject: [PATCH 03/21] xfs: add a xfs_buf_read_async buffer cache API
Date: Thu, 24 Sep 2026 11:59:35 +0200 [thread overview]
Message-ID: <20260924100032.2733101-4-hch@lst.de> (raw)
In-Reply-To: <20260924100032.2733101-1-hch@lst.de>
Add a new helper that reads a buffer asynchronously. This is similar
to readahead, but doesn't become a no-op under memory or I/O congestion
and returns the buffer to be read.
The intended use is to kick off a read of data checksum buffers at
roughly the same time as the data read so that they are available
in the I/O completion handler.
Signed-off-by: Christoph Hellwig <hch@lst.de>
---
fs/xfs/xfs_buf.c | 130 ++++++++++++++++++++++++++++++++++++---------
fs/xfs/xfs_buf.h | 4 ++
fs/xfs/xfs_trace.h | 3 ++
3 files changed, 113 insertions(+), 24 deletions(-)
diff --git a/fs/xfs/xfs_buf.c b/fs/xfs/xfs_buf.c
index eee491c01d8c..966c06bffeff 100644
--- a/fs/xfs/xfs_buf.c
+++ b/fs/xfs/xfs_buf.c
@@ -627,6 +627,40 @@ _xfs_buf_read(
return xfs_buf_iowait(bp);
}
+/*
+ * If we've had a read error, then the contents of the buffer are invalid and
+ * should not be used. To ensure that a followup read tries to pull the buffer
+ * from disk again, we clear the XBF_DONE flag and mark the buffer stale.
+ * This ensures that anyone who has a current reference to the buffer will
+ * interpret it's contents correctly and future cache lookups will also treat it
+ * as an empty, uninitialised buffer.
+ */
+static int
+xfs_buf_read_error(
+ struct xfs_buf *bp,
+ xfs_failaddr_t fa,
+ int error)
+{
+ /*
+ * Check against log shutdown for error reporting because metadata
+ * writeback may require a read first and we need to report errors in
+ * metadata writeback until the log is shut down.
+ * High level transaction read functions already check against mount
+ * shutdown, anyway, so we only need to be concerned about low level IO
+ * interactions here.
+ */
+ if (!xlog_is_shutdown(bp->b_mount->m_log))
+ xfs_buf_ioerror_alert(bp, fa);
+ xfs_buf_clear_flags(bp, XBF_DONE);
+ xfs_buf_stale(bp);
+ xfs_buf_relse(bp);
+
+ /* bad CRC means corrupted metadata */
+ if (error == -EFSBADCRC)
+ return -EFSCORRUPTED;
+ return error;
+}
+
int
xfs_buf_read_map(
struct xfs_buftarg *target,
@@ -699,39 +733,87 @@ xfs_buf_read_map(
}
if (error)
- goto out_ioerror;
-
+ return xfs_buf_read_error(bp, fa, error);
*bpp = bp;
return 0;
+}
-out_ioerror:
+int
+xfs_buf_read_async_wait(
+ struct xfs_buf *bp)
+{
/*
- * Check against log shutdown for error reporting because metadata
- * writeback may require a read first and we need to report errors in
- * metadata writeback until the log is shut down. High level
- * transaction read functions already check against mount shutdown, so
- * we only need to be concerned about low level/ IO interactions here.
+ * Protect against the case where the checksum read is slower than the
+ * data read.
*/
- if (!xlog_is_shutdown(target->bt_mount->m_log))
- xfs_buf_ioerror_alert(bp, fa);
+ if ((READ_ONCE(bp->b_flags) & (XBF_DONE | XBF_STALE)) == XBF_DONE &&
+ !bp->b_error) {
+ trace_xfs_buf_read_async_wait(bp, 0, _RET_IP_);
+ return 0;
+ }
+
+ /* xfs_buf_find_lock can't return an error with 0 flags */
+ xfs_buf_find_lock(bp, 0);
+ trace_xfs_buf_read_async_lock(bp, 0, _RET_IP_);
+ if (bp->b_error)
+ return xfs_buf_read_error(bp, __builtin_return_address(0),
+ bp->b_error);
+ ASSERT(bp->b_ops);
+ xfs_buf_clear_flags(bp, XBF_READ);
+ xfs_buf_unlock(bp);
+ return 0;
+}
+
+/*
+ * Kick off an asynchronous read. Unlike readahead, this returns a reference
+ * to the buffer, and reliably reads the data instead of skipping the read on
+ * memory pressure.
+ *
+ * The buffer may be locked when I/O is kicked off, but the I/O completion
+ * handler will unlock it. The caller needs to lock itself if need to prevent
+ * concurrent access or to synchronize with I/O completion.
+ */
+int
+xfs_buf_read_async(
+ struct xfs_buftarg *btp,
+ xfs_daddr_t daddr,
+ size_t numblks,
+ const struct xfs_buf_ops *ops,
+ struct xfs_buf **bpp)
+{
+ DEFINE_SINGLE_BUF_MAP(map, daddr, numblks);
+ struct xfs_buf *bp;
+ int error;
+
+ ASSERT(!xfs_buftarg_is_mem(btp));
+
+ error = xfs_find_get_buf(btp, &map, 1, XBF_READ, &bp);
+ if (error)
+ return error;
/*
- * If we've had a read error, then the contents of the buffer are
- * invalid and should not be used. To ensure that a followup read tries
- * to pull the buffer from disk again, we clear the XBF_DONE flag and
- * mark the buffer stale. This ensures that anyone who has a current
- * reference to the buffer will interpret it's contents correctly and
- * future cache lookups will also treat it as an empty, uninitialised
- * buffer.
+ * Do a lockless fast path check for a valid uptodate buffer and avoid
+ * locking entirely in this case.
*/
- xfs_buf_clear_flags(bp, XBF_DONE);
- xfs_buf_stale(bp);
- xfs_buf_relse(bp);
+ if ((READ_ONCE(bp->b_flags) & (XBF_DONE | XBF_STALE)) == XBF_DONE)
+ goto done;
- /* bad CRC means corrupted metadata */
- if (error == -EFSBADCRC)
- return -EFSCORRUPTED;
- return error;
+ /* xfs_buf_find_lock can't return an error with 0 flags */
+ xfs_buf_find_lock(bp, 0);
+ if (bp->b_flags & XBF_DONE) {
+ xfs_buf_unlock(bp);
+ goto done;
+ }
+ trace_xfs_buf_read_async(bp, 0, _RET_IP_);
+ XFS_STATS_INC(btp->bt_mount, xb_get_read);
+ xfs_buf_hold(bp);
+ bp->b_ops = ops;
+ xfs_buf_clear_flags(bp, XBF_WRITE);
+ xfs_buf_set_flags(bp, XBF_READ | XBF_ASYNC);
+ xfs_buf_submit(bp);
+done:
+ *bpp = bp;
+ return 0;
}
/*
diff --git a/fs/xfs/xfs_buf.h b/fs/xfs/xfs_buf.h
index a4729253b56f..1b352ec91aa6 100644
--- a/fs/xfs/xfs_buf.h
+++ b/fs/xfs/xfs_buf.h
@@ -255,6 +255,10 @@ xfs_buf_readahead(
return xfs_buf_readahead_map(target, &map, 1, ops);
}
+int xfs_buf_read_async(struct xfs_buftarg *btp, xfs_daddr_t daddr,
+ size_t numblks, const struct xfs_buf_ops *ops,
+ struct xfs_buf **bpp);
+int xfs_buf_read_async_wait(struct xfs_buf *bp);
int xfs_buf_get_uncached(struct xfs_buftarg *target, size_t numblks,
struct xfs_buf **bpp);
int xfs_buf_read_uncached(struct xfs_buftarg *target, xfs_daddr_t daddr,
diff --git a/fs/xfs/xfs_trace.h b/fs/xfs/xfs_trace.h
index 2af9a1429ae9..afaabd3adc73 100644
--- a/fs/xfs/xfs_trace.h
+++ b/fs/xfs/xfs_trace.h
@@ -840,6 +840,9 @@ DEFINE_EVENT(xfs_buf_flags_class, name, \
TP_ARGS(bp, flags, caller_ip))
DEFINE_BUF_FLAGS_EVENT(xfs_buf_get);
DEFINE_BUF_FLAGS_EVENT(xfs_buf_read);
+DEFINE_BUF_FLAGS_EVENT(xfs_buf_read_async);
+DEFINE_BUF_FLAGS_EVENT(xfs_buf_read_async_wait);
+DEFINE_BUF_FLAGS_EVENT(xfs_buf_read_async_lock);
DEFINE_BUF_FLAGS_EVENT(xfs_buf_readahead);
TRACE_EVENT(xfs_buf_ioerror,
--
2.53.0
next prev parent reply other threads:[~2026-09-24 10:00 UTC|newest]
Thread overview: 69+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-24 9:59 support for RT data checksums Christoph Hellwig
2026-09-24 9:59 ` [PATCH 01/21] block: export fs_bio_integrity_verify Christoph Hellwig
2026-09-24 20:29 ` Darrick J. Wong
2026-09-24 9:59 ` [PATCH 02/21] iomap: add support for data checksumming Christoph Hellwig
2026-09-24 21:39 ` Darrick J. Wong
2026-09-25 5:53 ` Christoph Hellwig
2026-09-24 9:59 ` Christoph Hellwig [this message]
2026-09-24 21:43 ` [PATCH 03/21] xfs: add a xfs_buf_read_async buffer cache API Darrick J. Wong
2026-09-25 5:54 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 04/21] xfs: add xfs_daddr_to_rgno and xfs_daddr_to_rgbno helpers Christoph Hellwig
2026-09-24 21:44 ` Darrick J. Wong
2026-09-24 9:59 ` [PATCH 05/21] xfs: introduce XFS_BLI_PREALLOC Christoph Hellwig
2026-09-24 21:49 ` Darrick J. Wong
2026-09-25 5:57 ` Christoph Hellwig
2026-10-08 11:46 ` Anuj gupta
2026-09-24 9:59 ` [PATCH 06/21] xfs: prepare xfs_rtfile_initialize_blocks for larger than FSB blocks Christoph Hellwig
2026-09-24 22:03 ` Darrick J. Wong
2026-09-25 5:58 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 07/21] xfs: relase zi_open_zones_lock over xfs_open_zone_put on unmount Christoph Hellwig
2026-09-24 9:59 ` [PATCH 08/21] xfs: define the RT data checksum on-disk format Christoph Hellwig
2026-09-24 22:13 ` Darrick J. Wong
2026-09-25 0:04 ` Eric Biggers
2026-09-25 6:01 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 09/21] xfs: add support for per-RTG csum files Christoph Hellwig
2026-09-24 22:24 ` Darrick J. Wong
2026-09-25 6:10 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 10/21] xfs: calculate the log reservation for logging data checksum buffers Christoph Hellwig
2026-09-24 22:30 ` Darrick J. Wong
2026-09-25 6:12 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 11/21] xfs: core RT data checksum support Christoph Hellwig
2026-09-25 23:20 ` Darrick J. Wong
2026-09-26 6:13 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 12/21] xfs: data checksums require stable writes Christoph Hellwig
2026-09-25 23:21 ` Darrick J. Wong
2026-09-24 9:59 ` [PATCH 13/21] xfs: require file system block size alignment when using data checksums Christoph Hellwig
2026-09-25 23:24 ` Darrick J. Wong
2026-09-26 6:15 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 14/21] xfs: add support for reading with " Christoph Hellwig
2026-09-29 0:42 ` Darrick J. Wong
2026-10-05 12:59 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 15/21] xfs: add support for writing " Christoph Hellwig
2026-09-29 1:01 ` Darrick J. Wong
2026-10-05 13:00 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 16/21] xfs: add data checksum support to zoned garbage collection Christoph Hellwig
2026-09-29 1:06 ` Darrick J. Wong
2026-10-05 13:11 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 17/21] xfs: verify data checksums during media verification Christoph Hellwig
2026-09-29 1:19 ` Darrick J. Wong
2026-10-05 13:13 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 18/21] xfs: don't try to verify checksums on empty zones Christoph Hellwig
2026-09-29 1:25 ` Darrick J. Wong
2026-10-05 13:14 ` Christoph Hellwig
2026-10-08 11:43 ` Anuj gupta
2026-09-24 9:59 ` [PATCH 19/21] xfs: report RT data checksum information via XFS_FSOP_GEOM Christoph Hellwig
2026-09-29 1:26 ` Darrick J. Wong
2026-09-24 9:59 ` [PATCH 20/21] xfs: add an experimental feature warning for RT data checksums Christoph Hellwig
2026-09-29 1:27 ` Darrick J. Wong
2026-09-24 9:59 ` [PATCH 21/21] xfs: enable " Christoph Hellwig
2026-09-29 1:27 ` Darrick J. Wong
2026-10-05 13:16 ` Christoph Hellwig
2026-09-24 22:52 ` support for " Dave Chinner
2026-09-25 6:27 ` Christoph Hellwig
2026-09-27 22:59 ` Dave Chinner
2026-09-28 5:24 ` Christoph Hellwig
2026-09-29 14:11 ` Dave Chinner
2026-09-30 7:11 ` Dave Chinner
2026-10-05 13:53 ` Christoph Hellwig
2026-10-06 5:31 ` Dave Chinner
2026-10-07 13:46 ` Christoph Hellwig
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260924100032.2733101-4-hch@lst.de \
--to=hch@lst.de \
--cc=axboe@kernel.dk \
--cc=brauner@kernel.org \
--cc=cem@kernel.org \
--cc=djwong@kernel.org \
--cc=linux-fsdevel@vger.kernel.org \
--cc=linux-xfs@vger.kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.