From: Christoph Hellwig <hch@lst.de>
To: Carlos Maiolino <cem@kernel.org>
Cc: "Darrick J . Wong" <djwong@kernel.org>,
Jens Axboe <axboe@kernel.dk>,
Christian Brauner <brauner@kernel.org>,
linux-xfs@vger.kernel.org, linux-fsdevel@vger.kernel.org
Subject: [PATCH 03/21] xfs: add a xfs_buf_read_async buffer cache API
Date: Thu, 24 Sep 2026 11:59:35 +0200 [thread overview]
Message-ID: <20260924100032.2733101-4-hch@lst.de> (raw)
In-Reply-To: <20260924100032.2733101-1-hch@lst.de>
Add a new helper that reads a buffer asynchronously. This is similar
to readahead, but doesn't become a no-op under memory or I/O congestion
and returns the buffer to be read.
The intended use is to kick off a read of data checksum buffers at
roughly the same time as the data read so that they are available
in the I/O completion handler.
Signed-off-by: Christoph Hellwig <hch@lst.de>
---
fs/xfs/xfs_buf.c | 130 ++++++++++++++++++++++++++++++++++++---------
fs/xfs/xfs_buf.h | 4 ++
fs/xfs/xfs_trace.h | 3 ++
3 files changed, 113 insertions(+), 24 deletions(-)
diff --git a/fs/xfs/xfs_buf.c b/fs/xfs/xfs_buf.c
index eee491c01d8c..966c06bffeff 100644
--- a/fs/xfs/xfs_buf.c
+++ b/fs/xfs/xfs_buf.c
@@ -627,6 +627,40 @@ _xfs_buf_read(
return xfs_buf_iowait(bp);
}
+/*
+ * If we've had a read error, then the contents of the buffer are invalid and
+ * should not be used. To ensure that a followup read tries to pull the buffer
+ * from disk again, we clear the XBF_DONE flag and mark the buffer stale.
+ * This ensures that anyone who has a current reference to the buffer will
+ * interpret it's contents correctly and future cache lookups will also treat it
+ * as an empty, uninitialised buffer.
+ */
+static int
+xfs_buf_read_error(
+ struct xfs_buf *bp,
+ xfs_failaddr_t fa,
+ int error)
+{
+ /*
+ * Check against log shutdown for error reporting because metadata
+ * writeback may require a read first and we need to report errors in
+ * metadata writeback until the log is shut down.
+ * High level transaction read functions already check against mount
+ * shutdown, anyway, so we only need to be concerned about low level IO
+ * interactions here.
+ */
+ if (!xlog_is_shutdown(bp->b_mount->m_log))
+ xfs_buf_ioerror_alert(bp, fa);
+ xfs_buf_clear_flags(bp, XBF_DONE);
+ xfs_buf_stale(bp);
+ xfs_buf_relse(bp);
+
+ /* bad CRC means corrupted metadata */
+ if (error == -EFSBADCRC)
+ return -EFSCORRUPTED;
+ return error;
+}
+
int
xfs_buf_read_map(
struct xfs_buftarg *target,
@@ -699,39 +733,87 @@ xfs_buf_read_map(
}
if (error)
- goto out_ioerror;
-
+ return xfs_buf_read_error(bp, fa, error);
*bpp = bp;
return 0;
+}
-out_ioerror:
+int
+xfs_buf_read_async_wait(
+ struct xfs_buf *bp)
+{
/*
- * Check against log shutdown for error reporting because metadata
- * writeback may require a read first and we need to report errors in
- * metadata writeback until the log is shut down. High level
- * transaction read functions already check against mount shutdown, so
- * we only need to be concerned about low level/ IO interactions here.
+ * Protect against the case where the checksum read is slower than the
+ * data read.
*/
- if (!xlog_is_shutdown(target->bt_mount->m_log))
- xfs_buf_ioerror_alert(bp, fa);
+ if ((READ_ONCE(bp->b_flags) & (XBF_DONE | XBF_STALE)) == XBF_DONE &&
+ !bp->b_error) {
+ trace_xfs_buf_read_async_wait(bp, 0, _RET_IP_);
+ return 0;
+ }
+
+ /* xfs_buf_find_lock can't return an error with 0 flags */
+ xfs_buf_find_lock(bp, 0);
+ trace_xfs_buf_read_async_lock(bp, 0, _RET_IP_);
+ if (bp->b_error)
+ return xfs_buf_read_error(bp, __builtin_return_address(0),
+ bp->b_error);
+ ASSERT(bp->b_ops);
+ xfs_buf_clear_flags(bp, XBF_READ);
+ xfs_buf_unlock(bp);
+ return 0;
+}
+
+/*
+ * Kick off an asynchronous read. Unlike readahead, this returns a reference
+ * to the buffer, and reliably reads the data instead of skipping the read on
+ * memory pressure.
+ *
+ * The buffer may be locked when I/O is kicked off, but the I/O completion
+ * handler will unlock it. The caller needs to lock itself if need to prevent
+ * concurrent access or to synchronize with I/O completion.
+ */
+int
+xfs_buf_read_async(
+ struct xfs_buftarg *btp,
+ xfs_daddr_t daddr,
+ size_t numblks,
+ const struct xfs_buf_ops *ops,
+ struct xfs_buf **bpp)
+{
+ DEFINE_SINGLE_BUF_MAP(map, daddr, numblks);
+ struct xfs_buf *bp;
+ int error;
+
+ ASSERT(!xfs_buftarg_is_mem(btp));
+
+ error = xfs_find_get_buf(btp, &map, 1, XBF_READ, &bp);
+ if (error)
+ return error;
/*
- * If we've had a read error, then the contents of the buffer are
- * invalid and should not be used. To ensure that a followup read tries
- * to pull the buffer from disk again, we clear the XBF_DONE flag and
- * mark the buffer stale. This ensures that anyone who has a current
- * reference to the buffer will interpret it's contents correctly and
- * future cache lookups will also treat it as an empty, uninitialised
- * buffer.
+ * Do a lockless fast path check for a valid uptodate buffer and avoid
+ * locking entirely in this case.
*/
- xfs_buf_clear_flags(bp, XBF_DONE);
- xfs_buf_stale(bp);
- xfs_buf_relse(bp);
+ if ((READ_ONCE(bp->b_flags) & (XBF_DONE | XBF_STALE)) == XBF_DONE)
+ goto done;
- /* bad CRC means corrupted metadata */
- if (error == -EFSBADCRC)
- return -EFSCORRUPTED;
- return error;
+ /* xfs_buf_find_lock can't return an error with 0 flags */
+ xfs_buf_find_lock(bp, 0);
+ if (bp->b_flags & XBF_DONE) {
+ xfs_buf_unlock(bp);
+ goto done;
+ }
+ trace_xfs_buf_read_async(bp, 0, _RET_IP_);
+ XFS_STATS_INC(btp->bt_mount, xb_get_read);
+ xfs_buf_hold(bp);
+ bp->b_ops = ops;
+ xfs_buf_clear_flags(bp, XBF_WRITE);
+ xfs_buf_set_flags(bp, XBF_READ | XBF_ASYNC);
+ xfs_buf_submit(bp);
+done:
+ *bpp = bp;
+ return 0;
}
/*
diff --git a/fs/xfs/xfs_buf.h b/fs/xfs/xfs_buf.h
index a4729253b56f..1b352ec91aa6 100644
--- a/fs/xfs/xfs_buf.h
+++ b/fs/xfs/xfs_buf.h
@@ -255,6 +255,10 @@ xfs_buf_readahead(
return xfs_buf_readahead_map(target, &map, 1, ops);
}
+int xfs_buf_read_async(struct xfs_buftarg *btp, xfs_daddr_t daddr,
+ size_t numblks, const struct xfs_buf_ops *ops,
+ struct xfs_buf **bpp);
+int xfs_buf_read_async_wait(struct xfs_buf *bp);
int xfs_buf_get_uncached(struct xfs_buftarg *target, size_t numblks,
struct xfs_buf **bpp);
int xfs_buf_read_uncached(struct xfs_buftarg *target, xfs_daddr_t daddr,
diff --git a/fs/xfs/xfs_trace.h b/fs/xfs/xfs_trace.h
index 2af9a1429ae9..afaabd3adc73 100644
--- a/fs/xfs/xfs_trace.h
+++ b/fs/xfs/xfs_trace.h
@@ -840,6 +840,9 @@ DEFINE_EVENT(xfs_buf_flags_class, name, \
TP_ARGS(bp, flags, caller_ip))
DEFINE_BUF_FLAGS_EVENT(xfs_buf_get);
DEFINE_BUF_FLAGS_EVENT(xfs_buf_read);
+DEFINE_BUF_FLAGS_EVENT(xfs_buf_read_async);
+DEFINE_BUF_FLAGS_EVENT(xfs_buf_read_async_wait);
+DEFINE_BUF_FLAGS_EVENT(xfs_buf_read_async_lock);
DEFINE_BUF_FLAGS_EVENT(xfs_buf_readahead);
TRACE_EVENT(xfs_buf_ioerror,
--
2.53.0
next prev parent reply other threads:[~2026-09-24 10:00 UTC|newest]
Thread overview: 69+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-24 9:59 support for RT data checksums Christoph Hellwig
2026-09-24 9:59 ` [PATCH 01/21] block: export fs_bio_integrity_verify Christoph Hellwig
2026-09-24 20:29 ` Darrick J. Wong
2026-09-24 9:59 ` [PATCH 02/21] iomap: add support for data checksumming Christoph Hellwig
2026-09-24 21:39 ` Darrick J. Wong
2026-09-25 5:53 ` Christoph Hellwig
2026-09-24 9:59 ` Christoph Hellwig [this message]
2026-09-24 21:43 ` [PATCH 03/21] xfs: add a xfs_buf_read_async buffer cache API Darrick J. Wong
2026-09-25 5:54 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 04/21] xfs: add xfs_daddr_to_rgno and xfs_daddr_to_rgbno helpers Christoph Hellwig
2026-09-24 21:44 ` Darrick J. Wong
2026-09-24 9:59 ` [PATCH 05/21] xfs: introduce XFS_BLI_PREALLOC Christoph Hellwig
2026-09-24 21:49 ` Darrick J. Wong
2026-09-25 5:57 ` Christoph Hellwig
2026-10-08 11:46 ` Anuj gupta
2026-09-24 9:59 ` [PATCH 06/21] xfs: prepare xfs_rtfile_initialize_blocks for larger than FSB blocks Christoph Hellwig
2026-09-24 22:03 ` Darrick J. Wong
2026-09-25 5:58 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 07/21] xfs: relase zi_open_zones_lock over xfs_open_zone_put on unmount Christoph Hellwig
2026-09-24 9:59 ` [PATCH 08/21] xfs: define the RT data checksum on-disk format Christoph Hellwig
2026-09-24 22:13 ` Darrick J. Wong
2026-09-25 0:04 ` Eric Biggers
2026-09-25 6:01 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 09/21] xfs: add support for per-RTG csum files Christoph Hellwig
2026-09-24 22:24 ` Darrick J. Wong
2026-09-25 6:10 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 10/21] xfs: calculate the log reservation for logging data checksum buffers Christoph Hellwig
2026-09-24 22:30 ` Darrick J. Wong
2026-09-25 6:12 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 11/21] xfs: core RT data checksum support Christoph Hellwig
2026-09-25 23:20 ` Darrick J. Wong
2026-09-26 6:13 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 12/21] xfs: data checksums require stable writes Christoph Hellwig
2026-09-25 23:21 ` Darrick J. Wong
2026-09-24 9:59 ` [PATCH 13/21] xfs: require file system block size alignment when using data checksums Christoph Hellwig
2026-09-25 23:24 ` Darrick J. Wong
2026-09-26 6:15 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 14/21] xfs: add support for reading with " Christoph Hellwig
2026-09-29 0:42 ` Darrick J. Wong
2026-10-05 12:59 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 15/21] xfs: add support for writing " Christoph Hellwig
2026-09-29 1:01 ` Darrick J. Wong
2026-10-05 13:00 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 16/21] xfs: add data checksum support to zoned garbage collection Christoph Hellwig
2026-09-29 1:06 ` Darrick J. Wong
2026-10-05 13:11 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 17/21] xfs: verify data checksums during media verification Christoph Hellwig
2026-09-29 1:19 ` Darrick J. Wong
2026-10-05 13:13 ` Christoph Hellwig
2026-09-24 9:59 ` [PATCH 18/21] xfs: don't try to verify checksums on empty zones Christoph Hellwig
2026-09-29 1:25 ` Darrick J. Wong
2026-10-05 13:14 ` Christoph Hellwig
2026-10-08 11:43 ` Anuj gupta
2026-09-24 9:59 ` [PATCH 19/21] xfs: report RT data checksum information via XFS_FSOP_GEOM Christoph Hellwig
2026-09-29 1:26 ` Darrick J. Wong
2026-09-24 9:59 ` [PATCH 20/21] xfs: add an experimental feature warning for RT data checksums Christoph Hellwig
2026-09-29 1:27 ` Darrick J. Wong
2026-09-24 9:59 ` [PATCH 21/21] xfs: enable " Christoph Hellwig
2026-09-29 1:27 ` Darrick J. Wong
2026-10-05 13:16 ` Christoph Hellwig
2026-09-24 22:52 ` support for " Dave Chinner
2026-09-25 6:27 ` Christoph Hellwig
2026-09-27 22:59 ` Dave Chinner
2026-09-28 5:24 ` Christoph Hellwig
2026-09-29 14:11 ` Dave Chinner
2026-09-30 7:11 ` Dave Chinner
2026-10-05 13:53 ` Christoph Hellwig
2026-10-06 5:31 ` Dave Chinner
2026-10-07 13:46 ` Christoph Hellwig
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260924100032.2733101-4-hch@lst.de \
--to=hch@lst.de \
--cc=axboe@kernel.dk \
--cc=brauner@kernel.org \
--cc=cem@kernel.org \
--cc=djwong@kernel.org \
--cc=linux-fsdevel@vger.kernel.org \
--cc=linux-xfs@vger.kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox