Linux filesystem development
 help / color / mirror / Atom feed
From: "Darrick J. Wong" <djwong@kernel.org>
To: Christoph Hellwig <hch@lst.de>
Cc: Carlos Maiolino <cem@kernel.org>, Jens Axboe <axboe@kernel.dk>,
	Christian Brauner <brauner@kernel.org>,
	linux-xfs@vger.kernel.org, linux-fsdevel@vger.kernel.org
Subject: Re: [PATCH 11/21] xfs: core RT data checksum support
Date: Fri, 25 Sep 2026 16:20:20 -0700	[thread overview]
Message-ID: <20260925232020.GF2705364@frogsfrogsfrogs> (raw)
In-Reply-To: <20260924100032.2733101-12-hch@lst.de>

On Thu, Sep 24, 2026 at 11:59:43AM +0200, Christoph Hellwig wrote:
> Support reading and writing of data checksum buffers, and generating
> and verifying the checksum.
> 
> All data checksum buffers for an open zone are pre-allocated at zone open
> time, so that we never have to read in a partially written buffer as part
> of a data write, which would otherwise impose very expensive seeks and
> stall writes.
> 
> Signed-off-by: Christoph Hellwig <hch@lst.de>
> ---
>  fs/xfs/Kconfig          |   1 +
>  fs/xfs/Makefile         |   1 +
>  fs/xfs/xfs_inode.h      |   5 +
>  fs/xfs/xfs_rtcsum.c     | 317 ++++++++++++++++++++++++++++++++++++++++
>  fs/xfs/xfs_rtcsum.h     |  23 +++
>  fs/xfs/xfs_super.c      |  14 ++
>  fs/xfs/xfs_sysfs.c      |   2 +
>  fs/xfs/xfs_zone_alloc.c |  32 +++-
>  fs/xfs/xfs_zone_priv.h  |   8 +
>  9 files changed, 399 insertions(+), 4 deletions(-)
>  create mode 100644 fs/xfs/xfs_rtcsum.c
>  create mode 100644 fs/xfs/xfs_rtcsum.h
> 
> diff --git a/fs/xfs/Kconfig b/fs/xfs/Kconfig
> index b99da294e9a3..424a04b507a6 100644
> --- a/fs/xfs/Kconfig
> +++ b/fs/xfs/Kconfig
> @@ -106,6 +106,7 @@ config XFS_RT
>  	bool "XFS Realtime subvolume support"
>  	depends on XFS_FS
>  	default BLK_DEV_ZONED
> +	select CRC64
>  	help
>  	  If you say Y here you will be able to mount and use XFS filesystems
>  	  which contain a realtime subvolume.  The realtime subvolume is a
> diff --git a/fs/xfs/Makefile b/fs/xfs/Makefile
> index 79ea4136fbba..0d57bf0701ec 100644
> --- a/fs/xfs/Makefile
> +++ b/fs/xfs/Makefile
> @@ -142,6 +142,7 @@ xfs-$(CONFIG_XFS_QUOTA)		+= xfs_dquot.o \
>  
>  # xfs_rtbitmap is shared with libxfs
>  xfs-$(CONFIG_XFS_RT)		+= xfs_rtalloc.o \
> +				   xfs_rtcsum.o \
>  				   xfs_zone_alloc.o \
>  				   xfs_zone_gc.o \
>  				   xfs_zone_info.o \
> diff --git a/fs/xfs/xfs_inode.h b/fs/xfs/xfs_inode.h
> index 34c1038ebfcd..9ed2fbfe86ff 100644
> --- a/fs/xfs/xfs_inode.h
> +++ b/fs/xfs/xfs_inode.h
> @@ -376,6 +376,11 @@ static inline bool xfs_inode_can_sw_atomic_write(const struct xfs_inode *ip)
>  	return xfs_can_sw_atomic_write(ip->i_mount);
>  }
>  
> +static inline bool xfs_is_rtcsum_inode(const struct xfs_inode *ip)
> +{
> +	return xfs_has_rtcsum(ip->i_mount) && XFS_IS_REALTIME_INODE(ip);
> +}
> +
>  /*
>   * In-core inode flags.
>   */
> diff --git a/fs/xfs/xfs_rtcsum.c b/fs/xfs/xfs_rtcsum.c
> new file mode 100644
> index 000000000000..7cdc5a5029eb
> --- /dev/null
> +++ b/fs/xfs/xfs_rtcsum.c
> @@ -0,0 +1,317 @@
> +// SPDX-License-Identifier: GPL-2.0
> +/*
> + * Copyright (c) 2026 Christoph Hellwig.
> + */
> +#include "xfs_platform.h"
> +#include "xfs_fs.h"
> +#include "xfs_format.h"
> +#include "xfs_log_format.h"
> +#include "xfs_shared.h"
> +#include "xfs_trans_resv.h"
> +#include "xfs_bit.h"
> +#include "xfs_mount.h"
> +#include "xfs_inode.h"
> +#include "xfs_bmap.h"
> +#include "xfs_rtgroup.h"
> +#include "xfs_rtbitmap.h"
> +#include "xfs_bmap_btree.h"
> +#include "xfs_trans.h"
> +#include "xfs_buf_item.h"
> +#include "xfs_trans_space.h"
> +#include "xfs_error.h"
> +#include "xfs_health.h"
> +#include "xfs_rtcsum.h"
> +#include "xfs_zone_priv.h"
> +#include <linux/iomap.h>
> +
> +static_assert(IOMAP_CSUM_MAX_SIZE <= XFS_RTCSUM_MAX_WRITE);
> +
> +static const char *xfs_data_csum_names[XFS_CSUM_TYPE_MAX] = {
> +	[XFS_CSUM_TYPE_CRC32C]	= "crc32c",
> +	[XFS_CSUM_TYPE_CRC64]	= "crc64",
> +};
> +
> +int
> +xfs_csum_verify(
> +	struct xfs_mount	*mp,
> +	struct bio		*bio,
> +	struct bvec_iter	*iter,
> +	void			*csum_buf,
> +	xfs_fsblock_t		bno,
> +	bool			verbose)
> +{
> +	unsigned int		bsize = mp->m_sb.sb_blocksize;
> +	unsigned int		csum_size = 1u << mp->m_rtcsum_shift;
> +	unsigned int		offset = 0;
> +	union xfs_csum		csum;
> +	union xfs_disk_csum	dsum;
> +
> +	do {
> +		struct bio_vec	bv = mp_bvec_iter_bvec(bio->bi_io_vec, *iter);
> +
> +		if (offset == 0)
> +			xfs_csum_seed(mp, &csum);
> +		bv.bv_len = min(bv.bv_len, bsize - offset);
> +		xfs_csum_gen(mp, bvec_virt(&bv), bv.bv_len, &csum);
> +		offset += bv.bv_len;
> +		if (offset == bsize) {
> +			xfs_csum_finalize(mp, &dsum, &csum);
> +			if (unlikely(memcmp(&dsum, csum_buf, csum_size) != 0))
> +				goto mismatch;
> +			bno++;
> +			csum_buf += csum_size;
> +			offset = 0;
> +		}
> +		bio_advance_iter_single(bio, iter, bv.bv_len);
> +	} while (iter->bi_size);
> +
> +	return 0;
> +
> +mismatch:
> +	if (verbose) {
> +		xfs_warn_ratelimited(mp,
> +"data csum mismatch for rtblock 0x%llx: 0x%*phN (expected 0x%*phN)",
> +			bno, csum_size, &csum, csum_size, csum_buf);
> +	}
> +	return -EIO;
> +}
> +
> +void
> +xfs_csum_generate(
> +	struct xfs_mount	*mp,
> +	struct bio		*bio,
> +	void			*csum_buf)
> +{
> +	struct bvec_iter	iter = bio->bi_iter;
> +	unsigned int		bsize = mp->m_sb.sb_blocksize;
> +	unsigned int		csum_size = 1u << mp->m_rtcsum_shift;
> +	unsigned int		offset = 0;
> +	union xfs_csum		csum;
> +
> +	do {
> +		struct bio_vec	bv = mp_bvec_iter_bvec(bio->bi_io_vec, iter);
> +
> +		if (offset == 0)
> +			xfs_csum_seed(mp, &csum);
> +		bv.bv_len = min(bv.bv_len, bsize - offset);
> +		xfs_csum_gen(mp, bvec_virt(&bv), bv.bv_len, &csum);
> +		offset += bv.bv_len;
> +		if (offset == bsize) {
> +			xfs_csum_finalize(mp, csum_buf, &csum);
> +			csum_buf += csum_size;
> +			offset = 0;
> +		}
> +		bio_advance_iter_single(bio, &iter, bv.bv_len);
> +	} while (iter.bi_size);
> +}
> +
> +/*
> + * Read the checksum buffer for @rtg/@csum_off and return it unlocked.

      Read the checksum buffer for @fsbno?

> + *
> + * We don't need to lock read access to the buffer because checksums will not
> + * change until the @rtg is reset.
> + */
> +int
> +xfs_rtcsum_read_async(
> +	struct xfs_mount	*mp,
> +	xfs_rtblock_t		fsbno,
> +	struct xfs_buf		**bpp)
> +{
> +	xfs_daddr_t		csum_daddr;
> +	struct xfs_rtgroup	*rtg;
> +	int			error;
> +
> +	rtg = xfs_rtgroup_get(mp, xfs_rtb_to_rgno(mp, fsbno));
> +	if (!rtg)
> +		return -EFSCORRUPTED;
> +	error = xfs_rtcsum_bmap(rtg, xfs_rtb_to_rgbno(mp, fsbno), &csum_daddr);
> +	if (!error)
> +		error = xfs_buf_read_async(mp->m_ddev_targp, csum_daddr,
> +				BTOBB(mp->m_rtcsum_bsize), &xfs_rtcsum_buf_ops,
> +				bpp);
> +	xfs_rtgroup_put(rtg);
> +	return error;
> +}
> +
> +/*
> + * When opening a zone for writing, do a speculative buf_get for each csum
> + * buffer.  This ensures we usually have a buffer in-memory when we actually
> + * start writing to it.
> + *
> + * Without this we'd have to read the buffer from disk, as we don't know if
> + * anyone has already written to it by the time we get to the buffer due to
> + * completion reordering.
> + *
> + * When reopening a partially written zone at mount time, just read ahead
> + * the entire csums for the zone.
> + */
> +int
> +xfs_rtcsum_open_zone(
> +	struct xfs_open_zone	*oz)
> +{
> +	struct xfs_rtgroup	*rtg = oz->oz_rtg;
> +	struct xfs_mount	*mp = rtg_mount(rtg);
> +	xfs_rgblock_t		rgbno = 0;
> +	int			i, error;
> +
> +	for (i = 0; i < oz->oz_nr_csum_bufs; i++) {
> +		xfs_daddr_t	csum_daddr;
> +		struct xfs_buf	*bp;
> +
> +		error = xfs_rtcsum_bmap(rtg, rgbno, &csum_daddr);
> +		if (error)
> +			goto out_error;
> +
> +		if (oz->oz_allocated) {
> +			error = xfs_buf_read_async(mp->m_ddev_targp, csum_daddr,
> +					BTOBB(mp->m_rtcsum_bsize),
> +					&xfs_rtcsum_buf_ops, &bp);
> +		} else {
> +			error = xfs_buf_get(mp->m_ddev_targp, csum_daddr,
> +					BTOBB(mp->m_rtcsum_bsize), &bp);
> +			if (!error) {
> +				bp->b_flags = XBF_DONE;
> +				xfs_rtfile_initialize_buf(rtg, XFS_RTGI_CSUM,
> +						bp, NULL);
> +			}
> +			xfs_buf_unlock(bp);
> +		}
> +		if (error)
> +			goto out_error;
> +		oz->oz_csum_bufs[i] = bp;
> +		rgbno += (xfs_rtcsum_payload_size(mp) /
> +			  (1u << mp->m_rtcsum_shift));
> +	}
> +
> +	return 0;
> +
> +out_error:
> +	while (--i >= 0)
> +		xfs_buf_rele(oz->oz_csum_bufs[i]);

/me wonders if this should null out oz_csum_bufs to avoid the
possibility of dangling pointers?  Though I think the only caller will
free the oz if this function returns error so it might not matter much.

--D

  reply	other threads:[~2026-09-25 23:20 UTC|newest]

Thread overview: 69+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-24  9:59 support for RT data checksums Christoph Hellwig
2026-09-24  9:59 ` [PATCH 01/21] block: export fs_bio_integrity_verify Christoph Hellwig
2026-09-24 20:29   ` Darrick J. Wong
2026-09-24  9:59 ` [PATCH 02/21] iomap: add support for data checksumming Christoph Hellwig
2026-09-24 21:39   ` Darrick J. Wong
2026-09-25  5:53     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 03/21] xfs: add a xfs_buf_read_async buffer cache API Christoph Hellwig
2026-09-24 21:43   ` Darrick J. Wong
2026-09-25  5:54     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 04/21] xfs: add xfs_daddr_to_rgno and xfs_daddr_to_rgbno helpers Christoph Hellwig
2026-09-24 21:44   ` Darrick J. Wong
2026-09-24  9:59 ` [PATCH 05/21] xfs: introduce XFS_BLI_PREALLOC Christoph Hellwig
2026-09-24 21:49   ` Darrick J. Wong
2026-09-25  5:57     ` Christoph Hellwig
2026-10-08 11:46   ` Anuj gupta
2026-09-24  9:59 ` [PATCH 06/21] xfs: prepare xfs_rtfile_initialize_blocks for larger than FSB blocks Christoph Hellwig
2026-09-24 22:03   ` Darrick J. Wong
2026-09-25  5:58     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 07/21] xfs: relase zi_open_zones_lock over xfs_open_zone_put on unmount Christoph Hellwig
2026-09-24  9:59 ` [PATCH 08/21] xfs: define the RT data checksum on-disk format Christoph Hellwig
2026-09-24 22:13   ` Darrick J. Wong
2026-09-25  0:04     ` Eric Biggers
2026-09-25  6:01     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 09/21] xfs: add support for per-RTG csum files Christoph Hellwig
2026-09-24 22:24   ` Darrick J. Wong
2026-09-25  6:10     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 10/21] xfs: calculate the log reservation for logging data checksum buffers Christoph Hellwig
2026-09-24 22:30   ` Darrick J. Wong
2026-09-25  6:12     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 11/21] xfs: core RT data checksum support Christoph Hellwig
2026-09-25 23:20   ` Darrick J. Wong [this message]
2026-09-26  6:13     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 12/21] xfs: data checksums require stable writes Christoph Hellwig
2026-09-25 23:21   ` Darrick J. Wong
2026-09-24  9:59 ` [PATCH 13/21] xfs: require file system block size alignment when using data checksums Christoph Hellwig
2026-09-25 23:24   ` Darrick J. Wong
2026-09-26  6:15     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 14/21] xfs: add support for reading with " Christoph Hellwig
2026-09-29  0:42   ` Darrick J. Wong
2026-10-05 12:59     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 15/21] xfs: add support for writing " Christoph Hellwig
2026-09-29  1:01   ` Darrick J. Wong
2026-10-05 13:00     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 16/21] xfs: add data checksum support to zoned garbage collection Christoph Hellwig
2026-09-29  1:06   ` Darrick J. Wong
2026-10-05 13:11     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 17/21] xfs: verify data checksums during media verification Christoph Hellwig
2026-09-29  1:19   ` Darrick J. Wong
2026-10-05 13:13     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 18/21] xfs: don't try to verify checksums on empty zones Christoph Hellwig
2026-09-29  1:25   ` Darrick J. Wong
2026-10-05 13:14     ` Christoph Hellwig
2026-10-08 11:43   ` Anuj gupta
2026-09-24  9:59 ` [PATCH 19/21] xfs: report RT data checksum information via XFS_FSOP_GEOM Christoph Hellwig
2026-09-29  1:26   ` Darrick J. Wong
2026-09-24  9:59 ` [PATCH 20/21] xfs: add an experimental feature warning for RT data checksums Christoph Hellwig
2026-09-29  1:27   ` Darrick J. Wong
2026-09-24  9:59 ` [PATCH 21/21] xfs: enable " Christoph Hellwig
2026-09-29  1:27   ` Darrick J. Wong
2026-10-05 13:16     ` Christoph Hellwig
2026-09-24 22:52 ` support for " Dave Chinner
2026-09-25  6:27   ` Christoph Hellwig
2026-09-27 22:59     ` Dave Chinner
2026-09-28  5:24       ` Christoph Hellwig
2026-09-29 14:11         ` Dave Chinner
2026-09-30  7:11           ` Dave Chinner
2026-10-05 13:53             ` Christoph Hellwig
2026-10-06  5:31               ` Dave Chinner
2026-10-07 13:46                 ` Christoph Hellwig

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260925232020.GF2705364@frogsfrogsfrogs \
    --to=djwong@kernel.org \
    --cc=axboe@kernel.dk \
    --cc=brauner@kernel.org \
    --cc=cem@kernel.org \
    --cc=hch@lst.de \
    --cc=linux-fsdevel@vger.kernel.org \
    --cc=linux-xfs@vger.kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox