All of lore.kernel.org
 help / color / mirror / Atom feed
From: "Darrick J. Wong" <djwong@kernel.org>
To: Christoph Hellwig <hch@lst.de>
Cc: Carlos Maiolino <cem@kernel.org>, Jens Axboe <axboe@kernel.dk>,
	Christian Brauner <brauner@kernel.org>,
	linux-xfs@vger.kernel.org, linux-fsdevel@vger.kernel.org
Subject: Re: [PATCH 11/21] xfs: core RT data checksum support
Date: Fri, 25 Sep 2026 16:20:20 -0700	[thread overview]
Message-ID: <20260925232020.GF2705364@frogsfrogsfrogs> (raw)
In-Reply-To: <20260924100032.2733101-12-hch@lst.de>

On Thu, Sep 24, 2026 at 11:59:43AM +0200, Christoph Hellwig wrote:
> Support reading and writing of data checksum buffers, and generating
> and verifying the checksum.
> 
> All data checksum buffers for an open zone are pre-allocated at zone open
> time, so that we never have to read in a partially written buffer as part
> of a data write, which would otherwise impose very expensive seeks and
> stall writes.
> 
> Signed-off-by: Christoph Hellwig <hch@lst.de>
> ---
>  fs/xfs/Kconfig          |   1 +
>  fs/xfs/Makefile         |   1 +
>  fs/xfs/xfs_inode.h      |   5 +
>  fs/xfs/xfs_rtcsum.c     | 317 ++++++++++++++++++++++++++++++++++++++++
>  fs/xfs/xfs_rtcsum.h     |  23 +++
>  fs/xfs/xfs_super.c      |  14 ++
>  fs/xfs/xfs_sysfs.c      |   2 +
>  fs/xfs/xfs_zone_alloc.c |  32 +++-
>  fs/xfs/xfs_zone_priv.h  |   8 +
>  9 files changed, 399 insertions(+), 4 deletions(-)
>  create mode 100644 fs/xfs/xfs_rtcsum.c
>  create mode 100644 fs/xfs/xfs_rtcsum.h
> 
> diff --git a/fs/xfs/Kconfig b/fs/xfs/Kconfig
> index b99da294e9a3..424a04b507a6 100644
> --- a/fs/xfs/Kconfig
> +++ b/fs/xfs/Kconfig
> @@ -106,6 +106,7 @@ config XFS_RT
>  	bool "XFS Realtime subvolume support"
>  	depends on XFS_FS
>  	default BLK_DEV_ZONED
> +	select CRC64
>  	help
>  	  If you say Y here you will be able to mount and use XFS filesystems
>  	  which contain a realtime subvolume.  The realtime subvolume is a
> diff --git a/fs/xfs/Makefile b/fs/xfs/Makefile
> index 79ea4136fbba..0d57bf0701ec 100644
> --- a/fs/xfs/Makefile
> +++ b/fs/xfs/Makefile
> @@ -142,6 +142,7 @@ xfs-$(CONFIG_XFS_QUOTA)		+= xfs_dquot.o \
>  
>  # xfs_rtbitmap is shared with libxfs
>  xfs-$(CONFIG_XFS_RT)		+= xfs_rtalloc.o \
> +				   xfs_rtcsum.o \
>  				   xfs_zone_alloc.o \
>  				   xfs_zone_gc.o \
>  				   xfs_zone_info.o \
> diff --git a/fs/xfs/xfs_inode.h b/fs/xfs/xfs_inode.h
> index 34c1038ebfcd..9ed2fbfe86ff 100644
> --- a/fs/xfs/xfs_inode.h
> +++ b/fs/xfs/xfs_inode.h
> @@ -376,6 +376,11 @@ static inline bool xfs_inode_can_sw_atomic_write(const struct xfs_inode *ip)
>  	return xfs_can_sw_atomic_write(ip->i_mount);
>  }
>  
> +static inline bool xfs_is_rtcsum_inode(const struct xfs_inode *ip)
> +{
> +	return xfs_has_rtcsum(ip->i_mount) && XFS_IS_REALTIME_INODE(ip);
> +}
> +
>  /*
>   * In-core inode flags.
>   */
> diff --git a/fs/xfs/xfs_rtcsum.c b/fs/xfs/xfs_rtcsum.c
> new file mode 100644
> index 000000000000..7cdc5a5029eb
> --- /dev/null
> +++ b/fs/xfs/xfs_rtcsum.c
> @@ -0,0 +1,317 @@
> +// SPDX-License-Identifier: GPL-2.0
> +/*
> + * Copyright (c) 2026 Christoph Hellwig.
> + */
> +#include "xfs_platform.h"
> +#include "xfs_fs.h"
> +#include "xfs_format.h"
> +#include "xfs_log_format.h"
> +#include "xfs_shared.h"
> +#include "xfs_trans_resv.h"
> +#include "xfs_bit.h"
> +#include "xfs_mount.h"
> +#include "xfs_inode.h"
> +#include "xfs_bmap.h"
> +#include "xfs_rtgroup.h"
> +#include "xfs_rtbitmap.h"
> +#include "xfs_bmap_btree.h"
> +#include "xfs_trans.h"
> +#include "xfs_buf_item.h"
> +#include "xfs_trans_space.h"
> +#include "xfs_error.h"
> +#include "xfs_health.h"
> +#include "xfs_rtcsum.h"
> +#include "xfs_zone_priv.h"
> +#include <linux/iomap.h>
> +
> +static_assert(IOMAP_CSUM_MAX_SIZE <= XFS_RTCSUM_MAX_WRITE);
> +
> +static const char *xfs_data_csum_names[XFS_CSUM_TYPE_MAX] = {
> +	[XFS_CSUM_TYPE_CRC32C]	= "crc32c",
> +	[XFS_CSUM_TYPE_CRC64]	= "crc64",
> +};
> +
> +int
> +xfs_csum_verify(
> +	struct xfs_mount	*mp,
> +	struct bio		*bio,
> +	struct bvec_iter	*iter,
> +	void			*csum_buf,
> +	xfs_fsblock_t		bno,
> +	bool			verbose)
> +{
> +	unsigned int		bsize = mp->m_sb.sb_blocksize;
> +	unsigned int		csum_size = 1u << mp->m_rtcsum_shift;
> +	unsigned int		offset = 0;
> +	union xfs_csum		csum;
> +	union xfs_disk_csum	dsum;
> +
> +	do {
> +		struct bio_vec	bv = mp_bvec_iter_bvec(bio->bi_io_vec, *iter);
> +
> +		if (offset == 0)
> +			xfs_csum_seed(mp, &csum);
> +		bv.bv_len = min(bv.bv_len, bsize - offset);
> +		xfs_csum_gen(mp, bvec_virt(&bv), bv.bv_len, &csum);
> +		offset += bv.bv_len;
> +		if (offset == bsize) {
> +			xfs_csum_finalize(mp, &dsum, &csum);
> +			if (unlikely(memcmp(&dsum, csum_buf, csum_size) != 0))
> +				goto mismatch;
> +			bno++;
> +			csum_buf += csum_size;
> +			offset = 0;
> +		}
> +		bio_advance_iter_single(bio, iter, bv.bv_len);
> +	} while (iter->bi_size);
> +
> +	return 0;
> +
> +mismatch:
> +	if (verbose) {
> +		xfs_warn_ratelimited(mp,
> +"data csum mismatch for rtblock 0x%llx: 0x%*phN (expected 0x%*phN)",
> +			bno, csum_size, &csum, csum_size, csum_buf);
> +	}
> +	return -EIO;
> +}
> +
> +void
> +xfs_csum_generate(
> +	struct xfs_mount	*mp,
> +	struct bio		*bio,
> +	void			*csum_buf)
> +{
> +	struct bvec_iter	iter = bio->bi_iter;
> +	unsigned int		bsize = mp->m_sb.sb_blocksize;
> +	unsigned int		csum_size = 1u << mp->m_rtcsum_shift;
> +	unsigned int		offset = 0;
> +	union xfs_csum		csum;
> +
> +	do {
> +		struct bio_vec	bv = mp_bvec_iter_bvec(bio->bi_io_vec, iter);
> +
> +		if (offset == 0)
> +			xfs_csum_seed(mp, &csum);
> +		bv.bv_len = min(bv.bv_len, bsize - offset);
> +		xfs_csum_gen(mp, bvec_virt(&bv), bv.bv_len, &csum);
> +		offset += bv.bv_len;
> +		if (offset == bsize) {
> +			xfs_csum_finalize(mp, csum_buf, &csum);
> +			csum_buf += csum_size;
> +			offset = 0;
> +		}
> +		bio_advance_iter_single(bio, &iter, bv.bv_len);
> +	} while (iter.bi_size);
> +}
> +
> +/*
> + * Read the checksum buffer for @rtg/@csum_off and return it unlocked.

      Read the checksum buffer for @fsbno?

> + *
> + * We don't need to lock read access to the buffer because checksums will not
> + * change until the @rtg is reset.
> + */
> +int
> +xfs_rtcsum_read_async(
> +	struct xfs_mount	*mp,
> +	xfs_rtblock_t		fsbno,
> +	struct xfs_buf		**bpp)
> +{
> +	xfs_daddr_t		csum_daddr;
> +	struct xfs_rtgroup	*rtg;
> +	int			error;
> +
> +	rtg = xfs_rtgroup_get(mp, xfs_rtb_to_rgno(mp, fsbno));
> +	if (!rtg)
> +		return -EFSCORRUPTED;
> +	error = xfs_rtcsum_bmap(rtg, xfs_rtb_to_rgbno(mp, fsbno), &csum_daddr);
> +	if (!error)
> +		error = xfs_buf_read_async(mp->m_ddev_targp, csum_daddr,
> +				BTOBB(mp->m_rtcsum_bsize), &xfs_rtcsum_buf_ops,
> +				bpp);
> +	xfs_rtgroup_put(rtg);
> +	return error;
> +}
> +
> +/*
> + * When opening a zone for writing, do a speculative buf_get for each csum
> + * buffer.  This ensures we usually have a buffer in-memory when we actually
> + * start writing to it.
> + *
> + * Without this we'd have to read the buffer from disk, as we don't know if
> + * anyone has already written to it by the time we get to the buffer due to
> + * completion reordering.
> + *
> + * When reopening a partially written zone at mount time, just read ahead
> + * the entire csums for the zone.
> + */
> +int
> +xfs_rtcsum_open_zone(
> +	struct xfs_open_zone	*oz)
> +{
> +	struct xfs_rtgroup	*rtg = oz->oz_rtg;
> +	struct xfs_mount	*mp = rtg_mount(rtg);
> +	xfs_rgblock_t		rgbno = 0;
> +	int			i, error;
> +
> +	for (i = 0; i < oz->oz_nr_csum_bufs; i++) {
> +		xfs_daddr_t	csum_daddr;
> +		struct xfs_buf	*bp;
> +
> +		error = xfs_rtcsum_bmap(rtg, rgbno, &csum_daddr);
> +		if (error)
> +			goto out_error;
> +
> +		if (oz->oz_allocated) {
> +			error = xfs_buf_read_async(mp->m_ddev_targp, csum_daddr,
> +					BTOBB(mp->m_rtcsum_bsize),
> +					&xfs_rtcsum_buf_ops, &bp);
> +		} else {
> +			error = xfs_buf_get(mp->m_ddev_targp, csum_daddr,
> +					BTOBB(mp->m_rtcsum_bsize), &bp);
> +			if (!error) {
> +				bp->b_flags = XBF_DONE;
> +				xfs_rtfile_initialize_buf(rtg, XFS_RTGI_CSUM,
> +						bp, NULL);
> +			}
> +			xfs_buf_unlock(bp);
> +		}
> +		if (error)
> +			goto out_error;
> +		oz->oz_csum_bufs[i] = bp;
> +		rgbno += (xfs_rtcsum_payload_size(mp) /
> +			  (1u << mp->m_rtcsum_shift));
> +	}
> +
> +	return 0;
> +
> +out_error:
> +	while (--i >= 0)
> +		xfs_buf_rele(oz->oz_csum_bufs[i]);

/me wonders if this should null out oz_csum_bufs to avoid the
possibility of dangling pointers?  Though I think the only caller will
free the oz if this function returns error so it might not matter much.

--D

  reply	other threads:[~2026-09-25 23:20 UTC|newest]

Thread overview: 69+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-24  9:59 support for RT data checksums Christoph Hellwig
2026-09-24  9:59 ` [PATCH 01/21] block: export fs_bio_integrity_verify Christoph Hellwig
2026-09-24 20:29   ` Darrick J. Wong
2026-09-24  9:59 ` [PATCH 02/21] iomap: add support for data checksumming Christoph Hellwig
2026-09-24 21:39   ` Darrick J. Wong
2026-09-25  5:53     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 03/21] xfs: add a xfs_buf_read_async buffer cache API Christoph Hellwig
2026-09-24 21:43   ` Darrick J. Wong
2026-09-25  5:54     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 04/21] xfs: add xfs_daddr_to_rgno and xfs_daddr_to_rgbno helpers Christoph Hellwig
2026-09-24 21:44   ` Darrick J. Wong
2026-09-24  9:59 ` [PATCH 05/21] xfs: introduce XFS_BLI_PREALLOC Christoph Hellwig
2026-09-24 21:49   ` Darrick J. Wong
2026-09-25  5:57     ` Christoph Hellwig
2026-10-08 11:46   ` Anuj gupta
2026-09-24  9:59 ` [PATCH 06/21] xfs: prepare xfs_rtfile_initialize_blocks for larger than FSB blocks Christoph Hellwig
2026-09-24 22:03   ` Darrick J. Wong
2026-09-25  5:58     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 07/21] xfs: relase zi_open_zones_lock over xfs_open_zone_put on unmount Christoph Hellwig
2026-09-24  9:59 ` [PATCH 08/21] xfs: define the RT data checksum on-disk format Christoph Hellwig
2026-09-24 22:13   ` Darrick J. Wong
2026-09-25  0:04     ` Eric Biggers
2026-09-25  6:01     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 09/21] xfs: add support for per-RTG csum files Christoph Hellwig
2026-09-24 22:24   ` Darrick J. Wong
2026-09-25  6:10     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 10/21] xfs: calculate the log reservation for logging data checksum buffers Christoph Hellwig
2026-09-24 22:30   ` Darrick J. Wong
2026-09-25  6:12     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 11/21] xfs: core RT data checksum support Christoph Hellwig
2026-09-25 23:20   ` Darrick J. Wong [this message]
2026-09-26  6:13     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 12/21] xfs: data checksums require stable writes Christoph Hellwig
2026-09-25 23:21   ` Darrick J. Wong
2026-09-24  9:59 ` [PATCH 13/21] xfs: require file system block size alignment when using data checksums Christoph Hellwig
2026-09-25 23:24   ` Darrick J. Wong
2026-09-26  6:15     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 14/21] xfs: add support for reading with " Christoph Hellwig
2026-09-29  0:42   ` Darrick J. Wong
2026-10-05 12:59     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 15/21] xfs: add support for writing " Christoph Hellwig
2026-09-29  1:01   ` Darrick J. Wong
2026-10-05 13:00     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 16/21] xfs: add data checksum support to zoned garbage collection Christoph Hellwig
2026-09-29  1:06   ` Darrick J. Wong
2026-10-05 13:11     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 17/21] xfs: verify data checksums during media verification Christoph Hellwig
2026-09-29  1:19   ` Darrick J. Wong
2026-10-05 13:13     ` Christoph Hellwig
2026-09-24  9:59 ` [PATCH 18/21] xfs: don't try to verify checksums on empty zones Christoph Hellwig
2026-09-29  1:25   ` Darrick J. Wong
2026-10-05 13:14     ` Christoph Hellwig
2026-10-08 11:43   ` Anuj gupta
2026-09-24  9:59 ` [PATCH 19/21] xfs: report RT data checksum information via XFS_FSOP_GEOM Christoph Hellwig
2026-09-29  1:26   ` Darrick J. Wong
2026-09-24  9:59 ` [PATCH 20/21] xfs: add an experimental feature warning for RT data checksums Christoph Hellwig
2026-09-29  1:27   ` Darrick J. Wong
2026-09-24  9:59 ` [PATCH 21/21] xfs: enable " Christoph Hellwig
2026-09-29  1:27   ` Darrick J. Wong
2026-10-05 13:16     ` Christoph Hellwig
2026-09-24 22:52 ` support for " Dave Chinner
2026-09-25  6:27   ` Christoph Hellwig
2026-09-27 22:59     ` Dave Chinner
2026-09-28  5:24       ` Christoph Hellwig
2026-09-29 14:11         ` Dave Chinner
2026-09-30  7:11           ` Dave Chinner
2026-10-05 13:53             ` Christoph Hellwig
2026-10-06  5:31               ` Dave Chinner
2026-10-07 13:46                 ` Christoph Hellwig

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260925232020.GF2705364@frogsfrogsfrogs \
    --to=djwong@kernel.org \
    --cc=axboe@kernel.dk \
    --cc=brauner@kernel.org \
    --cc=cem@kernel.org \
    --cc=hch@lst.de \
    --cc=linux-fsdevel@vger.kernel.org \
    --cc=linux-xfs@vger.kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.