From: Christoph Hellwig <hch@lst.de>
To: Andrey Albershteyn <aalbersh@kernel.org>
Cc: "Darrick J . Wong" <djwong@kernel.org>, linux-xfs@vger.kernel.org
Subject: [PATCH 19/32] xfs: define the RT data checksum on-disk format
Date: Thu, 24 Sep 2026 12:04:09 +0200 [thread overview]
Message-ID: <20260924100512.2733748-20-hch@lst.de> (raw)
In-Reply-To: <20260924100512.2733748-1-hch@lst.de>
Source kernel commit: 5f420198c30c87b4273bc18b1d85ce4f340a1d74
Add the on-disk format for the new RT data checksum format.
Keyed off a new read-only compat feature flag, this adds new fields to
the superblock to indicate the checksum algorithm used and the size of
the blocks containing the checksums. These new fields reuse the
previously reserved padding to make efficient use of the space in the
on-disk superblock.
Data checksums are only supported on zoned RT devices, because they
require out of places writes to safely update the checksums for file
overwrites and a data/metadata split to be able to store the checksums
for a group in a file without causing recursion. This means they can't
be supported directly on the data device at all, and only when using
the always_cow mode on regular RT devices, but that has no benefit
over the zoned allocator which is designed for out of place writes.
The initially supported data checksum algorithms are crc32c and crc64 as
specified by NVMe. Both have extremely fast kernel implementations and
the strong data protection guarantees offered by CRC-style algorithms.
Both also happen to be support by NVMe for protection information so that
the userspace PI passthrough support (once extended to files on file
systems) can be reused to expose the checksums to applications and thus
provide true end-to-end data integrity.
Signed-off-by: Christoph Hellwig <hch@lst.de>
---
libxfs/xfs_format.h | 43 +++++++++++++++++++++--
libxfs/xfs_log_format.h | 1 +
libxfs/xfs_ondisk.h | 4 ++-
libxfs/xfs_sb.c | 75 +++++++++++++++++++++++++++++++++++++++++
libxfs/xfs_sb.h | 1 +
5 files changed, 120 insertions(+), 4 deletions(-)
diff --git a/libxfs/xfs_format.h b/libxfs/xfs_format.h
index dd0ed046fbe9..1be3d21910a7 100644
--- a/libxfs/xfs_format.h
+++ b/libxfs/xfs_format.h
@@ -179,7 +179,9 @@ typedef struct xfs_sb {
xfs_rgnumber_t sb_rgcount; /* number of realtime groups */
xfs_rtxlen_t sb_rgextents; /* size of a realtime group in rtx */
uint8_t sb_rgblklog; /* rt group number shift */
- uint8_t sb_pad[7]; /* zeroes */
+ uint8_t sb_rtcsum_type; /* RT device data checksum type */
+ uint8_t sb_rtcsum_blklog; /* log2 of rtcsum bsize */
+ uint8_t sb_pad[5]; /* zero */
xfs_rfsblock_t sb_rtstart; /* start of internal RT section (FSB) */
xfs_filblks_t sb_rtreserved; /* reserved (zoned) RT blocks */
@@ -272,7 +274,9 @@ struct xfs_dsb {
__be32 sb_rgcount; /* # of realtime groups */
__be32 sb_rgextents; /* size of rtgroup in rtx */
__u8 sb_rgblklog; /* rt group number shift */
- __u8 sb_pad[7]; /* zeroes */
+ __u8 sb_rtcsum_type; /* RT device data checksum type */
+ __u8 sb_rtcsum_blklog; /* log2 of rtcsum bsize */
+ __u8 sb_pad[5]; /* zero */
__be64 sb_rtstart; /* start of internal RT section (FSB) */
__be64 sb_rtreserved; /* reserved (zoned) RT blocks */
@@ -374,6 +378,8 @@ xfs_sb_has_compat_feature(
#define XFS_SB_FEAT_RO_COMPAT_RMAPBT (1 << 1) /* reverse map btree */
#define XFS_SB_FEAT_RO_COMPAT_REFLINK (1 << 2) /* reflinked files */
#define XFS_SB_FEAT_RO_COMPAT_INOBTCNT (1 << 3) /* inobt block counts */
+#define XFS_SB_FEAT_RO_COMPAT_RTCSUM (1 << 5) /* RT data checksums */
+
#define XFS_SB_FEAT_RO_COMPAT_ALL \
(XFS_SB_FEAT_RO_COMPAT_FINOBT | \
XFS_SB_FEAT_RO_COMPAT_RMAPBT | \
@@ -866,6 +872,7 @@ enum xfs_metafile_type {
XFS_METAFILE_RTSUMMARY, /* rt summary */
XFS_METAFILE_RTRMAP, /* rt rmap */
XFS_METAFILE_RTREFCOUNT, /* rt refcount */
+ XFS_METAFILE_RTCSUM, /* rt data checksums */
XFS_METAFILE_MAX
} __packed;
@@ -879,7 +886,8 @@ enum xfs_metafile_type {
{ XFS_METAFILE_RTBITMAP, "rtbitmap" }, \
{ XFS_METAFILE_RTSUMMARY, "rtsummary" }, \
{ XFS_METAFILE_RTRMAP, "rtrmap" }, \
- { XFS_METAFILE_RTREFCOUNT, "rtrefcount" }
+ { XFS_METAFILE_RTREFCOUNT, "rtrefcount" }, \
+ { XFS_METAFILE_RTCSUM, "rtcsum", }
/*
* On-disk inode structure.
@@ -1318,6 +1326,7 @@ static inline bool xfs_dinode_is_metadir(const struct xfs_dinode *dip)
*/
#define XFS_RTBITMAP_MAGIC 0x424D505A /* BMPZ */
#define XFS_RTSUMMARY_MAGIC 0x53554D59 /* SUMY */
+#define XFS_RTCSUM_MAGIC 0x4353554D /* CSUM */
struct xfs_rtbuf_blkinfo {
__be32 rt_magic; /* validity check on block */
@@ -2027,4 +2036,32 @@ struct xfs_acl {
#define SGI_ACL_FILE_SIZE (sizeof(SGI_ACL_FILE)-1)
#define SGI_ACL_DEFAULT_SIZE (sizeof(SGI_ACL_DEFAULT)-1)
+/*
+ * Size of a RT data checksum block. Data reads must be contained in a single
+ * block, so this should be fairly large.
+ *
+ * The default is 32k, matching the default inode cluster size and the maximum
+ * memory allocation the Linux MM can handle in the fast path. 64k is primarily
+ * there so that his value never needs to be below the FSB size, even for 64k
+ * blocks.
+ */
+#define XFS_RTCSUM_BSIZE_LOG_MIN 15
+#define XFS_RTCSUM_BSIZE_LOG_MAX 16
+
+/*
+ * Data checksum types.
+ */
+#define XFS_CSUM_TYPE_NONE 0u
+#define XFS_CSUM_TYPE_CRC32C 1u
+#define XFS_CSUM_TYPE_CRC64 2u
+#define XFS_CSUM_TYPE_MAX 3u
+
+/*
+ * On-disk data checksums.
+ */
+union xfs_disk_csum {
+ __le32 crc32c;
+ __le64 crc64;
+};
+
#endif /* __XFS_FORMAT_H__ */
diff --git a/libxfs/xfs_log_format.h b/libxfs/xfs_log_format.h
index a4e1b3eb425c..b1037b77338b 100644
--- a/libxfs/xfs_log_format.h
+++ b/libxfs/xfs_log_format.h
@@ -581,6 +581,7 @@ enum xfs_blft {
XFS_BLFT_SB_BUF,
XFS_BLFT_RTBITMAP_BUF,
XFS_BLFT_RTSUMMARY_BUF,
+ XFS_BLFT_RTCSUM_BUF,
XFS_BLFT_MAX_BUF = (1 << XFS_BLFT_BITS),
};
diff --git a/libxfs/xfs_ondisk.h b/libxfs/xfs_ondisk.h
index 23cde1248f01..17ab9366b3b9 100644
--- a/libxfs/xfs_ondisk.h
+++ b/libxfs/xfs_ondisk.h
@@ -284,7 +284,9 @@ xfs_check_ondisk_structs(void)
XFS_CHECK_SB_OFFSET(sb_rgcount, 272);
XFS_CHECK_SB_OFFSET(sb_rgextents, 276);
XFS_CHECK_SB_OFFSET(sb_rgblklog, 280);
- XFS_CHECK_SB_OFFSET(sb_pad, 281);
+ XFS_CHECK_SB_OFFSET(sb_rtcsum_type, 281);
+ XFS_CHECK_SB_OFFSET(sb_rtcsum_blklog, 282);
+ XFS_CHECK_SB_OFFSET(sb_pad, 283);
XFS_CHECK_SB_OFFSET(sb_rtstart, 288);
XFS_CHECK_SB_OFFSET(sb_rtreserved, 296);
diff --git a/libxfs/xfs_sb.c b/libxfs/xfs_sb.c
index cccbcd153316..2d64412363b4 100644
--- a/libxfs/xfs_sb.c
+++ b/libxfs/xfs_sb.c
@@ -484,6 +484,40 @@ xfs_validate_sb_zoned(
return 0;
}
+static int
+xfs_validate_sb_csum(
+ struct xfs_mount *mp,
+ struct xfs_sb *sbp)
+{
+ unsigned int rtcsum_bsize = 1u << sbp->sb_rtcsum_blklog;
+
+ if (!(sbp->sb_features_incompat & XFS_SB_FEAT_INCOMPAT_ZONED)) {
+ xfs_warn(mp, "data checksum required the zone allocator");
+ return -EINVAL;
+ }
+ if (sbp->sb_rtcsum_type >= XFS_CSUM_TYPE_MAX) {
+ xfs_warn(mp, "invalid data checksum type: 0x%x",
+ sbp->sb_rtcsum_type);
+ return -EINVAL;
+ }
+ if (sbp->sb_rtcsum_blklog < XFS_RTCSUM_BSIZE_LOG_MIN ||
+ sbp->sb_rtcsum_blklog > XFS_RTCSUM_BSIZE_LOG_MAX) {
+ xfs_warn(mp,
+"invalid data checksum block log: %u (min %u/max %u)",
+ sbp->sb_rtcsum_blklog,
+ XFS_RTCSUM_BSIZE_LOG_MIN,
+ XFS_RTCSUM_BSIZE_LOG_MAX);
+ return -EINVAL;
+ }
+ if (rtcsum_bsize < sbp->sb_blocksize) {
+ xfs_warn(mp,
+"checksum block size must not be smaller than file system block size: %u/%u",
+ rtcsum_bsize, sbp->sb_blocksize);
+ return -EINVAL;
+ }
+ return 0;
+}
+
/* Check the validity of the SB. */
STATIC int
xfs_validate_sb_common(
@@ -577,6 +611,17 @@ xfs_validate_sb_common(
if (error)
return error;
}
+ if (sbp->sb_features_ro_compat & XFS_SB_FEAT_RO_COMPAT_RTCSUM) {
+ error = xfs_validate_sb_csum(mp, sbp);
+ if (error)
+ return error;
+ } else {
+ if (sbp->sb_rtcsum_type || sbp->sb_rtcsum_blklog) {
+ xfs_warn(mp,
+"rtcsum superblock fields must be zero for non-RTCSUM file systems.");
+ return -EINVAL;
+ }
+ }
} else if (sbp->sb_qflags & (XFS_PQUOTA_ENFD | XFS_GQUOTA_ENFD |
XFS_PQUOTA_CHKD | XFS_GQUOTA_CHKD)) {
xfs_notice(mp,
@@ -897,6 +942,14 @@ __xfs_sb_from_disk(
to->sb_rtstart = 0;
to->sb_rtreserved = 0;
}
+
+ if (to->sb_features_ro_compat & XFS_SB_FEAT_RO_COMPAT_RTCSUM) {
+ to->sb_rtcsum_type = from->sb_rtcsum_type;
+ to->sb_rtcsum_blklog = from->sb_rtcsum_blklog;
+ } else {
+ to->sb_rtcsum_type = XFS_CSUM_TYPE_NONE;
+ to->sb_rtcsum_blklog = 0;
+ }
}
void
@@ -1068,6 +1121,11 @@ xfs_sb_to_disk(
to->sb_rtstart = cpu_to_be64(from->sb_rtstart);
to->sb_rtreserved = cpu_to_be64(from->sb_rtreserved);
}
+
+ if (from->sb_features_ro_compat & XFS_SB_FEAT_RO_COMPAT_RTCSUM) {
+ to->sb_rtcsum_type = from->sb_rtcsum_type;
+ to->sb_rtcsum_blklog = from->sb_rtcsum_blklog;
+ }
}
/*
@@ -1240,6 +1298,20 @@ xfs_mount_sb_set_rextsize(
xfs_sb_mount_rextsize(mp, sbp);
}
+uint8_t
+xfs_data_csum_shift(
+ uint8_t csum)
+{
+ switch (csum) {
+ case XFS_CSUM_TYPE_CRC32C:
+ return 2;
+ case XFS_CSUM_TYPE_CRC64:
+ return 3;
+ default:
+ return 0;
+ }
+}
+
/*
* xfs_mount_common
*
@@ -1308,6 +1380,9 @@ xfs_sb_mount_common(
mp->m_bsize = XFS_FSB_TO_BB(mp, 1);
mp->m_alloc_set_aside = xfs_alloc_set_aside(mp);
mp->m_ag_max_usable = xfs_alloc_ag_max_usable(mp);
+
+ mp->m_rtcsum_shift = xfs_data_csum_shift(mp->m_sb.sb_rtcsum_type);
+ mp->m_rtcsum_bsize = 1u << mp->m_sb.sb_rtcsum_blklog;
}
/*
diff --git a/libxfs/xfs_sb.h b/libxfs/xfs_sb.h
index 34d0dd374e9b..16f300c12e37 100644
--- a/libxfs/xfs_sb.h
+++ b/libxfs/xfs_sb.h
@@ -20,6 +20,7 @@ extern void xfs_sb_mount_common(struct xfs_mount *mp, struct xfs_sb *sbp);
void xfs_sb_mount_rextsize(struct xfs_mount *mp, struct xfs_sb *sbp);
void xfs_mount_sb_set_rextsize(struct xfs_mount *mp,
struct xfs_sb *sbp, xfs_agblock_t rextsize);
+uint8_t xfs_data_csum_shift(uint8_t csum);
extern void xfs_sb_from_disk(struct xfs_sb *to, struct xfs_dsb *from);
extern void xfs_sb_to_disk(struct xfs_dsb *to, struct xfs_sb *from);
extern void xfs_sb_quota_from_disk(struct xfs_sb *sbp);
--
2.53.0
next prev parent reply other threads:[~2026-09-24 10:06 UTC|newest]
Thread overview: 57+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-24 10:03 xfsprogs support for RT data checksums Christoph Hellwig
2026-09-24 10:03 ` [PATCH 01/32] man: fix alignment of the rtstart field in ioctl_xfs_fsgeometry.2 Christoph Hellwig
2026-09-24 20:30 ` Darrick J. Wong
2026-09-26 6:15 ` Christoph Hellwig
2026-09-29 10:14 ` Andrey Albershteyn
2026-09-24 10:03 ` [PATCH 02/32] add cpu_to_le64 and le64_to_cpu_helpers Christoph Hellwig
2026-09-24 20:30 ` Darrick J. Wong
2026-09-24 10:03 ` [PATCH 03/32] libfrog: add a crc64_nvme implementation Christoph Hellwig
2026-09-24 10:03 ` [PATCH 04/32] libxfs: add DIV_ROUND_UP_ULL Christoph Hellwig
2026-09-24 20:44 ` Darrick J. Wong
2026-09-25 6:13 ` Christoph Hellwig
2026-09-24 10:03 ` [PATCH 05/32] libxfs: remove a spurious xfs_rtbitmap.h include in xfs_sb.c Christoph Hellwig
2026-09-25 23:27 ` Darrick J. Wong
2026-09-24 10:03 ` [PATCH 06/32] libxfs: add SZ_* constants Christoph Hellwig
2026-09-25 23:27 ` Darrick J. Wong
2026-09-24 10:03 ` [PATCH 07/32] xfs: remove spurious XBF_DONE clearing on readahead validation failure Christoph Hellwig
2026-09-24 10:03 ` [PATCH 08/32] xfs: hide b_flags manipulation from code outside of xfs_buf.c Christoph Hellwig
2026-09-24 10:03 ` [PATCH 09/32] FIXUP Christoph Hellwig
2026-09-25 23:29 ` Darrick J. Wong
2026-09-26 6:17 ` Christoph Hellwig
2026-09-24 10:04 ` [PATCH 10/32] xfs: add error injection for lazy bounce buffering Christoph Hellwig
2026-09-24 10:04 ` [PATCH 11/32] xfs: add xfs_daddr_to_rgno and xfs_daddr_to_rgbno helpers Christoph Hellwig
2026-09-24 10:04 ` [PATCH 12/32] FIXUP Christoph Hellwig
2026-09-24 10:04 ` [PATCH 13/32] xfs: introduce XFS_BLI_PREALLOC Christoph Hellwig
2026-09-24 10:04 ` [PATCH 14/32] xfs: centralize setting of buf_ops/buf_type/magic for rtblocks Christoph Hellwig
2026-09-24 10:04 ` [PATCH 15/32] xfs: factor out a xfs_rtfile_initialize_buf helper Christoph Hellwig
2026-09-24 10:04 ` [PATCH 16/32] xfs: add a xfs_rtblock_payload helper Christoph Hellwig
2026-09-24 10:04 ` [PATCH 17/32] xfs: prepare xfs_rtfile_initialize_blocks for larger than FSB blocks Christoph Hellwig
2026-09-24 10:04 ` [PATCH 18/32] FIXUP Christoph Hellwig
2026-09-24 10:04 ` Christoph Hellwig [this message]
2026-09-24 10:04 ` [PATCH 20/32] FIXUP Christoph Hellwig
2026-09-24 10:04 ` [PATCH 21/32] xfs: add support for per-RTG csum files Christoph Hellwig
2026-09-24 10:04 ` [PATCH 22/32] FIXUP Christoph Hellwig
2026-09-24 10:04 ` [PATCH 23/32] xfs: calculate the log reservation for logging data checksum buffers Christoph Hellwig
2026-09-24 10:04 ` [PATCH 24/32] xfs: report RT data checksum information via XFS_FSOP_GEOM Christoph Hellwig
2026-09-24 10:04 ` [PATCH 25/32] xfs: enable RT data checksums Christoph Hellwig
2026-09-24 10:04 ` [PATCH 26/32] man: document the rtcsum geom fields Christoph Hellwig
2026-09-25 23:31 ` Darrick J. Wong
2026-09-26 6:17 ` Christoph Hellwig
2026-09-24 10:04 ` [PATCH 27/32] libfrog: print csum geometry information Christoph Hellwig
2026-09-25 23:32 ` Darrick J. Wong
2026-09-24 10:04 ` [PATCH 28/32] xfs_io: report checksum information from fs geometry in statfs Christoph Hellwig
2026-09-25 23:33 ` Darrick J. Wong
2026-09-26 6:18 ` Christoph Hellwig
2026-09-24 10:04 ` [PATCH 29/32] mkfs: support RT data checksums Christoph Hellwig
2026-09-25 23:39 ` Darrick J. Wong
2026-09-26 6:19 ` Christoph Hellwig
2026-09-24 10:04 ` [PATCH 30/32] xfs_db: support RT data checksum Christoph Hellwig
2026-09-25 23:43 ` Darrick J. Wong
2026-09-26 6:21 ` Christoph Hellwig
2026-09-26 18:20 ` Darrick J. Wong
2026-09-24 10:04 ` [PATCH 31/32] repair: support RT data checksums Christoph Hellwig
2026-09-25 23:54 ` Darrick J. Wong
2026-09-26 6:23 ` Christoph Hellwig
2026-09-27 0:01 ` Darrick J. Wong
2026-09-24 10:04 ` [PATCH 32/32] xfs_scrub: don't merge over unused space when data checksums are enabled Christoph Hellwig
2026-09-25 23:51 ` Darrick J. Wong
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260924100512.2733748-20-hch@lst.de \
--to=hch@lst.de \
--cc=aalbersh@kernel.org \
--cc=djwong@kernel.org \
--cc=linux-xfs@vger.kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox