linux-xfs.vger.kernel.org archive mirror
 help / color / mirror / Atom feed
* [PATCH] xfs: don't limit software atomic writes by the group alignment
@ 2026-09-25 10:36 Pankaj Raghav
  2026-09-25 11:44 ` John Garry
                   ` (2 more replies)
  0 siblings, 3 replies; 19+ messages in thread
From: Pankaj Raghav @ 2026-09-25 10:36 UTC (permalink / raw)
  To: cem, linux-xfs
  Cc: John Garry, Darrick J . Wong, p.raghav, gost.dev, pankaj.raghav

xfs_calc_group_awu_max() clamps the atomic write unit maximum to the
greatest power-of-two factor of the group size when the device advertises
atomic writes, so that allocations can be made naturally aligned for
REQ_ATOMIC. But that value is the software limit: it is only used when
reflink is enabled, and out of place writes through the COW fork have no
alignment requirement.

mkfs caps agsize one block below 1T, so any filesystem with maximum sized
AGs has an odd agsize. max_pow_of_two_factor() will return 1 for odd
agsize.

A filesystem on a > 4TB device that advertises 16k atomic writes then reports
an atomic write unit maximum of a single fsblock and an optimal maximum of 0,
and rejects any larger RWF_ATOMIC write. That is worse than the same filesystem
on a device with no atomic write support at all.

Compute the software limit from the group size alone, and apply the
alignment constraint in xfs_get_atomic_write_max_opt() instead, which is
what reports the size that can be offloaded to the hardware.

Results on a 8TB device with 16k hardware atomic support with 4k
blocksize:

Before patches:
/media/test/hello.txt:
  stx_atomic_write_unit_min:            4096
  stx_atomic_write_unit_max:            4096
  stx_atomic_write_unit_max_opt:        0
  stx_atomic_write_segments_max:        1

After patches:
/media/test/hello.txt:
  stx_atomic_write_unit_min:            4096
  stx_atomic_write_unit_max:            2097152
  stx_atomic_write_unit_max_opt:        4096
  stx_atomic_write_segments_max:        1

Results on a 8TB device with 16k hardware atomic support with 16k
blocksize:

Before patches:
/media/test/hello.txt:
  stx_atomic_write_unit_min:            16384
  stx_atomic_write_unit_max:            16384
  stx_atomic_write_unit_max_opt:        0
  stx_atomic_write_segments_max:        1

After patches:
/media/test/hello.txt:
  stx_atomic_write_unit_min:            16384
  stx_atomic_write_unit_max:            33554432
  stx_atomic_write_unit_max_opt:        16384
  stx_atomic_write_segments_max:        1

Fixes: 0c438dcc3150 ("xfs: add xfs_calc_atomic_write_unit_max()")
Assisted-by: LLM
Signed-off-by: Pankaj Raghav <p.raghav@samsung.com>
---
 fs/xfs/xfs_iops.c  | 27 ++++++++++++++++++++++-----
 fs/xfs/xfs_mount.c | 35 ++++++++++++++++++++++++-----------
 fs/xfs/xfs_mount.h |  2 ++
 3 files changed, 48 insertions(+), 16 deletions(-)

diff --git a/fs/xfs/xfs_iops.c b/fs/xfs/xfs_iops.c
index d1306e723899..8c8b14f94ede 100644
--- a/fs/xfs/xfs_iops.c
+++ b/fs/xfs/xfs_iops.c
@@ -620,6 +620,12 @@ xfs_get_atomic_write_min(
 	return 0;
 }
 
+static inline enum xfs_group_type
+xfs_inode_group_type(struct xfs_inode *ip)
+{
+	return XFS_IS_REALTIME_INODE(ip) ? XG_TYPE_RTG : XG_TYPE_AG;
+}
+
 unsigned int
 xfs_get_atomic_write_max(
 	struct xfs_inode	*ip)
@@ -642,19 +648,20 @@ xfs_get_atomic_write_max(
 	 * then advertise a maximum size of whatever we can complete through
 	 * that means.  Hardware support is reported via max_opt, not here.
 	 */
-	if (XFS_IS_REALTIME_INODE(ip))
-		return XFS_FSB_TO_B(mp, mp->m_groups[XG_TYPE_RTG].awu_max);
-	return XFS_FSB_TO_B(mp, mp->m_groups[XG_TYPE_AG].awu_max);
+	return XFS_FSB_TO_B(mp, mp->m_groups[xfs_inode_group_type(ip)].awu_max);
 }
 
 unsigned int
 xfs_get_atomic_write_max_opt(
 	struct xfs_inode	*ip)
 {
+	struct xfs_mount	*mp = ip->i_mount;
 	unsigned int		awu_max = xfs_get_atomic_write_max(ip);
+	xfs_extlen_t		align_max_fsb;
+	unsigned int		opt;
 
 	/* if the max is 1x block, then just keep behaviour that opt is 0 */
-	if (awu_max <= ip->i_mount->m_sb.sb_blocksize)
+	if (awu_max <= mp->m_sb.sb_blocksize)
 		return 0;
 
 	/*
@@ -663,7 +670,17 @@ xfs_get_atomic_write_max_opt(
 	 * less than our out of place write limit, but we don't want to exceed
 	 * the awu_max.
 	 */
-	return min(awu_max, xfs_inode_buftarg(ip)->bt_awu_max);
+	opt = min(awu_max, xfs_inode_buftarg(ip)->bt_awu_max);
+
+	/*
+	 * REQ_ATOMIC writes also have to be naturally aligned on disk, so we
+	 * cannot promise more than the largest extent that the allocator is
+	 * able to align within a group.
+	 */
+	align_max_fsb = xfs_calc_group_awu_align_max(mp,
+						xfs_inode_group_type(ip));
+
+	return min_t(xfs_fsize_t, opt, XFS_FSB_TO_B(mp, align_max_fsb));
 }
 
 static void
diff --git a/fs/xfs/xfs_mount.c b/fs/xfs/xfs_mount.c
index be90c7b03994..b7110b3d023b 100644
--- a/fs/xfs/xfs_mount.c
+++ b/fs/xfs/xfs_mount.c
@@ -674,15 +674,13 @@ static inline xfs_extlen_t xfs_calc_atomic_write_max(struct xfs_mount *mp)
 }
 
 /*
- * If the underlying device advertises atomic write support, limit the size of
- * atomic writes to the greatest power-of-two factor of the group size so
- * that every atomic write unit aligns with the start of every group.  This is
- * required so that the allocations for an atomic write will always be
- * aligned compatibly with the alignment requirements of the storage.
+ * The largest out of place write is the number of blocks that user files can
+ * allocate from any group.
  *
- * If the device doesn't advertise atomic writes, then there are no alignment
- * restrictions and the largest out-of-place write we can do ourselves is the
- * number of blocks that user files can allocate from any group.
+ * This is a software limit, so it deliberately does not take the alignment
+ * requirements of the storage into account. An atomic write that cannot be
+ * handed to the device as a single naturally aligned REQ_ATOMIC bio is
+ * completed through the COW fork instead, which has no such requirement.
  */
 static xfs_extlen_t
 xfs_calc_group_awu_max(
@@ -690,15 +688,30 @@ xfs_calc_group_awu_max(
 	enum xfs_group_type	type)
 {
 	struct xfs_groups	*g = &mp->m_groups[type];
-	struct xfs_buftarg	*btp = xfs_group_type_buftarg(mp, type);
 
 	if (g->blocks == 0)
 		return 0;
-	if (btp && btp->bt_awu_min > 0)
-		return max_pow_of_two_factor(g->blocks);
 	return rounddown_pow_of_two(g->blocks);
 }
 
+/*
+ * Compute the largest atomic write unit for which the allocator can guarantee
+ * naturally aligned extents in this group type.
+ *
+ * Hardware atomic writes have to be naturally aligned on disk.
+ */
+xfs_extlen_t
+xfs_calc_group_awu_align_max(
+	struct xfs_mount	*mp,
+	enum xfs_group_type	type)
+{
+	struct xfs_groups	*g = &mp->m_groups[type];
+
+	if (g->blocks == 0)
+		return 0;
+	return max_pow_of_two_factor(g->blocks);
+}
+
 /* Compute the maximum atomic write unit size for each section. */
 static inline void
 xfs_calc_atomic_write_unit_max(
diff --git a/fs/xfs/xfs_mount.h b/fs/xfs/xfs_mount.h
index 216a38a354e7..a2894c18e3c1 100644
--- a/fs/xfs/xfs_mount.h
+++ b/fs/xfs/xfs_mount.h
@@ -812,6 +812,8 @@ static inline void xfs_mod_sb_delalloc(struct xfs_mount *mp, int64_t delta)
 
 int xfs_set_max_atomic_write_opt(struct xfs_mount *mp,
 		unsigned long long new_max_bytes);
+xfs_extlen_t xfs_calc_group_awu_align_max(struct xfs_mount *mp,
+		enum xfs_group_type type);
 
 static inline struct xfs_buftarg *
 xfs_group_type_buftarg(

base-commit: 1282269a5ddb044481e7f8fd43b3195e211d5475
-- 
2.51.2


^ permalink raw reply related	[flat|nested] 19+ messages in thread

end of thread, other threads:[~2026-10-07 17:43 UTC | newest]

Thread overview: 19+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-25 10:36 [PATCH] xfs: don't limit software atomic writes by the group alignment Pankaj Raghav
2026-09-25 11:44 ` John Garry
2026-09-29  7:17   ` Pankaj Raghav
2026-09-29 14:47     ` John Garry
2026-09-30  5:10       ` Pankaj Raghav (Samsung)
2026-09-30  8:42         ` John Garry
2026-09-30 12:27           ` Pankaj Raghav
2026-09-30 12:48             ` John Garry
2026-10-01 16:30               ` Pankaj Raghav
2026-10-02  8:45                 ` John Garry
2026-10-02 11:24                   ` Pankaj Raghav
2026-09-29 16:40     ` Darrick J. Wong
2026-09-29 20:42       ` Darrick J. Wong
2026-09-30  8:48         ` Pankaj Raghav (Samsung)
2026-10-07 17:43           ` Darrick J. Wong
2026-10-05 11:02 ` John Garry
2026-10-05 11:02 ` John Garry
2026-10-05 16:24   ` Pankaj Raghav (Samsung)
2026-10-06  7:06     ` John Garry

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox;
as well as URLs for NNTP newsgroup(s).