Linux Btrfs filesystem development
 help / color / mirror / Atom feed
* [PATCH v1] btrfs: add LOGICAL_INO_V2 flag for avoiding the tree modification log
@ 2026-09-22 20:12 Joanne Koong
  2026-09-22 21:41 ` Qu Wenruo
  2026-09-23  2:28 ` Zygo Blaxell
  0 siblings, 2 replies; 3+ messages in thread
From: Joanne Koong @ 2026-09-22 20:12 UTC (permalink / raw)
  To: dsterba; +Cc: boris, loemra.dev, fdmanana, wqu, linux-btrfs

At Meta, we use the LOGICAL_INO_V2 ioctl to estimate sharing-aware space
usage. We sample the logical address space and resolve each sample back
to the inodes which reference it. The measurement is statistical, so
with enough samples it gives good bounds on real usage.

Currently in the LOGICAL_INO_V2 ioctl path, iterate_extent_inodes()
(which can walk backrefs either against the current trees or against the
commit roots) acquires a tree modification log sequence number when
walking the current trees, since those trees are being modified while
the walk is in progress. The log is then enabled filesystem wide for as
long as any such walk is running, and is only disabled again once the
last walk has finished.

While it is enabled the cost impacts the whole filesystem, not only the
caller. Modifications of interior nodes of the extent tree and of
subvolume trees have to be recorded, serialized on a single filesystem
wide lock. Delayed refs with a sequence number at or above the oldest
active one can not be run (the run path bails out with -EAGAIN in
btrfs_run_delayed_refs_for_head()), and metadata refs above it can not
be merged, so they accumulate and that work is deferred to transaction
commit.

Currently, userspace has no way to ask for anything cheaper. Whenever a
transaction is running, the walk uses the tree modification log, even if
the caller would tolerate results that could be up to one commit interval
stale.

Give userspace this option by adding BTRFS_LOGICAL_INO_ARGS_COMMIT_ROOT,
which makes the backref walk use the commit roots and therefore never
enables the tree modification log. Behavior is unchanged unless the flag
gets explicitly passed by userspace. iterate_extent_inodes() already
takes a search_commit_root argument for this, which scrub, send and the
data relocation warning path all pass as true. The flag simply plumbs it
through to userspace.

Assisted-by: LLM
Signed-off-by: Joanne Koong <joannelkoong@gmail.com>
---
 fs/btrfs/backref.c         |  6 ++++--
 fs/btrfs/backref.h         |  3 ++-
 fs/btrfs/ioctl.c           |  9 +++++++--
 include/uapi/linux/btrfs.h | 10 ++++++++++
 4 files changed, 23 insertions(+), 5 deletions(-)

diff --git a/fs/btrfs/backref.c b/fs/btrfs/backref.c
index 1be632c742bd..364ec3ad7d8a 100644
--- a/fs/btrfs/backref.c
+++ b/fs/btrfs/backref.c
@@ -2549,7 +2549,8 @@ static int build_ino_list(u64 inum, u64 offset, u64 num_bytes, u64 root, void *c
 }
 
 int iterate_inodes_from_logical(u64 logical, struct btrfs_fs_info *fs_info,
-				void *ctx, bool ignore_offset)
+				void *ctx, bool ignore_offset,
+				bool search_commit_root)
 {
 	struct btrfs_backref_walk_ctx walk_ctx = { 0 };
 	int ret;
@@ -2575,7 +2576,8 @@ int iterate_inodes_from_logical(u64 logical, struct btrfs_fs_info *fs_info,
 		walk_ctx.extent_item_pos = logical - found_key.objectid;
 	walk_ctx.fs_info = fs_info;
 
-	return iterate_extent_inodes(&walk_ctx, false, build_ino_list, ctx);
+	return iterate_extent_inodes(&walk_ctx, search_commit_root,
+				     build_ino_list, ctx);
 }
 
 static int inode_to_path(u64 inum, u32 name_len, unsigned long name_off,
diff --git a/fs/btrfs/backref.h b/fs/btrfs/backref.h
index 179791de6b19..f8dc8a3723cf 100644
--- a/fs/btrfs/backref.h
+++ b/fs/btrfs/backref.h
@@ -226,7 +226,8 @@ int iterate_extent_inodes(struct btrfs_backref_walk_ctx *ctx,
 			  iterate_extent_inodes_t *iterate, void *user_ctx);
 
 int iterate_inodes_from_logical(u64 logical, struct btrfs_fs_info *fs_info,
-				void *ctx, bool ignore_offset);
+				void *ctx, bool ignore_offset,
+				bool search_commit_root);
 
 int paths_from_inode(u64 inum, struct inode_fs_paths *ipath);
 
diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c
index 52aab510aea0..63257f8fb897 100644
--- a/fs/btrfs/ioctl.c
+++ b/fs/btrfs/ioctl.c
@@ -3286,6 +3286,7 @@ static long btrfs_ioctl_logical_to_ino(struct btrfs_fs_info *fs_info,
 	struct btrfs_ioctl_logical_ino_args AUTO_KFREE(loi);
 	struct btrfs_data_container AUTO_KVFREE(inodes);
 	bool ignore_offset;
+	bool search_commit_root;
 
 	if (!capable(CAP_SYS_ADMIN))
 		return -EPERM;
@@ -3296,6 +3297,7 @@ static long btrfs_ioctl_logical_to_ino(struct btrfs_fs_info *fs_info,
 
 	if (version == 1) {
 		ignore_offset = false;
+		search_commit_root = false;
 		size = min_t(u32, loi->size, SZ_64K);
 	} else {
 		/* All reserved bits must be 0 for now */
@@ -3303,10 +3305,12 @@ static long btrfs_ioctl_logical_to_ino(struct btrfs_fs_info *fs_info,
 			return -EINVAL;
 
 		/* Only accept flags we have defined so far */
-		if (loi->flags & ~(BTRFS_LOGICAL_INO_ARGS_IGNORE_OFFSET))
+		if (loi->flags & ~(BTRFS_LOGICAL_INO_ARGS_IGNORE_OFFSET |
+				   BTRFS_LOGICAL_INO_ARGS_COMMIT_ROOT))
 			return -EINVAL;
 
 		ignore_offset = loi->flags & BTRFS_LOGICAL_INO_ARGS_IGNORE_OFFSET;
+		search_commit_root = loi->flags & BTRFS_LOGICAL_INO_ARGS_COMMIT_ROOT;
 		size = min_t(u32, loi->size, SZ_16M);
 	}
 
@@ -3314,7 +3318,8 @@ static long btrfs_ioctl_logical_to_ino(struct btrfs_fs_info *fs_info,
 	if (IS_ERR(inodes))
 		return PTR_ERR(inodes);
 
-	ret = iterate_inodes_from_logical(loi->logical, fs_info, inodes, ignore_offset);
+	ret = iterate_inodes_from_logical(loi->logical, fs_info, inodes,
+					  ignore_offset, search_commit_root);
 	if (ret == -EINVAL)
 		return -ENOENT;
 	if (ret < 0)
diff --git a/include/uapi/linux/btrfs.h b/include/uapi/linux/btrfs.h
index 0a13baf3d8d1..b6f196bfa600 100644
--- a/include/uapi/linux/btrfs.h
+++ b/include/uapi/linux/btrfs.h
@@ -732,6 +732,16 @@ struct btrfs_ioctl_logical_ino_args {
  */
 #define BTRFS_LOGICAL_INO_ARGS_IGNORE_OFFSET	(1ULL << 0)
 
+/*
+ * Resolve backrefs against the commit roots instead of the current trees.
+ *
+ * Resolving against the current trees is more expensive for the filesystem as
+ * a whole, not only for the caller. Callers that do not need to observe the
+ * currently running transaction should set this. The result can be up to one
+ * commit interval stale.
+ */
+#define BTRFS_LOGICAL_INO_ARGS_COMMIT_ROOT	(1ULL << 1)
+
 enum btrfs_dev_stat_values {
 	/* disk I/O failure stats */
 	BTRFS_DEV_STAT_WRITE_ERRS, /* EIO or EREMOTEIO from lower layers */
-- 
2.52.0


^ permalink raw reply related	[flat|nested] 3+ messages in thread

end of thread, other threads:[~2026-09-23  2:36 UTC | newest]

Thread overview: 3+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-22 20:12 [PATCH v1] btrfs: add LOGICAL_INO_V2 flag for avoiding the tree modification log Joanne Koong
2026-09-22 21:41 ` Qu Wenruo
2026-09-23  2:28 ` Zygo Blaxell

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox