From: Qu Wenruo <wqu@suse.com>
To: linux-btrfs@vger.kernel.org
Subject: [PATCH v4 RESEND 3/6] btrfs: introduce the skeleton of delayed bbio endio function
Date: Fri, 18 Sep 2026 10:14:02 +0930 [thread overview]
Message-ID: <1de09905a5581ab755d30f64fb8055aaa1c90b60.1789692182.git.wqu@suse.com> (raw)
In-Reply-To: <cover.1789692182.git.wqu@suse.com>
A delayed bbio will not be directly submitted, but queued into a
workqueue, to perform compression there.
There is also a new workqueue, delayed_write_workers, for this workload.
It will eventually replace the existing delallow_workers.
The compression and uncompressed fallback are not implemented in this
patch.
Only the main endio function and a helper to queue workload into a
workqueue are implemented.
The endio function is mostly the same as end_bbio_data_write(), except
for the extra memory allocation/freeing for the bbio->private.
Another point to note is that, we reuse btrfs_bio::pending_ios to track
the lifespan of the parent delayed bbio.
We can reuse it because the delayed bbio is never submitted for IO, thus
it will never be split by btrfs, so we can reuse that atomic to avoid
extra space inside delayed_bio_private.
Signed-off-by: Qu Wenruo <wqu@suse.com>
---
fs/btrfs/disk-io.c | 8 +++--
fs/btrfs/fs.h | 5 +++
fs/btrfs/inode.c | 78 +++++++++++++++++++++++++++++++++++++++++++---
3 files changed, 84 insertions(+), 7 deletions(-)
diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c
index a8535e309b57..64eaa7698be2 100644
--- a/fs/btrfs/disk-io.c
+++ b/fs/btrfs/disk-io.c
@@ -1774,6 +1774,8 @@ static void btrfs_stop_all_workers(struct btrfs_fs_info *fs_info)
if (fs_info->fixup_workers)
destroy_workqueue(fs_info->fixup_workers);
btrfs_destroy_workqueue(fs_info->delalloc_workers);
+ if (fs_info->delayed_write_workers)
+ destroy_workqueue(fs_info->delayed_write_workers);
btrfs_destroy_workqueue(fs_info->workers);
if (fs_info->endio_workers)
destroy_workqueue(fs_info->endio_workers);
@@ -1972,7 +1974,8 @@ static int btrfs_init_workqueues(struct btrfs_fs_info *fs_info)
fs_info->delalloc_workers =
btrfs_alloc_workqueue(fs_info, "delalloc",
flags, max_active, 2);
-
+ fs_info->delayed_write_workers =
+ alloc_workqueue("btrfs-delayed-write", flags, max_active);
fs_info->flush_workers =
btrfs_alloc_workqueue(fs_info, "flush_delalloc",
flags, max_active, 0);
@@ -2003,7 +2006,7 @@ static int btrfs_init_workqueues(struct btrfs_fs_info *fs_info)
fs_info->discard_ctl.discard_workers =
alloc_ordered_workqueue("btrfs-discard", WQ_FREEZABLE);
- if (!(fs_info->workers &&
+ if (!(fs_info->workers && fs_info->delayed_write_workers &&
fs_info->delalloc_workers && fs_info->flush_workers &&
fs_info->endio_workers && fs_info->endio_meta_workers &&
fs_info->endio_write_workers &&
@@ -4436,6 +4439,7 @@ void __cold close_ctree(struct btrfs_fs_info *fs_info)
* when we call kthread_stop().
*/
btrfs_flush_workqueue(fs_info->delalloc_workers);
+ flush_workqueue(fs_info->delayed_write_workers);
/*
* We can have ordered extents getting their last reference dropped from
diff --git a/fs/btrfs/fs.h b/fs/btrfs/fs.h
index 3eba8438593c..72e3ef1de423 100644
--- a/fs/btrfs/fs.h
+++ b/fs/btrfs/fs.h
@@ -707,6 +707,11 @@ struct btrfs_fs_info {
*/
struct btrfs_workqueue *workers;
struct btrfs_workqueue *delalloc_workers;
+ /*
+ * This is for delayed writes, for now only compressed writes
+ * on non-zoned experimental builds utilize this.
+ */
+ struct workqueue_struct *delayed_write_workers;
struct btrfs_workqueue *flush_workers;
struct workqueue_struct *endio_workers;
struct workqueue_struct *endio_meta_workers;
diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c
index 9d20af32eb2e..9568118e65b4 100644
--- a/fs/btrfs/inode.c
+++ b/fs/btrfs/inode.c
@@ -97,6 +97,11 @@ struct data_reloc_warn {
int mirror_num;
};
+struct delayed_bio_private {
+ struct work_struct work;
+ struct btrfs_bio *delayed_bbio;
+};
+
/*
* For the file_extent_tree, we want to hold the inode lock when we lookup and
* update the disk_i_size, but lockdep will complain because our io_tree we hold
@@ -7707,18 +7712,81 @@ struct extent_map *btrfs_create_delayed_em(struct btrfs_inode *inode,
return em;
}
-void btrfs_submit_delayed_write(struct btrfs_bio *bbio)
+static void run_delayed_bbio(struct work_struct *work)
{
- ASSERT(bbio->is_delayed);
+ struct delayed_bio_private *dbp = container_of(work, struct delayed_bio_private, work);
+ struct btrfs_bio *parent = dbp->delayed_bbio;
+
+ /* Compressed and uncompressed fallback is not yet implemented. */
+ ASSERT(0);
/*
- * Not yet implemented, and should not hit this path as we have no
- * caller to create delayed extent map.
+ * Any real compressed/uncompressed bios have increased
+ * parent->pending_ios, the last caller of btrfs_bio_end_io()
+ * will execute the real endio function.
*/
- ASSERT(0);
+ btrfs_bio_end_io(parent, BLK_STS_OK);
+}
+
+static void end_bbio_delayed(struct btrfs_bio *bbio)
+{
+ struct delayed_bio_private *dbp = bbio->private;
+ struct btrfs_inode *inode = bbio->inode;
+ struct btrfs_fs_info *fs_info = inode->root->fs_info;
+ struct folio_iter fi;
+ const u32 bio_size = bio_get_size(&bbio->bio);
+ const bool uptodate = bbio->status == BLK_STS_OK;
+
+ ASSERT(bbio->is_delayed);
+
+ bio_for_each_folio_all(fi, &bbio->bio) {
+ u64 start = folio_pos(fi.folio) + fi.offset;
+ u32 len = fi.length;
+
+ btrfs_folio_clear_writeback(fs_info, fi.folio, start, len);
+ }
+ btrfs_mark_ordered_io_finished(inode, bbio->file_offset, bio_size, uptodate);
+ kfree(dbp);
bio_put(&bbio->bio);
}
+void btrfs_submit_delayed_write(struct btrfs_bio *bbio)
+{
+ const struct btrfs_fs_info *fs_info = bbio->inode->root->fs_info;
+ struct delayed_bio_private *dbp;
+
+ ASSERT(bbio->is_delayed);
+
+ bbio->end_io = end_bbio_delayed;
+ dbp = kzalloc_obj(*dbp, GFP_NOFS);
+ if (!dbp) {
+ btrfs_bio_end_io(bbio, errno_to_blk_status(-ENOMEM));
+ return;
+ }
+ dbp->delayed_bbio = bbio;
+ bbio->private = dbp;
+ /*
+ * TODO: find a way to properly allow sequential extent allocation.
+ *
+ * Currently compression and submission happen in the same workload.
+ * We want to run compression in parallel but run allocation sequentially.
+ *
+ * The existing btrfs workqueue will execute the sequential workload
+ * twice, the second one to free the structure, thus the btrfs_work
+ * structure can have a lifespan longer than the delayed bbio.
+ *
+ * But the current structure's lifespan is bounded to the original
+ * bbio, if using btrfs workqueue, we need extra lifespan management
+ * to ensure the work structure doesn't disappear when the original
+ * bbio is finished.
+ *
+ * This means we have to either delay the original bbio's finish,
+ * or have a very complex synchronization.
+ */
+ INIT_WORK(&dbp->work, run_delayed_bbio);
+ queue_work(fs_info->delayed_write_workers, &dbp->work);
+}
+
/*
* For release_folio() and invalidate_folio() we have a race window where
* folio_end_writeback() is called but the subpage spinlock is not yet released.
--
2.55.0
next prev parent reply other threads:[~2026-09-18 0:44 UTC|newest]
Thread overview: 7+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-18 0:43 [PATCH v4 RESEND 0/6] btrfs: delay compression to bbio submission time Qu Wenruo
2026-09-18 0:44 ` [PATCH v4 RESEND 1/6] btrfs: add delayed ordered extent support Qu Wenruo
2026-09-18 0:44 ` [PATCH v4 RESEND 2/6] btrfs: add skeleton for delayed btrfs bio Qu Wenruo
2026-09-18 0:44 ` Qu Wenruo [this message]
2026-09-18 0:44 ` [PATCH v4 RESEND 4/6] btrfs: introduce compression for delayed bbio Qu Wenruo
2026-09-18 0:44 ` [PATCH v4 RESEND 5/6] btrfs: implement uncompressed fallback " Qu Wenruo
2026-09-18 0:44 ` [PATCH v4 RESEND 6/6] btrfs: enable experimental delayed compression support Qu Wenruo
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=1de09905a5581ab755d30f64fb8055aaa1c90b60.1789692182.git.wqu@suse.com \
--to=wqu@suse.com \
--cc=linux-btrfs@vger.kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox