Linux Btrfs filesystem development
 help / color / mirror / Atom feed
From: Johannes Thumshirn <johannes.thumshirn@wdc.com>
To: Richard Weinberger <richard@nod.at>
Cc: linux-btrfs@vger.kernel.org
Subject: Re: qgroup rescan worker makes suspend fail
Date: Sat, 19 Sep 2026 19:34:09 +0200	[thread overview]
Message-ID: <aq7HYO9SKdIaqvr_@mayhem.fritz.box> (raw)
In-Reply-To: <mu8fq8a8.b9a9a1ae-dd6b-45a4-a415-5db853d61536@nod.at>

On Sat, Sep 19, 2026 at 01:40:27PM +0000, Richard Weinberger wrote:
> Once in a while, suspending my laptop just causes the screen to freeze.
> Initially, I thought Linux had crashed, but it usually recovers after 2 minutes,
> though the suspend doesn't actually happen. You can imagine this can be very
> unfortunate when you just close the laptop lid and pack the laptop into your
> bag...
> 
> After the problem started occurring more frequently, I investigated and found
> that the qgroup rescan worker is the problem. In dmesg, logs like these can
> usually be found:
> 
> [246013.777637] [ T278489] Freezing remaining freezable tasks
> [246033.780538] [ T278489] Freezing remaining freezable tasks failed after 20.003 seconds (0 tasks refusing to freeze, wq_busy=1):
> [246033.780576] [ T278489] Showing freezable workqueues that are still busy:
> [246033.780582] [ T278489] workqueue events_freezable: flags=0x104
> [246033.780590] [ T278489]   pwq 10: cpus=2 node=0 flags=0x0 nice=0 active=0 refcnt=2
> [246033.780609] [ T278489]     inactive: pci_pme_list_scan
> [246033.780642] [ T278489] workqueue btrfs-endio-meta: flags=0xe
> [246033.780649] [ T278489]   pwq 57: cpus=0-13 node=0 flags=0x4 nice=0 active=0 refcnt=2
> [246033.780660] [ T278489]     inactive: simple_end_io_work [btrfs]
> [246033.781183] [ T278489] workqueue btrfs-qgroup-rescan: flags=0x2000e
> [246033.781189] [ T278489]   pwq 56: cpus=0-13 flags=0x4 nice=0 active=1 refcnt=16
> [246033.781199] [ T278489]     in-flight: 205929:btrfs_work_helper [btrfs] for 111s
> [246033.781721] [ T278489] workqueue wg-kex-wginterproc: flags=0x6
> [246033.781726] [ T278489]   pwq 57: cpus=0-13 node=0 flags=0x4 nice=0 active=0 refcnt=2
> [246033.781736] [ T278489]     inactive: wg_packet_handshake_send_worker [wireguard]
> 
> My first thought was that the worker is likely not freezable, but it is.
> The problem is that the whole qgroup rescan is a single work item.
> In my case, such a scan can take up to 10 minutes, even though I have a
> fast NVMe SSD installed...
> 
> Wouldn't it make sense to have rescan_should_stop() return true when
> suspend starts? I think using a pm notifier could help here.
> What do you think?

Something like this (completely untested):

diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c
index a1d83ad9a4c0..541735d8fdb6 100644
--- a/fs/btrfs/disk-io.c
+++ b/fs/btrfs/disk-io.c
@@ -17,6 +17,7 @@
 #include <linux/error-injection.h>
 #include <linux/crc32c.h>
 #include <linux/sched/mm.h>
+#include <linux/suspend.h>
 #include <linux/unaligned.h>
 #include "ctree.h"
 #include "disk-io.h"
@@ -3198,6 +3199,29 @@ int btrfs_start_pre_rw_mount(struct btrfs_fs_info *fs_info)
 	return 0;
 }
 
+static int btrfs_pm_notifier(struct notifier_block *nb, unsigned long action,
+			     void *data)
+{
+	struct btrfs_fs_info *fs_info = container_of(nb, struct btrfs_fs_info,
+						     pm_notifier);
+
+	switch (action) {
+	case PM_HIBERNATION_PREPARE:
+	case PM_SUSPEND_PREPARE:
+	case PM_RESTORE_PREPARE:
+		set_bit(BTRFS_FS_PM_SUSPENDING, &fs_info->flags);
+		break;
+	case PM_POST_HIBERNATION:
+	case PM_POST_SUSPEND:
+	case PM_POST_RESTORE:
+		clear_bit(BTRFS_FS_PM_SUSPENDING, &fs_info->flags);
+		btrfs_qgroup_rescan_resume(fs_info);
+		break;
+	}
+
+	return NOTIFY_DONE;
+}
+
 /*
  * Do various sanity and dependency checks of different features.
  *
@@ -3794,6 +3818,9 @@ int __cold open_ctree(struct super_block *sb, struct btrfs_fs_devices *fs_device
 
 	set_bit(BTRFS_FS_OPEN, &fs_info->flags);
 
+	fs_info->pm_notifier.notifier_call = btrfs_pm_notifier;
+	register_pm_notifier(&fs_info->pm_notifier);
+
 	/* Kick the cleaner thread so it'll start deleting snapshots. */
 	if (test_bit(BTRFS_FS_UNFINISHED_DROPS, &fs_info->flags))
 		wake_up_process(fs_info->cleaner_kthread);
@@ -4370,6 +4397,8 @@ void __cold close_ctree(struct btrfs_fs_info *fs_info)
 	 */
 	kthread_park(fs_info->cleaner_kthread);
 
+	unregister_pm_notifier(&fs_info->pm_notifier);
+
 	/* wait for the qgroup rescan worker to stop */
 	btrfs_qgroup_wait_for_completion(fs_info, false);
 
diff --git a/fs/btrfs/fs.h b/fs/btrfs/fs.h
index 3eba8438593c..caa90dc98e59 100644
--- a/fs/btrfs/fs.h
+++ b/fs/btrfs/fs.h
@@ -26,6 +26,7 @@
 #include <linux/wait_bit.h>
 #include <linux/sched.h>
 #include <linux/rbtree.h>
+#include <linux/notifier.h>
 #include <linux/xxhash.h>
 #include <linux/fserror.h>
 #include <uapi/linux/btrfs.h>
@@ -234,6 +235,8 @@ enum {
 	 */
 	BTRFS_FS_UNALIGNED_TREE_BLOCK,
 
+	BTRFS_FS_PM_SUSPENDING,
+
 #if BITS_PER_LONG == 32
 	/* Indicate if we have error/warn message printed on 32bit systems */
 	BTRFS_FS_32BIT_ERROR,
@@ -841,6 +844,8 @@ struct btrfs_fs_info {
 	u8 qgroup_drop_subtree_thres;
 	u64 qgroup_enable_gen;
 
+	struct notifier_block pm_notifier;
+
 	/*
 	 * If this is not 0, then it indicates a serious filesystem error has
 	 * happened and it contains that error (negative errno value).
diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c
index 05e35eb126dc..b4f1290d1e14 100644
--- a/fs/btrfs/qgroup.c
+++ b/fs/btrfs/qgroup.c
@@ -3883,6 +3883,7 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work)
 	struct btrfs_trans_handle *trans = NULL;
 	int ret = 0;
 	bool stopped = false;
+	bool pm_paused = false;
 	bool did_leaf_rescans = false;
 
 	if (btrfs_qgroup_mode(fs_info) == BTRFS_QGROUP_MODE_SIMPLE)
@@ -3900,7 +3901,18 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work)
 	path->search_commit_root = true;
 	path->skip_locking = true;
 
-	while (!ret && !(stopped = rescan_should_stop(fs_info))) {
+	while (!ret) {
+		if (rescan_should_stop(fs_info)) {
+			stopped = true;
+			break;
+		}
+
+		if (test_bit(BTRFS_FS_PM_SUSPENDING, &fs_info->flags)) {
+			stopped = true;
+			pm_paused = true;
+			break;
+		}
+
 		trans = btrfs_start_transaction(fs_info->fs_root, 0);
 		if (IS_ERR(trans)) {
 			ret = PTR_ERR(trans);
@@ -3963,12 +3975,17 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work)
 	complete_all(&fs_info->qgroup_rescan_completion);
 	mutex_unlock(&fs_info->qgroup_rescan_lock);
 
+	if (pm_paused && !test_bit(BTRFS_FS_PM_SUSPENDING, &fs_info->flags))
+		btrfs_qgroup_rescan_resume(fs_info);
+
 	if (!trans)
 		return;
 
 	btrfs_end_transaction(trans);
 
-	if (stopped) {
+	if (pm_paused) {
+		btrfs_info(fs_info, "qgroup scan paused for system suspend");
+	} else if (stopped) {
 		btrfs_info(fs_info, "qgroup scan paused");
 	} else if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags)) {
 		btrfs_info(fs_info, "qgroup scan cancelled");
@@ -4142,13 +4159,20 @@ int btrfs_qgroup_wait_for_completion(struct btrfs_fs_info *fs_info,
 void
 btrfs_qgroup_rescan_resume(struct btrfs_fs_info *fs_info)
 {
-	if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) {
-		mutex_lock(&fs_info->qgroup_rescan_lock);
+	if (!test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags))
+		return;
+
+	if (btrfs_fs_closing(fs_info))
+		return;
+
+	mutex_lock(&fs_info->qgroup_rescan_lock);
+	if (!fs_info->qgroup_rescan_running) {
+		reinit_completion(&fs_info->qgroup_rescan_completion);
 		fs_info->qgroup_rescan_running = true;
 		btrfs_queue_work(fs_info->qgroup_rescan_workers,
 				 &fs_info->qgroup_rescan_work);
-		mutex_unlock(&fs_info->qgroup_rescan_lock);
 	}
+	mutex_unlock(&fs_info->qgroup_rescan_lock);
 }
 
 #define rbtree_iterate_from_safe(node, next, start)				\


  reply	other threads:[~2026-09-19 17:34 UTC|newest]

Thread overview: 9+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-19 13:40 qgroup rescan worker makes suspend fail Richard Weinberger
2026-09-19 17:34 ` Johannes Thumshirn [this message]
2026-09-19 18:33   ` AW: " Richard Weinberger
2026-09-19 22:39   ` Qu Wenruo
2026-09-19 22:48     ` Qu Wenruo
2026-09-20  4:56     ` Andrei Borzenkov
2026-09-20  5:28       ` Qu Wenruo
2026-09-19 23:38 ` Qu Wenruo
2026-09-20 13:04   ` AW: " Richard Weinberger

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=aq7HYO9SKdIaqvr_@mayhem.fritz.box \
    --to=johannes.thumshirn@wdc.com \
    --cc=linux-btrfs@vger.kernel.org \
    --cc=richard@nod.at \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox