From: Johannes Thumshirn <johannes.thumshirn@wdc.com>
To: Richard Weinberger <richard@nod.at>
Cc: linux-btrfs@vger.kernel.org
Subject: Re: qgroup rescan worker makes suspend fail
Date: Sat, 19 Sep 2026 19:34:09 +0200 [thread overview]
Message-ID: <aq7HYO9SKdIaqvr_@mayhem.fritz.box> (raw)
In-Reply-To: <mu8fq8a8.b9a9a1ae-dd6b-45a4-a415-5db853d61536@nod.at>
On Sat, Sep 19, 2026 at 01:40:27PM +0000, Richard Weinberger wrote:
> Once in a while, suspending my laptop just causes the screen to freeze.
> Initially, I thought Linux had crashed, but it usually recovers after 2 minutes,
> though the suspend doesn't actually happen. You can imagine this can be very
> unfortunate when you just close the laptop lid and pack the laptop into your
> bag...
>
> After the problem started occurring more frequently, I investigated and found
> that the qgroup rescan worker is the problem. In dmesg, logs like these can
> usually be found:
>
> [246013.777637] [ T278489] Freezing remaining freezable tasks
> [246033.780538] [ T278489] Freezing remaining freezable tasks failed after 20.003 seconds (0 tasks refusing to freeze, wq_busy=1):
> [246033.780576] [ T278489] Showing freezable workqueues that are still busy:
> [246033.780582] [ T278489] workqueue events_freezable: flags=0x104
> [246033.780590] [ T278489] pwq 10: cpus=2 node=0 flags=0x0 nice=0 active=0 refcnt=2
> [246033.780609] [ T278489] inactive: pci_pme_list_scan
> [246033.780642] [ T278489] workqueue btrfs-endio-meta: flags=0xe
> [246033.780649] [ T278489] pwq 57: cpus=0-13 node=0 flags=0x4 nice=0 active=0 refcnt=2
> [246033.780660] [ T278489] inactive: simple_end_io_work [btrfs]
> [246033.781183] [ T278489] workqueue btrfs-qgroup-rescan: flags=0x2000e
> [246033.781189] [ T278489] pwq 56: cpus=0-13 flags=0x4 nice=0 active=1 refcnt=16
> [246033.781199] [ T278489] in-flight: 205929:btrfs_work_helper [btrfs] for 111s
> [246033.781721] [ T278489] workqueue wg-kex-wginterproc: flags=0x6
> [246033.781726] [ T278489] pwq 57: cpus=0-13 node=0 flags=0x4 nice=0 active=0 refcnt=2
> [246033.781736] [ T278489] inactive: wg_packet_handshake_send_worker [wireguard]
>
> My first thought was that the worker is likely not freezable, but it is.
> The problem is that the whole qgroup rescan is a single work item.
> In my case, such a scan can take up to 10 minutes, even though I have a
> fast NVMe SSD installed...
>
> Wouldn't it make sense to have rescan_should_stop() return true when
> suspend starts? I think using a pm notifier could help here.
> What do you think?
Something like this (completely untested):
diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c
index a1d83ad9a4c0..541735d8fdb6 100644
--- a/fs/btrfs/disk-io.c
+++ b/fs/btrfs/disk-io.c
@@ -17,6 +17,7 @@
#include <linux/error-injection.h>
#include <linux/crc32c.h>
#include <linux/sched/mm.h>
+#include <linux/suspend.h>
#include <linux/unaligned.h>
#include "ctree.h"
#include "disk-io.h"
@@ -3198,6 +3199,29 @@ int btrfs_start_pre_rw_mount(struct btrfs_fs_info *fs_info)
return 0;
}
+static int btrfs_pm_notifier(struct notifier_block *nb, unsigned long action,
+ void *data)
+{
+ struct btrfs_fs_info *fs_info = container_of(nb, struct btrfs_fs_info,
+ pm_notifier);
+
+ switch (action) {
+ case PM_HIBERNATION_PREPARE:
+ case PM_SUSPEND_PREPARE:
+ case PM_RESTORE_PREPARE:
+ set_bit(BTRFS_FS_PM_SUSPENDING, &fs_info->flags);
+ break;
+ case PM_POST_HIBERNATION:
+ case PM_POST_SUSPEND:
+ case PM_POST_RESTORE:
+ clear_bit(BTRFS_FS_PM_SUSPENDING, &fs_info->flags);
+ btrfs_qgroup_rescan_resume(fs_info);
+ break;
+ }
+
+ return NOTIFY_DONE;
+}
+
/*
* Do various sanity and dependency checks of different features.
*
@@ -3794,6 +3818,9 @@ int __cold open_ctree(struct super_block *sb, struct btrfs_fs_devices *fs_device
set_bit(BTRFS_FS_OPEN, &fs_info->flags);
+ fs_info->pm_notifier.notifier_call = btrfs_pm_notifier;
+ register_pm_notifier(&fs_info->pm_notifier);
+
/* Kick the cleaner thread so it'll start deleting snapshots. */
if (test_bit(BTRFS_FS_UNFINISHED_DROPS, &fs_info->flags))
wake_up_process(fs_info->cleaner_kthread);
@@ -4370,6 +4397,8 @@ void __cold close_ctree(struct btrfs_fs_info *fs_info)
*/
kthread_park(fs_info->cleaner_kthread);
+ unregister_pm_notifier(&fs_info->pm_notifier);
+
/* wait for the qgroup rescan worker to stop */
btrfs_qgroup_wait_for_completion(fs_info, false);
diff --git a/fs/btrfs/fs.h b/fs/btrfs/fs.h
index 3eba8438593c..caa90dc98e59 100644
--- a/fs/btrfs/fs.h
+++ b/fs/btrfs/fs.h
@@ -26,6 +26,7 @@
#include <linux/wait_bit.h>
#include <linux/sched.h>
#include <linux/rbtree.h>
+#include <linux/notifier.h>
#include <linux/xxhash.h>
#include <linux/fserror.h>
#include <uapi/linux/btrfs.h>
@@ -234,6 +235,8 @@ enum {
*/
BTRFS_FS_UNALIGNED_TREE_BLOCK,
+ BTRFS_FS_PM_SUSPENDING,
+
#if BITS_PER_LONG == 32
/* Indicate if we have error/warn message printed on 32bit systems */
BTRFS_FS_32BIT_ERROR,
@@ -841,6 +844,8 @@ struct btrfs_fs_info {
u8 qgroup_drop_subtree_thres;
u64 qgroup_enable_gen;
+ struct notifier_block pm_notifier;
+
/*
* If this is not 0, then it indicates a serious filesystem error has
* happened and it contains that error (negative errno value).
diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c
index 05e35eb126dc..b4f1290d1e14 100644
--- a/fs/btrfs/qgroup.c
+++ b/fs/btrfs/qgroup.c
@@ -3883,6 +3883,7 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work)
struct btrfs_trans_handle *trans = NULL;
int ret = 0;
bool stopped = false;
+ bool pm_paused = false;
bool did_leaf_rescans = false;
if (btrfs_qgroup_mode(fs_info) == BTRFS_QGROUP_MODE_SIMPLE)
@@ -3900,7 +3901,18 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work)
path->search_commit_root = true;
path->skip_locking = true;
- while (!ret && !(stopped = rescan_should_stop(fs_info))) {
+ while (!ret) {
+ if (rescan_should_stop(fs_info)) {
+ stopped = true;
+ break;
+ }
+
+ if (test_bit(BTRFS_FS_PM_SUSPENDING, &fs_info->flags)) {
+ stopped = true;
+ pm_paused = true;
+ break;
+ }
+
trans = btrfs_start_transaction(fs_info->fs_root, 0);
if (IS_ERR(trans)) {
ret = PTR_ERR(trans);
@@ -3963,12 +3975,17 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work)
complete_all(&fs_info->qgroup_rescan_completion);
mutex_unlock(&fs_info->qgroup_rescan_lock);
+ if (pm_paused && !test_bit(BTRFS_FS_PM_SUSPENDING, &fs_info->flags))
+ btrfs_qgroup_rescan_resume(fs_info);
+
if (!trans)
return;
btrfs_end_transaction(trans);
- if (stopped) {
+ if (pm_paused) {
+ btrfs_info(fs_info, "qgroup scan paused for system suspend");
+ } else if (stopped) {
btrfs_info(fs_info, "qgroup scan paused");
} else if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags)) {
btrfs_info(fs_info, "qgroup scan cancelled");
@@ -4142,13 +4159,20 @@ int btrfs_qgroup_wait_for_completion(struct btrfs_fs_info *fs_info,
void
btrfs_qgroup_rescan_resume(struct btrfs_fs_info *fs_info)
{
- if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) {
- mutex_lock(&fs_info->qgroup_rescan_lock);
+ if (!test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags))
+ return;
+
+ if (btrfs_fs_closing(fs_info))
+ return;
+
+ mutex_lock(&fs_info->qgroup_rescan_lock);
+ if (!fs_info->qgroup_rescan_running) {
+ reinit_completion(&fs_info->qgroup_rescan_completion);
fs_info->qgroup_rescan_running = true;
btrfs_queue_work(fs_info->qgroup_rescan_workers,
&fs_info->qgroup_rescan_work);
- mutex_unlock(&fs_info->qgroup_rescan_lock);
}
+ mutex_unlock(&fs_info->qgroup_rescan_lock);
}
#define rbtree_iterate_from_safe(node, next, start) \
next prev parent reply other threads:[~2026-09-19 17:34 UTC|newest]
Thread overview: 9+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-19 13:40 qgroup rescan worker makes suspend fail Richard Weinberger
2026-09-19 17:34 ` Johannes Thumshirn [this message]
2026-09-19 18:33 ` AW: " Richard Weinberger
2026-09-19 22:39 ` Qu Wenruo
2026-09-19 22:48 ` Qu Wenruo
2026-09-20 4:56 ` Andrei Borzenkov
2026-09-20 5:28 ` Qu Wenruo
2026-09-19 23:38 ` Qu Wenruo
2026-09-20 13:04 ` AW: " Richard Weinberger
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=aq7HYO9SKdIaqvr_@mayhem.fritz.box \
--to=johannes.thumshirn@wdc.com \
--cc=linux-btrfs@vger.kernel.org \
--cc=richard@nod.at \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox