fs/f2fs/f2fs.h | 9 +++++++++ fs/f2fs/gc.c | 13 ++++++++----- fs/f2fs/segment.c | 12 ++++++++---- fs/f2fs/super.c | 25 +++++++++++++++++++++++++ 4 files changed, 50 insertions(+), 9 deletions(-)
From: Daeho Jeong <daehojeong@google.com>
During system suspend, a race condition can cause f2fs_gc and f2fs_discard
threads to call submit_bio() while the underlying block device (e.g., UFS)
is in Runtime PM suspend. Because Runtime PM worker threads are already
frozen during task freezing, the threads become trapped in
__bio_queue_enter() waiting on mq_freeze_wq, leading to a PM freezer
timeout.
To prevent this deadlock, register a PM notifier to set SBI_IS_SUSPENDING
during PM_SUSPEND_PREPARE. Background GC and discard threads check this
flag and immediately stop issuing new bios, allowing them to enter a
freezable sleep state cleanly before process freezing begins.
In addition, check freezing() as a fast path to stop issuing new I/O
when non-PM freezing (e.g. dm-snapshot or cgroup freezer) is requested.
Signed-off-by: Daeho Jeong <daehojeong@google.com>
---
v2: check freezing() together for non-PM freezing.
---
fs/f2fs/f2fs.h | 9 +++++++++
fs/f2fs/gc.c | 13 ++++++++-----
fs/f2fs/segment.c | 12 ++++++++----
fs/f2fs/super.c | 25 +++++++++++++++++++++++++
4 files changed, 50 insertions(+), 9 deletions(-)
diff --git a/fs/f2fs/f2fs.h b/fs/f2fs/f2fs.h
index f1774d4e18d2..2a7b0fe8419b 100644
--- a/fs/f2fs/f2fs.h
+++ b/fs/f2fs/f2fs.h
@@ -25,6 +25,7 @@
#include <linux/quotaops.h>
#include <linux/part_stat.h>
#include <linux/rw_hint.h>
+#include <linux/suspend.h>
#include <linux/fscrypt.h>
#include <linux/fsverity.h>
@@ -1492,6 +1493,7 @@ enum {
SBI_IS_FREEZING, /* freezefs is in process */
SBI_IS_WRITABLE, /* remove ro mountoption transiently */
SBI_ENABLE_CHECKPOINT, /* indicate it's during f2fs_enable_checkpoint() */
+ SBI_IS_SUSPENDING, /* system suspend is in progress */
MAX_SBI_FLAG,
};
@@ -1755,6 +1757,7 @@ struct f2fs_sb_info {
struct f2fs_rwsem sb_lock; /* lock for raw super block */
int valid_super_block; /* valid super block no */
unsigned long s_flag; /* flags for sbi */
+ struct notifier_block pm_nb; /* for PM notifier */
struct mutex writepages; /* mutex for writepages() */
#ifdef CONFIG_BLK_DEV_ZONED
@@ -2309,6 +2312,12 @@ static inline void clear_sbi_flag(struct f2fs_sb_info *sbi, unsigned int type)
clear_bit(type, &sbi->s_flag);
}
+static inline bool f2fs_is_suspending(struct f2fs_sb_info *sbi)
+{
+ return is_sbi_flag_set(sbi, SBI_IS_SUSPENDING) ||
+ unlikely(freezing(current));
+}
+
static inline unsigned long long cur_cp_version(struct f2fs_checkpoint *cp)
{
return le64_to_cpu(cp->checkpoint_ver);
diff --git a/fs/f2fs/gc.c b/fs/f2fs/gc.c
index ffaa7ba76a1b..6c7ea38eb70d 100644
--- a/fs/f2fs/gc.c
+++ b/fs/f2fs/gc.c
@@ -71,7 +71,8 @@ static int gc_thread_func(void *data)
if (kthread_should_stop())
break;
- if (sbi->sb->s_writers.frozen >= SB_FREEZE_WRITE) {
+ if (sbi->sb->s_writers.frozen >= SB_FREEZE_WRITE ||
+ f2fs_is_suspending(sbi)) {
increase_sleep_time(gc_th, &wait_ms);
stat_other_skip_bggc_count(sbi);
continue;
@@ -1064,8 +1065,9 @@ static int gc_node_segment(struct f2fs_sb_info *sbi,
struct node_info ni;
int err;
- /* stop BG_GC if there is not enough free sections. */
- if (gc_type == BG_GC && has_not_enough_free_secs(sbi, 0, 0))
+ /* stop BG_GC if there is not enough free sections or suspending/freezing. */
+ if (gc_type == BG_GC && (has_not_enough_free_secs(sbi, 0, 0) ||
+ f2fs_is_suspending(sbi)))
return submitted;
if (check_valid_map(sbi, segno, off) == 0)
@@ -1611,7 +1613,8 @@ static int gc_data_segment(struct f2fs_sb_info *sbi, struct f2fs_summary *sum,
* Or, stop GC if the segment becomes fully valid caused by
* race condition along with SSR block allocation.
*/
- if ((gc_type == BG_GC && has_not_enough_free_secs(sbi, 0, 0)) ||
+ if ((gc_type == BG_GC && (has_not_enough_free_secs(sbi, 0, 0) ||
+ f2fs_is_suspending(sbi))) ||
(!force_migrate && get_valid_blocks(sbi, segno, true) ==
CAP_BLKS_PER_SEC(sbi)))
return submitted;
@@ -2015,7 +2018,7 @@ int f2fs_gc(struct f2fs_sb_info *sbi, struct f2fs_gc_control *gc_control)
goto stop;
}
retry:
- if (unlikely(freezing(current))) {
+ if (f2fs_is_suspending(sbi)) {
ret = 0;
goto stop;
}
diff --git a/fs/f2fs/segment.c b/fs/f2fs/segment.c
index d71ddb3ee918..7b03b3d06161 100644
--- a/fs/f2fs/segment.c
+++ b/fs/f2fs/segment.c
@@ -1300,7 +1300,7 @@ static int __submit_discard_cmd(struct f2fs_sb_info *sbi,
if (dc->state != D_PREP)
return 0;
- if (is_sbi_flag_set(sbi, SBI_NEED_FSCK))
+ if (is_sbi_flag_set(sbi, SBI_NEED_FSCK) || f2fs_is_suspending(sbi))
return 0;
#ifdef CONFIG_BLK_DEV_ZONED
@@ -1341,6 +1341,9 @@ static int __submit_discard_cmd(struct f2fs_sb_info *sbi,
unsigned long flags;
bool last = true;
+ if (f2fs_is_suspending(sbi))
+ break;
+
if (len > max_discard_blocks) {
len = max_discard_blocks;
last = false;
@@ -1615,7 +1618,7 @@ static void __issue_discard_cmd_orderly(struct f2fs_sb_info *sbi,
if (dc->state != D_PREP)
goto next;
- if (*issued > 0 && unlikely(freezing(current)))
+ if (f2fs_is_suspending(sbi))
break;
if (dpolicy->io_aware && !is_idle(sbi, DISCARD_TIME)) {
@@ -1688,7 +1691,7 @@ static int __issue_discard_cmd(struct f2fs_sb_info *sbi,
list_for_each_entry_safe(dc, tmp, pend_list, list) {
f2fs_bug_on(sbi, dc->state != D_PREP);
- if (issued > 0 && unlikely(freezing(current))) {
+ if (f2fs_is_suspending(sbi)) {
suspended = true;
break;
}
@@ -1955,7 +1958,8 @@ static int issue_discard_thread(void *data)
continue;
if (kthread_should_stop())
return 0;
- if (is_sbi_flag_set(sbi, SBI_NEED_FSCK) ||
+ if (f2fs_is_suspending(sbi) ||
+ is_sbi_flag_set(sbi, SBI_NEED_FSCK) ||
!atomic_read(&dcc->discard_cmd_cnt)) {
wait_ms = dpolicy.max_interval;
continue;
diff --git a/fs/f2fs/super.c b/fs/f2fs/super.c
index c448d992ff2a..8c97b5d1ee68 100644
--- a/fs/f2fs/super.c
+++ b/fs/f2fs/super.c
@@ -1979,6 +1979,26 @@ static void destroy_device_list(struct f2fs_sb_info *sbi)
kvfree(sbi->devs);
}
+static int f2fs_pm_notifier(struct notifier_block *nb,
+ unsigned long action, void *ptr)
+{
+ struct f2fs_sb_info *sbi = container_of(nb, struct f2fs_sb_info, pm_nb);
+
+ switch (action) {
+ case PM_HIBERNATION_PREPARE:
+ case PM_SUSPEND_PREPARE:
+ case PM_RESTORE_PREPARE:
+ set_sbi_flag(sbi, SBI_IS_SUSPENDING);
+ break;
+ case PM_POST_SUSPEND:
+ case PM_POST_HIBERNATION:
+ case PM_POST_RESTORE:
+ clear_sbi_flag(sbi, SBI_IS_SUSPENDING);
+ break;
+ }
+ return NOTIFY_OK;
+}
+
static void f2fs_put_super(struct super_block *sb)
{
struct f2fs_sb_info *sbi = F2FS_SB(sb);
@@ -1986,6 +2006,8 @@ static void f2fs_put_super(struct super_block *sb)
int err = 0;
bool done;
+ unregister_pm_notifier(&sbi->pm_nb);
+
/* unregister procfs/sysfs entries in advance to avoid race case */
f2fs_unregister_sysfs(sbi);
@@ -5436,6 +5458,9 @@ static int f2fs_fill_super(struct super_block *sb, struct fs_context *fc)
f2fs_update_time(sbi, REQ_TIME);
clear_sbi_flag(sbi, SBI_CP_DISABLED_QUICK);
+ sbi->pm_nb.notifier_call = f2fs_pm_notifier;
+ register_pm_notifier(&sbi->pm_nb);
+
sbi->umount_lock_holder = NULL;
return 0;
--
2.55.0.654.g21b8a5bc05-goog
On Thu, Aug 06, 2026 at 10:01:08AM -0700, Daeho Jeong wrote:
> From: Daeho Jeong <daehojeong@google.com>
>
> During system suspend, a race condition can cause f2fs_gc and f2fs_discard
> threads to call submit_bio() while the underlying block device (e.g., UFS)
> is in Runtime PM suspend. Because Runtime PM worker threads are already
> frozen during task freezing, the threads become trapped in
> __bio_queue_enter() waiting on mq_freeze_wq, leading to a PM freezer
> timeout.
That does sound like a general issue with our block device / threading
handling.
> To prevent this deadlock, register a PM notifier to set SBI_IS_SUSPENDING
> during PM_SUSPEND_PREPARE. Background GC and discard threads check this
> flag and immediately stop issuing new bios, allowing them to enter a
> freezable sleep state cleanly before process freezing begins.
>
> In addition, check freezing() as a fast path to stop issuing new I/O
> when non-PM freezing (e.g. dm-snapshot or cgroup freezer) is requested.
.. which means that we really sould have all the relevant parties
invited into figuring out whast is happening here, rather than
band-aiding something that looks like a horrible hack inside a
file system.
Unfortunately I see this a lot with f2fs. Please reach out to all
relevant maintainers for something that does not look strictly local
to f2fs.
> Signed-off-by: Daeho Jeong <daehojeong@google.com>
> ---
> v2: check freezing() together for non-PM freezing.
> ---
> fs/f2fs/f2fs.h | 9 +++++++++
> fs/f2fs/gc.c | 13 ++++++++-----
> fs/f2fs/segment.c | 12 ++++++++----
> fs/f2fs/super.c | 25 +++++++++++++++++++++++++
> 4 files changed, 50 insertions(+), 9 deletions(-)
>
> diff --git a/fs/f2fs/f2fs.h b/fs/f2fs/f2fs.h
> index f1774d4e18d2..2a7b0fe8419b 100644
> --- a/fs/f2fs/f2fs.h
> +++ b/fs/f2fs/f2fs.h
> @@ -25,6 +25,7 @@
> #include <linux/quotaops.h>
> #include <linux/part_stat.h>
> #include <linux/rw_hint.h>
> +#include <linux/suspend.h>
>
> #include <linux/fscrypt.h>
> #include <linux/fsverity.h>
> @@ -1492,6 +1493,7 @@ enum {
> SBI_IS_FREEZING, /* freezefs is in process */
> SBI_IS_WRITABLE, /* remove ro mountoption transiently */
> SBI_ENABLE_CHECKPOINT, /* indicate it's during f2fs_enable_checkpoint() */
> + SBI_IS_SUSPENDING, /* system suspend is in progress */
> MAX_SBI_FLAG,
> };
>
> @@ -1755,6 +1757,7 @@ struct f2fs_sb_info {
> struct f2fs_rwsem sb_lock; /* lock for raw super block */
> int valid_super_block; /* valid super block no */
> unsigned long s_flag; /* flags for sbi */
> + struct notifier_block pm_nb; /* for PM notifier */
> struct mutex writepages; /* mutex for writepages() */
>
> #ifdef CONFIG_BLK_DEV_ZONED
> @@ -2309,6 +2312,12 @@ static inline void clear_sbi_flag(struct f2fs_sb_info *sbi, unsigned int type)
> clear_bit(type, &sbi->s_flag);
> }
>
> +static inline bool f2fs_is_suspending(struct f2fs_sb_info *sbi)
> +{
> + return is_sbi_flag_set(sbi, SBI_IS_SUSPENDING) ||
> + unlikely(freezing(current));
> +}
> +
> static inline unsigned long long cur_cp_version(struct f2fs_checkpoint *cp)
> {
> return le64_to_cpu(cp->checkpoint_ver);
> diff --git a/fs/f2fs/gc.c b/fs/f2fs/gc.c
> index ffaa7ba76a1b..6c7ea38eb70d 100644
> --- a/fs/f2fs/gc.c
> +++ b/fs/f2fs/gc.c
> @@ -71,7 +71,8 @@ static int gc_thread_func(void *data)
> if (kthread_should_stop())
> break;
>
> - if (sbi->sb->s_writers.frozen >= SB_FREEZE_WRITE) {
> + if (sbi->sb->s_writers.frozen >= SB_FREEZE_WRITE ||
> + f2fs_is_suspending(sbi)) {
> increase_sleep_time(gc_th, &wait_ms);
> stat_other_skip_bggc_count(sbi);
> continue;
> @@ -1064,8 +1065,9 @@ static int gc_node_segment(struct f2fs_sb_info *sbi,
> struct node_info ni;
> int err;
>
> - /* stop BG_GC if there is not enough free sections. */
> - if (gc_type == BG_GC && has_not_enough_free_secs(sbi, 0, 0))
> + /* stop BG_GC if there is not enough free sections or suspending/freezing. */
> + if (gc_type == BG_GC && (has_not_enough_free_secs(sbi, 0, 0) ||
> + f2fs_is_suspending(sbi)))
> return submitted;
>
> if (check_valid_map(sbi, segno, off) == 0)
> @@ -1611,7 +1613,8 @@ static int gc_data_segment(struct f2fs_sb_info *sbi, struct f2fs_summary *sum,
> * Or, stop GC if the segment becomes fully valid caused by
> * race condition along with SSR block allocation.
> */
> - if ((gc_type == BG_GC && has_not_enough_free_secs(sbi, 0, 0)) ||
> + if ((gc_type == BG_GC && (has_not_enough_free_secs(sbi, 0, 0) ||
> + f2fs_is_suspending(sbi))) ||
> (!force_migrate && get_valid_blocks(sbi, segno, true) ==
> CAP_BLKS_PER_SEC(sbi)))
> return submitted;
> @@ -2015,7 +2018,7 @@ int f2fs_gc(struct f2fs_sb_info *sbi, struct f2fs_gc_control *gc_control)
> goto stop;
> }
> retry:
> - if (unlikely(freezing(current))) {
> + if (f2fs_is_suspending(sbi)) {
> ret = 0;
> goto stop;
> }
> diff --git a/fs/f2fs/segment.c b/fs/f2fs/segment.c
> index d71ddb3ee918..7b03b3d06161 100644
> --- a/fs/f2fs/segment.c
> +++ b/fs/f2fs/segment.c
> @@ -1300,7 +1300,7 @@ static int __submit_discard_cmd(struct f2fs_sb_info *sbi,
> if (dc->state != D_PREP)
> return 0;
>
> - if (is_sbi_flag_set(sbi, SBI_NEED_FSCK))
> + if (is_sbi_flag_set(sbi, SBI_NEED_FSCK) || f2fs_is_suspending(sbi))
> return 0;
>
> #ifdef CONFIG_BLK_DEV_ZONED
> @@ -1341,6 +1341,9 @@ static int __submit_discard_cmd(struct f2fs_sb_info *sbi,
> unsigned long flags;
> bool last = true;
>
> + if (f2fs_is_suspending(sbi))
> + break;
> +
> if (len > max_discard_blocks) {
> len = max_discard_blocks;
> last = false;
> @@ -1615,7 +1618,7 @@ static void __issue_discard_cmd_orderly(struct f2fs_sb_info *sbi,
> if (dc->state != D_PREP)
> goto next;
>
> - if (*issued > 0 && unlikely(freezing(current)))
> + if (f2fs_is_suspending(sbi))
> break;
>
> if (dpolicy->io_aware && !is_idle(sbi, DISCARD_TIME)) {
> @@ -1688,7 +1691,7 @@ static int __issue_discard_cmd(struct f2fs_sb_info *sbi,
> list_for_each_entry_safe(dc, tmp, pend_list, list) {
> f2fs_bug_on(sbi, dc->state != D_PREP);
>
> - if (issued > 0 && unlikely(freezing(current))) {
> + if (f2fs_is_suspending(sbi)) {
> suspended = true;
> break;
> }
> @@ -1955,7 +1958,8 @@ static int issue_discard_thread(void *data)
> continue;
> if (kthread_should_stop())
> return 0;
> - if (is_sbi_flag_set(sbi, SBI_NEED_FSCK) ||
> + if (f2fs_is_suspending(sbi) ||
> + is_sbi_flag_set(sbi, SBI_NEED_FSCK) ||
> !atomic_read(&dcc->discard_cmd_cnt)) {
> wait_ms = dpolicy.max_interval;
> continue;
> diff --git a/fs/f2fs/super.c b/fs/f2fs/super.c
> index c448d992ff2a..8c97b5d1ee68 100644
> --- a/fs/f2fs/super.c
> +++ b/fs/f2fs/super.c
> @@ -1979,6 +1979,26 @@ static void destroy_device_list(struct f2fs_sb_info *sbi)
> kvfree(sbi->devs);
> }
>
> +static int f2fs_pm_notifier(struct notifier_block *nb,
> + unsigned long action, void *ptr)
> +{
> + struct f2fs_sb_info *sbi = container_of(nb, struct f2fs_sb_info, pm_nb);
> +
> + switch (action) {
> + case PM_HIBERNATION_PREPARE:
> + case PM_SUSPEND_PREPARE:
> + case PM_RESTORE_PREPARE:
> + set_sbi_flag(sbi, SBI_IS_SUSPENDING);
> + break;
> + case PM_POST_SUSPEND:
> + case PM_POST_HIBERNATION:
> + case PM_POST_RESTORE:
> + clear_sbi_flag(sbi, SBI_IS_SUSPENDING);
> + break;
> + }
> + return NOTIFY_OK;
> +}
> +
> static void f2fs_put_super(struct super_block *sb)
> {
> struct f2fs_sb_info *sbi = F2FS_SB(sb);
> @@ -1986,6 +2006,8 @@ static void f2fs_put_super(struct super_block *sb)
> int err = 0;
> bool done;
>
> + unregister_pm_notifier(&sbi->pm_nb);
> +
> /* unregister procfs/sysfs entries in advance to avoid race case */
> f2fs_unregister_sysfs(sbi);
>
> @@ -5436,6 +5458,9 @@ static int f2fs_fill_super(struct super_block *sb, struct fs_context *fc)
> f2fs_update_time(sbi, REQ_TIME);
> clear_sbi_flag(sbi, SBI_CP_DISABLED_QUICK);
>
> + sbi->pm_nb.notifier_call = f2fs_pm_notifier;
> + register_pm_notifier(&sbi->pm_nb);
> +
> sbi->umount_lock_holder = NULL;
> return 0;
>
> --
> 2.55.0.654.g21b8a5bc05-goog
>
>
---end quoted text---
On 8/10/26 8:54 AM, Christoph Hellwig wrote: > On Thu, Aug 06, 2026 at 10:01:08AM -0700, Daeho Jeong wrote: >> From: Daeho Jeong <daehojeong@google.com> >> >> During system suspend, a race condition can cause f2fs_gc and f2fs_discard >> threads to call submit_bio() while the underlying block device (e.g., UFS) >> is in Runtime PM suspend. Because Runtime PM worker threads are already >> frozen during task freezing, the threads become trapped in >> __bio_queue_enter() waiting on mq_freeze_wq, leading to a PM freezer >> timeout. > > That does sound like a general issue with our block device / threading > handling. Daeho's description above mixes up unrelated topics. A runtime suspended UFS device is resumed automatically by the block layer if necessary. The issue Daeho is trying to solve is unrelated to runtime suspend according to my understanding. >> To prevent this deadlock, register a PM notifier to set SBI_IS_SUSPENDING >> during PM_SUSPEND_PREPARE. Background GC and discard threads check this >> flag and immediately stop issuing new bios, allowing them to enter a >> freezable sleep state cleanly before process freezing begins. >> >> In addition, check freezing() as a fast path to stop issuing new I/O >> when non-PM freezing (e.g. dm-snapshot or cgroup freezer) is requested. > > .. which means that we really sould have all the relevant parties > invited into figuring out whast is happening here, rather than > band-aiding something that looks like a horrible hack inside a > file system. > > Unfortunately I see this a lot with f2fs. Please reach out to all > relevant maintainers for something that does not look strictly local > to f2fs. This information was shared earlier with Daeho (https://b.corp.google.com/issues/515470309#comment45): [ ... ] Register a Freezable Kernel Thread (Recommended) [ ... ] Why this works: During suspend, freezer stops freezable kernel threads before devices and block queues enter their PM suspend phases. When the queue freezes later, your thread is already safely sleeping in try_to_freeze() and will not attempt submit_bio(). [ ... ] PS: I'm no longer subscribed to the f2fs-devel mailing list. Bart.
On Mon, Aug 10, 2026 at 9:34 AM Bart Van Assche <bvanassche@acm.org> wrote: > > On 8/10/26 8:54 AM, Christoph Hellwig wrote: > > On Thu, Aug 06, 2026 at 10:01:08AM -0700, Daeho Jeong wrote: > >> From: Daeho Jeong <daehojeong@google.com> > >> > >> During system suspend, a race condition can cause f2fs_gc and f2fs_discard > >> threads to call submit_bio() while the underlying block device (e.g., UFS) > >> is in Runtime PM suspend. Because Runtime PM worker threads are already > >> frozen during task freezing, the threads become trapped in > >> __bio_queue_enter() waiting on mq_freeze_wq, leading to a PM freezer > >> timeout. > > > > That does sound like a general issue with our block device / threading > > handling. > > Daeho's description above mixes up unrelated topics. A runtime suspended > UFS device is resumed automatically by the block layer if necessary. The > issue Daeho is trying to solve is unrelated to runtime suspend according > to my understanding. > > >> To prevent this deadlock, register a PM notifier to set SBI_IS_SUSPENDING > >> during PM_SUSPEND_PREPARE. Background GC and discard threads check this > >> flag and immediately stop issuing new bios, allowing them to enter a > >> freezable sleep state cleanly before process freezing begins. > >> > >> In addition, check freezing() as a fast path to stop issuing new I/O > >> when non-PM freezing (e.g. dm-snapshot or cgroup freezer) is requested. > > > > .. which means that we really sould have all the relevant parties > > invited into figuring out whast is happening here, rather than > > band-aiding something that looks like a horrible hack inside a > > file system. > > > > Unfortunately I see this a lot with f2fs. Please reach out to all > > relevant maintainers for something that does not look strictly local > > to f2fs. > > This information was shared earlier with Daeho > (https://b.corp.google.com/issues/515470309#comment45): [ ... ] > Register a Freezable Kernel Thread (Recommended) [ ... ] > Why this works: During suspend, freezer stops freezable kernel threads > before devices and block queues enter their PM suspend phases. When the > queue freezes later, your thread is already safely sleeping in > try_to_freeze() and will not attempt submit_bio(). [ ... ] Hi Bart, To clarify it: f2fs threads are already freezable: f2fs_gc and f2fs_discard are already registered with set_freezable() and call try_to_freeze(). However, a race window exists: a thread checks freezing() (false), calls submit_bio(), and gets trapped inside __bio_queue_enter(). Because it gets blocked before reaching try_to_freeze(), it triggers a PM freezer timeout. Hi Christoph, Point taken. Aside from this f2fs patch, I agree that addressing this at the block layer or PM subsystem level would be a much cleaner, system-wide solution. I’ll give more thought to how we can properly solve this race condition for the entire system, and I'll loop in the relevant PM/block maintainers if a viable generic approach emerges. Thank you, > > PS: I'm no longer subscribed to the f2fs-devel mailing list. > > Bart.
On Mon, Aug 10, 2026 at 09:53:14AM -0700, Daeho Jeong wrote: > On Mon, Aug 10, 2026 at 9:34 AM Bart Van Assche <bvanassche@acm.org> wrote: > > This information was shared earlier with Daeho > > (https://b.corp.google.com/issues/515470309#comment45): [ ... ] This does not seem to be information available to the public. > f2fs threads are already freezable: f2fs_gc and f2fs_discard are > already registered with set_freezable() and call try_to_freeze(). > However, a race window exists: a thread checks freezing() (false), > calls submit_bio(), and gets trapped inside __bio_queue_enter(). > Because it gets blocked before reaching try_to_freeze(), it triggers a > PM freezer timeout. This does sound very much like a block layer freezing issue. > Point taken. Aside from this f2fs patch, I agree that addressing this > at the block layer or PM subsystem level would be a much cleaner, > system-wide solution. > I’ll give more thought to how we can properly solve this race > condition for the entire system, and I'll loop in the relevant > PM/block maintainers if a viable generic approach emerges. Thanks, it would be great to fix this properly.
On Thu, Aug 13, 2026 at 11:30 PM Christoph Hellwig <hch@infradead.org> wrote: > > On Mon, Aug 10, 2026 at 09:53:14AM -0700, Daeho Jeong wrote: > > On Mon, Aug 10, 2026 at 9:34 AM Bart Van Assche <bvanassche@acm.org> wrote: > > > This information was shared earlier with Daeho > > > (https://b.corp.google.com/issues/515470309#comment45): [ ... ] > > This does not seem to be information available to the public. > > > f2fs threads are already freezable: f2fs_gc and f2fs_discard are > > already registered with set_freezable() and call try_to_freeze(). > > However, a race window exists: a thread checks freezing() (false), > > calls submit_bio(), and gets trapped inside __bio_queue_enter(). > > Because it gets blocked before reaching try_to_freeze(), it triggers a > > PM freezer timeout. > > This does sound very much like a block layer freezing issue. > > > Point taken. Aside from this f2fs patch, I agree that addressing this > > at the block layer or PM subsystem level would be a much cleaner, > > system-wide solution. > > I’ll give more thought to how we can properly solve this race > > condition for the entire system, and I'll loop in the relevant > > PM/block maintainers if a viable generic approach emerges. > > Thanks, it would be great to fix this properly. Now, I am considering modifying block/blk-core.c to introduce freezer-aware error handling in __bio_queue_enter(). The idea is to have __bio_queue_enter() check whether task freezing is active before it decides to block on mq_freeze_wq. If the system is in the process of freezing, it would immediately return an error instead of waiting indefinitely on a frozen pm_wq. I would greatly appreciate your thoughts on this approach. Thank you in advance for your time and insights. >
On 8/7/26 01:01, Daeho Jeong wrote: > From: Daeho Jeong <daehojeong@google.com> > > During system suspend, a race condition can cause f2fs_gc and f2fs_discard > threads to call submit_bio() while the underlying block device (e.g., UFS) > is in Runtime PM suspend. Because Runtime PM worker threads are already > frozen during task freezing, the threads become trapped in > __bio_queue_enter() waiting on mq_freeze_wq, leading to a PM freezer > timeout. > > To prevent this deadlock, register a PM notifier to set SBI_IS_SUSPENDING > during PM_SUSPEND_PREPARE. Background GC and discard threads check this > flag and immediately stop issuing new bios, allowing them to enter a > freezable sleep state cleanly before process freezing begins. > > In addition, check freezing() as a fast path to stop issuing new I/O > when non-PM freezing (e.g. dm-snapshot or cgroup freezer) is requested. > > Signed-off-by: Daeho Jeong <daehojeong@google.com> Reviewed-by: Chao Yu <chao@kernel.org> Thanks,
© 2016 - 2026 Red Hat, Inc.