kernel/softirq.c | 48 ++++++++++++++++++++++++++++++++++++------------ 1 file changed, 36 insertions(+), 12 deletions(-)
__irq_exit_rcu() drops HARDIRQ_OFFSET before deferred hrtimer rearm,
softirq dispatch, and timersd wakeup. This work is still on the IRQ
return path, but in_task() reports task context.
Context-sensitive code called from this window therefore sees task
context. ftrace records normal-context flags and selects its normal
recursion slot. KCSAN attributes IRQ-exit accesses to the interrupted
task, while KMSAN can select and modify that task's metadata.
Keep HARDIRQ_OFFSET until the IRQ-exit work is done. For direct softirq
handling, replace it with SOFTIRQ_OFFSET and restore it afterwards. Other
__do_softirq() call paths keep their existing accounting.
With hardirq context retained, ftrace records hardirq context, KCSAN uses
interrupt attribution, and KMSAN no longer uses the interrupted task's
state.
Since softirq eligibility is now tested before HARDIRQ_OFFSET is removed,
use irq_count() == HARDIRQ_OFFSET. This preserves the old !in_interrupt()
semantics, including PREEMPT_RT's task-local softirq-disable state.
Drop HARDIRQ_OFFSET before tick_irq_exit(), as before.
Suggested-by: Peter Zijlstra <peterz@infradead.org>
Link: https://lore.kernel.org/r/20260813130826.GW687043@noisy.programming.kicks-ass.net
Assisted-by: LLM
Signed-off-by: Karl Mehltretter <kmehltretter@gmail.com>
---
I reworked Peter's draft linked above into this version and tested it.
Changes:
- Use in_hardirq() in softirq_handle_begin() instead of ksirqd, since
__do_softirq() also has task-context callers, including ktimerd.
- Use irq_count() for the eligibility test so PREEMPT_RT's task-local
BH-disabled state remains part of the decision.
Tested with non-RT, threadirqs and PREEMPT_RT x86-64 QEMU boot/stress.
The QEMU kernels had lockdep and IRQ tracing enabled and reported no new
warnings.
A focused RT test rejected every BH-disabled IRQ-exit observation. The
old-mask negative control admitted every one.
ftrace marked direct IRQ-exit work as hardirq rather than normal context.
A separate A/B changed printk caller attribution from task to CPU,
in_task() from 1 to 0 and interrupt_context_level() from 0 to 2. Fault
injection no longer consumed the interrupted task's fail_nth state.
KCSAN attributed all 16 target reports to interrupt context.
KMSAN did not select or change task state in 64K IRQ-exit windows. All
28 KMSAN KUnit tests passed.
The exact TIP source also passed A/B boot/stress on a Pi 400
(Cortex-A72, arm64).
For additional coverage, the mainline adaptation passed A/B boot/stress
on a Microchip SAM9X75 Curiosity (ARM926EJ-S/ARMv5TEJ).
vmlinux linked successfully for arm64, ARM, RISC-V and s390.
kernel/softirq.c | 48 ++++++++++++++++++++++++++++++++++++------------
1 file changed, 36 insertions(+), 12 deletions(-)
diff --git a/kernel/softirq.c b/kernel/softirq.c
index 5d02c36c40e3..63aeaa5f62e9 100644
--- a/kernel/softirq.c
+++ b/kernel/softirq.c
@@ -350,8 +350,8 @@ static inline void ksoftirqd_run_end(void)
local_irq_enable();
}
-static inline void softirq_handle_begin(void) { }
-static inline void softirq_handle_end(void) { }
+static inline bool softirq_handle_begin(void) { return false; }
+static inline void softirq_handle_end(bool from_hardirq) { }
static inline bool should_wake_ksoftirqd(void)
{
@@ -481,15 +481,35 @@ void __local_bh_enable_ip(unsigned long ip, unsigned int cnt)
}
EXPORT_SYMBOL(__local_bh_enable_ip);
-static inline void softirq_handle_begin(void)
+static inline bool softirq_handle_begin(void)
{
- __local_bh_disable_ip(_RET_IP_, SOFTIRQ_OFFSET);
+ bool from_hardirq = in_hardirq();
+
+ if (!from_hardirq) {
+ __local_bh_disable_ip(_RET_IP_, SOFTIRQ_OFFSET);
+ return false;
+ }
+
+ /* Replace the retained hardirq context with normal softirq context. */
+ __preempt_count_add((int)SOFTIRQ_OFFSET - (int)HARDIRQ_OFFSET);
+ if (softirq_count() == SOFTIRQ_OFFSET)
+ lockdep_softirqs_off(_RET_IP_);
+
+ return true;
}
-static inline void softirq_handle_end(void)
+static inline void softirq_handle_end(bool from_hardirq)
{
- __local_bh_enable(SOFTIRQ_OFFSET);
- WARN_ON_ONCE(in_interrupt());
+ if (!from_hardirq) {
+ __local_bh_enable(SOFTIRQ_OFFSET);
+ WARN_ON_ONCE(in_interrupt());
+ return;
+ }
+
+ if (softirq_count() == SOFTIRQ_OFFSET)
+ lockdep_softirqs_on(_RET_IP_);
+ __preempt_count_sub((int)SOFTIRQ_OFFSET - (int)HARDIRQ_OFFSET);
+ WARN_ON_ONCE(!in_hardirq());
}
static inline void ksoftirqd_run_begin(void)
@@ -605,6 +625,7 @@ static void handle_softirqs(bool ksirqd)
unsigned long old_flags = current->flags;
int max_restart = MAX_SOFTIRQ_RESTART;
struct softirq_action *h;
+ bool from_hardirq;
bool in_hardirq;
__u32 pending;
int softirq_bit;
@@ -618,7 +639,7 @@ static void handle_softirqs(bool ksirqd)
pending = local_softirq_pending();
- softirq_handle_begin();
+ from_hardirq = softirq_handle_begin();
in_hardirq = lockdep_softirq_start();
account_softirq_enter(current);
@@ -670,7 +691,7 @@ restart:
account_softirq_exit(current);
lockdep_softirq_end(in_hardirq);
- softirq_handle_end();
+ softirq_handle_end(from_hardirq);
current_restore_flags(old_flags, PF_MEMALLOC);
}
@@ -740,6 +761,8 @@ static inline void wake_timersd(void) { }
#endif
+#define IRQ_EXIT_TIMERS (NMI_MASK | HARDIRQ_MASK)
+
static inline void __irq_exit_rcu(void)
{
#ifndef __ARCH_IRQ_EXIT_IRQS_DISABLED
@@ -748,8 +771,7 @@ static inline void __irq_exit_rcu(void)
lockdep_assert_irqs_disabled();
#endif
account_hardirq_exit(current);
- preempt_count_sub(HARDIRQ_OFFSET);
- if (!in_interrupt() && local_softirq_pending()) {
+ if (irq_count() == HARDIRQ_OFFSET && local_softirq_pending()) {
/*
* If we left hrtimers unarmed, make sure to arm them now,
* before enabling interrupts to run softirq.
@@ -759,9 +781,11 @@ static inline void __irq_exit_rcu(void)
}
if (IS_ENABLED(CONFIG_IRQ_FORCED_THREADING) && force_irqthreads() &&
- local_timers_pending_force_th() && !(in_nmi() | in_hardirq()))
+ local_timers_pending_force_th() &&
+ (preempt_count() & IRQ_EXIT_TIMERS) == HARDIRQ_OFFSET)
wake_timersd();
+ preempt_count_sub(HARDIRQ_OFFSET);
tick_irq_exit();
}
base-commit: 2af470916a208b576ac9975d221d9a378cf8ace9
--
2.53.0
Le Thu, Sep 03, 2026 at 01:27:37PM +0200, Karl Mehltretter a écrit :
> __irq_exit_rcu() drops HARDIRQ_OFFSET before deferred hrtimer rearm,
> softirq dispatch, and timersd wakeup. This work is still on the IRQ
> return path, but in_task() reports task context.
>
> Context-sensitive code called from this window therefore sees task
> context. ftrace records normal-context flags and selects its normal
> recursion slot. KCSAN attributes IRQ-exit accesses to the interrupted
> task, while KMSAN can select and modify that task's metadata.
>
> Keep HARDIRQ_OFFSET until the IRQ-exit work is done. For direct softirq
> handling, replace it with SOFTIRQ_OFFSET and restore it afterwards. Other
> __do_softirq() call paths keep their existing accounting.
>
> With hardirq context retained, ftrace records hardirq context, KCSAN uses
> interrupt attribution, and KMSAN no longer uses the interrupted task's
> state.
>
> Since softirq eligibility is now tested before HARDIRQ_OFFSET is removed,
> use irq_count() == HARDIRQ_OFFSET. This preserves the old !in_interrupt()
> semantics, including PREEMPT_RT's task-local softirq-disable state.
>
> Drop HARDIRQ_OFFSET before tick_irq_exit(), as before.
>
> Suggested-by: Peter Zijlstra <peterz@infradead.org>
> Link: https://lore.kernel.org/r/20260813130826.GW687043@noisy.programming.kicks-ass.net
> Assisted-by: LLM
> Signed-off-by: Karl Mehltretter <kmehltretter@gmail.com>
> ---
> I reworked Peter's draft linked above into this version and tested it.
>
> Changes:
> - Use in_hardirq() in softirq_handle_begin() instead of ksirqd, since
> __do_softirq() also has task-context callers, including ktimerd.
> - Use irq_count() for the eligibility test so PREEMPT_RT's task-local
> BH-disabled state remains part of the decision.
>
> Tested with non-RT, threadirqs and PREEMPT_RT x86-64 QEMU boot/stress.
> The QEMU kernels had lockdep and IRQ tracing enabled and reported no new
> warnings.
>
> A focused RT test rejected every BH-disabled IRQ-exit observation. The
> old-mask negative control admitted every one.
>
> ftrace marked direct IRQ-exit work as hardirq rather than normal context.
>
> A separate A/B changed printk caller attribution from task to CPU,
> in_task() from 1 to 0 and interrupt_context_level() from 0 to 2. Fault
> injection no longer consumed the interrupted task's fail_nth state.
>
> KCSAN attributed all 16 target reports to interrupt context.
>
> KMSAN did not select or change task state in 64K IRQ-exit windows. All
> 28 KMSAN KUnit tests passed.
>
> The exact TIP source also passed A/B boot/stress on a Pi 400
> (Cortex-A72, arm64).
>
> For additional coverage, the mainline adaptation passed A/B boot/stress
> on a Microchip SAM9X75 Curiosity (ARM926EJ-S/ARMv5TEJ).
>
> vmlinux linked successfully for arm64, ARM, RISC-V and s390.
>
> kernel/softirq.c | 48 ++++++++++++++++++++++++++++++++++++------------
> 1 file changed, 36 insertions(+), 12 deletions(-)
>
> diff --git a/kernel/softirq.c b/kernel/softirq.c
> index 5d02c36c40e3..63aeaa5f62e9 100644
> --- a/kernel/softirq.c
> +++ b/kernel/softirq.c
> @@ -350,8 +350,8 @@ static inline void ksoftirqd_run_end(void)
> local_irq_enable();
> }
>
> -static inline void softirq_handle_begin(void) { }
> -static inline void softirq_handle_end(void) { }
> +static inline bool softirq_handle_begin(void) { return false; }
> +static inline void softirq_handle_end(bool from_hardirq) { }
>
> static inline bool should_wake_ksoftirqd(void)
> {
> @@ -481,15 +481,35 @@ void __local_bh_enable_ip(unsigned long ip, unsigned int cnt)
> }
> EXPORT_SYMBOL(__local_bh_enable_ip);
>
> -static inline void softirq_handle_begin(void)
> +static inline bool softirq_handle_begin(void)
> {
> - __local_bh_disable_ip(_RET_IP_, SOFTIRQ_OFFSET);
> + bool from_hardirq = in_hardirq();
> +
> + if (!from_hardirq) {
> + __local_bh_disable_ip(_RET_IP_, SOFTIRQ_OFFSET);
> + return false;
> + }
> +
> + /* Replace the retained hardirq context with normal softirq context. */
> + __preempt_count_add((int)SOFTIRQ_OFFSET - (int)HARDIRQ_OFFSET);
So it skips the whole RT locking and processing because softirqs don't
happen anyway on hard IRQ tail there. Looks good.
> + if (softirq_count() == SOFTIRQ_OFFSET)
Any other value should be forbidden here.
It should just warn.
> + lockdep_softirqs_off(_RET_IP_);
> + return true;
> }
>
> -static inline void softirq_handle_end(void)
> +static inline void softirq_handle_end(bool from_hardirq)
> {
> - __local_bh_enable(SOFTIRQ_OFFSET);
> - WARN_ON_ONCE(in_interrupt());
> + if (!from_hardirq) {
> + __local_bh_enable(SOFTIRQ_OFFSET);
> + WARN_ON_ONCE(in_interrupt());
> + return;
> + }
> +
> + if (softirq_count() == SOFTIRQ_OFFSET)
> + lockdep_softirqs_on(_RET_IP_);
Same here, you should warn if softirq_count() != SOFTIRQ_OFFSET
> + __preempt_count_sub((int)SOFTIRQ_OFFSET - (int)HARDIRQ_OFFSET);
> + WARN_ON_ONCE(!in_hardirq());
> }
Thanks!
--
Frederic Weisbecker
SUSE Labs
On Thu, Sep 03, 2026 at 05:22:21PM +0100, Frederic Weisbecker wrote: > > + if (softirq_count() == SOFTIRQ_OFFSET) > > Any other value should be forbidden here. > It should just warn. > > + > > + if (softirq_count() == SOFTIRQ_OFFSET) > > + lockdep_softirqs_on(_RET_IP_); > > Same here, you should warn if softirq_count() != SOFTIRQ_OFFSET > Good point! I'll send a v2. Thanks, Karl
On 2026-09-03 13:27:37 [+0200], Karl Mehltretter wrote: > __irq_exit_rcu() drops HARDIRQ_OFFSET before deferred hrtimer rearm, > softirq dispatch, and timersd wakeup. This work is still on the IRQ > return path, but in_task() reports task context. Is this new? I don't remember seeing this before. Sebastian
On Thu, Sep 03, 2026 at 02:20:17PM +0100, Sebastian Andrzej Siewior wrote: > On 2026-09-03 13:27:37 [+0200], Karl Mehltretter wrote: > > __irq_exit_rcu() drops HARDIRQ_OFFSET before deferred hrtimer rearm, > > softirq dispatch, and timersd wakeup. This work is still on the IRQ > > return path, but in_task() reports task context. > > Is this new? I don't remember seeing this before. > > Sebastian irq_exit() dropped the hardirq count before invoking softirqs in v2.6.12-rc2, the beginning of git history. The newer additions to that window are the timersd wakeup added by 49a17639508c and the deferred hrtimer-rearm hook added by 7e641e52cf5f. Thanks, Karl
© 2016 - 2026 Red Hat, Inc.