kernel/sched/fair.c | 31 ++++++++++++++++++++++++++++++- 1 file changed, 30 insertions(+), 1 deletion(-)
Pipe and futex ping-pong workloads can be dominated by handoff cost. In
such cases, placing the wakee on an idle CPU can be slower than keeping
the pair on the same runqueue.
Use the existing last_wakee and wake_wide() state to identify narrow
reciprocal WF_SYNC wakeups:
A wakes B
B wakes A
A wakes B
...
For these wakeups, keep the wakee on the waker CPU only when this looks
like a narrow ping-pong: the wakee recently woke the current task,
wake_wide() does not classify the relationship as wide, and the waker CPU
has no other runnable CFS task. The wakee must also be allowed to run on
that CPU, and on asymmetric-capacity systems it must fit there.
Wakeups that do not match this pattern continue through the existing
wake_affine() and select_idle_sibling() path.
Signed-off-by: Shubhang Kaushik (Ampere) <sh@gentwo.org>
---
Tested on 80-core Ampere Altra: perf bench sched pipe -l 1000000 improved
about 21%, averaged over 20 runs. Hackbench, schbench and SPECjBB showed
no material regression.
Baseline ~ v7.2-rc4 (mainline origin/master at b95f03f04d47)
---
kernel/sched/fair.c | 31 ++++++++++++++++++++++++++++++-
1 file changed, 30 insertions(+), 1 deletion(-)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index d78467ec6ee1343050fcc2794dafb38ade3599e5..d9afc89bc7a45b0b05ebad536f9a31d5757a2753 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -8794,6 +8794,28 @@ static inline bool asym_fits_cpu(unsigned long util,
return true;
}
+/* For reciprocal WF_SYNC handoffs, prefer the waker CPU when it has no
+ * other runnable fair task.
+ */
+static bool prefer_sync_pair_cpu(struct task_struct *p, int cpu)
+{
+ struct rq *rq = cpu_rq(cpu);
+
+ if ((rq->nr_running - cfs_h_nr_delayed(rq)) != 1)
+ return false;
+
+ if (!cpumask_test_cpu(cpu, p->cpus_ptr))
+ return false;
+
+ if (sched_asym_cpucap_active()) {
+ sync_entity_load_avg(&p->se);
+ if (!task_fits_cpu(p, cpu))
+ return false;
+ }
+
+ return true;
+}
+
/*
* Try and locate an idle core/thread in the LLC cache domain.
*/
@@ -9548,6 +9570,7 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
int cpu = smp_processor_id();
int new_cpu = prev_cpu;
int want_affine = 0;
+ int wide = 0;
/* SD_flags and WF_flags share the first nibble */
int sd_flag = wake_flags & 0xF;
@@ -9557,6 +9580,7 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
lockdep_assert_held(&p->pi_lock);
if (wake_flags & WF_TTWU) {
record_wakee(p);
+ wide = wake_wide(p);
if ((wake_flags & WF_CURRENT_CPU) &&
cpumask_test_cpu(cpu, p->cpus_ptr))
@@ -9569,7 +9593,12 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
new_cpu = prev_cpu;
}
- want_affine = !wake_wide(p) && cpumask_test_cpu(cpu, p->cpus_ptr);
+ if (sync && !wide &&
+ READ_ONCE(p->last_wakee) == current &&
+ prefer_sync_pair_cpu(p, cpu))
+ return cpu;
+
+ want_affine = !wide && cpumask_test_cpu(cpu, p->cpus_ptr);
}
for_each_domain(cpu, tmp) {
---
base-commit: b95f03f04d475aa6719d15a636ddf32222d55657
change-id: 20260721-b4-sched-sync-wakeup-04d40cbeb1da
Best regards,
--
Shubhang Kaushik (Ampere) <sh@gentwo.org>
On 7/22/26 01:04, Shubhang Kaushik (Ampere) wrote:
> Pipe and futex ping-pong workloads can be dominated by handoff cost. In
> such cases, placing the wakee on an idle CPU can be slower than keeping
> the pair on the same runqueue.
>
> Use the existing last_wakee and wake_wide() state to identify narrow
> reciprocal WF_SYNC wakeups:
>
> A wakes B
> B wakes A
> A wakes B
> ...
But the futex ping-pong doesn't set WF_SYNC in the first place or am I missing
something?
>
> For these wakeups, keep the wakee on the waker CPU only when this looks
> like a narrow ping-pong: the wakee recently woke the current task,
> wake_wide() does not classify the relationship as wide, and the waker CPU
> has no other runnable CFS task. The wakee must also be allowed to run on
> that CPU, and on asymmetric-capacity systems it must fit there.
>
> Wakeups that do not match this pattern continue through the existing
> wake_affine() and select_idle_sibling() path.
>
> Signed-off-by: Shubhang Kaushik (Ampere) <sh@gentwo.org>
> ---
> Tested on 80-core Ampere Altra: perf bench sched pipe -l 1000000 improved
> about 21%, averaged over 20 runs. Hackbench, schbench and SPECjBB showed
> no material regression.
>
> Baseline ~ v7.2-rc4 (mainline origin/master at b95f03f04d47)
> ---
> kernel/sched/fair.c | 31 ++++++++++++++++++++++++++++++-
> 1 file changed, 30 insertions(+), 1 deletion(-)
>
> diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
> index d78467ec6ee1343050fcc2794dafb38ade3599e5..d9afc89bc7a45b0b05ebad536f9a31d5757a2753 100644
> --- a/kernel/sched/fair.c
> +++ b/kernel/sched/fair.c
> @@ -8794,6 +8794,28 @@ static inline bool asym_fits_cpu(unsigned long util,
> return true;
> }
>
> +/* For reciprocal WF_SYNC handoffs, prefer the waker CPU when it has no
> + * other runnable fair task.
> + */
> +static bool prefer_sync_pair_cpu(struct task_struct *p, int cpu)
> +{
> + struct rq *rq = cpu_rq(cpu);
> +
> + if ((rq->nr_running - cfs_h_nr_delayed(rq)) != 1)
> + return false;
> +
> + if (!cpumask_test_cpu(cpu, p->cpus_ptr))
> + return false;
> +
> + if (sched_asym_cpucap_active()) {
> + sync_entity_load_avg(&p->se);
> + if (!task_fits_cpu(p, cpu))
> + return false;
> + }
> +
> + return true;
> +}
> +
> /*
> * Try and locate an idle core/thread in the LLC cache domain.
> */
> @@ -9548,6 +9570,7 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
> int cpu = smp_processor_id();
> int new_cpu = prev_cpu;
> int want_affine = 0;
> + int wide = 0;
> /* SD_flags and WF_flags share the first nibble */
> int sd_flag = wake_flags & 0xF;
>
> @@ -9557,6 +9580,7 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
> lockdep_assert_held(&p->pi_lock);
> if (wake_flags & WF_TTWU) {
> record_wakee(p);
> + wide = wake_wide(p);
>
> if ((wake_flags & WF_CURRENT_CPU) &&
> cpumask_test_cpu(cpu, p->cpus_ptr))
> @@ -9569,7 +9593,12 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
> new_cpu = prev_cpu;
> }
>
> - want_affine = !wake_wide(p) && cpumask_test_cpu(cpu, p->cpus_ptr);
> + if (sync && !wide &&
> + READ_ONCE(p->last_wakee) == current &&
> + prefer_sync_pair_cpu(p, cpu))
> + return cpu;
This must move further below (to check for SD_WAKE_AFFINE) otherwise it ignores isolcpus.
> +
> + want_affine = !wide && cpumask_test_cpu(cpu, p->cpus_ptr);
> }
>
> for_each_domain(cpu, tmp) {
>
> ---
> base-commit: b95f03f04d475aa6719d15a636ddf32222d55657
> change-id: 20260721-b4-sched-sync-wakeup-04d40cbeb1da
>
> Best regards,
Hi Christian,
Thanks for the review.
On Wed, 22 Jul 2026, Christian Loehle wrote:
> On 7/22/26 01:04, Shubhang Kaushik (Ampere) wrote:
>> Pipe and futex ping-pong workloads can be dominated by handoff cost. In
>> such cases, placing the wakee on an idle CPU can be slower than keeping
>> the pair on the same runqueue.
>>
>> Use the existing last_wakee and wake_wide() state to identify narrow
>> reciprocal WF_SYNC wakeups:
>>
>> A wakes B
>> B wakes A
>> A wakes B
>> ...
>
> But the futex ping-pong doesn't set WF_SYNC in the first place or am I missing
> something?
Yes, the futex runs were only regression checks here. I'll drop the futex
wording from the description and keep the motivation to pipe sync
handoffs.
>> @@ -9548,6 +9570,7 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
>> int cpu = smp_processor_id();
>> int new_cpu = prev_cpu;
>> int want_affine = 0;
>> + int wide = 0;
>> /* SD_flags and WF_flags share the first nibble */
>> int sd_flag = wake_flags & 0xF;
>>
>> @@ -9557,6 +9580,7 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
>> lockdep_assert_held(&p->pi_lock);
>> if (wake_flags & WF_TTWU) {
>> record_wakee(p);
>> + wide = wake_wide(p);
>>
>> if ((wake_flags & WF_CURRENT_CPU) &&
>> cpumask_test_cpu(cpu, p->cpus_ptr))
>> @@ -9569,7 +9593,12 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
>> new_cpu = prev_cpu;
>> }
>>
>> - want_affine = !wake_wide(p) && cpumask_test_cpu(cpu, p->cpus_ptr);
>> + if (sync && !wide &&
>> + READ_ONCE(p->last_wakee) == current &&
>> + prefer_sync_pair_cpu(p, cpu))
>> + return cpu;
>
> This must move further below (to check for SD_WAKE_AFFINE) otherwise it ignores isolcpus.
Agreed. The early return bypasses the existing SD_WAKE_AFFINE check,
including the isolcpus handling that comes through the sched-domain
setup. I'll move it into the wake-affine block so it only applies after
that check.
Regards,
Shubhang Kaushik
© 2016 - 2026 Red Hat, Inc.