kernel/events/core.c | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-)
The following commit has been merged into the perf/urgent branch of tip:
Commit-ID: 3d8d74100954a3b17e5c5e37adfe14e16b1db103
Gitweb: https://git.kernel.org/tip/3d8d74100954a3b17e5c5e37adfe14e16b1db103
Author: Puranjay Mohan <puranjay@kernel.org>
AuthorDate: Mon, 10 Aug 2026 06:35:35 -07:00
Committer: Peter Zijlstra <peterz@infradead.org>
CommitterDate: Wed, 23 Sep 2026 11:48:35 +02:00
perf/core: Run sched_task() for PMUs with only CPU-wide events
perf_pmu_sched_task() returns early when cpuctx->task_ctx is set and
leaves the work to perf_ctx_sched_task_cb(), which only walks
ctx->pmu_ctx_list. A PMU whose events are all CPU-wide is not on that
list, so nothing calls its sched_task(). With
perf record -b -e cycles -a -- ls
armv8pmu_sched_task() is skipped on every switch to a task that has a
perf context but no event on that PMU, and BRBE records leak across the
task boundary. intel_pmu_lbr_add() calls perf_sched_cb_inc()
unconditionally too, so LBR records leak the same way on x86.
Drop the early return and skip only the CPCs that
perf_ctx_sched_task_cb() handles. That one needs a gate of its own to
make the split exact: it tests cpc->sched_cb_usage, which
perf_sched_cb_inc() sets per CPU for every branch stack user, so a task
with an event for that PMU pinned to another CPU would be handled twice.
On x86 the second __intel_pmu_lbr_restore() finds lbr_stack_state ==
LBR_NONE and calls intel_pmu_lbr_reset(), throwing away the callstack
the first one restored.
cpc->task_epc is set only while a task context is scheduled in, and
there is one epc per PMU on ctx->pmu_ctx_list, so the two gates are
inverses.
For the CPCs perf_pmu_sched_task() picks up, the callback now runs
outside the perf_ctx_disable() and perf_ctx_enable() pair in
perf_event_context_sched_in(). __perf_pmu_sched_task() disables the PMU
around the call itself.
Fixes: bd2756811766 ("perf: Rewrite core context handling")
Signed-off-by: Puranjay Mohan <puranjay@kernel.org>
Signed-off-by: Peter Zijlstra (Intel) <peterz@infradead.org>
Tested-by: Yifan Wu <wuyifan50@huawei.com>
Link: https://patch.msgid.link/20260810133540.1947118-3-puranjay@kernel.org
Cc: stable@vger.kernel.org
---
kernel/events/core.c | 13 +++++++++----
1 file changed, 9 insertions(+), 4 deletions(-)
diff --git a/kernel/events/core.c b/kernel/events/core.c
index 7ce72f3..634d2cc 100644
--- a/kernel/events/core.c
+++ b/kernel/events/core.c
@@ -3764,6 +3764,9 @@ static void perf_ctx_sched_task_cb(struct perf_event_context *ctx,
list_for_each_entry(pmu_ctx, &ctx->pmu_ctx_list, pmu_ctx_entry) {
cpc = this_cpc(pmu_ctx->pmu);
+ if (cpc->task_epc != pmu_ctx)
+ continue;
+
if (cpc->sched_cb_usage && pmu_ctx->pmu->sched_task)
pmu_ctx->pmu->sched_task(pmu_ctx, task, sched_in);
}
@@ -3924,15 +3927,17 @@ static void perf_pmu_sched_task(struct task_struct *prev,
struct task_struct *next,
bool sched_in)
{
- struct perf_cpu_context *cpuctx = this_cpu_ptr(&perf_cpu_context);
struct perf_cpu_pmu_context *cpc, *cpc2;
- /* cpuctx->task_ctx will be handled in perf_event_context_sched_in/out */
- if (prev == next || cpuctx->task_ctx)
+ if (prev == next)
return;
- list_for_each_entry_safe(cpc, cpc2, this_cpu_ptr(&sched_cb_list), sched_cb_entry)
+ list_for_each_entry_safe(cpc, cpc2, this_cpu_ptr(&sched_cb_list), sched_cb_entry) {
+ if (cpc->task_epc)
+ continue;
+
__perf_pmu_sched_task(cpc, sched_in ? next : prev, sched_in);
+ }
}
static void perf_event_switch(struct task_struct *task,
© 2016 - 2026 Red Hat, Inc.