From cca4980630b3c7a85f53cb43c6018184ce5d4e37 Mon Sep 17 00:00:00 2001 From: Namhyung Kim Date: Sun, 20 Sep 2026 16:16:39 -0700 Subject: perf/core: Fix a refcount leak in attach_perf_ctx_data() The attach_perf_ctx_data() can race on global and !global cases. The global case is protected by global_ctx_data_rwsem and shares a single reference count using perf_ctx_data.global field. But when it races with !global case, it may miss to set the global field and result in a reference count leak. CPU1 CPU2 ---------------------------------------------------------------- attach_task_ctx_data(.global=1) attach_task_ctx_data(.global=0) cd1 = alloc_perf_ctx_data(); cd2 = alloc_perf_ctx_data(); // { .global = 0, .refcount = 1 }; try_cmpxchg(); // success, // task->perf_ctx_data = cd2 try_cmpxhg(); // fail; old = cd2 refcount_inc_not_zero(&old->refcount); // success // old.refcount = 2 free_perf_ctx_data(cd1); Then later detach_global_ctx_data() will see the data but it's not marked as global, so it won't call detach_task_ctx_data(). Fixes: 506e64e710ff ("perf: attach/detach PMU specific data") Assisted-by: Sashiko.dev:Gemini-3.1-pro Signed-off-by: Namhyung Kim Signed-off-by: Peter Zijlstra (Intel) Link: https://patch.msgid.link/20260920231639.11910-1-namhyung@kernel.org --- kernel/events/core.c | 2 ++ 1 file changed, 2 insertions(+) (limited to 'kernel') diff --git a/kernel/events/core.c b/kernel/events/core.c index db7b76d6b68a..e180134bad5e 100644 --- a/kernel/events/core.c +++ b/kernel/events/core.c @@ -5454,6 +5454,8 @@ attach_task_ctx_data(struct task_struct *task, struct kmem_cache *ctx_cache, } if (refcount_inc_not_zero(&old->refcount)) { + if (global) + old->global = true; free_perf_ctx_data(cd); /* unused */ return 0; } -- cgit v1.2.3 From 36bb85cf36cab15fb611cb44b78a5df06e4e69a2 Mon Sep 17 00:00:00 2001 From: Puranjay Mohan Date: Mon, 10 Aug 2026 06:35:34 -0700 Subject: perf/core: Fix NULL pmu_ctx passed to pmu->sched_task() perf_pmu_sched_task() returns early when cpuctx->task_ctx is set, and cpc->task_epc is only non-NULL while a task context is scheduled in on this CPU. __perf_pmu_sched_task() therefore always passes NULL: Unable to handle kernel NULL pointer dereference at virtual address 00 pc : armv8pmu_sched_task+0x14/0x50 Call trace: armv8pmu_sched_task+0x14/0x50 (P) perf_pmu_sched_task+0xac/0x108 __perf_event_task_sched_out+0x6c/0xe0 Pass &cpc->epc instead, the CPU-wide context for this PMU, which the function already dereferences a few lines up to find pmu. armv8pmu_sched_task() is the only in-tree implementation that dereferences the argument, and it only reads ->pmu, so the oops needs BRBE, added in v6.17. Fixes: bd2756811766 ("perf: Rewrite core context handling") Signed-off-by: Puranjay Mohan Signed-off-by: Peter Zijlstra (Intel) Tested-by: Yifan Wu Cc: stable@vger.kernel.org Link: https://patch.msgid.link/20260810133540.1947118-2-puranjay@kernel.org --- kernel/events/core.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) (limited to 'kernel') diff --git a/kernel/events/core.c b/kernel/events/core.c index e180134bad5e..7ce72f366f6c 100644 --- a/kernel/events/core.c +++ b/kernel/events/core.c @@ -3914,7 +3914,7 @@ static void __perf_pmu_sched_task(struct perf_cpu_pmu_context *cpc, perf_ctx_lock(cpuctx, cpuctx->task_ctx); perf_pmu_disable(pmu); - pmu->sched_task(cpc->task_epc, task, sched_in); + pmu->sched_task(&cpc->epc, task, sched_in); perf_pmu_enable(pmu); perf_ctx_unlock(cpuctx, cpuctx->task_ctx); -- cgit v1.2.3 From 3d8d74100954a3b17e5c5e37adfe14e16b1db103 Mon Sep 17 00:00:00 2001 From: Puranjay Mohan Date: Mon, 10 Aug 2026 06:35:35 -0700 Subject: perf/core: Run sched_task() for PMUs with only CPU-wide events perf_pmu_sched_task() returns early when cpuctx->task_ctx is set and leaves the work to perf_ctx_sched_task_cb(), which only walks ctx->pmu_ctx_list. A PMU whose events are all CPU-wide is not on that list, so nothing calls its sched_task(). With perf record -b -e cycles -a -- ls armv8pmu_sched_task() is skipped on every switch to a task that has a perf context but no event on that PMU, and BRBE records leak across the task boundary. intel_pmu_lbr_add() calls perf_sched_cb_inc() unconditionally too, so LBR records leak the same way on x86. Drop the early return and skip only the CPCs that perf_ctx_sched_task_cb() handles. That one needs a gate of its own to make the split exact: it tests cpc->sched_cb_usage, which perf_sched_cb_inc() sets per CPU for every branch stack user, so a task with an event for that PMU pinned to another CPU would be handled twice. On x86 the second __intel_pmu_lbr_restore() finds lbr_stack_state == LBR_NONE and calls intel_pmu_lbr_reset(), throwing away the callstack the first one restored. cpc->task_epc is set only while a task context is scheduled in, and there is one epc per PMU on ctx->pmu_ctx_list, so the two gates are inverses. For the CPCs perf_pmu_sched_task() picks up, the callback now runs outside the perf_ctx_disable() and perf_ctx_enable() pair in perf_event_context_sched_in(). __perf_pmu_sched_task() disables the PMU around the call itself. Fixes: bd2756811766 ("perf: Rewrite core context handling") Signed-off-by: Puranjay Mohan Signed-off-by: Peter Zijlstra (Intel) Tested-by: Yifan Wu Link: https://patch.msgid.link/20260810133540.1947118-3-puranjay@kernel.org Cc: stable@vger.kernel.org --- kernel/events/core.c | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) (limited to 'kernel') diff --git a/kernel/events/core.c b/kernel/events/core.c index 7ce72f366f6c..634d2ccbab82 100644 --- a/kernel/events/core.c +++ b/kernel/events/core.c @@ -3764,6 +3764,9 @@ static void perf_ctx_sched_task_cb(struct perf_event_context *ctx, list_for_each_entry(pmu_ctx, &ctx->pmu_ctx_list, pmu_ctx_entry) { cpc = this_cpc(pmu_ctx->pmu); + if (cpc->task_epc != pmu_ctx) + continue; + if (cpc->sched_cb_usage && pmu_ctx->pmu->sched_task) pmu_ctx->pmu->sched_task(pmu_ctx, task, sched_in); } @@ -3924,15 +3927,17 @@ static void perf_pmu_sched_task(struct task_struct *prev, struct task_struct *next, bool sched_in) { - struct perf_cpu_context *cpuctx = this_cpu_ptr(&perf_cpu_context); struct perf_cpu_pmu_context *cpc, *cpc2; - /* cpuctx->task_ctx will be handled in perf_event_context_sched_in/out */ - if (prev == next || cpuctx->task_ctx) + if (prev == next) return; - list_for_each_entry_safe(cpc, cpc2, this_cpu_ptr(&sched_cb_list), sched_cb_entry) + list_for_each_entry_safe(cpc, cpc2, this_cpu_ptr(&sched_cb_list), sched_cb_entry) { + if (cpc->task_epc) + continue; + __perf_pmu_sched_task(cpc, sched_in ? next : prev, sched_in); + } } static void perf_event_switch(struct task_struct *task, -- cgit v1.2.3