From d35a535d3e3e71f92a051d94e415f43612c4edfc Mon Sep 17 00:00:00 2001 From: Tejun Heo Date: Wed, 23 Sep 2026 12:39:20 -1000 Subject: sched_ext: Fix CPU hotplug hang when a dying CPU's tasks sit in the BPF scheduler A CPU going down has to empty its own rq. Its hotplug thread waits in sched_cpu_wait_empty() until nothing else is left, and the only thing that wakes it is balance_push(), which runs from __schedule() on the dying CPU and pushes off the migratable tasks the CPU picks. sched_ext breaks this. A task sitting on a user DSQ or held by the BPF scheduler is still counted on its rq, but an inactive CPU no longer calls ops.dispatch() and can't pull the task back. The dying CPU goes idle with the task still counted, and one of two things happens: - Another CPU consumes the task. The rq empties without a __schedule() on the dying CPU, so the hotplug thread is never woken and cpu_down() hangs holding cpu_hotplug_lock. - The task is affine only to the dying CPU. Nothing can move it, and the offline stalls until the watchdog ejects the BPF scheduler. When the rq goes offline, re-enqueue every task on it that isn't already on the local DSQ. An enqueue on an offline rq lands on the local DSQ, so the dying CPU itself picks the tasks and balance_push() pushes them off, the same as for the other sched classes. From then on no sched_ext path on another CPU can pull a task off the rq. ops.dispatch() currently stops as soon as the CPU goes inactive, and CPU hotplug then waits for an RCU grace period before taking the rq offline. A BPF-held task affine only to the dying CPU and preempted inside an RCU read-side critical section would block that grace period, and the rq would never go offline. Test SCX_RQ_ONLINE directly so that ops.dispatch() keeps running until the rq goes offline. Only the dying CPU's own hotplug thread clears the flag during teardown, and task_can_run_on_remote_rq() still keeps other rqs' tasks off the CPU. Fixes: f0e1a0643a59 ("sched_ext: Implement BPF extensible scheduler class") Fixes: 991ef53a4832 ("sched_ext: Make scx_rq_online() also test cpu_active() in addition to SCX_RQ_ONLINE") Cc: stable@vger.kernel.org # v6.12+ Reported-by: Alap Mohan Reported-by: Joonwoo Park Signed-off-by: Tejun Heo Reviewed-by: Andrea Righi --- kernel/sched/ext/ext.c | 22 ++++++++++++++++++++-- kernel/sched/ext/inlines.h | 8 +++++++- kernel/sched/ext/sub.c | 4 ---- kernel/sched/sched.h | 5 +++-- 4 files changed, 30 insertions(+), 9 deletions(-) (limited to 'kernel') diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c index ad391a8cbd05..493b7aac7087 100644 --- a/kernel/sched/ext/ext.c +++ b/kernel/sched/ext/ext.c @@ -2123,8 +2123,8 @@ static void set_task_runnable(struct rq *rq, struct task_struct *p) } /* - * list_add_tail() must be used. scx_bypass() depends on tasks being - * appended to the runnable_list. + * list_add_tail() must be used. scx_bypass() and rq_offline_scx() + * depend on tasks being appended to the runnable_list. */ list_add_tail(&p->scx.runnable_node, &rq->scx.runnable_list); @@ -3731,8 +3731,26 @@ static void rq_online_scx(struct rq *rq) static void rq_offline_scx(struct rq *rq) { + struct task_struct *p, *n; + rq->scx.flags &= ~SCX_RQ_ONLINE; + + /* sched domain rebuilds call rq_offline with the CPU staying alive */ + if (cpu_active(cpu_of(rq))) + return; + scx_rescue_flush(rq); + + /* + * An offline CPU no longer calls ops.dispatch(). Re-enqueue its tasks + * onto the local DSQ so that they run here and balance_push() moves + * them off. + */ + list_for_each_entry_safe_reverse(p, n, &rq->scx.runnable_list, scx.runnable_node) { + if (p->scx.dsq == &rq->scx.local_dsq) + continue; + guard(sched_change)(p, DEQUEUE_SAVE | DEQUEUE_MOVE | DEQUEUE_NOCLOCK); + } } static bool check_rq_for_timeouts(struct rq *rq) diff --git a/kernel/sched/ext/inlines.h b/kernel/sched/ext/inlines.h index 2ff5479334cb..8de9d8fb4b25 100644 --- a/kernel/sched/ext/inlines.h +++ b/kernel/sched/ext/inlines.h @@ -70,7 +70,13 @@ scx_dispatch_sched(struct scx_sched *sch, struct rq *rq, #endif /* CONFIG_EXT_SUB_SCHED */ } - if (unlikely(!SCX_HAS_OP(sch, dispatch)) || !scx_rq_online(rq)) + /* + * scx_rq_online() can't be used. Its cpu_active() test goes false + * before CPU hotplug waits for an RCU grace period, and + * rq_offline_scx() moves this CPU's tasks to the local DSQ only after + * the wait. The grace period can depend on those tasks running. + */ + if (unlikely(!SCX_HAS_OP(sch, dispatch)) || !(rq->scx.flags & SCX_RQ_ONLINE)) return SCX_DSP_NONE; dspc->rq = rq; diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c index 3036190b0662..c9b44ab3bd0a 100644 --- a/kernel/sched/ext/sub.c +++ b/kernel/sched/ext/sub.c @@ -585,10 +585,6 @@ void scx_rescue_flush(struct rq *rq) lockdep_assert_rq_held(rq); - /* sched domain rebuilds call rq_offline with the CPU staying alive */ - if (cpu_active(cpu_of(rq))) - return; - /* end the current rescue */ if (rq->scx.rescue.curr) scx_task_slice_ended(rq, rq->scx.rescue.curr); diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index 7701a5a60972..742cce9da4e9 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -4205,8 +4205,9 @@ extern void balance_callbacks(struct rq *rq, struct balance_callback *head); * after which it is enqueued again. * * Typically this must be called while holding task_rq_lock, since most/all - * properties are serialized under those locks. There is currently one - * exception to this rule in sched/ext which only holds rq->lock. + * properties are serialized under those locks. There are currently two + * exceptions to this rule in sched/ext which only hold rq->lock: scx_bypass() + * and rq_offline_scx(). */ /* -- cgit v1.2.3 From af701b8d3238b9eb7cd450b0491300452c700bf5 Mon Sep 17 00:00:00 2001 From: Kuba Piecuch Date: Wed, 30 Sep 2026 14:23:58 +0000 Subject: sched_ext: Call ops.dequeue() when a task arrives on a remote local DSQ When a task in the BPF scheduler's custody is moved to another CPU's local DSQ, ops.dequeue() is only called once the task is picked for execution, from set_next_task_scx() with SCX_DEQ_CORE_SCHED_EXEC, even though no core-sched pick took place. It should instead be called without flags when the task is inserted into the destination DSQ, as it is for same-rq dispatches. move_remote_task_to_local_dsq() sets p->scx.sticky_cpu across the migration so that deactivate_task() on the source rq doesn't end custody. Since commit b75aaea24c9f ("sched_ext: Properly mark SCX-internal migrations via sticky_cpu"), enqueue_task_scx() only clears it after inserting the task, so task_leave_custody() still sees the migration in progress and skips the custody exit. Clear p->scx.sticky_cpu as soon as enqueue_task_scx() has read it, as was done before that commit. The source side is unaffected. Fixes: ebf1ccff79c4 ("sched_ext: Fix ops.dequeue() semantics") Cc: stable@vger.kernel.org # 7.1.x: 18d62044cda7: sched_ext: Preserve rq tracking across local DSQ dispatch Cc: stable@vger.kernel.org # 7.1.x Reviewed-by: Andrea Righi Assisted-by: Claude:claude-opus-5.5 Signed-off-by: Kuba Piecuch Signed-off-by: Tejun Heo --- kernel/sched/ext/ext.c | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) (limited to 'kernel') diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c index 493b7aac7087..5209510d6a5e 100644 --- a/kernel/sched/ext/ext.c +++ b/kernel/sched/ext/ext.c @@ -2152,6 +2152,13 @@ static void enqueue_task_scx(struct rq *rq, struct task_struct *p, int core_enq_ int sticky_cpu = p->scx.sticky_cpu; u64 enq_flags = core_enq_flags | rq->scx.remote_activate_enq_flags; + /* + * An SCX-internal migration ends on arrival. Clear sticky_cpu so @p can + * leave custody when inserted into the destination DSQ. + */ + if (sticky_cpu >= 0) + p->scx.sticky_cpu = -1; + /* * SCX_RQ_IN_WAKEUP promises a task_woken_scx() call once this enqueue * returns. Only the core's wakeup path delivers one. The flags stashed @@ -2190,9 +2197,6 @@ static void enqueue_task_scx(struct rq *rq, struct task_struct *p, int core_enq_ dl_server_start(&rq->ext_server); scx_do_enqueue_task(rq, p, enq_flags, sticky_cpu); - - if (sticky_cpu >= 0) - p->scx.sticky_cpu = -1; out: rq->scx.flags &= ~SCX_RQ_IN_WAKEUP; -- cgit v1.2.3 From e6ac89b8b1c12e9104df45a14a26e2dfa96ded06 Mon Sep 17 00:00:00 2001 From: Kuba Piecuch Date: Sat, 3 Oct 2026 11:53:17 +0000 Subject: sched_ext: Generate qseq from a per-task counter finish_dispatch() uses the qseq embedded in p->scx.ops_state to tell whether the QUEUED instance of a task it's about to claim is the one scx_bpf_dsq_insert() saw. qseq is generated from rq->scx.ops_qseq, but the counters of different rqs are independent, so if a task is dequeued and re-enqueued on a different rq between scx_bpf_dsq_insert() and finish_dispatch(), the new QUEUED instance can end up with the same qseq as the old one: CPU X CPU Z ----- ----- enqueue p on rq A, qseq = N ops.dispatch() scx_bpf_dsq_insert(p) records qseq N sched_setaffinity(p) dequeue p from rq A enqueue p on rq B, qseq = N finish_dispatch(p, N) qseq matches, p is claimed The claim itself is still atomic so the core stays consistent, but an insert issued for a previous QUEUED instance gets applied to a new one which the BPF scheduler has just received through ops.enqueue(). This breaks the guarantee that dispatches targeting a stale instance are ignored. Generate qseq from a per-task counter, p->scx.ops_qseq, instead so that consecutive QUEUED instances of a task never share a qseq regardless of which rq they're on. The counter is only updated in scx_do_enqueue_task() with the task's rq locked, so no additional synchronization is needed, and it fits in an existing hole in struct sched_ext_entity on 64bit. Remove the now unused rq->scx.ops_qseq. Never generate qseq 0. NONE and DISPATCHING don't carry a qseq, so scx_bpf_dsq_insert() on a task in either state records 0. With a per-task counter, every task's first QUEUED instance would otherwise get qseq 0 and could be claimed by such an insert. Wrap the counter where the QSEQ field wraps so that it can't reach a value that shifts to 0 on 32bit either. Fixes: f0e1a0643a59 ("sched_ext: Implement BPF extensible scheduler class") Cc: stable@vger.kernel.org # v6.12+ Assisted-by: Claude:claude-opus-5.5 Signed-off-by: Kuba Piecuch Signed-off-by: Tejun Heo --- kernel/sched/ext/ext.c | 14 ++++++++++---- kernel/sched/ext/internal.h | 5 +++++ kernel/sched/sched.h | 1 - 3 files changed, 15 insertions(+), 5 deletions(-) (limited to 'kernel') diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c index 5209510d6a5e..ac8fe70a9fb6 100644 --- a/kernel/sched/ext/ext.c +++ b/kernel/sched/ext/ext.c @@ -2057,8 +2057,14 @@ void scx_do_enqueue_task(struct rq *rq, struct task_struct *p, u64 enq_flags, if (unlikely(!SCX_HAS_OP(sch, enqueue))) goto global; - /* DSQ bypass didn't trigger, enqueue on the BPF scheduler */ - qseq = rq->scx.ops_qseq++ << SCX_OPSS_QSEQ_SHIFT; + /* + * DSQ bypass didn't trigger, enqueue on the BPF scheduler. Wrap the + * per-task qseq counter where the QSEQ field wraps and skip 0, which is + * what scx_bpf_dsq_insert() records for a task in NONE or DISPATCHING. + */ + p->scx.ops_qseq = ((p->scx.ops_qseq + 1) & + (SCX_OPSS_QSEQ_MASK >> SCX_OPSS_QSEQ_SHIFT)) ?: 1; + qseq = (unsigned long)p->scx.ops_qseq << SCX_OPSS_QSEQ_SHIFT; WARN_ON_ONCE(atomic_long_read(&p->scx.ops_state) != SCX_OPSS_NONE); atomic_long_set(&p->scx.ops_state, SCX_OPSS_QUEUEING | qseq); @@ -6969,9 +6975,9 @@ static void scx_dump_cpu(struct scx_sched *sch, struct seq_buf *s, seq_buf_init(&ns, buf, avail); dump_newline(&ns); - scx_dump_line(&ns, "CPU %-4d: nr_run=%u flags=0x%x cpu_rel=%d ops_qseq=%lu ksync=%lu", + scx_dump_line(&ns, "CPU %-4d: nr_run=%u flags=0x%x cpu_rel=%d ksync=%lu", cpu, rq->scx.nr_running, rq->scx.flags, rq->scx.cpu_released, - rq->scx.ops_qseq, rq->scx.kick_sync); + rq->scx.kick_sync); scx_rescue_dump(&ns, rq); scx_dump_line(&ns, " curr=%s[%d] class=%ps", rq->curr->comm, rq->curr->pid, rq->curr->sched_class); diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h index 1df8f583b0ec..5b37faa532d6 100644 --- a/kernel/sched/ext/internal.h +++ b/kernel/sched/ext/internal.h @@ -1998,6 +1998,11 @@ enum scx_ops_state { * dequeue/requeue, the dispatcher can tell whether it still has a claim * on the task being dispatched. * + * QSEQ is generated from the per-task p->scx.ops_qseq counter so that + * it doesn't repeat across QUEUED instances of the same task even if + * the task moves between rqs. 0 is never used as a valid QSEQ since + * NONE and DISPATCHING map to this value. + * * As some 32bit archs can't do 64bit store_release/load_acquire, * p->scx.ops_state is atomic_long_t which leaves 30 bits for QSEQ on * 32bit machines. The dispatch race window QSEQ protects is very narrow diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index 742cce9da4e9..9ed0cb605c33 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -813,7 +813,6 @@ struct scx_rq { #endif struct list_head runnable_list; /* runnable tasks on this rq */ struct list_head ddsp_deferred_locals; /* deferred ddsps from enq */ - unsigned long ops_qseq; /* both stashed across the activate_task() in move_remote_task_to_local_dsq() */ u64 remote_activate_enq_flags; struct scx_sched *remote_activate_sch; -- cgit v1.2.3