aboutsummaryrefslogtreecommitdiff
path: root/kernel
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-10-07 02:03:46 +0200
committerLinus Torvalds <torvalds@linux-foundation.org>2026-10-07 02:03:46 +0200
commit762122d75e50e4919ef77afc2dffdc4931c5be87 (patch)
treec6a4e5e53e813d62b79a3bcfee8d42246234017b /kernel
parent69f80fef3153299d9c72c53d1d71eef6354b6926 (diff)
parente6ac89b8b1c12e9104df45a14a26e2dfa96ded06 (diff)
Merge tag 'sched_ext-for-7.3-rc6-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext
Pull sched_ext fixes from Tejun Heo: - Taking a CPU offline could hang, or stall until the watchdog ejected the BPF scheduler, when tasks on the dying CPU were still held by the scheduler or sitting on a user dispatch queue. Re-enqueue them onto the local queue when the runqueue goes offline so that the CPU pushes them off like the other sched classes. - The sequence number guarding against stale dispatches was per runqueue, so a task re-enqueued on another CPU could get the same number and a dispatch meant for its earlier instance was applied to the new one. Use a per-task counter. - A task dispatched to another CPU's local queue got its ops.dequeue() only when picked to run and flagged as a core-sched pick. Call it at insertion like for same-CPU dispatches. * tag 'sched_ext-for-7.3-rc6-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext: sched_ext: Generate qseq from a per-task counter selftests/sched_ext: Add a test for ops.dequeue() on remote local DSQ moves sched_ext: Call ops.dequeue() when a task arrives on a remote local DSQ sched_ext: Fix CPU hotplug hang when a dying CPU's tasks sit in the BPF scheduler
Diffstat (limited to 'kernel')
-rw-r--r--kernel/sched/ext/ext.c46
-rw-r--r--kernel/sched/ext/inlines.h8
-rw-r--r--kernel/sched/ext/internal.h5
-rw-r--r--kernel/sched/ext/sub.c4
-rw-r--r--kernel/sched/sched.h6
5 files changed, 52 insertions, 17 deletions
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index e56c3c95018f..5fe980da545d 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -2057,8 +2057,14 @@ void scx_do_enqueue_task(struct rq *rq, struct task_struct *p, u64 enq_flags,
if (unlikely(!SCX_HAS_OP(sch, enqueue)))
goto global;
- /* DSQ bypass didn't trigger, enqueue on the BPF scheduler */
- qseq = rq->scx.ops_qseq++ << SCX_OPSS_QSEQ_SHIFT;
+ /*
+ * DSQ bypass didn't trigger, enqueue on the BPF scheduler. Wrap the
+ * per-task qseq counter where the QSEQ field wraps and skip 0, which is
+ * what scx_bpf_dsq_insert() records for a task in NONE or DISPATCHING.
+ */
+ p->scx.ops_qseq = ((p->scx.ops_qseq + 1) &
+ (SCX_OPSS_QSEQ_MASK >> SCX_OPSS_QSEQ_SHIFT)) ?: 1;
+ qseq = (unsigned long)p->scx.ops_qseq << SCX_OPSS_QSEQ_SHIFT;
WARN_ON_ONCE(atomic_long_read(&p->scx.ops_state) != SCX_OPSS_NONE);
atomic_long_set(&p->scx.ops_state, SCX_OPSS_QUEUEING | qseq);
@@ -2123,8 +2129,8 @@ static void set_task_runnable(struct rq *rq, struct task_struct *p)
}
/*
- * list_add_tail() must be used. scx_bypass() depends on tasks being
- * appended to the runnable_list.
+ * list_add_tail() must be used. scx_bypass() and rq_offline_scx()
+ * depend on tasks being appended to the runnable_list.
*/
list_add_tail(&p->scx.runnable_node, &rq->scx.runnable_list);
@@ -2153,6 +2159,13 @@ static void enqueue_task_scx(struct rq *rq, struct task_struct *p, int core_enq_
u64 enq_flags = core_enq_flags | rq->scx.remote_activate_enq_flags;
/*
+ * An SCX-internal migration ends on arrival. Clear sticky_cpu so @p can
+ * leave custody when inserted into the destination DSQ.
+ */
+ if (sticky_cpu >= 0)
+ p->scx.sticky_cpu = -1;
+
+ /*
* SCX_RQ_IN_WAKEUP promises a task_woken_scx() call once this enqueue
* returns. Only the core's wakeup path delivers one. The flags stashed
* for a remote activation may carry the wakeup bit without it.
@@ -2190,9 +2203,6 @@ static void enqueue_task_scx(struct rq *rq, struct task_struct *p, int core_enq_
dl_server_start(&rq->ext_server);
scx_do_enqueue_task(rq, p, enq_flags, sticky_cpu);
-
- if (sticky_cpu >= 0)
- p->scx.sticky_cpu = -1;
out:
rq->scx.flags &= ~SCX_RQ_IN_WAKEUP;
@@ -3731,8 +3741,26 @@ static void rq_online_scx(struct rq *rq)
static void rq_offline_scx(struct rq *rq)
{
+ struct task_struct *p, *n;
+
rq->scx.flags &= ~SCX_RQ_ONLINE;
+
+ /* sched domain rebuilds call rq_offline with the CPU staying alive */
+ if (cpu_active(cpu_of(rq)))
+ return;
+
scx_rescue_flush(rq);
+
+ /*
+ * An offline CPU no longer calls ops.dispatch(). Re-enqueue its tasks
+ * onto the local DSQ so that they run here and balance_push() moves
+ * them off.
+ */
+ list_for_each_entry_safe_reverse(p, n, &rq->scx.runnable_list, scx.runnable_node) {
+ if (p->scx.dsq == &rq->scx.local_dsq)
+ continue;
+ guard(sched_change)(p, DEQUEUE_SAVE | DEQUEUE_MOVE | DEQUEUE_NOCLOCK);
+ }
}
static bool check_rq_for_timeouts(struct rq *rq)
@@ -6947,9 +6975,9 @@ static void scx_dump_cpu(struct scx_sched *sch, struct seq_buf *s,
seq_buf_init(&ns, buf, avail);
dump_newline(&ns);
- scx_dump_line(&ns, "CPU %-4d: nr_run=%u flags=0x%x cpu_rel=%d ops_qseq=%lu ksync=%lu",
+ scx_dump_line(&ns, "CPU %-4d: nr_run=%u flags=0x%x cpu_rel=%d ksync=%lu",
cpu, rq->scx.nr_running, rq->scx.flags, rq->scx.cpu_released,
- rq->scx.ops_qseq, rq->scx.kick_sync);
+ rq->scx.kick_sync);
scx_rescue_dump(&ns, rq);
scx_dump_line(&ns, " curr=%s[%d] class=%ps",
rq->curr->comm, rq->curr->pid, rq->curr->sched_class);
diff --git a/kernel/sched/ext/inlines.h b/kernel/sched/ext/inlines.h
index 2ff5479334cb..8de9d8fb4b25 100644
--- a/kernel/sched/ext/inlines.h
+++ b/kernel/sched/ext/inlines.h
@@ -70,7 +70,13 @@ scx_dispatch_sched(struct scx_sched *sch, struct rq *rq,
#endif /* CONFIG_EXT_SUB_SCHED */
}
- if (unlikely(!SCX_HAS_OP(sch, dispatch)) || !scx_rq_online(rq))
+ /*
+ * scx_rq_online() can't be used. Its cpu_active() test goes false
+ * before CPU hotplug waits for an RCU grace period, and
+ * rq_offline_scx() moves this CPU's tasks to the local DSQ only after
+ * the wait. The grace period can depend on those tasks running.
+ */
+ if (unlikely(!SCX_HAS_OP(sch, dispatch)) || !(rq->scx.flags & SCX_RQ_ONLINE))
return SCX_DSP_NONE;
dspc->rq = rq;
diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h
index 1df8f583b0ec..5b37faa532d6 100644
--- a/kernel/sched/ext/internal.h
+++ b/kernel/sched/ext/internal.h
@@ -1998,6 +1998,11 @@ enum scx_ops_state {
* dequeue/requeue, the dispatcher can tell whether it still has a claim
* on the task being dispatched.
*
+ * QSEQ is generated from the per-task p->scx.ops_qseq counter so that
+ * it doesn't repeat across QUEUED instances of the same task even if
+ * the task moves between rqs. 0 is never used as a valid QSEQ since
+ * NONE and DISPATCHING map to this value.
+ *
* As some 32bit archs can't do 64bit store_release/load_acquire,
* p->scx.ops_state is atomic_long_t which leaves 30 bits for QSEQ on
* 32bit machines. The dispatch race window QSEQ protects is very narrow
diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c
index 0472eaf41c7c..90441206d3b9 100644
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -585,10 +585,6 @@ void scx_rescue_flush(struct rq *rq)
lockdep_assert_rq_held(rq);
- /* sched domain rebuilds call rq_offline with the CPU staying alive */
- if (cpu_active(cpu_of(rq)))
- return;
-
/* end the current rescue */
if (rq->scx.rescue.curr)
scx_task_slice_ended(rq, rq->scx.rescue.curr);
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index e656c7059bf8..4c25fbe84fb5 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -816,7 +816,6 @@ struct scx_rq {
#endif
struct list_head runnable_list; /* runnable tasks on this rq */
struct list_head ddsp_deferred_locals; /* deferred ddsps from enq */
- unsigned long ops_qseq;
/* both stashed across the activate_task() in move_remote_task_to_local_dsq() */
u64 remote_activate_enq_flags;
struct scx_sched *remote_activate_sch;
@@ -4222,8 +4221,9 @@ extern void balance_callbacks(struct rq *rq, struct balance_callback *head);
* after which it is enqueued again.
*
* Typically this must be called while holding task_rq_lock, since most/all
- * properties are serialized under those locks. There is currently one
- * exception to this rule in sched/ext which only holds rq->lock.
+ * properties are serialized under those locks. There are currently two
+ * exceptions to this rule in sched/ext which only hold rq->lock: scx_bypass()
+ * and rq_offline_scx().
*/
/*