diff options
| author | Linus Torvalds <torvalds@linux-foundation.org> | 2026-10-07 02:03:46 +0200 |
|---|---|---|
| committer | Linus Torvalds <torvalds@linux-foundation.org> | 2026-10-07 02:03:46 +0200 |
| commit | 762122d75e50e4919ef77afc2dffdc4931c5be87 (patch) | |
| tree | c6a4e5e53e813d62b79a3bcfee8d42246234017b /kernel | |
| parent | 69f80fef3153299d9c72c53d1d71eef6354b6926 (diff) | |
| parent | e6ac89b8b1c12e9104df45a14a26e2dfa96ded06 (diff) | |
Merge tag 'sched_ext-for-7.3-rc6-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext
Pull sched_ext fixes from Tejun Heo:
- Taking a CPU offline could hang, or stall until the watchdog ejected
the BPF scheduler, when tasks on the dying CPU were still held by the
scheduler or sitting on a user dispatch queue. Re-enqueue them onto
the local queue when the runqueue goes offline so that the CPU pushes
them off like the other sched classes.
- The sequence number guarding against stale dispatches was per
runqueue, so a task re-enqueued on another CPU could get the same
number and a dispatch meant for its earlier instance was applied to
the new one. Use a per-task counter.
- A task dispatched to another CPU's local queue got its ops.dequeue()
only when picked to run and flagged as a core-sched pick. Call it at
insertion like for same-CPU dispatches.
* tag 'sched_ext-for-7.3-rc6-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext:
sched_ext: Generate qseq from a per-task counter
selftests/sched_ext: Add a test for ops.dequeue() on remote local DSQ moves
sched_ext: Call ops.dequeue() when a task arrives on a remote local DSQ
sched_ext: Fix CPU hotplug hang when a dying CPU's tasks sit in the BPF scheduler
Diffstat (limited to 'kernel')
| -rw-r--r-- | kernel/sched/ext/ext.c | 46 | ||||
| -rw-r--r-- | kernel/sched/ext/inlines.h | 8 | ||||
| -rw-r--r-- | kernel/sched/ext/internal.h | 5 | ||||
| -rw-r--r-- | kernel/sched/ext/sub.c | 4 | ||||
| -rw-r--r-- | kernel/sched/sched.h | 6 |
5 files changed, 52 insertions, 17 deletions
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c index e56c3c95018f..5fe980da545d 100644 --- a/kernel/sched/ext/ext.c +++ b/kernel/sched/ext/ext.c @@ -2057,8 +2057,14 @@ void scx_do_enqueue_task(struct rq *rq, struct task_struct *p, u64 enq_flags, if (unlikely(!SCX_HAS_OP(sch, enqueue))) goto global; - /* DSQ bypass didn't trigger, enqueue on the BPF scheduler */ - qseq = rq->scx.ops_qseq++ << SCX_OPSS_QSEQ_SHIFT; + /* + * DSQ bypass didn't trigger, enqueue on the BPF scheduler. Wrap the + * per-task qseq counter where the QSEQ field wraps and skip 0, which is + * what scx_bpf_dsq_insert() records for a task in NONE or DISPATCHING. + */ + p->scx.ops_qseq = ((p->scx.ops_qseq + 1) & + (SCX_OPSS_QSEQ_MASK >> SCX_OPSS_QSEQ_SHIFT)) ?: 1; + qseq = (unsigned long)p->scx.ops_qseq << SCX_OPSS_QSEQ_SHIFT; WARN_ON_ONCE(atomic_long_read(&p->scx.ops_state) != SCX_OPSS_NONE); atomic_long_set(&p->scx.ops_state, SCX_OPSS_QUEUEING | qseq); @@ -2123,8 +2129,8 @@ static void set_task_runnable(struct rq *rq, struct task_struct *p) } /* - * list_add_tail() must be used. scx_bypass() depends on tasks being - * appended to the runnable_list. + * list_add_tail() must be used. scx_bypass() and rq_offline_scx() + * depend on tasks being appended to the runnable_list. */ list_add_tail(&p->scx.runnable_node, &rq->scx.runnable_list); @@ -2153,6 +2159,13 @@ static void enqueue_task_scx(struct rq *rq, struct task_struct *p, int core_enq_ u64 enq_flags = core_enq_flags | rq->scx.remote_activate_enq_flags; /* + * An SCX-internal migration ends on arrival. Clear sticky_cpu so @p can + * leave custody when inserted into the destination DSQ. + */ + if (sticky_cpu >= 0) + p->scx.sticky_cpu = -1; + + /* * SCX_RQ_IN_WAKEUP promises a task_woken_scx() call once this enqueue * returns. Only the core's wakeup path delivers one. The flags stashed * for a remote activation may carry the wakeup bit without it. @@ -2190,9 +2203,6 @@ static void enqueue_task_scx(struct rq *rq, struct task_struct *p, int core_enq_ dl_server_start(&rq->ext_server); scx_do_enqueue_task(rq, p, enq_flags, sticky_cpu); - - if (sticky_cpu >= 0) - p->scx.sticky_cpu = -1; out: rq->scx.flags &= ~SCX_RQ_IN_WAKEUP; @@ -3731,8 +3741,26 @@ static void rq_online_scx(struct rq *rq) static void rq_offline_scx(struct rq *rq) { + struct task_struct *p, *n; + rq->scx.flags &= ~SCX_RQ_ONLINE; + + /* sched domain rebuilds call rq_offline with the CPU staying alive */ + if (cpu_active(cpu_of(rq))) + return; + scx_rescue_flush(rq); + + /* + * An offline CPU no longer calls ops.dispatch(). Re-enqueue its tasks + * onto the local DSQ so that they run here and balance_push() moves + * them off. + */ + list_for_each_entry_safe_reverse(p, n, &rq->scx.runnable_list, scx.runnable_node) { + if (p->scx.dsq == &rq->scx.local_dsq) + continue; + guard(sched_change)(p, DEQUEUE_SAVE | DEQUEUE_MOVE | DEQUEUE_NOCLOCK); + } } static bool check_rq_for_timeouts(struct rq *rq) @@ -6947,9 +6975,9 @@ static void scx_dump_cpu(struct scx_sched *sch, struct seq_buf *s, seq_buf_init(&ns, buf, avail); dump_newline(&ns); - scx_dump_line(&ns, "CPU %-4d: nr_run=%u flags=0x%x cpu_rel=%d ops_qseq=%lu ksync=%lu", + scx_dump_line(&ns, "CPU %-4d: nr_run=%u flags=0x%x cpu_rel=%d ksync=%lu", cpu, rq->scx.nr_running, rq->scx.flags, rq->scx.cpu_released, - rq->scx.ops_qseq, rq->scx.kick_sync); + rq->scx.kick_sync); scx_rescue_dump(&ns, rq); scx_dump_line(&ns, " curr=%s[%d] class=%ps", rq->curr->comm, rq->curr->pid, rq->curr->sched_class); diff --git a/kernel/sched/ext/inlines.h b/kernel/sched/ext/inlines.h index 2ff5479334cb..8de9d8fb4b25 100644 --- a/kernel/sched/ext/inlines.h +++ b/kernel/sched/ext/inlines.h @@ -70,7 +70,13 @@ scx_dispatch_sched(struct scx_sched *sch, struct rq *rq, #endif /* CONFIG_EXT_SUB_SCHED */ } - if (unlikely(!SCX_HAS_OP(sch, dispatch)) || !scx_rq_online(rq)) + /* + * scx_rq_online() can't be used. Its cpu_active() test goes false + * before CPU hotplug waits for an RCU grace period, and + * rq_offline_scx() moves this CPU's tasks to the local DSQ only after + * the wait. The grace period can depend on those tasks running. + */ + if (unlikely(!SCX_HAS_OP(sch, dispatch)) || !(rq->scx.flags & SCX_RQ_ONLINE)) return SCX_DSP_NONE; dspc->rq = rq; diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h index 1df8f583b0ec..5b37faa532d6 100644 --- a/kernel/sched/ext/internal.h +++ b/kernel/sched/ext/internal.h @@ -1998,6 +1998,11 @@ enum scx_ops_state { * dequeue/requeue, the dispatcher can tell whether it still has a claim * on the task being dispatched. * + * QSEQ is generated from the per-task p->scx.ops_qseq counter so that + * it doesn't repeat across QUEUED instances of the same task even if + * the task moves between rqs. 0 is never used as a valid QSEQ since + * NONE and DISPATCHING map to this value. + * * As some 32bit archs can't do 64bit store_release/load_acquire, * p->scx.ops_state is atomic_long_t which leaves 30 bits for QSEQ on * 32bit machines. The dispatch race window QSEQ protects is very narrow diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c index 0472eaf41c7c..90441206d3b9 100644 --- a/kernel/sched/ext/sub.c +++ b/kernel/sched/ext/sub.c @@ -585,10 +585,6 @@ void scx_rescue_flush(struct rq *rq) lockdep_assert_rq_held(rq); - /* sched domain rebuilds call rq_offline with the CPU staying alive */ - if (cpu_active(cpu_of(rq))) - return; - /* end the current rescue */ if (rq->scx.rescue.curr) scx_task_slice_ended(rq, rq->scx.rescue.curr); diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index e656c7059bf8..4c25fbe84fb5 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -816,7 +816,6 @@ struct scx_rq { #endif struct list_head runnable_list; /* runnable tasks on this rq */ struct list_head ddsp_deferred_locals; /* deferred ddsps from enq */ - unsigned long ops_qseq; /* both stashed across the activate_task() in move_remote_task_to_local_dsq() */ u64 remote_activate_enq_flags; struct scx_sched *remote_activate_sch; @@ -4222,8 +4221,9 @@ extern void balance_callbacks(struct rq *rq, struct balance_callback *head); * after which it is enqueued again. * * Typically this must be called while holding task_rq_lock, since most/all - * properties are serialized under those locks. There is currently one - * exception to this rule in sched/ext which only holds rq->lock. + * properties are serialized under those locks. There are currently two + * exceptions to this rule in sched/ext which only hold rq->lock: scx_bypass() + * and rq_offline_scx(). */ /* |
