diff options
| author | Kuba Piecuch <jpiecuch@google.com> | 2026-10-03 11:53:17 +0000 |
|---|---|---|
| committer | Tejun Heo <tj@kernel.org> | 2026-10-03 13:23:31 -1000 |
| commit | e6ac89b8b1c12e9104df45a14a26e2dfa96ded06 (patch) | |
| tree | 31e9f8b717945df4cd2452e00f1c865d4a87ed9e /kernel | |
| parent | cc07fae1de5dbf514a9a3978050ee3cf5fe20d3c (diff) | |
sched_ext: Generate qseq from a per-task counter
finish_dispatch() uses the qseq embedded in p->scx.ops_state to tell
whether the QUEUED instance of a task it's about to claim is the one
scx_bpf_dsq_insert() saw. qseq is generated from rq->scx.ops_qseq, but
the counters of different rqs are independent, so if a task is dequeued
and re-enqueued on a different rq between scx_bpf_dsq_insert() and
finish_dispatch(), the new QUEUED instance can end up with the same
qseq as the old one:
CPU X CPU Z
----- -----
enqueue p on rq A, qseq = N
ops.dispatch()
scx_bpf_dsq_insert(p)
records qseq N
sched_setaffinity(p)
dequeue p from rq A
enqueue p on rq B, qseq = N
finish_dispatch(p, N)
qseq matches, p is claimed
The claim itself is still atomic so the core stays consistent, but an
insert issued for a previous QUEUED instance gets applied to a new one
which the BPF scheduler has just received through ops.enqueue(). This
breaks the guarantee that dispatches targeting a stale instance are
ignored.
Generate qseq from a per-task counter, p->scx.ops_qseq, instead so that
consecutive QUEUED instances of a task never share a qseq regardless of
which rq they're on. The counter is only updated in
scx_do_enqueue_task() with the task's rq locked, so no additional
synchronization is needed, and it fits in an existing hole in struct
sched_ext_entity on 64bit. Remove the now unused rq->scx.ops_qseq.
Never generate qseq 0. NONE and DISPATCHING don't carry a qseq, so
scx_bpf_dsq_insert() on a task in either state records 0. With a
per-task counter, every task's first QUEUED instance would otherwise get
qseq 0 and could be claimed by such an insert. Wrap the counter where
the QSEQ field wraps so that it can't reach a value that shifts to 0 on
32bit either.
Fixes: f0e1a0643a59 ("sched_ext: Implement BPF extensible scheduler class")
Cc: stable@vger.kernel.org # v6.12+
Assisted-by: Claude:claude-opus-5.5
Signed-off-by: Kuba Piecuch <jpiecuch@google.com>
Signed-off-by: Tejun Heo <tj@kernel.org>
Diffstat (limited to 'kernel')
| -rw-r--r-- | kernel/sched/ext/ext.c | 14 | ||||
| -rw-r--r-- | kernel/sched/ext/internal.h | 5 | ||||
| -rw-r--r-- | kernel/sched/sched.h | 1 |
3 files changed, 15 insertions, 5 deletions
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c index 5209510d6a5e..ac8fe70a9fb6 100644 --- a/kernel/sched/ext/ext.c +++ b/kernel/sched/ext/ext.c @@ -2057,8 +2057,14 @@ void scx_do_enqueue_task(struct rq *rq, struct task_struct *p, u64 enq_flags, if (unlikely(!SCX_HAS_OP(sch, enqueue))) goto global; - /* DSQ bypass didn't trigger, enqueue on the BPF scheduler */ - qseq = rq->scx.ops_qseq++ << SCX_OPSS_QSEQ_SHIFT; + /* + * DSQ bypass didn't trigger, enqueue on the BPF scheduler. Wrap the + * per-task qseq counter where the QSEQ field wraps and skip 0, which is + * what scx_bpf_dsq_insert() records for a task in NONE or DISPATCHING. + */ + p->scx.ops_qseq = ((p->scx.ops_qseq + 1) & + (SCX_OPSS_QSEQ_MASK >> SCX_OPSS_QSEQ_SHIFT)) ?: 1; + qseq = (unsigned long)p->scx.ops_qseq << SCX_OPSS_QSEQ_SHIFT; WARN_ON_ONCE(atomic_long_read(&p->scx.ops_state) != SCX_OPSS_NONE); atomic_long_set(&p->scx.ops_state, SCX_OPSS_QUEUEING | qseq); @@ -6969,9 +6975,9 @@ static void scx_dump_cpu(struct scx_sched *sch, struct seq_buf *s, seq_buf_init(&ns, buf, avail); dump_newline(&ns); - scx_dump_line(&ns, "CPU %-4d: nr_run=%u flags=0x%x cpu_rel=%d ops_qseq=%lu ksync=%lu", + scx_dump_line(&ns, "CPU %-4d: nr_run=%u flags=0x%x cpu_rel=%d ksync=%lu", cpu, rq->scx.nr_running, rq->scx.flags, rq->scx.cpu_released, - rq->scx.ops_qseq, rq->scx.kick_sync); + rq->scx.kick_sync); scx_rescue_dump(&ns, rq); scx_dump_line(&ns, " curr=%s[%d] class=%ps", rq->curr->comm, rq->curr->pid, rq->curr->sched_class); diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h index 1df8f583b0ec..5b37faa532d6 100644 --- a/kernel/sched/ext/internal.h +++ b/kernel/sched/ext/internal.h @@ -1998,6 +1998,11 @@ enum scx_ops_state { * dequeue/requeue, the dispatcher can tell whether it still has a claim * on the task being dispatched. * + * QSEQ is generated from the per-task p->scx.ops_qseq counter so that + * it doesn't repeat across QUEUED instances of the same task even if + * the task moves between rqs. 0 is never used as a valid QSEQ since + * NONE and DISPATCHING map to this value. + * * As some 32bit archs can't do 64bit store_release/load_acquire, * p->scx.ops_state is atomic_long_t which leaves 30 bits for QSEQ on * 32bit machines. The dispatch race window QSEQ protects is very narrow diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index 742cce9da4e9..9ed0cb605c33 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -813,7 +813,6 @@ struct scx_rq { #endif struct list_head runnable_list; /* runnable tasks on this rq */ struct list_head ddsp_deferred_locals; /* deferred ddsps from enq */ - unsigned long ops_qseq; /* both stashed across the activate_task() in move_remote_task_to_local_dsq() */ u64 remote_activate_enq_flags; struct scx_sched *remote_activate_sch; |
