aboutsummaryrefslogtreecommitdiff
path: root/kernel
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-09-24 15:04:11 -0700
committerLinus Torvalds <torvalds@linux-foundation.org>2026-09-24 15:04:11 -0700
commitee9c669f9bf5fd2c24206746ded9382fe810df89 (patch)
treedfc46791f0ed3fb6455e9144f8245456486ed2fe /kernel
parente8dfd03a1c51c4e84fbf2e5ef5cd7d33cc8b7578 (diff)
parent4409a85735cdca5c4c400c0dad1feee091872eee (diff)
Merge tag 'sched_ext-for-7.3-rc4-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext
Pull sched_ext fixes from Tejun Heo: - A task reenqueued while its dispatch was still completing had its queued state clobbered by the dispatcher, dropping every later dispatch of the task. Wait for the in-flight dispatch to settle first - A wakeup activation on another CPU marked the destination runqueue as mid-wakeup, stranding a pending local reenqueue. If the scheduler was unloaded first, the stale request pointed into freed memory that the next scheduler dereferenced - ops.dequeue() ran with the source dispatch queue's lock held, so a scheduler iterating that queue from the callback deadlocked the CPU - Schedulers with their own CPU ID mapping had no way to learn a task's initial CPU mask and rebuilt it themselves, which went wrong across sub-scheduler enable and re-home. Pass it to ops.enable() - A bypass dispatch event counter missed the dispatches made by the end-of-dispatch fallback and under-reported - Selftests for the dequeue locking and initial mask changes * tag 'sched_ext-for-7.3-rc4-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext: sched_ext: Count SCX_EV_SUB_BYPASS_DISPATCH in the dispatch fallback selftests/sched_ext: Check the cmask cid-form ops.enable() receives sched_ext: Pass the initial cmask to cid-form ops.enable() selftests/sched_ext: Test that ops.dequeue() can iterate the consumed DSQ sched_ext: Don't run ops.dequeue() with a DSQ lock held sched_ext: Derive SCX_RQ_IN_WAKEUP from the core enqueue flags sched_ext: Wait for SCX_OPSS_DISPATCHING before reenqueueing a task
Diffstat (limited to 'kernel')
-rw-r--r--kernel/sched/ext/ext.c156
-rw-r--r--kernel/sched/ext/inlines.h4
-rw-r--r--kernel/sched/ext/internal.h60
-rw-r--r--kernel/sched/ext/sub.c11
4 files changed, 162 insertions, 69 deletions
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 3219f0da0fe4..e56c3c95018f 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -447,37 +447,45 @@ static void switch_rq_lock(struct rq *from, struct rq *to)
DEFINE_STATIC_KEY_FALSE(__scx_is_cid_type);
/**
- * scx_call_op_set_cpumask - invoke ops.set_cpumask / ops_cid.set_cmask for @task
+ * scx_fill_cmask_scratch - Build this cpu's arena cmask from @cpumask
+ * @sch: scx_sched whose scratch to fill
+ * @cpumask: cpus to translate into cids
+ *
+ * The scratch lives in BPF-writable arena memory and its header can't be
+ * trusted, so it is rewritten from kernel geometry rather than read. Caller
+ * must hold an rq lock so this cpu is the sole kernel writer for as long as the
+ * returned address is in use.
+ */
+static struct scx_cmask *scx_fill_cmask_scratch(struct scx_sched *sch,
+ const struct cpumask *cpumask)
+{
+ struct scx_cmask *kern_va = *this_cpu_ptr(sch->set_cmask_scratch);
+ struct scx_cmask_ref ref;
+
+ scx_cmask_ref_init_kern(sch, kern_va, 0, num_possible_cpus(), &ref);
+ scx_cmask_ref_from_cpumask(&ref, cpumask);
+ return kern_va;
+}
+
+/**
+ * scx_call_op_set_cpumask - Invoke the set_cpumask or set_cmask op for @task
* @sch: scx_sched being invoked
* @rq: rq to update as the currently-locked rq, or NULL
* @task: task whose affinity is changing
* @cpumask: new cpumask
*
- * For cid-form schedulers, translate @cpumask to a cmask via the per-cpu
- * scratch in cid.c and dispatch through the ops_cid union view. Caller
- * must hold @rq's rq lock so this_cpu_ptr is stable across the call.
+ * For cid-form schedulers, translate @cpumask to a cmask in the per-cpu scratch
+ * and dispatch through the ops_cid union view. Caller must hold @rq's rq lock.
*/
static inline void scx_call_op_set_cpumask(struct scx_sched *sch, struct rq *rq,
struct task_struct *task,
const struct cpumask *cpumask)
{
- if (scx_is_cid_type()) {
- struct scx_cmask *kern_va = *this_cpu_ptr(sch->set_cmask_scratch);
- struct scx_cmask_ref ref;
-
- /*
- * Build the per-cpu arena cmask from kernel geometry via @ref,
- * never reading its BPF-writable header. set_cmask()'s __arena
- * argument takes the kernel address and the struct_ops
- * trampoline rebases it into BPF's arena pointer form. The rq
- * lock makes this cpu the sole kernel writer.
- */
- scx_cmask_ref_init_kern(sch, kern_va, 0, num_possible_cpus(), &ref);
- scx_cmask_ref_from_cpumask(&ref, cpumask);
- SCX_CALL_CID_OP_TASK(sch, set_cmask, rq, task, kern_va);
- } else {
+ if (scx_is_cid_type())
+ SCX_CALL_CID_OP_TASK(sch, set_cmask, rq, task,
+ scx_fill_cmask_scratch(sch, cpumask));
+ else
SCX_CALL_OP_TASK(sch, set_cpumask, rq, task, cpumask);
- }
}
enum scx_dsq_iter_flags {
@@ -1499,27 +1507,22 @@ static inline bool task_scx_migrating(struct task_struct *p)
return p->scx.sticky_cpu >= 0;
}
-/*
- * Call ops.dequeue() if the task is in BPF custody and not migrating.
- * Clears %SCX_TASK_IN_CUSTODY when the callback is invoked.
- */
-static void call_task_dequeue(struct scx_sched *sch, struct rq *rq,
- struct task_struct *p, u64 deq_flags)
+/* Must be called under the lock serializing @p's custody transfers. */
+static bool task_leave_custody(struct task_struct *p)
{
if (!(p->scx.flags & SCX_TASK_IN_CUSTODY) || task_scx_migrating(p))
- return;
-
- if (SCX_HAS_OP(sch, dequeue))
- SCX_CALL_OP_TASK(sch, dequeue, rq, p, deq_flags);
+ return false;
p->scx.flags &= ~SCX_TASK_IN_CUSTODY;
+ return true;
}
static void rq_owned_post_enq(struct scx_sched *sch, struct rq *rq,
struct scx_dispatch_q *dsq, struct task_struct *p,
u64 enq_flags)
{
- call_task_dequeue(sch, rq, p, 0);
+ if (task_leave_custody(p) && SCX_HAS_OP(sch, dequeue))
+ SCX_CALL_OP_TASK(sch, dequeue, rq, p, 0);
/*
* Only local inserts get the wakeup treatment below. Rejects kick the
@@ -1705,20 +1708,28 @@ static void scx_dispatch_enqueue(struct scx_sched *sch, struct rq *rq,
if (is_rq_owned) {
rq_owned_post_enq(sch, rq, dsq, p, enq_flags);
} else {
+ bool call_dequeue = false;
+
/*
* Global and bypass DSQs are terminal - the task leaves the
- * scheduler's custody, so ops.dequeue() fires here. It can run
+ * scheduler's custody, so ops.dequeue() fires. It can run
* without @p's rq lock (finish_dispatch() passes the dispatch
* rq); that's safe because dequeue_task_scx() waits on
* SCX_OPSS_DISPATCHING (see the ops_state note above) and so
* can't race it. A non-terminal DSQ keeps the task in custody.
+ * The custody transfer happens under @dsq->lock so that later
+ * consumers see the flag clear; the callback runs after
+ * @dsq->lock is dropped because it may lock a DSQ itself.
*/
if (dsq->id == SCX_DSQ_GLOBAL || dsq->id == SCX_DSQ_BYPASS)
- call_task_dequeue(sch, rq, p, 0);
+ call_dequeue = task_leave_custody(p);
else
p->scx.flags |= SCX_TASK_IN_CUSTODY;
raw_spin_unlock(&dsq->lock);
+
+ if (call_dequeue && SCX_HAS_OP(sch, dequeue))
+ SCX_CALL_OP_TASK(sch, dequeue, rq, p, 0);
}
/*
@@ -2141,7 +2152,12 @@ static void enqueue_task_scx(struct rq *rq, struct task_struct *p, int core_enq_
int sticky_cpu = p->scx.sticky_cpu;
u64 enq_flags = core_enq_flags | rq->scx.remote_activate_enq_flags;
- if (enq_flags & ENQUEUE_WAKEUP)
+ /*
+ * SCX_RQ_IN_WAKEUP promises a task_woken_scx() call once this enqueue
+ * returns. Only the core's wakeup path delivers one. The flags stashed
+ * for a remote activation may carry the wakeup bit without it.
+ */
+ if (core_enq_flags & ENQUEUE_WAKEUP)
rq->scx.flags |= SCX_RQ_IN_WAKEUP;
/*
@@ -2210,7 +2226,7 @@ retry:
/*
* A queued task must always be in BPF scheduler's custody. If
* SCX_TASK_IN_CUSTODY is clear, finish_dispatch() on another
- * CPU has already passed call_task_dequeue() (which clears the
+ * CPU has already passed task_leave_custody() (which clears the
* flag), but has not yet written SCX_OPSS_NONE. That final
* store does not require this rq's lock, so retrying with
* cpu_relax() is bounded: we will observe NONE (or DISPATCHING,
@@ -2258,7 +2274,8 @@ retry:
* NONE but the task may still have %SCX_TASK_IN_CUSTODY set until
* it is enqueued on the destination.
*/
- call_task_dequeue(sch, rq, p, deq_flags);
+ if (task_leave_custody(p) && SCX_HAS_OP(sch, dequeue))
+ SCX_CALL_OP_TASK(sch, dequeue, rq, p, deq_flags);
}
static bool dequeue_task_scx(struct rq *rq, struct task_struct *p, int core_deq_flags)
@@ -2374,14 +2391,10 @@ static void wakeup_preempt_scx(struct rq *rq, struct task_struct *p, int wake_fl
}
void scx_move_local_task_to_local_dsq(struct scx_sched *sch, struct task_struct *p,
- u64 enq_flags, struct scx_dispatch_q *src_dsq,
- struct rq *dst_rq)
+ u64 enq_flags, struct rq *dst_rq)
{
struct scx_dispatch_q *dst_dsq = scx_resolve_local_dsq(sch, dst_rq, p, &enq_flags);
- /* @p is on @dst_rq, an rq-owned @src_dsq is covered by the rq lock */
- if (!dsq_is_rq_owned(src_dsq))
- lockdep_assert_held(&src_dsq->lock);
lockdep_assert_rq_held(dst_rq);
WARN_ON_ONCE(p->scx.holding_cpu >= 0);
@@ -2629,8 +2642,8 @@ static struct rq *move_task_between_dsqs(struct scx_sched *sch,
/* @p is going from a non-local DSQ to a local DSQ */
if (src_rq == dst_rq) {
scx_task_unlink_from_dsq(p, src_dsq);
- scx_move_local_task_to_local_dsq(sch, p, enq_flags, src_dsq, dst_rq);
raw_spin_unlock(&src_dsq->lock);
+ scx_move_local_task_to_local_dsq(sch, p, enq_flags, dst_rq);
} else {
raw_spin_unlock(&src_dsq->lock);
move_remote_task_to_local_dsq(sch, p, enq_flags, src_rq, dst_rq);
@@ -2680,8 +2693,8 @@ retry:
if (rq == task_rq) {
scx_task_unlink_from_dsq(p, dsq);
- scx_move_local_task_to_local_dsq(sch, p, enq_flags, dsq, rq);
raw_spin_unlock(&dsq->lock);
+ scx_move_local_task_to_local_dsq(sch, p, enq_flags, rq);
return true;
}
@@ -3629,8 +3642,12 @@ static void set_cpus_allowed_scx(struct task_struct *p,
*
* Fine-grained memory write control is enforced by BPF making the const
* designation pointless. Cast it away when calling the operation.
+ *
+ * The cid form receives the initial mask when the task is enabled and
+ * hears about changes only afterwards, see struct scx_enable_args.
*/
- if (SCX_HAS_OP(sch, set_cpumask))
+ if (SCX_HAS_OP(sch, set_cpumask) &&
+ (!scx_is_cid_type() || scx_get_task_state(p) == SCX_TASK_ENABLED))
scx_call_op_set_cpumask(sch, task_rq(p), p, (struct cpumask *)p->cpus_ptr);
}
@@ -3939,8 +3956,27 @@ static void __scx_enable_task(struct scx_sched *sch, struct task_struct *p)
p->scx.weight = sched_weight_to_cgroup(weight);
- if (SCX_HAS_OP(sch, enable))
- SCX_CALL_OP_TASK(sch, enable, rq, p);
+ if (SCX_HAS_OP(sch, enable)) {
+ if (scx_is_cid_type()) {
+ struct scx_cmask *cmask = scx_fill_cmask_scratch(sch, p->cpus_ptr);
+ struct scx_enable_args args = {
+ .cmask_arena_addr = scx_kaddr_to_arena(sch, cmask),
+ };
+
+ SCX_CALL_CID_OP_TASK(sch, enable, rq, p, &args);
+ } else {
+ SCX_CALL_OP_TASK(sch, enable, rq, p);
+ }
+ }
+
+ /*
+ * The initial mask also goes out through set_cmask() so a scheduler can
+ * track affinity there alone, and before set_weight() so that the mask
+ * is in place when weight-dependent state is derived, see struct
+ * scx_enable_args.
+ */
+ if (scx_is_cid_type() && SCX_HAS_OP(sch, set_cpumask))
+ scx_call_op_set_cpumask(sch, rq, p, p->cpus_ptr);
if (SCX_HAS_OP(sch, set_weight))
SCX_CALL_OP_TASK(sch, set_weight, rq, p, p->scx.weight);
@@ -4283,9 +4319,10 @@ static void switching_to_scx(struct rq *rq, struct task_struct *p)
/*
* set_cpus_allowed_scx() is not called while @p is associated with a
- * different scheduler class. Keep the BPF scheduler up-to-date.
+ * different scheduler class. Keep the BPF scheduler up-to-date. The cid
+ * form gets its mask from scx_enable_task().
*/
- if (SCX_HAS_OP(sch, set_cpumask))
+ if (!scx_is_cid_type() && SCX_HAS_OP(sch, set_cpumask))
scx_call_op_set_cpumask(sch, rq, p, (struct cpumask *)p->cpus_ptr);
}
@@ -4404,6 +4441,17 @@ static bool local_task_should_reenq(struct rq *rq, struct task_struct *p,
return *reenq_flags & SCX_REENQ_ANY;
}
+/*
+ * The dispatcher stores the final ops_state after dropping the DSQ lock, so @p
+ * can be found on a DSQ while still %SCX_OPSS_DISPATCHING. Reenqueueing @p
+ * before that store lands would have it clobber the new %SCX_OPSS_QUEUED.
+ */
+void scx_reenq_wait_dispatching(struct task_struct *p)
+{
+ if (unlikely(atomic_long_read_acquire(&p->scx.ops_state) == SCX_OPSS_DISPATCHING))
+ wait_ops_state(p, SCX_OPSS_DISPATCHING);
+}
+
static u32 reenq_local(struct scx_sched *sch, struct rq *rq, u64 reenq_flags)
{
LIST_HEAD(tasks);
@@ -4447,6 +4495,7 @@ static u32 reenq_local(struct scx_sched *sch, struct rq *rq, u64 reenq_flags)
if (!local_task_should_reenq(rq, p, &reenq_flags, &reason))
continue;
+ scx_reenq_wait_dispatching(p);
scx_dispatch_dequeue(rq, p);
if (WARN_ON_ONCE(p->scx.flags & SCX_TASK_REENQ_REASON_MASK))
@@ -4570,6 +4619,7 @@ static void reenq_user(struct rq *rq, struct scx_dispatch_q *dsq, u64 reenq_flag
}
/* @p is on @dsq, its rq and @dsq are locked */
+ scx_reenq_wait_dispatching(p);
dispatch_dequeue_locked(p, dsq);
raw_spin_unlock(&dsq->lock);
@@ -8356,10 +8406,11 @@ static struct bpf_struct_ops bpf_sched_ext_ops = {
/*
* cid-form cfi stubs. Stubs whose signatures match the cpu-form (param types
* identical, only param names differ across structs) are reused. Some need
- * fresh stubs, set_cmask due to an argument type difference and the sub-sched
- * notifiers because no cpu-form stub exists to reuse.
+ * fresh stubs, set_cmask and enable due to argument differences and the
+ * sub-sched notifiers because no cpu-form stub exists to reuse.
*/
static void sched_ext_ops_cid__set_cmask(struct task_struct *p, const struct scx_cmask *cmask__arena) {}
+static void sched_ext_ops_cid__enable(struct task_struct *p, struct scx_enable_args *args) {}
static void sched_ext_ops__sub_caps_updated(const struct scx_cmask *cmask__arena, u64 caps) {}
static void sched_ext_ops__sub_ecaps_updated(s32 cid, u64 before, u64 after) {}
@@ -8380,7 +8431,7 @@ static struct sched_ext_ops_cid __bpf_ops_sched_ext_ops_cid = {
.update_idle = sched_ext_ops__update_idle,
.init_task = sched_ext_ops__init_task,
.exit_task = sched_ext_ops__exit_task,
- .enable = sched_ext_ops__enable,
+ .enable = sched_ext_ops_cid__enable,
.disable = sched_ext_ops__disable,
#ifdef CONFIG_EXT_GROUP_SCHED
.cpuctl_init = sched_ext_ops__cgroup_init,
@@ -10403,8 +10454,7 @@ __bpf_kfunc const void *scx_bpf_online_cmask(const struct bpf_prog_aux *aux)
if (unlikely(!online))
return NULL;
- /* BPF rebases by the low 32 bits, like __arena callback args */
- return (void *)((unsigned long)online - sch->arena_kern_base);
+ return (void *)scx_kaddr_to_arena(sch, online);
}
/**
diff --git a/kernel/sched/ext/inlines.h b/kernel/sched/ext/inlines.h
index ed423bcc26b8..2ff5479334cb 100644
--- a/kernel/sched/ext/inlines.h
+++ b/kernel/sched/ext/inlines.h
@@ -129,8 +129,10 @@ scx_dispatch_sched(struct scx_sched *sch, struct rq *rq,
* scheduler's ops.dispatch() doesn't yield any tasks.
*/
if (scx_bypass_dsp_enabled(sch) &&
- scx_consume_dispatch_q(sch, rq, scx_bypass_dsq(sch, cpu), 0))
+ scx_consume_dispatch_q(sch, rq, scx_bypass_dsq(sch, cpu), 0)) {
+ __scx_add_event(sch, SCX_EV_SUB_BYPASS_DISPATCH, 1);
return SCX_DSP_LOCAL;
+ }
return SCX_DSP_NONE;
}
diff --git a/kernel/sched/ext/internal.h b/kernel/sched/ext/internal.h
index 3464e0f113c1..1df8f583b0ec 100644
--- a/kernel/sched/ext/internal.h
+++ b/kernel/sched/ext/internal.h
@@ -250,6 +250,31 @@ struct scx_exit_task_args {
bool cancelled;
};
+/**
+ * struct scx_enable_args - Argument container for cid-form ops.enable()
+ * @cmask_arena_addr: BPF arena address of the cmask of cids the task may run on
+ *
+ * @cmask_arena_addr is the task's affinity as it enters the scheduler.
+ * set_cmask() delivers the same mask right after enable(), before set_weight()
+ * and the first enqueue, then every affinity change afterwards, and is never
+ * called before enable(). A scheduler may therefore track affinity in
+ * set_cmask() alone.
+ *
+ * The kernel builds the mask in the scheduler arena from its own geometry, so
+ * the header is valid regardless of what the scheduler last wrote there. The
+ * memory is per-cpu scratch reused once the callback returns: copy the bits
+ * out, don't keep the address. The set_cmask() argument follows the same rules.
+ *
+ * The address is a plain value rather than a typed pointer because BTF can't
+ * mark a struct member as an arena pointer yet and a pointer member would reach
+ * the program typed as a kernel pointer. Cast it to struct scx_cmask __arena *
+ * before use. Once arena members can be typed, a typed alias will join this
+ * field in an anonymous union at the same offset.
+ */
+struct scx_enable_args {
+ u64 cmask_arena_addr;
+};
+
/* argument container for ops.cgroup_init() */
struct scx_cgroup_init_args {
/* the weight of the cgroup [1..10000] */
@@ -1037,6 +1062,7 @@ struct sched_ext_ops {
* - dispatch -> dispatch (cpu arg is now cid)
* - update_idle -> update_idle (cpu arg is now cid)
* - set_cpumask -> set_cmask (cmask instead of cpumask)
+ * - enable -> enable (takes struct scx_enable_args)
* - cpu_online -> cid_online
* - cpu_offline -> cid_offline
* - dump_cpu -> dump_cid
@@ -1070,7 +1096,7 @@ struct sched_ext_ops_cid {
struct scx_init_task_args *args);
void (*exit_task)(struct task_struct *p,
struct scx_exit_task_args *args);
- void (*enable)(struct task_struct *p);
+ void (*enable)(struct task_struct *p, struct scx_enable_args *args);
void (*disable)(struct task_struct *p);
void (*dump)(struct scx_dump_ctx *ctx);
void (*dump_cid)(struct scx_dump_ctx *ctx, s32 cid, bool idle);
@@ -1533,7 +1559,8 @@ struct scx_sched {
* by BUILD_BUG_ON in scx_init()). The anonymous union lets the kernel
* access either view of the same storage without function-pointer
* casts: use .ops for cpu-form and shared fields, .ops_cid for the
- * cid-renamed callbacks (set_cmask, select_cid, cid_online, ...).
+ * callbacks whose cid-form signature differs (set_cmask, enable,
+ * select_cid, cid_online, ...).
*/
union {
struct sched_ext_ops ops;
@@ -1556,9 +1583,9 @@ struct scx_sched {
uintptr_t arena_kern_base;
/*
- * Per-CPU arena cmask used by scx_call_op_set_cpumask() to hand a cmask
- * to ops_cid.set_cmask(). The kernel writes through the stored kern_va
- * and passes it to the callback's __arena argument.
+ * Per-CPU arena cmask the kernel fills from a task's cpumask and hands
+ * to ops_cid.enable() and ops_cid.set_cmask(). The stored pointers are
+ * the kernel addresses.
*/
struct scx_cmask * __percpu *set_cmask_scratch;
struct scx_cmask *online_cmask;
@@ -1669,6 +1696,19 @@ static inline void *scx_arena_to_kaddr(struct scx_sched *sch, const void *bpf_pt
return (void *)(sch->arena_kern_base + (u32)(uintptr_t)bpf_ptr);
}
+/**
+ * scx_kaddr_to_arena - Translate a kernel arena address to the BPF form
+ * @sch: scheduler whose arena hosts @kaddr
+ * @kaddr: kernel address inside @sch's arena
+ *
+ * __arena callback arguments need no translation. Addresses handed to BPF any
+ * other way, such as struct fields and kfunc return values, go through this.
+ */
+static inline uintptr_t scx_kaddr_to_arena(struct scx_sched *sch, const void *kaddr)
+{
+ return (uintptr_t)kaddr - sch->arena_kern_base;
+}
+
enum scx_wake_flags {
/* expose select WF_* flags as enums */
SCX_WAKE_FORK = WF_FORK,
@@ -2065,8 +2105,7 @@ void scx_dispatch_dequeue(struct rq *rq, struct task_struct *p);
void scx_do_enqueue_task(struct rq *rq, struct task_struct *p, u64 enq_flags,
int sticky_cpu);
void scx_move_local_task_to_local_dsq(struct scx_sched *sch, struct task_struct *p,
- u64 enq_flags, struct scx_dispatch_q *src_dsq,
- struct rq *dst_rq);
+ u64 enq_flags, struct rq *dst_rq);
bool scx_consume_dispatch_q(struct scx_sched *sch, struct rq *rq,
struct scx_dispatch_q *dsq, u64 enq_flags);
bool scx_consume_global_dsq(struct scx_sched *sch, struct rq *rq);
@@ -2078,6 +2117,7 @@ void scx_kick_cpu(struct scx_sched *sch, s32 cpu, u64 flags);
u64 __scx_bpf_now(struct rq *rq);
void schedule_dsq_reenq(struct scx_sched *sch, struct scx_dispatch_q *dsq,
u64 reenq_flags, struct rq *locked_rq);
+void scx_reenq_wait_dispatching(struct task_struct *p);
int __scx_init_task(struct scx_sched *sch, struct task_struct *p,
struct cgroup *cgrp, bool fork);
void scx_enable_task(struct scx_sched *sch, struct task_struct *p);
@@ -2302,9 +2342,9 @@ do { \
} while (0)
/*
- * Dispatch a task op through the cid-form ops_cid table. Only set_cmask() needs
- * this: it takes an arena cmask address instead of a cpumask, so it cannot be
- * invoked via its cpu-form set_cpumask() slot.
+ * Dispatch a task op through the cid-form ops_cid table, for the ops whose
+ * cid-form signature differs from the cpu-form slot: set_cmask() takes an arena
+ * cmask instead of a cpumask and enable() takes scx_enable_args.
*/
#define SCX_CALL_CID_OP_TASK(sch, op, locked_rq, task, args...) \
__SCX_CALL_OP_TASK(sch, ops_cid, op, locked_rq, task, ##args)
diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c
index 34e642a1a403..0472eaf41c7c 100644
--- a/kernel/sched/ext/sub.c
+++ b/kernel/sched/ext/sub.c
@@ -555,8 +555,8 @@ static void scx_rescue_timerfn(struct timer_list *timer)
scx.dsq_list.node);
scx_task_unlink_from_dsq(p, &rq->scx.rescue.dsq);
scx_rescue_admit(rq, p, slice);
- scx_move_local_task_to_local_dsq(scx_task_sched(p), p, SCX_ENQ_IGNORE_CAPS,
- &rq->scx.rescue.dsq, rq);
+ scx_move_local_task_to_local_dsq(scx_task_sched(p), p,
+ SCX_ENQ_IGNORE_CAPS, rq);
if (sched_class_above(&ext_sched_class, rq->curr->sched_class))
resched_curr(rq);
} else if (p->scx.dsq && rq->scx.rescue.budget > 2 * scx_rescue_quantum_ns) {
@@ -572,7 +572,7 @@ static void scx_rescue_timerfn(struct timer_list *timer)
scx_task_unlink_from_dsq(p, &rq->scx.local_dsq);
scx_move_local_task_to_local_dsq(scx_task_sched(p), p,
SCX_ENQ_HEAD | SCX_ENQ_PREEMPT | SCX_ENQ_IGNORE_CAPS,
- &rq->scx.local_dsq, rq);
+ rq);
}
out_arm:
scx_rescue_timer_arm(rq);
@@ -596,8 +596,8 @@ void scx_rescue_flush(struct rq *rq)
/* and flush out all pending ones */
list_for_each_entry_safe(p, n, &rq->scx.rescue.dsq.list, scx.dsq_list.node) {
scx_task_unlink_from_dsq(p, &rq->scx.rescue.dsq);
- scx_move_local_task_to_local_dsq(scx_task_sched(p), p, SCX_ENQ_IGNORE_CAPS,
- &rq->scx.rescue.dsq, rq);
+ scx_move_local_task_to_local_dsq(scx_task_sched(p), p,
+ SCX_ENQ_IGNORE_CAPS, rq);
}
timer_delete(&rq->scx.rescue.timer);
@@ -801,6 +801,7 @@ void scx_reenq_reject(struct rq *rq)
if (WARN_ON_ONCE(p->migration_pending))
continue;
+ scx_reenq_wait_dispatching(p);
scx_dispatch_dequeue(rq, p);
if (WARN_ON_ONCE(p->scx.flags & SCX_TASK_REENQ_REASON_MASK))