[PATCH 04/14] sched: Add prepare_switch() class callback
Andrea Righi <[email protected]> Sat, 25 Jul 2026 18:04:10 +0200
| Newsgroups | dev.linux.lists.sched-ext,org.kernel.vger.linux-kernel |
|---|---|
| Message-ID | <[email protected]> |
sched_change_begin() only invokes callbacks on the outgoing class before recording the task queueing state. An incoming class may need to prepare a task while that state is still live, before the normal dequeue and class transition take place. Pass the incoming class to sched_change_begin() and invoke its optional prepare_switch() callback before recording the queued and running state. This is a preparatory change to support proxy execution with sched_ext. Suggested-by: Tejun Heo <[email protected]> Signed-off-by: Andrea Righi <[email protected]> --- kernel/sched/core.c | 16 +++++++++++----- kernel/sched/ext/ext.c | 7 ++++--- kernel/sched/ext/sub.c | 9 ++++++--- kernel/sched/sched.h | 10 +++++++--- kernel/sched/syscalls.c | 4 ++-- 5 files changed, 30 insertions(+), 16 deletions(-) diff --git a/kernel/sched/core.c b/kernel/sched/core.c index 9a9d4b5040089..474628f520a29 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -2792,7 +2792,7 @@ void set_cpus_allowed_common(struct task_struct *p, struct affinity_context *ctx static void do_set_cpus_allowed(struct task_struct *p, struct affinity_context *ctx) { - scoped_guard (sched_change, p, DEQUEUE_SAVE) + scoped_guard (sched_change, p, p->sched_class, DEQUEUE_SAVE) p->sched_class->set_cpus_allowed(p, ctx); } @@ -7722,7 +7722,7 @@ void rt_mutex_setprio(struct task_struct *p, struct task_struct *pi_task) if (prev_class != next_class) queue_flag |= DEQUEUE_CLASS; - scoped_guard (sched_change, p, queue_flag) { + scoped_guard (sched_change, p, next_class, queue_flag) { /* * Boosting condition are: * 1. -rt task is running and holds mutex A @@ -8399,7 +8399,7 @@ int migrate_task_to(struct task_struct *p, int target_cpu) void sched_setnuma(struct task_struct *p, int nid) { guard(task_rq_lock)(p); - scoped_guard (sched_change, p, DEQUEUE_SAVE) + scoped_guard (sched_change, p, p->sched_class, DEQUEUE_SAVE) p->numa_preferred_nid = nid; } #endif /* CONFIG_NUMA_BALANCING */ @@ -9516,7 +9516,7 @@ void sched_move_task(struct task_struct *tsk, bool for_autogroup) CLASS(task_rq_lock, rq_guard)(tsk); rq = rq_guard.rq; - scoped_guard (sched_change, tsk, queue_flags) { + scoped_guard (sched_change, tsk, tsk->sched_class, queue_flags) { sched_change_group(tsk); if (!for_autogroup) scx_cgroup_move_task(tsk); @@ -11220,7 +11220,9 @@ static inline void sched_mm_cid_fork(struct task_struct *t) { } static DEFINE_PER_CPU(struct sched_change_ctx, sched_change_ctx); -struct sched_change_ctx *sched_change_begin(struct task_struct *p, unsigned int flags) +struct sched_change_ctx * +sched_change_begin(struct task_struct *p, const struct sched_class *next_class, + unsigned int flags) { struct sched_change_ctx *ctx = this_cpu_ptr(&sched_change_ctx); struct rq *rq = task_rq(p); @@ -11238,6 +11240,10 @@ struct sched_change_ctx *sched_change_begin(struct task_struct *p, unsigned int flags |= DEQUEUE_NOCLOCK; } + if ((flags & DEQUEUE_CLASS) && next_class != p->sched_class && + next_class->prepare_switch) + next_class->prepare_switch(rq, p); + if ((flags & DEQUEUE_CLASS) && p->sched_class->switching_from) p->sched_class->switching_from(rq, p); diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c index 81c2d8eeae410..285bd8422193d 100644 --- a/kernel/sched/ext/ext.c +++ b/kernel/sched/ext/ext.c @@ -5859,7 +5859,8 @@ void scx_bypass(struct scx_sched *sch, bool bypass) continue; /* cycling deq/enq is enough, see the function comment */ - scoped_guard (sched_change, p, DEQUEUE_SAVE | DEQUEUE_MOVE) { + scoped_guard (sched_change, p, p->sched_class, + DEQUEUE_SAVE | DEQUEUE_MOVE) { /* nothing */ ; } } @@ -6141,7 +6142,7 @@ static void scx_root_disable(struct scx_sched *sch) if (old_class != new_class) queue_flags |= DEQUEUE_CLASS; - scoped_guard (sched_change, p, queue_flags) { + scoped_guard (sched_change, p, new_class, queue_flags) { p->sched_class = new_class; } @@ -7477,7 +7478,7 @@ static void scx_root_enable_workfn(struct kthread_work *work) if (old_class != new_class) queue_flags |= DEQUEUE_CLASS; - scoped_guard (sched_change, p, queue_flags) { + scoped_guard (sched_change, p, new_class, queue_flags) { set_task_slice(p, READ_ONCE(sch->slice_dfl)); p->sched_class = new_class; } diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c index 6da6c91e42870..765b516e06c95 100644 --- a/kernel/sched/ext/sub.c +++ b/kernel/sched/ext/sub.c @@ -791,7 +791,8 @@ static void scx_rehome_task(struct scx_sched *to, struct task_struct *p) lockdep_assert_held(&p->pi_lock); lockdep_assert_rq_held(task_rq(p)); - scoped_guard (sched_change, p, DEQUEUE_SAVE | DEQUEUE_MOVE) { + scoped_guard (sched_change, p, p->sched_class, + DEQUEUE_SAVE | DEQUEUE_MOVE) { scx_disable_and_exit_task(scx_task_sched(p), p); scx_set_task_state(p, SCX_TASK_INIT_BEGIN); scx_set_task_state(p, SCX_TASK_INIT); @@ -821,7 +822,8 @@ static void scx_punt_task(struct scx_sched *to, struct task_struct *p) lockdep_assert_rq_held(task_rq(p)); WARN_ON_ONCE(!READ_ONCE(to->bypass_depth)); - scoped_guard (sched_change, p, DEQUEUE_SAVE | DEQUEUE_MOVE) { + scoped_guard (sched_change, p, p->sched_class, + DEQUEUE_SAVE | DEQUEUE_MOVE) { scx_disable_and_exit_task(scx_task_sched(p), p); scx_set_task_sched(p, to); } @@ -1471,7 +1473,8 @@ void scx_sub_enable_workfn(struct kthread_work *work) if (!(p->scx.flags & SCX_TASK_SUB_INIT)) continue; - scoped_guard (sched_change, p, DEQUEUE_SAVE | DEQUEUE_MOVE) { + scoped_guard (sched_change, p, p->sched_class, + DEQUEUE_SAVE | DEQUEUE_MOVE) { /* * $p must be either READY or ENABLED. If ENABLED, * __scx_disabled_and_exit_task() first disables and diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index 437c4f68c9e53..ca436a8174d30 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -2705,6 +2705,7 @@ struct sched_class { /* * sched_change */ + void (*prepare_switch)(struct rq *this_rq, struct task_struct *task); void (*switching_from)(struct rq *this_rq, struct task_struct *task); void (*switched_from) (struct rq *this_rq, struct task_struct *task); void (*switching_to) (struct rq *this_rq, struct task_struct *task); @@ -4214,13 +4215,16 @@ struct sched_change_ctx { bool running; }; -struct sched_change_ctx *sched_change_begin(struct task_struct *p, unsigned int flags); +struct sched_change_ctx * +sched_change_begin(struct task_struct *p, const struct sched_class *next_class, + unsigned int flags); void sched_change_end(struct sched_change_ctx *ctx); DEFINE_CLASS(sched_change, struct sched_change_ctx *, sched_change_end(_T), - sched_change_begin(p, flags), - struct task_struct *p, unsigned int flags) + sched_change_begin(p, next_class, flags), + struct task_struct *p, const struct sched_class *next_class, + unsigned int flags) DEFINE_CLASS_IS_UNCONDITIONAL(sched_change) diff --git a/kernel/sched/syscalls.c b/kernel/sched/syscalls.c index b215b0ead9a60..bc32ce76ff4fe 100644 --- a/kernel/sched/syscalls.c +++ b/kernel/sched/syscalls.c @@ -85,7 +85,7 @@ void set_user_nice(struct task_struct *p, long nice) return; } - scoped_guard (sched_change, p, DEQUEUE_SAVE) { + scoped_guard (sched_change, p, p->sched_class, DEQUEUE_SAVE) { p->static_prio = NICE_TO_PRIO(nice); set_load_weight(p, true); old_prio = p->prio; @@ -678,7 +678,7 @@ int __sched_setscheduler(struct task_struct *p, if (prev_class != next_class) queue_flags |= DEQUEUE_CLASS; - scoped_guard (sched_change, p, queue_flags) { + scoped_guard (sched_change, p, next_class, queue_flags) { if (!(attr->sched_flags & SCHED_FLAG_KEEP_PARAMS)) { __setscheduler_params(p, attr); -- 2.55.0