[PATCH 05/15] sched: Add prepare_switch() class callback
Andrea Righi <[email protected]> Tue, 28 Jul 2026 17:43:23 +0200
| Newsgroups | dev.linux.lists.sched-ext,org.kernel.vger.linux-kernel |
|---|---|
| Message-ID | <[email protected]> |
sched_change_begin() only invokes callbacks on the outgoing class before recording the task queueing state. An incoming class may need to prepare a task while that state is still live, before the normal dequeue and class transition take place. Pass the incoming class to sched_change_begin() and invoke its optional prepare_switch() callback before recording the queued and running state. This is a preparatory change to support proxy execution with sched_ext. Suggested-by: Tejun Heo <[email protected]> Signed-off-by: Andrea Righi <[email protected]> --- kernel/sched/core.c | 16 +++++++++++----- kernel/sched/ext/ext.c | 7 ++++--- kernel/sched/ext/sub.c | 9 ++++++--- kernel/sched/sched.h | 10 +++++++--- kernel/sched/syscalls.c | 4 ++-- 5 files changed, 30 insertions(+), 16 deletions(-) diff --git a/kernel/sched/core.c b/kernel/sched/core.c index 2cc6c9d16200c..cff007d8b5e26 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -2785,7 +2785,7 @@ void set_cpus_allowed_common(struct task_struct *p, struct affinity_context *ctx static void do_set_cpus_allowed(struct task_struct *p, struct affinity_context *ctx) { - scoped_guard (sched_change, p, DEQUEUE_SAVE) + scoped_guard (sched_change, p, p->sched_class, DEQUEUE_SAVE) p->sched_class->set_cpus_allowed(p, ctx); } @@ -7715,7 +7715,7 @@ void rt_mutex_setprio(struct task_struct *p, struct task_struct *pi_task) if (prev_class != next_class) queue_flag |= DEQUEUE_CLASS; - scoped_guard (sched_change, p, queue_flag) { + scoped_guard (sched_change, p, next_class, queue_flag) { /* * Boosting condition are: * 1. -rt task is running and holds mutex A @@ -8392,7 +8392,7 @@ int migrate_task_to(struct task_struct *p, int target_cpu) void sched_setnuma(struct task_struct *p, int nid) { guard(task_rq_lock)(p); - scoped_guard (sched_change, p, DEQUEUE_SAVE) + scoped_guard (sched_change, p, p->sched_class, DEQUEUE_SAVE) p->numa_preferred_nid = nid; } #endif /* CONFIG_NUMA_BALANCING */ @@ -9509,7 +9509,7 @@ void sched_move_task(struct task_struct *tsk, bool for_autogroup) CLASS(task_rq_lock, rq_guard)(tsk); rq = rq_guard.rq; - scoped_guard (sched_change, tsk, queue_flags) { + scoped_guard (sched_change, tsk, tsk->sched_class, queue_flags) { sched_change_group(tsk); if (!for_autogroup) scx_cgroup_move_task(tsk); @@ -11213,7 +11213,9 @@ static inline void sched_mm_cid_fork(struct task_struct *t) { } static DEFINE_PER_CPU(struct sched_change_ctx, sched_change_ctx); -struct sched_change_ctx *sched_change_begin(struct task_struct *p, unsigned int flags) +struct sched_change_ctx * +sched_change_begin(struct task_struct *p, const struct sched_class *next_class, + unsigned int flags) { struct sched_change_ctx *ctx = this_cpu_ptr(&sched_change_ctx); struct rq *rq = task_rq(p); @@ -11231,6 +11233,10 @@ struct sched_change_ctx *sched_change_begin(struct task_struct *p, unsigned int flags |= DEQUEUE_NOCLOCK; } + if ((flags & DEQUEUE_CLASS) && next_class != p->sched_class && + next_class->prepare_switch) + next_class->prepare_switch(rq, p); + if ((flags & DEQUEUE_CLASS) && p->sched_class->switching_from) p->sched_class->switching_from(rq, p); diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c index 312938179e524..69ab59d176ce0 100644 --- a/kernel/sched/ext/ext.c +++ b/kernel/sched/ext/ext.c @@ -5861,7 +5861,8 @@ void scx_bypass(struct scx_sched *sch, bool bypass) continue; /* cycling deq/enq is enough, see the function comment */ - scoped_guard (sched_change, p, DEQUEUE_SAVE | DEQUEUE_MOVE) { + scoped_guard (sched_change, p, p->sched_class, + DEQUEUE_SAVE | DEQUEUE_MOVE) { /* nothing */ ; } } @@ -6140,7 +6141,7 @@ static void scx_root_disable(struct scx_sched *sch) if (old_class != new_class) queue_flags |= DEQUEUE_CLASS; - scoped_guard (sched_change, p, queue_flags) { + scoped_guard (sched_change, p, new_class, queue_flags) { p->sched_class = new_class; } @@ -7508,7 +7509,7 @@ static void scx_root_enable_workfn(struct kthread_work *work) if (old_class != new_class) queue_flags |= DEQUEUE_CLASS; - scoped_guard (sched_change, p, queue_flags) { + scoped_guard (sched_change, p, new_class, queue_flags) { set_task_slice(p, READ_ONCE(sch->slice_dfl)); p->sched_class = new_class; } diff --git a/kernel/sched/ext/sub.c b/kernel/sched/ext/sub.c index daf7fbd3d0c3b..8758317035754 100644 --- a/kernel/sched/ext/sub.c +++ b/kernel/sched/ext/sub.c @@ -794,7 +794,8 @@ static void scx_rehome_task(struct scx_sched *to, struct task_struct *p) lockdep_assert_held(&p->pi_lock); lockdep_assert_rq_held(task_rq(p)); - scoped_guard (sched_change, p, DEQUEUE_SAVE | DEQUEUE_MOVE) { + scoped_guard (sched_change, p, p->sched_class, + DEQUEUE_SAVE | DEQUEUE_MOVE) { scx_disable_and_exit_task(scx_task_sched(p), p); scx_set_task_state(p, SCX_TASK_INIT_BEGIN); scx_set_task_state(p, SCX_TASK_INIT); @@ -824,7 +825,8 @@ static void scx_punt_task(struct scx_sched *to, struct task_struct *p) lockdep_assert_rq_held(task_rq(p)); WARN_ON_ONCE(!READ_ONCE(to->bypass_depth)); - scoped_guard (sched_change, p, DEQUEUE_SAVE | DEQUEUE_MOVE) { + scoped_guard (sched_change, p, p->sched_class, + DEQUEUE_SAVE | DEQUEUE_MOVE) { scx_disable_and_exit_task(scx_task_sched(p), p); scx_set_task_sched(p, to); } @@ -1476,7 +1478,8 @@ void scx_sub_enable_workfn(struct kthread_work *work) if (!(p->scx.flags & SCX_TASK_SUB_INIT)) continue; - scoped_guard (sched_change, p, DEQUEUE_SAVE | DEQUEUE_MOVE) { + scoped_guard (sched_change, p, p->sched_class, + DEQUEUE_SAVE | DEQUEUE_MOVE) { /* * $p must be either READY or ENABLED. If ENABLED, * __scx_disabled_and_exit_task() first disables and diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index 6119bfd2de60d..79daacc2389e1 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -2704,6 +2704,7 @@ struct sched_class { /* * sched_change */ + void (*prepare_switch)(struct rq *this_rq, struct task_struct *task); void (*switching_from)(struct rq *this_rq, struct task_struct *task); void (*switched_from) (struct rq *this_rq, struct task_struct *task); void (*switching_to) (struct rq *this_rq, struct task_struct *task); @@ -4213,13 +4214,16 @@ struct sched_change_ctx { bool running; }; -struct sched_change_ctx *sched_change_begin(struct task_struct *p, unsigned int flags); +struct sched_change_ctx * +sched_change_begin(struct task_struct *p, const struct sched_class *next_class, + unsigned int flags); void sched_change_end(struct sched_change_ctx *ctx); DEFINE_CLASS(sched_change, struct sched_change_ctx *, sched_change_end(_T), - sched_change_begin(p, flags), - struct task_struct *p, unsigned int flags) + sched_change_begin(p, next_class, flags), + struct task_struct *p, const struct sched_class *next_class, + unsigned int flags) DEFINE_CLASS_IS_UNCONDITIONAL(sched_change) diff --git a/kernel/sched/syscalls.c b/kernel/sched/syscalls.c index 903b47f5d0b74..9ab23f765f7a7 100644 --- a/kernel/sched/syscalls.c +++ b/kernel/sched/syscalls.c @@ -85,7 +85,7 @@ void set_user_nice(struct task_struct *p, long nice) return; } - scoped_guard (sched_change, p, DEQUEUE_SAVE) { + scoped_guard (sched_change, p, p->sched_class, DEQUEUE_SAVE) { p->static_prio = NICE_TO_PRIO(nice); set_load_weight(p, true); old_prio = p->prio; @@ -679,7 +679,7 @@ int __sched_setscheduler(struct task_struct *p, prev_class != next_class) queue_flags |= DEQUEUE_CLASS; - scoped_guard (sched_change, p, queue_flags) { + scoped_guard (sched_change, p, next_class, queue_flags) { if (!(attr->sched_flags & SCHED_FLAG_KEEP_PARAMS)) { __setscheduler_params(p, attr); -- 2.55.0