[PATCH 08/18] sched/fair: Optimize push task mechanism for fair

From: Vincent Guittot

Date: Fri Oct 02 2026 - 11:50:04 EST


Instead of always unlocking local rq in order to lock both task and rq
in a safe order, just try to lock the task. If the task is already locked
by something else its state will probably change and the conditions used
add it in the pushable list are probably not true anymore. As a result
skipping the push sequence seems like a good choice.

Signed-off-by: Vincent Guittot <vincent.guittot@xxxxxxxxxx>
---
kernel/sched/core.c | 4 ++--
kernel/sched/fair.c | 48 ++++++++++++++++++++++++--------------------
kernel/sched/sched.h | 3 ---
3 files changed, 28 insertions(+), 27 deletions(-)

diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 18692b752814..837dc74c9a8d 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -2553,8 +2553,8 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu)
*
* Returns (locked) new rq. Old rq's lock is released.
*/
-struct rq *move_queued_task(struct rq *rq, struct rq_flags *rf,
- struct task_struct *p, int new_cpu)
+static struct rq *move_queued_task(struct rq *rq, struct rq_flags *rf,
+ struct task_struct *p, int new_cpu)
__must_hold(__rq_lockp(rq))
{
lockdep_assert_rq_held(rq);
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 00078ac7fada..6ba2efeba435 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -9871,8 +9871,7 @@ static bool fair_push_task(struct rq *rq)
{
struct task_struct *next_task;
int prev_cpu, new_cpu;
- struct rq_flags rf;
- struct rq *cur_rq;
+ struct rq *new_rq;

next_task = pick_next_pushable_fair_task(rq);
if (!next_task)
@@ -9881,38 +9880,41 @@ static bool fair_push_task(struct rq *rq)
if (is_migration_disabled(next_task))
return true;

- /* We might release rq lock */
- get_task_struct(next_task);
-
prev_cpu = rq->cpu;

/*
- * We need to release rq lock and take both task and rq w/o
- * triggering a deadlock.
+ * The safe lock ordering for task and rq is task 1st then rq but we
+ * already get the rq so just try to get task too. If task is already
+ * locked, it is waiting for the rq's lock and it is about to change
+ * task state so skipping the push sequence in order to speed up the
+ * release of the lock is the best choice.
*/
- raw_spin_rq_unlock(rq);
-
- cur_rq = task_rq_lock(next_task, &rf);
-
- /* Task already migrated */
- if (cur_rq->cpu != prev_cpu)
- goto skip_push;
+ if (!raw_spin_trylock(&next_task->pi_lock))
+ return true;

new_cpu = select_task_rq_fair(next_task, prev_cpu, 0);

/* Task doesn't need to migrate */
if (new_cpu == prev_cpu)
- goto skip_push;
+ goto no_push;
+
+ new_rq = cpu_rq(new_cpu);

- update_rq_clock(cur_rq);
- cur_rq = move_queued_task(cur_rq, &rf, next_task, new_cpu);
+ deactivate_task(rq, next_task, 0);
+ set_task_cpu(next_task, new_cpu);
+ raw_spin_rq_unlock(rq);

-skip_push:
- task_rq_unlock(cur_rq, next_task, &rf);
+ raw_spin_rq_lock(new_rq);
+ WARN_ON_ONCE(task_cpu(next_task) != new_cpu);
+ activate_task(new_rq, next_task, 0);
+ wakeup_preempt(new_rq, next_task, 0);
+ raw_spin_rq_unlock(new_rq);

- /* Restore rq state */
+ /* Restore rq lock state */
raw_spin_rq_lock(rq);
- put_task_struct(next_task);
+
+no_push:
+ raw_spin_unlock(&next_task->pi_lock);

return true;
}
@@ -10432,6 +10434,9 @@ static void put_prev_task_fair(struct rq *rq, struct task_struct *prev, struct t
* if it is still active.
*/
fair_add_pushable_prev(rq, prev, next);
+
+ if (next && next != prev)
+ fair_queue_push_tasks(rq);
}

/*
@@ -15584,7 +15589,6 @@ static void set_next_task_fair(struct rq *rq, struct task_struct *p, enum snt_e

update_misfit_status(p, rq);
sched_fair_update_stop_tick(rq, p);
- fair_queue_push_tasks(rq);

repick:
/*
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 74130bd2a2c8..293f23620282 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -4213,9 +4213,6 @@ void move_queued_task_locked(struct rq *src_rq, struct rq *dst_rq, struct task_s
wakeup_preempt(dst_rq, task, 0);
}

-extern struct rq *move_queued_task(struct rq *rq, struct rq_flags *rf,
- struct task_struct *p, int new_cpu);
-
static inline
bool task_is_pushable(struct rq *rq, struct task_struct *p, int cpu)
{
--
2.53.0