* [PATCH 01/18 v2] sched/eevdf: Decay positive lag of sleeping entities
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
@ 2026-10-02 15:43 ` Vincent Guittot
2026-10-02 15:43 ` [PATCH 02/18] sched/eevdf: Reset lag when waking up on idle cpu Vincent Guittot
` (16 subsequent siblings)
17 siblings, 0 replies; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:43 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
Similarly to delayed dequeue that enables an entity to decay its negative
lag while sleeping, a task should not keep a positive lag forever.
The sleep duration and the weight of the entity is used to decay the
positive lag at wakeup.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/fair.c | 51 +++++++++++++++++++++++++++++++++++++++++++--
1 file changed, 49 insertions(+), 2 deletions(-)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 03206e15e6fe..8cda1d39b037 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -897,6 +897,51 @@ bool update_entity_lag(struct cfs_rq *cfs_rq, struct sched_entity *se)
return avruntime - vlag != se->vruntime;
}
+static inline unsigned long cfs_rq_load_avg(struct cfs_rq *cfs_rq);
+
+static __always_inline
+void decay_entity_lag(struct cfs_rq *cfs_rq, struct sched_entity *se, int flags)
+{
+ s64 delta_exec, vlag = se->vlag;
+ unsigned long cfs_load;
+ struct rq *rq;
+
+ WARN_ON_ONCE(se->on_rq);
+
+ /* Negative lag implies delayed dequeue */
+ if (vlag <= 0)
+ return;
+
+ rq = rq_of(cfs_rq);
+
+ if (flags & ENQUEUE_MIGRATED)
+ return;
+
+ /* Compute sleep time */
+ delta_exec = rq_clock_task(rq) - se->exec_start;
+ if (unlikely(delta_exec <= 0))
+ return;
+
+ /* For anything above ~4 seconds, save computation and clear the lag */
+ if (unlikely(delta_exec >> 32)) {
+ se->vlag = 0;
+ return;
+ }
+
+ cfs_load = cfs_rq_load_avg(cfs_rq);
+ if (cfs_load) {
+ unsigned long weight = scale_load_down(se->h_load.weight);
+
+ delta_exec *= weight;
+ delta_exec = div64_long(delta_exec, cfs_load + weight);
+ }
+
+ vlag -= calc_delta_fair(delta_exec, se);
+
+ /* vlag can't become neg while sleeping */
+ se->vlag = max(0, vlag);
+}
+
/*
* Entity is eligible once it received less service than it ought to have,
* eg. lag >= 0.
@@ -7987,7 +8032,7 @@ static void
enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
{
int rq_h_nr_queued = rq->cfs.h_nr_queued;
- int task_new = !(flags & ENQUEUE_WAKEUP);
+ int task_wake = flags & ENQUEUE_WAKEUP;
struct sched_entity *se = &p->se;
struct cfs_rq *cfs_rq = &rq->cfs;
unsigned long weight;
@@ -8019,6 +8064,8 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
if (p->in_iowait)
cpufreq_update_util(rq, SCHED_CPUFREQ_IOWAIT);
+ if (task_wake)
+ decay_entity_lag(cfs_rq, se, flags);
if (se->on_rq && se->sched_delayed)
requeue_delayed_entity(cfs_rq, se);
@@ -8048,7 +8095,7 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
* into account, but that is not straightforward to implement,
* and the following generally works well enough in practice.
*/
- if (!task_new)
+ if (task_wake)
check_update_overutilized_status(rq);
assert_list_leaf_cfs_rq(rq);
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* [PATCH 02/18] sched/eevdf: Reset lag when waking up on idle cpu
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
2026-10-02 15:43 ` [PATCH 01/18 v2] sched/eevdf: Decay positive lag of sleeping entities Vincent Guittot
@ 2026-10-02 15:43 ` Vincent Guittot
2026-10-04 17:30 ` Kayra Cizmeci
2026-10-02 15:44 ` [PATCH 03/18 v2] sched/eevdf: Add per cpu cached min_slice Vincent Guittot
` (15 subsequent siblings)
17 siblings, 1 reply; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:43 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
When several tasks wake up simultaneously on an idle CPU, their final vlag
will depend of the ordering as the first one will lose its lag but not
the next ones.
Reset the lag when the enqueue happens while no fair task has already been
picked et set as the running task.
As a typical example:
CPU0 is idle
TA with vlag 0ms and TB with vlag 5ms wake up on CPU0 simultaneously.
Depending which grab the lock 1st the behavior will be different:
If TA is enqueued 1st, TB will be enqueued with a positive lag and will
be picked 1st.
But if TB is enqueued 1st, it will loose its positive vlag and both TA and
TB will have 0 vlag when fair will pick a task.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
include/linux/sched.h | 1 +
kernel/sched/fair.c | 17 +++++++++++++++++
kernel/sched/sched.h | 1 +
3 files changed, 19 insertions(+)
diff --git a/include/linux/sched.h b/include/linux/sched.h
index d7cc77181ef9..32d7077ef148 100644
--- a/include/linux/sched.h
+++ b/include/linux/sched.h
@@ -592,6 +592,7 @@ struct sched_entity {
u64 vruntime;
/* Approximated virtual lag: */
s64 vlag;
+ u32 vlag_seq;
/* 'Protected' deadline, to give out minimum quantums: */
u64 vprot;
u64 slice;
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 8cda1d39b037..4c8f12fc8869 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -893,6 +893,7 @@ bool update_entity_lag(struct cfs_rq *cfs_rq, struct sched_entity *se)
vlag = min(vlag, 0);
}
se->vlag = vlag;
+ se->vlag_seq = cfs_rq->idle_seq;
return avruntime - vlag != se->vruntime;
}
@@ -914,9 +915,22 @@ void decay_entity_lag(struct cfs_rq *cfs_rq, struct sched_entity *se, int flags)
rq = rq_of(cfs_rq);
+ /* You can't claim any lag when waking on idle CPU */
+ if (rq->curr == rq->idle) {
+ se->vlag = 0;
+ return;
+ }
+
+ /* Accessing remote rq task clock is a cost */
if (flags & ENQUEUE_MIGRATED)
return;
+ /* CPU has been idle in between so the lag has been removed */
+ if (se->vlag_seq != cfs_rq->idle_seq) {
+ se->vlag = 0;
+ return;
+ }
+
/* Compute sleep time */
delta_exec = rq_clock_task(rq) - se->exec_start;
if (unlikely(delta_exec <= 0))
@@ -8190,6 +8204,9 @@ static bool __dequeue_task(struct rq *rq, struct task_struct *p, int flags)
dequeue_hierarchy(p, flags);
+ if (!cfs_rq->h_nr_queued)
+ cfs_rq->idle_seq++;
+
if (sched_feat(PLACE_REL_DEADLINE) && !task_sleep) {
se->deadline -= se->vruntime;
se->rel_deadline = 1;
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index b98084e1f5b0..69a2a749e188 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -689,6 +689,7 @@ struct cfs_rq {
u64 sum_weight;
u64 zero_vruntime;
unsigned int sum_shift;
+ u32 idle_seq;
#ifdef CONFIG_SCHED_CORE
unsigned int forceidle_seq;
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* Re: [PATCH 02/18] sched/eevdf: Reset lag when waking up on idle cpu
2026-10-02 15:43 ` [PATCH 02/18] sched/eevdf: Reset lag when waking up on idle cpu Vincent Guittot
@ 2026-10-04 17:30 ` Kayra Cizmeci
0 siblings, 0 replies; 24+ messages in thread
From: Kayra Cizmeci @ 2026-10-04 17:30 UTC (permalink / raw)
To: vincent.guittot
Cc: arighi, bsegall, changwoo, christian.loehle, dietmar.eggemann,
juri.lelli, kprateek.nayak, linux-kernel, linux-pm, lukasz.luba,
mgorman, mingo, peterz, pierre.gondois, qyousef, rafael, rostedt,
sched-ext, sshegde, tj, void, vschneid
> When several tasks wake up simultaneously on an idle CPU, their final vlag
> will depend of the ordering as the first one will lose its lag but not
> the next ones.
> Reset the lag when the enqueue happens while no fair task has already been
> picked et set as the running task.
There is a typo. ('et', also 'depend' and 'loose' Not sure that's all :>)
> As a typical example:
> CPU0 is idle
> TA with vlag 0ms and TB with vlag 5ms wake up on CPU0 simultaneously.
> Depending which grab the lock 1st the behavior will be different:
> If TA is enqueued 1st, TB will be enqueued with a positive lag and will
> be picked 1st.
> But if TB is enqueued 1st, it will loose its positive vlag and both TA and
> TB will have 0 vlag when fair will pick a task.
> diff --git a/include/linux/sched.h b/include/linux/sched.h
> index d7cc77181ef9..32d7077ef148 100644
> --- a/include/linux/sched.h
> +++ b/include/linux/sched.h
> @@ -592,6 +592,7 @@ struct sched_entity {
> u64 vruntime;
> /* Approximated virtual lag: */
> s64 vlag;
> + u32 vlag_seq;
> /* 'Protected' deadline, to give out minimum quantums: */
> u64 vprot;
> u64 slice;
> diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
> index 8cda1d39b037..4c8f12fc8869 100644
> --- a/kernel/sched/fair.c
> +++ b/kernel/sched/fair.c
> @@ -893,6 +893,7 @@ bool update_entity_lag(struct cfs_rq *cfs_rq, struct sched_entity *se)
> vlag = min(vlag, 0);
> }
> se->vlag = vlag;
> + se->vlag_seq = cfs_rq->idle_seq;
>
> return avruntime - vlag != se->vruntime;
> }
> @@ -914,9 +915,22 @@ void decay_entity_lag(struct cfs_rq *cfs_rq, struct sched_entity *se, int flags)
>
> rq = rq_of(cfs_rq);
>
> + /* You can't claim any lag when waking on idle CPU */
> + if (rq->curr == rq->idle) {
> + se->vlag = 0;
> + return;
> + }
> +
> + /* Accessing remote rq task clock is a cost */
> if (flags & ENQUEUE_MIGRATED)
> return;
>
> + /* CPU has been idle in between so the lag has been removed */
> + if (se->vlag_seq != cfs_rq->idle_seq) {
> + se->vlag = 0;
> + return;
> + }
> +
> /* Compute sleep time */
> delta_exec = rq_clock_task(rq) - se->exec_start;
> if (unlikely(delta_exec <= 0))
> @@ -8190,6 +8204,9 @@ static bool __dequeue_task(struct rq *rq, struct task_struct *p, int flags)
>
> dequeue_hierarchy(p, flags);
>
> + if (!cfs_rq->h_nr_queued)
> + cfs_rq->idle_seq++;
Shouldn't this skip DEQUEUE_SAVE?
> +
> if (sched_feat(PLACE_REL_DEADLINE) && !task_sleep) {
> se->deadline -= se->vruntime;
> se->rel_deadline = 1;
> diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
> index b98084e1f5b0..69a2a749e188 100644
> --- a/kernel/sched/sched.h
> +++ b/kernel/sched/sched.h
> @@ -689,6 +689,7 @@ struct cfs_rq {
> u64 sum_weight;
> u64 zero_vruntime;
> unsigned int sum_shift;
> + u32 idle_seq;
>
> #ifdef CONFIG_SCHED_CORE
> unsigned int forceidle_seq;
Yeah, well AFAICT other things seems OK.
Also, while trying to get this patch series on my tip
branch I had a hard time. I was at the caves
of git and b4 figthing with... Everything.
Like there were no v2 tags on some of the patches
so b4 didn't tracked them. I had to do some
things by hand and finally it worked. (Not too greatly tho.)
But who cares? :>.
Thanks,
Kayra
^ permalink raw reply [flat|nested] 24+ messages in thread
* [PATCH 03/18 v2] sched/eevdf: Add per cpu cached min_slice
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
2026-10-02 15:43 ` [PATCH 01/18 v2] sched/eevdf: Decay positive lag of sleeping entities Vincent Guittot
2026-10-02 15:43 ` [PATCH 02/18] sched/eevdf: Reset lag when waking up on idle cpu Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 04/18 v2] sched/eevdf: Compare min slice during wake_affine Vincent Guittot
` (14 subsequent siblings)
17 siblings, 0 replies; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
In order to use min_slice during cpu selection, update a cached value when
needed after en/dequeing a new task. This cached value includes current
task.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/core.c | 1 +
kernel/sched/fair.c | 20 ++++++++++++++++++++
kernel/sched/sched.h | 1 +
3 files changed, 22 insertions(+)
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index ee9b443f760d..f38cf5a37a8a 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -8902,6 +8902,7 @@ void __init sched_init(void)
rq->sd = NULL;
rq->rd = NULL;
rq->cpu_capacity = SCHED_CAPACITY_SCALE;
+ rq->min_slice = ULONG_MAX;
rq->balance_callback = &balance_push_callback;
rq->active_balance = 0;
rq->next_balance = jiffies;
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 4c8f12fc8869..ad72b8536d6c 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -1116,6 +1116,22 @@ static inline bool min_vruntime_update(struct sched_entity *se, bool exit)
RB_DECLARE_CALLBACKS(static, min_vruntime_cb, struct sched_entity,
run_node, min_vruntime, min_vruntime_update);
+/*
+ * Entity's slice is in the range [100us:100ms].
+ */
+static unsigned long get_rq_min_slice(struct rq *rq)
+{
+ return READ_ONCE(rq->min_slice);
+}
+
+static void __update_rq_min_slice(struct rq *rq)
+{
+ unsigned long min = cfs_rq_min_slice(&rq->cfs);
+
+ if (min != get_rq_min_slice(rq))
+ WRITE_ONCE(rq->min_slice, min);
+}
+
/*
* Enqueue an entity into the rb-tree:
*/
@@ -8089,6 +8105,8 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED);
__enqueue_entity(cfs_rq, se);
+ __update_rq_min_slice(rq);
+
if (!rq_h_nr_queued && rq->cfs.h_nr_queued)
dl_server_start(&rq->fair_server);
@@ -8214,6 +8232,8 @@ static bool __dequeue_task(struct rq *rq, struct task_struct *p, int flags)
if (se != cfs_rq->curr)
__dequeue_entity(cfs_rq, se);
+ __update_rq_min_slice(rq);
+
sub_nr_running(rq, 1);
/* balance early to pull high priority tasks */
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 69a2a749e188..e025d2a6c302 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -1173,6 +1173,7 @@ struct rq {
#endif
unsigned int ttwu_pending;
unsigned long cpu_capacity;
+ unsigned long min_slice;
#ifdef CONFIG_SCHED_PROXY_EXEC
struct task_struct __rcu *donor; /* Scheduling context */
struct task_struct __rcu *curr; /* Execution context */
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* [PATCH 04/18 v2] sched/eevdf: Compare min slice during wake_affine
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (2 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 03/18 v2] sched/eevdf: Add per cpu cached min_slice Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-05 15:58 ` Kayra Cizmeci
2026-10-02 15:44 ` [PATCH 05/18 v2] sched/eevdf: Add min slice check when selecting CPU Vincent Guittot
` (13 subsequent siblings)
17 siblings, 1 reply; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
Add a new level in wake affine where we check on which CPU the task
would most probably run 1st between this and prev CPUs.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/fair.c | 20 ++++++++++++++++++++
1 file changed, 20 insertions(+)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index ad72b8536d6c..eeac0aaba3cd 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -8422,6 +8422,9 @@ static int wake_wide(struct task_struct *p)
* wake_affine_idle() - only considers 'now', it check if the waking CPU is
* cache-affine and is (or will be) idle.
*
+ * wake_affine_slice() - only considers 'now', it check if the waking CPU can
+ * be preempted becaus using longerslice.
+ *
* wake_affine_weight() - considers the weight to reflect the average
* scheduling latency of the CPUs. This seems to work
* for the overloaded case.
@@ -8457,6 +8460,20 @@ wake_affine_idle(int this_cpu, int prev_cpu, int sync)
return nr_cpumask_bits;
}
+static int
+wake_affine_slice(struct task_struct *p, int this_cpu, int prev_cpu)
+{
+ struct sched_entity *se = &p->se;
+
+ if (se->slice < get_rq_min_slice(cpu_rq(prev_cpu)))
+ return prev_cpu;
+
+ if (se->slice < get_rq_min_slice(cpu_rq(this_cpu)))
+ return this_cpu;
+
+ return nr_cpumask_bits;
+}
+
static int
wake_affine_weight(struct sched_domain *sd, struct task_struct *p,
int this_cpu, int prev_cpu, int sync)
@@ -8508,6 +8525,9 @@ static int wake_affine(struct sched_domain *sd, struct task_struct *p,
if (sched_feat(WA_IDLE))
target = wake_affine_idle(this_cpu, prev_cpu, sync);
+ if (sched_feat(PREEMPT_SHORT) && target == nr_cpumask_bits)
+ target = wake_affine_slice(p, this_cpu, prev_cpu);
+
if (sched_feat(WA_WEIGHT) && target == nr_cpumask_bits)
target = wake_affine_weight(sd, p, this_cpu, prev_cpu, sync);
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* Re: [PATCH 04/18 v2] sched/eevdf: Compare min slice during wake_affine
2026-10-02 15:44 ` [PATCH 04/18 v2] sched/eevdf: Compare min slice during wake_affine Vincent Guittot
@ 2026-10-05 15:58 ` Kayra Cizmeci
0 siblings, 0 replies; 24+ messages in thread
From: Kayra Cizmeci @ 2026-10-05 15:58 UTC (permalink / raw)
To: vincent.guittot
Cc: arighi, bsegall, changwoo, christian.loehle, dietmar.eggemann,
juri.lelli, kprateek.nayak, linux-kernel, linux-pm, lukasz.luba,
mgorman, mingo, peterz, pierre.gondois, qyousef, rafael, rostedt,
sched-ext, sshegde, tj, void, vschneid
> Add a new level in wake affine where we check on which CPU the task
> would most probably run 1st between this and prev CPUs.
> Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
> diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
> index ad72b8536d6c..eeac0aaba3cd 100644
> --- a/kernel/sched/fair.c
> +++ b/kernel/sched/fair.c
> @@ -8422,6 +8422,9 @@ static int wake_wide(struct task_struct *p)
> * wake_affine_idle() - only considers 'now', it check if the waking CPU is
> * cache-affine and is (or will be) idle.
> *
> + * wake_affine_slice() - only considers 'now', it check if the waking CPU can
> + * be preempted becaus using longerslice.
> + *
> * wake_affine_weight() - considers the weight to reflect the average
> * scheduling latency of the CPUs. This seems to work
> * for the overloaded case.
>
Nice. A typo. 'becaus'. 'be' wanted to be part of Santa Claus instead of 'cause'.
> +static int
> +wake_affine_slice(struct task_struct *p, int this_cpu, int prev_cpu)
> +{
> + struct sched_entity *se = &p->se;
> +
> + if (se->slice < get_rq_min_slice(cpu_rq(prev_cpu)))
> + return prev_cpu;
> +
> + if (se->slice < get_rq_min_slice(cpu_rq(this_cpu)))
> + return this_cpu;
> +
> + return nr_cpumask_bits;
> +}
> +
> static int
> wake_affine_weight(struct sched_domain *sd, struct task_struct *p,
> int this_cpu, int prev_cpu, int sync)
> @@ -8508,6 +8525,9 @@ static int wake_affine(struct sched_domain *sd, struct task_struct *p,
> if (sched_feat(WA_IDLE))
> target = wake_affine_idle(this_cpu, prev_cpu, sync);
>
> + if (sched_feat(PREEMPT_SHORT) && target == nr_cpumask_bits)
> + target = wake_affine_slice(p, this_cpu, prev_cpu);
> +
> if (sched_feat(WA_WEIGHT) && target == nr_cpumask_bits)
> target = wake_affine_weight(sd, p, this_cpu, prev_cpu, sync);
>
Also,
Scene (Why I always start like this? Answer is... IDK neither. :>):
CPU0 has an RT task running named TA, and CPU1 has a fair task running
named TB that has 100 ms slice.
cfs_rq_min_slice() looks only to rq's cfs_rq skipping others like dl
and rt. The value that's coming from cfs_rq_min_slice() is then checked if it
equals to the currently saved value.
So if the cfs_rq is empty, cfs_rq_min_slice() just returns the starting value of min,
that is ~0ULL. And we write this.
In the case of CPU0 and CPU1 the behavior will change whenether or not which one
of these is prev or this CPU. If CPU0 is prev it will be chosen, if not CPU1 will.
IDK if this is tolerated tho. But shouldn't this be changed?
Or Am I getting something wrong?
(To y'all that are currently attending to LPC, enjoy! I'm sadly only enjoying my room.)
Thanks,
Kayra :>
^ permalink raw reply [flat|nested] 24+ messages in thread
* [PATCH 05/18 v2] sched/eevdf: Add min slice check when selecting CPU
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (3 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 04/18 v2] sched/eevdf: Compare min slice during wake_affine Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-04 19:18 ` Kayra Cizmeci
2026-10-02 15:44 ` [PATCH 06/18 v2] sched/fair: Prepare select_task_rq_fair() to be called for new cases Vincent Guittot
` (12 subsequent siblings)
17 siblings, 1 reply; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
Add a new level for selecting CPU when select_task_rq_fair() fails to find
an idle CPU. This last level will compare the slice to select a CPU where
the task could run 1st.
This helps a waking task to select a CPU where a longer slice runs
instead of one where a task with the same or shorter slice already run.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/fair.c | 106 ++++++++++++++++++++++++++++----------------
1 file changed, 67 insertions(+), 39 deletions(-)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index eeac0aaba3cd..19a0e67827f7 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -8771,10 +8771,11 @@ static int select_idle_smt(struct task_struct *p, struct sched_domain *sd, int t
* comparing the average scan cost (tracked in sd->avg_scan_cost) against the
* average idle time for this rq (as found in rq->avg_idle).
*/
-static int select_idle_cpu(struct task_struct *p, struct sched_domain *sd, bool has_idle_core, int target)
+static int select_idle_cpu(struct task_struct *p, struct sched_domain *sd, bool has_idle_core, int *best)
{
struct cpumask *cpus = this_cpu_cpumask_var_ptr(select_rq_mask);
- int i, cpu, idle_cpu = -1, nr = INT_MAX;
+ int i, cpu, idle_cpu = -1, slice_cpu = -1, target = *best, nr = INT_MAX;
+ unsigned long task_slice;
if (sched_feat(SIS_UTIL) && sd->shared) {
/*
@@ -8795,6 +8796,8 @@ static int select_idle_cpu(struct task_struct *p, struct sched_domain *sd, bool
if (!cpumask_and(cpus, sched_domain_span(sd), p->cpus_ptr))
return -1;
+ task_slice = p->se.slice;
+
if (static_branch_unlikely(&sched_cluster_active)) {
struct sched_group *sg = sd->groups;
@@ -8814,6 +8817,10 @@ static int select_idle_cpu(struct task_struct *p, struct sched_domain *sd, bool
if ((unsigned int)idle_cpu < nr_cpumask_bits)
return idle_cpu;
}
+
+ if (slice_cpu == -1 &&
+ task_slice < get_rq_min_slice(cpu_rq(cpu)))
+ slice_cpu = cpu;
}
cpumask_andnot(cpus, cpus, sched_group_span(sg));
}
@@ -8832,11 +8839,18 @@ static int select_idle_cpu(struct task_struct *p, struct sched_domain *sd, bool
if ((unsigned int)idle_cpu < nr_cpumask_bits)
break;
}
+
+ if (slice_cpu == -1 &&
+ task_slice < get_rq_min_slice(cpu_rq(cpu)))
+ slice_cpu = cpu;
}
if (has_idle_core)
set_idle_cores(target, false);
+ if (slice_cpu != -1)
+ *best = slice_cpu;
+
return idle_cpu;
}
@@ -8851,14 +8865,19 @@ static int select_idle_cpu(struct task_struct *p, struct sched_domain *sd, bool
*
* Rank Val Tier Meaning
* ------------------------------ --- ------ ---------------------------
- * ASYM_IDLE_UCLAMP_MISFIT -4 core Idle core; capacity fits
+ * ASYM_BUSY_FITS -6 core Busy core but capacity fits
+ * and task can preempt current.
+ * ASYM_IDLE_UCLAMP_MISFIT -5 core Idle core; capacity fits
* util but uclamp_min misses.
- * ASYM_IDLE_COMPLETE_MISFIT -3 core Idle core; capacity does
+ * ASYM_IDLE_COMPLETE_MISFIT -4 core Idle core; capacity does
* not fit. Still beats every
* thread-tier rank: a busy
* sibling cuts effective
* capacity more than a
* misfit hurts a quiet core.
+ * ASYM_BUSY_THREAD_FITS -3 thread Busy CPU and SMT sibling but
+ * capacity fits and task can
+ * preempt current.
* ASYM_IDLE_THREAD_FITS -2 thread Busy SMT sibling; capacity
* fits util + uclamp.
* ASYM_IDLE_THREAD_UCLAMP_MISFIT -1 thread Busy SMT sibling; capacity
@@ -8868,24 +8887,26 @@ static int select_idle_cpu(struct task_struct *p, struct sched_domain *sd, bool
* ASYM_IDLE_THREAD_MISFIT 0 thread Busy SMT sibling; capacity
* does not fit.
*
- * ASYM_IDLE_CORE_BIAS (-3) is an offset, not a state. On an idle core,
+ * ASYM_IDLE_CORE_BIAS (-4) is an offset, not a state. On an idle core,
* fits += ASYM_IDLE_CORE_BIAS rebases thread-tier ranks into the core tier:
*
- * ASYM_IDLE_THREAD_UCLAMP_MISFIT (-1) + BIAS -> ASYM_IDLE_UCLAMP_MISFIT (-4)
- * ASYM_IDLE_THREAD_MISFIT (0) + BIAS -> ASYM_IDLE_COMPLETE_MISFIT (-3)
+ * ASYM_IDLE_THREAD_UCLAMP_MISFIT (-1) + BIAS -> ASYM_IDLE_UCLAMP_MISFIT (-5)
+ * ASYM_IDLE_THREAD_MISFIT (0) + BIAS -> ASYM_IDLE_COMPLETE_MISFIT (-4)
*
* ASYM_IDLE_THREAD_FITS (-2) is never rebased because a fully-fitting idle-core
* candidate early-returns from select_idle_capacity().
*/
enum asym_fits_state {
- ASYM_IDLE_UCLAMP_MISFIT = -4,
+ ASYM_BUSY_FITS = -6,
+ ASYM_IDLE_UCLAMP_MISFIT,
ASYM_IDLE_COMPLETE_MISFIT,
ASYM_IDLE_THREAD_FITS,
+ ASYM_BUSY_THREAD_FITS,
ASYM_IDLE_THREAD_UCLAMP_MISFIT,
ASYM_IDLE_THREAD_MISFIT,
/* util_fits_cpu() bias for idle core */
- ASYM_IDLE_CORE_BIAS = -3,
+ ASYM_IDLE_CORE_BIAS = -4,
};
/*
@@ -8904,6 +8925,7 @@ select_idle_capacity(struct task_struct *p, struct sched_domain *sd, int target)
bool has_idle_core = sched_smt_active() && test_idle_cores(target);
unsigned long task_util, util_min, util_max, best_cap = 0;
int fits, best_fits = ASYM_IDLE_THREAD_MISFIT;
+ unsigned long task_slice;
int cpu, best_cpu = -1;
struct cpumask *cpus;
int nr = INT_MAX;
@@ -8914,6 +8936,7 @@ select_idle_capacity(struct task_struct *p, struct sched_domain *sd, int target)
task_util = task_util_est(p);
util_min = uclamp_eff_value(p, UCLAMP_MIN);
util_max = uclamp_eff_value(p, UCLAMP_MAX);
+ task_slice = p->se.slice;
if (sched_feat(SIS_UTIL) && sd->shared) {
/*
@@ -8937,42 +8960,47 @@ select_idle_capacity(struct task_struct *p, struct sched_domain *sd, int target)
if (!has_idle_core && --nr <= 0)
return best_cpu;
- if (!choose_idle_cpu(cpu, p))
- continue;
-
fits = util_fits_cpu(task_util, util_min, util_max, cpu);
- /*
- * Perfect fit: capacity satisfies util + uclamp and the CPU
- * sits on a fully-idle SMT core, this is a !SMT system, or
- * there is no idle core to find.
- * Short-circuit the rank-based selection and return
- * immediately.
- */
- if (fits > 0 && preferred_core)
- return cpu;
- /*
- * Only the min performance hint (i.e. uclamp_min) doesn't fit.
- * Look for the CPU with best capacity.
- */
- else if (fits < 0)
+ if (choose_idle_cpu(cpu, p)) {
+ /*
+ * Perfect fit: capacity satisfies util + uclamp and the CPU
+ * sits on a fully-idle SMT core, this is a !SMT system, or
+ * there is no idle core to find.
+ * Short-circuit the rank-based selection and return
+ * immediately.
+ */
+ if (fits > 0 && preferred_core)
+ return cpu;
+ /*
+ * Only the min performance hint (i.e. uclamp_min) doesn't fit.
+ * Look for the CPU with best capacity.
+ */
+ else if (fits < 0)
+ cpu_cap = get_actual_cpu_capacity(cpu);
+ /*
+ * fits > 0 implies we are not on a preferred core, but the util
+ * fits CPU capacity. Set fits to ASYM_IDLE_THREAD_FITS
+ * so the effective range becomes
+ * [ASYM_IDLE_THREAD_FITS, ASYM_IDLE_THREAD_MISFIT], where:
+ * ASYM_IDLE_THREAD_MISFIT - does not fit
+ * ASYM_IDLE_THREAD_UCLAMP_MISFIT - fits with the exception of UCLAMP_MIN
+ * ASYM_IDLE_THREAD_FITS - fits with the exception of preferred_core
+ */
+ else if (fits > 0)
+ fits = ASYM_IDLE_THREAD_FITS;
+
+ } else if (fits > 0 && task_slice < get_rq_min_slice(cpu_rq(cpu))) {
+ fits = ASYM_BUSY_THREAD_FITS;
cpu_cap = get_actual_cpu_capacity(cpu);
- /*
- * fits > 0 implies we are not on a preferred core, but the util
- * fits CPU capacity. Set fits to ASYM_IDLE_THREAD_FITS
- * so the effective range becomes
- * [ASYM_IDLE_THREAD_FITS, ASYM_IDLE_THREAD_MISFIT], where:
- * ASYM_IDLE_THREAD_MISFIT - does not fit
- * ASYM_IDLE_THREAD_UCLAMP_MISFIT - fits with the exception of UCLAMP_MIN
- * ASYM_IDLE_THREAD_FITS - fits with the exception of preferred_core
- */
- else if (fits > 0)
- fits = ASYM_IDLE_THREAD_FITS;
+ } else {
+ continue;
+ }
/*
* If we are on a preferred core, translate the range of fits
* of [ASYM_IDLE_THREAD_UCLAMP_MISFIT, ASYM_IDLE_THREAD_MISFIT] to
- * [ASYM_IDLE_UCLAMP_MISFIT, ASYM_IDLE_COMPLETE_MISFIT].
+ * [ASYM_IDLE_THREAD_MISFIT_IDLE_UCLAMP_MISFIT, ASYM_IDLE_COMPLETE_MISFIT].
* This ensures that an idle core is always given priority over
* (partially) busy core.
*
@@ -9147,7 +9175,7 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
}
}
- i = select_idle_cpu(p, sd, has_idle_core, target);
+ i = select_idle_cpu(p, sd, has_idle_core, &target);
if ((unsigned)i < nr_cpumask_bits)
return i;
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* Re: [PATCH 05/18 v2] sched/eevdf: Add min slice check when selecting CPU
2026-10-02 15:44 ` [PATCH 05/18 v2] sched/eevdf: Add min slice check when selecting CPU Vincent Guittot
@ 2026-10-04 19:18 ` Kayra Cizmeci
0 siblings, 0 replies; 24+ messages in thread
From: Kayra Cizmeci @ 2026-10-04 19:18 UTC (permalink / raw)
To: vincent.guittot
Cc: arighi, bsegall, changwoo, christian.loehle, dietmar.eggemann,
juri.lelli, kprateek.nayak, linux-kernel, linux-pm, lukasz.luba,
mgorman, mingo, peterz, pierre.gondois, qyousef, rafael, rostedt,
sched-ext, sshegde, tj, void, vschneid
Add a new level for selecting CPU when select_task_rq_fair() fails to find
an idle CPU. This last level will compare the slice to select a CPU where
the task could run 1st.
This helps a waking task to select a CPU where a longer slice runs
instead of one where a task with the same or shorter slice already run.
There's a typo. 'run'
> @@ -8771,10 +8771,11 @@ static int select_idle_smt(struct task_struct *p, struct sched_domain *sd, int t
> * comparing the average scan cost (tracked in sd->avg_scan_cost) against the
> * average idle time for this rq (as found in rq->avg_idle).
> */
> -static int select_idle_cpu(struct task_struct *p, struct sched_domain *sd, bool has_idle_core, int target)
> +static int select_idle_cpu(struct task_struct *p, struct sched_domain *sd, bool has_idle_core, int *best)
> {
> struct cpumask *cpus = this_cpu_cpumask_var_ptr(select_rq_mask);
> - int i, cpu, idle_cpu = -1, nr = INT_MAX;
> + int i, cpu, idle_cpu = -1, slice_cpu = -1, target = *best, nr = INT_MAX;
> + unsigned long task_slice;
>
> if (sched_feat(SIS_UTIL) && sd->shared) {
> /*
> @@ -8795,6 +8796,8 @@ static int select_idle_cpu(struct task_struct *p, struct sched_domain *sd, bool
> if (!cpumask_and(cpus, sched_domain_span(sd), p->cpus_ptr))
> return -1;
>
> + task_slice = p->se.slice;
> +
> if (static_branch_unlikely(&sched_cluster_active)) {
> struct sched_group *sg = sd->groups;
>
> @@ -8814,6 +8817,10 @@ static int select_idle_cpu(struct task_struct *p, struct sched_domain *sd, bool
> if ((unsigned int)idle_cpu < nr_cpumask_bits)
> return idle_cpu;
> }
> +
> + if (slice_cpu == -1 &&
> + task_slice < get_rq_min_slice(cpu_rq(cpu)))
> + slice_cpu = cpu;
> }
> cpumask_andnot(cpus, cpus, sched_group_span(sg));
> }
> @@ -8832,11 +8839,18 @@ static int select_idle_cpu(struct task_struct *p, struct sched_domain *sd, bool
> if ((unsigned int)idle_cpu < nr_cpumask_bits)
> break;
> }
> +
> + if (slice_cpu == -1 &&
> + task_slice < get_rq_min_slice(cpu_rq(cpu)))
> + slice_cpu = cpu;
> }
>
> if (has_idle_core)
> set_idle_cores(target, false);
>
> + if (slice_cpu != -1)
> + *best = slice_cpu;
> +
> return idle_cpu;
> }
Also,
Scene:
Let's say that in our domain there are 8 CPU's and has_idle_core is false. And nr equals 4.
And sched_cluster_active is true. select_idle_cpu() enters the for_each_cpu_wrap block
and the else branch. Because nr decrases each time, and if there are not any idle_cpu's
even if we found any slice_cpu it's not set to best. We just leave the loop after checking 3 CPU's.
The same thing a bit differently happens below on the main loop too.
But I don't think is more important than typos. Typos is the reason we're here. They are the or nevermind.
This is a joke btw. (that, destroyed my masterpiece and created a new one. Saying a joke is a joke because it is,
is one thing while saying a joke is a joke as a joke is one thing. I wanna go on with this but it'll be to long. :<)
I could be getting something wrong tho. (No one is perfect, but I'm not perfect at all so ya know.)
Thanks,
Kayra
^ permalink raw reply [flat|nested] 24+ messages in thread
* [PATCH 06/18 v2] sched/fair: Prepare select_task_rq_fair() to be called for new cases
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (4 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 05/18 v2] sched/eevdf: Add min slice check when selecting CPU Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-06 22:25 ` Tim Chen
2026-10-02 15:44 ` [PATCH 07/18] sched/fair: Add push task mechanism for fair Vincent Guittot
` (11 subsequent siblings)
17 siblings, 1 reply; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
Update select_task_rq_fair() to be called out of the 3 current cases which
are :
- wake up
- exec
- fork
We wants to select a rq in some new cases like pushing a runnable task on a
better CPU than the local one. In such case, it's not a wakeup , nor an
exec nor a fork. We make sure to not distrub these cases but still
go through EAS and fast-path.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/core.c | 2 +-
kernel/sched/fair.c | 57 ++++++++++++++++++++++++++-------------------
2 files changed, 34 insertions(+), 25 deletions(-)
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index f38cf5a37a8a..837dc74c9a8d 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -3618,7 +3618,7 @@ static int select_fallback_rq(int cpu, struct task_struct *p)
}
/*
- * The caller (fork, wakeup) owns p->pi_lock, ->cpus_ptr is stable.
+ * The caller (fork, wakeup, push) owns p->pi_lock, ->cpus_ptr is stable.
*/
static inline
int select_task_rq(struct task_struct *p, int cpu, int *wake_flags)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 19a0e67827f7..f12678850ce2 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -9800,46 +9800,55 @@ static int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu)
}
/*
- * select_task_rq_fair: Select target runqueue for the waking task in domains
- * that have the relevant SD flag set. In practice, this is SD_BALANCE_WAKE,
- * SD_BALANCE_FORK, or SD_BALANCE_EXEC.
+ * select_task_rq_fair: Select a target runqueue for the task.
+ * There are 2 ways to select the target runqueue:
+ * - The fast path which only looks for an idle CPU in the LLC or the smallest
+ * asymmetric domain (i.e. the lowest domain with all compute capacities).
+ * - The slow path which looks for the idlest CPU in the highest domain with
+ * the relevant SD flag set.
*
- * Balances load by selecting the idlest CPU in the idlest group, or under
- * certain conditions an idle sibling CPU if the domain has SD_WAKE_AFFINE set.
+ * In practice, WF_EXEC and WF_FORK uses the slow path whereas WF_TTWU and no
+ * flag (Push) uses the fast path.
*
- * Returns the target CPU number.
*/
static int
-select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
+select_task_rq_fair(struct task_struct *p, int prev_cpu, int select_flags)
{
- int sync = (wake_flags & WF_SYNC) && !(current->flags & PF_EXITING);
+ int sync = (select_flags & WF_SYNC) && !(current->flags & PF_EXITING);
+ int want_sibling = !(select_flags & (WF_EXEC | WF_FORK));
+ int new_cpu, cpu = smp_processor_id();
struct sched_domain *tmp, *sd = NULL;
- int cpu = smp_processor_id();
- int new_cpu = prev_cpu;
- int want_affine = 0;
/* SD_flags and WF_flags share the first nibble */
- int sd_flag = wake_flags & 0xF;
+ int sd_flag = select_flags & 0xF;
+ int want_affine = 0;
/*
- * required for stable ->cpus_allowed
+ * Required for stable ->cpus_allowed
*/
lockdep_assert_held(&p->pi_lock);
- if (wake_flags & WF_TTWU) {
+
+ if (select_flags & WF_TTWU) {
record_wakee(p);
- if ((wake_flags & WF_CURRENT_CPU) &&
+ if ((select_flags & WF_CURRENT_CPU) &&
cpumask_test_cpu(cpu, p->cpus_ptr))
return cpu;
+ }
- if (!is_rd_overutilized(this_rq()->rd)) {
- new_cpu = find_energy_efficient_cpu(p, prev_cpu);
- if (new_cpu >= 0)
- return new_cpu;
- new_cpu = prev_cpu;
- }
+ /*
+ * We don't want EAS to be called for exec or fork but it should be
+ * called for any other case such as wake up or push callback.
+ */
+ if (!is_rd_overutilized(this_rq()->rd) && want_sibling) {
+ new_cpu = find_energy_efficient_cpu(p, prev_cpu);
+ if (new_cpu >= 0)
+ return new_cpu;
+ }
+ if (select_flags & WF_TTWU)
want_affine = !wake_wide(p) && cpumask_test_cpu(cpu, p->cpus_ptr);
- }
+
+ new_cpu = prev_cpu;
for_each_domain(cpu, tmp) {
/*
@@ -9871,8 +9880,8 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
return sched_balance_find_dst_cpu(sd, p, cpu, prev_cpu, sd_flag);
/* Fast path */
- if (wake_flags & WF_TTWU)
- return select_idle_sibling(p, prev_cpu, new_cpu);
+ if (want_sibling)
+ new_cpu = select_idle_sibling(p, prev_cpu, new_cpu);
return new_cpu;
}
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* Re: [PATCH 06/18 v2] sched/fair: Prepare select_task_rq_fair() to be called for new cases
2026-10-02 15:44 ` [PATCH 06/18 v2] sched/fair: Prepare select_task_rq_fair() to be called for new cases Vincent Guittot
@ 2026-10-06 22:25 ` Tim Chen
0 siblings, 0 replies; 24+ messages in thread
From: Tim Chen @ 2026-10-06 22:25 UTC (permalink / raw)
To: Vincent Guittot, mingo, peterz, juri.lelli, dietmar.eggemann,
rostedt, bsegall, mgorman, vschneid, kprateek.nayak,
linux-kernel, lukasz.luba, rafael, linux-pm, tj, void, arighi,
changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde
On Fri, 2026-10-02 at 17:44 +0200, Vincent Guittot wrote:
> Update select_task_rq_fair() to be called out of the 3 current cases which
> are :
> - wake up
> - exec
> - fork
>
> We wants to select a rq in some new cases like pushing a runnable task on a
> better CPU than the local one. In such case, it's not a wakeup , nor an
> exec nor a fork. We make sure to not distrub these cases but still
> go through EAS and fast-path.
>
> Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
> ---
> kernel/sched/core.c | 2 +-
> kernel/sched/fair.c | 57 ++++++++++++++++++++++++++-------------------
> 2 files changed, 34 insertions(+), 25 deletions(-)
>
> diff --git a/kernel/sched/core.c b/kernel/sched/core.c
> index f38cf5a37a8a..837dc74c9a8d 100644
> --- a/kernel/sched/core.c
> +++ b/kernel/sched/core.c
> @@ -3618,7 +3618,7 @@ static int select_fallback_rq(int cpu, struct task_struct *p)
> }
>
> /*
> - * The caller (fork, wakeup) owns p->pi_lock, ->cpus_ptr is stable.
> + * The caller (fork, wakeup, push) owns p->pi_lock, ->cpus_ptr is stable.
> */
> static inline
> int select_task_rq(struct task_struct *p, int cpu, int *wake_flags)
> diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
> index 19a0e67827f7..f12678850ce2 100644
> --- a/kernel/sched/fair.c
> +++ b/kernel/sched/fair.c
> @@ -9800,46 +9800,55 @@ static int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu)
> }
>
> /*
> - * select_task_rq_fair: Select target runqueue for the waking task in domains
> - * that have the relevant SD flag set. In practice, this is SD_BALANCE_WAKE,
> - * SD_BALANCE_FORK, or SD_BALANCE_EXEC.
> + * select_task_rq_fair: Select a target runqueue for the task.
> + * There are 2 ways to select the target runqueue:
> + * - The fast path which only looks for an idle CPU in the LLC or the smallest
> + * asymmetric domain (i.e. the lowest domain with all compute capacities).
> + * - The slow path which looks for the idlest CPU in the highest domain with
> + * the relevant SD flag set.
> *
> - * Balances load by selecting the idlest CPU in the idlest group, or under
> - * certain conditions an idle sibling CPU if the domain has SD_WAKE_AFFINE set.
> + * In practice, WF_EXEC and WF_FORK uses the slow path whereas WF_TTWU and no
> + * flag (Push) uses the fast path.
> *
> - * Returns the target CPU number.
> */
> static int
> -select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
> +select_task_rq_fair(struct task_struct *p, int prev_cpu, int select_flags)
> {
> - int sync = (wake_flags & WF_SYNC) && !(current->flags & PF_EXITING);
> + int sync = (select_flags & WF_SYNC) && !(current->flags & PF_EXITING);
> + int want_sibling = !(select_flags & (WF_EXEC | WF_FORK));
> + int new_cpu, cpu = smp_processor_id();
> struct sched_domain *tmp, *sd = NULL;
> - int cpu = smp_processor_id();
> - int new_cpu = prev_cpu;
> - int want_affine = 0;
> /* SD_flags and WF_flags share the first nibble */
> - int sd_flag = wake_flags & 0xF;
> + int sd_flag = select_flags & 0xF;
> + int want_affine = 0;
>
> /*
> - * required for stable ->cpus_allowed
> + * Required for stable ->cpus_allowed
> */
> lockdep_assert_held(&p->pi_lock);
> - if (wake_flags & WF_TTWU) {
> +
> + if (select_flags & WF_TTWU) {
> record_wakee(p);
>
> - if ((wake_flags & WF_CURRENT_CPU) &&
> + if ((select_flags & WF_CURRENT_CPU) &&
> cpumask_test_cpu(cpu, p->cpus_ptr))
> return cpu;
> + }
>
> - if (!is_rd_overutilized(this_rq()->rd)) {
> - new_cpu = find_energy_efficient_cpu(p, prev_cpu);
> - if (new_cpu >= 0)
> - return new_cpu;
> - new_cpu = prev_cpu;
> - }
> + /*
> + * We don't want EAS to be called for exec or fork but it should be
> + * called for any other case such as wake up or push callback.
> + */
> + if (!is_rd_overutilized(this_rq()->rd) && want_sibling) {
> + new_cpu = find_energy_efficient_cpu(p, prev_cpu);
> + if (new_cpu >= 0)
> + return new_cpu;
> + }
>
> + if (select_flags & WF_TTWU)
> want_affine = !wake_wide(p) && cpumask_test_cpu(cpu, p->cpus_ptr);
> - }
> +
> + new_cpu = prev_cpu;
>
> for_each_domain(cpu, tmp) {
> /*
> @@ -9871,8 +9880,8 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
> return sched_balance_find_dst_cpu(sd, p, cpu, prev_cpu, sd_flag);
>
> /* Fast path */
> - if (wake_flags & WF_TTWU)
> - return select_idle_sibling(p, prev_cpu, new_cpu);
> + if (want_sibling)
> + new_cpu = select_idle_sibling(p, prev_cpu, new_cpu);
Hi Vincent,
With this change, a push also goes through select_idle_sibling(), and
select_idle_sibling() sets p->recent_used_cpu = prev on every call.
For a wakeup, prev is the CPU where the task ran last, and the old
value of the hint becomes a second candidate for the next wakeup. For
a push, prev is the CPU where the task is queued now. If the push
doesn't move the task, the hint becomes the current CPU. When the task
later sleeps and wakes up on this CPU, recent_used_cpu is equal to
prev, so the wakeup has no second candidate. Each push attempt that
fails does this again.
Perhaps something like the following fix below on top of the series. It
passes the select flags to select_idle_sibling() and
updates the hint only for WF_TTWU. When fair_push_task() moves the
task, it saves the source CPU as the hint, which is what the hint
holds after a wakeup migration.
Tim
---
kernel/sched/fair.c | 11 +++++++----
1 file changed, 7 insertions(+), 4 deletions(-)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index fd22731949c5..3eb8a0170902 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -1381,7 +1381,7 @@ static bool update_deadline(struct cfs_rq *cfs_rq, struct sched_entity *se)
#include "pelt.h"
-static int select_idle_sibling(struct task_struct *p, int prev_cpu, int cpu);
+static int select_idle_sibling(struct task_struct *p, int prev_cpu, int cpu, int select_flags);
static unsigned long task_h_load(struct task_struct *p);
static unsigned long capacity_of(int cpu);
@@ -9072,7 +9072,7 @@ static inline bool asym_fits_cpu(unsigned long util,
/*
* Try and locate an idle core/thread in the LLC cache domain.
*/
-static int select_idle_sibling(struct task_struct *p, int prev, int target)
+static int select_idle_sibling(struct task_struct *p, int prev, int target, int select_flags)
{
bool has_idle_core = false;
struct sched_domain *sd;
@@ -9131,7 +9131,9 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
/* Check a recently used CPU as a potential idle candidate: */
recent_used_cpu = p->recent_used_cpu;
- p->recent_used_cpu = prev;
+ /* A push only updates the hint when it moves the task */
+ if (select_flags & WF_TTWU)
+ p->recent_used_cpu = prev;
if (recent_used_cpu != prev &&
recent_used_cpu != target &&
cpus_share_cache(recent_used_cpu, target) &&
@@ -10015,6 +10017,7 @@ static bool fair_push_task(struct rq *rq)
deactivate_task(rq, next_task, 0);
set_task_cpu(next_task, new_cpu);
+ next_task->recent_used_cpu = prev_cpu;
raw_spin_rq_unlock(rq);
raw_spin_rq_lock(new_rq);
@@ -10225,7 +10228,7 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int select_flags)
/* Fast path */
if (want_sibling)
- new_cpu = select_idle_sibling(p, prev_cpu, new_cpu);
+ new_cpu = select_idle_sibling(p, prev_cpu, new_cpu, select_flags);
return new_cpu;
}
^ permalink raw reply [flat|nested] 24+ messages in thread
* [PATCH 07/18] sched/fair: Add push task mechanism for fair
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (5 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 06/18 v2] sched/fair: Prepare select_task_rq_fair() to be called for new cases Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 08/18] sched/fair: Optimize " Vincent Guittot
` (10 subsequent siblings)
17 siblings, 0 replies; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
EAS is based on wakeup events to efficiently place tasks on the system, but
there are cases where a task doesn't have wakeup events anymore or at a far
too low pace. For such situation, we can take advantage of the task being
put back in the enqueued list to check if it should be pushed on another
CPU.
Add a push task mechanism that enables fair scheduler to push runnable
tasks. EAS will be one user but other feature like filling idle CPUs or
short slice tasks can also take advantage of it.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/core.c | 4 +-
kernel/sched/fair.c | 163 +++++++++++++++++++++++++++++++++++++++++++
kernel/sched/sched.h | 7 ++
3 files changed, 172 insertions(+), 2 deletions(-)
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 837dc74c9a8d..18692b752814 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -2553,8 +2553,8 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu)
*
* Returns (locked) new rq. Old rq's lock is released.
*/
-static struct rq *move_queued_task(struct rq *rq, struct rq_flags *rf,
- struct task_struct *p, int new_cpu)
+struct rq *move_queued_task(struct rq *rq, struct rq_flags *rf,
+ struct task_struct *p, int new_cpu)
__must_hold(__rq_lockp(rq))
{
lockdep_assert_rq_held(rq);
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index f12678850ce2..00078ac7fada 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -8179,6 +8179,8 @@ static void dequeue_hierarchy(struct task_struct *p, int flags)
}
}
+static void fair_remove_pushable_task(struct rq *rq, struct task_struct *p);
+
/*
* The part of dequeue_task_fair() that is needed to dequeue delayed tasks.
*
@@ -8195,6 +8197,7 @@ static bool __dequeue_task(struct rq *rq, struct task_struct *p, int flags)
bool task_delayed = flags & DEQUEUE_DELAYED;
clear_buddies(cfs_rq, se);
+ fair_remove_pushable_task(rq, p);
update_curr_eevdf(cfs_rq);
update_entity_lag(cfs_rq, se);
@@ -9799,6 +9802,157 @@ static int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu)
return target;
}
+DEFINE_STATIC_KEY_FALSE(sched_push_task);
+
+static inline bool sched_push_task_enabled(void)
+{
+ return static_branch_unlikely(&sched_push_task);
+}
+
+static bool __check_pushable_fair_task(struct rq *rq, struct task_struct *p)
+{
+ if (!task_on_rq_queued(p))
+ return false;
+
+ if (p->se.sched_delayed)
+ return false;
+
+ if (p->nr_cpus_allowed <= 1)
+ return false;
+
+ return true;
+}
+
+static bool fair_check_pushable_task(struct rq *rq, struct task_struct *p, struct task_struct *next)
+{
+ if (!__check_pushable_fair_task(rq, p))
+ return false;
+
+ return false;
+}
+
+static inline int has_pushable_tasks(struct rq *rq)
+{
+ return !plist_head_empty(&rq->cfs.pushable_tasks);
+}
+
+static struct task_struct *pick_next_pushable_fair_task(struct rq *rq)
+{
+ struct task_struct *p;
+
+ if (!has_pushable_tasks(rq))
+ return NULL;
+
+ p = plist_first_entry(&rq->cfs.pushable_tasks,
+ struct task_struct, pushable_tasks);
+
+ WARN_ON_ONCE(rq->cpu != task_cpu(p));
+ WARN_ON_ONCE(task_current(rq, p));
+ WARN_ON_ONCE(p->nr_cpus_allowed <= 1);
+ WARN_ON_ONCE(!task_on_rq_queued(p));
+
+ /*
+ * Remove task from the pushable list as we try only once after that
+ * the task has been put back in enqueued list.
+ */
+ plist_del(&p->pushable_tasks, &rq->cfs.pushable_tasks);
+
+ return p;
+}
+
+static int
+select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags);
+
+/*
+ * See if the non running fair tasks on this rq can be sent on other CPUs
+ * that fits better with their profile.
+ */
+static bool fair_push_task(struct rq *rq)
+{
+ struct task_struct *next_task;
+ int prev_cpu, new_cpu;
+ struct rq_flags rf;
+ struct rq *cur_rq;
+
+ next_task = pick_next_pushable_fair_task(rq);
+ if (!next_task)
+ return false;
+
+ if (is_migration_disabled(next_task))
+ return true;
+
+ /* We might release rq lock */
+ get_task_struct(next_task);
+
+ prev_cpu = rq->cpu;
+
+ /*
+ * We need to release rq lock and take both task and rq w/o
+ * triggering a deadlock.
+ */
+ raw_spin_rq_unlock(rq);
+
+ cur_rq = task_rq_lock(next_task, &rf);
+
+ /* Task already migrated */
+ if (cur_rq->cpu != prev_cpu)
+ goto skip_push;
+
+ new_cpu = select_task_rq_fair(next_task, prev_cpu, 0);
+
+ /* Task doesn't need to migrate */
+ if (new_cpu == prev_cpu)
+ goto skip_push;
+
+ update_rq_clock(cur_rq);
+ cur_rq = move_queued_task(cur_rq, &rf, next_task, new_cpu);
+
+skip_push:
+ task_rq_unlock(cur_rq, next_task, &rf);
+
+ /* Restore rq state */
+ raw_spin_rq_lock(rq);
+ put_task_struct(next_task);
+
+ return true;
+}
+
+static void fair_push_tasks(struct rq *rq)
+{
+ /* fair_push_task() will return true if it moved a fair task */
+ while (fair_push_task(rq))
+ ;
+}
+
+static DEFINE_PER_CPU(struct balance_callback, fair_push_head);
+
+static inline void fair_queue_push_tasks(struct rq *rq)
+{
+ if (!sched_push_task_enabled() || !has_pushable_tasks(rq))
+ return;
+
+ queue_balance_callback(rq, &per_cpu(fair_push_head, rq->cpu), fair_push_tasks);
+}
+
+static void fair_remove_pushable_task(struct rq *rq, struct task_struct *p)
+{
+ if (sched_push_task_enabled())
+ plist_del(&p->pushable_tasks, &rq->cfs.pushable_tasks);
+}
+
+static void __fair_add_pushable_task(struct rq *rq, struct task_struct *p)
+{
+ plist_del(&p->pushable_tasks, &rq->cfs.pushable_tasks);
+ plist_node_init(&p->pushable_tasks, p->prio);
+ plist_add(&p->pushable_tasks, &rq->cfs.pushable_tasks);
+}
+
+static void fair_add_pushable_prev(struct rq *rq, struct task_struct *prev, struct task_struct *next)
+{
+ if (sched_push_task_enabled() && fair_check_pushable_task(rq, prev, next))
+ __fair_add_pushable_task(rq, prev);
+}
+
/*
* select_task_rq_fair: Select a target runqueue for the task.
* There are 2 ways to select the target runqueue:
@@ -10272,6 +10426,12 @@ static void put_prev_task_fair(struct rq *rq, struct task_struct *prev, struct t
cfs_rq->curr = NULL;
if (se->on_rq)
__enqueue_entity(cfs_rq, se);
+
+ /*
+ * The previous task might be eligible for being pushed on another cpu
+ * if it is still active.
+ */
+ fair_add_pushable_prev(rq, prev, next);
}
/*
@@ -15379,6 +15539,7 @@ static void set_next_task_fair(struct rq *rq, struct task_struct *p, enum snt_e
goto repick;
clear_buddies(cfs_rq, se);
+ fair_remove_pushable_task(rq, p);
if (on_rq)
__dequeue_entity(cfs_rq, se);
@@ -15423,6 +15584,7 @@ static void set_next_task_fair(struct rq *rq, struct task_struct *p, enum snt_e
update_misfit_status(p, rq);
sched_fair_update_stop_tick(rq, p);
+ fair_queue_push_tasks(rq);
repick:
/*
@@ -15439,6 +15601,7 @@ void init_cfs_rq(struct cfs_rq *cfs_rq)
{
cfs_rq->tasks_timeline = RB_ROOT_CACHED;
cfs_rq->zero_vruntime = (u64)(-(1LL << 20));
+ plist_head_init(&cfs_rq->pushable_tasks);
raw_spin_lock_init(&cfs_rq->removed.lock);
}
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index e025d2a6c302..74130bd2a2c8 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -721,6 +721,8 @@ struct cfs_rq {
unsigned long runnable_avg;
} removed;
+ struct plist_head pushable_tasks;
+
#ifdef CONFIG_FAIR_GROUP_SCHED
u64 last_update_tg_load_avg;
unsigned long tg_load_avg_contrib;
@@ -3877,6 +3879,8 @@ static inline bool sched_energy_enabled(void) { return false; }
#endif /* !(CONFIG_ENERGY_MODEL && CONFIG_CPU_FREQ_GOV_SCHEDUTIL) */
+DECLARE_STATIC_KEY_FALSE(sched_push_task);
+
#ifdef CONFIG_MEMBARRIER
/*
@@ -4209,6 +4213,9 @@ void move_queued_task_locked(struct rq *src_rq, struct rq *dst_rq, struct task_s
wakeup_preempt(dst_rq, task, 0);
}
+extern struct rq *move_queued_task(struct rq *rq, struct rq_flags *rf,
+ struct task_struct *p, int new_cpu);
+
static inline
bool task_is_pushable(struct rq *rq, struct task_struct *p, int cpu)
{
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* [PATCH 08/18] sched/fair: Optimize push task mechanism for fair
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (6 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 07/18] sched/fair: Add push task mechanism for fair Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 09/18 v2] sched/core: Add rq flag to tick parameters Vincent Guittot
` (9 subsequent siblings)
17 siblings, 0 replies; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
Instead of always unlocking local rq in order to lock both task and rq
in a safe order, just try to lock the task. If the task is already locked
by something else its state will probably change and the conditions used
add it in the pushable list are probably not true anymore. As a result
skipping the push sequence seems like a good choice.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/core.c | 4 ++--
kernel/sched/fair.c | 48 ++++++++++++++++++++++++--------------------
kernel/sched/sched.h | 3 ---
3 files changed, 28 insertions(+), 27 deletions(-)
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 18692b752814..837dc74c9a8d 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -2553,8 +2553,8 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu)
*
* Returns (locked) new rq. Old rq's lock is released.
*/
-struct rq *move_queued_task(struct rq *rq, struct rq_flags *rf,
- struct task_struct *p, int new_cpu)
+static struct rq *move_queued_task(struct rq *rq, struct rq_flags *rf,
+ struct task_struct *p, int new_cpu)
__must_hold(__rq_lockp(rq))
{
lockdep_assert_rq_held(rq);
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 00078ac7fada..6ba2efeba435 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -9871,8 +9871,7 @@ static bool fair_push_task(struct rq *rq)
{
struct task_struct *next_task;
int prev_cpu, new_cpu;
- struct rq_flags rf;
- struct rq *cur_rq;
+ struct rq *new_rq;
next_task = pick_next_pushable_fair_task(rq);
if (!next_task)
@@ -9881,38 +9880,41 @@ static bool fair_push_task(struct rq *rq)
if (is_migration_disabled(next_task))
return true;
- /* We might release rq lock */
- get_task_struct(next_task);
-
prev_cpu = rq->cpu;
/*
- * We need to release rq lock and take both task and rq w/o
- * triggering a deadlock.
+ * The safe lock ordering for task and rq is task 1st then rq but we
+ * already get the rq so just try to get task too. If task is already
+ * locked, it is waiting for the rq's lock and it is about to change
+ * task state so skipping the push sequence in order to speed up the
+ * release of the lock is the best choice.
*/
- raw_spin_rq_unlock(rq);
-
- cur_rq = task_rq_lock(next_task, &rf);
-
- /* Task already migrated */
- if (cur_rq->cpu != prev_cpu)
- goto skip_push;
+ if (!raw_spin_trylock(&next_task->pi_lock))
+ return true;
new_cpu = select_task_rq_fair(next_task, prev_cpu, 0);
/* Task doesn't need to migrate */
if (new_cpu == prev_cpu)
- goto skip_push;
+ goto no_push;
+
+ new_rq = cpu_rq(new_cpu);
- update_rq_clock(cur_rq);
- cur_rq = move_queued_task(cur_rq, &rf, next_task, new_cpu);
+ deactivate_task(rq, next_task, 0);
+ set_task_cpu(next_task, new_cpu);
+ raw_spin_rq_unlock(rq);
-skip_push:
- task_rq_unlock(cur_rq, next_task, &rf);
+ raw_spin_rq_lock(new_rq);
+ WARN_ON_ONCE(task_cpu(next_task) != new_cpu);
+ activate_task(new_rq, next_task, 0);
+ wakeup_preempt(new_rq, next_task, 0);
+ raw_spin_rq_unlock(new_rq);
- /* Restore rq state */
+ /* Restore rq lock state */
raw_spin_rq_lock(rq);
- put_task_struct(next_task);
+
+no_push:
+ raw_spin_unlock(&next_task->pi_lock);
return true;
}
@@ -10432,6 +10434,9 @@ static void put_prev_task_fair(struct rq *rq, struct task_struct *prev, struct t
* if it is still active.
*/
fair_add_pushable_prev(rq, prev, next);
+
+ if (next && next != prev)
+ fair_queue_push_tasks(rq);
}
/*
@@ -15584,7 +15589,6 @@ static void set_next_task_fair(struct rq *rq, struct task_struct *p, enum snt_e
update_misfit_status(p, rq);
sched_fair_update_stop_tick(rq, p);
- fair_queue_push_tasks(rq);
repick:
/*
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 74130bd2a2c8..293f23620282 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -4213,9 +4213,6 @@ void move_queued_task_locked(struct rq *src_rq, struct rq *dst_rq, struct task_s
wakeup_preempt(dst_rq, task, 0);
}
-extern struct rq *move_queued_task(struct rq *rq, struct rq_flags *rf,
- struct task_struct *p, int new_cpu);
-
static inline
bool task_is_pushable(struct rq *rq, struct task_struct *p, int cpu)
{
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* [PATCH 09/18 v2] sched/core: Add rq flag to tick parameters
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (7 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 08/18] sched/fair: Optimize " Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 10/18 v2] sched/fair: Add force push task mechanism for fair Vincent Guittot
` (8 subsequent siblings)
17 siblings, 0 replies; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
In order to force migration in tick, we need to unlock the rq
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/core.c | 6 +++---
kernel/sched/deadline.c | 2 +-
kernel/sched/ext/ext.c | 2 +-
kernel/sched/fair.c | 2 +-
kernel/sched/idle.c | 2 +-
kernel/sched/rt.c | 2 +-
kernel/sched/sched.h | 2 +-
kernel/sched/stop_task.c | 2 +-
8 files changed, 10 insertions(+), 10 deletions(-)
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 837dc74c9a8d..1e34c8fdbee8 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -923,7 +923,7 @@ static enum hrtimer_restart hrtick(struct hrtimer *timer)
rq_lock(rq, &rf);
update_rq_clock(rq);
- rq->donor->sched_class->task_tick(rq, rq->donor, 1);
+ rq->donor->sched_class->task_tick(rq, rq->donor, &rf, 1);
rq_unlock(rq, &rf);
return HRTIMER_NORESTART;
@@ -5800,7 +5800,7 @@ void sched_tick(void)
if (dynamic_preempt_lazy() && tif_test_bit(TIF_NEED_RESCHED_LAZY))
resched_curr(rq);
- donor->sched_class->task_tick(rq, donor, 0);
+ donor->sched_class->task_tick(rq, donor, &rf, 0);
if (sched_feat(LATENCY_WARN))
resched_latency = cpu_resched_latency(rq);
calc_global_load_tick(rq);
@@ -5896,7 +5896,7 @@ static void sched_tick_remote(struct work_struct *work)
u64 delta = rq_clock_task(rq) - curr->se.exec_start;
WARN_ON_ONCE(delta > (u64)NSEC_PER_SEC * 30);
}
- curr->sched_class->task_tick(rq, curr, 0);
+ curr->sched_class->task_tick(rq, curr, NULL, 0);
calc_load_nohz_remote(rq);
}
diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c
index c0ebdcde5fe5..1f6bc63b3110 100644
--- a/kernel/sched/deadline.c
+++ b/kernel/sched/deadline.c
@@ -2879,7 +2879,7 @@ static void put_prev_task_dl(struct rq *rq, struct task_struct *p, struct task_s
* and everything must be accessed through the @rq and @curr passed in
* parameters.
*/
-static void task_tick_dl(struct rq *rq, struct task_struct *p, int queued)
+static void task_tick_dl(struct rq *rq, struct task_struct *p, struct rq_flags *rf, int queued)
{
update_curr_dl(rq);
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 53275bc7c029..cdf48af9f05e 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -3790,7 +3790,7 @@ void scx_tick(struct rq *rq)
update_other_load_avgs(rq);
}
-static void task_tick_scx(struct rq *rq, struct task_struct *curr, int queued)
+static void task_tick_scx(struct rq *rq, struct task_struct *curr, struct rq_flags *rf, int queued)
{
struct scx_sched *sch = scx_task_sched(curr);
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 6ba2efeba435..483ed5807dfc 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -15346,7 +15346,7 @@ static inline void task_tick_core(struct rq *rq, struct task_struct *curr) {}
* and everything must be accessed through the @rq and @curr passed in
* parameters.
*/
-static void task_tick_fair(struct rq *rq, struct task_struct *curr, int queued)
+static void task_tick_fair(struct rq *rq, struct task_struct *curr, struct rq_flags *rf, int queued)
{
struct sched_entity *se = &curr->se;
diff --git a/kernel/sched/idle.c b/kernel/sched/idle.c
index 76f3c84ca684..1ba0303aacb4 100644
--- a/kernel/sched/idle.c
+++ b/kernel/sched/idle.c
@@ -538,7 +538,7 @@ dequeue_task_idle(struct rq *rq, struct task_struct *p, int flags)
* and everything must be accessed through the @rq and @curr passed in
* parameters.
*/
-static void task_tick_idle(struct rq *rq, struct task_struct *curr, int queued)
+static void task_tick_idle(struct rq *rq, struct task_struct *curr, struct rq_flags *rf, int queued)
{
update_curr_idle(rq);
}
diff --git a/kernel/sched/rt.c b/kernel/sched/rt.c
index 1535046a23ff..35b3831a72cf 100644
--- a/kernel/sched/rt.c
+++ b/kernel/sched/rt.c
@@ -2541,7 +2541,7 @@ static inline void watchdog(struct rq *rq, struct task_struct *p) { }
* and everything must be accessed through the @rq and @curr passed in
* parameters.
*/
-static void task_tick_rt(struct rq *rq, struct task_struct *p, int queued)
+static void task_tick_rt(struct rq *rq, struct task_struct *p, struct rq_flags *rf, int queued)
{
struct sched_rt_entity *rt_se = &p->rt;
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 293f23620282..c1d7e04d2899 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -2741,7 +2741,7 @@ struct sched_class {
* sched_tick: rq->lock
* sched_tick_remote: rq->lock
*/
- void (*task_tick)(struct rq *rq, struct task_struct *p, int queued);
+ void (*task_tick)(struct rq *rq, struct task_struct *p, struct rq_flags *rf, int queued);
/*
* sched_cgroup_fork: p->pi_lock
*/
diff --git a/kernel/sched/stop_task.c b/kernel/sched/stop_task.c
index 1e0109ec36b3..3d99b23afd89 100644
--- a/kernel/sched/stop_task.c
+++ b/kernel/sched/stop_task.c
@@ -74,7 +74,7 @@ static void put_prev_task_stop(struct rq *rq, struct task_struct *prev, struct t
* and everything must be accessed through the @rq and @curr passed in
* parameters.
*/
-static void task_tick_stop(struct rq *rq, struct task_struct *curr, int queued)
+static void task_tick_stop(struct rq *rq, struct task_struct *curr, struct rq_flags *rf, int queued)
{
}
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* [PATCH 10/18 v2] sched/fair: Add force push task mechanism for fair
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (8 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 09/18 v2] sched/core: Add rq flag to tick parameters Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-06 19:24 ` Kayra Cizmeci
2026-10-02 15:44 ` [PATCH 11/18 v2] sched/fair: Support not wakeup case in select_idle_sibling Vincent Guittot
` (7 subsequent siblings)
17 siblings, 1 reply; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
When the task is alone on the CPU, it's never put back in the enqueued
list; In this special case, we use the tick to run the check used to push
runnable task on a btter CPU.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/fair.c | 71 +++++++++++++++++++++++++++++++++++++++++++--
1 file changed, 69 insertions(+), 2 deletions(-)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 483ed5807dfc..6ed3e5bb7fd6 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -9955,6 +9955,71 @@ static void fair_add_pushable_prev(struct rq *rq, struct task_struct *prev, stru
__fair_add_pushable_task(rq, prev);
}
+static int active_load_balance_cpu_stop(void *data);
+
+/*
+ * See if the alone task running on the CPU should migrate on a better than
+ * the local one.
+ */
+static inline bool tick_pushable_task(struct task_struct *p, struct rq *rq, struct rq_flags *rf)
+{
+ int new_cpu, cpu = cpu_of(rq);
+
+ if (!sched_push_task_enabled())
+ return false;
+
+ if (!rf)
+ return false;
+
+ if (WARN_ON(!p))
+ return false;
+
+ if (WARN_ON(!task_current(rq, p)))
+ return false;
+
+ if (is_migration_disabled(p))
+ return false;
+
+ /* If there are several task, wait for being put back */
+ if (rq->nr_running > 1)
+ return false;
+
+ if (!fair_check_pushable_task(rq, p, NULL))
+ return false;
+
+ if (!raw_spin_trylock(&p->pi_lock))
+ return false;
+
+ new_cpu = select_task_rq_fair(p, cpu, 0);
+
+ raw_spin_unlock(&p->pi_lock);
+
+ if (new_cpu == cpu)
+ return false;
+
+ /*
+ * ->active_balance synchronizes accesses to
+ * ->active_balance_work. Once set, it's cleared
+ * only after active load balance is finished.
+ */
+ if (!rq->active_balance) {
+ rq->active_balance = 1;
+ rq->push_cpu = new_cpu;
+ } else {
+ return false;
+ }
+
+ preempt_disable();
+ rq_unlock(rq, rf);
+ stop_one_cpu_nowait(cpu,
+ active_load_balance_cpu_stop, rq,
+ &rq->active_balance_work);
+ preempt_enable();
+ rq_lock(rq, rf);
+
+ return true;
+}
+
/*
* select_task_rq_fair: Select a target runqueue for the task.
* There are 2 ways to select the target runqueue:
@@ -15373,8 +15438,10 @@ static void task_tick_fair(struct rq *rq, struct task_struct *curr, struct rq_fl
task_tick_cache(rq, curr);
- update_misfit_status(curr, rq);
- check_update_overutilized_status(task_rq(curr));
+ if (!tick_pushable_task(curr, rq, rf)) {
+ update_misfit_status(curr, rq);
+ check_update_overutilized_status(task_rq(curr));
+ }
task_tick_core(rq, curr);
}
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* Re: [PATCH 10/18 v2] sched/fair: Add force push task mechanism for fair
2026-10-02 15:44 ` [PATCH 10/18 v2] sched/fair: Add force push task mechanism for fair Vincent Guittot
@ 2026-10-06 19:24 ` Kayra Cizmeci
0 siblings, 0 replies; 24+ messages in thread
From: Kayra Cizmeci @ 2026-10-06 19:24 UTC (permalink / raw)
To: vincent.guittot
Cc: arighi, bsegall, changwoo, christian.loehle, dietmar.eggemann,
juri.lelli, kprateek.nayak, linux-kernel, linux-pm, lukasz.luba,
mgorman, mingo, peterz, pierre.gondois, qyousef, rafael, rostedt,
sched-ext, sshegde, tj, void, vschneid
> When the task is alone on the CPU, it's never put back in the enqueued
> list; In this special case, we use the tick to run the check used to push
> runnable task on a btter CPU.
I love typos. Like, in here. 'btter'. I think it should be bitter. Like:
"In this special case, we use the tick to run the check used to push runnable task on a bitter
CPU."
(Or maybe by a really low chance, it could be 'better' too. Maybe tho... :>)
> +static int active_load_balance_cpu_stop(void *data);
> +
> +/*
> + * See if the alone task running on the CPU should migrate on a better than
> + * the local one.
> + */
> +static inline bool tick_pushable_task(struct task_struct *p, struct rq *rq, struct rq_flags *rf)
> +{
> + int new_cpu, cpu = cpu_of(rq);
> +
> + if (!sched_push_task_enabled())
> + return false;
> +
> + if (!rf)
> + return false;
> +
> + if (WARN_ON(!p))
> + return false;
> +
> + if (WARN_ON(!task_current(rq, p)))
> + return false;
> +
OK. So, this is called from task_tick_fair() and task_tick_fair() is
called from sched_tick() with rq->donor. So on Proxy bla bla
this could get true. Or maybe I'm missing something. Dunno.
NOTE: Some more places could have the same case. Again, dunno.
I need more tea. And sleep for not having to drink that much tea.
Thanks,
Kayra :>
^ permalink raw reply [flat|nested] 24+ messages in thread
* [PATCH 11/18 v2] sched/fair: Support not wakeup case in select_idle_sibling
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (9 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 10/18 v2] sched/fair: Add force push task mechanism for fair Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 12/18 v2] sched/eevdf: Try to push short slice task on a better CPU Vincent Guittot
` (6 subsequent siblings)
17 siblings, 0 replies; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
With push callback mecanism, select_task_rq_fair can be called for a task
that is already enqueued. Task into account this case when choosing idle
CPU.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/fair.c | 5 ++++-
1 file changed, 4 insertions(+), 1 deletion(-)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 6ed3e5bb7fd6..38eb80dbd1a4 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -7974,10 +7974,13 @@ static int choose_sched_idle_rq(struct rq *rq, struct task_struct *p)
return sched_idle_rq(rq) && !task_has_idle_policy(p);
}
+static int idle_cpu_without(int cpu, struct task_struct *p);
+
static int choose_idle_cpu(int cpu, struct task_struct *p)
{
return available_idle_cpu(cpu) ||
- choose_sched_idle_rq(cpu_rq(cpu), p);
+ choose_sched_idle_rq(cpu_rq(cpu), p) ||
+ idle_cpu_without(cpu, p);
}
static void
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* [PATCH 12/18 v2] sched/eevdf: Try to push short slice task on a better CPU
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (10 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 11/18 v2] sched/fair: Support not wakeup case in select_idle_sibling Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 13/18 v2] sched/eevdf: Push short slice task that are not picked Vincent Guittot
` (5 subsequent siblings)
17 siblings, 0 replies; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
When a task is put back in the runnable list, we check if we should try to
push on a better CPU, i.e. when fair is preempted by a higher class or the
task is preempted by another fair task.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/fair.c | 23 +++++++++++++++++++++++
1 file changed, 23 insertions(+)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 38eb80dbd1a4..186f10173eed 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -9826,11 +9826,34 @@ static bool __check_pushable_fair_task(struct rq *rq, struct task_struct *p)
return true;
}
+static bool check_pushable_short_task(struct rq *rq, struct task_struct *p)
+{
+ struct sched_entity *pse = &p->se;
+ struct cfs_rq *cfs_rq = &rq->cfs;
+
+ if (cfs_rq->h_nr_runnable <= 1)
+ return false;
+
+ if (!entity_eligible(cfs_rq, pse))
+ return false;
+
+ if (pse->slice < cfs_rq_max_slice(cfs_rq))
+ return true;
+
+ return false;
+}
+
static bool fair_check_pushable_task(struct rq *rq, struct task_struct *p, struct task_struct *next)
{
if (!__check_pushable_fair_task(rq, p))
return false;
+ if (next && next->sched_class != p->sched_class)
+ return true;
+
+ if (check_pushable_short_task(rq, p))
+ return true;
+
return false;
}
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* [PATCH 13/18 v2] sched/eevdf: Push short slice task that are not picked
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (11 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 12/18 v2] sched/eevdf: Try to push short slice task on a better CPU Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 14/18 v2] sched/fair: Enable push task for preempt short Vincent Guittot
` (4 subsequent siblings)
17 siblings, 0 replies; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
Even when set as next buddy, a number of event can prevent short slice
task to not be picked. Add the task in the pushable list in case it is not
picked so it get a chance to be pushed on a better CPU.
This can typically happen when several short slice tasks are pushed on the
same CPU at wake up.
If the task is picked, it will be removed from the pushable list before we
queue the callback.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/fair.c | 10 +++++++++-
1 file changed, 9 insertions(+), 1 deletion(-)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 186f10173eed..cccdc40b8a24 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -9981,6 +9981,12 @@ static void fair_add_pushable_prev(struct rq *rq, struct task_struct *prev, stru
__fair_add_pushable_task(rq, prev);
}
+static void fair_add_pushable_short(struct rq *rq, struct task_struct *p)
+{
+ if (sched_push_task_enabled() && __check_pushable_fair_task(rq, p))
+ __fair_add_pushable_task(rq, p);
+}
+
static int active_load_balance_cpu_stop(void *data);
/*
@@ -10419,8 +10425,10 @@ static void wakeup_preempt_fair(struct rq *rq, struct task_struct *p, int wake_f
preempt:
cancel_protect_slice(se);
- if (preempt_action == PREEMPT_WAKEUP_SHORT)
+ if (preempt_action == PREEMPT_WAKEUP_SHORT) {
set_short_buddy(cfs_rq, pse);
+ fair_add_pushable_short(rq, p);
+ }
resched_curr_lazy(rq);
}
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* [PATCH 14/18 v2] sched/fair: Enable push task for preempt short
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (12 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 13/18 v2] sched/eevdf: Push short slice task that are not picked Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 15/18 v2] energy model: Add a get previous state function Vincent Guittot
` (3 subsequent siblings)
17 siblings, 0 replies; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
Enable push mecanism for the preempt short feature which is the
1st feature using it.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/topology.c | 3 +++
1 file changed, 3 insertions(+)
diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c
index 0248227d983a..6b1058268e6d 100644
--- a/kernel/sched/topology.c
+++ b/kernel/sched/topology.c
@@ -3475,6 +3475,9 @@ static void partition_sched_domains_locked(int ndoms_new, cpumask_var_t doms_new
sched_energy_set(has_eas);
#endif
+ if (sched_feat(PREEMPT_SHORT))
+ static_branch_inc_cpuslocked(&sched_push_task);
+
/* Remember the new sched domains: */
if (doms_cur != &fallback_doms)
free_sched_domains(doms_cur, ndoms_cur);
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* [PATCH 15/18 v2] energy model: Add a get previous state function
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (13 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 14/18 v2] sched/fair: Enable push task for preempt short Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 16/18 v2] sched/fair: Rework feec() to use cost instead of spare capacity Vincent Guittot
` (2 subsequent siblings)
17 siblings, 0 replies; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
Instead of parsing the entire EM table everytime, add a function to get the
previous state.
Will be used in the scheduler feec() function.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
include/linux/energy_model.h | 32 ++++++++++++++++++++++++++++++++
1 file changed, 32 insertions(+)
diff --git a/include/linux/energy_model.h b/include/linux/energy_model.h
index c909a8ba22e8..89cf109cbca6 100644
--- a/include/linux/energy_model.h
+++ b/include/linux/energy_model.h
@@ -224,6 +224,26 @@ em_pd_get_efficient_state(struct em_perf_state *table,
return max_ps;
}
+static inline int
+em_pd_get_previous_state(struct em_perf_state *table,
+ struct em_perf_domain *pd, int idx)
+{
+ unsigned long pd_flags = pd->flags;
+ int min_ps = pd->min_perf_state;
+ struct em_perf_state *ps;
+ int i;
+
+ for (i = idx - 1; i >= min_ps; i--) {
+ ps = &table[i];
+ if (pd_flags & EM_PERF_DOMAIN_SKIP_INEFFICIENCIES &&
+ ps->flags & EM_PERF_STATE_INEFFICIENT)
+ continue;
+ return i;
+ }
+
+ return -1;
+}
+
/**
* em_cpu_energy() - Estimates the energy consumed by the CPUs of a
* performance domain
@@ -375,6 +395,18 @@ static inline struct em_perf_domain *em_pd_get(struct device *dev)
{
return NULL;
}
+static inline int
+em_pd_get_efficient_state(struct em_perf_state *table,
+ struct em_perf_domain *pd, unsigned long max_util)
+{
+ return 0;
+}
+static inline int
+em_pd_get_previous_state(struct em_perf_state *table,
+ struct em_perf_domain *pd, int idx)
+{
+ return -1;
+}
static inline unsigned long em_cpu_energy(struct em_perf_domain *pd,
unsigned long max_util, unsigned long sum_util,
unsigned long allowed_cpu_cap)
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* [PATCH 16/18 v2] sched/fair: Rework feec() to use cost instead of spare capacity
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (14 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 15/18 v2] energy model: Add a get previous state function Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 17/18 v2] energy model: Remove unused em_cpu_energy() Vincent Guittot
2026-10-02 15:44 ` [PATCH 18/18 v2] sched/fair: Take into account slice in EAS Vincent Guittot
17 siblings, 0 replies; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
feec() looks for the CPU with highest spare capacity in a PD assuming that
it will be the best CPU from a energy efficiency PoV because it will
require the smallest increase of OPP. Although this is true generally
speaking, this policy also filters some others CPUs which will be as
efficients because of using the same OPP.
In fact, we really care about the cost of the new OPP that will be
selected to handle the waking task. In many cases, several CPUs will end
up selecting the same OPP and as a result using the same energy cost. In
these cases, we can use other metrics to select the best CPU for the same
energy cost.
Rework feec() to look 1st for the lowest cost in a PD and then the most
performant CPU between CPUs. The cost of the OPP remains the only
comparison criteria between Performance Domains.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/fair.c | 469 +++++++++++++++++++++++---------------------
1 file changed, 250 insertions(+), 219 deletions(-)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index cccdc40b8a24..94554f165f42 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -9444,29 +9444,37 @@ unsigned long sched_cpu_util(int cpu)
}
/*
- * energy_env - Utilization landscape for energy estimation.
- * @task_busy_time: Utilization contribution by the task for which we test the
- * placement. Given by eenv_task_busy_time().
- * @pd_busy_time: Utilization of the whole perf domain without the task
- * contribution. Given by eenv_pd_busy_time().
- * @cpu_cap: Maximum CPU capacity for the perf domain.
- * @pd_cap: Entire perf domain capacity. (pd->nr_cpus * cpu_cap).
- */
-struct energy_env {
- unsigned long task_busy_time;
- unsigned long pd_busy_time;
- unsigned long cpu_cap;
- unsigned long pd_cap;
+ * energy_cpu_stat - Utilization landscape for energy estimation.
+ * @idx : Index of the OPP in the performance domain
+ * @cost : Cost of the OPP
+ * @max_perf : Compute capacity of OPP
+ * @min_perf : Compute capacity of the previous OPP
+ * @capa : Capacity of the CPU
+ * @runnable : runnable_avg of the CPU
+ * @nr_running : Number of cfs running task
+ * @fits : Fits level of the CPU
+ * @cpu : Current best CPU
+ */
+struct energy_cpu_stat {
+ unsigned long idx;
+ unsigned long cost;
+ unsigned long max_perf;
+ unsigned long min_perf;
+ unsigned long capa;
+ unsigned long util;
+ unsigned long runnable;
+ unsigned int nr_running;
+ int fits;
+ int cpu;
};
/*
- * Compute the task busy time for compute_energy(). This time cannot be
- * injected directly into effective_cpu_util() because of the IRQ scaling.
+ * Compute the task busy time for computing its energy impact. This time cannot
+ * be injected directly into effective_cpu_util() because of the IRQ scaling.
* The latter only makes sense with the most recent CPUs where the task has
* run.
*/
-static inline void eenv_task_busy_time(struct energy_env *eenv,
- struct task_struct *p, int prev_cpu)
+static inline unsigned long task_busy_time(struct task_struct *p, int prev_cpu)
{
unsigned long busy_time, max_cap = arch_scale_cpu_capacity(prev_cpu);
unsigned long irq = cpu_util_irq(cpu_rq(prev_cpu));
@@ -9476,124 +9484,152 @@ static inline void eenv_task_busy_time(struct energy_env *eenv,
else
busy_time = scale_irq_capacity(task_util_est(p), irq, max_cap);
- eenv->task_busy_time = busy_time;
+ return busy_time;
}
-/*
- * Compute the perf_domain (PD) busy time for compute_energy(). Based on the
- * utilization for each @pd_cpus, it however doesn't take into account
- * clamping since the ratio (utilization / cpu_capacity) is already enough to
- * scale the EM reported power consumption at the (eventually clamped)
- * cpu_capacity.
- *
- * The contribution of the task @p for which we want to estimate the
- * energy cost is removed (by cpu_util()) and must be calculated
- * separately (see eenv_task_busy_time). This ensures:
- *
- * - A stable PD utilization, no matter which CPU of that PD we want to place
- * the task on.
- *
- * - A fair comparison between CPUs as the task contribution (task_util())
- * will always be the same no matter which CPU utilization we rely on
- * (util_avg or util_est).
- *
- * Set @eenv busy time for the PD that spans @pd_cpus. This busy time can't
- * exceed @eenv->pd_cap.
- */
-static inline void eenv_pd_busy_time(struct energy_env *eenv,
- struct cpumask *pd_cpus,
- struct task_struct *p)
+/* Estimate the utilization of the CPU that is then used to select the OPP */
+static unsigned long find_cpu_max_util(int cpu, struct task_struct *p, int dst_cpu)
{
- unsigned long busy_time = 0;
- int cpu;
+ unsigned long util = cpu_util(cpu, p, dst_cpu, 1);
+ unsigned long eff_util, min, max;
+
+ /*
+ * Performance domain frequency: utilization clamping
+ * must be considered since it affects the selection
+ * of the performance domain frequency.
+ */
+ eff_util = effective_cpu_util(cpu, util, &min, &max);
- for_each_cpu(cpu, pd_cpus) {
- unsigned long util = cpu_util(cpu, p, -1, 0);
+ /* Task's uclamp can modify min and max value */
+ if (uclamp_is_used() && cpu == dst_cpu) {
+ min = max(min, uclamp_eff_value(p, UCLAMP_MIN));
- busy_time += effective_cpu_util(cpu, util, NULL, NULL);
+ /*
+ * If there is no active max uclamp constraint,
+ * directly use task's one, otherwise keep max.
+ */
+ if (uclamp_rq_is_idle(cpu_rq(cpu)))
+ max = uclamp_eff_value(p, UCLAMP_MAX);
+ else
+ max = max(max, uclamp_eff_value(p, UCLAMP_MAX));
}
- eenv->pd_busy_time = min(eenv->pd_cap, busy_time);
+ eff_util = sugov_effective_cpu_perf(cpu, eff_util, min, max);
+ return eff_util;
}
-/*
- * Compute the maximum utilization for compute_energy() when the task @p
- * is placed on the cpu @dst_cpu.
- *
- * Returns the maximum utilization among @eenv->cpus. This utilization can't
- * exceed @eenv->cpu_cap.
- */
-static inline unsigned long
-eenv_pd_max_util(struct energy_env *eenv, struct cpumask *pd_cpus,
- struct task_struct *p, int dst_cpu)
+/* Estimate the utilization of the CPU without the task */
+static unsigned long find_cpu_actual_util(int cpu, struct task_struct *p)
{
- unsigned long max_util = 0;
- int cpu;
+ unsigned long util = cpu_util(cpu, p, -1, 0);
+ unsigned long eff_util;
- for_each_cpu(cpu, pd_cpus) {
- struct task_struct *tsk = (cpu == dst_cpu) ? p : NULL;
- unsigned long util = cpu_util(cpu, p, dst_cpu, 1);
- unsigned long eff_util, min, max;
+ eff_util = effective_cpu_util(cpu, util, NULL, NULL);
- /*
- * Performance domain frequency: utilization clamping
- * must be considered since it affects the selection
- * of the performance domain frequency.
- * NOTE: in case RT tasks are running, by default the min
- * utilization can be max OPP.
- */
- eff_util = effective_cpu_util(cpu, util, &min, &max);
+ return eff_util;
+}
- /* Task's uclamp can modify min and max value */
- if (tsk && uclamp_is_used()) {
- min = max(min, uclamp_eff_value(p, UCLAMP_MIN));
+/* Find the cost of a performance domain for the estimated utilization */
+static inline void find_pd_cost(struct em_perf_domain *pd,
+ unsigned long max_util,
+ struct energy_cpu_stat *stat)
+{
+ struct em_perf_table *em_table;
+ struct em_perf_state *ps;
+ int i;
- /*
- * If there is no active max uclamp constraint,
- * directly use task's one, otherwise keep max.
- */
- if (uclamp_rq_is_idle(cpu_rq(cpu)))
- max = uclamp_eff_value(p, UCLAMP_MAX);
- else
- max = max(max, uclamp_eff_value(p, UCLAMP_MAX));
- }
+ /*
+ * Find the lowest performance state of the Energy Model above the
+ * requested performance.
+ */
+ em_table = rcu_dereference(pd->em_table);
+ i = em_pd_get_efficient_state(em_table->state, pd, max_util);
+ ps = &em_table->state[i];
- eff_util = sugov_effective_cpu_perf(cpu, eff_util, min, max);
- max_util = max(max_util, eff_util);
+ /* Save the cost and performance range of the OPP */
+ stat->max_perf = ps->performance;
+ stat->cost = ps->cost;
+ i = em_pd_get_previous_state(em_table->state, pd, i);
+ if (i < 0) {
+ stat->min_perf = 0;
+ } else {
+ ps = &em_table->state[i];
+ stat->min_perf = ps->performance;
}
+}
- return min(max_util, eenv->cpu_cap);
+/*Check if the CPU can handle the waking task */
+static int check_cpu_with_task(struct task_struct *p, int cpu)
+{
+ unsigned long p_util_min = uclamp_is_used() ? uclamp_eff_value(p, UCLAMP_MIN) : 0;
+ unsigned long p_util_max = uclamp_is_used() ? uclamp_eff_value(p, UCLAMP_MAX) : 1024;
+ unsigned long util_min = p_util_min;
+ unsigned long util_max = p_util_max;
+ unsigned long util = cpu_util(cpu, p, cpu, 0);
+ struct rq *rq = cpu_rq(cpu);
+
+ /*
+ * Skip CPUs that cannot satisfy the capacity request.
+ * IOW, placing the task there would make the CPU
+ * overutilized. Take uclamp into account to see how
+ * much capacity we can get out of the CPU; this is
+ * aligned with sched_cpu_util().
+ */
+ if (uclamp_is_used() && !uclamp_rq_is_idle(rq)) {
+ unsigned long rq_util_min, rq_util_max;
+ /*
+ * Open code uclamp_rq_util_with() except for
+ * the clamp() part. I.e.: apply max aggregation
+ * only. util_fits_cpu() logic requires to
+ * operate on non clamped util but must use the
+ * max-aggregated uclamp_{min, max}.
+ */
+ rq_util_min = uclamp_rq_get(rq, UCLAMP_MIN);
+ rq_util_max = uclamp_rq_get(rq, UCLAMP_MAX);
+ util_min = max(rq_util_min, p_util_min);
+ util_max = max(rq_util_max, p_util_max);
+ }
+ return util_fits_cpu(util, util_min, util_max, cpu);
}
/*
- * compute_energy(): Use the Energy Model to estimate the energy that @pd would
- * consume for a given utilization landscape @eenv. When @dst_cpu < 0, the task
- * contribution is ignored.
+ * For the same cost, select the CPU that will povide best performance for the
+ * task.
*/
-static inline unsigned long
-compute_energy(struct energy_env *eenv, struct perf_domain *pd,
- struct cpumask *pd_cpus, struct task_struct *p, int dst_cpu)
+static bool update_best_cpu(struct energy_cpu_stat *target,
+ struct energy_cpu_stat *min,
+ int prev, struct sched_domain *sd)
{
- unsigned long max_util = eenv_pd_max_util(eenv, pd_cpus, p, dst_cpu);
- unsigned long busy_time = eenv->pd_busy_time;
- unsigned long energy;
-
- if (dst_cpu >= 0)
- busy_time = min(eenv->pd_cap, busy_time + eenv->task_busy_time);
+ if (target->cpu == prev)
+ return true;
+ if (min->cpu == prev)
+ return false;
- energy = em_cpu_energy(pd->em_pd, max_util, busy_time, eenv->cpu_cap);
+ /* Select the one with the least number of running tasks otherwise */
+ if (target->nr_running < min->nr_running)
+ return true;
+ if (target->nr_running > min->nr_running)
+ return false;
- trace_sched_compute_energy_tp(p, dst_cpu, energy, max_util, busy_time);
+ /*
+ * Choose CPU with lowest contention. One might want to consider load
+ * instead of runnable but we are supposed to not be overutilized so
+ * there is enough compute capacity for everybody.
+ */
+ if ((target->runnable * min->capa * sd->imbalance_pct) >=
+ (min->runnable * target->capa * 100))
+ return false;
- return energy;
+ return true;
}
/*
* find_energy_efficient_cpu(): Find most energy-efficient target CPU for the
- * waking task. find_energy_efficient_cpu() looks for the CPU with maximum
- * spare capacity in each performance domain and uses it as a potential
- * candidate to execute the task. Then, it uses the Energy Model to figure
- * out which of the CPU candidates is the most energy-efficient.
+ * waking task. find_energy_efficient_cpu() looks for the CPU with the lowest
+ * power cost (usually with maximum spare capacity but not always) in each
+ * performance domain and uses it as a potential candidate to execute the task.
+ * Then, it uses the Energy Model to figure out which of the CPU candidates is
+ * the most energy-efficient.
*
* The rationale for this heuristic is as follows. In a performance domain,
* all the most energy efficient CPU candidates (according to the Energy
@@ -9630,17 +9666,14 @@ compute_energy(struct energy_env *eenv, struct perf_domain *pd,
static int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu)
{
struct cpumask *cpus = this_cpu_cpumask_var_ptr(select_rq_mask);
- unsigned long prev_delta = ULONG_MAX, best_delta = ULONG_MAX;
- unsigned long p_util_min = uclamp_is_used() ? uclamp_eff_value(p, UCLAMP_MIN) : 0;
- unsigned long p_util_max = uclamp_is_used() ? uclamp_eff_value(p, UCLAMP_MAX) : 1024;
struct root_domain *rd = this_rq()->rd;
- int cpu, best_energy_cpu, target = -1;
- int prev_fits = -1, best_fits = -1;
- unsigned long best_actual_cap = 0;
- unsigned long prev_actual_cap = 0;
+ unsigned long best_nrg = ULONG_MAX;
+ unsigned long task_util;
struct sched_domain *sd;
struct perf_domain *pd;
- struct energy_env eenv;
+ int cpu, target = -1;
+ int best_fits = -1;
+ int best_cpu = -1;
pd = rcu_dereference_all(rd->pd);
if (!pd)
@@ -9659,19 +9692,19 @@ static int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu)
target = prev_cpu;
sync_entity_load_avg(&p->se);
- if (!task_util_est(p) && p_util_min == 0)
- return target;
-
- eenv_task_busy_time(&eenv, p, prev_cpu);
+ task_util = task_busy_time(p, prev_cpu);
for (; pd; pd = pd->next) {
- unsigned long util_min = p_util_min, util_max = p_util_max;
- unsigned long cpu_cap, cpu_actual_cap, util;
- long prev_spare_cap = -1, max_spare_cap = -1;
- unsigned long rq_util_min, rq_util_max;
- unsigned long cur_delta, base_energy;
- int max_spare_cap_cpu = -1;
- int fits, max_fits = -1;
+ unsigned long pd_actual_util = 0, delta_nrg = 0;
+ unsigned long cpu_actual_cap, max_cost = 0;
+ struct energy_cpu_stat target_stat;
+ struct energy_cpu_stat min_stat = {
+ .cost = ULONG_MAX,
+ .max_perf = ULONG_MAX,
+ .min_perf = ULONG_MAX,
+ .fits = -2,
+ .cpu = -1,
+ };
if (!cpumask_and(cpus, perf_domain_span(pd), cpu_online_mask))
continue;
@@ -9680,13 +9713,9 @@ static int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu)
cpu = cpumask_first(cpus);
cpu_actual_cap = get_actual_cpu_capacity(cpu);
- eenv.cpu_cap = cpu_actual_cap;
- eenv.pd_cap = 0;
-
+ /* In a PD, the CPU with the lowest cost will be the most efficient */
for_each_cpu(cpu, cpus) {
- struct rq *rq = cpu_rq(cpu);
-
- eenv.pd_cap += cpu_actual_cap;
+ unsigned long target_perf;
if (!cpumask_test_cpu(cpu, sched_domain_span(sd)))
continue;
@@ -9694,113 +9723,115 @@ static int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu)
if (!cpumask_test_cpu(cpu, p->cpus_ptr))
continue;
- util = cpu_util(cpu, p, cpu, 0);
- cpu_cap = capacity_of(cpu);
+ target_stat.fits = check_cpu_with_task(p, cpu);
- /*
- * Skip CPUs that cannot satisfy the capacity request.
- * IOW, placing the task there would make the CPU
- * overutilized. Take uclamp into account to see how
- * much capacity we can get out of the CPU; this is
- * aligned with sched_cpu_util().
- */
- if (uclamp_is_used() && !uclamp_rq_is_idle(rq)) {
- /*
- * Open code uclamp_rq_util_with() except for
- * the clamp() part. I.e.: apply max aggregation
- * only. util_fits_cpu() logic requires to
- * operate on non clamped util but must use the
- * max-aggregated uclamp_{min, max}.
- */
- rq_util_min = uclamp_rq_get(rq, UCLAMP_MIN);
- rq_util_max = uclamp_rq_get(rq, UCLAMP_MAX);
+ if (!target_stat.fits)
+ continue;
- util_min = max(rq_util_min, p_util_min);
- util_max = max(rq_util_max, p_util_max);
- }
+ /* 1st select the CPU that fits best */
+ if (target_stat.fits < min_stat.fits)
+ continue;
+
+ /* Then select the CPU with lowest cost */
+
+ /* Get the performance of the CPU w/ the waking task */
+ target_perf = find_cpu_max_util(cpu, p, cpu);
+ target_perf = min(target_perf, cpu_actual_cap);
- fits = util_fits_cpu(util, util_min, util_max, cpu);
- if (!fits)
+ /* Needing a higher OPP means a higher cost */
+ if (target_perf > min_stat.max_perf)
continue;
- lsub_positive(&cpu_cap, util);
+ /*
+ * At this point, target's cost can be either equal or
+ * lower than the current minimum cost.
+ */
- if (cpu == prev_cpu) {
- /* Always use prev_cpu as a candidate. */
- prev_spare_cap = cpu_cap;
- prev_fits = fits;
- } else if ((fits > max_fits) ||
- ((fits == max_fits) && ((long)cpu_cap > max_spare_cap))) {
- /*
- * Find the CPU with the maximum spare capacity
- * among the remaining CPUs in the performance
- * domain.
- */
- max_spare_cap = cpu_cap;
- max_spare_cap_cpu = cpu;
- max_fits = fits;
- }
+ /* Gather more statistics */
+ target_stat.cpu = cpu;
+ target_stat.runnable = cpu_runnable(cpu_rq(cpu));
+ target_stat.capa = capacity_of(cpu);
+ target_stat.nr_running = cpu_rq(cpu)->cfs.h_nr_runnable;
+ if (p->on_rq && !p->se.sched_delayed && cpu == prev_cpu)
+ target_stat.nr_running--;
+
+ /* If the target needs a lower OPP, then look up for
+ * the corresponding OPP and its associated cost.
+ * Otherwise at same cost level, select the CPU which
+ * provides best performance.
+ */
+ if (target_perf < min_stat.min_perf)
+ find_pd_cost(pd->em_pd, target_perf, &target_stat);
+ else if (!update_best_cpu(&target_stat, &min_stat, prev_cpu, sd))
+ continue;
+
+ /* Save the new most efficient CPU of the PD */
+ min_stat = target_stat;
}
- if (max_spare_cap_cpu < 0 && prev_spare_cap < 0)
+ if (min_stat.cpu == -1)
continue;
- eenv_pd_busy_time(&eenv, cpus, p);
- /* Compute the 'base' energy of the pd, without @p */
- base_energy = compute_energy(&eenv, pd, cpus, p, -1);
-
- /* Evaluate the energy impact of using prev_cpu. */
- if (prev_spare_cap > -1) {
- prev_delta = compute_energy(&eenv, pd, cpus, p,
- prev_cpu);
- /* CPU utilization has changed */
- if (prev_delta < base_energy)
- return target;
- prev_delta -= base_energy;
- prev_actual_cap = cpu_actual_cap;
- best_delta = min(best_delta, prev_delta);
- }
+ if (min_stat.fits < best_fits)
+ continue;
- /* Evaluate the energy impact of using max_spare_cap_cpu. */
- if (max_spare_cap_cpu >= 0 && max_spare_cap > prev_spare_cap) {
- /* Current best energy cpu fits better */
- if (max_fits < best_fits)
- continue;
+ /* Idle system costs nothing */
+ target_stat.max_perf = 0;
+ target_stat.cost = 0;
- /*
- * Both don't fit performance hint (i.e. uclamp_min)
- * but best energy cpu has better capacity.
- */
- if ((max_fits < 0) &&
- (cpu_actual_cap <= best_actual_cap))
- continue;
+ /* Estimate utilization and cost without p */
+ for_each_cpu(cpu, cpus) {
+ unsigned long target_util;
- cur_delta = compute_energy(&eenv, pd, cpus, p,
- max_spare_cap_cpu);
- /* CPU utilization has changed */
- if (cur_delta < base_energy)
- return target;
- cur_delta -= base_energy;
+ /* Accumulate actual utilization w/o task p */
+ pd_actual_util += find_cpu_actual_util(cpu, p);
- /*
- * Both fit for the task but best energy cpu has lower
- * energy impact.
- */
- if ((max_fits > 0) && (best_fits > 0) &&
- (cur_delta >= best_delta))
+ /* Get the max utilization of the CPU w/o task p */
+ target_util = find_cpu_max_util(cpu, p, -1);
+ target_util = min(target_util, cpu_actual_cap);
+
+ /* Current OPP is enough */
+ if (target_util <= target_stat.max_perf)
continue;
- best_delta = cur_delta;
- best_energy_cpu = max_spare_cap_cpu;
- best_fits = max_fits;
- best_actual_cap = cpu_actual_cap;
+ /* Compute and save the cost of the OPP */
+ find_pd_cost(pd->em_pd, target_util, &target_stat);
+ max_cost = target_stat.cost;
}
+
+ /* Add the energy cost of p */
+ delta_nrg = task_util * min_stat.cost;
+
+ /*
+ * Compute the energy cost of others running at higher OPP
+ * because of p.
+ */
+ if (min_stat.cost > max_cost)
+ delta_nrg += pd_actual_util * (min_stat.cost - max_cost);
+
+ /* Delta energy with p */
+ trace_sched_compute_energy_tp(p, min_stat.cpu, delta_nrg,
+ min_stat.max_perf,
+ pd_actual_util + task_util);
+
+ /*
+ * The probability that delta energies are equals is almost
+ * null. PDs being sorted by max capacity, keep the one with
+ * highest max capacity if this happens.
+ * TODO: add a margin in energy cost and take into account
+ * other stats.
+ */
+ if (min_stat.fits == best_fits &&
+ delta_nrg >= best_nrg)
+ continue;
+
+ best_fits = min_stat.fits;
+ best_nrg = delta_nrg;
+ best_cpu = min_stat.cpu;
}
- if ((best_fits > prev_fits) ||
- ((best_fits > 0) && (best_delta < prev_delta)) ||
- ((best_fits < 0) && (best_actual_cap > prev_actual_cap)))
- target = best_energy_cpu;
+ if (best_cpu >= 0)
+ target = best_cpu;
return target;
}
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* [PATCH 17/18 v2] energy model: Remove unused em_cpu_energy()
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (15 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 16/18 v2] sched/fair: Rework feec() to use cost instead of spare capacity Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
2026-10-02 15:44 ` [PATCH 18/18 v2] sched/fair: Take into account slice in EAS Vincent Guittot
17 siblings, 0 replies; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
Remove the unused function em_cpu_energy()
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
include/linux/energy_model.h | 97 ------------------------------------
1 file changed, 97 deletions(-)
diff --git a/include/linux/energy_model.h b/include/linux/energy_model.h
index 89cf109cbca6..c4f9558b58d1 100644
--- a/include/linux/energy_model.h
+++ b/include/linux/energy_model.h
@@ -244,97 +244,6 @@ em_pd_get_previous_state(struct em_perf_state *table,
return -1;
}
-/**
- * em_cpu_energy() - Estimates the energy consumed by the CPUs of a
- * performance domain
- * @pd : performance domain for which energy has to be estimated
- * @max_util : highest utilization among CPUs of the domain
- * @sum_util : sum of the utilization of all CPUs in the domain
- * @allowed_cpu_cap : maximum allowed CPU capacity for the @pd, which
- * might reflect reduced frequency (due to thermal)
- *
- * This function must be used only for CPU devices. There is no validation,
- * i.e. if the EM is a CPU type and has cpumask allocated. It is called from
- * the scheduler code quite frequently and that is why there is not checks.
- *
- * Return: the sum of the energy consumed by the CPUs of the domain assuming
- * a capacity state satisfying the max utilization of the domain.
- */
-static inline unsigned long em_cpu_energy(struct em_perf_domain *pd,
- unsigned long max_util, unsigned long sum_util,
- unsigned long allowed_cpu_cap)
-{
- struct em_perf_table *em_table;
- struct em_perf_state *ps;
- int i;
-
- lockdep_assert(rcu_read_lock_any_held());
-
- if (!sum_util)
- return 0;
-
- /*
- * In order to predict the performance state, map the utilization of
- * the most utilized CPU of the performance domain to a requested
- * performance, like schedutil. Take also into account that the real
- * performance might be set lower (due to thermal capping). Thus, clamp
- * max utilization to the allowed CPU capacity before calculating
- * effective performance.
- */
- max_util = min(max_util, allowed_cpu_cap);
-
- /*
- * Find the lowest performance state of the Energy Model above the
- * requested performance.
- */
- em_table = rcu_dereference_all(pd->em_table);
- i = em_pd_get_efficient_state(em_table->state, pd, max_util);
- ps = &em_table->state[i];
-
- /*
- * The performance (capacity) of a CPU in the domain at the performance
- * state (ps) can be computed as:
- *
- * ps->freq * scale_cpu
- * ps->performance = -------------------- (1)
- * cpu_max_freq
- *
- * So, ignoring the costs of idle states (which are not available in
- * the EM), the energy consumed by this CPU at that performance state
- * is estimated as:
- *
- * ps->power * cpu_util
- * cpu_nrg = -------------------- (2)
- * ps->performance
- *
- * since 'cpu_util / ps->performance' represents its percentage of busy
- * time.
- *
- * NOTE: Although the result of this computation actually is in
- * units of power, it can be manipulated as an energy value
- * over a scheduling period, since it is assumed to be
- * constant during that interval.
- *
- * By injecting (1) in (2), 'cpu_nrg' can be re-expressed as a product
- * of two terms:
- *
- * ps->power * cpu_max_freq
- * cpu_nrg = ------------------------ * cpu_util (3)
- * ps->freq * scale_cpu
- *
- * The first term is static, and is stored in the em_perf_state struct
- * as 'ps->cost'.
- *
- * Since all CPUs of the domain have the same micro-architecture, they
- * share the same 'ps->cost', and the same CPU capacity. Hence, the
- * total energy of the domain (which is the simple sum of the energy of
- * all of its CPUs) can be factorized as:
- *
- * pd_nrg = ps->cost * \Sum cpu_util (4)
- */
- return ps->cost * sum_util;
-}
-
/**
* em_pd_nr_perf_states() - Get the number of performance states of a perf.
* domain
@@ -407,12 +316,6 @@ em_pd_get_previous_state(struct em_perf_state *table,
{
return -1;
}
-static inline unsigned long em_cpu_energy(struct em_perf_domain *pd,
- unsigned long max_util, unsigned long sum_util,
- unsigned long allowed_cpu_cap)
-{
- return 0;
-}
static inline int em_pd_nr_perf_states(struct em_perf_domain *pd)
{
return 0;
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread* [PATCH 18/18 v2] sched/fair: Take into account slice in EAS
2026-10-02 15:43 [PATCH 00/18 v2] Improving latency of short slice tasks Vincent Guittot
` (16 preceding siblings ...)
2026-10-02 15:44 ` [PATCH 17/18 v2] energy model: Remove unused em_cpu_energy() Vincent Guittot
@ 2026-10-02 15:44 ` Vincent Guittot
17 siblings, 0 replies; 24+ messages in thread
From: Vincent Guittot @ 2026-10-02 15:44 UTC (permalink / raw)
To: mingo, peterz, juri.lelli, dietmar.eggemann, rostedt, bsegall,
mgorman, vschneid, kprateek.nayak, linux-kernel, lukasz.luba,
rafael, linux-pm, tj, void, arighi, changwoo, sched-ext
Cc: qyousef, christian.loehle, pierre.gondois, sshegde, Vincent Guittot
When the cost is the same, take into account the slice of a task to try to
select a CPU where is will run first.
Signed-off-by: Vincent Guittot <vincent.guittot@linaro.org>
---
kernel/sched/fair.c | 13 +++++++++++--
1 file changed, 11 insertions(+), 2 deletions(-)
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 94554f165f42..d60bb6db4ce9 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -9598,8 +9598,17 @@ static int check_cpu_with_task(struct task_struct *p, int cpu)
*/
static bool update_best_cpu(struct energy_cpu_stat *target,
struct energy_cpu_stat *min,
- int prev, struct sched_domain *sd)
+ int prev, struct sched_domain *sd,
+ struct task_struct *p)
{
+ unsigned long task_slice = p->se.slice;
+
+ /* Select the one where you can run first */
+ if (task_slice < get_rq_min_slice(cpu_rq(target->cpu)) &&
+ task_slice >= get_rq_min_slice(cpu_rq(min->cpu)))
+ return true;
+
+ /* Favor previous CPU */
if (target->cpu == prev)
return true;
if (min->cpu == prev)
@@ -9762,7 +9771,7 @@ static int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu)
*/
if (target_perf < min_stat.min_perf)
find_pd_cost(pd->em_pd, target_perf, &target_stat);
- else if (!update_best_cpu(&target_stat, &min_stat, prev_cpu, sd))
+ else if (!update_best_cpu(&target_stat, &min_stat, prev_cpu, sd, p))
continue;
/* Save the new most efficient CPU of the PD */
--
2.53.0
^ permalink raw reply [flat|nested] 24+ messages in thread