Actively push out the current running task on a non-preferred CPU. Since
the task is currently running, a stopper thread must be queued to push the
task out. However, if the task is pinned only to non-preferred CPUs,
it will continue running there. This helps to maintain userspace
affinities, unlike CPU hotplug or isolated cpusets.

Though the code is similar to __balance_push_cpu_stop and quite close to
push_cpu_stop, it is kept separate as it provides a cleaner
implementation specifically for CONFIG_PREFERRED_CPU.

Add the push_task_work_done flag to protect the work buffer.

For now, only the currently running task is pushed out. This keeps the code
simpler. In the future, an optimization may be added to move all queued
tasks on the runqueue.

This works only for the FAIR scheduling class.

Signed-off-by: Shrikanth Hegde <[email protected]>
---
 kernel/sched/core.c  | 78 ++++++++++++++++++++++++++++++++++++++++++++
 kernel/sched/sched.h |  8 +++++
 2 files changed, 86 insertions(+)

diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 1c90bcad0a75..c51ba2979496 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -5786,6 +5786,9 @@ void sched_tick(void)
        unsigned long hw_pressure;
        u64 resched_latency;
 
+       if (!cpu_preferred(cpu))
+               sched_push_current_non_preferred_cpu(rq);
+
        if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE))
                arch_scale_freq_tick();
 
@@ -11304,3 +11307,78 @@ void sched_change_end(struct sched_change_ctx *ctx)
                p->sched_class->prio_changed(rq, p, ctx->prio);
        }
 }
+
+#ifdef CONFIG_PREFERRED_CPU
+static DEFINE_PER_CPU(struct cpu_stop_work, npc_push_task_work);
+
+static int sched_non_preferred_cpu_push_stop(void *arg)
+{
+       struct task_struct *p = arg;
+       struct rq *rq = this_rq();
+       struct rq_flags rf;
+       int cpu;
+
+       if (cpu_preferred(rq->cpu)) {
+               scoped_guard(rq_lock, rq)
+                       rq->push_task_work_done = false;
+               put_task_struct(p);
+               return 0;
+       }
+
+       raw_spin_lock_irq(&p->pi_lock);
+
+       /* This could take rq lock. So call it before rq lock is taken */
+       cpu = select_fallback_rq(rq->cpu, p);
+       rq_lock(rq, &rf);
+       rq->push_task_work_done = false;
+       update_rq_clock(rq);
+
+       context_unsafe_alias(rq);
+
+       if (task_rq(p) == rq && task_on_rq_queued(p) &&
+           !is_migration_disabled(p))
+               rq = __migrate_task(rq, &rf, p, cpu);
+
+       rq_unlock(rq, &rf);
+       raw_spin_unlock_irq(&p->pi_lock);
+       put_task_struct(p);
+
+       return 0;
+}
+
+/*
+ * Push the current task running on non-preferred CPU(npc).
+ * Using this non preferred CPU will lead to more contention
+ * in the host. So it is better not to use this CPU.
+ *
+ * Since task is running, call a stopper to push the task out. This is
+ * similar to how task moves during hotplug. In select_fallback_rq a
+ * preferred CPU will be chosen and henceforth task shouldn't come back to
+ * this CPU again.
+ *
+ * Works for FAIR class only.
+ *
+ * If task is affined only on non-preferred CPUs, no point in moving it out.
+ */
+void sched_push_current_non_preferred_cpu(struct rq *rq)
+{
+       struct task_struct *push_task = rq->curr;
+
+       scoped_guard(rq_lock, rq) {
+               /* Push the task if its explicit affinity allows */
+               if (!task_can_sched_on_preferred(rq->cpu, push_task))
+                       return;
+
+               /* There is already a stopper thread. Don't race with it. */
+               if (rq->push_task_work_done)
+                       return;
+
+               rq->push_task_work_done = true;
+       }
+
+       /* sched_tick runs with interrupts disabled. */
+       get_task_struct(push_task);
+       stop_one_cpu_nowait(rq->cpu, sched_non_preferred_cpu_push_stop,
+                           push_task, this_cpu_ptr(&npc_push_task_work));
+}
+#endif
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 26ae13c86b69..6cee31466087 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -1277,6 +1277,8 @@ struct rq {
 
        struct list_head cfs_tasks;
 
+       bool                    push_task_work_done;
+
        struct sched_avg        avg_rt;
        struct sched_avg        avg_dl;
 #ifdef CONFIG_HAVE_SCHED_AVG_IRQ
@@ -4230,4 +4232,10 @@ DEFINE_CLASS_IS_UNCONDITIONAL(sched_change)
 
 #include "ext/ext.h"
 
+#ifdef CONFIG_PREFERRED_CPU
+void sched_push_current_non_preferred_cpu(struct rq *rq);
+#else  /* !CONFIG_PREFERRED_CPU */
+static inline void sched_push_current_non_preferred_cpu(struct rq *rq) { }
+#endif
+
 #endif /* _KERNEL_SCHED_SCHED_H */
-- 
2.47.3


Reply via email to