summaryrefslogtreecommitdiff
path: root/kernel
diff options
context:
space:
mode:
authorIngo Molnar <mingo@kernel.org>2026-10-07 18:25:55 +0200
committerIngo Molnar <mingo@kernel.org>2026-10-07 18:25:56 +0200
commit3be71701f785587f115eed063cbb148d9d577b95 (patch)
tree3ae7f72b34cb6cd046e4373beb03ff8c1c910c87 /kernel
parent959fd5e1c34667dd68c02fa5bad44bacca9b0acf (diff)
parent4a3b51aab6e25244d97936aa65e6d5425adf98e1 (diff)
downloadlinux-next-3be71701f785587f115eed063cbb148d9d577b95.tar.gz
linux-next-3be71701f785587f115eed063cbb148d9d577b95.zip
Merge branch into tip/master: 'sched/core'
# New commits in sched/core: 4a3b51aab6e2 ("smpboot: Don't park the thread if work is pending") 40dcc9bdbef3 ("irq_work: Flush lazy work CPU down on PREEMPT_RT") 791b1760accd ("irq_work: Update a comment regarding CPU hotplug invocation") 648d44bda731 ("sched/topology: Add asymmetric SMT packing override") c8fc4136fd3c ("sched/fair: Honor asymmetric SMT priority in idle selection") 53bc5c556b82 ("sched: Set TIF_NEED_RESCHED before calling __trace_set_need_resched()") 4b1f75be23c4 ("sched/core: Fix context analysis errors in non-preferred CPU push") 1fb28c664a19 ("virt/steal_governor: Enable the driver") 27d47ebce4d6 ("virt/steal_governor: Implement steal_governor policy loop") 4b9302d494ff ("virt/steal_governor: Add control knobs for handling steal values") 9a8e740ee9f6 ("virt: Introduce steal governor driver") 68957caaa9c0 ("sched/debug: Add migration stats due to non preferred CPUs") 74699f56ebcf ("sched/core: Push current task from non preferred CPU") 4ee29b029058 ("sched/fair: Load balance only among preferred CPUs") d8a3da0de843 ("sched/core: Try to use a preferred CPU in is_cpu_allowed") 620824516557 ("sysfs: Add preferred CPU file") 518b32bd5bb3 ("cpumask: Introduce cpu_preferred_mask") 06a49ef784ac ("sched/docs: Document cpu_preferred_mask and Preferred CPU concept") cfb463b7172d ("cpumask: Introduce cpumask_intersects_and") a8d0854a76a8 ("sched/cputime: Add kcpustat_field_total helper") be100c77178e ("sched: Add sched_ext hooks for proxy execution") 57c75e3ae38c ("sched: Add helper to block retained proxy donors") a49653d0abeb ("sched/core: Mark wakeups completed through ttwu_runnable()") 8f8c0417e973 ("sched/core: Dequeue waking proxy donors before reset") 313b652837d0 ("sched/core: Drop mutex locks before proxy rescheduling") 627ea30aca3b ("sched/wait: Clarify WF_SYNC wakeup semantics") d2e010082757 ("sched/eevdf: Handle more short slice waking cases") 4bf32ec3327d ("sched/eevdf: Align update_protect_slice to set_protect_slice") aae2a33ea662 ("sched/eevdf: Ensure that vprot will never go above a min slice") c9ce69fc43bd ("sched/fair: Randomize equally shallow slow-path candidates") abe440b3770f ("sched/fair: Drop idle recency from slow-path CPU selection") fbbc63fed0b0 ("sched/core: Remove redundant core_sched_seq") 819224e506bc ("sched/fair: Remove dead code on enqueue_task_fair()") c72945693b90 ("sched: Restart fair hrtick after same-task repicks") a9b3c7570564 ("sched/headers: Replace __ASSEMBLY__ with __ASSEMBLER__ in the <uapi/linux/sched.h> header") e81ee0630837 ("sched/fair: Reset NUMA fault locality after scan period update") ef9293b3b797 ("sched: dynamic: Fix preemption model strings") 879eaa76e608 ("sched: Remove unneeded function type cast in do_balance_callbacks()") f549101187c8 ("sched/deadline: check start_dl_timer expiry with ktime_before()") 2a672daa4b27 ("sched/feat: Use the new static key API for sched_feat") a5576ebce920 ("sched: Convert paravirt_steal to new static key APIs") 9650ce11f2e3 ("sched: dynamic: Simplify preempt model accessors") 5b9a28eeed37 ("sched: dynamic: Remove HAVE_PREEMPT_DYNAMIC_{CALL,KEY}") aa4178f63847 ("sched: dynamic: Simplify irqentry_exit_cond_resched()") b9d267b9d632 ("sched: dynamic: Simplify preempt_schedule{,_notrace}()") 88e0b3bb9930 ("sched: dynamic: Simplify {cond,might}_resched()") d3d16750693b ("sched: dynamic: Make PREEMPT_DYNAMIC depend on ARCH_HAS_PREEMPT_LAZY") 772d9ffbfd26 ("sched: Migrate whole chain in proxy_migrate_task()") 6b73a09e943f ("sched: Break out core of attach_tasks() helper into sched.h") 1f8805138593 ("sched: Switch rq->next_class in proxy_reset_donor()") 09351db90a28 ("sched/core: Don't proxy-exec unmatched cookie lock owners") 9be817f991e2 ("sched/core: Avoid migrating blocked_on tasks") 3dd95f077371 ("sched/core: Don't steal a proxy-exec donor") Signed-off-by: Ingo Molnar <mingo@kernel.org>
Diffstat (limited to 'kernel')
-rw-r--r--kernel/Kconfig.preempt13
-rw-r--r--kernel/cpu.c6
-rw-r--r--kernel/entry/common.c17
-rw-r--r--kernel/irq_work.c16
-rw-r--r--kernel/sched/core.c451
-rw-r--r--kernel/sched/cputime.c4
-rw-r--r--kernel/sched/deadline.c9
-rw-r--r--kernel/sched/debug.c21
-rw-r--r--kernel/sched/ext/ext.c18
-rw-r--r--kernel/sched/ext/ext.h7
-rw-r--r--kernel/sched/fair.c216
-rw-r--r--kernel/sched/idle.c5
-rw-r--r--kernel/sched/rt.c7
-rw-r--r--kernel/sched/sched.h80
-rw-r--r--kernel/sched/stop_task.c5
-rw-r--r--kernel/sched/topology.c18
-rw-r--r--kernel/sched/wait.c22
-rw-r--r--kernel/smp.c1
-rw-r--r--kernel/smpboot.c6
19 files changed, 539 insertions, 383 deletions
diff --git a/kernel/Kconfig.preempt b/kernel/Kconfig.preempt
index f294dad43bd7..edc067a0c422 100644
--- a/kernel/Kconfig.preempt
+++ b/kernel/Kconfig.preempt
@@ -132,10 +132,9 @@ config PREEMPTION
config PREEMPT_DYNAMIC
bool "Preemption behaviour defined on boot"
- depends on HAVE_PREEMPT_DYNAMIC
- select JUMP_LABEL if HAVE_PREEMPT_DYNAMIC_KEY
+ depends on ARCH_HAS_PREEMPT_LAZY
select PREEMPT_BUILD
- default y if HAVE_PREEMPT_DYNAMIC_CALL
+ default y
help
This option allows to define the preemption model on the kernel
command line parameter and thus override the default preemption
@@ -145,9 +144,7 @@ config PREEMPT_DYNAMIC
provide a pre-built kernel binary to reduce the number of kernel
flavors they offer while still offering different usecases.
- The runtime overhead is negligible with HAVE_STATIC_CALL_INLINE enabled
- but if runtime patching is not available for the specific architecture
- then the potential overhead should be considered.
+ The runtime overhead is negligible.
Interesting if you want the same pre-built kernel should be used for
both Server and Desktop workloads.
@@ -197,3 +194,7 @@ config SCHED_CLASS_EXT
For more information:
Documentation/scheduler/sched-ext.rst
https://github.com/sched-ext/scx
+
+config PREFERRED_CPU
+ bool
+ depends on SMP && PARAVIRT
diff --git a/kernel/cpu.c b/kernel/cpu.c
index b3c8553d7bd6..376d297a6292 100644
--- a/kernel/cpu.c
+++ b/kernel/cpu.c
@@ -3103,6 +3103,11 @@ EXPORT_SYMBOL(__cpu_dying_mask);
atomic_t __num_online_cpus __read_mostly;
EXPORT_SYMBOL(__num_online_cpus);
+#ifdef CONFIG_PREFERRED_CPU
+struct cpumask __cpu_preferred_mask __read_mostly;
+EXPORT_SYMBOL_GPL(__cpu_preferred_mask);
+#endif
+
void init_cpu_present(const struct cpumask *src)
{
cpumask_copy(&__cpu_present_mask, src);
@@ -3160,6 +3165,7 @@ void __init boot_cpu_init(void)
/* Mark the boot cpu "present", "online" etc for SMP and UP case */
set_cpu_online(cpu, true);
set_cpu_active(cpu, true);
+ set_cpu_preferred(cpu, true);
set_cpu_present(cpu, true);
set_cpu_possible(cpu, true);
diff --git a/kernel/entry/common.c b/kernel/entry/common.c
index e3d381fd3d25..e234b04373fe 100644
--- a/kernel/entry/common.c
+++ b/kernel/entry/common.c
@@ -123,7 +123,7 @@ noinstr irqentry_state_t irqentry_enter(struct pt_regs *regs)
/**
* arch_irqentry_exit_need_resched - Architecture specific need resched function
*
- * Invoked from raw_irqentry_exit_cond_resched() to check if resched is needed.
+ * Invoked from irqentry_exit_cond_resched() to check if resched is needed.
* Defaults return true.
*
* The main purpose is to permit arch to avoid preemption of a task from an IRQ.
@@ -134,7 +134,7 @@ static inline bool arch_irqentry_exit_need_resched(void);
static inline bool arch_irqentry_exit_need_resched(void) { return true; }
#endif
-void raw_irqentry_exit_cond_resched(void)
+void irqentry_exit_cond_resched(void)
{
if (!preempt_count()) {
/* Sanity check RCU and thread stack */
@@ -145,19 +145,6 @@ void raw_irqentry_exit_cond_resched(void)
preempt_schedule_irq();
}
}
-#ifdef CONFIG_PREEMPT_DYNAMIC
-#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-DEFINE_STATIC_CALL(irqentry_exit_cond_resched, raw_irqentry_exit_cond_resched);
-#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-DEFINE_STATIC_KEY_TRUE(sk_dynamic_irqentry_exit_cond_resched);
-void dynamic_irqentry_exit_cond_resched(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_irqentry_exit_cond_resched))
- return;
- raw_irqentry_exit_cond_resched();
-}
-#endif
-#endif
noinstr void irqentry_exit(struct pt_regs *regs, irqentry_state_t state)
{
diff --git a/kernel/irq_work.c b/kernel/irq_work.c
index f7e2dc2c30c6..73eabcbdcd50 100644
--- a/kernel/irq_work.c
+++ b/kernel/irq_work.c
@@ -252,10 +252,7 @@ static void irq_work_run_list(struct llist_head *list)
irq_work_single(work);
}
-/*
- * hotplug calls this through:
- * hotplug_cfd() -> flush_smp_call_function_queue()
- */
+/* CPU hotplug calls this through smpcfd_dying_cpu() */
void irq_work_run(void)
{
irq_work_run_list(this_cpu_ptr(&raised_list));
@@ -266,6 +263,17 @@ void irq_work_run(void)
}
EXPORT_SYMBOL_GPL(irq_work_run);
+void irq_work_run_cpu(unsigned int cpu)
+{
+ if (WARN_ON_ONCE(!cpumask_test_cpu(cpu, cpu_dying_mask)))
+ return;
+
+ if (!IS_ENABLED(CONFIG_PREEMPT_RT))
+ return;
+
+ irq_work_run_list(per_cpu_ptr(&lazy_list, cpu));
+}
+
void irq_work_tick(void)
{
struct llist_head *raised = this_cpu_ptr(&raised_list);
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 1fe40de6ebe3..047fbe8ef298 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -805,7 +805,7 @@ struct rq *_task_rq_lock(struct task_struct *p, struct rq_flags *rf)
/* Use CONFIG_PARAVIRT as this will avoid more #ifdef in arch code. */
#ifdef CONFIG_PARAVIRT
-struct static_key paravirt_steal_rq_enabled;
+DEFINE_STATIC_KEY_FALSE(paravirt_steal_rq_enabled);
#endif
static void update_rq_clock_task(struct rq *rq, s64 delta)
@@ -844,7 +844,7 @@ static void update_rq_clock_task(struct rq *rq, s64 delta)
}
#endif
#ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING
- if (static_key_false((&paravirt_steal_rq_enabled))) {
+ if (static_branch_unlikely(&paravirt_steal_rq_enabled)) {
u64 prev_steal;
steal = prev_steal = paravirt_steal_clock(cpu_of(rq));
@@ -1196,6 +1196,7 @@ static void __resched_curr(struct rq *rq, int tif)
{
struct task_struct *curr = rq->curr;
struct thread_info *cti = task_thread_info(curr);
+ bool need_ipi;
int cpu;
lockdep_assert_rq_held(rq);
@@ -1212,15 +1213,17 @@ static void __resched_curr(struct rq *rq, int tif)
cpu = cpu_of(rq);
- trace_sched_set_need_resched_tp(curr, cpu, tif);
if (cpu == smp_processor_id()) {
set_ti_thread_flag(cti, tif);
if (tif == TIF_NEED_RESCHED)
set_preempt_need_resched();
+ trace_sched_set_need_resched_tp(curr, cpu, tif);
return;
}
- if (set_nr_and_not_polling(cti, tif)) {
+ need_ipi = set_nr_and_not_polling(cti, tif);
+ trace_sched_set_need_resched_tp(curr, cpu, tif);
+ if (need_ipi) {
if (tif == TIF_NEED_RESCHED)
smp_send_reschedule(cpu);
} else {
@@ -2252,7 +2255,8 @@ void deactivate_task(struct rq *rq, struct task_struct *p, int flags)
dequeue_task(rq, p, flags);
}
-static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_state)
+static bool dequeue_block_task(struct rq *rq, struct task_struct *p,
+ unsigned long task_state)
{
int flags = DEQUEUE_NOCLOCK;
@@ -2273,9 +2277,15 @@ static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_
*
* Where __schedule() and ttwu() have matching control dependencies.
*
- * After this, schedule() must not care about p->state any more.
+ * Once the caller invokes __block_task(), schedule() must not care about
+ * p->state any more.
*/
- if (dequeue_task(rq, p, DEQUEUE_SLEEP | flags))
+ return dequeue_task(rq, p, DEQUEUE_SLEEP | flags);
+}
+
+static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_state)
+{
+ if (dequeue_block_task(rq, p, task_state))
__block_task(rq, p);
}
@@ -2504,6 +2514,24 @@ static inline bool rq_has_pinned_tasks(struct rq *rq)
return rq->nr_pinned;
}
+static inline bool task_can_migrate_to_preferred(struct task_struct *p, int cpu)
+{
+ /* No need to migrate from a preferred CPU */
+ if (cpu_preferred(cpu))
+ return false;
+
+ /* Only FAIR tasks honor preferred CPU state */
+ if (unlikely(p->sched_class != &fair_sched_class))
+ return false;
+
+ /* Ignore preferred state if task affinity is changing */
+ if (unlikely(!cpumask_test_cpu(task_cpu(p), p->cpus_ptr)))
+ return false;
+
+ return cpumask_intersects_and(p->cpus_ptr, cpu_preferred_mask,
+ task_cpu_possible_mask(p));
+}
+
/*
* Per-CPU kthreads are allowed to run on !active && online CPUs, see
* __set_cpus_allowed_ptr() and select_fallback_rq().
@@ -2519,8 +2547,12 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu)
return cpu_online(cpu);
/* Non kernel threads are not allowed during either online or offline. */
- if (!(p->flags & PF_KTHREAD))
+ if (!(p->flags & PF_KTHREAD)) {
+ /* Try to use preferred CPU if task's affinity allows */
+ if (task_can_migrate_to_preferred(p, cpu))
+ return false;
return cpu_active(cpu);
+ }
/* KTHREAD_IS_PER_CPU is always allowed. */
if (kthread_is_per_cpu(p))
@@ -2530,7 +2562,11 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu)
if (cpu_dying(cpu))
return false;
- /* But are allowed during online. */
+ /* Try to keep unbound kthreads on a preferred CPU if possible. */
+ if (task_can_migrate_to_preferred(p, cpu))
+ return false;
+
+ /* Otherwise, they are allowed to run on online CPU. */
return cpu_online(cpu);
}
@@ -3773,6 +3809,7 @@ static inline void proxy_reset_donor(struct rq *rq)
WARN_ON_ONCE(rq->donor == rq->curr);
put_prev_set_next_task(rq, rq->donor, rq->curr);
+ rq->next_class = rq->curr->sched_class;
rq_set_donor(rq, rq->curr);
zap_balance_callbacks(rq);
resched_curr(rq);
@@ -3787,6 +3824,8 @@ static inline void proxy_reset_donor(struct rq *rq)
*/
static inline bool proxy_needs_return(struct rq *rq, struct task_struct *p)
{
+ bool dequeued;
+
/*
* Typically per __set_task_cpu(), task_cpu(p) == p->wake_cpu.
*
@@ -3809,12 +3848,23 @@ static inline bool proxy_needs_return(struct rq *rq, struct task_struct *p)
/* If already current, don't need to return migrate */
if (task_current(rq, p))
return false;
-
- /* If we're return migrating the rq->donor, switch it out for idle */
- if (task_current_donor(rq, p))
- proxy_reset_donor(rq);
}
- block_task(rq, p, TASK_WAKING);
+
+ dequeued = dequeue_block_task(rq, p, TASK_WAKING);
+
+ /*
+ * Dequeue @p from its scheduling class before resetting rq->donor.
+ * In particular, sched_ext needs to end the donor's running session
+ * and clear SCX_TASK_QUEUED before put_prev_task_scx() is called by
+ * proxy_reset_donor(); otherwise it would reenqueue the blocked donor.
+ *
+ * Keep on_rq set until all donor references have been replaced.
+ */
+ if (task_current_donor(rq, p))
+ proxy_reset_donor(rq);
+
+ if (dequeued)
+ __block_task(rq, p);
return true;
}
#else /* !CONFIG_SCHED_PROXY_EXEC */
@@ -3905,7 +3955,7 @@ static int ttwu_runnable(struct task_struct *p, int wake_flags)
* When on_rq && !on_cpu the task is preempted, see if
* it should preempt the task that is current now.
*/
- wakeup_preempt(rq, p, wake_flags);
+ wakeup_preempt(rq, p, wake_flags | WF_TTWU_RQ);
}
ttwu_do_wakeup(p);
return 1;
@@ -5149,7 +5199,7 @@ static void do_balance_callbacks(struct rq *rq, struct balance_callback *head)
lockdep_assert_rq_held(rq);
while (head) {
- func = (void (*)(struct rq *))head->func;
+ func = head->func;
next = head->next;
head->next = NULL;
head = next;
@@ -5789,6 +5839,9 @@ void sched_tick(void)
unsigned long hw_pressure;
u64 resched_latency;
+ if (!cpu_preferred(cpu))
+ sched_push_current_non_preferred_cpu(rq);
+
if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE))
arch_scale_freq_tick();
@@ -6283,10 +6336,7 @@ pick_next_task(struct rq *rq, struct rq_flags *rf)
* selection. In this case, do a core-wide selection.
*/
if (rq->core->core_pick_seq == rq->core->core_task_seq &&
- rq->core->core_pick_seq != rq->core_sched_seq &&
rq->core_pick) {
- WRITE_ONCE(rq->core_sched_seq, rq->core->core_pick_seq);
-
next = rq->core_pick;
rq->dl_server = rq->core_dl_server;
rq->core_pick = NULL;
@@ -6318,11 +6368,13 @@ restart:
}
/*
- * core->core_task_seq, core->core_pick_seq, rq->core_sched_seq
+ * core->core_task_seq, core->core_pick_seq
*
* @task_seq guards the task state ({en,de}queues)
* @pick_seq is the @task_seq we did a selection on
- * @sched_seq is the @pick_seq we scheduled
+ *
+ * Once a core-wide selection is committed, a non-NULL core_pick denotes
+ * a pick which still needs to be consumed on this CPU.
*
* However, preemptions can cause multiple picks on the same task set.
* 'Fix' this by also increasing @task_seq for every pick.
@@ -6429,7 +6481,6 @@ restart:
rq->core->core_pick_seq = rq->core->core_task_seq;
next = rq->core_pick;
- rq->core_sched_seq = rq->core->core_pick_seq;
/* Something should have been selected for current CPU */
WARN_ON_ONCE(!next);
@@ -6517,7 +6568,10 @@ static bool try_steal_cookie(int this, int that)
return false;
do {
- if (p == src->core_pick || p == src->curr)
+ if (p == src->core_pick || p == src->curr || p == src->donor)
+ goto next;
+
+ if (task_is_blocked(p))
goto next;
if (!is_cpu_allowed(p, this))
@@ -6820,6 +6874,34 @@ static void proxy_deactivate(struct rq *rq, struct task_struct *donor)
block_task(rq, donor, state);
}
+/*
+ * Remove a retained proxy donor before changing its scheduler ownership.
+ * The caller holds p->pi_lock, so p cannot wake and migrate if block_task()
+ * drops it from the runqueue. If DELAY_DEQUEUE keeps a blocked fair task
+ * queued, switching_from_fair() completes the dequeue in the immediately
+ * following sched_change_begin().
+ */
+void sched_proxy_block_task(struct rq *rq, struct task_struct *p)
+{
+ unsigned long state = READ_ONCE(p->__state);
+
+ lockdep_assert_held(&p->pi_lock);
+ lockdep_assert_rq_held(rq);
+
+ if (!p->is_blocked || !task_on_rq_queued(p))
+ return;
+ if (WARN_ON_ONCE(state == TASK_RUNNING))
+ return;
+
+ if (task_current_donor(rq, p))
+ proxy_reset_donor(rq);
+
+ if (!p->se.sched_delayed)
+ block_task(rq, p, state);
+
+ WARN_ON_ONCE(task_on_rq_queued(p) && !p->se.sched_delayed);
+}
+
static inline void proxy_release_rq_lock(struct rq *rq, struct rq_flags *rf)
__releases(__rq_lockp(rq))
{
@@ -6865,9 +6947,9 @@ static void proxy_migrate_task(struct rq *rq, struct rq_flags *rf,
__must_hold(__rq_lockp(rq))
{
struct rq *target_rq = cpu_rq(target_cpu);
+ LIST_HEAD(migrate_list);
lockdep_assert_rq_held(rq);
- WARN_ON(p == rq->curr);
/*
* Since we are migrating a blocked donor, it could be rq->donor,
* and we want to make sure there aren't any references from this
@@ -6880,13 +6962,20 @@ static void proxy_migrate_task(struct rq *rq, struct rq_flags *rf,
* before we release the lock.
*/
proxy_resched_idle(rq);
-
- deactivate_task(rq, p, DEQUEUE_NOCLOCK);
- proxy_set_task_cpu(p, target_cpu);
-
+ for (; p; p = p->blocked_donor) {
+ WARN_ON(p == rq->curr);
+ deactivate_task(rq, p, DEQUEUE_NOCLOCK);
+ proxy_set_task_cpu(p, target_cpu);
+ /*
+ * We can re-use se.group_node to migrate the thing,
+ * because @p is deactivated (won't be balanced) and
+ * we hold the rq_lock.
+ */
+ list_add(&p->se.group_node, &migrate_list);
+ }
proxy_release_rq_lock(rq, rf);
- attach_one_task(target_rq, p);
+ __attach_tasks(target_rq, &migrate_list);
proxy_reacquire_rq_lock(rq, rf);
}
@@ -6979,7 +7068,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
if (!READ_ONCE(owner->on_rq) || owner->se.sched_delayed) {
/* XXX Don't handle blocked owners/delayed dequeue yet */
if (curr_in_chain)
- return proxy_resched_idle(rq);
+ goto resched_idle;
__clear_task_blocked_on(p, NULL);
goto deactivate;
}
@@ -6991,7 +7080,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
* and leave that CPU to sort things out.
*/
if (curr_in_chain)
- return proxy_resched_idle(rq);
+ goto resched_idle;
goto migrate_task;
}
@@ -7004,7 +7093,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
* case we should end up back in find_proxy_task(), this time
* hopefully with all relevant tasks already enqueued.
*/
- return proxy_resched_idle(rq);
+ goto resched_idle;
}
/*
@@ -7041,7 +7130,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
* So schedule rq->idle so that ttwu_runnable() can get the rq
* lock and mark owner as running.
*/
- return proxy_resched_idle(rq);
+ goto resched_idle;
}
/*
* OK, now we're absolutely sure @owner is on this
@@ -7051,8 +7140,18 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
owner->blocked_donor = p;
}
WARN_ON_ONCE(owner && !owner->on_rq);
+
+ if (owner && !sched_cpu_cookie_match(rq, owner)) {
+ if (curr_in_chain)
+ return proxy_resched_idle(rq);
+ p = donor; /* Deactivate the donor, not the runnable owner */
+ clear_task_blocked_on(p, NULL);
+ goto deactivate;
+ }
return owner;
+resched_idle:
+ return proxy_resched_idle(rq);
deactivate:
proxy_deactivate(rq, p);
return NULL;
@@ -7184,13 +7283,12 @@ static void __sched notrace __schedule(int sched_mode)
}
} else if (!preempt && prev_state) {
/*
- * We pass task_is_blocked() as the should_block arg
- * in order to keep mutex-blocked tasks on the runqueue
- * for slection with proxy-exec (without proxy-exec
- * task_is_blocked() will always be false).
+ * Keep mutex-blocked tasks on the runqueue for proxy execution
+ * only when their scheduling class allows it. Without proxy
+ * execution, task_is_blocked() always returns false.
*/
try_to_block_task(rq, prev, &prev_state,
- !task_is_blocked(prev));
+ !task_is_blocked(prev) || !scx_allow_proxy_exec(prev));
switch_count = &prev->nvcsw;
}
@@ -7211,6 +7309,7 @@ pick_again:
}
if (next == rq->idle) {
zap_balance_callbacks(rq);
+ scx_proxy_reenqueue_retry(rq, next);
goto keep_resched;
}
}
@@ -7229,8 +7328,10 @@ pick_again:
* on_cpu.
*/
donor->sched_class->put_prev_task(rq, donor, donor);
- donor->sched_class->set_next_task(rq, donor, true);
+ donor->sched_class->set_next_task(rq, donor, SNT_PICK);
}
+ scx_proxy_donor_start(rq);
+ scx_proxy_reenqueue_retry(rq, next);
} else {
rq_set_donor(rq, next);
}
@@ -7489,27 +7590,6 @@ asmlinkage __visible void __sched notrace preempt_schedule(void)
NOKPROBE_SYMBOL(preempt_schedule);
EXPORT_SYMBOL(preempt_schedule);
-#ifdef CONFIG_PREEMPT_DYNAMIC
-# ifdef CONFIG_HAVE_PREEMPT_DYNAMIC_CALL
-# ifndef preempt_schedule_dynamic_enabled
-# define preempt_schedule_dynamic_enabled preempt_schedule
-# define preempt_schedule_dynamic_disabled NULL
-# endif
-DEFINE_STATIC_CALL(preempt_schedule, preempt_schedule_dynamic_enabled);
-EXPORT_STATIC_CALL_TRAMP(preempt_schedule);
-# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule);
-void __sched notrace dynamic_preempt_schedule(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_preempt_schedule))
- return;
- preempt_schedule();
-}
-NOKPROBE_SYMBOL(dynamic_preempt_schedule);
-EXPORT_SYMBOL(dynamic_preempt_schedule);
-# endif
-#endif /* CONFIG_PREEMPT_DYNAMIC */
-
/**
* preempt_schedule_notrace - preempt_schedule called by tracing
*
@@ -7562,27 +7642,6 @@ asmlinkage __visible void __sched notrace preempt_schedule_notrace(void)
}
EXPORT_SYMBOL_GPL(preempt_schedule_notrace);
-#ifdef CONFIG_PREEMPT_DYNAMIC
-# if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-# ifndef preempt_schedule_notrace_dynamic_enabled
-# define preempt_schedule_notrace_dynamic_enabled preempt_schedule_notrace
-# define preempt_schedule_notrace_dynamic_disabled NULL
-# endif
-DEFINE_STATIC_CALL(preempt_schedule_notrace, preempt_schedule_notrace_dynamic_enabled);
-EXPORT_STATIC_CALL_TRAMP(preempt_schedule_notrace);
-# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule_notrace);
-void __sched notrace dynamic_preempt_schedule_notrace(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_preempt_schedule_notrace))
- return;
- preempt_schedule_notrace();
-}
-NOKPROBE_SYMBOL(dynamic_preempt_schedule_notrace);
-EXPORT_SYMBOL(dynamic_preempt_schedule_notrace);
-# endif
-#endif
-
#endif /* CONFIG_PREEMPTION */
/*
@@ -7799,7 +7858,7 @@ out_unlock:
}
#endif /* CONFIG_RT_MUTEXES */
-#if !defined(CONFIG_PREEMPTION) || defined(CONFIG_PREEMPT_DYNAMIC)
+#if !defined(CONFIG_PREEMPTION)
int __sched __cond_resched(void)
{
if (should_resched(0) && !irqs_disabled()) {
@@ -7827,38 +7886,6 @@ int __sched __cond_resched(void)
EXPORT_SYMBOL(__cond_resched);
#endif
-#ifdef CONFIG_PREEMPT_DYNAMIC
-# ifdef CONFIG_HAVE_PREEMPT_DYNAMIC_CALL
-# define cond_resched_dynamic_enabled __cond_resched
-# define cond_resched_dynamic_disabled ((void *)&__static_call_return0)
-DEFINE_STATIC_CALL_RET0(cond_resched, __cond_resched);
-EXPORT_STATIC_CALL_TRAMP(cond_resched);
-
-# define might_resched_dynamic_enabled __cond_resched
-# define might_resched_dynamic_disabled ((void *)&__static_call_return0)
-DEFINE_STATIC_CALL_RET0(might_resched, __cond_resched);
-EXPORT_STATIC_CALL_TRAMP(might_resched);
-# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-static DEFINE_STATIC_KEY_FALSE(sk_dynamic_cond_resched);
-int __sched dynamic_cond_resched(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_cond_resched))
- return 0;
- return __cond_resched();
-}
-EXPORT_SYMBOL(dynamic_cond_resched);
-
-static DEFINE_STATIC_KEY_FALSE(sk_dynamic_might_resched);
-int __sched dynamic_might_resched(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_might_resched))
- return 0;
- return __cond_resched();
-}
-EXPORT_SYMBOL(dynamic_might_resched);
-# endif
-#endif /* CONFIG_PREEMPT_DYNAMIC */
-
/*
* __cond_resched_lock() - if a reschedule is pending, drop the given lock,
* call schedule, and on return reacquire the lock.
@@ -7928,50 +7955,21 @@ EXPORT_SYMBOL(__cond_resched_rwlock_write);
# endif
/*
- * SC:cond_resched
- * SC:might_resched
- * SC:preempt_schedule
- * SC:preempt_schedule_notrace
- * SC:irqentry_exit_cond_resched
- *
- *
* NONE:
- * cond_resched <- __cond_resched
- * might_resched <- RET0
- * preempt_schedule <- NOP
- * preempt_schedule_notrace <- NOP
- * irqentry_exit_cond_resched <- NOP
- * dynamic_preempt_lazy <- false
+ * (unselectable)
*
* VOLUNTARY:
- * cond_resched <- __cond_resched
- * might_resched <- __cond_resched
- * preempt_schedule <- NOP
- * preempt_schedule_notrace <- NOP
- * irqentry_exit_cond_resched <- NOP
- * dynamic_preempt_lazy <- false
+ * (unselectable)
*
* FULL:
- * cond_resched <- RET0
- * might_resched <- RET0
- * preempt_schedule <- preempt_schedule
- * preempt_schedule_notrace <- preempt_schedule_notrace
- * irqentry_exit_cond_resched <- irqentry_exit_cond_resched
* dynamic_preempt_lazy <- false
*
* LAZY:
- * cond_resched <- RET0
- * might_resched <- RET0
- * preempt_schedule <- preempt_schedule
- * preempt_schedule_notrace <- preempt_schedule_notrace
- * irqentry_exit_cond_resched <- irqentry_exit_cond_resched
* dynamic_preempt_lazy <- true
*/
enum {
preempt_dynamic_undefined = -1,
- preempt_dynamic_none,
- preempt_dynamic_voluntary,
preempt_dynamic_full,
preempt_dynamic_lazy,
};
@@ -7980,21 +7978,11 @@ int preempt_dynamic_mode = preempt_dynamic_undefined;
int sched_dynamic_mode(const char *str)
{
-# if !(defined(CONFIG_PREEMPT_RT) || defined(CONFIG_ARCH_HAS_PREEMPT_LAZY))
- if (!strcmp(str, "none"))
- return preempt_dynamic_none;
-
- if (!strcmp(str, "voluntary"))
- return preempt_dynamic_voluntary;
-# endif
-
if (!strcmp(str, "full"))
return preempt_dynamic_full;
-# ifdef CONFIG_ARCH_HAS_PREEMPT_LAZY
if (!strcmp(str, "lazy"))
return preempt_dynamic_lazy;
-# endif
return -EINVAL;
}
@@ -8002,71 +7990,18 @@ int sched_dynamic_mode(const char *str)
# define preempt_dynamic_key_enable(f) static_key_enable(&sk_dynamic_##f.key)
# define preempt_dynamic_key_disable(f) static_key_disable(&sk_dynamic_##f.key)
-# if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-# define preempt_dynamic_enable(f) static_call_update(f, f##_dynamic_enabled)
-# define preempt_dynamic_disable(f) static_call_update(f, f##_dynamic_disabled)
-# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-# define preempt_dynamic_enable(f) preempt_dynamic_key_enable(f)
-# define preempt_dynamic_disable(f) preempt_dynamic_key_disable(f)
-# else
-# error "Unsupported PREEMPT_DYNAMIC mechanism"
-# endif
-
static DEFINE_MUTEX(sched_dynamic_mutex);
static void __sched_dynamic_update(int mode)
{
- /*
- * Avoid {NONE,VOLUNTARY} -> FULL transitions from ever ending up in
- * the ZERO state, which is invalid.
- */
- preempt_dynamic_enable(cond_resched);
- preempt_dynamic_enable(might_resched);
- preempt_dynamic_enable(preempt_schedule);
- preempt_dynamic_enable(preempt_schedule_notrace);
- preempt_dynamic_enable(irqentry_exit_cond_resched);
- preempt_dynamic_key_disable(preempt_lazy);
-
switch (mode) {
- case preempt_dynamic_none:
- preempt_dynamic_enable(cond_resched);
- preempt_dynamic_disable(might_resched);
- preempt_dynamic_disable(preempt_schedule);
- preempt_dynamic_disable(preempt_schedule_notrace);
- preempt_dynamic_disable(irqentry_exit_cond_resched);
- preempt_dynamic_key_disable(preempt_lazy);
- if (mode != preempt_dynamic_mode)
- pr_info("Dynamic Preempt: none\n");
- break;
-
- case preempt_dynamic_voluntary:
- preempt_dynamic_enable(cond_resched);
- preempt_dynamic_enable(might_resched);
- preempt_dynamic_disable(preempt_schedule);
- preempt_dynamic_disable(preempt_schedule_notrace);
- preempt_dynamic_disable(irqentry_exit_cond_resched);
- preempt_dynamic_key_disable(preempt_lazy);
- if (mode != preempt_dynamic_mode)
- pr_info("Dynamic Preempt: voluntary\n");
- break;
-
case preempt_dynamic_full:
- preempt_dynamic_disable(cond_resched);
- preempt_dynamic_disable(might_resched);
- preempt_dynamic_enable(preempt_schedule);
- preempt_dynamic_enable(preempt_schedule_notrace);
- preempt_dynamic_enable(irqentry_exit_cond_resched);
preempt_dynamic_key_disable(preempt_lazy);
if (mode != preempt_dynamic_mode)
pr_info("Dynamic Preempt: full\n");
break;
case preempt_dynamic_lazy:
- preempt_dynamic_disable(cond_resched);
- preempt_dynamic_disable(might_resched);
- preempt_dynamic_enable(preempt_schedule);
- preempt_dynamic_enable(preempt_schedule_notrace);
- preempt_dynamic_enable(irqentry_exit_cond_resched);
preempt_dynamic_key_enable(preempt_lazy);
if (mode != preempt_dynamic_mode)
pr_info("Dynamic Preempt: lazy\n");
@@ -8099,11 +8034,7 @@ __setup("preempt=", setup_preempt_mode);
static void __init preempt_dynamic_init(void)
{
if (preempt_dynamic_mode == preempt_dynamic_undefined) {
- if (IS_ENABLED(CONFIG_PREEMPT_NONE)) {
- sched_dynamic_update(preempt_dynamic_none);
- } else if (IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY)) {
- sched_dynamic_update(preempt_dynamic_voluntary);
- } else if (IS_ENABLED(CONFIG_PREEMPT_LAZY)) {
+ if (IS_ENABLED(CONFIG_PREEMPT_LAZY)) {
sched_dynamic_update(preempt_dynamic_lazy);
} else {
/* Default static call setting, nothing to do */
@@ -8123,8 +8054,6 @@ static void __init preempt_dynamic_init(void)
} \
EXPORT_SYMBOL_GPL(preempt_model_##mode)
-PREEMPT_MODEL_ACCESSOR(none);
-PREEMPT_MODEL_ACCESSOR(voluntary);
PREEMPT_MODEL_ACCESSOR(full);
PREEMPT_MODEL_ACCESSOR(lazy);
@@ -8137,7 +8066,7 @@ static inline void preempt_dynamic_init(void) { }
#endif /* CONFIG_PREEMPT_DYNAMIC */
const char *preempt_modes[] = {
- "none", "voluntary", "full", "lazy", NULL,
+ "full", "lazy", NULL,
};
const char *preempt_model_str(void)
@@ -8759,6 +8688,9 @@ int sched_cpu_activate(unsigned int cpu)
*/
sched_set_rq_online(rq, cpu);
+ /* preferred is subset of active and follows its state */
+ set_cpu_preferred(cpu, true);
+
return 0;
}
@@ -8772,6 +8704,8 @@ int sched_cpu_deactivate(unsigned int cpu)
if (ret)
return ret;
+ set_cpu_preferred(cpu, false);
+
/*
* Remove CPU from nohz.idle_cpus_mask to prevent participating in
* load balancing when not active
@@ -11349,3 +11283,88 @@ void sched_change_end(struct sched_change_ctx *ctx)
p->sched_class->prio_changed(rq, p, ctx->prio);
}
}
+
+#ifdef CONFIG_PREFERRED_CPU
+static DEFINE_PER_CPU(struct cpu_stop_work, npc_push_task_work);
+
+static int sched_non_preferred_cpu_push_stop(void *arg)
+{
+ struct task_struct *p = arg;
+ struct rq *rq = this_rq();
+ struct rq_flags rf;
+ int cpu;
+
+ if (cpu_preferred(rq->cpu)) {
+ scoped_guard(rq_lock_irqsave, rq)
+ rq->npc_push_work_pending = false;
+ put_task_struct(p);
+ return 0;
+ }
+
+ scoped_guard (raw_spinlock_irq, &p->pi_lock) {
+ /*
+ * select_fallback_rq() may acquire the rq lock in case of
+ * fallback. So call it before grabbing rq lock. If the task
+ * migrates to another CPU before the rq lock is acquired,
+ * subsequent validation of task's current rq will help to
+ * safely bail out.
+ */
+ cpu = select_fallback_rq(rq->cpu, p);
+ context_unsafe_alias(rq);
+ rq_lock(rq, &rf);
+ rq->npc_push_work_pending = false;
+ update_rq_clock(rq);
+
+ if (task_rq(p) == rq && task_on_rq_queued(p)) {
+ struct rq *dest_rq = __migrate_task(rq, &rf, p, cpu);
+
+ if (rq != dest_rq)
+ schedstat_inc(p->stats.nr_migrations_cpu_non_preferred);
+ rq = dest_rq;
+ }
+ rq_unlock(rq, &rf);
+ }
+
+ put_task_struct(p);
+ return 0;
+}
+
+/*
+ * Push the current task running on non-preferred CPU(npc).
+ * Using this non preferred CPU will lead to more contention
+ * in the host. So it is better not to use this CPU.
+ *
+ * Since task is running, call a stopper to push the task out. This is
+ * similar to how task moves during hotplug. In select_fallback_rq() a
+ * preferred CPU will be chosen and henceforth task shouldn't come back to
+ * this CPU again.
+ *
+ * Works for FAIR class only.
+ *
+ * If task is affined only on non-preferred CPUs, no point in moving it out.
+ */
+void sched_push_current_non_preferred_cpu(struct rq *rq)
+{
+ struct task_struct *push_task = rq->curr;
+
+ scoped_guard(rq_lock, rq) {
+ /* Push the task if its explicit affinity allows */
+ if (!task_can_migrate_to_preferred(push_task, rq->cpu))
+ return;
+
+ /* There is already a stopper thread. Don't race with it. */
+ if (rq->npc_push_work_pending)
+ return;
+
+ if (is_migration_disabled(push_task))
+ return;
+
+ rq->npc_push_work_pending = true;
+ }
+
+ /* sched_tick runs with interrupts disabled. */
+ get_task_struct(push_task);
+ stop_one_cpu_nowait(rq->cpu, sched_non_preferred_cpu_push_stop,
+ push_task, this_cpu_ptr(&npc_push_task_work));
+}
+#endif
diff --git a/kernel/sched/cputime.c b/kernel/sched/cputime.c
index 06bddaa738e5..f16970ca81d0 100644
--- a/kernel/sched/cputime.c
+++ b/kernel/sched/cputime.c
@@ -255,7 +255,7 @@ void __account_forceidle_time(struct task_struct *p, u64 delta)
* occasion account more time than the calling functions think elapsed.
*/
#ifdef CONFIG_PARAVIRT
-struct static_key paravirt_steal_enabled;
+DEFINE_STATIC_KEY_FALSE(paravirt_steal_enabled);
#ifdef CONFIG_HAVE_PV_STEAL_CLOCK_GEN
static u64 native_steal_clock(int cpu)
@@ -270,7 +270,7 @@ DEFINE_STATIC_CALL(pv_steal_clock, native_steal_clock);
static __always_inline u64 steal_account_process_time(u64 maxtime)
{
#ifdef CONFIG_PARAVIRT
- if (static_key_false(&paravirt_steal_enabled)) {
+ if (static_branch_unlikely(&paravirt_steal_enabled)) {
u64 steal;
steal = paravirt_steal_clock(smp_processor_id());
diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c
index 0663c00c41c0..c0ebdcde5fe5 100644
--- a/kernel/sched/deadline.c
+++ b/kernel/sched/deadline.c
@@ -1097,7 +1097,7 @@ static int start_dl_timer(struct sched_dl_entity *dl_se)
* chosen as the deadline is too small, don't even try to
* start the timer in the past!
*/
- if (ktime_us_delta(act, now) < 0)
+ if (ktime_before(act, now))
return 0;
/*
@@ -2773,11 +2773,14 @@ static void start_hrtick_dl(struct rq *rq, struct sched_dl_entity *dl_se)
* DL keeps current in tree, because ->deadline is not typically changed while
* a task is runnable.
*/
-static void set_next_task_dl(struct rq *rq, struct task_struct *p, bool first)
+static void set_next_task_dl(struct rq *rq, struct task_struct *p, enum snt_e type)
{
struct sched_dl_entity *dl_se = &p->dl;
struct dl_rq *dl_rq = &rq->dl;
+ if (type == SNT_REPICK)
+ return;
+
p->se.exec_start = rq_clock_task(rq);
if (on_dl_rq(&p->dl))
update_stats_wait_end_dl(dl_rq, dl_se);
@@ -2788,7 +2791,7 @@ static void set_next_task_dl(struct rq *rq, struct task_struct *p, bool first)
WARN_ON_ONCE(dl_rq->curr);
dl_rq->curr = dl_se;
- if (!first)
+ if (type != SNT_PICK)
return;
if (rq->donor->sched_class != &dl_sched_class)
diff --git a/kernel/sched/debug.c b/kernel/sched/debug.c
index 72236db67983..e6a3b516c703 100644
--- a/kernel/sched/debug.c
+++ b/kernel/sched/debug.c
@@ -73,13 +73,13 @@ static int sched_feat_show(struct seq_file *m, void *v)
#ifdef CONFIG_JUMP_LABEL
-#define jump_label_key__true STATIC_KEY_INIT_TRUE
-#define jump_label_key__false STATIC_KEY_INIT_FALSE
+#define jump_label_key__true { .key_true = STATIC_KEY_TRUE_INIT }
+#define jump_label_key__false { .key_false = STATIC_KEY_FALSE_INIT }
#define SCHED_FEAT(name, enabled) \
jump_label_key__##enabled ,
-struct static_key sched_feat_keys[__SCHED_FEAT_NR] = {
+union sched_feat_key sched_feat_keys[__SCHED_FEAT_NR] = {
#include "features.h"
};
@@ -87,12 +87,12 @@ struct static_key sched_feat_keys[__SCHED_FEAT_NR] = {
static void sched_feat_disable(int i)
{
- static_key_disable_cpuslocked(&sched_feat_keys[i]);
+ static_branch_disable_cpuslocked(&sched_feat_keys[i].key_true);
}
static void sched_feat_enable(int i)
{
- static_key_enable_cpuslocked(&sched_feat_keys[i]);
+ static_branch_enable_cpuslocked(&sched_feat_keys[i].key_false);
}
#else /* !CONFIG_JUMP_LABEL: */
static void sched_feat_disable(int i) { };
@@ -280,16 +280,10 @@ static ssize_t sched_dynamic_write(struct file *filp, const char __user *ubuf,
static int sched_dynamic_show(struct seq_file *m, void *v)
{
- int i = (IS_ENABLED(CONFIG_PREEMPT_RT) || IS_ENABLED(CONFIG_ARCH_HAS_PREEMPT_LAZY)) * 2;
int mode = READ_ONCE(preempt_dynamic_mode);
- int j;
- /* Count entries in NULL terminated preempt_modes */
- for (j = 0; preempt_modes[j]; j++)
- ;
- j -= !IS_ENABLED(CONFIG_ARCH_HAS_PREEMPT_LAZY);
-
- for (; i < j; i++) {
+ /* Stop at NULL terminator */
+ for (int i = 0; preempt_modes[i]; i++) {
if (mode == i)
seq_puts(m, "(");
seq_puts(m, preempt_modes[i]);
@@ -1446,6 +1440,7 @@ void proc_sched_show_task(struct task_struct *p, struct pid_namespace *ns,
P_SCHEDSTAT(nr_failed_migrations_running);
P_SCHEDSTAT(nr_failed_migrations_hot);
P_SCHEDSTAT(nr_forced_migrations);
+ P_SCHEDSTAT(nr_migrations_cpu_non_preferred);
P_SCHEDSTAT(nr_wakeups);
P_SCHEDSTAT(nr_wakeups_sync);
P_SCHEDSTAT(nr_wakeups_migrate);
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 5fe980da545d..522cf6e98e31 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -24,6 +24,11 @@
DEFINE_RAW_SPINLOCK(scx_sched_lock);
+bool scx_allow_proxy_exec(const struct task_struct *p)
+{
+ return true;
+}
+
/*
* NOTE: sched_ext is in the process of growing multiple scheduler support and
* scx_root usage is in a transitional state. Naked dereferences are safe if the
@@ -1087,6 +1092,10 @@ static void schedule_deferred_locked(struct rq *rq)
schedule_deferred(rq);
}
+void scx_proxy_reenqueue_retry(struct rq *rq, struct task_struct *next)
+{
+}
+
void schedule_dsq_reenq(struct scx_sched *sch, struct scx_dispatch_q *dsq,
u64 reenq_flags, struct rq *locked_rq)
{
@@ -3031,10 +3040,13 @@ has_tasks:
return verdict;
}
-static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first)
+static void set_next_task_scx(struct rq *rq, struct task_struct *p, enum snt_e type)
{
struct scx_sched *sch = scx_task_sched(p);
+ if (type == SNT_REPICK)
+ return;
+
if (p->scx.flags & SCX_TASK_QUEUED) {
/*
* Core-sched might decide to execute @p before it is
@@ -3092,6 +3104,10 @@ static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first)
}
}
+void scx_proxy_donor_start(struct rq *rq)
+{
+}
+
static enum scx_cpu_preempt_reason
preempt_reason_from_class(const struct sched_class *class)
{
diff --git a/kernel/sched/ext/ext.h b/kernel/sched/ext/ext.h
index 0b7fc46aee08..3cfbfeb1bf9d 100644
--- a/kernel/sched/ext/ext.h
+++ b/kernel/sched/ext/ext.h
@@ -20,6 +20,9 @@ void scx_rq_deactivate(struct rq *rq);
int scx_check_setscheduler(struct task_struct *p, int policy);
bool task_should_scx(int policy);
bool scx_allow_ttwu_queue(const struct task_struct *p);
+bool scx_allow_proxy_exec(const struct task_struct *p);
+void scx_proxy_donor_start(struct rq *rq);
+void scx_proxy_reenqueue_retry(struct rq *rq, struct task_struct *next);
void init_sched_ext_class(void);
static inline u32 scx_cpuperf_target(s32 cpu)
@@ -54,6 +57,10 @@ static inline void scx_rq_deactivate(struct rq *rq) {}
static inline int scx_check_setscheduler(struct task_struct *p, int policy) { return 0; }
static inline bool task_on_scx(const struct task_struct *p) { return false; }
static inline bool scx_allow_ttwu_queue(const struct task_struct *p) { return true; }
+static inline bool scx_allow_proxy_exec(const struct task_struct *p) { return true; }
+static inline void scx_proxy_donor_start(struct rq *rq) {}
+static inline void scx_proxy_reenqueue_retry(struct rq *rq,
+ struct task_struct *next) {}
static inline void init_sched_ext_class(void) {}
#endif /* CONFIG_SCHED_CLASS_EXT */
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 57360f5cdde4..5c98d8dfce5d 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -24,6 +24,7 @@
#include <linux/mmap_lock.h>
#include <linux/hugetlb_inline.h>
#include <linux/jiffies.h>
+#include <linux/math.h>
#include <linux/mm_api.h>
#include <linux/highmem.h>
#include <linux/hrtimer.h>
@@ -820,12 +821,6 @@ static u64 ineligible_vruntime(struct cfs_rq *cfs_rq)
if (curr && !curr->on_rq)
curr = NULL;
- /*
- * This is called from set_next_task_fair(.first=true) /
- * set_protect_slice() so curr had better be set and on_rq.
- */
- WARN_ON_ONCE(!curr);
-
if (weight) {
s64 runtime = cfs_rq->sum_w_vruntime;
@@ -1137,10 +1132,9 @@ static inline void set_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity
/* If there are shorter slices than se's one */
if (slice != se->slice) {
+ vprot = min_vruntime(vprot, se->vruntime + calc_delta_fair(slice, se));
if (sched_feat(PREEMPT_SHORT))
vprot = min_vruntime(vprot, ineligible_vruntime(cfs_rq));
- else
- vprot = min_vruntime(vprot, se->vruntime + calc_delta_fair(slice, se));
}
se->vprot = vprot;
@@ -1148,10 +1142,19 @@ static inline void set_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity
static inline void update_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity *se)
{
- u64 slice = cfs_rq_min_slice(cfs_rq);
u64 vruntime = min_vruntime(se->vruntime, avg_vruntime(cfs_rq));
+ u64 slice = normalized_sysctl_sched_base_slice;
+ u64 vprot;
- se->vprot = min_vruntime(se->vprot, vruntime + calc_delta_fair(slice, se));
+ if (sched_feat(RUN_TO_PARITY))
+ slice = cfs_rq_min_slice(cfs_rq);
+
+ vprot = min_vruntime(se->vprot, vruntime + calc_delta_fair(slice, se));
+
+ if (sched_feat(PREEMPT_SHORT) && slice != se->slice)
+ vprot = min_vruntime(vprot, ineligible_vruntime(cfs_rq));
+
+ se->vprot = vprot;
}
static inline bool protect_slice(struct sched_entity *se)
@@ -3713,7 +3716,7 @@ static void update_task_scan_period(struct task_struct *p,
p->mm->numa_next_scan = jiffies +
msecs_to_jiffies(p->numa_scan_period);
- return;
+ goto out;
}
/*
@@ -3757,7 +3760,10 @@ static void update_task_scan_period(struct task_struct *p,
p->numa_scan_period = clamp(p->numa_scan_period + diff,
task_scan_min(p), task_scan_max(p));
- memset(p->numa_faults_locality, 0, sizeof(p->numa_faults_locality));
+
+out:
+ memset(p->numa_faults_locality, 0,
+ sizeof(p->numa_faults_locality));
}
/*
@@ -8209,7 +8215,6 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
struct sched_entity *se = &p->se;
struct cfs_rq *cfs_rq = &rq->cfs;
unsigned long weight;
- bool curr;
if (task_is_throttled(p) && enqueue_throttled_task(p))
return;
@@ -8238,23 +8243,14 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
if (p->in_iowait)
cpufreq_update_util(rq, SCHED_CPUFREQ_IOWAIT);
- /*
- * XXX comment on the curr thing
- */
- curr = (cfs_rq->curr == se);
- if (curr)
- place_entity(cfs_rq, se, flags);
if (se->on_rq && se->sched_delayed)
requeue_delayed_entity(cfs_rq, se);
weight = enqueue_hierarchy(p, flags);
-
- if (!curr) {
- reweight_eevdf(cfs_rq, se, weight, false);
- place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED);
- __enqueue_entity(cfs_rq, se);
- }
+ reweight_eevdf(cfs_rq, se, weight, false);
+ place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED);
+ __enqueue_entity(cfs_rq, se);
if (!rq_h_nr_queued && rq->cfs.h_nr_queued)
dl_server_start(&rq->fair_server);
@@ -8674,8 +8670,8 @@ static int
sched_balance_find_dst_group_cpu(struct sched_group *group, struct task_struct *p, int this_cpu)
{
unsigned long load, min_load = ULONG_MAX;
- unsigned int min_exit_latency = UINT_MAX;
- u64 latest_idle_timestamp = 0;
+ u64 min_exit_latency = U64_MAX;
+ unsigned int nr_candidates = 0;
int least_loaded_cpu = this_cpu;
int shallowest_idle_cpu = -1;
int i;
@@ -8696,24 +8692,16 @@ sched_balance_find_dst_group_cpu(struct sched_group *group, struct task_struct *
if (available_idle_cpu(i)) {
struct cpuidle_state *idle = idle_get_state(rq);
- if (idle && idle->exit_latency < min_exit_latency) {
- /*
- * We give priority to a CPU whose idle state
- * has the smallest exit latency irrespective
- * of any idle timestamp.
- */
- min_exit_latency = idle->exit_latency;
- latest_idle_timestamp = rq->idle_stamp;
- shallowest_idle_cpu = i;
- } else if ((!idle || idle->exit_latency == min_exit_latency) &&
- rq->idle_stamp > latest_idle_timestamp) {
- /*
- * If equal or no active idle state, then
- * the most recently idled CPU might have
- * a warmer cache.
- */
- latest_idle_timestamp = rq->idle_stamp;
+ u64 exit_latency = idle ? idle->exit_latency : U64_MAX;
+
+ if (shallowest_idle_cpu == -1 || exit_latency < min_exit_latency) {
+ min_exit_latency = exit_latency;
shallowest_idle_cpu = i;
+ nr_candidates = 1;
+ } else if (exit_latency == min_exit_latency) {
+ nr_candidates++;
+ if (!reciprocal_scale(sched_rng(), nr_candidates))
+ shallowest_idle_cpu = i;
}
} else if (shallowest_idle_cpu == -1) {
load = cpu_load(cpu_rq(i));
@@ -8812,6 +8800,35 @@ static inline bool test_idle_cores(int cpu)
}
/*
+ * Redirect a CPU to a higher-priority available sibling in its SMT domain,
+ * subject to task affinity.
+ */
+static inline int select_idle_smt_cpu(struct task_struct *p, int cpu)
+{
+ struct sched_domain *sd;
+ int best = cpu;
+ int sibling;
+
+ if (!sched_smt_active())
+ return cpu;
+
+ sd = rcu_dereference_all(cpu_rq(cpu)->sd);
+ if (!sd || !(sd->flags & SD_SHARE_CPUCAPACITY) ||
+ !(sd->flags & SD_ASYM_PACKING))
+ return cpu;
+
+ for_each_cpu_and(sibling, sched_domain_span(sd), p->cpus_ptr) {
+ if (sibling == best || !choose_idle_cpu(sibling, p))
+ continue;
+
+ if (sched_asym_prefer(sibling, best))
+ best = sibling;
+ }
+
+ return best;
+}
+
+/*
* Scans the local SMT mask to see if the entire core is idle, and records this
* information in sd_balance_shared->has_idle_cores.
*
@@ -9195,7 +9212,7 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
if (choose_idle_cpu(target, p) &&
asym_fits_cpu(task_util, util_min, util_max, target))
- return target;
+ goto select_smt_priority;
/*
* If the previous CPU is cache affine and idle, don't be stupid:
@@ -9205,8 +9222,10 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
asym_fits_cpu(task_util, util_min, util_max, prev)) {
if (!static_branch_unlikely(&sched_cluster_active) ||
- cpus_share_resources(prev, target))
- return prev;
+ cpus_share_resources(prev, target)) {
+ target = prev;
+ goto select_smt_priority;
+ }
prev_aff = prev;
}
@@ -9224,7 +9243,8 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
prev == smp_processor_id() &&
this_rq()->nr_running <= 1 &&
asym_fits_cpu(task_util, util_min, util_max, prev)) {
- return prev;
+ target = prev;
+ goto select_smt_priority;
}
/* Check a recently used CPU as a potential idle candidate: */
@@ -9238,8 +9258,10 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
asym_fits_cpu(task_util, util_min, util_max, recent_used_cpu)) {
if (!static_branch_unlikely(&sched_cluster_active) ||
- cpus_share_resources(recent_used_cpu, target))
- return recent_used_cpu;
+ cpus_share_resources(recent_used_cpu, target)) {
+ target = recent_used_cpu;
+ goto select_smt_priority;
+ }
} else {
recent_used_cpu = -1;
@@ -9261,7 +9283,11 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
*/
if (sd) {
i = select_idle_capacity(p, sd, target);
- return ((unsigned)i < nr_cpumask_bits) ? i : target;
+ if ((unsigned int)i < nr_cpumask_bits) {
+ target = i;
+ goto select_smt_priority;
+ }
+ return target;
}
}
@@ -9274,14 +9300,18 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
if (!has_idle_core && cpus_share_cache(prev, target)) {
i = select_idle_smt(p, sd, prev);
- if ((unsigned int)i < nr_cpumask_bits)
- return i;
+ if ((unsigned int)i < nr_cpumask_bits) {
+ target = i;
+ goto select_smt_priority;
+ }
}
}
i = select_idle_cpu(p, sd, has_idle_core, target);
- if ((unsigned)i < nr_cpumask_bits)
- return i;
+ if ((unsigned int)i < nr_cpumask_bits) {
+ target = i;
+ goto select_smt_priority;
+ }
/*
* For cluster machines which have lower sharing cache like L2 or
@@ -9289,12 +9319,19 @@ static int select_idle_sibling(struct task_struct *p, int prev, int target)
* first. But prev_cpu or recent_used_cpu may also be a good candidate,
* use them if possible when no idle CPU found in select_idle_cpu().
*/
- if ((unsigned int)prev_aff < nr_cpumask_bits)
- return prev_aff;
- if ((unsigned int)recent_used_cpu < nr_cpumask_bits)
- return recent_used_cpu;
+ if ((unsigned int)prev_aff < nr_cpumask_bits) {
+ target = prev_aff;
+ goto select_smt_priority;
+ }
+ if ((unsigned int)recent_used_cpu < nr_cpumask_bits) {
+ target = recent_used_cpu;
+ goto select_smt_priority;
+ }
return target;
+
+select_smt_priority:
+ return select_idle_smt_cpu(p, target);
}
/**
@@ -9971,8 +10008,10 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
}
/* Slow path */
- if (unlikely(sd))
- return sched_balance_find_dst_cpu(sd, p, cpu, prev_cpu, sd_flag);
+ if (unlikely(sd)) {
+ new_cpu = sched_balance_find_dst_cpu(sd, p, cpu, prev_cpu, sd_flag);
+ return select_idle_smt_cpu(p, new_cpu);
+ }
/* Fast path */
if (wake_flags & WF_TTWU)
@@ -10072,8 +10111,14 @@ static inline bool set_preempt_buddy(struct cfs_rq *cfs_rq, struct sched_entity
static inline bool set_short_buddy(struct cfs_rq *cfs_rq, struct sched_entity *pse)
{
- if (cfs_rq->next && cfs_rq->next->slice < pse->slice)
- return false;
+ if (cfs_rq->next) {
+ if (cfs_rq->next->slice < pse->slice)
+ return false;
+
+ if (cfs_rq->next->slice == pse->slice &&
+ entity_before(cfs_rq->next, pse))
+ return false;
+ }
set_next_buddy(cfs_rq, pse);
return true;
@@ -11439,21 +11484,7 @@ next:
*/
static void attach_tasks(struct lb_env *env)
{
- struct list_head *tasks = &env->tasks;
- struct task_struct *p;
- struct rq_flags rf;
-
- rq_lock(env->dst_rq, &rf);
- update_rq_clock(env->dst_rq);
-
- while (!list_empty(tasks)) {
- p = list_first_entry(tasks, struct task_struct, se.group_node);
- list_del_init(&p->se.group_node);
-
- attach_task(env->dst_rq, p);
- }
-
- rq_unlock(env->dst_rq, &rf);
+ __attach_tasks(env->dst_rq, &env->tasks);
}
#ifdef CONFIG_NO_HZ_COMMON
@@ -13746,7 +13777,7 @@ static int sched_balance_rq(int this_cpu, struct rq *this_rq,
};
bool need_unlock = false;
- cpumask_and(cpus, sched_domain_span(sd), cpu_active_mask);
+ cpumask_and(cpus, sched_domain_span(sd), cpu_preferred_mask);
schedstat_inc(sd->lb_count[idle]);
@@ -14871,10 +14902,8 @@ static int sched_balance_newidle(struct rq *this_rq, struct rq_flags *rf)
*/
this_rq->idle_stamp = rq_clock(this_rq);
- /*
- * Do not pull tasks towards !active CPUs...
- */
- if (!cpu_active(this_cpu))
+ /* Do not pull tasks towards !preferred CPUs */
+ if (!cpu_preferred(this_cpu))
return 0;
/*
@@ -15514,14 +15543,18 @@ static void switched_to_fair(struct rq *rq, struct task_struct *p)
}
}
-static void set_next_task_fair(struct rq *rq, struct task_struct *p, bool first)
+static void set_next_task_fair(struct rq *rq, struct task_struct *p, enum snt_e type)
{
struct sched_entity *se = &p->se;
- bool throttled = false;
struct cfs_rq *cfs_rq = &rq->cfs;
unsigned long weight = NICE_0_LOAD;
+ bool first = type == SNT_PICK;
+ bool throttled = false;
bool on_rq = se->on_rq;
+ if (type == SNT_REPICK)
+ goto repick;
+
clear_buddies(cfs_rq, se);
if (on_rq)
@@ -15565,11 +15598,18 @@ static void set_next_task_fair(struct rq *rq, struct task_struct *p, bool first)
WARN_ON_ONCE(se->sched_delayed);
- if (hrtick_enabled_fair(rq))
- hrtick_start_fair(rq, p);
-
update_misfit_status(p, rq);
sched_fair_update_stop_tick(rq, p);
+
+repick:
+ /*
+ * A same-task repick skips put_prev_task_fair(), but
+ * pick_task_fair() refreshed the entity hrtick_start_fair() reads
+ * before selecting it again. rq->cfs.curr identifies that entity,
+ * including with group scheduling.
+ */
+ if (hrtick_enabled_fair(rq))
+ hrtick_start_fair(rq, p);
}
void init_cfs_rq(struct cfs_rq *cfs_rq)
diff --git a/kernel/sched/idle.c b/kernel/sched/idle.c
index eb73b65ce6c4..76f3c84ca684 100644
--- a/kernel/sched/idle.c
+++ b/kernel/sched/idle.c
@@ -487,8 +487,11 @@ static void put_prev_task_idle(struct rq *rq, struct task_struct *prev, struct t
update_rq_avg_idle(rq);
}
-static void set_next_task_idle(struct rq *rq, struct task_struct *next, bool first)
+static void set_next_task_idle(struct rq *rq, struct task_struct *next, enum snt_e type)
{
+ if (type == SNT_REPICK)
+ return;
+
update_idle_core(rq);
scx_update_idle(rq, true, true);
schedstat_inc(rq->sched_goidle);
diff --git a/kernel/sched/rt.c b/kernel/sched/rt.c
index 85303add726d..1535046a23ff 100644
--- a/kernel/sched/rt.c
+++ b/kernel/sched/rt.c
@@ -1654,11 +1654,14 @@ static void wakeup_preempt_rt(struct rq *rq, struct task_struct *p, int flags)
check_preempt_equal_prio(rq, p);
}
-static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, bool first)
+static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, enum snt_e type)
{
struct sched_rt_entity *rt_se = &p->rt;
struct rt_rq *rt_rq = &rq->rt;
+ if (type == SNT_REPICK)
+ return;
+
p->se.exec_start = rq_clock_task(rq);
if (on_rt_rq(&p->rt))
update_stats_wait_end_rt(rt_rq, rt_se);
@@ -1666,7 +1669,7 @@ static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, bool f
/* The running task is never eligible for pushing */
dequeue_pushable_task(rq, p);
- if (!first)
+ if (type != SNT_PICK)
return;
/*
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 4c25fbe84fb5..499a5649582e 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -1325,6 +1325,9 @@ struct rq {
#ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING
u64 prev_steal_time_rq;
#endif
+#ifdef CONFIG_PREFERRED_CPU
+ bool npc_push_work_pending;
+#endif
/* calc_load related fields */
unsigned long calc_load_update;
@@ -1370,7 +1373,6 @@ struct rq {
struct task_struct *core_pick;
struct sched_dl_entity *core_dl_server;
unsigned int core_enabled;
- unsigned int core_sched_seq;
struct rb_root core_tree;
/* shared state -- careful with sched_core_cpu_deactivate() */
@@ -2446,16 +2448,25 @@ extern __read_mostly unsigned int sysctl_sched_features;
#ifdef CONFIG_JUMP_LABEL
-#define SCHED_FEAT(name, enabled) \
-static __always_inline bool static_branch_##name(struct static_key *key) \
-{ \
- return static_key_##enabled(key); \
+union sched_feat_key {
+ struct static_key_true key_true;
+ struct static_key_false key_false;
+};
+
+#define sched_feat_branch_true(key) static_branch_likely(&(key)->key_true)
+#define sched_feat_branch_false(key) static_branch_unlikely(&(key)->key_false)
+
+#define SCHED_FEAT(name, enabled) \
+static __always_inline bool \
+static_branch_##name(union sched_feat_key *key) \
+{ \
+ return sched_feat_branch_##enabled(key); \
}
#include "features.h"
#undef SCHED_FEAT
-extern struct static_key sched_feat_keys[__SCHED_FEAT_NR];
+extern union sched_feat_key sched_feat_keys[__SCHED_FEAT_NR];
#define sched_feat(x) (static_branch_##x(&sched_feat_keys[__SCHED_FEAT_##x]))
#else /* !CONFIG_JUMP_LABEL: */
@@ -2507,6 +2518,12 @@ static inline bool task_is_blocked(struct task_struct *p)
return !!p->blocked_on;
}
+#ifdef CONFIG_SCHED_PROXY_EXEC
+void sched_proxy_block_task(struct rq *rq, struct task_struct *p);
+#else
+static inline void sched_proxy_block_task(struct rq *rq, struct task_struct *p) {}
+#endif
+
static inline int task_on_cpu(struct rq *rq, struct task_struct *p)
{
return p->on_cpu;
@@ -2526,11 +2543,17 @@ static inline int task_on_rq_migrating(struct task_struct *p)
#define WF_EXEC 0x02 /* Wakeup after exec; maps to SD_BALANCE_EXEC */
#define WF_FORK 0x04 /* Wakeup after fork; maps to SD_BALANCE_FORK */
#define WF_TTWU 0x08 /* Wakeup; maps to SD_BALANCE_WAKE */
-
-#define WF_SYNC 0x10 /* Waker goes to sleep after wakeup */
+/*
+ * Hint that the caller expects the waker to sleep soon.
+ * Scheduler classes may use it for placement or preemption.
+ * Callers must not rely on it to prevent migration,
+ * preserve CPU locality or make the wakee run next.
+ */
+#define WF_SYNC 0x10
#define WF_MIGRATED 0x20 /* Internal use, task got migrated */
#define WF_CURRENT_CPU 0x40 /* Prefer to move the wakee to the current CPU. */
#define WF_RQ_SELECTED 0x80 /* ->select_task_rq() was called */
+#define WF_TTWU_RQ 0x100 /* Wakeup completed through ttwu_runnable() */
static_assert(WF_EXEC == SD_BALANCE_EXEC);
static_assert(WF_FORK == SD_BALANCE_FORK);
@@ -2620,6 +2643,12 @@ struct affinity_context {
extern s64 update_curr_common(struct rq *rq);
+enum snt_e {
+ SNT_NORMAL, /* set_next_task() */
+ SNT_PICK, /* put_prev_set_next_task(): prev != next */
+ SNT_REPICK, /* put_prev_set_next_task(): prev == next */
+};
+
struct sched_class {
#ifdef CONFIG_UCLAMP_TASK
@@ -2677,7 +2706,7 @@ struct sched_class {
* __schedule: rq->lock
*/
void (*put_prev_task)(struct rq *rq, struct task_struct *p, struct task_struct *next);
- void (*set_next_task)(struct rq *rq, struct task_struct *p, bool first);
+ void (*set_next_task)(struct rq *rq, struct task_struct *p, enum snt_e type);
/*
* select_task_rq: p->pi_lock
@@ -2780,7 +2809,7 @@ static inline void put_prev_task(struct rq *rq, struct task_struct *prev)
static inline void set_next_task(struct rq *rq, struct task_struct *next)
{
- next->sched_class->set_next_task(rq, next, false);
+ next->sched_class->set_next_task(rq, next, SNT_NORMAL);
}
static inline void
@@ -2801,11 +2830,13 @@ static inline void put_prev_set_next_task(struct rq *rq,
__put_prev_set_next_dl_server(rq, prev, next);
- if (next == prev)
+ if (next == prev) {
+ next->sched_class->set_next_task(rq, next, SNT_REPICK);
return;
+ }
prev->sched_class->put_prev_task(rq, prev, next);
- next->sched_class->set_next_task(rq, next, true);
+ next->sched_class->set_next_task(rq, next, SNT_PICK);
}
/*
@@ -3138,6 +3169,25 @@ static inline void attach_one_task(struct rq *rq, struct task_struct *p)
attach_task(rq, p);
}
+/*
+ * __attach_tasks() - attaches a list of tasks (using se.group_node) to
+ * the new rq
+ */
+static inline void __attach_tasks(struct rq *rq, struct list_head *tasks)
+{
+ guard(rq_lock)(rq);
+ update_rq_clock(rq);
+
+ while (!list_empty(tasks)) {
+ struct task_struct *p;
+
+ p = list_first_entry(tasks, struct task_struct, se.group_node);
+ list_del_init(&p->se.group_node);
+
+ attach_task(rq, p);
+ }
+}
+
#ifdef CONFIG_PREEMPT_RT
# define SCHED_NR_MIGRATE_BREAK 8
#else
@@ -4252,4 +4302,10 @@ DEFINE_CLASS_IS_UNCONDITIONAL(sched_change)
#include "ext/ext.h"
+#ifdef CONFIG_PREFERRED_CPU
+void sched_push_current_non_preferred_cpu(struct rq *rq);
+#else /* !CONFIG_PREFERRED_CPU */
+static inline void sched_push_current_non_preferred_cpu(struct rq *rq) { }
+#endif
+
#endif /* _KERNEL_SCHED_SCHED_H */
diff --git a/kernel/sched/stop_task.c b/kernel/sched/stop_task.c
index c909ca0d8c87..1e0109ec36b3 100644
--- a/kernel/sched/stop_task.c
+++ b/kernel/sched/stop_task.c
@@ -27,8 +27,11 @@ wakeup_preempt_stop(struct rq *rq, struct task_struct *p, int flags)
/* we're never preempted */
}
-static void set_next_task_stop(struct rq *rq, struct task_struct *stop, bool first)
+static void set_next_task_stop(struct rq *rq, struct task_struct *stop, enum snt_e type)
{
+ if (type == SNT_REPICK)
+ return;
+
stop->se.exec_start = rq_clock_task(rq);
}
diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c
index 3dab0253976f..919d0fb00bd9 100644
--- a/kernel/sched/topology.c
+++ b/kernel/sched/topology.c
@@ -32,6 +32,20 @@ static int __init sched_debug_setup(char *str)
}
early_param("sched_verbose", sched_debug_setup);
+#ifdef CONFIG_SCHED_SMT
+static bool sched_smt_asym_packing __read_mostly;
+
+static int __init setup_sched_smt_asym_packing(char *str)
+{
+ if (strcmp(str, "on"))
+ return 0;
+
+ sched_smt_asym_packing = true;
+ return 1;
+}
+__setup("sched_smt_asym_packing=", setup_sched_smt_asym_packing);
+#endif
+
static inline bool sched_debug(void)
{
return sched_debug_verbose;
@@ -1954,6 +1968,10 @@ sd_init(struct sched_domain_topology_level *tl,
if (WARN_ONCE(sd_flags & ~TOPOLOGY_SD_FLAGS,
"wrong sd_flags in topology description\n"))
sd_flags &= TOPOLOGY_SD_FLAGS;
+#ifdef CONFIG_SCHED_SMT
+ if (sched_smt_asym_packing && (sd_flags & SD_SHARE_CPUCAPACITY))
+ sd_flags |= SD_ASYM_PACKING;
+#endif
sd_flags |= asym_cpu_capacity_classify(sd_span, cpu_map);
*sd = (struct sched_domain){
diff --git a/kernel/sched/wait.c b/kernel/sched/wait.c
index d033f600f48c..477e4bf9c01e 100644
--- a/kernel/sched/wait.c
+++ b/kernel/sched/wait.c
@@ -174,15 +174,11 @@ EXPORT_SYMBOL_GPL(__wake_up_locked_key);
* @mode: which threads
* @key: opaque value to be passed to wakeup targets
*
- * The sync wakeup differs that the waker knows that it will schedule
- * away soon, so while the target thread will be woken up, it will not
- * be migrated to another CPU - ie. the two threads are 'synchronized'
- * with each other. This can prevent needless bouncing between CPUs.
+ * Passes WF_SYNC to waitqueue wake functions. The default wake function
+ * forwards it to the scheduler; see WF_SYNC for the hint's semantics.
*
- * On UP it can prevent extra preemption.
- *
- * If this function wakes up a task, it executes a full memory barrier before
- * accessing the task state.
+ * If this function wakes up a task, it executes a full memory barrier
+ * before accessing the task state.
*/
void __wake_up_sync_key(struct wait_queue_head *wq_head, unsigned int mode,
void *key)
@@ -200,15 +196,7 @@ EXPORT_SYMBOL_GPL(__wake_up_sync_key);
* @mode: which threads
* @key: opaque value to be passed to wakeup targets
*
- * The sync wakeup differs in that the waker knows that it will schedule
- * away soon, so while the target thread will be woken up, it will not
- * be migrated to another CPU - ie. the two threads are 'synchronized'
- * with each other. This can prevent needless bouncing between CPUs.
- *
- * On UP it can prevent extra preemption.
- *
- * If this function wakes up a task, it executes a full memory barrier before
- * accessing the task state.
+ * Same as __wake_up_sync_key(), but called with @wq_head->lock held.
*/
void __wake_up_locked_sync_key(struct wait_queue_head *wq_head,
unsigned int mode, void *key)
diff --git a/kernel/smp.c b/kernel/smp.c
index b696bcc60c08..8f530a094c30 100644
--- a/kernel/smp.c
+++ b/kernel/smp.c
@@ -87,6 +87,7 @@ int smpcfd_dead_cpu(unsigned int cpu)
free_cpumask_var(cfd->cpumask);
free_cpumask_var(cfd->cpumask_ipi);
+ irq_work_run_cpu(cpu);
return 0;
}
diff --git a/kernel/smpboot.c b/kernel/smpboot.c
index 4503b60ce9bd..3f60e8c6dd30 100644
--- a/kernel/smpboot.c
+++ b/kernel/smpboot.c
@@ -103,6 +103,7 @@ static int smpboot_thread_fn(void *data)
{
struct smpboot_thread_data *td = data;
struct smp_hotplug_thread *ht = td->ht;
+ bool should_run;
while (1) {
set_current_state(TASK_INTERRUPTIBLE);
@@ -117,7 +118,8 @@ static int smpboot_thread_fn(void *data)
return 0;
}
- if (kthread_should_park()) {
+ should_run = td->status == HP_THREAD_ACTIVE && ht->thread_should_run(td->cpu);
+ if (kthread_should_park() && !should_run) {
__set_current_state(TASK_RUNNING);
preempt_enable();
if (ht->park && td->status == HP_THREAD_ACTIVE) {
@@ -151,7 +153,7 @@ static int smpboot_thread_fn(void *data)
continue;
}
- if (!ht->thread_should_run(td->cpu)) {
+ if (!should_run) {
preempt_enable_no_resched();
schedule();
} else {