diff options
| author | Mark Brown <broonie@kernel.org> | 2026-09-30 13:15:48 +0100 |
|---|---|---|
| committer | Mark Brown <broonie@kernel.org> | 2026-09-30 13:15:48 +0100 |
| commit | 24bf019cbe7e44d1e933480933b8410886bf4c60 (patch) | |
| tree | 68ba5a2536df9ec6d93b3f524b60b53011988d08 /kernel/sched | |
| parent | d366f5b1dbc9ab26c4575690274dd8f6412805b0 (diff) | |
| parent | 1aeb52f7869a680c042fc9ae806281e8f60469f7 (diff) | |
| download | linux-next-24bf019cbe7e44d1e933480933b8410886bf4c60.tar.gz linux-next-24bf019cbe7e44d1e933480933b8410886bf4c60.zip | |
Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git
# Conflicts:
# Documentation/scheduler/index.rst
# arch/arm64/configs/defconfig
Diffstat (limited to 'kernel/sched')
| -rw-r--r-- | kernel/sched/core.c | 444 | ||||
| -rw-r--r-- | kernel/sched/cputime.c | 4 | ||||
| -rw-r--r-- | kernel/sched/deadline.c | 9 | ||||
| -rw-r--r-- | kernel/sched/debug.c | 21 | ||||
| -rw-r--r-- | kernel/sched/ext/ext.c | 18 | ||||
| -rw-r--r-- | kernel/sched/ext/ext.h | 7 | ||||
| -rw-r--r-- | kernel/sched/fair.c | 131 | ||||
| -rw-r--r-- | kernel/sched/idle.c | 5 | ||||
| -rw-r--r-- | kernel/sched/rt.c | 7 | ||||
| -rw-r--r-- | kernel/sched/sched.h | 80 | ||||
| -rw-r--r-- | kernel/sched/stop_task.c | 5 | ||||
| -rw-r--r-- | kernel/sched/wait.c | 22 |
12 files changed, 416 insertions, 337 deletions
diff --git a/kernel/sched/core.c b/kernel/sched/core.c index 1fe40de6ebe3..84313c9c4ba9 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -805,7 +805,7 @@ struct rq *_task_rq_lock(struct task_struct *p, struct rq_flags *rf) /* Use CONFIG_PARAVIRT as this will avoid more #ifdef in arch code. */ #ifdef CONFIG_PARAVIRT -struct static_key paravirt_steal_rq_enabled; +DEFINE_STATIC_KEY_FALSE(paravirt_steal_rq_enabled); #endif static void update_rq_clock_task(struct rq *rq, s64 delta) @@ -844,7 +844,7 @@ static void update_rq_clock_task(struct rq *rq, s64 delta) } #endif #ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING - if (static_key_false((¶virt_steal_rq_enabled))) { + if (static_branch_unlikely(¶virt_steal_rq_enabled)) { u64 prev_steal; steal = prev_steal = paravirt_steal_clock(cpu_of(rq)); @@ -2252,7 +2252,8 @@ void deactivate_task(struct rq *rq, struct task_struct *p, int flags) dequeue_task(rq, p, flags); } -static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_state) +static bool dequeue_block_task(struct rq *rq, struct task_struct *p, + unsigned long task_state) { int flags = DEQUEUE_NOCLOCK; @@ -2273,9 +2274,15 @@ static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_ * * Where __schedule() and ttwu() have matching control dependencies. * - * After this, schedule() must not care about p->state any more. + * Once the caller invokes __block_task(), schedule() must not care about + * p->state any more. */ - if (dequeue_task(rq, p, DEQUEUE_SLEEP | flags)) + return dequeue_task(rq, p, DEQUEUE_SLEEP | flags); +} + +static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_state) +{ + if (dequeue_block_task(rq, p, task_state)) __block_task(rq, p); } @@ -2504,6 +2511,24 @@ static inline bool rq_has_pinned_tasks(struct rq *rq) return rq->nr_pinned; } +static inline bool task_can_migrate_to_preferred(struct task_struct *p, int cpu) +{ + /* No need to migrate from a preferred CPU */ + if (cpu_preferred(cpu)) + return false; + + /* Only FAIR tasks honor preferred CPU state */ + if (unlikely(p->sched_class != &fair_sched_class)) + return false; + + /* Ignore preferred state if task affinity is changing */ + if (unlikely(!cpumask_test_cpu(task_cpu(p), p->cpus_ptr))) + return false; + + return cpumask_intersects_and(p->cpus_ptr, cpu_preferred_mask, + task_cpu_possible_mask(p)); +} + /* * Per-CPU kthreads are allowed to run on !active && online CPUs, see * __set_cpus_allowed_ptr() and select_fallback_rq(). @@ -2519,8 +2544,12 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu) return cpu_online(cpu); /* Non kernel threads are not allowed during either online or offline. */ - if (!(p->flags & PF_KTHREAD)) + if (!(p->flags & PF_KTHREAD)) { + /* Try to use preferred CPU if task's affinity allows */ + if (task_can_migrate_to_preferred(p, cpu)) + return false; return cpu_active(cpu); + } /* KTHREAD_IS_PER_CPU is always allowed. */ if (kthread_is_per_cpu(p)) @@ -2530,7 +2559,11 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu) if (cpu_dying(cpu)) return false; - /* But are allowed during online. */ + /* Try to keep unbound kthreads on a preferred CPU if possible. */ + if (task_can_migrate_to_preferred(p, cpu)) + return false; + + /* Otherwise, they are allowed to run on online CPU. */ return cpu_online(cpu); } @@ -3773,6 +3806,7 @@ static inline void proxy_reset_donor(struct rq *rq) WARN_ON_ONCE(rq->donor == rq->curr); put_prev_set_next_task(rq, rq->donor, rq->curr); + rq->next_class = rq->curr->sched_class; rq_set_donor(rq, rq->curr); zap_balance_callbacks(rq); resched_curr(rq); @@ -3787,6 +3821,8 @@ static inline void proxy_reset_donor(struct rq *rq) */ static inline bool proxy_needs_return(struct rq *rq, struct task_struct *p) { + bool dequeued; + /* * Typically per __set_task_cpu(), task_cpu(p) == p->wake_cpu. * @@ -3809,12 +3845,23 @@ static inline bool proxy_needs_return(struct rq *rq, struct task_struct *p) /* If already current, don't need to return migrate */ if (task_current(rq, p)) return false; - - /* If we're return migrating the rq->donor, switch it out for idle */ - if (task_current_donor(rq, p)) - proxy_reset_donor(rq); } - block_task(rq, p, TASK_WAKING); + + dequeued = dequeue_block_task(rq, p, TASK_WAKING); + + /* + * Dequeue @p from its scheduling class before resetting rq->donor. + * In particular, sched_ext needs to end the donor's running session + * and clear SCX_TASK_QUEUED before put_prev_task_scx() is called by + * proxy_reset_donor(); otherwise it would reenqueue the blocked donor. + * + * Keep on_rq set until all donor references have been replaced. + */ + if (task_current_donor(rq, p)) + proxy_reset_donor(rq); + + if (dequeued) + __block_task(rq, p); return true; } #else /* !CONFIG_SCHED_PROXY_EXEC */ @@ -3905,7 +3952,7 @@ static int ttwu_runnable(struct task_struct *p, int wake_flags) * When on_rq && !on_cpu the task is preempted, see if * it should preempt the task that is current now. */ - wakeup_preempt(rq, p, wake_flags); + wakeup_preempt(rq, p, wake_flags | WF_TTWU_RQ); } ttwu_do_wakeup(p); return 1; @@ -5149,7 +5196,7 @@ static void do_balance_callbacks(struct rq *rq, struct balance_callback *head) lockdep_assert_rq_held(rq); while (head) { - func = (void (*)(struct rq *))head->func; + func = head->func; next = head->next; head->next = NULL; head = next; @@ -5789,6 +5836,9 @@ void sched_tick(void) unsigned long hw_pressure; u64 resched_latency; + if (!cpu_preferred(cpu)) + sched_push_current_non_preferred_cpu(rq); + if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE)) arch_scale_freq_tick(); @@ -6283,10 +6333,7 @@ pick_next_task(struct rq *rq, struct rq_flags *rf) * selection. In this case, do a core-wide selection. */ if (rq->core->core_pick_seq == rq->core->core_task_seq && - rq->core->core_pick_seq != rq->core_sched_seq && rq->core_pick) { - WRITE_ONCE(rq->core_sched_seq, rq->core->core_pick_seq); - next = rq->core_pick; rq->dl_server = rq->core_dl_server; rq->core_pick = NULL; @@ -6318,11 +6365,13 @@ restart: } /* - * core->core_task_seq, core->core_pick_seq, rq->core_sched_seq + * core->core_task_seq, core->core_pick_seq * * @task_seq guards the task state ({en,de}queues) * @pick_seq is the @task_seq we did a selection on - * @sched_seq is the @pick_seq we scheduled + * + * Once a core-wide selection is committed, a non-NULL core_pick denotes + * a pick which still needs to be consumed on this CPU. * * However, preemptions can cause multiple picks on the same task set. * 'Fix' this by also increasing @task_seq for every pick. @@ -6429,7 +6478,6 @@ restart: rq->core->core_pick_seq = rq->core->core_task_seq; next = rq->core_pick; - rq->core_sched_seq = rq->core->core_pick_seq; /* Something should have been selected for current CPU */ WARN_ON_ONCE(!next); @@ -6517,7 +6565,10 @@ static bool try_steal_cookie(int this, int that) return false; do { - if (p == src->core_pick || p == src->curr) + if (p == src->core_pick || p == src->curr || p == src->donor) + goto next; + + if (task_is_blocked(p)) goto next; if (!is_cpu_allowed(p, this)) @@ -6820,6 +6871,34 @@ static void proxy_deactivate(struct rq *rq, struct task_struct *donor) block_task(rq, donor, state); } +/* + * Remove a retained proxy donor before changing its scheduler ownership. + * The caller holds p->pi_lock, so p cannot wake and migrate if block_task() + * drops it from the runqueue. If DELAY_DEQUEUE keeps a blocked fair task + * queued, switching_from_fair() completes the dequeue in the immediately + * following sched_change_begin(). + */ +void sched_proxy_block_task(struct rq *rq, struct task_struct *p) +{ + unsigned long state = READ_ONCE(p->__state); + + lockdep_assert_held(&p->pi_lock); + lockdep_assert_rq_held(rq); + + if (!p->is_blocked || !task_on_rq_queued(p)) + return; + if (WARN_ON_ONCE(state == TASK_RUNNING)) + return; + + if (task_current_donor(rq, p)) + proxy_reset_donor(rq); + + if (!p->se.sched_delayed) + block_task(rq, p, state); + + WARN_ON_ONCE(task_on_rq_queued(p) && !p->se.sched_delayed); +} + static inline void proxy_release_rq_lock(struct rq *rq, struct rq_flags *rf) __releases(__rq_lockp(rq)) { @@ -6865,9 +6944,9 @@ static void proxy_migrate_task(struct rq *rq, struct rq_flags *rf, __must_hold(__rq_lockp(rq)) { struct rq *target_rq = cpu_rq(target_cpu); + LIST_HEAD(migrate_list); lockdep_assert_rq_held(rq); - WARN_ON(p == rq->curr); /* * Since we are migrating a blocked donor, it could be rq->donor, * and we want to make sure there aren't any references from this @@ -6880,13 +6959,20 @@ static void proxy_migrate_task(struct rq *rq, struct rq_flags *rf, * before we release the lock. */ proxy_resched_idle(rq); - - deactivate_task(rq, p, DEQUEUE_NOCLOCK); - proxy_set_task_cpu(p, target_cpu); - + for (; p; p = p->blocked_donor) { + WARN_ON(p == rq->curr); + deactivate_task(rq, p, DEQUEUE_NOCLOCK); + proxy_set_task_cpu(p, target_cpu); + /* + * We can re-use se.group_node to migrate the thing, + * because @p is deactivated (won't be balanced) and + * we hold the rq_lock. + */ + list_add(&p->se.group_node, &migrate_list); + } proxy_release_rq_lock(rq, rf); - attach_one_task(target_rq, p); + __attach_tasks(target_rq, &migrate_list); proxy_reacquire_rq_lock(rq, rf); } @@ -6979,7 +7065,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) if (!READ_ONCE(owner->on_rq) || owner->se.sched_delayed) { /* XXX Don't handle blocked owners/delayed dequeue yet */ if (curr_in_chain) - return proxy_resched_idle(rq); + goto resched_idle; __clear_task_blocked_on(p, NULL); goto deactivate; } @@ -6991,7 +7077,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) * and leave that CPU to sort things out. */ if (curr_in_chain) - return proxy_resched_idle(rq); + goto resched_idle; goto migrate_task; } @@ -7004,7 +7090,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) * case we should end up back in find_proxy_task(), this time * hopefully with all relevant tasks already enqueued. */ - return proxy_resched_idle(rq); + goto resched_idle; } /* @@ -7041,7 +7127,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) * So schedule rq->idle so that ttwu_runnable() can get the rq * lock and mark owner as running. */ - return proxy_resched_idle(rq); + goto resched_idle; } /* * OK, now we're absolutely sure @owner is on this @@ -7051,8 +7137,18 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) owner->blocked_donor = p; } WARN_ON_ONCE(owner && !owner->on_rq); + + if (owner && !sched_cpu_cookie_match(rq, owner)) { + if (curr_in_chain) + return proxy_resched_idle(rq); + p = donor; /* Deactivate the donor, not the runnable owner */ + clear_task_blocked_on(p, NULL); + goto deactivate; + } return owner; +resched_idle: + return proxy_resched_idle(rq); deactivate: proxy_deactivate(rq, p); return NULL; @@ -7184,13 +7280,12 @@ static void __sched notrace __schedule(int sched_mode) } } else if (!preempt && prev_state) { /* - * We pass task_is_blocked() as the should_block arg - * in order to keep mutex-blocked tasks on the runqueue - * for slection with proxy-exec (without proxy-exec - * task_is_blocked() will always be false). + * Keep mutex-blocked tasks on the runqueue for proxy execution + * only when their scheduling class allows it. Without proxy + * execution, task_is_blocked() always returns false. */ try_to_block_task(rq, prev, &prev_state, - !task_is_blocked(prev)); + !task_is_blocked(prev) || !scx_allow_proxy_exec(prev)); switch_count = &prev->nvcsw; } @@ -7211,6 +7306,7 @@ pick_again: } if (next == rq->idle) { zap_balance_callbacks(rq); + scx_proxy_reenqueue_retry(rq, next); goto keep_resched; } } @@ -7229,8 +7325,10 @@ pick_again: * on_cpu. */ donor->sched_class->put_prev_task(rq, donor, donor); - donor->sched_class->set_next_task(rq, donor, true); + donor->sched_class->set_next_task(rq, donor, SNT_PICK); } + scx_proxy_donor_start(rq); + scx_proxy_reenqueue_retry(rq, next); } else { rq_set_donor(rq, next); } @@ -7489,27 +7587,6 @@ asmlinkage __visible void __sched notrace preempt_schedule(void) NOKPROBE_SYMBOL(preempt_schedule); EXPORT_SYMBOL(preempt_schedule); -#ifdef CONFIG_PREEMPT_DYNAMIC -# ifdef CONFIG_HAVE_PREEMPT_DYNAMIC_CALL -# ifndef preempt_schedule_dynamic_enabled -# define preempt_schedule_dynamic_enabled preempt_schedule -# define preempt_schedule_dynamic_disabled NULL -# endif -DEFINE_STATIC_CALL(preempt_schedule, preempt_schedule_dynamic_enabled); -EXPORT_STATIC_CALL_TRAMP(preempt_schedule); -# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule); -void __sched notrace dynamic_preempt_schedule(void) -{ - if (!static_branch_unlikely(&sk_dynamic_preempt_schedule)) - return; - preempt_schedule(); -} -NOKPROBE_SYMBOL(dynamic_preempt_schedule); -EXPORT_SYMBOL(dynamic_preempt_schedule); -# endif -#endif /* CONFIG_PREEMPT_DYNAMIC */ - /** * preempt_schedule_notrace - preempt_schedule called by tracing * @@ -7562,27 +7639,6 @@ asmlinkage __visible void __sched notrace preempt_schedule_notrace(void) } EXPORT_SYMBOL_GPL(preempt_schedule_notrace); -#ifdef CONFIG_PREEMPT_DYNAMIC -# if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) -# ifndef preempt_schedule_notrace_dynamic_enabled -# define preempt_schedule_notrace_dynamic_enabled preempt_schedule_notrace -# define preempt_schedule_notrace_dynamic_disabled NULL -# endif -DEFINE_STATIC_CALL(preempt_schedule_notrace, preempt_schedule_notrace_dynamic_enabled); -EXPORT_STATIC_CALL_TRAMP(preempt_schedule_notrace); -# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule_notrace); -void __sched notrace dynamic_preempt_schedule_notrace(void) -{ - if (!static_branch_unlikely(&sk_dynamic_preempt_schedule_notrace)) - return; - preempt_schedule_notrace(); -} -NOKPROBE_SYMBOL(dynamic_preempt_schedule_notrace); -EXPORT_SYMBOL(dynamic_preempt_schedule_notrace); -# endif -#endif - #endif /* CONFIG_PREEMPTION */ /* @@ -7799,7 +7855,7 @@ out_unlock: } #endif /* CONFIG_RT_MUTEXES */ -#if !defined(CONFIG_PREEMPTION) || defined(CONFIG_PREEMPT_DYNAMIC) +#if !defined(CONFIG_PREEMPTION) int __sched __cond_resched(void) { if (should_resched(0) && !irqs_disabled()) { @@ -7827,38 +7883,6 @@ int __sched __cond_resched(void) EXPORT_SYMBOL(__cond_resched); #endif -#ifdef CONFIG_PREEMPT_DYNAMIC -# ifdef CONFIG_HAVE_PREEMPT_DYNAMIC_CALL -# define cond_resched_dynamic_enabled __cond_resched -# define cond_resched_dynamic_disabled ((void *)&__static_call_return0) -DEFINE_STATIC_CALL_RET0(cond_resched, __cond_resched); -EXPORT_STATIC_CALL_TRAMP(cond_resched); - -# define might_resched_dynamic_enabled __cond_resched -# define might_resched_dynamic_disabled ((void *)&__static_call_return0) -DEFINE_STATIC_CALL_RET0(might_resched, __cond_resched); -EXPORT_STATIC_CALL_TRAMP(might_resched); -# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -static DEFINE_STATIC_KEY_FALSE(sk_dynamic_cond_resched); -int __sched dynamic_cond_resched(void) -{ - if (!static_branch_unlikely(&sk_dynamic_cond_resched)) - return 0; - return __cond_resched(); -} -EXPORT_SYMBOL(dynamic_cond_resched); - -static DEFINE_STATIC_KEY_FALSE(sk_dynamic_might_resched); -int __sched dynamic_might_resched(void) -{ - if (!static_branch_unlikely(&sk_dynamic_might_resched)) - return 0; - return __cond_resched(); -} -EXPORT_SYMBOL(dynamic_might_resched); -# endif -#endif /* CONFIG_PREEMPT_DYNAMIC */ - /* * __cond_resched_lock() - if a reschedule is pending, drop the given lock, * call schedule, and on return reacquire the lock. @@ -7928,50 +7952,21 @@ EXPORT_SYMBOL(__cond_resched_rwlock_write); # endif /* - * SC:cond_resched - * SC:might_resched - * SC:preempt_schedule - * SC:preempt_schedule_notrace - * SC:irqentry_exit_cond_resched - * - * * NONE: - * cond_resched <- __cond_resched - * might_resched <- RET0 - * preempt_schedule <- NOP - * preempt_schedule_notrace <- NOP - * irqentry_exit_cond_resched <- NOP - * dynamic_preempt_lazy <- false + * (unselectable) * * VOLUNTARY: - * cond_resched <- __cond_resched - * might_resched <- __cond_resched - * preempt_schedule <- NOP - * preempt_schedule_notrace <- NOP - * irqentry_exit_cond_resched <- NOP - * dynamic_preempt_lazy <- false + * (unselectable) * * FULL: - * cond_resched <- RET0 - * might_resched <- RET0 - * preempt_schedule <- preempt_schedule - * preempt_schedule_notrace <- preempt_schedule_notrace - * irqentry_exit_cond_resched <- irqentry_exit_cond_resched * dynamic_preempt_lazy <- false * * LAZY: - * cond_resched <- RET0 - * might_resched <- RET0 - * preempt_schedule <- preempt_schedule - * preempt_schedule_notrace <- preempt_schedule_notrace - * irqentry_exit_cond_resched <- irqentry_exit_cond_resched * dynamic_preempt_lazy <- true */ enum { preempt_dynamic_undefined = -1, - preempt_dynamic_none, - preempt_dynamic_voluntary, preempt_dynamic_full, preempt_dynamic_lazy, }; @@ -7980,21 +7975,11 @@ int preempt_dynamic_mode = preempt_dynamic_undefined; int sched_dynamic_mode(const char *str) { -# if !(defined(CONFIG_PREEMPT_RT) || defined(CONFIG_ARCH_HAS_PREEMPT_LAZY)) - if (!strcmp(str, "none")) - return preempt_dynamic_none; - - if (!strcmp(str, "voluntary")) - return preempt_dynamic_voluntary; -# endif - if (!strcmp(str, "full")) return preempt_dynamic_full; -# ifdef CONFIG_ARCH_HAS_PREEMPT_LAZY if (!strcmp(str, "lazy")) return preempt_dynamic_lazy; -# endif return -EINVAL; } @@ -8002,71 +7987,18 @@ int sched_dynamic_mode(const char *str) # define preempt_dynamic_key_enable(f) static_key_enable(&sk_dynamic_##f.key) # define preempt_dynamic_key_disable(f) static_key_disable(&sk_dynamic_##f.key) -# if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) -# define preempt_dynamic_enable(f) static_call_update(f, f##_dynamic_enabled) -# define preempt_dynamic_disable(f) static_call_update(f, f##_dynamic_disabled) -# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -# define preempt_dynamic_enable(f) preempt_dynamic_key_enable(f) -# define preempt_dynamic_disable(f) preempt_dynamic_key_disable(f) -# else -# error "Unsupported PREEMPT_DYNAMIC mechanism" -# endif - static DEFINE_MUTEX(sched_dynamic_mutex); static void __sched_dynamic_update(int mode) { - /* - * Avoid {NONE,VOLUNTARY} -> FULL transitions from ever ending up in - * the ZERO state, which is invalid. - */ - preempt_dynamic_enable(cond_resched); - preempt_dynamic_enable(might_resched); - preempt_dynamic_enable(preempt_schedule); - preempt_dynamic_enable(preempt_schedule_notrace); - preempt_dynamic_enable(irqentry_exit_cond_resched); - preempt_dynamic_key_disable(preempt_lazy); - switch (mode) { - case preempt_dynamic_none: - preempt_dynamic_enable(cond_resched); - preempt_dynamic_disable(might_resched); - preempt_dynamic_disable(preempt_schedule); - preempt_dynamic_disable(preempt_schedule_notrace); - preempt_dynamic_disable(irqentry_exit_cond_resched); - preempt_dynamic_key_disable(preempt_lazy); - if (mode != preempt_dynamic_mode) - pr_info("Dynamic Preempt: none\n"); - break; - - case preempt_dynamic_voluntary: - preempt_dynamic_enable(cond_resched); - preempt_dynamic_enable(might_resched); - preempt_dynamic_disable(preempt_schedule); - preempt_dynamic_disable(preempt_schedule_notrace); - preempt_dynamic_disable(irqentry_exit_cond_resched); - preempt_dynamic_key_disable(preempt_lazy); - if (mode != preempt_dynamic_mode) - pr_info("Dynamic Preempt: voluntary\n"); - break; - case preempt_dynamic_full: - preempt_dynamic_disable(cond_resched); - preempt_dynamic_disable(might_resched); - preempt_dynamic_enable(preempt_schedule); - preempt_dynamic_enable(preempt_schedule_notrace); - preempt_dynamic_enable(irqentry_exit_cond_resched); preempt_dynamic_key_disable(preempt_lazy); if (mode != preempt_dynamic_mode) pr_info("Dynamic Preempt: full\n"); break; case preempt_dynamic_lazy: - preempt_dynamic_disable(cond_resched); - preempt_dynamic_disable(might_resched); - preempt_dynamic_enable(preempt_schedule); - preempt_dynamic_enable(preempt_schedule_notrace); - preempt_dynamic_enable(irqentry_exit_cond_resched); preempt_dynamic_key_enable(preempt_lazy); if (mode != preempt_dynamic_mode) pr_info("Dynamic Preempt: lazy\n"); @@ -8099,11 +8031,7 @@ __setup("preempt=", setup_preempt_mode); static void __init preempt_dynamic_init(void) { if (preempt_dynamic_mode == preempt_dynamic_undefined) { - if (IS_ENABLED(CONFIG_PREEMPT_NONE)) { - sched_dynamic_update(preempt_dynamic_none); - } else if (IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY)) { - sched_dynamic_update(preempt_dynamic_voluntary); - } else if (IS_ENABLED(CONFIG_PREEMPT_LAZY)) { + if (IS_ENABLED(CONFIG_PREEMPT_LAZY)) { sched_dynamic_update(preempt_dynamic_lazy); } else { /* Default static call setting, nothing to do */ @@ -8123,8 +8051,6 @@ static void __init preempt_dynamic_init(void) } \ EXPORT_SYMBOL_GPL(preempt_model_##mode) -PREEMPT_MODEL_ACCESSOR(none); -PREEMPT_MODEL_ACCESSOR(voluntary); PREEMPT_MODEL_ACCESSOR(full); PREEMPT_MODEL_ACCESSOR(lazy); @@ -8137,7 +8063,7 @@ static inline void preempt_dynamic_init(void) { } #endif /* CONFIG_PREEMPT_DYNAMIC */ const char *preempt_modes[] = { - "none", "voluntary", "full", "lazy", NULL, + "full", "lazy", NULL, }; const char *preempt_model_str(void) @@ -8759,6 +8685,9 @@ int sched_cpu_activate(unsigned int cpu) */ sched_set_rq_online(rq, cpu); + /* preferred is subset of active and follows its state */ + set_cpu_preferred(cpu, true); + return 0; } @@ -8772,6 +8701,8 @@ int sched_cpu_deactivate(unsigned int cpu) if (ret) return ret; + set_cpu_preferred(cpu, false); + /* * Remove CPU from nohz.idle_cpus_mask to prevent participating in * load balancing when not active @@ -11349,3 +11280,88 @@ void sched_change_end(struct sched_change_ctx *ctx) p->sched_class->prio_changed(rq, p, ctx->prio); } } + +#ifdef CONFIG_PREFERRED_CPU +static DEFINE_PER_CPU(struct cpu_stop_work, npc_push_task_work); + +static int sched_non_preferred_cpu_push_stop(void *arg) +{ + struct task_struct *p = arg; + struct rq *rq = this_rq(); + struct rq_flags rf; + int cpu; + + if (cpu_preferred(rq->cpu)) { + scoped_guard(rq_lock_irqsave, rq) + rq->npc_push_work_pending = false; + put_task_struct(p); + return 0; + } + + scoped_guard (raw_spinlock_irq, &p->pi_lock) { + /* + * select_fallback_rq() may acquire the rq lock in case of + * fallback. So call it before grabbing rq lock. If the task + * migrates to another CPU before the rq lock is acquired, + * subsequent validation of task's current rq will help to + * safely bail out. + */ + cpu = select_fallback_rq(rq->cpu, p); + rq_lock(rq, &rf); + rq->npc_push_work_pending = false; + update_rq_clock(rq); + context_unsafe_alias(rq); + + if (task_rq(p) == rq && task_on_rq_queued(p)) { + struct rq *dest_rq = __migrate_task(rq, &rf, p, cpu); + + if (rq != dest_rq) + schedstat_inc(p->stats.nr_migrations_cpu_non_preferred); + rq = dest_rq; + } + rq_unlock(rq, &rf); + } + + put_task_struct(p); + return 0; +} + +/* + * Push the current task running on non-preferred CPU(npc). + * Using this non preferred CPU will lead to more contention + * in the host. So it is better not to use this CPU. + * + * Since task is running, call a stopper to push the task out. This is + * similar to how task moves during hotplug. In select_fallback_rq() a + * preferred CPU will be chosen and henceforth task shouldn't come back to + * this CPU again. + * + * Works for FAIR class only. + * + * If task is affined only on non-preferred CPUs, no point in moving it out. + */ +void sched_push_current_non_preferred_cpu(struct rq *rq) +{ + struct task_struct *push_task = rq->curr; + + scoped_guard(rq_lock, rq) { + /* Push the task if its explicit affinity allows */ + if (!task_can_migrate_to_preferred(push_task, rq->cpu)) + return; + + /* There is already a stopper thread. Don't race with it. */ + if (rq->npc_push_work_pending) + return; + + if (is_migration_disabled(push_task)) + return; + + rq->npc_push_work_pending = true; + } + + /* sched_tick runs with interrupts disabled. */ + get_task_struct(push_task); + stop_one_cpu_nowait(rq->cpu, sched_non_preferred_cpu_push_stop, + push_task, this_cpu_ptr(&npc_push_task_work)); +} +#endif diff --git a/kernel/sched/cputime.c b/kernel/sched/cputime.c index 06bddaa738e5..f16970ca81d0 100644 --- a/kernel/sched/cputime.c +++ b/kernel/sched/cputime.c @@ -255,7 +255,7 @@ void __account_forceidle_time(struct task_struct *p, u64 delta) * occasion account more time than the calling functions think elapsed. */ #ifdef CONFIG_PARAVIRT -struct static_key paravirt_steal_enabled; +DEFINE_STATIC_KEY_FALSE(paravirt_steal_enabled); #ifdef CONFIG_HAVE_PV_STEAL_CLOCK_GEN static u64 native_steal_clock(int cpu) @@ -270,7 +270,7 @@ DEFINE_STATIC_CALL(pv_steal_clock, native_steal_clock); static __always_inline u64 steal_account_process_time(u64 maxtime) { #ifdef CONFIG_PARAVIRT - if (static_key_false(¶virt_steal_enabled)) { + if (static_branch_unlikely(¶virt_steal_enabled)) { u64 steal; steal = paravirt_steal_clock(smp_processor_id()); diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c index 0663c00c41c0..c0ebdcde5fe5 100644 --- a/kernel/sched/deadline.c +++ b/kernel/sched/deadline.c @@ -1097,7 +1097,7 @@ static int start_dl_timer(struct sched_dl_entity *dl_se) * chosen as the deadline is too small, don't even try to * start the timer in the past! */ - if (ktime_us_delta(act, now) < 0) + if (ktime_before(act, now)) return 0; /* @@ -2773,11 +2773,14 @@ static void start_hrtick_dl(struct rq *rq, struct sched_dl_entity *dl_se) * DL keeps current in tree, because ->deadline is not typically changed while * a task is runnable. */ -static void set_next_task_dl(struct rq *rq, struct task_struct *p, bool first) +static void set_next_task_dl(struct rq *rq, struct task_struct *p, enum snt_e type) { struct sched_dl_entity *dl_se = &p->dl; struct dl_rq *dl_rq = &rq->dl; + if (type == SNT_REPICK) + return; + p->se.exec_start = rq_clock_task(rq); if (on_dl_rq(&p->dl)) update_stats_wait_end_dl(dl_rq, dl_se); @@ -2788,7 +2791,7 @@ static void set_next_task_dl(struct rq *rq, struct task_struct *p, bool first) WARN_ON_ONCE(dl_rq->curr); dl_rq->curr = dl_se; - if (!first) + if (type != SNT_PICK) return; if (rq->donor->sched_class != &dl_sched_class) diff --git a/kernel/sched/debug.c b/kernel/sched/debug.c index 72236db67983..e6a3b516c703 100644 --- a/kernel/sched/debug.c +++ b/kernel/sched/debug.c @@ -73,13 +73,13 @@ static int sched_feat_show(struct seq_file *m, void *v) #ifdef CONFIG_JUMP_LABEL -#define jump_label_key__true STATIC_KEY_INIT_TRUE -#define jump_label_key__false STATIC_KEY_INIT_FALSE +#define jump_label_key__true { .key_true = STATIC_KEY_TRUE_INIT } +#define jump_label_key__false { .key_false = STATIC_KEY_FALSE_INIT } #define SCHED_FEAT(name, enabled) \ jump_label_key__##enabled , -struct static_key sched_feat_keys[__SCHED_FEAT_NR] = { +union sched_feat_key sched_feat_keys[__SCHED_FEAT_NR] = { #include "features.h" }; @@ -87,12 +87,12 @@ struct static_key sched_feat_keys[__SCHED_FEAT_NR] = { static void sched_feat_disable(int i) { - static_key_disable_cpuslocked(&sched_feat_keys[i]); + static_branch_disable_cpuslocked(&sched_feat_keys[i].key_true); } static void sched_feat_enable(int i) { - static_key_enable_cpuslocked(&sched_feat_keys[i]); + static_branch_enable_cpuslocked(&sched_feat_keys[i].key_false); } #else /* !CONFIG_JUMP_LABEL: */ static void sched_feat_disable(int i) { }; @@ -280,16 +280,10 @@ static ssize_t sched_dynamic_write(struct file *filp, const char __user *ubuf, static int sched_dynamic_show(struct seq_file *m, void *v) { - int i = (IS_ENABLED(CONFIG_PREEMPT_RT) || IS_ENABLED(CONFIG_ARCH_HAS_PREEMPT_LAZY)) * 2; int mode = READ_ONCE(preempt_dynamic_mode); - int j; - /* Count entries in NULL terminated preempt_modes */ - for (j = 0; preempt_modes[j]; j++) - ; - j -= !IS_ENABLED(CONFIG_ARCH_HAS_PREEMPT_LAZY); - - for (; i < j; i++) { + /* Stop at NULL terminator */ + for (int i = 0; preempt_modes[i]; i++) { if (mode == i) seq_puts(m, "("); seq_puts(m, preempt_modes[i]); @@ -1446,6 +1440,7 @@ void proc_sched_show_task(struct task_struct *p, struct pid_namespace *ns, P_SCHEDSTAT(nr_failed_migrations_running); P_SCHEDSTAT(nr_failed_migrations_hot); P_SCHEDSTAT(nr_forced_migrations); + P_SCHEDSTAT(nr_migrations_cpu_non_preferred); P_SCHEDSTAT(nr_wakeups); P_SCHEDSTAT(nr_wakeups_sync); P_SCHEDSTAT(nr_wakeups_migrate); diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c index e56c3c95018f..aed5286b82aa 100644 --- a/kernel/sched/ext/ext.c +++ b/kernel/sched/ext/ext.c @@ -24,6 +24,11 @@ DEFINE_RAW_SPINLOCK(scx_sched_lock); +bool scx_allow_proxy_exec(const struct task_struct *p) +{ + return true; +} + /* * NOTE: sched_ext is in the process of growing multiple scheduler support and * scx_root usage is in a transitional state. Naked dereferences are safe if the @@ -1087,6 +1092,10 @@ static void schedule_deferred_locked(struct rq *rq) schedule_deferred(rq); } +void scx_proxy_reenqueue_retry(struct rq *rq, struct task_struct *next) +{ +} + void schedule_dsq_reenq(struct scx_sched *sch, struct scx_dispatch_q *dsq, u64 reenq_flags, struct rq *locked_rq) { @@ -3021,10 +3030,13 @@ has_tasks: return verdict; } -static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first) +static void set_next_task_scx(struct rq *rq, struct task_struct *p, enum snt_e type) { struct scx_sched *sch = scx_task_sched(p); + if (type == SNT_REPICK) + return; + if (p->scx.flags & SCX_TASK_QUEUED) { /* * Core-sched might decide to execute @p before it is @@ -3082,6 +3094,10 @@ static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first) } } +void scx_proxy_donor_start(struct rq *rq) +{ +} + static enum scx_cpu_preempt_reason preempt_reason_from_class(const struct sched_class *class) { diff --git a/kernel/sched/ext/ext.h b/kernel/sched/ext/ext.h index 0b7fc46aee08..3cfbfeb1bf9d 100644 --- a/kernel/sched/ext/ext.h +++ b/kernel/sched/ext/ext.h @@ -20,6 +20,9 @@ void scx_rq_deactivate(struct rq *rq); int scx_check_setscheduler(struct task_struct *p, int policy); bool task_should_scx(int policy); bool scx_allow_ttwu_queue(const struct task_struct *p); +bool scx_allow_proxy_exec(const struct task_struct *p); +void scx_proxy_donor_start(struct rq *rq); +void scx_proxy_reenqueue_retry(struct rq *rq, struct task_struct *next); void init_sched_ext_class(void); static inline u32 scx_cpuperf_target(s32 cpu) @@ -54,6 +57,10 @@ static inline void scx_rq_deactivate(struct rq *rq) {} static inline int scx_check_setscheduler(struct task_struct *p, int policy) { return 0; } static inline bool task_on_scx(const struct task_struct *p) { return false; } static inline bool scx_allow_ttwu_queue(const struct task_struct *p) { return true; } +static inline bool scx_allow_proxy_exec(const struct task_struct *p) { return true; } +static inline void scx_proxy_donor_start(struct rq *rq) {} +static inline void scx_proxy_reenqueue_retry(struct rq *rq, + struct task_struct *next) {} static inline void init_sched_ext_class(void) {} #endif /* CONFIG_SCHED_CLASS_EXT */ diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 8d38c3b7d792..56f4ab6d9ada 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -23,6 +23,7 @@ #include <linux/energy_model.h> #include <linux/mmap_lock.h> #include <linux/jiffies.h> +#include <linux/math.h> #include <linux/mm_api.h> #include <linux/highmem.h> #include <linux/hrtimer.h> @@ -819,12 +820,6 @@ static u64 ineligible_vruntime(struct cfs_rq *cfs_rq) if (curr && !curr->on_rq) curr = NULL; - /* - * This is called from set_next_task_fair(.first=true) / - * set_protect_slice() so curr had better be set and on_rq. - */ - WARN_ON_ONCE(!curr); - if (weight) { s64 runtime = cfs_rq->sum_w_vruntime; @@ -1136,10 +1131,9 @@ static inline void set_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity /* If there are shorter slices than se's one */ if (slice != se->slice) { + vprot = min_vruntime(vprot, se->vruntime + calc_delta_fair(slice, se)); if (sched_feat(PREEMPT_SHORT)) vprot = min_vruntime(vprot, ineligible_vruntime(cfs_rq)); - else - vprot = min_vruntime(vprot, se->vruntime + calc_delta_fair(slice, se)); } se->vprot = vprot; @@ -1147,10 +1141,19 @@ static inline void set_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity static inline void update_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity *se) { - u64 slice = cfs_rq_min_slice(cfs_rq); u64 vruntime = min_vruntime(se->vruntime, avg_vruntime(cfs_rq)); + u64 slice = normalized_sysctl_sched_base_slice; + u64 vprot; - se->vprot = min_vruntime(se->vprot, vruntime + calc_delta_fair(slice, se)); + if (sched_feat(RUN_TO_PARITY)) + slice = cfs_rq_min_slice(cfs_rq); + + vprot = min_vruntime(se->vprot, vruntime + calc_delta_fair(slice, se)); + + if (sched_feat(PREEMPT_SHORT) && slice != se->slice) + vprot = min_vruntime(vprot, ineligible_vruntime(cfs_rq)); + + se->vprot = vprot; } static inline bool protect_slice(struct sched_entity *se) @@ -3712,7 +3715,7 @@ static void update_task_scan_period(struct task_struct *p, p->mm->numa_next_scan = jiffies + msecs_to_jiffies(p->numa_scan_period); - return; + goto out; } /* @@ -3756,7 +3759,10 @@ static void update_task_scan_period(struct task_struct *p, p->numa_scan_period = clamp(p->numa_scan_period + diff, task_scan_min(p), task_scan_max(p)); - memset(p->numa_faults_locality, 0, sizeof(p->numa_faults_locality)); + +out: + memset(p->numa_faults_locality, 0, + sizeof(p->numa_faults_locality)); } /* @@ -8208,7 +8214,6 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags) struct sched_entity *se = &p->se; struct cfs_rq *cfs_rq = &rq->cfs; unsigned long weight; - bool curr; if (task_is_throttled(p) && enqueue_throttled_task(p)) return; @@ -8237,23 +8242,14 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags) if (p->in_iowait) cpufreq_update_util(rq, SCHED_CPUFREQ_IOWAIT); - /* - * XXX comment on the curr thing - */ - curr = (cfs_rq->curr == se); - if (curr) - place_entity(cfs_rq, se, flags); if (se->on_rq && se->sched_delayed) requeue_delayed_entity(cfs_rq, se); weight = enqueue_hierarchy(p, flags); - - if (!curr) { - reweight_eevdf(cfs_rq, se, weight, false); - place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED); - __enqueue_entity(cfs_rq, se); - } + reweight_eevdf(cfs_rq, se, weight, false); + place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED); + __enqueue_entity(cfs_rq, se); if (!rq_h_nr_queued && rq->cfs.h_nr_queued) dl_server_start(&rq->fair_server); @@ -8673,8 +8669,8 @@ static int sched_balance_find_dst_group_cpu(struct sched_group *group, struct task_struct *p, int this_cpu) { unsigned long load, min_load = ULONG_MAX; - unsigned int min_exit_latency = UINT_MAX; - u64 latest_idle_timestamp = 0; + u64 min_exit_latency = U64_MAX; + unsigned int nr_candidates = 0; int least_loaded_cpu = this_cpu; int shallowest_idle_cpu = -1; int i; @@ -8695,24 +8691,16 @@ sched_balance_find_dst_group_cpu(struct sched_group *group, struct task_struct * if (available_idle_cpu(i)) { struct cpuidle_state *idle = idle_get_state(rq); - if (idle && idle->exit_latency < min_exit_latency) { - /* - * We give priority to a CPU whose idle state - * has the smallest exit latency irrespective - * of any idle timestamp. - */ - min_exit_latency = idle->exit_latency; - latest_idle_timestamp = rq->idle_stamp; - shallowest_idle_cpu = i; - } else if ((!idle || idle->exit_latency == min_exit_latency) && - rq->idle_stamp > latest_idle_timestamp) { - /* - * If equal or no active idle state, then - * the most recently idled CPU might have - * a warmer cache. - */ - latest_idle_timestamp = rq->idle_stamp; + u64 exit_latency = idle ? idle->exit_latency : U64_MAX; + + if (shallowest_idle_cpu == -1 || exit_latency < min_exit_latency) { + min_exit_latency = exit_latency; shallowest_idle_cpu = i; + nr_candidates = 1; + } else if (exit_latency == min_exit_latency) { + nr_candidates++; + if (!reciprocal_scale(sched_rng(), nr_candidates)) + shallowest_idle_cpu = i; } } else if (shallowest_idle_cpu == -1) { load = cpu_load(cpu_rq(i)); @@ -10071,8 +10059,14 @@ static inline bool set_preempt_buddy(struct cfs_rq *cfs_rq, struct sched_entity static inline bool set_short_buddy(struct cfs_rq *cfs_rq, struct sched_entity *pse) { - if (cfs_rq->next && cfs_rq->next->slice < pse->slice) - return false; + if (cfs_rq->next) { + if (cfs_rq->next->slice < pse->slice) + return false; + + if (cfs_rq->next->slice == pse->slice && + entity_before(cfs_rq->next, pse)) + return false; + } set_next_buddy(cfs_rq, pse); return true; @@ -11438,21 +11432,7 @@ next: */ static void attach_tasks(struct lb_env *env) { - struct list_head *tasks = &env->tasks; - struct task_struct *p; - struct rq_flags rf; - - rq_lock(env->dst_rq, &rf); - update_rq_clock(env->dst_rq); - - while (!list_empty(tasks)) { - p = list_first_entry(tasks, struct task_struct, se.group_node); - list_del_init(&p->se.group_node); - - attach_task(env->dst_rq, p); - } - - rq_unlock(env->dst_rq, &rf); + __attach_tasks(env->dst_rq, &env->tasks); } #ifdef CONFIG_NO_HZ_COMMON @@ -13745,7 +13725,7 @@ static int sched_balance_rq(int this_cpu, struct rq *this_rq, }; bool need_unlock = false; - cpumask_and(cpus, sched_domain_span(sd), cpu_active_mask); + cpumask_and(cpus, sched_domain_span(sd), cpu_preferred_mask); schedstat_inc(sd->lb_count[idle]); @@ -14870,10 +14850,8 @@ static int sched_balance_newidle(struct rq *this_rq, struct rq_flags *rf) */ this_rq->idle_stamp = rq_clock(this_rq); - /* - * Do not pull tasks towards !active CPUs... - */ - if (!cpu_active(this_cpu)) + /* Do not pull tasks towards !preferred CPUs */ + if (!cpu_preferred(this_cpu)) return 0; /* @@ -15513,14 +15491,18 @@ static void switched_to_fair(struct rq *rq, struct task_struct *p) } } -static void set_next_task_fair(struct rq *rq, struct task_struct *p, bool first) +static void set_next_task_fair(struct rq *rq, struct task_struct *p, enum snt_e type) { struct sched_entity *se = &p->se; - bool throttled = false; struct cfs_rq *cfs_rq = &rq->cfs; unsigned long weight = NICE_0_LOAD; + bool first = type == SNT_PICK; + bool throttled = false; bool on_rq = se->on_rq; + if (type == SNT_REPICK) + goto repick; + clear_buddies(cfs_rq, se); if (on_rq) @@ -15564,11 +15546,18 @@ static void set_next_task_fair(struct rq *rq, struct task_struct *p, bool first) WARN_ON_ONCE(se->sched_delayed); - if (hrtick_enabled_fair(rq)) - hrtick_start_fair(rq, p); - update_misfit_status(p, rq); sched_fair_update_stop_tick(rq, p); + +repick: + /* + * A same-task repick skips put_prev_task_fair(), but + * pick_task_fair() refreshed the entity hrtick_start_fair() reads + * before selecting it again. rq->cfs.curr identifies that entity, + * including with group scheduling. + */ + if (hrtick_enabled_fair(rq)) + hrtick_start_fair(rq, p); } void init_cfs_rq(struct cfs_rq *cfs_rq) diff --git a/kernel/sched/idle.c b/kernel/sched/idle.c index eb73b65ce6c4..76f3c84ca684 100644 --- a/kernel/sched/idle.c +++ b/kernel/sched/idle.c @@ -487,8 +487,11 @@ static void put_prev_task_idle(struct rq *rq, struct task_struct *prev, struct t update_rq_avg_idle(rq); } -static void set_next_task_idle(struct rq *rq, struct task_struct *next, bool first) +static void set_next_task_idle(struct rq *rq, struct task_struct *next, enum snt_e type) { + if (type == SNT_REPICK) + return; + update_idle_core(rq); scx_update_idle(rq, true, true); schedstat_inc(rq->sched_goidle); diff --git a/kernel/sched/rt.c b/kernel/sched/rt.c index 85303add726d..1535046a23ff 100644 --- a/kernel/sched/rt.c +++ b/kernel/sched/rt.c @@ -1654,11 +1654,14 @@ static void wakeup_preempt_rt(struct rq *rq, struct task_struct *p, int flags) check_preempt_equal_prio(rq, p); } -static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, bool first) +static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, enum snt_e type) { struct sched_rt_entity *rt_se = &p->rt; struct rt_rq *rt_rq = &rq->rt; + if (type == SNT_REPICK) + return; + p->se.exec_start = rq_clock_task(rq); if (on_rt_rq(&p->rt)) update_stats_wait_end_rt(rt_rq, rt_se); @@ -1666,7 +1669,7 @@ static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, bool f /* The running task is never eligible for pushing */ dequeue_pushable_task(rq, p); - if (!first) + if (type != SNT_PICK) return; /* diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index e656c7059bf8..7d2ec527b8a2 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -1326,6 +1326,9 @@ struct rq { #ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING u64 prev_steal_time_rq; #endif +#ifdef CONFIG_PREFERRED_CPU + bool npc_push_work_pending; +#endif /* calc_load related fields */ unsigned long calc_load_update; @@ -1371,7 +1374,6 @@ struct rq { struct task_struct *core_pick; struct sched_dl_entity *core_dl_server; unsigned int core_enabled; - unsigned int core_sched_seq; struct rb_root core_tree; /* shared state -- careful with sched_core_cpu_deactivate() */ @@ -2447,16 +2449,25 @@ extern __read_mostly unsigned int sysctl_sched_features; #ifdef CONFIG_JUMP_LABEL -#define SCHED_FEAT(name, enabled) \ -static __always_inline bool static_branch_##name(struct static_key *key) \ -{ \ - return static_key_##enabled(key); \ +union sched_feat_key { + struct static_key_true key_true; + struct static_key_false key_false; +}; + +#define sched_feat_branch_true(key) static_branch_likely(&(key)->key_true) +#define sched_feat_branch_false(key) static_branch_unlikely(&(key)->key_false) + +#define SCHED_FEAT(name, enabled) \ +static __always_inline bool \ +static_branch_##name(union sched_feat_key *key) \ +{ \ + return sched_feat_branch_##enabled(key); \ } #include "features.h" #undef SCHED_FEAT -extern struct static_key sched_feat_keys[__SCHED_FEAT_NR]; +extern union sched_feat_key sched_feat_keys[__SCHED_FEAT_NR]; #define sched_feat(x) (static_branch_##x(&sched_feat_keys[__SCHED_FEAT_##x])) #else /* !CONFIG_JUMP_LABEL: */ @@ -2508,6 +2519,12 @@ static inline bool task_is_blocked(struct task_struct *p) return !!p->blocked_on; } +#ifdef CONFIG_SCHED_PROXY_EXEC +void sched_proxy_block_task(struct rq *rq, struct task_struct *p); +#else +static inline void sched_proxy_block_task(struct rq *rq, struct task_struct *p) {} +#endif + static inline int task_on_cpu(struct rq *rq, struct task_struct *p) { return p->on_cpu; @@ -2527,11 +2544,17 @@ static inline int task_on_rq_migrating(struct task_struct *p) #define WF_EXEC 0x02 /* Wakeup after exec; maps to SD_BALANCE_EXEC */ #define WF_FORK 0x04 /* Wakeup after fork; maps to SD_BALANCE_FORK */ #define WF_TTWU 0x08 /* Wakeup; maps to SD_BALANCE_WAKE */ - -#define WF_SYNC 0x10 /* Waker goes to sleep after wakeup */ +/* + * Hint that the caller expects the waker to sleep soon. + * Scheduler classes may use it for placement or preemption. + * Callers must not rely on it to prevent migration, + * preserve CPU locality or make the wakee run next. + */ +#define WF_SYNC 0x10 #define WF_MIGRATED 0x20 /* Internal use, task got migrated */ #define WF_CURRENT_CPU 0x40 /* Prefer to move the wakee to the current CPU. */ #define WF_RQ_SELECTED 0x80 /* ->select_task_rq() was called */ +#define WF_TTWU_RQ 0x100 /* Wakeup completed through ttwu_runnable() */ static_assert(WF_EXEC == SD_BALANCE_EXEC); static_assert(WF_FORK == SD_BALANCE_FORK); @@ -2621,6 +2644,12 @@ struct affinity_context { extern s64 update_curr_common(struct rq *rq); +enum snt_e { + SNT_NORMAL, /* set_next_task() */ + SNT_PICK, /* put_prev_set_next_task(): prev != next */ + SNT_REPICK, /* put_prev_set_next_task(): prev == next */ +}; + struct sched_class { #ifdef CONFIG_UCLAMP_TASK @@ -2678,7 +2707,7 @@ struct sched_class { * __schedule: rq->lock */ void (*put_prev_task)(struct rq *rq, struct task_struct *p, struct task_struct *next); - void (*set_next_task)(struct rq *rq, struct task_struct *p, bool first); + void (*set_next_task)(struct rq *rq, struct task_struct *p, enum snt_e type); /* * select_task_rq: p->pi_lock @@ -2781,7 +2810,7 @@ static inline void put_prev_task(struct rq *rq, struct task_struct *prev) static inline void set_next_task(struct rq *rq, struct task_struct *next) { - next->sched_class->set_next_task(rq, next, false); + next->sched_class->set_next_task(rq, next, SNT_NORMAL); } static inline void @@ -2802,11 +2831,13 @@ static inline void put_prev_set_next_task(struct rq *rq, __put_prev_set_next_dl_server(rq, prev, next); - if (next == prev) + if (next == prev) { + next->sched_class->set_next_task(rq, next, SNT_REPICK); return; + } prev->sched_class->put_prev_task(rq, prev, next); - next->sched_class->set_next_task(rq, next, true); + next->sched_class->set_next_task(rq, next, SNT_PICK); } /* @@ -3139,6 +3170,25 @@ static inline void attach_one_task(struct rq *rq, struct task_struct *p) attach_task(rq, p); } +/* + * __attach_tasks() - attaches a list of tasks (using se.group_node) to + * the new rq + */ +static inline void __attach_tasks(struct rq *rq, struct list_head *tasks) +{ + guard(rq_lock)(rq); + update_rq_clock(rq); + + while (!list_empty(tasks)) { + struct task_struct *p; + + p = list_first_entry(tasks, struct task_struct, se.group_node); + list_del_init(&p->se.group_node); + + attach_task(rq, p); + } +} + #ifdef CONFIG_PREEMPT_RT # define SCHED_NR_MIGRATE_BREAK 8 #else @@ -4252,4 +4302,10 @@ DEFINE_CLASS_IS_UNCONDITIONAL(sched_change) #include "ext/ext.h" +#ifdef CONFIG_PREFERRED_CPU +void sched_push_current_non_preferred_cpu(struct rq *rq); +#else /* !CONFIG_PREFERRED_CPU */ +static inline void sched_push_current_non_preferred_cpu(struct rq *rq) { } +#endif + #endif /* _KERNEL_SCHED_SCHED_H */ diff --git a/kernel/sched/stop_task.c b/kernel/sched/stop_task.c index c909ca0d8c87..1e0109ec36b3 100644 --- a/kernel/sched/stop_task.c +++ b/kernel/sched/stop_task.c @@ -27,8 +27,11 @@ wakeup_preempt_stop(struct rq *rq, struct task_struct *p, int flags) /* we're never preempted */ } -static void set_next_task_stop(struct rq *rq, struct task_struct *stop, bool first) +static void set_next_task_stop(struct rq *rq, struct task_struct *stop, enum snt_e type) { + if (type == SNT_REPICK) + return; + stop->se.exec_start = rq_clock_task(rq); } diff --git a/kernel/sched/wait.c b/kernel/sched/wait.c index d033f600f48c..477e4bf9c01e 100644 --- a/kernel/sched/wait.c +++ b/kernel/sched/wait.c @@ -174,15 +174,11 @@ EXPORT_SYMBOL_GPL(__wake_up_locked_key); * @mode: which threads * @key: opaque value to be passed to wakeup targets * - * The sync wakeup differs that the waker knows that it will schedule - * away soon, so while the target thread will be woken up, it will not - * be migrated to another CPU - ie. the two threads are 'synchronized' - * with each other. This can prevent needless bouncing between CPUs. + * Passes WF_SYNC to waitqueue wake functions. The default wake function + * forwards it to the scheduler; see WF_SYNC for the hint's semantics. * - * On UP it can prevent extra preemption. - * - * If this function wakes up a task, it executes a full memory barrier before - * accessing the task state. + * If this function wakes up a task, it executes a full memory barrier + * before accessing the task state. */ void __wake_up_sync_key(struct wait_queue_head *wq_head, unsigned int mode, void *key) @@ -200,15 +196,7 @@ EXPORT_SYMBOL_GPL(__wake_up_sync_key); * @mode: which threads * @key: opaque value to be passed to wakeup targets * - * The sync wakeup differs in that the waker knows that it will schedule - * away soon, so while the target thread will be woken up, it will not - * be migrated to another CPU - ie. the two threads are 'synchronized' - * with each other. This can prevent needless bouncing between CPUs. - * - * On UP it can prevent extra preemption. - * - * If this function wakes up a task, it executes a full memory barrier before - * accessing the task state. + * Same as __wake_up_sync_key(), but called with @wq_head->lock held. */ void __wake_up_locked_sync_key(struct wait_queue_head *wq_head, unsigned int mode, void *key) |
