diff options
Diffstat (limited to 'kernel')
35 files changed, 784 insertions, 461 deletions
diff --git a/kernel/Kconfig.kexec b/kernel/Kconfig.kexec index 15632358bcf7..a97ed9605602 100644 --- a/kernel/Kconfig.kexec +++ b/kernel/Kconfig.kexec @@ -167,7 +167,7 @@ config CRASH_MAX_MEMORY_RANGES memory regions that the elfcorehdr buffer/segment can accommodate. These regions are obtained via walk_system_ram_res(); eg. the 'System RAM' entries in /proc/iomem. - This value is combined with NR_CPUS_DEFAULT and multiplied by + This value is combined with NR_CPUS and multiplied by sizeof(Elf64_Phdr) to determine the final elfcorehdr memory buffer/ segment size. The value 8192, for example, covers a (sparsely populated) 1TiB system diff --git a/kernel/Kconfig.preempt b/kernel/Kconfig.preempt index f294dad43bd7..edc067a0c422 100644 --- a/kernel/Kconfig.preempt +++ b/kernel/Kconfig.preempt @@ -132,10 +132,9 @@ config PREEMPTION config PREEMPT_DYNAMIC bool "Preemption behaviour defined on boot" - depends on HAVE_PREEMPT_DYNAMIC - select JUMP_LABEL if HAVE_PREEMPT_DYNAMIC_KEY + depends on ARCH_HAS_PREEMPT_LAZY select PREEMPT_BUILD - default y if HAVE_PREEMPT_DYNAMIC_CALL + default y help This option allows to define the preemption model on the kernel command line parameter and thus override the default preemption @@ -145,9 +144,7 @@ config PREEMPT_DYNAMIC provide a pre-built kernel binary to reduce the number of kernel flavors they offer while still offering different usecases. - The runtime overhead is negligible with HAVE_STATIC_CALL_INLINE enabled - but if runtime patching is not available for the specific architecture - then the potential overhead should be considered. + The runtime overhead is negligible. Interesting if you want the same pre-built kernel should be used for both Server and Desktop workloads. @@ -197,3 +194,7 @@ config SCHED_CLASS_EXT For more information: Documentation/scheduler/sched-ext.rst https://github.com/sched-ext/scx + +config PREFERRED_CPU + bool + depends on SMP && PARAVIRT diff --git a/kernel/cpu.c b/kernel/cpu.c index b3c8553d7bd6..376d297a6292 100644 --- a/kernel/cpu.c +++ b/kernel/cpu.c @@ -3103,6 +3103,11 @@ EXPORT_SYMBOL(__cpu_dying_mask); atomic_t __num_online_cpus __read_mostly; EXPORT_SYMBOL(__num_online_cpus); +#ifdef CONFIG_PREFERRED_CPU +struct cpumask __cpu_preferred_mask __read_mostly; +EXPORT_SYMBOL_GPL(__cpu_preferred_mask); +#endif + void init_cpu_present(const struct cpumask *src) { cpumask_copy(&__cpu_present_mask, src); @@ -3160,6 +3165,7 @@ void __init boot_cpu_init(void) /* Mark the boot cpu "present", "online" etc for SMP and UP case */ set_cpu_online(cpu, true); set_cpu_active(cpu, true); + set_cpu_preferred(cpu, true); set_cpu_present(cpu, true); set_cpu_possible(cpu, true); diff --git a/kernel/crash_core.c b/kernel/crash_core.c index 2b36aa9fade0..d0bd2d0cf899 100644 --- a/kernel/crash_core.c +++ b/kernel/crash_core.c @@ -648,7 +648,7 @@ int crash_check_hotplug_support(void) * new list of CPUs and memory. To make changes to the elfcorehdr, it * should be large enough to permit a growing number of CPU and Memory * resources. One can estimate the elfcorehdr memory size based on - * NR_CPUS_DEFAULT and CRASH_MAX_MEMORY_RANGES. The elfcorehdr is + * NR_CPUS and CRASH_MAX_MEMORY_RANGES. The elfcorehdr is * excluded from SHA verification by default if the architecture * supports crash hotplug. */ diff --git a/kernel/entry/common.c b/kernel/entry/common.c index e3d381fd3d25..e234b04373fe 100644 --- a/kernel/entry/common.c +++ b/kernel/entry/common.c @@ -123,7 +123,7 @@ noinstr irqentry_state_t irqentry_enter(struct pt_regs *regs) /** * arch_irqentry_exit_need_resched - Architecture specific need resched function * - * Invoked from raw_irqentry_exit_cond_resched() to check if resched is needed. + * Invoked from irqentry_exit_cond_resched() to check if resched is needed. * Defaults return true. * * The main purpose is to permit arch to avoid preemption of a task from an IRQ. @@ -134,7 +134,7 @@ static inline bool arch_irqentry_exit_need_resched(void); static inline bool arch_irqentry_exit_need_resched(void) { return true; } #endif -void raw_irqentry_exit_cond_resched(void) +void irqentry_exit_cond_resched(void) { if (!preempt_count()) { /* Sanity check RCU and thread stack */ @@ -145,19 +145,6 @@ void raw_irqentry_exit_cond_resched(void) preempt_schedule_irq(); } } -#ifdef CONFIG_PREEMPT_DYNAMIC -#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) -DEFINE_STATIC_CALL(irqentry_exit_cond_resched, raw_irqentry_exit_cond_resched); -#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -DEFINE_STATIC_KEY_TRUE(sk_dynamic_irqentry_exit_cond_resched); -void dynamic_irqentry_exit_cond_resched(void) -{ - if (!static_branch_unlikely(&sk_dynamic_irqentry_exit_cond_resched)) - return; - raw_irqentry_exit_cond_resched(); -} -#endif -#endif noinstr void irqentry_exit(struct pt_regs *regs, irqentry_state_t state) { diff --git a/kernel/events/core.c b/kernel/events/core.c index 601e8d944c24..a34ff4cb410d 100644 --- a/kernel/events/core.c +++ b/kernel/events/core.c @@ -7828,22 +7828,82 @@ unsigned long perf_instruction_pointer(struct perf_event *event, 0 : perf_arch_instruction_pointer(regs); } +u64 __weak perf_reg_value(struct pt_regs *regs, int idx) +{ + return 0; +} + +int __weak perf_reg_validate(u64 mask, bool simd_enabled) +{ + return mask ? -ENOSYS : 0; +} + +u64 __weak perf_reg_abi(struct task_struct *task) +{ + return PERF_SAMPLE_REGS_ABI_NONE; +} + +void __weak perf_get_regs_user(struct perf_regs *regs_user, + struct pt_regs *regs) +{ + regs_user->regs = task_pt_regs(current); + regs_user->abi = perf_reg_abi(current); +} + +#define word_for_each_set_bit(bit, val) \ + for (unsigned long long __v = (val); \ + __v && ((bit = __builtin_ctzll(__v)), 1); \ + __v &= __v - 1) + static void perf_output_sample_regs(struct perf_output_handle *handle, struct pt_regs *regs, u64 mask) { int bit; - DECLARE_BITMAP(_mask, 64); - bitmap_from_u64(_mask, mask); - for_each_set_bit(bit, _mask, sizeof(mask) * BITS_PER_BYTE) { - u64 val; - - val = perf_reg_value(regs, bit); + word_for_each_set_bit(bit, mask) { + u64 val = perf_reg_value(regs, bit); perf_output_put(handle, val); } } +static void +perf_output_sample_simd_regs(struct perf_output_handle *handle, + struct perf_event *event, + struct pt_regs *regs, + u64 mask, u32 pred_mask) +{ + u64 pred_qwords = event->attr.sample_simd_pred_reg_qwords; + u64 vec_qwords = event->attr.sample_simd_vec_reg_qwords; + u64 nr_vectors = hweight64(mask); + u64 nr_pred = hweight32(pred_mask); + int bit; + + perf_output_put(handle, nr_vectors); + perf_output_put(handle, vec_qwords); + perf_output_put(handle, nr_pred); + perf_output_put(handle, pred_qwords); + + if (nr_vectors) { + word_for_each_set_bit(bit, mask) { + for (int i = 0; i < vec_qwords; i++) { + u64 val = perf_simd_reg_value(regs, bit, + i, false); + perf_output_put(handle, val); + } + } + } + if (nr_pred) { + word_for_each_set_bit(bit, pred_mask) { + for (int i = 0; i < pred_qwords; i++) { + u64 val = perf_simd_reg_value(regs, bit, + i, true); + perf_output_put(handle, val); + } + } + } +} + static void perf_sample_regs_user(struct perf_regs *regs_user, struct pt_regs *regs) { @@ -7877,6 +7937,17 @@ static void perf_sample_regs_intr(struct perf_regs *regs_intr, } } +int __weak perf_simd_reg_validate(u16 vec_qwords, u64 vec_mask, + u16 pred_qwords, u32 pred_mask) +{ + return -EINVAL; +} + +u64 __weak perf_simd_reg_value(struct pt_regs *regs, int idx, + u16 qwords_idx, bool pred) +{ + return 0; +} /* * Get remaining task size from user stack pointer. @@ -8407,10 +8478,17 @@ void perf_output_sample(struct perf_output_handle *handle, perf_output_put(handle, abi); if (abi) { - u64 mask = event->attr.sample_regs_user; + struct perf_event_attr *attr = &event->attr; + u64 mask = attr->sample_regs_user; perf_output_sample_regs(handle, data->regs_user.regs, mask); + if (abi & PERF_SAMPLE_REGS_ABI_SIMD) { + perf_output_sample_simd_regs(handle, event, + data->regs_user.regs, + attr->sample_simd_vec_reg_user, + attr->sample_simd_pred_reg_user); + } } } @@ -8438,11 +8516,18 @@ void perf_output_sample(struct perf_output_handle *handle, perf_output_put(handle, abi); if (abi) { - u64 mask = event->attr.sample_regs_intr; + struct perf_event_attr *attr = &event->attr; + u64 mask = attr->sample_regs_intr; perf_output_sample_regs(handle, data->regs_intr.regs, mask); + if (abi & PERF_SAMPLE_REGS_ABI_SIMD) { + perf_output_sample_simd_regs(handle, event, + data->regs_intr.regs, + attr->sample_simd_vec_reg_intr, + attr->sample_simd_pred_reg_intr); + } } } @@ -8645,6 +8730,29 @@ static __always_inline u64 __cond_set(u64 flags, u64 s, u64 d) return d * !!(flags & s); } +u64 perf_update_xregs_size(struct perf_event *event, bool intr) +{ + u16 pred_qwords = event->attr.sample_simd_pred_reg_qwords; + u16 vec_qwords = event->attr.sample_simd_vec_reg_qwords; + u64 pred_mask; + u64 mask; + int size; + + if (intr) { + mask = event->attr.sample_simd_vec_reg_intr; + pred_mask = event->attr.sample_simd_pred_reg_intr; + } else { + mask = event->attr.sample_simd_vec_reg_user; + pred_mask = event->attr.sample_simd_pred_reg_user; + } + + size = sizeof(u64) * 4; + size += (hweight64(mask) * vec_qwords + + hweight64(pred_mask) * pred_qwords) * sizeof(u64); + + return size; +} + void perf_prepare_sample(struct perf_sample_data *data, struct perf_event *event, struct pt_regs *regs) @@ -8707,7 +8815,12 @@ void perf_prepare_sample(struct perf_sample_data *data, if (data->regs_user.regs) { u64 mask = event->attr.sample_regs_user; + size += hweight64(mask) * sizeof(u64); + if (event_has_simd_regs(event)) { + size += perf_update_xregs_size(event, false); + data->regs_user.abi |= PERF_SAMPLE_REGS_ABI_SIMD; + } } data->dyn_size += size; @@ -8772,6 +8885,10 @@ void perf_prepare_sample(struct perf_sample_data *data, u64 mask = event->attr.sample_regs_intr; size += hweight64(mask) * sizeof(u64); + if (event_has_simd_regs(event)) { + size += perf_update_xregs_size(event, true); + data->regs_intr.abi |= PERF_SAMPLE_REGS_ABI_SIMD; + } } data->dyn_size += size; @@ -13116,12 +13233,6 @@ int perf_pmu_unregister(struct pmu *pmu) } EXPORT_SYMBOL_GPL(perf_pmu_unregister); -static inline bool has_extended_regs(struct perf_event *event) -{ - return (event->attr.sample_regs_user & PERF_REG_EXTENDED_MASK) || - (event->attr.sample_regs_intr & PERF_REG_EXTENDED_MASK); -} - static int perf_try_init_event(struct pmu *pmu, struct perf_event *event) { struct perf_event_context *ctx = NULL; @@ -13155,8 +13266,14 @@ static int perf_try_init_event(struct pmu *pmu, struct perf_event *event) if (ret) goto err_pmu; + if (!(pmu->capabilities & PERF_PMU_CAP_SIMD_REGS) && + event_has_simd_regs(event)) { + ret = -EOPNOTSUPP; + goto err_destroy; + } + if (!(pmu->capabilities & PERF_PMU_CAP_EXTENDED_REGS) && - has_extended_regs(event)) { + event_has_extended_regs(event)) { ret = -EOPNOTSUPP; goto err_destroy; } @@ -13650,7 +13767,8 @@ static int perf_copy_attr(struct perf_event_attr __user *uattr, attr->size = size; - if (attr->__reserved_1 || attr->__reserved_2 || attr->__reserved_3) + if (attr->__reserved_1 || attr->__reserved_2 || + attr->__reserved_3 || attr->__reserved_4) return -EINVAL; if (attr->sample_type & ~(PERF_SAMPLE_MAX-1)) @@ -13696,9 +13814,18 @@ static int perf_copy_attr(struct perf_event_attr __user *uattr, } if (attr->sample_type & PERF_SAMPLE_REGS_USER) { - ret = perf_reg_validate(attr->sample_regs_user); + ret = perf_reg_validate(attr->sample_regs_user, + attr->sample_simd_regs_enabled); if (ret) return ret; + if (attr->sample_simd_regs_enabled) { + ret = perf_simd_reg_validate(attr->sample_simd_vec_reg_qwords, + attr->sample_simd_vec_reg_user, + attr->sample_simd_pred_reg_qwords, + attr->sample_simd_pred_reg_user); + if (ret) + return ret; + } } if (attr->sample_type & PERF_SAMPLE_STACK_USER) { @@ -13719,8 +13846,20 @@ static int perf_copy_attr(struct perf_event_attr __user *uattr, if (!attr->sample_max_stack) attr->sample_max_stack = sysctl_perf_event_max_stack; - if (attr->sample_type & PERF_SAMPLE_REGS_INTR) - ret = perf_reg_validate(attr->sample_regs_intr); + if (attr->sample_type & PERF_SAMPLE_REGS_INTR) { + ret = perf_reg_validate(attr->sample_regs_intr, + attr->sample_simd_regs_enabled); + if (ret) + return ret; + if (attr->sample_simd_regs_enabled) { + ret = perf_simd_reg_validate(attr->sample_simd_vec_reg_qwords, + attr->sample_simd_vec_reg_intr, + attr->sample_simd_pred_reg_qwords, + attr->sample_simd_pred_reg_intr); + if (ret) + return ret; + } + } #ifndef CONFIG_CGROUP_PERF if (attr->sample_type & PERF_SAMPLE_CGROUP) diff --git a/kernel/exit.c b/kernel/exit.c index 29e853a36602..9ff1fa7b30ea 100644 --- a/kernel/exit.c +++ b/kernel/exit.c @@ -168,12 +168,6 @@ static void __exit_signal(struct release_task_post *post, struct task_struct *ts lockdep_tasklist_lock_is_held()); spin_lock(&sighand->siglock); -#ifdef CONFIG_POSIX_TIMERS - posix_cpu_timers_exit(tsk); - if (group_dead) - posix_cpu_timers_exit_group(tsk); -#endif - if (group_dead) { tty = sig->tty; sig->tty = NULL; @@ -940,13 +934,12 @@ void __noreturn do_exit(long code) panic("Attempted to kill init! exitcode=0x%08x\n", tsk->signal->group_exit_code ?: (int)code); -#ifdef CONFIG_POSIX_TIMERS - hrtimer_cancel(&tsk->signal->real_timer); - exit_itimers(tsk); -#endif if (tsk->mm) setmax_mm_hiwater_rss(&tsk->signal->maxrss, tsk->mm); } + + posixtimer_exit(group_dead); + acct_collect(code, group_dead); if (group_dead) tty_audit_exit(); diff --git a/kernel/futex/requeue.c b/kernel/futex/requeue.c index b3f4a4bccb12..842d852302dd 100644 --- a/kernel/futex/requeue.c +++ b/kernel/futex/requeue.c @@ -744,7 +744,7 @@ int handle_early_requeue_pi_wakeup(struct futex_hash_bucket *hb, /* Handle spurious wakeups gracefully */ ret = -EWOULDBLOCK; - if (timeout && !timeout->task) + if (timeout && !hrtimer_sleeper_task_get(timeout)) ret = -ETIMEDOUT; else if (signal_pending(current)) ret = -ERESTARTNOINTR; diff --git a/kernel/futex/waitwake.c b/kernel/futex/waitwake.c index d4483d15d30a..cf18309e5770 100644 --- a/kernel/futex/waitwake.c +++ b/kernel/futex/waitwake.c @@ -383,7 +383,7 @@ void futex_do_wait(struct futex_q *q, struct hrtimer_sleeper *timeout) * flagged for rescheduling. Only call schedule if there * is no timeout, or if it has yet to expire. */ - if (!timeout || timeout->task) + if (!timeout || hrtimer_sleeper_task_get(timeout)) schedule(); } __set_current_state(TASK_RUNNING); @@ -539,7 +539,7 @@ retry: static void futex_sleep_multiple(struct futex_vector *vs, unsigned int count, struct hrtimer_sleeper *to) { - if (to && !to->task) + if (to && !hrtimer_sleeper_task_get(to)) return; for (; count; count--, vs++) { @@ -590,7 +590,7 @@ int futex_wait_multiple(struct futex_vector *vs, unsigned int count, if (ret >= 0) return ret; - if (to && !to->task) + if (to && !hrtimer_sleeper_task_get(to)) return -ETIMEDOUT; else if (signal_pending(current)) return -ERESTARTSYS; @@ -725,7 +725,7 @@ retry: if (!futex_unqueue(&q)) return 0; - if (to && !to->task) + if (to && !hrtimer_sleeper_task_get(to)) return -ETIMEDOUT; /* diff --git a/kernel/irq/irqdomain.c b/kernel/irq/irqdomain.c index 57c819da30c2..4fdcb6df5306 100644 --- a/kernel/irq/irqdomain.c +++ b/kernel/irq/irqdomain.c @@ -344,6 +344,7 @@ static struct irq_domain *__irq_domain_instantiate(const struct irq_domain_info err = irq_domain_alloc_generic_chips(domain, info->dgc_info); if (err) goto err_domain_free; + domain->flags |= IRQ_DOMAIN_FLAG_DESTROY_GC; } if (info->init) { diff --git a/kernel/locking/rtmutex.c b/kernel/locking/rtmutex.c index 4728631ae719..5a9534c715b8 100644 --- a/kernel/locking/rtmutex.c +++ b/kernel/locking/rtmutex.c @@ -1644,7 +1644,7 @@ static int __sched rt_mutex_slowlock_block(struct rt_mutex_base *lock, break; } - if (timeout && !timeout->task) { + if (timeout && !hrtimer_sleeper_task_get(timeout)) { ret = -ETIMEDOUT; break; } diff --git a/kernel/sched/core.c b/kernel/sched/core.c index 1fe40de6ebe3..84313c9c4ba9 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -805,7 +805,7 @@ struct rq *_task_rq_lock(struct task_struct *p, struct rq_flags *rf) /* Use CONFIG_PARAVIRT as this will avoid more #ifdef in arch code. */ #ifdef CONFIG_PARAVIRT -struct static_key paravirt_steal_rq_enabled; +DEFINE_STATIC_KEY_FALSE(paravirt_steal_rq_enabled); #endif static void update_rq_clock_task(struct rq *rq, s64 delta) @@ -844,7 +844,7 @@ static void update_rq_clock_task(struct rq *rq, s64 delta) } #endif #ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING - if (static_key_false((¶virt_steal_rq_enabled))) { + if (static_branch_unlikely(¶virt_steal_rq_enabled)) { u64 prev_steal; steal = prev_steal = paravirt_steal_clock(cpu_of(rq)); @@ -2252,7 +2252,8 @@ void deactivate_task(struct rq *rq, struct task_struct *p, int flags) dequeue_task(rq, p, flags); } -static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_state) +static bool dequeue_block_task(struct rq *rq, struct task_struct *p, + unsigned long task_state) { int flags = DEQUEUE_NOCLOCK; @@ -2273,9 +2274,15 @@ static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_ * * Where __schedule() and ttwu() have matching control dependencies. * - * After this, schedule() must not care about p->state any more. + * Once the caller invokes __block_task(), schedule() must not care about + * p->state any more. */ - if (dequeue_task(rq, p, DEQUEUE_SLEEP | flags)) + return dequeue_task(rq, p, DEQUEUE_SLEEP | flags); +} + +static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_state) +{ + if (dequeue_block_task(rq, p, task_state)) __block_task(rq, p); } @@ -2504,6 +2511,24 @@ static inline bool rq_has_pinned_tasks(struct rq *rq) return rq->nr_pinned; } +static inline bool task_can_migrate_to_preferred(struct task_struct *p, int cpu) +{ + /* No need to migrate from a preferred CPU */ + if (cpu_preferred(cpu)) + return false; + + /* Only FAIR tasks honor preferred CPU state */ + if (unlikely(p->sched_class != &fair_sched_class)) + return false; + + /* Ignore preferred state if task affinity is changing */ + if (unlikely(!cpumask_test_cpu(task_cpu(p), p->cpus_ptr))) + return false; + + return cpumask_intersects_and(p->cpus_ptr, cpu_preferred_mask, + task_cpu_possible_mask(p)); +} + /* * Per-CPU kthreads are allowed to run on !active && online CPUs, see * __set_cpus_allowed_ptr() and select_fallback_rq(). @@ -2519,8 +2544,12 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu) return cpu_online(cpu); /* Non kernel threads are not allowed during either online or offline. */ - if (!(p->flags & PF_KTHREAD)) + if (!(p->flags & PF_KTHREAD)) { + /* Try to use preferred CPU if task's affinity allows */ + if (task_can_migrate_to_preferred(p, cpu)) + return false; return cpu_active(cpu); + } /* KTHREAD_IS_PER_CPU is always allowed. */ if (kthread_is_per_cpu(p)) @@ -2530,7 +2559,11 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu) if (cpu_dying(cpu)) return false; - /* But are allowed during online. */ + /* Try to keep unbound kthreads on a preferred CPU if possible. */ + if (task_can_migrate_to_preferred(p, cpu)) + return false; + + /* Otherwise, they are allowed to run on online CPU. */ return cpu_online(cpu); } @@ -3773,6 +3806,7 @@ static inline void proxy_reset_donor(struct rq *rq) WARN_ON_ONCE(rq->donor == rq->curr); put_prev_set_next_task(rq, rq->donor, rq->curr); + rq->next_class = rq->curr->sched_class; rq_set_donor(rq, rq->curr); zap_balance_callbacks(rq); resched_curr(rq); @@ -3787,6 +3821,8 @@ static inline void proxy_reset_donor(struct rq *rq) */ static inline bool proxy_needs_return(struct rq *rq, struct task_struct *p) { + bool dequeued; + /* * Typically per __set_task_cpu(), task_cpu(p) == p->wake_cpu. * @@ -3809,12 +3845,23 @@ static inline bool proxy_needs_return(struct rq *rq, struct task_struct *p) /* If already current, don't need to return migrate */ if (task_current(rq, p)) return false; - - /* If we're return migrating the rq->donor, switch it out for idle */ - if (task_current_donor(rq, p)) - proxy_reset_donor(rq); } - block_task(rq, p, TASK_WAKING); + + dequeued = dequeue_block_task(rq, p, TASK_WAKING); + + /* + * Dequeue @p from its scheduling class before resetting rq->donor. + * In particular, sched_ext needs to end the donor's running session + * and clear SCX_TASK_QUEUED before put_prev_task_scx() is called by + * proxy_reset_donor(); otherwise it would reenqueue the blocked donor. + * + * Keep on_rq set until all donor references have been replaced. + */ + if (task_current_donor(rq, p)) + proxy_reset_donor(rq); + + if (dequeued) + __block_task(rq, p); return true; } #else /* !CONFIG_SCHED_PROXY_EXEC */ @@ -3905,7 +3952,7 @@ static int ttwu_runnable(struct task_struct *p, int wake_flags) * When on_rq && !on_cpu the task is preempted, see if * it should preempt the task that is current now. */ - wakeup_preempt(rq, p, wake_flags); + wakeup_preempt(rq, p, wake_flags | WF_TTWU_RQ); } ttwu_do_wakeup(p); return 1; @@ -5149,7 +5196,7 @@ static void do_balance_callbacks(struct rq *rq, struct balance_callback *head) lockdep_assert_rq_held(rq); while (head) { - func = (void (*)(struct rq *))head->func; + func = head->func; next = head->next; head->next = NULL; head = next; @@ -5789,6 +5836,9 @@ void sched_tick(void) unsigned long hw_pressure; u64 resched_latency; + if (!cpu_preferred(cpu)) + sched_push_current_non_preferred_cpu(rq); + if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE)) arch_scale_freq_tick(); @@ -6283,10 +6333,7 @@ pick_next_task(struct rq *rq, struct rq_flags *rf) * selection. In this case, do a core-wide selection. */ if (rq->core->core_pick_seq == rq->core->core_task_seq && - rq->core->core_pick_seq != rq->core_sched_seq && rq->core_pick) { - WRITE_ONCE(rq->core_sched_seq, rq->core->core_pick_seq); - next = rq->core_pick; rq->dl_server = rq->core_dl_server; rq->core_pick = NULL; @@ -6318,11 +6365,13 @@ restart: } /* - * core->core_task_seq, core->core_pick_seq, rq->core_sched_seq + * core->core_task_seq, core->core_pick_seq * * @task_seq guards the task state ({en,de}queues) * @pick_seq is the @task_seq we did a selection on - * @sched_seq is the @pick_seq we scheduled + * + * Once a core-wide selection is committed, a non-NULL core_pick denotes + * a pick which still needs to be consumed on this CPU. * * However, preemptions can cause multiple picks on the same task set. * 'Fix' this by also increasing @task_seq for every pick. @@ -6429,7 +6478,6 @@ restart: rq->core->core_pick_seq = rq->core->core_task_seq; next = rq->core_pick; - rq->core_sched_seq = rq->core->core_pick_seq; /* Something should have been selected for current CPU */ WARN_ON_ONCE(!next); @@ -6517,7 +6565,10 @@ static bool try_steal_cookie(int this, int that) return false; do { - if (p == src->core_pick || p == src->curr) + if (p == src->core_pick || p == src->curr || p == src->donor) + goto next; + + if (task_is_blocked(p)) goto next; if (!is_cpu_allowed(p, this)) @@ -6820,6 +6871,34 @@ static void proxy_deactivate(struct rq *rq, struct task_struct *donor) block_task(rq, donor, state); } +/* + * Remove a retained proxy donor before changing its scheduler ownership. + * The caller holds p->pi_lock, so p cannot wake and migrate if block_task() + * drops it from the runqueue. If DELAY_DEQUEUE keeps a blocked fair task + * queued, switching_from_fair() completes the dequeue in the immediately + * following sched_change_begin(). + */ +void sched_proxy_block_task(struct rq *rq, struct task_struct *p) +{ + unsigned long state = READ_ONCE(p->__state); + + lockdep_assert_held(&p->pi_lock); + lockdep_assert_rq_held(rq); + + if (!p->is_blocked || !task_on_rq_queued(p)) + return; + if (WARN_ON_ONCE(state == TASK_RUNNING)) + return; + + if (task_current_donor(rq, p)) + proxy_reset_donor(rq); + + if (!p->se.sched_delayed) + block_task(rq, p, state); + + WARN_ON_ONCE(task_on_rq_queued(p) && !p->se.sched_delayed); +} + static inline void proxy_release_rq_lock(struct rq *rq, struct rq_flags *rf) __releases(__rq_lockp(rq)) { @@ -6865,9 +6944,9 @@ static void proxy_migrate_task(struct rq *rq, struct rq_flags *rf, __must_hold(__rq_lockp(rq)) { struct rq *target_rq = cpu_rq(target_cpu); + LIST_HEAD(migrate_list); lockdep_assert_rq_held(rq); - WARN_ON(p == rq->curr); /* * Since we are migrating a blocked donor, it could be rq->donor, * and we want to make sure there aren't any references from this @@ -6880,13 +6959,20 @@ static void proxy_migrate_task(struct rq *rq, struct rq_flags *rf, * before we release the lock. */ proxy_resched_idle(rq); - - deactivate_task(rq, p, DEQUEUE_NOCLOCK); - proxy_set_task_cpu(p, target_cpu); - + for (; p; p = p->blocked_donor) { + WARN_ON(p == rq->curr); + deactivate_task(rq, p, DEQUEUE_NOCLOCK); + proxy_set_task_cpu(p, target_cpu); + /* + * We can re-use se.group_node to migrate the thing, + * because @p is deactivated (won't be balanced) and + * we hold the rq_lock. + */ + list_add(&p->se.group_node, &migrate_list); + } proxy_release_rq_lock(rq, rf); - attach_one_task(target_rq, p); + __attach_tasks(target_rq, &migrate_list); proxy_reacquire_rq_lock(rq, rf); } @@ -6979,7 +7065,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) if (!READ_ONCE(owner->on_rq) || owner->se.sched_delayed) { /* XXX Don't handle blocked owners/delayed dequeue yet */ if (curr_in_chain) - return proxy_resched_idle(rq); + goto resched_idle; __clear_task_blocked_on(p, NULL); goto deactivate; } @@ -6991,7 +7077,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) * and leave that CPU to sort things out. */ if (curr_in_chain) - return proxy_resched_idle(rq); + goto resched_idle; goto migrate_task; } @@ -7004,7 +7090,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) * case we should end up back in find_proxy_task(), this time * hopefully with all relevant tasks already enqueued. */ - return proxy_resched_idle(rq); + goto resched_idle; } /* @@ -7041,7 +7127,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) * So schedule rq->idle so that ttwu_runnable() can get the rq * lock and mark owner as running. */ - return proxy_resched_idle(rq); + goto resched_idle; } /* * OK, now we're absolutely sure @owner is on this @@ -7051,8 +7137,18 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf) owner->blocked_donor = p; } WARN_ON_ONCE(owner && !owner->on_rq); + + if (owner && !sched_cpu_cookie_match(rq, owner)) { + if (curr_in_chain) + return proxy_resched_idle(rq); + p = donor; /* Deactivate the donor, not the runnable owner */ + clear_task_blocked_on(p, NULL); + goto deactivate; + } return owner; +resched_idle: + return proxy_resched_idle(rq); deactivate: proxy_deactivate(rq, p); return NULL; @@ -7184,13 +7280,12 @@ static void __sched notrace __schedule(int sched_mode) } } else if (!preempt && prev_state) { /* - * We pass task_is_blocked() as the should_block arg - * in order to keep mutex-blocked tasks on the runqueue - * for slection with proxy-exec (without proxy-exec - * task_is_blocked() will always be false). + * Keep mutex-blocked tasks on the runqueue for proxy execution + * only when their scheduling class allows it. Without proxy + * execution, task_is_blocked() always returns false. */ try_to_block_task(rq, prev, &prev_state, - !task_is_blocked(prev)); + !task_is_blocked(prev) || !scx_allow_proxy_exec(prev)); switch_count = &prev->nvcsw; } @@ -7211,6 +7306,7 @@ pick_again: } if (next == rq->idle) { zap_balance_callbacks(rq); + scx_proxy_reenqueue_retry(rq, next); goto keep_resched; } } @@ -7229,8 +7325,10 @@ pick_again: * on_cpu. */ donor->sched_class->put_prev_task(rq, donor, donor); - donor->sched_class->set_next_task(rq, donor, true); + donor->sched_class->set_next_task(rq, donor, SNT_PICK); } + scx_proxy_donor_start(rq); + scx_proxy_reenqueue_retry(rq, next); } else { rq_set_donor(rq, next); } @@ -7489,27 +7587,6 @@ asmlinkage __visible void __sched notrace preempt_schedule(void) NOKPROBE_SYMBOL(preempt_schedule); EXPORT_SYMBOL(preempt_schedule); -#ifdef CONFIG_PREEMPT_DYNAMIC -# ifdef CONFIG_HAVE_PREEMPT_DYNAMIC_CALL -# ifndef preempt_schedule_dynamic_enabled -# define preempt_schedule_dynamic_enabled preempt_schedule -# define preempt_schedule_dynamic_disabled NULL -# endif -DEFINE_STATIC_CALL(preempt_schedule, preempt_schedule_dynamic_enabled); -EXPORT_STATIC_CALL_TRAMP(preempt_schedule); -# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule); -void __sched notrace dynamic_preempt_schedule(void) -{ - if (!static_branch_unlikely(&sk_dynamic_preempt_schedule)) - return; - preempt_schedule(); -} -NOKPROBE_SYMBOL(dynamic_preempt_schedule); -EXPORT_SYMBOL(dynamic_preempt_schedule); -# endif -#endif /* CONFIG_PREEMPT_DYNAMIC */ - /** * preempt_schedule_notrace - preempt_schedule called by tracing * @@ -7562,27 +7639,6 @@ asmlinkage __visible void __sched notrace preempt_schedule_notrace(void) } EXPORT_SYMBOL_GPL(preempt_schedule_notrace); -#ifdef CONFIG_PREEMPT_DYNAMIC -# if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) -# ifndef preempt_schedule_notrace_dynamic_enabled -# define preempt_schedule_notrace_dynamic_enabled preempt_schedule_notrace -# define preempt_schedule_notrace_dynamic_disabled NULL -# endif -DEFINE_STATIC_CALL(preempt_schedule_notrace, preempt_schedule_notrace_dynamic_enabled); -EXPORT_STATIC_CALL_TRAMP(preempt_schedule_notrace); -# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule_notrace); -void __sched notrace dynamic_preempt_schedule_notrace(void) -{ - if (!static_branch_unlikely(&sk_dynamic_preempt_schedule_notrace)) - return; - preempt_schedule_notrace(); -} -NOKPROBE_SYMBOL(dynamic_preempt_schedule_notrace); -EXPORT_SYMBOL(dynamic_preempt_schedule_notrace); -# endif -#endif - #endif /* CONFIG_PREEMPTION */ /* @@ -7799,7 +7855,7 @@ out_unlock: } #endif /* CONFIG_RT_MUTEXES */ -#if !defined(CONFIG_PREEMPTION) || defined(CONFIG_PREEMPT_DYNAMIC) +#if !defined(CONFIG_PREEMPTION) int __sched __cond_resched(void) { if (should_resched(0) && !irqs_disabled()) { @@ -7827,38 +7883,6 @@ int __sched __cond_resched(void) EXPORT_SYMBOL(__cond_resched); #endif -#ifdef CONFIG_PREEMPT_DYNAMIC -# ifdef CONFIG_HAVE_PREEMPT_DYNAMIC_CALL -# define cond_resched_dynamic_enabled __cond_resched -# define cond_resched_dynamic_disabled ((void *)&__static_call_return0) -DEFINE_STATIC_CALL_RET0(cond_resched, __cond_resched); -EXPORT_STATIC_CALL_TRAMP(cond_resched); - -# define might_resched_dynamic_enabled __cond_resched -# define might_resched_dynamic_disabled ((void *)&__static_call_return0) -DEFINE_STATIC_CALL_RET0(might_resched, __cond_resched); -EXPORT_STATIC_CALL_TRAMP(might_resched); -# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -static DEFINE_STATIC_KEY_FALSE(sk_dynamic_cond_resched); -int __sched dynamic_cond_resched(void) -{ - if (!static_branch_unlikely(&sk_dynamic_cond_resched)) - return 0; - return __cond_resched(); -} -EXPORT_SYMBOL(dynamic_cond_resched); - -static DEFINE_STATIC_KEY_FALSE(sk_dynamic_might_resched); -int __sched dynamic_might_resched(void) -{ - if (!static_branch_unlikely(&sk_dynamic_might_resched)) - return 0; - return __cond_resched(); -} -EXPORT_SYMBOL(dynamic_might_resched); -# endif -#endif /* CONFIG_PREEMPT_DYNAMIC */ - /* * __cond_resched_lock() - if a reschedule is pending, drop the given lock, * call schedule, and on return reacquire the lock. @@ -7928,50 +7952,21 @@ EXPORT_SYMBOL(__cond_resched_rwlock_write); # endif /* - * SC:cond_resched - * SC:might_resched - * SC:preempt_schedule - * SC:preempt_schedule_notrace - * SC:irqentry_exit_cond_resched - * - * * NONE: - * cond_resched <- __cond_resched - * might_resched <- RET0 - * preempt_schedule <- NOP - * preempt_schedule_notrace <- NOP - * irqentry_exit_cond_resched <- NOP - * dynamic_preempt_lazy <- false + * (unselectable) * * VOLUNTARY: - * cond_resched <- __cond_resched - * might_resched <- __cond_resched - * preempt_schedule <- NOP - * preempt_schedule_notrace <- NOP - * irqentry_exit_cond_resched <- NOP - * dynamic_preempt_lazy <- false + * (unselectable) * * FULL: - * cond_resched <- RET0 - * might_resched <- RET0 - * preempt_schedule <- preempt_schedule - * preempt_schedule_notrace <- preempt_schedule_notrace - * irqentry_exit_cond_resched <- irqentry_exit_cond_resched * dynamic_preempt_lazy <- false * * LAZY: - * cond_resched <- RET0 - * might_resched <- RET0 - * preempt_schedule <- preempt_schedule - * preempt_schedule_notrace <- preempt_schedule_notrace - * irqentry_exit_cond_resched <- irqentry_exit_cond_resched * dynamic_preempt_lazy <- true */ enum { preempt_dynamic_undefined = -1, - preempt_dynamic_none, - preempt_dynamic_voluntary, preempt_dynamic_full, preempt_dynamic_lazy, }; @@ -7980,21 +7975,11 @@ int preempt_dynamic_mode = preempt_dynamic_undefined; int sched_dynamic_mode(const char *str) { -# if !(defined(CONFIG_PREEMPT_RT) || defined(CONFIG_ARCH_HAS_PREEMPT_LAZY)) - if (!strcmp(str, "none")) - return preempt_dynamic_none; - - if (!strcmp(str, "voluntary")) - return preempt_dynamic_voluntary; -# endif - if (!strcmp(str, "full")) return preempt_dynamic_full; -# ifdef CONFIG_ARCH_HAS_PREEMPT_LAZY if (!strcmp(str, "lazy")) return preempt_dynamic_lazy; -# endif return -EINVAL; } @@ -8002,71 +7987,18 @@ int sched_dynamic_mode(const char *str) # define preempt_dynamic_key_enable(f) static_key_enable(&sk_dynamic_##f.key) # define preempt_dynamic_key_disable(f) static_key_disable(&sk_dynamic_##f.key) -# if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL) -# define preempt_dynamic_enable(f) static_call_update(f, f##_dynamic_enabled) -# define preempt_dynamic_disable(f) static_call_update(f, f##_dynamic_disabled) -# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY) -# define preempt_dynamic_enable(f) preempt_dynamic_key_enable(f) -# define preempt_dynamic_disable(f) preempt_dynamic_key_disable(f) -# else -# error "Unsupported PREEMPT_DYNAMIC mechanism" -# endif - static DEFINE_MUTEX(sched_dynamic_mutex); static void __sched_dynamic_update(int mode) { - /* - * Avoid {NONE,VOLUNTARY} -> FULL transitions from ever ending up in - * the ZERO state, which is invalid. - */ - preempt_dynamic_enable(cond_resched); - preempt_dynamic_enable(might_resched); - preempt_dynamic_enable(preempt_schedule); - preempt_dynamic_enable(preempt_schedule_notrace); - preempt_dynamic_enable(irqentry_exit_cond_resched); - preempt_dynamic_key_disable(preempt_lazy); - switch (mode) { - case preempt_dynamic_none: - preempt_dynamic_enable(cond_resched); - preempt_dynamic_disable(might_resched); - preempt_dynamic_disable(preempt_schedule); - preempt_dynamic_disable(preempt_schedule_notrace); - preempt_dynamic_disable(irqentry_exit_cond_resched); - preempt_dynamic_key_disable(preempt_lazy); - if (mode != preempt_dynamic_mode) - pr_info("Dynamic Preempt: none\n"); - break; - - case preempt_dynamic_voluntary: - preempt_dynamic_enable(cond_resched); - preempt_dynamic_enable(might_resched); - preempt_dynamic_disable(preempt_schedule); - preempt_dynamic_disable(preempt_schedule_notrace); - preempt_dynamic_disable(irqentry_exit_cond_resched); - preempt_dynamic_key_disable(preempt_lazy); - if (mode != preempt_dynamic_mode) - pr_info("Dynamic Preempt: voluntary\n"); - break; - case preempt_dynamic_full: - preempt_dynamic_disable(cond_resched); - preempt_dynamic_disable(might_resched); - preempt_dynamic_enable(preempt_schedule); - preempt_dynamic_enable(preempt_schedule_notrace); - preempt_dynamic_enable(irqentry_exit_cond_resched); preempt_dynamic_key_disable(preempt_lazy); if (mode != preempt_dynamic_mode) pr_info("Dynamic Preempt: full\n"); break; case preempt_dynamic_lazy: - preempt_dynamic_disable(cond_resched); - preempt_dynamic_disable(might_resched); - preempt_dynamic_enable(preempt_schedule); - preempt_dynamic_enable(preempt_schedule_notrace); - preempt_dynamic_enable(irqentry_exit_cond_resched); preempt_dynamic_key_enable(preempt_lazy); if (mode != preempt_dynamic_mode) pr_info("Dynamic Preempt: lazy\n"); @@ -8099,11 +8031,7 @@ __setup("preempt=", setup_preempt_mode); static void __init preempt_dynamic_init(void) { if (preempt_dynamic_mode == preempt_dynamic_undefined) { - if (IS_ENABLED(CONFIG_PREEMPT_NONE)) { - sched_dynamic_update(preempt_dynamic_none); - } else if (IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY)) { - sched_dynamic_update(preempt_dynamic_voluntary); - } else if (IS_ENABLED(CONFIG_PREEMPT_LAZY)) { + if (IS_ENABLED(CONFIG_PREEMPT_LAZY)) { sched_dynamic_update(preempt_dynamic_lazy); } else { /* Default static call setting, nothing to do */ @@ -8123,8 +8051,6 @@ static void __init preempt_dynamic_init(void) } \ EXPORT_SYMBOL_GPL(preempt_model_##mode) -PREEMPT_MODEL_ACCESSOR(none); -PREEMPT_MODEL_ACCESSOR(voluntary); PREEMPT_MODEL_ACCESSOR(full); PREEMPT_MODEL_ACCESSOR(lazy); @@ -8137,7 +8063,7 @@ static inline void preempt_dynamic_init(void) { } #endif /* CONFIG_PREEMPT_DYNAMIC */ const char *preempt_modes[] = { - "none", "voluntary", "full", "lazy", NULL, + "full", "lazy", NULL, }; const char *preempt_model_str(void) @@ -8759,6 +8685,9 @@ int sched_cpu_activate(unsigned int cpu) */ sched_set_rq_online(rq, cpu); + /* preferred is subset of active and follows its state */ + set_cpu_preferred(cpu, true); + return 0; } @@ -8772,6 +8701,8 @@ int sched_cpu_deactivate(unsigned int cpu) if (ret) return ret; + set_cpu_preferred(cpu, false); + /* * Remove CPU from nohz.idle_cpus_mask to prevent participating in * load balancing when not active @@ -11349,3 +11280,88 @@ void sched_change_end(struct sched_change_ctx *ctx) p->sched_class->prio_changed(rq, p, ctx->prio); } } + +#ifdef CONFIG_PREFERRED_CPU +static DEFINE_PER_CPU(struct cpu_stop_work, npc_push_task_work); + +static int sched_non_preferred_cpu_push_stop(void *arg) +{ + struct task_struct *p = arg; + struct rq *rq = this_rq(); + struct rq_flags rf; + int cpu; + + if (cpu_preferred(rq->cpu)) { + scoped_guard(rq_lock_irqsave, rq) + rq->npc_push_work_pending = false; + put_task_struct(p); + return 0; + } + + scoped_guard (raw_spinlock_irq, &p->pi_lock) { + /* + * select_fallback_rq() may acquire the rq lock in case of + * fallback. So call it before grabbing rq lock. If the task + * migrates to another CPU before the rq lock is acquired, + * subsequent validation of task's current rq will help to + * safely bail out. + */ + cpu = select_fallback_rq(rq->cpu, p); + rq_lock(rq, &rf); + rq->npc_push_work_pending = false; + update_rq_clock(rq); + context_unsafe_alias(rq); + + if (task_rq(p) == rq && task_on_rq_queued(p)) { + struct rq *dest_rq = __migrate_task(rq, &rf, p, cpu); + + if (rq != dest_rq) + schedstat_inc(p->stats.nr_migrations_cpu_non_preferred); + rq = dest_rq; + } + rq_unlock(rq, &rf); + } + + put_task_struct(p); + return 0; +} + +/* + * Push the current task running on non-preferred CPU(npc). + * Using this non preferred CPU will lead to more contention + * in the host. So it is better not to use this CPU. + * + * Since task is running, call a stopper to push the task out. This is + * similar to how task moves during hotplug. In select_fallback_rq() a + * preferred CPU will be chosen and henceforth task shouldn't come back to + * this CPU again. + * + * Works for FAIR class only. + * + * If task is affined only on non-preferred CPUs, no point in moving it out. + */ +void sched_push_current_non_preferred_cpu(struct rq *rq) +{ + struct task_struct *push_task = rq->curr; + + scoped_guard(rq_lock, rq) { + /* Push the task if its explicit affinity allows */ + if (!task_can_migrate_to_preferred(push_task, rq->cpu)) + return; + + /* There is already a stopper thread. Don't race with it. */ + if (rq->npc_push_work_pending) + return; + + if (is_migration_disabled(push_task)) + return; + + rq->npc_push_work_pending = true; + } + + /* sched_tick runs with interrupts disabled. */ + get_task_struct(push_task); + stop_one_cpu_nowait(rq->cpu, sched_non_preferred_cpu_push_stop, + push_task, this_cpu_ptr(&npc_push_task_work)); +} +#endif diff --git a/kernel/sched/cputime.c b/kernel/sched/cputime.c index 06bddaa738e5..f16970ca81d0 100644 --- a/kernel/sched/cputime.c +++ b/kernel/sched/cputime.c @@ -255,7 +255,7 @@ void __account_forceidle_time(struct task_struct *p, u64 delta) * occasion account more time than the calling functions think elapsed. */ #ifdef CONFIG_PARAVIRT -struct static_key paravirt_steal_enabled; +DEFINE_STATIC_KEY_FALSE(paravirt_steal_enabled); #ifdef CONFIG_HAVE_PV_STEAL_CLOCK_GEN static u64 native_steal_clock(int cpu) @@ -270,7 +270,7 @@ DEFINE_STATIC_CALL(pv_steal_clock, native_steal_clock); static __always_inline u64 steal_account_process_time(u64 maxtime) { #ifdef CONFIG_PARAVIRT - if (static_key_false(¶virt_steal_enabled)) { + if (static_branch_unlikely(¶virt_steal_enabled)) { u64 steal; steal = paravirt_steal_clock(smp_processor_id()); diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c index 0663c00c41c0..c0ebdcde5fe5 100644 --- a/kernel/sched/deadline.c +++ b/kernel/sched/deadline.c @@ -1097,7 +1097,7 @@ static int start_dl_timer(struct sched_dl_entity *dl_se) * chosen as the deadline is too small, don't even try to * start the timer in the past! */ - if (ktime_us_delta(act, now) < 0) + if (ktime_before(act, now)) return 0; /* @@ -2773,11 +2773,14 @@ static void start_hrtick_dl(struct rq *rq, struct sched_dl_entity *dl_se) * DL keeps current in tree, because ->deadline is not typically changed while * a task is runnable. */ -static void set_next_task_dl(struct rq *rq, struct task_struct *p, bool first) +static void set_next_task_dl(struct rq *rq, struct task_struct *p, enum snt_e type) { struct sched_dl_entity *dl_se = &p->dl; struct dl_rq *dl_rq = &rq->dl; + if (type == SNT_REPICK) + return; + p->se.exec_start = rq_clock_task(rq); if (on_dl_rq(&p->dl)) update_stats_wait_end_dl(dl_rq, dl_se); @@ -2788,7 +2791,7 @@ static void set_next_task_dl(struct rq *rq, struct task_struct *p, bool first) WARN_ON_ONCE(dl_rq->curr); dl_rq->curr = dl_se; - if (!first) + if (type != SNT_PICK) return; if (rq->donor->sched_class != &dl_sched_class) diff --git a/kernel/sched/debug.c b/kernel/sched/debug.c index 72236db67983..e6a3b516c703 100644 --- a/kernel/sched/debug.c +++ b/kernel/sched/debug.c @@ -73,13 +73,13 @@ static int sched_feat_show(struct seq_file *m, void *v) #ifdef CONFIG_JUMP_LABEL -#define jump_label_key__true STATIC_KEY_INIT_TRUE -#define jump_label_key__false STATIC_KEY_INIT_FALSE +#define jump_label_key__true { .key_true = STATIC_KEY_TRUE_INIT } +#define jump_label_key__false { .key_false = STATIC_KEY_FALSE_INIT } #define SCHED_FEAT(name, enabled) \ jump_label_key__##enabled , -struct static_key sched_feat_keys[__SCHED_FEAT_NR] = { +union sched_feat_key sched_feat_keys[__SCHED_FEAT_NR] = { #include "features.h" }; @@ -87,12 +87,12 @@ struct static_key sched_feat_keys[__SCHED_FEAT_NR] = { static void sched_feat_disable(int i) { - static_key_disable_cpuslocked(&sched_feat_keys[i]); + static_branch_disable_cpuslocked(&sched_feat_keys[i].key_true); } static void sched_feat_enable(int i) { - static_key_enable_cpuslocked(&sched_feat_keys[i]); + static_branch_enable_cpuslocked(&sched_feat_keys[i].key_false); } #else /* !CONFIG_JUMP_LABEL: */ static void sched_feat_disable(int i) { }; @@ -280,16 +280,10 @@ static ssize_t sched_dynamic_write(struct file *filp, const char __user *ubuf, static int sched_dynamic_show(struct seq_file *m, void *v) { - int i = (IS_ENABLED(CONFIG_PREEMPT_RT) || IS_ENABLED(CONFIG_ARCH_HAS_PREEMPT_LAZY)) * 2; int mode = READ_ONCE(preempt_dynamic_mode); - int j; - /* Count entries in NULL terminated preempt_modes */ - for (j = 0; preempt_modes[j]; j++) - ; - j -= !IS_ENABLED(CONFIG_ARCH_HAS_PREEMPT_LAZY); - - for (; i < j; i++) { + /* Stop at NULL terminator */ + for (int i = 0; preempt_modes[i]; i++) { if (mode == i) seq_puts(m, "("); seq_puts(m, preempt_modes[i]); @@ -1446,6 +1440,7 @@ void proc_sched_show_task(struct task_struct *p, struct pid_namespace *ns, P_SCHEDSTAT(nr_failed_migrations_running); P_SCHEDSTAT(nr_failed_migrations_hot); P_SCHEDSTAT(nr_forced_migrations); + P_SCHEDSTAT(nr_migrations_cpu_non_preferred); P_SCHEDSTAT(nr_wakeups); P_SCHEDSTAT(nr_wakeups_sync); P_SCHEDSTAT(nr_wakeups_migrate); diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c index e56c3c95018f..aed5286b82aa 100644 --- a/kernel/sched/ext/ext.c +++ b/kernel/sched/ext/ext.c @@ -24,6 +24,11 @@ DEFINE_RAW_SPINLOCK(scx_sched_lock); +bool scx_allow_proxy_exec(const struct task_struct *p) +{ + return true; +} + /* * NOTE: sched_ext is in the process of growing multiple scheduler support and * scx_root usage is in a transitional state. Naked dereferences are safe if the @@ -1087,6 +1092,10 @@ static void schedule_deferred_locked(struct rq *rq) schedule_deferred(rq); } +void scx_proxy_reenqueue_retry(struct rq *rq, struct task_struct *next) +{ +} + void schedule_dsq_reenq(struct scx_sched *sch, struct scx_dispatch_q *dsq, u64 reenq_flags, struct rq *locked_rq) { @@ -3021,10 +3030,13 @@ has_tasks: return verdict; } -static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first) +static void set_next_task_scx(struct rq *rq, struct task_struct *p, enum snt_e type) { struct scx_sched *sch = scx_task_sched(p); + if (type == SNT_REPICK) + return; + if (p->scx.flags & SCX_TASK_QUEUED) { /* * Core-sched might decide to execute @p before it is @@ -3082,6 +3094,10 @@ static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first) } } +void scx_proxy_donor_start(struct rq *rq) +{ +} + static enum scx_cpu_preempt_reason preempt_reason_from_class(const struct sched_class *class) { diff --git a/kernel/sched/ext/ext.h b/kernel/sched/ext/ext.h index 0b7fc46aee08..3cfbfeb1bf9d 100644 --- a/kernel/sched/ext/ext.h +++ b/kernel/sched/ext/ext.h @@ -20,6 +20,9 @@ void scx_rq_deactivate(struct rq *rq); int scx_check_setscheduler(struct task_struct *p, int policy); bool task_should_scx(int policy); bool scx_allow_ttwu_queue(const struct task_struct *p); +bool scx_allow_proxy_exec(const struct task_struct *p); +void scx_proxy_donor_start(struct rq *rq); +void scx_proxy_reenqueue_retry(struct rq *rq, struct task_struct *next); void init_sched_ext_class(void); static inline u32 scx_cpuperf_target(s32 cpu) @@ -54,6 +57,10 @@ static inline void scx_rq_deactivate(struct rq *rq) {} static inline int scx_check_setscheduler(struct task_struct *p, int policy) { return 0; } static inline bool task_on_scx(const struct task_struct *p) { return false; } static inline bool scx_allow_ttwu_queue(const struct task_struct *p) { return true; } +static inline bool scx_allow_proxy_exec(const struct task_struct *p) { return true; } +static inline void scx_proxy_donor_start(struct rq *rq) {} +static inline void scx_proxy_reenqueue_retry(struct rq *rq, + struct task_struct *next) {} static inline void init_sched_ext_class(void) {} #endif /* CONFIG_SCHED_CLASS_EXT */ diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 8d38c3b7d792..56f4ab6d9ada 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -23,6 +23,7 @@ #include <linux/energy_model.h> #include <linux/mmap_lock.h> #include <linux/jiffies.h> +#include <linux/math.h> #include <linux/mm_api.h> #include <linux/highmem.h> #include <linux/hrtimer.h> @@ -819,12 +820,6 @@ static u64 ineligible_vruntime(struct cfs_rq *cfs_rq) if (curr && !curr->on_rq) curr = NULL; - /* - * This is called from set_next_task_fair(.first=true) / - * set_protect_slice() so curr had better be set and on_rq. - */ - WARN_ON_ONCE(!curr); - if (weight) { s64 runtime = cfs_rq->sum_w_vruntime; @@ -1136,10 +1131,9 @@ static inline void set_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity /* If there are shorter slices than se's one */ if (slice != se->slice) { + vprot = min_vruntime(vprot, se->vruntime + calc_delta_fair(slice, se)); if (sched_feat(PREEMPT_SHORT)) vprot = min_vruntime(vprot, ineligible_vruntime(cfs_rq)); - else - vprot = min_vruntime(vprot, se->vruntime + calc_delta_fair(slice, se)); } se->vprot = vprot; @@ -1147,10 +1141,19 @@ static inline void set_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity static inline void update_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity *se) { - u64 slice = cfs_rq_min_slice(cfs_rq); u64 vruntime = min_vruntime(se->vruntime, avg_vruntime(cfs_rq)); + u64 slice = normalized_sysctl_sched_base_slice; + u64 vprot; - se->vprot = min_vruntime(se->vprot, vruntime + calc_delta_fair(slice, se)); + if (sched_feat(RUN_TO_PARITY)) + slice = cfs_rq_min_slice(cfs_rq); + + vprot = min_vruntime(se->vprot, vruntime + calc_delta_fair(slice, se)); + + if (sched_feat(PREEMPT_SHORT) && slice != se->slice) + vprot = min_vruntime(vprot, ineligible_vruntime(cfs_rq)); + + se->vprot = vprot; } static inline bool protect_slice(struct sched_entity *se) @@ -3712,7 +3715,7 @@ static void update_task_scan_period(struct task_struct *p, p->mm->numa_next_scan = jiffies + msecs_to_jiffies(p->numa_scan_period); - return; + goto out; } /* @@ -3756,7 +3759,10 @@ static void update_task_scan_period(struct task_struct *p, p->numa_scan_period = clamp(p->numa_scan_period + diff, task_scan_min(p), task_scan_max(p)); - memset(p->numa_faults_locality, 0, sizeof(p->numa_faults_locality)); + +out: + memset(p->numa_faults_locality, 0, + sizeof(p->numa_faults_locality)); } /* @@ -8208,7 +8214,6 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags) struct sched_entity *se = &p->se; struct cfs_rq *cfs_rq = &rq->cfs; unsigned long weight; - bool curr; if (task_is_throttled(p) && enqueue_throttled_task(p)) return; @@ -8237,23 +8242,14 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags) if (p->in_iowait) cpufreq_update_util(rq, SCHED_CPUFREQ_IOWAIT); - /* - * XXX comment on the curr thing - */ - curr = (cfs_rq->curr == se); - if (curr) - place_entity(cfs_rq, se, flags); if (se->on_rq && se->sched_delayed) requeue_delayed_entity(cfs_rq, se); weight = enqueue_hierarchy(p, flags); - - if (!curr) { - reweight_eevdf(cfs_rq, se, weight, false); - place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED); - __enqueue_entity(cfs_rq, se); - } + reweight_eevdf(cfs_rq, se, weight, false); + place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED); + __enqueue_entity(cfs_rq, se); if (!rq_h_nr_queued && rq->cfs.h_nr_queued) dl_server_start(&rq->fair_server); @@ -8673,8 +8669,8 @@ static int sched_balance_find_dst_group_cpu(struct sched_group *group, struct task_struct *p, int this_cpu) { unsigned long load, min_load = ULONG_MAX; - unsigned int min_exit_latency = UINT_MAX; - u64 latest_idle_timestamp = 0; + u64 min_exit_latency = U64_MAX; + unsigned int nr_candidates = 0; int least_loaded_cpu = this_cpu; int shallowest_idle_cpu = -1; int i; @@ -8695,24 +8691,16 @@ sched_balance_find_dst_group_cpu(struct sched_group *group, struct task_struct * if (available_idle_cpu(i)) { struct cpuidle_state *idle = idle_get_state(rq); - if (idle && idle->exit_latency < min_exit_latency) { - /* - * We give priority to a CPU whose idle state - * has the smallest exit latency irrespective - * of any idle timestamp. - */ - min_exit_latency = idle->exit_latency; - latest_idle_timestamp = rq->idle_stamp; - shallowest_idle_cpu = i; - } else if ((!idle || idle->exit_latency == min_exit_latency) && - rq->idle_stamp > latest_idle_timestamp) { - /* - * If equal or no active idle state, then - * the most recently idled CPU might have - * a warmer cache. - */ - latest_idle_timestamp = rq->idle_stamp; + u64 exit_latency = idle ? idle->exit_latency : U64_MAX; + + if (shallowest_idle_cpu == -1 || exit_latency < min_exit_latency) { + min_exit_latency = exit_latency; shallowest_idle_cpu = i; + nr_candidates = 1; + } else if (exit_latency == min_exit_latency) { + nr_candidates++; + if (!reciprocal_scale(sched_rng(), nr_candidates)) + shallowest_idle_cpu = i; } } else if (shallowest_idle_cpu == -1) { load = cpu_load(cpu_rq(i)); @@ -10071,8 +10059,14 @@ static inline bool set_preempt_buddy(struct cfs_rq *cfs_rq, struct sched_entity static inline bool set_short_buddy(struct cfs_rq *cfs_rq, struct sched_entity *pse) { - if (cfs_rq->next && cfs_rq->next->slice < pse->slice) - return false; + if (cfs_rq->next) { + if (cfs_rq->next->slice < pse->slice) + return false; + + if (cfs_rq->next->slice == pse->slice && + entity_before(cfs_rq->next, pse)) + return false; + } set_next_buddy(cfs_rq, pse); return true; @@ -11438,21 +11432,7 @@ next: */ static void attach_tasks(struct lb_env *env) { - struct list_head *tasks = &env->tasks; - struct task_struct *p; - struct rq_flags rf; - - rq_lock(env->dst_rq, &rf); - update_rq_clock(env->dst_rq); - - while (!list_empty(tasks)) { - p = list_first_entry(tasks, struct task_struct, se.group_node); - list_del_init(&p->se.group_node); - - attach_task(env->dst_rq, p); - } - - rq_unlock(env->dst_rq, &rf); + __attach_tasks(env->dst_rq, &env->tasks); } #ifdef CONFIG_NO_HZ_COMMON @@ -13745,7 +13725,7 @@ static int sched_balance_rq(int this_cpu, struct rq *this_rq, }; bool need_unlock = false; - cpumask_and(cpus, sched_domain_span(sd), cpu_active_mask); + cpumask_and(cpus, sched_domain_span(sd), cpu_preferred_mask); schedstat_inc(sd->lb_count[idle]); @@ -14870,10 +14850,8 @@ static int sched_balance_newidle(struct rq *this_rq, struct rq_flags *rf) */ this_rq->idle_stamp = rq_clock(this_rq); - /* - * Do not pull tasks towards !active CPUs... - */ - if (!cpu_active(this_cpu)) + /* Do not pull tasks towards !preferred CPUs */ + if (!cpu_preferred(this_cpu)) return 0; /* @@ -15513,14 +15491,18 @@ static void switched_to_fair(struct rq *rq, struct task_struct *p) } } -static void set_next_task_fair(struct rq *rq, struct task_struct *p, bool first) +static void set_next_task_fair(struct rq *rq, struct task_struct *p, enum snt_e type) { struct sched_entity *se = &p->se; - bool throttled = false; struct cfs_rq *cfs_rq = &rq->cfs; unsigned long weight = NICE_0_LOAD; + bool first = type == SNT_PICK; + bool throttled = false; bool on_rq = se->on_rq; + if (type == SNT_REPICK) + goto repick; + clear_buddies(cfs_rq, se); if (on_rq) @@ -15564,11 +15546,18 @@ static void set_next_task_fair(struct rq *rq, struct task_struct *p, bool first) WARN_ON_ONCE(se->sched_delayed); - if (hrtick_enabled_fair(rq)) - hrtick_start_fair(rq, p); - update_misfit_status(p, rq); sched_fair_update_stop_tick(rq, p); + +repick: + /* + * A same-task repick skips put_prev_task_fair(), but + * pick_task_fair() refreshed the entity hrtick_start_fair() reads + * before selecting it again. rq->cfs.curr identifies that entity, + * including with group scheduling. + */ + if (hrtick_enabled_fair(rq)) + hrtick_start_fair(rq, p); } void init_cfs_rq(struct cfs_rq *cfs_rq) diff --git a/kernel/sched/idle.c b/kernel/sched/idle.c index eb73b65ce6c4..76f3c84ca684 100644 --- a/kernel/sched/idle.c +++ b/kernel/sched/idle.c @@ -487,8 +487,11 @@ static void put_prev_task_idle(struct rq *rq, struct task_struct *prev, struct t update_rq_avg_idle(rq); } -static void set_next_task_idle(struct rq *rq, struct task_struct *next, bool first) +static void set_next_task_idle(struct rq *rq, struct task_struct *next, enum snt_e type) { + if (type == SNT_REPICK) + return; + update_idle_core(rq); scx_update_idle(rq, true, true); schedstat_inc(rq->sched_goidle); diff --git a/kernel/sched/rt.c b/kernel/sched/rt.c index 85303add726d..1535046a23ff 100644 --- a/kernel/sched/rt.c +++ b/kernel/sched/rt.c @@ -1654,11 +1654,14 @@ static void wakeup_preempt_rt(struct rq *rq, struct task_struct *p, int flags) check_preempt_equal_prio(rq, p); } -static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, bool first) +static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, enum snt_e type) { struct sched_rt_entity *rt_se = &p->rt; struct rt_rq *rt_rq = &rq->rt; + if (type == SNT_REPICK) + return; + p->se.exec_start = rq_clock_task(rq); if (on_rt_rq(&p->rt)) update_stats_wait_end_rt(rt_rq, rt_se); @@ -1666,7 +1669,7 @@ static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, bool f /* The running task is never eligible for pushing */ dequeue_pushable_task(rq, p); - if (!first) + if (type != SNT_PICK) return; /* diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index e656c7059bf8..7d2ec527b8a2 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -1326,6 +1326,9 @@ struct rq { #ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING u64 prev_steal_time_rq; #endif +#ifdef CONFIG_PREFERRED_CPU + bool npc_push_work_pending; +#endif /* calc_load related fields */ unsigned long calc_load_update; @@ -1371,7 +1374,6 @@ struct rq { struct task_struct *core_pick; struct sched_dl_entity *core_dl_server; unsigned int core_enabled; - unsigned int core_sched_seq; struct rb_root core_tree; /* shared state -- careful with sched_core_cpu_deactivate() */ @@ -2447,16 +2449,25 @@ extern __read_mostly unsigned int sysctl_sched_features; #ifdef CONFIG_JUMP_LABEL -#define SCHED_FEAT(name, enabled) \ -static __always_inline bool static_branch_##name(struct static_key *key) \ -{ \ - return static_key_##enabled(key); \ +union sched_feat_key { + struct static_key_true key_true; + struct static_key_false key_false; +}; + +#define sched_feat_branch_true(key) static_branch_likely(&(key)->key_true) +#define sched_feat_branch_false(key) static_branch_unlikely(&(key)->key_false) + +#define SCHED_FEAT(name, enabled) \ +static __always_inline bool \ +static_branch_##name(union sched_feat_key *key) \ +{ \ + return sched_feat_branch_##enabled(key); \ } #include "features.h" #undef SCHED_FEAT -extern struct static_key sched_feat_keys[__SCHED_FEAT_NR]; +extern union sched_feat_key sched_feat_keys[__SCHED_FEAT_NR]; #define sched_feat(x) (static_branch_##x(&sched_feat_keys[__SCHED_FEAT_##x])) #else /* !CONFIG_JUMP_LABEL: */ @@ -2508,6 +2519,12 @@ static inline bool task_is_blocked(struct task_struct *p) return !!p->blocked_on; } +#ifdef CONFIG_SCHED_PROXY_EXEC +void sched_proxy_block_task(struct rq *rq, struct task_struct *p); +#else +static inline void sched_proxy_block_task(struct rq *rq, struct task_struct *p) {} +#endif + static inline int task_on_cpu(struct rq *rq, struct task_struct *p) { return p->on_cpu; @@ -2527,11 +2544,17 @@ static inline int task_on_rq_migrating(struct task_struct *p) #define WF_EXEC 0x02 /* Wakeup after exec; maps to SD_BALANCE_EXEC */ #define WF_FORK 0x04 /* Wakeup after fork; maps to SD_BALANCE_FORK */ #define WF_TTWU 0x08 /* Wakeup; maps to SD_BALANCE_WAKE */ - -#define WF_SYNC 0x10 /* Waker goes to sleep after wakeup */ +/* + * Hint that the caller expects the waker to sleep soon. + * Scheduler classes may use it for placement or preemption. + * Callers must not rely on it to prevent migration, + * preserve CPU locality or make the wakee run next. + */ +#define WF_SYNC 0x10 #define WF_MIGRATED 0x20 /* Internal use, task got migrated */ #define WF_CURRENT_CPU 0x40 /* Prefer to move the wakee to the current CPU. */ #define WF_RQ_SELECTED 0x80 /* ->select_task_rq() was called */ +#define WF_TTWU_RQ 0x100 /* Wakeup completed through ttwu_runnable() */ static_assert(WF_EXEC == SD_BALANCE_EXEC); static_assert(WF_FORK == SD_BALANCE_FORK); @@ -2621,6 +2644,12 @@ struct affinity_context { extern s64 update_curr_common(struct rq *rq); +enum snt_e { + SNT_NORMAL, /* set_next_task() */ + SNT_PICK, /* put_prev_set_next_task(): prev != next */ + SNT_REPICK, /* put_prev_set_next_task(): prev == next */ +}; + struct sched_class { #ifdef CONFIG_UCLAMP_TASK @@ -2678,7 +2707,7 @@ struct sched_class { * __schedule: rq->lock */ void (*put_prev_task)(struct rq *rq, struct task_struct *p, struct task_struct *next); - void (*set_next_task)(struct rq *rq, struct task_struct *p, bool first); + void (*set_next_task)(struct rq *rq, struct task_struct *p, enum snt_e type); /* * select_task_rq: p->pi_lock @@ -2781,7 +2810,7 @@ static inline void put_prev_task(struct rq *rq, struct task_struct *prev) static inline void set_next_task(struct rq *rq, struct task_struct *next) { - next->sched_class->set_next_task(rq, next, false); + next->sched_class->set_next_task(rq, next, SNT_NORMAL); } static inline void @@ -2802,11 +2831,13 @@ static inline void put_prev_set_next_task(struct rq *rq, __put_prev_set_next_dl_server(rq, prev, next); - if (next == prev) + if (next == prev) { + next->sched_class->set_next_task(rq, next, SNT_REPICK); return; + } prev->sched_class->put_prev_task(rq, prev, next); - next->sched_class->set_next_task(rq, next, true); + next->sched_class->set_next_task(rq, next, SNT_PICK); } /* @@ -3139,6 +3170,25 @@ static inline void attach_one_task(struct rq *rq, struct task_struct *p) attach_task(rq, p); } +/* + * __attach_tasks() - attaches a list of tasks (using se.group_node) to + * the new rq + */ +static inline void __attach_tasks(struct rq *rq, struct list_head *tasks) +{ + guard(rq_lock)(rq); + update_rq_clock(rq); + + while (!list_empty(tasks)) { + struct task_struct *p; + + p = list_first_entry(tasks, struct task_struct, se.group_node); + list_del_init(&p->se.group_node); + + attach_task(rq, p); + } +} + #ifdef CONFIG_PREEMPT_RT # define SCHED_NR_MIGRATE_BREAK 8 #else @@ -4252,4 +4302,10 @@ DEFINE_CLASS_IS_UNCONDITIONAL(sched_change) #include "ext/ext.h" +#ifdef CONFIG_PREFERRED_CPU +void sched_push_current_non_preferred_cpu(struct rq *rq); +#else /* !CONFIG_PREFERRED_CPU */ +static inline void sched_push_current_non_preferred_cpu(struct rq *rq) { } +#endif + #endif /* _KERNEL_SCHED_SCHED_H */ diff --git a/kernel/sched/stop_task.c b/kernel/sched/stop_task.c index c909ca0d8c87..1e0109ec36b3 100644 --- a/kernel/sched/stop_task.c +++ b/kernel/sched/stop_task.c @@ -27,8 +27,11 @@ wakeup_preempt_stop(struct rq *rq, struct task_struct *p, int flags) /* we're never preempted */ } -static void set_next_task_stop(struct rq *rq, struct task_struct *stop, bool first) +static void set_next_task_stop(struct rq *rq, struct task_struct *stop, enum snt_e type) { + if (type == SNT_REPICK) + return; + stop->se.exec_start = rq_clock_task(rq); } diff --git a/kernel/sched/wait.c b/kernel/sched/wait.c index d033f600f48c..477e4bf9c01e 100644 --- a/kernel/sched/wait.c +++ b/kernel/sched/wait.c @@ -174,15 +174,11 @@ EXPORT_SYMBOL_GPL(__wake_up_locked_key); * @mode: which threads * @key: opaque value to be passed to wakeup targets * - * The sync wakeup differs that the waker knows that it will schedule - * away soon, so while the target thread will be woken up, it will not - * be migrated to another CPU - ie. the two threads are 'synchronized' - * with each other. This can prevent needless bouncing between CPUs. + * Passes WF_SYNC to waitqueue wake functions. The default wake function + * forwards it to the scheduler; see WF_SYNC for the hint's semantics. * - * On UP it can prevent extra preemption. - * - * If this function wakes up a task, it executes a full memory barrier before - * accessing the task state. + * If this function wakes up a task, it executes a full memory barrier + * before accessing the task state. */ void __wake_up_sync_key(struct wait_queue_head *wq_head, unsigned int mode, void *key) @@ -200,15 +196,7 @@ EXPORT_SYMBOL_GPL(__wake_up_sync_key); * @mode: which threads * @key: opaque value to be passed to wakeup targets * - * The sync wakeup differs in that the waker knows that it will schedule - * away soon, so while the target thread will be woken up, it will not - * be migrated to another CPU - ie. the two threads are 'synchronized' - * with each other. This can prevent needless bouncing between CPUs. - * - * On UP it can prevent extra preemption. - * - * If this function wakes up a task, it executes a full memory barrier before - * accessing the task state. + * Same as __wake_up_sync_key(), but called with @wq_head->lock held. */ void __wake_up_locked_sync_key(struct wait_queue_head *wq_head, unsigned int mode, void *key) diff --git a/kernel/time/hrtimer.c b/kernel/time/hrtimer.c index cbf1693c86b3..17dd38a6cee7 100644 --- a/kernel/time/hrtimer.c +++ b/kernel/time/hrtimer.c @@ -780,7 +780,8 @@ static void hrtimer_switch_to_hres(void) return; } base->hres_active = true; - hrtimer_resolution = HIGH_RES_NSEC; + if (hrtimer_resolution != HIGH_RES_NSEC) + hrtimer_resolution = HIGH_RES_NSEC; tick_setup_sched_timer(true); /* "Retrigger" the interrupt to get things going */ @@ -2003,7 +2004,7 @@ bool hrtimer_active(const struct hrtimer *timer) base = READ_ONCE(timer->base); seq = raw_read_seqcount_begin(&base->seq); - if (timer->is_queued || base->running == timer) + if (timer->is_queued || READ_ONCE(base->running) == timer) return true; } while (read_seqcount_retry(&base->seq, seq) || base != READ_ONCE(timer->base)); @@ -2040,7 +2041,7 @@ static void __run_hrtimer(struct hrtimer_cpu_base *cpu_base, struct hrtimer_cloc lockdep_assert_held(&cpu_base->lock); debug_hrtimer_deactivate(timer); - base->running = timer; + WRITE_ONCE(base->running, timer); /* * Separate the ->running assignment from the ->is_queued assignment. @@ -2099,7 +2100,7 @@ static void __run_hrtimer(struct hrtimer_cpu_base *cpu_base, struct hrtimer_cloc raw_write_seqcount_barrier(&base->seq); WARN_ON_ONCE(base->running != timer); - base->running = NULL; + WRITE_ONCE(base->running, NULL); } static void __hrtimer_run_queues(struct hrtimer_cpu_base *cpu_base, ktime_t now, @@ -2323,9 +2324,9 @@ void hrtimer_run_queues(void) static enum hrtimer_restart hrtimer_wakeup(struct hrtimer *timer) { struct hrtimer_sleeper *t = container_of(timer, struct hrtimer_sleeper, timer); - struct task_struct *task = t->task; + struct task_struct *task = hrtimer_sleeper_task_get(t); - t->task = NULL; + hrtimer_sleeper_task_set(t, NULL); if (task) wake_up_process(task); @@ -2354,7 +2355,7 @@ void hrtimer_sleeper_start_expires(struct hrtimer_sleeper *sl, enum hrtimer_mode /* If already expired, clear the task pointer and set current state to running */ if (!hrtimer_start_expires_user(&sl->timer, mode)) { - sl->task = NULL; + hrtimer_sleeper_task_set(sl, NULL); __set_current_state(TASK_RUNNING); } } @@ -2388,7 +2389,7 @@ static void __hrtimer_setup_sleeper(struct hrtimer_sleeper *sl, clockid_t clock_ } __hrtimer_setup(&sl->timer, hrtimer_wakeup, clock_id, mode); - sl->task = current; + hrtimer_sleeper_task_set(sl, current); } /** @@ -2432,17 +2433,17 @@ static int __sched do_nanosleep(struct hrtimer_sleeper *t, enum hrtimer_mode mod set_current_state(TASK_INTERRUPTIBLE|TASK_FREEZABLE); hrtimer_sleeper_start_expires(t, mode); - if (likely(t->task)) + if (likely(hrtimer_sleeper_task_get(t))) schedule(); hrtimer_cancel(&t->timer); mode = HRTIMER_MODE_ABS; - } while (t->task && !signal_pending(current)); + } while (hrtimer_sleeper_task_get(t) && !signal_pending(current)); __set_current_state(TASK_RUNNING); - if (!t->task) + if (!hrtimer_sleeper_task_get(t)) return 0; restart = ¤t->restart_block; diff --git a/kernel/time/posix-cpu-timers.c b/kernel/time/posix-cpu-timers.c index 0bf4fcd969c8..cd75d4bb5b64 100644 --- a/kernel/time/posix-cpu-timers.c +++ b/kernel/time/posix-cpu-timers.c @@ -439,6 +439,38 @@ static void trigger_base_recalc_expires(struct k_itimer *timer, base->nextevt = 0; } +static inline bool cpu_timer_enqueue(struct timerqueue_head *head, + struct cpu_timer *ctmr) +{ + ctmr->head = head; + return timerqueue_add(head, &ctmr->node); +} + +static inline bool cpu_timer_queued(struct cpu_timer *ctmr) +{ + return !!ctmr->head; +} + +static inline bool cpu_timer_dequeue(struct cpu_timer *ctmr) +{ + if (cpu_timer_queued(ctmr)) { + timerqueue_del(ctmr->head, &ctmr->node); + ctmr->head = NULL; + return true; + } + return false; +} + +static inline u64 cpu_timer_getexpires(struct cpu_timer *ctmr) +{ + return ctmr->node.expires; +} + +static inline void cpu_timer_setexpires(struct cpu_timer *ctmr, u64 exp) +{ + ctmr->node.expires = exp; +} + /* * Dequeue the timer and reset the base if it was its earliest expiration. * It makes sure the next tick recalculates the base next expiration so we @@ -607,6 +639,7 @@ static int posix_cpu_timer_del(struct k_itimer *timer) } if (!ret) { + WARN_ON_ONCE(cpu_timer_queued(&timer->it.cpu)); put_pid(timer->it.cpu.pid); timer->it_status = POSIX_TIMER_DISARMED; } @@ -639,18 +672,50 @@ static void cleanup_timers(struct posix_cputimers *pct) cleanup_timerqueue(&pct->bases[CPUCLOCK_SCHED].tqhead); } +static inline void posix_cpu_timers_exit_work(void); + /* - * These are both called with the siglock held, when the current thread - * is being reaped. When the final (leader) thread in the group is reaped, - * posix_cpu_timers_exit_group will be called after posix_cpu_timers_exit. + * Invoked from posixtimer_exit_task() after PF_EXITING was set in tsk::flags or + * from posixtimer_exec_cleanup(). */ -void posix_cpu_timers_exit(struct task_struct *tsk) +void posix_cpu_timers_exit_task(void) { - cleanup_timers(&tsk->posix_cputimers); + posix_cpu_timers_exit_work(); + + guard(spinlock_irq)(¤t->sighand->siglock); + cleanup_timers(¤t->posix_cputimers); } -void posix_cpu_timers_exit_group(struct task_struct *tsk) + +/* + * Invoked from posixtimer_exit_group() after PF_EXITING was set in tsk::flags. + */ +void posix_cpu_timers_exit_group(void) { - cleanup_timers(&tsk->signal->posix_cputimers); + posix_cpu_timers_exit_task(); + + guard(spinlock_irq)(¤t->sighand->siglock); + cleanup_timers(¤t->signal->posix_cputimers); +} + +/* + * This function validates that POSIX CPU timers can be safely enqueued on the + * target task. + * + * Enqueue is allowed when PF_EXITING is not set. If set then it is only allowed + * for process shared timers (type = PIDTYPE_TGID) as long as tsk::signal::flags + * does not have SIGNAL_GROUP_EXIT set. PIDTYPE_PID targets are not allowed at + * all when the task has PF_EXITING set. + * + * This guarantees that after the POSIX timer cleanup in posixtimer_exit() no + * POSIX CPU timers are queued on the task or in case of a group exit on the + * process. + */ +static inline bool task_can_enqueue_timer(struct task_struct *tsk, enum pid_type type) +{ + if (likely(!(tsk->flags & PF_EXITING))) + return true; + + return type == PIDTYPE_TGID && !(tsk->signal->flags & SIGNAL_GROUP_EXIT); } /* @@ -663,7 +728,13 @@ static void arm_timer(struct k_itimer *timer, struct task_struct *p) struct cpu_timer *ctmr = &timer->it.cpu; u64 newexp = cpu_timer_getexpires(ctmr); + lockdep_assert_held(&p->sighand->siglock); + timer->it_status = POSIX_TIMER_ARMED; + + if (unlikely(!task_can_enqueue_timer(p, clock_pid_type(timer->it_clock)))) + return; + if (!cpu_timer_enqueue(&base->tqhead, ctmr)) return; @@ -1201,6 +1272,20 @@ static void posix_cpu_timers_work(struct callback_head *work) mutex_unlock(&cw->mutex); } +static inline void posix_cpu_timers_exit_work(void) +{ + /* Canceling the work is only valid for exit() but not for exec() */ + if (!(current->flags & PF_EXITING)) + return; + /* + * current->flags has PF_EXITING set so this can be done lockless and + * with interrupts enabled as PF_EXITING prevents the interrupt from + * scheduling the work. + */ + if (current->posix_cputimers_work.scheduled) + task_work_cancel(current, ¤t->posix_cputimers_work.work); +} + /* * Invoked from the posix-timer core when a cancel operation failed because * the timer is marked firing. The caller holds rcu_read_lock(), which @@ -1331,6 +1416,8 @@ static inline void __run_posix_cpu_timers(struct task_struct *tsk) lockdep_posixtimer_exit(); } +static inline void posix_cpu_timers_exit_work(void) { } + static void posix_cpu_timer_wait_running(struct k_itimer *timr) { cpu_relax(); @@ -1477,7 +1564,7 @@ void run_posix_cpu_timers(void) * posix_cpu_timer_del() may fail to lock_task_sighand(tsk) and * miss timer->it.cpu.firing != 0. */ - if (tsk->exit_state) + if (tsk->flags & PF_EXITING) return; /* diff --git a/kernel/time/posix-timers.c b/kernel/time/posix-timers.c index 436ba794cc0b..188dbedbffca 100644 --- a/kernel/time/posix-timers.c +++ b/kernel/time/posix-timers.c @@ -1077,13 +1077,9 @@ SYSCALL_DEFINE1(timer_delete, timer_t, timer_id) return 0; } -/* - * Invoked from do_exit() when the last thread of a thread group exits. - * At that point no other task can access the timers of the dying - * task anymore. - */ -void exit_itimers(struct task_struct *tsk) +static void posixtimer_delete_timers(void) { + struct task_struct *tsk = current; struct hlist_head timers; struct hlist_node *next; struct k_itimer *timer; @@ -1120,6 +1116,24 @@ void exit_itimers(struct task_struct *tsk) } } +void posixtimer_exit(bool group_dead) +{ + if (group_dead) { + hrtimer_cancel(¤t->signal->real_timer); + posix_cpu_timers_exit_group(); + posixtimer_delete_timers(); + } else { + posix_cpu_timers_exit_task(); + } +} + +void posixtimer_exec(void) +{ + posix_cpu_timers_exit_task(); + posixtimer_delete_timers(); + flush_itimer_signals(); +} + SYSCALL_DEFINE2(clock_settime, const clockid_t, which_clock, const struct __kernel_timespec __user *, tp) { diff --git a/kernel/time/posix-timers.h b/kernel/time/posix-timers.h index 4ea9611dd716..79fd7ea71046 100644 --- a/kernel/time/posix-timers.h +++ b/kernel/time/posix-timers.h @@ -51,3 +51,6 @@ int common_timer_set(struct k_itimer *timr, int flags, struct itimerspec64 *old_setting); void posix_timer_set_common(struct k_itimer *timer, struct itimerspec64 *new_setting); int common_timer_del(struct k_itimer *timer); + +void posix_cpu_timers_exit_task(void); +void posix_cpu_timers_exit_group(void); diff --git a/kernel/time/sleep_timeout.c b/kernel/time/sleep_timeout.c index 3c90574bd904..ad8c415851ae 100644 --- a/kernel/time/sleep_timeout.c +++ b/kernel/time/sleep_timeout.c @@ -212,7 +212,7 @@ int __sched schedule_hrtimeout_range_clock(ktime_t *expires, u64 delta, hrtimer_set_expires_range_ns(&t.timer, *expires, delta); hrtimer_sleeper_start_expires(&t, mode); - if (likely(t.task)) + if (likely(hrtimer_sleeper_task_get(&t))) schedule(); hrtimer_cancel(&t.timer); @@ -220,7 +220,7 @@ int __sched schedule_hrtimeout_range_clock(ktime_t *expires, u64 delta, __set_current_state(TASK_RUNNING); - return !t.task ? 0 : -EINTR; + return !hrtimer_sleeper_task_get(&t) ? 0 : -EINTR; } EXPORT_SYMBOL_GPL(schedule_hrtimeout_range_clock); diff --git a/kernel/time/tick-sched.c b/kernel/time/tick-sched.c index 6c3fea386713..a7893a079a83 100644 --- a/kernel/time/tick-sched.c +++ b/kernel/time/tick-sched.c @@ -738,14 +738,11 @@ bool tick_nohz_tick_stopped_cpu(int cpu) */ static void tick_nohz_update_jiffies(ktime_t now) { - unsigned long flags; + /* Reached only from irq_enter_rcu(), i.e. hard interrupt entry. */ + lockdep_assert_irqs_disabled(); __this_cpu_write(tick_cpu_sched.idle_waketime, now); - - local_irq_save(flags); tick_do_update_jiffies64(now); - local_irq_restore(flags); - touch_softlockup_watchdog_sched(); } @@ -819,7 +816,7 @@ u64 get_jiffies_update(unsigned long *basej) */ static ktime_t tick_nohz_next_event(struct tick_sched *ts, int cpu) { - u64 basemono, next_tick, delta, expires; + u64 basemono, next_tick, expires; unsigned long basejiff; int tick_cpu; @@ -859,8 +856,7 @@ static ktime_t tick_nohz_next_event(struct tick_sched *ts, int cpu) * If the tick is due in the next period, keep it ticking or * force prod the timer. */ - delta = next_tick - basemono; - if (delta <= (u64)TICK_NSEC) { + if (next_tick - basemono <= (u64)TICK_NSEC) { /* * We've not stopped the tick yet, and there's a timer in the * next period, so no point in stopping it either, bail. @@ -876,17 +872,19 @@ static ktime_t tick_nohz_next_event(struct tick_sched *ts, int cpu) * the sleep time to the timekeeping 'max_deferment' value. * Otherwise we can sleep as long as we want. */ - delta = timekeeping_max_deferment(); tick_cpu = READ_ONCE(tick_do_timer_cpu); if (tick_cpu != cpu && - (tick_cpu != TICK_DO_TIMER_NONE || !tick_sched_flag_test(ts, TS_FLAG_DO_TIMER_LAST))) - delta = KTIME_MAX; - - /* Calculate the next expiry time */ - if (delta < (KTIME_MAX - basemono)) - expires = basemono + delta; - else + (tick_cpu != TICK_DO_TIMER_NONE || !tick_sched_flag_test(ts, TS_FLAG_DO_TIMER_LAST))) { expires = KTIME_MAX; + } else { + expires = timekeeping_max_deferment(); + + /* Calculate the next expiry time */ + if (expires < (KTIME_MAX - basemono)) + expires += basemono; + else + expires = KTIME_MAX; + } ts->timer_expires = min_t(u64, expires, next_tick); diff --git a/kernel/time/time_test.c b/kernel/time/time_test.c index 1b99180da288..8b718767b3ba 100644 --- a/kernel/time/time_test.c +++ b/kernel/time/time_test.c @@ -87,8 +87,24 @@ static void time64_to_tm_test_date_range(struct kunit *test) } } +static void time64_to_tm_test_wide_day_count(struct kunit *test) +{ + /* 2^31 days: the first count that does not fit in a 32-bit long. */ + time64_t timestamp = (1LL << 31) * 86400; + struct tm result; + + time64_to_tm(timestamp, 0, &result); + + KUNIT_EXPECT_EQ(test, result.tm_year, 5879680); + KUNIT_EXPECT_EQ(test, result.tm_mon, 6); + KUNIT_EXPECT_EQ(test, result.tm_mday, 12); + KUNIT_EXPECT_EQ(test, result.tm_yday, 193); + KUNIT_EXPECT_EQ(test, result.tm_wday, 6); +} + static struct kunit_case time_test_cases[] = { KUNIT_CASE_SLOW(time64_to_tm_test_date_range), + KUNIT_CASE(time64_to_tm_test_wide_day_count), {} }; diff --git a/kernel/time/timeconv.c b/kernel/time/timeconv.c index 59b922c826e7..aed3af950fa0 100644 --- a/kernel/time/timeconv.c +++ b/kernel/time/timeconv.c @@ -49,8 +49,9 @@ void time64_to_tm(time64_t totalsecs, int offset, struct tm *result) u32 u32tmp, day_of_century, year_of_century, day_of_year, month, day; u64 u64tmp, udays, century, year; bool is_Jan_or_Feb, is_leap_year; - long days, rem; int remainder; + long rem; + s64 days; days = div_s64_rem(totalsecs, SECS_PER_DAY, &remainder); rem = remainder; @@ -70,7 +71,8 @@ void time64_to_tm(time64_t totalsecs, int offset, struct tm *result) result->tm_sec = rem % 60; /* January 1, 1970 was a Thursday. */ - result->tm_wday = (4 + days) % 7; + div_s64_rem(days + 4, 7, &remainder); + result->tm_wday = remainder; if (result->tm_wday < 0) result->tm_wday += 7; diff --git a/kernel/time/timekeeping.c b/kernel/time/timekeeping.c index ea2e6e55f37b..d54c4d303db6 100644 --- a/kernel/time/timekeeping.c +++ b/kernel/time/timekeeping.c @@ -861,8 +861,10 @@ static void timekeeping_update_from_shadow(struct tk_data *tkd, unsigned int act * * Write xtime_sec first so that even if the memcpy() tears the store * data integrity is provided for ktime_get_real_seconds(). + * The same goes for ktime_sec and ktime_get_seconds(). */ WRITE_ONCE(tkd->timekeeper.xtime_sec, tk->xtime_sec); + WRITE_ONCE(tkd->timekeeper.ktime_sec, tk->ktime_sec); memcpy(&tkd->timekeeper, tk, sizeof(*tk)); write_seqcount_end(&tkd->seq); } @@ -1169,7 +1171,7 @@ time64_t ktime_get_seconds(void) struct timekeeper *tk = &tk_core.timekeeper; WARN_ON(timekeeping_suspended); - return tk->ktime_sec; + return READ_ONCE(tk->ktime_sec); } EXPORT_SYMBOL_GPL(ktime_get_seconds); diff --git a/kernel/time/timer.c b/kernel/time/timer.c index ae9abf14688e..42afdcb229d8 100644 --- a/kernel/time/timer.c +++ b/kernel/time/timer.c @@ -890,7 +890,7 @@ static inline void detach_timer(struct timer_list *timer, bool clear_pending) __hlist_del(entry); if (clear_pending) - entry->pprev = NULL; + WRITE_ONCE(entry->pprev, NULL); entry->next = LIST_POISON2; } diff --git a/kernel/time/timer_migration.c b/kernel/time/timer_migration.c index 059d43355e65..f920e73fff51 100644 --- a/kernel/time/timer_migration.c +++ b/kernel/time/timer_migration.c @@ -715,7 +715,7 @@ static void __tmigr_cpu_activate(struct tmigr_cpu *tmc) trace_tmigr_cpu_active(tmc); - tmc->cpuevt.ignore = true; + WRITE_ONCE(tmc->cpuevt.ignore, true); WRITE_ONCE(tmc->wakeup, KTIME_MAX); walk_groups(&tmigr_active_up, &data, tmc); @@ -1258,7 +1258,7 @@ u64 tmigr_cpu_new_timer(u64 nextexp) ret = READ_ONCE(tmc->wakeup); if (nextexp != KTIME_MAX) { if (nextexp != tmc->cpuevt.nextevt.expires || - tmc->cpuevt.ignore) { + READ_ONCE(tmc->cpuevt.ignore)) { ret = tmigr_new_timer(tmc, nextexp); /* * Make sure the reevaluation of timers in idle path @@ -1362,7 +1362,7 @@ static u64 __tmigr_cpu_deactivate(struct tmigr_cpu *tmc, u64 nextexp) * or CPU goes offline. */ if (nextexp != KTIME_MAX) - tmc->cpuevt.ignore = false; + WRITE_ONCE(tmc->cpuevt.ignore, false); walk_groups(&tmigr_inactive_up, &data, tmc); return data.firstexp; diff --git a/kernel/time/vsyscall.c b/kernel/time/vsyscall.c index aa59919b8f2c..0e4b499328c0 100644 --- a/kernel/time/vsyscall.c +++ b/kernel/time/vsyscall.c @@ -41,14 +41,12 @@ static inline void update_vdso_time_data(struct vdso_time_data *vdata, struct ti nsec = tk->tkr_mono.xtime_nsec; nsec += ((u64)tk->wall_to_monotonic.tv_nsec << tk->tkr_mono.shift); - while (nsec >= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift)) { - nsec -= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift); - vdso_ts->sec++; - } - vdso_ts->nsec = nsec; + vdso_ts->sec += __iter_div64_u64_rem(nsec, (u64)NSEC_PER_SEC << tk->tkr_mono.shift, + &vdso_ts->nsec); /* Copy MONOTONIC time for BOOTTIME */ sec = vdso_ts->sec; + nsec = vdso_ts->nsec; /* Add the boot offset */ sec += tk->monotonic_to_boot.tv_sec; nsec += (u64)tk->monotonic_to_boot.tv_nsec << tk->tkr_mono.shift; @@ -56,12 +54,8 @@ static inline void update_vdso_time_data(struct vdso_time_data *vdata, struct ti /* CLOCK_BOOTTIME */ vdso_ts = &vc[CS_HRES_COARSE].basetime[CLOCK_BOOTTIME]; vdso_ts->sec = sec; - - while (nsec >= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift)) { - nsec -= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift); - vdso_ts->sec++; - } - vdso_ts->nsec = nsec; + vdso_ts->sec += __iter_div64_u64_rem(nsec, (u64)NSEC_PER_SEC << tk->tkr_mono.shift, + &vdso_ts->nsec); /* CLOCK_MONOTONIC_RAW */ vdso_ts = &vc[CS_RAW].basetime[CLOCK_MONOTONIC_RAW]; @@ -161,11 +155,11 @@ void vdso_time_update_aux(struct timekeeper *tk) vdso_ts->sec = tk->xtime_sec + tk->monotonic_to_aux.tv_sec; - nsec = tk->tkr_mono.xtime_nsec >> tk->tkr_mono.shift; - nsec += tk->monotonic_to_aux.tv_nsec; - vdso_ts->sec += __iter_div_u64_rem(nsec, NSEC_PER_SEC, &nsec); - nsec = nsec << tk->tkr_mono.shift; - vdso_ts->nsec = nsec; + nsec = tk->tkr_mono.xtime_nsec; + nsec += (u64)tk->monotonic_to_aux.tv_nsec << tk->tkr_mono.shift; + vdso_ts->sec += __iter_div64_u64_rem(nsec, + (u64)NSEC_PER_SEC << tk->tkr_mono.shift, + &vdso_ts->nsec); } __arch_update_vdso_clock(vc); |
