summaryrefslogtreecommitdiff
path: root/kernel
diff options
context:
space:
mode:
Diffstat (limited to 'kernel')
-rw-r--r--kernel/Kconfig.kexec2
-rw-r--r--kernel/Kconfig.preempt13
-rw-r--r--kernel/cpu.c6
-rw-r--r--kernel/crash_core.c2
-rw-r--r--kernel/entry/common.c17
-rw-r--r--kernel/events/core.c177
-rw-r--r--kernel/exit.c13
-rw-r--r--kernel/futex/requeue.c2
-rw-r--r--kernel/futex/waitwake.c8
-rw-r--r--kernel/irq/irqdomain.c1
-rw-r--r--kernel/locking/rtmutex.c2
-rw-r--r--kernel/sched/core.c444
-rw-r--r--kernel/sched/cputime.c4
-rw-r--r--kernel/sched/deadline.c9
-rw-r--r--kernel/sched/debug.c21
-rw-r--r--kernel/sched/ext/ext.c18
-rw-r--r--kernel/sched/ext/ext.h7
-rw-r--r--kernel/sched/fair.c131
-rw-r--r--kernel/sched/idle.c5
-rw-r--r--kernel/sched/rt.c7
-rw-r--r--kernel/sched/sched.h80
-rw-r--r--kernel/sched/stop_task.c5
-rw-r--r--kernel/sched/wait.c22
-rw-r--r--kernel/time/hrtimer.c23
-rw-r--r--kernel/time/posix-cpu-timers.c103
-rw-r--r--kernel/time/posix-timers.c26
-rw-r--r--kernel/time/posix-timers.h3
-rw-r--r--kernel/time/sleep_timeout.c4
-rw-r--r--kernel/time/tick-sched.c30
-rw-r--r--kernel/time/time_test.c16
-rw-r--r--kernel/time/timeconv.c6
-rw-r--r--kernel/time/timekeeping.c4
-rw-r--r--kernel/time/timer.c2
-rw-r--r--kernel/time/timer_migration.c6
-rw-r--r--kernel/time/vsyscall.c26
35 files changed, 784 insertions, 461 deletions
diff --git a/kernel/Kconfig.kexec b/kernel/Kconfig.kexec
index 15632358bcf7..a97ed9605602 100644
--- a/kernel/Kconfig.kexec
+++ b/kernel/Kconfig.kexec
@@ -167,7 +167,7 @@ config CRASH_MAX_MEMORY_RANGES
memory regions that the elfcorehdr buffer/segment can accommodate.
These regions are obtained via walk_system_ram_res(); eg. the
'System RAM' entries in /proc/iomem.
- This value is combined with NR_CPUS_DEFAULT and multiplied by
+ This value is combined with NR_CPUS and multiplied by
sizeof(Elf64_Phdr) to determine the final elfcorehdr memory buffer/
segment size.
The value 8192, for example, covers a (sparsely populated) 1TiB system
diff --git a/kernel/Kconfig.preempt b/kernel/Kconfig.preempt
index f294dad43bd7..edc067a0c422 100644
--- a/kernel/Kconfig.preempt
+++ b/kernel/Kconfig.preempt
@@ -132,10 +132,9 @@ config PREEMPTION
config PREEMPT_DYNAMIC
bool "Preemption behaviour defined on boot"
- depends on HAVE_PREEMPT_DYNAMIC
- select JUMP_LABEL if HAVE_PREEMPT_DYNAMIC_KEY
+ depends on ARCH_HAS_PREEMPT_LAZY
select PREEMPT_BUILD
- default y if HAVE_PREEMPT_DYNAMIC_CALL
+ default y
help
This option allows to define the preemption model on the kernel
command line parameter and thus override the default preemption
@@ -145,9 +144,7 @@ config PREEMPT_DYNAMIC
provide a pre-built kernel binary to reduce the number of kernel
flavors they offer while still offering different usecases.
- The runtime overhead is negligible with HAVE_STATIC_CALL_INLINE enabled
- but if runtime patching is not available for the specific architecture
- then the potential overhead should be considered.
+ The runtime overhead is negligible.
Interesting if you want the same pre-built kernel should be used for
both Server and Desktop workloads.
@@ -197,3 +194,7 @@ config SCHED_CLASS_EXT
For more information:
Documentation/scheduler/sched-ext.rst
https://github.com/sched-ext/scx
+
+config PREFERRED_CPU
+ bool
+ depends on SMP && PARAVIRT
diff --git a/kernel/cpu.c b/kernel/cpu.c
index b3c8553d7bd6..376d297a6292 100644
--- a/kernel/cpu.c
+++ b/kernel/cpu.c
@@ -3103,6 +3103,11 @@ EXPORT_SYMBOL(__cpu_dying_mask);
atomic_t __num_online_cpus __read_mostly;
EXPORT_SYMBOL(__num_online_cpus);
+#ifdef CONFIG_PREFERRED_CPU
+struct cpumask __cpu_preferred_mask __read_mostly;
+EXPORT_SYMBOL_GPL(__cpu_preferred_mask);
+#endif
+
void init_cpu_present(const struct cpumask *src)
{
cpumask_copy(&__cpu_present_mask, src);
@@ -3160,6 +3165,7 @@ void __init boot_cpu_init(void)
/* Mark the boot cpu "present", "online" etc for SMP and UP case */
set_cpu_online(cpu, true);
set_cpu_active(cpu, true);
+ set_cpu_preferred(cpu, true);
set_cpu_present(cpu, true);
set_cpu_possible(cpu, true);
diff --git a/kernel/crash_core.c b/kernel/crash_core.c
index 2b36aa9fade0..d0bd2d0cf899 100644
--- a/kernel/crash_core.c
+++ b/kernel/crash_core.c
@@ -648,7 +648,7 @@ int crash_check_hotplug_support(void)
* new list of CPUs and memory. To make changes to the elfcorehdr, it
* should be large enough to permit a growing number of CPU and Memory
* resources. One can estimate the elfcorehdr memory size based on
- * NR_CPUS_DEFAULT and CRASH_MAX_MEMORY_RANGES. The elfcorehdr is
+ * NR_CPUS and CRASH_MAX_MEMORY_RANGES. The elfcorehdr is
* excluded from SHA verification by default if the architecture
* supports crash hotplug.
*/
diff --git a/kernel/entry/common.c b/kernel/entry/common.c
index e3d381fd3d25..e234b04373fe 100644
--- a/kernel/entry/common.c
+++ b/kernel/entry/common.c
@@ -123,7 +123,7 @@ noinstr irqentry_state_t irqentry_enter(struct pt_regs *regs)
/**
* arch_irqentry_exit_need_resched - Architecture specific need resched function
*
- * Invoked from raw_irqentry_exit_cond_resched() to check if resched is needed.
+ * Invoked from irqentry_exit_cond_resched() to check if resched is needed.
* Defaults return true.
*
* The main purpose is to permit arch to avoid preemption of a task from an IRQ.
@@ -134,7 +134,7 @@ static inline bool arch_irqentry_exit_need_resched(void);
static inline bool arch_irqentry_exit_need_resched(void) { return true; }
#endif
-void raw_irqentry_exit_cond_resched(void)
+void irqentry_exit_cond_resched(void)
{
if (!preempt_count()) {
/* Sanity check RCU and thread stack */
@@ -145,19 +145,6 @@ void raw_irqentry_exit_cond_resched(void)
preempt_schedule_irq();
}
}
-#ifdef CONFIG_PREEMPT_DYNAMIC
-#if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-DEFINE_STATIC_CALL(irqentry_exit_cond_resched, raw_irqentry_exit_cond_resched);
-#elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-DEFINE_STATIC_KEY_TRUE(sk_dynamic_irqentry_exit_cond_resched);
-void dynamic_irqentry_exit_cond_resched(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_irqentry_exit_cond_resched))
- return;
- raw_irqentry_exit_cond_resched();
-}
-#endif
-#endif
noinstr void irqentry_exit(struct pt_regs *regs, irqentry_state_t state)
{
diff --git a/kernel/events/core.c b/kernel/events/core.c
index 601e8d944c24..a34ff4cb410d 100644
--- a/kernel/events/core.c
+++ b/kernel/events/core.c
@@ -7828,22 +7828,82 @@ unsigned long perf_instruction_pointer(struct perf_event *event,
0 : perf_arch_instruction_pointer(regs);
}
+u64 __weak perf_reg_value(struct pt_regs *regs, int idx)
+{
+ return 0;
+}
+
+int __weak perf_reg_validate(u64 mask, bool simd_enabled)
+{
+ return mask ? -ENOSYS : 0;
+}
+
+u64 __weak perf_reg_abi(struct task_struct *task)
+{
+ return PERF_SAMPLE_REGS_ABI_NONE;
+}
+
+void __weak perf_get_regs_user(struct perf_regs *regs_user,
+ struct pt_regs *regs)
+{
+ regs_user->regs = task_pt_regs(current);
+ regs_user->abi = perf_reg_abi(current);
+}
+
+#define word_for_each_set_bit(bit, val) \
+ for (unsigned long long __v = (val); \
+ __v && ((bit = __builtin_ctzll(__v)), 1); \
+ __v &= __v - 1)
+
static void
perf_output_sample_regs(struct perf_output_handle *handle,
struct pt_regs *regs, u64 mask)
{
int bit;
- DECLARE_BITMAP(_mask, 64);
- bitmap_from_u64(_mask, mask);
- for_each_set_bit(bit, _mask, sizeof(mask) * BITS_PER_BYTE) {
- u64 val;
-
- val = perf_reg_value(regs, bit);
+ word_for_each_set_bit(bit, mask) {
+ u64 val = perf_reg_value(regs, bit);
perf_output_put(handle, val);
}
}
+static void
+perf_output_sample_simd_regs(struct perf_output_handle *handle,
+ struct perf_event *event,
+ struct pt_regs *regs,
+ u64 mask, u32 pred_mask)
+{
+ u64 pred_qwords = event->attr.sample_simd_pred_reg_qwords;
+ u64 vec_qwords = event->attr.sample_simd_vec_reg_qwords;
+ u64 nr_vectors = hweight64(mask);
+ u64 nr_pred = hweight32(pred_mask);
+ int bit;
+
+ perf_output_put(handle, nr_vectors);
+ perf_output_put(handle, vec_qwords);
+ perf_output_put(handle, nr_pred);
+ perf_output_put(handle, pred_qwords);
+
+ if (nr_vectors) {
+ word_for_each_set_bit(bit, mask) {
+ for (int i = 0; i < vec_qwords; i++) {
+ u64 val = perf_simd_reg_value(regs, bit,
+ i, false);
+ perf_output_put(handle, val);
+ }
+ }
+ }
+ if (nr_pred) {
+ word_for_each_set_bit(bit, pred_mask) {
+ for (int i = 0; i < pred_qwords; i++) {
+ u64 val = perf_simd_reg_value(regs, bit,
+ i, true);
+ perf_output_put(handle, val);
+ }
+ }
+ }
+}
+
static void perf_sample_regs_user(struct perf_regs *regs_user,
struct pt_regs *regs)
{
@@ -7877,6 +7937,17 @@ static void perf_sample_regs_intr(struct perf_regs *regs_intr,
}
}
+int __weak perf_simd_reg_validate(u16 vec_qwords, u64 vec_mask,
+ u16 pred_qwords, u32 pred_mask)
+{
+ return -EINVAL;
+}
+
+u64 __weak perf_simd_reg_value(struct pt_regs *regs, int idx,
+ u16 qwords_idx, bool pred)
+{
+ return 0;
+}
/*
* Get remaining task size from user stack pointer.
@@ -8407,10 +8478,17 @@ void perf_output_sample(struct perf_output_handle *handle,
perf_output_put(handle, abi);
if (abi) {
- u64 mask = event->attr.sample_regs_user;
+ struct perf_event_attr *attr = &event->attr;
+ u64 mask = attr->sample_regs_user;
perf_output_sample_regs(handle,
data->regs_user.regs,
mask);
+ if (abi & PERF_SAMPLE_REGS_ABI_SIMD) {
+ perf_output_sample_simd_regs(handle, event,
+ data->regs_user.regs,
+ attr->sample_simd_vec_reg_user,
+ attr->sample_simd_pred_reg_user);
+ }
}
}
@@ -8438,11 +8516,18 @@ void perf_output_sample(struct perf_output_handle *handle,
perf_output_put(handle, abi);
if (abi) {
- u64 mask = event->attr.sample_regs_intr;
+ struct perf_event_attr *attr = &event->attr;
+ u64 mask = attr->sample_regs_intr;
perf_output_sample_regs(handle,
data->regs_intr.regs,
mask);
+ if (abi & PERF_SAMPLE_REGS_ABI_SIMD) {
+ perf_output_sample_simd_regs(handle, event,
+ data->regs_intr.regs,
+ attr->sample_simd_vec_reg_intr,
+ attr->sample_simd_pred_reg_intr);
+ }
}
}
@@ -8645,6 +8730,29 @@ static __always_inline u64 __cond_set(u64 flags, u64 s, u64 d)
return d * !!(flags & s);
}
+u64 perf_update_xregs_size(struct perf_event *event, bool intr)
+{
+ u16 pred_qwords = event->attr.sample_simd_pred_reg_qwords;
+ u16 vec_qwords = event->attr.sample_simd_vec_reg_qwords;
+ u64 pred_mask;
+ u64 mask;
+ int size;
+
+ if (intr) {
+ mask = event->attr.sample_simd_vec_reg_intr;
+ pred_mask = event->attr.sample_simd_pred_reg_intr;
+ } else {
+ mask = event->attr.sample_simd_vec_reg_user;
+ pred_mask = event->attr.sample_simd_pred_reg_user;
+ }
+
+ size = sizeof(u64) * 4;
+ size += (hweight64(mask) * vec_qwords +
+ hweight64(pred_mask) * pred_qwords) * sizeof(u64);
+
+ return size;
+}
+
void perf_prepare_sample(struct perf_sample_data *data,
struct perf_event *event,
struct pt_regs *regs)
@@ -8707,7 +8815,12 @@ void perf_prepare_sample(struct perf_sample_data *data,
if (data->regs_user.regs) {
u64 mask = event->attr.sample_regs_user;
+
size += hweight64(mask) * sizeof(u64);
+ if (event_has_simd_regs(event)) {
+ size += perf_update_xregs_size(event, false);
+ data->regs_user.abi |= PERF_SAMPLE_REGS_ABI_SIMD;
+ }
}
data->dyn_size += size;
@@ -8772,6 +8885,10 @@ void perf_prepare_sample(struct perf_sample_data *data,
u64 mask = event->attr.sample_regs_intr;
size += hweight64(mask) * sizeof(u64);
+ if (event_has_simd_regs(event)) {
+ size += perf_update_xregs_size(event, true);
+ data->regs_intr.abi |= PERF_SAMPLE_REGS_ABI_SIMD;
+ }
}
data->dyn_size += size;
@@ -13116,12 +13233,6 @@ int perf_pmu_unregister(struct pmu *pmu)
}
EXPORT_SYMBOL_GPL(perf_pmu_unregister);
-static inline bool has_extended_regs(struct perf_event *event)
-{
- return (event->attr.sample_regs_user & PERF_REG_EXTENDED_MASK) ||
- (event->attr.sample_regs_intr & PERF_REG_EXTENDED_MASK);
-}
-
static int perf_try_init_event(struct pmu *pmu, struct perf_event *event)
{
struct perf_event_context *ctx = NULL;
@@ -13155,8 +13266,14 @@ static int perf_try_init_event(struct pmu *pmu, struct perf_event *event)
if (ret)
goto err_pmu;
+ if (!(pmu->capabilities & PERF_PMU_CAP_SIMD_REGS) &&
+ event_has_simd_regs(event)) {
+ ret = -EOPNOTSUPP;
+ goto err_destroy;
+ }
+
if (!(pmu->capabilities & PERF_PMU_CAP_EXTENDED_REGS) &&
- has_extended_regs(event)) {
+ event_has_extended_regs(event)) {
ret = -EOPNOTSUPP;
goto err_destroy;
}
@@ -13650,7 +13767,8 @@ static int perf_copy_attr(struct perf_event_attr __user *uattr,
attr->size = size;
- if (attr->__reserved_1 || attr->__reserved_2 || attr->__reserved_3)
+ if (attr->__reserved_1 || attr->__reserved_2 ||
+ attr->__reserved_3 || attr->__reserved_4)
return -EINVAL;
if (attr->sample_type & ~(PERF_SAMPLE_MAX-1))
@@ -13696,9 +13814,18 @@ static int perf_copy_attr(struct perf_event_attr __user *uattr,
}
if (attr->sample_type & PERF_SAMPLE_REGS_USER) {
- ret = perf_reg_validate(attr->sample_regs_user);
+ ret = perf_reg_validate(attr->sample_regs_user,
+ attr->sample_simd_regs_enabled);
if (ret)
return ret;
+ if (attr->sample_simd_regs_enabled) {
+ ret = perf_simd_reg_validate(attr->sample_simd_vec_reg_qwords,
+ attr->sample_simd_vec_reg_user,
+ attr->sample_simd_pred_reg_qwords,
+ attr->sample_simd_pred_reg_user);
+ if (ret)
+ return ret;
+ }
}
if (attr->sample_type & PERF_SAMPLE_STACK_USER) {
@@ -13719,8 +13846,20 @@ static int perf_copy_attr(struct perf_event_attr __user *uattr,
if (!attr->sample_max_stack)
attr->sample_max_stack = sysctl_perf_event_max_stack;
- if (attr->sample_type & PERF_SAMPLE_REGS_INTR)
- ret = perf_reg_validate(attr->sample_regs_intr);
+ if (attr->sample_type & PERF_SAMPLE_REGS_INTR) {
+ ret = perf_reg_validate(attr->sample_regs_intr,
+ attr->sample_simd_regs_enabled);
+ if (ret)
+ return ret;
+ if (attr->sample_simd_regs_enabled) {
+ ret = perf_simd_reg_validate(attr->sample_simd_vec_reg_qwords,
+ attr->sample_simd_vec_reg_intr,
+ attr->sample_simd_pred_reg_qwords,
+ attr->sample_simd_pred_reg_intr);
+ if (ret)
+ return ret;
+ }
+ }
#ifndef CONFIG_CGROUP_PERF
if (attr->sample_type & PERF_SAMPLE_CGROUP)
diff --git a/kernel/exit.c b/kernel/exit.c
index 29e853a36602..9ff1fa7b30ea 100644
--- a/kernel/exit.c
+++ b/kernel/exit.c
@@ -168,12 +168,6 @@ static void __exit_signal(struct release_task_post *post, struct task_struct *ts
lockdep_tasklist_lock_is_held());
spin_lock(&sighand->siglock);
-#ifdef CONFIG_POSIX_TIMERS
- posix_cpu_timers_exit(tsk);
- if (group_dead)
- posix_cpu_timers_exit_group(tsk);
-#endif
-
if (group_dead) {
tty = sig->tty;
sig->tty = NULL;
@@ -940,13 +934,12 @@ void __noreturn do_exit(long code)
panic("Attempted to kill init! exitcode=0x%08x\n",
tsk->signal->group_exit_code ?: (int)code);
-#ifdef CONFIG_POSIX_TIMERS
- hrtimer_cancel(&tsk->signal->real_timer);
- exit_itimers(tsk);
-#endif
if (tsk->mm)
setmax_mm_hiwater_rss(&tsk->signal->maxrss, tsk->mm);
}
+
+ posixtimer_exit(group_dead);
+
acct_collect(code, group_dead);
if (group_dead)
tty_audit_exit();
diff --git a/kernel/futex/requeue.c b/kernel/futex/requeue.c
index b3f4a4bccb12..842d852302dd 100644
--- a/kernel/futex/requeue.c
+++ b/kernel/futex/requeue.c
@@ -744,7 +744,7 @@ int handle_early_requeue_pi_wakeup(struct futex_hash_bucket *hb,
/* Handle spurious wakeups gracefully */
ret = -EWOULDBLOCK;
- if (timeout && !timeout->task)
+ if (timeout && !hrtimer_sleeper_task_get(timeout))
ret = -ETIMEDOUT;
else if (signal_pending(current))
ret = -ERESTARTNOINTR;
diff --git a/kernel/futex/waitwake.c b/kernel/futex/waitwake.c
index d4483d15d30a..cf18309e5770 100644
--- a/kernel/futex/waitwake.c
+++ b/kernel/futex/waitwake.c
@@ -383,7 +383,7 @@ void futex_do_wait(struct futex_q *q, struct hrtimer_sleeper *timeout)
* flagged for rescheduling. Only call schedule if there
* is no timeout, or if it has yet to expire.
*/
- if (!timeout || timeout->task)
+ if (!timeout || hrtimer_sleeper_task_get(timeout))
schedule();
}
__set_current_state(TASK_RUNNING);
@@ -539,7 +539,7 @@ retry:
static void futex_sleep_multiple(struct futex_vector *vs, unsigned int count,
struct hrtimer_sleeper *to)
{
- if (to && !to->task)
+ if (to && !hrtimer_sleeper_task_get(to))
return;
for (; count; count--, vs++) {
@@ -590,7 +590,7 @@ int futex_wait_multiple(struct futex_vector *vs, unsigned int count,
if (ret >= 0)
return ret;
- if (to && !to->task)
+ if (to && !hrtimer_sleeper_task_get(to))
return -ETIMEDOUT;
else if (signal_pending(current))
return -ERESTARTSYS;
@@ -725,7 +725,7 @@ retry:
if (!futex_unqueue(&q))
return 0;
- if (to && !to->task)
+ if (to && !hrtimer_sleeper_task_get(to))
return -ETIMEDOUT;
/*
diff --git a/kernel/irq/irqdomain.c b/kernel/irq/irqdomain.c
index 57c819da30c2..4fdcb6df5306 100644
--- a/kernel/irq/irqdomain.c
+++ b/kernel/irq/irqdomain.c
@@ -344,6 +344,7 @@ static struct irq_domain *__irq_domain_instantiate(const struct irq_domain_info
err = irq_domain_alloc_generic_chips(domain, info->dgc_info);
if (err)
goto err_domain_free;
+ domain->flags |= IRQ_DOMAIN_FLAG_DESTROY_GC;
}
if (info->init) {
diff --git a/kernel/locking/rtmutex.c b/kernel/locking/rtmutex.c
index 4728631ae719..5a9534c715b8 100644
--- a/kernel/locking/rtmutex.c
+++ b/kernel/locking/rtmutex.c
@@ -1644,7 +1644,7 @@ static int __sched rt_mutex_slowlock_block(struct rt_mutex_base *lock,
break;
}
- if (timeout && !timeout->task) {
+ if (timeout && !hrtimer_sleeper_task_get(timeout)) {
ret = -ETIMEDOUT;
break;
}
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index 1fe40de6ebe3..84313c9c4ba9 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -805,7 +805,7 @@ struct rq *_task_rq_lock(struct task_struct *p, struct rq_flags *rf)
/* Use CONFIG_PARAVIRT as this will avoid more #ifdef in arch code. */
#ifdef CONFIG_PARAVIRT
-struct static_key paravirt_steal_rq_enabled;
+DEFINE_STATIC_KEY_FALSE(paravirt_steal_rq_enabled);
#endif
static void update_rq_clock_task(struct rq *rq, s64 delta)
@@ -844,7 +844,7 @@ static void update_rq_clock_task(struct rq *rq, s64 delta)
}
#endif
#ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING
- if (static_key_false((&paravirt_steal_rq_enabled))) {
+ if (static_branch_unlikely(&paravirt_steal_rq_enabled)) {
u64 prev_steal;
steal = prev_steal = paravirt_steal_clock(cpu_of(rq));
@@ -2252,7 +2252,8 @@ void deactivate_task(struct rq *rq, struct task_struct *p, int flags)
dequeue_task(rq, p, flags);
}
-static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_state)
+static bool dequeue_block_task(struct rq *rq, struct task_struct *p,
+ unsigned long task_state)
{
int flags = DEQUEUE_NOCLOCK;
@@ -2273,9 +2274,15 @@ static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_
*
* Where __schedule() and ttwu() have matching control dependencies.
*
- * After this, schedule() must not care about p->state any more.
+ * Once the caller invokes __block_task(), schedule() must not care about
+ * p->state any more.
*/
- if (dequeue_task(rq, p, DEQUEUE_SLEEP | flags))
+ return dequeue_task(rq, p, DEQUEUE_SLEEP | flags);
+}
+
+static void block_task(struct rq *rq, struct task_struct *p, unsigned long task_state)
+{
+ if (dequeue_block_task(rq, p, task_state))
__block_task(rq, p);
}
@@ -2504,6 +2511,24 @@ static inline bool rq_has_pinned_tasks(struct rq *rq)
return rq->nr_pinned;
}
+static inline bool task_can_migrate_to_preferred(struct task_struct *p, int cpu)
+{
+ /* No need to migrate from a preferred CPU */
+ if (cpu_preferred(cpu))
+ return false;
+
+ /* Only FAIR tasks honor preferred CPU state */
+ if (unlikely(p->sched_class != &fair_sched_class))
+ return false;
+
+ /* Ignore preferred state if task affinity is changing */
+ if (unlikely(!cpumask_test_cpu(task_cpu(p), p->cpus_ptr)))
+ return false;
+
+ return cpumask_intersects_and(p->cpus_ptr, cpu_preferred_mask,
+ task_cpu_possible_mask(p));
+}
+
/*
* Per-CPU kthreads are allowed to run on !active && online CPUs, see
* __set_cpus_allowed_ptr() and select_fallback_rq().
@@ -2519,8 +2544,12 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu)
return cpu_online(cpu);
/* Non kernel threads are not allowed during either online or offline. */
- if (!(p->flags & PF_KTHREAD))
+ if (!(p->flags & PF_KTHREAD)) {
+ /* Try to use preferred CPU if task's affinity allows */
+ if (task_can_migrate_to_preferred(p, cpu))
+ return false;
return cpu_active(cpu);
+ }
/* KTHREAD_IS_PER_CPU is always allowed. */
if (kthread_is_per_cpu(p))
@@ -2530,7 +2559,11 @@ static inline bool is_cpu_allowed(struct task_struct *p, int cpu)
if (cpu_dying(cpu))
return false;
- /* But are allowed during online. */
+ /* Try to keep unbound kthreads on a preferred CPU if possible. */
+ if (task_can_migrate_to_preferred(p, cpu))
+ return false;
+
+ /* Otherwise, they are allowed to run on online CPU. */
return cpu_online(cpu);
}
@@ -3773,6 +3806,7 @@ static inline void proxy_reset_donor(struct rq *rq)
WARN_ON_ONCE(rq->donor == rq->curr);
put_prev_set_next_task(rq, rq->donor, rq->curr);
+ rq->next_class = rq->curr->sched_class;
rq_set_donor(rq, rq->curr);
zap_balance_callbacks(rq);
resched_curr(rq);
@@ -3787,6 +3821,8 @@ static inline void proxy_reset_donor(struct rq *rq)
*/
static inline bool proxy_needs_return(struct rq *rq, struct task_struct *p)
{
+ bool dequeued;
+
/*
* Typically per __set_task_cpu(), task_cpu(p) == p->wake_cpu.
*
@@ -3809,12 +3845,23 @@ static inline bool proxy_needs_return(struct rq *rq, struct task_struct *p)
/* If already current, don't need to return migrate */
if (task_current(rq, p))
return false;
-
- /* If we're return migrating the rq->donor, switch it out for idle */
- if (task_current_donor(rq, p))
- proxy_reset_donor(rq);
}
- block_task(rq, p, TASK_WAKING);
+
+ dequeued = dequeue_block_task(rq, p, TASK_WAKING);
+
+ /*
+ * Dequeue @p from its scheduling class before resetting rq->donor.
+ * In particular, sched_ext needs to end the donor's running session
+ * and clear SCX_TASK_QUEUED before put_prev_task_scx() is called by
+ * proxy_reset_donor(); otherwise it would reenqueue the blocked donor.
+ *
+ * Keep on_rq set until all donor references have been replaced.
+ */
+ if (task_current_donor(rq, p))
+ proxy_reset_donor(rq);
+
+ if (dequeued)
+ __block_task(rq, p);
return true;
}
#else /* !CONFIG_SCHED_PROXY_EXEC */
@@ -3905,7 +3952,7 @@ static int ttwu_runnable(struct task_struct *p, int wake_flags)
* When on_rq && !on_cpu the task is preempted, see if
* it should preempt the task that is current now.
*/
- wakeup_preempt(rq, p, wake_flags);
+ wakeup_preempt(rq, p, wake_flags | WF_TTWU_RQ);
}
ttwu_do_wakeup(p);
return 1;
@@ -5149,7 +5196,7 @@ static void do_balance_callbacks(struct rq *rq, struct balance_callback *head)
lockdep_assert_rq_held(rq);
while (head) {
- func = (void (*)(struct rq *))head->func;
+ func = head->func;
next = head->next;
head->next = NULL;
head = next;
@@ -5789,6 +5836,9 @@ void sched_tick(void)
unsigned long hw_pressure;
u64 resched_latency;
+ if (!cpu_preferred(cpu))
+ sched_push_current_non_preferred_cpu(rq);
+
if (housekeeping_cpu(cpu, HK_TYPE_KERNEL_NOISE))
arch_scale_freq_tick();
@@ -6283,10 +6333,7 @@ pick_next_task(struct rq *rq, struct rq_flags *rf)
* selection. In this case, do a core-wide selection.
*/
if (rq->core->core_pick_seq == rq->core->core_task_seq &&
- rq->core->core_pick_seq != rq->core_sched_seq &&
rq->core_pick) {
- WRITE_ONCE(rq->core_sched_seq, rq->core->core_pick_seq);
-
next = rq->core_pick;
rq->dl_server = rq->core_dl_server;
rq->core_pick = NULL;
@@ -6318,11 +6365,13 @@ restart:
}
/*
- * core->core_task_seq, core->core_pick_seq, rq->core_sched_seq
+ * core->core_task_seq, core->core_pick_seq
*
* @task_seq guards the task state ({en,de}queues)
* @pick_seq is the @task_seq we did a selection on
- * @sched_seq is the @pick_seq we scheduled
+ *
+ * Once a core-wide selection is committed, a non-NULL core_pick denotes
+ * a pick which still needs to be consumed on this CPU.
*
* However, preemptions can cause multiple picks on the same task set.
* 'Fix' this by also increasing @task_seq for every pick.
@@ -6429,7 +6478,6 @@ restart:
rq->core->core_pick_seq = rq->core->core_task_seq;
next = rq->core_pick;
- rq->core_sched_seq = rq->core->core_pick_seq;
/* Something should have been selected for current CPU */
WARN_ON_ONCE(!next);
@@ -6517,7 +6565,10 @@ static bool try_steal_cookie(int this, int that)
return false;
do {
- if (p == src->core_pick || p == src->curr)
+ if (p == src->core_pick || p == src->curr || p == src->donor)
+ goto next;
+
+ if (task_is_blocked(p))
goto next;
if (!is_cpu_allowed(p, this))
@@ -6820,6 +6871,34 @@ static void proxy_deactivate(struct rq *rq, struct task_struct *donor)
block_task(rq, donor, state);
}
+/*
+ * Remove a retained proxy donor before changing its scheduler ownership.
+ * The caller holds p->pi_lock, so p cannot wake and migrate if block_task()
+ * drops it from the runqueue. If DELAY_DEQUEUE keeps a blocked fair task
+ * queued, switching_from_fair() completes the dequeue in the immediately
+ * following sched_change_begin().
+ */
+void sched_proxy_block_task(struct rq *rq, struct task_struct *p)
+{
+ unsigned long state = READ_ONCE(p->__state);
+
+ lockdep_assert_held(&p->pi_lock);
+ lockdep_assert_rq_held(rq);
+
+ if (!p->is_blocked || !task_on_rq_queued(p))
+ return;
+ if (WARN_ON_ONCE(state == TASK_RUNNING))
+ return;
+
+ if (task_current_donor(rq, p))
+ proxy_reset_donor(rq);
+
+ if (!p->se.sched_delayed)
+ block_task(rq, p, state);
+
+ WARN_ON_ONCE(task_on_rq_queued(p) && !p->se.sched_delayed);
+}
+
static inline void proxy_release_rq_lock(struct rq *rq, struct rq_flags *rf)
__releases(__rq_lockp(rq))
{
@@ -6865,9 +6944,9 @@ static void proxy_migrate_task(struct rq *rq, struct rq_flags *rf,
__must_hold(__rq_lockp(rq))
{
struct rq *target_rq = cpu_rq(target_cpu);
+ LIST_HEAD(migrate_list);
lockdep_assert_rq_held(rq);
- WARN_ON(p == rq->curr);
/*
* Since we are migrating a blocked donor, it could be rq->donor,
* and we want to make sure there aren't any references from this
@@ -6880,13 +6959,20 @@ static void proxy_migrate_task(struct rq *rq, struct rq_flags *rf,
* before we release the lock.
*/
proxy_resched_idle(rq);
-
- deactivate_task(rq, p, DEQUEUE_NOCLOCK);
- proxy_set_task_cpu(p, target_cpu);
-
+ for (; p; p = p->blocked_donor) {
+ WARN_ON(p == rq->curr);
+ deactivate_task(rq, p, DEQUEUE_NOCLOCK);
+ proxy_set_task_cpu(p, target_cpu);
+ /*
+ * We can re-use se.group_node to migrate the thing,
+ * because @p is deactivated (won't be balanced) and
+ * we hold the rq_lock.
+ */
+ list_add(&p->se.group_node, &migrate_list);
+ }
proxy_release_rq_lock(rq, rf);
- attach_one_task(target_rq, p);
+ __attach_tasks(target_rq, &migrate_list);
proxy_reacquire_rq_lock(rq, rf);
}
@@ -6979,7 +7065,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
if (!READ_ONCE(owner->on_rq) || owner->se.sched_delayed) {
/* XXX Don't handle blocked owners/delayed dequeue yet */
if (curr_in_chain)
- return proxy_resched_idle(rq);
+ goto resched_idle;
__clear_task_blocked_on(p, NULL);
goto deactivate;
}
@@ -6991,7 +7077,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
* and leave that CPU to sort things out.
*/
if (curr_in_chain)
- return proxy_resched_idle(rq);
+ goto resched_idle;
goto migrate_task;
}
@@ -7004,7 +7090,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
* case we should end up back in find_proxy_task(), this time
* hopefully with all relevant tasks already enqueued.
*/
- return proxy_resched_idle(rq);
+ goto resched_idle;
}
/*
@@ -7041,7 +7127,7 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
* So schedule rq->idle so that ttwu_runnable() can get the rq
* lock and mark owner as running.
*/
- return proxy_resched_idle(rq);
+ goto resched_idle;
}
/*
* OK, now we're absolutely sure @owner is on this
@@ -7051,8 +7137,18 @@ find_proxy_task(struct rq *rq, struct task_struct *donor, struct rq_flags *rf)
owner->blocked_donor = p;
}
WARN_ON_ONCE(owner && !owner->on_rq);
+
+ if (owner && !sched_cpu_cookie_match(rq, owner)) {
+ if (curr_in_chain)
+ return proxy_resched_idle(rq);
+ p = donor; /* Deactivate the donor, not the runnable owner */
+ clear_task_blocked_on(p, NULL);
+ goto deactivate;
+ }
return owner;
+resched_idle:
+ return proxy_resched_idle(rq);
deactivate:
proxy_deactivate(rq, p);
return NULL;
@@ -7184,13 +7280,12 @@ static void __sched notrace __schedule(int sched_mode)
}
} else if (!preempt && prev_state) {
/*
- * We pass task_is_blocked() as the should_block arg
- * in order to keep mutex-blocked tasks on the runqueue
- * for slection with proxy-exec (without proxy-exec
- * task_is_blocked() will always be false).
+ * Keep mutex-blocked tasks on the runqueue for proxy execution
+ * only when their scheduling class allows it. Without proxy
+ * execution, task_is_blocked() always returns false.
*/
try_to_block_task(rq, prev, &prev_state,
- !task_is_blocked(prev));
+ !task_is_blocked(prev) || !scx_allow_proxy_exec(prev));
switch_count = &prev->nvcsw;
}
@@ -7211,6 +7306,7 @@ pick_again:
}
if (next == rq->idle) {
zap_balance_callbacks(rq);
+ scx_proxy_reenqueue_retry(rq, next);
goto keep_resched;
}
}
@@ -7229,8 +7325,10 @@ pick_again:
* on_cpu.
*/
donor->sched_class->put_prev_task(rq, donor, donor);
- donor->sched_class->set_next_task(rq, donor, true);
+ donor->sched_class->set_next_task(rq, donor, SNT_PICK);
}
+ scx_proxy_donor_start(rq);
+ scx_proxy_reenqueue_retry(rq, next);
} else {
rq_set_donor(rq, next);
}
@@ -7489,27 +7587,6 @@ asmlinkage __visible void __sched notrace preempt_schedule(void)
NOKPROBE_SYMBOL(preempt_schedule);
EXPORT_SYMBOL(preempt_schedule);
-#ifdef CONFIG_PREEMPT_DYNAMIC
-# ifdef CONFIG_HAVE_PREEMPT_DYNAMIC_CALL
-# ifndef preempt_schedule_dynamic_enabled
-# define preempt_schedule_dynamic_enabled preempt_schedule
-# define preempt_schedule_dynamic_disabled NULL
-# endif
-DEFINE_STATIC_CALL(preempt_schedule, preempt_schedule_dynamic_enabled);
-EXPORT_STATIC_CALL_TRAMP(preempt_schedule);
-# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule);
-void __sched notrace dynamic_preempt_schedule(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_preempt_schedule))
- return;
- preempt_schedule();
-}
-NOKPROBE_SYMBOL(dynamic_preempt_schedule);
-EXPORT_SYMBOL(dynamic_preempt_schedule);
-# endif
-#endif /* CONFIG_PREEMPT_DYNAMIC */
-
/**
* preempt_schedule_notrace - preempt_schedule called by tracing
*
@@ -7562,27 +7639,6 @@ asmlinkage __visible void __sched notrace preempt_schedule_notrace(void)
}
EXPORT_SYMBOL_GPL(preempt_schedule_notrace);
-#ifdef CONFIG_PREEMPT_DYNAMIC
-# if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-# ifndef preempt_schedule_notrace_dynamic_enabled
-# define preempt_schedule_notrace_dynamic_enabled preempt_schedule_notrace
-# define preempt_schedule_notrace_dynamic_disabled NULL
-# endif
-DEFINE_STATIC_CALL(preempt_schedule_notrace, preempt_schedule_notrace_dynamic_enabled);
-EXPORT_STATIC_CALL_TRAMP(preempt_schedule_notrace);
-# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-static DEFINE_STATIC_KEY_TRUE(sk_dynamic_preempt_schedule_notrace);
-void __sched notrace dynamic_preempt_schedule_notrace(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_preempt_schedule_notrace))
- return;
- preempt_schedule_notrace();
-}
-NOKPROBE_SYMBOL(dynamic_preempt_schedule_notrace);
-EXPORT_SYMBOL(dynamic_preempt_schedule_notrace);
-# endif
-#endif
-
#endif /* CONFIG_PREEMPTION */
/*
@@ -7799,7 +7855,7 @@ out_unlock:
}
#endif /* CONFIG_RT_MUTEXES */
-#if !defined(CONFIG_PREEMPTION) || defined(CONFIG_PREEMPT_DYNAMIC)
+#if !defined(CONFIG_PREEMPTION)
int __sched __cond_resched(void)
{
if (should_resched(0) && !irqs_disabled()) {
@@ -7827,38 +7883,6 @@ int __sched __cond_resched(void)
EXPORT_SYMBOL(__cond_resched);
#endif
-#ifdef CONFIG_PREEMPT_DYNAMIC
-# ifdef CONFIG_HAVE_PREEMPT_DYNAMIC_CALL
-# define cond_resched_dynamic_enabled __cond_resched
-# define cond_resched_dynamic_disabled ((void *)&__static_call_return0)
-DEFINE_STATIC_CALL_RET0(cond_resched, __cond_resched);
-EXPORT_STATIC_CALL_TRAMP(cond_resched);
-
-# define might_resched_dynamic_enabled __cond_resched
-# define might_resched_dynamic_disabled ((void *)&__static_call_return0)
-DEFINE_STATIC_CALL_RET0(might_resched, __cond_resched);
-EXPORT_STATIC_CALL_TRAMP(might_resched);
-# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-static DEFINE_STATIC_KEY_FALSE(sk_dynamic_cond_resched);
-int __sched dynamic_cond_resched(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_cond_resched))
- return 0;
- return __cond_resched();
-}
-EXPORT_SYMBOL(dynamic_cond_resched);
-
-static DEFINE_STATIC_KEY_FALSE(sk_dynamic_might_resched);
-int __sched dynamic_might_resched(void)
-{
- if (!static_branch_unlikely(&sk_dynamic_might_resched))
- return 0;
- return __cond_resched();
-}
-EXPORT_SYMBOL(dynamic_might_resched);
-# endif
-#endif /* CONFIG_PREEMPT_DYNAMIC */
-
/*
* __cond_resched_lock() - if a reschedule is pending, drop the given lock,
* call schedule, and on return reacquire the lock.
@@ -7928,50 +7952,21 @@ EXPORT_SYMBOL(__cond_resched_rwlock_write);
# endif
/*
- * SC:cond_resched
- * SC:might_resched
- * SC:preempt_schedule
- * SC:preempt_schedule_notrace
- * SC:irqentry_exit_cond_resched
- *
- *
* NONE:
- * cond_resched <- __cond_resched
- * might_resched <- RET0
- * preempt_schedule <- NOP
- * preempt_schedule_notrace <- NOP
- * irqentry_exit_cond_resched <- NOP
- * dynamic_preempt_lazy <- false
+ * (unselectable)
*
* VOLUNTARY:
- * cond_resched <- __cond_resched
- * might_resched <- __cond_resched
- * preempt_schedule <- NOP
- * preempt_schedule_notrace <- NOP
- * irqentry_exit_cond_resched <- NOP
- * dynamic_preempt_lazy <- false
+ * (unselectable)
*
* FULL:
- * cond_resched <- RET0
- * might_resched <- RET0
- * preempt_schedule <- preempt_schedule
- * preempt_schedule_notrace <- preempt_schedule_notrace
- * irqentry_exit_cond_resched <- irqentry_exit_cond_resched
* dynamic_preempt_lazy <- false
*
* LAZY:
- * cond_resched <- RET0
- * might_resched <- RET0
- * preempt_schedule <- preempt_schedule
- * preempt_schedule_notrace <- preempt_schedule_notrace
- * irqentry_exit_cond_resched <- irqentry_exit_cond_resched
* dynamic_preempt_lazy <- true
*/
enum {
preempt_dynamic_undefined = -1,
- preempt_dynamic_none,
- preempt_dynamic_voluntary,
preempt_dynamic_full,
preempt_dynamic_lazy,
};
@@ -7980,21 +7975,11 @@ int preempt_dynamic_mode = preempt_dynamic_undefined;
int sched_dynamic_mode(const char *str)
{
-# if !(defined(CONFIG_PREEMPT_RT) || defined(CONFIG_ARCH_HAS_PREEMPT_LAZY))
- if (!strcmp(str, "none"))
- return preempt_dynamic_none;
-
- if (!strcmp(str, "voluntary"))
- return preempt_dynamic_voluntary;
-# endif
-
if (!strcmp(str, "full"))
return preempt_dynamic_full;
-# ifdef CONFIG_ARCH_HAS_PREEMPT_LAZY
if (!strcmp(str, "lazy"))
return preempt_dynamic_lazy;
-# endif
return -EINVAL;
}
@@ -8002,71 +7987,18 @@ int sched_dynamic_mode(const char *str)
# define preempt_dynamic_key_enable(f) static_key_enable(&sk_dynamic_##f.key)
# define preempt_dynamic_key_disable(f) static_key_disable(&sk_dynamic_##f.key)
-# if defined(CONFIG_HAVE_PREEMPT_DYNAMIC_CALL)
-# define preempt_dynamic_enable(f) static_call_update(f, f##_dynamic_enabled)
-# define preempt_dynamic_disable(f) static_call_update(f, f##_dynamic_disabled)
-# elif defined(CONFIG_HAVE_PREEMPT_DYNAMIC_KEY)
-# define preempt_dynamic_enable(f) preempt_dynamic_key_enable(f)
-# define preempt_dynamic_disable(f) preempt_dynamic_key_disable(f)
-# else
-# error "Unsupported PREEMPT_DYNAMIC mechanism"
-# endif
-
static DEFINE_MUTEX(sched_dynamic_mutex);
static void __sched_dynamic_update(int mode)
{
- /*
- * Avoid {NONE,VOLUNTARY} -> FULL transitions from ever ending up in
- * the ZERO state, which is invalid.
- */
- preempt_dynamic_enable(cond_resched);
- preempt_dynamic_enable(might_resched);
- preempt_dynamic_enable(preempt_schedule);
- preempt_dynamic_enable(preempt_schedule_notrace);
- preempt_dynamic_enable(irqentry_exit_cond_resched);
- preempt_dynamic_key_disable(preempt_lazy);
-
switch (mode) {
- case preempt_dynamic_none:
- preempt_dynamic_enable(cond_resched);
- preempt_dynamic_disable(might_resched);
- preempt_dynamic_disable(preempt_schedule);
- preempt_dynamic_disable(preempt_schedule_notrace);
- preempt_dynamic_disable(irqentry_exit_cond_resched);
- preempt_dynamic_key_disable(preempt_lazy);
- if (mode != preempt_dynamic_mode)
- pr_info("Dynamic Preempt: none\n");
- break;
-
- case preempt_dynamic_voluntary:
- preempt_dynamic_enable(cond_resched);
- preempt_dynamic_enable(might_resched);
- preempt_dynamic_disable(preempt_schedule);
- preempt_dynamic_disable(preempt_schedule_notrace);
- preempt_dynamic_disable(irqentry_exit_cond_resched);
- preempt_dynamic_key_disable(preempt_lazy);
- if (mode != preempt_dynamic_mode)
- pr_info("Dynamic Preempt: voluntary\n");
- break;
-
case preempt_dynamic_full:
- preempt_dynamic_disable(cond_resched);
- preempt_dynamic_disable(might_resched);
- preempt_dynamic_enable(preempt_schedule);
- preempt_dynamic_enable(preempt_schedule_notrace);
- preempt_dynamic_enable(irqentry_exit_cond_resched);
preempt_dynamic_key_disable(preempt_lazy);
if (mode != preempt_dynamic_mode)
pr_info("Dynamic Preempt: full\n");
break;
case preempt_dynamic_lazy:
- preempt_dynamic_disable(cond_resched);
- preempt_dynamic_disable(might_resched);
- preempt_dynamic_enable(preempt_schedule);
- preempt_dynamic_enable(preempt_schedule_notrace);
- preempt_dynamic_enable(irqentry_exit_cond_resched);
preempt_dynamic_key_enable(preempt_lazy);
if (mode != preempt_dynamic_mode)
pr_info("Dynamic Preempt: lazy\n");
@@ -8099,11 +8031,7 @@ __setup("preempt=", setup_preempt_mode);
static void __init preempt_dynamic_init(void)
{
if (preempt_dynamic_mode == preempt_dynamic_undefined) {
- if (IS_ENABLED(CONFIG_PREEMPT_NONE)) {
- sched_dynamic_update(preempt_dynamic_none);
- } else if (IS_ENABLED(CONFIG_PREEMPT_VOLUNTARY)) {
- sched_dynamic_update(preempt_dynamic_voluntary);
- } else if (IS_ENABLED(CONFIG_PREEMPT_LAZY)) {
+ if (IS_ENABLED(CONFIG_PREEMPT_LAZY)) {
sched_dynamic_update(preempt_dynamic_lazy);
} else {
/* Default static call setting, nothing to do */
@@ -8123,8 +8051,6 @@ static void __init preempt_dynamic_init(void)
} \
EXPORT_SYMBOL_GPL(preempt_model_##mode)
-PREEMPT_MODEL_ACCESSOR(none);
-PREEMPT_MODEL_ACCESSOR(voluntary);
PREEMPT_MODEL_ACCESSOR(full);
PREEMPT_MODEL_ACCESSOR(lazy);
@@ -8137,7 +8063,7 @@ static inline void preempt_dynamic_init(void) { }
#endif /* CONFIG_PREEMPT_DYNAMIC */
const char *preempt_modes[] = {
- "none", "voluntary", "full", "lazy", NULL,
+ "full", "lazy", NULL,
};
const char *preempt_model_str(void)
@@ -8759,6 +8685,9 @@ int sched_cpu_activate(unsigned int cpu)
*/
sched_set_rq_online(rq, cpu);
+ /* preferred is subset of active and follows its state */
+ set_cpu_preferred(cpu, true);
+
return 0;
}
@@ -8772,6 +8701,8 @@ int sched_cpu_deactivate(unsigned int cpu)
if (ret)
return ret;
+ set_cpu_preferred(cpu, false);
+
/*
* Remove CPU from nohz.idle_cpus_mask to prevent participating in
* load balancing when not active
@@ -11349,3 +11280,88 @@ void sched_change_end(struct sched_change_ctx *ctx)
p->sched_class->prio_changed(rq, p, ctx->prio);
}
}
+
+#ifdef CONFIG_PREFERRED_CPU
+static DEFINE_PER_CPU(struct cpu_stop_work, npc_push_task_work);
+
+static int sched_non_preferred_cpu_push_stop(void *arg)
+{
+ struct task_struct *p = arg;
+ struct rq *rq = this_rq();
+ struct rq_flags rf;
+ int cpu;
+
+ if (cpu_preferred(rq->cpu)) {
+ scoped_guard(rq_lock_irqsave, rq)
+ rq->npc_push_work_pending = false;
+ put_task_struct(p);
+ return 0;
+ }
+
+ scoped_guard (raw_spinlock_irq, &p->pi_lock) {
+ /*
+ * select_fallback_rq() may acquire the rq lock in case of
+ * fallback. So call it before grabbing rq lock. If the task
+ * migrates to another CPU before the rq lock is acquired,
+ * subsequent validation of task's current rq will help to
+ * safely bail out.
+ */
+ cpu = select_fallback_rq(rq->cpu, p);
+ rq_lock(rq, &rf);
+ rq->npc_push_work_pending = false;
+ update_rq_clock(rq);
+ context_unsafe_alias(rq);
+
+ if (task_rq(p) == rq && task_on_rq_queued(p)) {
+ struct rq *dest_rq = __migrate_task(rq, &rf, p, cpu);
+
+ if (rq != dest_rq)
+ schedstat_inc(p->stats.nr_migrations_cpu_non_preferred);
+ rq = dest_rq;
+ }
+ rq_unlock(rq, &rf);
+ }
+
+ put_task_struct(p);
+ return 0;
+}
+
+/*
+ * Push the current task running on non-preferred CPU(npc).
+ * Using this non preferred CPU will lead to more contention
+ * in the host. So it is better not to use this CPU.
+ *
+ * Since task is running, call a stopper to push the task out. This is
+ * similar to how task moves during hotplug. In select_fallback_rq() a
+ * preferred CPU will be chosen and henceforth task shouldn't come back to
+ * this CPU again.
+ *
+ * Works for FAIR class only.
+ *
+ * If task is affined only on non-preferred CPUs, no point in moving it out.
+ */
+void sched_push_current_non_preferred_cpu(struct rq *rq)
+{
+ struct task_struct *push_task = rq->curr;
+
+ scoped_guard(rq_lock, rq) {
+ /* Push the task if its explicit affinity allows */
+ if (!task_can_migrate_to_preferred(push_task, rq->cpu))
+ return;
+
+ /* There is already a stopper thread. Don't race with it. */
+ if (rq->npc_push_work_pending)
+ return;
+
+ if (is_migration_disabled(push_task))
+ return;
+
+ rq->npc_push_work_pending = true;
+ }
+
+ /* sched_tick runs with interrupts disabled. */
+ get_task_struct(push_task);
+ stop_one_cpu_nowait(rq->cpu, sched_non_preferred_cpu_push_stop,
+ push_task, this_cpu_ptr(&npc_push_task_work));
+}
+#endif
diff --git a/kernel/sched/cputime.c b/kernel/sched/cputime.c
index 06bddaa738e5..f16970ca81d0 100644
--- a/kernel/sched/cputime.c
+++ b/kernel/sched/cputime.c
@@ -255,7 +255,7 @@ void __account_forceidle_time(struct task_struct *p, u64 delta)
* occasion account more time than the calling functions think elapsed.
*/
#ifdef CONFIG_PARAVIRT
-struct static_key paravirt_steal_enabled;
+DEFINE_STATIC_KEY_FALSE(paravirt_steal_enabled);
#ifdef CONFIG_HAVE_PV_STEAL_CLOCK_GEN
static u64 native_steal_clock(int cpu)
@@ -270,7 +270,7 @@ DEFINE_STATIC_CALL(pv_steal_clock, native_steal_clock);
static __always_inline u64 steal_account_process_time(u64 maxtime)
{
#ifdef CONFIG_PARAVIRT
- if (static_key_false(&paravirt_steal_enabled)) {
+ if (static_branch_unlikely(&paravirt_steal_enabled)) {
u64 steal;
steal = paravirt_steal_clock(smp_processor_id());
diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c
index 0663c00c41c0..c0ebdcde5fe5 100644
--- a/kernel/sched/deadline.c
+++ b/kernel/sched/deadline.c
@@ -1097,7 +1097,7 @@ static int start_dl_timer(struct sched_dl_entity *dl_se)
* chosen as the deadline is too small, don't even try to
* start the timer in the past!
*/
- if (ktime_us_delta(act, now) < 0)
+ if (ktime_before(act, now))
return 0;
/*
@@ -2773,11 +2773,14 @@ static void start_hrtick_dl(struct rq *rq, struct sched_dl_entity *dl_se)
* DL keeps current in tree, because ->deadline is not typically changed while
* a task is runnable.
*/
-static void set_next_task_dl(struct rq *rq, struct task_struct *p, bool first)
+static void set_next_task_dl(struct rq *rq, struct task_struct *p, enum snt_e type)
{
struct sched_dl_entity *dl_se = &p->dl;
struct dl_rq *dl_rq = &rq->dl;
+ if (type == SNT_REPICK)
+ return;
+
p->se.exec_start = rq_clock_task(rq);
if (on_dl_rq(&p->dl))
update_stats_wait_end_dl(dl_rq, dl_se);
@@ -2788,7 +2791,7 @@ static void set_next_task_dl(struct rq *rq, struct task_struct *p, bool first)
WARN_ON_ONCE(dl_rq->curr);
dl_rq->curr = dl_se;
- if (!first)
+ if (type != SNT_PICK)
return;
if (rq->donor->sched_class != &dl_sched_class)
diff --git a/kernel/sched/debug.c b/kernel/sched/debug.c
index 72236db67983..e6a3b516c703 100644
--- a/kernel/sched/debug.c
+++ b/kernel/sched/debug.c
@@ -73,13 +73,13 @@ static int sched_feat_show(struct seq_file *m, void *v)
#ifdef CONFIG_JUMP_LABEL
-#define jump_label_key__true STATIC_KEY_INIT_TRUE
-#define jump_label_key__false STATIC_KEY_INIT_FALSE
+#define jump_label_key__true { .key_true = STATIC_KEY_TRUE_INIT }
+#define jump_label_key__false { .key_false = STATIC_KEY_FALSE_INIT }
#define SCHED_FEAT(name, enabled) \
jump_label_key__##enabled ,
-struct static_key sched_feat_keys[__SCHED_FEAT_NR] = {
+union sched_feat_key sched_feat_keys[__SCHED_FEAT_NR] = {
#include "features.h"
};
@@ -87,12 +87,12 @@ struct static_key sched_feat_keys[__SCHED_FEAT_NR] = {
static void sched_feat_disable(int i)
{
- static_key_disable_cpuslocked(&sched_feat_keys[i]);
+ static_branch_disable_cpuslocked(&sched_feat_keys[i].key_true);
}
static void sched_feat_enable(int i)
{
- static_key_enable_cpuslocked(&sched_feat_keys[i]);
+ static_branch_enable_cpuslocked(&sched_feat_keys[i].key_false);
}
#else /* !CONFIG_JUMP_LABEL: */
static void sched_feat_disable(int i) { };
@@ -280,16 +280,10 @@ static ssize_t sched_dynamic_write(struct file *filp, const char __user *ubuf,
static int sched_dynamic_show(struct seq_file *m, void *v)
{
- int i = (IS_ENABLED(CONFIG_PREEMPT_RT) || IS_ENABLED(CONFIG_ARCH_HAS_PREEMPT_LAZY)) * 2;
int mode = READ_ONCE(preempt_dynamic_mode);
- int j;
- /* Count entries in NULL terminated preempt_modes */
- for (j = 0; preempt_modes[j]; j++)
- ;
- j -= !IS_ENABLED(CONFIG_ARCH_HAS_PREEMPT_LAZY);
-
- for (; i < j; i++) {
+ /* Stop at NULL terminator */
+ for (int i = 0; preempt_modes[i]; i++) {
if (mode == i)
seq_puts(m, "(");
seq_puts(m, preempt_modes[i]);
@@ -1446,6 +1440,7 @@ void proc_sched_show_task(struct task_struct *p, struct pid_namespace *ns,
P_SCHEDSTAT(nr_failed_migrations_running);
P_SCHEDSTAT(nr_failed_migrations_hot);
P_SCHEDSTAT(nr_forced_migrations);
+ P_SCHEDSTAT(nr_migrations_cpu_non_preferred);
P_SCHEDSTAT(nr_wakeups);
P_SCHEDSTAT(nr_wakeups_sync);
P_SCHEDSTAT(nr_wakeups_migrate);
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index e56c3c95018f..aed5286b82aa 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -24,6 +24,11 @@
DEFINE_RAW_SPINLOCK(scx_sched_lock);
+bool scx_allow_proxy_exec(const struct task_struct *p)
+{
+ return true;
+}
+
/*
* NOTE: sched_ext is in the process of growing multiple scheduler support and
* scx_root usage is in a transitional state. Naked dereferences are safe if the
@@ -1087,6 +1092,10 @@ static void schedule_deferred_locked(struct rq *rq)
schedule_deferred(rq);
}
+void scx_proxy_reenqueue_retry(struct rq *rq, struct task_struct *next)
+{
+}
+
void schedule_dsq_reenq(struct scx_sched *sch, struct scx_dispatch_q *dsq,
u64 reenq_flags, struct rq *locked_rq)
{
@@ -3021,10 +3030,13 @@ has_tasks:
return verdict;
}
-static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first)
+static void set_next_task_scx(struct rq *rq, struct task_struct *p, enum snt_e type)
{
struct scx_sched *sch = scx_task_sched(p);
+ if (type == SNT_REPICK)
+ return;
+
if (p->scx.flags & SCX_TASK_QUEUED) {
/*
* Core-sched might decide to execute @p before it is
@@ -3082,6 +3094,10 @@ static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first)
}
}
+void scx_proxy_donor_start(struct rq *rq)
+{
+}
+
static enum scx_cpu_preempt_reason
preempt_reason_from_class(const struct sched_class *class)
{
diff --git a/kernel/sched/ext/ext.h b/kernel/sched/ext/ext.h
index 0b7fc46aee08..3cfbfeb1bf9d 100644
--- a/kernel/sched/ext/ext.h
+++ b/kernel/sched/ext/ext.h
@@ -20,6 +20,9 @@ void scx_rq_deactivate(struct rq *rq);
int scx_check_setscheduler(struct task_struct *p, int policy);
bool task_should_scx(int policy);
bool scx_allow_ttwu_queue(const struct task_struct *p);
+bool scx_allow_proxy_exec(const struct task_struct *p);
+void scx_proxy_donor_start(struct rq *rq);
+void scx_proxy_reenqueue_retry(struct rq *rq, struct task_struct *next);
void init_sched_ext_class(void);
static inline u32 scx_cpuperf_target(s32 cpu)
@@ -54,6 +57,10 @@ static inline void scx_rq_deactivate(struct rq *rq) {}
static inline int scx_check_setscheduler(struct task_struct *p, int policy) { return 0; }
static inline bool task_on_scx(const struct task_struct *p) { return false; }
static inline bool scx_allow_ttwu_queue(const struct task_struct *p) { return true; }
+static inline bool scx_allow_proxy_exec(const struct task_struct *p) { return true; }
+static inline void scx_proxy_donor_start(struct rq *rq) {}
+static inline void scx_proxy_reenqueue_retry(struct rq *rq,
+ struct task_struct *next) {}
static inline void init_sched_ext_class(void) {}
#endif /* CONFIG_SCHED_CLASS_EXT */
diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c
index 8d38c3b7d792..56f4ab6d9ada 100644
--- a/kernel/sched/fair.c
+++ b/kernel/sched/fair.c
@@ -23,6 +23,7 @@
#include <linux/energy_model.h>
#include <linux/mmap_lock.h>
#include <linux/jiffies.h>
+#include <linux/math.h>
#include <linux/mm_api.h>
#include <linux/highmem.h>
#include <linux/hrtimer.h>
@@ -819,12 +820,6 @@ static u64 ineligible_vruntime(struct cfs_rq *cfs_rq)
if (curr && !curr->on_rq)
curr = NULL;
- /*
- * This is called from set_next_task_fair(.first=true) /
- * set_protect_slice() so curr had better be set and on_rq.
- */
- WARN_ON_ONCE(!curr);
-
if (weight) {
s64 runtime = cfs_rq->sum_w_vruntime;
@@ -1136,10 +1131,9 @@ static inline void set_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity
/* If there are shorter slices than se's one */
if (slice != se->slice) {
+ vprot = min_vruntime(vprot, se->vruntime + calc_delta_fair(slice, se));
if (sched_feat(PREEMPT_SHORT))
vprot = min_vruntime(vprot, ineligible_vruntime(cfs_rq));
- else
- vprot = min_vruntime(vprot, se->vruntime + calc_delta_fair(slice, se));
}
se->vprot = vprot;
@@ -1147,10 +1141,19 @@ static inline void set_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity
static inline void update_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity *se)
{
- u64 slice = cfs_rq_min_slice(cfs_rq);
u64 vruntime = min_vruntime(se->vruntime, avg_vruntime(cfs_rq));
+ u64 slice = normalized_sysctl_sched_base_slice;
+ u64 vprot;
- se->vprot = min_vruntime(se->vprot, vruntime + calc_delta_fair(slice, se));
+ if (sched_feat(RUN_TO_PARITY))
+ slice = cfs_rq_min_slice(cfs_rq);
+
+ vprot = min_vruntime(se->vprot, vruntime + calc_delta_fair(slice, se));
+
+ if (sched_feat(PREEMPT_SHORT) && slice != se->slice)
+ vprot = min_vruntime(vprot, ineligible_vruntime(cfs_rq));
+
+ se->vprot = vprot;
}
static inline bool protect_slice(struct sched_entity *se)
@@ -3712,7 +3715,7 @@ static void update_task_scan_period(struct task_struct *p,
p->mm->numa_next_scan = jiffies +
msecs_to_jiffies(p->numa_scan_period);
- return;
+ goto out;
}
/*
@@ -3756,7 +3759,10 @@ static void update_task_scan_period(struct task_struct *p,
p->numa_scan_period = clamp(p->numa_scan_period + diff,
task_scan_min(p), task_scan_max(p));
- memset(p->numa_faults_locality, 0, sizeof(p->numa_faults_locality));
+
+out:
+ memset(p->numa_faults_locality, 0,
+ sizeof(p->numa_faults_locality));
}
/*
@@ -8208,7 +8214,6 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
struct sched_entity *se = &p->se;
struct cfs_rq *cfs_rq = &rq->cfs;
unsigned long weight;
- bool curr;
if (task_is_throttled(p) && enqueue_throttled_task(p))
return;
@@ -8237,23 +8242,14 @@ enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
if (p->in_iowait)
cpufreq_update_util(rq, SCHED_CPUFREQ_IOWAIT);
- /*
- * XXX comment on the curr thing
- */
- curr = (cfs_rq->curr == se);
- if (curr)
- place_entity(cfs_rq, se, flags);
if (se->on_rq && se->sched_delayed)
requeue_delayed_entity(cfs_rq, se);
weight = enqueue_hierarchy(p, flags);
-
- if (!curr) {
- reweight_eevdf(cfs_rq, se, weight, false);
- place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED);
- __enqueue_entity(cfs_rq, se);
- }
+ reweight_eevdf(cfs_rq, se, weight, false);
+ place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED);
+ __enqueue_entity(cfs_rq, se);
if (!rq_h_nr_queued && rq->cfs.h_nr_queued)
dl_server_start(&rq->fair_server);
@@ -8673,8 +8669,8 @@ static int
sched_balance_find_dst_group_cpu(struct sched_group *group, struct task_struct *p, int this_cpu)
{
unsigned long load, min_load = ULONG_MAX;
- unsigned int min_exit_latency = UINT_MAX;
- u64 latest_idle_timestamp = 0;
+ u64 min_exit_latency = U64_MAX;
+ unsigned int nr_candidates = 0;
int least_loaded_cpu = this_cpu;
int shallowest_idle_cpu = -1;
int i;
@@ -8695,24 +8691,16 @@ sched_balance_find_dst_group_cpu(struct sched_group *group, struct task_struct *
if (available_idle_cpu(i)) {
struct cpuidle_state *idle = idle_get_state(rq);
- if (idle && idle->exit_latency < min_exit_latency) {
- /*
- * We give priority to a CPU whose idle state
- * has the smallest exit latency irrespective
- * of any idle timestamp.
- */
- min_exit_latency = idle->exit_latency;
- latest_idle_timestamp = rq->idle_stamp;
- shallowest_idle_cpu = i;
- } else if ((!idle || idle->exit_latency == min_exit_latency) &&
- rq->idle_stamp > latest_idle_timestamp) {
- /*
- * If equal or no active idle state, then
- * the most recently idled CPU might have
- * a warmer cache.
- */
- latest_idle_timestamp = rq->idle_stamp;
+ u64 exit_latency = idle ? idle->exit_latency : U64_MAX;
+
+ if (shallowest_idle_cpu == -1 || exit_latency < min_exit_latency) {
+ min_exit_latency = exit_latency;
shallowest_idle_cpu = i;
+ nr_candidates = 1;
+ } else if (exit_latency == min_exit_latency) {
+ nr_candidates++;
+ if (!reciprocal_scale(sched_rng(), nr_candidates))
+ shallowest_idle_cpu = i;
}
} else if (shallowest_idle_cpu == -1) {
load = cpu_load(cpu_rq(i));
@@ -10071,8 +10059,14 @@ static inline bool set_preempt_buddy(struct cfs_rq *cfs_rq, struct sched_entity
static inline bool set_short_buddy(struct cfs_rq *cfs_rq, struct sched_entity *pse)
{
- if (cfs_rq->next && cfs_rq->next->slice < pse->slice)
- return false;
+ if (cfs_rq->next) {
+ if (cfs_rq->next->slice < pse->slice)
+ return false;
+
+ if (cfs_rq->next->slice == pse->slice &&
+ entity_before(cfs_rq->next, pse))
+ return false;
+ }
set_next_buddy(cfs_rq, pse);
return true;
@@ -11438,21 +11432,7 @@ next:
*/
static void attach_tasks(struct lb_env *env)
{
- struct list_head *tasks = &env->tasks;
- struct task_struct *p;
- struct rq_flags rf;
-
- rq_lock(env->dst_rq, &rf);
- update_rq_clock(env->dst_rq);
-
- while (!list_empty(tasks)) {
- p = list_first_entry(tasks, struct task_struct, se.group_node);
- list_del_init(&p->se.group_node);
-
- attach_task(env->dst_rq, p);
- }
-
- rq_unlock(env->dst_rq, &rf);
+ __attach_tasks(env->dst_rq, &env->tasks);
}
#ifdef CONFIG_NO_HZ_COMMON
@@ -13745,7 +13725,7 @@ static int sched_balance_rq(int this_cpu, struct rq *this_rq,
};
bool need_unlock = false;
- cpumask_and(cpus, sched_domain_span(sd), cpu_active_mask);
+ cpumask_and(cpus, sched_domain_span(sd), cpu_preferred_mask);
schedstat_inc(sd->lb_count[idle]);
@@ -14870,10 +14850,8 @@ static int sched_balance_newidle(struct rq *this_rq, struct rq_flags *rf)
*/
this_rq->idle_stamp = rq_clock(this_rq);
- /*
- * Do not pull tasks towards !active CPUs...
- */
- if (!cpu_active(this_cpu))
+ /* Do not pull tasks towards !preferred CPUs */
+ if (!cpu_preferred(this_cpu))
return 0;
/*
@@ -15513,14 +15491,18 @@ static void switched_to_fair(struct rq *rq, struct task_struct *p)
}
}
-static void set_next_task_fair(struct rq *rq, struct task_struct *p, bool first)
+static void set_next_task_fair(struct rq *rq, struct task_struct *p, enum snt_e type)
{
struct sched_entity *se = &p->se;
- bool throttled = false;
struct cfs_rq *cfs_rq = &rq->cfs;
unsigned long weight = NICE_0_LOAD;
+ bool first = type == SNT_PICK;
+ bool throttled = false;
bool on_rq = se->on_rq;
+ if (type == SNT_REPICK)
+ goto repick;
+
clear_buddies(cfs_rq, se);
if (on_rq)
@@ -15564,11 +15546,18 @@ static void set_next_task_fair(struct rq *rq, struct task_struct *p, bool first)
WARN_ON_ONCE(se->sched_delayed);
- if (hrtick_enabled_fair(rq))
- hrtick_start_fair(rq, p);
-
update_misfit_status(p, rq);
sched_fair_update_stop_tick(rq, p);
+
+repick:
+ /*
+ * A same-task repick skips put_prev_task_fair(), but
+ * pick_task_fair() refreshed the entity hrtick_start_fair() reads
+ * before selecting it again. rq->cfs.curr identifies that entity,
+ * including with group scheduling.
+ */
+ if (hrtick_enabled_fair(rq))
+ hrtick_start_fair(rq, p);
}
void init_cfs_rq(struct cfs_rq *cfs_rq)
diff --git a/kernel/sched/idle.c b/kernel/sched/idle.c
index eb73b65ce6c4..76f3c84ca684 100644
--- a/kernel/sched/idle.c
+++ b/kernel/sched/idle.c
@@ -487,8 +487,11 @@ static void put_prev_task_idle(struct rq *rq, struct task_struct *prev, struct t
update_rq_avg_idle(rq);
}
-static void set_next_task_idle(struct rq *rq, struct task_struct *next, bool first)
+static void set_next_task_idle(struct rq *rq, struct task_struct *next, enum snt_e type)
{
+ if (type == SNT_REPICK)
+ return;
+
update_idle_core(rq);
scx_update_idle(rq, true, true);
schedstat_inc(rq->sched_goidle);
diff --git a/kernel/sched/rt.c b/kernel/sched/rt.c
index 85303add726d..1535046a23ff 100644
--- a/kernel/sched/rt.c
+++ b/kernel/sched/rt.c
@@ -1654,11 +1654,14 @@ static void wakeup_preempt_rt(struct rq *rq, struct task_struct *p, int flags)
check_preempt_equal_prio(rq, p);
}
-static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, bool first)
+static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, enum snt_e type)
{
struct sched_rt_entity *rt_se = &p->rt;
struct rt_rq *rt_rq = &rq->rt;
+ if (type == SNT_REPICK)
+ return;
+
p->se.exec_start = rq_clock_task(rq);
if (on_rt_rq(&p->rt))
update_stats_wait_end_rt(rt_rq, rt_se);
@@ -1666,7 +1669,7 @@ static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, bool f
/* The running task is never eligible for pushing */
dequeue_pushable_task(rq, p);
- if (!first)
+ if (type != SNT_PICK)
return;
/*
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index e656c7059bf8..7d2ec527b8a2 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -1326,6 +1326,9 @@ struct rq {
#ifdef CONFIG_PARAVIRT_TIME_ACCOUNTING
u64 prev_steal_time_rq;
#endif
+#ifdef CONFIG_PREFERRED_CPU
+ bool npc_push_work_pending;
+#endif
/* calc_load related fields */
unsigned long calc_load_update;
@@ -1371,7 +1374,6 @@ struct rq {
struct task_struct *core_pick;
struct sched_dl_entity *core_dl_server;
unsigned int core_enabled;
- unsigned int core_sched_seq;
struct rb_root core_tree;
/* shared state -- careful with sched_core_cpu_deactivate() */
@@ -2447,16 +2449,25 @@ extern __read_mostly unsigned int sysctl_sched_features;
#ifdef CONFIG_JUMP_LABEL
-#define SCHED_FEAT(name, enabled) \
-static __always_inline bool static_branch_##name(struct static_key *key) \
-{ \
- return static_key_##enabled(key); \
+union sched_feat_key {
+ struct static_key_true key_true;
+ struct static_key_false key_false;
+};
+
+#define sched_feat_branch_true(key) static_branch_likely(&(key)->key_true)
+#define sched_feat_branch_false(key) static_branch_unlikely(&(key)->key_false)
+
+#define SCHED_FEAT(name, enabled) \
+static __always_inline bool \
+static_branch_##name(union sched_feat_key *key) \
+{ \
+ return sched_feat_branch_##enabled(key); \
}
#include "features.h"
#undef SCHED_FEAT
-extern struct static_key sched_feat_keys[__SCHED_FEAT_NR];
+extern union sched_feat_key sched_feat_keys[__SCHED_FEAT_NR];
#define sched_feat(x) (static_branch_##x(&sched_feat_keys[__SCHED_FEAT_##x]))
#else /* !CONFIG_JUMP_LABEL: */
@@ -2508,6 +2519,12 @@ static inline bool task_is_blocked(struct task_struct *p)
return !!p->blocked_on;
}
+#ifdef CONFIG_SCHED_PROXY_EXEC
+void sched_proxy_block_task(struct rq *rq, struct task_struct *p);
+#else
+static inline void sched_proxy_block_task(struct rq *rq, struct task_struct *p) {}
+#endif
+
static inline int task_on_cpu(struct rq *rq, struct task_struct *p)
{
return p->on_cpu;
@@ -2527,11 +2544,17 @@ static inline int task_on_rq_migrating(struct task_struct *p)
#define WF_EXEC 0x02 /* Wakeup after exec; maps to SD_BALANCE_EXEC */
#define WF_FORK 0x04 /* Wakeup after fork; maps to SD_BALANCE_FORK */
#define WF_TTWU 0x08 /* Wakeup; maps to SD_BALANCE_WAKE */
-
-#define WF_SYNC 0x10 /* Waker goes to sleep after wakeup */
+/*
+ * Hint that the caller expects the waker to sleep soon.
+ * Scheduler classes may use it for placement or preemption.
+ * Callers must not rely on it to prevent migration,
+ * preserve CPU locality or make the wakee run next.
+ */
+#define WF_SYNC 0x10
#define WF_MIGRATED 0x20 /* Internal use, task got migrated */
#define WF_CURRENT_CPU 0x40 /* Prefer to move the wakee to the current CPU. */
#define WF_RQ_SELECTED 0x80 /* ->select_task_rq() was called */
+#define WF_TTWU_RQ 0x100 /* Wakeup completed through ttwu_runnable() */
static_assert(WF_EXEC == SD_BALANCE_EXEC);
static_assert(WF_FORK == SD_BALANCE_FORK);
@@ -2621,6 +2644,12 @@ struct affinity_context {
extern s64 update_curr_common(struct rq *rq);
+enum snt_e {
+ SNT_NORMAL, /* set_next_task() */
+ SNT_PICK, /* put_prev_set_next_task(): prev != next */
+ SNT_REPICK, /* put_prev_set_next_task(): prev == next */
+};
+
struct sched_class {
#ifdef CONFIG_UCLAMP_TASK
@@ -2678,7 +2707,7 @@ struct sched_class {
* __schedule: rq->lock
*/
void (*put_prev_task)(struct rq *rq, struct task_struct *p, struct task_struct *next);
- void (*set_next_task)(struct rq *rq, struct task_struct *p, bool first);
+ void (*set_next_task)(struct rq *rq, struct task_struct *p, enum snt_e type);
/*
* select_task_rq: p->pi_lock
@@ -2781,7 +2810,7 @@ static inline void put_prev_task(struct rq *rq, struct task_struct *prev)
static inline void set_next_task(struct rq *rq, struct task_struct *next)
{
- next->sched_class->set_next_task(rq, next, false);
+ next->sched_class->set_next_task(rq, next, SNT_NORMAL);
}
static inline void
@@ -2802,11 +2831,13 @@ static inline void put_prev_set_next_task(struct rq *rq,
__put_prev_set_next_dl_server(rq, prev, next);
- if (next == prev)
+ if (next == prev) {
+ next->sched_class->set_next_task(rq, next, SNT_REPICK);
return;
+ }
prev->sched_class->put_prev_task(rq, prev, next);
- next->sched_class->set_next_task(rq, next, true);
+ next->sched_class->set_next_task(rq, next, SNT_PICK);
}
/*
@@ -3139,6 +3170,25 @@ static inline void attach_one_task(struct rq *rq, struct task_struct *p)
attach_task(rq, p);
}
+/*
+ * __attach_tasks() - attaches a list of tasks (using se.group_node) to
+ * the new rq
+ */
+static inline void __attach_tasks(struct rq *rq, struct list_head *tasks)
+{
+ guard(rq_lock)(rq);
+ update_rq_clock(rq);
+
+ while (!list_empty(tasks)) {
+ struct task_struct *p;
+
+ p = list_first_entry(tasks, struct task_struct, se.group_node);
+ list_del_init(&p->se.group_node);
+
+ attach_task(rq, p);
+ }
+}
+
#ifdef CONFIG_PREEMPT_RT
# define SCHED_NR_MIGRATE_BREAK 8
#else
@@ -4252,4 +4302,10 @@ DEFINE_CLASS_IS_UNCONDITIONAL(sched_change)
#include "ext/ext.h"
+#ifdef CONFIG_PREFERRED_CPU
+void sched_push_current_non_preferred_cpu(struct rq *rq);
+#else /* !CONFIG_PREFERRED_CPU */
+static inline void sched_push_current_non_preferred_cpu(struct rq *rq) { }
+#endif
+
#endif /* _KERNEL_SCHED_SCHED_H */
diff --git a/kernel/sched/stop_task.c b/kernel/sched/stop_task.c
index c909ca0d8c87..1e0109ec36b3 100644
--- a/kernel/sched/stop_task.c
+++ b/kernel/sched/stop_task.c
@@ -27,8 +27,11 @@ wakeup_preempt_stop(struct rq *rq, struct task_struct *p, int flags)
/* we're never preempted */
}
-static void set_next_task_stop(struct rq *rq, struct task_struct *stop, bool first)
+static void set_next_task_stop(struct rq *rq, struct task_struct *stop, enum snt_e type)
{
+ if (type == SNT_REPICK)
+ return;
+
stop->se.exec_start = rq_clock_task(rq);
}
diff --git a/kernel/sched/wait.c b/kernel/sched/wait.c
index d033f600f48c..477e4bf9c01e 100644
--- a/kernel/sched/wait.c
+++ b/kernel/sched/wait.c
@@ -174,15 +174,11 @@ EXPORT_SYMBOL_GPL(__wake_up_locked_key);
* @mode: which threads
* @key: opaque value to be passed to wakeup targets
*
- * The sync wakeup differs that the waker knows that it will schedule
- * away soon, so while the target thread will be woken up, it will not
- * be migrated to another CPU - ie. the two threads are 'synchronized'
- * with each other. This can prevent needless bouncing between CPUs.
+ * Passes WF_SYNC to waitqueue wake functions. The default wake function
+ * forwards it to the scheduler; see WF_SYNC for the hint's semantics.
*
- * On UP it can prevent extra preemption.
- *
- * If this function wakes up a task, it executes a full memory barrier before
- * accessing the task state.
+ * If this function wakes up a task, it executes a full memory barrier
+ * before accessing the task state.
*/
void __wake_up_sync_key(struct wait_queue_head *wq_head, unsigned int mode,
void *key)
@@ -200,15 +196,7 @@ EXPORT_SYMBOL_GPL(__wake_up_sync_key);
* @mode: which threads
* @key: opaque value to be passed to wakeup targets
*
- * The sync wakeup differs in that the waker knows that it will schedule
- * away soon, so while the target thread will be woken up, it will not
- * be migrated to another CPU - ie. the two threads are 'synchronized'
- * with each other. This can prevent needless bouncing between CPUs.
- *
- * On UP it can prevent extra preemption.
- *
- * If this function wakes up a task, it executes a full memory barrier before
- * accessing the task state.
+ * Same as __wake_up_sync_key(), but called with @wq_head->lock held.
*/
void __wake_up_locked_sync_key(struct wait_queue_head *wq_head,
unsigned int mode, void *key)
diff --git a/kernel/time/hrtimer.c b/kernel/time/hrtimer.c
index cbf1693c86b3..17dd38a6cee7 100644
--- a/kernel/time/hrtimer.c
+++ b/kernel/time/hrtimer.c
@@ -780,7 +780,8 @@ static void hrtimer_switch_to_hres(void)
return;
}
base->hres_active = true;
- hrtimer_resolution = HIGH_RES_NSEC;
+ if (hrtimer_resolution != HIGH_RES_NSEC)
+ hrtimer_resolution = HIGH_RES_NSEC;
tick_setup_sched_timer(true);
/* "Retrigger" the interrupt to get things going */
@@ -2003,7 +2004,7 @@ bool hrtimer_active(const struct hrtimer *timer)
base = READ_ONCE(timer->base);
seq = raw_read_seqcount_begin(&base->seq);
- if (timer->is_queued || base->running == timer)
+ if (timer->is_queued || READ_ONCE(base->running) == timer)
return true;
} while (read_seqcount_retry(&base->seq, seq) || base != READ_ONCE(timer->base));
@@ -2040,7 +2041,7 @@ static void __run_hrtimer(struct hrtimer_cpu_base *cpu_base, struct hrtimer_cloc
lockdep_assert_held(&cpu_base->lock);
debug_hrtimer_deactivate(timer);
- base->running = timer;
+ WRITE_ONCE(base->running, timer);
/*
* Separate the ->running assignment from the ->is_queued assignment.
@@ -2099,7 +2100,7 @@ static void __run_hrtimer(struct hrtimer_cpu_base *cpu_base, struct hrtimer_cloc
raw_write_seqcount_barrier(&base->seq);
WARN_ON_ONCE(base->running != timer);
- base->running = NULL;
+ WRITE_ONCE(base->running, NULL);
}
static void __hrtimer_run_queues(struct hrtimer_cpu_base *cpu_base, ktime_t now,
@@ -2323,9 +2324,9 @@ void hrtimer_run_queues(void)
static enum hrtimer_restart hrtimer_wakeup(struct hrtimer *timer)
{
struct hrtimer_sleeper *t = container_of(timer, struct hrtimer_sleeper, timer);
- struct task_struct *task = t->task;
+ struct task_struct *task = hrtimer_sleeper_task_get(t);
- t->task = NULL;
+ hrtimer_sleeper_task_set(t, NULL);
if (task)
wake_up_process(task);
@@ -2354,7 +2355,7 @@ void hrtimer_sleeper_start_expires(struct hrtimer_sleeper *sl, enum hrtimer_mode
/* If already expired, clear the task pointer and set current state to running */
if (!hrtimer_start_expires_user(&sl->timer, mode)) {
- sl->task = NULL;
+ hrtimer_sleeper_task_set(sl, NULL);
__set_current_state(TASK_RUNNING);
}
}
@@ -2388,7 +2389,7 @@ static void __hrtimer_setup_sleeper(struct hrtimer_sleeper *sl, clockid_t clock_
}
__hrtimer_setup(&sl->timer, hrtimer_wakeup, clock_id, mode);
- sl->task = current;
+ hrtimer_sleeper_task_set(sl, current);
}
/**
@@ -2432,17 +2433,17 @@ static int __sched do_nanosleep(struct hrtimer_sleeper *t, enum hrtimer_mode mod
set_current_state(TASK_INTERRUPTIBLE|TASK_FREEZABLE);
hrtimer_sleeper_start_expires(t, mode);
- if (likely(t->task))
+ if (likely(hrtimer_sleeper_task_get(t)))
schedule();
hrtimer_cancel(&t->timer);
mode = HRTIMER_MODE_ABS;
- } while (t->task && !signal_pending(current));
+ } while (hrtimer_sleeper_task_get(t) && !signal_pending(current));
__set_current_state(TASK_RUNNING);
- if (!t->task)
+ if (!hrtimer_sleeper_task_get(t))
return 0;
restart = &current->restart_block;
diff --git a/kernel/time/posix-cpu-timers.c b/kernel/time/posix-cpu-timers.c
index 0bf4fcd969c8..cd75d4bb5b64 100644
--- a/kernel/time/posix-cpu-timers.c
+++ b/kernel/time/posix-cpu-timers.c
@@ -439,6 +439,38 @@ static void trigger_base_recalc_expires(struct k_itimer *timer,
base->nextevt = 0;
}
+static inline bool cpu_timer_enqueue(struct timerqueue_head *head,
+ struct cpu_timer *ctmr)
+{
+ ctmr->head = head;
+ return timerqueue_add(head, &ctmr->node);
+}
+
+static inline bool cpu_timer_queued(struct cpu_timer *ctmr)
+{
+ return !!ctmr->head;
+}
+
+static inline bool cpu_timer_dequeue(struct cpu_timer *ctmr)
+{
+ if (cpu_timer_queued(ctmr)) {
+ timerqueue_del(ctmr->head, &ctmr->node);
+ ctmr->head = NULL;
+ return true;
+ }
+ return false;
+}
+
+static inline u64 cpu_timer_getexpires(struct cpu_timer *ctmr)
+{
+ return ctmr->node.expires;
+}
+
+static inline void cpu_timer_setexpires(struct cpu_timer *ctmr, u64 exp)
+{
+ ctmr->node.expires = exp;
+}
+
/*
* Dequeue the timer and reset the base if it was its earliest expiration.
* It makes sure the next tick recalculates the base next expiration so we
@@ -607,6 +639,7 @@ static int posix_cpu_timer_del(struct k_itimer *timer)
}
if (!ret) {
+ WARN_ON_ONCE(cpu_timer_queued(&timer->it.cpu));
put_pid(timer->it.cpu.pid);
timer->it_status = POSIX_TIMER_DISARMED;
}
@@ -639,18 +672,50 @@ static void cleanup_timers(struct posix_cputimers *pct)
cleanup_timerqueue(&pct->bases[CPUCLOCK_SCHED].tqhead);
}
+static inline void posix_cpu_timers_exit_work(void);
+
/*
- * These are both called with the siglock held, when the current thread
- * is being reaped. When the final (leader) thread in the group is reaped,
- * posix_cpu_timers_exit_group will be called after posix_cpu_timers_exit.
+ * Invoked from posixtimer_exit_task() after PF_EXITING was set in tsk::flags or
+ * from posixtimer_exec_cleanup().
*/
-void posix_cpu_timers_exit(struct task_struct *tsk)
+void posix_cpu_timers_exit_task(void)
{
- cleanup_timers(&tsk->posix_cputimers);
+ posix_cpu_timers_exit_work();
+
+ guard(spinlock_irq)(&current->sighand->siglock);
+ cleanup_timers(&current->posix_cputimers);
}
-void posix_cpu_timers_exit_group(struct task_struct *tsk)
+
+/*
+ * Invoked from posixtimer_exit_group() after PF_EXITING was set in tsk::flags.
+ */
+void posix_cpu_timers_exit_group(void)
{
- cleanup_timers(&tsk->signal->posix_cputimers);
+ posix_cpu_timers_exit_task();
+
+ guard(spinlock_irq)(&current->sighand->siglock);
+ cleanup_timers(&current->signal->posix_cputimers);
+}
+
+/*
+ * This function validates that POSIX CPU timers can be safely enqueued on the
+ * target task.
+ *
+ * Enqueue is allowed when PF_EXITING is not set. If set then it is only allowed
+ * for process shared timers (type = PIDTYPE_TGID) as long as tsk::signal::flags
+ * does not have SIGNAL_GROUP_EXIT set. PIDTYPE_PID targets are not allowed at
+ * all when the task has PF_EXITING set.
+ *
+ * This guarantees that after the POSIX timer cleanup in posixtimer_exit() no
+ * POSIX CPU timers are queued on the task or in case of a group exit on the
+ * process.
+ */
+static inline bool task_can_enqueue_timer(struct task_struct *tsk, enum pid_type type)
+{
+ if (likely(!(tsk->flags & PF_EXITING)))
+ return true;
+
+ return type == PIDTYPE_TGID && !(tsk->signal->flags & SIGNAL_GROUP_EXIT);
}
/*
@@ -663,7 +728,13 @@ static void arm_timer(struct k_itimer *timer, struct task_struct *p)
struct cpu_timer *ctmr = &timer->it.cpu;
u64 newexp = cpu_timer_getexpires(ctmr);
+ lockdep_assert_held(&p->sighand->siglock);
+
timer->it_status = POSIX_TIMER_ARMED;
+
+ if (unlikely(!task_can_enqueue_timer(p, clock_pid_type(timer->it_clock))))
+ return;
+
if (!cpu_timer_enqueue(&base->tqhead, ctmr))
return;
@@ -1201,6 +1272,20 @@ static void posix_cpu_timers_work(struct callback_head *work)
mutex_unlock(&cw->mutex);
}
+static inline void posix_cpu_timers_exit_work(void)
+{
+ /* Canceling the work is only valid for exit() but not for exec() */
+ if (!(current->flags & PF_EXITING))
+ return;
+ /*
+ * current->flags has PF_EXITING set so this can be done lockless and
+ * with interrupts enabled as PF_EXITING prevents the interrupt from
+ * scheduling the work.
+ */
+ if (current->posix_cputimers_work.scheduled)
+ task_work_cancel(current, &current->posix_cputimers_work.work);
+}
+
/*
* Invoked from the posix-timer core when a cancel operation failed because
* the timer is marked firing. The caller holds rcu_read_lock(), which
@@ -1331,6 +1416,8 @@ static inline void __run_posix_cpu_timers(struct task_struct *tsk)
lockdep_posixtimer_exit();
}
+static inline void posix_cpu_timers_exit_work(void) { }
+
static void posix_cpu_timer_wait_running(struct k_itimer *timr)
{
cpu_relax();
@@ -1477,7 +1564,7 @@ void run_posix_cpu_timers(void)
* posix_cpu_timer_del() may fail to lock_task_sighand(tsk) and
* miss timer->it.cpu.firing != 0.
*/
- if (tsk->exit_state)
+ if (tsk->flags & PF_EXITING)
return;
/*
diff --git a/kernel/time/posix-timers.c b/kernel/time/posix-timers.c
index 436ba794cc0b..188dbedbffca 100644
--- a/kernel/time/posix-timers.c
+++ b/kernel/time/posix-timers.c
@@ -1077,13 +1077,9 @@ SYSCALL_DEFINE1(timer_delete, timer_t, timer_id)
return 0;
}
-/*
- * Invoked from do_exit() when the last thread of a thread group exits.
- * At that point no other task can access the timers of the dying
- * task anymore.
- */
-void exit_itimers(struct task_struct *tsk)
+static void posixtimer_delete_timers(void)
{
+ struct task_struct *tsk = current;
struct hlist_head timers;
struct hlist_node *next;
struct k_itimer *timer;
@@ -1120,6 +1116,24 @@ void exit_itimers(struct task_struct *tsk)
}
}
+void posixtimer_exit(bool group_dead)
+{
+ if (group_dead) {
+ hrtimer_cancel(&current->signal->real_timer);
+ posix_cpu_timers_exit_group();
+ posixtimer_delete_timers();
+ } else {
+ posix_cpu_timers_exit_task();
+ }
+}
+
+void posixtimer_exec(void)
+{
+ posix_cpu_timers_exit_task();
+ posixtimer_delete_timers();
+ flush_itimer_signals();
+}
+
SYSCALL_DEFINE2(clock_settime, const clockid_t, which_clock,
const struct __kernel_timespec __user *, tp)
{
diff --git a/kernel/time/posix-timers.h b/kernel/time/posix-timers.h
index 4ea9611dd716..79fd7ea71046 100644
--- a/kernel/time/posix-timers.h
+++ b/kernel/time/posix-timers.h
@@ -51,3 +51,6 @@ int common_timer_set(struct k_itimer *timr, int flags,
struct itimerspec64 *old_setting);
void posix_timer_set_common(struct k_itimer *timer, struct itimerspec64 *new_setting);
int common_timer_del(struct k_itimer *timer);
+
+void posix_cpu_timers_exit_task(void);
+void posix_cpu_timers_exit_group(void);
diff --git a/kernel/time/sleep_timeout.c b/kernel/time/sleep_timeout.c
index 3c90574bd904..ad8c415851ae 100644
--- a/kernel/time/sleep_timeout.c
+++ b/kernel/time/sleep_timeout.c
@@ -212,7 +212,7 @@ int __sched schedule_hrtimeout_range_clock(ktime_t *expires, u64 delta,
hrtimer_set_expires_range_ns(&t.timer, *expires, delta);
hrtimer_sleeper_start_expires(&t, mode);
- if (likely(t.task))
+ if (likely(hrtimer_sleeper_task_get(&t)))
schedule();
hrtimer_cancel(&t.timer);
@@ -220,7 +220,7 @@ int __sched schedule_hrtimeout_range_clock(ktime_t *expires, u64 delta,
__set_current_state(TASK_RUNNING);
- return !t.task ? 0 : -EINTR;
+ return !hrtimer_sleeper_task_get(&t) ? 0 : -EINTR;
}
EXPORT_SYMBOL_GPL(schedule_hrtimeout_range_clock);
diff --git a/kernel/time/tick-sched.c b/kernel/time/tick-sched.c
index 6c3fea386713..a7893a079a83 100644
--- a/kernel/time/tick-sched.c
+++ b/kernel/time/tick-sched.c
@@ -738,14 +738,11 @@ bool tick_nohz_tick_stopped_cpu(int cpu)
*/
static void tick_nohz_update_jiffies(ktime_t now)
{
- unsigned long flags;
+ /* Reached only from irq_enter_rcu(), i.e. hard interrupt entry. */
+ lockdep_assert_irqs_disabled();
__this_cpu_write(tick_cpu_sched.idle_waketime, now);
-
- local_irq_save(flags);
tick_do_update_jiffies64(now);
- local_irq_restore(flags);
-
touch_softlockup_watchdog_sched();
}
@@ -819,7 +816,7 @@ u64 get_jiffies_update(unsigned long *basej)
*/
static ktime_t tick_nohz_next_event(struct tick_sched *ts, int cpu)
{
- u64 basemono, next_tick, delta, expires;
+ u64 basemono, next_tick, expires;
unsigned long basejiff;
int tick_cpu;
@@ -859,8 +856,7 @@ static ktime_t tick_nohz_next_event(struct tick_sched *ts, int cpu)
* If the tick is due in the next period, keep it ticking or
* force prod the timer.
*/
- delta = next_tick - basemono;
- if (delta <= (u64)TICK_NSEC) {
+ if (next_tick - basemono <= (u64)TICK_NSEC) {
/*
* We've not stopped the tick yet, and there's a timer in the
* next period, so no point in stopping it either, bail.
@@ -876,17 +872,19 @@ static ktime_t tick_nohz_next_event(struct tick_sched *ts, int cpu)
* the sleep time to the timekeeping 'max_deferment' value.
* Otherwise we can sleep as long as we want.
*/
- delta = timekeeping_max_deferment();
tick_cpu = READ_ONCE(tick_do_timer_cpu);
if (tick_cpu != cpu &&
- (tick_cpu != TICK_DO_TIMER_NONE || !tick_sched_flag_test(ts, TS_FLAG_DO_TIMER_LAST)))
- delta = KTIME_MAX;
-
- /* Calculate the next expiry time */
- if (delta < (KTIME_MAX - basemono))
- expires = basemono + delta;
- else
+ (tick_cpu != TICK_DO_TIMER_NONE || !tick_sched_flag_test(ts, TS_FLAG_DO_TIMER_LAST))) {
expires = KTIME_MAX;
+ } else {
+ expires = timekeeping_max_deferment();
+
+ /* Calculate the next expiry time */
+ if (expires < (KTIME_MAX - basemono))
+ expires += basemono;
+ else
+ expires = KTIME_MAX;
+ }
ts->timer_expires = min_t(u64, expires, next_tick);
diff --git a/kernel/time/time_test.c b/kernel/time/time_test.c
index 1b99180da288..8b718767b3ba 100644
--- a/kernel/time/time_test.c
+++ b/kernel/time/time_test.c
@@ -87,8 +87,24 @@ static void time64_to_tm_test_date_range(struct kunit *test)
}
}
+static void time64_to_tm_test_wide_day_count(struct kunit *test)
+{
+ /* 2^31 days: the first count that does not fit in a 32-bit long. */
+ time64_t timestamp = (1LL << 31) * 86400;
+ struct tm result;
+
+ time64_to_tm(timestamp, 0, &result);
+
+ KUNIT_EXPECT_EQ(test, result.tm_year, 5879680);
+ KUNIT_EXPECT_EQ(test, result.tm_mon, 6);
+ KUNIT_EXPECT_EQ(test, result.tm_mday, 12);
+ KUNIT_EXPECT_EQ(test, result.tm_yday, 193);
+ KUNIT_EXPECT_EQ(test, result.tm_wday, 6);
+}
+
static struct kunit_case time_test_cases[] = {
KUNIT_CASE_SLOW(time64_to_tm_test_date_range),
+ KUNIT_CASE(time64_to_tm_test_wide_day_count),
{}
};
diff --git a/kernel/time/timeconv.c b/kernel/time/timeconv.c
index 59b922c826e7..aed3af950fa0 100644
--- a/kernel/time/timeconv.c
+++ b/kernel/time/timeconv.c
@@ -49,8 +49,9 @@ void time64_to_tm(time64_t totalsecs, int offset, struct tm *result)
u32 u32tmp, day_of_century, year_of_century, day_of_year, month, day;
u64 u64tmp, udays, century, year;
bool is_Jan_or_Feb, is_leap_year;
- long days, rem;
int remainder;
+ long rem;
+ s64 days;
days = div_s64_rem(totalsecs, SECS_PER_DAY, &remainder);
rem = remainder;
@@ -70,7 +71,8 @@ void time64_to_tm(time64_t totalsecs, int offset, struct tm *result)
result->tm_sec = rem % 60;
/* January 1, 1970 was a Thursday. */
- result->tm_wday = (4 + days) % 7;
+ div_s64_rem(days + 4, 7, &remainder);
+ result->tm_wday = remainder;
if (result->tm_wday < 0)
result->tm_wday += 7;
diff --git a/kernel/time/timekeeping.c b/kernel/time/timekeeping.c
index ea2e6e55f37b..d54c4d303db6 100644
--- a/kernel/time/timekeeping.c
+++ b/kernel/time/timekeeping.c
@@ -861,8 +861,10 @@ static void timekeeping_update_from_shadow(struct tk_data *tkd, unsigned int act
*
* Write xtime_sec first so that even if the memcpy() tears the store
* data integrity is provided for ktime_get_real_seconds().
+ * The same goes for ktime_sec and ktime_get_seconds().
*/
WRITE_ONCE(tkd->timekeeper.xtime_sec, tk->xtime_sec);
+ WRITE_ONCE(tkd->timekeeper.ktime_sec, tk->ktime_sec);
memcpy(&tkd->timekeeper, tk, sizeof(*tk));
write_seqcount_end(&tkd->seq);
}
@@ -1169,7 +1171,7 @@ time64_t ktime_get_seconds(void)
struct timekeeper *tk = &tk_core.timekeeper;
WARN_ON(timekeeping_suspended);
- return tk->ktime_sec;
+ return READ_ONCE(tk->ktime_sec);
}
EXPORT_SYMBOL_GPL(ktime_get_seconds);
diff --git a/kernel/time/timer.c b/kernel/time/timer.c
index ae9abf14688e..42afdcb229d8 100644
--- a/kernel/time/timer.c
+++ b/kernel/time/timer.c
@@ -890,7 +890,7 @@ static inline void detach_timer(struct timer_list *timer, bool clear_pending)
__hlist_del(entry);
if (clear_pending)
- entry->pprev = NULL;
+ WRITE_ONCE(entry->pprev, NULL);
entry->next = LIST_POISON2;
}
diff --git a/kernel/time/timer_migration.c b/kernel/time/timer_migration.c
index 059d43355e65..f920e73fff51 100644
--- a/kernel/time/timer_migration.c
+++ b/kernel/time/timer_migration.c
@@ -715,7 +715,7 @@ static void __tmigr_cpu_activate(struct tmigr_cpu *tmc)
trace_tmigr_cpu_active(tmc);
- tmc->cpuevt.ignore = true;
+ WRITE_ONCE(tmc->cpuevt.ignore, true);
WRITE_ONCE(tmc->wakeup, KTIME_MAX);
walk_groups(&tmigr_active_up, &data, tmc);
@@ -1258,7 +1258,7 @@ u64 tmigr_cpu_new_timer(u64 nextexp)
ret = READ_ONCE(tmc->wakeup);
if (nextexp != KTIME_MAX) {
if (nextexp != tmc->cpuevt.nextevt.expires ||
- tmc->cpuevt.ignore) {
+ READ_ONCE(tmc->cpuevt.ignore)) {
ret = tmigr_new_timer(tmc, nextexp);
/*
* Make sure the reevaluation of timers in idle path
@@ -1362,7 +1362,7 @@ static u64 __tmigr_cpu_deactivate(struct tmigr_cpu *tmc, u64 nextexp)
* or CPU goes offline.
*/
if (nextexp != KTIME_MAX)
- tmc->cpuevt.ignore = false;
+ WRITE_ONCE(tmc->cpuevt.ignore, false);
walk_groups(&tmigr_inactive_up, &data, tmc);
return data.firstexp;
diff --git a/kernel/time/vsyscall.c b/kernel/time/vsyscall.c
index aa59919b8f2c..0e4b499328c0 100644
--- a/kernel/time/vsyscall.c
+++ b/kernel/time/vsyscall.c
@@ -41,14 +41,12 @@ static inline void update_vdso_time_data(struct vdso_time_data *vdata, struct ti
nsec = tk->tkr_mono.xtime_nsec;
nsec += ((u64)tk->wall_to_monotonic.tv_nsec << tk->tkr_mono.shift);
- while (nsec >= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift)) {
- nsec -= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift);
- vdso_ts->sec++;
- }
- vdso_ts->nsec = nsec;
+ vdso_ts->sec += __iter_div64_u64_rem(nsec, (u64)NSEC_PER_SEC << tk->tkr_mono.shift,
+ &vdso_ts->nsec);
/* Copy MONOTONIC time for BOOTTIME */
sec = vdso_ts->sec;
+ nsec = vdso_ts->nsec;
/* Add the boot offset */
sec += tk->monotonic_to_boot.tv_sec;
nsec += (u64)tk->monotonic_to_boot.tv_nsec << tk->tkr_mono.shift;
@@ -56,12 +54,8 @@ static inline void update_vdso_time_data(struct vdso_time_data *vdata, struct ti
/* CLOCK_BOOTTIME */
vdso_ts = &vc[CS_HRES_COARSE].basetime[CLOCK_BOOTTIME];
vdso_ts->sec = sec;
-
- while (nsec >= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift)) {
- nsec -= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift);
- vdso_ts->sec++;
- }
- vdso_ts->nsec = nsec;
+ vdso_ts->sec += __iter_div64_u64_rem(nsec, (u64)NSEC_PER_SEC << tk->tkr_mono.shift,
+ &vdso_ts->nsec);
/* CLOCK_MONOTONIC_RAW */
vdso_ts = &vc[CS_RAW].basetime[CLOCK_MONOTONIC_RAW];
@@ -161,11 +155,11 @@ void vdso_time_update_aux(struct timekeeper *tk)
vdso_ts->sec = tk->xtime_sec + tk->monotonic_to_aux.tv_sec;
- nsec = tk->tkr_mono.xtime_nsec >> tk->tkr_mono.shift;
- nsec += tk->monotonic_to_aux.tv_nsec;
- vdso_ts->sec += __iter_div_u64_rem(nsec, NSEC_PER_SEC, &nsec);
- nsec = nsec << tk->tkr_mono.shift;
- vdso_ts->nsec = nsec;
+ nsec = tk->tkr_mono.xtime_nsec;
+ nsec += (u64)tk->monotonic_to_aux.tv_nsec << tk->tkr_mono.shift;
+ vdso_ts->sec += __iter_div64_u64_rem(nsec,
+ (u64)NSEC_PER_SEC << tk->tkr_mono.shift,
+ &vdso_ts->nsec);
}
__arch_update_vdso_clock(vc);