summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorHui Su <sh_def@163.com>2026-09-16 01:31:38 +0900
committerPeter Zijlstra <peterz@infradead.org>2026-09-25 12:45:58 +0200
commitfbbc63fed0b09c8c5cf3972db8922ae98406ceec (patch)
tree1cb39cbacc5aba4c6f3a1d2f030ccafff3cc36eb
parent819224e506bc7c2d61ec6a58ec6e876b505abce9 (diff)
downloadlinux-next-fbbc63fed0b09c8c5cf3972db8922ae98406ceec.tar.gz
linux-next-fbbc63fed0b09c8c5cf3972db8922ae98406ceec.zip
sched/core: Remove redundant core_sched_seq
core_sched_seq records whether an rq has consumed the pick made by the last core-wide selection. However, rq->core_pick already carries the same per-rq state. A core-wide selection leaves core_pick populated only for siblings which still need to consume their picks. It is cleared when the current CPU consumes its pick, when a sibling is already running the selected task, or when a pending pick is consumed through the fastpath. The CPU offline path clears it as well. core_task_seq and core_pick_seq serve a separate purpose. core_task_seq changes when the task set changes and for each new core-wide pick, while core_pick_seq records the task sequence on which the selection was made. Their equality therefore establishes that a pending core_pick is still valid. Consequently, once core_pick_seq == core_task_seq, a non-NULL core_pick is sufficient to tell that this rq still has a pick to consume. core_sched_seq duplicates that pending/consumed state. Remove core_sched_seq and use core_pick directly as the pending marker. Testing included an instrumented comparison of the old and new pending predicates under core-scheduling stress. Additional guest tests exercised repeated core-cookie lifecycles, forced idle, and CPU hotplug under load. No predicate divergence or kernel warning was observed. Signed-off-by: Hui Su <sh_def@163.com> Signed-off-by: Peter Zijlstra (Intel) <peterz@infradead.org> Link: https://patch.msgid.link/20260915163138.2973969-1-sh_def@163.com
-rw-r--r--kernel/sched/core.c10
-rw-r--r--kernel/sched/sched.h1
2 files changed, 4 insertions, 7 deletions
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index d368aaf68c8e..ee9b443f760d 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -6276,10 +6276,7 @@ pick_next_task(struct rq *rq, struct rq_flags *rf)
* selection. In this case, do a core-wide selection.
*/
if (rq->core->core_pick_seq == rq->core->core_task_seq &&
- rq->core->core_pick_seq != rq->core_sched_seq &&
rq->core_pick) {
- WRITE_ONCE(rq->core_sched_seq, rq->core->core_pick_seq);
-
next = rq->core_pick;
rq->dl_server = rq->core_dl_server;
rq->core_pick = NULL;
@@ -6311,11 +6308,13 @@ restart:
}
/*
- * core->core_task_seq, core->core_pick_seq, rq->core_sched_seq
+ * core->core_task_seq, core->core_pick_seq
*
* @task_seq guards the task state ({en,de}queues)
* @pick_seq is the @task_seq we did a selection on
- * @sched_seq is the @pick_seq we scheduled
+ *
+ * Once a core-wide selection is committed, a non-NULL core_pick denotes
+ * a pick which still needs to be consumed on this CPU.
*
* However, preemptions can cause multiple picks on the same task set.
* 'Fix' this by also increasing @task_seq for every pick.
@@ -6422,7 +6421,6 @@ restart:
rq->core->core_pick_seq = rq->core->core_task_seq;
next = rq->core_pick;
- rq->core_sched_seq = rq->core->core_pick_seq;
/* Something should have been selected for current CPU */
WARN_ON_ONCE(!next);
diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index 1b64be5771ca..f7d0b64b9d35 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -1371,7 +1371,6 @@ struct rq {
struct task_struct *core_pick;
struct sched_dl_entity *core_dl_server;
unsigned int core_enabled;
- unsigned int core_sched_seq;
struct rb_root core_tree;
/* shared state -- careful with sched_core_cpu_deactivate() */