summaryrefslogtreecommitdiff
path: root/include/linux
diff options
context:
space:
mode:
authorTim Chen <tim.c.chen@linux.intel.com>2026-09-21 17:37:24 -0700
committerIngo Molnar <mingo@kernel.org>2026-09-22 10:50:35 +0200
commit28f9c0e0a0b94c5d3e1b634db545f6e1f94858c5 (patch)
treef88a88f5a27a419d4d7e612016bae9468516ea1f /include/linux
parentd6013e2465d98d524b030a81c1223882a1bb7e4c (diff)
downloadlwn-28f9c0e0a0b94c5d3e1b634db545f6e1f94858c5.tar.gz
lwn-28f9c0e0a0b94c5d3e1b634db545f6e1f94858c5.zip
sched/cache: Decouple sched_cache_group from mm to fix UAF
Currently the sched cache grouping is by mm and the scheduling statistics sched_cache_stat lives in the mm structure. This ties the life cycle of scheduling stats with mm. In account_mm_sched(), the scheduling stats are accessed by task->mm->sc_stat. However, a task may be switching mm on one CPU when another CPU is running account_mm_sched(), and possibly accessing the old mm that was freed. This problem was found when running tests with KASAN by Hyunwoo: https://lore.kernel.org/lkml/apPb-Dr4nPYuHQOK@v4bel/ Instead of serializing the mm access by introducing extra acquisition of rq lock in the mm free path, extract sched_cache_stat from mm_struct, rename it as sched_cache_group and manage its life cycle apart from mm_struct with its own ref counting. This allows us in the next patch access sched_cache_group directly from task, and add a refcount on sched_cache_group when a task links to it. This prevents the use after free issue when accessing stale and released old mm and its sched cache stat a task switches to a new mm while account_mm_sched() is done elsewhere. The other benefit of this restructure is in the future, the grouping of tasks to a LLC would have the flexibility to be associated with a user defined grouping, or cgroup, cookie group, numa_group or others instead of just with a single mm address space. Rename sched_cache_stat to sched_cache_group and turn it into a refcounted object allocated from mm_struct. The mm_struct now holds a pointer (sched_cache_grp) to this object instead of embedding it. Fixes: df0d98475954 ("sched/cache: Introduce infrastructure for cache-aware load balancing") Closes: https://lore.kernel.org/lkml/apPb-Dr4nPYuHQOK@v4bel/ Closes: https://lore.kernel.org/all/343a7e07-7fad-4979-9c9b-82ec038c293c@linux.dev/ Reported-by: Hyunwoo Kim <imv4bel@gmail.com> Reported-by: Zenghui Yu (Huawei) <zenghui.yu@linux.dev> Co-developed-by: Chen Yu <yu.c.chen@intel.com> Signed-off-by: Chen Yu <yu.c.chen@intel.com> Signed-off-by: Tim Chen <tim.c.chen@linux.intel.com> Signed-off-by: Peter Zijlstra (Intel) <peterz@infradead.org> Signed-off-by: Ingo Molnar <mingo@kernel.org> Cc: <stable@kernel.org> #7.2.x Link: https://patch.msgid.link/91fd1e3266707c865bc9abecfb3e17bc676712df.1790035273.git.tim.c.chen@linux.intel.com
Diffstat (limited to 'include/linux')
-rw-r--r--include/linux/mm_types.h15
-rw-r--r--include/linux/sched.h6
2 files changed, 9 insertions, 12 deletions
diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h
index 6d815f6440c9..f3e5a2fadbe5 100644
--- a/include/linux/mm_types.h
+++ b/include/linux/mm_types.h
@@ -1226,7 +1226,7 @@ struct mm_struct {
struct mm_mm_cid mm_cid;
/* sched_cache related statistics */
- struct sched_cache_stat sc_stat;
+ struct sched_cache_group *sched_cache_grp;
#ifdef CONFIG_MMU
atomic_long_t pgtables_bytes; /* size of all page tables */
#endif
@@ -1624,8 +1624,9 @@ static inline unsigned int mm_cid_size(void)
#endif /* CONFIG_SCHED_MM_CID */
#ifdef CONFIG_SCHED_CACHE
-void mm_init_sched(struct mm_struct *mm,
- struct sched_cache_time __percpu *pcpu_sched);
+int mm_init_sched(struct mm_struct *mm,
+ struct sched_cache_time __percpu *pcpu_sched);
+void mm_destroy_sched(struct mm_struct *mm);
static inline int mm_alloc_sched_noprof(struct mm_struct *mm)
{
@@ -1635,17 +1636,11 @@ static inline int mm_alloc_sched_noprof(struct mm_struct *mm)
if (!pcpu_sched)
return -ENOMEM;
- mm_init_sched(mm, pcpu_sched);
- return 0;
+ return mm_init_sched(mm, pcpu_sched);
}
#define mm_alloc_sched(...) alloc_hooks(mm_alloc_sched_noprof(__VA_ARGS__))
-static inline void mm_destroy_sched(struct mm_struct *mm)
-{
- free_percpu(mm->sc_stat.pcpu_sched);
- mm->sc_stat.pcpu_sched = NULL;
-}
#else /* !CONFIG_SCHED_CACHE */
static inline int mm_alloc_sched(struct mm_struct *mm) { return 0; }
diff --git a/include/linux/sched.h b/include/linux/sched.h
index 705970d07614..e14ad43522c8 100644
--- a/include/linux/sched.h
+++ b/include/linux/sched.h
@@ -2405,7 +2405,7 @@ struct sched_cache_time {
unsigned long epoch;
};
-struct sched_cache_stat {
+struct sched_cache_group {
struct sched_cache_time __percpu *pcpu_sched;
raw_spinlock_t lock;
unsigned long epoch;
@@ -2413,11 +2413,13 @@ struct sched_cache_stat {
unsigned long next_scan;
unsigned long footprint;
int cpu;
+ refcount_t refcnt;
+ struct rcu_head rcu;
} ____cacheline_aligned_in_smp;
#else
-struct sched_cache_stat { };
+struct sched_cache_group { };
#endif