diff options
| author | Tim Chen <tim.c.chen@linux.intel.com> | 2026-09-21 17:37:24 -0700 |
|---|---|---|
| committer | Ingo Molnar <mingo@kernel.org> | 2026-09-22 10:50:35 +0200 |
| commit | 28f9c0e0a0b94c5d3e1b634db545f6e1f94858c5 (patch) | |
| tree | f88a88f5a27a419d4d7e612016bae9468516ea1f /include/linux | |
| parent | d6013e2465d98d524b030a81c1223882a1bb7e4c (diff) | |
| download | lwn-28f9c0e0a0b94c5d3e1b634db545f6e1f94858c5.tar.gz lwn-28f9c0e0a0b94c5d3e1b634db545f6e1f94858c5.zip | |
sched/cache: Decouple sched_cache_group from mm to fix UAF
Currently the sched cache grouping is by mm and the scheduling statistics
sched_cache_stat lives in the mm structure. This ties the life cycle
of scheduling stats with mm.
In account_mm_sched(), the scheduling stats are accessed by
task->mm->sc_stat. However, a task may be switching mm on one CPU when
another CPU is running account_mm_sched(), and possibly accessing the
old mm that was freed. This problem was found when running tests with
KASAN by Hyunwoo:
https://lore.kernel.org/lkml/apPb-Dr4nPYuHQOK@v4bel/
Instead of serializing the mm access by introducing extra acquisition of
rq lock in the mm free path, extract sched_cache_stat from mm_struct,
rename it as sched_cache_group and manage its life cycle apart from
mm_struct with its own ref counting. This allows us in the next patch
access sched_cache_group directly from task, and add a refcount
on sched_cache_group when a task links to it. This prevents the use
after free issue when accessing stale and released old mm and its
sched cache stat a task switches to a new mm while account_mm_sched()
is done elsewhere.
The other benefit of this restructure is in the future, the grouping of
tasks to a LLC would have the flexibility to be associated with a user
defined grouping, or cgroup, cookie group, numa_group or others instead
of just with a single mm address space.
Rename sched_cache_stat to sched_cache_group and turn it into a refcounted
object allocated from mm_struct. The mm_struct now holds a pointer
(sched_cache_grp) to this object instead of embedding it.
Fixes: df0d98475954 ("sched/cache: Introduce infrastructure for cache-aware load balancing")
Closes: https://lore.kernel.org/lkml/apPb-Dr4nPYuHQOK@v4bel/
Closes: https://lore.kernel.org/all/343a7e07-7fad-4979-9c9b-82ec038c293c@linux.dev/
Reported-by: Hyunwoo Kim <imv4bel@gmail.com>
Reported-by: Zenghui Yu (Huawei) <zenghui.yu@linux.dev>
Co-developed-by: Chen Yu <yu.c.chen@intel.com>
Signed-off-by: Chen Yu <yu.c.chen@intel.com>
Signed-off-by: Tim Chen <tim.c.chen@linux.intel.com>
Signed-off-by: Peter Zijlstra (Intel) <peterz@infradead.org>
Signed-off-by: Ingo Molnar <mingo@kernel.org>
Cc: <stable@kernel.org> #7.2.x
Link: https://patch.msgid.link/91fd1e3266707c865bc9abecfb3e17bc676712df.1790035273.git.tim.c.chen@linux.intel.com
Diffstat (limited to 'include/linux')
| -rw-r--r-- | include/linux/mm_types.h | 15 | ||||
| -rw-r--r-- | include/linux/sched.h | 6 |
2 files changed, 9 insertions, 12 deletions
diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 6d815f6440c9..f3e5a2fadbe5 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -1226,7 +1226,7 @@ struct mm_struct { struct mm_mm_cid mm_cid; /* sched_cache related statistics */ - struct sched_cache_stat sc_stat; + struct sched_cache_group *sched_cache_grp; #ifdef CONFIG_MMU atomic_long_t pgtables_bytes; /* size of all page tables */ #endif @@ -1624,8 +1624,9 @@ static inline unsigned int mm_cid_size(void) #endif /* CONFIG_SCHED_MM_CID */ #ifdef CONFIG_SCHED_CACHE -void mm_init_sched(struct mm_struct *mm, - struct sched_cache_time __percpu *pcpu_sched); +int mm_init_sched(struct mm_struct *mm, + struct sched_cache_time __percpu *pcpu_sched); +void mm_destroy_sched(struct mm_struct *mm); static inline int mm_alloc_sched_noprof(struct mm_struct *mm) { @@ -1635,17 +1636,11 @@ static inline int mm_alloc_sched_noprof(struct mm_struct *mm) if (!pcpu_sched) return -ENOMEM; - mm_init_sched(mm, pcpu_sched); - return 0; + return mm_init_sched(mm, pcpu_sched); } #define mm_alloc_sched(...) alloc_hooks(mm_alloc_sched_noprof(__VA_ARGS__)) -static inline void mm_destroy_sched(struct mm_struct *mm) -{ - free_percpu(mm->sc_stat.pcpu_sched); - mm->sc_stat.pcpu_sched = NULL; -} #else /* !CONFIG_SCHED_CACHE */ static inline int mm_alloc_sched(struct mm_struct *mm) { return 0; } diff --git a/include/linux/sched.h b/include/linux/sched.h index 705970d07614..e14ad43522c8 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -2405,7 +2405,7 @@ struct sched_cache_time { unsigned long epoch; }; -struct sched_cache_stat { +struct sched_cache_group { struct sched_cache_time __percpu *pcpu_sched; raw_spinlock_t lock; unsigned long epoch; @@ -2413,11 +2413,13 @@ struct sched_cache_stat { unsigned long next_scan; unsigned long footprint; int cpu; + refcount_t refcnt; + struct rcu_head rcu; } ____cacheline_aligned_in_smp; #else -struct sched_cache_stat { }; +struct sched_cache_group { }; #endif |
