From 0d6526f82c3cdefcca47f73f5fc08dc6f335eac6 Mon Sep 17 00:00:00 2001 From: Tim Chen Date: Mon, 21 Sep 2026 17:37:22 -0700 Subject: [PATCH 1/7] sched/cache: Keep nr_pref_llc_running in the runnable domain, to fix LLC mis-scheduling bug alb_break_llc() decides whether to break LLC preference during active load balance. It does so by testing that every runnable fair task on the source rq prefers its LLC: env->src_rq->nr_pref_llc_running == env->src_rq->cfs.h_nr_runnable But the two counters cover different sets. nr_pref_llc_running is updated in account_llc_enqueue()/account_llc_dequeue(), next to cfs_rq->nr_queued, so it follows queued tasks. h_nr_runnable is updated in set_delayed()/ clear_delayed() and drops delay-dequeued tasks. So under DELAY_DEQUEUE, a preferring task that goes to sleep stays counted in nr_pref_llc_running while h_nr_runnable falls. The equality then breaks, alb_break_llc() returns false, and active balance is free to pull a task off its preferred LLC. Active balance only moves runnable tasks, and this is the only LLC check it consults: once the stopper runs, LBF_ACTIVE_LB skips the per-task test in can_migrate_task(). The runnable set is the one we want. Fix it on the counter side. A task should be counted in nr_pref_llc_running exactly while it is both queued on its preferred LLC (pref_llc_queued) and runnable (!sched_delayed). Define that membership once in task_pref_llc_runnable(), and adjust the counter only through pref_llc_running_inc()/pref_llc_running_dec() from the four sites that change either input: account_llc_enqueue(), account_llc_dequeue(), set_delayed() and clear_delayed(). Gating every update on the same predicate keeps the delay, wake and dequeue paths from double-counting or underflowing; see the comments at those sites for the ordering. nr_llc_running and sd->llc_counts are not touched and stay on queued semantics. Fixes: 714059f79ff0 ("sched/cache: Handle moving single tasks to/from their preferred LLC") Closes: https://lore.kernel.org/lkml/20260827135000.735138-1-zhanxusheng@xiaomi.com/ Reported-by: Zhan Xusheng Suggested-by: Chen Yu Signed-off-by: Tim Chen Signed-off-by: Peter Zijlstra (Intel) Signed-off-by: Ingo Molnar Reviewed-by: Kayra Cizmeci Cc: # v7.2.x Link: https://patch.msgid.link/06af61afedac32e6477f57feb4d658f6c411c3af.1790035273.git.tim.c.chen@linux.intel.com --- kernel/sched/fair.c | 63 +++++++++++++++++++++++++++++++++++++++++---- 1 file changed, 58 insertions(+), 5 deletions(-) diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 7455a83a6a99..de3d589fa8eb 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -1546,6 +1546,28 @@ static bool invalid_llc_nr(struct mm_struct *mm, struct task_struct *p, (scale * per_cpu(sd_llc_size, cpu))); } +/* + * A task counts in nr_pref_llc_running while it is queued on its preferred + * LLC (pref_llc_queued) and runnable (!sched_delayed), keeping the counter in + * the runnable domain so alb_break_llc() can compare it with h_nr_runnable. + */ +static bool task_pref_llc_runnable(struct task_struct *p) +{ + return p->pref_llc_queued && !p->se.sched_delayed; +} + +static void pref_llc_running_inc(struct rq *rq, struct task_struct *p) +{ + if (task_pref_llc_runnable(p)) + rq->nr_pref_llc_running++; +} + +static void pref_llc_running_dec(struct rq *rq, struct task_struct *p) +{ + if (task_pref_llc_runnable(p)) + rq->nr_pref_llc_running--; +} + static void account_llc_enqueue(struct rq *rq, struct task_struct *p) { int pref_llc, pref_llc_queued; @@ -1557,7 +1579,6 @@ static void account_llc_enqueue(struct rq *rq, struct task_struct *p) pref_llc_queued = (pref_llc == task_llc(p)); rq->nr_llc_running++; - rq->nr_pref_llc_running += pref_llc_queued; /* * Record whether p is enqueued on its preferred @@ -1575,6 +1596,9 @@ static void account_llc_enqueue(struct rq *rq, struct task_struct *p) */ p->pref_llc_queued = pref_llc_queued; + /* Skipped while delayed; clear_delayed() adds it back on wake. */ + pref_llc_running_inc(rq, p); + sd = rcu_dereference_all(rq->sd); if (sd && (unsigned int)pref_llc < sd->llc_max) sd->llc_counts[pref_llc]++; @@ -1591,7 +1615,12 @@ static void account_llc_dequeue(struct rq *rq, struct task_struct *p) rq->nr_llc_running--; if (p->pref_llc_queued) { - rq->nr_pref_llc_running--; + /* + * Skipped if still delayed (set_delayed() already removed it); + * clearing pref_llc_queued below also stops clear_delayed() + * from re-adding it. + */ + pref_llc_running_dec(rq, p); /* * Update the status in case * other logic might query @@ -1995,6 +2024,7 @@ void init_sched_mm(struct task_struct *p) * polluting account_llc_enqueue(). */ p->preferred_llc = -1; + p->pref_llc_queued = 0; } #else /* CONFIG_SCHED_CACHE */ @@ -2016,6 +2046,10 @@ static void account_llc_enqueue(struct rq *rq, struct task_struct *p) {} static void account_llc_dequeue(struct rq *rq, struct task_struct *p) {} +static void pref_llc_running_inc(struct rq *rq, struct task_struct *p) {} + +static void pref_llc_running_dec(struct rq *rq, struct task_struct *p) {} + #endif /* CONFIG_SCHED_CACHE */ /* @@ -6390,15 +6424,27 @@ static __always_inline void return_cfs_rq_runtime(struct cfs_rq *cfs_rq); static void set_delayed(struct sched_entity *se) { - se->sched_delayed = 1; - /* * Delayed se of cfs_rq have no tasks queued on them. * Do not adjust h_nr_runnable since __dequeue_task() * will account it for blocked tasks. + * + * This check can be removed because when flat pick + * patches get merged as only task can get delayed, + * same for clear_delayed(). */ - if (!entity_is_task(se)) + if (!entity_is_task(se)) { + se->sched_delayed = 1; return; + } + + /* + * Drop a task leaving the runnable set. + * Needs to be called before sched_delayed is set. + * clear_delayed() mirrors this after clearing the flag. + */ + pref_llc_running_dec(rq_of(cfs_rq_of(se)), task_of(se)); + se->sched_delayed = 1; for_each_sched_entity(se) { struct cfs_rq *cfs_rq = cfs_rq_of(se); @@ -6420,6 +6466,13 @@ static void clear_delayed(struct sched_entity *se) if (!entity_is_task(se)) return; + /* + * Re-add on wake, after sched_delayed is cleared. On a final delayed + * dequeue account_llc_dequeue() already cleared pref_llc_queued, so + * this does nothing. + */ + pref_llc_running_inc(rq_of(cfs_rq_of(se)), task_of(se)); + for_each_sched_entity(se) { struct cfs_rq *cfs_rq = cfs_rq_of(se); From d6013e2465d98d524b030a81c1223882a1bb7e4c Mon Sep 17 00:00:00 2001 From: Lu Wang Date: Mon, 21 Sep 2026 17:37:23 -0700 Subject: [PATCH 2/7] sched/cache: Honor migrate_llc_task semantics in active load balance, to fix LLC mis-scheduling bug Cache aware scheduling introduced the migrate_llc_task migration type to direct tasks toward their preferred LLC, but its semantics can be lost when passive load balance falls back to active load balance (ALB). This may allow ALB to select a candidate whose preferred LLC does not match the destination, moving it away from its preferred LLC. Example scenario: src_rq has two runnable tasks, p1 and p2. p1 prefers dst_rq (dst_llc), while p2 prefers src_rq (src_llc). In this case, migrate_llc_task is set because src_rq has at least one task, p1, that wants to migrate to dst_rq. In ALB, can_migrate_task() finds p2 and returns true for it, thus moving p2 out of its preferred LLC. Solution: The CPU stopper in ALB constructs a fresh lb_env that does not inherit migration_type from the passive load-balance pass. Two approaches are possible: (a) Add a new member to struct rq so ALB can inherit migrate_llc_task from the passive LB that triggered it. (b) Define a new flag LBF_ACTIVE_LB_LLC and select the stopper callback at kick time to preserve the migration semantics across the asynchronous boundary. We choose (b) because it avoids passing migration_type through the stopper, which would affect the meaning of migration_type for delayed-dequeue tasks. Fixes: e4c9a4cb244a ("sched/cache: Add migrate_llc_task migration type for cache-aware balancing") Suggested-by: Chen Yu Signed-off-by: Lu Wang Signed-off-by: Tim Chen Signed-off-by: Peter Zijlstra (Intel) Signed-off-by: Ingo Molnar Reviewed-by: Tim Chen Reviewed-by: Chen Yu Cc: # v7.2.x Link: https://patch.msgid.link/cb39f64a17fc2b76097264aaec74a2d6dfff4315.1790035273.git.tim.c.chen@linux.intel.com --- kernel/sched/fair.c | 57 ++++++++++++++++++++++++++++++++++++++++----- 1 file changed, 51 insertions(+), 6 deletions(-) diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index de3d589fa8eb..514bd54ccd56 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -10448,6 +10448,7 @@ enum migration_type { #define LBF_SOME_PINNED 0x08 #define LBF_ACTIVE_LB 0x10 #define LBF_LLC_PINNED 0x20 +#define LBF_ACTIVE_LB_LLC 0x40 struct lb_env { struct sched_domain *sd; @@ -10866,6 +10867,21 @@ alb_break_llc(struct lb_env *env) return false; } +/* + * Returns true if p's preferred LLC does not match the destination CPU + * under migrate_llc_task semantics. Passive LB passes migrate_llc_task + * in env->migration_type, while active LB carries LBF_ACTIVE_LB_LLC in + * env->flags to avoid overwriting env->migration_type. + */ +static inline bool +migrate_llc_task_wrong_dst(struct task_struct *p, struct lb_env *env) +{ + return sched_cache_enabled() && + (env->migration_type == migrate_llc_task || + env->flags & LBF_ACTIVE_LB_LLC) && + READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu); +} + /* * Check if migrating task p from env->src_cpu to * env->dst_cpu breaks LLC localiy. @@ -10894,8 +10910,7 @@ static bool migrate_degrades_llc(struct task_struct *p, struct lb_env *env) * run on env->dst_cpu, skip the tasks do not prefer * env->dst_cpu, and find the one that prefers. */ - if (env->migration_type == migrate_llc_task && - READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu)) + if (migrate_llc_task_wrong_dst(p, env)) return true; if (can_migrate_llc_task(env, p) != mig_forbid) @@ -10917,6 +10932,12 @@ alb_break_llc(struct lb_env *env) return false; } +static inline bool +migrate_llc_task_wrong_dst(struct task_struct *p, struct lb_env *env) +{ + return false; +} + static inline bool migrate_degrades_llc(struct task_struct *p, struct lb_env *env) { @@ -11016,7 +11037,7 @@ int can_migrate_task(struct task_struct *p, struct lb_env *env) * 4) too many balance attempts have failed. */ if (env->flags & LBF_ACTIVE_LB) - return 1; + return !migrate_llc_task_wrong_dst(p, env); degrades = migrate_degrades_locality(p, env); if (!degrades) { @@ -13415,6 +13436,20 @@ static int need_active_balance(struct lb_env *env) } static int active_load_balance_cpu_stop(void *data); +static int active_load_balance_llc_cpu_stop(void *data); + +/* + * migration_type is checked elsewhere to decide migration policy, so + * it shouldn't be repurposed just to flag an LLC-directed active + * balance across the stopper. Pick the callback here instead. + */ +static inline cpu_stop_fn_t alb_stop_fn(struct lb_env *env) +{ + if (env->migration_type == migrate_llc_task) + return active_load_balance_llc_cpu_stop; + + return active_load_balance_cpu_stop; +} static int should_we_balance(struct lb_env *env) { @@ -13760,7 +13795,7 @@ static int sched_balance_rq(int this_cpu, struct rq *this_rq, } if (active_balance) { stop_one_cpu_nowait(cpu_of(busiest), - active_load_balance_cpu_stop, busiest, + alb_stop_fn(&env), busiest, &busiest->active_balance_work); } preempt_enable(); @@ -13865,7 +13900,7 @@ update_next_balance(struct sched_domain *sd, unsigned long *next_balance) * least 1 task to be running on each physical CPU where possible, and * avoids physical / logical imbalances. */ -static int active_load_balance_cpu_stop(void *data) +static int __active_load_balance_cpu_stop(void *data, unsigned int lb_flags) { struct rq *busiest_rq = data; int busiest_cpu = cpu_of(busiest_rq); @@ -13915,7 +13950,7 @@ static int active_load_balance_cpu_stop(void *data) .src_cpu = busiest_rq->cpu, .src_rq = busiest_rq, .idle = CPU_IDLE, - .flags = LBF_ACTIVE_LB, + .flags = LBF_ACTIVE_LB | lb_flags, }; schedstat_inc(sd->alb_count); @@ -13943,6 +13978,16 @@ static int active_load_balance_cpu_stop(void *data) return 0; } +static int active_load_balance_cpu_stop(void *data) +{ + return __active_load_balance_cpu_stop(data, 0); +} + +static int active_load_balance_llc_cpu_stop(void *data) +{ + return __active_load_balance_cpu_stop(data, LBF_ACTIVE_LB_LLC); +} + /* * Scale the max sched_balance_rq interval with the number of CPUs in the system. * This trades load-balance latency on larger machines for less cross talk. From 28f9c0e0a0b94c5d3e1b634db545f6e1f94858c5 Mon Sep 17 00:00:00 2001 From: Tim Chen Date: Mon, 21 Sep 2026 17:37:24 -0700 Subject: [PATCH 3/7] sched/cache: Decouple sched_cache_group from mm to fix UAF Currently the sched cache grouping is by mm and the scheduling statistics sched_cache_stat lives in the mm structure. This ties the life cycle of scheduling stats with mm. In account_mm_sched(), the scheduling stats are accessed by task->mm->sc_stat. However, a task may be switching mm on one CPU when another CPU is running account_mm_sched(), and possibly accessing the old mm that was freed. This problem was found when running tests with KASAN by Hyunwoo: https://lore.kernel.org/lkml/apPb-Dr4nPYuHQOK@v4bel/ Instead of serializing the mm access by introducing extra acquisition of rq lock in the mm free path, extract sched_cache_stat from mm_struct, rename it as sched_cache_group and manage its life cycle apart from mm_struct with its own ref counting. This allows us in the next patch access sched_cache_group directly from task, and add a refcount on sched_cache_group when a task links to it. This prevents the use after free issue when accessing stale and released old mm and its sched cache stat a task switches to a new mm while account_mm_sched() is done elsewhere. The other benefit of this restructure is in the future, the grouping of tasks to a LLC would have the flexibility to be associated with a user defined grouping, or cgroup, cookie group, numa_group or others instead of just with a single mm address space. Rename sched_cache_stat to sched_cache_group and turn it into a refcounted object allocated from mm_struct. The mm_struct now holds a pointer (sched_cache_grp) to this object instead of embedding it. Fixes: df0d98475954 ("sched/cache: Introduce infrastructure for cache-aware load balancing") Closes: https://lore.kernel.org/lkml/apPb-Dr4nPYuHQOK@v4bel/ Closes: https://lore.kernel.org/all/343a7e07-7fad-4979-9c9b-82ec038c293c@linux.dev/ Reported-by: Hyunwoo Kim Reported-by: Zenghui Yu (Huawei) Co-developed-by: Chen Yu Signed-off-by: Chen Yu Signed-off-by: Tim Chen Signed-off-by: Peter Zijlstra (Intel) Signed-off-by: Ingo Molnar Cc: #7.2.x Link: https://patch.msgid.link/91fd1e3266707c865bc9abecfb3e17bc676712df.1790035273.git.tim.c.chen@linux.intel.com --- include/linux/mm_types.h | 15 ++-- include/linux/sched.h | 6 +- kernel/exit.c | 11 ++- kernel/sched/fair.c | 173 ++++++++++++++++++++++++++++----------- 4 files changed, 144 insertions(+), 61 deletions(-) diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 6d815f6440c9..f3e5a2fadbe5 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -1226,7 +1226,7 @@ struct mm_struct { struct mm_mm_cid mm_cid; /* sched_cache related statistics */ - struct sched_cache_stat sc_stat; + struct sched_cache_group *sched_cache_grp; #ifdef CONFIG_MMU atomic_long_t pgtables_bytes; /* size of all page tables */ #endif @@ -1624,8 +1624,9 @@ static inline unsigned int mm_cid_size(void) #endif /* CONFIG_SCHED_MM_CID */ #ifdef CONFIG_SCHED_CACHE -void mm_init_sched(struct mm_struct *mm, - struct sched_cache_time __percpu *pcpu_sched); +int mm_init_sched(struct mm_struct *mm, + struct sched_cache_time __percpu *pcpu_sched); +void mm_destroy_sched(struct mm_struct *mm); static inline int mm_alloc_sched_noprof(struct mm_struct *mm) { @@ -1635,17 +1636,11 @@ static inline int mm_alloc_sched_noprof(struct mm_struct *mm) if (!pcpu_sched) return -ENOMEM; - mm_init_sched(mm, pcpu_sched); - return 0; + return mm_init_sched(mm, pcpu_sched); } #define mm_alloc_sched(...) alloc_hooks(mm_alloc_sched_noprof(__VA_ARGS__)) -static inline void mm_destroy_sched(struct mm_struct *mm) -{ - free_percpu(mm->sc_stat.pcpu_sched); - mm->sc_stat.pcpu_sched = NULL; -} #else /* !CONFIG_SCHED_CACHE */ static inline int mm_alloc_sched(struct mm_struct *mm) { return 0; } diff --git a/include/linux/sched.h b/include/linux/sched.h index 705970d07614..e14ad43522c8 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -2405,7 +2405,7 @@ struct sched_cache_time { unsigned long epoch; }; -struct sched_cache_stat { +struct sched_cache_group { struct sched_cache_time __percpu *pcpu_sched; raw_spinlock_t lock; unsigned long epoch; @@ -2413,11 +2413,13 @@ struct sched_cache_stat { unsigned long next_scan; unsigned long footprint; int cpu; + refcount_t refcnt; + struct rcu_head rcu; } ____cacheline_aligned_in_smp; #else -struct sched_cache_stat { }; +struct sched_cache_group { }; #endif diff --git a/kernel/exit.c b/kernel/exit.c index 424c44a42a4d..024350e9b48c 100644 --- a/kernel/exit.c +++ b/kernel/exit.c @@ -558,18 +558,23 @@ void mm_update_next_owner(struct mm_struct *mm) */ static void exit_mm_sched_cache(struct mm_struct *mm) { + struct sched_cache_group *grp; unsigned long fp, sub; if (!current->total_numa_faults) return; /* * No lock protection due to performance considerations. - * Make sure mm->sc_stat.footprint does not become + * Make sure the group footprint does not become * negative. */ - fp = READ_ONCE(mm->sc_stat.footprint); + grp = READ_ONCE(mm->sched_cache_grp); + if (!grp) + return; + + fp = READ_ONCE(grp->footprint); sub = min(fp, current->total_numa_faults); - WRITE_ONCE(mm->sc_stat.footprint, fp - sub); + WRITE_ONCE(grp->footprint, fp - sub); } #else static inline void exit_mm_sched_cache(struct mm_struct *mm) diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 514bd54ccd56..f0a9586332fd 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -1492,12 +1492,17 @@ static bool exceed_llc_capacity(struct mm_struct *mm, int cpu) return true; if (static_branch_likely(&sched_numa_balancing)) { + struct sched_cache_group *grp = READ_ONCE(mm->sched_cache_grp); + + if (!grp) + return true; + /* * TBD: RDT exclusive LLC ways reserved should be * excluded. */ llc = sd->llc_bytes; - footprint = READ_ONCE(mm->sc_stat.footprint); + footprint = READ_ONCE(grp->footprint); /* * Scale the LLC size by 256*llc_aggr_tolerance @@ -1529,6 +1534,7 @@ static bool exceed_llc_capacity(struct mm_struct *mm, int cpu) static bool invalid_llc_nr(struct mm_struct *mm, struct task_struct *p, int cpu) { + struct sched_cache_group *grp; int scale; if (get_nr_threads(p) <= 1) @@ -1542,7 +1548,11 @@ static bool invalid_llc_nr(struct mm_struct *mm, struct task_struct *p, if (scale == INT_MAX) return false; - return !fits_capacity((mm->sc_stat.nr_running_avg * cpu_smt_num_threads), + grp = READ_ONCE(mm->sched_cache_grp); + if (!grp) + return true; + + return !fits_capacity((READ_ONCE(grp->nr_running_avg) * cpu_smt_num_threads), (scale * per_cpu(sd_llc_size, cpu))); } @@ -1648,12 +1658,20 @@ static void account_llc_dequeue(struct rq *rq, struct task_struct *p) } } -void mm_init_sched(struct mm_struct *mm, - struct sched_cache_time __percpu *_pcpu_sched) +int mm_init_sched(struct mm_struct *mm, + struct sched_cache_time __percpu *_pcpu_sched) { + struct sched_cache_group *grp; unsigned long epoch = 0; int i; + grp = kzalloc_obj(*grp); + if (!grp) { + free_percpu(_pcpu_sched); + mm->sched_cache_grp = NULL; + return -ENOMEM; + } + for_each_possible_cpu(i) { struct sched_cache_time *pcpu_sched = per_cpu_ptr(_pcpu_sched, i); struct rq *rq = cpu_rq(i); @@ -1664,18 +1682,51 @@ void mm_init_sched(struct mm_struct *mm, epoch = rq->cpu_epoch; } - raw_spin_lock_init(&mm->sc_stat.lock); - mm->sc_stat.epoch = epoch; - mm->sc_stat.cpu = -1; - mm->sc_stat.next_scan = jiffies; - mm->sc_stat.nr_running_avg = 0; - mm->sc_stat.footprint = 0; + raw_spin_lock_init(&grp->lock); + grp->epoch = epoch; + grp->cpu = -1; + grp->next_scan = jiffies; + grp->nr_running_avg = 0; + grp->footprint = 0; + refcount_set(&grp->refcnt, 1); /* - * The update to mm->sc_stat should not be reordered - * before initialization to mm's other fields, in case + * The update to grp->pcpu_sched should not be reordered + * before initialization to grp's other fields, in case * the readers may get invalid mm_sched_epoch, etc. */ - smp_store_release(&mm->sc_stat.pcpu_sched, _pcpu_sched); + smp_store_release(&grp->pcpu_sched, _pcpu_sched); + /* + * Publish the group last. Not every reader qualifies it by + * grp->pcpu_sched - can_migrate_llc_task() only checks that the + * pointer is non-NULL before reading grp->footprint and + * grp->nr_running_avg - so a reachable group must already be + * fully initialized. + */ + smp_store_release(&mm->sched_cache_grp, grp); + return 0; +} + +static void sched_cache_group_free_rcu(struct rcu_head *rcu) +{ + struct sched_cache_group *grp = + container_of(rcu, struct sched_cache_group, rcu); + + free_percpu(grp->pcpu_sched); + kfree(grp); +} + +static void sched_cache_group_put(struct sched_cache_group *grp) +{ + if (!grp || !refcount_dec_and_test(&grp->refcnt)) + return; + + call_rcu(&grp->rcu, sched_cache_group_free_rcu); +} + +void mm_destroy_sched(struct mm_struct *mm) +{ + sched_cache_group_put(mm->sched_cache_grp); + mm->sched_cache_grp = NULL; } /* because why would C be fully specified */ @@ -1729,11 +1780,16 @@ static unsigned long fraction_mm_sched(struct rq *rq, static int get_pref_llc(struct task_struct *p, struct mm_struct *mm) { int mm_sched_llc = -1, mm_sched_cpu; + struct sched_cache_group *grp; if (!mm) return -1; - mm_sched_cpu = READ_ONCE(mm->sc_stat.cpu); + grp = READ_ONCE(mm->sched_cache_grp); + if (!grp) + return -1; + + mm_sched_cpu = READ_ONCE(grp->cpu); if (mm_sched_cpu != -1) { mm_sched_llc = llc_id(mm_sched_cpu); @@ -1764,6 +1820,7 @@ static inline void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec) { struct sched_cache_time *pcpu_sched; + struct sched_cache_group *grp; struct mm_struct *mm = p->mm; int mm_sched_llc = -1; unsigned long epoch; @@ -1777,10 +1834,14 @@ void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec) * init_task, kthreads and user thread created * by user_mode_thread() don't have mm. */ - if (!mm || !mm->sc_stat.pcpu_sched) + if (!mm) return; - pcpu_sched = per_cpu_ptr(mm->sc_stat.pcpu_sched, cpu_of(rq)); + grp = READ_ONCE(mm->sched_cache_grp); + if (!grp || !grp->pcpu_sched) + return; + + pcpu_sched = per_cpu_ptr(grp->pcpu_sched, cpu_of(rq)); scoped_guard (raw_spinlock, &rq->cpu_epoch_lock) { __update_mm_sched(rq, pcpu_sched); @@ -1793,11 +1854,11 @@ void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec) * If this process hasn't hit task_cache_work() for a while invalidate * its preferred state. */ - if ((long)(epoch - READ_ONCE(mm->sc_stat.epoch)) > llc_epoch_affinity_timeout || + if ((long)(epoch - READ_ONCE(grp->epoch)) > llc_epoch_affinity_timeout || invalid_llc_nr(mm, p, cpu_of(rq)) || exceed_llc_capacity(mm, cpu_of(rq))) { - if (READ_ONCE(mm->sc_stat.cpu) != -1) - WRITE_ONCE(mm->sc_stat.cpu, -1); + if (READ_ONCE(grp->cpu) != -1) + WRITE_ONCE(grp->cpu, -1); } mm_sched_llc = get_pref_llc(p, mm); @@ -1814,30 +1875,35 @@ void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec) static void task_tick_cache(struct rq *rq, struct task_struct *p) { struct callback_head *work = &p->cache_work; + struct sched_cache_group *grp; struct mm_struct *mm = p->mm; unsigned long epoch; if (!sched_cache_enabled()) return; - if (!mm || p->flags & PF_KTHREAD || - !mm->sc_stat.pcpu_sched) + if (!mm || p->flags & PF_KTHREAD) + return; + + grp = READ_ONCE(mm->sched_cache_grp); + if (!grp || !grp->pcpu_sched) return; epoch = rq->cpu_epoch; /* avoid moving backwards */ - if (time_after_eq(mm->sc_stat.epoch, epoch)) + if (time_after_eq(grp->epoch, epoch)) return; - guard(raw_spinlock)(&mm->sc_stat.lock); + guard(raw_spinlock)(&grp->lock); if (work->next == work) { task_work_add(p, work, TWA_RESUME); - WRITE_ONCE(mm->sc_stat.epoch, epoch); + WRITE_ONCE(grp->epoch, epoch); } } -static void get_scan_cpumasks(cpumask_var_t cpus, struct task_struct *p) +static void get_scan_cpumasks(cpumask_var_t cpus, struct task_struct *p, + struct sched_cache_group *grp) { #ifdef CONFIG_NUMA_BALANCING int cpu, curr_cpu, nid, pref_nid; @@ -1845,7 +1911,7 @@ static void get_scan_cpumasks(cpumask_var_t cpus, struct task_struct *p) if (!static_branch_likely(&sched_numa_balancing)) goto out; - cpu = READ_ONCE(p->mm->sc_stat.cpu); + cpu = READ_ONCE(grp->cpu); if (cpu != -1) nid = cpu_to_node(cpu); curr_cpu = task_cpu(p); @@ -1906,6 +1972,7 @@ static void task_cache_work(struct callback_head *work) unsigned long next_scan, now = jiffies; struct task_struct *p = current, *cur; unsigned long curr_m_a_occ = 0; + struct sched_cache_group *grp; struct mm_struct *mm = p->mm; unsigned long m_a_occ = 0; cpumask_var_t cpus; @@ -1917,12 +1984,16 @@ static void task_cache_work(struct callback_head *work) if (p->flags & PF_EXITING) return; - next_scan = READ_ONCE(mm->sc_stat.next_scan); + grp = READ_ONCE(mm->sched_cache_grp); + if (!grp) + return; + + next_scan = READ_ONCE(grp->next_scan); if (time_before(now, next_scan)) return; /* only 1 thread is allowed to scan */ - if (!try_cmpxchg(&mm->sc_stat.next_scan, &next_scan, + if (!try_cmpxchg(&grp->next_scan, &next_scan, now + max_t(unsigned long, READ_ONCE(llc_epoch_period), 1))) return; @@ -1930,8 +2001,8 @@ static void task_cache_work(struct callback_head *work) curr_cpu = task_cpu(p); if (invalid_llc_nr(mm, p, curr_cpu) || exceed_llc_capacity(mm, curr_cpu)) { - if (READ_ONCE(mm->sc_stat.cpu) != -1) - WRITE_ONCE(mm->sc_stat.cpu, -1); + if (READ_ONCE(grp->cpu) != -1) + WRITE_ONCE(grp->cpu, -1); return; } @@ -1942,7 +2013,7 @@ static void task_cache_work(struct callback_head *work) scoped_guard (cpus_read_lock) { guard(rcu)(); - get_scan_cpumasks(cpus, p); + get_scan_cpumasks(cpus, p, grp); for_each_cpu(cpu, cpus) { /* XXX sched_cluster_active */ @@ -1955,7 +2026,7 @@ static void task_cache_work(struct callback_head *work) for_each_cpu(i, sched_domain_span(sd)) { occ = fraction_mm_sched(cpu_rq(i), - per_cpu_ptr(mm->sc_stat.pcpu_sched, i)); + per_cpu_ptr(grp->pcpu_sched, i)); a_occ += occ; if (occ > m_occ) { m_occ = occ; @@ -1988,7 +2059,7 @@ static void task_cache_work(struct callback_head *work) m_a_cpu = m_cpu; } - if (llc_id(cpu) == llc_id(READ_ONCE(mm->sc_stat.cpu))) + if (llc_id(cpu) == llc_id(READ_ONCE(grp->cpu))) curr_m_a_occ = a_occ; cpumask_andnot(cpus, cpus, sched_domain_span(sd)); @@ -1997,7 +2068,7 @@ static void task_cache_work(struct callback_head *work) if (m_a_occ > (2 * curr_m_a_occ)) { /* - * Avoid switching sc_stat.cpu too fast. + * Avoid switching sched_cache_grp->cpu too fast. * The reason to choose 2X is because: * 1. It is better to keep the preferred LLC stable, * rather than changing it frequently and cause migrations @@ -2006,10 +2077,10 @@ static void task_cache_work(struct callback_head *work) * 3. 2X is chosen based on test results, as it delivers * the optimal performance gain so far. */ - WRITE_ONCE(mm->sc_stat.cpu, m_a_cpu); + WRITE_ONCE(grp->cpu, m_a_cpu); } - update_avg_scale(&mm->sc_stat.nr_running_avg, nr_running); + update_avg_scale(&grp->nr_running_avg, nr_running); free_cpumask_var(cpus); } @@ -3726,6 +3797,7 @@ static int preferred_group_nid(struct task_struct *p, int nid) static void task_numa_placement(struct task_struct *p) __context_unsafe(/* conditional locking */) { + struct sched_cache_group __maybe_unused *grp; int seq, nid, max_nid = NUMA_NO_NODE; unsigned long max_faults = 0; unsigned long fault_types[2] = { 0, 0 }; @@ -3818,19 +3890,23 @@ static void task_numa_placement(struct task_struct *p) * heuristic and occasional lost updates are tolerable. * * If a task exits, its corresponding footprint must - * be subtracted from the mm->sc_stat.footprint, otherwise - * the mm->sc_stat.footprint will not converge: - * the exiting thread's footprint remains unchanged/undecayed - * in mm->sc_stat.footprint. See exit_mm(). + * be subtracted from the mm->sched_cache_grp->footprint, + * otherwise the mm->sched_cache_grp->footprint will not + * converge: the exiting thread's footprint remains + * unchanged/undecayed in mm->sched_cache_grp->footprint. + * See exit_mm(). * * Lost updates and unsynchronized subtraction * in exit_mm() can cause footprint + diff to * go negative. Clamp to zero to prevent the * unsigned footprint from wrapping. */ - new_fp = (long)READ_ONCE(p->mm->sc_stat.footprint) + diff; - WRITE_ONCE(p->mm->sc_stat.footprint, - max(new_fp, 0L)); + grp = READ_ONCE(p->mm->sched_cache_grp); + if (!grp) + continue; + + new_fp = (long)READ_ONCE(grp->footprint) + diff; + WRITE_ONCE(grp->footprint, max(new_fp, 0L)); #endif } @@ -10778,6 +10854,7 @@ static inline bool task_misfits_asym_cpu(struct lb_env *env, struct task_struct static enum llc_mig can_migrate_llc_task(struct lb_env *env, struct task_struct *p) { + struct sched_cache_group *grp; struct mm_struct *mm; bool to_pref; int cpu, src_cpu, dst_cpu; @@ -10791,15 +10868,19 @@ static enum llc_mig can_migrate_llc_task(struct lb_env *env, if (!mm) return mig_unrestricted; - cpu = READ_ONCE(mm->sc_stat.cpu); + grp = READ_ONCE(mm->sched_cache_grp); + if (!grp) + return mig_unrestricted; + + cpu = READ_ONCE(grp->cpu); if (cpu < 0 || cpus_share_cache(src_cpu, dst_cpu)) return mig_unrestricted; /* skip cache aware load balance for too many threads */ if (invalid_llc_nr(mm, p, dst_cpu) || exceed_llc_capacity(mm, dst_cpu)) { - if (READ_ONCE(mm->sc_stat.cpu) != -1) - WRITE_ONCE(mm->sc_stat.cpu, -1); + if (READ_ONCE(grp->cpu) != -1) + WRITE_ONCE(grp->cpu, -1); return mig_unrestricted; } From b636fef85bda7d1bab9c0a45067ab1508d79d946 Mon Sep 17 00:00:00 2001 From: Tim Chen Date: Mon, 21 Sep 2026 17:37:25 -0700 Subject: [PATCH 4/7] sched/cache: Introduce task_struct->sched_cache_grp to fix UAF Add a sched_cache_grp pointer to task_struct so that scheduler code can access the cache group directly via the task, without going through mm->sched_cache_grp. This decouples the scheduler's hot-path accesses from the mm_struct. Each task holds its own refcount on the sched_cache_group, separate from the reference held by its mm_struct. The reference is acquired in copy_mm() (fork) and exec_mmap() (exec), and released in exit_mm(). This fixes the use-after-free when account_mm_sched() reaches the group through a task whose mm is being switched, as reported by Hyunwoo: https://lore.kernel.org/lkml/apPb-Dr4nPYuHQOK@v4bel/ Convert all scheduler code in fair.c and exit.c to use p->sched_cache_grp instead of p->mm->sched_cache_grp. Keep the fork/exec/exit reference management out of the generic mm paths: add sched_cache_fork(), sched_cache_fork_cleanup(), sched_cache_exec_mmap() and sched_cache_exit_mm() in kernel/sched/cache_sched.c (with empty stubs for !CONFIG_SCHED_CACHE), so fs/exec.c, kernel/fork.c and kernel/exit.c each call one helper instead of open-coding the refcounting under #ifdef. Also add sched_cache_group_get() and task_cache_group_get(). Fixes: df0d98475954 ("sched/cache: Introduce infrastructure for cache-aware load balancing") Closes: https://lore.kernel.org/lkml/apPb-Dr4nPYuHQOK@v4bel/ Closes: https://lore.kernel.org/all/343a7e07-7fad-4979-9c9b-82ec038c293c@linux.dev/ Reported-by: Hyunwoo Kim Reported-by: Zenghui Yu (Huawei) Co-developed-by: Chen Yu Signed-off-by: Chen Yu Signed-off-by: Tim Chen Signed-off-by: Peter Zijlstra (Intel) Signed-off-by: Ingo Molnar Cc: #7.2.x Link: https://patch.msgid.link/ae7081dc54736bf115215f9867abb2711a7403fb.1790035273.git.tim.c.chen@linux.intel.com --- fs/exec.c | 1 + include/linux/sched.h | 14 +++ kernel/exit.c | 33 +------ kernel/fork.c | 2 + kernel/sched/fair.c | 196 +++++++++++++++++++++++++++++------------- 5 files changed, 154 insertions(+), 92 deletions(-) diff --git a/fs/exec.c b/fs/exec.c index 819643408e6d..a5269b5e00df 100644 --- a/fs/exec.c +++ b/fs/exec.c @@ -882,6 +882,7 @@ static int exec_mmap(struct linux_binprm *bprm) active_mm = tsk->active_mm; tsk->active_mm = mm; tsk->mm = mm; + sched_cache_exec_mmap(tsk, mm); mm_init_cid(mm, tsk); exec_state = task_exec_state_replace(tsk, exec_state); /* diff --git a/include/linux/sched.h b/include/linux/sched.h index e14ad43522c8..d35ae49a991f 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -1433,6 +1433,7 @@ struct task_struct { #ifdef CONFIG_SCHED_CACHE struct callback_head cache_work; + struct sched_cache_group __rcu *sched_cache_grp; int preferred_llc; /* 1: task was enqueued to its preferred LLC, 0 otherwise */ int pref_llc_queued; @@ -2417,10 +2418,23 @@ struct sched_cache_group { struct rcu_head rcu; } ____cacheline_aligned_in_smp; +struct sched_cache_group *sched_cache_group_get(struct sched_cache_group *grp); +struct sched_cache_group *task_cache_group_get(struct task_struct *p); + +void sched_cache_fork(struct task_struct *p); +void sched_cache_fork_cleanup(struct task_struct *p); +void sched_cache_exec_mmap(struct task_struct *p, struct mm_struct *mm); +void sched_cache_exit_mm(struct task_struct *p); + #else struct sched_cache_group { }; +static inline void sched_cache_fork(struct task_struct *p) { } +static inline void sched_cache_fork_cleanup(struct task_struct *p) { } +static inline void sched_cache_exec_mmap(struct task_struct *p, struct mm_struct *mm) { } +static inline void sched_cache_exit_mm(struct task_struct *p) { } + #endif #ifndef MODULE diff --git a/kernel/exit.c b/kernel/exit.c index 024350e9b48c..282328d2b4cf 100644 --- a/kernel/exit.c +++ b/kernel/exit.c @@ -551,37 +551,6 @@ void mm_update_next_owner(struct mm_struct *mm) } #endif /* CONFIG_MEMCG */ -#if defined(CONFIG_SCHED_CACHE) && defined(CONFIG_NUMA_BALANCING) -/* - * Subtract the memory footprint of the current task from - * mm. - */ -static void exit_mm_sched_cache(struct mm_struct *mm) -{ - struct sched_cache_group *grp; - unsigned long fp, sub; - - if (!current->total_numa_faults) - return; - /* - * No lock protection due to performance considerations. - * Make sure the group footprint does not become - * negative. - */ - grp = READ_ONCE(mm->sched_cache_grp); - if (!grp) - return; - - fp = READ_ONCE(grp->footprint); - sub = min(fp, current->total_numa_faults); - WRITE_ONCE(grp->footprint, fp - sub); -} -#else -static inline void exit_mm_sched_cache(struct mm_struct *mm) -{ -} -#endif /* CONFIG_SCHED_CACHE CONFIG_NUMA_BALANCING */ - /* * Turn us into a lazy TLB process if we * aren't already.. @@ -594,7 +563,7 @@ static void exit_mm(void) if (!mm) return; - exit_mm_sched_cache(mm); + sched_cache_exit_mm(current); mmap_read_lock(mm); mmgrab_lazy_tlb(mm); diff --git a/kernel/fork.c b/kernel/fork.c index 5ef413368912..10f2d05d816a 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -1599,6 +1599,7 @@ static int copy_mm(u64 clone_flags, struct task_struct *tsk) tsk->mm = mm; tsk->active_mm = mm; + sched_cache_fork(tsk); return 0; } @@ -2602,6 +2603,7 @@ __latent_entropy struct task_struct *copy_process( bad_fork_cleanup_namespaces: exit_nsproxy_namespaces(p); bad_fork_cleanup_mm: + sched_cache_fork_cleanup(p); if (p->mm) { mm_clear_owner(p->mm, p); mmput(p->mm); diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index f0a9586332fd..974a7dfe3215 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -1478,7 +1478,7 @@ static inline int get_sched_cache_scale(int mul) return (1 + (tol - 1) * mul); } -static bool exceed_llc_capacity(struct mm_struct *mm, int cpu) +static bool exceed_llc_capacity(struct sched_cache_group *grp, int cpu) { #ifdef CONFIG_NUMA_BALANCING unsigned long llc, footprint; @@ -1492,11 +1492,6 @@ static bool exceed_llc_capacity(struct mm_struct *mm, int cpu) return true; if (static_branch_likely(&sched_numa_balancing)) { - struct sched_cache_group *grp = READ_ONCE(mm->sched_cache_grp); - - if (!grp) - return true; - /* * TBD: RDT exclusive LLC ways reserved should be * excluded. @@ -1531,10 +1526,9 @@ static bool exceed_llc_capacity(struct mm_struct *mm, int cpu) return false; } -static bool invalid_llc_nr(struct mm_struct *mm, struct task_struct *p, +static bool invalid_llc_nr(struct sched_cache_group *grp, struct task_struct *p, int cpu) { - struct sched_cache_group *grp; int scale; if (get_nr_threads(p) <= 1) @@ -1548,10 +1542,6 @@ static bool invalid_llc_nr(struct mm_struct *mm, struct task_struct *p, if (scale == INT_MAX) return false; - grp = READ_ONCE(mm->sched_cache_grp); - if (!grp) - return true; - return !fits_capacity((READ_ONCE(grp->nr_running_avg) * cpu_smt_num_threads), (scale * per_cpu(sd_llc_size, cpu))); } @@ -1723,6 +1713,96 @@ static void sched_cache_group_put(struct sched_cache_group *grp) call_rcu(&grp->rcu, sched_cache_group_free_rcu); } +DEFINE_FREE(sched_cache_group_put, struct sched_cache_group *, + sched_cache_group_put(_T)); + +#define rcu_deref_sched_cache_grp(tsk) \ + rcu_dereference_check((tsk)->sched_cache_grp, (tsk) == current) + +static struct sched_cache_group *sched_cache_replace_grp(struct task_struct *p, + struct sched_cache_group *new) +{ + struct sched_cache_group *old; + + old = rcu_deref_sched_cache_grp(p); + rcu_assign_pointer(p->sched_cache_grp, new); + + return old; +} + +struct sched_cache_group *sched_cache_group_get(struct sched_cache_group *grp) +{ + /* + * refcount_inc_not_zero() is the acquire primitive for lockless + * (RCU) lookups; plain refcount_inc() would scribble the count if + * it already reached zero. Return NULL in that case. + */ + if (grp && !refcount_inc_not_zero(&grp->refcnt)) + grp = NULL; + + return grp; +} + +struct sched_cache_group *task_cache_group_get(struct task_struct *p) +{ + guard(rcu)(); + return sched_cache_group_get(rcu_dereference(p->sched_cache_grp)); +} + +void sched_cache_fork(struct task_struct *p) +{ + /* + * The child takes its own reference on the mm's cache group, separate + * from the reference held by the mm. @p is not yet visible to readers, + * so a plain initializing store is enough. + */ + RCU_INIT_POINTER(p->sched_cache_grp, + sched_cache_group_get(p->mm->sched_cache_grp)); +} + +void sched_cache_fork_cleanup(struct task_struct *p) +{ + /* + * A fork that fails after sched_cache_fork() never reaches exit_mm(), + * so drop the reference here. @p never became visible, so there are no + * concurrent readers and the reference we hold keeps the group alive. + */ + sched_cache_group_put(rcu_access_pointer(p->sched_cache_grp)); + RCU_INIT_POINTER(p->sched_cache_grp, NULL); +} + +void sched_cache_exec_mmap(struct task_struct *p, struct mm_struct *mm) +{ + struct sched_cache_group *old; + + /* + * Acquire the new reference before publishing the pointer, then drop + * the old one. @p is current and the only writer of its own pointer. + */ + old = sched_cache_replace_grp(p, sched_cache_group_get(mm->sched_cache_grp)); + sched_cache_group_put(old); +} + +void sched_cache_exit_mm(struct task_struct *p) +{ + struct sched_cache_group *grp = sched_cache_replace_grp(p, NULL); + +#ifdef CONFIG_NUMA_BALANCING + /* + * Subtract this task's footprint from the group before dropping the + * reference, so the group footprint converges as its threads exit. + * Unlocked for performance; clamp to avoid underflow. + */ + if (grp && p->total_numa_faults) { + unsigned long fp = READ_ONCE(grp->footprint); + unsigned long sub = min(fp, p->total_numa_faults); + + WRITE_ONCE(grp->footprint, fp - sub); + } +#endif + sched_cache_group_put(grp); +} + void mm_destroy_sched(struct mm_struct *mm) { sched_cache_group_put(mm->sched_cache_grp); @@ -1777,15 +1857,10 @@ static unsigned long fraction_mm_sched(struct rq *rq, return div64_u64(NICE_0_LOAD * pcpu_sched->runtime, rq->cpu_runtime + 1); } -static int get_pref_llc(struct task_struct *p, struct mm_struct *mm) +static int get_pref_llc(struct task_struct *p, struct sched_cache_group *grp) { int mm_sched_llc = -1, mm_sched_cpu; - struct sched_cache_group *grp; - if (!mm) - return -1; - - grp = READ_ONCE(mm->sched_cache_grp); if (!grp) return -1; @@ -1819,9 +1894,8 @@ static unsigned int task_running_on_cpu(int cpu, struct task_struct *p); static inline void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec) { + struct sched_cache_group *grp = rcu_dereference_all(p->sched_cache_grp); struct sched_cache_time *pcpu_sched; - struct sched_cache_group *grp; - struct mm_struct *mm = p->mm; int mm_sched_llc = -1; unsigned long epoch; @@ -1832,12 +1906,8 @@ void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec) return; /* * init_task, kthreads and user thread created - * by user_mode_thread() don't have mm. + * by user_mode_thread() don't have a cache group. */ - if (!mm) - return; - - grp = READ_ONCE(mm->sched_cache_grp); if (!grp || !grp->pcpu_sched) return; @@ -1855,13 +1925,13 @@ void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec) * its preferred state. */ if ((long)(epoch - READ_ONCE(grp->epoch)) > llc_epoch_affinity_timeout || - invalid_llc_nr(mm, p, cpu_of(rq)) || - exceed_llc_capacity(mm, cpu_of(rq))) { + invalid_llc_nr(grp, p, cpu_of(rq)) || + exceed_llc_capacity(grp, cpu_of(rq))) { if (READ_ONCE(grp->cpu) != -1) WRITE_ONCE(grp->cpu, -1); } - mm_sched_llc = get_pref_llc(p, mm); + mm_sched_llc = get_pref_llc(p, grp); /* task not on rq accounted later in account_entity_enqueue() */ if (task_running_on_cpu(rq->cpu, p) && @@ -1874,19 +1944,15 @@ void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec) static void task_tick_cache(struct rq *rq, struct task_struct *p) { + struct sched_cache_group *grp = rcu_dereference_all(p->sched_cache_grp); struct callback_head *work = &p->cache_work; - struct sched_cache_group *grp; - struct mm_struct *mm = p->mm; unsigned long epoch; if (!sched_cache_enabled()) return; - if (!mm || p->flags & PF_KTHREAD) - return; - - grp = READ_ONCE(mm->sched_cache_grp); - if (!grp || !grp->pcpu_sched) + if (!grp || p->flags & PF_KTHREAD || + !grp->pcpu_sched) return; epoch = rq->cpu_epoch; @@ -1968,14 +2034,13 @@ static inline void update_avg_scale(u64 *avg, u64 sample) static void task_cache_work(struct callback_head *work) { + struct sched_cache_group *grp __free(sched_cache_group_put) = NULL; + cpumask_var_t cpus __free(free_cpumask_var) = CPUMASK_VAR_NULL; int cpu, m_a_cpu = -1, nr_running = 0, curr_cpu; unsigned long next_scan, now = jiffies; struct task_struct *p = current, *cur; unsigned long curr_m_a_occ = 0; - struct sched_cache_group *grp; - struct mm_struct *mm = p->mm; unsigned long m_a_occ = 0; - cpumask_var_t cpus; WARN_ON_ONCE(work != &p->cache_work); @@ -1984,7 +2049,12 @@ static void task_cache_work(struct callback_head *work) if (p->flags & PF_EXITING) return; - grp = READ_ONCE(mm->sched_cache_grp); + /* + * A reference makes sure grp is not released by others. The rcu + * lock can not be held till after zalloc_cpumask_var() below, + * because the latter might sleep. + */ + grp = task_cache_group_get(p); if (!grp) return; @@ -1999,8 +2069,8 @@ static void task_cache_work(struct callback_head *work) return; curr_cpu = task_cpu(p); - if (invalid_llc_nr(mm, p, curr_cpu) || - exceed_llc_capacity(mm, curr_cpu)) { + if (invalid_llc_nr(grp, p, curr_cpu) || + exceed_llc_capacity(grp, curr_cpu)) { if (READ_ONCE(grp->cpu) != -1) WRITE_ONCE(grp->cpu, -1); @@ -2033,9 +2103,13 @@ static void task_cache_work(struct callback_head *work) m_cpu = i; } + /* + * rcu_access_pointer() is used because the + * pointer is only compared, never dereferenced. + */ cur = rcu_dereference_all(cpu_rq(i)->curr); if (cur && !(cur->flags & (PF_EXITING | PF_KTHREAD)) && - cur->mm == mm) + rcu_access_pointer(cur->sched_cache_grp) == grp) nr_running++; } @@ -2081,7 +2155,6 @@ static void task_cache_work(struct callback_head *work) } update_avg_scale(&grp->nr_running_avg, nr_running); - free_cpumask_var(cpus); } void init_sched_mm(struct task_struct *p) @@ -2090,6 +2163,13 @@ void init_sched_mm(struct task_struct *p) init_task_work(work, task_cache_work); work->next = work; + /* + * dup_task_struct() copies the parent's task_struct, including its + * sched_cache_grp, for which the child holds no reference. Clear it + * here - before copy_mm() runs - so the child never carries a + * borrowed pointer that the fork error path would put. + */ + RCU_INIT_POINTER(p->sched_cache_grp, NULL); /* * Reset new task's preference to avoid * polluting account_llc_enqueue(). @@ -3890,10 +3970,9 @@ static void task_numa_placement(struct task_struct *p) * heuristic and occasional lost updates are tolerable. * * If a task exits, its corresponding footprint must - * be subtracted from the mm->sched_cache_grp->footprint, - * otherwise the mm->sched_cache_grp->footprint will not - * converge: the exiting thread's footprint remains - * unchanged/undecayed in mm->sched_cache_grp->footprint. + * be subtracted from p->sched_cache_grp->footprint, + * otherwise the footprint will not converge: the + * exiting thread's footprint remains unchanged/undecayed. * See exit_mm(). * * Lost updates and unsynchronized subtraction @@ -3901,12 +3980,14 @@ static void task_numa_placement(struct task_struct *p) * go negative. Clamp to zero to prevent the * unsigned footprint from wrapping. */ - grp = READ_ONCE(p->mm->sched_cache_grp); - if (!grp) - continue; + scoped_guard(rcu) { + grp = rcu_dereference(p->sched_cache_grp); - new_fp = (long)READ_ONCE(grp->footprint) + diff; - WRITE_ONCE(grp->footprint, max(new_fp, 0L)); + if (grp) { + new_fp = (long)READ_ONCE(grp->footprint) + diff; + WRITE_ONCE(grp->footprint, max(new_fp, 0L)); + } + } #endif } @@ -10855,7 +10936,6 @@ static enum llc_mig can_migrate_llc_task(struct lb_env *env, struct task_struct *p) { struct sched_cache_group *grp; - struct mm_struct *mm; bool to_pref; int cpu, src_cpu, dst_cpu; @@ -10864,11 +10944,7 @@ static enum llc_mig can_migrate_llc_task(struct lb_env *env, src_cpu = env->src_cpu; dst_cpu = env->dst_cpu; - mm = p->mm; - if (!mm) - return mig_unrestricted; - - grp = READ_ONCE(mm->sched_cache_grp); + grp = rcu_dereference_all(p->sched_cache_grp); if (!grp) return mig_unrestricted; @@ -10877,8 +10953,8 @@ static enum llc_mig can_migrate_llc_task(struct lb_env *env, return mig_unrestricted; /* skip cache aware load balance for too many threads */ - if (invalid_llc_nr(mm, p, dst_cpu) || - exceed_llc_capacity(mm, dst_cpu)) { + if (invalid_llc_nr(grp, p, dst_cpu) || + exceed_llc_capacity(grp, dst_cpu)) { if (READ_ONCE(grp->cpu) != -1) WRITE_ONCE(grp->cpu, -1); return mig_unrestricted; From 65efcccddc83d6a19e8a2e2a6117811e39e589bf Mon Sep 17 00:00:00 2001 From: Chen Yu Date: Mon, 21 Sep 2026 17:37:26 -0700 Subject: [PATCH 5/7] sched/cache: Skip kernel threads for cache aware scheduling to rubustify the code Kernel thread should not be covered by cache aware scheduling as it borrows the statistics from the user space thread. Filter the kernel thread in account_mm_sched(). In theory a kernel thread does not have any valid cache group, so !grp should gate the kernel thread. Add the PF_KTHREAD check explicitly here for safety reasons, to guard against future modifications and to pair with task_tick_cache(). Fixes: df0d98475954 ("sched/cache: Introduce infrastructure for cache-aware load balancing") Signed-off-by: Chen Yu Signed-off-by: Tim Chen Signed-off-by: Peter Zijlstra (Intel) Signed-off-by: Ingo Molnar Cc: # 7.2.x Link: https://patch.msgid.link/058f0c6ea7b991c177a17de347fa3157f25489a7.1790035273.git.tim.c.chen@linux.intel.com --- kernel/sched/fair.c | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 974a7dfe3215..57360f5cdde4 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -1907,8 +1907,14 @@ void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec) /* * init_task, kthreads and user thread created * by user_mode_thread() don't have a cache group. + * In theory a kernel thread does not have any valid + * cache group, because sched_cache_fork() is not + * invoked for a kernel thread - !grp should gate the + * kernel thread. Use the PF_KTHREAD check explicitly + * here for safety reasons, to guard against future + * modifications and to pair with task_tick_cache(). */ - if (!grp || !grp->pcpu_sched) + if (p->flags & PF_KTHREAD || !grp || !grp->pcpu_sched) return; pcpu_sched = per_cpu_ptr(grp->pcpu_sched, cpu_of(rq)); From 3cb0243767fd033bdce95f4f1b5882172a2f8119 Mon Sep 17 00:00:00 2001 From: Davi Chaves Azevedo Date: Mon, 21 Sep 2026 17:37:27 -0700 Subject: [PATCH 6/7] sched/cache: Refresh LLC capacity across CPU hotplug, to fix capacity underestimation bug The scheduler scales LLC capacity by the fraction of cache-sharing CPUs covered by a domain: llc_bytes = cache_size * span_weight / shared_weight During CPU teardown, sched_cpu_deactivate() rebuilds scheduler domains before cacheinfo_cpu_pre_down() removes the CPU from shared_cpu_map. The new domains therefore use the old sharing weight. The later call to sched_update_llc_bytes() looks up the departing CPU's sd_llc, which has already been detached, and returns without correcting the surviving CPUs. On a Ryzen 5 7535U with twelve logical CPUs sharing a 16 MiB LLC, offlining one SMT sibling left the remaining CPUs with: llc_bytes = floor(16777216 * 11 / 12) = 15379114 bytes The correct capacity is still 16777216 bytes. On systems with active cache-aware scheduling, an underestimated capacity can cause exceed_llc_capacity() to reject aggregation for a process whose footprint would fit. Unchanged cpuset partitions sharing the physical cache can also retain stale capacity when a CPU comes online in another partition. Pass the cache-sharing mask already retained by cacheinfo to the scheduler update. Refresh every surviving CPU using its own LLC domain so that each partition receives the correct share. This also preserves the correction needed as cache-sharing maps grow during boot. Keep the existing CPU-hotplug and scheduler-domain synchronization. The update remains on the hotplug path; no steady-state scheduling operation or persistent allocation is added. Fixes: 7030513a0877 ("sched/cache: Calculate the LLC size and store it in sched_domain") Signed-off-by: Davi Chaves Azevedo Signed-off-by: Tim Chen Signed-off-by: Peter Zijlstra (Intel) Signed-off-by: Ingo Molnar Reviewed-by: Chen Yu Reviewed-by: Tim Chen Reviewed-by: K Prateek Nayak Tested-by: Chen Yu Tested-by: K Prateek Nayak Cc: # v7.2.x Link: https://patch.msgid.link/6751d93e15889e624796c74db0bfe66603d60b1b.1790035273.git.tim.c.chen@linux.intel.com --- drivers/base/cacheinfo.c | 11 ++++++----- include/linux/sched/topology.h | 4 ++-- kernel/sched/topology.c | 22 +++++++++++++--------- 3 files changed, 21 insertions(+), 16 deletions(-) diff --git a/drivers/base/cacheinfo.c b/drivers/base/cacheinfo.c index 9f9c72727a05..7a47a392568a 100644 --- a/drivers/base/cacheinfo.c +++ b/drivers/base/cacheinfo.c @@ -1040,9 +1040,10 @@ static int cacheinfo_cpu_online(unsigned int cpu) rc = cache_add_dev(cpu); if (rc) goto err; - if (cpu_map_shared_cache(true, cpu, &cpu_map)) + if (cpu_map_shared_cache(true, cpu, &cpu_map)) { update_per_cpu_data_slice_size(true, cpu, cpu_map); - sched_update_llc_bytes(cpu); + sched_update_llc_bytes(cpu_map); + } return 0; err: free_cache_attributes(cpu); @@ -1059,10 +1060,10 @@ static int cacheinfo_cpu_pre_down(unsigned int cpu) cpu_cache_sysfs_exit(cpu); free_cache_attributes(cpu); - if (nr_shared > 1) + if (nr_shared > 1) { update_per_cpu_data_slice_size(false, cpu, cpu_map); - - sched_update_llc_bytes(cpu); + sched_update_llc_bytes(cpu_map); + } return 0; } diff --git a/include/linux/sched/topology.h b/include/linux/sched/topology.h index b5d9d7c2b8ad..f96812d71c51 100644 --- a/include/linux/sched/topology.h +++ b/include/linux/sched/topology.h @@ -281,9 +281,9 @@ static inline int task_node(const struct task_struct *p) } #ifdef CONFIG_SCHED_CACHE -extern void sched_update_llc_bytes(unsigned int cpu); +extern void sched_update_llc_bytes(const struct cpumask *cpus); #else -static inline void sched_update_llc_bytes(unsigned int cpu) { } +static inline void sched_update_llc_bytes(const struct cpumask *cpus) { } #endif #endif /* _LINUX_SCHED_TOPOLOGY_H */ diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c index 0248227d983a..3dab0253976f 100644 --- a/kernel/sched/topology.c +++ b/kernel/sched/topology.c @@ -985,8 +985,8 @@ void sched_cache_active_set(void) } /* - * Update the bottom sched_domain's llc_bytes for @cpu and all its - * LLC siblings. Called from cacheinfo_cpu_online() or + * Update the bottom sched_domain's llc_bytes for @cpus sharing a physical + * LLC. Called from cacheinfo_cpu_online() or * cacheinfo_cpu_pre_down() with cpu hotplug lock held. * * Note: get_effective_llc_bytes() returns 0 on PowerPC. @@ -996,17 +996,13 @@ void sched_cache_active_set(void) * and does not populates the per-CPU struct cpu_cacheinfo array * that get_cpu_cacheinfo_llc() reads. */ -void sched_update_llc_bytes(unsigned int cpu) +void sched_update_llc_bytes(const struct cpumask *cpus) { struct sched_domain *sd, *sdp; unsigned int i; sched_domains_mutex_lock(); - sdp = rcu_dereference_sched_domain(per_cpu(sd_llc, cpu)); - if (!sdp) - goto unlock; - /* * ci->shared_cpu_map is built incrementally as CPUs come * online, so the first CPU in an LLC initially sees @@ -1014,14 +1010,22 @@ void sched_update_llc_bytes(unsigned int cpu) * get_effective_llc_bytes(). Re-evaluating every LLC * sibling on each online event corrects this once the full * shared_cpu_map is known. + * + * The departing CPU's domains have already been detached when + * cacheinfo removes it. Use the surviving cache siblings instead. + * They may belong to different cpuset partitions, so use each CPU's + * own LLC domain to scale its share of the physical cache. */ - for_each_cpu(i, sched_domain_span(sdp)) { + for_each_cpu(i, cpus) { + sdp = rcu_dereference_sched_domain(per_cpu(sd_llc, i)); + if (!sdp) + continue; + sd = rcu_dereference_sched_domain(cpu_rq(i)->sd); if (sd) sd->llc_bytes = get_effective_llc_bytes(i, sdp); } -unlock: sched_domains_mutex_unlock(); } From a0bb6fac53fa7cf1cadb487b43d4c9276a6b82e3 Mon Sep 17 00:00:00 2001 From: Zhan Xusheng Date: Fri, 18 Sep 2026 21:29:15 +0800 Subject: [PATCH 7/7] sched/core: Account PSI IRQ time to the execution context, not the scheduling context psi_account_irqtime() has two callers which share rq->psi_irq_time, and they disagree about the context: __schedule() passes the outgoing rq->curr, sched_tick() passes rq->donor. Under proxy execution the donor is blocked on a mutex while rq->curr burns the CPU. The tick charges PSI_IRQ_FULL to the donor's cgroup and advances the timestamp, so the call from __schedule() then finds delta <= 0 and charges nothing. The delta is not counted twice, it lands on the wrong cgroup. Pass rq->curr, which is what the call read before commit af0c8b2bf67b ("sched: Split scheduler and execution contexts") renamed 'curr' to 'donor' across sched_tick(). Without CONFIG_SCHED_PROXY_EXEC the two rq members are a union, so this only changes anything where that option is set, and it depends on EXPERT. Fixes: af0c8b2bf67b ("sched: Split scheduler and execution contexts") Signed-off-by: Zhan Xusheng Signed-off-by: Peter Zijlstra (Intel) Signed-off-by: Ingo Molnar Link: https://patch.msgid.link/20260918132915.1236312-1-zhanxusheng@xiaomi.com --- kernel/sched/core.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kernel/sched/core.c b/kernel/sched/core.c index 0b846a13c628..1fe40de6ebe3 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -5798,7 +5798,7 @@ void sched_tick(void) curr = rq->curr; donor = rq->donor; - psi_account_irqtime(rq, donor, NULL); + psi_account_irqtime(rq, curr, NULL); update_rq_clock(rq); hw_pressure = arch_scale_hw_pressure(cpu_of(rq));