sched_ext: RCU-protect the sub-sched tree's children/sibling lists

Future kfuncs need to walk descendants without scx_sched_lock. Make the
walker RCU-safe so that they can. A sub-sched's fields are initialized
before it is linked, so a walk that observes a linked node also observes its
setup. In-place changes after linking carry their own ordering.

Switch the children/sibling list ops to RCU and expand the descendant walker
to accept rcu_read_lock as a valid read-side context. Walkers that mutate
keep scx_sched_lock.

A sub-sched can be linked while an ancestor is bypassing, after the bypass
walk that propagates the depth has passed its parent. Bypass state is a
per-cpu flag plus a depth count and can't be established atomically at link
time, so refuse to link under a bypassing ancestor. Take scx_bypass_lock
across linking to check the parent's bypass state coherently.

v3: Reject linking under a bypassing ancestor instead of inheriting bypass_depth. (sashiko AI)
v2: Inherit bypass_depth before publishing @sch on the RCU sibling list.

Signed-off-by: Tejun Heo <tj@kernel.org>
Reviewed-by: Andrea Righi <arighi@nvidia.com>
This commit is contained in:
Tejun Heo 2026-07-13 22:18:43 -10:00
parent 33ffb56e85
commit 70f8b17853
3 changed files with 24 additions and 9 deletions

View File

@ -5502,7 +5502,8 @@ s32 scx_link_sched(struct scx_sched *sch)
const char *err_msg = "";
s32 ret = 0;
scoped_guard(raw_spinlock_irq, &scx_sched_lock) {
scoped_guard(raw_spinlock_irqsave, &scx_bypass_lock) /* for the parent bypass check */
scoped_guard(raw_spinlock, &scx_sched_lock) {
#ifdef CONFIG_EXT_SUB_SCHED
struct scx_sched *parent = scx_parent(sch);
@ -5519,6 +5520,17 @@ s32 scx_link_sched(struct scx_sched *sch)
break;
}
/*
* Bypass state is spread across per-cpu flags and a
* depth count, so inheriting it is tricky and has no
* valid use case. Refuse it.
*/
if (READ_ONCE(parent->bypass_depth)) {
err_msg = "parent bypassing";
ret = -EBUSY;
break;
}
ret = rhashtable_lookup_insert_fast(&scx_sched_hash,
&sch->hash_node, scx_sched_hash_params);
if (ret) {
@ -5526,7 +5538,7 @@ s32 scx_link_sched(struct scx_sched *sch)
break;
}
list_add_tail(&sch->sibling, &parent->children);
list_add_tail_rcu(&sch->sibling, &parent->children);
}
#endif /* CONFIG_EXT_SUB_SCHED */
@ -5553,7 +5565,7 @@ void scx_unlink_sched(struct scx_sched *sch)
if (scx_parent(sch)) {
rhashtable_remove_fast(&scx_sched_hash, &sch->hash_node,
scx_sched_hash_params);
list_del_init(&sch->sibling);
list_del_rcu(&sch->sibling);
}
#endif /* CONFIG_EXT_SUB_SCHED */
list_del_rcu(&sch->all);

View File

@ -35,21 +35,24 @@ struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sche
struct scx_sched *next;
lockdep_assert(lockdep_is_held(&scx_enable_mutex) ||
lockdep_is_held(&scx_sched_lock));
lockdep_is_held(&scx_sched_lock) ||
rcu_read_lock_any_held());
/* if first iteration, visit @root */
if (!pos)
return root;
/* visit the first child if exists */
next = list_first_entry_or_null(&pos->children, struct scx_sched, sibling);
next = list_first_or_null_rcu(&pos->children, struct scx_sched, sibling);
if (next)
return next;
/* no child, visit my or the closest ancestor's next sibling */
while (pos != root) {
if (!list_is_last(&pos->sibling, &scx_parent(pos)->children))
return list_next_entry(pos, sibling);
next = list_next_or_null_rcu(&scx_parent(pos)->children, &pos->sibling,
struct scx_sched, sibling);
if (next)
return next;
pos = scx_parent(pos);
}

View File

@ -52,8 +52,8 @@ static inline s32 scx_alloc_pshards(struct scx_sched *sch) { return 0; }
* @root: sched to walk the descendants of
*
* Walk @root's descendants. @root is included in the iteration and the first
* node to be visited. Must be called with either scx_enable_mutex or
* scx_sched_lock held.
* node to be visited. Must be called with scx_enable_mutex, scx_sched_lock, or
* RCU read lock.
*/
#define scx_for_each_descendant_pre(pos, root) \
for ((pos) = scx_next_descendant_pre(NULL, (root)); (pos); \