sched_ext: Defer scx_sched kobj sysfs add into the enable workfns

Split kobject_init_and_add() in scx_alloc_and_add_sched(): only
kobject_init() runs there. A new scx_sched_sysfs_add() helper does
kobject_add() (and creates sub_kset when the scheduler implements
ops.sub_attach), called by both enable workfns once @sch is linked and its
sysfs-visible state is initialized. Prep so a future caps attribute can rely
on @sch being fully built by the time it's sysfs-visible. Add early enough
that a stall later in enable still leaves sysfs inspectable.

Signed-off-by: Tejun Heo <tj@kernel.org>
Reviewed-by: Andrea Righi <arighi@nvidia.com>
This commit is contained in:
Tejun Heo 2026-07-13 22:18:42 -10:00
parent 30067643bc
commit 80e6adaa35
3 changed files with 53 additions and 29 deletions

View File

@ -5735,7 +5735,9 @@ static void scx_root_disable(struct scx_sched *sch)
if (sch->sub_kset)
kobject_del(&sch->sub_kset->kobj);
#endif
kobject_del(&sch->kobj);
/* not added if enable failed before scx_sched_sysfs_add() */
if (sch->kobj.state_in_sysfs)
kobject_del(&sch->kobj);
free_kick_syncs();
@ -6454,36 +6456,15 @@ struct scx_sched *scx_alloc_and_add_sched(struct scx_enable_cmd *cmd,
* disable. Released in scx_sched_free_rcu_work().
*/
kobject_get(&parent->kobj);
ret = kobject_init_and_add(&sch->kobj, &scx_ktype,
&parent->sub_kset->kobj,
"sub-%llu", cgroup_id(cgrp));
} else {
ret = kobject_init_and_add(&sch->kobj, &scx_ktype, NULL, "root");
}
if (ret < 0) {
RCU_INIT_POINTER(ops->priv, NULL);
kobject_put(&sch->kobj);
return ERR_PTR(ret);
}
if (ops->sub_attach) {
sch->sub_kset = kset_create_and_add("sub", NULL, &sch->kobj);
if (!sch->sub_kset) {
RCU_INIT_POINTER(ops->priv, NULL);
kobject_put(&sch->kobj);
return ERR_PTR(-ENOMEM);
}
}
#else /* CONFIG_EXT_SUB_SCHED */
ret = kobject_init_and_add(&sch->kobj, &scx_ktype, NULL, "root");
if (ret < 0) {
RCU_INIT_POINTER(ops->priv, NULL);
kobject_put(&sch->kobj);
return ERR_PTR(ret);
}
#endif /* CONFIG_EXT_SUB_SCHED */
/*
* Init the kobj but don't add to sysfs yet. The enable path calls
* scx_sched_sysfs_add() once @sch's sysfs-visible state is initialized.
*/
kobject_init(&sch->kobj, &scx_ktype);
/*
* Consume the arena_map ref bpf_scx_reg_cid() took. Defer to here so
* earlier failure paths leave cmd->arena_map set and bpf_scx_reg_cid
@ -6541,6 +6522,36 @@ struct scx_sched *scx_alloc_and_add_sched(struct scx_enable_cmd *cmd,
return ERR_PTR(ret);
}
/*
* Add @sch's kobject to sysfs, and create its sub_kset if the scheduler
* implements ops.sub_attach. Called by the enable workfns once @sch's
* sysfs-visible state is initialized.
*/
int scx_sched_sysfs_add(struct scx_sched *sch)
{
#ifdef CONFIG_EXT_SUB_SCHED
struct scx_sched *parent = scx_parent(sch);
int ret;
if (parent)
ret = kobject_add(&sch->kobj, &parent->sub_kset->kobj,
"sub-%llu", cgroup_id(sch_cgroup(sch)));
else
ret = kobject_add(&sch->kobj, NULL, "root");
if (ret < 0)
return ret;
if (sch->ops.sub_attach) {
sch->sub_kset = kset_create_and_add("sub", NULL, &sch->kobj);
if (!sch->sub_kset)
return -ENOMEM;
}
return 0;
#else
return kobject_add(&sch->kobj, NULL, "root");
#endif
}
static int check_hotplug_seq(struct scx_sched *sch,
const struct sched_ext_ops *ops)
{
@ -6763,6 +6774,12 @@ static void scx_root_enable_workfn(struct kthread_work *work)
sch->exit_info->flags |= SCX_EFLAG_INITIALIZED;
}
ret = scx_sched_sysfs_add(sch);
if (ret) {
cpus_read_unlock();
goto err_disable;
}
for (i = SCX_OPI_CPU_HOTPLUG_BEGIN; i < SCX_OPI_CPU_HOTPLUG_END; i++)
if (((void (**)(void))ops)[i])
set_bit(i, sch->has_op);

View File

@ -1690,6 +1690,7 @@ struct scx_sched *scx_alloc_and_add_sched(struct scx_enable_cmd *cmd,
struct cgroup *cgrp,
struct scx_sched *parent);
int scx_validate_ops(struct scx_sched *sch, const struct sched_ext_ops *ops);
int scx_sched_sysfs_add(struct scx_sched *sch);
extern raw_spinlock_t scx_sched_lock;
extern struct mutex scx_enable_mutex;

View File

@ -270,7 +270,9 @@ void scx_sub_disable(struct scx_sched *sch)
SCX_CALL_OP(sch, exit, NULL, sch->exit_info);
if (sch->sub_kset)
kobject_del(&sch->sub_kset->kobj);
kobject_del(&sch->kobj);
/* not added if enable failed before scx_sched_sysfs_add() */
if (sch->kobj.state_in_sysfs)
kobject_del(&sch->kobj);
}
/* verify that a scheduler can be attached to @cgrp and return the parent */
@ -363,6 +365,10 @@ void scx_sub_enable_workfn(struct kthread_work *work)
if (ret)
goto err_disable;
ret = scx_sched_sysfs_add(sch);
if (ret)
goto err_disable;
if (sch->level >= SCX_SUB_MAX_DEPTH) {
scx_error(sch, "max nesting depth %d violated",
SCX_SUB_MAX_DEPTH);