mirror of
https://github.com/torvalds/linux.git
synced 2026-07-28 10:09:10 +02:00
Merge branch 'for-7.1-fixes' into for-7.2
Conflict between: [1]41e3312861("sched_ext: add p->scx.tid and SCX_OPS_TID_TO_TASK lookup") [2]c941d7391f("sched_ext: Close root-enable vs sched_ext_dead() race with SCX_TASK_INIT_BEGIN") in scx_root_enable_workfn()'s post-init block. [1] added a tid hash insertion under a scoped_guard() for scx_tasks_lock; [2] wraps the same region in task_rq_lock() for a DEAD recheck. A naive merge would invert the iter's outer/inner order. [3]f25ad1e3cb("sched_ext: Add scx_task_iter_relock() and use it in scx_root_enable_workfn()") was added to for-7.2 for a clean resolution: scx_task_iter_relock(iter, p) takes both scx_tasks_lock and @p's rq lock in iter order. Resolved by routing both sides through [3]'s dual-lock helper: the post-init region runs under a single scx_task_iter_relock() acquisition, with [2]'s state machine and [1]'s hash insert in sequence inside it. Signed-off-by: Tejun Heo <tj@kernel.org>
This commit is contained in:
commit
b5646c6276
|
|
@ -53,6 +53,7 @@ struct kernel_clone_args;
|
|||
enum css_task_iter_flags {
|
||||
CSS_TASK_ITER_PROCS = (1U << 0), /* walk only threadgroup leaders */
|
||||
CSS_TASK_ITER_THREADED = (1U << 1), /* walk all threaded css_sets in the domain */
|
||||
CSS_TASK_ITER_WITH_DEAD = (1U << 2), /* include exiting tasks */
|
||||
CSS_TASK_ITER_SKIPPED = (1U << 16), /* internal flags */
|
||||
};
|
||||
|
||||
|
|
|
|||
|
|
@ -103,21 +103,25 @@ enum scx_ent_flags {
|
|||
SCX_TASK_IMMED = 1 << 5, /* task is on local DSQ with %SCX_ENQ_IMMED */
|
||||
|
||||
/*
|
||||
* Bits 8 and 9 are used to carry task state:
|
||||
* Bits 8 to 10 are used to carry task state:
|
||||
*
|
||||
* NONE ops.init_task() not called yet
|
||||
* INIT_BEGIN ops.init_task() in flight; see sched_ext_dead()
|
||||
* INIT ops.init_task() succeeded, but task can be cancelled
|
||||
* READY fully initialized, but not in sched_ext
|
||||
* ENABLED fully initialized and in sched_ext
|
||||
* DEAD terminal state set by sched_ext_dead()
|
||||
*/
|
||||
SCX_TASK_STATE_SHIFT = 8, /* bits 8 and 9 are used to carry task state */
|
||||
SCX_TASK_STATE_BITS = 2,
|
||||
SCX_TASK_STATE_SHIFT = 8,
|
||||
SCX_TASK_STATE_BITS = 3,
|
||||
SCX_TASK_STATE_MASK = ((1 << SCX_TASK_STATE_BITS) - 1) << SCX_TASK_STATE_SHIFT,
|
||||
|
||||
SCX_TASK_NONE = 0 << SCX_TASK_STATE_SHIFT,
|
||||
SCX_TASK_INIT = 1 << SCX_TASK_STATE_SHIFT,
|
||||
SCX_TASK_READY = 2 << SCX_TASK_STATE_SHIFT,
|
||||
SCX_TASK_ENABLED = 3 << SCX_TASK_STATE_SHIFT,
|
||||
SCX_TASK_INIT_BEGIN = 1 << SCX_TASK_STATE_SHIFT,
|
||||
SCX_TASK_INIT = 2 << SCX_TASK_STATE_SHIFT,
|
||||
SCX_TASK_READY = 3 << SCX_TASK_STATE_SHIFT,
|
||||
SCX_TASK_ENABLED = 4 << SCX_TASK_STATE_SHIFT,
|
||||
SCX_TASK_DEAD = 5 << SCX_TASK_STATE_SHIFT,
|
||||
|
||||
/*
|
||||
* Bits 12 and 13 are used to carry reenqueue reason. In addition to
|
||||
|
|
|
|||
|
|
@ -5059,10 +5059,12 @@ static void css_task_iter_advance(struct css_task_iter *it)
|
|||
|
||||
task = list_entry(it->task_pos, struct task_struct, cg_list);
|
||||
/*
|
||||
* Hide tasks that are exiting but not yet removed. Keep zombie
|
||||
* leaders with live threads visible.
|
||||
* Hide tasks that are exiting but not yet removed by default. Keep
|
||||
* zombie leaders with live threads visible. Usages that need to walk
|
||||
* every existing task can opt out via CSS_TASK_ITER_WITH_DEAD.
|
||||
*/
|
||||
if ((task->flags & PF_EXITING) && !atomic_read(&task->signal->live))
|
||||
if (!(it->flags & CSS_TASK_ITER_WITH_DEAD) &&
|
||||
(task->flags & PF_EXITING) && !atomic_read(&task->signal->live))
|
||||
goto repeat;
|
||||
|
||||
if (it->flags & CSS_TASK_ITER_PROCS) {
|
||||
|
|
|
|||
|
|
@ -325,7 +325,6 @@ static void scx_set_task_sched(struct task_struct *p, struct scx_sched *sch)
|
|||
#else /* CONFIG_EXT_SUB_SCHED */
|
||||
static struct scx_sched *scx_parent(struct scx_sched *sch) { return NULL; }
|
||||
static struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sched *root) { return pos ? NULL : root; }
|
||||
static struct scx_sched *scx_find_sub_sched(u64 cgroup_id) { return NULL; }
|
||||
static void scx_set_task_sched(struct task_struct *p, struct scx_sched *sch) {}
|
||||
#endif /* CONFIG_EXT_SUB_SCHED */
|
||||
|
||||
|
|
@ -802,6 +801,51 @@ struct bpf_iter_scx_dsq {
|
|||
} __attribute__((aligned(8)));
|
||||
|
||||
|
||||
static u32 scx_get_task_state(const struct task_struct *p)
|
||||
{
|
||||
return p->scx.flags & SCX_TASK_STATE_MASK;
|
||||
}
|
||||
|
||||
static void scx_set_task_state(struct task_struct *p, u32 state)
|
||||
{
|
||||
u32 prev_state = scx_get_task_state(p);
|
||||
bool warn = false;
|
||||
|
||||
switch (state) {
|
||||
case SCX_TASK_NONE:
|
||||
warn = prev_state == SCX_TASK_DEAD;
|
||||
break;
|
||||
case SCX_TASK_INIT_BEGIN:
|
||||
warn = prev_state != SCX_TASK_NONE;
|
||||
break;
|
||||
case SCX_TASK_INIT:
|
||||
warn = prev_state != SCX_TASK_INIT_BEGIN;
|
||||
p->scx.flags |= SCX_TASK_RESET_RUNNABLE_AT;
|
||||
break;
|
||||
case SCX_TASK_READY:
|
||||
warn = !(prev_state == SCX_TASK_INIT ||
|
||||
prev_state == SCX_TASK_ENABLED);
|
||||
break;
|
||||
case SCX_TASK_ENABLED:
|
||||
warn = prev_state != SCX_TASK_READY;
|
||||
break;
|
||||
case SCX_TASK_DEAD:
|
||||
warn = !(prev_state == SCX_TASK_NONE ||
|
||||
prev_state == SCX_TASK_INIT_BEGIN);
|
||||
break;
|
||||
default:
|
||||
WARN_ONCE(1, "sched_ext: Invalid task state %d -> %d for %s[%d]",
|
||||
prev_state, state, p->comm, p->pid);
|
||||
return;
|
||||
}
|
||||
|
||||
WARN_ONCE(warn, "sched_ext: Invalid task state transition 0x%x -> 0x%x for %s[%d]",
|
||||
prev_state, state, p->comm, p->pid);
|
||||
|
||||
p->scx.flags &= ~SCX_TASK_STATE_MASK;
|
||||
p->scx.flags |= state;
|
||||
}
|
||||
|
||||
/*
|
||||
* SCX task iterator.
|
||||
*/
|
||||
|
|
@ -856,7 +900,8 @@ static void scx_task_iter_start(struct scx_task_iter *iter, struct cgroup *cgrp)
|
|||
lockdep_assert_held(&cgroup_mutex);
|
||||
iter->cgrp = cgrp;
|
||||
iter->css_pos = css_next_descendant_pre(NULL, &iter->cgrp->self);
|
||||
css_task_iter_start(iter->css_pos, 0, &iter->css_iter);
|
||||
css_task_iter_start(iter->css_pos, CSS_TASK_ITER_WITH_DEAD,
|
||||
&iter->css_iter);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
|
|
@ -974,7 +1019,8 @@ static struct task_struct *scx_task_iter_next(struct scx_task_iter *iter)
|
|||
iter->css_pos = css_next_descendant_pre(iter->css_pos,
|
||||
&iter->cgrp->self);
|
||||
if (iter->css_pos)
|
||||
css_task_iter_start(iter->css_pos, 0, &iter->css_iter);
|
||||
css_task_iter_start(iter->css_pos, CSS_TASK_ITER_WITH_DEAD,
|
||||
&iter->css_iter);
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
|
@ -1034,16 +1080,27 @@ static struct task_struct *scx_task_iter_next_locked(struct scx_task_iter *iter)
|
|||
*
|
||||
* Test for idle_sched_class as only init_tasks are on it.
|
||||
*/
|
||||
if (p->sched_class != &idle_sched_class)
|
||||
break;
|
||||
if (p->sched_class == &idle_sched_class)
|
||||
continue;
|
||||
|
||||
iter->rq = task_rq_lock(p, &iter->rf);
|
||||
iter->locked_task = p;
|
||||
|
||||
/*
|
||||
* cgroup_task_dead() removes the dead tasks from cset->tasks
|
||||
* after sched_ext_dead() and cgroup iteration may see tasks
|
||||
* which already finished sched_ext_dead(). %SCX_TASK_DEAD is
|
||||
* set by sched_ext_dead() under @p's rq lock. Test it to
|
||||
* avoid visiting tasks which are already dead from SCX POV.
|
||||
*/
|
||||
if (scx_get_task_state(p) == SCX_TASK_DEAD) {
|
||||
__scx_task_iter_rq_unlock(iter);
|
||||
continue;
|
||||
}
|
||||
|
||||
return p;
|
||||
}
|
||||
if (!p)
|
||||
return NULL;
|
||||
|
||||
iter->rq = task_rq_lock(p, &iter->rf);
|
||||
iter->locked_task = p;
|
||||
|
||||
return p;
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -1510,14 +1567,51 @@ static void local_dsq_post_enq(struct scx_sched *sch, struct scx_dispatch_q *dsq
|
|||
struct task_struct *p, u64 enq_flags)
|
||||
{
|
||||
struct rq *rq = container_of(dsq, struct rq, scx.local_dsq);
|
||||
bool preempt = false;
|
||||
|
||||
call_task_dequeue(sch, rq, p, 0);
|
||||
|
||||
/*
|
||||
* Note that @rq's lock may be dropped between this enqueue and @p
|
||||
* actually getting on CPU. This gives higher-class tasks (e.g. RT)
|
||||
* an opportunity to wake up on @rq and prevent @p from running.
|
||||
* Here are some concrete examples:
|
||||
*
|
||||
* Example 1:
|
||||
*
|
||||
* We dispatch two tasks from a single ops.dispatch():
|
||||
* - First, a local task to this CPU's local DSQ;
|
||||
* - Second, a local/remote task to a remote CPU's local DSQ.
|
||||
* We must drop the local rq lock in order to finish the second
|
||||
* dispatch. In that time, an RT task can wake up on the local rq.
|
||||
*
|
||||
* Example 2:
|
||||
*
|
||||
* We dispatch a local/remote task to a remote CPU's local DSQ.
|
||||
* We must drop the remote rq lock before the dispatched task can run,
|
||||
* which gives an RT task an opportunity to wake up on the remote rq.
|
||||
*
|
||||
* Both examples work the same if we replace dispatching with moving
|
||||
* the tasks from a user-created DSQ.
|
||||
*
|
||||
* We must detect these wakeups so that we can re-enqueue IMMED tasks
|
||||
* from @rq's local DSQ. scx_wakeup_preempt() serves exactly this
|
||||
* purpose, but for it to be invoked, we must ensure that we bump
|
||||
* @rq->next_class to &ext_sched_class if it's currently idle.
|
||||
*
|
||||
* wakeup_preempt() does the bumping, and since we only invoke it if
|
||||
* @rq->next_class is below &ext_sched_class, it will also
|
||||
* resched_curr(rq).
|
||||
*/
|
||||
if (sched_class_above(p->sched_class, rq->next_class))
|
||||
wakeup_preempt(rq, p, 0);
|
||||
|
||||
/*
|
||||
* If @rq is in balance, the CPU is already vacant and looking for the
|
||||
* next task to run. No need to preempt or trigger resched after moving
|
||||
* @p into its local DSQ.
|
||||
* Note that the wakeup_preempt() above may have already triggered
|
||||
* a resched if @rq->next_class was idle. It's harmless, since
|
||||
* need_resched is cleared immediately after task pick.
|
||||
*/
|
||||
if (rq->scx.flags & SCX_RQ_IN_BALANCE)
|
||||
return;
|
||||
|
|
@ -1525,11 +1619,8 @@ static void local_dsq_post_enq(struct scx_sched *sch, struct scx_dispatch_q *dsq
|
|||
if ((enq_flags & SCX_ENQ_PREEMPT) && p != rq->curr &&
|
||||
rq->curr->sched_class == &ext_sched_class) {
|
||||
rq->curr->scx.slice = 0;
|
||||
preempt = true;
|
||||
}
|
||||
|
||||
if (preempt || sched_class_above(&ext_sched_class, rq->curr->sched_class))
|
||||
resched_curr(rq);
|
||||
}
|
||||
}
|
||||
|
||||
static void dispatch_enqueue(struct scx_sched *sch, struct rq *rq,
|
||||
|
|
@ -3566,41 +3657,6 @@ static struct cgroup *tg_cgrp(struct task_group *tg)
|
|||
|
||||
#endif /* CONFIG_EXT_GROUP_SCHED */
|
||||
|
||||
static u32 scx_get_task_state(const struct task_struct *p)
|
||||
{
|
||||
return p->scx.flags & SCX_TASK_STATE_MASK;
|
||||
}
|
||||
|
||||
static void scx_set_task_state(struct task_struct *p, u32 state)
|
||||
{
|
||||
u32 prev_state = scx_get_task_state(p);
|
||||
bool warn = false;
|
||||
|
||||
switch (state) {
|
||||
case SCX_TASK_NONE:
|
||||
break;
|
||||
case SCX_TASK_INIT:
|
||||
warn = prev_state != SCX_TASK_NONE;
|
||||
break;
|
||||
case SCX_TASK_READY:
|
||||
warn = prev_state == SCX_TASK_NONE;
|
||||
break;
|
||||
case SCX_TASK_ENABLED:
|
||||
warn = prev_state != SCX_TASK_READY;
|
||||
break;
|
||||
default:
|
||||
WARN_ONCE(1, "sched_ext: Invalid task state %d -> %d for %s[%d]",
|
||||
prev_state, state, p->comm, p->pid);
|
||||
return;
|
||||
}
|
||||
|
||||
WARN_ONCE(warn, "sched_ext: Invalid task state transition 0x%x -> 0x%x for %s[%d]",
|
||||
prev_state, state, p->comm, p->pid);
|
||||
|
||||
p->scx.flags &= ~SCX_TASK_STATE_MASK;
|
||||
p->scx.flags |= state;
|
||||
}
|
||||
|
||||
static int __scx_init_task(struct scx_sched *sch, struct task_struct *p, bool fork)
|
||||
{
|
||||
int ret;
|
||||
|
|
@ -3652,22 +3708,6 @@ static int __scx_init_task(struct scx_sched *sch, struct task_struct *p, bool fo
|
|||
return 0;
|
||||
}
|
||||
|
||||
static int scx_init_task(struct scx_sched *sch, struct task_struct *p, bool fork)
|
||||
{
|
||||
int ret;
|
||||
|
||||
ret = __scx_init_task(sch, p, fork);
|
||||
if (!ret) {
|
||||
/*
|
||||
* While @p's rq is not locked. @p is not visible to the rest of
|
||||
* SCX yet and it's safe to update the flags and state.
|
||||
*/
|
||||
p->scx.flags |= SCX_TASK_RESET_RUNNABLE_AT;
|
||||
scx_set_task_state(p, SCX_TASK_INIT);
|
||||
}
|
||||
return ret;
|
||||
}
|
||||
|
||||
static void __scx_enable_task(struct scx_sched *sch, struct task_struct *p)
|
||||
{
|
||||
struct rq *rq = task_rq(p);
|
||||
|
|
@ -3776,6 +3816,15 @@ static void scx_sub_init_cancel_task(struct scx_sched *sch, struct task_struct *
|
|||
static void scx_disable_and_exit_task(struct scx_sched *sch,
|
||||
struct task_struct *p)
|
||||
{
|
||||
/*
|
||||
* %NONE means @p is already detached at the SCX level (e.g. handed
|
||||
* back to the parent by scx_fail_parent() with no init to undo).
|
||||
* Skip to avoid clobbering scx_task_sched() and writing %NONE again
|
||||
* on a state that's already %NONE.
|
||||
*/
|
||||
if (scx_get_task_state(p) == SCX_TASK_NONE)
|
||||
return;
|
||||
|
||||
__scx_disable_and_exit_task(sch, p);
|
||||
|
||||
/*
|
||||
|
|
@ -3859,10 +3908,14 @@ int scx_fork(struct task_struct *p, struct kernel_clone_args *kargs)
|
|||
#else
|
||||
struct scx_sched *sch = scx_root;
|
||||
#endif
|
||||
ret = scx_init_task(sch, p, true);
|
||||
if (!ret)
|
||||
scx_set_task_sched(p, sch);
|
||||
return ret;
|
||||
scx_set_task_state(p, SCX_TASK_INIT_BEGIN);
|
||||
ret = __scx_init_task(sch, p, true);
|
||||
if (unlikely(ret)) {
|
||||
scx_set_task_state(p, SCX_TASK_NONE);
|
||||
return ret;
|
||||
}
|
||||
scx_set_task_state(p, SCX_TASK_INIT);
|
||||
scx_set_task_sched(p, sch);
|
||||
}
|
||||
|
||||
return 0;
|
||||
|
|
@ -3960,13 +4013,24 @@ void sched_ext_dead(struct task_struct *p)
|
|||
/*
|
||||
* @p is off scx_tasks and wholly ours. scx_root_enable()'s READY ->
|
||||
* ENABLED transitions can't race us. Disable ops for @p.
|
||||
*
|
||||
* %SCX_TASK_DEAD synchronizes against cgroup task iteration - see
|
||||
* scx_task_iter_next_locked(). NONE tasks need no marking: cgroup
|
||||
* iteration is only used from sub-sched paths, which require root
|
||||
* enabled. Root enable transitions every live task to at least READY.
|
||||
*
|
||||
* %INIT_BEGIN means ops.init_task() is running for @p. Don't call
|
||||
* into ops; transition to %DEAD so the post-init recheck unwinds
|
||||
* via scx_sub_init_cancel_task().
|
||||
*/
|
||||
if (scx_get_task_state(p) != SCX_TASK_NONE) {
|
||||
struct rq_flags rf;
|
||||
struct rq *rq;
|
||||
|
||||
rq = task_rq_lock(p, &rf);
|
||||
scx_disable_and_exit_task(scx_task_sched(p), p);
|
||||
if (scx_get_task_state(p) != SCX_TASK_INIT_BEGIN)
|
||||
scx_disable_and_exit_task(scx_task_sched(p), p);
|
||||
scx_set_task_state(p, SCX_TASK_DEAD);
|
||||
task_rq_unlock(rq, p, &rf);
|
||||
}
|
||||
}
|
||||
|
|
@ -4012,6 +4076,16 @@ static void switched_from_scx(struct rq *rq, struct task_struct *p)
|
|||
if (task_dead_and_done(p))
|
||||
return;
|
||||
|
||||
/*
|
||||
* %NONE means SCX is no longer tracking @p at the task level (e.g.
|
||||
* scx_fail_parent() handed @p back to the parent at NONE pending the
|
||||
* parent's own teardown). There is nothing to disable; calling
|
||||
* scx_disable_task() would WARN on the non-%ENABLED state and trigger a
|
||||
* NONE -> READY validation failure.
|
||||
*/
|
||||
if (scx_get_task_state(p) == SCX_TASK_NONE)
|
||||
return;
|
||||
|
||||
scx_disable_task(scx_task_sched(p), p);
|
||||
}
|
||||
|
||||
|
|
@ -5679,10 +5753,12 @@ static void refresh_watchdog(void)
|
|||
|
||||
static s32 scx_link_sched(struct scx_sched *sch)
|
||||
{
|
||||
const char *err_msg = "";
|
||||
s32 ret = 0;
|
||||
|
||||
scoped_guard(raw_spinlock_irq, &scx_sched_lock) {
|
||||
#ifdef CONFIG_EXT_SUB_SCHED
|
||||
struct scx_sched *parent = scx_parent(sch);
|
||||
s32 ret;
|
||||
|
||||
if (parent) {
|
||||
/*
|
||||
|
|
@ -5692,15 +5768,16 @@ static s32 scx_link_sched(struct scx_sched *sch)
|
|||
* parent can shoot us down.
|
||||
*/
|
||||
if (atomic_read(&parent->exit_kind) != SCX_EXIT_NONE) {
|
||||
scx_error(sch, "parent disabled");
|
||||
return -ENOENT;
|
||||
err_msg = "parent disabled";
|
||||
ret = -ENOENT;
|
||||
break;
|
||||
}
|
||||
|
||||
ret = rhashtable_lookup_insert_fast(&scx_sched_hash,
|
||||
&sch->hash_node, scx_sched_hash_params);
|
||||
if (ret) {
|
||||
scx_error(sch, "failed to insert into scx_sched_hash (%d)", ret);
|
||||
return ret;
|
||||
err_msg = "failed to insert into scx_sched_hash";
|
||||
break;
|
||||
}
|
||||
|
||||
list_add_tail(&sch->sibling, &parent->children);
|
||||
|
|
@ -5710,6 +5787,15 @@ static s32 scx_link_sched(struct scx_sched *sch)
|
|||
list_add_tail_rcu(&sch->all, &scx_sched_all);
|
||||
}
|
||||
|
||||
/*
|
||||
* scx_error() takes scx_sched_lock via scx_claim_exit(), so it must run after
|
||||
* the guard above is released.
|
||||
*/
|
||||
if (ret) {
|
||||
scx_error(sch, "%s (%d)", err_msg, ret);
|
||||
return ret;
|
||||
}
|
||||
|
||||
refresh_watchdog();
|
||||
return 0;
|
||||
}
|
||||
|
|
@ -5799,7 +5885,7 @@ static void scx_fail_parent(struct scx_sched *sch,
|
|||
|
||||
scoped_guard (sched_change, p, DEQUEUE_SAVE | DEQUEUE_MOVE) {
|
||||
scx_disable_and_exit_task(sch, p);
|
||||
rcu_assign_pointer(p->scx.sched, parent);
|
||||
scx_set_task_sched(p, parent);
|
||||
}
|
||||
}
|
||||
scx_task_iter_stop(&sti);
|
||||
|
|
@ -5877,6 +5963,21 @@ static void scx_sub_disable(struct scx_sched *sch)
|
|||
}
|
||||
|
||||
rq = task_rq_lock(p, &rf);
|
||||
|
||||
if (scx_get_task_state(p) == SCX_TASK_DEAD) {
|
||||
/*
|
||||
* sched_ext_dead() raced us between __scx_init_task()
|
||||
* and this rq lock and ran exit_task() on @sch (the
|
||||
* sched @p was on at that point), not on $parent.
|
||||
* $parent's just-completed init is owed an exit_task()
|
||||
* and we issue it here.
|
||||
*/
|
||||
scx_sub_init_cancel_task(parent, p);
|
||||
task_rq_unlock(rq, p, &rf);
|
||||
put_task_struct(p);
|
||||
continue;
|
||||
}
|
||||
|
||||
scoped_guard (sched_change, p, DEQUEUE_SAVE | DEQUEUE_MOVE) {
|
||||
/*
|
||||
* $p is initialized for $parent and still attached to
|
||||
|
|
@ -5885,13 +5986,14 @@ static void scx_sub_disable(struct scx_sched *sch)
|
|||
* $p having already been initialized, and then enable.
|
||||
*/
|
||||
scx_disable_and_exit_task(sch, p);
|
||||
scx_set_task_state(p, SCX_TASK_INIT_BEGIN);
|
||||
scx_set_task_state(p, SCX_TASK_INIT);
|
||||
rcu_assign_pointer(p->scx.sched, parent);
|
||||
scx_set_task_sched(p, parent);
|
||||
scx_set_task_state(p, SCX_TASK_READY);
|
||||
scx_enable_task(parent, p);
|
||||
}
|
||||
task_rq_unlock(rq, p, &rf);
|
||||
|
||||
task_rq_unlock(rq, p, &rf);
|
||||
put_task_struct(p);
|
||||
}
|
||||
scx_task_iter_stop(&sti);
|
||||
|
|
@ -6166,8 +6268,13 @@ static void scx_disable(struct scx_sched *sch, enum scx_exit_kind kind)
|
|||
*/
|
||||
static void scx_flush_disable_work(struct scx_sched *sch)
|
||||
{
|
||||
irq_work_sync(&sch->disable_irq_work);
|
||||
kthread_flush_work(&sch->disable_work);
|
||||
int kind;
|
||||
|
||||
do {
|
||||
irq_work_sync(&sch->disable_irq_work);
|
||||
kthread_flush_work(&sch->disable_work);
|
||||
kind = atomic_read(&sch->exit_kind);
|
||||
} while (kind != SCX_EXIT_NONE && kind != SCX_EXIT_DONE);
|
||||
}
|
||||
|
||||
static void dump_newline(struct seq_buf *s)
|
||||
|
|
@ -6719,7 +6826,7 @@ static struct scx_sched *scx_alloc_and_add_sched(struct scx_enable_cmd *cmd,
|
|||
|
||||
sch->slice_dfl = SCX_SLICE_DFL;
|
||||
atomic_set(&sch->exit_kind, SCX_EXIT_NONE);
|
||||
init_irq_work(&sch->disable_irq_work, scx_disable_irq_workfn);
|
||||
sch->disable_irq_work = IRQ_WORK_INIT_HARD(scx_disable_irq_workfn);
|
||||
kthread_init_work(&sch->disable_work, scx_disable_workfn);
|
||||
timer_setup(&sch->bypass_lb_timer, scx_bypass_lb_timerfn, 0);
|
||||
|
||||
|
|
@ -6793,8 +6900,10 @@ static struct scx_sched *scx_alloc_and_add_sched(struct scx_enable_cmd *cmd,
|
|||
#endif /* CONFIG_EXT_SUB_SCHED */
|
||||
return sch;
|
||||
|
||||
#ifdef CONFIG_EXT_SUB_SCHED
|
||||
err_free_lb_resched:
|
||||
free_cpumask_var(sch->bypass_lb_resched_cpumask);
|
||||
#endif
|
||||
err_free_lb_cpumask:
|
||||
free_cpumask_var(sch->bypass_lb_donee_cpumask);
|
||||
err_stop_helper:
|
||||
|
|
@ -7079,13 +7188,25 @@ static void scx_root_enable_workfn(struct kthread_work *work)
|
|||
if (!tryget_task_struct(p))
|
||||
continue;
|
||||
|
||||
/*
|
||||
* Set %INIT_BEGIN under the iter's rq lock so that a concurrent
|
||||
* sched_ext_dead() does not call ops.exit_task() on @p while
|
||||
* ops.init_task() is running. If sched_ext_dead() runs before
|
||||
* this store, it has already removed @p from scx_tasks and the
|
||||
* iter won't visit @p; if it runs after, it observes
|
||||
* %INIT_BEGIN and transitions to %DEAD without calling ops,
|
||||
* leaving the post-init recheck below to unwind.
|
||||
*/
|
||||
scx_set_task_state(p, SCX_TASK_INIT_BEGIN);
|
||||
scx_task_iter_unlock(&sti);
|
||||
|
||||
ret = scx_init_task(sch, p, false);
|
||||
ret = __scx_init_task(sch, p, false);
|
||||
|
||||
scx_task_iter_relock(&sti, p);
|
||||
|
||||
if (ret) {
|
||||
if (unlikely(ret)) {
|
||||
if (scx_get_task_state(p) != SCX_TASK_DEAD)
|
||||
scx_set_task_state(p, SCX_TASK_NONE);
|
||||
put_task_struct(p);
|
||||
scx_task_iter_stop(&sti);
|
||||
scx_error(sch, "ops.init_task() failed (%d) for %s[%d]",
|
||||
|
|
@ -7093,8 +7214,18 @@ static void scx_root_enable_workfn(struct kthread_work *work)
|
|||
goto err_disable_unlock_all;
|
||||
}
|
||||
|
||||
scx_set_task_sched(p, sch);
|
||||
scx_set_task_state(p, SCX_TASK_READY);
|
||||
if (scx_get_task_state(p) == SCX_TASK_DEAD) {
|
||||
/*
|
||||
* sched_ext_dead() observed %INIT_BEGIN and set %DEAD.
|
||||
* ops.exit_task() is owed to the sched __scx_init_task()
|
||||
* ran against; call it now.
|
||||
*/
|
||||
scx_sub_init_cancel_task(sch, p);
|
||||
} else {
|
||||
scx_set_task_state(p, SCX_TASK_INIT);
|
||||
scx_set_task_sched(p, sch);
|
||||
scx_set_task_state(p, SCX_TASK_READY);
|
||||
}
|
||||
|
||||
/*
|
||||
* Insert into the tid hash. scx_tasks_lock is held by the iter;
|
||||
|
|
@ -7376,6 +7507,21 @@ static void scx_sub_enable_workfn(struct kthread_work *work)
|
|||
goto abort;
|
||||
|
||||
rq = task_rq_lock(p, &rf);
|
||||
|
||||
if (scx_get_task_state(p) == SCX_TASK_DEAD) {
|
||||
/*
|
||||
* sched_ext_dead() raced us between __scx_init_task()
|
||||
* and this rq lock and ran exit_task() on $parent (the
|
||||
* sched @p was on at that point), not on @sch. @sch's
|
||||
* just-completed init is owed an exit_task() and we
|
||||
* issue it here.
|
||||
*/
|
||||
scx_sub_init_cancel_task(sch, p);
|
||||
task_rq_unlock(rq, p, &rf);
|
||||
put_task_struct(p);
|
||||
continue;
|
||||
}
|
||||
|
||||
p->scx.flags |= SCX_TASK_SUB_INIT;
|
||||
task_rq_unlock(rq, p, &rf);
|
||||
|
||||
|
|
@ -7410,7 +7556,7 @@ static void scx_sub_enable_workfn(struct kthread_work *work)
|
|||
* $p is now only initialized for @sch and READY, which
|
||||
* is what we want. Assign it to @sch and enable.
|
||||
*/
|
||||
rcu_assign_pointer(p->scx.sched, sch);
|
||||
scx_set_task_sched(p, sch);
|
||||
scx_enable_task(sch, p);
|
||||
|
||||
p->scx.flags &= ~SCX_TASK_SUB_INIT;
|
||||
|
|
|
|||
|
|
@ -464,12 +464,6 @@ s32 scx_select_cpu_dfl(struct task_struct *p, s32 prev_cpu, u64 wake_flags,
|
|||
|
||||
preempt_disable();
|
||||
|
||||
/*
|
||||
* Check whether @prev_cpu is still within the allowed set. If not,
|
||||
* we can still try selecting a nearby CPU.
|
||||
*/
|
||||
is_prev_allowed = cpumask_test_cpu(prev_cpu, allowed);
|
||||
|
||||
/*
|
||||
* Determine the subset of CPUs usable by @p within @cpus_allowed.
|
||||
*/
|
||||
|
|
@ -486,6 +480,12 @@ s32 scx_select_cpu_dfl(struct task_struct *p, s32 prev_cpu, u64 wake_flags,
|
|||
}
|
||||
}
|
||||
|
||||
/*
|
||||
* Check whether @prev_cpu is still within the allowed set. If not,
|
||||
* we can still try selecting a nearby CPU.
|
||||
*/
|
||||
is_prev_allowed = cpumask_test_cpu(prev_cpu, allowed);
|
||||
|
||||
/*
|
||||
* This is necessary to protect llc_cpus.
|
||||
*/
|
||||
|
|
|
|||
Loading…
Reference in New Issue
Block a user