sched_ext: Source tree reorganization for v7.2

Follow-up to the v7.2 sched_ext pull. Pure source reorganization with no
 functional change: the kernel/sched/ext* files move into a new
 kernel/sched/ext/ subdirectory, and the headers and sources are made
 self-contained so editor tooling can parse each file on its own.
 -----BEGIN PGP SIGNATURE-----
 
 iIQEABYKACwWIQTfIjM1kS57o3GsC/uxYfJx3gVYGQUCajmn3w4cdGpAa2VybmVs
 Lm9yZwAKCRCxYfJx3gVYGYRyAP4qs7Xlkva8ppSiJX/nja4OpQixpCZ1zi1yPDGK
 iaJ+tQEA+bnkb2QI0Cwd9l+kY12/kl8gHUfqG0UPVyyEOvAeXgs=
 =MvGz
 -----END PGP SIGNATURE-----

Merge tag 'sched_ext-for-7.2-1' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext

Pull sched_ext tree reorg from Tejun Heo:
 "Pure source reorganization with no functional change:

   - the kernel/sched/ext* files move into a new kernel/sched/ext/
     subdirectory

   - the headers and sources are made self-contained so editor tooling
     can parse each file on its own"

* tag 'sched_ext-for-7.2-1' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext:
  sched_ext: Move shared helpers from ext.c into internal.h and cid.h
  sched_ext: Make kernel/sched/ext/ sources self-contained for clangd
  sched_ext: Move sources under kernel/sched/ext/
This commit is contained in:
Linus Torvalds 2026-06-23 13:36:09 -07:00
commit 7603d8e780
15 changed files with 209 additions and 163 deletions

View File

@ -114,7 +114,7 @@ counters. Each counter occupies one ``name value`` line:
SCX_EV_INSERT_NOT_OWNED 0
SCX_EV_SUB_BYPASS_DISPATCH 0
The counters are described in ``kernel/sched/ext_internal.h``; briefly:
The counters are described in ``kernel/sched/ext/internal.h``; briefly:
* ``SCX_EV_SELECT_CPU_FALLBACK``: ops.select_cpu() returned a CPU unusable by
the task and the core scheduler silently picked a fallback CPU.
@ -496,11 +496,11 @@ Where to Look
* ``include/linux/sched/ext.h`` defines the core data structures, ops table
and constants.
* ``kernel/sched/ext.c`` contains sched_ext core implementation and helpers.
* ``kernel/sched/ext/ext.c`` contains sched_ext core implementation and helpers.
The functions prefixed with ``scx_bpf_`` can be called from the BPF
scheduler.
* ``kernel/sched/ext_idle.c`` contains the built-in idle CPU selection policy.
* ``kernel/sched/ext/idle.c`` contains the built-in idle CPU selection policy.
* ``tools/sched_ext/`` hosts example BPF scheduler implementations.
@ -557,7 +557,7 @@ ABI Instability
The APIs provided by sched_ext to BPF schedulers programs have no stability
guarantees. This includes the ops table callbacks and constants defined in
``include/linux/sched/ext.h``, as well as the ``scx_bpf_`` kfuncs defined in
``kernel/sched/ext.c`` and ``kernel/sched/ext_idle.c``.
``kernel/sched/ext/ext.c`` and ``kernel/sched/ext/idle.c``.
While we will attempt to provide a relatively stable API surface when
possible, they are subject to change without warning between kernel

View File

@ -24204,7 +24204,7 @@ S: Maintained
W: https://github.com/sched-ext/scx
T: git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext.git
F: include/linux/sched/ext.h
F: kernel/sched/ext*
F: kernel/sched/ext/
F: tools/sched_ext/
F: tools/testing/selftests/sched_ext

View File

@ -61,15 +61,15 @@
# include <linux/btf_ids.h>
# include <linux/find.h>
# include <linux/genalloc.h>
# include "ext_types.h"
# include "ext_internal.h"
# include "ext_cid.h"
# include "ext_arena.h"
# include "ext_idle.h"
# include "ext.c"
# include "ext_cid.c"
# include "ext_arena.c"
# include "ext_idle.c"
# include "ext/types.h"
# include "ext/internal.h"
# include "ext/cid.h"
# include "ext/arena.h"
# include "ext/idle.h"
# include "ext/ext.c"
# include "ext/cid.c"
# include "ext/arena.c"
# include "ext/idle.c"
#endif
#include "syscalls.c"

View File

@ -15,6 +15,10 @@
* Copyright (c) 2026 Meta Platforms, Inc. and affiliates.
* Copyright (c) 2026 Tejun Heo <tj@kernel.org>
*/
#include <linux/genalloc.h>
#include "internal.h"
#include "arena.h"
enum scx_arena_consts {
SCX_ARENA_MIN_ORDER = 3, /* 8-byte minimum sub-allocation */

View File

@ -8,6 +8,8 @@
#ifndef _KERNEL_SCHED_EXT_ARENA_H
#define _KERNEL_SCHED_EXT_ARENA_H
#include <linux/types.h>
struct scx_sched;
s32 scx_arena_pool_init(struct scx_sched *sch);

View File

@ -7,6 +7,9 @@
*/
#include <linux/cacheinfo.h>
#include "internal.h"
#include "cid.h"
/*
* cid tables.
*
@ -71,7 +74,7 @@ static s32 scx_cid_arrays_alloc(void)
* scx_cid_init - build the cid mapping
* @sch: the scx_sched being initialized; used as the scx_error() target
*
* See "Topological CPU IDs" in ext_cid.h for the model. Walk online cpus by
* See "Topological CPU IDs" in cid.h for the model. Walk online cpus by
* intersection at each level (parent_scratch & this_level_mask), which keeps
* containment correct by construction and naturally splits a physical LLC
* straddling two NUMA nodes into two LLC units. The caller must hold

View File

@ -33,6 +33,8 @@
#ifndef _KERNEL_SCHED_EXT_CID_H
#define _KERNEL_SCHED_EXT_CID_H
#include "internal.h"
struct scx_sched;
/*
@ -43,7 +45,7 @@ struct scx_sched;
* possible-but-not-online cpus and carries all-(-1) topo info (see
* scx_cid_topo); callers detect it via the -1 sentinels.
*
* See the comment above the table definitions in ext_cid.c for the
* See the comment above the table definitions in cid.c for the
* memory-ordering and visibility contract.
*/
extern s16 *scx_cid_to_cpu_tbl;
@ -268,4 +270,25 @@ static inline u32 scx_cmask_nr_used_words(const struct scx_cmask *m)
__w && ((cid) = __bs + __wi * 64 + __ffs64(__w), true); \
__w &= __w - 1)
/*
* scx_cpu_arg() wraps a cpu arg being handed to an SCX op. For cid-form
* schedulers it resolves to the matching cid; for cpu-form it passes @cpu
* through. scx_cpu_ret() is the inverse for a cpu/cid returned from an op
* (currently only ops.select_cpu); it validates the BPF-supplied cid and
* triggers scx_error() on @sch if invalid.
*/
static inline s32 scx_cpu_arg(s32 cpu)
{
if (scx_is_cid_type())
return __scx_cpu_to_cid(cpu);
return cpu;
}
static inline s32 scx_cpu_ret(struct scx_sched *sch, s32 cpu_or_cid)
{
if (cpu_or_cid < 0 || !scx_is_cid_type())
return cpu_or_cid;
return scx_cid_to_cpu(sch, cpu_or_cid);
}
#endif /* _KERNEL_SCHED_EXT_CID_H */

View File

@ -6,6 +6,19 @@
* Copyright (c) 2022 Tejun Heo <tj@kernel.org>
* Copyright (c) 2022 David Vernet <dvernet@meta.com>
*/
#include <linux/bitmap.h>
#include <linux/btf_ids.h>
#include <linux/rhashtable.h>
#include <linux/sched/clock.h>
#include <linux/sched/isolation.h>
#include <linux/suspend.h>
#include <linux/sysrq.h>
#include "../pelt.h"
#include "internal.h"
#include "cid.h"
#include "arena.h"
#include "idle.h"
static DEFINE_RAW_SPINLOCK(scx_sched_lock);
@ -246,8 +259,6 @@ __printf(5, 6) bool __scx_exit(struct scx_sched *sch,
return ret;
}
#define SCX_HAS_OP(sch, op) test_bit(SCX_OP_IDX(op), (sch)->has_op)
static long jiffies_delta_msecs(unsigned long at, unsigned long now)
{
if (time_after(at, now))
@ -262,20 +273,6 @@ static bool u32_before(u32 a, u32 b)
}
#ifdef CONFIG_EXT_SUB_SCHED
/**
* scx_parent - Find the parent sched
* @sch: sched to find the parent of
*
* Returns the parent scheduler or %NULL if @sch is root.
*/
static struct scx_sched *scx_parent(struct scx_sched *sch)
{
if (sch->level)
return sch->ancestors[sch->level - 1];
else
return NULL;
}
/**
* scx_next_descendant_pre - find the next descendant for pre-order walk
* @pos: the current position (%NULL to initiate traversal)
@ -323,7 +320,6 @@ static void scx_set_task_sched(struct task_struct *p, struct scx_sched *sch)
rcu_assign_pointer(p->scx.sched, sch);
}
#else /* CONFIG_EXT_SUB_SCHED */
static inline struct scx_sched *scx_parent(struct scx_sched *sch) { return NULL; }
static inline struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sched *root) { return pos ? NULL : root; }
static inline void scx_set_task_sched(struct task_struct *p, struct scx_sched *sch) {}
#endif /* CONFIG_EXT_SUB_SCHED */
@ -483,123 +479,12 @@ static bool rq_is_open(struct rq *rq, u64 enq_flags)
*/
DEFINE_PER_CPU(struct rq *, scx_locked_rq_state);
static inline void update_locked_rq(struct rq *rq)
{
/*
* Check whether @rq is actually locked. This can help expose bugs
* or incorrect assumptions about the context in which a kfunc or
* callback is executed.
*/
if (rq)
lockdep_assert_rq_held(rq);
__this_cpu_write(scx_locked_rq_state, rq);
}
/*
* SCX ops can recurse via scx_bpf_sub_dispatch() - the inner call must not
* clobber the outer's scx_locked_rq_state. Save it on entry, restore on exit.
*/
#define SCX_CALL_OP(sch, op, locked_rq, args...) \
do { \
struct rq *__prev_locked_rq; \
\
if (locked_rq) { \
__prev_locked_rq = scx_locked_rq(); \
update_locked_rq(locked_rq); \
} \
(sch)->ops.op(args); \
if (locked_rq) \
update_locked_rq(__prev_locked_rq); \
} while (0)
/*
* Flipped on enable per sch->is_cid_type. Declared in ext_internal.h so
* Flipped on enable per sch->is_cid_type. Declared in internal.h so
* subsystem inlines can read it.
*/
DEFINE_STATIC_KEY_FALSE(__scx_is_cid_type);
/*
* scx_cpu_arg() wraps a cpu arg being handed to an SCX op. For cid-form
* schedulers it resolves to the matching cid; for cpu-form it passes @cpu
* through. scx_cpu_ret() is the inverse for a cpu/cid returned from an op
* (currently only ops.select_cpu); it validates the BPF-supplied cid and
* triggers scx_error() on @sch if invalid.
*/
static s32 scx_cpu_arg(s32 cpu)
{
if (scx_is_cid_type())
return __scx_cpu_to_cid(cpu);
return cpu;
}
static s32 scx_cpu_ret(struct scx_sched *sch, s32 cpu_or_cid)
{
if (cpu_or_cid < 0 || !scx_is_cid_type())
return cpu_or_cid;
return scx_cid_to_cpu(sch, cpu_or_cid);
}
#define SCX_CALL_OP_RET(sch, op, locked_rq, args...) \
({ \
struct rq *__prev_locked_rq; \
__typeof__((sch)->ops.op(args)) __ret; \
\
if (locked_rq) { \
__prev_locked_rq = scx_locked_rq(); \
update_locked_rq(locked_rq); \
} \
__ret = (sch)->ops.op(args); \
if (locked_rq) \
update_locked_rq(__prev_locked_rq); \
__ret; \
})
/*
* SCX_CALL_OP_TASK*() invokes an SCX op that takes one or two task arguments
* and records them in current->scx.kf_tasks[] for the duration of the call. A
* kfunc invoked from inside such an op can then use
* scx_kf_arg_task_ok() to verify that its task argument is one of
* those subject tasks.
*
* Every SCX_CALL_OP_TASK*() call site invokes its op with @p's rq lock held -
* either via the @locked_rq argument here, or (for ops.select_cpu()) via @p's
* pi_lock held by try_to_wake_up() with rq tracking via scx_rq.in_select_cpu.
* So if kf_tasks[] is set, @p's scheduler-protected fields are stable.
*
* kf_tasks[] can not stack, so task-based SCX ops must not nest. The
* WARN_ON_ONCE() in each macro catches a re-entry of any of the three variants
* while a previous one is still in progress.
*/
#define SCX_CALL_OP_TASK(sch, op, locked_rq, task, args...) \
do { \
WARN_ON_ONCE(current->scx.kf_tasks[0]); \
current->scx.kf_tasks[0] = task; \
SCX_CALL_OP((sch), op, locked_rq, task, ##args); \
current->scx.kf_tasks[0] = NULL; \
} while (0)
#define SCX_CALL_OP_TASK_RET(sch, op, locked_rq, task, args...) \
({ \
__typeof__((sch)->ops.op(task, ##args)) __ret; \
WARN_ON_ONCE(current->scx.kf_tasks[0]); \
current->scx.kf_tasks[0] = task; \
__ret = SCX_CALL_OP_RET((sch), op, locked_rq, task, ##args); \
current->scx.kf_tasks[0] = NULL; \
__ret; \
})
#define SCX_CALL_OP_2TASKS_RET(sch, op, locked_rq, task0, task1, args...) \
({ \
__typeof__((sch)->ops.op(task0, task1, ##args)) __ret; \
WARN_ON_ONCE(current->scx.kf_tasks[0]); \
current->scx.kf_tasks[0] = task0; \
current->scx.kf_tasks[1] = task1; \
__ret = SCX_CALL_OP_RET((sch), op, locked_rq, task0, task1, ##args); \
current->scx.kf_tasks[0] = NULL; \
current->scx.kf_tasks[1] = NULL; \
__ret; \
})
/**
* scx_call_op_set_cpumask - invoke ops.set_cpumask / ops_cid.set_cmask for @task
* @sch: scx_sched being invoked
@ -608,7 +493,7 @@ do { \
* @cpumask: new cpumask
*
* For cid-form schedulers, translate @cpumask to a cmask via the per-cpu
* scratch in ext_cid.c and dispatch through the ops_cid union view. Caller
* scratch in cid.c and dispatch through the ops_cid union view. Caller
* must hold @rq's rq lock so this_cpu_ptr is stable across the call.
*/
static inline void scx_call_op_set_cpumask(struct scx_sched *sch, struct rq *rq,
@ -638,19 +523,6 @@ static inline void scx_call_op_set_cpumask(struct scx_sched *sch, struct rq *rq,
current->scx.kf_tasks[0] = NULL;
}
/* see SCX_CALL_OP_TASK() */
static __always_inline bool scx_kf_arg_task_ok(struct scx_sched *sch,
struct task_struct *p)
{
if (unlikely((p != current->scx.kf_tasks[0] &&
p != current->scx.kf_tasks[1]))) {
scx_error(sch, "called on a task not being operated on");
return false;
}
return true;
}
enum scx_dsq_iter_flags {
/* iterate in the reverse dispatch order */
SCX_DSQ_ITER_REV = 1U << 16,

View File

@ -9,6 +9,9 @@
* Copyright (c) 2022 David Vernet <dvernet@meta.com>
* Copyright (c) 2024 Andrea Righi <arighi@nvidia.com>
*/
#include "internal.h"
#include "cid.h"
#include "idle.h"
/* Enable/disable built-in idle CPU selection policy */
static DEFINE_STATIC_KEY_FALSE(scx_builtin_idle_enabled);

View File

@ -10,7 +10,11 @@
#ifndef _KERNEL_SCHED_EXT_IDLE_H
#define _KERNEL_SCHED_EXT_IDLE_H
#include <linux/btf_ids.h>
struct cpumask;
struct sched_ext_ops;
struct task_struct;
extern struct btf_id_set8 scx_kfunc_ids_idle;
extern struct btf_id_set8 scx_kfunc_ids_select_cpu;

View File

@ -5,6 +5,12 @@
* Copyright (c) 2025 Meta Platforms, Inc. and affiliates.
* Copyright (c) 2025 Tejun Heo <tj@kernel.org>
*/
#ifndef _KERNEL_SCHED_EXT_INTERNAL_H
#define _KERNEL_SCHED_EXT_INTERNAL_H
#include "../sched.h"
#include "types.h"
#define SCX_OP_IDX(op) (offsetof(struct sched_ext_ops, op) / sizeof(void (*)(void)))
#define SCX_MOFF_IDX(moff) ((moff) / sizeof(void (*)(void)))
@ -1547,6 +1553,111 @@ static inline struct rq *scx_locked_rq(void)
return __this_cpu_read(scx_locked_rq_state);
}
static inline void update_locked_rq(struct rq *rq)
{
/*
* Check whether @rq is actually locked. This can help expose bugs
* or incorrect assumptions about the context in which a kfunc or
* callback is executed.
*/
if (rq)
lockdep_assert_rq_held(rq);
__this_cpu_write(scx_locked_rq_state, rq);
}
#define SCX_HAS_OP(sch, op) test_bit(SCX_OP_IDX(op), (sch)->has_op)
/*
* SCX ops can recurse via scx_bpf_sub_dispatch() - the inner call must not
* clobber the outer's scx_locked_rq_state. Save it on entry, restore on exit.
*/
#define SCX_CALL_OP(sch, op, locked_rq, args...) \
do { \
struct rq *__prev_locked_rq; \
\
if (locked_rq) { \
__prev_locked_rq = scx_locked_rq(); \
update_locked_rq(locked_rq); \
} \
(sch)->ops.op(args); \
if (locked_rq) \
update_locked_rq(__prev_locked_rq); \
} while (0)
#define SCX_CALL_OP_RET(sch, op, locked_rq, args...) \
({ \
struct rq *__prev_locked_rq; \
__typeof__((sch)->ops.op(args)) __ret; \
\
if (locked_rq) { \
__prev_locked_rq = scx_locked_rq(); \
update_locked_rq(locked_rq); \
} \
__ret = (sch)->ops.op(args); \
if (locked_rq) \
update_locked_rq(__prev_locked_rq); \
__ret; \
})
/*
* SCX_CALL_OP_TASK*() invokes an SCX op that takes one or two task arguments
* and records them in current->scx.kf_tasks[] for the duration of the call. A
* kfunc invoked from inside such an op can then use
* scx_kf_arg_task_ok() to verify that its task argument is one of
* those subject tasks.
*
* Every SCX_CALL_OP_TASK*() call site invokes its op with @p's rq lock held -
* either via the @locked_rq argument here, or (for ops.select_cpu()) via @p's
* pi_lock held by try_to_wake_up() with rq tracking via scx_rq.in_select_cpu.
* So if kf_tasks[] is set, @p's scheduler-protected fields are stable.
*
* kf_tasks[] can not stack, so task-based SCX ops must not nest. The
* WARN_ON_ONCE() in each macro catches a re-entry of any of the three variants
* while a previous one is still in progress.
*/
#define SCX_CALL_OP_TASK(sch, op, locked_rq, task, args...) \
do { \
WARN_ON_ONCE(current->scx.kf_tasks[0]); \
current->scx.kf_tasks[0] = task; \
SCX_CALL_OP((sch), op, locked_rq, task, ##args); \
current->scx.kf_tasks[0] = NULL; \
} while (0)
#define SCX_CALL_OP_TASK_RET(sch, op, locked_rq, task, args...) \
({ \
__typeof__((sch)->ops.op(task, ##args)) __ret; \
WARN_ON_ONCE(current->scx.kf_tasks[0]); \
current->scx.kf_tasks[0] = task; \
__ret = SCX_CALL_OP_RET((sch), op, locked_rq, task, ##args); \
current->scx.kf_tasks[0] = NULL; \
__ret; \
})
#define SCX_CALL_OP_2TASKS_RET(sch, op, locked_rq, task0, task1, args...) \
({ \
__typeof__((sch)->ops.op(task0, task1, ##args)) __ret; \
WARN_ON_ONCE(current->scx.kf_tasks[0]); \
current->scx.kf_tasks[0] = task0; \
current->scx.kf_tasks[1] = task1; \
__ret = SCX_CALL_OP_RET((sch), op, locked_rq, task0, task1, ##args); \
current->scx.kf_tasks[0] = NULL; \
current->scx.kf_tasks[1] = NULL; \
__ret; \
})
/* see SCX_CALL_OP_TASK() */
static __always_inline bool scx_kf_arg_task_ok(struct scx_sched *sch,
struct task_struct *p)
{
if (unlikely((p != current->scx.kf_tasks[0] &&
p != current->scx.kf_tasks[1]))) {
scx_error(sch, "called on a task not being operated on");
return false;
}
return true;
}
static inline bool scx_bypassing(struct scx_sched *sch, s32 cpu)
{
return unlikely(per_cpu_ptr(sch->pcpu, cpu)->flags &
@ -1627,6 +1738,20 @@ static inline struct scx_sched *scx_prog_sched(const struct bpf_prog_aux *aux)
return NULL;
}
/**
* scx_parent - Find the parent sched
* @sch: sched to find the parent of
*
* Returns the parent scheduler or %NULL if @sch is root.
*/
static inline struct scx_sched *scx_parent(struct scx_sched *sch)
{
if (sch->level)
return sch->ancestors[sch->level - 1];
else
return NULL;
}
#else /* CONFIG_EXT_SUB_SCHED */
static inline struct scx_sched *scx_task_sched(const struct task_struct *p)
{
@ -1650,4 +1775,8 @@ static inline struct scx_sched *scx_prog_sched(const struct bpf_prog_aux *aux)
{
return rcu_dereference_all(scx_root);
}
static inline struct scx_sched *scx_parent(struct scx_sched *sch) { return NULL; }
#endif /* CONFIG_EXT_SUB_SCHED */
#endif /* _KERNEL_SCHED_EXT_INTERNAL_H */

View File

@ -8,6 +8,12 @@
#ifndef _KERNEL_SCHED_EXT_TYPES_H
#define _KERNEL_SCHED_EXT_TYPES_H
#include <linux/types.h>
#include <linux/jiffies.h>
#include <linux/overflow.h>
#include <linux/time64.h>
#include <linux/sched/topology.h>
enum scx_consts {
SCX_DSP_DFL_MAX_BATCH = 32,
SCX_DSP_MAX_LOOPS = 32,

View File

@ -4211,6 +4211,6 @@ DEFINE_CLASS(sched_change, struct sched_change_ctx *,
DEFINE_CLASS_IS_UNCONDITIONAL(sched_change)
#include "ext.h"
#include "ext/ext.h"
#endif /* _KERNEL_SCHED_SCHED_H */

View File

@ -1,6 +1,6 @@
/* SPDX-License-Identifier: GPL-2.0 */
/*
* BPF-side helpers for cids and cmasks. See kernel/sched/ext_cid.h for the
* BPF-side helpers for cids and cmasks. See kernel/sched/ext/cid.h for the
* authoritative layout and semantics. The BPF-side helpers use the cmask_*
* naming (no scx_ prefix); cmask is the SCX bitmap type so the prefix is
* redundant in BPF code. Atomics use __sync_val_compare_and_swap and every
@ -33,7 +33,7 @@
#endif
/*
* Mirrors SCX_CMASK_NR_WORDS in kernel/sched/ext_types.h. The u64 cast keeps
* Mirrors SCX_CMASK_NR_WORDS in kernel/sched/ext/types.h. The u64 cast keeps
* the +63 from wrapping when @nr_cids is near U32_MAX, so cmask_reframe()
* bounds-checking the result against alloc_words catches the overflow instead
* of seeing a small value.
@ -281,7 +281,7 @@ static __always_inline void cmask_zero(struct scx_cmask __arena *m)
/*
* BPF_-prefixed to avoid colliding with the kernel's anonymous CMASK_OP_*
* enum in ext_cid.c, which is exported via BTF and reachable through
* enum in ext/cid.c, which is exported via BTF and reachable through
* vmlinux.h.
*/
enum {