sched_ext: Changes for v7.2

This depends on the BPF arena work in bpf-next and should be pulled after
 it. for-7.1-fixes was merged in a few times to avoid conflicts with the
 feature changes.
 
 - Most of this continues the in-development sub-scheduler support, which
   lets a root BPF scheduler delegate to nested sub-schedulers. The
   dispatch-path building blocks landed in 7.1. A follow-up patchset in
   development will complete enqueue-path support for hierarchical
   scheduling. This cycle adds most of that infrastructure:
 
   - Topological CPU IDs (cids): a dense, topology-ordered CPU numbering
     where the CPUs of a core, LLC, or NUMA node form contiguous ranges,
     so a topology unit becomes a (start, length) slice. Raw CPU numbers
     are sparse and don't track topological closeness, which makes them
     clumsy for sharding work across sub-schedulers and awkward in BPF.
 
   - cmask: bitmaps windowed over a slice of cid space, so a sub-scheduler
     can track, for example, the idle cids of its shard without a full
     NR_CPUS cpumask.
 
   - A struct_ops variant that cid-form sub-schedulers register with,
     along with the cid-form kfuncs they call.
 
   - BPF arena integration, which sub-scheduler support is built on. The
     bpf-next additions let the kernel read and write the BPF scheduler's
     arena directly, turning it into a real kernel/BPF shared-memory
     channel. Shared state like the per-CPU cmask now lives there.
 
   - scx_qmap is reworked to exercise the new arena and cid interfaces.
 
 - Exit-dump improvements: dump the faulting CPU first, expose the exit
   CPU to BPF and userspace, and normalize the dump header.
 
 - Misc kfuncs and cleanups: a task-ID lookup kfunc, __printf checking on
   the error and dump formatters, header reorganization, and assorted
   fixes.
 -----BEGIN PGP SIGNATURE-----
 
 iIQEABYKACwWIQTfIjM1kS57o3GsC/uxYfJx3gVYGQUCajCCYw4cdGpAa2VybmVs
 Lm9yZwAKCRCxYfJx3gVYGSnKAP4/v5Y5VCaB/C6a4RaWuhCKI5tR8hLd9zMwJY1h
 0KrhQAEAtVVe8FoTQFFZhP12HEuebvNARqJ8E8bDQCb41B/FfQ4=
 =Xrkh
 -----END PGP SIGNATURE-----

Merge tag 'sched_ext-for-7.2' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext

Pull sched_ext updates from Tejun Heo:
 "Most of this continues the in-development sub-scheduler support, which
  lets a root BPF scheduler delegate to nested sub-schedulers. The
  dispatch-path building blocks landed in 7.1. A follow-up patchset in
  development will complete enqueue-path support for hierarchical
  scheduling. This cycle adds most of that infrastructure:

   - Topological CPU IDs (cids): a dense, topology-ordered CPU numbering
     where the CPUs of a core, LLC, or NUMA node form contiguous ranges,
     so a topology unit becomes a (start, length) slice. Raw CPU numbers
     are sparse and don't track topological closeness, which makes them
     clumsy for sharding work across sub-schedulers and awkward in BPF.

   - cmask: bitmaps windowed over a slice of cid space, so a
     sub-scheduler can track, for example, the idle cids of its shard
     without a full NR_CPUS cpumask.

   - A struct_ops variant that cid-form sub-schedulers register with,
     along with the cid-form kfuncs they call.

   - BPF arena integration, which sub-scheduler support is built on. The
     bpf-next additions let the kernel read and write the BPF
     scheduler's arena directly, turning it into a real kernel/BPF
     shared-memory channel. Shared state like the per-CPU cmask now
     lives there.

   - scx_qmap is reworked to exercise the new arena and cid interfaces.

  Additionally:

   - Exit-dump improvements: dump the faulting CPU first, expose the
     exit CPU to BPF and userspace, and normalize the dump header.

   - Misc kfuncs and cleanups: a task-ID lookup kfunc, __printf checking
     on the error and dump formatters, header reorganization, and
     assorted fixes"

* tag 'sched_ext-for-7.2' of git://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext: (59 commits)
  sched_ext: Add scx_arena_to_kaddr() / scx_kaddr_to_arena()
  sched_ext: Make scx_bpf_kick_cid() return s32
  sched_ext: Add scx_cmask_test() and scx_cmask_for_each_cid()
  tools/sched_ext: Order single-cid cmask helpers as (cid, mask)
  sched_ext: Order single-cid cmask helpers as (cid, mask)
  selftests/sched_ext: Fix dsq_move_to_local check
  sched_ext: Guard BPF arena helper calls to fix 32-bit build
  sched_ext: idle: Fix errno loss in scx_idle_init()
  sched_ext: Convert ops.set_cmask() to arena-resident cmask
  sched_ext: Sub-allocator over kernel-claimed BPF arena pages
  sched_ext: Require an arena for cid-form schedulers
  sched_ext: Add cmask mask ops
  sched_ext: Track bits[] storage size in struct scx_cmask
  sched_ext: Rename scx_cmask.nr_bits to nr_cids
  tools/sched_ext: scx_qmap: Fix qa arena placement
  sched_ext: Mark !CONFIG_EXT_SUB_SCHED dummy stubs static inline
  sched_ext: Replace tryget_task_struct() with get_task_struct()
  sched_ext: Add scx_task_iter_relock() and use it in scx_root_enable_workfn()
  sched_ext: Fix ops_cid layout assert
  sched_ext: Use offsetofend on both sides of the ops_cid layout assert
  ...
This commit is contained in:
Linus Torvalds 2026-06-17 12:10:11 +01:00
commit 5b33fc6492
29 changed files with 4093 additions and 707 deletions

View File

@ -339,6 +339,11 @@ The following briefly shows how a waking task is scheduled and executed.
leaves (e.g., when ``ops.dispatch()`` moves it to a terminal DSQ, or
on property change / sleep).
Note that ``ops.enqueue()`` can be called multiple times in a row without
an intervening call to ``ops.dequeue()``. This can happen, for example,
when a task on a user-created DSQ is re-enqueued using
``scx_bpf_dsq_reenq()``. The task stays in BPF custody the entire time.
When a task leaves BPF scheduler custody, ``ops.dequeue()`` is invoked.
The dequeue can happen for different reasons, distinguished by flags:
@ -503,7 +508,7 @@ Where to Look
custom DSQ.
* ``scx_qmap[.bpf].c``: A multi-level FIFO scheduler supporting five
levels of priority implemented with ``BPF_MAP_TYPE_QUEUE``.
levels of priority implemented with arena-backed doubly-linked lists.
* ``scx_central[.bpf].c``: A central FIFO scheduler where all scheduling
decisions are made on one CPU, demonstrating ``LOCAL_ON`` dispatching,

View File

@ -207,6 +207,15 @@ struct sched_ext_entity {
u64 core_sched_at; /* see scx_prio_less() */
#endif
/*
* Unique non-zero task ID assigned at fork. Persists across exec and
* is never reused. Lets BPF schedulers identify tasks without storing
* kernel pointers - arena-backed schedulers being one example. See
* scx_bpf_tid_to_task().
*/
u64 tid;
struct rhash_head tid_hash_node; /* see SCX_OPS_TID_TO_TASK */
/* BPF scheduler modifiable fields */
/*

View File

@ -58,8 +58,17 @@
#include "deadline.c"
#ifdef CONFIG_SCHED_CLASS_EXT
# include <linux/btf_ids.h>
# include <linux/find.h>
# include <linux/genalloc.h>
# include "ext_types.h"
# include "ext_internal.h"
# include "ext_cid.h"
# include "ext_arena.h"
# include "ext_idle.h"
# include "ext.c"
# include "ext_cid.c"
# include "ext_arena.c"
# include "ext_idle.c"
#endif

File diff suppressed because it is too large Load Diff

131
kernel/sched/ext_arena.c Normal file
View File

@ -0,0 +1,131 @@
/* SPDX-License-Identifier: GPL-2.0 */
/*
* BPF extensible scheduler class: Documentation/scheduler/sched-ext.rst
*
* scx_arena_pool: kernel-side sub-allocator over BPF-arena pages.
*
* Each chunk added to @sch->arena_pool comes from one
* bpf_arena_alloc_pages_sleepable() call and is registered at the
* kernel-side mapping address. Callers translate to the BPF-arena form
* themselves if needed.
*
* Allocations grow the pool on demand. Underlying arena pages are released
* when the arena map itself is torn down.
*
* Copyright (c) 2026 Meta Platforms, Inc. and affiliates.
* Copyright (c) 2026 Tejun Heo <tj@kernel.org>
*/
enum scx_arena_consts {
SCX_ARENA_MIN_ORDER = 3, /* 8-byte minimum sub-allocation */
SCX_ARENA_GROW_PAGES = 4, /* per growth */
};
s32 scx_arena_pool_init(struct scx_sched *sch)
{
if (!sch->arena_map)
return 0;
sch->arena_pool = gen_pool_create(SCX_ARENA_MIN_ORDER, NUMA_NO_NODE);
if (!sch->arena_pool)
return -ENOMEM;
return 0;
}
static void scx_arena_clear_chunk(struct gen_pool *pool, struct gen_pool_chunk *chunk,
void *data)
{
int order = pool->min_alloc_order;
size_t chunk_sz = chunk->end_addr - chunk->start_addr + 1;
unsigned long end_bit = chunk_sz >> order;
unsigned long b, e;
for_each_set_bitrange(b, e, chunk->bits, end_bit)
gen_pool_free(pool, chunk->start_addr + (b << order),
(e - b) << order);
}
/*
* Tear down the pool. Outstanding gen_pool allocations are freed via
* scx_arena_clear_chunk() so gen_pool_destroy() doesn't BUG. The underlying
* arena pages are released when the arena map itself is torn down.
*/
void scx_arena_pool_destroy(struct scx_sched *sch)
{
if (!sch->arena_pool)
return;
gen_pool_for_each_chunk(sch->arena_pool, scx_arena_clear_chunk, NULL);
gen_pool_destroy(sch->arena_pool);
sch->arena_pool = NULL;
}
/*
* Grow the pool by @page_cnt pages. bpf_arena_alloc_pages_sleepable() and
* gen_pool_add() (which calls vzalloc(GFP_KERNEL)) require a sleepable
* context.
*/
static int scx_arena_grow(struct scx_sched *sch, u32 page_cnt)
{
u64 kern_vm_start;
u32 uaddr32;
void *p;
int ret;
if (!sch->arena_map || !sch->arena_pool)
return -EINVAL;
p = bpf_arena_alloc_pages_sleepable(sch->arena_map, NULL,
page_cnt, NUMA_NO_NODE, 0);
if (!p)
return -ENOMEM;
uaddr32 = (u32)(unsigned long)p;
/* arena.o, which defines these, is built only on MMU && 64BIT */
#if defined(CONFIG_MMU) && defined(CONFIG_64BIT)
kern_vm_start = bpf_arena_map_kern_vm_start(sch->arena_map);
#else
kern_vm_start = 0;
#endif
ret = gen_pool_add(sch->arena_pool, kern_vm_start + uaddr32,
page_cnt * PAGE_SIZE, NUMA_NO_NODE);
if (ret) {
bpf_arena_free_pages_non_sleepable(sch->arena_map, p, page_cnt);
return ret;
}
return 0;
}
/*
* Allocate @size bytes from the arena pool. Returns kernel VA on success, NULL
* on failure. May grow the pool via scx_arena_grow() which sleeps. Caller must
* be in a GFP_KERNEL context.
*/
void *scx_arena_alloc(struct scx_sched *sch, size_t size)
{
unsigned long kern_va;
u32 page_cnt;
might_sleep();
if (!sch->arena_pool)
return NULL;
while (true) {
kern_va = gen_pool_alloc(sch->arena_pool, size);
if (kern_va)
break;
page_cnt = max_t(u32, SCX_ARENA_GROW_PAGES,
(size + PAGE_SIZE - 1) >> PAGE_SHIFT);
if (scx_arena_grow(sch, page_cnt))
return NULL;
}
return (void *)kern_va;
}
void scx_arena_free(struct scx_sched *sch, void *kern_va, size_t size)
{
if (sch->arena_pool && kern_va)
gen_pool_free(sch->arena_pool, (unsigned long)kern_va, size);
}

18
kernel/sched/ext_arena.h Normal file
View File

@ -0,0 +1,18 @@
/* SPDX-License-Identifier: GPL-2.0 */
/*
* BPF extensible scheduler class: Documentation/scheduler/sched-ext.rst
*
* Copyright (c) 2025 Meta Platforms, Inc. and affiliates.
* Copyright (c) 2025 Tejun Heo <tj@kernel.org>
*/
#ifndef _KERNEL_SCHED_EXT_ARENA_H
#define _KERNEL_SCHED_EXT_ARENA_H
struct scx_sched;
s32 scx_arena_pool_init(struct scx_sched *sch);
void scx_arena_pool_destroy(struct scx_sched *sch);
void *scx_arena_alloc(struct scx_sched *sch, size_t size);
void scx_arena_free(struct scx_sched *sch, void *kern_va, size_t size);
#endif /* _KERNEL_SCHED_EXT_ARENA_H */

707
kernel/sched/ext_cid.c Normal file
View File

@ -0,0 +1,707 @@
/* SPDX-License-Identifier: GPL-2.0 */
/*
* BPF extensible scheduler class: Documentation/scheduler/sched-ext.rst
*
* Copyright (c) 2026 Meta Platforms, Inc. and affiliates.
* Copyright (c) 2026 Tejun Heo <tj@kernel.org>
*/
#include <linux/cacheinfo.h>
/*
* cid tables.
*
* Pointers are published once on first enable and never revoked. The default
* mapping is populated before ops.init() runs; scx_bpf_cid_override() commits
* before it returns. As long as the BPF scheduler only uses the tables from
* those points onward, it sees a consistent view.
*/
s16 *scx_cid_to_cpu_tbl;
s16 *scx_cpu_to_cid_tbl;
struct scx_cid_topo *scx_cid_topo;
#define SCX_CID_TOPO_NEG (struct scx_cid_topo) { \
.core_cid = -1, .core_idx = -1, .llc_cid = -1, .llc_idx = -1, \
.node_cid = -1, .node_idx = -1, \
}
/*
* Return @cpu's LLC shared_cpu_map. If cacheinfo isn't populated (offline or
* !present), record @cpu in @fallbacks and return its node mask instead - the
* worst that can happen is that the cpu's LLC becomes coarser than reality.
*/
static const struct cpumask *cpu_llc_mask(int cpu, struct cpumask *fallbacks)
{
struct cpu_cacheinfo *ci = get_cpu_cacheinfo(cpu);
if (!ci || !ci->info_list || !ci->num_leaves) {
cpumask_set_cpu(cpu, fallbacks);
return cpumask_of_node(cpu_to_node(cpu));
}
return &ci->info_list[ci->num_leaves - 1].shared_cpu_map;
}
/* Allocate the cid tables once on first enable; never freed. */
static s32 scx_cid_arrays_alloc(void)
{
u32 npossible = num_possible_cpus();
s16 *cid_to_cpu, *cpu_to_cid;
struct scx_cid_topo *cid_topo;
if (scx_cid_to_cpu_tbl)
return 0;
cid_to_cpu = kzalloc_objs(*scx_cid_to_cpu_tbl, npossible, GFP_KERNEL);
cpu_to_cid = kzalloc_objs(*scx_cpu_to_cid_tbl, nr_cpu_ids, GFP_KERNEL);
cid_topo = kmalloc_objs(*scx_cid_topo, npossible, GFP_KERNEL);
if (!cid_to_cpu || !cpu_to_cid || !cid_topo) {
kfree(cid_to_cpu);
kfree(cpu_to_cid);
kfree(cid_topo);
return -ENOMEM;
}
WRITE_ONCE(scx_cid_to_cpu_tbl, cid_to_cpu);
WRITE_ONCE(scx_cpu_to_cid_tbl, cpu_to_cid);
WRITE_ONCE(scx_cid_topo, cid_topo);
return 0;
}
/**
* scx_cid_init - build the cid mapping
* @sch: the scx_sched being initialized; used as the scx_error() target
*
* See "Topological CPU IDs" in ext_cid.h for the model. Walk online cpus by
* intersection at each level (parent_scratch & this_level_mask), which keeps
* containment correct by construction and naturally splits a physical LLC
* straddling two NUMA nodes into two LLC units. The caller must hold
* cpus_read_lock.
*/
s32 scx_cid_init(struct scx_sched *sch)
{
cpumask_var_t to_walk __free(free_cpumask_var) = CPUMASK_VAR_NULL;
cpumask_var_t node_scratch __free(free_cpumask_var) = CPUMASK_VAR_NULL;
cpumask_var_t llc_scratch __free(free_cpumask_var) = CPUMASK_VAR_NULL;
cpumask_var_t core_scratch __free(free_cpumask_var) = CPUMASK_VAR_NULL;
cpumask_var_t llc_fallback __free(free_cpumask_var) = CPUMASK_VAR_NULL;
cpumask_var_t online_no_topo __free(free_cpumask_var) = CPUMASK_VAR_NULL;
u32 next_cid = 0;
s32 next_node_idx = 0, next_llc_idx = 0, next_core_idx = 0;
s32 cpu, ret;
/* CMASK_MAX_WORDS in cid.bpf.h covers NR_CPUS up to 8192 */
BUILD_BUG_ON(NR_CPUS > 8192);
lockdep_assert_cpus_held();
ret = scx_cid_arrays_alloc();
if (ret)
return ret;
if (!zalloc_cpumask_var(&to_walk, GFP_KERNEL) ||
!zalloc_cpumask_var(&node_scratch, GFP_KERNEL) ||
!zalloc_cpumask_var(&llc_scratch, GFP_KERNEL) ||
!zalloc_cpumask_var(&core_scratch, GFP_KERNEL) ||
!zalloc_cpumask_var(&llc_fallback, GFP_KERNEL) ||
!zalloc_cpumask_var(&online_no_topo, GFP_KERNEL))
return -ENOMEM;
/* -1 sentinels for sparse-possible cpu id holes (0 is a valid cid) */
for (cpu = 0; cpu < nr_cpu_ids; cpu++)
scx_cpu_to_cid_tbl[cpu] = -1;
cpumask_copy(to_walk, cpu_online_mask);
while (!cpumask_empty(to_walk)) {
s32 next_cpu = cpumask_first(to_walk);
s32 nid = cpu_to_node(next_cpu);
s32 node_cid = next_cid;
s32 node_idx;
/*
* No NUMA info: skip and let the tail loop assign a no-topo
* cid. cpumask_of_node(-1) is undefined.
*/
if (nid < 0) {
cpumask_clear_cpu(next_cpu, to_walk);
continue;
}
node_idx = next_node_idx++;
/* node_scratch = to_walk & this node */
cpumask_and(node_scratch, to_walk, cpumask_of_node(nid));
if (WARN_ON_ONCE(!cpumask_test_cpu(next_cpu, node_scratch)))
return -EINVAL;
while (!cpumask_empty(node_scratch)) {
s32 ncpu = cpumask_first(node_scratch);
const struct cpumask *llc_mask = cpu_llc_mask(ncpu, llc_fallback);
s32 llc_cid = next_cid;
s32 llc_idx = next_llc_idx++;
/* llc_scratch = node_scratch & this llc */
cpumask_and(llc_scratch, node_scratch, llc_mask);
if (WARN_ON_ONCE(!cpumask_test_cpu(ncpu, llc_scratch)))
return -EINVAL;
while (!cpumask_empty(llc_scratch)) {
s32 lcpu = cpumask_first(llc_scratch);
const struct cpumask *sib = topology_sibling_cpumask(lcpu);
s32 core_cid = next_cid;
s32 core_idx = next_core_idx++;
s32 ccpu;
/* core_scratch = llc_scratch & this core */
cpumask_and(core_scratch, llc_scratch, sib);
if (WARN_ON_ONCE(!cpumask_test_cpu(lcpu, core_scratch)))
return -EINVAL;
for_each_cpu(ccpu, core_scratch) {
s32 cid = next_cid++;
scx_cid_to_cpu_tbl[cid] = ccpu;
scx_cpu_to_cid_tbl[ccpu] = cid;
scx_cid_topo[cid] = (struct scx_cid_topo){
.core_cid = core_cid,
.core_idx = core_idx,
.llc_cid = llc_cid,
.llc_idx = llc_idx,
.node_cid = node_cid,
.node_idx = node_idx,
};
cpumask_clear_cpu(ccpu, llc_scratch);
cpumask_clear_cpu(ccpu, node_scratch);
cpumask_clear_cpu(ccpu, to_walk);
}
}
}
}
/*
* No-topo section: any possible cpu without a cid - normally just the
* not-online ones. Collect any currently-online cpus that land here in
* @online_no_topo so we can warn about them at the end.
*/
for_each_cpu(cpu, cpu_possible_mask) {
s32 cid;
if (__scx_cpu_to_cid(cpu) != -1)
continue;
if (cpu_online(cpu))
cpumask_set_cpu(cpu, online_no_topo);
cid = next_cid++;
scx_cid_to_cpu_tbl[cid] = cpu;
scx_cpu_to_cid_tbl[cpu] = cid;
scx_cid_topo[cid] = SCX_CID_TOPO_NEG;
}
if (!cpumask_empty(llc_fallback))
pr_warn("scx_cid: cpus without cacheinfo, using node mask as llc: %*pbl\n",
cpumask_pr_args(llc_fallback));
if (!cpumask_empty(online_no_topo))
pr_warn("scx_cid: online cpus with no usable topology: %*pbl\n",
cpumask_pr_args(online_no_topo));
return 0;
}
/**
* scx_cmask_clear - Zero every bit in @m's active range
* @m: cmask to clear
*
* Storage past the active range is left as is.
*/
void scx_cmask_clear(struct scx_cmask *m)
{
u32 nr_words;
if (!m->nr_cids)
return;
nr_words = (m->base + m->nr_cids - 1) / 64 - m->base / 64 + 1;
memset(m->bits, 0, nr_words * sizeof(u64));
}
/**
* scx_cmask_fill - Set every bit in @m's active range
* @m: cmask to fill
*
* Counterpart to scx_cmask_clear(). Storage past the active range is left as is.
*/
void scx_cmask_fill(struct scx_cmask *m)
{
u32 nr_words, head_bits, tail_bits;
if (!m->nr_cids)
return;
nr_words = (m->base + m->nr_cids - 1) / 64 - m->base / 64 + 1;
memset(m->bits, 0xff, nr_words * sizeof(u64));
/* clear word-0 bits below base */
head_bits = m->base & 63;
if (head_bits)
m->bits[0] &= ~((1ULL << head_bits) - 1);
/* clear last-word bits at or past base + nr_cids */
tail_bits = (m->base + m->nr_cids) & 63;
if (tail_bits)
m->bits[nr_words - 1] &= (1ULL << tail_bits) - 1;
}
/**
* scx_cpumask_to_cmask - Translate a kernel cpumask into a cmask
* @src: source cpumask
* @dst: cmask to write
*
* Clear @dst's active range and set the bit for each cid whose cpu is in
* @src and lies within that range. Out-of-range cids are silently ignored.
*/
void scx_cpumask_to_cmask(const struct cpumask *src, struct scx_cmask *dst)
{
s32 cpu;
scx_cmask_clear(dst);
for_each_cpu(cpu, src) {
s32 cid = __scx_cpu_to_cid(cpu);
if (cid >= 0)
__scx_cmask_set(cid, dst);
}
}
__bpf_kfunc_start_defs();
/**
* scx_bpf_cid_override - Install an explicit cpu->cid mapping
* @cpu_to_cid: array of nr_cpu_ids s32 entries (cid for each cpu)
* @cpu_to_cid__sz: must be nr_cpu_ids * sizeof(s32) bytes
* @aux: implicit BPF argument to access bpf_prog_aux hidden from BPF progs
*
* May only be called from ops.init() of the root scheduler. Replace the
* topology-probed cid mapping with the caller-provided one. Each possible cpu
* must map to a unique cid in [0, num_possible_cpus()). Topo info is cleared.
* On invalid input, trigger scx_error() to abort the scheduler.
*/
__bpf_kfunc void scx_bpf_cid_override(const s32 *cpu_to_cid, u32 cpu_to_cid__sz,
const struct bpf_prog_aux *aux)
{
cpumask_var_t seen __free(free_cpumask_var) = CPUMASK_VAR_NULL;
struct scx_sched *sch;
bool alloced;
s32 cpu, cid;
/* GFP_KERNEL alloc must happen before the rcu read section */
alloced = zalloc_cpumask_var(&seen, GFP_KERNEL);
guard(rcu)();
sch = scx_prog_sched(aux);
if (unlikely(!sch))
return;
if (!alloced) {
scx_error(sch, "scx_bpf_cid_override: failed to allocate cpumask");
return;
}
if (scx_parent(sch)) {
scx_error(sch, "scx_bpf_cid_override() only allowed from root sched");
return;
}
if (cpu_to_cid__sz != nr_cpu_ids * sizeof(s32)) {
scx_error(sch, "scx_bpf_cid_override: expected %zu bytes, got %u",
nr_cpu_ids * sizeof(s32), cpu_to_cid__sz);
return;
}
for_each_possible_cpu(cpu) {
s32 c = cpu_to_cid[cpu];
if (!cid_valid(sch, c))
return;
if (cpumask_test_and_set_cpu(c, seen)) {
scx_error(sch, "cid %d assigned to multiple cpus", c);
return;
}
scx_cpu_to_cid_tbl[cpu] = c;
scx_cid_to_cpu_tbl[c] = cpu;
}
/* Invalidate stale topo info - the override carries no topology. */
for (cid = 0; cid < num_possible_cpus(); cid++)
scx_cid_topo[cid] = SCX_CID_TOPO_NEG;
}
/**
* scx_bpf_cid_to_cpu - Return the raw CPU id for @cid
* @cid: cid to look up
* @aux: implicit BPF argument to access bpf_prog_aux hidden from BPF progs
*
* Return the raw CPU id for @cid. Trigger scx_error() and return -EINVAL if
* @cid is invalid. The cid<->cpu mapping is static for the lifetime of the
* loaded scheduler, so the BPF side can cache the result to avoid repeated
* kfunc invocations.
*/
__bpf_kfunc s32 scx_bpf_cid_to_cpu(s32 cid, const struct bpf_prog_aux *aux)
{
struct scx_sched *sch;
guard(rcu)();
sch = scx_prog_sched(aux);
if (unlikely(!sch))
return -EINVAL;
return scx_cid_to_cpu(sch, cid);
}
/**
* scx_bpf_cpu_to_cid - Return the cid for @cpu
* @cpu: cpu to look up
* @aux: implicit BPF argument to access bpf_prog_aux hidden from BPF progs
*
* Return the cid for @cpu. Trigger scx_error() and return -EINVAL if @cpu is
* invalid. The cid<->cpu mapping is static for the lifetime of the loaded
* scheduler, so the BPF side can cache the result to avoid repeated kfunc
* invocations.
*/
__bpf_kfunc s32 scx_bpf_cpu_to_cid(s32 cpu, const struct bpf_prog_aux *aux)
{
struct scx_sched *sch;
guard(rcu)();
sch = scx_prog_sched(aux);
if (unlikely(!sch))
return -EINVAL;
return scx_cpu_to_cid(sch, cpu);
}
/*
* Set ops on cmasks. cmask_walk_op2() shares one walk across mutating
* (and/or/copy/andnot) and predicate (subset/intersects) two-cmask forms;
* cmask_walk_op1() does the same shape over a single cmask range. Every public
* entry passes a compile-time-constant @op; cmask_walk_op{1,2}() and
* cmask_word_op{1,2}() are __always_inline so the inner switch collapses to the
* selected op and cmask_op2_is_pred() folds the predicate early-exit out of
* mutating ops.
*
* Two-cmask ops only touch @dst bits inside the intersection of the two ranges;
* bits outside stay untouched. In particular, scx_cmask_copy() does NOT zero
* @dst bits that lie outside @src's range.
*
* The _RACY variants are otherwise identical to their non-racy counterpart but
* read @src word-by-word via data_race(). Memory ordering with concurrent
* writers is the caller's responsibility.
*/
enum cmask_op2 {
/* mutating */
CMASK_OP2_AND,
CMASK_OP2_OR,
CMASK_OP2_OR_RACY,
CMASK_OP2_COPY,
CMASK_OP2_COPY_RACY,
CMASK_OP2_ANDNOT,
/* predicates - short-circuit when the per-word result is true */
CMASK_OP2_SUBSET,
CMASK_OP2_INTERSECTS,
};
static __always_inline bool cmask_op2_is_pred(const enum cmask_op2 op)
{
return op == CMASK_OP2_SUBSET || op == CMASK_OP2_INTERSECTS;
}
static __always_inline bool cmask_word_op2(u64 *av, const u64 *bp, u64 mask,
const enum cmask_op2 op)
{
switch (op) {
case CMASK_OP2_AND:
*av &= ~mask | *bp;
return false;
case CMASK_OP2_OR:
*av |= *bp & mask;
return false;
case CMASK_OP2_OR_RACY:
*av |= data_race(*bp) & mask;
return false;
case CMASK_OP2_COPY:
*av = (*av & ~mask) | (*bp & mask);
return false;
case CMASK_OP2_COPY_RACY:
*av = (*av & ~mask) | (data_race(*bp) & mask);
return false;
case CMASK_OP2_ANDNOT:
*av &= ~(*bp & mask);
return false;
case CMASK_OP2_SUBSET:
/* stop on the first bit in @sub not set in @super */
return (*bp & ~*av) & mask;
case CMASK_OP2_INTERSECTS:
return (*av & *bp) & mask;
}
unreachable();
}
/*
* Walk the intersection of [@a_base, @a_base + @a_nr_cids) with [@b_base,
* @b_base + @b_nr_cids) word by word, applying @op. Mutating ops walk all words
* and return false; predicates return true on the first word whose per-word
* test is true. Empty intersection returns false (matches "no bits to consider"
* for both mutate and predicate).
*
* Base/nr_cids are taken as parameters so callers with snapshotted bounds can
* drive the walk with values independent of the cmask's header.
*/
static __always_inline bool cmask_walk_op2(u64 *a_bits, u32 a_base, u32 a_nr_cids,
const u64 *b_bits, u32 b_base, u32 b_nr_cids,
const enum cmask_op2 op)
{
u32 lo = max(a_base, b_base);
u32 hi = min(a_base + a_nr_cids, b_base + b_nr_cids);
u32 a_word_off = a_base / 64;
u32 b_word_off = b_base / 64;
u32 lo_word = lo / 64;
u32 hi_word = (hi - 1) / 64;
u64 head_mask = GENMASK_U64(63, lo & 63);
u64 tail_mask = GENMASK_U64((hi - 1) & 63, 0);
u32 w;
if (lo >= hi)
return false;
if (lo_word == hi_word)
return cmask_word_op2(&a_bits[lo_word - a_word_off],
&b_bits[lo_word - b_word_off],
head_mask & tail_mask, op);
if (cmask_word_op2(&a_bits[lo_word - a_word_off],
&b_bits[lo_word - b_word_off], head_mask, op) &&
cmask_op2_is_pred(op))
return true;
for (w = lo_word + 1; w < hi_word; w++)
if (cmask_word_op2(&a_bits[w - a_word_off],
&b_bits[w - b_word_off], ~0ULL, op) &&
cmask_op2_is_pred(op))
return true;
return cmask_word_op2(&a_bits[hi_word - a_word_off],
&b_bits[hi_word - b_word_off], tail_mask, op);
}
enum cmask_op1 {
CMASK_OP1_ANY_SET,
};
static __always_inline bool cmask_word_op1(const u64 *ap, u64 mask,
const enum cmask_op1 op)
{
switch (op) {
case CMASK_OP1_ANY_SET:
return *ap & mask;
}
unreachable();
}
/*
* Walk [@a_base, @a_base + @a_nr_cids) of @a_bits word by word, applying @op.
* Returns true on the first word whose per-word test is true; returns false if
* no word matches or the range is empty. All current op1s short-circuit on
* per-word true; if a non-predicate op1 lands here, add a cmask_op1_is_pred()
* guard analogous to cmask_op2_is_pred().
*/
static __always_inline bool cmask_walk_op1(const u64 *a_bits, u32 a_base,
u32 a_nr_cids,
const enum cmask_op1 op)
{
u32 lo = a_base;
u32 hi = a_base + a_nr_cids;
u32 a_word_off = a_base / 64;
u32 lo_word = lo / 64;
u32 hi_word = (hi - 1) / 64;
u64 head_mask = GENMASK_U64(63, lo & 63);
u64 tail_mask = GENMASK_U64((hi - 1) & 63, 0);
u32 w;
if (lo >= hi)
return false;
if (lo_word == hi_word)
return cmask_word_op1(&a_bits[lo_word - a_word_off],
head_mask & tail_mask, op);
if (cmask_word_op1(&a_bits[lo_word - a_word_off], head_mask, op))
return true;
for (w = lo_word + 1; w < hi_word; w++)
if (cmask_word_op1(&a_bits[w - a_word_off], ~0ULL, op))
return true;
return cmask_word_op1(&a_bits[hi_word - a_word_off], tail_mask, op);
}
void scx_cmask_and(struct scx_cmask *dst, const struct scx_cmask *src)
{
cmask_walk_op2(dst->bits, dst->base, dst->nr_cids,
src->bits, src->base, src->nr_cids, CMASK_OP2_AND);
}
void scx_cmask_or(struct scx_cmask *dst, const struct scx_cmask *src)
{
cmask_walk_op2(dst->bits, dst->base, dst->nr_cids,
src->bits, src->base, src->nr_cids, CMASK_OP2_OR);
}
/**
* scx_cmask_or_racy - OR @src into @dst, reading @src without locking
*
* @src is read word-by-word through data_race(). Same per-bit independence
* rationale as scx_cmask_copy_racy(). Memory ordering with writers is the
* caller's responsibility.
*/
void scx_cmask_or_racy(struct scx_cmask *dst, const struct scx_cmask *src)
{
cmask_walk_op2(dst->bits, dst->base, dst->nr_cids,
src->bits, src->base, src->nr_cids, CMASK_OP2_OR_RACY);
}
void scx_cmask_copy(struct scx_cmask *dst, const struct scx_cmask *src)
{
cmask_walk_op2(dst->bits, dst->base, dst->nr_cids,
src->bits, src->base, src->nr_cids, CMASK_OP2_COPY);
}
/**
* scx_cmask_copy_racy - Snapshot @src into @dst without locking
*
* @src is read word-by-word through data_race(). Head/tail masking matches
* scx_cmask_copy(). Each bit in a cmask is independent, so partial updates
* just leave some bits fresher than others. Memory ordering with writers is
* the caller's responsibility.
*/
void scx_cmask_copy_racy(struct scx_cmask *dst, const struct scx_cmask *src)
{
cmask_walk_op2(dst->bits, dst->base, dst->nr_cids,
src->bits, src->base, src->nr_cids, CMASK_OP2_COPY_RACY);
}
void scx_cmask_andnot(struct scx_cmask *dst, const struct scx_cmask *src)
{
cmask_walk_op2(dst->bits, dst->base, dst->nr_cids,
src->bits, src->base, src->nr_cids, CMASK_OP2_ANDNOT);
}
/*
* Return true if @cm has any bit set in [@lo, @hi). Caller must ensure
* [@lo, @hi) is contained in @cm's range.
*/
static bool cmask_any_set_in_range(const struct scx_cmask *cm, u32 lo, u32 hi)
{
if (lo >= hi)
return false;
return cmask_walk_op1(&cm->bits[lo / 64 - cm->base / 64], lo, hi - lo,
CMASK_OP1_ANY_SET);
}
/**
* scx_cmask_subset - test whether @sub is a subset of @super
* @sub: cmask to test
* @super: cmask to test against
*
* Return true iff every set bit of @sub is also set in @super.
*/
bool scx_cmask_subset(const struct scx_cmask *sub, const struct scx_cmask *super)
{
u32 super_end = super->base + super->nr_cids;
u32 sub_end = sub->base + sub->nr_cids;
/*
* Set bits in @sub outside @super's range can't be in @super, so any
* such bit means not a subset. The walk below only visits words
* common to both ranges, so these need a separate scan.
*/
if (sub->base < super->base &&
cmask_any_set_in_range(sub, sub->base, min(super->base, sub_end)))
return false;
if (sub_end > super_end &&
cmask_any_set_in_range(sub, max(sub->base, super_end), sub_end))
return false;
return !cmask_walk_op2((u64 *)super->bits, super->base, super->nr_cids,
sub->bits, sub->base, sub->nr_cids, CMASK_OP2_SUBSET);
}
bool scx_cmask_intersects(const struct scx_cmask *a, const struct scx_cmask *b)
{
return cmask_walk_op2((u64 *)a->bits, a->base, a->nr_cids,
b->bits, b->base, b->nr_cids, CMASK_OP2_INTERSECTS);
}
/**
* scx_cmask_empty - Test whether @m has no bits set
* @m: cmask to test
*
* Return true iff @m's active range has no bits set.
*/
bool scx_cmask_empty(const struct scx_cmask *m)
{
return !cmask_any_set_in_range(m, m->base, m->base + m->nr_cids);
}
/**
* scx_bpf_cid_topo - Copy out per-cid topology info
* @cid: cid to look up
* @out__uninit: where to copy the topology info; fully written by this call
* @aux: implicit BPF argument to access bpf_prog_aux hidden from BPF progs
*
* Fill @out__uninit with the topology info for @cid. Trigger scx_error() if
* @cid is out of range. If @cid is valid but in the no-topo section, all fields
* are set to -1.
*/
__bpf_kfunc void scx_bpf_cid_topo(s32 cid, struct scx_cid_topo *out__uninit,
const struct bpf_prog_aux *aux)
{
struct scx_sched *sch;
guard(rcu)();
sch = scx_prog_sched(aux);
if (unlikely(!sch) || !cid_valid(sch, cid)) {
*out__uninit = SCX_CID_TOPO_NEG;
return;
}
*out__uninit = READ_ONCE(scx_cid_topo)[cid];
}
__bpf_kfunc_end_defs();
BTF_KFUNCS_START(scx_kfunc_ids_init)
BTF_ID_FLAGS(func, scx_bpf_cid_override, KF_IMPLICIT_ARGS | KF_SLEEPABLE)
BTF_KFUNCS_END(scx_kfunc_ids_init)
static const struct btf_kfunc_id_set scx_kfunc_set_init = {
.owner = THIS_MODULE,
.set = &scx_kfunc_ids_init,
.filter = scx_kfunc_context_filter,
};
BTF_KFUNCS_START(scx_kfunc_ids_cid)
BTF_ID_FLAGS(func, scx_bpf_cid_to_cpu, KF_IMPLICIT_ARGS)
BTF_ID_FLAGS(func, scx_bpf_cpu_to_cid, KF_IMPLICIT_ARGS)
BTF_ID_FLAGS(func, scx_bpf_cid_topo, KF_IMPLICIT_ARGS)
BTF_KFUNCS_END(scx_kfunc_ids_cid)
static const struct btf_kfunc_id_set scx_kfunc_set_cid = {
.owner = THIS_MODULE,
.set = &scx_kfunc_ids_cid,
};
int scx_cid_kfunc_init(void)
{
return register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS, &scx_kfunc_set_init) ?:
register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS, &scx_kfunc_set_cid) ?:
register_btf_kfunc_id_set(BPF_PROG_TYPE_TRACING, &scx_kfunc_set_cid) ?:
register_btf_kfunc_id_set(BPF_PROG_TYPE_SYSCALL, &scx_kfunc_set_cid);
}

271
kernel/sched/ext_cid.h Normal file
View File

@ -0,0 +1,271 @@
/* SPDX-License-Identifier: GPL-2.0 */
/*
* Topological CPU IDs (cids)
* --------------------------
*
* Raw cpu numbers are clumsy for sharding work and communication across
* topology units, especially from BPF: the space can be sparse, numerical
* closeness doesn't imply topological closeness (x86 hyperthreading often puts
* SMT siblings far apart), and a range of cpu ids doesn't mean anything.
* Sub-scheds make this acute - cpu allocation, revocation and other state are
* constantly communicated across sub-scheds, and passing whole cpumasks scales
* poorly with cpu count. cpumasks are also awkward in BPF: a variable-length
* kernel type sized for the maximum NR_CPUS (4k), with verbose helper sequences
* for every op.
*
* cids give every cpu a dense, topology-ordered id. CPUs sharing a core, LLC or
* NUMA node get contiguous cid ranges, so a topology unit becomes a (start,
* length) slice of cid space. Communication can pass a slice instead of a
* cpumask, and BPF code can process, for example, a u64 word's worth of cids at
* a time.
*
* The mapping is built once at root scheduler enable time by walking the
* topology of online cpus only. Going by online cpus is out of necessity:
* depending on the arch, topology info isn't reliably available for offline
* cpus. The expected usage model is restarting the scheduler on hotplug events
* so the mapping is rebuilt against the new online set. A scheduler that wants
* to handle hotplug without a restart can provide its own cid and shard mapping
* through the override interface.
*
* Copyright (c) 2026 Meta Platforms, Inc. and affiliates.
* Copyright (c) 2026 Tejun Heo <tj@kernel.org>
*/
#ifndef _KERNEL_SCHED_EXT_CID_H
#define _KERNEL_SCHED_EXT_CID_H
struct scx_sched;
/*
* Cid space (total is always num_possible_cpus()) is laid out with
* topology-annotated cids first, then no-topo cids at the tail. The
* topology-annotated block covers the cpus that were online when scx_cid_init()
* ran and remains valid even after those cpus go offline. The tail block covers
* possible-but-not-online cpus and carries all-(-1) topo info (see
* scx_cid_topo); callers detect it via the -1 sentinels.
*
* See the comment above the table definitions in ext_cid.c for the
* memory-ordering and visibility contract.
*/
extern s16 *scx_cid_to_cpu_tbl;
extern s16 *scx_cpu_to_cid_tbl;
extern struct scx_cid_topo *scx_cid_topo;
extern struct btf_id_set8 scx_kfunc_ids_init;
void scx_cmask_clear(struct scx_cmask *m);
void scx_cmask_fill(struct scx_cmask *m);
void scx_cmask_and(struct scx_cmask *dst, const struct scx_cmask *src);
void scx_cmask_or(struct scx_cmask *dst, const struct scx_cmask *src);
void scx_cmask_or_racy(struct scx_cmask *dst, const struct scx_cmask *src);
void scx_cmask_copy(struct scx_cmask *dst, const struct scx_cmask *src);
void scx_cmask_copy_racy(struct scx_cmask *dst, const struct scx_cmask *src);
void scx_cmask_andnot(struct scx_cmask *dst, const struct scx_cmask *src);
bool scx_cmask_subset(const struct scx_cmask *sub, const struct scx_cmask *super);
bool scx_cmask_intersects(const struct scx_cmask *a, const struct scx_cmask *b);
bool scx_cmask_empty(const struct scx_cmask *m);
s32 scx_cid_init(struct scx_sched *sch);
int scx_cid_kfunc_init(void);
void scx_cpumask_to_cmask(const struct cpumask *src, struct scx_cmask *dst);
/**
* cid_valid - Verify a cid value, to be used on ops input args
* @sch: scx_sched to abort on error
* @cid: cid which came from a BPF ops
*
* Return true if @cid is in [0, num_possible_cpus()). On failure, trigger
* scx_error() and return false.
*/
static inline bool cid_valid(struct scx_sched *sch, s32 cid)
{
if (likely(cid >= 0 && cid < num_possible_cpus()))
return true;
scx_error(sch, "invalid cid %d", cid);
return false;
}
/**
* __scx_cid_to_cpu - Unchecked cid->cpu table lookup
* @cid: cid to look up. Must be in [0, num_possible_cpus()).
*
* Intended for callsites that have already validated @cid and that hold a
* non-NULL @sch from scx_prog_sched() - a live sched implies the table has
* been allocated, so no NULL check is needed here.
*/
static inline s32 __scx_cid_to_cpu(s32 cid)
{
/* READ_ONCE pairs with WRITE_ONCE in scx_cid_arrays_alloc() */
return READ_ONCE(scx_cid_to_cpu_tbl)[cid];
}
/**
* __scx_cpu_to_cid - Unchecked cpu->cid table lookup
* @cpu: cpu to look up. Must be a valid possible cpu id.
*
* Same usage constraints as __scx_cid_to_cpu().
*/
static inline s32 __scx_cpu_to_cid(s32 cpu)
{
return READ_ONCE(scx_cpu_to_cid_tbl)[cpu];
}
/**
* scx_cid_to_cpu - Translate @cid to its cpu
* @sch: scx_sched for error reporting
* @cid: cid to look up
*
* Return the cpu for @cid or a negative errno on failure. Invalid cid triggers
* scx_error() on @sch. The cid arrays are allocated on first scheduler enable
* and never freed, so the returned cpu is stable for the lifetime of the loaded
* scheduler.
*/
static inline s32 scx_cid_to_cpu(struct scx_sched *sch, s32 cid)
{
if (!cid_valid(sch, cid))
return -EINVAL;
return __scx_cid_to_cpu(cid);
}
/**
* scx_cpu_to_cid - Translate @cpu to its cid
* @sch: scx_sched for error reporting
* @cpu: cpu to look up
*
* Return the cid for @cpu or a negative errno on failure. Invalid cpu triggers
* scx_error() on @sch. Same lifetime guarantee as scx_cid_to_cpu().
*/
static inline s32 scx_cpu_to_cid(struct scx_sched *sch, s32 cpu)
{
if (!scx_cpu_valid(sch, cpu, NULL))
return -EINVAL;
return __scx_cpu_to_cid(cpu);
}
/**
* scx_is_cid_type - Test whether the active scheduler hierarchy is cid-form
*/
static inline bool scx_is_cid_type(void)
{
return static_branch_unlikely(&__scx_is_cid_type);
}
static inline bool __scx_cmask_contains(u32 cid, const struct scx_cmask *m)
{
return likely(cid >= m->base && cid < m->base + m->nr_cids);
}
/* Word in bits[] covering @cid. @cid must satisfy __scx_cmask_contains(). */
static inline u64 *__scx_cmask_word(u32 cid, const struct scx_cmask *m)
{
return (u64 *)&m->bits[cid / 64 - m->base / 64];
}
/**
* __scx_cmask_init - Initialize @m with explicit storage capacity
* @m: cmask to initialize
* @base: first cid of the active range
* @nr_cids: number of cids in the active range
* @alloc_cids: storage capacity in cids, at least @nr_cids
*
* Use when storage is sized larger than the initial active range. All of
* bits[] is zeroed.
*/
static inline void __scx_cmask_init(struct scx_cmask *m, u32 base, u32 nr_cids,
u32 alloc_cids)
{
if (WARN_ON_ONCE(alloc_cids < nr_cids))
nr_cids = alloc_cids;
m->base = base;
m->nr_cids = nr_cids;
m->alloc_words = SCX_CMASK_NR_WORDS(alloc_cids);
memset(m->bits, 0, m->alloc_words * sizeof(u64));
}
/**
* scx_cmask_init - Initialize @m on tight storage
* @m: cmask to initialize
* @base: first cid of the active range
* @nr_cids: number of cids in the active range
*
* All of bits[] is zeroed.
*/
static inline void scx_cmask_init(struct scx_cmask *m, u32 base, u32 nr_cids)
{
__scx_cmask_init(m, base, nr_cids, nr_cids);
}
/**
* scx_cmask_reframe - Reshape @m's active range without resizing storage
* @m: cmask to reframe
* @base: new active range base
* @nr_cids: new active range length, must fit within @m->alloc_words
*
* Body bits within the new range become garbage - only the head and tail
* words are zeroed to keep the padding invariant.
*/
static inline void scx_cmask_reframe(struct scx_cmask *m, u32 base, u32 nr_cids)
{
if (WARN_ON_ONCE(SCX_CMASK_NR_WORDS(nr_cids) > m->alloc_words))
return;
if (nr_cids) {
u32 last_word = ((base & 63) + nr_cids - 1) / 64;
m->bits[0] = 0;
m->bits[last_word] = 0;
}
m->base = base;
m->nr_cids = nr_cids;
}
static inline void __scx_cmask_set(u32 cid, struct scx_cmask *m)
{
if (!__scx_cmask_contains(cid, m))
return;
*__scx_cmask_word(cid, m) |= BIT_U64(cid & 63);
}
/**
* scx_cmask_test - test whether @cid is set in @m
* @cid: cid to test
* @m: cmask to test
*
* Return %false if @cid is outside @m's active range. Otherwise return the
* bit's value. Read via READ_ONCE so callers can race set/clear writers.
*/
static inline bool scx_cmask_test(u32 cid, const struct scx_cmask *m)
{
if (!__scx_cmask_contains(cid, m))
return false;
return READ_ONCE(*__scx_cmask_word(cid, m)) & BIT_U64(cid & 63);
}
/*
* Words of bits[] the active range spans, 0 if empty. Tighter than the storage
* SCX_CMASK_NR_WORDS() sizes for the worst-case base alignment.
*/
static inline u32 scx_cmask_nr_used_words(const struct scx_cmask *m)
{
if (!m->nr_cids)
return 0;
return ((m->base & 63) + m->nr_cids - 1) / 64 + 1;
}
/**
* scx_cmask_for_each_cid - iterate set cids in @m
* @cid: s32 loop var that receives each set cid in turn
* @m: cmask to iterate
*
* Visits set bits within @m's active range in ascending order. Scans only the
* words the active range spans, where head and tail padding is kept zero, so
* no per-cid range check is needed.
*/
#define scx_cmask_for_each_cid(cid, m) \
for (u64 __bs = (m)->base & ~63u, __wi = 0, \
__nw = scx_cmask_nr_used_words(m); \
__wi < __nw; __wi++) \
for (u64 __w = READ_ONCE((m)->bits[__wi]); \
__w && ((cid) = __bs + __wi * 64 + __ffs64(__w), true); \
__w &= __w - 1)
#endif /* _KERNEL_SCHED_EXT_CID_H */

View File

@ -9,7 +9,6 @@
* Copyright (c) 2022 David Vernet <dvernet@meta.com>
* Copyright (c) 2024 Andrea Righi <arighi@nvidia.com>
*/
#include "ext_idle.h"
/* Enable/disable built-in idle CPU selection policy */
static DEFINE_STATIC_KEY_FALSE(scx_builtin_idle_enabled);
@ -783,7 +782,7 @@ void __scx_update_idle(struct rq *rq, bool idle, bool do_notify)
*/
if (SCX_HAS_OP(sch, update_idle) && do_notify &&
!scx_bypassing(sch, cpu_of(rq)))
SCX_CALL_OP(sch, update_idle, rq, cpu_of(rq), idle);
SCX_CALL_OP(sch, update_idle, rq, scx_cpu_arg(cpu_of(rq)), idle);
}
static void reset_idle_masks(struct sched_ext_ops *ops)
@ -911,7 +910,7 @@ static s32 select_cpu_from_kfunc(struct scx_sched *sch, struct task_struct *p,
bool we_locked = false;
s32 cpu;
if (!ops_cpu_valid(sch, prev_cpu, NULL))
if (!scx_cpu_valid(sch, prev_cpu, NULL))
return -EINVAL;
if (!check_builtin_idle_enabled(sch))
@ -984,7 +983,7 @@ __bpf_kfunc s32 scx_bpf_cpu_node(s32 cpu, const struct bpf_prog_aux *aux)
guard(rcu)();
sch = scx_prog_sched(aux);
if (unlikely(!sch) || !ops_cpu_valid(sch, cpu, NULL))
if (unlikely(!sch) || !scx_cpu_valid(sch, cpu, NULL))
return NUMA_NO_NODE;
return cpu_to_node(cpu);
}
@ -1266,7 +1265,7 @@ __bpf_kfunc bool scx_bpf_test_and_clear_cpu_idle(s32 cpu, const struct bpf_prog_
if (!check_builtin_idle_enabled(sch))
return false;
if (!ops_cpu_valid(sch, cpu, NULL))
if (!scx_cpu_valid(sch, cpu, NULL))
return false;
return scx_idle_test_and_clear_cpu(cpu);
@ -1504,13 +1503,9 @@ static const struct btf_kfunc_id_set scx_kfunc_set_select_cpu = {
int scx_idle_init(void)
{
int ret;
ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS, &scx_kfunc_set_idle) ||
register_btf_kfunc_id_set(BPF_PROG_TYPE_TRACING, &scx_kfunc_set_idle) ||
register_btf_kfunc_id_set(BPF_PROG_TYPE_SYSCALL, &scx_kfunc_set_idle) ||
register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS, &scx_kfunc_set_select_cpu) ||
register_btf_kfunc_id_set(BPF_PROG_TYPE_SYSCALL, &scx_kfunc_set_select_cpu);
return ret;
return register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS, &scx_kfunc_set_idle) ?:
register_btf_kfunc_id_set(BPF_PROG_TYPE_TRACING, &scx_kfunc_set_idle) ?:
register_btf_kfunc_id_set(BPF_PROG_TYPE_SYSCALL, &scx_kfunc_set_idle) ?:
register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS, &scx_kfunc_set_select_cpu) ?:
register_btf_kfunc_id_set(BPF_PROG_TYPE_SYSCALL, &scx_kfunc_set_select_cpu);
}

View File

@ -8,35 +8,6 @@
#define SCX_OP_IDX(op) (offsetof(struct sched_ext_ops, op) / sizeof(void (*)(void)))
#define SCX_MOFF_IDX(moff) ((moff) / sizeof(void (*)(void)))
enum scx_consts {
SCX_DSP_DFL_MAX_BATCH = 32,
SCX_DSP_MAX_LOOPS = 32,
SCX_WATCHDOG_MAX_TIMEOUT = 30 * HZ,
SCX_EXIT_BT_LEN = 64,
SCX_EXIT_MSG_LEN = 1024,
SCX_EXIT_DUMP_DFL_LEN = 32768,
SCX_CPUPERF_ONE = SCHED_CAPACITY_SCALE,
/*
* Iterating all tasks may take a while. Periodically drop
* scx_tasks_lock to avoid causing e.g. CSD and RCU stalls.
*/
SCX_TASK_ITER_BATCH = 32,
SCX_BYPASS_HOST_NTH = 2,
SCX_BYPASS_LB_DFL_INTV_US = 500 * USEC_PER_MSEC,
SCX_BYPASS_LB_DONOR_PCT = 125,
SCX_BYPASS_LB_MIN_DELTA_DIV = 4,
SCX_BYPASS_LB_BATCH = 256,
SCX_REENQ_LOCAL_MAX_REPEAT = 256,
SCX_SUB_MAX_DEPTH = 4,
};
enum scx_exit_kind {
SCX_EXIT_NONE,
SCX_EXIT_DONE,
@ -94,6 +65,12 @@ struct scx_exit_info {
/* %SCX_EXIT_* - broad category of the exit reason */
enum scx_exit_kind kind;
/*
* CPU that initiated the exit, valid once @kind has been set.
* Negative if the exit path didn't identify a CPU.
*/
s32 exit_cpu;
/* exit code if gracefully exiting */
s64 exit_code;
@ -138,7 +115,8 @@ enum scx_ops_flags {
* To mask this problem, by default, unhashed tasks are automatically
* dispatched to the local DSQ on enqueue. If the BPF scheduler doesn't
* depend on pid lookups and wants to handle these tasks directly, the
* following flag can be used.
* following flag can be used. With %SCX_OPS_TID_TO_TASK,
* scx_bpf_tid_to_task() can find exiting tasks reliably.
*/
SCX_OPS_ENQ_EXITING = 1LLU << 2,
@ -189,6 +167,17 @@ enum scx_ops_flags {
*/
SCX_OPS_ALWAYS_ENQ_IMMED = 1LLU << 7,
/*
* Maintain a mapping from p->scx.tid to task_struct so the BPF
* scheduler can recover task pointers from stored tids via
* scx_bpf_tid_to_task().
*
* Only the root scheduler turns this on. A sub-sched may set the flag
* to declare a dependency on the lookup; if the root scheduler hasn't
* enabled it, attaching the sub-sched is rejected.
*/
SCX_OPS_TID_TO_TASK = 1LLU << 8,
SCX_OPS_ALL_FLAGS = SCX_OPS_KEEP_BUILTIN_IDLE |
SCX_OPS_ENQ_LAST |
SCX_OPS_ENQ_EXITING |
@ -196,7 +185,8 @@ enum scx_ops_flags {
SCX_OPS_ALLOW_QUEUED_WAKEUP |
SCX_OPS_SWITCH_PARTIAL |
SCX_OPS_BUILTIN_IDLE_PER_NODE |
SCX_OPS_ALWAYS_ENQ_IMMED,
SCX_OPS_ALWAYS_ENQ_IMMED |
SCX_OPS_TID_TO_TASK,
/* high 8 bits are internal, don't include in SCX_OPS_ALL_FLAGS */
__SCX_OPS_INTERNAL_MASK = 0xffLLU << 56,
@ -539,28 +529,6 @@ struct sched_ext_ops {
*/
void (*update_idle)(s32 cpu, bool idle);
/**
* @cpu_acquire: A CPU is becoming available to the BPF scheduler
* @cpu: The CPU being acquired by the BPF scheduler.
* @args: Acquire arguments, see the struct definition.
*
* A CPU that was previously released from the BPF scheduler is now once
* again under its control.
*/
void (*cpu_acquire)(s32 cpu, struct scx_cpu_acquire_args *args);
/**
* @cpu_release: A CPU is taken away from the BPF scheduler
* @cpu: The CPU being released by the BPF scheduler.
* @args: Release arguments, see the struct definition.
*
* The specified CPU is no longer under the control of the BPF
* scheduler. This could be because it was preempted by a higher
* priority sched_class, though there may be other reasons as well. The
* caller should consult @args->reason to determine the cause.
*/
void (*cpu_release)(s32 cpu, struct scx_cpu_release_args *args);
/**
* @init_task: Initialize a task to run in a BPF scheduler
* @p: task to initialize for BPF scheduling
@ -851,6 +819,128 @@ struct sched_ext_ops {
/* internal use only, must be NULL */
void __rcu *priv;
/*
* Deprecated callbacks. Kept at the end of the struct so the cid-form
* struct (sched_ext_ops_cid) can omit them without affecting the
* shared field offsets. Use SCX_ENQ_IMMED instead. Sitting past
* SCX_OPI_END means has_op doesn't cover them, so SCX_HAS_OP() cannot
* be used; callers must test sch->ops.cpu_acquire / cpu_release
* directly.
*/
/**
* @cpu_acquire: A CPU is becoming available to the BPF scheduler
* @cpu: The CPU being acquired by the BPF scheduler.
* @args: Acquire arguments, see the struct definition.
*
* A CPU that was previously released from the BPF scheduler is now once
* again under its control. Deprecated; use SCX_ENQ_IMMED instead.
*/
void (*cpu_acquire)(s32 cpu, struct scx_cpu_acquire_args *args);
/**
* @cpu_release: A CPU is taken away from the BPF scheduler
* @cpu: The CPU being released by the BPF scheduler.
* @args: Release arguments, see the struct definition.
*
* The specified CPU is no longer under the control of the BPF
* scheduler. This could be because it was preempted by a higher
* priority sched_class, though there may be other reasons as well. The
* caller should consult @args->reason to determine the cause.
* Deprecated; use SCX_ENQ_IMMED instead.
*/
void (*cpu_release)(s32 cpu, struct scx_cpu_release_args *args);
};
/**
* struct sched_ext_ops_cid - cid-form alternative to struct sched_ext_ops
*
* Mirrors struct sched_ext_ops with cpu/cpumask substituted with cid/cmask
* where applicable. Layout up to and including @priv matches sched_ext_ops
* byte-for-byte (verified by BUILD_BUG_ON checks at scx_init() time) so
* shared field offsets work for both struct types in bpf_scx_init_member()
* and bpf_scx_check_member(). The deprecated cpu_acquire/cpu_release
* callbacks at the tail of sched_ext_ops are omitted here entirely.
*
* Differences from sched_ext_ops:
* - select_cpu -> select_cid (returns cid)
* - dispatch -> dispatch (cpu arg is now cid)
* - update_idle -> update_idle (cpu arg is now cid)
* - set_cpumask -> set_cmask (cmask instead of cpumask)
* - cpu_online -> cid_online
* - cpu_offline -> cid_offline
* - dump_cpu -> dump_cid
* - cpu_acquire/cpu_release -> not present (deprecated in sched_ext_ops)
*
* BPF schedulers using this type cannot call cpu-form scx_bpf_* kfuncs;
* use the cid-form variants instead. Enforced at BPF verifier time via
* scx_kfunc_context_filter() branching on prog->aux->st_ops.
*
* See sched_ext_ops for callback documentation.
*/
struct sched_ext_ops_cid {
s32 (*select_cid)(struct task_struct *p, s32 prev_cid, u64 wake_flags);
void (*enqueue)(struct task_struct *p, u64 enq_flags);
void (*dequeue)(struct task_struct *p, u64 deq_flags);
void (*dispatch)(s32 cid, struct task_struct *prev);
void (*tick)(struct task_struct *p);
void (*runnable)(struct task_struct *p, u64 enq_flags);
void (*running)(struct task_struct *p);
void (*stopping)(struct task_struct *p, bool runnable);
void (*quiescent)(struct task_struct *p, u64 deq_flags);
bool (*yield)(struct task_struct *from, struct task_struct *to);
bool (*core_sched_before)(struct task_struct *a,
struct task_struct *b);
void (*set_weight)(struct task_struct *p, u32 weight);
void (*set_cmask)(struct task_struct *p,
const struct scx_cmask *cmask);
void (*update_idle)(s32 cid, bool idle);
s32 (*init_task)(struct task_struct *p,
struct scx_init_task_args *args);
void (*exit_task)(struct task_struct *p,
struct scx_exit_task_args *args);
void (*enable)(struct task_struct *p);
void (*disable)(struct task_struct *p);
void (*dump)(struct scx_dump_ctx *ctx);
void (*dump_cid)(struct scx_dump_ctx *ctx, s32 cid, bool idle);
void (*dump_task)(struct scx_dump_ctx *ctx, struct task_struct *p);
#ifdef CONFIG_EXT_GROUP_SCHED
s32 (*cgroup_init)(struct cgroup *cgrp,
struct scx_cgroup_init_args *args);
void (*cgroup_exit)(struct cgroup *cgrp);
s32 (*cgroup_prep_move)(struct task_struct *p,
struct cgroup *from, struct cgroup *to);
void (*cgroup_move)(struct task_struct *p,
struct cgroup *from, struct cgroup *to);
void (*cgroup_cancel_move)(struct task_struct *p,
struct cgroup *from, struct cgroup *to);
void (*cgroup_set_weight)(struct cgroup *cgrp, u32 weight);
void (*cgroup_set_bandwidth)(struct cgroup *cgrp,
u64 period_us, u64 quota_us, u64 burst_us);
void (*cgroup_set_idle)(struct cgroup *cgrp, bool idle);
#endif /* CONFIG_EXT_GROUP_SCHED */
s32 (*sub_attach)(struct scx_sub_attach_args *args);
void (*sub_detach)(struct scx_sub_detach_args *args);
void (*cid_online)(s32 cid);
void (*cid_offline)(s32 cid);
s32 (*init)(void);
void (*exit)(struct scx_exit_info *info);
/* Data fields - must match sched_ext_ops layout exactly */
u32 dispatch_max_batch;
u64 flags;
u32 timeout_ms;
u32 exit_dump_len;
u64 hotplug_seq;
u64 sub_cgroup_id;
char name[SCX_OPS_NAME_LEN];
/* internal use only, must be NULL */
void __rcu *priv;
/* layout end anchor for the BUILD_BUG_ON in scx_init(); keep last */
char __end[0];
};
enum scx_opi {
@ -1009,7 +1099,40 @@ struct scx_sched_pnode {
};
struct scx_sched {
struct sched_ext_ops ops;
/*
* cpu-form and cid-form ops share field offsets up to .priv (verified
* by BUILD_BUG_ON in scx_init()). The anonymous union lets the kernel
* access either view of the same storage without function-pointer
* casts: use .ops for cpu-form and shared fields, .ops_cid for the
* cid-renamed callbacks (set_cmask, select_cid, cid_online, ...).
*/
union {
struct sched_ext_ops ops;
struct sched_ext_ops_cid ops_cid;
};
bool is_cid_type; /* true if registered via bpf_sched_ext_ops_cid */
/*
* Arena map auto-discovered from member progs at struct_ops attach.
* cid-form schedulers must use exactly one arena across all member
* progs. NULL on cpu-form.
*
* @arena_pool sub-allocates @arena_map. Each gen_pool chunk is added
* at the kernel-side mapping address. @arena_kern_base is the start
* of the arena's kern_vm range. See scx_arena_to_kaddr() and
* scx_kaddr_to_arena().
*/
struct bpf_map *arena_map;
struct gen_pool *arena_pool;
uintptr_t arena_kern_base;
/*
* Per-CPU arena cmask used by scx_call_op_set_cpumask() to hand a cmask
* to ops_cid.set_cmask(). The kernel writes through the stored kern_va
* and hands BPF its arena pointer via scx_kaddr_to_arena().
*/
struct scx_cmask * __percpu *set_cmask_scratch;
DECLARE_BITMAP(has_op, SCX_OPI_END);
/*
@ -1083,6 +1206,31 @@ struct scx_sched {
struct scx_sched *ancestors[];
};
/**
* scx_arena_to_kaddr - Translate a BPF-arena pointer to its kernel address
* @sch: scheduler whose arena hosts @bpf_ptr
* @bpf_ptr: BPF-arena pointer, only the low 32 bits are used
*
* The (u32) cast normalizes any input into the arena's 4 GiB kern_vm range,
* which combined with scratch-page fault recovery makes the returned pointer
* safe to dereference up to GUARD_SZ / 2 past the intended object. Accesses
* larger than GUARD_SZ / 2 must be explicitly bounds-checked.
*/
static inline void *scx_arena_to_kaddr(struct scx_sched *sch, const void *bpf_ptr)
{
return (void *)(sch->arena_kern_base + (u32)(uintptr_t)bpf_ptr);
}
/**
* scx_kaddr_to_arena - Translate a kernel arena address to its BPF form
* @sch: scheduler whose arena hosts @kaddr
* @kaddr: kernel-side arena address, supplied by trusted kernel code
*/
static inline void *scx_kaddr_to_arena(struct scx_sched *sch, const void *kaddr)
{
return (void *)((uintptr_t)kaddr - sch->arena_kern_base);
}
enum scx_wake_flags {
/* expose select WF_* flags as enums */
SCX_WAKE_FORK = WF_FORK,
@ -1366,8 +1514,30 @@ enum scx_ops_state {
extern struct scx_sched __rcu *scx_root;
DECLARE_PER_CPU(struct rq *, scx_locked_rq_state);
/*
* True when the currently loaded scheduler hierarchy is cid-form. All scheds
* in a hierarchy share one form, so this single key tells callsites which
* view to use without per-sch dereferences. Use scx_is_cid_type() to test.
*/
DECLARE_STATIC_KEY_FALSE(__scx_is_cid_type);
int scx_kfunc_context_filter(const struct bpf_prog *prog, u32 kfunc_id);
bool scx_cpu_valid(struct scx_sched *sch, s32 cpu, const char *where);
__printf(5, 0) bool scx_vexit(struct scx_sched *sch, enum scx_exit_kind kind,
s64 exit_code, s32 exit_cpu, const char *fmt,
va_list args);
__printf(5, 6) bool __scx_exit(struct scx_sched *sch, enum scx_exit_kind kind,
s64 exit_code, s32 exit_cpu, const char *fmt, ...);
#define scx_exit(sch, kind, exit_code, fmt, args...) \
__scx_exit(sch, kind, exit_code, raw_smp_processor_id(), fmt, ##args)
#define scx_error(sch, fmt, args...) \
scx_exit((sch), SCX_EXIT_ERROR, 0, fmt, ##args)
#define scx_verror(sch, fmt, args) \
scx_vexit((sch), SCX_EXIT_ERROR, 0, raw_smp_processor_id(), fmt, args)
/*
* Return the rq currently locked from an scx callback, or NULL if no rq is
* locked.
@ -1476,7 +1646,7 @@ static inline bool scx_task_on_sched(struct scx_sched *sch,
return true;
}
static struct scx_sched *scx_prog_sched(const struct bpf_prog_aux *aux)
static inline struct scx_sched *scx_prog_sched(const struct bpf_prog_aux *aux)
{
return rcu_dereference_all(scx_root);
}

144
kernel/sched/ext_types.h Normal file
View File

@ -0,0 +1,144 @@
/* SPDX-License-Identifier: GPL-2.0 */
/*
* Early sched_ext type definitions.
*
* Copyright (c) 2026 Meta Platforms, Inc. and affiliates.
* Copyright (c) 2026 Tejun Heo <tj@kernel.org>
*/
#ifndef _KERNEL_SCHED_EXT_TYPES_H
#define _KERNEL_SCHED_EXT_TYPES_H
enum scx_consts {
SCX_DSP_DFL_MAX_BATCH = 32,
SCX_DSP_MAX_LOOPS = 32,
SCX_WATCHDOG_MAX_TIMEOUT = 30 * HZ,
/* per-CPU chunk size for p->scx.tid allocation, see scx_alloc_tid() */
SCX_TID_CHUNK = 1024,
SCX_EXIT_BT_LEN = 64,
SCX_EXIT_MSG_LEN = 1024,
SCX_EXIT_DUMP_DFL_LEN = 32768,
SCX_CPUPERF_ONE = SCHED_CAPACITY_SCALE,
/*
* Iterating all tasks may take a while. Periodically drop
* scx_tasks_lock to avoid causing e.g. CSD and RCU stalls.
*/
SCX_TASK_ITER_BATCH = 32,
SCX_BYPASS_HOST_NTH = 2,
SCX_BYPASS_LB_DFL_INTV_US = 500 * USEC_PER_MSEC,
SCX_BYPASS_LB_DONOR_PCT = 125,
SCX_BYPASS_LB_MIN_DELTA_DIV = 4,
SCX_BYPASS_LB_BATCH = 256,
SCX_REENQ_LOCAL_MAX_REPEAT = 256,
SCX_SUB_MAX_DEPTH = 4,
};
/*
* Per-cid topology info. For each topology level (core, LLC, node), records
* the first cid in the unit and its global index. Global indices are
* consecutive integers assigned in cid-walk order, so e.g. core_idx ranges
* over [0, nr_cores_at_init) with no gaps. No-topo cids have all fields set
* to -1.
*
* @core_cid: first cid of this cid's core (smt-sibling group)
* @core_idx: global index of that core, in [0, nr_cores_at_init)
* @llc_cid: first cid of this cid's LLC
* @llc_idx: global index of that LLC, in [0, nr_llcs_at_init)
* @node_cid: first cid of this cid's NUMA node
* @node_idx: global index of that node, in [0, nr_nodes_at_init)
*/
struct scx_cid_topo {
s32 core_cid;
s32 core_idx;
s32 llc_cid;
s32 llc_idx;
s32 node_cid;
s32 node_idx;
};
/*
* cmask: variable-length, base-windowed bitmap over cid space
* -----------------------------------------------------------
*
* A cmask covers the cid range [base, base + nr_cids). bits[] is aligned to the
* global 64-cid grid: bits[0] spans [base & ~63, (base & ~63) + 64), so the
* first (base & 63) bits of bits[0] are head padding and the trailing bits of
* the last active word past base + nr_cids are tail padding. Both stay zero;
* all mutating helpers preserve that. Words past the last active word are not
* read by any helper and have no constraint.
*
* Grid alignment means two cmasks always address bits[] against the same global
* 64-cid windows, so cross-cmask word ops (AND, OR, ...) reduce to
*
* dst->bits[i] OP= src->bits[i - delta]
*
* with no bit-shifting, regardless of how the two bases relate mod 64.
*/
struct scx_cmask {
u32 base;
u32 nr_cids;
u32 alloc_words;
u64 bits[] __counted_by(alloc_words);
};
/*
* Number of u64 words of bits[] storage that covers @nr_cids regardless of base
* alignment. The +1 absorbs up to 63 bits of head padding when base is not
* 64-aligned - always allocating one extra word beats branching on base or
* splitting the compute. The u64 cast keeps the +63 from wrapping when @nr_cids
* is near U32_MAX, so callers bounds-checking the result against @alloc_words
* catch the overflow instead of seeing a small value.
*/
#define SCX_CMASK_NR_WORDS(nr_cids) ((u32)(((u64)(nr_cids) + 63) / 64 + 1))
/**
* __SCX_CMASK_DEFINE - Define an on-stack cmask with explicit storage capacity
* @NAME: variable name to define
* @BASE: first cid of the active range
* @NR_CIDS: active range length
* @ALLOC_CIDS: storage capacity in cids, at least @NR_CIDS
*
* @NAME aliases zero-initialized storage with the active range set to
* [BASE, BASE + NR_CIDS). Use scx_cmask_reframe() to reshape later, up to
* @ALLOC_CIDS.
*/
#define __SCX_CMASK_DEFINE(NAME, BASE, NR_CIDS, ALLOC_CIDS) \
_DEFINE_FLEX(struct scx_cmask, NAME, bits, SCX_CMASK_NR_WORDS(ALLOC_CIDS), \
= { .base = (BASE), \
.nr_cids = (NR_CIDS), \
.alloc_words = SCX_CMASK_NR_WORDS(ALLOC_CIDS) })
/**
* SCX_CMASK_DEFINE - Define an on-stack cmask on tight storage
* @NAME: variable name to define
* @BASE: first cid of the active range
* @NR_CIDS: active range length, also storage capacity
*
* @NAME aliases zero-initialized storage with the active range and storage
* both [BASE, BASE + NR_CIDS).
*/
#define SCX_CMASK_DEFINE(NAME, BASE, NR_CIDS) \
__SCX_CMASK_DEFINE(NAME, BASE, NR_CIDS, NR_CIDS)
/**
* SCX_CMASK_DEFINE_SHARD - Define an on-stack cmask sized to one shard
* @NAME: variable name to define
* @BASE: first cid of the active range
* @NR_CIDS: active range length, must be <= SCX_CID_SHARD_MAX_CPUS
*
* Storage is fixed at SCX_CID_SHARD_MAX_CPUS, active range framed by
* (BASE, NR_CIDS). Passing NR_CIDS > SCX_CID_SHARD_MAX_CPUS leaves the
* cmask claiming more bits than storage holds and subsequent cmask
* operations will overrun.
*/
#define SCX_CMASK_DEFINE_SHARD(NAME, BASE, NR_CIDS) \
__SCX_CMASK_DEFINE(NAME, BASE, NR_CIDS, SCX_CID_SHARD_MAX_CPUS)
#endif /* _KERNEL_SCHED_EXT_TYPES_H */

View File

@ -168,9 +168,9 @@ well on single-socket systems with a unified L3 cache.
Another simple, yet slightly more complex scheduler that provides an example of
a basic weighted FIFO queuing policy. It also provides examples of some common
useful BPF features, such as sleepable per-task storage allocation in the
`ops.prep_enable()` callback, and using the `BPF_MAP_TYPE_QUEUE` map type to
enqueue tasks. It also illustrates how core-sched support could be implemented.
useful BPF features, such as arena-backed doubly-linked lists threaded through
per-task context and `bpf_res_spin_lock` for per-queue synchronization. It also
illustrates how core-sched support could be implemented.
## scx_central

View File

@ -0,0 +1,678 @@
/* SPDX-License-Identifier: GPL-2.0 */
/*
* BPF-side helpers for cids and cmasks. See kernel/sched/ext_cid.h for the
* authoritative layout and semantics. The BPF-side helpers use the cmask_*
* naming (no scx_ prefix); cmask is the SCX bitmap type so the prefix is
* redundant in BPF code. Atomics use __sync_val_compare_and_swap and every
* helper is inline (no .c counterpart).
*
* Included by scx/common.bpf.h; don't include directly.
*
* Copyright (c) 2026 Meta Platforms, Inc. and affiliates.
* Copyright (c) 2026 Tejun Heo <tj@kernel.org>
*/
#ifndef __SCX_CID_BPF_H
#define __SCX_CID_BPF_H
#include "bpf_arena_common.bpf.h"
#ifndef BIT_U64
#define BIT_U64(nr) (1ULL << (nr))
#endif
#ifndef GENMASK_U64
#define GENMASK_U64(h, l) ((~0ULL << (l)) & (~0ULL >> (63 - (h))))
#endif
/*
* Storage cap for bounded loops over bits[]. Sized to cover NR_CPUS=8192 with
* one extra word for head-misalignment. Increase if deployment targets larger
* NR_CPUS.
*/
#ifndef CMASK_MAX_WORDS
#define CMASK_MAX_WORDS 129
#endif
/*
* Mirrors SCX_CMASK_NR_WORDS in kernel/sched/ext_types.h. The u64 cast keeps
* the +63 from wrapping when @nr_cids is near U32_MAX, so cmask_reframe()
* bounds-checking the result against alloc_words catches the overflow instead
* of seeing a small value.
*/
#define CMASK_NR_WORDS(nr_cids) ((u32)(((u64)(nr_cids) + 63) / 64 + 1))
static __always_inline bool __cmask_contains(u32 cid, const struct scx_cmask __arena *m)
{
return cid >= m->base && cid < m->base + m->nr_cids;
}
static __always_inline u64 __arena *__cmask_word(u32 cid, const struct scx_cmask __arena *m)
{
return (u64 __arena *)&m->bits[cid / 64 - m->base / 64];
}
/**
* __cmask_init - Initialize @m with explicit storage capacity
* @m: cmask to initialize
* @base: first cid of the active range
* @nr_cids: number of cids in the active range
* @alloc_cids: storage capacity in cids, at least @nr_cids
*
* Use when storage is sized larger than the initial active range. All of
* bits[] is zeroed.
*/
static __always_inline void __cmask_init(struct scx_cmask __arena *m, u32 base,
u32 nr_cids, u32 alloc_cids)
{
u32 alloc_words, i;
if (unlikely(nr_cids > alloc_cids)) {
scx_bpf_error("__cmask_init: nr_cids=%u exceeds alloc_cids=%u",
nr_cids, alloc_cids);
return;
}
alloc_words = CMASK_NR_WORDS(alloc_cids);
m->base = base;
m->nr_cids = nr_cids;
m->alloc_words = alloc_words;
bpf_for(i, 0, CMASK_MAX_WORDS) {
if (i >= alloc_words)
break;
m->bits[i] = 0;
}
}
/**
* cmask_init - Initialize @m on tight storage
* @m: cmask to initialize
* @base: first cid of the active range
* @nr_cids: number of cids in the active range
*
* All of bits[] is zeroed.
*/
static __always_inline void cmask_init(struct scx_cmask __arena *m, u32 base, u32 nr_cids)
{
__cmask_init(m, base, nr_cids, nr_cids);
}
/**
* cmask_reframe - Reshape @m's active range without resizing storage
* @m: cmask to reframe
* @base: new active range base
* @nr_cids: new active range length, must fit within @m->alloc_words
*
* Body bits within the new range become garbage - only the head and tail
* words are zeroed to keep the padding invariant.
*/
static __always_inline void cmask_reframe(struct scx_cmask __arena *m, u32 base, u32 nr_cids)
{
if (CMASK_NR_WORDS(nr_cids) > m->alloc_words) {
scx_bpf_error("cmask_reframe: nr_cids=%u exceeds alloc_words=%u",
nr_cids, m->alloc_words);
return;
}
if (nr_cids) {
u32 last_word = ((base & 63) + nr_cids - 1) / 64;
m->bits[0] = 0;
m->bits[last_word] = 0;
}
m->base = base;
m->nr_cids = nr_cids;
}
static __always_inline bool cmask_test(u32 cid, const struct scx_cmask __arena *m)
{
if (!__cmask_contains(cid, m))
return false;
return *__cmask_word(cid, m) & BIT_U64(cid & 63);
}
/*
* x86 BPF JIT rejects BPF_OR | BPF_FETCH and BPF_AND | BPF_FETCH on arena
* pointers (see bpf_jit_supports_insn() in arch/x86/net/bpf_jit_comp.c). Only
* BPF_CMPXCHG / BPF_XCHG / BPF_ADD with FETCH are allowed. Implement
* test_and_{set,clear} and the atomic set/clear via a cmpxchg loop.
*
* CMASK_CAS_TRIES is sized so exhausting it means seconds of real spinning
* on one word - past any plausible contention. Abort hard.
*/
#define CMASK_CAS_TRIES (1U << 23)
static __always_inline void cmask_set(u32 cid, struct scx_cmask __arena *m)
{
u64 __arena *w;
u64 bit, old, new;
u32 i;
if (!__cmask_contains(cid, m))
return;
w = __cmask_word(cid, m);
bit = BIT_U64(cid & 63);
bpf_for(i, 0, CMASK_CAS_TRIES) {
old = *w;
if (old & bit)
return;
new = old | bit;
if (__sync_val_compare_and_swap(w, old, new) == old)
return;
}
scx_bpf_error("cmask_set CAS exhausted at cid %u", cid);
}
static __always_inline void cmask_clear(u32 cid, struct scx_cmask __arena *m)
{
u64 __arena *w;
u64 bit, old, new;
u32 i;
if (!__cmask_contains(cid, m))
return;
w = __cmask_word(cid, m);
bit = BIT_U64(cid & 63);
bpf_for(i, 0, CMASK_CAS_TRIES) {
old = *w;
if (!(old & bit))
return;
new = old & ~bit;
if (__sync_val_compare_and_swap(w, old, new) == old)
return;
}
scx_bpf_error("cmask_clear CAS exhausted at cid %u", cid);
}
static __always_inline bool cmask_test_and_set(u32 cid, struct scx_cmask __arena *m)
{
u64 __arena *w;
u64 bit, old, new;
u32 i;
if (!__cmask_contains(cid, m))
return false;
w = __cmask_word(cid, m);
bit = BIT_U64(cid & 63);
bpf_for(i, 0, CMASK_CAS_TRIES) {
old = *w;
if (old & bit)
return true;
new = old | bit;
if (__sync_val_compare_and_swap(w, old, new) == old)
return false;
}
scx_bpf_error("cmask_test_and_set CAS exhausted at cid %u", cid);
return false;
}
static __always_inline bool cmask_test_and_clear(u32 cid, struct scx_cmask __arena *m)
{
u64 __arena *w;
u64 bit, old, new;
u32 i;
if (!__cmask_contains(cid, m))
return false;
w = __cmask_word(cid, m);
bit = BIT_U64(cid & 63);
bpf_for(i, 0, CMASK_CAS_TRIES) {
old = *w;
if (!(old & bit))
return false;
new = old & ~bit;
if (__sync_val_compare_and_swap(w, old, new) == old)
return true;
}
scx_bpf_error("cmask_test_and_clear CAS exhausted at cid %u", cid);
return false;
}
static __always_inline void __cmask_set(u32 cid, struct scx_cmask __arena *m)
{
if (!__cmask_contains(cid, m))
return;
*__cmask_word(cid, m) |= BIT_U64(cid & 63);
}
static __always_inline void __cmask_clear(u32 cid, struct scx_cmask __arena *m)
{
if (!__cmask_contains(cid, m))
return;
*__cmask_word(cid, m) &= ~BIT_U64(cid & 63);
}
static __always_inline bool __cmask_test_and_set(u32 cid, struct scx_cmask __arena *m)
{
u64 bit = BIT_U64(cid & 63);
u64 __arena *w;
u64 prev;
if (!__cmask_contains(cid, m))
return false;
w = __cmask_word(cid, m);
prev = *w & bit;
*w |= bit;
return prev;
}
static __always_inline bool __cmask_test_and_clear(u32 cid, struct scx_cmask __arena *m)
{
u64 bit = BIT_U64(cid & 63);
u64 __arena *w;
u64 prev;
if (!__cmask_contains(cid, m))
return false;
w = __cmask_word(cid, m);
prev = *w & bit;
*w &= ~bit;
return prev;
}
static __always_inline void cmask_zero(struct scx_cmask __arena *m)
{
u32 nr_words = CMASK_NR_WORDS(m->nr_cids), i;
bpf_for(i, 0, CMASK_MAX_WORDS) {
if (i >= nr_words)
break;
m->bits[i] = 0;
}
}
/*
* BPF_-prefixed to avoid colliding with the kernel's anonymous CMASK_OP_*
* enum in ext_cid.c, which is exported via BTF and reachable through
* vmlinux.h.
*/
enum {
BPF_CMASK_OP_AND,
BPF_CMASK_OP_OR,
BPF_CMASK_OP_COPY,
BPF_CMASK_OP_ANDNOT,
};
static __always_inline void cmask_op_word(struct scx_cmask __arena *dst,
const struct scx_cmask __arena *src,
u32 di, u32 si, u64 mask, int op)
{
u64 dv = dst->bits[di];
u64 sv = src->bits[si];
u64 rv;
if (op == BPF_CMASK_OP_AND)
rv = dv & sv;
else if (op == BPF_CMASK_OP_OR)
rv = dv | sv;
else if (op == BPF_CMASK_OP_ANDNOT)
rv = dv & ~sv;
else
rv = sv;
dst->bits[di] = (dv & ~mask) | (rv & mask);
}
static __always_inline void cmask_op(struct scx_cmask __arena *dst,
const struct scx_cmask __arena *src, int op)
{
u32 d_end = dst->base + dst->nr_cids;
u32 s_end = src->base + src->nr_cids;
u32 lo = dst->base > src->base ? dst->base : src->base;
u32 hi = d_end < s_end ? d_end : s_end;
u32 d_base = dst->base / 64;
u32 s_base = src->base / 64;
u32 lo_word, hi_word, i;
u64 head_mask, tail_mask;
if (lo >= hi)
return;
lo_word = lo / 64;
hi_word = (hi - 1) / 64;
head_mask = GENMASK_U64(63, lo & 63);
tail_mask = GENMASK_U64((hi - 1) & 63, 0);
bpf_for(i, 0, CMASK_MAX_WORDS) {
u32 w = lo_word + i;
u64 m;
if (w > hi_word)
break;
m = GENMASK_U64(63, 0);
if (w == lo_word)
m &= head_mask;
if (w == hi_word)
m &= tail_mask;
cmask_op_word(dst, src, w - d_base, w - s_base, m, op);
}
}
/*
* cmask_and/or/copy only modify @dst bits that lie in the intersection of
* [@dst->base, @dst->base + @dst->nr_cids) and [@src->base,
* @src->base + @src->nr_cids). Bits in @dst outside that window
* keep their prior values - in particular, cmask_copy() does NOT zero @dst
* bits that lie outside @src's range.
*/
static __always_inline void cmask_and(struct scx_cmask __arena *dst,
const struct scx_cmask __arena *src)
{
cmask_op(dst, src, BPF_CMASK_OP_AND);
}
static __always_inline void cmask_or(struct scx_cmask __arena *dst,
const struct scx_cmask __arena *src)
{
cmask_op(dst, src, BPF_CMASK_OP_OR);
}
static __always_inline void cmask_copy(struct scx_cmask __arena *dst,
const struct scx_cmask __arena *src)
{
cmask_op(dst, src, BPF_CMASK_OP_COPY);
}
static __always_inline void cmask_andnot(struct scx_cmask __arena *dst,
const struct scx_cmask __arena *src)
{
cmask_op(dst, src, BPF_CMASK_OP_ANDNOT);
}
/*
* True iff @a and @b have identical bits over their (assumed equal) range.
* Callers are expected to pass same-shape cmasks; differing shapes always
* compare unequal.
*/
static __always_inline bool cmask_equal(const struct scx_cmask __arena *a,
const struct scx_cmask __arena *b)
{
u32 nr_words, i;
if (a->base != b->base || a->nr_cids != b->nr_cids)
return false;
nr_words = CMASK_NR_WORDS(a->nr_cids);
bpf_for(i, 0, CMASK_MAX_WORDS) {
if (i >= nr_words)
break;
if (a->bits[i] != b->bits[i])
return false;
}
return true;
}
/*
* True iff every bit set in @a is also set in @b over the intersection of
* their ranges. Bits of @a outside @b's range fail the test.
*/
static __always_inline bool cmask_subset(const struct scx_cmask __arena *a,
const struct scx_cmask __arena *b)
{
u32 a_end = a->base + a->nr_cids;
u32 b_end = b->base + b->nr_cids;
u32 a_wbase = a->base / 64;
u32 b_wbase = b->base / 64;
u32 nr_words, i;
/* any bit of @a outside @b's range is a subset violation */
if (a->base < b->base || a_end > b_end)
return false;
nr_words = CMASK_NR_WORDS(a->nr_cids);
bpf_for(i, 0, CMASK_MAX_WORDS) {
u32 wi_b;
if (i >= nr_words)
break;
wi_b = a_wbase + i - b_wbase;
if (a->bits[i] & ~b->bits[wi_b])
return false;
}
return true;
}
/**
* cmask_next_set - find the first set bit at or after @cid
* @m: cmask to search
* @cid: starting cid (clamped to @m->base if below)
*
* Returns the smallest set cid in [@cid, @m->base + @m->nr_cids), or
* @m->base + @m->nr_cids if none (the out-of-range sentinel matches the
* termination condition used by cmask_for_each()).
*/
static __always_inline u32 cmask_next_set(const struct scx_cmask __arena *m, u32 cid)
{
u32 end = m->base + m->nr_cids;
u32 base = m->base / 64;
u32 last_wi = (end - 1) / 64 - base;
u32 start_wi, start_bit, i;
if (cid < m->base)
cid = m->base;
if (cid >= end)
return end;
start_wi = cid / 64 - base;
start_bit = cid & 63;
bpf_for(i, 0, CMASK_MAX_WORDS) {
u32 wi = start_wi + i;
u64 word;
u32 found;
if (wi > last_wi)
break;
word = m->bits[wi];
if (i == 0)
word &= GENMASK_U64(63, start_bit);
if (!word)
continue;
found = (base + wi) * 64 + ctzll(word);
if (found >= end)
return end;
return found;
}
return end;
}
static __always_inline u32 cmask_first_set(const struct scx_cmask __arena *m)
{
return cmask_next_set(m, m->base);
}
#define cmask_for_each(cid, m) \
for ((cid) = cmask_first_set(m); \
(cid) < (m)->base + (m)->nr_cids; \
(cid) = cmask_next_set((m), (cid) + 1))
/*
* Population count over [base, base + nr_cids). Padding bits in the head/tail
* words are guaranteed zero by the mutating helpers, so a flat popcount over
* all words is correct.
*/
static __always_inline u32 cmask_weight(const struct scx_cmask __arena *m)
{
u32 nr_words = CMASK_NR_WORDS(m->nr_cids), i;
u32 count = 0;
bpf_for(i, 0, CMASK_MAX_WORDS) {
if (i >= nr_words)
break;
count += __builtin_popcountll(m->bits[i]);
}
return count;
}
/*
* True if @a and @b share any set bit. Walk only the intersection of their
* ranges, matching the semantics of cmask_and().
*/
static __always_inline bool cmask_intersects(const struct scx_cmask __arena *a,
const struct scx_cmask __arena *b)
{
u32 a_end = a->base + a->nr_cids;
u32 b_end = b->base + b->nr_cids;
u32 lo = a->base > b->base ? a->base : b->base;
u32 hi = a_end < b_end ? a_end : b_end;
u32 a_base = a->base / 64;
u32 b_base = b->base / 64;
u32 lo_word, hi_word, i;
u64 head_mask, tail_mask;
if (lo >= hi)
return false;
lo_word = lo / 64;
hi_word = (hi - 1) / 64;
head_mask = GENMASK_U64(63, lo & 63);
tail_mask = GENMASK_U64((hi - 1) & 63, 0);
bpf_for(i, 0, CMASK_MAX_WORDS) {
u32 w = lo_word + i;
u64 mask, av, bv;
if (w > hi_word)
break;
mask = GENMASK_U64(63, 0);
if (w == lo_word)
mask &= head_mask;
if (w == hi_word)
mask &= tail_mask;
av = a->bits[w - a_base] & mask;
bv = b->bits[w - b_base] & mask;
if (av & bv)
return true;
}
return false;
}
/*
* Find the next cid set in both @a and @b at or after @start, bounded by the
* intersection of the two ranges. Return a->base + a->nr_cids if none found.
*
* Building block for cmask_next_and_set_wrap(). Callers that want a bounded
* scan without wrap call this directly.
*/
static __always_inline u32 cmask_next_and_set(const struct scx_cmask __arena *a,
const struct scx_cmask __arena *b,
u32 start)
{
u32 a_end = a->base + a->nr_cids;
u32 b_end = b->base + b->nr_cids;
u32 a_wbase = a->base / 64;
u32 b_wbase = b->base / 64;
u32 lo = a->base > b->base ? a->base : b->base;
u32 hi = a_end < b_end ? a_end : b_end;
u32 last_wi, start_wi, start_bit, i;
if (lo >= hi)
return a_end;
if (start < lo)
start = lo;
if (start >= hi)
return a_end;
last_wi = (hi - 1) / 64;
start_wi = start / 64;
start_bit = start & 63;
bpf_for(i, 0, CMASK_MAX_WORDS) {
u32 abs_wi = start_wi + i;
u64 word;
u32 found;
if (abs_wi > last_wi)
break;
word = a->bits[abs_wi - a_wbase] & b->bits[abs_wi - b_wbase];
if (i == 0)
word &= GENMASK_U64(63, start_bit);
if (!word)
continue;
found = abs_wi * 64 + ctzll(word);
if (found >= hi)
return a_end;
return found;
}
return a_end;
}
/*
* Find the next set cid in @m at or after @start, wrapping to @m->base if no
* set bit is found in [start, m->base + m->nr_cids). Return m->base +
* m->nr_cids if @m is empty.
*
* Callers do round-robin distribution by passing (last_cid + 1) as @start.
*/
static __always_inline u32 cmask_next_set_wrap(const struct scx_cmask __arena *m,
u32 start)
{
u32 end = m->base + m->nr_cids;
u32 found;
found = cmask_next_set(m, start);
if (found < end || start <= m->base)
return found;
found = cmask_next_set(m, m->base);
return found < start ? found : end;
}
/*
* Find the next cid set in both @a and @b at or after @start, wrapping to
* @a->base if none found in the forward half. Return a->base + a->nr_cids
* if the intersection is empty.
*
* Callers do round-robin distribution by passing (last_cid + 1) as @start.
*/
static __always_inline u32 cmask_next_and_set_wrap(const struct scx_cmask __arena *a,
const struct scx_cmask __arena *b,
u32 start)
{
u32 a_end = a->base + a->nr_cids;
u32 found;
found = cmask_next_and_set(a, b, start);
if (found < a_end || start <= a->base)
return found;
found = cmask_next_and_set(a, b, a->base);
return found < start ? found : a_end;
}
/**
* cmask_from_cpumask - translate a kernel cpumask to a cid-space cmask
* @m: cmask to fill. Zeroed first; only bits within [@m->base, @m->base +
* @m->nr_cids) are updated - cpus mapping to cids outside that range
* are ignored.
* @cpumask: kernel cpumask to translate
*
* For each cpu in @cpumask, set the cpu's cid in @m. Caller must ensure
* @cpumask stays stable across the call (e.g. RCU read lock for
* task->cpus_ptr).
*/
static __always_inline void cmask_from_cpumask(struct scx_cmask __arena *m,
const struct cpumask *cpumask)
{
u32 nr_cpu_ids = scx_bpf_nr_cpu_ids();
s32 cpu;
cmask_zero(m);
bpf_for(cpu, 0, nr_cpu_ids) {
s32 cid;
if (!bpf_cpumask_test_cpu(cpu, cpumask))
continue;
cid = scx_bpf_cpu_to_cid(cpu);
if (cid >= 0)
__cmask_set(cid, m);
}
}
#endif /* __SCX_CID_BPF_H */

View File

@ -99,8 +99,21 @@ s32 scx_bpf_task_cpu(const struct task_struct *p) __ksym;
struct rq *scx_bpf_cpu_rq(s32 cpu) __ksym;
struct rq *scx_bpf_locked_rq(void) __ksym;
struct task_struct *scx_bpf_cpu_curr(s32 cpu) __ksym __weak;
struct task_struct *scx_bpf_tid_to_task(u64 tid) __ksym __weak;
u64 scx_bpf_now(void) __ksym __weak;
void scx_bpf_events(struct scx_event_stats *events, size_t events__sz) __ksym __weak;
s32 scx_bpf_cpu_to_cid(s32 cpu) __ksym __weak;
s32 scx_bpf_cid_to_cpu(s32 cid) __ksym __weak;
void scx_bpf_cid_topo(s32 cid, struct scx_cid_topo *out) __ksym __weak;
s32 scx_bpf_kick_cid(s32 cid, u64 flags) __ksym __weak;
s32 scx_bpf_task_cid(const struct task_struct *p) __ksym __weak;
s32 scx_bpf_this_cid(void) __ksym __weak;
struct task_struct *scx_bpf_cid_curr(s32 cid) __ksym __weak;
u32 scx_bpf_nr_cids(void) __ksym __weak;
u32 scx_bpf_nr_online_cids(void) __ksym __weak;
u32 scx_bpf_cidperf_cap(s32 cid) __ksym __weak;
u32 scx_bpf_cidperf_cur(s32 cid) __ksym __weak;
void scx_bpf_cidperf_set(s32 cid, u32 perf) __ksym __weak;
/*
* Use the following as @it__iter when calling scx_bpf_dsq_move[_vtime]() from
@ -526,6 +539,10 @@ static inline bool is_migration_disabled(const struct task_struct *p)
void bpf_rcu_read_lock(void) __ksym;
void bpf_rcu_read_unlock(void) __ksym;
/* resilient qspinlock */
int bpf_res_spin_lock(struct bpf_res_spin_lock *lock) __ksym __weak;
void bpf_res_spin_unlock(struct bpf_res_spin_lock *lock) __ksym __weak;
/*
* Time helpers, most of which are from jiffies.h.
*/
@ -1035,7 +1052,18 @@ static inline u64 scx_clock_irq(u32 cpu)
return irqt ? BPF_CORE_READ(irqt, total) : 0;
}
/* Abbreviated forms of <linux/overflow.h>'s struct_size() family. */
#define flex_array_size(p, member, count) \
((count) * sizeof(*(p)->member))
#define struct_size(p, member, count) \
(offsetof(typeof(*(p)), member) + flex_array_size(p, member, count))
#define struct_size_t(type, member, count) \
struct_size((type *)NULL, member, count)
#include "compat.bpf.h"
#include "enums.bpf.h"
#include "cid.bpf.h"
#endif /* __SCX_COMMON_BPF_H */

View File

@ -121,6 +121,18 @@ static inline bool scx_bpf_sub_dispatch(u64 cgroup_id)
return false;
}
/*
* v7.2: scx_bpf_cid_override() for explicit cpu->cid mapping. Ignore if
* missing.
*/
void scx_bpf_cid_override___compat(const s32 *cpu_to_cid, u32 cpu_to_cid__sz) __ksym __weak;
static inline void scx_bpf_cid_override(const s32 *cpu_to_cid, u32 cpu_to_cid__sz)
{
if (bpf_ksym_exists(scx_bpf_cid_override___compat))
return scx_bpf_cid_override___compat(cpu_to_cid, cpu_to_cid__sz);
}
/**
* __COMPAT_is_enq_cpu_selected - Test if SCX_ENQ_CPU_SELECTED is on
* in a compatible way. We will preserve this __COMPAT helper until v6.16.
@ -423,8 +435,10 @@ static inline void scx_bpf_dsq_reenq(u64 dsq_id, u64 reenq_flags)
}
/*
* Define sched_ext_ops. This may be expanded to define multiple variants for
* backward compatibility. See compat.h::SCX_OPS_LOAD/ATTACH().
* Define sched_ext_ops. See compat.h::SCX_OPS_OPEN() for how backward
* compatibility is handled (this macro can be expanded to emit multiple
* variants for incompatible op changes; SCX_OPS_OPEN() handles purely
* additive changes at load time).
*/
#define SCX_OPS_DEFINE(__name, ...) \
SEC(".struct_ops.link") \
@ -432,4 +446,16 @@ static inline void scx_bpf_dsq_reenq(u64 dsq_id, u64 reenq_flags)
__VA_ARGS__, \
};
/*
* Define a cid-form sched_ext_ops. Programs targeting this struct_ops type
* use cid-form callback signatures (select_cid, set_cmask, cid_online/offline,
* dispatch with cid arg, etc.) and may only call the cid-form scx_bpf_*
* kfuncs (kick_cid, task_cid, this_cid, ...).
*/
#define SCX_OPS_CID_DEFINE(__name, ...) \
SEC(".struct_ops.link") \
struct sched_ext_ops_cid __name = { \
__VA_ARGS__, \
};
#endif /* __SCX_COMPAT_BPF_H */

View File

@ -149,10 +149,24 @@ static inline long scx_hotplug_seq(void)
}
/*
* struct sched_ext_ops can change over time. If compat.bpf.h::SCX_OPS_DEFINE()
* is used to define ops and compat.h::SCX_OPS_LOAD/ATTACH() are used to load
* and attach it, backward compatibility is automatically maintained where
* reasonable.
* Open the sched_ext_ops skeleton.
*
* struct sched_ext_ops can change over time. Two complementary mechanisms
* keep BPF schedulers built against newer headers running on older kernels:
*
* 1. Load-time fix-up (this macro). For each optional ops callback or field
* added to struct sched_ext_ops, an explicit stanza below probes the
* running kernel's BTF via __COMPAT_struct_has_field() and, if the field
* is missing, clears it in the in-memory struct_ops (with a warning to
* stderr) before load. Handles additive changes - a new stanza must be
* added here for each new optional field.
*
* 2. Multi-variant struct_ops via compat.bpf.h::SCX_OPS_DEFINE(). That
* macro can be expanded to emit several variants of struct sched_ext_ops,
* and SCX_OPS_LOAD()/ATTACH() can pick the right one based on what the
* kernel supports. Needed when an existing operation has to change
* incompatibly (e.g. a callback signature changes); the load-time
* fix-up above only handles purely additive changes.
*
* ec7e3b0463e1 ("implement-ops") in https://github.com/sched-ext/sched_ext is
* the current minimum required kernel version.
@ -225,6 +239,7 @@ static inline void __scx_ops_assoc_prog(struct bpf_program *prog,
}
#endif
/* See SCX_OPS_OPEN() above for backward-compatibility handling. */
#define SCX_OPS_LOAD(__skel, __ops_name, __scx_name, __uei_name) ({ \
struct bpf_program *__prog; \
UEI_SET_SIZE(__skel, __ops_name, __uei_name); \

View File

@ -32,6 +32,9 @@
__uei_name##_dump_len, (__ei)->dump); \
if (bpf_core_field_exists((__ei)->exit_code)) \
__uei_name.exit_code = (__ei)->exit_code; \
__uei_name.exit_cpu = -1; \
if (bpf_core_field_exists((__ei)->exit_cpu)) \
__uei_name.exit_cpu = (__ei)->exit_cpu; \
/* use __sync to force memory barrier */ \
__sync_val_compare_and_swap(&__uei_name.kind, __uei_name.kind, \
(__ei)->kind); \

View File

@ -39,6 +39,8 @@
fprintf(stderr, "EXIT: %s", __uei->reason); \
if (__uei->msg[0] != '\0') \
fprintf(stderr, " (%s)", __uei->msg); \
if (__uei->exit_cpu >= 0) \
fprintf(stderr, " on CPU %d", __uei->exit_cpu); \
fputs("\n", stderr); \
__uei->exit_code; \
})

View File

@ -22,6 +22,11 @@ enum uei_sizes {
struct user_exit_info {
int kind;
/*
* CPU that triggered the exit, or -1 if unset (e.g. running on an
* older kernel that does not expose this field).
*/
s32 exit_cpu;
s64 exit_code;
char reason[UEI_REASON_LEN];
char msg[UEI_MSG_LEN];

View File

@ -149,10 +149,14 @@ static bool dispatch_to_cpu(s32 cpu)
}
/*
* If we can't run the task at the top, do the dumb thing and
* bounce it to the fallback dsq.
* If we can't run the task at the top for whatever reason,
* bounce it to the fallback dsq. Also check
* is_migration_disabled() explicitly as p->cpus_ptr may not
* reflect the migration-disabled state yet if
* migrate_disable_switch() hasn't run.
*/
if (!bpf_cpumask_test_cpu(cpu, p->cpus_ptr)) {
if (!bpf_cpumask_test_cpu(cpu, p->cpus_ptr) ||
(is_migration_disabled(p) && scx_bpf_task_cpu(p) != cpu)) {
__sync_fetch_and_add(&nr_mismatches, 1);
scx_bpf_dsq_insert(p, FALLBACK_DSQ_ID, SCX_SLICE_INF, 0);
bpf_task_release(p);

View File

@ -18,8 +18,6 @@
char _license[] SEC("license") = "GPL";
const volatile u32 nr_cpus = 32; /* !0 for veristat, set during init */
UEI_DEFINE(uei);
/*

View File

@ -72,8 +72,6 @@ int main(int argc, char **argv)
optind = 1;
skel = SCX_OPS_OPEN(cpu0_ops, scx_cpu0);
skel->rodata->nr_cpus = libbpf_num_possible_cpus();
while ((opt = getopt(argc, argv, "vh")) != -1) {
switch (opt) {
case 'v':

View File

@ -130,7 +130,6 @@ int main(int argc, char **argv)
struct scx_flatcg *skel;
struct bpf_link *link;
struct timespec intv_ts = { .tv_sec = 2, .tv_nsec = 0 };
bool dump_cgrps = false;
__u64 last_cpu_sum = 0, last_cpu_idle = 0;
__u64 last_stats[FCG_NR_STATS] = {};
unsigned long seq = 0;
@ -148,7 +147,7 @@ int main(int argc, char **argv)
assert(skel->rodata->nr_cpus > 0);
skel->rodata->cgrp_slice_ns = __COMPAT_ENUM_OR_ZERO("scx_public_consts", "SCX_SLICE_DFL");
while ((opt = getopt(argc, argv, "s:i:dfvh")) != -1) {
while ((opt = getopt(argc, argv, "s:i:fvh")) != -1) {
double v;
switch (opt) {
@ -161,9 +160,6 @@ int main(int argc, char **argv)
intv_ts.tv_sec = v;
intv_ts.tv_nsec = (v - (float)intv_ts.tv_sec) * 1000000000;
break;
case 'd':
dump_cgrps = true;
break;
case 'f':
skel->rodata->fifo_sched = true;
break;
@ -177,10 +173,10 @@ int main(int argc, char **argv)
}
}
printf("slice=%.1lfms intv=%.1lfs dump_cgrps=%d",
printf("slice=%.1lfms intv=%.1lfs",
(double)skel->rodata->cgrp_slice_ns / 1000000.0,
(double)intv_ts.tv_sec + (double)intv_ts.tv_nsec / 1000000000.0,
dump_cgrps);
(double)intv_ts.tv_sec + (double)intv_ts.tv_nsec / 1000000000.0);
SCX_OPS_LOAD(skel, flatcg_ops, scx_flatcg, uei);
link = SCX_OPS_ATTACH(skel, flatcg_ops, scx_flatcg);

File diff suppressed because it is too large Load Diff

View File

@ -10,9 +10,11 @@
#include <inttypes.h>
#include <signal.h>
#include <libgen.h>
#include <sys/mman.h>
#include <sys/stat.h>
#include <bpf/bpf.h>
#include <scx/common.h>
#include "scx_qmap.h"
#include "scx_qmap.bpf.skel.h"
const char help_fmt[] =
@ -21,23 +23,27 @@ const char help_fmt[] =
"See the top-level comment in .bpf.c for more details.\n"
"\n"
"Usage: %s [-s SLICE_US] [-e COUNT] [-t COUNT] [-T COUNT] [-l COUNT] [-b COUNT]\n"
" [-P] [-M] [-H] [-d PID] [-D LEN] [-S] [-p] [-I] [-F COUNT] [-v]\n"
" [-N COUNT] [-P] [-M] [-H] [-c CG_PATH] [-d PID] [-D LEN] [-S] [-p] [-I]\n"
" [-F COUNT] [-v]\n"
"\n"
" -s SLICE_US Override slice duration\n"
" -e COUNT Trigger scx_bpf_error() after COUNT enqueues\n"
" -t COUNT Stall every COUNT'th user thread\n"
" -T COUNT Stall every COUNT'th kernel thread\n"
" -N COUNT Size of the task_ctx arena slab (default 16384)\n"
" -l COUNT Trigger dispatch infinite looping after COUNT dispatches\n"
" -b COUNT Dispatch upto COUNT tasks together\n"
" -P Print out DSQ content and event counters to trace_pipe every second\n"
" -M Print out debug messages to trace_pipe\n"
" -H Boost nice -20 tasks in SHARED_DSQ, use with -b\n"
" -c CG_PATH Cgroup path to attach as sub-scheduler, must run parent scheduler first\n"
" -d PID Disallow a process from switching into SCHED_EXT (-1 for self)\n"
" -D LEN Set scx_exit_info.dump buffer length\n"
" -S Suppress qmap-specific debug dump\n"
" -p Switch only tasks on SCHED_EXT policy instead of all\n"
" -I Turn on SCX_OPS_ALWAYS_ENQ_IMMED\n"
" -F COUNT IMMED stress: force every COUNT'th enqueue to a busy local DSQ (use with -I)\n"
" -C MODE cid-override test (shuffle|bad-dup|bad-range)\n"
" -v Print libbpf debug messages\n"
" -h Display this help and exit\n";
@ -60,23 +66,36 @@ int main(int argc, char **argv)
{
struct scx_qmap *skel;
struct bpf_link *link;
struct qmap_arena *qa;
__u32 test_error_cnt = 0;
__u64 ecode;
int opt;
libbpf_set_print(libbpf_print_fn);
signal(SIGINT, sigint_handler);
signal(SIGTERM, sigint_handler);
if (libbpf_num_possible_cpus() > SCX_QMAP_MAX_CPUS) {
fprintf(stderr,
"scx_qmap: %d possible CPUs exceeds compile-time cap %d; "
"rebuild with larger SCX_QMAP_MAX_CPUS\n",
libbpf_num_possible_cpus(), SCX_QMAP_MAX_CPUS);
return 1;
}
restart:
optind = 1;
skel = SCX_OPS_OPEN(qmap_ops, scx_qmap);
skel->rodata->slice_ns = __COMPAT_ENUM_OR_ZERO("scx_public_consts", "SCX_SLICE_DFL");
skel->rodata->max_tasks = 16384;
while ((opt = getopt(argc, argv, "s:e:t:T:l:b:PMHc:d:D:SpIF:vh")) != -1) {
while ((opt = getopt(argc, argv, "s:e:t:T:l:b:N:PMHc:d:D:SpIF:C:vh")) != -1) {
switch (opt) {
case 's':
skel->rodata->slice_ns = strtoull(optarg, NULL, 0) * 1000;
break;
case 'e':
skel->bss->test_error_cnt = strtoul(optarg, NULL, 0);
test_error_cnt = strtoul(optarg, NULL, 0);
break;
case 't':
skel->rodata->stall_user_nth = strtoul(optarg, NULL, 0);
@ -90,6 +109,9 @@ int main(int argc, char **argv)
case 'b':
skel->rodata->dsp_batch = strtoul(optarg, NULL, 0);
break;
case 'N':
skel->rodata->max_tasks = strtoul(optarg, NULL, 0);
break;
case 'P':
skel->rodata->print_dsqs_and_events = true;
break;
@ -130,6 +152,35 @@ int main(int argc, char **argv)
case 'F':
skel->rodata->immed_stress_nth = strtoul(optarg, NULL, 0);
break;
case 'C': {
u32 nr_cpus = libbpf_num_possible_cpus();
u32 mode, i;
if (!strcmp(optarg, "shuffle"))
mode = 1;
else if (!strcmp(optarg, "bad-dup"))
mode = 2;
else if (!strcmp(optarg, "bad-range"))
mode = 3;
else {
fprintf(stderr, "unknown cid-override mode '%s'\n", optarg);
return 1;
}
skel->rodata->cid_override_mode = mode;
/* shuffle: reversed cpu_to_cid, bad-dup: dup cid 0, bad-range: identity */
for (i = 0; i < nr_cpus; i++) {
if (mode == 1)
skel->bss->cid_override_cpu_to_cid[i] = nr_cpus - 1 - i;
else
skel->bss->cid_override_cpu_to_cid[i] = i;
}
if (mode == 2 && nr_cpus >= 2)
skel->bss->cid_override_cpu_to_cid[1] = 0;
if (mode == 3)
skel->bss->cid_override_cpu_to_cid[0] = (s32)nr_cpus;
break;
}
case 'v':
verbose = true;
break;
@ -142,39 +193,41 @@ int main(int argc, char **argv)
SCX_OPS_LOAD(skel, qmap_ops, scx_qmap, uei);
link = SCX_OPS_ATTACH(skel, qmap_ops, scx_qmap);
while (!exit_req && !UEI_EXITED(skel, uei)) {
long nr_enqueued = skel->bss->nr_enqueued;
long nr_dispatched = skel->bss->nr_dispatched;
qa = &skel->arena->qa;
qa->test_error_cnt = test_error_cnt;
printf("stats : enq=%lu dsp=%lu delta=%ld reenq/cpu0=%"PRIu64"/%"PRIu64" deq=%"PRIu64" core=%"PRIu64" enq_ddsp=%"PRIu64"\n",
while (!exit_req && !UEI_EXITED(skel, uei)) {
long nr_enqueued = qa->nr_enqueued;
long nr_dispatched = qa->nr_dispatched;
printf("stats : enq=%lu dsp=%lu delta=%ld reenq/cid0=%llu/%llu deq=%llu core=%llu enq_ddsp=%llu\n",
nr_enqueued, nr_dispatched, nr_enqueued - nr_dispatched,
skel->bss->nr_reenqueued, skel->bss->nr_reenqueued_cpu0,
skel->bss->nr_dequeued,
skel->bss->nr_core_sched_execed,
skel->bss->nr_ddsp_from_enq);
printf(" exp_local=%"PRIu64" exp_remote=%"PRIu64" exp_timer=%"PRIu64" exp_lost=%"PRIu64"\n",
skel->bss->nr_expedited_local,
skel->bss->nr_expedited_remote,
skel->bss->nr_expedited_from_timer,
skel->bss->nr_expedited_lost);
if (__COMPAT_has_ksym("scx_bpf_cpuperf_cur"))
qa->nr_reenqueued, qa->nr_reenqueued_cid0,
qa->nr_dequeued,
qa->nr_core_sched_execed,
qa->nr_ddsp_from_enq);
printf(" exp_local=%llu exp_remote=%llu exp_timer=%llu exp_lost=%llu\n",
qa->nr_expedited_local,
qa->nr_expedited_remote,
qa->nr_expedited_from_timer,
qa->nr_expedited_lost);
if (__COMPAT_has_ksym("scx_bpf_cidperf_cur"))
printf("cpuperf: cur min/avg/max=%u/%u/%u target min/avg/max=%u/%u/%u\n",
skel->bss->cpuperf_min,
skel->bss->cpuperf_avg,
skel->bss->cpuperf_max,
skel->bss->cpuperf_target_min,
skel->bss->cpuperf_target_avg,
skel->bss->cpuperf_target_max);
qa->cpuperf_min,
qa->cpuperf_avg,
qa->cpuperf_max,
qa->cpuperf_target_min,
qa->cpuperf_target_avg,
qa->cpuperf_target_max);
fflush(stdout);
sleep(1);
}
bpf_link__destroy(link);
UEI_REPORT(skel, uei);
ecode = UEI_REPORT(skel, uei);
scx_qmap__destroy(skel);
/*
* scx_qmap implements ops.cpu_on/offline() and doesn't need to restart
* on CPU hotplug events.
*/
if (UEI_ECODE_RESTART(ecode))
goto restart;
return 0;
}

View File

@ -0,0 +1,73 @@
/* SPDX-License-Identifier: GPL-2.0 */
/*
* Shared definitions between scx_qmap.bpf.c and scx_qmap.c.
*
* The scheduler keeps all state in a single BPF arena map. struct
* qmap_arena is the one object that lives at the base of the arena and is
* mmap'd into userspace so the loader can read counters directly.
*
* Copyright (c) 2026 Meta Platforms, Inc. and affiliates.
* Copyright (c) 2026 Tejun Heo <tj@kernel.org>
*/
#ifndef __SCX_QMAP_H
#define __SCX_QMAP_H
#ifdef __BPF__
#include <scx/bpf_arena_common.bpf.h>
#else
#include <linux/types.h>
#include <scx/bpf_arena_common.h>
#endif
#define MAX_SUB_SCHEDS 8
/*
* cpu_ctxs[] is sized to a fixed cap so the layout is shared between BPF and
* userspace. Keep this in sync with NR_CPUS used by the BPF side.
*/
#define SCX_QMAP_MAX_CPUS 1024
struct cpu_ctx {
__u64 dsp_idx; /* dispatch index */
__u64 dsp_cnt; /* remaining count */
__u32 avg_weight;
__u32 cpuperf_target;
};
/* Opaque to userspace; defined in scx_qmap.bpf.c. */
struct task_ctx;
struct qmap_fifo {
struct task_ctx __arena *head;
struct task_ctx __arena *tail;
__s32 idx;
};
struct qmap_arena {
/* userspace-visible stats */
__u64 nr_enqueued, nr_dispatched, nr_reenqueued, nr_reenqueued_cid0;
__u64 nr_dequeued, nr_ddsp_from_enq;
__u64 nr_core_sched_execed;
__u64 nr_expedited_local, nr_expedited_remote;
__u64 nr_expedited_lost, nr_expedited_from_timer;
__u64 nr_highpri_queued;
__u32 test_error_cnt;
__u32 cpuperf_min, cpuperf_avg, cpuperf_max;
__u32 cpuperf_target_min, cpuperf_target_avg, cpuperf_target_max;
/* kernel-side runtime state */
__u64 sub_sched_cgroup_ids[MAX_SUB_SCHEDS];
__u64 core_sched_head_seqs[5];
__u64 core_sched_tail_seqs[5];
struct cpu_ctx cpu_ctxs[SCX_QMAP_MAX_CPUS];
/* task_ctx slab; allocated and threaded by qmap_init() */
struct task_ctx __arena *task_ctxs;
struct task_ctx __arena *task_free_head;
/* five priority FIFOs, each a doubly-linked list through task_ctx */
struct qmap_fifo fifos[5];
};
#endif /* __SCX_QMAP_H */

View File

@ -9,12 +9,7 @@
* Copyright (C) 2026 Cheng-Yang Chou <yphbchou0911@gmail.com>
*/
#include <vmlinux.h>
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_tracing.h>
/* SCX kfunc from scx_kfunc_ids_any set */
void scx_bpf_kick_cpu(s32 cpu, u64 flags) __ksym;
#include <scx/common.bpf.h>
SEC("struct_ops/ssthresh")
__u32 BPF_PROG(tcp_ca_ssthresh, struct sock *sk)

View File

@ -95,7 +95,7 @@ static int scan_dsq_pool(void)
record_peek_result(task->pid);
/* Try to move this task to local */
if (!moved && scx_bpf_dsq_move_to_local(dsq_id, 0) == 0) {
if (!moved && scx_bpf_dsq_move_to_local(dsq_id, 0)) {
moved = 1;
break;
}

View File

@ -6,6 +6,7 @@
*/
#include <bpf/bpf.h>
#include <scx/common.h>
#include <stdlib.h>
#include <sys/wait.h>
#include <unistd.h>
#include "select_cpu_dfl.bpf.skel.h"
@ -13,29 +14,44 @@
#define NUM_CHILDREN 1028
struct select_cpu_dfl_ctx {
struct select_cpu_dfl *skel;
struct bpf_link *link;
};
static enum scx_test_status setup(void **ctx)
{
struct select_cpu_dfl *skel;
struct select_cpu_dfl_ctx *tctx;
skel = select_cpu_dfl__open();
SCX_FAIL_IF(!skel, "Failed to open");
SCX_ENUM_INIT(skel);
SCX_FAIL_IF(select_cpu_dfl__load(skel), "Failed to load skel");
tctx = malloc(sizeof(*tctx));
SCX_FAIL_IF(!tctx, "Failed to allocate test context");
tctx->link = NULL;
*ctx = skel;
tctx->skel = select_cpu_dfl__open();
if (!tctx->skel) {
free(tctx);
SCX_FAIL("Failed to open");
}
SCX_ENUM_INIT(tctx->skel);
if (select_cpu_dfl__load(tctx->skel)) {
select_cpu_dfl__destroy(tctx->skel);
free(tctx);
SCX_FAIL("Failed to load skel");
}
*ctx = tctx;
return SCX_TEST_PASS;
}
static enum scx_test_status run(void *ctx)
{
struct select_cpu_dfl *skel = ctx;
struct bpf_link *link;
struct select_cpu_dfl_ctx *tctx = ctx;
pid_t pids[NUM_CHILDREN];
int i, status;
int i, status, nforked = 0;
link = bpf_map__attach_struct_ops(skel->maps.select_cpu_dfl_ops);
SCX_FAIL_IF(!link, "Failed to attach scheduler");
tctx->link = bpf_map__attach_struct_ops(tctx->skel->maps.select_cpu_dfl_ops);
SCX_FAIL_IF(!tctx->link, "Failed to attach scheduler");
for (i = 0; i < NUM_CHILDREN; i++) {
pids[i] = fork();
@ -43,25 +59,31 @@ static enum scx_test_status run(void *ctx)
sleep(1);
exit(0);
}
if (pids[i] > 0)
nforked++;
}
for (i = 0; i < NUM_CHILDREN; i++) {
if (pids[i] <= 0)
continue;
SCX_EQ(waitpid(pids[i], &status, 0), pids[i]);
SCX_EQ(status, 0);
}
SCX_ASSERT(!skel->bss->saw_local);
bpf_link__destroy(link);
SCX_GT(nforked, 0);
SCX_ASSERT(!tctx->skel->bss->saw_local);
return SCX_TEST_PASS;
}
static void cleanup(void *ctx)
{
struct select_cpu_dfl *skel = ctx;
struct select_cpu_dfl_ctx *tctx = ctx;
select_cpu_dfl__destroy(skel);
if (tctx->link)
bpf_link__destroy(tctx->link);
select_cpu_dfl__destroy(tctx->skel);
free(tctx);
}
struct scx_test select_cpu_dfl = {