Merge branch 'bpf-fix-per-cpu-initialization-of-a-bpf_f_cpu-created-hash-element'

Donggeun Yoo says:

====================
bpf: fix per-cpu initialization of a BPF_F_CPU created hash element

A BPF_F_CPU update that creates a [lru_]percpu_hash element writes the
named CPU's slot and leaves the others holding the recycled element's
values, so a lookup of the new key returns a deleted key's per-cpu
values.

Patch 1 zero-fills the other CPUs.  Patch 2 adds the selftest: the
existing cpu_flag subtests always prime a key with BPF_F_ALL_CPUS
first, so the create path is not covered today.

v1: https://lore.kernel.org/bpf/20260920093153.439743-1-donggeunyoo.kernel@gmail.com/
v2: https://lore.kernel.org/bpf/20260923000801.1764758-1-donggeunyoo.kernel@gmail.com/

Changes in v3:
 - patch 1: key init_cpu on BPF_F_CPU rather than on onallcpus (Leon
   Hwang)
 - patch 1: re-flow the paragraph naming pcpu_copy_value() (BPF CI AI
   review)
 - patch 2: pin the thread across the delete and the create, and name a
   CPU other than the pinned one, so the BPF_F_NO_PREALLOC arm does not
   depend on staying put (BPF CI AI review)

Changes in v2:
 - patch 1: cover the BPF_F_CPU entry condition in the block comment
   above pcpu_init_value() (BPF CI AI review, Alexei Starovoitov)
 - patch 2: skip the new subtests instead of failing them on a
   uniprocessor machine (Sashiko AI review)
====================

Link: https://patch.msgid.link/20260924102321.2120434-1-donggeunyoo.kernel@gmail.com
Signed-off-by: Alexei Starovoitov <ast@kernel.org>
This commit is contained in:
Alexei Starovoitov 2026-09-24 14:24:52 +00:00
commit 1b1dca5682
No known key found for this signature in database
2 changed files with 112 additions and 4 deletions

View File

@ -1054,14 +1054,17 @@ static void pcpu_init_value(struct bpf_htab *htab, void __percpu *pptr,
/* When not setting the initial value on all cpus, zero-fill element
* values for other cpus. Otherwise, bpf program has no way to ensure
* known initial values for cpus other than current one
* (onallcpus=false always when coming from bpf prog).
* (onallcpus=false always when coming from bpf prog,
* map_flags & BPF_F_CPU when coming from syscall but setting
* only one cpu).
*/
if (!onallcpus) {
int current_cpu = raw_smp_processor_id();
if (!onallcpus || (map_flags & BPF_F_CPU)) {
int init_cpu = (map_flags & BPF_F_CPU) ? map_flags >> 32 :
raw_smp_processor_id();
int cpu;
for_each_possible_cpu(cpu) {
if (cpu == current_cpu)
if (cpu == init_cpu)
copy_map_value(&htab->map, per_cpu_ptr(pptr, cpu), value);
else /* Since elem is preallocated, we cannot touch special fields */
zero_map_value(&htab->map, per_cpu_ptr(pptr, cpu));

View File

@ -1,4 +1,6 @@
// SPDX-License-Identifier: GPL-2.0
#define _GNU_SOURCE
#include <sched.h>
#include <test_progs.h>
#include "cgroup_helpers.h"
#include "percpu_alloc_array.skel.h"
@ -350,6 +352,103 @@ static void test_lru_percpu_hash_cpu_flag(void)
test_percpu_map_cpu_flag(BPF_MAP_TYPE_LRU_PERCPU_HASH);
}
/*
* A BPF_F_CPU update that creates an element must zero the value on the other
* cpus, rather than leave them holding whatever the recycled element last
* contained. max_entries is 1 so the second key can only reuse the element
* the first one released.
*/
static void test_percpu_map_cpu_flag_create(enum bpf_map_type map_type, __u32 map_flags)
{
LIBBPF_OPTS(bpf_map_create_opts, opts, .map_flags = map_flags);
const u32 stale = 0xDEADC0DE, fresh = 0xC0FFEE;
int nr_cpus, cpu, map_fd, err, key;
int pinned_cpu, value_cpu;
cpu_set_t old_mask, new_mask;
bool restore_mask = false;
u32 value;
u64 flags;
nr_cpus = libbpf_num_possible_cpus();
if (!ASSERT_GT(nr_cpus, 0, "libbpf_num_possible_cpus"))
return;
if (nr_cpus < 2) {
test__skip();
return;
}
map_fd = bpf_map_create(map_type, "cpu_flag_create", sizeof(key), sizeof(value), 1, &opts);
if (!ASSERT_GE(map_fd, 0, "bpf_map_create"))
return;
/* NO_PREALLOC recycles per cpu, so keep the delete and the create on one cpu. */
err = sched_getaffinity(0, sizeof(old_mask), &old_mask);
if (!ASSERT_OK(err, "sched_getaffinity"))
goto out;
pinned_cpu = sched_getcpu();
if (!ASSERT_GE(pinned_cpu, 0, "sched_getcpu"))
goto out;
CPU_ZERO(&new_mask);
CPU_SET(pinned_cpu, &new_mask);
err = sched_setaffinity(0, sizeof(new_mask), &new_mask);
if (!ASSERT_OK(err, "sched_setaffinity"))
goto out;
restore_mask = true;
value_cpu = pinned_cpu ? 0 : 1;
key = 1;
value = stale;
err = bpf_map_update_elem(map_fd, &key, &value, BPF_F_ALL_CPUS);
if (!ASSERT_OK(err, "bpf_map_update_elem all_cpus"))
goto out;
err = bpf_map_delete_elem(map_fd, &key);
if (!ASSERT_OK(err, "bpf_map_delete_elem"))
goto out;
key = 2;
value = fresh;
flags = (u64)value_cpu << 32 | BPF_F_CPU;
err = bpf_map_update_elem(map_fd, &key, &value, flags);
if (!ASSERT_OK(err, "bpf_map_update_elem specified cpu"))
goto out;
for (cpu = 0; cpu < nr_cpus; cpu++) {
value = 0;
flags = (u64)cpu << 32 | BPF_F_CPU;
err = bpf_map_lookup_elem_flags(map_fd, &key, &value, flags);
if (!ASSERT_OK(err, "bpf_map_lookup_elem_flags specified cpu"))
goto out;
if (!ASSERT_EQ(value, cpu == value_cpu ? fresh : 0, "value on specified cpu"))
goto out;
}
out:
if (restore_mask)
sched_setaffinity(0, sizeof(old_mask), &old_mask);
close(map_fd);
}
static void test_percpu_hash_cpu_flag_create(void)
{
test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_PERCPU_HASH, 0);
}
static void test_percpu_hash_cpu_flag_create_malloc(void)
{
test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_PERCPU_HASH, BPF_F_NO_PREALLOC);
}
static void test_lru_percpu_hash_cpu_flag_create(void)
{
/* lru without prealloc is -ENOTSUPP, so there is no malloc variant */
test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_LRU_PERCPU_HASH, 0);
}
static void test_percpu_cgroup_storage_cpu_flag(void)
{
struct percpu_alloc_array *skel = NULL;
@ -454,6 +553,12 @@ void test_percpu_alloc(void)
test_percpu_hash_cpu_flag();
if (test__start_subtest("cpu_flag_lru_percpu_hash"))
test_lru_percpu_hash_cpu_flag();
if (test__start_subtest("cpu_flag_create_percpu_hash"))
test_percpu_hash_cpu_flag_create();
if (test__start_subtest("cpu_flag_create_percpu_hash_malloc"))
test_percpu_hash_cpu_flag_create_malloc();
if (test__start_subtest("cpu_flag_create_lru_percpu_hash"))
test_lru_percpu_hash_cpu_flag_create();
if (test__start_subtest("cpu_flag_percpu_cgroup_storage"))
test_percpu_cgroup_storage_cpu_flag();
if (test__start_subtest("cpu_flag_array"))