bpf-fixes

-----BEGIN PGP SIGNATURE-----
 
 iQJRBAABCgA7FiEE+soXsSLHKoYyzcli6rmadz2vbToFAmq1NvAdHGFsZXhlaS5z
 dGFyb3ZvaXRvdkBnbWFpbC5jb20ACgkQ6rmadz2vbTpmpA//fqEoC6Sq1zxo3ADH
 hV0Z9ewkNTjjH85QnispjcRkAhSHG3JscNXKRXm1NmNkwJsHJ4TDIKcjABYDjquD
 wqfvL9hLXPsvud0M/PR6/CZeBAWXpukkaxeYedY+83ttTHjzR0tDq0Ne9yvIV+Nc
 hS0qFIIHs8C6l2nyuNSxvgrv216orG8qd0Bi3tpDfsLqCsLLEmDyQ1H+ZzpJF2xW
 RV3oMcggCeo305m8+uiofQGf8hmmRrmA7SEfF+Qe08ab2GOn9glINfVTbTE3AdQW
 fkvAk8Zuio3hwwMBHDWWYXrKO0N3ykzcDk4V6JPUWiH1dOANf1tS3G1uJyxb4tsv
 dHVdA0xL3sg7YnuSywfb82vTXvQz5QEFeDYxxLB2fMJe8LcCfgF1lkIr1DbdrMjI
 hlozGCs3p/GVIVNhjGVPezWsUvzK4PIuKVM6U1qWojtAhXU/bJCvP0iBU83RlBL7
 S0GGSC/qvkuPed9hAp2pnrrRo/9GEp1PZq8AXsp3nti0OaRxgxpjjQQFzZEWlOrR
 bErtbKziHzl2xARvCRycNZ+QWT4ZdjmK++8pba7hOuFkRclOV4KRWk4OthBvcADu
 NfNsVXdXpOnesg7TNIUJ8heVAYPjKaBHJBuG2Prghrnc95uqEQwOfbYT15zAbe6I
 l3HqGerSa7WtbrnCwsf1DvyPPII=
 =ZxKg
 -----END PGP SIGNATURE-----

Merge tag 'bpf-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf

Pull bpf fixes from Alexei Starovoitov:

 - Fix bpf_skb_change_tail() to drop the checksum offload instead of
   rejecting the trim of CHECKSUM_PARTIAL skbs (Daniel Borkmann)

 - Add KF_PERFMON kfunc flag and require CAP_PERFMON for kfuncs that
   read arbitrary memory and for untrusted read-only memory reads
   (Daniel Borkmann)

 - Clear scalar delta on narrowing stack spill (Daniel Borkmann)

 - Set up the frame pointer for the exception callback in arm64 JIT, and
   zero-fill other CPUs when BPF_F_CPU update creates a per-cpu hash
   element (Donggeun Yoo)

 - Various fixes (Emil Tsalapatis):
     - Fix bounds check underflow for skb-backed dynptrs
     - Fix rx_queue_mapping context access code generation in bpf_sock
     - Reject packet pointer arguments to subprogs that may mutate the
       packet
     - Reject ALU instructions that see arena and non-arena operands on
       different code paths

 - Fix copied_seq double-counting on sockmap self-redirect
   (Geliang Tang)

 - Fix divide-by-zero in btf_struct_walk() on a flexible array of
   zero-sized elements, fix out-of-bounds read of rtt_min in sock_ops
   (Jiayuan Chen)

 - Fix bpf_sock_destroy() out-of-bounds read of sk_protocol on TIME_WAIT
   and request socks, and sleeping under RCU when destroying a listener
   with pending children (Jiayuan Chen)

 - Fix JEQ/JNE with immediate operand in MIPS32 JIT and missing zero
   extension of BSWAP 16/32 in MIPS64 JIT (Johan Almbladh)

 - Avoid soft lockup in htab lookup[_and_delete] batch operations on
   large maps (Jose Fernandez)

 - Various fixes (Kumar Kartikeya Dwivedi):
     - Verify global subprogs in each sleepability context they are
       called from
     - Make post-verification instruction rewrites killable
     - Preserve packet pointer displacement in regsafe()
     - Apply CO-RE relocations before subprogram validation, restrict
       CO-RE poisoning to relocatable instructions, and reject truncated
       ldimm64 CO-RE relocations in libbpf
     - Assign lock identity to callback map values
     - Compare stack frames in regs_exact()
     - Bound ownership depth through local kptrs and graph roots

 - Fix u32 overflow in map batch operations when the map size exceeds
   4GB (Masoud Aghasi)

 - Fix UAF in bpf memalloc due to concurrent consumption of ttrace lists
   in alloc_bulk() (Pu Lehui)

 - Allow gotox as the terminal instruction of a program or a subprogram
   (Siddharth Chintamaneni)

 - Disallow bpf_skb_pull_data() for LWT_SEG6LOCAL, skip unsettled links
   in link iterator, and reject dev-bound-only programs on other devices
   (Weiming Shi)

 - Reject non-negative stack offsets in stack_slot_obj_get_spi()
   (Xu Yunxiang)

 - Check params size before reading reserved fields in
   bpf_crypto_ctx_create() (Yuqi Xu)

 - Reject max_entries > INT_MAX in sock_map_alloc() (Zhao Gongyi)

 - Use a 32-bit compare in xsk_map_gen_lookup() (Zhiling Zou)

 - Use kvfree() in xdp_test_run_teardown() (Zhixing Chen)

* tag 'bpf-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf: (58 commits)
  selftests/bpf: Test per-cpu initialization of a BPF_F_CPU created element
  bpf: Zero-fill other CPUs when BPF_F_CPU creates a per-cpu hash element
  bpf: Fix BSWAP 32 and 16 on MIPS64
  bpf: Fix immediate JMP JEQ/JNE on MIPS32
  bpf: Reject dev-bound-only programs on other devices
  bpf, sockmap: Reject max_entries > INT_MAX in sock_map_alloc
  selftests/bpf: Test for mixed arena/nonarena code paths
  bpf: Prevent variable arena/non-arena register contents
  selftests/bpf: Test rejection of pkt args to mutating subprogs
  bpf: Reject pkt arguments in mutating subprogs
  selftests/bpf: Add selftests for rx_queue_mapping context access
  bpf: Fix bpf_sock context code generation
  selftests/bpf: Test dynptr slices past end of skb
  bpf: Fix bounds check for skb-backed dynptrs
  selftests/bpf: Reject iterator destruction through fp+0
  bpf: Reject non-negative offsets in stack_slot_obj_get_spi()
  bpf: Check params size before reading reserved fields
  selftests/bpf: Check local object ownership depth
  bpf: Bound ownership depth through local kptrs and graph roots
  selftests/bpf: Cover frame changes in bounded loops
  ...
This commit is contained in:
Linus Torvalds 2026-09-24 08:25:26 -07:00
commit 5fc5768c7c
54 changed files with 2134 additions and 270 deletions

View File

@ -486,6 +486,16 @@ Example usage in BPF program:
/* note that the last argument is omitted */
bpf_task_work_schedule_signal(task, &work->tw, &arrmap, task_work_callback);
2.5.10 KF_PERFMON flag
----------------------
The KF_PERFMON flag is used for kfuncs that can expose kernel memory or kernel
addresses to the BPF program, for example by reading through a pointer that the
verifier does not check. Calling such a kfunc requires CAP_PERFMON, or
CAP_SYS_ADMIN, in the same way that the equivalent BPF helpers are gated in
bpf_base_func_proto(). A program loaded with CAP_BPF alone is rejected at load
time.
2.6 Registering the kfuncs
--------------------------

View File

@ -600,6 +600,8 @@ static int build_prologue(struct jit_ctx *ctx, bool ebpf_from_cbpf)
* 12 registers are on the stack
*/
emit(A64_SUB_I(1, A64_SP, A64_FP, 96), ctx);
/* The callback may use its own BPF stack, set up fp for it. */
ctx->fp_used = true;
}
/* Stack must be multiples of 16B */

View File

@ -1111,7 +1111,7 @@ static void emit_jmp_i64(struct jit_context *ctx,
emit(ctx, xor, tmp, lo(dst), tmp);
}
if (imm < 0) { /* Compare sign extension */
emit(ctx, addu, MIPS_R_T9, hi(dst), 1);
emit(ctx, addiu, MIPS_R_T9, hi(dst), 1);
emit(ctx, or, tmp, tmp, MIPS_R_T9);
} else { /* Compare zero extension */
emit(ctx, or, tmp, tmp, hi(dst));

View File

@ -305,8 +305,7 @@ static void emit_bswap_r64(struct jit_context *ctx, u8 dst, u32 width)
case 16:
emit_sext(ctx, dst, dst);
emit_bswap_r(ctx, dst, width);
if (cpu_has_mips64r2 || cpu_has_mips64r6)
emit_zext(ctx, dst);
emit_zext(ctx, dst);
break;
}
clobber_reg(ctx, dst);

View File

@ -1651,8 +1651,9 @@ static inline void bpf_trampoline_set_flags(struct bpf_trampoline *tr, u32 flags
struct bpf_func_info_aux {
u16 linkage;
bool unreliable;
bool called : 1;
bool verified : 1;
/* Indexed by in_sleepable. */
bool called[2];
bool verified[2];
};
enum bpf_jit_poke_reason {

View File

@ -45,18 +45,20 @@ struct bpf_reg_state {
union {
/* valid when type == PTR_TO_PACKET */
int range;
/* valid when type == CONST_PTR_TO_MAP | PTR_TO_MAP_VALUE |
* PTR_TO_MAP_VALUE_OR_NULL
/*
* Valid when type == PTR_TO_STACK. Inside the callee two registers
* can be both PTR_TO_STACK like R1=fp-8 and R2=fp-8, but one of them
* points to this function stack while another to the caller's stack.
* To differentiate them 'frameno' is used which is an index in
* bpf_verifier_state->frame[] array pointing to bpf_func_state.
*/
struct {
struct bpf_map *map_ptr;
/* To distinguish map lookups from outer map
* the map_uid is non-zero for registers
* pointing to inner maps.
*/
u32 map_uid;
};
u8 frameno;
/*
* For CONST_PTR_TO_MAP, PTR_TO_MAP_KEY, PTR_TO_MAP_VALUE and
* PTR_TO_INSN.
*/
struct bpf_map *map_ptr;
/* for PTR_TO_BTF_ID */
struct {
@ -155,13 +157,12 @@ struct bpf_reg_state {
* gets parent_id set to the dynptr's id.
*/
u32 parent_id;
/* Inside the callee two registers can be both PTR_TO_STACK like
* R1=fp-8 and R2=fp-8, but one of them points to this function stack
* while another to the caller's stack. To differentiate them 'frameno'
* is used which is an index in bpf_verifier_state->frame[] array
* pointing to bpf_func_state.
/*
* Distinguishes inner-map lookups and their keys and values. Zero for
* other registers. Kept outside the metadata union for ID remapping
* during state comparisons.
*/
u32 frameno;
u32 map_uid;
/* if (!precise && SCALAR_VALUE) min/max/tnum don't affect safety */
bool precise;
};
@ -679,6 +680,7 @@ struct bpf_insn_aux_data {
bool nospec_result; /* result is unsafe under speculation, nospec must follow */
bool zext_dst; /* this insn zero extends dst reg */
bool needs_zext; /* alu op needs to clear upper bits */
bool prevent_zext; /* alu op cannot be zext (already used with 64-bit scalars) */
bool non_sleepable; /* helper/kfunc may be called from non-sleepable context */
bool is_iter_next; /* bpf_iter_<type>_next() kfunc call */
bool call_with_percpu_alloc_ptr; /* {this,per}_cpu_ptr() with prog percpu alloc */
@ -1197,6 +1199,8 @@ static inline void bpf_trampoline_unpack_key(u64 key, u32 *obj_id, u32 *btf_id)
int bpf_prepare_btf_info(struct bpf_verifier_env *env,
const union bpf_attr *attr, bpfptr_t uattr);
int bpf_check_core_relo(struct bpf_verifier_env *env,
const union bpf_attr *attr, bpfptr_t uattr);
int bpf_check_btf_info(struct bpf_verifier_env *env,
const union bpf_attr *attr, bpfptr_t uattr);
@ -1236,11 +1240,18 @@ static inline int bpf_get_spi(s32 off)
return (-off - 1) / BPF_REG_SIZE;
}
/*
* Return the function state a stack pointer register refers to. frameno
* shares storage with other pointer metadata, so return NULL for any
* other register type instead of indexing frame[] with aliased bytes.
*/
static inline struct bpf_func_state *bpf_func(struct bpf_verifier_env *env,
const struct bpf_reg_state *reg)
{
struct bpf_verifier_state *cur = env->cur_state;
if (reg->type != PTR_TO_STACK)
return NULL;
return cur->frame[reg->frameno];
}

View File

@ -80,6 +80,7 @@
#define KF_ARENA_ARG2 (1 << 15) /* kfunc takes an arena pointer as its second argument */
#define KF_IMPLICIT_ARGS (1 << 16) /* kfunc has implicit arguments supplied by the verifier */
#define KF_SPINLOCK_SAFE (1 << 17) /* kfunc is allowed inside bpf_spin_lock-ed region */
#define KF_PERFMON (1 << 18) /* kfunc requires CAP_PERFMON */
/*
* Tag marking a kernel function as a kfunc. This is meant to minimize the

View File

@ -4372,7 +4372,10 @@ skb_header_pointer_careful(const struct sk_buff *skb, int offset,
static inline void * __must_check
skb_pointer_if_linear(const struct sk_buff *skb, int offset, int len)
{
if (likely(skb_headlen(skb) - offset >= len))
unsigned int uoffset = (unsigned int)offset;
if (likely(uoffset <= skb_headlen(skb) &&
(unsigned int)len <= skb_headlen(skb) - uoffset))
return skb->data + offset;
return NULL;
}

View File

@ -4268,13 +4268,10 @@ int btf_check_and_fixup_fields(const struct btf *btf, struct btf_record *rec)
{
int i;
/* There are three types that signify ownership of some other type:
* kptr_ref, bpf_list_head, bpf_rb_root.
* kptr_ref only supports storing kernel types, which can't store
* references to program allocated local types.
*
* Hence we only need to ensure that bpf_{list_head,rb_root} ownership
* does not form cycles.
/*
* Check fields which require the complete BTF and initialize runtime
* metadata. Ownership relationships are validated after every record has
* been fixed up.
*/
if (IS_ERR_OR_NULL(rec) || !(rec->field_mask & (BPF_GRAPH_ROOT | BPF_UPTR)))
return 0;
@ -4305,53 +4302,90 @@ int btf_check_and_fixup_fields(const struct btf *btf, struct btf_record *rec)
if (!meta)
return -EFAULT;
rec->fields[i].graph_root.value_rec = meta->record;
/* We need to set value_rec for all root types, but no need
* to check ownership cycle for a type unless it's also a
* node type.
*/
if (!(rec->field_mask & BPF_GRAPH_NODE))
continue;
/* We need to ensure ownership acyclicity among all types. The
* proper way to do it would be to topologically sort all BTF
* IDs based on the ownership edges, since there can be multiple
* bpf_{list_head,rb_node} in a type. Instead, we use the
* following resaoning:
*
* - A type can only be owned by another type in user BTF if it
* has a bpf_{list,rb}_node. Let's call these node types.
* - A type can only _own_ another type in user BTF if it has a
* bpf_{list_head,rb_root}. Let's call these root types.
*
* We ensure that if a type is both a root and node, its
* element types cannot be root types.
*
* To ensure acyclicity:
*
* When A is an root type but not a node, its ownership
* chain can be:
* A -> B -> C
* Where:
* - A is an root, e.g. has bpf_rb_root.
* - B is both a root and node, e.g. has bpf_rb_node and
* bpf_list_head.
* - C is only an root, e.g. has bpf_list_node
*
* When A is both a root and node, some other type already
* owns it in the BTF domain, hence it can not own
* another root type through any of the ownership edges.
* A -> B
* Where:
* - A is both an root and node.
* - B is only an node.
*/
if (meta->record->field_mask & BPF_GRAPH_ROOT)
return -ELOOP;
}
return 0;
}
static int btf_owned_type_idx(const struct btf *btf, struct btf_struct_metas *tab,
const struct btf_field *field)
{
struct btf_struct_meta *meta;
u32 btf_id;
if (field->type & BPF_GRAPH_ROOT) {
btf_id = field->graph_root.value_btf_id;
} else if (field->type == BPF_KPTR_REF || field->type == BPF_KPTR_PERCPU) {
if (btf_is_kernel(field->kptr.btf))
return -ENOENT;
btf_id = field->kptr.btf_id;
} else {
return -ENOENT;
}
meta = btf_find_struct_meta(btf, btf_id);
if (!meta)
return field->type & BPF_GRAPH_ROOT ? -EFAULT : -ENOENT;
return meta - tab->types;
}
/*
* Each ownership edge adds kernel frames through bpf_obj_free_fields() and
* __bpf_obj_drop_impl(). Keep the bound deliberately small because object
* destruction can itself run below a BPF call chain. A final pointee without
* special fields is not present in the struct metadata table and adds only a
* non-recursing drop.
*/
#define BTF_MAX_OWNERSHIP_DEPTH 8
static int btf_ownership_depth(const struct btf *btf,
struct btf_struct_metas *tab, u8 *depth,
int idx, int depth_left)
{
const struct btf_record *rec = tab->types[idx].record;
int i, ret, max_depth = 0;
if (!depth_left)
return -ELOOP;
if (depth[idx])
goto done;
for (i = 0; i < rec->cnt; i++) {
ret = btf_owned_type_idx(btf, tab, &rec->fields[i]);
if (ret == -ENOENT)
continue;
if (ret < 0)
return ret;
ret = btf_ownership_depth(btf, tab, depth, ret, depth_left - 1);
if (ret < 0)
return ret;
max_depth = max(max_depth, ret);
}
depth[idx] = max_depth + 1;
done:
return depth[idx] > depth_left ? -ELOOP : depth[idx];
}
static int btf_check_ownership_depth(const struct btf *btf,
struct btf_struct_metas *tab)
{
u8 *depth;
int i, ret = 0;
depth = kvcalloc(tab->cnt, sizeof(*depth), GFP_KERNEL | __GFP_NOWARN);
if (!depth)
return -ENOMEM;
for (i = 0; i < tab->cnt; i++) {
ret = btf_ownership_depth(btf, tab, depth, i,
BTF_MAX_OWNERSHIP_DEPTH);
if (ret < 0)
break;
ret = 0;
}
kvfree(depth);
return ret;
}
static void __btf_struct_show(const struct btf *btf, const struct btf_type *t,
u32 type_id, void *data, u8 bits_offset,
struct btf_show *show)
@ -6044,6 +6078,10 @@ static struct btf *btf_parse(const union bpf_attr *attr, bpfptr_t uattr,
if (err < 0)
goto errout_meta;
}
err = btf_check_ownership_depth(btf, struct_meta_tab);
if (err < 0)
goto errout_meta;
}
err = bpf_log_attr_finalize(attr_log, &env->log);
@ -7187,7 +7225,7 @@ static int btf_struct_walk(struct bpf_verifier_log *log, const struct btf *btf,
if (btf_type_is_int(t))
return WALK_SCALAR;
if (!btf_type_is_struct(t))
if (!btf_type_is_struct(t) || !t->size)
goto error;
off = (off - moff) % t->size;

View File

@ -338,9 +338,9 @@ static int check_btf_line(struct bpf_verifier_env *env,
#define MIN_CORE_RELO_SIZE sizeof(struct bpf_core_relo)
#define MAX_CORE_RELO_SIZE MAX_FUNCINFO_REC_SIZE
static int check_core_relo(struct bpf_verifier_env *env,
const union bpf_attr *attr,
bpfptr_t uattr)
int bpf_check_core_relo(struct bpf_verifier_env *env,
const union bpf_attr *attr,
bpfptr_t uattr)
{
u32 i, nr_core_relo, ncopy, expected_size, rec_size;
struct bpf_core_relo core_relo = {};
@ -414,7 +414,7 @@ int bpf_prepare_btf_info(struct bpf_verifier_env *env,
struct btf *btf;
int err;
if (!attr->func_info_cnt && !attr->line_info_cnt) {
if (!attr->func_info_cnt && !attr->line_info_cnt && !attr->core_relo_cnt) {
if (check_abnormal_return(env))
return -EINVAL;
return 0;
@ -455,9 +455,5 @@ int bpf_check_btf_info(struct bpf_verifier_env *env,
if (err)
return err;
err = check_core_relo(env, attr, uattr);
if (err)
return err;
return 0;
}

View File

@ -19,6 +19,7 @@
#include <uapi/linux/btf.h>
#include <linux/filter.h>
#include <linux/sched/signal.h>
#include <linux/skbuff.h>
#include <linux/static_call.h>
#include <linux/vmalloc.h>
@ -1619,6 +1620,8 @@ struct bpf_prog *bpf_jit_blind_constants(struct bpf_verifier_env *env, struct bp
* fix it up here on error.
*/
bpf_jit_prog_release_other(prog, clone);
if (env && fatal_signal_pending(current))
return ERR_PTR(-EINTR);
return IS_ERR(tmp) ? tmp : ERR_PTR(-ENOMEM);
}
@ -2636,11 +2639,14 @@ static struct bpf_prog *bpf_prog_jit_compile(struct bpf_verifier_env *env, struc
orig_prog = prog;
prog = bpf_jit_blind_constants(env, prog);
/*
* If blinding was requested and we failed during blinding, we must fall
* back to the interpreter.
* Fall back to the interpreter after blinding failures, except when
* the loader was killed.
*/
if (IS_ERR(prog))
if (IS_ERR(prog)) {
if (PTR_ERR(prog) == -EINTR)
return prog;
goto out_restore;
}
prog = bpf_int_jit_compile(env, prog);
if (prog->jited) {
@ -2659,6 +2665,8 @@ static struct bpf_prog *bpf_prog_jit_compile(struct bpf_verifier_env *env, struc
struct bpf_prog *__bpf_prog_select_runtime(struct bpf_verifier_env *env, struct bpf_prog *fp,
int *err)
{
struct bpf_prog *jit_prog;
/* In case of BPF to BPF calls, verifier did all the prep
* work with regards to JITing, etc.
*/
@ -2681,7 +2689,12 @@ struct bpf_prog *__bpf_prog_select_runtime(struct bpf_verifier_env *env, struct
if (*err)
return fp;
fp = bpf_prog_jit_compile(env, fp);
jit_prog = bpf_prog_jit_compile(env, fp);
if (IS_ERR(jit_prog)) {
*err = PTR_ERR(jit_prog);
return fp;
}
fp = jit_prog;
bpf_prog_jit_attempt_done(fp);
if (!fp->jited && jit_needed) {
*err = -ENOTSUPP;

View File

@ -149,8 +149,9 @@ bpf_crypto_ctx_create(const struct bpf_crypto_params *params, u32 params__sz,
const struct bpf_crypto_type *type;
struct bpf_crypto_ctx *ctx;
if (!params || params->reserved[0] || params->reserved[1] ||
params__sz != sizeof(struct bpf_crypto_params)) {
if (!params ||
params__sz != sizeof(struct bpf_crypto_params) ||
params->reserved[0] || params->reserved[1]) {
*err = -EINVAL;
return NULL;
}

View File

@ -8,6 +8,7 @@
#include <linux/bsearch.h>
#include <linux/sort.h>
#include <linux/perf_event.h>
#include <linux/sched/signal.h>
#include <net/xdp.h>
#include "disasm.h"
@ -306,12 +307,28 @@ static void adjust_poke_descs(struct bpf_prog *prog, u32 off, u32 len)
}
}
/*
* Some post-verification instruction rewriting passes require an
* O(prog->len) operation per instruction. Keep their shared primitives
* killable and preemptible.
*/
static bool bpf_rewrite_must_abort(void)
{
if (fatal_signal_pending(current))
return true;
cond_resched();
return false;
}
struct bpf_prog *bpf_patch_insn_data(struct bpf_verifier_env *env, u32 off,
const struct bpf_insn *patch, u32 len)
{
struct bpf_prog *new_prog;
struct bpf_insn_aux_data *new_data = NULL;
if (bpf_rewrite_must_abort())
return NULL;
if (len > 1) {
new_data = vrealloc(env->insn_aux_data,
array_size(env->prog->len + len - 1,
@ -523,6 +540,9 @@ static int verifier_remove_insns(struct bpf_verifier_env *env, u32 off, u32 cnt)
unsigned int orig_prog_len = env->prog->len;
int err;
if (bpf_rewrite_must_abort())
return -EINTR;
if (bpf_prog_is_offloaded(env->prog->aux))
bpf_prog_offload_remove_insns(env, off, cnt);
@ -1356,7 +1376,7 @@ int bpf_jit_subprogs(struct bpf_verifier_env *env)
}
prog = bpf_jit_blind_constants(env, prog);
if (IS_ERR(prog)) {
err = -ENOMEM;
err = PTR_ERR(prog);
prog = orig_prog;
goto out_restore;
}
@ -1433,7 +1453,7 @@ int bpf_fixup_call_args(struct bpf_verifier_env *env)
err = bpf_jit_subprogs(env);
if (err == 0)
return 0;
if (err == -EFAULT)
if (err == -EFAULT || err == -EINTR)
return err;
}
#ifndef CONFIG_BPF_JIT_ALWAYS_ON

View File

@ -1054,14 +1054,17 @@ static void pcpu_init_value(struct bpf_htab *htab, void __percpu *pptr,
/* When not setting the initial value on all cpus, zero-fill element
* values for other cpus. Otherwise, bpf program has no way to ensure
* known initial values for cpus other than current one
* (onallcpus=false always when coming from bpf prog).
* (onallcpus=false always when coming from bpf prog,
* map_flags & BPF_F_CPU when coming from syscall but setting
* only one cpu).
*/
if (!onallcpus) {
int current_cpu = raw_smp_processor_id();
if (!onallcpus || (map_flags & BPF_F_CPU)) {
int init_cpu = (map_flags & BPF_F_CPU) ? map_flags >> 32 :
raw_smp_processor_id();
int cpu;
for_each_possible_cpu(cpu) {
if (cpu == current_cpu)
if (cpu == init_cpu)
copy_map_value(&htab->map, per_cpu_ptr(pptr, cpu), value);
else /* Since elem is preallocated, we cannot touch special fields */
zero_map_value(&htab->map, per_cpu_ptr(pptr, cpu));
@ -1772,6 +1775,12 @@ static int htab_lru_percpu_map_lookup_and_delete_elem(struct bpf_map *map,
flags);
}
/*
* Max consecutive empty buckets to walk in one RCU +
* instrumentation-disabled section before rescheduling.
*/
#define HTAB_BATCH_EMPTY_RESCHED 64
static int
__htab_map_lookup_and_delete_batch(struct bpf_map *map,
const union bpf_attr *attr,
@ -1793,6 +1802,7 @@ __htab_map_lookup_and_delete_batch(struct bpf_map *map,
unsigned long flags = 0;
bool locked = false;
struct htab_elem *l;
u32 empty_cnt = 0;
struct bucket *b;
int ret = 0;
@ -1971,30 +1981,41 @@ __htab_map_lookup_and_delete_batch(struct bpf_map *map,
}
next_batch:
/* If we are not copying data, we can go to next bucket and avoid
* unlocking the rcu.
/*
* If we are not copying data, we can go to next bucket and avoid
* unlocking the rcu. Bound the walk though: after
* HTAB_BATCH_EMPTY_RESCHED consecutive empty buckets, fully exit
* the critical section (no locks are held here) and reschedule.
*/
if (!bucket_cnt && (batch + 1 < htab->n_buckets)) {
batch++;
goto again_nocopy;
if (++empty_cnt < HTAB_BATCH_EMPTY_RESCHED)
goto again_nocopy;
empty_cnt = 0;
rcu_read_unlock();
bpf_enable_instrumentation();
cond_resched_tasks_rcu_qs();
goto again;
}
rcu_read_unlock();
bpf_enable_instrumentation();
if (bucket_cnt && (copy_to_user(ukeys + total * key_size, keys,
key_size * bucket_cnt) ||
copy_to_user(uvalues + total * value_size, values,
value_size * bucket_cnt))) {
if (bucket_cnt && (copy_to_user(ukeys + (size_t)total * key_size, keys,
(size_t)key_size * bucket_cnt) ||
copy_to_user(uvalues + (size_t)total * value_size, values,
(size_t)value_size * bucket_cnt))) {
ret = -EFAULT;
goto after_loop;
}
total += bucket_cnt;
empty_cnt = 0;
batch++;
if (batch >= htab->n_buckets) {
ret = -ENOENT;
goto after_loop;
}
cond_resched_tasks_rcu_qs();
goto again;
after_loop:

View File

@ -4883,7 +4883,7 @@ BTF_ID(func, bpf_cgroup_release_dtor)
BTF_KFUNCS_START(common_btf_ids)
BTF_ID_FLAGS(func, bpf_cast_to_kern_ctx, KF_FASTCALL)
BTF_ID_FLAGS(func, bpf_rdonly_cast, KF_FASTCALL)
BTF_ID_FLAGS(func, bpf_rdonly_cast, KF_FASTCALL | KF_PERFMON)
BTF_ID_FLAGS(func, bpf_rcu_read_lock)
BTF_ID_FLAGS(func, bpf_rcu_read_unlock)
BTF_ID_FLAGS(func, bpf_dynptr_slice, KF_RET_NULL)
@ -4920,26 +4920,26 @@ BTF_ID_FLAGS(func, bpf_wq_set_callback, KF_IMPLICIT_ARGS)
BTF_ID_FLAGS(func, bpf_wq_start)
BTF_ID_FLAGS(func, bpf_preempt_disable)
BTF_ID_FLAGS(func, bpf_preempt_enable)
BTF_ID_FLAGS(func, bpf_iter_bits_new, KF_ITER_NEW)
BTF_ID_FLAGS(func, bpf_iter_bits_new, KF_ITER_NEW | KF_PERFMON)
BTF_ID_FLAGS(func, bpf_iter_bits_next, KF_ITER_NEXT | KF_RET_NULL)
BTF_ID_FLAGS(func, bpf_iter_bits_destroy, KF_ITER_DESTROY)
BTF_ID_FLAGS(func, bpf_copy_from_user_str, KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_copy_from_user_task_str, KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_get_kmem_cache)
BTF_ID_FLAGS(func, bpf_copy_from_user_str, KF_SLEEPABLE | KF_PERFMON)
BTF_ID_FLAGS(func, bpf_copy_from_user_task_str, KF_SLEEPABLE | KF_PERFMON)
BTF_ID_FLAGS(func, bpf_get_kmem_cache, KF_PERFMON)
BTF_ID_FLAGS(func, bpf_iter_kmem_cache_new, KF_ITER_NEW | KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_iter_kmem_cache_next, KF_ITER_NEXT | KF_RET_NULL | KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_iter_kmem_cache_destroy, KF_ITER_DESTROY | KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_local_irq_save)
BTF_ID_FLAGS(func, bpf_local_irq_restore)
#ifdef CONFIG_BPF_EVENTS
BTF_ID_FLAGS(func, bpf_probe_read_user_dynptr)
BTF_ID_FLAGS(func, bpf_probe_read_kernel_dynptr)
BTF_ID_FLAGS(func, bpf_probe_read_user_str_dynptr)
BTF_ID_FLAGS(func, bpf_probe_read_kernel_str_dynptr)
BTF_ID_FLAGS(func, bpf_copy_from_user_dynptr, KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_copy_from_user_str_dynptr, KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_copy_from_user_task_dynptr, KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_copy_from_user_task_str_dynptr, KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_probe_read_user_dynptr, KF_PERFMON)
BTF_ID_FLAGS(func, bpf_probe_read_kernel_dynptr, KF_PERFMON)
BTF_ID_FLAGS(func, bpf_probe_read_user_str_dynptr, KF_PERFMON)
BTF_ID_FLAGS(func, bpf_probe_read_kernel_str_dynptr, KF_PERFMON)
BTF_ID_FLAGS(func, bpf_copy_from_user_dynptr, KF_SLEEPABLE | KF_PERFMON)
BTF_ID_FLAGS(func, bpf_copy_from_user_str_dynptr, KF_SLEEPABLE | KF_PERFMON)
BTF_ID_FLAGS(func, bpf_copy_from_user_task_dynptr, KF_SLEEPABLE | KF_PERFMON)
BTF_ID_FLAGS(func, bpf_copy_from_user_task_str_dynptr, KF_SLEEPABLE | KF_PERFMON)
#endif
#ifdef CONFIG_DMA_SHARED_BUFFER
BTF_ID_FLAGS(func, bpf_iter_dmabuf_new, KF_ITER_NEW | KF_SLEEPABLE)
@ -4947,26 +4947,26 @@ BTF_ID_FLAGS(func, bpf_iter_dmabuf_next, KF_ITER_NEXT | KF_RET_NULL | KF_SLEEPAB
BTF_ID_FLAGS(func, bpf_iter_dmabuf_destroy, KF_ITER_DESTROY | KF_SLEEPABLE)
#endif
BTF_ID_FLAGS(func, __bpf_trap)
BTF_ID_FLAGS(func, bpf_strcmp);
BTF_ID_FLAGS(func, bpf_strcasecmp);
BTF_ID_FLAGS(func, bpf_strncasecmp);
BTF_ID_FLAGS(func, bpf_strchr);
BTF_ID_FLAGS(func, bpf_strchrnul);
BTF_ID_FLAGS(func, bpf_strnchr);
BTF_ID_FLAGS(func, bpf_strrchr);
BTF_ID_FLAGS(func, bpf_strlen);
BTF_ID_FLAGS(func, bpf_strnlen);
BTF_ID_FLAGS(func, bpf_strspn);
BTF_ID_FLAGS(func, bpf_strcspn);
BTF_ID_FLAGS(func, bpf_strstr);
BTF_ID_FLAGS(func, bpf_strcasestr);
BTF_ID_FLAGS(func, bpf_strnstr);
BTF_ID_FLAGS(func, bpf_strncasestr);
BTF_ID_FLAGS(func, bpf_strcmp, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strcasecmp, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strncasecmp, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strchr, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strchrnul, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strnchr, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strrchr, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strlen, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strnlen, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strspn, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strcspn, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strstr, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strcasestr, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strnstr, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strncasestr, KF_PERFMON);
#if defined(CONFIG_BPF_LSM) && defined(CONFIG_CGROUPS)
BTF_ID_FLAGS(func, bpf_cgroup_read_xattr, KF_RCU)
#endif
BTF_ID_FLAGS(func, bpf_stream_vprintk, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE)
BTF_ID_FLAGS(func, bpf_stream_print_stack, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE)
BTF_ID_FLAGS(func, bpf_stream_vprintk, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE | KF_PERFMON)
BTF_ID_FLAGS(func, bpf_stream_print_stack, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE | KF_PERFMON)
BTF_ID_FLAGS(func, bpf_task_work_schedule_signal, KF_IMPLICIT_ARGS)
BTF_ID_FLAGS(func, bpf_task_work_schedule_resume, KF_IMPLICIT_ARGS)
BTF_ID_FLAGS(func, bpf_dynptr_from_file)

View File

@ -119,6 +119,7 @@ struct bpf_mem_cache {
struct llist_head waiting_for_gp_ttrace;
struct rcu_head rcu_ttrace;
atomic_t call_rcu_ttrace_in_progress;
raw_spinlock_t lock;
};
struct bpf_mem_caches {
@ -214,25 +215,24 @@ static void alloc_bulk(struct bpf_mem_cache *c, int cnt, int node, bool atomic)
gfp = __GFP_NOWARN | __GFP_ACCOUNT;
gfp |= atomic ? GFP_NOWAIT : GFP_KERNEL;
for (i = 0; i < cnt; i++) {
/*
* For every 'c' llist_del_first(&c->free_by_rcu_ttrace); is
* done only by one CPU == current CPU. Other CPUs might
* llist_add() and llist_del_all() in parallel.
*/
obj = llist_del_first(&c->free_by_rcu_ttrace);
if (!obj)
break;
add_obj_to_free_list(c, obj);
}
if (i >= cnt)
return;
/*
* c->lock serializes concurrent llist_del_first() against
* llist_del_all() in __free_rcu() and do_call_rcu_ttrace().
*/
scoped_guard(raw_spinlock_irqsave, &c->lock) {
for (i = 0; i < cnt; i++) {
obj = llist_del_first(&c->free_by_rcu_ttrace);
if (!obj)
break;
add_obj_to_free_list(c, obj);
}
for (; i < cnt; i++) {
obj = llist_del_first(&c->waiting_for_gp_ttrace);
if (!obj)
break;
add_obj_to_free_list(c, obj);
for (; i < cnt; i++) {
obj = llist_del_first(&c->waiting_for_gp_ttrace);
if (!obj)
break;
add_obj_to_free_list(c, obj);
}
}
if (i >= cnt)
return;
@ -279,8 +279,12 @@ static int free_all(struct bpf_mem_cache *c, struct llist_node *llnode, bool per
static void __free_rcu(struct rcu_head *head)
{
struct bpf_mem_cache *c = container_of(head, struct bpf_mem_cache, rcu_ttrace);
struct llist_node *llnode;
free_all(c, llist_del_all(&c->waiting_for_gp_ttrace), !!c->percpu_size);
scoped_guard(raw_spinlock_irqsave, &c->lock)
llnode = llist_del_all(&c->waiting_for_gp_ttrace);
free_all(c, llnode, !!c->percpu_size);
atomic_set(&c->call_rcu_ttrace_in_progress, 0);
}
@ -300,7 +304,8 @@ static void do_call_rcu_ttrace(struct bpf_mem_cache *c)
if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 1)) {
if (unlikely(READ_ONCE(c->draining))) {
llnode = llist_del_all(&c->free_by_rcu_ttrace);
scoped_guard(raw_spinlock_irqsave, &c->lock)
llnode = llist_del_all(&c->free_by_rcu_ttrace);
free_all(c, llnode, !!c->percpu_size);
}
return;
@ -535,6 +540,7 @@ int bpf_mem_alloc_init(struct bpf_mem_alloc *ma, int size, bool percpu)
c->objcg = objcg;
c->percpu_size = percpu_size;
c->tgt = c;
raw_spin_lock_init(&c->lock);
init_refill_work(c);
prefill_mem_cache(c, cpu);
}
@ -557,7 +563,7 @@ int bpf_mem_alloc_init(struct bpf_mem_alloc *ma, int size, bool percpu)
c->objcg = objcg;
c->percpu_size = percpu_size;
c->tgt = c;
raw_spin_lock_init(&c->lock);
init_refill_work(c);
prefill_mem_cache(c, cpu);
}
@ -609,7 +615,7 @@ int bpf_mem_alloc_percpu_unit_init(struct bpf_mem_alloc *ma, int size)
c->objcg = objcg;
c->percpu_size = percpu_size;
c->tgt = c;
raw_spin_lock_init(&c->lock);
init_refill_work(c);
prefill_mem_cache(c, cpu);
}

View File

@ -698,6 +698,8 @@ static bool __bpf_offload_dev_match(struct bpf_prog *prog,
return false;
if (offload->netdev == netdev)
return true;
if (!bpf_prog_is_offloaded(prog->aux))
return false;
ondev1 = bpf_offload_find_netdev(offload->netdev);
ondev2 = bpf_offload_find_netdev(netdev);

View File

@ -491,7 +491,8 @@ static bool regs_exact(const struct bpf_reg_state *rold,
{
return memcmp(rold, rcur, offsetof(struct bpf_reg_state, id)) == 0 &&
check_ids(rold->id, rcur->id, idmap) &&
check_ids(rold->parent_id, rcur->parent_id, idmap);
check_ids(rold->parent_id, rcur->parent_id, idmap) &&
check_ids(rold->map_uid, rcur->map_uid, idmap);
}
enum exact_level {
@ -616,7 +617,8 @@ static bool regsafe(struct bpf_verifier_env *env, struct bpf_reg_state *rold,
range_within(rold, rcur) &&
tnum_in(rold->var_off, rcur->var_off) &&
check_ids(rold->id, rcur->id, idmap) &&
check_ids(rold->parent_id, rcur->parent_id, idmap);
check_ids(rold->parent_id, rcur->parent_id, idmap) &&
check_ids(rold->map_uid, rcur->map_uid, idmap);
case PTR_TO_PACKET_META:
case PTR_TO_PACKET:
/* We must have at least as much range as the old ptr
@ -635,14 +637,14 @@ static bool regsafe(struct bpf_verifier_env *env, struct bpf_reg_state *rold,
/* id relations must be preserved */
if (!check_ids(rold->id, rcur->id, idmap))
return false;
/* Preserve displacements between pointers sharing an ID. */
if (rold->id && rold->r64.base != rcur->r64.base)
return false;
/* new val must satisfy old val knowledge */
return range_within(rold, rcur) &&
tnum_in(rold->var_off, rcur->var_off);
case PTR_TO_STACK:
/* two stack pointers are equal only if they're pointing to
* the same stack frame, since fp-8 in foo != fp-8 in bar
*/
return regs_exact(rold, rcur, idmap) && rold->frameno == rcur->frameno;
return regs_exact(rold, rcur, idmap);
case PTR_TO_ARENA:
return true;
case PTR_TO_INSN:
@ -1121,7 +1123,7 @@ static bool states_maybe_looping(struct bpf_verifier_state *old,
fcur = cur->frame[fr];
for (i = 0; i < MAX_BPF_REG; i++)
if (memcmp(&fold->regs[i], &fcur->regs[i],
offsetof(struct bpf_reg_state, frameno)))
offsetof(struct bpf_reg_state, precise)))
return false;
return true;
}

View File

@ -2036,7 +2036,7 @@ int generic_map_delete_batch(struct bpf_map *map,
for (cp = 0; cp < max_count; cp++) {
err = -EFAULT;
if (copy_from_user(key, keys + cp * map->key_size,
if (copy_from_user(key, keys + (size_t)cp * map->key_size,
map->key_size))
break;
@ -2098,9 +2098,9 @@ int generic_map_update_batch(struct bpf_map *map, struct file *map_file,
for (cp = 0; cp < max_count; cp++) {
err = -EFAULT;
if (copy_from_user(key, keys + cp * map->key_size,
if (copy_from_user(key, keys + (size_t)cp * map->key_size,
map->key_size) ||
copy_from_user(value, values + cp * value_size, value_size))
copy_from_user(value, values + (size_t)cp * value_size, value_size))
break;
err = bpf_map_update_value(map, map_file, key, value,
@ -2179,12 +2179,12 @@ int generic_map_lookup_batch(struct bpf_map *map,
if (err)
goto free_buf;
if (copy_to_user(keys + cp * map->key_size, key,
if (copy_to_user(keys + (size_t)cp * map->key_size, key,
map->key_size)) {
err = -EFAULT;
goto free_buf;
}
if (copy_to_user(values + cp * value_size, value, value_size)) {
if (copy_to_user(values + (size_t)cp * value_size, value, value_size)) {
err = -EFAULT;
goto free_buf;
}
@ -6042,7 +6042,10 @@ struct bpf_link *bpf_link_get_curr_or_next(u32 *id)
again:
link = idr_get_next(&link_idr, id);
if (link) {
link = bpf_link_inc_not_zero(link);
if (link->id)
link = bpf_link_inc_not_zero(link);
else
link = ERR_PTR(-EAGAIN);
if (IS_ERR(link)) {
(*id)++;
goto again;

View File

@ -567,7 +567,7 @@ static int stack_slot_obj_get_spi(struct bpf_verifier_env *env, struct bpf_reg_s
}
off = reg->var_off.value;
if (off % BPF_REG_SIZE) {
if (off >= 0 || off % BPF_REG_SIZE) {
verbose(env, "cannot pass in %s at an offset=%d\n", obj_kind, off);
return -EINVAL;
}
@ -1864,6 +1864,7 @@ static void __mark_reg_known(struct bpf_reg_state *reg, u64 imm)
offsetof(struct bpf_reg_state, var_off) - sizeof(reg->type));
reg->id = 0;
reg->parent_id = 0;
reg->map_uid = 0;
___mark_reg_known(reg, imm);
}
@ -1925,17 +1926,18 @@ static void refine_map_lookup_value(struct bpf_reg_state *reg)
if (map->inner_map_meta) {
reg->type = CONST_PTR_TO_MAP | maybe_null;
reg->map_ptr = map->inner_map_meta;
/* transfer reg's id which is unique for every map_lookup_elem
/*
* transfer reg's id which is unique for every map_lookup_elem
* as UID of the inner map.
*/
if (btf_record_has_field(map->inner_map_meta->record,
BPF_TIMER | BPF_WORKQUEUE | BPF_TASK_WORK))
reg->map_uid = reg->id;
reg->map_uid = reg->id;
} else if (map->map_type == BPF_MAP_TYPE_XSKMAP) {
reg->type = PTR_TO_XDP_SOCK | maybe_null;
reg->map_uid = 0;
} else if (map->map_type == BPF_MAP_TYPE_SOCKMAP ||
map->map_type == BPF_MAP_TYPE_SOCKHASH) {
reg->type = PTR_TO_SOCKET | maybe_null;
reg->map_uid = 0;
}
}
@ -3042,6 +3044,8 @@ static int check_subprogs(struct bpf_verifier_env *env)
subprog[cur_subprog].exit_idx = i;
goto next;
}
if (insn_is_gotox(&insn[i]))
goto next;
off = i + bpf_jmp_offset(&insn[i]) + 1;
if (off < subprog_start || off >= subprog_end) {
verbose(env, "jump out of range from insn %d to %d\n", i, off);
@ -3061,7 +3065,8 @@ static int check_subprogs(struct bpf_verifier_env *env)
*/
if (code != (BPF_JMP | BPF_EXIT) &&
code != (BPF_JMP32 | BPF_JA) &&
code != (BPF_JMP | BPF_JA)) {
code != (BPF_JMP | BPF_JA) &&
!insn_is_gotox(&insn[i])) {
verbose(env, "last insn is not an exit or jmp\n");
bpf_diag_program_structure(
env, i, "subprogram can fall through",
@ -3582,7 +3587,7 @@ static int check_stack_write_fixed_off(struct bpf_verifier_env *env,
save_register_state(env, state, spi, reg, size);
/* Break the relation on a narrowing spill. */
if (!reg_value_fits)
state->stack[spi].spilled_ptr.id = 0;
clear_scalar_id(&state->stack[spi].spilled_ptr);
} else if (!reg && !(off % BPF_REG_SIZE) && is_bpf_st_mem(insn) &&
env->bpf_capable) {
struct bpf_reg_state *tmp_reg = &env->fake_reg[0];
@ -6453,6 +6458,15 @@ static int check_mem_access(struct bpf_verifier_env *env, int insn_idx, struct b
return -EACCES;
}
if (rdonly_untrusted && !env->allow_ptr_leaks) {
verbose(env, "%s access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n",
reg_type_str(env, reg->type));
bpf_diag_policy(env, insn_idx, "read from untrusted read-only memory",
"the access requires CAP_PERFMON",
"Load the program with CAP_PERFMON, or avoid dereferencing untrusted pointers.");
return -EPERM;
}
/*
* Accesses to untrusted PTR_TO_MEM are done through probe
* instructions, hence no need to check bounds in that case.
@ -9736,6 +9750,16 @@ static int btf_check_func_arg_match(struct bpf_verifier_env *env, int subprog,
if (check_mem_reg(env, reg, argno, arg->mem_size, BPF_READ | BPF_WRITE, NULL,
NULL))
return -EINVAL;
/*
* PTR_TO_PACKET get passed as PTR_TO_MEM, preventing
* us from adjusting bounds tracking info.
*/
if ((reg_is_pkt_pointer_any(reg) || reg_is_dynptr_slice_pkt(reg)) &&
sub->changes_pkt_data) {
bpf_log(log, "%s is a packet pointer, but func#%d may change packet data\n",
reg_arg_name(env, argno), subprog);
return -EINVAL;
}
if (!(arg->arg_type & PTR_MAYBE_NULL) &&
(type_may_be_null(reg->type) || bpf_register_is_null(reg))) {
bpf_log(log, "%s is expected to be non-NULL\n",
@ -9919,6 +9943,7 @@ static int check_func_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
if (err == -EFAULT)
return err;
if (bpf_subprog_is_global(env, subprog)) {
struct bpf_func_info_aux *sub_aux = subprog_aux(env, subprog);
const char *sub_name = bpf_subprog_name(env, subprog);
const char *operation;
bool returns_void;
@ -9950,11 +9975,10 @@ static int check_func_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
if (env->log.level & BPF_LOG_LEVEL)
verbose(env, "Func#%d ('%s') is global and assumed valid.\n",
subprog, sub_name);
sub_aux->called[in_sleepable_context(env)] = true;
returns_void = subprog_returns_void(env, subprog);
if (env->subprog_info[subprog].changes_pkt_data)
clear_all_pkt_pointers(env);
/* mark global subprog for verifying after main prog */
subprog_aux(env, subprog)->called = true;
if (returns_void)
bpf_diag_record_scrub(env, &caller->regs[BPF_REG_0], BPF_DIAG_MOD_CALLER_SAVED);
else
@ -10036,6 +10060,7 @@ int map_set_for_each_callback_args(struct bpf_verifier_env *env,
__mark_reg_known_zero(&callee->regs[BPF_REG_3]);
callee->regs[BPF_REG_3].map_ptr = caller->regs[BPF_REG_1].map_ptr;
callee->regs[BPF_REG_3].map_uid = caller->regs[BPF_REG_1].map_uid;
callee->regs[BPF_REG_3].id = ++env->id_gen;
/* pointer to stack or null */
callee->regs[BPF_REG_4] = caller->regs[BPF_REG_3];
@ -10132,6 +10157,7 @@ static int set_timer_callback_state(struct bpf_verifier_env *env,
__mark_reg_known_zero(&callee->regs[BPF_REG_3]);
callee->regs[BPF_REG_3].map_ptr = map_ptr;
callee->regs[BPF_REG_3].map_uid = map_uid;
callee->regs[BPF_REG_3].id = ++env->id_gen;
/* unused */
bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
@ -10250,6 +10276,7 @@ static int set_task_work_schedule_callback_state(struct bpf_verifier_env *env,
__mark_reg_known_zero(&callee->regs[BPF_REG_3]);
callee->regs[BPF_REG_3].map_ptr = map_ptr;
callee->regs[BPF_REG_3].map_uid = map_uid;
callee->regs[BPF_REG_3].id = ++env->id_gen;
/* unused */
bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
@ -10772,11 +10799,7 @@ int bpf_get_helper_proto(struct bpf_verifier_env *env, int func_id,
/* Check if we're in a sleepable context. */
static inline bool in_sleepable_context(struct bpf_verifier_env *env)
{
return !env->cur_state->active_rcu_locks &&
!env->cur_state->active_preempt_locks &&
!env->cur_state->active_locks &&
!env->cur_state->active_irq_id &&
in_sleepable(env);
return !in_rcu_cs(env);
}
static const char *non_sleepable_context_description(struct bpf_verifier_env *env)
@ -11356,6 +11379,11 @@ static bool is_kfunc_destructive(struct bpf_call_arg_meta *meta)
return meta->kfunc_flags & KF_DESTRUCTIVE;
}
static bool is_kfunc_perfmon(struct bpf_call_arg_meta *meta)
{
return meta->kfunc_flags & KF_PERFMON;
}
static bool is_kfunc_rcu(struct bpf_call_arg_meta *meta)
{
return meta->kfunc_flags & KF_RCU;
@ -13834,6 +13862,15 @@ static int check_kfunc_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
return -EACCES;
}
if (is_kfunc_perfmon(&meta) && !env->allow_ptr_leaks) {
verbose(env, "%s is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n",
func_name);
operation = bpf_diag_fmt(env, "kfunc %s", func_name);
bpf_diag_policy(env, insn_idx, operation, "the kfunc requires CAP_PERFMON",
"Load the program with CAP_PERFMON, or avoid the kfunc.");
return -EPERM;
}
sleepable = bpf_is_kfunc_sleepable(&meta);
if (sleepable && !in_sleepable(env)) {
verbose(env, "program must be sleepable to call sleepable kfunc %s\n", func_name);
@ -15704,6 +15741,7 @@ static int adjust_reg_min_max_vals(struct bpf_verifier_env *env,
struct bpf_reg_state *regs = state->regs, *dst_reg, *src_reg;
struct bpf_reg_state *ptr_reg = NULL, off_reg = {0};
bool alu32 = (BPF_CLASS(insn->code) != BPF_ALU64);
struct bpf_insn_aux_data *aux = cur_aux(env);
u8 opcode = BPF_OP(insn->code);
int err;
@ -15715,12 +15753,23 @@ static int adjust_reg_min_max_vals(struct bpf_verifier_env *env,
/* Case where at least one operand is an arena. */
if (dst_reg->type == PTR_TO_ARENA || (src_reg && src_reg->type == PTR_TO_ARENA)) {
struct bpf_insn_aux_data *aux = cur_aux(env);
if (dst_reg->type != PTR_TO_ARENA)
*dst_reg = *src_reg;
if (BPF_CLASS(insn->code) == BPF_ALU64) {
/*
* Only arena pointers set needs_zext, but doing so
* modifies the instruction at fixup time to an ALU32
* and makes it unsuitable for 64-bit scalar args. We
* prevent zext from being set if the instruction has
* been previously called with non-arena registers.
*/
if (aux->prevent_zext) {
verbose(env, "same insn cannot be used with and without arena pointer\n");
return -EINVAL;
}
/*
* 32-bit operations zero upper bits automatically.
* 64-bit operations need to be converted to 32.
@ -15733,6 +15782,16 @@ static int adjust_reg_min_max_vals(struct bpf_verifier_env *env,
return 0;
}
/* Prevent the instruction from being used with arena pointers (see above). */
if (env->prog->aux->arena && BPF_CLASS(insn->code) == BPF_ALU64) {
if (aux->needs_zext) {
verbose(env, "same insn cannot be used with and without arena pointer\n");
return -EINVAL;
}
aux->prevent_zext = true;
}
if (dst_reg->type != SCALAR_VALUE)
ptr_reg = dst_reg;
@ -19421,13 +19480,14 @@ static void free_states(struct bpf_verifier_env *env)
}
}
static int do_check_common(struct bpf_verifier_env *env, int subprog)
static int do_check_common(struct bpf_verifier_env *env, int subprog, bool is_sleepable)
{
bool pop_log = !(env->log.level & BPF_LOG_LEVEL2);
struct bpf_subprog_info *sub = subprog_info(env, subprog);
struct bpf_prog_aux *aux = env->prog->aux;
struct bpf_verifier_state *state;
struct bpf_reg_state *regs;
u32 old_insns_total = sub->insns_total;
u32 insn_processed = env->insn_processed;
int ret, i;
@ -19440,7 +19500,7 @@ static int do_check_common(struct bpf_verifier_env *env, int subprog)
state->curframe = 0;
state->speculative = false;
state->branches = 1;
state->in_sleepable = env->prog->sleepable;
state->in_sleepable = is_sleepable;
state->frame[0] = kzalloc_obj(struct bpf_func_state, GFP_KERNEL_ACCOUNT);
if (!state->frame[0]) {
kfree(state);
@ -19581,8 +19641,10 @@ static int do_check_common(struct bpf_verifier_env *env, int subprog)
* not accounted as callees by account_current_path().
* Accumulate their total counts as total counts of the main or
* global subprog hosting the async call.
* Start from the saved total of earlier contexts: adding to the current
* total would count this pass's synchronous paths twice.
*/
env->subprog_info[subprog].insns_total = env->insn_processed - insn_processed;
sub->insns_total = old_insns_total + (env->insn_processed - insn_processed);
return ret;
}
@ -19610,14 +19672,19 @@ static int do_check_subprogs(struct bpf_verifier_env *env)
{
struct bpf_prog_aux *aux = env->prog->aux;
struct bpf_func_info_aux *sub_aux;
int i, ret, new_cnt;
int context, i, ret, new_cnt;
if (!aux->func_info)
return 0;
/* exception callback is presumed to be always called */
if (env->exception_callback_subprog)
subprog_aux(env, env->exception_callback_subprog)->called = true;
/*
* Callbacks cannot throw, so the exception callback always runs in the
* main program's context. It is presumed to be always called.
*/
if (env->exception_callback_subprog) {
sub_aux = subprog_aux(env, env->exception_callback_subprog);
sub_aux->called[env->prog->sleepable] = true;
}
again:
new_cnt = 0;
@ -19626,29 +19693,28 @@ static int do_check_subprogs(struct bpf_verifier_env *env)
continue;
sub_aux = subprog_aux(env, i);
if (!sub_aux->called || sub_aux->verified)
continue;
for (context = 0; context < ARRAY_SIZE(sub_aux->called); context++) {
if (!sub_aux->called[context] || sub_aux->verified[context])
continue;
env->insn_idx = env->subprog_info[i].start;
WARN_ON_ONCE(env->insn_idx == 0);
ret = do_check_common(env, i);
if (ret) {
return ret;
} else if (env->log.level & BPF_LOG_LEVEL) {
verbose(env, "Func#%d ('%s') is safe for any args that match its prototype\n",
i, bpf_subprog_name(env, i));
env->insn_idx = env->subprog_info[i].start;
WARN_ON_ONCE(env->insn_idx == 0);
ret = do_check_common(env, i, context);
if (ret)
return ret;
if (env->log.level & BPF_LOG_LEVEL)
verbose(env, "Func#%d ('%s') is safe for any args "
"that match its prototype\n",
i, bpf_subprog_name(env, i));
sub_aux->verified[context] = true;
new_cnt++;
}
/* We verified new global subprog, it might have called some
* more global subprogs that we haven't verified yet, so we
* need to do another pass over subprogs to verify those.
*/
sub_aux->verified = true;
new_cnt++;
}
/* We can't loop forever as we verify at least one global subprog on
* each pass.
/*
* We can't loop forever as each pass verifies at least one new context,
* and there are only two contexts per global subprog.
*/
if (new_cnt)
goto again;
@ -19661,7 +19727,7 @@ static int do_check_main(struct bpf_verifier_env *env)
int ret;
env->insn_idx = 0;
ret = do_check_common(env, 0);
ret = do_check_common(env, 0, env->prog->sleepable);
if (!ret)
env->prog->aux->stack_depth = env->subprog_info[0].stack_depth;
return ret;
@ -21170,6 +21236,11 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr,
ret = bpf_diag_init(env);
if (ret)
goto err_prep;
if (env->prog->insnsi[env->prog->len - 1].code == (BPF_LD | BPF_IMM | BPF_DW)) {
verbose(env, "invalid bpf_ld_imm64 insn\n");
ret = -EINVAL;
goto err_prep;
}
if (env->signature) {
ret = bpf_prog_calc_tag(env->prog);
if (ret < 0)
@ -21245,6 +21316,11 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr,
if (ret < 0)
goto skip_full_check;
/* Apply CO-RE before validating the program's instruction layout. */
ret = bpf_check_core_relo(env, attr, uattr);
if (ret < 0)
goto skip_full_check;
/* Discover all subprograms before validating their layout and BTF. */
ret = add_subprogs(env);
if (ret < 0)
@ -21254,7 +21330,7 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr,
if (ret < 0)
goto skip_full_check;
/* Validate BTF against the complete subprogram layout and apply CO-RE. */
/* Validate BTF against the complete subprogram layout. */
ret = bpf_check_btf_info(env, attr, uattr);
if (ret < 0)
goto skip_full_check;

View File

@ -205,8 +205,8 @@ static void xdp_test_run_teardown(struct xdp_test_data *xdp)
{
xdp_unreg_mem_model(&xdp->mem);
page_pool_destroy(xdp->pp);
kfree(xdp->frames);
kfree(xdp->skbs);
kvfree(xdp->frames);
kvfree(xdp->skbs);
}
static bool frame_was_changed(const struct xdp_page_head *head)

View File

@ -3961,12 +3961,6 @@ static u32 __bpf_skb_min_len(const struct sk_buff *skb)
if (offset > 0)
min_len = offset;
}
if (skb->ip_summed == CHECKSUM_PARTIAL) {
offset = skb_checksum_start_offset(skb) +
skb->csum_offset + sizeof(__sum16);
if (offset > 0)
min_len = offset;
}
return min_len;
}
@ -3983,6 +3977,11 @@ static int bpf_skb_grow_rcsum(struct sk_buff *skb, unsigned int new_len)
static int bpf_skb_trim_rcsum(struct sk_buff *skb, unsigned int new_len)
{
if (skb->ip_summed == CHECKSUM_PARTIAL &&
new_len < skb_checksum_start_offset(skb) + skb->csum_offset +
sizeof(__sum16))
skb->ip_summed = CHECKSUM_NONE;
return __skb_trim_rcsum(skb, new_len);
}
@ -9045,6 +9044,8 @@ static const struct bpf_func_proto *
lwt_seg6local_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog)
{
switch (func_id) {
case BPF_FUNC_skb_pull_data:
return NULL;
#if IS_ENABLED(CONFIG_IPV6_SEG6_BPF)
case BPF_FUNC_lwt_seg6_store_bytes:
return &bpf_lwt_seg6_store_bytes_proto;
@ -10565,11 +10566,12 @@ u32 bpf_sock_convert_ctx_access(enum bpf_access_type type,
target_size));
*insn++ = BPF_JMP_IMM(BPF_JNE, si->dst_reg, NO_QUEUE_MAPPING,
1);
*insn++ = BPF_MOV64_IMM(si->dst_reg, -1);
*insn++ = BPF_MOV32_IMM(si->dst_reg, -1);
#else
*insn++ = BPF_MOV64_IMM(si->dst_reg, -1);
*target_size = 2;
*insn++ = BPF_MOV32_IMM(si->dst_reg, -1);
#endif
*target_size = sizeof_field(struct bpf_sock, rx_queue_mapping);
break;
}
@ -11105,18 +11107,7 @@ static u32 sock_ops_convert_ctx_access(enum bpf_access_type type,
break;
case offsetof(struct bpf_sock_ops, rtt_min):
BUILD_BUG_ON(sizeof_field(struct tcp_sock, rtt_min) !=
sizeof(struct minmax));
BUILD_BUG_ON(sizeof(struct minmax) <
sizeof(struct minmax_sample));
*insn++ = BPF_LDX_MEM(BPF_FIELD_SIZEOF(
struct bpf_sock_ops_kern, sk),
si->dst_reg, si->src_reg,
offsetof(struct bpf_sock_ops_kern, sk));
*insn++ = BPF_LDX_MEM(BPF_W, si->dst_reg, si->dst_reg,
offsetof(struct tcp_sock, rtt_min) +
sizeof_field(struct minmax_sample, t));
SOCK_OPS_GET_FIELD(rtt_min, rtt_min.s[0].v, struct tcp_sock);
break;
case offsetof(struct bpf_sock_ops, bpf_sock_ops_cb_flags):
@ -12912,8 +12903,9 @@ __bpf_kfunc_start_defs();
* @sock: Pointer to socket to be destroyed
*
* Return:
* On error, may return EPROTONOSUPPORT, EINVAL.
* EPROTONOSUPPORT if protocol specific destroy handler is not supported.
* On error, may return EOPNOTSUPP, or whatever the protocol specific
* destroy handler returns.
* EOPNOTSUPP if protocol specific destroy handler is not supported.
* 0 otherwise
*/
__bpf_kfunc int bpf_sock_destroy(struct sock_common *sock)
@ -12925,8 +12917,12 @@ __bpf_kfunc int bpf_sock_destroy(struct sock_common *sock)
* Supporting protocols will need to acquire sock lock in the BPF context
* prior to invoking this kfunc.
*/
if (!sk->sk_prot->diag_destroy || (sk->sk_protocol != IPPROTO_TCP &&
sk->sk_protocol != IPPROTO_UDP))
if (!sk->sk_prot->diag_destroy)
return -EOPNOTSUPP;
if (sk_fullsock(sk) &&
sk->sk_protocol != IPPROTO_TCP &&
sk->sk_protocol != IPPROTO_UDP)
return -EOPNOTSUPP;
return sk->sk_prot->diag_destroy(sk, ECONNABORTED);

View File

@ -1000,6 +1000,10 @@ static int sk_psock_verdict_apply(struct sk_psock *psock, struct sk_buff *skb,
int err = 0;
u32 len, off;
if (verdict == __SK_REDIRECT && skb_bpf_ingress(skb) &&
skb_bpf_redirect_fetch(skb) == psock->sk)
verdict = __SK_PASS;
switch (verdict) {
case __SK_PASS:
err = -EIO;

View File

@ -41,6 +41,7 @@ static struct bpf_map *sock_map_alloc(union bpf_attr *attr)
struct bpf_stab *stab;
if (attr->max_entries == 0 ||
attr->max_entries > INT_MAX ||
attr->key_size != 4 ||
(attr->value_size != sizeof(u32) &&
attr->value_size != sizeof(u64)) ||

View File

@ -1520,7 +1520,8 @@ void inet_csk_listen_stop(struct sock *sk)
local_bh_enable();
sock_put(child);
cond_resched();
if (!has_current_bpf_ctx())
cond_resched();
}
if (queue->fastopenq.rskq_rst_head) {
/* Free all the reqs queued in rskq_rst_head. */

View File

@ -124,7 +124,7 @@ static int xsk_map_gen_lookup(struct bpf_map *map, struct bpf_insn *insn_buf)
struct bpf_insn *insn = insn_buf;
*insn++ = BPF_LDX_MEM(BPF_W, ret, index, 0);
*insn++ = BPF_JMP_IMM(BPF_JGE, ret, map->max_entries, 5);
*insn++ = BPF_JMP32_IMM(BPF_JGE, ret, map->max_entries, 5);
*insn++ = BPF_ALU64_IMM(BPF_LSH, ret, ilog2(sizeof(struct xsk_sock *)));
*insn++ = BPF_ALU64_IMM(BPF_ADD, mp, offsetof(struct xsk_map, xsk_map));
*insn++ = BPF_ALU64_REG(BPF_ADD, ret, mp);

View File

@ -6206,6 +6206,13 @@ bpf_object__relocate_core(struct bpf_object *obj, const char *targ_btf_path)
return -EINVAL;
insn = &prog->insns[insn_idx];
if (is_ldimm64_insn(insn) && (size_t)insn_idx + 1 >= prog->insns_cnt) {
pr_warn("prog '%s': relo #%d: insn #%d (LDIMM64) is truncated\n",
prog->name, i, insn_idx);
err = -EINVAL;
goto out;
}
err = record_relo_core(prog, rec, insn_idx);
if (err) {
pr_warn("prog '%s': relo #%d: failed to record relocation: %s\n",

View File

@ -980,23 +980,30 @@ static int bpf_core_calc_relo(const char *prog_name,
}
/*
* Turn instruction for which CO_RE relocation failed into invalid one with
* Turn instruction for which CO-RE relocation failed into invalid one with
* distinct signature.
*/
static void bpf_core_poison_insn(const char *prog_name, int relo_idx,
int insn_idx, struct bpf_insn *insn)
static int bpf_core_poison_insn(const char *prog_name, int relo_idx,
struct bpf_insn *insn, int insn_idx)
{
pr_debug("prog '%s': relo #%d: substituting insn #%d w/ invalid insn\n",
prog_name, relo_idx, insn_idx);
insn->code = BPF_JMP | BPF_CALL;
insn->dst_reg = 0;
insn->src_reg = 0;
insn->off = 0;
/* if this instruction is reachable (not a dead code),
* verifier will complain with the following message:
* invalid func unknown#195896080
*/
insn->imm = 195896080; /* => 0xbad2310 => "bad relo" */
int insn_cnt = is_ldimm64_insn(insn) ? 2 : 1;
int i;
for (i = 0; i < insn_cnt; i++) {
pr_debug("prog '%s': relo #%d: substituting insn #%d w/ invalid insn\n",
prog_name, relo_idx, insn_idx + i);
insn[i].code = BPF_JMP | BPF_CALL;
insn[i].dst_reg = 0;
insn[i].src_reg = 0;
insn[i].off = 0;
/*
* If this instruction is reachable (not dead code), the verifier
* will complain with "invalid func unknown#195896080".
*/
insn[i].imm = 195896080; /* => 0xbad2310 => "bad relo" */
}
return 0;
}
static int insn_bpf_size_to_bytes(struct bpf_insn *insn)
@ -1047,17 +1054,6 @@ int bpf_core_patch_insn(const char *prog_name, struct bpf_insn *insn,
class = BPF_CLASS(insn->code);
if (res->poison) {
poison:
/* poison second part of ldimm64 to avoid confusing error from
* verifier about "unknown opcode 00"
*/
if (is_ldimm64_insn(insn))
bpf_core_poison_insn(prog_name, relo_idx, insn_idx + 1, insn + 1);
bpf_core_poison_insn(prog_name, relo_idx, insn_idx, insn);
return 0;
}
orig_val = res->orig_val;
new_val = res->new_val;
@ -1065,7 +1061,9 @@ int bpf_core_patch_insn(const char *prog_name, struct bpf_insn *insn,
case BPF_ALU:
case BPF_ALU64:
if (BPF_SRC(insn->code) != BPF_K)
return -EINVAL;
goto bad_insn;
if (res->poison)
return bpf_core_poison_insn(prog_name, relo_idx, insn, insn_idx);
if (res->validate && insn->imm != orig_val) {
pr_warn("prog '%s': relo #%d: unexpected insn #%d (ALU/ALU64) value: got %d, exp %llu -> %llu\n",
prog_name, relo_idx,
@ -1082,6 +1080,8 @@ int bpf_core_patch_insn(const char *prog_name, struct bpf_insn *insn,
case BPF_LDX:
case BPF_ST:
case BPF_STX:
if (res->poison)
return bpf_core_poison_insn(prog_name, relo_idx, insn, insn_idx);
if (res->validate && insn->off != orig_val) {
pr_warn("prog '%s': relo #%d: unexpected insn #%d (LDX/ST/STX) value: got %d, exp %llu -> %llu\n",
prog_name, relo_idx, insn_idx, insn->off, (unsigned long long)orig_val,
@ -1097,7 +1097,7 @@ int bpf_core_patch_insn(const char *prog_name, struct bpf_insn *insn,
pr_warn("prog '%s': relo #%d: insn #%d (LDX/ST/STX) accesses field incorrectly. "
"Make sure you are accessing pointers, unsigned integers, or fields of matching type and size.\n",
prog_name, relo_idx, insn_idx);
goto poison;
return bpf_core_poison_insn(prog_name, relo_idx, insn, insn_idx);
}
orig_val = insn->off;
@ -1140,6 +1140,9 @@ int bpf_core_patch_insn(const char *prog_name, struct bpf_insn *insn,
return -EINVAL;
}
if (res->poison)
return bpf_core_poison_insn(prog_name, relo_idx, insn, insn_idx);
imm = (__u32)insn[0].imm | ((__u64)insn[1].imm << 32);
if (res->validate && imm != orig_val) {
pr_warn("prog '%s': relo #%d: unexpected insn #%d (LDIMM64) value: got %llu, exp %llu -> %llu\n",
@ -1157,6 +1160,7 @@ int bpf_core_patch_insn(const char *prog_name, struct bpf_insn *insn,
break;
}
default:
bad_insn:
pr_warn("prog '%s': relo #%d: trying to relocate unrecognized insn #%d, code:0x%x, src:0x%x, dst:0x%x, off:0x%x, imm:0x%x\n",
prog_name, relo_idx, insn_idx, insn->code,
(unsigned)insn->src_reg, (unsigned)insn->dst_reg, (unsigned)insn->off, (unsigned)insn->imm);

View File

@ -13,7 +13,7 @@ struct {
} cb_refs_tests[] = {
{ "underflow_prog", "release kfunc bpf_kfunc_call_test_release expects referenced PTR_TO_BTF_ID passed to R1" },
{ "leak_prog", "Possibly NULL pointer passed to helper R2" },
{ "nested_cb", "Unreleased reference id=4 alloc_insn=2" }, /* alloc_insn=2{4,5} */
{ "nested_cb", "Unreleased reference id=5 alloc_insn=2" }, /* alloc_insn=2{4,5} */
{ "non_cb_transfer_ref", "Unreleased reference id=4 alloc_insn=1" }, /* alloc_insn=1{1,2} */
};

View File

@ -14,6 +14,197 @@
static char log[16 * 1024];
static int load_core_relo_insns(int btf_fd, struct bpf_insn *insns, int insn_cnt,
struct bpf_func_info *funcs, int func_cnt,
int enum_id, int access_str_off, int insn_idx,
bool relocate)
{
struct bpf_core_relo relo = {
.insn_off = insn_idx * sizeof(struct bpf_insn),
.type_id = enum_id,
.access_str_off = access_str_off,
.kind = BPF_CORE_ENUMVAL_VALUE,
};
union bpf_attr attr = {
.prog_type = BPF_PROG_TYPE_SOCKET_FILTER,
.insn_cnt = insn_cnt,
.insns = (__u64)insns,
.license = (__u64)"GPL",
.log_buf = (__u64)log,
.log_size = sizeof(log),
.log_level = 2,
.prog_btf_fd = btf_fd,
.func_info_rec_size = sizeof(struct bpf_func_info),
.func_info = (__u64)funcs,
.func_info_cnt = func_cnt,
};
if (relocate) {
attr.core_relo_cnt = 1;
attr.core_relos = (__u64)&relo;
attr.core_relo_rec_size = sizeof(relo);
}
memset(log, 0, sizeof(log));
return sys_bpf_prog_load(&attr, sizeof(attr), 1);
}
static void test_early_core_relo(void)
{
static const char unrecognized[] = "trying to relocate unrecognized insn #2";
static const struct {
const char *name;
struct bpf_insn insns[2];
const char *err_msg;
} tests[] = {
{ "poison_exit", { BPF_EXIT_INSN() }, unrecognized },
{ "poison_ja", { BPF_JMP_A(1) }, unrecognized },
{ "poison_jmp", { BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 1) }, unrecognized },
{ "poison_jmp32", { BPF_JMP32_IMM(BPF_JEQ, BPF_REG_0, 0, 1) }, unrecognized },
{ "poison_call", { BPF_EMIT_CALL(BPF_FUNC_get_prandom_u32) }, unrecognized },
{ "poison_alu_reg", { BPF_MOV32_REG(BPF_REG_0, BPF_REG_1) }, unrecognized },
{ "poison_alu64_reg", { BPF_MOV64_REG(BPF_REG_0, BPF_REG_1) }, unrecognized },
{ "poison_ld_abs", { BPF_LD_ABS(BPF_W, 0) },
"insn #2 (LDIMM64) has unexpected form" },
{ "poison_alu_imm", { BPF_MOV32_IMM(BPF_REG_0, 0) } },
{ "poison_alu64_imm", { BPF_MOV64_IMM(BPF_REG_0, 0) } },
{ "poison_ldx", { BPF_LDX_MEM(BPF_W, BPF_REG_0, BPF_REG_1, 0) } },
{ "poison_st", { BPF_ST_MEM(BPF_W, BPF_REG_10, -4, 0) } },
{ "poison_stx", { BPF_STX_MEM(BPF_W, BPF_REG_10, BPF_REG_0, -4) } },
{ "poison_ldimm64", { BPF_LD_IMM64(BPF_REG_0, 0) } },
};
struct test_btf {
struct btf_header hdr;
__u32 types[18];
char strings[64];
} raw_btf = {
.hdr = {
.magic = BTF_MAGIC,
.version = BTF_VERSION,
.hdr_len = sizeof(struct btf_header),
.type_off = 0,
.type_len = sizeof(raw_btf.types),
.str_off = offsetof(struct test_btf, strings) -
offsetof(struct test_btf, types),
.str_len = sizeof(raw_btf.strings),
},
.types = {
BTF_TYPE_INT_ENC(1, BTF_INT_SIGNED, 0, 32, 4), /* [1] int */
BTF_FUNC_PROTO_ENC(1, 0), /* [2] int (*)(void) */
BTF_FUNC_ENC(5, 2), /* [3] main_fn */
BTF_FUNC_ENC(13, 2), /* [4] sub_fn */
BTF_TYPE_ENC(20, BTF_INFO_ENC(BTF_KIND_ENUM, 0, 1), 4), /* [5] enum */
BTF_ENUM_ENC(45, 0), /* value = 0 */
},
.strings = "\0int\0main_fn\0sub_fn\0core_relo_poison_missing\0value\0" "0",
};
struct bpf_func_info funcs[] = {
{ .insn_off = 0, .type_id = 3 },
{ .insn_off = 3, .type_id = 4 },
};
struct bpf_insn core_only[] = {
BPF_MOV64_IMM(BPF_REG_0, 0),
BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 1),
BPF_MOV64_IMM(BPF_REG_0, 0),
BPF_EXIT_INSN(),
};
struct bpf_insn subprog[] = {
BPF_CALL_REL(2),
BPF_MOV64_IMM(BPF_REG_0, 0),
BPF_MOV64_IMM(BPF_REG_0, 0),
BPF_MOV64_IMM(BPF_REG_0, 0),
BPF_EXIT_INSN(),
};
struct bpf_insn truncated_ldimm64[] = {
BPF_RAW_INSN(BPF_LD | BPF_IMM | BPF_DW, 0, 0, 0, 0),
};
int access_str_off = 51; /* offset of "0" */
int enum_id = 5;
int btf_fd, prog_fd = -1, i;
btf_fd = bpf_btf_load(&raw_btf, sizeof(raw_btf), NULL);
if (!ASSERT_GE(btf_fd, 0, "btf_load"))
goto cleanup;
if (test__start_subtest("without_func_info")) {
prog_fd = load_core_relo_insns(btf_fd, core_only, ARRAY_SIZE(core_only), NULL, 0,
enum_id, access_str_off, 2, false);
if (!ASSERT_GE(prog_fd, 0, "control_load"))
goto cleanup;
close(prog_fd);
prog_fd = load_core_relo_insns(btf_fd, core_only, ARRAY_SIZE(core_only), NULL, 0,
enum_id, access_str_off, 2, true);
if (!ASSERT_GE(prog_fd, 0, "poisoned_load"))
goto cleanup;
ASSERT_HAS_SUBSTR(log, "substituting insn #2", "poison_log");
close(prog_fd);
prog_fd = -1;
}
if (test__start_subtest("before_subprog_validation")) {
prog_fd = load_core_relo_insns(btf_fd, subprog, ARRAY_SIZE(subprog), funcs, 2,
enum_id, access_str_off, 2, true);
if (!ASSERT_LT(prog_fd, 0, "poisoned_load"))
goto cleanup;
ASSERT_HAS_SUBSTR(log, "substituting insn #2", "poison_log");
ASSERT_HAS_SUBSTR(log, "last insn is not an exit or jmp", "poisoned_load_log");
}
if (test__start_subtest("truncated_ldimm64")) {
prog_fd = load_core_relo_insns(btf_fd, truncated_ldimm64,
ARRAY_SIZE(truncated_ldimm64), NULL, 0,
enum_id, access_str_off, 0, true);
if (!ASSERT_LT(prog_fd, 0, "truncated_load"))
goto cleanup;
ASSERT_HAS_SUBSTR(log, "invalid bpf_ld_imm64 insn", "truncated_load_log");
}
for (i = 0; i < ARRAY_SIZE(tests); i++) {
struct bpf_insn insns[] = {
BPF_MOV64_IMM(BPF_REG_0, 0),
BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 1),
tests[i].insns[0],
BPF_MOV64_IMM(BPF_REG_0, 0),
BPF_EXIT_INSN(),
};
bool is_ldimm64 = insns[2].code == (BPF_LD | BPF_DW | BPF_IMM);
if (!test__start_subtest(tests[i].name))
continue;
if (is_ldimm64) {
insns[1].off = 2;
insns[3] = tests[i].insns[1];
}
prog_fd = load_core_relo_insns(btf_fd, insns, ARRAY_SIZE(insns), funcs, 1,
enum_id, access_str_off, 2, false);
if (!ASSERT_GE(prog_fd, 0, "control_load"))
goto cleanup;
close(prog_fd);
prog_fd = load_core_relo_insns(btf_fd, insns, ARRAY_SIZE(insns), funcs, 1,
enum_id, access_str_off, 2, true);
if (!tests[i].err_msg) {
ASSERT_GE(prog_fd, 0, "dead_poison_load");
ASSERT_HAS_SUBSTR(log, "substituting insn #2", "poison_log");
if (is_ldimm64)
ASSERT_HAS_SUBSTR(log, "substituting insn #3", "poison_ldimm64_log");
} else {
ASSERT_LT(prog_fd, 0, "invalid_poison_load");
ASSERT_HAS_SUBSTR(log, tests[i].err_msg, "invalid_poison_log");
ASSERT_NULL(strstr(log, "substituting insn"), "invalid_poison_substitution");
}
close(prog_fd);
prog_fd = -1;
}
cleanup:
if (env.verbosity > VERBOSE_NORMAL && log[0]) {
printf("-------- program load log start --------\n");
printf("%s", log);
printf("-------- program load log end ----------\n");
}
close(prog_fd);
close(btf_fd);
}
/* Check that verifier rejects BPF program containing relocation
* pointing to non-existent BTF type.
*/
@ -120,6 +311,7 @@ static void test_bad_local_id(void)
void test_core_reloc_raw(void)
{
test_early_core_relo();
if (test__start_subtest("bad_local_id"))
test_bad_local_id();
}

View File

@ -9,6 +9,7 @@
enum test_setup_type {
SETUP_SYSCALL_SLEEP,
SETUP_SKB_PROG,
SETUP_SKB_PROG_NONLINEAR,
SETUP_SKB_PROG_TP,
SETUP_XDP_PROG,
};
@ -32,6 +33,7 @@ static struct {
{"test_ringbuf", SETUP_SYSCALL_SLEEP},
{"test_skb_readonly", SETUP_SKB_PROG},
{"test_dynptr_skb_data", SETUP_SKB_PROG},
{"test_dynptr_skb_slice_non_linear", SETUP_SKB_PROG_NONLINEAR},
{"test_dynptr_skb_meta_data", SETUP_SKB_PROG},
{"test_dynptr_skb_meta_flags", SETUP_SKB_PROG},
{"test_adjust", SETUP_SYSCALL_SLEEP},
@ -94,7 +96,9 @@ static void verify_success(const char *prog_name, enum test_setup_type setup_typ
bpf_link__destroy(link);
break;
case SETUP_SKB_PROG:
case SETUP_SKB_PROG_NONLINEAR:
{
struct __sk_buff ctx = {};
int prog_fd;
char buf[64];
@ -106,6 +110,12 @@ static void verify_success(const char *prog_name, enum test_setup_type setup_typ
.repeat = 1,
);
if (setup_type == SETUP_SKB_PROG_NONLINEAR) {
ctx.data_end = ETH_HLEN + sizeof(struct iphdr);
topts.ctx_in = &ctx;
topts.ctx_size_in = sizeof(ctx);
}
prog_fd = bpf_program__fd(prog);
if (!ASSERT_GE(prog_fd, 0, "prog_fd"))
goto cleanup;

View File

@ -55,6 +55,7 @@ static void test_exceptions_success(void)
RUN_SUCCESS(exception_ext, 0);
RUN_SUCCESS(exception_ext_mod_cb_runtime, 35);
RUN_SUCCESS(exception_throw_subprog, 1);
RUN_SUCCESS(exception_throw_subprog_stack_cb, 0x1234);
RUN_SUCCESS(exception_assert_nz_gfunc, 1);
RUN_SUCCESS(exception_assert_zero_gfunc, 1);
RUN_SUCCESS(exception_assert_neg_gfunc, 1);

View File

@ -714,7 +714,7 @@ static void test_btf(void)
break;
err = btf__load_into_kernel(btf);
ASSERT_EQ(err, -ELOOP, "check btf");
ASSERT_EQ(err, 0, "check btf");
btf__free(btf);
break;
}
@ -773,7 +773,7 @@ static void test_btf(void)
break;
err = btf__load_into_kernel(btf);
ASSERT_EQ(err, -ELOOP, "check btf");
ASSERT_EQ(err, 0, "check btf");
btf__free(btf);
break;
}

View File

@ -0,0 +1,296 @@
// SPDX-License-Identifier: GPL-2.0
/* Copyright (c) 2026 Meta Platforms, Inc. and affiliates. */
#include <bpf/btf.h>
#include <linux/btf.h>
#include <test_progs.h>
#define SPIN_LOCK 2
#define LIST_HEAD 3
#define LIST_NODE 4
/* Keep in sync with BTF_MAX_OWNERSHIP_DEPTH. */
#define MAX_OWNERSHIP_DEPTH 8
static struct btf *init_btf(void)
{
struct btf *btf;
int id;
btf = btf__new_empty();
if (!ASSERT_OK_PTR(btf, "btf__new_empty"))
return NULL;
id = btf__add_int(btf, "int", 4, BTF_INT_SIGNED);
if (!ASSERT_EQ(id, 1, "btf__add_int"))
goto err_out;
id = btf__add_struct(btf, "bpf_spin_lock", 4);
if (!ASSERT_EQ(id, SPIN_LOCK, "btf__add_struct bpf_spin_lock"))
goto err_out;
id = btf__add_struct(btf, "bpf_list_head", 16);
if (!ASSERT_EQ(id, LIST_HEAD, "btf__add_struct bpf_list_head"))
goto err_out;
id = btf__add_struct(btf, "bpf_list_node", 24);
if (!ASSERT_EQ(id, LIST_NODE, "btf__add_struct bpf_list_node"))
goto err_out;
return btf;
err_out:
btf__free(btf);
return NULL;
}
static int add_local_kptr(struct btf *btf, int pointee_id, const char *tag)
{
int id;
id = btf__add_type_tag(btf, tag, pointee_id);
if (!ASSERT_GT(id, 0, "btf__add_type_tag"))
return id;
id = btf__add_ptr(btf, id);
ASSERT_GT(id, 0, "btf__add_ptr");
return id;
}
static void test_self_cycle(const char *tag, int expected_err)
{
struct btf *btf;
int id, err;
btf = init_btf();
if (!ASSERT_OK_PTR(btf, "init_btf"))
return;
id = add_local_kptr(btf, 7, tag);
if (id <= 0)
goto out;
id = btf__add_struct(btf, "self_cycle", 8);
if (!ASSERT_EQ(id, 7, "btf__add_struct self_cycle"))
goto out;
err = btf__add_field(btf, "next", 6, 0, 0);
if (!ASSERT_OK(err, "btf__add_field self_cycle::next"))
goto out;
err = btf__load_into_kernel(btf);
ASSERT_EQ(err, expected_err, "check btf");
out:
btf__free(btf);
}
static void test_aba_cycle(void)
{
struct btf *btf;
int id, err;
btf = init_btf();
if (!ASSERT_OK_PTR(btf, "init_btf"))
return;
id = add_local_kptr(btf, 10, "kptr");
if (id <= 0)
goto out;
id = add_local_kptr(btf, 9, "kptr");
if (id <= 0)
goto out;
id = btf__add_struct(btf, "cycle_a", 8);
if (!ASSERT_EQ(id, 9, "btf__add_struct cycle_a"))
goto out;
err = btf__add_field(btf, "b", 6, 0, 0);
if (!ASSERT_OK(err, "btf__add_field cycle_a::b"))
goto out;
id = btf__add_struct(btf, "cycle_b", 8);
if (!ASSERT_EQ(id, 10, "btf__add_struct cycle_b"))
goto out;
err = btf__add_field(btf, "a", 8, 0, 0);
if (!ASSERT_OK(err, "btf__add_field cycle_b::a"))
goto out;
err = btf__load_into_kernel(btf);
ASSERT_EQ(err, -ELOOP, "check btf");
out:
btf__free(btf);
}
static void test_mixed_cycle(void)
{
struct btf *btf;
int id, err;
btf = init_btf();
if (!ASSERT_OK_PTR(btf, "init_btf"))
return;
id = add_local_kptr(btf, 7, "kptr");
if (id <= 0)
goto out;
id = btf__add_struct(btf, "mixed_owner", 20);
if (!ASSERT_EQ(id, 7, "btf__add_struct mixed_owner"))
goto out;
err = btf__add_field(btf, "root", LIST_HEAD, 0, 0);
if (!ASSERT_OK(err, "btf__add_field mixed_owner::root"))
goto out;
err = btf__add_field(btf, "lock", SPIN_LOCK, 128, 0);
if (!ASSERT_OK(err, "btf__add_field mixed_owner::lock"))
goto out;
id = btf__add_decl_tag(btf, "contains:mixed_node:node", 7, 0);
if (!ASSERT_EQ(id, 8, "btf__add_decl_tag mixed_owner"))
goto out;
id = btf__add_struct(btf, "mixed_node", 32);
if (!ASSERT_EQ(id, 9, "btf__add_struct mixed_node"))
goto out;
err = btf__add_field(btf, "node", LIST_NODE, 0, 0);
if (!ASSERT_OK(err, "btf__add_field mixed_node::node"))
goto out;
err = btf__add_field(btf, "owner", 6, 192, 0);
if (!ASSERT_OK(err, "btf__add_field mixed_node::owner"))
goto out;
err = btf__load_into_kernel(btf);
ASSERT_EQ(err, -ELOOP, "check btf");
out:
btf__free(btf);
}
static void test_acyclic_depth(int depth, bool child_first, bool shared_suffix, int expected_err)
{
int ptr_id[MAX_OWNERSHIP_DEPTH + 1];
int first_struct_id;
struct btf *btf;
int id, err, i, n, pointee_id;
btf = init_btf();
if (!ASSERT_OK_PTR(btf, "init_btf"))
return;
first_struct_id = 5 + 2 * depth;
for (i = 0; i < depth; i++) {
if (i == depth - 1)
pointee_id = first_struct_id + depth;
else
pointee_id = first_struct_id + (child_first ? depth - 2 - i : i + 1);
ptr_id[i] = add_local_kptr(btf, pointee_id, "kptr");
if (ptr_id[i] <= 0)
goto out;
}
for (n = 0; n < depth; n++) {
char name[32];
int offset = 0;
i = child_first ? depth - 1 - n : n;
snprintf(name, sizeof(name), "owner_%d", i);
id = btf__add_struct(btf, name, shared_suffix && !i ? 16 : 8);
if (!ASSERT_EQ(id, first_struct_id + n, "btf__add_struct owner"))
goto out;
if (shared_suffix && !i) {
/*
* Visit the shared suffix through the shorter path before
* reaching it again with less remaining depth.
*/
err = btf__add_field(btf, "suffix", ptr_id[1], 0, 0);
if (!ASSERT_OK(err, "btf__add_field owner::suffix"))
goto out;
offset = 64;
}
err = btf__add_field(btf, "next", ptr_id[i], offset, 0);
if (!ASSERT_OK(err, "btf__add_field owner::next"))
goto out;
}
id = btf__add_struct(btf, "plain_leaf", 4);
if (!ASSERT_EQ(id, first_struct_id + depth, "btf__add_struct plain_leaf"))
goto out;
err = btf__load_into_kernel(btf);
ASSERT_EQ(err, expected_err, "check btf");
out:
btf__free(btf);
}
static void test_graph_depth(bool rbtree, int depth, int expected_err)
{
int root_type = LIST_HEAD, node_type = LIST_NODE, node_size = 24;
int id, err, i, lock_off, root_off, size;
struct btf *btf;
btf = init_btf();
if (!ASSERT_OK_PTR(btf, "init_btf"))
return;
if (rbtree) {
root_type = btf__add_struct(btf, "bpf_rb_root", 16);
if (!ASSERT_GT(root_type, 0, "btf__add_struct bpf_rb_root"))
goto out;
node_type = btf__add_struct(btf, "bpf_rb_node", 32);
if (!ASSERT_GT(node_type, 0, "btf__add_struct bpf_rb_node"))
goto out;
node_size = 32;
}
for (i = 0; i < depth; i++) {
char name[32], tag[64];
lock_off = i ? node_size : 0;
root_off = lock_off + 8;
size = i == depth - 1 ? node_size : root_off + 16;
snprintf(name, sizeof(name), "graph_owner_%d", i);
id = btf__add_struct(btf, name, size);
if (!ASSERT_GT(id, 0, "btf__add_struct graph_owner"))
goto out;
if (i) {
err = btf__add_field(btf, "node", node_type, 0, 0);
if (!ASSERT_OK(err, "btf__add_field graph_owner::node"))
goto out;
}
if (i == depth - 1)
continue;
err = btf__add_field(btf, "lock", SPIN_LOCK, lock_off * 8, 0);
if (!ASSERT_OK(err, "btf__add_field graph_owner::lock"))
goto out;
err = btf__add_field(btf, "root", root_type, root_off * 8, 0);
if (!ASSERT_OK(err, "btf__add_field graph_owner::root"))
goto out;
snprintf(tag, sizeof(tag), "contains:graph_owner_%d:node", i + 1);
err = btf__add_decl_tag(btf, tag, id, i ? 2 : 1);
if (!ASSERT_GT(err, 0, "btf__add_decl_tag graph_owner"))
goto out;
}
err = btf__load_into_kernel(btf);
ASSERT_EQ(err, expected_err, "check btf");
out:
btf__free(btf);
}
void test_local_kptr_ownership(void)
{
if (test__start_subtest("self_cycle"))
test_self_cycle("kptr", -ELOOP);
if (test__start_subtest("untrusted_self_cycle"))
test_self_cycle("kptr_untrusted", 0);
if (test__start_subtest("percpu_self_cycle"))
test_self_cycle("percpu_kptr", -ELOOP);
if (test__start_subtest("ABA_cycle"))
test_aba_cycle();
if (test__start_subtest("mixed_graph_root_cycle"))
test_mixed_cycle();
if (test__start_subtest("max_acyclic"))
test_acyclic_depth(MAX_OWNERSHIP_DEPTH, false, false, 0);
if (test__start_subtest("too_deep_acyclic"))
test_acyclic_depth(MAX_OWNERSHIP_DEPTH + 1, false, false, -ELOOP);
if (test__start_subtest("max_acyclic_child_first"))
test_acyclic_depth(MAX_OWNERSHIP_DEPTH, true, false, 0);
if (test__start_subtest("too_deep_acyclic_child_first"))
test_acyclic_depth(MAX_OWNERSHIP_DEPTH + 1, true, false, -ELOOP);
if (test__start_subtest("max_acyclic_shared_suffix"))
test_acyclic_depth(MAX_OWNERSHIP_DEPTH, false, true, 0);
if (test__start_subtest("too_deep_acyclic_shared_suffix"))
test_acyclic_depth(MAX_OWNERSHIP_DEPTH + 1, false, true, -ELOOP);
if (test__start_subtest("list_three_types"))
test_graph_depth(false, 3, 0);
if (test__start_subtest("list_four_types"))
test_graph_depth(false, 4, 0);
if (test__start_subtest("list_max_depth"))
test_graph_depth(false, MAX_OWNERSHIP_DEPTH, 0);
if (test__start_subtest("list_too_deep"))
test_graph_depth(false, MAX_OWNERSHIP_DEPTH + 1, -ELOOP);
if (test__start_subtest("rbtree_three_types"))
test_graph_depth(true, 3, 0);
if (test__start_subtest("rbtree_four_types"))
test_graph_depth(true, 4, 0);
if (test__start_subtest("rbtree_max_depth"))
test_graph_depth(true, MAX_OWNERSHIP_DEPTH, 0);
if (test__start_subtest("rbtree_too_deep"))
test_graph_depth(true, MAX_OWNERSHIP_DEPTH + 1, -ELOOP);
}

View File

@ -1,4 +1,6 @@
// SPDX-License-Identifier: GPL-2.0
#define _GNU_SOURCE
#include <sched.h>
#include <test_progs.h>
#include "cgroup_helpers.h"
#include "percpu_alloc_array.skel.h"
@ -350,6 +352,103 @@ static void test_lru_percpu_hash_cpu_flag(void)
test_percpu_map_cpu_flag(BPF_MAP_TYPE_LRU_PERCPU_HASH);
}
/*
* A BPF_F_CPU update that creates an element must zero the value on the other
* cpus, rather than leave them holding whatever the recycled element last
* contained. max_entries is 1 so the second key can only reuse the element
* the first one released.
*/
static void test_percpu_map_cpu_flag_create(enum bpf_map_type map_type, __u32 map_flags)
{
LIBBPF_OPTS(bpf_map_create_opts, opts, .map_flags = map_flags);
const u32 stale = 0xDEADC0DE, fresh = 0xC0FFEE;
int nr_cpus, cpu, map_fd, err, key;
int pinned_cpu, value_cpu;
cpu_set_t old_mask, new_mask;
bool restore_mask = false;
u32 value;
u64 flags;
nr_cpus = libbpf_num_possible_cpus();
if (!ASSERT_GT(nr_cpus, 0, "libbpf_num_possible_cpus"))
return;
if (nr_cpus < 2) {
test__skip();
return;
}
map_fd = bpf_map_create(map_type, "cpu_flag_create", sizeof(key), sizeof(value), 1, &opts);
if (!ASSERT_GE(map_fd, 0, "bpf_map_create"))
return;
/* NO_PREALLOC recycles per cpu, so keep the delete and the create on one cpu. */
err = sched_getaffinity(0, sizeof(old_mask), &old_mask);
if (!ASSERT_OK(err, "sched_getaffinity"))
goto out;
pinned_cpu = sched_getcpu();
if (!ASSERT_GE(pinned_cpu, 0, "sched_getcpu"))
goto out;
CPU_ZERO(&new_mask);
CPU_SET(pinned_cpu, &new_mask);
err = sched_setaffinity(0, sizeof(new_mask), &new_mask);
if (!ASSERT_OK(err, "sched_setaffinity"))
goto out;
restore_mask = true;
value_cpu = pinned_cpu ? 0 : 1;
key = 1;
value = stale;
err = bpf_map_update_elem(map_fd, &key, &value, BPF_F_ALL_CPUS);
if (!ASSERT_OK(err, "bpf_map_update_elem all_cpus"))
goto out;
err = bpf_map_delete_elem(map_fd, &key);
if (!ASSERT_OK(err, "bpf_map_delete_elem"))
goto out;
key = 2;
value = fresh;
flags = (u64)value_cpu << 32 | BPF_F_CPU;
err = bpf_map_update_elem(map_fd, &key, &value, flags);
if (!ASSERT_OK(err, "bpf_map_update_elem specified cpu"))
goto out;
for (cpu = 0; cpu < nr_cpus; cpu++) {
value = 0;
flags = (u64)cpu << 32 | BPF_F_CPU;
err = bpf_map_lookup_elem_flags(map_fd, &key, &value, flags);
if (!ASSERT_OK(err, "bpf_map_lookup_elem_flags specified cpu"))
goto out;
if (!ASSERT_EQ(value, cpu == value_cpu ? fresh : 0, "value on specified cpu"))
goto out;
}
out:
if (restore_mask)
sched_setaffinity(0, sizeof(old_mask), &old_mask);
close(map_fd);
}
static void test_percpu_hash_cpu_flag_create(void)
{
test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_PERCPU_HASH, 0);
}
static void test_percpu_hash_cpu_flag_create_malloc(void)
{
test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_PERCPU_HASH, BPF_F_NO_PREALLOC);
}
static void test_lru_percpu_hash_cpu_flag_create(void)
{
/* lru without prealloc is -ENOTSUPP, so there is no malloc variant */
test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_LRU_PERCPU_HASH, 0);
}
static void test_percpu_cgroup_storage_cpu_flag(void)
{
struct percpu_alloc_array *skel = NULL;
@ -454,6 +553,12 @@ void test_percpu_alloc(void)
test_percpu_hash_cpu_flag();
if (test__start_subtest("cpu_flag_lru_percpu_hash"))
test_lru_percpu_hash_cpu_flag();
if (test__start_subtest("cpu_flag_create_percpu_hash"))
test_percpu_hash_cpu_flag_create();
if (test__start_subtest("cpu_flag_create_percpu_hash_malloc"))
test_percpu_hash_cpu_flag_create_malloc();
if (test__start_subtest("cpu_flag_create_lru_percpu_hash"))
test_lru_percpu_hash_cpu_flag_create();
if (test__start_subtest("cpu_flag_percpu_cgroup_storage"))
test_percpu_cgroup_storage_cpu_flag();
if (test__start_subtest("cpu_flag_array"))

View File

@ -1,4 +1,5 @@
// SPDX-License-Identifier: GPL-2.0
#include <poll.h>
#include <test_progs.h>
#include <bpf/bpf_endian.h>
@ -110,6 +111,122 @@ static void test_tcp_server(struct sock_destroy_prog *skel)
close(serv);
}
static void test_tcp_listen_pending(struct sock_destroy_prog *skel)
{
int serv = -1, clien = -1, accept_serv = -1, n, serv_port;
struct pollfd pfd = { .events = POLLIN };
char buf[1];
serv = start_server(AF_INET6, SOCK_STREAM, NULL, 0, 0);
if (!ASSERT_GE(serv, 0, "start_server"))
goto cleanup;
serv_port = get_socket_local_port(serv);
if (!ASSERT_GE(serv_port, 0, "get_sock_local_port"))
goto cleanup;
skel->bss->serv_port = (__be16)serv_port;
/*
* Connect but never accept, so the child sits in the accept queue
* of the listener. Wait until it's actually there.
*/
clien = connect_to_fd(serv, 0);
if (!ASSERT_GE(clien, 0, "connect_to_fd"))
goto cleanup;
pfd.fd = serv;
if (!ASSERT_EQ(poll(&pfd, 1, -1), 1, "poll listener"))
goto cleanup;
/* Run iterator program that destroys server sockets. */
start_iter_sockets(skel->progs.iter_tcp6_server);
accept_serv = accept(serv, NULL, NULL);
if (!ASSERT_LT(accept_serv, 0, "accept on destroyed listener"))
goto cleanup;
ASSERT_EQ(errno, EINVAL, "error code on destroyed listener");
/* The unaccepted child was reset along with the listener. */
n = recv(clien, buf, sizeof(buf), 0);
if (!ASSERT_LT(n, 0, "client recv on reset child"))
goto cleanup;
ASSERT_EQ(errno, ECONNRESET, "error code on reset child");
cleanup:
if (clien != -1)
close(clien);
if (accept_serv != -1)
close(accept_serv);
if (serv != -1)
close(serv);
}
static void test_tcp_timewait(struct sock_destroy_prog *skel)
{
int serv = -1, clien = -1, accept_serv = -1, n;
struct timeval tv = {};
char buf[1];
serv = start_server(AF_INET6, SOCK_STREAM, NULL, 0, 0);
if (!ASSERT_GE(serv, 0, "start_server"))
goto cleanup;
clien = connect_to_fd(serv, 0);
if (!ASSERT_GE(clien, 0, "connect_to_fd"))
goto cleanup;
accept_serv = accept(serv, NULL, NULL);
if (!ASSERT_GE(accept_serv, 0, "serv accept"))
goto cleanup;
/*
* Active close from the client, then close the server side. Once
* recv() sees EOF the server FIN has been processed and the client
* sock is in TIME_WAIT. Block without timeout so a loaded CI box
* can't race us.
*/
if (!ASSERT_OK(setsockopt(clien, SOL_SOCKET, SO_RCVTIMEO, &tv,
sizeof(tv)), "clear rcvtimeo"))
goto cleanup;
if (!ASSERT_OK(shutdown(clien, SHUT_WR), "client shutdown"))
goto cleanup;
/*
* Make sure the server has seen the client FIN before it closes,
* so the two FINs never cross.
*/
n = recv(accept_serv, buf, sizeof(buf), 0);
if (!ASSERT_EQ(n, 0, "server recv EOF"))
goto cleanup;
close(accept_serv);
accept_serv = -1;
/* block until return EOF */
n = recv(clien, buf, sizeof(buf), 0);
if (!ASSERT_EQ(n, 0, "client recv EOF"))
goto cleanup;
/* Run iterator program that destroys the timewait client sock. */
skel->bss->tw_found = 0;
start_iter_sockets(skel->progs.iter_tcp6_timewait);
if (!ASSERT_EQ(skel->bss->tw_found, 1, "timewait sock found"))
goto cleanup;
ASSERT_OK(skel->bss->tw_destroy_err, "destroy timewait sock");
/* The destroyed timewait sock must be gone. */
skel->bss->tw_found = 0;
start_iter_sockets(skel->progs.iter_tcp6_timewait);
ASSERT_EQ(skel->bss->tw_found, 0, "timewait sock destroyed");
cleanup:
if (clien != -1)
close(clien);
if (accept_serv != -1)
close(accept_serv);
if (serv != -1)
close(serv);
}
static void test_udp_client(struct sock_destroy_prog *skel)
{
int serv = -1, clien = -1, n = 0;
@ -204,6 +321,10 @@ void test_sock_destroy(void)
test_tcp_client(skel);
if (test__start_subtest("tcp_server"))
test_tcp_server(skel);
if (test__start_subtest("tcp_listen_pending"))
test_tcp_listen_pending(skel);
if (test__start_subtest("tcp_timewait"))
test_tcp_timewait(skel);
if (test__start_subtest("udp_client"))
test_udp_client(skel);
if (test__start_subtest("udp_server"))

View File

@ -54,6 +54,8 @@ static struct {
{ "lock_global_sleepable_helper_subprog", "global function calls are not allowed while holding a lock" },
{ "lock_global_sleepable_kfunc_subprog", "global function calls are not allowed while holding a lock" },
{ "lock_global_sleepable_subprog_indirect", "global function calls are not allowed while holding a lock" },
{ "callback_value_lock_identity", "bpf_spin_unlock of different lock" },
{ "callback_inner_map_value_lock_identity", "bpf_spin_unlock of different lock" },
};
static int match_regex(const char *pattern, const char *string)

View File

@ -0,0 +1,125 @@
// SPDX-License-Identifier: GPL-2.0
#include <netinet/tcp.h>
#include "test_progs.h"
#include "network_helpers.h"
#include "test_tc_change_tail_pmtu.skel.h"
#define CLIENT_NS "tc-change-tail-cli-ns"
#define SERVER_NS "tc-change-tail-srv-ns"
#define CLIENT_IP "192.168.1.1"
#define SERVER_IP "192.168.1.2"
#define TEST_PMTU 1000
#define TEST_MSS_MAX (TEST_PMTU - 20 - 20)
#define TIMEOUT_MS 3000
#define XFER_BYTES 8192
void test_tc_change_tail_pmtu(void)
{
LIBBPF_OPTS(bpf_tcx_opts, tcx_opts);
int mss_before = 0, mss_after = 0, ifindex, port;
int srv_fd = -1, srv_conn_fd = -1, cli_fd = -1;
struct test_tc_change_tail_pmtu *skel = NULL;
struct nstoken *nstoken = NULL;
static char buf[XFER_BYTES];
socklen_t optlen;
ssize_t bytes;
size_t total;
if (!ASSERT_OK(make_netns(CLIENT_NS), "make client ns"))
return;
if (!ASSERT_OK(make_netns(SERVER_NS), "make server ns"))
goto out_client_ns;
nstoken = open_netns(CLIENT_NS);
if (!ASSERT_OK_PTR(nstoken, "open client ns"))
goto out;
SYS(out, "ip link add veth1 type veth peer name veth2 netns " SERVER_NS);
SYS(out, "ip -4 addr add " CLIENT_IP "/24 dev veth1");
SYS(out, "ip link set veth1 up");
ifindex = if_nametoindex("veth1");
if (!ASSERT_NEQ(ifindex, 0, "if_nametoindex"))
goto out;
close_netns(nstoken);
nstoken = NULL;
nstoken = open_netns(SERVER_NS);
if (!ASSERT_OK_PTR(nstoken, "open server ns"))
goto out;
SYS(out, "ip -4 addr add " SERVER_IP "/24 dev veth2");
SYS(out, "ip link set veth2 up");
srv_fd = start_server(AF_INET, SOCK_STREAM, SERVER_IP, 0, TIMEOUT_MS);
if (!ASSERT_OK_FD(srv_fd, "start server"))
goto out;
close_netns(nstoken);
nstoken = NULL;
skel = test_tc_change_tail_pmtu__open_and_load();
if (!ASSERT_OK_PTR(skel, "open and load skeleton"))
goto out;
port = get_socket_local_port(srv_fd);
if (!ASSERT_GE(port, 0, "get server port"))
goto out;
skel->bss->server_port = port;
skel->bss->pmtu = TEST_PMTU;
nstoken = open_netns(CLIENT_NS);
if (!ASSERT_OK_PTR(nstoken, "open client ns"))
goto out;
skel->links.change_tail_icmp =
bpf_program__attach_tcx(skel->progs.change_tail_icmp, ifindex,
&tcx_opts);
if (!ASSERT_OK_PTR(skel->links.change_tail_icmp, "attach tcx"))
goto out;
cli_fd = connect_to_fd(srv_fd, TIMEOUT_MS);
if (!ASSERT_OK_FD(cli_fd, "connect to server"))
goto out;
srv_conn_fd = accept(srv_fd, NULL, NULL);
if (!ASSERT_OK_FD(srv_conn_fd, "accept connection"))
goto out;
if (!ASSERT_OK(settimeo(srv_conn_fd, TIMEOUT_MS), "set server timeout"))
goto out;
optlen = sizeof(mss_before);
if (!ASSERT_OK(getsockopt(cli_fd, IPPROTO_TCP, TCP_MAXSEG, &mss_before,
&optlen), "get mss before"))
goto out;
bytes = send(cli_fd, buf, sizeof(buf), 0);
if (!ASSERT_EQ(bytes, (ssize_t)sizeof(buf), "send data"))
goto out;
for (total = 0; total < sizeof(buf); total += bytes) {
bytes = recv(srv_conn_fd, buf, sizeof(buf), 0);
if (bytes <= 0)
break;
}
ASSERT_EQ(total, sizeof(buf), "receive data");
ASSERT_OK(skel->data->change_tail_ret, "change tail");
ASSERT_OK(skel->bss->adjust_room_ret, "adjust room");
ASSERT_TRUE(skel->bss->icmp_sent, "icmp sent");
optlen = sizeof(mss_after);
if (!ASSERT_OK(getsockopt(cli_fd, IPPROTO_TCP, TCP_MAXSEG, &mss_after,
&optlen), "get mss after"))
goto out;
ASSERT_LT(mss_after, mss_before, "mss reduced");
ASSERT_LE(mss_after, TEST_MSS_MAX, "mss below pmtu");
out:
close(srv_conn_fd);
close(cli_fd);
close(srv_fd);
test_tc_change_tail_pmtu__destroy(skel);
close_netns(nstoken);
remove_netns(SERVER_NS);
out_client_ns:
remove_netns(CLIENT_NS);
}

View File

@ -23,6 +23,7 @@
#include "verifier_bpf_trap.skel.h"
#include "verifier_bswap.skel.h"
#include "verifier_btf_ctx_access.skel.h"
#include "verifier_btf_flex_array.skel.h"
#include "verifier_btf_unreliable_prog.skel.h"
#include "verifier_call_large_imm.skel.h"
#include "verifier_cfg.skel.h"
@ -53,6 +54,7 @@
#include "verifier_iterating_callbacks.skel.h"
#include "verifier_jeq_infer_not_null.skel.h"
#include "verifier_jit_convergence.skel.h"
#include "verifier_kfunc_perfmon.skel.h"
#include "verifier_ld_ind.skel.h"
#include "verifier_ldsx.skel.h"
#include "verifier_leak_ptr.skel.h"
@ -186,6 +188,7 @@ void test_verifier_bpf_get_stack(void) { RUN(verifier_bpf_get_stack); }
void test_verifier_bpf_trap(void) { RUN(verifier_bpf_trap); }
void test_verifier_bswap(void) { RUN(verifier_bswap); }
void test_verifier_btf_ctx_access(void) { RUN(verifier_btf_ctx_access); }
void test_verifier_btf_flex_array(void) { RUN(verifier_btf_flex_array); }
void test_verifier_btf_unreliable_prog(void) { RUN(verifier_btf_unreliable_prog); }
void test_verifier_call_large_imm(void) { RUN(verifier_call_large_imm); }
void test_verifier_cfg(void) { RUN(verifier_cfg); }
@ -216,6 +219,7 @@ void test_verifier_int_ptr(void) { RUN(verifier_int_ptr); }
void test_verifier_iterating_callbacks(void) { RUN(verifier_iterating_callbacks); }
void test_verifier_jeq_infer_not_null(void) { RUN(verifier_jeq_infer_not_null); }
void test_verifier_jit_convergence(void) { RUN(verifier_jit_convergence); }
void test_verifier_kfunc_perfmon(void) { RUN(verifier_kfunc_perfmon); }
void test_verifier_load_acquire(void) { RUN(verifier_load_acquire); }
void test_verifier_ld_ind(void) { RUN(verifier_ld_ind); }
void test_verifier_ldsx(void) { RUN(verifier_ldsx); }

View File

@ -10,6 +10,7 @@
#include "errno.h"
#define PAGE_SIZE_64K 65536
#define TEST_SKB_LINEAR_SIZE (sizeof(struct ethhdr) + sizeof(struct iphdr))
char _license[] SEC("license") = "GPL";
@ -211,6 +212,25 @@ int test_dynptr_skb_data(struct __sk_buff *skb)
return 1;
}
SEC("?tc")
int test_dynptr_skb_slice_non_linear(struct __sk_buff *skb)
{
struct bpf_dynptr ptr;
void *data;
if (bpf_dynptr_from_skb(skb, 0, &ptr)) {
err = 1;
return 1;
}
/* Ensure we cannot read past the end of the buffer. */
data = bpf_dynptr_slice(&ptr, TEST_SKB_LINEAR_SIZE + 1, NULL, 1);
if (data)
err = 2;
return 1;
}
SEC("?tc")
int test_dynptr_skb_meta_data(struct __sk_buff *skb)
{

View File

@ -212,6 +212,36 @@ int exception_throw_subprog(struct __sk_buff *ctx)
return 0;
}
u64 exception_cb_stack_src = 0x1234;
/*
* The address handed to the helper has to be this callback's own stack
* slot, not one from a frame that is already gone.
*/
__noinline int exception_cb_stack(u64 cookie)
{
volatile u64 val = 0xdead;
bpf_probe_read_kernel((void *)&val, sizeof(val), &exception_cb_stack_src);
return val;
}
/* Throws from a subprogram that has a stack of its own. */
__noinline static int throwing_subprog_stack(struct __sk_buff *ctx)
{
volatile u64 pad[4] = {};
bpf_throw(pad[0]);
return 0;
}
SEC("tc")
__exception_cb(exception_cb_stack)
int exception_throw_subprog_stack_cb(struct __sk_buff *ctx)
{
return throwing_subprog_stack(ctx);
}
__noinline int assert_nz_gfunc(u64 c)
{
volatile u64 cookie = c;

View File

@ -52,6 +52,28 @@ int create_and_destroy(void *ctx)
return 0;
}
/* fp+0 is not a stack slot. bpf_get_spi(0) used to alias spi 0 (fp-8). */
SEC("?raw_tp")
__failure __msg("cannot pass in iter at an offset=0")
int destroy_fp0_fail(void *ctx)
{
struct bpf_iter_num iter;
asm volatile ("r1 = %[iter];"
"r2 = 0;"
"r3 = 1000;"
"call %[bpf_iter_num_new];"
/* r10 is fp+0, one byte above the top of the BPF stack */
"r1 = r10;"
"call %[bpf_iter_num_destroy];"
:
: __imm_ptr(iter), ITER_HELPERS
: __clobber_common
);
return 0;
}
SEC("?raw_tp")
__failure __msg("Unreleased reference id=1")
int create_and_forget_to_destroy_fail(void *ctx)

View File

@ -7,6 +7,8 @@
#include "bpf_tracing_net.h"
__be16 serv_port = 0;
int tw_found = 0;
int tw_destroy_err = 0;
int bpf_sock_destroy(struct sock_common *sk) __ksym;
@ -100,6 +102,34 @@ int iter_tcp6_server(struct bpf_iter__tcp *ctx)
return 0;
}
SEC("iter/tcp")
int iter_tcp6_timewait(struct bpf_iter__tcp *ctx)
{
struct sock_common *sk_common = ctx->sk_common;
__u64 *val;
int key = 0;
if (!sk_common)
return 0;
if (sk_common->skc_family != AF_INET6)
return 0;
if (!bpf_skc_to_tcp_timewait_sock(sk_common))
return 0;
val = bpf_map_lookup_elem(&tcp_conn_sockets, &key);
if (!val)
return 0;
/* The timewait sock inherits the cookie of the closed client sock. */
if (bpf_get_socket_cookie(sk_common) != *val)
return 0;
tw_found++;
tw_destroy_err = bpf_sock_destroy(sk_common);
return 0;
}
SEC("iter/udp")
int iter_udp6_client(struct bpf_iter__udp *ctx)

View File

@ -14,17 +14,18 @@ struct array_map {
__type(key, int);
__type(value, struct foo);
__uint(max_entries, 1);
} array_map SEC(".maps");
} array_map SEC(".maps"), array_map_b SEC(".maps");
struct {
__uint(type, BPF_MAP_TYPE_ARRAY_OF_MAPS);
__uint(max_entries, 1);
__uint(max_entries, 2);
__type(key, int);
__type(value, int);
__array(values, struct array_map);
} map_of_maps SEC(".maps") = {
.values = {
[0] = &array_map,
[1] = &array_map_b,
},
};
@ -314,4 +315,66 @@ int lock_global_sleepable_subprog_indirect(struct __sk_buff *ctx)
return ret;
}
struct {
__uint(type, BPF_MAP_TYPE_ARRAY);
__uint(max_entries, 2);
__type(key, int);
__type(value, struct foo);
} callback_array_map SEC(".maps");
struct callback_ctx {
struct foo *value;
};
static long lock_different_value(struct bpf_map *map, int *key,
struct foo *value, struct callback_ctx *ctx)
{
bpf_spin_lock(&value->lock);
bpf_spin_unlock(&ctx->value->lock);
return 0;
}
static long nest_lock_different_value(struct bpf_map *map, int *key,
struct foo *value, void *data)
{
struct callback_ctx ctx = { .value = value };
bpf_for_each_map_elem(&callback_array_map, lock_different_value, &ctx, 0);
return 0;
}
SEC("?tc")
int callback_value_lock_identity(void *ctx)
{
bpf_for_each_map_elem(&callback_array_map, nest_lock_different_value, NULL, 0);
return 0;
}
static long nest_lock_different_inner_value(struct bpf_map *map, int *key,
struct foo *value, void *data)
{
struct callback_ctx ctx = { .value = value };
int inner_key = 1;
void *inner_map;
inner_map = bpf_map_lookup_elem(&map_of_maps, &inner_key);
if (!inner_map)
return 0;
bpf_for_each_map_elem(inner_map, lock_different_value, &ctx, 0);
return 0;
}
SEC("?tc")
int callback_inner_map_value_lock_identity(void *ctx)
{
int inner_key = 0;
void *inner_map;
inner_map = bpf_map_lookup_elem(&map_of_maps, &inner_key);
if (!inner_map)
return 0;
bpf_for_each_map_elem(inner_map, nest_lock_different_inner_value, NULL, 0);
return 0;
}
char _license[] SEC("license") = "GPL";

View File

@ -0,0 +1,129 @@
// SPDX-License-Identifier: GPL-2.0
#include <stdbool.h>
#include <stddef.h>
#include <linux/bpf.h>
#include <linux/icmp.h>
#include <linux/if_ether.h>
#include <linux/in.h>
#include <linux/ip.h>
#include <linux/tcp.h>
#include <bpf/bpf_helpers.h>
#include <bpf/bpf_endian.h>
#define ICMP_SAMPLE_LEN (sizeof(struct iphdr) + 8)
#define ICMP_HDRS_LEN (sizeof(struct iphdr) + sizeof(struct icmphdr))
__be16 server_port = 0;
__u16 pmtu = 0;
long change_tail_ret = 1;
long adjust_room_ret = 0;
bool icmp_sent = false;
bool icmp_err = false;
static __always_inline __sum16 csum_fold(__wsum csum)
{
csum = (csum & 0xffff) + (csum >> 16);
csum = (csum & 0xffff) + (csum >> 16);
return (__sum16)~csum;
}
SEC("tc/egress")
int change_tail_icmp(struct __sk_buff *skb)
{
__u8 smac[ETH_ALEN], dmac[ETH_ALEN];
void *data, *data_end;
struct icmphdr *icmp;
struct ethhdr *eth;
struct tcphdr *tcp;
__be32 saddr, daddr;
struct iphdr *ip;
__wsum csum;
if (icmp_sent || icmp_err)
return TCX_PASS;
data = (void *)(long)skb->data;
data_end = (void *)(long)skb->data_end;
eth = data;
if ((void *)(eth + 1) > data_end)
return TCX_PASS;
if (eth->h_proto != bpf_htons(ETH_P_IP))
return TCX_PASS;
ip = (void *)(eth + 1);
if ((void *)(ip + 1) > data_end)
return TCX_PASS;
if (ip->ihl != 5 || ip->protocol != IPPROTO_TCP)
return TCX_PASS;
tcp = (void *)(ip + 1);
if ((void *)(tcp + 1) > data_end)
return TCX_PASS;
if (tcp->dest != server_port)
return TCX_PASS;
if (bpf_ntohs(ip->tot_len) <= sizeof(*ip) + tcp->doff * 4)
return TCX_PASS;
__builtin_memcpy(smac, eth->h_source, ETH_ALEN);
__builtin_memcpy(dmac, eth->h_dest, ETH_ALEN);
saddr = ip->saddr;
daddr = ip->daddr;
change_tail_ret = bpf_skb_change_tail(skb, ETH_HLEN + ICMP_SAMPLE_LEN, 0);
if (change_tail_ret) {
icmp_err = true;
return TCX_PASS;
}
adjust_room_ret = bpf_skb_adjust_room(skb, ICMP_HDRS_LEN,
BPF_ADJ_ROOM_MAC,
BPF_F_ADJ_ROOM_NO_CSUM_RESET);
if (adjust_room_ret) {
icmp_err = true;
return TCX_DROP;
}
data = (void *)(long)skb->data;
data_end = (void *)(long)skb->data_end;
eth = data;
ip = (void *)(eth + 1);
icmp = (void *)(ip + 1);
if ((void *)icmp + sizeof(*icmp) + ICMP_SAMPLE_LEN > data_end) {
icmp_err = true;
return TCX_DROP;
}
__builtin_memcpy(eth->h_dest, smac, ETH_ALEN);
__builtin_memcpy(eth->h_source, dmac, ETH_ALEN);
__builtin_memset(icmp, 0, sizeof(*icmp));
icmp->type = ICMP_DEST_UNREACH;
icmp->code = ICMP_FRAG_NEEDED;
icmp->un.frag.mtu = bpf_htons(pmtu);
__builtin_memset(ip, 0, sizeof(*ip));
ip->version = 4;
ip->ihl = 5;
ip->ttl = 64;
ip->protocol = IPPROTO_ICMP;
ip->tot_len = bpf_htons(ICMP_HDRS_LEN + ICMP_SAMPLE_LEN);
ip->saddr = daddr;
ip->daddr = saddr;
csum = bpf_csum_diff(NULL, 0, (__be32 *)icmp,
sizeof(*icmp) + ICMP_SAMPLE_LEN, 0);
icmp->checksum = csum_fold(csum);
csum = bpf_csum_diff(NULL, 0, (__be32 *)ip, sizeof(*ip), 0);
ip->check = csum_fold(csum);
icmp_sent = true;
return bpf_redirect(skb->ifindex, BPF_F_INGRESS);
}
char _license[] SEC("license") = "GPL";

View File

@ -561,6 +561,55 @@ int arena_ptr_add_arena_ptr(void *ctx)
return 0;
}
SEC("syscall")
__failure __msg("same insn cannot be used with and without arena pointer")
int mixed_arena_scalar_alu64_scalar_first(void *ctx)
{
volatile register __u64 reg asm("r3");
__u32 pick_arena = bpf_get_prandom_u32();
reg = 1ULL << 32;
if (pick_arena) {
asm volatile (
"r9 = %[arena] ll;"
"%[reg] = 0;"
"%[reg] = addr_space_cast(%[reg], 0x0, 0x1);"
: [reg] "=r"(reg)
: __imm_addr(arena)
: "r9"
);
}
reg += 1;
return 0;
}
SEC("syscall")
__failure __msg("same insn cannot be used with and without arena pointer")
int mixed_arena_scalar_alu64_arena_first(void *ctx)
{
volatile register __u64 reg asm("r3");
__u32 pick_scalar = bpf_get_prandom_u32();
asm volatile (
"r9 = %[arena] ll;"
"%[reg] = 0;"
"%[reg] = addr_space_cast(%[reg], 0x0, 0x1);"
: [reg] "=r"(reg)
: __imm_addr(arena)
: "r9"
);
if (pick_scalar)
reg = 1ULL << 32;
reg += 1;
return 0;
}
SEC("syscall")
__success __retval(0)
int scalar_xor_arena_ptr(void *ctx)

View File

@ -9,6 +9,11 @@
char _license[] SEC("license") = "GPL";
struct task_struct *bpf_task_acquire(struct task_struct *p) __ksym;
void bpf_task_release(struct task_struct *p) __ksym;
void bpf_rcu_read_lock(void) __ksym;
void bpf_rcu_read_unlock(void) __ksym;
/* Timer tests */
struct timer_elem {
@ -164,6 +169,7 @@ int syscall_btf_find_prog(void *ctx)
struct wq_elem {
struct bpf_wq w;
struct task_struct __kptr *task;
};
struct {
@ -217,6 +223,106 @@ int wq_sleepable_prog(void *ctx)
return 0;
}
__noinline int wq_global_acquire(void)
{
struct task_struct *task, *acquired;
struct wq_elem *val;
int key = 0;
val = bpf_map_lookup_elem(&wq_map, &key);
if (!val)
return 0;
task = val->task;
if (!task)
return 0;
acquired = bpf_task_acquire(task);
if (acquired)
bpf_task_release(acquired);
return 0;
}
static int wq_global_rcu_cb(void *map, int *key, void *value)
{
wq_global_acquire();
return 0;
}
SEC("fentry/bpf_fentry_test1")
__failure __msg("R1 must be a rcu pointer")
int wq_global_rcu_prog(void *ctx)
{
struct wq_elem *val;
int key = 0;
val = bpf_map_lookup_elem(&wq_map, &key);
if (!val)
return 0;
bpf_wq_init(&val->w, &wq_map, 0);
bpf_wq_set_callback(&val->w, wq_global_rcu_cb, 0);
return 0;
}
static int wq_global_rcu_lock_cb(void *map, int *key, void *value)
{
bpf_rcu_read_lock();
wq_global_acquire();
bpf_rcu_read_unlock();
return 0;
}
SEC("fentry/bpf_fentry_test1")
__success
int wq_global_rcu_lock_prog(void *ctx)
{
struct wq_elem *val;
int key = 0;
/* Verify the same global subprog in non-sleepable and protected contexts. */
wq_global_acquire();
val = bpf_map_lookup_elem(&wq_map, &key);
if (!val)
return 0;
bpf_wq_init(&val->w, &wq_map, 0);
bpf_wq_set_callback(&val->w, wq_global_rcu_lock_cb, 0);
return 0;
}
__weak __noinline int wq_global_no_rcu(void)
{
return 0;
}
static int wq_global_no_rcu_cb(void *map, int *key, void *value)
{
wq_global_no_rcu();
return 0;
}
SEC("fentry/bpf_fentry_test1")
__success __log_level(4)
__msg("subprog {{[0-9]+}} (wq_global_no_rcu) global insns_self 4 insns_total 4 stack 0")
int wq_global_no_rcu_prog(void *ctx)
{
struct wq_elem *val;
int key = 0;
/* Verify the same global in non-sleepable and unprotected contexts. */
wq_global_no_rcu();
val = bpf_map_lookup_elem(&wq_map, &key);
if (!val)
return 0;
bpf_wq_init(&val->w, &wq_map, 0);
bpf_wq_set_callback(&val->w, wq_global_no_rcu_cb, 0);
return 0;
}
/* Task work tests */
struct task_work_elem {

View File

@ -0,0 +1,56 @@
// SPDX-License-Identifier: GPL-2.0
#include <vmlinux.h>
#include <bpf/bpf_helpers.h>
#include "bpf_experimental.h"
#include "bpf_misc.h"
struct test_empty_event {};
struct test_flex_batch {
int nr;
struct test_empty_event events[];
};
struct map_value {
struct test_flex_batch __kptr *batch;
};
struct {
__uint(type, BPF_MAP_TYPE_ARRAY);
__type(key, int);
__type(value, struct map_value);
__uint(max_entries, 1);
} batches SEC(".maps");
SEC("syscall")
__description("btf walk into flexible array of zero-sized elements")
__failure __msg("access beyond struct test_flex_batch at off 4 size 1")
int stash_and_peek(void *ctx)
{
struct test_flex_batch *b, *old;
struct map_value *v;
int key = 0;
v = bpf_map_lookup_elem(&batches, &key);
if (!v)
return 0;
b = bpf_obj_new(struct test_flex_batch);
if (!b)
return 0;
b->nr = 1;
old = bpf_kptr_xchg(&v->batch, b);
if (old)
bpf_obj_drop(old);
b = v->batch;
if (!b)
return 0;
return b->nr + *(char *)&b->events[0];
}
char _license[] SEC("license") = "GPL";

View File

@ -350,4 +350,57 @@ int anything_to_untrusted_mem(void *ctx)
return 0;
}
struct pkt_arg {
__u64 x;
__u8 pad[56];
};
__weak int subprog_pkt_ptr_no_change(struct pkt_arg *p)
{
if (!p)
return 0;
return p->x;
}
SEC("?tc")
__success
int pkt_ptr_to_global_mem_arg_no_change(struct __sk_buff *skb)
{
void *data = (void *)(long)skb->data;
void *data_end = (void *)(long)skb->data_end;
struct pkt_arg *p = data;
if ((void *)(p + 1) > data_end)
return 0;
return subprog_pkt_ptr_no_change(p);
}
__weak int subprog_pkt_ptr_changes_data(struct __sk_buff *skb __arg_ctx,
struct pkt_arg *p)
{
if (!p)
return 0;
bpf_skb_pull_data(skb, 0);
return p->x;
}
SEC("?tc")
__failure __log_level(2)
__msg("R2 is a packet pointer, but func#{{[0-9]+}} may change packet data")
__msg("Caller passes invalid args into func#{{[0-9]+}} ('subprog_pkt_ptr_changes_data')")
int pkt_ptr_to_global_mem_arg_changes_data(struct __sk_buff *skb)
{
void *data = (void *)(long)skb->data;
void *data_end = (void *)(long)skb->data_end;
struct pkt_arg *p = data;
if ((void *)(p + 1) > data_end)
return 0;
return subprog_pkt_ptr_changes_data(skb, p);
}
char _license[] SEC("license") = "GPL";

View File

@ -47,6 +47,54 @@ DEFINE_SIMPLE_JUMP_TABLE_PROG(reserved_field_src_reg, BPF_REG_1, 0, 0, __fa
DEFINE_SIMPLE_JUMP_TABLE_PROG(reserved_field_non_zero_off, BPF_REG_0, 1, 0, __failure __msg("BPF_JA|BPF_X uses reserved fields"))
DEFINE_SIMPLE_JUMP_TABLE_PROG(reserved_field_non_zero_imm, BPF_REG_0, 0, 1, __failure __msg("BPF_JA|BPF_X uses reserved fields"))
#define DEFINE_TERMINAL_GOTOX_PROG(NAME, BASE) \
__naked void NAME(void) \
{ \
asm volatile (" \
.pushsection .jumptables,\"\",@progbits; \
jt0_%=: \
.quad ret0_%= - " BASE "; \
.size jt0_%=, 8; \
.global jt0_%=; \
.popsection; \
\
r0 = jt0_%= ll; \
r0 = *(u64 *)(r0 + 0); \
goto end_%=; \
ret0_%=: \
r0 = 0; \
exit; \
end_%=: \
.8byte %[gotox_r0]; \
" : \
: __imm_insn(gotox_r0, BPF_RAW_INSN(BPF_JMP | BPF_JA | BPF_X, \
BPF_REG_0, 0, 0, 0)) \
: __clobber_all); \
}
SEC("socket")
__success __retval(0)
DEFINE_TERMINAL_GOTOX_PROG(jump_table_terminal_gotox, "socket")
static __noinline __used
DEFINE_TERMINAL_GOTOX_PROG(terminal_gotox_subprog1, ".text")
static __noinline __used int terminal_gotox_subprog2(void)
{
return 0;
}
SEC("socket")
__success __retval(0)
__naked void jump_table_terminal_gotox_subprog(void)
{
asm volatile (" \
call terminal_gotox_subprog1; \
call terminal_gotox_subprog2; \
exit; \
" ::: __clobber_all);
}
/*
* Gotox is forbidden when there is no jump table loaded
* which points to the sub-function where the gotox is used

View File

@ -0,0 +1,75 @@
// SPDX-License-Identifier: GPL-2.0
#include <vmlinux.h>
#include <bpf/bpf_helpers.h>
#include "bpf_misc.h"
void *user_ptr;
char dynptr_buf[8];
u64 kaddr;
extern struct kmem_cache *bpf_get_kmem_cache(u64 addr) __ksym;
SEC("socket")
__success
__caps_unpriv(CAP_BPF)
__failure_unpriv
__msg_unpriv("bpf_rdonly_cast is allowed only to CAP_PERFMON and CAP_SYS_ADMIN")
int rdonly_cast_noperfmon(void *ctx)
{
char *p = bpf_rdonly_cast(0, 0);
return p[0x7fff];
}
SEC("socket")
__success
__caps_unpriv(CAP_BPF)
__failure_unpriv
__msg_unpriv("bpf_probe_read_kernel_dynptr is allowed only to CAP_PERFMON and CAP_SYS_ADMIN")
int probe_read_kernel_dynptr_noperfmon(void *ctx)
{
struct bpf_dynptr dptr;
bpf_dynptr_from_mem(dynptr_buf, sizeof(dynptr_buf), 0, &dptr);
bpf_probe_read_kernel_dynptr(&dptr, 0, sizeof(dynptr_buf), user_ptr);
return 0;
}
SEC("socket")
__success
__caps_unpriv(CAP_BPF)
__failure_unpriv
__msg_unpriv("bpf_stream_vprintk is allowed only to CAP_PERFMON and CAP_SYS_ADMIN")
int stream_vprintk_noperfmon(void *ctx)
{
bpf_stream_printk(BPF_STDOUT, "%pB", (void *)kaddr);
return 0;
}
SEC("socket")
__success
__caps_unpriv(CAP_BPF)
__failure_unpriv
__msg_unpriv("bpf_get_kmem_cache is allowed only to CAP_PERFMON and CAP_SYS_ADMIN")
int get_kmem_cache_noperfmon(void *ctx)
{
return !!bpf_get_kmem_cache(kaddr);
}
__weak int subprog_untrusted_read(void *p __arg_untrusted)
{
return *(char *)p;
}
SEC("socket")
__success
__caps_unpriv(CAP_BPF)
__failure_unpriv
__msg_unpriv("rdonly_untrusted_mem access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN")
int arg_untrusted_read_noperfmon(void *ctx)
{
return subprog_untrusted_read(0);
}
char _license[] SEC("license") = "GPL";

View File

@ -303,4 +303,40 @@ __naked void maybe_exit_scc_bug1(void)
::: __clobber_all);
}
/*
* The loop reads zero from the caller's stack on its first iteration and
* one from the callee's stack on its second iteration. At the loop header,
* only the frame number of the pointer in r1 changes.
*/
static __naked __noinline __used
void loop_stack_frames_reg(void)
{
asm volatile (
"*(u64 *)(r10 - 8) = 1;"
"1:"
"r0 = *(u64 *)(r1 + 0);"
"if r0 != 0 goto 2f;"
"r1 = r10;"
"r1 += -8;"
"goto 1b;"
"2:"
"exit;"
::: __clobber_all);
}
SEC("xdp")
__description("bounded loop changing stack frame in a register")
__success __retval(1)
__flag(BPF_F_TEST_STATE_FREQ)
__naked void bounded_loop_stack_frames_reg(void)
{
asm volatile (
"*(u64 *)(r10 - 8) = 0;"
"r1 = r10;"
"r1 += -8;"
"call loop_stack_frames_reg;"
"exit;"
::: __clobber_all);
}
char _license[] SEC("license") = "GPL";

View File

@ -88,6 +88,44 @@ l0_%=: r0 = *(u32*)(r1 + %[bpf_sock_family]); \
: __clobber_all);
}
SEC("socket")
__description("skb->sk: sk->rx_queue_mapping [no sign extension]")
__success __success_unpriv __retval(0)
__naked void sk_rx_queue_mapping_no_sign_ext(void)
{
asm volatile (" \
r1 = *(u64*)(r1 + %[__sk_buff_sk]); \
if r1 != 0 goto l0_%=; \
r0 = 0xdead; \
exit; \
l0_%=: r0 = *(u32*)(r1 + %[bpf_sock_rx_queue_mapping]); \
r0 >>= 32; \
exit; \
" :
: __imm_const(__sk_buff_sk, offsetof(struct __sk_buff, sk)),
__imm_const(bpf_sock_rx_queue_mapping, offsetof(struct bpf_sock, rx_queue_mapping))
: __clobber_all);
}
SEC("socket")
__description("skb->sk: sk->rx_queue_mapping [narrow load mask]")
__success __success_unpriv __retval(0)
__naked void sk_rx_queue_mapping_narrow_load_mask(void)
{
asm volatile (" \
r1 = *(u64*)(r1 + %[__sk_buff_sk]); \
if r1 != 0 goto l0_%=; \
r0 = 0xdead; \
exit; \
l0_%=: r0 = *(u16*)(r1 + %[bpf_sock_rx_queue_mapping]); \
r0 >>= 16; \
exit; \
" :
: __imm_const(__sk_buff_sk, offsetof(struct __sk_buff, sk)),
__imm_const(bpf_sock_rx_queue_mapping, offsetof(struct bpf_sock, rx_queue_mapping))
: __clobber_all);
}
SEC("cgroup/skb")
__description("skb->sk: sk->type [fullsock field]")
__failure __msg("invalid sock_common access")

View File

@ -1719,4 +1719,39 @@ l0_%=: r0 = 0; \
: __clobber_all);
}
SEC("xdp")
__description("XDP pkt regsafe preserves packet pointer class displacement")
__failure __msg("R2 min value is outside of the allowed memory range")
__flag(BPF_F_ANY_ALIGNMENT) __flag(BPF_F_TEST_STATE_FREQ)
__naked void pkt_regsafe_class_displacement(void)
{
asm volatile (" \
r8 = *(u32 *)(r1 + %[xdp_md_data_end]); \
r9 = *(u32 *)(r1 + %[xdp_md_data]); \
r4 = *(u32 *)(r1 + %[xdp_md_rx_queue_index]); \
r4 &= 15; \
r0 = *(u32 *)(r1 + %[xdp_md_ingress_ifindex]); \
if r0 != 0 goto l0_%=; \
r2 = r9; \
r2 += r4; \
r3 = r2; \
r3 += 8; \
goto l1_%=; \
l0_%=: r4 &= 3; \
r4 += 8; \
r2 = r9; \
r2 += r4; \
r3 = r2; \
l1_%=: if r3 > r8 goto l2_%=; \
r0 = *(u64 *)(r2 + 0); \
l2_%=: r0 = 0; \
exit; \
" :
: __imm_const(xdp_md_data, offsetof(struct xdp_md, data)),
__imm_const(xdp_md_data_end, offsetof(struct xdp_md, data_end)),
__imm_const(xdp_md_rx_queue_index, offsetof(struct xdp_md, rx_queue_index)),
__imm_const(xdp_md_ingress_ifindex, offsetof(struct xdp_md, ingress_ifindex))
: __clobber_all);
}
char _license[] SEC("license") = "GPL";