diff --git a/Documentation/bpf/kfuncs.rst b/Documentation/bpf/kfuncs.rst index 85f73e0bbd0f..6691fe8a32c3 100644 --- a/Documentation/bpf/kfuncs.rst +++ b/Documentation/bpf/kfuncs.rst @@ -486,6 +486,16 @@ Example usage in BPF program: /* note that the last argument is omitted */ bpf_task_work_schedule_signal(task, &work->tw, &arrmap, task_work_callback); +2.5.10 KF_PERFMON flag +---------------------- + +The KF_PERFMON flag is used for kfuncs that can expose kernel memory or kernel +addresses to the BPF program, for example by reading through a pointer that the +verifier does not check. Calling such a kfunc requires CAP_PERFMON, or +CAP_SYS_ADMIN, in the same way that the equivalent BPF helpers are gated in +bpf_base_func_proto(). A program loaded with CAP_BPF alone is rejected at load +time. + 2.6 Registering the kfuncs -------------------------- diff --git a/arch/arm64/net/bpf_jit_comp.c b/arch/arm64/net/bpf_jit_comp.c index c18e005a41db..c5f55d6161fe 100644 --- a/arch/arm64/net/bpf_jit_comp.c +++ b/arch/arm64/net/bpf_jit_comp.c @@ -600,6 +600,8 @@ static int build_prologue(struct jit_ctx *ctx, bool ebpf_from_cbpf) * 12 registers are on the stack */ emit(A64_SUB_I(1, A64_SP, A64_FP, 96), ctx); + /* The callback may use its own BPF stack, set up fp for it. */ + ctx->fp_used = true; } /* Stack must be multiples of 16B */ diff --git a/arch/mips/net/bpf_jit_comp32.c b/arch/mips/net/bpf_jit_comp32.c index 40a878b672f5..15a2a153dc87 100644 --- a/arch/mips/net/bpf_jit_comp32.c +++ b/arch/mips/net/bpf_jit_comp32.c @@ -1111,7 +1111,7 @@ static void emit_jmp_i64(struct jit_context *ctx, emit(ctx, xor, tmp, lo(dst), tmp); } if (imm < 0) { /* Compare sign extension */ - emit(ctx, addu, MIPS_R_T9, hi(dst), 1); + emit(ctx, addiu, MIPS_R_T9, hi(dst), 1); emit(ctx, or, tmp, tmp, MIPS_R_T9); } else { /* Compare zero extension */ emit(ctx, or, tmp, tmp, hi(dst)); diff --git a/arch/mips/net/bpf_jit_comp64.c b/arch/mips/net/bpf_jit_comp64.c index fa7e9aa37f49..6681ccac9dd9 100644 --- a/arch/mips/net/bpf_jit_comp64.c +++ b/arch/mips/net/bpf_jit_comp64.c @@ -305,8 +305,7 @@ static void emit_bswap_r64(struct jit_context *ctx, u8 dst, u32 width) case 16: emit_sext(ctx, dst, dst); emit_bswap_r(ctx, dst, width); - if (cpu_has_mips64r2 || cpu_has_mips64r6) - emit_zext(ctx, dst); + emit_zext(ctx, dst); break; } clobber_reg(ctx, dst); diff --git a/include/linux/bpf.h b/include/linux/bpf.h index e57af902560c..1d2676782d70 100644 --- a/include/linux/bpf.h +++ b/include/linux/bpf.h @@ -1651,8 +1651,9 @@ static inline void bpf_trampoline_set_flags(struct bpf_trampoline *tr, u32 flags struct bpf_func_info_aux { u16 linkage; bool unreliable; - bool called : 1; - bool verified : 1; + /* Indexed by in_sleepable. */ + bool called[2]; + bool verified[2]; }; enum bpf_jit_poke_reason { diff --git a/include/linux/bpf_verifier.h b/include/linux/bpf_verifier.h index 36b65797877d..b83ec99f1a13 100644 --- a/include/linux/bpf_verifier.h +++ b/include/linux/bpf_verifier.h @@ -45,18 +45,20 @@ struct bpf_reg_state { union { /* valid when type == PTR_TO_PACKET */ int range; - - /* valid when type == CONST_PTR_TO_MAP | PTR_TO_MAP_VALUE | - * PTR_TO_MAP_VALUE_OR_NULL + /* + * Valid when type == PTR_TO_STACK. Inside the callee two registers + * can be both PTR_TO_STACK like R1=fp-8 and R2=fp-8, but one of them + * points to this function stack while another to the caller's stack. + * To differentiate them 'frameno' is used which is an index in + * bpf_verifier_state->frame[] array pointing to bpf_func_state. */ - struct { - struct bpf_map *map_ptr; - /* To distinguish map lookups from outer map - * the map_uid is non-zero for registers - * pointing to inner maps. - */ - u32 map_uid; - }; + u8 frameno; + + /* + * For CONST_PTR_TO_MAP, PTR_TO_MAP_KEY, PTR_TO_MAP_VALUE and + * PTR_TO_INSN. + */ + struct bpf_map *map_ptr; /* for PTR_TO_BTF_ID */ struct { @@ -155,13 +157,12 @@ struct bpf_reg_state { * gets parent_id set to the dynptr's id. */ u32 parent_id; - /* Inside the callee two registers can be both PTR_TO_STACK like - * R1=fp-8 and R2=fp-8, but one of them points to this function stack - * while another to the caller's stack. To differentiate them 'frameno' - * is used which is an index in bpf_verifier_state->frame[] array - * pointing to bpf_func_state. + /* + * Distinguishes inner-map lookups and their keys and values. Zero for + * other registers. Kept outside the metadata union for ID remapping + * during state comparisons. */ - u32 frameno; + u32 map_uid; /* if (!precise && SCALAR_VALUE) min/max/tnum don't affect safety */ bool precise; }; @@ -679,6 +680,7 @@ struct bpf_insn_aux_data { bool nospec_result; /* result is unsafe under speculation, nospec must follow */ bool zext_dst; /* this insn zero extends dst reg */ bool needs_zext; /* alu op needs to clear upper bits */ + bool prevent_zext; /* alu op cannot be zext (already used with 64-bit scalars) */ bool non_sleepable; /* helper/kfunc may be called from non-sleepable context */ bool is_iter_next; /* bpf_iter__next() kfunc call */ bool call_with_percpu_alloc_ptr; /* {this,per}_cpu_ptr() with prog percpu alloc */ @@ -1197,6 +1199,8 @@ static inline void bpf_trampoline_unpack_key(u64 key, u32 *obj_id, u32 *btf_id) int bpf_prepare_btf_info(struct bpf_verifier_env *env, const union bpf_attr *attr, bpfptr_t uattr); +int bpf_check_core_relo(struct bpf_verifier_env *env, + const union bpf_attr *attr, bpfptr_t uattr); int bpf_check_btf_info(struct bpf_verifier_env *env, const union bpf_attr *attr, bpfptr_t uattr); @@ -1236,11 +1240,18 @@ static inline int bpf_get_spi(s32 off) return (-off - 1) / BPF_REG_SIZE; } +/* + * Return the function state a stack pointer register refers to. frameno + * shares storage with other pointer metadata, so return NULL for any + * other register type instead of indexing frame[] with aliased bytes. + */ static inline struct bpf_func_state *bpf_func(struct bpf_verifier_env *env, const struct bpf_reg_state *reg) { struct bpf_verifier_state *cur = env->cur_state; + if (reg->type != PTR_TO_STACK) + return NULL; return cur->frame[reg->frameno]; } diff --git a/include/linux/btf.h b/include/linux/btf.h index 89d5a5c4f117..7c62ea17b116 100644 --- a/include/linux/btf.h +++ b/include/linux/btf.h @@ -80,6 +80,7 @@ #define KF_ARENA_ARG2 (1 << 15) /* kfunc takes an arena pointer as its second argument */ #define KF_IMPLICIT_ARGS (1 << 16) /* kfunc has implicit arguments supplied by the verifier */ #define KF_SPINLOCK_SAFE (1 << 17) /* kfunc is allowed inside bpf_spin_lock-ed region */ +#define KF_PERFMON (1 << 18) /* kfunc requires CAP_PERFMON */ /* * Tag marking a kernel function as a kfunc. This is meant to minimize the diff --git a/include/linux/skbuff.h b/include/linux/skbuff.h index 421f6fc45451..c8e21903074c 100644 --- a/include/linux/skbuff.h +++ b/include/linux/skbuff.h @@ -4372,7 +4372,10 @@ skb_header_pointer_careful(const struct sk_buff *skb, int offset, static inline void * __must_check skb_pointer_if_linear(const struct sk_buff *skb, int offset, int len) { - if (likely(skb_headlen(skb) - offset >= len)) + unsigned int uoffset = (unsigned int)offset; + + if (likely(uoffset <= skb_headlen(skb) && + (unsigned int)len <= skb_headlen(skb) - uoffset)) return skb->data + offset; return NULL; } diff --git a/kernel/bpf/btf.c b/kernel/bpf/btf.c index 9f33e95d5741..d870bc5e50bc 100644 --- a/kernel/bpf/btf.c +++ b/kernel/bpf/btf.c @@ -4268,13 +4268,10 @@ int btf_check_and_fixup_fields(const struct btf *btf, struct btf_record *rec) { int i; - /* There are three types that signify ownership of some other type: - * kptr_ref, bpf_list_head, bpf_rb_root. - * kptr_ref only supports storing kernel types, which can't store - * references to program allocated local types. - * - * Hence we only need to ensure that bpf_{list_head,rb_root} ownership - * does not form cycles. + /* + * Check fields which require the complete BTF and initialize runtime + * metadata. Ownership relationships are validated after every record has + * been fixed up. */ if (IS_ERR_OR_NULL(rec) || !(rec->field_mask & (BPF_GRAPH_ROOT | BPF_UPTR))) return 0; @@ -4305,53 +4302,90 @@ int btf_check_and_fixup_fields(const struct btf *btf, struct btf_record *rec) if (!meta) return -EFAULT; rec->fields[i].graph_root.value_rec = meta->record; - - /* We need to set value_rec for all root types, but no need - * to check ownership cycle for a type unless it's also a - * node type. - */ - if (!(rec->field_mask & BPF_GRAPH_NODE)) - continue; - - /* We need to ensure ownership acyclicity among all types. The - * proper way to do it would be to topologically sort all BTF - * IDs based on the ownership edges, since there can be multiple - * bpf_{list_head,rb_node} in a type. Instead, we use the - * following resaoning: - * - * - A type can only be owned by another type in user BTF if it - * has a bpf_{list,rb}_node. Let's call these node types. - * - A type can only _own_ another type in user BTF if it has a - * bpf_{list_head,rb_root}. Let's call these root types. - * - * We ensure that if a type is both a root and node, its - * element types cannot be root types. - * - * To ensure acyclicity: - * - * When A is an root type but not a node, its ownership - * chain can be: - * A -> B -> C - * Where: - * - A is an root, e.g. has bpf_rb_root. - * - B is both a root and node, e.g. has bpf_rb_node and - * bpf_list_head. - * - C is only an root, e.g. has bpf_list_node - * - * When A is both a root and node, some other type already - * owns it in the BTF domain, hence it can not own - * another root type through any of the ownership edges. - * A -> B - * Where: - * - A is both an root and node. - * - B is only an node. - */ - if (meta->record->field_mask & BPF_GRAPH_ROOT) - return -ELOOP; } return 0; } +static int btf_owned_type_idx(const struct btf *btf, struct btf_struct_metas *tab, + const struct btf_field *field) +{ + struct btf_struct_meta *meta; + u32 btf_id; + + if (field->type & BPF_GRAPH_ROOT) { + btf_id = field->graph_root.value_btf_id; + } else if (field->type == BPF_KPTR_REF || field->type == BPF_KPTR_PERCPU) { + if (btf_is_kernel(field->kptr.btf)) + return -ENOENT; + btf_id = field->kptr.btf_id; + } else { + return -ENOENT; + } + + meta = btf_find_struct_meta(btf, btf_id); + if (!meta) + return field->type & BPF_GRAPH_ROOT ? -EFAULT : -ENOENT; + return meta - tab->types; +} + +/* + * Each ownership edge adds kernel frames through bpf_obj_free_fields() and + * __bpf_obj_drop_impl(). Keep the bound deliberately small because object + * destruction can itself run below a BPF call chain. A final pointee without + * special fields is not present in the struct metadata table and adds only a + * non-recursing drop. + */ +#define BTF_MAX_OWNERSHIP_DEPTH 8 + +static int btf_ownership_depth(const struct btf *btf, + struct btf_struct_metas *tab, u8 *depth, + int idx, int depth_left) +{ + const struct btf_record *rec = tab->types[idx].record; + int i, ret, max_depth = 0; + + if (!depth_left) + return -ELOOP; + if (depth[idx]) + goto done; + + for (i = 0; i < rec->cnt; i++) { + ret = btf_owned_type_idx(btf, tab, &rec->fields[i]); + if (ret == -ENOENT) + continue; + if (ret < 0) + return ret; + ret = btf_ownership_depth(btf, tab, depth, ret, depth_left - 1); + if (ret < 0) + return ret; + max_depth = max(max_depth, ret); + } + depth[idx] = max_depth + 1; +done: + return depth[idx] > depth_left ? -ELOOP : depth[idx]; +} + +static int btf_check_ownership_depth(const struct btf *btf, + struct btf_struct_metas *tab) +{ + u8 *depth; + int i, ret = 0; + + depth = kvcalloc(tab->cnt, sizeof(*depth), GFP_KERNEL | __GFP_NOWARN); + if (!depth) + return -ENOMEM; + + for (i = 0; i < tab->cnt; i++) { + ret = btf_ownership_depth(btf, tab, depth, i, + BTF_MAX_OWNERSHIP_DEPTH); + if (ret < 0) + break; + ret = 0; + } + kvfree(depth); + return ret; +} + static void __btf_struct_show(const struct btf *btf, const struct btf_type *t, u32 type_id, void *data, u8 bits_offset, struct btf_show *show) @@ -6044,6 +6078,10 @@ static struct btf *btf_parse(const union bpf_attr *attr, bpfptr_t uattr, if (err < 0) goto errout_meta; } + + err = btf_check_ownership_depth(btf, struct_meta_tab); + if (err < 0) + goto errout_meta; } err = bpf_log_attr_finalize(attr_log, &env->log); @@ -7187,7 +7225,7 @@ static int btf_struct_walk(struct bpf_verifier_log *log, const struct btf *btf, if (btf_type_is_int(t)) return WALK_SCALAR; - if (!btf_type_is_struct(t)) + if (!btf_type_is_struct(t) || !t->size) goto error; off = (off - moff) % t->size; diff --git a/kernel/bpf/check_btf.c b/kernel/bpf/check_btf.c index 0e8b3ccc7a5b..4c1ed842f661 100644 --- a/kernel/bpf/check_btf.c +++ b/kernel/bpf/check_btf.c @@ -338,9 +338,9 @@ static int check_btf_line(struct bpf_verifier_env *env, #define MIN_CORE_RELO_SIZE sizeof(struct bpf_core_relo) #define MAX_CORE_RELO_SIZE MAX_FUNCINFO_REC_SIZE -static int check_core_relo(struct bpf_verifier_env *env, - const union bpf_attr *attr, - bpfptr_t uattr) +int bpf_check_core_relo(struct bpf_verifier_env *env, + const union bpf_attr *attr, + bpfptr_t uattr) { u32 i, nr_core_relo, ncopy, expected_size, rec_size; struct bpf_core_relo core_relo = {}; @@ -414,7 +414,7 @@ int bpf_prepare_btf_info(struct bpf_verifier_env *env, struct btf *btf; int err; - if (!attr->func_info_cnt && !attr->line_info_cnt) { + if (!attr->func_info_cnt && !attr->line_info_cnt && !attr->core_relo_cnt) { if (check_abnormal_return(env)) return -EINVAL; return 0; @@ -455,9 +455,5 @@ int bpf_check_btf_info(struct bpf_verifier_env *env, if (err) return err; - err = check_core_relo(env, attr, uattr); - if (err) - return err; - return 0; } diff --git a/kernel/bpf/core.c b/kernel/bpf/core.c index 8b294dfc1ad4..2e3bf8113ae9 100644 --- a/kernel/bpf/core.c +++ b/kernel/bpf/core.c @@ -19,6 +19,7 @@ #include #include +#include #include #include #include @@ -1619,6 +1620,8 @@ struct bpf_prog *bpf_jit_blind_constants(struct bpf_verifier_env *env, struct bp * fix it up here on error. */ bpf_jit_prog_release_other(prog, clone); + if (env && fatal_signal_pending(current)) + return ERR_PTR(-EINTR); return IS_ERR(tmp) ? tmp : ERR_PTR(-ENOMEM); } @@ -2636,11 +2639,14 @@ static struct bpf_prog *bpf_prog_jit_compile(struct bpf_verifier_env *env, struc orig_prog = prog; prog = bpf_jit_blind_constants(env, prog); /* - * If blinding was requested and we failed during blinding, we must fall - * back to the interpreter. + * Fall back to the interpreter after blinding failures, except when + * the loader was killed. */ - if (IS_ERR(prog)) + if (IS_ERR(prog)) { + if (PTR_ERR(prog) == -EINTR) + return prog; goto out_restore; + } prog = bpf_int_jit_compile(env, prog); if (prog->jited) { @@ -2659,6 +2665,8 @@ static struct bpf_prog *bpf_prog_jit_compile(struct bpf_verifier_env *env, struc struct bpf_prog *__bpf_prog_select_runtime(struct bpf_verifier_env *env, struct bpf_prog *fp, int *err) { + struct bpf_prog *jit_prog; + /* In case of BPF to BPF calls, verifier did all the prep * work with regards to JITing, etc. */ @@ -2681,7 +2689,12 @@ struct bpf_prog *__bpf_prog_select_runtime(struct bpf_verifier_env *env, struct if (*err) return fp; - fp = bpf_prog_jit_compile(env, fp); + jit_prog = bpf_prog_jit_compile(env, fp); + if (IS_ERR(jit_prog)) { + *err = PTR_ERR(jit_prog); + return fp; + } + fp = jit_prog; bpf_prog_jit_attempt_done(fp); if (!fp->jited && jit_needed) { *err = -ENOTSUPP; diff --git a/kernel/bpf/crypto.c b/kernel/bpf/crypto.c index 51f89cecefb4..3f3fe2450fc6 100644 --- a/kernel/bpf/crypto.c +++ b/kernel/bpf/crypto.c @@ -149,8 +149,9 @@ bpf_crypto_ctx_create(const struct bpf_crypto_params *params, u32 params__sz, const struct bpf_crypto_type *type; struct bpf_crypto_ctx *ctx; - if (!params || params->reserved[0] || params->reserved[1] || - params__sz != sizeof(struct bpf_crypto_params)) { + if (!params || + params__sz != sizeof(struct bpf_crypto_params) || + params->reserved[0] || params->reserved[1]) { *err = -EINVAL; return NULL; } diff --git a/kernel/bpf/fixups.c b/kernel/bpf/fixups.c index 52d3cec33672..d6f83521fc78 100644 --- a/kernel/bpf/fixups.c +++ b/kernel/bpf/fixups.c @@ -8,6 +8,7 @@ #include #include #include +#include #include #include "disasm.h" @@ -306,12 +307,28 @@ static void adjust_poke_descs(struct bpf_prog *prog, u32 off, u32 len) } } +/* + * Some post-verification instruction rewriting passes require an + * O(prog->len) operation per instruction. Keep their shared primitives + * killable and preemptible. + */ +static bool bpf_rewrite_must_abort(void) +{ + if (fatal_signal_pending(current)) + return true; + cond_resched(); + return false; +} + struct bpf_prog *bpf_patch_insn_data(struct bpf_verifier_env *env, u32 off, const struct bpf_insn *patch, u32 len) { struct bpf_prog *new_prog; struct bpf_insn_aux_data *new_data = NULL; + if (bpf_rewrite_must_abort()) + return NULL; + if (len > 1) { new_data = vrealloc(env->insn_aux_data, array_size(env->prog->len + len - 1, @@ -523,6 +540,9 @@ static int verifier_remove_insns(struct bpf_verifier_env *env, u32 off, u32 cnt) unsigned int orig_prog_len = env->prog->len; int err; + if (bpf_rewrite_must_abort()) + return -EINTR; + if (bpf_prog_is_offloaded(env->prog->aux)) bpf_prog_offload_remove_insns(env, off, cnt); @@ -1356,7 +1376,7 @@ int bpf_jit_subprogs(struct bpf_verifier_env *env) } prog = bpf_jit_blind_constants(env, prog); if (IS_ERR(prog)) { - err = -ENOMEM; + err = PTR_ERR(prog); prog = orig_prog; goto out_restore; } @@ -1433,7 +1453,7 @@ int bpf_fixup_call_args(struct bpf_verifier_env *env) err = bpf_jit_subprogs(env); if (err == 0) return 0; - if (err == -EFAULT) + if (err == -EFAULT || err == -EINTR) return err; } #ifndef CONFIG_BPF_JIT_ALWAYS_ON diff --git a/kernel/bpf/hashtab.c b/kernel/bpf/hashtab.c index 6f331c80130d..f9464e566f10 100644 --- a/kernel/bpf/hashtab.c +++ b/kernel/bpf/hashtab.c @@ -1054,14 +1054,17 @@ static void pcpu_init_value(struct bpf_htab *htab, void __percpu *pptr, /* When not setting the initial value on all cpus, zero-fill element * values for other cpus. Otherwise, bpf program has no way to ensure * known initial values for cpus other than current one - * (onallcpus=false always when coming from bpf prog). + * (onallcpus=false always when coming from bpf prog, + * map_flags & BPF_F_CPU when coming from syscall but setting + * only one cpu). */ - if (!onallcpus) { - int current_cpu = raw_smp_processor_id(); + if (!onallcpus || (map_flags & BPF_F_CPU)) { + int init_cpu = (map_flags & BPF_F_CPU) ? map_flags >> 32 : + raw_smp_processor_id(); int cpu; for_each_possible_cpu(cpu) { - if (cpu == current_cpu) + if (cpu == init_cpu) copy_map_value(&htab->map, per_cpu_ptr(pptr, cpu), value); else /* Since elem is preallocated, we cannot touch special fields */ zero_map_value(&htab->map, per_cpu_ptr(pptr, cpu)); @@ -1772,6 +1775,12 @@ static int htab_lru_percpu_map_lookup_and_delete_elem(struct bpf_map *map, flags); } +/* + * Max consecutive empty buckets to walk in one RCU + + * instrumentation-disabled section before rescheduling. + */ +#define HTAB_BATCH_EMPTY_RESCHED 64 + static int __htab_map_lookup_and_delete_batch(struct bpf_map *map, const union bpf_attr *attr, @@ -1793,6 +1802,7 @@ __htab_map_lookup_and_delete_batch(struct bpf_map *map, unsigned long flags = 0; bool locked = false; struct htab_elem *l; + u32 empty_cnt = 0; struct bucket *b; int ret = 0; @@ -1971,30 +1981,41 @@ __htab_map_lookup_and_delete_batch(struct bpf_map *map, } next_batch: - /* If we are not copying data, we can go to next bucket and avoid - * unlocking the rcu. + /* + * If we are not copying data, we can go to next bucket and avoid + * unlocking the rcu. Bound the walk though: after + * HTAB_BATCH_EMPTY_RESCHED consecutive empty buckets, fully exit + * the critical section (no locks are held here) and reschedule. */ if (!bucket_cnt && (batch + 1 < htab->n_buckets)) { batch++; - goto again_nocopy; + if (++empty_cnt < HTAB_BATCH_EMPTY_RESCHED) + goto again_nocopy; + empty_cnt = 0; + rcu_read_unlock(); + bpf_enable_instrumentation(); + cond_resched_tasks_rcu_qs(); + goto again; } rcu_read_unlock(); bpf_enable_instrumentation(); - if (bucket_cnt && (copy_to_user(ukeys + total * key_size, keys, - key_size * bucket_cnt) || - copy_to_user(uvalues + total * value_size, values, - value_size * bucket_cnt))) { + if (bucket_cnt && (copy_to_user(ukeys + (size_t)total * key_size, keys, + (size_t)key_size * bucket_cnt) || + copy_to_user(uvalues + (size_t)total * value_size, values, + (size_t)value_size * bucket_cnt))) { ret = -EFAULT; goto after_loop; } total += bucket_cnt; + empty_cnt = 0; batch++; if (batch >= htab->n_buckets) { ret = -ENOENT; goto after_loop; } + cond_resched_tasks_rcu_qs(); goto again; after_loop: diff --git a/kernel/bpf/helpers.c b/kernel/bpf/helpers.c index b3cc5c8fc875..712dca5a2c5b 100644 --- a/kernel/bpf/helpers.c +++ b/kernel/bpf/helpers.c @@ -4883,7 +4883,7 @@ BTF_ID(func, bpf_cgroup_release_dtor) BTF_KFUNCS_START(common_btf_ids) BTF_ID_FLAGS(func, bpf_cast_to_kern_ctx, KF_FASTCALL) -BTF_ID_FLAGS(func, bpf_rdonly_cast, KF_FASTCALL) +BTF_ID_FLAGS(func, bpf_rdonly_cast, KF_FASTCALL | KF_PERFMON) BTF_ID_FLAGS(func, bpf_rcu_read_lock) BTF_ID_FLAGS(func, bpf_rcu_read_unlock) BTF_ID_FLAGS(func, bpf_dynptr_slice, KF_RET_NULL) @@ -4920,26 +4920,26 @@ BTF_ID_FLAGS(func, bpf_wq_set_callback, KF_IMPLICIT_ARGS) BTF_ID_FLAGS(func, bpf_wq_start) BTF_ID_FLAGS(func, bpf_preempt_disable) BTF_ID_FLAGS(func, bpf_preempt_enable) -BTF_ID_FLAGS(func, bpf_iter_bits_new, KF_ITER_NEW) +BTF_ID_FLAGS(func, bpf_iter_bits_new, KF_ITER_NEW | KF_PERFMON) BTF_ID_FLAGS(func, bpf_iter_bits_next, KF_ITER_NEXT | KF_RET_NULL) BTF_ID_FLAGS(func, bpf_iter_bits_destroy, KF_ITER_DESTROY) -BTF_ID_FLAGS(func, bpf_copy_from_user_str, KF_SLEEPABLE) -BTF_ID_FLAGS(func, bpf_copy_from_user_task_str, KF_SLEEPABLE) -BTF_ID_FLAGS(func, bpf_get_kmem_cache) +BTF_ID_FLAGS(func, bpf_copy_from_user_str, KF_SLEEPABLE | KF_PERFMON) +BTF_ID_FLAGS(func, bpf_copy_from_user_task_str, KF_SLEEPABLE | KF_PERFMON) +BTF_ID_FLAGS(func, bpf_get_kmem_cache, KF_PERFMON) BTF_ID_FLAGS(func, bpf_iter_kmem_cache_new, KF_ITER_NEW | KF_SLEEPABLE) BTF_ID_FLAGS(func, bpf_iter_kmem_cache_next, KF_ITER_NEXT | KF_RET_NULL | KF_SLEEPABLE) BTF_ID_FLAGS(func, bpf_iter_kmem_cache_destroy, KF_ITER_DESTROY | KF_SLEEPABLE) BTF_ID_FLAGS(func, bpf_local_irq_save) BTF_ID_FLAGS(func, bpf_local_irq_restore) #ifdef CONFIG_BPF_EVENTS -BTF_ID_FLAGS(func, bpf_probe_read_user_dynptr) -BTF_ID_FLAGS(func, bpf_probe_read_kernel_dynptr) -BTF_ID_FLAGS(func, bpf_probe_read_user_str_dynptr) -BTF_ID_FLAGS(func, bpf_probe_read_kernel_str_dynptr) -BTF_ID_FLAGS(func, bpf_copy_from_user_dynptr, KF_SLEEPABLE) -BTF_ID_FLAGS(func, bpf_copy_from_user_str_dynptr, KF_SLEEPABLE) -BTF_ID_FLAGS(func, bpf_copy_from_user_task_dynptr, KF_SLEEPABLE) -BTF_ID_FLAGS(func, bpf_copy_from_user_task_str_dynptr, KF_SLEEPABLE) +BTF_ID_FLAGS(func, bpf_probe_read_user_dynptr, KF_PERFMON) +BTF_ID_FLAGS(func, bpf_probe_read_kernel_dynptr, KF_PERFMON) +BTF_ID_FLAGS(func, bpf_probe_read_user_str_dynptr, KF_PERFMON) +BTF_ID_FLAGS(func, bpf_probe_read_kernel_str_dynptr, KF_PERFMON) +BTF_ID_FLAGS(func, bpf_copy_from_user_dynptr, KF_SLEEPABLE | KF_PERFMON) +BTF_ID_FLAGS(func, bpf_copy_from_user_str_dynptr, KF_SLEEPABLE | KF_PERFMON) +BTF_ID_FLAGS(func, bpf_copy_from_user_task_dynptr, KF_SLEEPABLE | KF_PERFMON) +BTF_ID_FLAGS(func, bpf_copy_from_user_task_str_dynptr, KF_SLEEPABLE | KF_PERFMON) #endif #ifdef CONFIG_DMA_SHARED_BUFFER BTF_ID_FLAGS(func, bpf_iter_dmabuf_new, KF_ITER_NEW | KF_SLEEPABLE) @@ -4947,26 +4947,26 @@ BTF_ID_FLAGS(func, bpf_iter_dmabuf_next, KF_ITER_NEXT | KF_RET_NULL | KF_SLEEPAB BTF_ID_FLAGS(func, bpf_iter_dmabuf_destroy, KF_ITER_DESTROY | KF_SLEEPABLE) #endif BTF_ID_FLAGS(func, __bpf_trap) -BTF_ID_FLAGS(func, bpf_strcmp); -BTF_ID_FLAGS(func, bpf_strcasecmp); -BTF_ID_FLAGS(func, bpf_strncasecmp); -BTF_ID_FLAGS(func, bpf_strchr); -BTF_ID_FLAGS(func, bpf_strchrnul); -BTF_ID_FLAGS(func, bpf_strnchr); -BTF_ID_FLAGS(func, bpf_strrchr); -BTF_ID_FLAGS(func, bpf_strlen); -BTF_ID_FLAGS(func, bpf_strnlen); -BTF_ID_FLAGS(func, bpf_strspn); -BTF_ID_FLAGS(func, bpf_strcspn); -BTF_ID_FLAGS(func, bpf_strstr); -BTF_ID_FLAGS(func, bpf_strcasestr); -BTF_ID_FLAGS(func, bpf_strnstr); -BTF_ID_FLAGS(func, bpf_strncasestr); +BTF_ID_FLAGS(func, bpf_strcmp, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strcasecmp, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strncasecmp, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strchr, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strchrnul, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strnchr, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strrchr, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strlen, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strnlen, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strspn, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strcspn, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strstr, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strcasestr, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strnstr, KF_PERFMON); +BTF_ID_FLAGS(func, bpf_strncasestr, KF_PERFMON); #if defined(CONFIG_BPF_LSM) && defined(CONFIG_CGROUPS) BTF_ID_FLAGS(func, bpf_cgroup_read_xattr, KF_RCU) #endif -BTF_ID_FLAGS(func, bpf_stream_vprintk, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE) -BTF_ID_FLAGS(func, bpf_stream_print_stack, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE) +BTF_ID_FLAGS(func, bpf_stream_vprintk, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE | KF_PERFMON) +BTF_ID_FLAGS(func, bpf_stream_print_stack, KF_IMPLICIT_ARGS | KF_SPINLOCK_SAFE | KF_PERFMON) BTF_ID_FLAGS(func, bpf_task_work_schedule_signal, KF_IMPLICIT_ARGS) BTF_ID_FLAGS(func, bpf_task_work_schedule_resume, KF_IMPLICIT_ARGS) BTF_ID_FLAGS(func, bpf_dynptr_from_file) diff --git a/kernel/bpf/memalloc.c b/kernel/bpf/memalloc.c index e9662db7198f..8a8f088e83e6 100644 --- a/kernel/bpf/memalloc.c +++ b/kernel/bpf/memalloc.c @@ -119,6 +119,7 @@ struct bpf_mem_cache { struct llist_head waiting_for_gp_ttrace; struct rcu_head rcu_ttrace; atomic_t call_rcu_ttrace_in_progress; + raw_spinlock_t lock; }; struct bpf_mem_caches { @@ -214,25 +215,24 @@ static void alloc_bulk(struct bpf_mem_cache *c, int cnt, int node, bool atomic) gfp = __GFP_NOWARN | __GFP_ACCOUNT; gfp |= atomic ? GFP_NOWAIT : GFP_KERNEL; - for (i = 0; i < cnt; i++) { - /* - * For every 'c' llist_del_first(&c->free_by_rcu_ttrace); is - * done only by one CPU == current CPU. Other CPUs might - * llist_add() and llist_del_all() in parallel. - */ - obj = llist_del_first(&c->free_by_rcu_ttrace); - if (!obj) - break; - add_obj_to_free_list(c, obj); - } - if (i >= cnt) - return; + /* + * c->lock serializes concurrent llist_del_first() against + * llist_del_all() in __free_rcu() and do_call_rcu_ttrace(). + */ + scoped_guard(raw_spinlock_irqsave, &c->lock) { + for (i = 0; i < cnt; i++) { + obj = llist_del_first(&c->free_by_rcu_ttrace); + if (!obj) + break; + add_obj_to_free_list(c, obj); + } - for (; i < cnt; i++) { - obj = llist_del_first(&c->waiting_for_gp_ttrace); - if (!obj) - break; - add_obj_to_free_list(c, obj); + for (; i < cnt; i++) { + obj = llist_del_first(&c->waiting_for_gp_ttrace); + if (!obj) + break; + add_obj_to_free_list(c, obj); + } } if (i >= cnt) return; @@ -279,8 +279,12 @@ static int free_all(struct bpf_mem_cache *c, struct llist_node *llnode, bool per static void __free_rcu(struct rcu_head *head) { struct bpf_mem_cache *c = container_of(head, struct bpf_mem_cache, rcu_ttrace); + struct llist_node *llnode; - free_all(c, llist_del_all(&c->waiting_for_gp_ttrace), !!c->percpu_size); + scoped_guard(raw_spinlock_irqsave, &c->lock) + llnode = llist_del_all(&c->waiting_for_gp_ttrace); + + free_all(c, llnode, !!c->percpu_size); atomic_set(&c->call_rcu_ttrace_in_progress, 0); } @@ -300,7 +304,8 @@ static void do_call_rcu_ttrace(struct bpf_mem_cache *c) if (atomic_xchg(&c->call_rcu_ttrace_in_progress, 1)) { if (unlikely(READ_ONCE(c->draining))) { - llnode = llist_del_all(&c->free_by_rcu_ttrace); + scoped_guard(raw_spinlock_irqsave, &c->lock) + llnode = llist_del_all(&c->free_by_rcu_ttrace); free_all(c, llnode, !!c->percpu_size); } return; @@ -535,6 +540,7 @@ int bpf_mem_alloc_init(struct bpf_mem_alloc *ma, int size, bool percpu) c->objcg = objcg; c->percpu_size = percpu_size; c->tgt = c; + raw_spin_lock_init(&c->lock); init_refill_work(c); prefill_mem_cache(c, cpu); } @@ -557,7 +563,7 @@ int bpf_mem_alloc_init(struct bpf_mem_alloc *ma, int size, bool percpu) c->objcg = objcg; c->percpu_size = percpu_size; c->tgt = c; - + raw_spin_lock_init(&c->lock); init_refill_work(c); prefill_mem_cache(c, cpu); } @@ -609,7 +615,7 @@ int bpf_mem_alloc_percpu_unit_init(struct bpf_mem_alloc *ma, int size) c->objcg = objcg; c->percpu_size = percpu_size; c->tgt = c; - + raw_spin_lock_init(&c->lock); init_refill_work(c); prefill_mem_cache(c, cpu); } diff --git a/kernel/bpf/offload.c b/kernel/bpf/offload.c index 0d6f5569588c..d855399812ee 100644 --- a/kernel/bpf/offload.c +++ b/kernel/bpf/offload.c @@ -698,6 +698,8 @@ static bool __bpf_offload_dev_match(struct bpf_prog *prog, return false; if (offload->netdev == netdev) return true; + if (!bpf_prog_is_offloaded(prog->aux)) + return false; ondev1 = bpf_offload_find_netdev(offload->netdev); ondev2 = bpf_offload_find_netdev(netdev); diff --git a/kernel/bpf/states.c b/kernel/bpf/states.c index 66fb11b6c6a7..012b82513a3b 100644 --- a/kernel/bpf/states.c +++ b/kernel/bpf/states.c @@ -491,7 +491,8 @@ static bool regs_exact(const struct bpf_reg_state *rold, { return memcmp(rold, rcur, offsetof(struct bpf_reg_state, id)) == 0 && check_ids(rold->id, rcur->id, idmap) && - check_ids(rold->parent_id, rcur->parent_id, idmap); + check_ids(rold->parent_id, rcur->parent_id, idmap) && + check_ids(rold->map_uid, rcur->map_uid, idmap); } enum exact_level { @@ -616,7 +617,8 @@ static bool regsafe(struct bpf_verifier_env *env, struct bpf_reg_state *rold, range_within(rold, rcur) && tnum_in(rold->var_off, rcur->var_off) && check_ids(rold->id, rcur->id, idmap) && - check_ids(rold->parent_id, rcur->parent_id, idmap); + check_ids(rold->parent_id, rcur->parent_id, idmap) && + check_ids(rold->map_uid, rcur->map_uid, idmap); case PTR_TO_PACKET_META: case PTR_TO_PACKET: /* We must have at least as much range as the old ptr @@ -635,14 +637,14 @@ static bool regsafe(struct bpf_verifier_env *env, struct bpf_reg_state *rold, /* id relations must be preserved */ if (!check_ids(rold->id, rcur->id, idmap)) return false; + /* Preserve displacements between pointers sharing an ID. */ + if (rold->id && rold->r64.base != rcur->r64.base) + return false; /* new val must satisfy old val knowledge */ return range_within(rold, rcur) && tnum_in(rold->var_off, rcur->var_off); case PTR_TO_STACK: - /* two stack pointers are equal only if they're pointing to - * the same stack frame, since fp-8 in foo != fp-8 in bar - */ - return regs_exact(rold, rcur, idmap) && rold->frameno == rcur->frameno; + return regs_exact(rold, rcur, idmap); case PTR_TO_ARENA: return true; case PTR_TO_INSN: @@ -1121,7 +1123,7 @@ static bool states_maybe_looping(struct bpf_verifier_state *old, fcur = cur->frame[fr]; for (i = 0; i < MAX_BPF_REG; i++) if (memcmp(&fold->regs[i], &fcur->regs[i], - offsetof(struct bpf_reg_state, frameno))) + offsetof(struct bpf_reg_state, precise))) return false; return true; } diff --git a/kernel/bpf/syscall.c b/kernel/bpf/syscall.c index c7bc9ba9b331..244a939b9d2d 100644 --- a/kernel/bpf/syscall.c +++ b/kernel/bpf/syscall.c @@ -2036,7 +2036,7 @@ int generic_map_delete_batch(struct bpf_map *map, for (cp = 0; cp < max_count; cp++) { err = -EFAULT; - if (copy_from_user(key, keys + cp * map->key_size, + if (copy_from_user(key, keys + (size_t)cp * map->key_size, map->key_size)) break; @@ -2098,9 +2098,9 @@ int generic_map_update_batch(struct bpf_map *map, struct file *map_file, for (cp = 0; cp < max_count; cp++) { err = -EFAULT; - if (copy_from_user(key, keys + cp * map->key_size, + if (copy_from_user(key, keys + (size_t)cp * map->key_size, map->key_size) || - copy_from_user(value, values + cp * value_size, value_size)) + copy_from_user(value, values + (size_t)cp * value_size, value_size)) break; err = bpf_map_update_value(map, map_file, key, value, @@ -2179,12 +2179,12 @@ int generic_map_lookup_batch(struct bpf_map *map, if (err) goto free_buf; - if (copy_to_user(keys + cp * map->key_size, key, + if (copy_to_user(keys + (size_t)cp * map->key_size, key, map->key_size)) { err = -EFAULT; goto free_buf; } - if (copy_to_user(values + cp * value_size, value, value_size)) { + if (copy_to_user(values + (size_t)cp * value_size, value, value_size)) { err = -EFAULT; goto free_buf; } @@ -6042,7 +6042,10 @@ struct bpf_link *bpf_link_get_curr_or_next(u32 *id) again: link = idr_get_next(&link_idr, id); if (link) { - link = bpf_link_inc_not_zero(link); + if (link->id) + link = bpf_link_inc_not_zero(link); + else + link = ERR_PTR(-EAGAIN); if (IS_ERR(link)) { (*id)++; goto again; diff --git a/kernel/bpf/verifier.c b/kernel/bpf/verifier.c index 72a3f5998dd2..41b49c56e123 100644 --- a/kernel/bpf/verifier.c +++ b/kernel/bpf/verifier.c @@ -567,7 +567,7 @@ static int stack_slot_obj_get_spi(struct bpf_verifier_env *env, struct bpf_reg_s } off = reg->var_off.value; - if (off % BPF_REG_SIZE) { + if (off >= 0 || off % BPF_REG_SIZE) { verbose(env, "cannot pass in %s at an offset=%d\n", obj_kind, off); return -EINVAL; } @@ -1864,6 +1864,7 @@ static void __mark_reg_known(struct bpf_reg_state *reg, u64 imm) offsetof(struct bpf_reg_state, var_off) - sizeof(reg->type)); reg->id = 0; reg->parent_id = 0; + reg->map_uid = 0; ___mark_reg_known(reg, imm); } @@ -1925,17 +1926,18 @@ static void refine_map_lookup_value(struct bpf_reg_state *reg) if (map->inner_map_meta) { reg->type = CONST_PTR_TO_MAP | maybe_null; reg->map_ptr = map->inner_map_meta; - /* transfer reg's id which is unique for every map_lookup_elem + /* + * transfer reg's id which is unique for every map_lookup_elem * as UID of the inner map. */ - if (btf_record_has_field(map->inner_map_meta->record, - BPF_TIMER | BPF_WORKQUEUE | BPF_TASK_WORK)) - reg->map_uid = reg->id; + reg->map_uid = reg->id; } else if (map->map_type == BPF_MAP_TYPE_XSKMAP) { reg->type = PTR_TO_XDP_SOCK | maybe_null; + reg->map_uid = 0; } else if (map->map_type == BPF_MAP_TYPE_SOCKMAP || map->map_type == BPF_MAP_TYPE_SOCKHASH) { reg->type = PTR_TO_SOCKET | maybe_null; + reg->map_uid = 0; } } @@ -3042,6 +3044,8 @@ static int check_subprogs(struct bpf_verifier_env *env) subprog[cur_subprog].exit_idx = i; goto next; } + if (insn_is_gotox(&insn[i])) + goto next; off = i + bpf_jmp_offset(&insn[i]) + 1; if (off < subprog_start || off >= subprog_end) { verbose(env, "jump out of range from insn %d to %d\n", i, off); @@ -3061,7 +3065,8 @@ static int check_subprogs(struct bpf_verifier_env *env) */ if (code != (BPF_JMP | BPF_EXIT) && code != (BPF_JMP32 | BPF_JA) && - code != (BPF_JMP | BPF_JA)) { + code != (BPF_JMP | BPF_JA) && + !insn_is_gotox(&insn[i])) { verbose(env, "last insn is not an exit or jmp\n"); bpf_diag_program_structure( env, i, "subprogram can fall through", @@ -3582,7 +3587,7 @@ static int check_stack_write_fixed_off(struct bpf_verifier_env *env, save_register_state(env, state, spi, reg, size); /* Break the relation on a narrowing spill. */ if (!reg_value_fits) - state->stack[spi].spilled_ptr.id = 0; + clear_scalar_id(&state->stack[spi].spilled_ptr); } else if (!reg && !(off % BPF_REG_SIZE) && is_bpf_st_mem(insn) && env->bpf_capable) { struct bpf_reg_state *tmp_reg = &env->fake_reg[0]; @@ -6453,6 +6458,15 @@ static int check_mem_access(struct bpf_verifier_env *env, int insn_idx, struct b return -EACCES; } + if (rdonly_untrusted && !env->allow_ptr_leaks) { + verbose(env, "%s access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n", + reg_type_str(env, reg->type)); + bpf_diag_policy(env, insn_idx, "read from untrusted read-only memory", + "the access requires CAP_PERFMON", + "Load the program with CAP_PERFMON, or avoid dereferencing untrusted pointers."); + return -EPERM; + } + /* * Accesses to untrusted PTR_TO_MEM are done through probe * instructions, hence no need to check bounds in that case. @@ -9736,6 +9750,16 @@ static int btf_check_func_arg_match(struct bpf_verifier_env *env, int subprog, if (check_mem_reg(env, reg, argno, arg->mem_size, BPF_READ | BPF_WRITE, NULL, NULL)) return -EINVAL; + /* + * PTR_TO_PACKET get passed as PTR_TO_MEM, preventing + * us from adjusting bounds tracking info. + */ + if ((reg_is_pkt_pointer_any(reg) || reg_is_dynptr_slice_pkt(reg)) && + sub->changes_pkt_data) { + bpf_log(log, "%s is a packet pointer, but func#%d may change packet data\n", + reg_arg_name(env, argno), subprog); + return -EINVAL; + } if (!(arg->arg_type & PTR_MAYBE_NULL) && (type_may_be_null(reg->type) || bpf_register_is_null(reg))) { bpf_log(log, "%s is expected to be non-NULL\n", @@ -9919,6 +9943,7 @@ static int check_func_call(struct bpf_verifier_env *env, struct bpf_insn *insn, if (err == -EFAULT) return err; if (bpf_subprog_is_global(env, subprog)) { + struct bpf_func_info_aux *sub_aux = subprog_aux(env, subprog); const char *sub_name = bpf_subprog_name(env, subprog); const char *operation; bool returns_void; @@ -9950,11 +9975,10 @@ static int check_func_call(struct bpf_verifier_env *env, struct bpf_insn *insn, if (env->log.level & BPF_LOG_LEVEL) verbose(env, "Func#%d ('%s') is global and assumed valid.\n", subprog, sub_name); + sub_aux->called[in_sleepable_context(env)] = true; returns_void = subprog_returns_void(env, subprog); if (env->subprog_info[subprog].changes_pkt_data) clear_all_pkt_pointers(env); - /* mark global subprog for verifying after main prog */ - subprog_aux(env, subprog)->called = true; if (returns_void) bpf_diag_record_scrub(env, &caller->regs[BPF_REG_0], BPF_DIAG_MOD_CALLER_SAVED); else @@ -10036,6 +10060,7 @@ int map_set_for_each_callback_args(struct bpf_verifier_env *env, __mark_reg_known_zero(&callee->regs[BPF_REG_3]); callee->regs[BPF_REG_3].map_ptr = caller->regs[BPF_REG_1].map_ptr; callee->regs[BPF_REG_3].map_uid = caller->regs[BPF_REG_1].map_uid; + callee->regs[BPF_REG_3].id = ++env->id_gen; /* pointer to stack or null */ callee->regs[BPF_REG_4] = caller->regs[BPF_REG_3]; @@ -10132,6 +10157,7 @@ static int set_timer_callback_state(struct bpf_verifier_env *env, __mark_reg_known_zero(&callee->regs[BPF_REG_3]); callee->regs[BPF_REG_3].map_ptr = map_ptr; callee->regs[BPF_REG_3].map_uid = map_uid; + callee->regs[BPF_REG_3].id = ++env->id_gen; /* unused */ bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]); @@ -10250,6 +10276,7 @@ static int set_task_work_schedule_callback_state(struct bpf_verifier_env *env, __mark_reg_known_zero(&callee->regs[BPF_REG_3]); callee->regs[BPF_REG_3].map_ptr = map_ptr; callee->regs[BPF_REG_3].map_uid = map_uid; + callee->regs[BPF_REG_3].id = ++env->id_gen; /* unused */ bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]); @@ -10772,11 +10799,7 @@ int bpf_get_helper_proto(struct bpf_verifier_env *env, int func_id, /* Check if we're in a sleepable context. */ static inline bool in_sleepable_context(struct bpf_verifier_env *env) { - return !env->cur_state->active_rcu_locks && - !env->cur_state->active_preempt_locks && - !env->cur_state->active_locks && - !env->cur_state->active_irq_id && - in_sleepable(env); + return !in_rcu_cs(env); } static const char *non_sleepable_context_description(struct bpf_verifier_env *env) @@ -11356,6 +11379,11 @@ static bool is_kfunc_destructive(struct bpf_call_arg_meta *meta) return meta->kfunc_flags & KF_DESTRUCTIVE; } +static bool is_kfunc_perfmon(struct bpf_call_arg_meta *meta) +{ + return meta->kfunc_flags & KF_PERFMON; +} + static bool is_kfunc_rcu(struct bpf_call_arg_meta *meta) { return meta->kfunc_flags & KF_RCU; @@ -13834,6 +13862,15 @@ static int check_kfunc_call(struct bpf_verifier_env *env, struct bpf_insn *insn, return -EACCES; } + if (is_kfunc_perfmon(&meta) && !env->allow_ptr_leaks) { + verbose(env, "%s is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n", + func_name); + operation = bpf_diag_fmt(env, "kfunc %s", func_name); + bpf_diag_policy(env, insn_idx, operation, "the kfunc requires CAP_PERFMON", + "Load the program with CAP_PERFMON, or avoid the kfunc."); + return -EPERM; + } + sleepable = bpf_is_kfunc_sleepable(&meta); if (sleepable && !in_sleepable(env)) { verbose(env, "program must be sleepable to call sleepable kfunc %s\n", func_name); @@ -15704,6 +15741,7 @@ static int adjust_reg_min_max_vals(struct bpf_verifier_env *env, struct bpf_reg_state *regs = state->regs, *dst_reg, *src_reg; struct bpf_reg_state *ptr_reg = NULL, off_reg = {0}; bool alu32 = (BPF_CLASS(insn->code) != BPF_ALU64); + struct bpf_insn_aux_data *aux = cur_aux(env); u8 opcode = BPF_OP(insn->code); int err; @@ -15715,12 +15753,23 @@ static int adjust_reg_min_max_vals(struct bpf_verifier_env *env, /* Case where at least one operand is an arena. */ if (dst_reg->type == PTR_TO_ARENA || (src_reg && src_reg->type == PTR_TO_ARENA)) { - struct bpf_insn_aux_data *aux = cur_aux(env); if (dst_reg->type != PTR_TO_ARENA) *dst_reg = *src_reg; if (BPF_CLASS(insn->code) == BPF_ALU64) { + /* + * Only arena pointers set needs_zext, but doing so + * modifies the instruction at fixup time to an ALU32 + * and makes it unsuitable for 64-bit scalar args. We + * prevent zext from being set if the instruction has + * been previously called with non-arena registers. + */ + if (aux->prevent_zext) { + verbose(env, "same insn cannot be used with and without arena pointer\n"); + return -EINVAL; + } + /* * 32-bit operations zero upper bits automatically. * 64-bit operations need to be converted to 32. @@ -15733,6 +15782,16 @@ static int adjust_reg_min_max_vals(struct bpf_verifier_env *env, return 0; } + /* Prevent the instruction from being used with arena pointers (see above). */ + if (env->prog->aux->arena && BPF_CLASS(insn->code) == BPF_ALU64) { + if (aux->needs_zext) { + verbose(env, "same insn cannot be used with and without arena pointer\n"); + return -EINVAL; + } + + aux->prevent_zext = true; + } + if (dst_reg->type != SCALAR_VALUE) ptr_reg = dst_reg; @@ -19421,13 +19480,14 @@ static void free_states(struct bpf_verifier_env *env) } } -static int do_check_common(struct bpf_verifier_env *env, int subprog) +static int do_check_common(struct bpf_verifier_env *env, int subprog, bool is_sleepable) { bool pop_log = !(env->log.level & BPF_LOG_LEVEL2); struct bpf_subprog_info *sub = subprog_info(env, subprog); struct bpf_prog_aux *aux = env->prog->aux; struct bpf_verifier_state *state; struct bpf_reg_state *regs; + u32 old_insns_total = sub->insns_total; u32 insn_processed = env->insn_processed; int ret, i; @@ -19440,7 +19500,7 @@ static int do_check_common(struct bpf_verifier_env *env, int subprog) state->curframe = 0; state->speculative = false; state->branches = 1; - state->in_sleepable = env->prog->sleepable; + state->in_sleepable = is_sleepable; state->frame[0] = kzalloc_obj(struct bpf_func_state, GFP_KERNEL_ACCOUNT); if (!state->frame[0]) { kfree(state); @@ -19581,8 +19641,10 @@ static int do_check_common(struct bpf_verifier_env *env, int subprog) * not accounted as callees by account_current_path(). * Accumulate their total counts as total counts of the main or * global subprog hosting the async call. + * Start from the saved total of earlier contexts: adding to the current + * total would count this pass's synchronous paths twice. */ - env->subprog_info[subprog].insns_total = env->insn_processed - insn_processed; + sub->insns_total = old_insns_total + (env->insn_processed - insn_processed); return ret; } @@ -19610,14 +19672,19 @@ static int do_check_subprogs(struct bpf_verifier_env *env) { struct bpf_prog_aux *aux = env->prog->aux; struct bpf_func_info_aux *sub_aux; - int i, ret, new_cnt; + int context, i, ret, new_cnt; if (!aux->func_info) return 0; - /* exception callback is presumed to be always called */ - if (env->exception_callback_subprog) - subprog_aux(env, env->exception_callback_subprog)->called = true; + /* + * Callbacks cannot throw, so the exception callback always runs in the + * main program's context. It is presumed to be always called. + */ + if (env->exception_callback_subprog) { + sub_aux = subprog_aux(env, env->exception_callback_subprog); + sub_aux->called[env->prog->sleepable] = true; + } again: new_cnt = 0; @@ -19626,29 +19693,28 @@ static int do_check_subprogs(struct bpf_verifier_env *env) continue; sub_aux = subprog_aux(env, i); - if (!sub_aux->called || sub_aux->verified) - continue; + for (context = 0; context < ARRAY_SIZE(sub_aux->called); context++) { + if (!sub_aux->called[context] || sub_aux->verified[context]) + continue; - env->insn_idx = env->subprog_info[i].start; - WARN_ON_ONCE(env->insn_idx == 0); - ret = do_check_common(env, i); - if (ret) { - return ret; - } else if (env->log.level & BPF_LOG_LEVEL) { - verbose(env, "Func#%d ('%s') is safe for any args that match its prototype\n", - i, bpf_subprog_name(env, i)); + env->insn_idx = env->subprog_info[i].start; + WARN_ON_ONCE(env->insn_idx == 0); + ret = do_check_common(env, i, context); + if (ret) + return ret; + if (env->log.level & BPF_LOG_LEVEL) + verbose(env, "Func#%d ('%s') is safe for any args " + "that match its prototype\n", + i, bpf_subprog_name(env, i)); + + sub_aux->verified[context] = true; + new_cnt++; } - - /* We verified new global subprog, it might have called some - * more global subprogs that we haven't verified yet, so we - * need to do another pass over subprogs to verify those. - */ - sub_aux->verified = true; - new_cnt++; } - /* We can't loop forever as we verify at least one global subprog on - * each pass. + /* + * We can't loop forever as each pass verifies at least one new context, + * and there are only two contexts per global subprog. */ if (new_cnt) goto again; @@ -19661,7 +19727,7 @@ static int do_check_main(struct bpf_verifier_env *env) int ret; env->insn_idx = 0; - ret = do_check_common(env, 0); + ret = do_check_common(env, 0, env->prog->sleepable); if (!ret) env->prog->aux->stack_depth = env->subprog_info[0].stack_depth; return ret; @@ -21170,6 +21236,11 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr, ret = bpf_diag_init(env); if (ret) goto err_prep; + if (env->prog->insnsi[env->prog->len - 1].code == (BPF_LD | BPF_IMM | BPF_DW)) { + verbose(env, "invalid bpf_ld_imm64 insn\n"); + ret = -EINVAL; + goto err_prep; + } if (env->signature) { ret = bpf_prog_calc_tag(env->prog); if (ret < 0) @@ -21245,6 +21316,11 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr, if (ret < 0) goto skip_full_check; + /* Apply CO-RE before validating the program's instruction layout. */ + ret = bpf_check_core_relo(env, attr, uattr); + if (ret < 0) + goto skip_full_check; + /* Discover all subprograms before validating their layout and BTF. */ ret = add_subprogs(env); if (ret < 0) @@ -21254,7 +21330,7 @@ int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr, if (ret < 0) goto skip_full_check; - /* Validate BTF against the complete subprogram layout and apply CO-RE. */ + /* Validate BTF against the complete subprogram layout. */ ret = bpf_check_btf_info(env, attr, uattr); if (ret < 0) goto skip_full_check; diff --git a/net/bpf/test_run.c b/net/bpf/test_run.c index 5d51f6cb7d15..513354e928cb 100644 --- a/net/bpf/test_run.c +++ b/net/bpf/test_run.c @@ -205,8 +205,8 @@ static void xdp_test_run_teardown(struct xdp_test_data *xdp) { xdp_unreg_mem_model(&xdp->mem); page_pool_destroy(xdp->pp); - kfree(xdp->frames); - kfree(xdp->skbs); + kvfree(xdp->frames); + kvfree(xdp->skbs); } static bool frame_was_changed(const struct xdp_page_head *head) diff --git a/net/core/filter.c b/net/core/filter.c index 61940e753552..70dc621672f2 100644 --- a/net/core/filter.c +++ b/net/core/filter.c @@ -3961,12 +3961,6 @@ static u32 __bpf_skb_min_len(const struct sk_buff *skb) if (offset > 0) min_len = offset; } - if (skb->ip_summed == CHECKSUM_PARTIAL) { - offset = skb_checksum_start_offset(skb) + - skb->csum_offset + sizeof(__sum16); - if (offset > 0) - min_len = offset; - } return min_len; } @@ -3983,6 +3977,11 @@ static int bpf_skb_grow_rcsum(struct sk_buff *skb, unsigned int new_len) static int bpf_skb_trim_rcsum(struct sk_buff *skb, unsigned int new_len) { + if (skb->ip_summed == CHECKSUM_PARTIAL && + new_len < skb_checksum_start_offset(skb) + skb->csum_offset + + sizeof(__sum16)) + skb->ip_summed = CHECKSUM_NONE; + return __skb_trim_rcsum(skb, new_len); } @@ -9045,6 +9044,8 @@ static const struct bpf_func_proto * lwt_seg6local_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog) { switch (func_id) { + case BPF_FUNC_skb_pull_data: + return NULL; #if IS_ENABLED(CONFIG_IPV6_SEG6_BPF) case BPF_FUNC_lwt_seg6_store_bytes: return &bpf_lwt_seg6_store_bytes_proto; @@ -10565,11 +10566,12 @@ u32 bpf_sock_convert_ctx_access(enum bpf_access_type type, target_size)); *insn++ = BPF_JMP_IMM(BPF_JNE, si->dst_reg, NO_QUEUE_MAPPING, 1); - *insn++ = BPF_MOV64_IMM(si->dst_reg, -1); + *insn++ = BPF_MOV32_IMM(si->dst_reg, -1); #else - *insn++ = BPF_MOV64_IMM(si->dst_reg, -1); - *target_size = 2; + *insn++ = BPF_MOV32_IMM(si->dst_reg, -1); #endif + *target_size = sizeof_field(struct bpf_sock, rx_queue_mapping); + break; } @@ -11105,18 +11107,7 @@ static u32 sock_ops_convert_ctx_access(enum bpf_access_type type, break; case offsetof(struct bpf_sock_ops, rtt_min): - BUILD_BUG_ON(sizeof_field(struct tcp_sock, rtt_min) != - sizeof(struct minmax)); - BUILD_BUG_ON(sizeof(struct minmax) < - sizeof(struct minmax_sample)); - - *insn++ = BPF_LDX_MEM(BPF_FIELD_SIZEOF( - struct bpf_sock_ops_kern, sk), - si->dst_reg, si->src_reg, - offsetof(struct bpf_sock_ops_kern, sk)); - *insn++ = BPF_LDX_MEM(BPF_W, si->dst_reg, si->dst_reg, - offsetof(struct tcp_sock, rtt_min) + - sizeof_field(struct minmax_sample, t)); + SOCK_OPS_GET_FIELD(rtt_min, rtt_min.s[0].v, struct tcp_sock); break; case offsetof(struct bpf_sock_ops, bpf_sock_ops_cb_flags): @@ -12912,8 +12903,9 @@ __bpf_kfunc_start_defs(); * @sock: Pointer to socket to be destroyed * * Return: - * On error, may return EPROTONOSUPPORT, EINVAL. - * EPROTONOSUPPORT if protocol specific destroy handler is not supported. + * On error, may return EOPNOTSUPP, or whatever the protocol specific + * destroy handler returns. + * EOPNOTSUPP if protocol specific destroy handler is not supported. * 0 otherwise */ __bpf_kfunc int bpf_sock_destroy(struct sock_common *sock) @@ -12925,8 +12917,12 @@ __bpf_kfunc int bpf_sock_destroy(struct sock_common *sock) * Supporting protocols will need to acquire sock lock in the BPF context * prior to invoking this kfunc. */ - if (!sk->sk_prot->diag_destroy || (sk->sk_protocol != IPPROTO_TCP && - sk->sk_protocol != IPPROTO_UDP)) + if (!sk->sk_prot->diag_destroy) + return -EOPNOTSUPP; + + if (sk_fullsock(sk) && + sk->sk_protocol != IPPROTO_TCP && + sk->sk_protocol != IPPROTO_UDP) return -EOPNOTSUPP; return sk->sk_prot->diag_destroy(sk, ECONNABORTED); diff --git a/net/core/skmsg.c b/net/core/skmsg.c index 2521b643fa05..df385a5a961e 100644 --- a/net/core/skmsg.c +++ b/net/core/skmsg.c @@ -1000,6 +1000,10 @@ static int sk_psock_verdict_apply(struct sk_psock *psock, struct sk_buff *skb, int err = 0; u32 len, off; + if (verdict == __SK_REDIRECT && skb_bpf_ingress(skb) && + skb_bpf_redirect_fetch(skb) == psock->sk) + verdict = __SK_PASS; + switch (verdict) { case __SK_PASS: err = -EIO; diff --git a/net/core/sock_map.c b/net/core/sock_map.c index ca49bc7f8687..38df84284328 100644 --- a/net/core/sock_map.c +++ b/net/core/sock_map.c @@ -41,6 +41,7 @@ static struct bpf_map *sock_map_alloc(union bpf_attr *attr) struct bpf_stab *stab; if (attr->max_entries == 0 || + attr->max_entries > INT_MAX || attr->key_size != 4 || (attr->value_size != sizeof(u32) && attr->value_size != sizeof(u64)) || diff --git a/net/ipv4/inet_connection_sock.c b/net/ipv4/inet_connection_sock.c index 6257459bcee2..6a30f1138454 100644 --- a/net/ipv4/inet_connection_sock.c +++ b/net/ipv4/inet_connection_sock.c @@ -1520,7 +1520,8 @@ void inet_csk_listen_stop(struct sock *sk) local_bh_enable(); sock_put(child); - cond_resched(); + if (!has_current_bpf_ctx()) + cond_resched(); } if (queue->fastopenq.rskq_rst_head) { /* Free all the reqs queued in rskq_rst_head. */ diff --git a/net/xdp/xskmap.c b/net/xdp/xskmap.c index 3bff346308d0..bf00d6463c19 100644 --- a/net/xdp/xskmap.c +++ b/net/xdp/xskmap.c @@ -124,7 +124,7 @@ static int xsk_map_gen_lookup(struct bpf_map *map, struct bpf_insn *insn_buf) struct bpf_insn *insn = insn_buf; *insn++ = BPF_LDX_MEM(BPF_W, ret, index, 0); - *insn++ = BPF_JMP_IMM(BPF_JGE, ret, map->max_entries, 5); + *insn++ = BPF_JMP32_IMM(BPF_JGE, ret, map->max_entries, 5); *insn++ = BPF_ALU64_IMM(BPF_LSH, ret, ilog2(sizeof(struct xsk_sock *))); *insn++ = BPF_ALU64_IMM(BPF_ADD, mp, offsetof(struct xsk_map, xsk_map)); *insn++ = BPF_ALU64_REG(BPF_ADD, ret, mp); diff --git a/tools/lib/bpf/libbpf.c b/tools/lib/bpf/libbpf.c index b749c01742ee..bfa64ae6c94d 100644 --- a/tools/lib/bpf/libbpf.c +++ b/tools/lib/bpf/libbpf.c @@ -6206,6 +6206,13 @@ bpf_object__relocate_core(struct bpf_object *obj, const char *targ_btf_path) return -EINVAL; insn = &prog->insns[insn_idx]; + if (is_ldimm64_insn(insn) && (size_t)insn_idx + 1 >= prog->insns_cnt) { + pr_warn("prog '%s': relo #%d: insn #%d (LDIMM64) is truncated\n", + prog->name, i, insn_idx); + err = -EINVAL; + goto out; + } + err = record_relo_core(prog, rec, insn_idx); if (err) { pr_warn("prog '%s': relo #%d: failed to record relocation: %s\n", diff --git a/tools/lib/bpf/relo_core.c b/tools/lib/bpf/relo_core.c index 8ad2715721cf..2672623a4198 100644 --- a/tools/lib/bpf/relo_core.c +++ b/tools/lib/bpf/relo_core.c @@ -980,23 +980,30 @@ static int bpf_core_calc_relo(const char *prog_name, } /* - * Turn instruction for which CO_RE relocation failed into invalid one with + * Turn instruction for which CO-RE relocation failed into invalid one with * distinct signature. */ -static void bpf_core_poison_insn(const char *prog_name, int relo_idx, - int insn_idx, struct bpf_insn *insn) +static int bpf_core_poison_insn(const char *prog_name, int relo_idx, + struct bpf_insn *insn, int insn_idx) { - pr_debug("prog '%s': relo #%d: substituting insn #%d w/ invalid insn\n", - prog_name, relo_idx, insn_idx); - insn->code = BPF_JMP | BPF_CALL; - insn->dst_reg = 0; - insn->src_reg = 0; - insn->off = 0; - /* if this instruction is reachable (not a dead code), - * verifier will complain with the following message: - * invalid func unknown#195896080 - */ - insn->imm = 195896080; /* => 0xbad2310 => "bad relo" */ + int insn_cnt = is_ldimm64_insn(insn) ? 2 : 1; + int i; + + for (i = 0; i < insn_cnt; i++) { + pr_debug("prog '%s': relo #%d: substituting insn #%d w/ invalid insn\n", + prog_name, relo_idx, insn_idx + i); + insn[i].code = BPF_JMP | BPF_CALL; + insn[i].dst_reg = 0; + insn[i].src_reg = 0; + insn[i].off = 0; + /* + * If this instruction is reachable (not dead code), the verifier + * will complain with "invalid func unknown#195896080". + */ + insn[i].imm = 195896080; /* => 0xbad2310 => "bad relo" */ + } + + return 0; } static int insn_bpf_size_to_bytes(struct bpf_insn *insn) @@ -1047,17 +1054,6 @@ int bpf_core_patch_insn(const char *prog_name, struct bpf_insn *insn, class = BPF_CLASS(insn->code); - if (res->poison) { -poison: - /* poison second part of ldimm64 to avoid confusing error from - * verifier about "unknown opcode 00" - */ - if (is_ldimm64_insn(insn)) - bpf_core_poison_insn(prog_name, relo_idx, insn_idx + 1, insn + 1); - bpf_core_poison_insn(prog_name, relo_idx, insn_idx, insn); - return 0; - } - orig_val = res->orig_val; new_val = res->new_val; @@ -1065,7 +1061,9 @@ int bpf_core_patch_insn(const char *prog_name, struct bpf_insn *insn, case BPF_ALU: case BPF_ALU64: if (BPF_SRC(insn->code) != BPF_K) - return -EINVAL; + goto bad_insn; + if (res->poison) + return bpf_core_poison_insn(prog_name, relo_idx, insn, insn_idx); if (res->validate && insn->imm != orig_val) { pr_warn("prog '%s': relo #%d: unexpected insn #%d (ALU/ALU64) value: got %d, exp %llu -> %llu\n", prog_name, relo_idx, @@ -1082,6 +1080,8 @@ int bpf_core_patch_insn(const char *prog_name, struct bpf_insn *insn, case BPF_LDX: case BPF_ST: case BPF_STX: + if (res->poison) + return bpf_core_poison_insn(prog_name, relo_idx, insn, insn_idx); if (res->validate && insn->off != orig_val) { pr_warn("prog '%s': relo #%d: unexpected insn #%d (LDX/ST/STX) value: got %d, exp %llu -> %llu\n", prog_name, relo_idx, insn_idx, insn->off, (unsigned long long)orig_val, @@ -1097,7 +1097,7 @@ int bpf_core_patch_insn(const char *prog_name, struct bpf_insn *insn, pr_warn("prog '%s': relo #%d: insn #%d (LDX/ST/STX) accesses field incorrectly. " "Make sure you are accessing pointers, unsigned integers, or fields of matching type and size.\n", prog_name, relo_idx, insn_idx); - goto poison; + return bpf_core_poison_insn(prog_name, relo_idx, insn, insn_idx); } orig_val = insn->off; @@ -1140,6 +1140,9 @@ int bpf_core_patch_insn(const char *prog_name, struct bpf_insn *insn, return -EINVAL; } + if (res->poison) + return bpf_core_poison_insn(prog_name, relo_idx, insn, insn_idx); + imm = (__u32)insn[0].imm | ((__u64)insn[1].imm << 32); if (res->validate && imm != orig_val) { pr_warn("prog '%s': relo #%d: unexpected insn #%d (LDIMM64) value: got %llu, exp %llu -> %llu\n", @@ -1157,6 +1160,7 @@ int bpf_core_patch_insn(const char *prog_name, struct bpf_insn *insn, break; } default: +bad_insn: pr_warn("prog '%s': relo #%d: trying to relocate unrecognized insn #%d, code:0x%x, src:0x%x, dst:0x%x, off:0x%x, imm:0x%x\n", prog_name, relo_idx, insn_idx, insn->code, (unsigned)insn->src_reg, (unsigned)insn->dst_reg, (unsigned)insn->off, (unsigned)insn->imm); diff --git a/tools/testing/selftests/bpf/prog_tests/cb_refs.c b/tools/testing/selftests/bpf/prog_tests/cb_refs.c index 78566b817fd7..490e15e7126d 100644 --- a/tools/testing/selftests/bpf/prog_tests/cb_refs.c +++ b/tools/testing/selftests/bpf/prog_tests/cb_refs.c @@ -13,7 +13,7 @@ struct { } cb_refs_tests[] = { { "underflow_prog", "release kfunc bpf_kfunc_call_test_release expects referenced PTR_TO_BTF_ID passed to R1" }, { "leak_prog", "Possibly NULL pointer passed to helper R2" }, - { "nested_cb", "Unreleased reference id=4 alloc_insn=2" }, /* alloc_insn=2{4,5} */ + { "nested_cb", "Unreleased reference id=5 alloc_insn=2" }, /* alloc_insn=2{4,5} */ { "non_cb_transfer_ref", "Unreleased reference id=4 alloc_insn=1" }, /* alloc_insn=1{1,2} */ }; diff --git a/tools/testing/selftests/bpf/prog_tests/core_reloc_raw.c b/tools/testing/selftests/bpf/prog_tests/core_reloc_raw.c index a18d3680fb16..51f42b02a267 100644 --- a/tools/testing/selftests/bpf/prog_tests/core_reloc_raw.c +++ b/tools/testing/selftests/bpf/prog_tests/core_reloc_raw.c @@ -14,6 +14,197 @@ static char log[16 * 1024]; +static int load_core_relo_insns(int btf_fd, struct bpf_insn *insns, int insn_cnt, + struct bpf_func_info *funcs, int func_cnt, + int enum_id, int access_str_off, int insn_idx, + bool relocate) +{ + struct bpf_core_relo relo = { + .insn_off = insn_idx * sizeof(struct bpf_insn), + .type_id = enum_id, + .access_str_off = access_str_off, + .kind = BPF_CORE_ENUMVAL_VALUE, + }; + union bpf_attr attr = { + .prog_type = BPF_PROG_TYPE_SOCKET_FILTER, + .insn_cnt = insn_cnt, + .insns = (__u64)insns, + .license = (__u64)"GPL", + .log_buf = (__u64)log, + .log_size = sizeof(log), + .log_level = 2, + .prog_btf_fd = btf_fd, + .func_info_rec_size = sizeof(struct bpf_func_info), + .func_info = (__u64)funcs, + .func_info_cnt = func_cnt, + }; + + if (relocate) { + attr.core_relo_cnt = 1; + attr.core_relos = (__u64)&relo; + attr.core_relo_rec_size = sizeof(relo); + } + memset(log, 0, sizeof(log)); + return sys_bpf_prog_load(&attr, sizeof(attr), 1); +} + +static void test_early_core_relo(void) +{ + static const char unrecognized[] = "trying to relocate unrecognized insn #2"; + static const struct { + const char *name; + struct bpf_insn insns[2]; + const char *err_msg; + } tests[] = { + { "poison_exit", { BPF_EXIT_INSN() }, unrecognized }, + { "poison_ja", { BPF_JMP_A(1) }, unrecognized }, + { "poison_jmp", { BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 1) }, unrecognized }, + { "poison_jmp32", { BPF_JMP32_IMM(BPF_JEQ, BPF_REG_0, 0, 1) }, unrecognized }, + { "poison_call", { BPF_EMIT_CALL(BPF_FUNC_get_prandom_u32) }, unrecognized }, + { "poison_alu_reg", { BPF_MOV32_REG(BPF_REG_0, BPF_REG_1) }, unrecognized }, + { "poison_alu64_reg", { BPF_MOV64_REG(BPF_REG_0, BPF_REG_1) }, unrecognized }, + { "poison_ld_abs", { BPF_LD_ABS(BPF_W, 0) }, + "insn #2 (LDIMM64) has unexpected form" }, + { "poison_alu_imm", { BPF_MOV32_IMM(BPF_REG_0, 0) } }, + { "poison_alu64_imm", { BPF_MOV64_IMM(BPF_REG_0, 0) } }, + { "poison_ldx", { BPF_LDX_MEM(BPF_W, BPF_REG_0, BPF_REG_1, 0) } }, + { "poison_st", { BPF_ST_MEM(BPF_W, BPF_REG_10, -4, 0) } }, + { "poison_stx", { BPF_STX_MEM(BPF_W, BPF_REG_10, BPF_REG_0, -4) } }, + { "poison_ldimm64", { BPF_LD_IMM64(BPF_REG_0, 0) } }, + }; + struct test_btf { + struct btf_header hdr; + __u32 types[18]; + char strings[64]; + } raw_btf = { + .hdr = { + .magic = BTF_MAGIC, + .version = BTF_VERSION, + .hdr_len = sizeof(struct btf_header), + .type_off = 0, + .type_len = sizeof(raw_btf.types), + .str_off = offsetof(struct test_btf, strings) - + offsetof(struct test_btf, types), + .str_len = sizeof(raw_btf.strings), + }, + .types = { + BTF_TYPE_INT_ENC(1, BTF_INT_SIGNED, 0, 32, 4), /* [1] int */ + BTF_FUNC_PROTO_ENC(1, 0), /* [2] int (*)(void) */ + BTF_FUNC_ENC(5, 2), /* [3] main_fn */ + BTF_FUNC_ENC(13, 2), /* [4] sub_fn */ + BTF_TYPE_ENC(20, BTF_INFO_ENC(BTF_KIND_ENUM, 0, 1), 4), /* [5] enum */ + BTF_ENUM_ENC(45, 0), /* value = 0 */ + }, + .strings = "\0int\0main_fn\0sub_fn\0core_relo_poison_missing\0value\0" "0", + }; + struct bpf_func_info funcs[] = { + { .insn_off = 0, .type_id = 3 }, + { .insn_off = 3, .type_id = 4 }, + }; + struct bpf_insn core_only[] = { + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 1), + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_EXIT_INSN(), + }; + struct bpf_insn subprog[] = { + BPF_CALL_REL(2), + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_EXIT_INSN(), + }; + struct bpf_insn truncated_ldimm64[] = { + BPF_RAW_INSN(BPF_LD | BPF_IMM | BPF_DW, 0, 0, 0, 0), + }; + int access_str_off = 51; /* offset of "0" */ + int enum_id = 5; + int btf_fd, prog_fd = -1, i; + + btf_fd = bpf_btf_load(&raw_btf, sizeof(raw_btf), NULL); + if (!ASSERT_GE(btf_fd, 0, "btf_load")) + goto cleanup; + + if (test__start_subtest("without_func_info")) { + prog_fd = load_core_relo_insns(btf_fd, core_only, ARRAY_SIZE(core_only), NULL, 0, + enum_id, access_str_off, 2, false); + if (!ASSERT_GE(prog_fd, 0, "control_load")) + goto cleanup; + close(prog_fd); + prog_fd = load_core_relo_insns(btf_fd, core_only, ARRAY_SIZE(core_only), NULL, 0, + enum_id, access_str_off, 2, true); + if (!ASSERT_GE(prog_fd, 0, "poisoned_load")) + goto cleanup; + ASSERT_HAS_SUBSTR(log, "substituting insn #2", "poison_log"); + close(prog_fd); + prog_fd = -1; + } + + if (test__start_subtest("before_subprog_validation")) { + prog_fd = load_core_relo_insns(btf_fd, subprog, ARRAY_SIZE(subprog), funcs, 2, + enum_id, access_str_off, 2, true); + if (!ASSERT_LT(prog_fd, 0, "poisoned_load")) + goto cleanup; + ASSERT_HAS_SUBSTR(log, "substituting insn #2", "poison_log"); + ASSERT_HAS_SUBSTR(log, "last insn is not an exit or jmp", "poisoned_load_log"); + } + + if (test__start_subtest("truncated_ldimm64")) { + prog_fd = load_core_relo_insns(btf_fd, truncated_ldimm64, + ARRAY_SIZE(truncated_ldimm64), NULL, 0, + enum_id, access_str_off, 0, true); + if (!ASSERT_LT(prog_fd, 0, "truncated_load")) + goto cleanup; + ASSERT_HAS_SUBSTR(log, "invalid bpf_ld_imm64 insn", "truncated_load_log"); + } + + for (i = 0; i < ARRAY_SIZE(tests); i++) { + struct bpf_insn insns[] = { + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_JMP_IMM(BPF_JEQ, BPF_REG_0, 0, 1), + tests[i].insns[0], + BPF_MOV64_IMM(BPF_REG_0, 0), + BPF_EXIT_INSN(), + }; + bool is_ldimm64 = insns[2].code == (BPF_LD | BPF_DW | BPF_IMM); + + if (!test__start_subtest(tests[i].name)) + continue; + if (is_ldimm64) { + insns[1].off = 2; + insns[3] = tests[i].insns[1]; + } + prog_fd = load_core_relo_insns(btf_fd, insns, ARRAY_SIZE(insns), funcs, 1, + enum_id, access_str_off, 2, false); + if (!ASSERT_GE(prog_fd, 0, "control_load")) + goto cleanup; + close(prog_fd); + prog_fd = load_core_relo_insns(btf_fd, insns, ARRAY_SIZE(insns), funcs, 1, + enum_id, access_str_off, 2, true); + if (!tests[i].err_msg) { + ASSERT_GE(prog_fd, 0, "dead_poison_load"); + ASSERT_HAS_SUBSTR(log, "substituting insn #2", "poison_log"); + if (is_ldimm64) + ASSERT_HAS_SUBSTR(log, "substituting insn #3", "poison_ldimm64_log"); + } else { + ASSERT_LT(prog_fd, 0, "invalid_poison_load"); + ASSERT_HAS_SUBSTR(log, tests[i].err_msg, "invalid_poison_log"); + ASSERT_NULL(strstr(log, "substituting insn"), "invalid_poison_substitution"); + } + close(prog_fd); + prog_fd = -1; + } + +cleanup: + if (env.verbosity > VERBOSE_NORMAL && log[0]) { + printf("-------- program load log start --------\n"); + printf("%s", log); + printf("-------- program load log end ----------\n"); + } + close(prog_fd); + close(btf_fd); +} + /* Check that verifier rejects BPF program containing relocation * pointing to non-existent BTF type. */ @@ -120,6 +311,7 @@ static void test_bad_local_id(void) void test_core_reloc_raw(void) { + test_early_core_relo(); if (test__start_subtest("bad_local_id")) test_bad_local_id(); } diff --git a/tools/testing/selftests/bpf/prog_tests/dynptr.c b/tools/testing/selftests/bpf/prog_tests/dynptr.c index 5fda11590708..4396560365e8 100644 --- a/tools/testing/selftests/bpf/prog_tests/dynptr.c +++ b/tools/testing/selftests/bpf/prog_tests/dynptr.c @@ -9,6 +9,7 @@ enum test_setup_type { SETUP_SYSCALL_SLEEP, SETUP_SKB_PROG, + SETUP_SKB_PROG_NONLINEAR, SETUP_SKB_PROG_TP, SETUP_XDP_PROG, }; @@ -32,6 +33,7 @@ static struct { {"test_ringbuf", SETUP_SYSCALL_SLEEP}, {"test_skb_readonly", SETUP_SKB_PROG}, {"test_dynptr_skb_data", SETUP_SKB_PROG}, + {"test_dynptr_skb_slice_non_linear", SETUP_SKB_PROG_NONLINEAR}, {"test_dynptr_skb_meta_data", SETUP_SKB_PROG}, {"test_dynptr_skb_meta_flags", SETUP_SKB_PROG}, {"test_adjust", SETUP_SYSCALL_SLEEP}, @@ -94,7 +96,9 @@ static void verify_success(const char *prog_name, enum test_setup_type setup_typ bpf_link__destroy(link); break; case SETUP_SKB_PROG: + case SETUP_SKB_PROG_NONLINEAR: { + struct __sk_buff ctx = {}; int prog_fd; char buf[64]; @@ -106,6 +110,12 @@ static void verify_success(const char *prog_name, enum test_setup_type setup_typ .repeat = 1, ); + if (setup_type == SETUP_SKB_PROG_NONLINEAR) { + ctx.data_end = ETH_HLEN + sizeof(struct iphdr); + topts.ctx_in = &ctx; + topts.ctx_size_in = sizeof(ctx); + } + prog_fd = bpf_program__fd(prog); if (!ASSERT_GE(prog_fd, 0, "prog_fd")) goto cleanup; diff --git a/tools/testing/selftests/bpf/prog_tests/exceptions.c b/tools/testing/selftests/bpf/prog_tests/exceptions.c index 3588d6f97fd4..639866ce09a9 100644 --- a/tools/testing/selftests/bpf/prog_tests/exceptions.c +++ b/tools/testing/selftests/bpf/prog_tests/exceptions.c @@ -55,6 +55,7 @@ static void test_exceptions_success(void) RUN_SUCCESS(exception_ext, 0); RUN_SUCCESS(exception_ext_mod_cb_runtime, 35); RUN_SUCCESS(exception_throw_subprog, 1); + RUN_SUCCESS(exception_throw_subprog_stack_cb, 0x1234); RUN_SUCCESS(exception_assert_nz_gfunc, 1); RUN_SUCCESS(exception_assert_zero_gfunc, 1); RUN_SUCCESS(exception_assert_neg_gfunc, 1); diff --git a/tools/testing/selftests/bpf/prog_tests/linked_list.c b/tools/testing/selftests/bpf/prog_tests/linked_list.c index c3d133c6a00d..52fabbee3dd5 100644 --- a/tools/testing/selftests/bpf/prog_tests/linked_list.c +++ b/tools/testing/selftests/bpf/prog_tests/linked_list.c @@ -714,7 +714,7 @@ static void test_btf(void) break; err = btf__load_into_kernel(btf); - ASSERT_EQ(err, -ELOOP, "check btf"); + ASSERT_EQ(err, 0, "check btf"); btf__free(btf); break; } @@ -773,7 +773,7 @@ static void test_btf(void) break; err = btf__load_into_kernel(btf); - ASSERT_EQ(err, -ELOOP, "check btf"); + ASSERT_EQ(err, 0, "check btf"); btf__free(btf); break; } diff --git a/tools/testing/selftests/bpf/prog_tests/local_kptr_ownership.c b/tools/testing/selftests/bpf/prog_tests/local_kptr_ownership.c new file mode 100644 index 000000000000..a487aa68f2ee --- /dev/null +++ b/tools/testing/selftests/bpf/prog_tests/local_kptr_ownership.c @@ -0,0 +1,296 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Copyright (c) 2026 Meta Platforms, Inc. and affiliates. */ + +#include +#include +#include + +#define SPIN_LOCK 2 +#define LIST_HEAD 3 +#define LIST_NODE 4 +/* Keep in sync with BTF_MAX_OWNERSHIP_DEPTH. */ +#define MAX_OWNERSHIP_DEPTH 8 + +static struct btf *init_btf(void) +{ + struct btf *btf; + int id; + + btf = btf__new_empty(); + if (!ASSERT_OK_PTR(btf, "btf__new_empty")) + return NULL; + id = btf__add_int(btf, "int", 4, BTF_INT_SIGNED); + if (!ASSERT_EQ(id, 1, "btf__add_int")) + goto err_out; + id = btf__add_struct(btf, "bpf_spin_lock", 4); + if (!ASSERT_EQ(id, SPIN_LOCK, "btf__add_struct bpf_spin_lock")) + goto err_out; + id = btf__add_struct(btf, "bpf_list_head", 16); + if (!ASSERT_EQ(id, LIST_HEAD, "btf__add_struct bpf_list_head")) + goto err_out; + id = btf__add_struct(btf, "bpf_list_node", 24); + if (!ASSERT_EQ(id, LIST_NODE, "btf__add_struct bpf_list_node")) + goto err_out; + return btf; + +err_out: + btf__free(btf); + return NULL; +} + +static int add_local_kptr(struct btf *btf, int pointee_id, const char *tag) +{ + int id; + + id = btf__add_type_tag(btf, tag, pointee_id); + if (!ASSERT_GT(id, 0, "btf__add_type_tag")) + return id; + id = btf__add_ptr(btf, id); + ASSERT_GT(id, 0, "btf__add_ptr"); + return id; +} + +static void test_self_cycle(const char *tag, int expected_err) +{ + struct btf *btf; + int id, err; + + btf = init_btf(); + if (!ASSERT_OK_PTR(btf, "init_btf")) + return; + id = add_local_kptr(btf, 7, tag); + if (id <= 0) + goto out; + id = btf__add_struct(btf, "self_cycle", 8); + if (!ASSERT_EQ(id, 7, "btf__add_struct self_cycle")) + goto out; + err = btf__add_field(btf, "next", 6, 0, 0); + if (!ASSERT_OK(err, "btf__add_field self_cycle::next")) + goto out; + + err = btf__load_into_kernel(btf); + ASSERT_EQ(err, expected_err, "check btf"); +out: + btf__free(btf); +} + +static void test_aba_cycle(void) +{ + struct btf *btf; + int id, err; + + btf = init_btf(); + if (!ASSERT_OK_PTR(btf, "init_btf")) + return; + id = add_local_kptr(btf, 10, "kptr"); + if (id <= 0) + goto out; + id = add_local_kptr(btf, 9, "kptr"); + if (id <= 0) + goto out; + id = btf__add_struct(btf, "cycle_a", 8); + if (!ASSERT_EQ(id, 9, "btf__add_struct cycle_a")) + goto out; + err = btf__add_field(btf, "b", 6, 0, 0); + if (!ASSERT_OK(err, "btf__add_field cycle_a::b")) + goto out; + id = btf__add_struct(btf, "cycle_b", 8); + if (!ASSERT_EQ(id, 10, "btf__add_struct cycle_b")) + goto out; + err = btf__add_field(btf, "a", 8, 0, 0); + if (!ASSERT_OK(err, "btf__add_field cycle_b::a")) + goto out; + + err = btf__load_into_kernel(btf); + ASSERT_EQ(err, -ELOOP, "check btf"); +out: + btf__free(btf); +} + +static void test_mixed_cycle(void) +{ + struct btf *btf; + int id, err; + + btf = init_btf(); + if (!ASSERT_OK_PTR(btf, "init_btf")) + return; + id = add_local_kptr(btf, 7, "kptr"); + if (id <= 0) + goto out; + id = btf__add_struct(btf, "mixed_owner", 20); + if (!ASSERT_EQ(id, 7, "btf__add_struct mixed_owner")) + goto out; + err = btf__add_field(btf, "root", LIST_HEAD, 0, 0); + if (!ASSERT_OK(err, "btf__add_field mixed_owner::root")) + goto out; + err = btf__add_field(btf, "lock", SPIN_LOCK, 128, 0); + if (!ASSERT_OK(err, "btf__add_field mixed_owner::lock")) + goto out; + id = btf__add_decl_tag(btf, "contains:mixed_node:node", 7, 0); + if (!ASSERT_EQ(id, 8, "btf__add_decl_tag mixed_owner")) + goto out; + id = btf__add_struct(btf, "mixed_node", 32); + if (!ASSERT_EQ(id, 9, "btf__add_struct mixed_node")) + goto out; + err = btf__add_field(btf, "node", LIST_NODE, 0, 0); + if (!ASSERT_OK(err, "btf__add_field mixed_node::node")) + goto out; + err = btf__add_field(btf, "owner", 6, 192, 0); + if (!ASSERT_OK(err, "btf__add_field mixed_node::owner")) + goto out; + + err = btf__load_into_kernel(btf); + ASSERT_EQ(err, -ELOOP, "check btf"); +out: + btf__free(btf); +} + +static void test_acyclic_depth(int depth, bool child_first, bool shared_suffix, int expected_err) +{ + int ptr_id[MAX_OWNERSHIP_DEPTH + 1]; + int first_struct_id; + struct btf *btf; + int id, err, i, n, pointee_id; + + btf = init_btf(); + if (!ASSERT_OK_PTR(btf, "init_btf")) + return; + first_struct_id = 5 + 2 * depth; + for (i = 0; i < depth; i++) { + if (i == depth - 1) + pointee_id = first_struct_id + depth; + else + pointee_id = first_struct_id + (child_first ? depth - 2 - i : i + 1); + ptr_id[i] = add_local_kptr(btf, pointee_id, "kptr"); + if (ptr_id[i] <= 0) + goto out; + } + for (n = 0; n < depth; n++) { + char name[32]; + int offset = 0; + + i = child_first ? depth - 1 - n : n; + snprintf(name, sizeof(name), "owner_%d", i); + id = btf__add_struct(btf, name, shared_suffix && !i ? 16 : 8); + if (!ASSERT_EQ(id, first_struct_id + n, "btf__add_struct owner")) + goto out; + if (shared_suffix && !i) { + /* + * Visit the shared suffix through the shorter path before + * reaching it again with less remaining depth. + */ + err = btf__add_field(btf, "suffix", ptr_id[1], 0, 0); + if (!ASSERT_OK(err, "btf__add_field owner::suffix")) + goto out; + offset = 64; + } + err = btf__add_field(btf, "next", ptr_id[i], offset, 0); + if (!ASSERT_OK(err, "btf__add_field owner::next")) + goto out; + } + id = btf__add_struct(btf, "plain_leaf", 4); + if (!ASSERT_EQ(id, first_struct_id + depth, "btf__add_struct plain_leaf")) + goto out; + + err = btf__load_into_kernel(btf); + ASSERT_EQ(err, expected_err, "check btf"); +out: + btf__free(btf); +} + +static void test_graph_depth(bool rbtree, int depth, int expected_err) +{ + int root_type = LIST_HEAD, node_type = LIST_NODE, node_size = 24; + int id, err, i, lock_off, root_off, size; + struct btf *btf; + + btf = init_btf(); + if (!ASSERT_OK_PTR(btf, "init_btf")) + return; + if (rbtree) { + root_type = btf__add_struct(btf, "bpf_rb_root", 16); + if (!ASSERT_GT(root_type, 0, "btf__add_struct bpf_rb_root")) + goto out; + node_type = btf__add_struct(btf, "bpf_rb_node", 32); + if (!ASSERT_GT(node_type, 0, "btf__add_struct bpf_rb_node")) + goto out; + node_size = 32; + } + + for (i = 0; i < depth; i++) { + char name[32], tag[64]; + + lock_off = i ? node_size : 0; + root_off = lock_off + 8; + size = i == depth - 1 ? node_size : root_off + 16; + snprintf(name, sizeof(name), "graph_owner_%d", i); + id = btf__add_struct(btf, name, size); + if (!ASSERT_GT(id, 0, "btf__add_struct graph_owner")) + goto out; + if (i) { + err = btf__add_field(btf, "node", node_type, 0, 0); + if (!ASSERT_OK(err, "btf__add_field graph_owner::node")) + goto out; + } + if (i == depth - 1) + continue; + err = btf__add_field(btf, "lock", SPIN_LOCK, lock_off * 8, 0); + if (!ASSERT_OK(err, "btf__add_field graph_owner::lock")) + goto out; + err = btf__add_field(btf, "root", root_type, root_off * 8, 0); + if (!ASSERT_OK(err, "btf__add_field graph_owner::root")) + goto out; + snprintf(tag, sizeof(tag), "contains:graph_owner_%d:node", i + 1); + err = btf__add_decl_tag(btf, tag, id, i ? 2 : 1); + if (!ASSERT_GT(err, 0, "btf__add_decl_tag graph_owner")) + goto out; + } + + err = btf__load_into_kernel(btf); + ASSERT_EQ(err, expected_err, "check btf"); +out: + btf__free(btf); +} + +void test_local_kptr_ownership(void) +{ + if (test__start_subtest("self_cycle")) + test_self_cycle("kptr", -ELOOP); + if (test__start_subtest("untrusted_self_cycle")) + test_self_cycle("kptr_untrusted", 0); + if (test__start_subtest("percpu_self_cycle")) + test_self_cycle("percpu_kptr", -ELOOP); + if (test__start_subtest("ABA_cycle")) + test_aba_cycle(); + if (test__start_subtest("mixed_graph_root_cycle")) + test_mixed_cycle(); + if (test__start_subtest("max_acyclic")) + test_acyclic_depth(MAX_OWNERSHIP_DEPTH, false, false, 0); + if (test__start_subtest("too_deep_acyclic")) + test_acyclic_depth(MAX_OWNERSHIP_DEPTH + 1, false, false, -ELOOP); + if (test__start_subtest("max_acyclic_child_first")) + test_acyclic_depth(MAX_OWNERSHIP_DEPTH, true, false, 0); + if (test__start_subtest("too_deep_acyclic_child_first")) + test_acyclic_depth(MAX_OWNERSHIP_DEPTH + 1, true, false, -ELOOP); + if (test__start_subtest("max_acyclic_shared_suffix")) + test_acyclic_depth(MAX_OWNERSHIP_DEPTH, false, true, 0); + if (test__start_subtest("too_deep_acyclic_shared_suffix")) + test_acyclic_depth(MAX_OWNERSHIP_DEPTH + 1, false, true, -ELOOP); + if (test__start_subtest("list_three_types")) + test_graph_depth(false, 3, 0); + if (test__start_subtest("list_four_types")) + test_graph_depth(false, 4, 0); + if (test__start_subtest("list_max_depth")) + test_graph_depth(false, MAX_OWNERSHIP_DEPTH, 0); + if (test__start_subtest("list_too_deep")) + test_graph_depth(false, MAX_OWNERSHIP_DEPTH + 1, -ELOOP); + if (test__start_subtest("rbtree_three_types")) + test_graph_depth(true, 3, 0); + if (test__start_subtest("rbtree_four_types")) + test_graph_depth(true, 4, 0); + if (test__start_subtest("rbtree_max_depth")) + test_graph_depth(true, MAX_OWNERSHIP_DEPTH, 0); + if (test__start_subtest("rbtree_too_deep")) + test_graph_depth(true, MAX_OWNERSHIP_DEPTH + 1, -ELOOP); +} diff --git a/tools/testing/selftests/bpf/prog_tests/percpu_alloc.c b/tools/testing/selftests/bpf/prog_tests/percpu_alloc.c index a72ae0b29f6e..7b4a1e24363b 100644 --- a/tools/testing/selftests/bpf/prog_tests/percpu_alloc.c +++ b/tools/testing/selftests/bpf/prog_tests/percpu_alloc.c @@ -1,4 +1,6 @@ // SPDX-License-Identifier: GPL-2.0 +#define _GNU_SOURCE +#include #include #include "cgroup_helpers.h" #include "percpu_alloc_array.skel.h" @@ -350,6 +352,103 @@ static void test_lru_percpu_hash_cpu_flag(void) test_percpu_map_cpu_flag(BPF_MAP_TYPE_LRU_PERCPU_HASH); } +/* + * A BPF_F_CPU update that creates an element must zero the value on the other + * cpus, rather than leave them holding whatever the recycled element last + * contained. max_entries is 1 so the second key can only reuse the element + * the first one released. + */ +static void test_percpu_map_cpu_flag_create(enum bpf_map_type map_type, __u32 map_flags) +{ + LIBBPF_OPTS(bpf_map_create_opts, opts, .map_flags = map_flags); + const u32 stale = 0xDEADC0DE, fresh = 0xC0FFEE; + int nr_cpus, cpu, map_fd, err, key; + int pinned_cpu, value_cpu; + cpu_set_t old_mask, new_mask; + bool restore_mask = false; + u32 value; + u64 flags; + + nr_cpus = libbpf_num_possible_cpus(); + if (!ASSERT_GT(nr_cpus, 0, "libbpf_num_possible_cpus")) + return; + + if (nr_cpus < 2) { + test__skip(); + return; + } + + map_fd = bpf_map_create(map_type, "cpu_flag_create", sizeof(key), sizeof(value), 1, &opts); + if (!ASSERT_GE(map_fd, 0, "bpf_map_create")) + return; + + /* NO_PREALLOC recycles per cpu, so keep the delete and the create on one cpu. */ + err = sched_getaffinity(0, sizeof(old_mask), &old_mask); + if (!ASSERT_OK(err, "sched_getaffinity")) + goto out; + + pinned_cpu = sched_getcpu(); + if (!ASSERT_GE(pinned_cpu, 0, "sched_getcpu")) + goto out; + + CPU_ZERO(&new_mask); + CPU_SET(pinned_cpu, &new_mask); + err = sched_setaffinity(0, sizeof(new_mask), &new_mask); + if (!ASSERT_OK(err, "sched_setaffinity")) + goto out; + restore_mask = true; + + value_cpu = pinned_cpu ? 0 : 1; + + key = 1; + value = stale; + err = bpf_map_update_elem(map_fd, &key, &value, BPF_F_ALL_CPUS); + if (!ASSERT_OK(err, "bpf_map_update_elem all_cpus")) + goto out; + + err = bpf_map_delete_elem(map_fd, &key); + if (!ASSERT_OK(err, "bpf_map_delete_elem")) + goto out; + + key = 2; + value = fresh; + flags = (u64)value_cpu << 32 | BPF_F_CPU; + err = bpf_map_update_elem(map_fd, &key, &value, flags); + if (!ASSERT_OK(err, "bpf_map_update_elem specified cpu")) + goto out; + + for (cpu = 0; cpu < nr_cpus; cpu++) { + value = 0; + flags = (u64)cpu << 32 | BPF_F_CPU; + err = bpf_map_lookup_elem_flags(map_fd, &key, &value, flags); + if (!ASSERT_OK(err, "bpf_map_lookup_elem_flags specified cpu")) + goto out; + if (!ASSERT_EQ(value, cpu == value_cpu ? fresh : 0, "value on specified cpu")) + goto out; + } + +out: + if (restore_mask) + sched_setaffinity(0, sizeof(old_mask), &old_mask); + close(map_fd); +} + +static void test_percpu_hash_cpu_flag_create(void) +{ + test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_PERCPU_HASH, 0); +} + +static void test_percpu_hash_cpu_flag_create_malloc(void) +{ + test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_PERCPU_HASH, BPF_F_NO_PREALLOC); +} + +static void test_lru_percpu_hash_cpu_flag_create(void) +{ + /* lru without prealloc is -ENOTSUPP, so there is no malloc variant */ + test_percpu_map_cpu_flag_create(BPF_MAP_TYPE_LRU_PERCPU_HASH, 0); +} + static void test_percpu_cgroup_storage_cpu_flag(void) { struct percpu_alloc_array *skel = NULL; @@ -454,6 +553,12 @@ void test_percpu_alloc(void) test_percpu_hash_cpu_flag(); if (test__start_subtest("cpu_flag_lru_percpu_hash")) test_lru_percpu_hash_cpu_flag(); + if (test__start_subtest("cpu_flag_create_percpu_hash")) + test_percpu_hash_cpu_flag_create(); + if (test__start_subtest("cpu_flag_create_percpu_hash_malloc")) + test_percpu_hash_cpu_flag_create_malloc(); + if (test__start_subtest("cpu_flag_create_lru_percpu_hash")) + test_lru_percpu_hash_cpu_flag_create(); if (test__start_subtest("cpu_flag_percpu_cgroup_storage")) test_percpu_cgroup_storage_cpu_flag(); if (test__start_subtest("cpu_flag_array")) diff --git a/tools/testing/selftests/bpf/prog_tests/sock_destroy.c b/tools/testing/selftests/bpf/prog_tests/sock_destroy.c index 9c11938fe597..78d642a02bdb 100644 --- a/tools/testing/selftests/bpf/prog_tests/sock_destroy.c +++ b/tools/testing/selftests/bpf/prog_tests/sock_destroy.c @@ -1,4 +1,5 @@ // SPDX-License-Identifier: GPL-2.0 +#include #include #include @@ -110,6 +111,122 @@ static void test_tcp_server(struct sock_destroy_prog *skel) close(serv); } +static void test_tcp_listen_pending(struct sock_destroy_prog *skel) +{ + int serv = -1, clien = -1, accept_serv = -1, n, serv_port; + struct pollfd pfd = { .events = POLLIN }; + char buf[1]; + + serv = start_server(AF_INET6, SOCK_STREAM, NULL, 0, 0); + if (!ASSERT_GE(serv, 0, "start_server")) + goto cleanup; + serv_port = get_socket_local_port(serv); + if (!ASSERT_GE(serv_port, 0, "get_sock_local_port")) + goto cleanup; + skel->bss->serv_port = (__be16)serv_port; + + /* + * Connect but never accept, so the child sits in the accept queue + * of the listener. Wait until it's actually there. + */ + clien = connect_to_fd(serv, 0); + if (!ASSERT_GE(clien, 0, "connect_to_fd")) + goto cleanup; + pfd.fd = serv; + if (!ASSERT_EQ(poll(&pfd, 1, -1), 1, "poll listener")) + goto cleanup; + + /* Run iterator program that destroys server sockets. */ + start_iter_sockets(skel->progs.iter_tcp6_server); + + accept_serv = accept(serv, NULL, NULL); + if (!ASSERT_LT(accept_serv, 0, "accept on destroyed listener")) + goto cleanup; + ASSERT_EQ(errno, EINVAL, "error code on destroyed listener"); + + /* The unaccepted child was reset along with the listener. */ + n = recv(clien, buf, sizeof(buf), 0); + if (!ASSERT_LT(n, 0, "client recv on reset child")) + goto cleanup; + ASSERT_EQ(errno, ECONNRESET, "error code on reset child"); + +cleanup: + if (clien != -1) + close(clien); + if (accept_serv != -1) + close(accept_serv); + if (serv != -1) + close(serv); +} + +static void test_tcp_timewait(struct sock_destroy_prog *skel) +{ + int serv = -1, clien = -1, accept_serv = -1, n; + struct timeval tv = {}; + char buf[1]; + + serv = start_server(AF_INET6, SOCK_STREAM, NULL, 0, 0); + if (!ASSERT_GE(serv, 0, "start_server")) + goto cleanup; + + clien = connect_to_fd(serv, 0); + if (!ASSERT_GE(clien, 0, "connect_to_fd")) + goto cleanup; + + accept_serv = accept(serv, NULL, NULL); + if (!ASSERT_GE(accept_serv, 0, "serv accept")) + goto cleanup; + + /* + * Active close from the client, then close the server side. Once + * recv() sees EOF the server FIN has been processed and the client + * sock is in TIME_WAIT. Block without timeout so a loaded CI box + * can't race us. + */ + if (!ASSERT_OK(setsockopt(clien, SOL_SOCKET, SO_RCVTIMEO, &tv, + sizeof(tv)), "clear rcvtimeo")) + goto cleanup; + if (!ASSERT_OK(shutdown(clien, SHUT_WR), "client shutdown")) + goto cleanup; + + /* + * Make sure the server has seen the client FIN before it closes, + * so the two FINs never cross. + */ + n = recv(accept_serv, buf, sizeof(buf), 0); + if (!ASSERT_EQ(n, 0, "server recv EOF")) + goto cleanup; + + close(accept_serv); + accept_serv = -1; + + /* block until return EOF */ + n = recv(clien, buf, sizeof(buf), 0); + if (!ASSERT_EQ(n, 0, "client recv EOF")) + goto cleanup; + + /* Run iterator program that destroys the timewait client sock. */ + skel->bss->tw_found = 0; + start_iter_sockets(skel->progs.iter_tcp6_timewait); + if (!ASSERT_EQ(skel->bss->tw_found, 1, "timewait sock found")) + goto cleanup; + + ASSERT_OK(skel->bss->tw_destroy_err, "destroy timewait sock"); + + /* The destroyed timewait sock must be gone. */ + skel->bss->tw_found = 0; + start_iter_sockets(skel->progs.iter_tcp6_timewait); + ASSERT_EQ(skel->bss->tw_found, 0, "timewait sock destroyed"); + +cleanup: + if (clien != -1) + close(clien); + if (accept_serv != -1) + close(accept_serv); + if (serv != -1) + close(serv); +} + static void test_udp_client(struct sock_destroy_prog *skel) { int serv = -1, clien = -1, n = 0; @@ -204,6 +321,10 @@ void test_sock_destroy(void) test_tcp_client(skel); if (test__start_subtest("tcp_server")) test_tcp_server(skel); + if (test__start_subtest("tcp_listen_pending")) + test_tcp_listen_pending(skel); + if (test__start_subtest("tcp_timewait")) + test_tcp_timewait(skel); if (test__start_subtest("udp_client")) test_udp_client(skel); if (test__start_subtest("udp_server")) diff --git a/tools/testing/selftests/bpf/prog_tests/spin_lock.c b/tools/testing/selftests/bpf/prog_tests/spin_lock.c index 5c3579438427..e368370262c8 100644 --- a/tools/testing/selftests/bpf/prog_tests/spin_lock.c +++ b/tools/testing/selftests/bpf/prog_tests/spin_lock.c @@ -54,6 +54,8 @@ static struct { { "lock_global_sleepable_helper_subprog", "global function calls are not allowed while holding a lock" }, { "lock_global_sleepable_kfunc_subprog", "global function calls are not allowed while holding a lock" }, { "lock_global_sleepable_subprog_indirect", "global function calls are not allowed while holding a lock" }, + { "callback_value_lock_identity", "bpf_spin_unlock of different lock" }, + { "callback_inner_map_value_lock_identity", "bpf_spin_unlock of different lock" }, }; static int match_regex(const char *pattern, const char *string) diff --git a/tools/testing/selftests/bpf/prog_tests/tc_change_tail_pmtu.c b/tools/testing/selftests/bpf/prog_tests/tc_change_tail_pmtu.c new file mode 100644 index 000000000000..7acdbd5757a9 --- /dev/null +++ b/tools/testing/selftests/bpf/prog_tests/tc_change_tail_pmtu.c @@ -0,0 +1,125 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include + +#include "test_progs.h" +#include "network_helpers.h" +#include "test_tc_change_tail_pmtu.skel.h" + +#define CLIENT_NS "tc-change-tail-cli-ns" +#define SERVER_NS "tc-change-tail-srv-ns" +#define CLIENT_IP "192.168.1.1" +#define SERVER_IP "192.168.1.2" + +#define TEST_PMTU 1000 +#define TEST_MSS_MAX (TEST_PMTU - 20 - 20) +#define TIMEOUT_MS 3000 +#define XFER_BYTES 8192 + +void test_tc_change_tail_pmtu(void) +{ + LIBBPF_OPTS(bpf_tcx_opts, tcx_opts); + int mss_before = 0, mss_after = 0, ifindex, port; + int srv_fd = -1, srv_conn_fd = -1, cli_fd = -1; + struct test_tc_change_tail_pmtu *skel = NULL; + struct nstoken *nstoken = NULL; + static char buf[XFER_BYTES]; + socklen_t optlen; + ssize_t bytes; + size_t total; + + if (!ASSERT_OK(make_netns(CLIENT_NS), "make client ns")) + return; + if (!ASSERT_OK(make_netns(SERVER_NS), "make server ns")) + goto out_client_ns; + + nstoken = open_netns(CLIENT_NS); + if (!ASSERT_OK_PTR(nstoken, "open client ns")) + goto out; + SYS(out, "ip link add veth1 type veth peer name veth2 netns " SERVER_NS); + SYS(out, "ip -4 addr add " CLIENT_IP "/24 dev veth1"); + SYS(out, "ip link set veth1 up"); + ifindex = if_nametoindex("veth1"); + if (!ASSERT_NEQ(ifindex, 0, "if_nametoindex")) + goto out; + close_netns(nstoken); + nstoken = NULL; + + nstoken = open_netns(SERVER_NS); + if (!ASSERT_OK_PTR(nstoken, "open server ns")) + goto out; + SYS(out, "ip -4 addr add " SERVER_IP "/24 dev veth2"); + SYS(out, "ip link set veth2 up"); + srv_fd = start_server(AF_INET, SOCK_STREAM, SERVER_IP, 0, TIMEOUT_MS); + if (!ASSERT_OK_FD(srv_fd, "start server")) + goto out; + close_netns(nstoken); + nstoken = NULL; + + skel = test_tc_change_tail_pmtu__open_and_load(); + if (!ASSERT_OK_PTR(skel, "open and load skeleton")) + goto out; + + port = get_socket_local_port(srv_fd); + if (!ASSERT_GE(port, 0, "get server port")) + goto out; + + skel->bss->server_port = port; + skel->bss->pmtu = TEST_PMTU; + + nstoken = open_netns(CLIENT_NS); + if (!ASSERT_OK_PTR(nstoken, "open client ns")) + goto out; + + skel->links.change_tail_icmp = + bpf_program__attach_tcx(skel->progs.change_tail_icmp, ifindex, + &tcx_opts); + if (!ASSERT_OK_PTR(skel->links.change_tail_icmp, "attach tcx")) + goto out; + + cli_fd = connect_to_fd(srv_fd, TIMEOUT_MS); + if (!ASSERT_OK_FD(cli_fd, "connect to server")) + goto out; + srv_conn_fd = accept(srv_fd, NULL, NULL); + if (!ASSERT_OK_FD(srv_conn_fd, "accept connection")) + goto out; + if (!ASSERT_OK(settimeo(srv_conn_fd, TIMEOUT_MS), "set server timeout")) + goto out; + + optlen = sizeof(mss_before); + if (!ASSERT_OK(getsockopt(cli_fd, IPPROTO_TCP, TCP_MAXSEG, &mss_before, + &optlen), "get mss before")) + goto out; + + bytes = send(cli_fd, buf, sizeof(buf), 0); + if (!ASSERT_EQ(bytes, (ssize_t)sizeof(buf), "send data")) + goto out; + + for (total = 0; total < sizeof(buf); total += bytes) { + bytes = recv(srv_conn_fd, buf, sizeof(buf), 0); + if (bytes <= 0) + break; + } + + ASSERT_EQ(total, sizeof(buf), "receive data"); + ASSERT_OK(skel->data->change_tail_ret, "change tail"); + ASSERT_OK(skel->bss->adjust_room_ret, "adjust room"); + ASSERT_TRUE(skel->bss->icmp_sent, "icmp sent"); + + optlen = sizeof(mss_after); + if (!ASSERT_OK(getsockopt(cli_fd, IPPROTO_TCP, TCP_MAXSEG, &mss_after, + &optlen), "get mss after")) + goto out; + + ASSERT_LT(mss_after, mss_before, "mss reduced"); + ASSERT_LE(mss_after, TEST_MSS_MAX, "mss below pmtu"); +out: + close(srv_conn_fd); + close(cli_fd); + close(srv_fd); + test_tc_change_tail_pmtu__destroy(skel); + close_netns(nstoken); + remove_netns(SERVER_NS); +out_client_ns: + remove_netns(CLIENT_NS); +} diff --git a/tools/testing/selftests/bpf/prog_tests/verifier.c b/tools/testing/selftests/bpf/prog_tests/verifier.c index 64ac49ad67e6..8b439e194bcc 100644 --- a/tools/testing/selftests/bpf/prog_tests/verifier.c +++ b/tools/testing/selftests/bpf/prog_tests/verifier.c @@ -23,6 +23,7 @@ #include "verifier_bpf_trap.skel.h" #include "verifier_bswap.skel.h" #include "verifier_btf_ctx_access.skel.h" +#include "verifier_btf_flex_array.skel.h" #include "verifier_btf_unreliable_prog.skel.h" #include "verifier_call_large_imm.skel.h" #include "verifier_cfg.skel.h" @@ -53,6 +54,7 @@ #include "verifier_iterating_callbacks.skel.h" #include "verifier_jeq_infer_not_null.skel.h" #include "verifier_jit_convergence.skel.h" +#include "verifier_kfunc_perfmon.skel.h" #include "verifier_ld_ind.skel.h" #include "verifier_ldsx.skel.h" #include "verifier_leak_ptr.skel.h" @@ -186,6 +188,7 @@ void test_verifier_bpf_get_stack(void) { RUN(verifier_bpf_get_stack); } void test_verifier_bpf_trap(void) { RUN(verifier_bpf_trap); } void test_verifier_bswap(void) { RUN(verifier_bswap); } void test_verifier_btf_ctx_access(void) { RUN(verifier_btf_ctx_access); } +void test_verifier_btf_flex_array(void) { RUN(verifier_btf_flex_array); } void test_verifier_btf_unreliable_prog(void) { RUN(verifier_btf_unreliable_prog); } void test_verifier_call_large_imm(void) { RUN(verifier_call_large_imm); } void test_verifier_cfg(void) { RUN(verifier_cfg); } @@ -216,6 +219,7 @@ void test_verifier_int_ptr(void) { RUN(verifier_int_ptr); } void test_verifier_iterating_callbacks(void) { RUN(verifier_iterating_callbacks); } void test_verifier_jeq_infer_not_null(void) { RUN(verifier_jeq_infer_not_null); } void test_verifier_jit_convergence(void) { RUN(verifier_jit_convergence); } +void test_verifier_kfunc_perfmon(void) { RUN(verifier_kfunc_perfmon); } void test_verifier_load_acquire(void) { RUN(verifier_load_acquire); } void test_verifier_ld_ind(void) { RUN(verifier_ld_ind); } void test_verifier_ldsx(void) { RUN(verifier_ldsx); } diff --git a/tools/testing/selftests/bpf/progs/dynptr_success.c b/tools/testing/selftests/bpf/progs/dynptr_success.c index e0745b6e467e..b668ebd61fc7 100644 --- a/tools/testing/selftests/bpf/progs/dynptr_success.c +++ b/tools/testing/selftests/bpf/progs/dynptr_success.c @@ -10,6 +10,7 @@ #include "errno.h" #define PAGE_SIZE_64K 65536 +#define TEST_SKB_LINEAR_SIZE (sizeof(struct ethhdr) + sizeof(struct iphdr)) char _license[] SEC("license") = "GPL"; @@ -211,6 +212,25 @@ int test_dynptr_skb_data(struct __sk_buff *skb) return 1; } +SEC("?tc") +int test_dynptr_skb_slice_non_linear(struct __sk_buff *skb) +{ + struct bpf_dynptr ptr; + void *data; + + if (bpf_dynptr_from_skb(skb, 0, &ptr)) { + err = 1; + return 1; + } + + /* Ensure we cannot read past the end of the buffer. */ + data = bpf_dynptr_slice(&ptr, TEST_SKB_LINEAR_SIZE + 1, NULL, 1); + if (data) + err = 2; + + return 1; +} + SEC("?tc") int test_dynptr_skb_meta_data(struct __sk_buff *skb) { diff --git a/tools/testing/selftests/bpf/progs/exceptions.c b/tools/testing/selftests/bpf/progs/exceptions.c index c8d716fbd419..91c81971e58c 100644 --- a/tools/testing/selftests/bpf/progs/exceptions.c +++ b/tools/testing/selftests/bpf/progs/exceptions.c @@ -212,6 +212,36 @@ int exception_throw_subprog(struct __sk_buff *ctx) return 0; } +u64 exception_cb_stack_src = 0x1234; + +/* + * The address handed to the helper has to be this callback's own stack + * slot, not one from a frame that is already gone. + */ +__noinline int exception_cb_stack(u64 cookie) +{ + volatile u64 val = 0xdead; + + bpf_probe_read_kernel((void *)&val, sizeof(val), &exception_cb_stack_src); + return val; +} + +/* Throws from a subprogram that has a stack of its own. */ +__noinline static int throwing_subprog_stack(struct __sk_buff *ctx) +{ + volatile u64 pad[4] = {}; + + bpf_throw(pad[0]); + return 0; +} + +SEC("tc") +__exception_cb(exception_cb_stack) +int exception_throw_subprog_stack_cb(struct __sk_buff *ctx) +{ + return throwing_subprog_stack(ctx); +} + __noinline int assert_nz_gfunc(u64 c) { volatile u64 cookie = c; diff --git a/tools/testing/selftests/bpf/progs/iters_state_safety.c b/tools/testing/selftests/bpf/progs/iters_state_safety.c index 646026430e9b..e5bb9fe6d5e5 100644 --- a/tools/testing/selftests/bpf/progs/iters_state_safety.c +++ b/tools/testing/selftests/bpf/progs/iters_state_safety.c @@ -52,6 +52,28 @@ int create_and_destroy(void *ctx) return 0; } +/* fp+0 is not a stack slot. bpf_get_spi(0) used to alias spi 0 (fp-8). */ +SEC("?raw_tp") +__failure __msg("cannot pass in iter at an offset=0") +int destroy_fp0_fail(void *ctx) +{ + struct bpf_iter_num iter; + + asm volatile ("r1 = %[iter];" + "r2 = 0;" + "r3 = 1000;" + "call %[bpf_iter_num_new];" + /* r10 is fp+0, one byte above the top of the BPF stack */ + "r1 = r10;" + "call %[bpf_iter_num_destroy];" + : + : __imm_ptr(iter), ITER_HELPERS + : __clobber_common + ); + + return 0; +} + SEC("?raw_tp") __failure __msg("Unreleased reference id=1") int create_and_forget_to_destroy_fail(void *ctx) diff --git a/tools/testing/selftests/bpf/progs/sock_destroy_prog.c b/tools/testing/selftests/bpf/progs/sock_destroy_prog.c index 9e0bf7a54cec..0a8887543218 100644 --- a/tools/testing/selftests/bpf/progs/sock_destroy_prog.c +++ b/tools/testing/selftests/bpf/progs/sock_destroy_prog.c @@ -7,6 +7,8 @@ #include "bpf_tracing_net.h" __be16 serv_port = 0; +int tw_found = 0; +int tw_destroy_err = 0; int bpf_sock_destroy(struct sock_common *sk) __ksym; @@ -100,6 +102,34 @@ int iter_tcp6_server(struct bpf_iter__tcp *ctx) return 0; } +SEC("iter/tcp") +int iter_tcp6_timewait(struct bpf_iter__tcp *ctx) +{ + struct sock_common *sk_common = ctx->sk_common; + __u64 *val; + int key = 0; + + if (!sk_common) + return 0; + + if (sk_common->skc_family != AF_INET6) + return 0; + + if (!bpf_skc_to_tcp_timewait_sock(sk_common)) + return 0; + + val = bpf_map_lookup_elem(&tcp_conn_sockets, &key); + if (!val) + return 0; + /* The timewait sock inherits the cookie of the closed client sock. */ + if (bpf_get_socket_cookie(sk_common) != *val) + return 0; + + tw_found++; + tw_destroy_err = bpf_sock_destroy(sk_common); + + return 0; +} SEC("iter/udp") int iter_udp6_client(struct bpf_iter__udp *ctx) diff --git a/tools/testing/selftests/bpf/progs/test_spin_lock_fail.c b/tools/testing/selftests/bpf/progs/test_spin_lock_fail.c index f678ee6bd7ea..55282f20fa32 100644 --- a/tools/testing/selftests/bpf/progs/test_spin_lock_fail.c +++ b/tools/testing/selftests/bpf/progs/test_spin_lock_fail.c @@ -14,17 +14,18 @@ struct array_map { __type(key, int); __type(value, struct foo); __uint(max_entries, 1); -} array_map SEC(".maps"); +} array_map SEC(".maps"), array_map_b SEC(".maps"); struct { __uint(type, BPF_MAP_TYPE_ARRAY_OF_MAPS); - __uint(max_entries, 1); + __uint(max_entries, 2); __type(key, int); __type(value, int); __array(values, struct array_map); } map_of_maps SEC(".maps") = { .values = { [0] = &array_map, + [1] = &array_map_b, }, }; @@ -314,4 +315,66 @@ int lock_global_sleepable_subprog_indirect(struct __sk_buff *ctx) return ret; } +struct { + __uint(type, BPF_MAP_TYPE_ARRAY); + __uint(max_entries, 2); + __type(key, int); + __type(value, struct foo); +} callback_array_map SEC(".maps"); + +struct callback_ctx { + struct foo *value; +}; + +static long lock_different_value(struct bpf_map *map, int *key, + struct foo *value, struct callback_ctx *ctx) +{ + bpf_spin_lock(&value->lock); + bpf_spin_unlock(&ctx->value->lock); + return 0; +} + +static long nest_lock_different_value(struct bpf_map *map, int *key, + struct foo *value, void *data) +{ + struct callback_ctx ctx = { .value = value }; + + bpf_for_each_map_elem(&callback_array_map, lock_different_value, &ctx, 0); + return 0; +} + +SEC("?tc") +int callback_value_lock_identity(void *ctx) +{ + bpf_for_each_map_elem(&callback_array_map, nest_lock_different_value, NULL, 0); + return 0; +} + +static long nest_lock_different_inner_value(struct bpf_map *map, int *key, + struct foo *value, void *data) +{ + struct callback_ctx ctx = { .value = value }; + int inner_key = 1; + void *inner_map; + + inner_map = bpf_map_lookup_elem(&map_of_maps, &inner_key); + if (!inner_map) + return 0; + bpf_for_each_map_elem(inner_map, lock_different_value, &ctx, 0); + return 0; +} + +SEC("?tc") +int callback_inner_map_value_lock_identity(void *ctx) +{ + int inner_key = 0; + void *inner_map; + + inner_map = bpf_map_lookup_elem(&map_of_maps, &inner_key); + if (!inner_map) + return 0; + bpf_for_each_map_elem(inner_map, nest_lock_different_inner_value, NULL, 0); + return 0; +} + char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/test_tc_change_tail_pmtu.c b/tools/testing/selftests/bpf/progs/test_tc_change_tail_pmtu.c new file mode 100644 index 000000000000..5c4c07545bc9 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/test_tc_change_tail_pmtu.c @@ -0,0 +1,129 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include +#include + +#include +#include +#include +#include +#include +#include + +#include +#include + +#define ICMP_SAMPLE_LEN (sizeof(struct iphdr) + 8) +#define ICMP_HDRS_LEN (sizeof(struct iphdr) + sizeof(struct icmphdr)) + +__be16 server_port = 0; +__u16 pmtu = 0; + +long change_tail_ret = 1; +long adjust_room_ret = 0; +bool icmp_sent = false; +bool icmp_err = false; + +static __always_inline __sum16 csum_fold(__wsum csum) +{ + csum = (csum & 0xffff) + (csum >> 16); + csum = (csum & 0xffff) + (csum >> 16); + + return (__sum16)~csum; +} + +SEC("tc/egress") +int change_tail_icmp(struct __sk_buff *skb) +{ + __u8 smac[ETH_ALEN], dmac[ETH_ALEN]; + void *data, *data_end; + struct icmphdr *icmp; + struct ethhdr *eth; + struct tcphdr *tcp; + __be32 saddr, daddr; + struct iphdr *ip; + __wsum csum; + + if (icmp_sent || icmp_err) + return TCX_PASS; + + data = (void *)(long)skb->data; + data_end = (void *)(long)skb->data_end; + + eth = data; + if ((void *)(eth + 1) > data_end) + return TCX_PASS; + if (eth->h_proto != bpf_htons(ETH_P_IP)) + return TCX_PASS; + + ip = (void *)(eth + 1); + if ((void *)(ip + 1) > data_end) + return TCX_PASS; + if (ip->ihl != 5 || ip->protocol != IPPROTO_TCP) + return TCX_PASS; + + tcp = (void *)(ip + 1); + if ((void *)(tcp + 1) > data_end) + return TCX_PASS; + if (tcp->dest != server_port) + return TCX_PASS; + if (bpf_ntohs(ip->tot_len) <= sizeof(*ip) + tcp->doff * 4) + return TCX_PASS; + + __builtin_memcpy(smac, eth->h_source, ETH_ALEN); + __builtin_memcpy(dmac, eth->h_dest, ETH_ALEN); + saddr = ip->saddr; + daddr = ip->daddr; + + change_tail_ret = bpf_skb_change_tail(skb, ETH_HLEN + ICMP_SAMPLE_LEN, 0); + if (change_tail_ret) { + icmp_err = true; + return TCX_PASS; + } + + adjust_room_ret = bpf_skb_adjust_room(skb, ICMP_HDRS_LEN, + BPF_ADJ_ROOM_MAC, + BPF_F_ADJ_ROOM_NO_CSUM_RESET); + if (adjust_room_ret) { + icmp_err = true; + return TCX_DROP; + } + + data = (void *)(long)skb->data; + data_end = (void *)(long)skb->data_end; + + eth = data; + ip = (void *)(eth + 1); + icmp = (void *)(ip + 1); + if ((void *)icmp + sizeof(*icmp) + ICMP_SAMPLE_LEN > data_end) { + icmp_err = true; + return TCX_DROP; + } + + __builtin_memcpy(eth->h_dest, smac, ETH_ALEN); + __builtin_memcpy(eth->h_source, dmac, ETH_ALEN); + + __builtin_memset(icmp, 0, sizeof(*icmp)); + icmp->type = ICMP_DEST_UNREACH; + icmp->code = ICMP_FRAG_NEEDED; + icmp->un.frag.mtu = bpf_htons(pmtu); + + __builtin_memset(ip, 0, sizeof(*ip)); + ip->version = 4; + ip->ihl = 5; + ip->ttl = 64; + ip->protocol = IPPROTO_ICMP; + ip->tot_len = bpf_htons(ICMP_HDRS_LEN + ICMP_SAMPLE_LEN); + ip->saddr = daddr; + ip->daddr = saddr; + + csum = bpf_csum_diff(NULL, 0, (__be32 *)icmp, + sizeof(*icmp) + ICMP_SAMPLE_LEN, 0); + icmp->checksum = csum_fold(csum); + csum = bpf_csum_diff(NULL, 0, (__be32 *)ip, sizeof(*ip), 0); + ip->check = csum_fold(csum); + icmp_sent = true; + return bpf_redirect(skb->ifindex, BPF_F_INGRESS); +} + +char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/verifier_arena.c b/tools/testing/selftests/bpf/progs/verifier_arena.c index 815f342eb4b0..3e33766547c0 100644 --- a/tools/testing/selftests/bpf/progs/verifier_arena.c +++ b/tools/testing/selftests/bpf/progs/verifier_arena.c @@ -561,6 +561,55 @@ int arena_ptr_add_arena_ptr(void *ctx) return 0; } +SEC("syscall") +__failure __msg("same insn cannot be used with and without arena pointer") +int mixed_arena_scalar_alu64_scalar_first(void *ctx) +{ + volatile register __u64 reg asm("r3"); + __u32 pick_arena = bpf_get_prandom_u32(); + + reg = 1ULL << 32; + + if (pick_arena) { + asm volatile ( + "r9 = %[arena] ll;" + "%[reg] = 0;" + "%[reg] = addr_space_cast(%[reg], 0x0, 0x1);" + : [reg] "=r"(reg) + : __imm_addr(arena) + : "r9" + ); + } + + reg += 1; + + return 0; +} + +SEC("syscall") +__failure __msg("same insn cannot be used with and without arena pointer") +int mixed_arena_scalar_alu64_arena_first(void *ctx) +{ + volatile register __u64 reg asm("r3"); + __u32 pick_scalar = bpf_get_prandom_u32(); + + asm volatile ( + "r9 = %[arena] ll;" + "%[reg] = 0;" + "%[reg] = addr_space_cast(%[reg], 0x0, 0x1);" + : [reg] "=r"(reg) + : __imm_addr(arena) + : "r9" + ); + + if (pick_scalar) + reg = 1ULL << 32; + + reg += 1; + + return 0; +} + SEC("syscall") __success __retval(0) int scalar_xor_arena_ptr(void *ctx) diff --git a/tools/testing/selftests/bpf/progs/verifier_async_cb_context.c b/tools/testing/selftests/bpf/progs/verifier_async_cb_context.c index e0926767bbd3..1b653bfb63eb 100644 --- a/tools/testing/selftests/bpf/progs/verifier_async_cb_context.c +++ b/tools/testing/selftests/bpf/progs/verifier_async_cb_context.c @@ -9,6 +9,11 @@ char _license[] SEC("license") = "GPL"; +struct task_struct *bpf_task_acquire(struct task_struct *p) __ksym; +void bpf_task_release(struct task_struct *p) __ksym; +void bpf_rcu_read_lock(void) __ksym; +void bpf_rcu_read_unlock(void) __ksym; + /* Timer tests */ struct timer_elem { @@ -164,6 +169,7 @@ int syscall_btf_find_prog(void *ctx) struct wq_elem { struct bpf_wq w; + struct task_struct __kptr *task; }; struct { @@ -217,6 +223,106 @@ int wq_sleepable_prog(void *ctx) return 0; } +__noinline int wq_global_acquire(void) +{ + struct task_struct *task, *acquired; + struct wq_elem *val; + int key = 0; + + val = bpf_map_lookup_elem(&wq_map, &key); + if (!val) + return 0; + + task = val->task; + if (!task) + return 0; + + acquired = bpf_task_acquire(task); + if (acquired) + bpf_task_release(acquired); + return 0; +} + +static int wq_global_rcu_cb(void *map, int *key, void *value) +{ + wq_global_acquire(); + return 0; +} + +SEC("fentry/bpf_fentry_test1") +__failure __msg("R1 must be a rcu pointer") +int wq_global_rcu_prog(void *ctx) +{ + struct wq_elem *val; + int key = 0; + + val = bpf_map_lookup_elem(&wq_map, &key); + if (!val) + return 0; + + bpf_wq_init(&val->w, &wq_map, 0); + bpf_wq_set_callback(&val->w, wq_global_rcu_cb, 0); + return 0; +} + +static int wq_global_rcu_lock_cb(void *map, int *key, void *value) +{ + bpf_rcu_read_lock(); + wq_global_acquire(); + bpf_rcu_read_unlock(); + return 0; +} + +SEC("fentry/bpf_fentry_test1") +__success +int wq_global_rcu_lock_prog(void *ctx) +{ + struct wq_elem *val; + int key = 0; + + /* Verify the same global subprog in non-sleepable and protected contexts. */ + wq_global_acquire(); + + val = bpf_map_lookup_elem(&wq_map, &key); + if (!val) + return 0; + + bpf_wq_init(&val->w, &wq_map, 0); + bpf_wq_set_callback(&val->w, wq_global_rcu_lock_cb, 0); + return 0; +} + +__weak __noinline int wq_global_no_rcu(void) +{ + return 0; +} + +static int wq_global_no_rcu_cb(void *map, int *key, void *value) +{ + wq_global_no_rcu(); + return 0; +} + +SEC("fentry/bpf_fentry_test1") +__success __log_level(4) +__msg("subprog {{[0-9]+}} (wq_global_no_rcu) global insns_self 4 insns_total 4 stack 0") +int wq_global_no_rcu_prog(void *ctx) +{ + struct wq_elem *val; + int key = 0; + + /* Verify the same global in non-sleepable and unprotected contexts. */ + wq_global_no_rcu(); + + val = bpf_map_lookup_elem(&wq_map, &key); + if (!val) + return 0; + + bpf_wq_init(&val->w, &wq_map, 0); + bpf_wq_set_callback(&val->w, wq_global_no_rcu_cb, 0); + return 0; +} + /* Task work tests */ struct task_work_elem { diff --git a/tools/testing/selftests/bpf/progs/verifier_btf_flex_array.c b/tools/testing/selftests/bpf/progs/verifier_btf_flex_array.c new file mode 100644 index 000000000000..59b84261f622 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/verifier_btf_flex_array.c @@ -0,0 +1,56 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include +#include + +#include "bpf_experimental.h" +#include "bpf_misc.h" + +struct test_empty_event {}; + +struct test_flex_batch { + int nr; + struct test_empty_event events[]; +}; + +struct map_value { + struct test_flex_batch __kptr *batch; +}; + +struct { + __uint(type, BPF_MAP_TYPE_ARRAY); + __type(key, int); + __type(value, struct map_value); + __uint(max_entries, 1); +} batches SEC(".maps"); + +SEC("syscall") +__description("btf walk into flexible array of zero-sized elements") +__failure __msg("access beyond struct test_flex_batch at off 4 size 1") +int stash_and_peek(void *ctx) +{ + struct test_flex_batch *b, *old; + struct map_value *v; + int key = 0; + + v = bpf_map_lookup_elem(&batches, &key); + if (!v) + return 0; + + b = bpf_obj_new(struct test_flex_batch); + if (!b) + return 0; + b->nr = 1; + + old = bpf_kptr_xchg(&v->batch, b); + if (old) + bpf_obj_drop(old); + + b = v->batch; + if (!b) + return 0; + + return b->nr + *(char *)&b->events[0]; +} + +char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/verifier_global_ptr_args.c b/tools/testing/selftests/bpf/progs/verifier_global_ptr_args.c index a3d2af8dc839..dcc2dd46751a 100644 --- a/tools/testing/selftests/bpf/progs/verifier_global_ptr_args.c +++ b/tools/testing/selftests/bpf/progs/verifier_global_ptr_args.c @@ -350,4 +350,57 @@ int anything_to_untrusted_mem(void *ctx) return 0; } +struct pkt_arg { + __u64 x; + __u8 pad[56]; +}; + +__weak int subprog_pkt_ptr_no_change(struct pkt_arg *p) +{ + if (!p) + return 0; + + return p->x; +} + +SEC("?tc") +__success +int pkt_ptr_to_global_mem_arg_no_change(struct __sk_buff *skb) +{ + void *data = (void *)(long)skb->data; + void *data_end = (void *)(long)skb->data_end; + struct pkt_arg *p = data; + + if ((void *)(p + 1) > data_end) + return 0; + + return subprog_pkt_ptr_no_change(p); +} + +__weak int subprog_pkt_ptr_changes_data(struct __sk_buff *skb __arg_ctx, + struct pkt_arg *p) +{ + if (!p) + return 0; + + bpf_skb_pull_data(skb, 0); + return p->x; +} + +SEC("?tc") +__failure __log_level(2) +__msg("R2 is a packet pointer, but func#{{[0-9]+}} may change packet data") +__msg("Caller passes invalid args into func#{{[0-9]+}} ('subprog_pkt_ptr_changes_data')") +int pkt_ptr_to_global_mem_arg_changes_data(struct __sk_buff *skb) +{ + void *data = (void *)(long)skb->data; + void *data_end = (void *)(long)skb->data_end; + struct pkt_arg *p = data; + + if ((void *)(p + 1) > data_end) + return 0; + + return subprog_pkt_ptr_changes_data(skb, p); +} + char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/verifier_gotox.c b/tools/testing/selftests/bpf/progs/verifier_gotox.c index 5b18c9a27717..0e27c2c79c57 100644 --- a/tools/testing/selftests/bpf/progs/verifier_gotox.c +++ b/tools/testing/selftests/bpf/progs/verifier_gotox.c @@ -47,6 +47,54 @@ DEFINE_SIMPLE_JUMP_TABLE_PROG(reserved_field_src_reg, BPF_REG_1, 0, 0, __fa DEFINE_SIMPLE_JUMP_TABLE_PROG(reserved_field_non_zero_off, BPF_REG_0, 1, 0, __failure __msg("BPF_JA|BPF_X uses reserved fields")) DEFINE_SIMPLE_JUMP_TABLE_PROG(reserved_field_non_zero_imm, BPF_REG_0, 0, 1, __failure __msg("BPF_JA|BPF_X uses reserved fields")) +#define DEFINE_TERMINAL_GOTOX_PROG(NAME, BASE) \ + __naked void NAME(void) \ + { \ + asm volatile (" \ + .pushsection .jumptables,\"\",@progbits; \ +jt0_%=: \ + .quad ret0_%= - " BASE "; \ + .size jt0_%=, 8; \ + .global jt0_%=; \ + .popsection; \ + \ + r0 = jt0_%= ll; \ + r0 = *(u64 *)(r0 + 0); \ + goto end_%=; \ +ret0_%=: \ + r0 = 0; \ + exit; \ +end_%=: \ + .8byte %[gotox_r0]; \ +" : \ + : __imm_insn(gotox_r0, BPF_RAW_INSN(BPF_JMP | BPF_JA | BPF_X, \ + BPF_REG_0, 0, 0, 0)) \ + : __clobber_all); \ + } + +SEC("socket") +__success __retval(0) +DEFINE_TERMINAL_GOTOX_PROG(jump_table_terminal_gotox, "socket") + +static __noinline __used +DEFINE_TERMINAL_GOTOX_PROG(terminal_gotox_subprog1, ".text") + +static __noinline __used int terminal_gotox_subprog2(void) +{ + return 0; +} + +SEC("socket") +__success __retval(0) +__naked void jump_table_terminal_gotox_subprog(void) +{ + asm volatile (" \ + call terminal_gotox_subprog1; \ + call terminal_gotox_subprog2; \ + exit; \ +" ::: __clobber_all); +} + /* * Gotox is forbidden when there is no jump table loaded * which points to the sub-function where the gotox is used diff --git a/tools/testing/selftests/bpf/progs/verifier_kfunc_perfmon.c b/tools/testing/selftests/bpf/progs/verifier_kfunc_perfmon.c new file mode 100644 index 000000000000..76c39ef30e96 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/verifier_kfunc_perfmon.c @@ -0,0 +1,75 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include +#include +#include "bpf_misc.h" + +void *user_ptr; +char dynptr_buf[8]; +u64 kaddr; + +extern struct kmem_cache *bpf_get_kmem_cache(u64 addr) __ksym; + +SEC("socket") +__success +__caps_unpriv(CAP_BPF) +__failure_unpriv +__msg_unpriv("bpf_rdonly_cast is allowed only to CAP_PERFMON and CAP_SYS_ADMIN") +int rdonly_cast_noperfmon(void *ctx) +{ + char *p = bpf_rdonly_cast(0, 0); + + return p[0x7fff]; +} + +SEC("socket") +__success +__caps_unpriv(CAP_BPF) +__failure_unpriv +__msg_unpriv("bpf_probe_read_kernel_dynptr is allowed only to CAP_PERFMON and CAP_SYS_ADMIN") +int probe_read_kernel_dynptr_noperfmon(void *ctx) +{ + struct bpf_dynptr dptr; + + bpf_dynptr_from_mem(dynptr_buf, sizeof(dynptr_buf), 0, &dptr); + bpf_probe_read_kernel_dynptr(&dptr, 0, sizeof(dynptr_buf), user_ptr); + return 0; +} + +SEC("socket") +__success +__caps_unpriv(CAP_BPF) +__failure_unpriv +__msg_unpriv("bpf_stream_vprintk is allowed only to CAP_PERFMON and CAP_SYS_ADMIN") +int stream_vprintk_noperfmon(void *ctx) +{ + bpf_stream_printk(BPF_STDOUT, "%pB", (void *)kaddr); + return 0; +} + +SEC("socket") +__success +__caps_unpriv(CAP_BPF) +__failure_unpriv +__msg_unpriv("bpf_get_kmem_cache is allowed only to CAP_PERFMON and CAP_SYS_ADMIN") +int get_kmem_cache_noperfmon(void *ctx) +{ + return !!bpf_get_kmem_cache(kaddr); +} + +__weak int subprog_untrusted_read(void *p __arg_untrusted) +{ + return *(char *)p; +} + +SEC("socket") +__success +__caps_unpriv(CAP_BPF) +__failure_unpriv +__msg_unpriv("rdonly_untrusted_mem access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN") +int arg_untrusted_read_noperfmon(void *ctx) +{ + return subprog_untrusted_read(0); +} + +char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/verifier_loops1.c b/tools/testing/selftests/bpf/progs/verifier_loops1.c index d248ce877f14..48a966cda199 100644 --- a/tools/testing/selftests/bpf/progs/verifier_loops1.c +++ b/tools/testing/selftests/bpf/progs/verifier_loops1.c @@ -303,4 +303,40 @@ __naked void maybe_exit_scc_bug1(void) ::: __clobber_all); } +/* + * The loop reads zero from the caller's stack on its first iteration and + * one from the callee's stack on its second iteration. At the loop header, + * only the frame number of the pointer in r1 changes. + */ +static __naked __noinline __used +void loop_stack_frames_reg(void) +{ + asm volatile ( + "*(u64 *)(r10 - 8) = 1;" +"1:" + "r0 = *(u64 *)(r1 + 0);" + "if r0 != 0 goto 2f;" + "r1 = r10;" + "r1 += -8;" + "goto 1b;" +"2:" + "exit;" + ::: __clobber_all); +} + +SEC("xdp") +__description("bounded loop changing stack frame in a register") +__success __retval(1) +__flag(BPF_F_TEST_STATE_FREQ) +__naked void bounded_loop_stack_frames_reg(void) +{ + asm volatile ( + "*(u64 *)(r10 - 8) = 0;" + "r1 = r10;" + "r1 += -8;" + "call loop_stack_frames_reg;" + "exit;" + ::: __clobber_all); +} + char _license[] SEC("license") = "GPL"; diff --git a/tools/testing/selftests/bpf/progs/verifier_sock.c b/tools/testing/selftests/bpf/progs/verifier_sock.c index 4f2f3209eec8..bf9f6fb6582c 100644 --- a/tools/testing/selftests/bpf/progs/verifier_sock.c +++ b/tools/testing/selftests/bpf/progs/verifier_sock.c @@ -88,6 +88,44 @@ l0_%=: r0 = *(u32*)(r1 + %[bpf_sock_family]); \ : __clobber_all); } +SEC("socket") +__description("skb->sk: sk->rx_queue_mapping [no sign extension]") +__success __success_unpriv __retval(0) +__naked void sk_rx_queue_mapping_no_sign_ext(void) +{ + asm volatile (" \ + r1 = *(u64*)(r1 + %[__sk_buff_sk]); \ + if r1 != 0 goto l0_%=; \ + r0 = 0xdead; \ + exit; \ +l0_%=: r0 = *(u32*)(r1 + %[bpf_sock_rx_queue_mapping]); \ + r0 >>= 32; \ + exit; \ +" : + : __imm_const(__sk_buff_sk, offsetof(struct __sk_buff, sk)), + __imm_const(bpf_sock_rx_queue_mapping, offsetof(struct bpf_sock, rx_queue_mapping)) + : __clobber_all); +} + +SEC("socket") +__description("skb->sk: sk->rx_queue_mapping [narrow load mask]") +__success __success_unpriv __retval(0) +__naked void sk_rx_queue_mapping_narrow_load_mask(void) +{ + asm volatile (" \ + r1 = *(u64*)(r1 + %[__sk_buff_sk]); \ + if r1 != 0 goto l0_%=; \ + r0 = 0xdead; \ + exit; \ +l0_%=: r0 = *(u16*)(r1 + %[bpf_sock_rx_queue_mapping]); \ + r0 >>= 16; \ + exit; \ +" : + : __imm_const(__sk_buff_sk, offsetof(struct __sk_buff, sk)), + __imm_const(bpf_sock_rx_queue_mapping, offsetof(struct bpf_sock, rx_queue_mapping)) + : __clobber_all); +} + SEC("cgroup/skb") __description("skb->sk: sk->type [fullsock field]") __failure __msg("invalid sock_common access") diff --git a/tools/testing/selftests/bpf/progs/verifier_xdp_direct_packet_access.c b/tools/testing/selftests/bpf/progs/verifier_xdp_direct_packet_access.c index 0b86d95a4133..9866bc154194 100644 --- a/tools/testing/selftests/bpf/progs/verifier_xdp_direct_packet_access.c +++ b/tools/testing/selftests/bpf/progs/verifier_xdp_direct_packet_access.c @@ -1719,4 +1719,39 @@ l0_%=: r0 = 0; \ : __clobber_all); } +SEC("xdp") +__description("XDP pkt regsafe preserves packet pointer class displacement") +__failure __msg("R2 min value is outside of the allowed memory range") +__flag(BPF_F_ANY_ALIGNMENT) __flag(BPF_F_TEST_STATE_FREQ) +__naked void pkt_regsafe_class_displacement(void) +{ + asm volatile (" \ + r8 = *(u32 *)(r1 + %[xdp_md_data_end]); \ + r9 = *(u32 *)(r1 + %[xdp_md_data]); \ + r4 = *(u32 *)(r1 + %[xdp_md_rx_queue_index]); \ + r4 &= 15; \ + r0 = *(u32 *)(r1 + %[xdp_md_ingress_ifindex]); \ + if r0 != 0 goto l0_%=; \ + r2 = r9; \ + r2 += r4; \ + r3 = r2; \ + r3 += 8; \ + goto l1_%=; \ +l0_%=: r4 &= 3; \ + r4 += 8; \ + r2 = r9; \ + r2 += r4; \ + r3 = r2; \ +l1_%=: if r3 > r8 goto l2_%=; \ + r0 = *(u64 *)(r2 + 0); \ +l2_%=: r0 = 0; \ + exit; \ +" : + : __imm_const(xdp_md_data, offsetof(struct xdp_md, data)), + __imm_const(xdp_md_data_end, offsetof(struct xdp_md, data_end)), + __imm_const(xdp_md_rx_queue_index, offsetof(struct xdp_md, rx_queue_index)), + __imm_const(xdp_md_ingress_ifindex, offsetof(struct xdp_md, ingress_ifindex)) + : __clobber_all); +} + char _license[] SEC("license") = "GPL";