mirror of
https://github.com/torvalds/linux.git
synced 2026-07-28 10:09:10 +02:00
Merge branch 'bpf-implement-stack_map_get_build_id_offset_sleepable'
Ihor Solodrai says:
====================
bpf: Implement stack_map_get_build_id_offset_sleepable()
The series introduces stack_map_get_build_id_offset_sleepable(),
fixing a gap with parsing build_id in sleepable context in stackmap.c
In particular, this fixes a deadlock in
stack_map_get_build_id_offset() doing a blocking __kernel_read(),
which happens since commit 777a8560fd ("lib/buildid: use
__kernel_read() for sleepable context").
See previous revisions for more details.
---
v6->v7:
* Addressed feedback from Andrii (mostly patch #2):
* implement proper CONFIG_PER_VMA_LOCK=n support, following a
VMA locking pattern similar to one used in PROCMAP_QUERY
* change the contract of stack_map_lock_vma(): if a non-NULL VMA
is returned, then a read lock is held
* remove now unnecessary vma_locked flag
* and various other nits
* Add vma_is_anonymous() checks where appropriate (AIs)
v6: https://lore.kernel.org/bpf/20260521225022.2695755-1-ihor.solodrai@linux.dev/
v5->v6:
* Misc refactoring (Andrii):
* add stack_map_build_id_set_valid() helper
* simplify control flow in stack_map_get_build_id_offset_sleepable()
v5: https://lore.kernel.org/bpf/20260515005244.1333013-1-ihor.solodrai@linux.dev/
v4->v5:
* Add comments explaining mmap_read_trylock() (Shakeel)
* Rebase on bpf-next (Alexei)
v4: https://lore.kernel.org/bpf/20260514184727.1067141-1-ihor.solodrai@linux.dev/
v3->v4:
* Change Fixes tag in patch #2 (AI)
* Nit in caching implementation (Mykyta)
v3: https://lore.kernel.org/bpf/20260512032906.2670326-1-ihor.solodrai@linux.dev/
v2->v3:
* Split patch #2 in two: stack_map_get_build_id_offset_sleepable()
implementation, and then introduce caching
* Drop taking mmap_lock if CONFIG_PER_VMA_LOCK=n, fall back to raw
IPs instead
* Cache vm_{start,end} in addition to prev_file (Mykyta)
v2: https://lore.kernel.org/bpf/20260409010604.1439087-1-ihor.solodrai@linux.dev/
v1->v2:
* Addressed feedback from Puranjay:
* split out a small refactoring patch
* use mmap_read_trylock()
* take into account CONFIG_PER_VMA_LOCK
* replace find_vma() with vma_lookup()
* cache prev_build_id to avoid re-parsing the same file
* Snapshot vm_pgoff and vm_start before unlocking (AI)
* To avoid repetitive unlocking statements, introduce struct
stack_map_vma_lock to hold relevant lock state info and add an
unlock helper
v1: https://lore.kernel.org/bpf/20260407223003.720428-1-ihor.solodrai@linux.dev/
---
====================
Link: https://patch.msgid.link/20260525223948.1920986-1-ihor.solodrai@linux.dev
Signed-off-by: Andrii Nakryiko <andrii@kernel.org>
This commit is contained in:
commit
7f9ce282da
|
|
@ -9,6 +9,7 @@
|
|||
#include <linux/perf_event.h>
|
||||
#include <linux/btf_ids.h>
|
||||
#include <linux/buildid.h>
|
||||
#include <linux/mmap_lock.h>
|
||||
#include "percpu_freelist.h"
|
||||
#include "mmap_unlock_work.h"
|
||||
|
||||
|
|
@ -152,6 +153,180 @@ static int fetch_build_id(struct vm_area_struct *vma, unsigned char *build_id, b
|
|||
: build_id_parse_nofault(vma, build_id, NULL);
|
||||
}
|
||||
|
||||
static inline void stack_map_build_id_set_ip(struct bpf_stack_build_id *id)
|
||||
{
|
||||
id->status = BPF_STACK_BUILD_ID_IP;
|
||||
memset(id->build_id, 0, BUILD_ID_SIZE_MAX);
|
||||
}
|
||||
|
||||
static inline u64 stack_map_build_id_offset(unsigned long vm_pgoff,
|
||||
unsigned long vm_start, u64 ip)
|
||||
{
|
||||
return (vm_pgoff << PAGE_SHIFT) + ip - vm_start;
|
||||
}
|
||||
|
||||
static inline void stack_map_build_id_set_valid(struct bpf_stack_build_id *id,
|
||||
u64 offset,
|
||||
const unsigned char *build_id)
|
||||
{
|
||||
id->status = BPF_STACK_BUILD_ID_VALID;
|
||||
id->offset = offset;
|
||||
if (id->build_id != build_id)
|
||||
memcpy(id->build_id, build_id, BUILD_ID_SIZE_MAX);
|
||||
}
|
||||
|
||||
struct stack_map_vma_lock {
|
||||
struct vm_area_struct *vma;
|
||||
struct mm_struct *mm;
|
||||
};
|
||||
|
||||
/*
|
||||
* Acquire a stable read-side reference on the VMA covering @ip.
|
||||
*
|
||||
* With CONFIG_PER_VMA_LOCK=y this returns a VMA with its per-VMA read
|
||||
* lock held and mmap_lock dropped, so the caller may sleep.
|
||||
*
|
||||
* With CONFIG_PER_VMA_LOCK=n it returns a VMA with mmap_lock still
|
||||
* held; the caller must snapshot any fields it needs and pin vm_file
|
||||
* with get_file() before stack_map_unlock_vma() drops mmap_lock, as
|
||||
* the VMA may be split, merged, or freed after that.
|
||||
*
|
||||
* Returns NULL on failure, in which case no lock is held.
|
||||
*/
|
||||
static struct vm_area_struct *
|
||||
stack_map_lock_vma(struct stack_map_vma_lock *lock, unsigned long ip)
|
||||
{
|
||||
struct mm_struct *mm = lock->mm;
|
||||
struct vm_area_struct *vma;
|
||||
|
||||
/* noop under !CONFIG_PER_VMA_LOCK */
|
||||
vma = lock_vma_under_rcu(mm, ip);
|
||||
if (vma) {
|
||||
lock->vma = vma;
|
||||
return vma;
|
||||
}
|
||||
|
||||
/*
|
||||
* Taking mmap_read_lock() is unsafe here, because the caller BPF
|
||||
* program might already hold it, causing a deadlock.
|
||||
*/
|
||||
if (!mmap_read_trylock(mm))
|
||||
return NULL;
|
||||
|
||||
vma = vma_lookup(mm, ip);
|
||||
if (!vma) {
|
||||
mmap_read_unlock(mm);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
#ifdef CONFIG_PER_VMA_LOCK
|
||||
if (!vma_start_read_locked(vma)) {
|
||||
mmap_read_unlock(mm);
|
||||
return NULL;
|
||||
}
|
||||
mmap_read_unlock(mm);
|
||||
#endif
|
||||
|
||||
lock->vma = vma;
|
||||
return vma;
|
||||
}
|
||||
|
||||
static void stack_map_unlock_vma(struct stack_map_vma_lock *lock)
|
||||
{
|
||||
#ifdef CONFIG_PER_VMA_LOCK
|
||||
vma_end_read(lock->vma);
|
||||
#else
|
||||
mmap_read_unlock(lock->mm);
|
||||
#endif
|
||||
lock->vma = NULL;
|
||||
}
|
||||
|
||||
static void stack_map_get_build_id_offset_sleepable(struct bpf_stack_build_id *id_offs,
|
||||
u32 trace_nr)
|
||||
{
|
||||
struct mm_struct *mm = current->mm;
|
||||
struct stack_map_vma_lock lock = { .mm = mm };
|
||||
struct {
|
||||
struct file *file;
|
||||
const unsigned char *build_id;
|
||||
unsigned long vm_start;
|
||||
unsigned long vm_end;
|
||||
unsigned long vm_pgoff;
|
||||
} cache = {};
|
||||
unsigned long vm_pgoff, vm_start, vm_end;
|
||||
struct vm_area_struct *vma;
|
||||
struct file *file;
|
||||
u64 offset;
|
||||
u64 ip;
|
||||
|
||||
for (u32 i = 0; i < trace_nr; i++) {
|
||||
ip = READ_ONCE(id_offs[i].ip);
|
||||
|
||||
/*
|
||||
* Range cache fast path: if ip falls within the previously
|
||||
* resolved VMA range, reuse the cache build_id without
|
||||
* re-acquiring the VMA lock.
|
||||
*/
|
||||
if (cache.build_id && ip >= cache.vm_start && ip < cache.vm_end) {
|
||||
offset = stack_map_build_id_offset(cache.vm_pgoff, cache.vm_start, ip);
|
||||
stack_map_build_id_set_valid(&id_offs[i], offset, cache.build_id);
|
||||
continue;
|
||||
}
|
||||
|
||||
vma = stack_map_lock_vma(&lock, ip);
|
||||
if (!vma) {
|
||||
stack_map_build_id_set_ip(&id_offs[i]);
|
||||
continue;
|
||||
}
|
||||
if (vma_is_anonymous(vma) || !vma->vm_file) {
|
||||
stack_map_build_id_set_ip(&id_offs[i]);
|
||||
stack_map_unlock_vma(&lock);
|
||||
continue;
|
||||
}
|
||||
|
||||
file = vma->vm_file;
|
||||
vm_pgoff = vma->vm_pgoff;
|
||||
vm_start = vma->vm_start;
|
||||
vm_end = vma->vm_end;
|
||||
offset = stack_map_build_id_offset(vm_pgoff, vm_start, ip);
|
||||
|
||||
/*
|
||||
* Same backing file as previous (e.g. different VMAs
|
||||
* of the same ELF binary). Reuse the cache build_id.
|
||||
*/
|
||||
if (file == cache.file) {
|
||||
stack_map_unlock_vma(&lock);
|
||||
stack_map_build_id_set_valid(&id_offs[i], offset, cache.build_id);
|
||||
cache.vm_start = vm_start;
|
||||
cache.vm_end = vm_end;
|
||||
cache.vm_pgoff = vm_pgoff;
|
||||
continue;
|
||||
}
|
||||
|
||||
file = get_file(file);
|
||||
stack_map_unlock_vma(&lock);
|
||||
|
||||
/* build_id_parse_file() may block on filesystem reads */
|
||||
if (build_id_parse_file(file, id_offs[i].build_id, NULL)) {
|
||||
stack_map_build_id_set_ip(&id_offs[i]);
|
||||
fput(file);
|
||||
continue;
|
||||
}
|
||||
|
||||
stack_map_build_id_set_valid(&id_offs[i], offset, id_offs[i].build_id);
|
||||
if (cache.file)
|
||||
fput(cache.file);
|
||||
cache.file = file;
|
||||
cache.build_id = id_offs[i].build_id;
|
||||
cache.vm_start = vm_start;
|
||||
cache.vm_end = vm_end;
|
||||
cache.vm_pgoff = vm_pgoff;
|
||||
}
|
||||
|
||||
if (cache.file)
|
||||
fput(cache.file);
|
||||
}
|
||||
|
||||
/*
|
||||
* Expects all id_offs[i].ip values to be set to correct initial IPs.
|
||||
* They will be subsequently:
|
||||
|
|
@ -165,44 +340,50 @@ static int fetch_build_id(struct vm_area_struct *vma, unsigned char *build_id, b
|
|||
static void stack_map_get_build_id_offset(struct bpf_stack_build_id *id_offs,
|
||||
u32 trace_nr, bool user, bool may_fault)
|
||||
{
|
||||
int i;
|
||||
struct mmap_unlock_irq_work *work = NULL;
|
||||
bool irq_work_busy = bpf_mmap_unlock_get_irq_work(&work);
|
||||
bool has_user_ctx = user && current && current->mm;
|
||||
struct vm_area_struct *vma, *prev_vma = NULL;
|
||||
const char *prev_build_id;
|
||||
const unsigned char *prev_build_id = NULL;
|
||||
int i;
|
||||
|
||||
if (may_fault && has_user_ctx) {
|
||||
stack_map_get_build_id_offset_sleepable(id_offs, trace_nr);
|
||||
return;
|
||||
}
|
||||
|
||||
/* If the irq_work is in use, fall back to report ips. Same
|
||||
* fallback is used for kernel stack (!user) on a stackmap with
|
||||
* build_id.
|
||||
*/
|
||||
if (!user || !current || !current->mm || irq_work_busy ||
|
||||
!mmap_read_trylock(current->mm)) {
|
||||
if (!has_user_ctx || irq_work_busy || !mmap_read_trylock(current->mm)) {
|
||||
/* cannot access current->mm, fall back to ips */
|
||||
for (i = 0; i < trace_nr; i++) {
|
||||
id_offs[i].status = BPF_STACK_BUILD_ID_IP;
|
||||
memset(id_offs[i].build_id, 0, BUILD_ID_SIZE_MAX);
|
||||
}
|
||||
for (i = 0; i < trace_nr; i++)
|
||||
stack_map_build_id_set_ip(&id_offs[i]);
|
||||
return;
|
||||
}
|
||||
|
||||
for (i = 0; i < trace_nr; i++) {
|
||||
u64 ip = READ_ONCE(id_offs[i].ip);
|
||||
u64 offset;
|
||||
|
||||
if (range_in_vma(prev_vma, ip, ip)) {
|
||||
if (prev_build_id && range_in_vma(prev_vma, ip, ip)) {
|
||||
vma = prev_vma;
|
||||
memcpy(id_offs[i].build_id, prev_build_id, BUILD_ID_SIZE_MAX);
|
||||
goto build_id_valid;
|
||||
}
|
||||
vma = find_vma(current->mm, ip);
|
||||
if (!vma || fetch_build_id(vma, id_offs[i].build_id, may_fault)) {
|
||||
/* per entry fall back to ips */
|
||||
id_offs[i].status = BPF_STACK_BUILD_ID_IP;
|
||||
memset(id_offs[i].build_id, 0, BUILD_ID_SIZE_MAX);
|
||||
offset = stack_map_build_id_offset(vma->vm_pgoff, vma->vm_start, ip);
|
||||
stack_map_build_id_set_valid(&id_offs[i], offset, prev_build_id);
|
||||
continue;
|
||||
}
|
||||
build_id_valid:
|
||||
id_offs[i].offset = (vma->vm_pgoff << PAGE_SHIFT) + ip - vma->vm_start;
|
||||
id_offs[i].status = BPF_STACK_BUILD_ID_VALID;
|
||||
vma = find_vma(current->mm, ip);
|
||||
if (!vma || vma_is_anonymous(vma) ||
|
||||
fetch_build_id(vma, id_offs[i].build_id, may_fault)) {
|
||||
/* per entry fall back to ips */
|
||||
stack_map_build_id_set_ip(&id_offs[i]);
|
||||
prev_vma = vma;
|
||||
prev_build_id = NULL;
|
||||
continue;
|
||||
}
|
||||
offset = stack_map_build_id_offset(vma->vm_pgoff, vma->vm_start, ip);
|
||||
stack_map_build_id_set_valid(&id_offs[i], offset, id_offs[i].build_id);
|
||||
prev_vma = vma;
|
||||
prev_build_id = id_offs[i].build_id;
|
||||
}
|
||||
|
|
|
|||
Loading…
Reference in New Issue
Block a user