From d22c3e0088e85be8131f7a9283f759cdbb20726d Mon Sep 17 00:00:00 2001 From: Sven Schnelle Date: Wed, 9 Sep 2026 11:29:53 +0200 Subject: [PATCH 1/3] selftests/ftrace: Fix unique symbol check in kprobe_non_uniq_symbol.tc The current regex also matches symbols in modules, which makes the test fail on s390 where name_show is present only once in the kernel, but also multiple times in modules: 000001b1401cdc20 t name_show 000001b0c05e6c40 t name_show [mdev] 000001b0c0495f30 t name_show [i2c_core] Fix this by changing the regular expression to only match the function name. Link: https://lore.kernel.org/all/20260909092954.2200558-1-svens@linux.ibm.com/ Fixes: 03b80ff8023a ("selftests/ftrace: Add new test case which checks non unique symbol") Signed-off-by: Sven Schnelle Reviewed-by: Steven Rostedt Signed-off-by: Masami Hiramatsu (Google) --- .../selftests/ftrace/test.d/kprobe/kprobe_non_uniq_symbol.tc | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_non_uniq_symbol.tc b/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_non_uniq_symbol.tc index bc9514428dba..07b1177c1634 100644 --- a/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_non_uniq_symbol.tc +++ b/tools/testing/selftests/ftrace/test.d/kprobe/kprobe_non_uniq_symbol.tc @@ -6,7 +6,7 @@ SYMBOL='name_show' # We skip this test on kernel where SYMBOL is unique or does not exist. -if [ "$(grep -c -E "[[:alnum:]]+ t ${SYMBOL}" /proc/kallsyms)" -le '1' ]; then +if [ "$(grep -c -E "[[:alnum:]]+ t ${SYMBOL}$" /proc/kallsyms)" -le '1' ]; then exit_unsupported fi From 1d653a183973f5283a3db5a38cd5e195eb152244 Mon Sep 17 00:00:00 2001 From: David Carlier Date: Thu, 17 Sep 2026 22:24:07 +0100 Subject: [PATCH 2/3] fprobe: Terminate the fgraph_data list when the reservation is not filled fprobe_fgraph_entry() reserves shadow stack space for every fprobe with an exit handler, but only fills it for those whose entry handler returns 0. fgraph_reserve_data() does not clear the area, so fprobe_return() parses the unused tail as headers left over from an earlier call, and an exit handler can run twice or despite its entry handler asking to skip it. Write a zero word after the last entry to terminate the walk. A zeroed slot does not decode to a NULL fprobe on the arches that encode the header into one unsigned long, since arch_decode_fprobe_header_fp() ORs in FPROBE_HEADER_MSB_PATTERN, so make read_fprobe_header() return NULL for a zeroed slot. Link: https://lore.kernel.org/all/20260917212407.384468-1-devnexen@gmail.com/ Fixes: e0a384434ae1 ("tracing: fprobe: do not zero out unused fgraph_data") Cc: stable@vger.kernel.org Suggested-by: Masami Hiramatsu (Google) Signed-off-by: David Carlier Signed-off-by: Masami Hiramatsu (Google) --- kernel/trace/fprobe.c | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/kernel/trace/fprobe.c b/kernel/trace/fprobe.c index 1e9b00997ff2..9f2d98181779 100644 --- a/kernel/trace/fprobe.c +++ b/kernel/trace/fprobe.c @@ -171,6 +171,11 @@ static inline bool write_fprobe_header(unsigned long *stack, static inline void read_fprobe_header(unsigned long *stack, struct fprobe **fp, unsigned int *size_words) { + if (!*stack) { + *fp = NULL; + *size_words = 0; + return; + } *fp = arch_decode_fprobe_header_fp(*stack); *size_words = arch_decode_fprobe_header_size(*stack); } @@ -203,6 +208,12 @@ static inline void read_fprobe_header(unsigned long *stack, { struct __fprobe_header *fph = (struct __fprobe_header *)stack; + if (!*stack) { + *fp = NULL; + *size_words = 0; + return; + } + *fp = fph->fp; *size_words = fph->size_words; } @@ -635,6 +646,10 @@ static int fprobe_fgraph_entry(struct ftrace_graph_ent *trace, struct fgraph_ops } } + /* Terminate the list, fgraph_reserve_data() does not clear it. */ + if (used && used < reserved_words) + fgraph_data[used] = 0; + /* If any exit_handler is set, data must be used. */ return used != 0; } From 5bfa9f1a9dcb6ecb607adbc1c0226605c972935b Mon Sep 17 00:00:00 2001 From: Andrea Parri Date: Thu, 24 Sep 2026 11:21:39 +0200 Subject: [PATCH 3/3] kprobes: Fix permanent hang when flushing the kprobe optimizer Writing 0 to /proc/sys/debug/kprobes-optimization while a kprobe is jump-optimized never returns. The writer sleeps in D state forever with kprobe_sysctl_mutex held, so any later read or write of that sysctl hangs as well. For example, with vfs_read+9 as an optimizable address in this build: # cd /sys/kernel/tracing # echo 'p:myprobe vfs_read+9' >> kprobe_events # echo 1 > events/kprobes/myprobe/enable # # wait until /sys/kernel/debug/kprobes/list shows [OPTIMIZED] # echo 0 > /proc/sys/debug/kprobes-optimization INFO: task sh:246 blocked for more than 10 seconds. Call Trace: __schedule+0x1176/0x4f70 schedule+0xdc/0x2c0 schedule_timeout+0x17b/0x260 wait_for_completion+0x173/0x3c0 wait_for_kprobe_optimizer_locked+0xbc/0x130 proc_kprobes_optimization_handler+0x156/0x1b0 proc_sys_call_handler+0x324/0x490 vfs_write+0x52d/0xfe0 ksys_write+0xff/0x200 do_syscall_64+0x106/0x630 entry_SYSCALL_64_after_hwframe+0x77/0x7f ... INFO: task cat:265 is blocked on a mutex likely owned by task sh:246. wait_for_kprobe_optimizer_locked() reinitializes optimizer_completion, asks the optimizer thread to flush and sleeps in wait_for_completion(). The thread drains the (un)optimizing lists, but calls complete() only if completion_done() is true, i.e. if the completion is already done, which never happens while someone waits. disarm_all_kprobes() and kprobe_trace_self_tests_init() wait the same way. Calling complete() unconditionally would not be enough: the waiter drops kprobe_mutex while it sleeps, and nothing else serializes the sysctl handler against the debugfs "enabled" file. A second flusher that still finds the lists non-empty, e.g. because a disabled probe is queued for unoptimizing, reinitializes the completion under the first: sysctl write debugfs "enabled" write unoptimize_all_kprobes() wait_for_kprobe_optimizer_locked() init_completion(c) mutex_unlock(&kprobe_mutex) wait_for_completion(c) disarm_all_kprobes() wait_for_kprobe_optimizer_locked() init_completion(c) // c->wait is reset, the first // waiter is off the queue mutex_unlock(&kprobe_mutex) wait_for_completion(c) kprobe_optimizer() complete(c) // wakes the debugfs writer only where c is &optimizer_completion. Lining up the two writes during an optimizer pass loses the sysctl writer this way. Replace the completion with a counter of optimizer passes, bumped at the end of each pass and signalled with wake_up_var_locked(), both under kprobe_mutex. A flusher samples the count and waits with wait_var_event_mutex(), which drops kprobe_mutex only while sleeping, so a new count means a whole pass ran in the meantime. Nothing is reinitialized, so several flushers can sleep in the wait at once. Link: https://lore.kernel.org/all/20260924092142.199198-1-parri.andrea@gmail.com/ Fixes: 73c12f209462 ("kprobes: Use dedicated kthread for kprobe optimizer") Cc: stable@vger.kernel.org Assisted-by: LLM Signed-off-by: Andrea Parri Signed-off-by: Masami Hiramatsu (Google) --- kernel/kprobes.c | 22 ++++++++++++++-------- 1 file changed, 14 insertions(+), 8 deletions(-) diff --git a/kernel/kprobes.c b/kernel/kprobes.c index 6337da5cab9e..4edd8ca5c657 100644 --- a/kernel/kprobes.c +++ b/kernel/kprobes.c @@ -42,6 +42,7 @@ #include #include #include +#include #include #include @@ -526,7 +527,8 @@ enum { OPTIMIZER_ST_FLUSHING = 2, }; -static DECLARE_COMPLETION(optimizer_completion); +/* Bumped at the end of each kprobe_optimizer() pass, under 'kprobe_mutex' */ +static unsigned long optimizer_passes; #define OPTIMIZE_DELAY 5 @@ -654,9 +656,9 @@ static void kprobe_optimizer(void) do_free_cleaned_kprobes(); } - /* Step 5: Kick optimizer again if needed. But if there is a flush requested, */ - if (completion_done(&optimizer_completion)) - complete(&optimizer_completion); + /* Step 5: Wake up flushers, and kick optimizer again if needed. */ + optimizer_passes++; + wake_up_var_locked(&optimizer_passes, &kprobe_mutex); if (!list_empty(&optimizing_list) || !list_empty(&unoptimizing_list)) kick_kprobe_optimizer(); /*normal kick*/ @@ -708,7 +710,8 @@ static void wait_for_kprobe_optimizer_locked(void) lockdep_assert_held(&kprobe_mutex); while (!list_empty(&optimizing_list) || !list_empty(&unoptimizing_list)) { - init_completion(&optimizer_completion); + unsigned long passes = optimizer_passes; + /* * Set state to OPTIMIZER_ST_FLUSHING and wake up the thread if it's * idle. If it's already kicked, it will see the state change. @@ -717,9 +720,12 @@ static void wait_for_kprobe_optimizer_locked(void) OPTIMIZER_ST_FLUSHING) != OPTIMIZER_ST_FLUSHING) wake_up(&kprobe_optimizer_wait); - mutex_unlock(&kprobe_mutex); - wait_for_completion(&optimizer_completion); - mutex_lock(&kprobe_mutex); + /* + * kprobe_optimizer() holds 'kprobe_mutex' for a whole pass, which + * this drops while sleeping, so a new count means a full pass ran. + */ + wait_var_event_mutex(&optimizer_passes, + optimizer_passes != passes, &kprobe_mutex); } }