summaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
authorMasami Hiramatsu (Google) <mhiramat@kernel.org>2026-09-29 09:19:12 +0900
committerMasami Hiramatsu (Google) <mhiramat@kernel.org>2026-09-30 08:41:31 +0900
commite0a6249190402a28f1e2925a81acf573157d55ef (patch)
tree73efd3a4d505db6ac61b7d5035b056c2e7313f5e
parent72d3fcf802c45d00b300f25b848a93c3a2bd7c7e (diff)
downloadlinux-stable-e0a6249190402a28f1e2925a81acf573157d55ef.tar.gz
linux-stable-e0a6249190402a28f1e2925a81acf573157d55ef.zip
fprobe: Use guard(rcu_sched_notrace) and check rcu_is_watching()
unregister_fprobe() and unregister_fprobe_async() (used by BPF kprobe-multi) rely on standard RCU grace periods (synchronize_rcu() and call_rcu()) to wait until in-flight fprobe handlers complete before freeing the fprobe. However, if an fprobe handler executes while RCU is not watching (such as in the idle loop or nohz_full extended quiescent states), standard RCU does not track preemption-disabled sections. Consequently, synchronize_rcu() does not wait for those executions, which can lead to a use-after-free if the fprobe is freed immediately after unregistration. Ensure handlers exit early when !rcu_is_watching(). Furthermore, fprobe_fgraph_entry() and fprobe_ftrace_entry() previously used guard(rcu)() and rcu_read_lock(), which invoke lockdep on every hit under CONFIG_PROVE_LOCKING. This adds overhead and can cause lockdep recursion if probed functions interact with lockdep. Since rhltable_lookup() and rhl_for_each_entry_rcu() use rcu_dereference_all_check() (which checks rcu_read_lock_any_held()), holding preemption disabled via rcu_read_lock_sched_notrace() is fully valid and sufficient so long as rcu_is_watching() is true. Define and use guard(rcu_sched_notrace)() across fprobe_ftrace_entry(), fprobe_fgraph_entry(), and fprobe_return(). This eliminates fast-path rcu_read_lock() and lockdep overhead while guaranteeing safe grace period synchronization. Link: https://lore.kernel.org/all/179064115227.394389.16910234241400391996.stgit@devnote2/ Reported-by: Sashiko <sashiko-bot@kernel.org> Closes: https://sashiko.dev/#/bug/linux-e46bcd68-4a56-4f19-a255-e3772980e5e3 Fixes: 657b594b2084 ("fprobe: Fix unregister_fprobe() to wait for RCU grace period") Cc: stable@vger.kernel.org Assisted-by: LLM Signed-off-by: Masami Hiramatsu (Google) <mhiramat@kernel.org> Reviewed-by: Paul E. McKenney <paulmck@kernel.org>
-rw-r--r--kernel/trace/fprobe.c26
1 files changed, 16 insertions, 10 deletions
diff --git a/kernel/trace/fprobe.c b/kernel/trace/fprobe.c
index 9f2d98181..da286619c 100644
--- a/kernel/trace/fprobe.c
+++ b/kernel/trace/fprobe.c
@@ -47,6 +47,10 @@ static struct rhltable fprobe_ip_table;
static DEFINE_MUTEX(fprobe_mutex);
static struct fgraph_ops fprobe_graph_ops;
+DEFINE_LOCK_GUARD_0(rcu_sched_notrace,
+ rcu_read_lock_sched_notrace(),
+ rcu_read_unlock_sched_notrace())
+
static u32 fprobe_node_hashfn(const void *data, u32 len, u32 seed)
{
return hash_ptr(*(unsigned long **)data, 32);
@@ -329,16 +333,14 @@ static void fprobe_ftrace_entry(unsigned long ip, unsigned long parent_ip,
struct fprobe *fp;
int bit;
+ if (!rcu_is_watching())
+ return;
+
bit = ftrace_test_recursion_trylock(ip, parent_ip);
if (bit < 0)
return;
- /*
- * ftrace_test_recursion_trylock() disables preemption, but
- * rhltable_lookup() checks whether rcu_read_lcok is held.
- * So we take rcu_read_lock() here.
- */
- rcu_read_lock();
+ guard(rcu_sched_notrace)();
head = rhltable_lookup(&fprobe_ip_table, &ip, fprobe_rht_params);
rhl_for_each_entry_rcu(node, pos, head, hlist) {
@@ -353,7 +355,6 @@ static void fprobe_ftrace_entry(unsigned long ip, unsigned long parent_ip,
else
__fprobe_handler(ip, parent_ip, fp, fregs, NULL);
}
- rcu_read_unlock();
ftrace_test_recursion_unlock(bit);
}
NOKPROBE_SYMBOL(fprobe_ftrace_entry);
@@ -567,10 +568,13 @@ static int fprobe_fgraph_entry(struct ftrace_graph_ent *trace, struct fgraph_ops
struct fprobe *fp;
int used, ret;
+ if (!rcu_is_watching())
+ return 0;
+
if (WARN_ON_ONCE(!fregs))
return 0;
- guard(rcu)();
+ guard(rcu_sched_notrace)();
head = rhltable_lookup(&fprobe_ip_table, &func, fprobe_rht_params);
reserved_words = 0;
rhl_for_each_entry_rcu(node, pos, head, hlist) {
@@ -665,13 +669,16 @@ static void fprobe_return(struct ftrace_graph_ret *trace,
int size, curr;
int size_words;
+ if (!rcu_is_watching())
+ return;
+
fgraph_data = (unsigned long *)fgraph_retrieve_data(gops->idx, &size);
if (WARN_ON_ONCE(!fgraph_data))
return;
size_words = SIZE_IN_LONG(size);
ret_ip = ftrace_regs_get_instruction_pointer(fregs);
- preempt_disable_notrace();
+ guard(rcu_sched_notrace)();
curr = 0;
while (size_words > curr) {
@@ -687,7 +694,6 @@ static void fprobe_return(struct ftrace_graph_ret *trace,
}
curr += size;
}
- preempt_enable_notrace();
}
NOKPROBE_SYMBOL(fprobe_return);