summaryrefslogtreecommitdiffstats
path: root/include/net/inet_common.h
diff options
context:
space:
mode:
authorJinke Han <jinkehan@didiglobal.com>2026-09-08 15:37:42 +0800
committerIngo Molnar <mingo@kernel.org>2026-09-17 09:48:42 +0200
commita5f7a5bb3b7f28ba7e4fa246775b29a0e5537255 (patch)
tree73fef824f9a846cf935869d0dbc5070ef6e2f741 /include/net/inet_common.h
downloadlinux-stable-a5f7a5bb3b7f28ba7e4fa246775b29a0e5537255.tar.gz
linux-stable-a5f7a5bb3b7f28ba7e4fa246775b29a0e5537255.zip
x86/kprobes: Fix crash when probing CS CALL instructionsgrafted
When using eBPF to probe CS CALL instructions within a function, a crash can be triggered. The eBPF tool probes offset 257 of the __hrtimer_run_queues() function: <__hrtimer_run_queues+249>: nopl 0x0(%rax,%rax,1) <__hrtimer_run_queues+254>: mov %r14,%rdi <__hrtimer_run_queues+257>: cs call <__x86_indirect_thunk_r12> <__hrtimer_run_queues+263>: mov %eax,%r12d <__hrtimer_run_queues+266>: xchg %ax,%ax <__hrtimer_run_queues+268>: mov %r13,%rdi Which triggers this crash: BUG: unable to handle page fault for address: 00000000000f41c9 #PF: supervisor write access in kernel mode #PF: error_code(0x0002) - not-present page PGD 0 P4D 0 Oops: 0002 [#1] SMP NOPTI CPU: 1 PID: 0 Comm: swapper/1 Kdump: loaded Tainted: P RIP: 0010:__hrtimer_run_queues+0x106/0x230 Note that __hrtimer_run_queues+0x106 is __hrtimer_run_queues+262, which is at the 6th byte of the above CS CALL instruction. Since the CS CALL instruction occupies 6 bytes, the exception occurred in the middle of that call instruction. The root cause is that when using eBPF tools to probe in the middle of a function, a kprobe with INT3 is used as the underlying implementation. During single-step emulation of the original CALL instruction, int3_emulate_call() assumes that the probed CALL instruction is 5 bytes long. However, the actual CS-prefixed CALL instruction occupies 6 bytes, so it constructs an incorrect exception return address. When the CPU returns from the kprobe handler, the next instruction to be executed is at the address of the last byte of that CS CALL instruction. Coincidentally, starting from that address, the CPU fetches and decodes a completely different instruction, which ultimately triggers a kernel crash. Fix the issue by using the actual instruction length obtained from the instruction decoder when constructing the exception return address, rather than relying on the hardcoded CALL_INSN_SIZE macro. [ mingo: Refined the changelog ] Fixes: 6256e668b7af ("x86/kprobes: Use int3 instead of debug trap for single-step") Suggested-by: Masami Hiramatsu (Google) <mhiramat@kernel.org> Signed-off-by: Jinke Han <jinkehan@didiglobal.com> Signed-off-by: Ingo Molnar <mingo@kernel.org> Reviewed-by: Masami Hiramatsu (Google) <mhiramat@kernel.org> Acked-by: Yafang Shao <laoar.shao@gmail.com> Acked-by: Borislav Petkov <bp@alien8.de> Cc: Peter Zijlstra <peterz@infradead.org> Link: https://patch.msgid.link/20260908073742.GA10517@didi-ThinkCentre-M920t-N000
Diffstat (limited to 'include/net/inet_common.h')
-rw-r--r--include/net/inet_common.h82
1 files changed, 82 insertions, 0 deletions
diff --git a/include/net/inet_common.h b/include/net/inet_common.h
new file mode 100644
index 000000000..3d747896b
--- /dev/null
+++ b/include/net/inet_common.h
@@ -0,0 +1,82 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef _INET_COMMON_H
+#define _INET_COMMON_H
+
+#include <linux/indirect_call_wrapper.h>
+#include <linux/net.h>
+#include <linux/netdev_features.h>
+#include <linux/types.h>
+#include <net/sock.h>
+
+extern const struct proto_ops inet_stream_ops;
+extern const struct proto_ops inet_dgram_ops;
+
+/*
+ * INET4 prototypes used by INET6
+ */
+
+struct msghdr;
+struct net;
+struct page;
+struct sock;
+struct socket;
+
+int inet_release(struct socket *sock);
+int inet_stream_connect(struct socket *sock, struct sockaddr_unsized *uaddr,
+ int addr_len, int flags);
+int __inet_stream_connect(struct socket *sock, struct sockaddr_unsized *uaddr,
+ int addr_len, int flags, int is_sendmsg);
+int inet_dgram_connect(struct socket *sock, struct sockaddr_unsized *uaddr,
+ int addr_len, int flags);
+int inet_accept(struct socket *sock, struct socket *newsock,
+ struct proto_accept_arg *arg);
+void __inet_accept(struct socket *sock, struct socket *newsock,
+ struct sock *newsk);
+int inet_send_prepare(struct sock *sk);
+int inet_sendmsg(struct socket *sock, struct msghdr *msg, size_t size);
+void inet_splice_eof(struct socket *sock);
+int inet_recvmsg(struct socket *sock, struct msghdr *msg, size_t size,
+ int flags);
+int inet_shutdown(struct socket *sock, int how);
+int inet_listen(struct socket *sock, int backlog);
+int __inet_listen_sk(struct sock *sk, int backlog);
+void inet_sock_destruct(struct sock *sk);
+int inet_bind(struct socket *sock, struct sockaddr_unsized *uaddr, int addr_len);
+int inet_bind_sk(struct sock *sk, struct sockaddr_unsized *uaddr, int addr_len);
+/* Don't allocate port at this moment, defer to connect. */
+#define BIND_FORCE_ADDRESS_NO_PORT (1 << 0)
+/* Grab and release socket lock. */
+#define BIND_WITH_LOCK (1 << 1)
+/* Called from BPF program. */
+#define BIND_FROM_BPF (1 << 2)
+/* Skip CAP_NET_BIND_SERVICE check. */
+#define BIND_NO_CAP_NET_BIND_SERVICE (1 << 3)
+int __inet_bind(struct sock *sk, struct sockaddr_unsized *uaddr, int addr_len,
+ u32 flags);
+int inet_getname(struct socket *sock, struct sockaddr *uaddr,
+ int peer);
+int inet_ioctl(struct socket *sock, unsigned int cmd, unsigned long arg);
+int inet_ctl_sock_create(struct sock **sk, unsigned short family,
+ unsigned short type, unsigned char protocol,
+ struct net *net);
+int inet_recv_error(struct sock *sk, struct msghdr *msg, int len);
+
+struct sk_buff *inet_gro_receive(struct list_head *head, struct sk_buff *skb);
+int inet_gro_complete(struct sk_buff *skb, int nhoff);
+struct sk_buff *inet_gso_segment(struct sk_buff *skb,
+ netdev_features_t features);
+
+static inline void inet_ctl_sock_destroy(struct sock *sk)
+{
+ if (sk)
+ sock_release(sk->sk_socket);
+}
+
+#define indirect_call_gro_receive(f2, f1, cb, head, skb) \
+({ \
+ unlikely(gro_recursion_inc_test(skb)) ? \
+ NAPI_GRO_CB(skb)->flush |= 1, NULL : \
+ INDIRECT_CALL_2(cb, f2, f1, head, skb); \
+})
+
+#endif