diff options
| author | Alexei Starovoitov <ast@kernel.org> | 2026-10-01 14:52:55 +0000 |
|---|---|---|
| committer | Kumar Kartikeya Dwivedi <memxor@gmail.com> | 2026-10-01 18:39:27 +0200 |
| commit | 33a154a96e71a34a1bcca9f40da343dbbf7b38b4 (patch) | |
| tree | 543be8029ab3d8a3a63994ae23385a053ed30029 /include/net/busy_poll.h | |
| download | linux-stable-33a154a96e71a34a1bcca9f40da343dbbf7b38b4.tar.gz linux-stable-33a154a96e71a34a1bcca9f40da343dbbf7b38b4.zip | |
selftests/bpf: Test packet range of pointers sharing an idgrafted
Add tests where two packet pointers share an id and tightening one
pointer's umax from its var_off would put it less than their constant
distance from the other's umax: with an index & 0x38 capped at 50, the
base pointer keeps umax 50, so the pointer 8 bytes further on must keep
umax 58, even though its known bits allow at most 56.
These refused a valid program or accepted an out-of-bounds access before
the fix:
- check the advanced copy, load through the base: valid, was refused;
- check the base, load the byte at base + 1 through a copy advanced by
8: was accepted;
- check base + 4, load 4 bytes at base + 2 through base + 8: reads two
bytes past the checked range, was accepted;
- the same as the second with data_meta pointers checked against data:
was accepted.
These pass with and without the fix and cover nearby paths:
- subtract an unknown scalar from a checked pointer and load below it
(the range is kept across a new id);
- reach a load through two paths whose checks cover 8 and 7 bytes after
the loaded pointer; the second path must not be pruned by the first;
- spill a copy of a pointer, check the pointer, fill the copy and load
one byte past the checked range: the load is refused, and the copy
has the range of the check.
Signed-off-by: Alexei Starovoitov <ast@kernel.org>
Link: https://lore.kernel.org/bpf/20261001145255.855630-2-alexei.starovoitov@gmail.com
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
Diffstat (limited to 'include/net/busy_poll.h')
| -rw-r--r-- | include/net/busy_poll.h | 188 |
1 files changed, 188 insertions, 0 deletions
diff --git a/include/net/busy_poll.h b/include/net/busy_poll.h new file mode 100644 index 000000000..6e172d0f6 --- /dev/null +++ b/include/net/busy_poll.h @@ -0,0 +1,188 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * net busy poll support + * Copyright(c) 2013 Intel Corporation. + * + * Author: Eliezer Tamir + * + * Contact Information: + * e1000-devel Mailing List <e1000-devel@lists.sourceforge.net> + */ + +#ifndef _LINUX_NET_BUSY_POLL_H +#define _LINUX_NET_BUSY_POLL_H + +#include <linux/netdevice.h> +#include <linux/sched/clock.h> +#include <linux/sched/signal.h> +#include <net/ip.h> +#include <net/xdp.h> + +/* 0 - Reserved to indicate value not set + * 1..NR_CPUS - Reserved for sender_cpu + * NR_CPUS+1..~0 - Region available for NAPI IDs + */ +#define MIN_NAPI_ID ((unsigned int)(NR_CPUS + 1)) + +static inline bool napi_id_valid(unsigned int napi_id) +{ + return napi_id >= MIN_NAPI_ID; +} + +#define BUSY_POLL_BUDGET 8 + +#ifdef CONFIG_NET_RX_BUSY_POLL + +struct napi_struct; +extern unsigned int sysctl_net_busy_read __read_mostly; +extern unsigned int sysctl_net_busy_poll __read_mostly; + +static inline bool net_busy_loop_on(void) +{ + return READ_ONCE(sysctl_net_busy_poll); +} + +static inline bool sk_can_busy_loop(const struct sock *sk) +{ + return READ_ONCE(sk->sk_ll_usec) && !signal_pending(current); +} + +bool sk_busy_loop_end(void *p, unsigned long start_time); + +void napi_busy_loop(unsigned int napi_id, + bool (*loop_end)(void *, unsigned long), + void *loop_end_arg, bool prefer_busy_poll, u16 budget); + +void napi_busy_loop_rcu(unsigned int napi_id, + bool (*loop_end)(void *, unsigned long), + void *loop_end_arg, bool prefer_busy_poll, u16 budget); + +void napi_suspend_irqs(unsigned int napi_id); +void napi_resume_irqs(unsigned int napi_id); + +#else /* CONFIG_NET_RX_BUSY_POLL */ +static inline unsigned long net_busy_loop_on(void) +{ + return 0; +} + +static inline bool sk_can_busy_loop(struct sock *sk) +{ + return false; +} + +#endif /* CONFIG_NET_RX_BUSY_POLL */ + +static inline unsigned long busy_loop_current_time(void) +{ +#ifdef CONFIG_NET_RX_BUSY_POLL + return (unsigned long)(ktime_get_ns() >> 10); +#else + return 0; +#endif +} + +/* in poll/select we use the global sysctl_net_ll_poll value */ +static inline bool busy_loop_timeout(unsigned long start_time) +{ +#ifdef CONFIG_NET_RX_BUSY_POLL + unsigned long bp_usec = READ_ONCE(sysctl_net_busy_poll); + + if (bp_usec) { + unsigned long end_time = start_time + bp_usec; + unsigned long now = busy_loop_current_time(); + + return time_after(now, end_time); + } +#endif + return true; +} + +static inline bool sk_busy_loop_timeout(struct sock *sk, + unsigned long start_time) +{ +#ifdef CONFIG_NET_RX_BUSY_POLL + unsigned long bp_usec = READ_ONCE(sk->sk_ll_usec); + + if (bp_usec) { + unsigned long end_time = start_time + bp_usec; + unsigned long now = busy_loop_current_time(); + + return time_after(now, end_time); + } +#endif + return true; +} + +static inline void sk_busy_loop(struct sock *sk, int nonblock) +{ +#ifdef CONFIG_NET_RX_BUSY_POLL + unsigned int napi_id = READ_ONCE(sk->sk_napi_id); + + if (napi_id_valid(napi_id)) + napi_busy_loop(napi_id, nonblock ? NULL : sk_busy_loop_end, sk, + READ_ONCE(sk->sk_prefer_busy_poll), + READ_ONCE(sk->sk_busy_poll_budget) ?: BUSY_POLL_BUDGET); +#endif +} + +/* used in the NIC receive handler to mark the skb */ +static inline void __skb_mark_napi_id(struct sk_buff *skb, + const struct gro_node *gro) +{ +#ifdef CONFIG_NET_RX_BUSY_POLL + /* If the skb was already marked with a valid NAPI ID, avoid overwriting + * it. + */ + if (!napi_id_valid(skb->napi_id)) + skb->napi_id = gro->cached_napi_id; +#endif +} + +static inline void skb_mark_napi_id(struct sk_buff *skb, + const struct napi_struct *napi) +{ + __skb_mark_napi_id(skb, &napi->gro); +} + +/* used in the protocol handler to propagate the napi_id to the socket */ +static inline void sk_mark_napi_id(struct sock *sk, const struct sk_buff *skb) +{ +#ifdef CONFIG_NET_RX_BUSY_POLL + if (unlikely(READ_ONCE(sk->sk_napi_id) != skb->napi_id)) + WRITE_ONCE(sk->sk_napi_id, skb->napi_id); +#endif + sk_rx_queue_update(sk, skb); +} + +/* Variant of sk_mark_napi_id() for passive flow setup, + * as sk->sk_napi_id and sk->sk_rx_queue_mapping content + * needs to be set. + */ +static inline void sk_mark_napi_id_set(struct sock *sk, + const struct sk_buff *skb) +{ +#ifdef CONFIG_NET_RX_BUSY_POLL + WRITE_ONCE(sk->sk_napi_id, skb->napi_id); +#endif + sk_rx_queue_set(sk, skb); +} + +static inline void __sk_mark_napi_id_once(struct sock *sk, unsigned int napi_id) +{ +#ifdef CONFIG_NET_RX_BUSY_POLL + if (!READ_ONCE(sk->sk_napi_id)) + WRITE_ONCE(sk->sk_napi_id, napi_id); +#endif +} + +/* variant used for unconnected sockets */ +static inline void sk_mark_napi_id_once(struct sock *sk, + const struct sk_buff *skb) +{ +#ifdef CONFIG_NET_RX_BUSY_POLL + __sk_mark_napi_id_once(sk, skb->napi_id); +#endif +} + +#endif /* _LINUX_NET_BUSY_POLL_H */ |
