diff options
| author | Linus Torvalds <torvalds@linux-foundation.org> | 2026-10-02 12:17:24 -0700 |
|---|---|---|
| committer | Linus Torvalds <torvalds@linux-foundation.org> | 2026-10-02 12:17:24 -0700 |
| commit | 3f1fe48a36b0b6722dc3fd421d93512bac138e9a (patch) | |
| tree | b767d7f6e26bc64334d3f14f8422d187239dc45d /lib/math/gcd.c | |
| download | linux-stable-3f1fe48a36b0b6722dc3fd421d93512bac138e9a.tar.gz linux-stable-3f1fe48a36b0b6722dc3fd421d93512bac138e9a.zip | |
Merge tag 'io_uring-7.3-20261002' of git://git.kernel.org/pub/scm/linux/kernel/git/axboe/linuxgrafted
Pull io_uring fixes from Jens Axboe:
- Fix a task_work add use-after-free with SQPOLL.
The sqpoll thread could pop and complete the last request while
io_req_normal_work_add() was still looking at them after the mpscq
push.
Use the same approach as DEFER_TASKRUN to protect from that, holding
an RCU read lock across the add, and have exit wait for an RCU grace
period for SQPOLL rings as well.
- CQE32 ring fixes: correct the free entry check for 32b CQEs, zero the
big_cqe for aux CQEs, and only post the dummy skip CQE on CQE_MIXED
rings
- Mark the source filter table as COW when cloning bpf filters, so
registering another filter on the source doesn't modify the shared
table in place
- Initialize the task context before running the BPF loop
- Requeue zcrx multishot receives stopped by a local resource
- End a TX_TIMESTAMP multishot cmd when the CQ is full (lollipopkit)
* tag 'io_uring-7.3-20261002' of git://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux:
io_uring: fix task_work add use-after-free with SQPOLL
io_uring/cmd_net: end TX_TIMESTAMP multishot when the CQ is full
io_uring/zcrx: requeue multishot receives stopped by a local resource
io_uring: initialize task context before running the BPF loop
io_uring: zero big_cqe for aux CQEs on CQE32 rings
io_uring: fix free entry check for 32b CQEs on CQE32 rings
io_uring: only post the dummy skip CQE on CQE_MIXED rings
io_uring/bpf_filter: mark source as COW when cloning filters
Diffstat (limited to 'lib/math/gcd.c')
| -rw-r--r-- | lib/math/gcd.c | 88 |
1 files changed, 88 insertions, 0 deletions
diff --git a/lib/math/gcd.c b/lib/math/gcd.c new file mode 100644 index 000000000..62efca678 --- /dev/null +++ b/lib/math/gcd.c @@ -0,0 +1,88 @@ +// SPDX-License-Identifier: GPL-2.0-only +#include <linux/kernel.h> +#include <linux/gcd.h> +#include <linux/export.h> + +/* + * This implements the binary GCD algorithm. (Often attributed to Stein, + * but as Knuth has noted, appears in a first-century Chinese math text.) + * + * This is faster than the division-based algorithm even on x86, which + * has decent hardware division. + */ + +DEFINE_STATIC_KEY_TRUE(efficient_ffs_key); + +#if !defined(CONFIG_CPU_NO_EFFICIENT_FFS) + +/* If __ffs is available, the even/odd algorithm benchmarks slower. */ + +static unsigned long binary_gcd(unsigned long a, unsigned long b) +{ + unsigned long r = a | b; + + b >>= __ffs(b); + if (b == 1) + return r & -r; + + for (;;) { + a >>= __ffs(a); + if (a == 1) + return r & -r; + if (a == b) + return a << __ffs(r); + + if (a < b) + swap(a, b); + a -= b; + } +} + +#endif + +/* If normalization is done by loops, the even/odd algorithm is a win. */ + +/** + * gcd - calculate and return the greatest common divisor of 2 unsigned longs + * @a: first value + * @b: second value + */ +unsigned long gcd(unsigned long a, unsigned long b) +{ + unsigned long r = a | b; + + if (!a || !b) + return r; + +#if !defined(CONFIG_CPU_NO_EFFICIENT_FFS) + if (static_branch_likely(&efficient_ffs_key)) + return binary_gcd(a, b); +#endif + + /* Isolate lsbit of r */ + r &= -r; + + while (!(b & r)) + b >>= 1; + if (b == r) + return r; + + for (;;) { + while (!(a & r)) + a >>= 1; + if (a == r) + return r; + if (a == b) + return a; + + if (a < b) + swap(a, b); + a -= b; + a >>= 1; + if (a & r) + a += b; + a >>= 1; + } +} + +EXPORT_SYMBOL_GPL(gcd); |
