From 3f1fe48a36b0b6722dc3fd421d93512bac138e9a Mon Sep 17 00:00:00 2001 From: Linus Torvalds Date: Fri, 2 Oct 2026 12:17:24 -0700 Subject: Merge tag 'io_uring-7.3-20261002' of git://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux Pull io_uring fixes from Jens Axboe: - Fix a task_work add use-after-free with SQPOLL. The sqpoll thread could pop and complete the last request while io_req_normal_work_add() was still looking at them after the mpscq push. Use the same approach as DEFER_TASKRUN to protect from that, holding an RCU read lock across the add, and have exit wait for an RCU grace period for SQPOLL rings as well. - CQE32 ring fixes: correct the free entry check for 32b CQEs, zero the big_cqe for aux CQEs, and only post the dummy skip CQE on CQE_MIXED rings - Mark the source filter table as COW when cloning bpf filters, so registering another filter on the source doesn't modify the shared table in place - Initialize the task context before running the BPF loop - Requeue zcrx multishot receives stopped by a local resource - End a TX_TIMESTAMP multishot cmd when the CQ is full (lollipopkit) * tag 'io_uring-7.3-20261002' of git://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux: io_uring: fix task_work add use-after-free with SQPOLL io_uring/cmd_net: end TX_TIMESTAMP multishot when the CQ is full io_uring/zcrx: requeue multishot receives stopped by a local resource io_uring: initialize task context before running the BPF loop io_uring: zero big_cqe for aux CQEs on CQE32 rings io_uring: fix free entry check for 32b CQEs on CQE32 rings io_uring: only post the dummy skip CQE on CQE_MIXED rings io_uring/bpf_filter: mark source as COW when cloning filters --- io_uring/loop.c | 96 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 96 insertions(+) create mode 100644 io_uring/loop.c (limited to 'io_uring/loop.c') diff --git a/io_uring/loop.c b/io_uring/loop.c new file mode 100644 index 000000000..3d472b782 --- /dev/null +++ b/io_uring/loop.c @@ -0,0 +1,96 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#include "io_uring.h" +#include "wait.h" +#include "loop.h" +#include "tctx.h" + +static inline int io_loop_nr_cqes(const struct io_ring_ctx *ctx, + const struct iou_loop_params *lp) +{ + return lp->cq_wait_idx - READ_ONCE(ctx->rings->cq.tail); +} + +static inline void io_loop_wait_start(struct io_ring_ctx *ctx, unsigned nr_wait) +{ + atomic_set(&ctx->cq_wait_nr, nr_wait); + set_current_state(TASK_INTERRUPTIBLE); +} + +static inline void io_loop_wait_finish(struct io_ring_ctx *ctx) +{ + __set_current_state(TASK_RUNNING); + atomic_set(&ctx->cq_wait_nr, IO_CQ_WAKE_INIT); +} + +static void io_loop_wait(struct io_ring_ctx *ctx, struct iou_loop_params *lp, + unsigned nr_wait) +{ + io_loop_wait_start(ctx, nr_wait); + + if (unlikely(io_local_work_pending(ctx) || + io_loop_nr_cqes(ctx, lp) <= 0) || + READ_ONCE(ctx->check_cq)) { + io_loop_wait_finish(ctx); + return; + } + + mutex_unlock(&ctx->uring_lock); + schedule(); + io_loop_wait_finish(ctx); + mutex_lock(&ctx->uring_lock); +} + +static int __io_run_loop(struct io_ring_ctx *ctx) +{ + struct iou_loop_params lp = {}; + + while (true) { + int nr_wait, step_res; + + if (unlikely(!ctx->loop_step)) + return -EFAULT; + + step_res = ctx->loop_step(io_loop_mangle_ctx(ctx), &lp); + if (step_res == IOU_LOOP_STOP) + break; + if (step_res != IOU_LOOP_CONTINUE) + return -EINVAL; + + nr_wait = io_loop_nr_cqes(ctx, &lp); + if (nr_wait > 0) + io_loop_wait(ctx, &lp, nr_wait); + else + nr_wait = 0; + + if (task_work_pending(current)) { + mutex_unlock(&ctx->uring_lock); + io_run_task_work(); + mutex_lock(&ctx->uring_lock); + } + if (unlikely(task_sigpending(current))) + return -EINTR; + io_run_local_work_locked(ctx, nr_wait); + + if (READ_ONCE(ctx->check_cq) & BIT(IO_CHECK_CQ_OVERFLOW_BIT)) + io_cqring_overflow_flush_locked(ctx); + } + + return 0; +} + +int io_run_loop(struct io_ring_ctx *ctx) +{ + int ret; + + ret = io_uring_add_tctx_node(ctx); + if (unlikely(ret)) + return ret; + + if (!io_allowed_run_tw(ctx)) + return -EEXIST; + + mutex_lock(&ctx->uring_lock); + ret = __io_run_loop(ctx); + mutex_unlock(&ctx->uring_lock); + return ret; +} -- cgit v1.3.1