summaryrefslogtreecommitdiffstats
path: root/lib/syscall.c
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-10-02 12:17:24 -0700
committerLinus Torvalds <torvalds@linux-foundation.org>2026-10-02 12:17:24 -0700
commit3f1fe48a36b0b6722dc3fd421d93512bac138e9a (patch)
treeb767d7f6e26bc64334d3f14f8422d187239dc45d /lib/syscall.c
downloadlinux-stable-3f1fe48a36b0b6722dc3fd421d93512bac138e9a.tar.gz
linux-stable-3f1fe48a36b0b6722dc3fd421d93512bac138e9a.zip
Merge tag 'io_uring-7.3-20261002' of git://git.kernel.org/pub/scm/linux/kernel/git/axboe/linuxgrafted
Pull io_uring fixes from Jens Axboe: - Fix a task_work add use-after-free with SQPOLL. The sqpoll thread could pop and complete the last request while io_req_normal_work_add() was still looking at them after the mpscq push. Use the same approach as DEFER_TASKRUN to protect from that, holding an RCU read lock across the add, and have exit wait for an RCU grace period for SQPOLL rings as well. - CQE32 ring fixes: correct the free entry check for 32b CQEs, zero the big_cqe for aux CQEs, and only post the dummy skip CQE on CQE_MIXED rings - Mark the source filter table as COW when cloning bpf filters, so registering another filter on the source doesn't modify the shared table in place - Initialize the task context before running the BPF loop - Requeue zcrx multishot receives stopped by a local resource - End a TX_TIMESTAMP multishot cmd when the CQ is full (lollipopkit) * tag 'io_uring-7.3-20261002' of git://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux: io_uring: fix task_work add use-after-free with SQPOLL io_uring/cmd_net: end TX_TIMESTAMP multishot when the CQ is full io_uring/zcrx: requeue multishot receives stopped by a local resource io_uring: initialize task context before running the BPF loop io_uring: zero big_cqe for aux CQEs on CQE32 rings io_uring: fix free entry check for 32b CQEs on CQE32 rings io_uring: only post the dummy skip CQE on CQE_MIXED rings io_uring/bpf_filter: mark source as COW when cloning filters
Diffstat (limited to 'lib/syscall.c')
-rw-r--r--lib/syscall.c88
1 files changed, 88 insertions, 0 deletions
diff --git a/lib/syscall.c b/lib/syscall.c
new file mode 100644
index 000000000..006e256d2
--- /dev/null
+++ b/lib/syscall.c
@@ -0,0 +1,88 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <linux/ptrace.h>
+#include <linux/sched.h>
+#include <linux/sched/task_stack.h>
+#include <linux/export.h>
+#include <asm/syscall.h>
+
+static int collect_syscall(struct task_struct *target, struct syscall_info *info)
+{
+ unsigned long args[6] = { };
+ struct pt_regs *regs;
+
+ if (!try_get_task_stack(target)) {
+ /* Task has no stack, so the task isn't in a syscall. */
+ memset(info, 0, sizeof(*info));
+ info->data.nr = -1;
+ return 0;
+ }
+
+ regs = task_pt_regs(target);
+ if (unlikely(!regs)) {
+ put_task_stack(target);
+ return -EAGAIN;
+ }
+
+ info->sp = user_stack_pointer(regs);
+ info->data.instruction_pointer = instruction_pointer(regs);
+
+ info->data.nr = syscall_get_nr(target, regs);
+ if (info->data.nr != -1L)
+ syscall_get_arguments(target, regs, args);
+
+ info->data.args[0] = args[0];
+ info->data.args[1] = args[1];
+ info->data.args[2] = args[2];
+ info->data.args[3] = args[3];
+ info->data.args[4] = args[4];
+ info->data.args[5] = args[5];
+
+ put_task_stack(target);
+ return 0;
+}
+
+/**
+ * task_current_syscall - Discover what a blocked task is doing.
+ * @target: thread to examine
+ * @info: structure with the following fields:
+ * .sp - filled with user stack pointer
+ * .data.nr - filled with system call number or -1
+ * .data.args - filled with @maxargs system call arguments
+ * .data.instruction_pointer - filled with user PC
+ *
+ * If @target is blocked in a system call, returns zero with @info.data.nr
+ * set to the call's number and @info.data.args filled in with its
+ * arguments. Registers not used for system call arguments may not be available
+ * and it is not kosher to use &struct user_regset calls while the system
+ * call is still in progress. Note we may get this result if @target
+ * has finished its system call but not yet returned to user mode, such
+ * as when it's stopped for signal handling or syscall exit tracing.
+ *
+ * If @target is blocked in the kernel during a fault or exception,
+ * returns zero with *@info.data.nr set to -1 and does not fill in
+ * @info.data.args. If so, it's now safe to examine @target using
+ * &struct user_regset get() calls as long as we're sure @target won't return
+ * to user mode.
+ *
+ * Returns -%EAGAIN if @target does not remain blocked.
+ */
+int task_current_syscall(struct task_struct *target, struct syscall_info *info)
+{
+ unsigned long ncsw;
+ unsigned int state;
+
+ if (target == current)
+ return collect_syscall(target, info);
+
+ state = READ_ONCE(target->__state);
+ if (unlikely(!state))
+ return -EAGAIN;
+
+ ncsw = wait_task_inactive(target, state);
+ if (unlikely(!ncsw) ||
+ unlikely(collect_syscall(target, info)) ||
+ unlikely(wait_task_inactive(target, state) != ncsw))
+ return -EAGAIN;
+
+ return 0;
+}