diff options
| author | Fuad Tabba <fuad.tabba@linux.dev> | 2026-09-22 19:14:30 +0100 |
|---|---|---|
| committer | Will Deacon <will@kernel.org> | 2026-09-23 12:34:30 +0000 |
| commit | 2bc6b218717b9d08f466f88209251d54bc09b207 (patch) | |
| tree | cb91c941eccb15f94cbec4e96aa3a9c8736b5d66 /lib/syscall.c | |
| download | linux-stable-2bc6b218717b9d08f466f88209251d54bc09b207.tar.gz linux-stable-2bc6b218717b9d08f466f88209251d54bc09b207.zip | |
arm64/boot: Disable trapping of PMZR_EL0 writes to EL2grafted
__init_el2_fgt2() writes one mask to both HDFGRTR2_EL2 and HDFGWTR2_EL2.
PMZR_EL0 is write-only, so its trap bit, nPMZR_EL0, exists only in
HDFGWTR2_EL2 and is therefore never set: a PMZR_EL0 write from the host
traps to EL2, where the nVHE hypervisor has no handler and BUG()s. The
kernel never writes PMZR_EL0, but kernel.perf_user_access=1 has the PMU
driver set PMUSERENR_EL0.UEN for a task with a user-read event, so a
write from EL0 reaches the trap and takes the host down without a panic
message.
Accumulate the HDFGWTR2_EL2 bits separately, as __init_el2_fgt() already
does for HDFGWTR_EL2, and set nPMZR_EL0 with the other FEAT_PMUv3p9
bits.
Fixes: 858c7bfcb35e1 ("arm64/boot: Enable EL2 requirements for FEAT_PMUv3p9")
Cc: stable@vger.kernel.org
Signed-off-by: Fuad Tabba <fuad.tabba@linux.dev>
Reviewed-by: Anshuman Khandual <anshuman.khandual@arm.com>
Reviewed-by: Oliver Upton <oupton@kernel.org>
Signed-off-by: Will Deacon <will@kernel.org>
Diffstat (limited to 'lib/syscall.c')
| -rw-r--r-- | lib/syscall.c | 88 |
1 files changed, 88 insertions, 0 deletions
diff --git a/lib/syscall.c b/lib/syscall.c new file mode 100644 index 000000000..006e256d2 --- /dev/null +++ b/lib/syscall.c @@ -0,0 +1,88 @@ +// SPDX-License-Identifier: GPL-2.0 +#include <linux/ptrace.h> +#include <linux/sched.h> +#include <linux/sched/task_stack.h> +#include <linux/export.h> +#include <asm/syscall.h> + +static int collect_syscall(struct task_struct *target, struct syscall_info *info) +{ + unsigned long args[6] = { }; + struct pt_regs *regs; + + if (!try_get_task_stack(target)) { + /* Task has no stack, so the task isn't in a syscall. */ + memset(info, 0, sizeof(*info)); + info->data.nr = -1; + return 0; + } + + regs = task_pt_regs(target); + if (unlikely(!regs)) { + put_task_stack(target); + return -EAGAIN; + } + + info->sp = user_stack_pointer(regs); + info->data.instruction_pointer = instruction_pointer(regs); + + info->data.nr = syscall_get_nr(target, regs); + if (info->data.nr != -1L) + syscall_get_arguments(target, regs, args); + + info->data.args[0] = args[0]; + info->data.args[1] = args[1]; + info->data.args[2] = args[2]; + info->data.args[3] = args[3]; + info->data.args[4] = args[4]; + info->data.args[5] = args[5]; + + put_task_stack(target); + return 0; +} + +/** + * task_current_syscall - Discover what a blocked task is doing. + * @target: thread to examine + * @info: structure with the following fields: + * .sp - filled with user stack pointer + * .data.nr - filled with system call number or -1 + * .data.args - filled with @maxargs system call arguments + * .data.instruction_pointer - filled with user PC + * + * If @target is blocked in a system call, returns zero with @info.data.nr + * set to the call's number and @info.data.args filled in with its + * arguments. Registers not used for system call arguments may not be available + * and it is not kosher to use &struct user_regset calls while the system + * call is still in progress. Note we may get this result if @target + * has finished its system call but not yet returned to user mode, such + * as when it's stopped for signal handling or syscall exit tracing. + * + * If @target is blocked in the kernel during a fault or exception, + * returns zero with *@info.data.nr set to -1 and does not fill in + * @info.data.args. If so, it's now safe to examine @target using + * &struct user_regset get() calls as long as we're sure @target won't return + * to user mode. + * + * Returns -%EAGAIN if @target does not remain blocked. + */ +int task_current_syscall(struct task_struct *target, struct syscall_info *info) +{ + unsigned long ncsw; + unsigned int state; + + if (target == current) + return collect_syscall(target, info); + + state = READ_ONCE(target->__state); + if (unlikely(!state)) + return -EAGAIN; + + ncsw = wait_task_inactive(target, state); + if (unlikely(!ncsw) || + unlikely(collect_syscall(target, info)) || + unlikely(wait_task_inactive(target, state) != ncsw)) + return -EAGAIN; + + return 0; +} |
