From d2457a7e2727dc747a61a50d65c2f259afce8c45 Mon Sep 17 00:00:00 2001 From: Laurent Wandrebeck Date: Tue, 22 Sep 2026 10:50:32 +0200 Subject: x86/mm: Don't apply va_align to hugetlb mappings on AMD F15h get_align_mask() returns huge_page_mask_align() for hugetlbfs, but get_align_bits() adds va_align.bits regardless, so vm_unmapped_area() returns an address off the huge page boundary and __unmap_hugepage_range() hits BUG_ON(start & ~huge_page_mask(h)) at teardown. This can be triggered on Carrizo and FX-8370E, both hstates. Pass the file to get_align_bits() and skip the randomisation for hugetlbfs. [ bp: Massage commit message. ] Fixes: 1317a5e7f7b1 ("arch/x86: teach arch_get_unmapped_area_vmflags to handle hugetlb mappings") Suggested-by: Dave Hansen Acked-by: Dave Hansen Signed-off-by: Laurent Wandrebeck Signed-off-by: Borislav Petkov (AMD) Cc: stable@vger.kernel.org # 6.13+ Link: https://patch.msgid.link/20260922085032.46144-1-l.wandrebeck@quelquesmots.fr --- arch/x86/kernel/sys_x86_64.c | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/arch/x86/kernel/sys_x86_64.c b/arch/x86/kernel/sys_x86_64.c index 776ae6fa7..4c078827f 100644 --- a/arch/x86/kernel/sys_x86_64.c +++ b/arch/x86/kernel/sys_x86_64.c @@ -50,9 +50,16 @@ static unsigned long get_align_mask(struct file *filp) * value before calling vm_unmapped_area() or ORed directly to the * address. */ -static unsigned long get_align_bits(void) +static unsigned long get_align_bits(struct file *filp) { - return va_align.bits & get_align_mask(NULL); + /* + * va_align.bits is smaller than the huge page size and will + * lead to misaligned huge pages. Ignore it for huge mappings. + */ + if (is_file_hugepages(filp)) + return 0; + + return va_align.bits & get_align_mask(filp); } static int __init control_va_addr_alignment(char *str) @@ -157,7 +164,7 @@ arch_get_unmapped_area(struct file *filp, unsigned long addr, unsigned long len, } if (filp) { info.align_mask = get_align_mask(filp); - info.align_offset += get_align_bits(); + info.align_offset += get_align_bits(filp); } return vm_unmapped_area(&info); @@ -222,7 +229,7 @@ get_unmapped_area: if (filp) { info.align_mask = get_align_mask(filp); - info.align_offset += get_align_bits(); + info.align_offset += get_align_bits(filp); } addr = vm_unmapped_area(&info); if (!(addr & ~PAGE_MASK)) -- cgit v1.3.1 From e3ee38c1bc0b11a8d3e63dce7050759b61864286 Mon Sep 17 00:00:00 2001 From: Mikhail Gavrilov Date: Thu, 24 Sep 2026 03:31:16 +0500 Subject: x86/mm: Drop unnecessary PMD page copy when freeing On a box with a discrete GPU, lockdep reports a possible deadlock as soon as kswapd shrinks the TTM page pool. The immediate cause is an x86 commit that added an mmap_read_lock() to kernel page protection munging code. The huge vmap code holds the same lock over a GFP_KERNEL allocation, which is a no-no now that reclaim can take it. That allocation is in a page table *free* path and ends up being for dubious purposes[1]. Basically, it tries to avoid hardware setting Accessed=1 in page table entries that are unreachable by the hardware, a non-issue. Remove the PMD copy. Detach the original PMD page at the PUD, flush the mid-level caches, and free the PTE tables straight from the detached PMD page. With no allocation left, the locking issue is gone. Lockdep splat/analysis: WARNING: possible circular locking dependency detected 7.3.0-rc3-f6e7b42bf05b+ #183 Tainted: G U ------------------------------------------------------ kswapd0/269 is trying to acquire lock: ((init_mm).mmap_lock){++++}-{4:4}, at: change_page_attr_set_clr+0x29a/0x4a0 but task is already holding lock: (pool_shrink_rwsem){.+.+}-{4:4}, at: ttm_pool_shrink+0xb2/0x330 [ttm] Chain exists of: (init_mm).mmap_lock --> fs_reclaim --> pool_shrink_rwsem The cycle is built from three edges: 1) pool_shrink_rwsem -> (init_mm).mmap_lock The TTM shrinker restores the caching attribute of every page it frees, while holding pool_shrink_rwsem: ttm_pool_shrink() -> ttm_pool_dispose_list() -> ttm_pool_free_page() -> set_pages_wb() -> change_page_attr_set_clr() [ init_mm mmap read lock ] 2) fs_reclaim -> pool_shrink_rwsem The same shrinker, called from reclaim. 3) (init_mm).mmap_lock -> fs_reclaim ioremap() installing a huge PUD mapping over an existing PMD table: ioremap_page_range() -> vmap_range_noflush() -> vmap_try_huge_pud() [ init_mm mmap read lock ] -> pud_free_pmd_page() -> __get_free_page(GFP_KERNEL) [ enters reclaim ] [ dhansen: Lots of changelog munging/trimming and merged comments from my version of the fix. ] Fixes: d5d8b8662e6e ("x86/mm/pat: Acquire init_mm read lock on attribute changes to avoid UAF") Suggested-by: Pedro Falcato Signed-off-by: Mikhail Gavrilov Signed-off-by: Dave Hansen Reviewed-by: Pedro Falcato Link: https://lore.kernel.org/20260916062222.27347-1-mikhail.v.gavrilov@gmail.com Link: https://lore.kernel.org/all/e11449f0-d9ad-4d1b-ab21-2be7d71fe335@intel.com/ [1] Link: https://patch.msgid.link/20260923223116.20090-1-mikhail.v.gavrilov@gmail.com Cc: stable@vger.kernel.org --- arch/x86/mm/pgtable.c | 38 ++++++++++++++------------------------ 1 file changed, 14 insertions(+), 24 deletions(-) diff --git a/arch/x86/mm/pgtable.c b/arch/x86/mm/pgtable.c index cb03f5a2b..4a105f283 100644 --- a/arch/x86/mm/pgtable.c +++ b/arch/x86/mm/pgtable.c @@ -705,47 +705,37 @@ int pmd_clear_huge(pmd_t *pmd) } #ifdef CONFIG_X86_64 -/** - * pud_free_pmd_page - Clear PUD entry and free PMD page - * @pud: Pointer to a PUD - * @addr: Virtual address associated with PUD - * - * Context: The PUD range has been unmapped and TLB purged. - * Return: 1 if clearing the entry succeeded. 0 otherwise. - * - * NOTE: Callers must allow a single page allocation. +/* + * Given a PUD poitner, detach and free the pointed-to + * PMD page and any PTE page children. The entire range + * under the PUD must not have any valid translations + * and the TLB must have already been flushed. */ int pud_free_pmd_page(pud_t *pud, unsigned long addr) { - pmd_t *pmd, *pmd_sv; struct ptdesc *pt; + pmd_t *pmd; int i; pmd = pud_pgtable(*pud); - pmd_sv = (pmd_t *)__get_free_page(GFP_KERNEL); - if (!pmd_sv) - return 0; - - for (i = 0; i < PTRS_PER_PMD; i++) { - pmd_sv[i] = pmd[i]; - if (!pmd_none(pmd[i])) - pmd_clear(&pmd[i]); - } + /* Detach the PMD page: */ pud_clear(pud); - /* INVLPG to clear all paging-structure caches */ + /* + * PMD and all its descendents are unreachable + * via normal page walks. Make them unreachable + * in cached mid-level walks too: + */ flush_tlb_kernel_range(addr, addr + PAGE_SIZE-1); for (i = 0; i < PTRS_PER_PMD; i++) { - if (!pmd_none(pmd_sv[i])) { - pt = page_ptdesc(pmd_page(pmd_sv[i])); + if (!pmd_none(pmd[i])) { + pt = page_ptdesc(pmd_page(pmd[i])); pagetable_dtor_free(pt); } } - free_page((unsigned long)pmd_sv); - pmd_free(&init_mm, pmd); return 1; -- cgit v1.3.1 From a661c34fe693c8f9ef29b45ee71cfa1fe593502b Mon Sep 17 00:00:00 2001 From: Nick Desaulniers Date: Fri, 21 Aug 2026 15:45:42 -0700 Subject: {x86,um}/uapi/ptrace: Guard register offset macros with __ASSEMBLER__ or __FRAME_OFFSETS The register offset macros in are guarded by `defined(__ASSEMBLER__) || defined(__FRAME_OFFSETS)` for 64-bit, but were left unguarded for 32-bit. This causes havoc for userspace that happens to use identifiers colliding with these short macro names (e.g., EBX, ECX, EAX, DS, ES, FS, GS, CS, SS). Without this guard, userspace is forced to be super extra careful with include ordering to minimize the chance of collision. Wrap both the 32-bit and 64-bit register definitions under `#if defined(__ASSEMBLER__) || defined(__FRAME_OFFSETS)`, and ensure User-Mode Linux (UML) defines `__FRAME_OFFSETS` for 32-bit as well. Closes: https://github.com/llvm/llvm-project/issues/217413 Assisted-by: LLM Signed-off-by: Nick Desaulniers Signed-off-by: Borislav Petkov (AMD) Acked-by: Oleg Nesterov Acked-by: Johannes Berg Tested-by: Elliott Hughes Link: https://patch.msgid.link/20260821-ptrace_uapi-v1-1-3de8638a29f2@google.com --- arch/x86/include/uapi/asm/ptrace-abi.h | 4 ++-- arch/x86/um/asm/ptrace.h | 4 +--- arch/x86/um/ptrace_32.c | 1 + 3 files changed, 4 insertions(+), 5 deletions(-) diff --git a/arch/x86/include/uapi/asm/ptrace-abi.h b/arch/x86/include/uapi/asm/ptrace-abi.h index 5823584de..3656955c6 100644 --- a/arch/x86/include/uapi/asm/ptrace-abi.h +++ b/arch/x86/include/uapi/asm/ptrace-abi.h @@ -2,6 +2,7 @@ #ifndef _ASM_X86_PTRACE_ABI_H #define _ASM_X86_PTRACE_ABI_H +#if defined(__ASSEMBLER__) || defined(__FRAME_OFFSETS) #ifdef __i386__ #define EBX 0 @@ -25,7 +26,6 @@ #else /* __i386__ */ -#if defined(__ASSEMBLER__) || defined(__FRAME_OFFSETS) /* * C ABI says these regs are callee-preserved. They aren't saved on kernel entry * unless syscall needs a complete, fully filled "struct pt_regs". @@ -57,12 +57,12 @@ #define EFLAGS 144 #define RSP 152 #define SS 160 -#endif /* __ASSEMBLER__ */ /* top of stack page */ #define FRAME_SIZE 168 #endif /* !__i386__ */ +#endif /* defined(__ASSEMBLER__) || defined(__FRAME_OFFSETS) */ /* Arbitrarily choose the same ptrace numbers as used by the Sparc code. */ #define PTRACE_GETREGS 12 diff --git a/arch/x86/um/asm/ptrace.h b/arch/x86/um/asm/ptrace.h index 2641d28d1..439c4151f 100644 --- a/arch/x86/um/asm/ptrace.h +++ b/arch/x86/um/asm/ptrace.h @@ -13,9 +13,7 @@ enum { }; #include -#ifndef CONFIG_X86_32 -#define __FRAME_OFFSETS /* Needed to get the R* macros */ -#endif +#define __FRAME_OFFSETS /* Needed to get the register macros */ #include #define user_mode(r) UPT_IS_USER(&(r)->regs) diff --git a/arch/x86/um/ptrace_32.c b/arch/x86/um/ptrace_32.c index 3af3cb821..9e9155b0e 100644 --- a/arch/x86/um/ptrace_32.c +++ b/arch/x86/um/ptrace_32.c @@ -7,6 +7,7 @@ #include #include #include +#define __FRAME_OFFSETS #include #include #include -- cgit v1.3.1