mirror of
https://github.com/torvalds/linux.git
synced 2026-09-24 06:24:02 +02:00
x86/alternatives: Exclude text poking against change_page_attr()
From time to time, the following BUG can be observed
in the x86 alternatives patching code [0]:
> kernel BUG at arch/x86/kernel/alternative.c:2576!
> Oops: invalid opcode: 0000 [#1] SMP NOPTI
> CPU: 0 UID: 0 PID: 355 Comm: (udev-worker) Not tainted 7.1.3-1-default #1 PREEMPT(full) openSUSE Tumbleweed 8c1795b03ec64f997e57a8ad38b1161e3b98da64
> Hardware name: QEMU Standard PC (i440FX + PIIX, 1996), BIOS unknown 02/02/2022
> RIP: 0010:__text_poke+0x2aa/0x450
> Call Trace:
> <TASK>
> smp_text_poke_batch_finish+0x2a7/0x320
> __static_call_transform+0xb7/0x220
> arch_static_call_transform+0x5b/0xb0
> __static_call_init+0xe9/0x270
> static_call_module_notify+0x11f/0x150
> notifier_call_chain+0x61/0xe0
> blocking_notifier_call_chain_robust+0x63/0xc0
> load_module+0x1c92/0x20c0
> init_module_from_file+0xd8/0x140
> idempotent_init_module+0x100/0x2f0
> __x64_sys_finit_module+0x71/0xe0
> do_syscall_64+0xe1/0x610
> entry_SYSCALL_64_after_hwframe+0x76/0x7e
which matches the following BUG_ON() in alternative.c:
/*
* If something went wrong, crash and burn since recovery paths are not
* implemented.
*/
BUG_ON(!pages[0] || (cross_page_boundary && !pages[1]));
This can happen if vmalloc_to_page() fails, for any reason. Such can happen
if text poking races with CPA, which can possibly result in the collapsing
of page tables (or breaking of PMD hugepages). It is not a problem for most
users of vmalloc_to_page() (they solely own the vmalloc'd range) but, when
CONFIG_ARCH_HAS_EXECMEM_ROX=y, various modules own a single execmem vmalloc
range, and can call set_memory_*() in parallel on it. This can happen to
race against __text_poke and cause havoc in vmalloc_to_page().
Fix it by excluding against CPA using the init_mm mmap read lock.
[ dhansen: Fix up SoB ordering. The actual code flow here was:
Pedro=>Lorenzo=>Mike=>Me which is reflected in the SoB chain
now. I *believe* Mike simply picked up Lorenzo's update to
Pedro's post from the Link ]
Fixes: 64f6a4e10c ("x86: re-enable EXECMEM_ROX support")
Reported-by: Jiri Slaby <jirislaby@kernel.org>
Reported-by: Steffen Dirkwinkel <lists@steffen.cc>
Signed-off-by: Pedro Falcato <pfalcato@suse.de>
Signed-off-by: Lorenzo Stoakes (ARM) <ljs@kernel.org>
Co-developed-by: Lorenzo Stoakes (ARM) <ljs@kernel.org>
Signed-off-by: Mike Rapoport (Microsoft) <rppt@kernel.org>
Signed-off-by: Dave Hansen <dave.hansen@linux.intel.com>
Signed-off-by: Ingo Molnar <mingo@kernel.org>
Tested-by: Jiri Slaby <jirislaby@kernel.org>
Tested-by: Atish Patra <atishp@meta.com>
Tested-by: Nikunj A Dadhania <nikunj@amd.com>
Cc: stable@vger.kernel.org
Link: https://bugzilla.opensuse.org/show_bug.cgi?id=1271202 [0]
Link: https://lore.kernel.org/linux-mm/555ea1d43a12c30a8f1eaf10c899b3790d728f33.camel@dirkwinkel.cc/
Link: https://patch.msgid.link/20260813-cpa-fixes-v2-3-39b4ff90f91d@kernel.org
This commit is contained in:
parent
d5d8b8662e
commit
1587d3394e
|
|
@ -6,6 +6,9 @@
|
|||
#include <linux/vmalloc.h>
|
||||
#include <linux/memory.h>
|
||||
#include <linux/execmem.h>
|
||||
#include <linux/cleanup.h>
|
||||
#include <linux/kgdb.h>
|
||||
#include <linux/mmap_lock.h>
|
||||
|
||||
#include <asm/text-patching.h>
|
||||
#include <asm/insn.h>
|
||||
|
|
@ -2372,6 +2375,38 @@ static void text_poke_memset(void *dst, const void *src, size_t len)
|
|||
|
||||
typedef void text_poke_f(void *dst, const void *src, size_t len);
|
||||
|
||||
static void __poke_vmalloc_pages(struct page **pages, void *addr,
|
||||
bool cross_page_boundary)
|
||||
{
|
||||
pages[0] = vmalloc_to_page(addr);
|
||||
if (cross_page_boundary)
|
||||
pages[1] = vmalloc_to_page(addr + PAGE_SIZE);
|
||||
}
|
||||
|
||||
static void poke_vmalloc_pages(struct page **pages, void *addr,
|
||||
bool cross_page_boundary)
|
||||
{
|
||||
if (in_dbg_master()) {
|
||||
/*
|
||||
* If called from kgdb cannot sleep, but all other CPUs stopped
|
||||
* anyway so safe to proceed without locks
|
||||
*/
|
||||
__poke_vmalloc_pages(pages, addr, cross_page_boundary);
|
||||
} else {
|
||||
/*
|
||||
* execmem ROX ranges are shared between modules and can be
|
||||
* collapsed to huge PMD entries, and this collapse can happen
|
||||
* concurrently with a racing set_memory_rox().
|
||||
*
|
||||
* Prevent vmalloc_to_page() from racing by acquiring an
|
||||
* init_mm read lock which pairs with the init_mm write lock in
|
||||
* cpa_collapse_large_pages().
|
||||
*/
|
||||
guard(mmap_read_lock)(&init_mm);
|
||||
__poke_vmalloc_pages(pages, addr, cross_page_boundary);
|
||||
}
|
||||
}
|
||||
|
||||
static void *__text_poke(text_poke_f func, void *addr, const void *src, size_t len)
|
||||
{
|
||||
bool cross_page_boundary = offset_in_page(addr) + len > PAGE_SIZE;
|
||||
|
|
@ -2389,9 +2424,7 @@ static void *__text_poke(text_poke_f func, void *addr, const void *src, size_t l
|
|||
BUG_ON(!after_bootmem);
|
||||
|
||||
if (!core_kernel_text((unsigned long)addr)) {
|
||||
pages[0] = vmalloc_to_page(addr);
|
||||
if (cross_page_boundary)
|
||||
pages[1] = vmalloc_to_page(addr + PAGE_SIZE);
|
||||
poke_vmalloc_pages(pages, addr, cross_page_boundary);
|
||||
} else {
|
||||
pages[0] = virt_to_page(addr);
|
||||
WARN_ON(!PageReserved(pages[0]));
|
||||
|
|
|
|||
Loading…
Reference in New Issue
Block a user