From aab0734094cef631d611e404c389d89f0b869f5d Mon Sep 17 00:00:00 2001 From: Heiko Carstens Date: Mon, 20 Jul 2026 10:58:27 +0200 Subject: [PATCH] KVM: s390: pv: Use VM_SPARSE area for guest variable storage area The guest variable storage area is allocated with vmalloc and then donated to the ultravisor. Any kernel access to that area will result in a secure storage access exception (aka fault). This is a problem if such a memory area is read via /proc/kcore. This causes an exception via vread_iter() and results in an unexpected short read. Avoid this by allocating a custom VM_SPARSE area. If such an area is read, vread_iter() returns zeroes for the entire area. Note that the function which frees the area does not update ptes. This is intentional to allow for deferred / lazy pte updates and TLB flushing like the generic vfree() code is doing that. See vunmap_pte_range(). This assumes that s390 will gain full support for lazy_mmu_mode_enable() and lazy_mmu_mode_disable() in the future, since as of now the used ptep_get_and_clear() in vunmap_pte_range() does indeed invalidate and flush every single pte entry, but only for s390. Tested-by: Christian Borntraeger Signed-off-by: Heiko Carstens Reviewed-by: Christian Borntraeger Signed-off-by: Vasily Gorbik --- arch/s390/include/asm/uv.h | 2 ++ arch/s390/kernel/uv.c | 65 ++++++++++++++++++++++++++++++++++++++ arch/s390/kvm/pv.c | 6 ++-- 3 files changed, 70 insertions(+), 3 deletions(-) diff --git a/arch/s390/include/asm/uv.h b/arch/s390/include/asm/uv.h index d919e69662f5..153fed63adda 100644 --- a/arch/s390/include/asm/uv.h +++ b/arch/s390/include/asm/uv.h @@ -635,6 +635,8 @@ int s390_wiggle_split_folio(struct mm_struct *mm, struct folio *folio); int __make_folio_secure(struct folio *folio, struct uv_cb_header *uvcb); int uv_convert_from_secure(unsigned long paddr); int uv_convert_from_secure_folio(struct folio *folio); +void *uv_alloc_stor_var(unsigned long size); +void uv_free_stor_var(void *stor_var); void setup_uv(void); diff --git a/arch/s390/kernel/uv.c b/arch/s390/kernel/uv.c index a284f98d9716..d1cdc5ae5e27 100644 --- a/arch/s390/kernel/uv.c +++ b/arch/s390/kernel/uv.c @@ -16,6 +16,7 @@ #include #include #include +#include #include #include #include @@ -209,6 +210,70 @@ int uv_convert_from_secure_pte(pte_t pte) return uv_convert_from_secure_folio(pfn_folio(pte_pfn(pte))); } +static int uv_free_range_cb(pte_t *ptep, unsigned long addr, void *data) +{ + pte_t pte = ptep_get(ptep); + + if (!pte_present(pte)) + return 0; + /* + * Note: do not update the pte here, since there is no code which + * accesses the memory range, besides bugs. The invalidation of ptes + * and TLB flushing is deferred like for regular vfree() calls. + */ + __free_page(pte_page(pte)); + return 0; +} + +void uv_free_stor_var(void *stor_var) +{ + unsigned long addr, size; + struct vm_struct *area; + + if (!stor_var) + return; + area = find_vm_area(stor_var); + if (WARN_ON_ONCE(!area || !(area->flags & VM_SPARSE))) + return; + size = get_vm_area_size(area); + addr = (unsigned long)area->addr; + apply_to_existing_page_range(&init_mm, addr, size, uv_free_range_cb, NULL); + free_vm_area(area); +} +EXPORT_SYMBOL_FOR_MODULES(uv_free_stor_var, "kvm"); + +static int uv_alloc_range_cb(pte_t *ptep, unsigned long addr, void *data) +{ + struct page *page; + pte_t pte; + + page = alloc_page(GFP_KERNEL_ACCOUNT | __GFP_ZERO); + if (!page) + return -ENOMEM; + pte = __pte(page_to_phys(page) | pgprot_val(PAGE_KERNEL)); + set_pte(ptep, pte); + return 0; +} + +void *uv_alloc_stor_var(unsigned long size) +{ + struct vm_struct *area; + unsigned long addr; + + size = PAGE_ALIGN(size); + area = get_vm_area(size, VM_SPARSE); + if (!area) + return NULL; + addr = (unsigned long)area->addr; + if (apply_to_page_range(&init_mm, addr, size, uv_alloc_range_cb, NULL)) + goto out; + return area->addr; +out: + uv_free_stor_var(area->addr); + return NULL; +} +EXPORT_SYMBOL_FOR_MODULES(uv_alloc_stor_var, "kvm"); + /* * Calculate the expected ref_count for a folio that would otherwise have no * further pins. This was cribbed from similar functions in other places in diff --git a/arch/s390/kvm/pv.c b/arch/s390/kvm/pv.c index 1beacc841ca8..dc5ac29b4c31 100644 --- a/arch/s390/kvm/pv.c +++ b/arch/s390/kvm/pv.c @@ -337,7 +337,7 @@ int kvm_s390_pv_create_cpu(struct kvm_vcpu *vcpu, u16 *rc, u16 *rrc) /* only free resources when the destroy was successful */ static void kvm_s390_pv_dealloc_vm(struct kvm *kvm) { - vfree(kvm->arch.pv.stor_var); + uv_free_stor_var(kvm->arch.pv.stor_var); free_pages(kvm->arch.pv.stor_base, get_order(uv_info.guest_base_stor_len)); kvm_s390_clear_pv_state(kvm); @@ -369,7 +369,7 @@ static int kvm_s390_pv_alloc_vm(struct kvm *kvm) /* Allocate variable storage */ vlen = ALIGN(virt * ((npages * PAGE_SIZE) / HPAGE_SIZE), PAGE_SIZE); vlen += uv_info.guest_virt_base_stor_len; - kvm->arch.pv.stor_var = vzalloc(vlen); + kvm->arch.pv.stor_var = uv_alloc_stor_var(vlen); if (!kvm->arch.pv.stor_var) goto out_err; return 0; @@ -414,7 +414,7 @@ static int kvm_s390_pv_dispose_one_leftover(struct kvm *kvm, */ free_pages(leftover->stor_base, get_order(uv_info.guest_base_stor_len)); free_pages(leftover->old_gmap_table, CRST_ALLOC_ORDER); - vfree(leftover->stor_var); + uv_free_stor_var(leftover->stor_var); done_fast: atomic_dec(&kvm->mm->context.protected_count); return 0;