mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
* linux-next: manual merge of the kvm-x86 tree with the kvm-arm tree
@ 2026-10-02 12:54 Mark Brown
  0 siblings, 0 replies; 4+ messages in thread
From: Mark Brown @ 2026-10-02 12:54 UTC (permalink / raw)
  To: Sean Christopherson
  Cc: Ackerley Tng, Fuad Tabba, Linux Kernel Mailing List,
	Linux Next Mailing List, Marc Zyngier, Oliver Upton,
	Vincent Donnefort, Yan Zhao

[-- Attachment #1: Type: text/plain, Size: 35199 bytes --]

Hi all,

Today's linux-next merge of the kvm-x86 tree got a conflict in:

  arch/arm64/kvm/mmu.c

between commit:

  80b0b86dc5e43 ("KVM: arm64: Use kvm_s2_fault_vma_info in gmem_abort()")

from the kvm-arm tree and commit:

  89fbe3de01ccf ("KVM: guest_memfd: Stop returning struct page from PFN lookup")

from the kvm-x86 tree.

I fixed it up (see below) and can carry the fix as necessary. This
is now fixed as far as linux-next is concerned, but any non trivial
conflicts should be mentioned to your upstream maintainer when your tree
is submitted for merging.  You may also want to consider cooperating
with the maintainer of the conflicting tree to minimise any particularly
complex conflicts.

diff --combined arch/arm64/kvm/mmu.c
index 85745d13c08e0,df48df8d89800..0000000000000
--- a/arch/arm64/kvm/mmu.c
+++ b/arch/arm64/kvm/mmu.c
@@@ -5,10 -5,8 +5,10 @@@
   */
  
  #include <linux/acpi.h>
 +#include <linux/cleanup.h>
  #include <linux/mman.h>
  #include <linux/kvm_host.h>
 +#include <linux/interval_tree.h>
  #include <linux/io.h>
  #include <linux/hugetlb.h>
  #include <linux/sched/signal.h>
@@@ -324,19 -322,6 +324,19 @@@ static void invalidate_icache_guest_pag
   * we then fully enforce cacheability of RAM, no matter what the guest
   * does.
   */
 +
 +static int kvm_pgtable_stage2_unmap_tracked(struct kvm_pgtable *pgt, u64 addr, u64 size)
 +{
 +	int ret;
 +
 +	ret = kvm_pgtable_stage2_unmap(pgt, addr, size);
 +	if (ret)
 +		return ret;
 +
 +	kvm_remove_guest_s2_mappings(pgt->mmu, addr, size);
 +	return 0;
 +}
 +
  /**
   * __unmap_stage2_range -- Clear stage2 page table entries to unmap a range
   * @mmu:   The KVM stage-2 MMU pointer
@@@ -354,17 -339,11 +354,17 @@@ static void __unmap_stage2_range(struc
  {
  	struct kvm *kvm = kvm_s2_mmu_to_kvm(mmu);
  	phys_addr_t end = start + size;
 +	int (*fn)(struct kvm_pgtable *, u64, u64);
  
  	lockdep_assert_held_write(&kvm->mmu_lock);
  	WARN_ON(size & ~PAGE_MASK);
 -	WARN_ON(stage2_apply_range(mmu, start, end, KVM_PGT_FN(kvm_pgtable_stage2_unmap),
 -				   may_block));
 +
 +	if (kvm_is_nested_s2_mmu(kvm, mmu))
 +		fn = kvm_pgtable_stage2_unmap_tracked;
 +	else
 +		fn = KVM_PGT_FN(kvm_pgtable_stage2_unmap);
 +
 +	WARN_ON(stage2_apply_range(mmu, start, end, fn, may_block));
  }
  
  void kvm_stage2_unmap_range(struct kvm_s2_mmu *mmu, phys_addr_t start,
@@@ -896,7 -875,7 +896,7 @@@ static int get_user_mapping_size(struc
  	 * IPI-ing threads).
  	 */
  	local_irq_save(flags);
 -	ret = kvm_pgtable_get_leaf(&pgt, addr, &pte, &level);
 +	ret = kvm_pgtable_get_leaf(&pgt, addr, &pte, &level, 0);
  	local_irq_restore(flags);
  
  	if (ret)
@@@ -1063,8 -1042,6 +1063,8 @@@ int kvm_init_stage2_mmu(struct kvm *kvm
  
  	mmu->pgd_phys = __pa(pgt->pgd);
  
 +	mmu->guest_s2_mappings = RB_ROOT_CACHED;
 +
  	if (kvm_is_nested_s2_mmu(kvm, mmu))
  		kvm_init_nested_s2_mmu(mmu);
  
@@@ -1154,25 -1131,10 +1154,25 @@@ void stage2_unmap_vm(struct kvm *kvm
  	srcu_read_unlock(&kvm->srcu, idx);
  }
  
 +static void guest_s2_tracking_destroy(struct rb_root_cached *tree)
 +{
 +	struct kvm_guest_s2_mapping *mapping;
 +	struct interval_tree_node *node;
 +
 +	while ((node = interval_tree_iter_first(tree, 0, ULONG_MAX))) {
 +		interval_tree_remove(node, tree);
 +		mapping = container_of(node, struct kvm_guest_s2_mapping,
 +				       canonical);
 +		kfree(mapping);
 +		cond_resched();
 +	}
 +}
 +
  void kvm_free_stage2_pgd(struct kvm_s2_mmu *mmu)
  {
  	struct kvm *kvm = kvm_s2_mmu_to_kvm(mmu);
  	struct kvm_pgtable *pgt = NULL;
 +	struct rb_root_cached mappings_tree;
  
  	write_lock(&kvm->mmu_lock);
  	pgt = mmu->pgt;
@@@ -1185,18 -1147,12 +1185,18 @@@
  	if (kvm_is_nested_s2_mmu(kvm, mmu))
  		kvm_init_nested_s2_mmu(mmu);
  
 +	mappings_tree = mmu->guest_s2_mappings;
 +	mmu->guest_s2_mappings = RB_ROOT_CACHED;
 +
  	write_unlock(&kvm->mmu_lock);
  
  	if (pgt) {
  		kvm_stage2_destroy(pgt);
  		kfree(pgt);
  	}
 +
 +	if (!kvm_is_nested_s2_mmu(kvm, mmu))
 +		guest_s2_tracking_destroy(&mappings_tree);
  }
  
  static void hyp_mc_free_fn(void *addr, void *mc)
@@@ -1406,53 -1362,21 +1406,53 @@@ static void kvm_send_hwpoison_signal(un
  	send_sig_mceerr(BUS_MCEERR_AR, (void __user *)address, lsb, current);
  }
  
 -static bool fault_supports_stage2_huge_mapping(struct kvm_memory_slot *memslot,
 -					       unsigned long hva,
 +struct kvm_s2_fault_desc {
 +	struct kvm_vcpu		*vcpu;
 +	phys_addr_t		fault_ipa;
 +	struct kvm_s2_trans	*nested;
 +	struct kvm_memory_slot	*memslot;
 +	unsigned long		hva;
 +	unsigned long		esr;
 +	struct kvm_s2_mmu	*mmu;
 +};
 +
 +struct kvm_s2_fault_vma_info {
 +	unsigned long	mmu_seq;
 +	long		vma_pagesize;
 +	vm_flags_t	vm_flags;
 +	unsigned long	max_map_size;
 +	struct page	*page;
 +	kvm_pfn_t	pfn;
 +	gfn_t		gfn;
 +	bool		device;
 +	bool		mte_allowed;
 +	bool		is_vma_cacheable;
 +	bool		map_writable;
 +	bool		map_non_cacheable;
 +};
 +
 +struct kvm_s2_fault_result {
 +	unsigned long mapping_size;
 +};
 +
 +static bool fault_supports_stage2_huge_mapping(const struct kvm_s2_fault_desc *s2fd,
  					       unsigned long map_size)
  {
 -	gpa_t gpa_start;
 +	struct kvm_memory_slot *memslot = s2fd->memslot;
 +	unsigned long hva = s2fd->hva;
  	hva_t uaddr_start, uaddr_end;
 +	gpa_t gpa_start;
  	size_t size;
  
  	/* The memslot and the VMA are guaranteed to be aligned to PAGE_SIZE */
  	if (map_size == PAGE_SIZE)
  		return true;
  
 -	/* pKVM only supports PMD_SIZE huge-mappings */
 -	if (is_protected_kvm_enabled() && map_size != PMD_SIZE)
 -		return false;
 +	/* pKVM only supports PMD_SIZE huge-mappings for non-protected VMs */
 +	if (is_protected_kvm_enabled()) {
 +		if (vcpu_is_protected(s2fd->vcpu) || map_size != PMD_SIZE)
 +			return false;
 +	}
  
  	size = memslot->npages * PAGE_SIZE;
  
@@@ -1512,9 -1436,9 +1512,9 @@@
   * Returns the size of the mapping.
   */
  static long
 -transparent_hugepage_adjust(struct kvm *kvm, struct kvm_memory_slot *memslot,
 -			    unsigned long hva, kvm_pfn_t *pfnp, gfn_t *gfnp)
 +transparent_hugepage_adjust(const struct kvm_s2_fault_desc *s2fd, kvm_pfn_t *pfnp, gfn_t *gfnp)
  {
 +	struct kvm *kvm = s2fd->vcpu->kvm;
  	kvm_pfn_t pfn = *pfnp;
  	gfn_t gfn = *gfnp;
  
@@@ -1523,8 -1447,8 +1523,8 @@@
  	 * sure that the HVA and IPA are sufficiently aligned and that the
  	 * block map is contained within the memslot.
  	 */
 -	if (fault_supports_stage2_huge_mapping(memslot, hva, PMD_SIZE)) {
 -		int sz = get_user_mapping_size(kvm, hva);
 +	if (fault_supports_stage2_huge_mapping(s2fd, PMD_SIZE)) {
 +		int sz = get_user_mapping_size(kvm, s2fd->hva);
  
  		if (sz < 0)
  			return sz;
@@@ -1544,11 -1468,32 +1544,11 @@@
  	return PAGE_SIZE;
  }
  
 -static int get_vma_page_shift(struct vm_area_struct *vma, unsigned long hva)
 +static int get_vma_page_shift(struct vm_area_struct *vma)
  {
 -	unsigned long pa;
 -
 -	if (is_vm_hugetlb_page(vma) && !(vma->vm_flags & VM_PFNMAP))
 +	if (vma_is_hugetlb(vma))
  		return huge_page_shift(hstate_vma(vma));
  
 -	if (!(vma->vm_flags & VM_PFNMAP))
 -		return PAGE_SHIFT;
 -
 -	VM_BUG_ON(is_vm_hugetlb_page(vma));
 -
 -	pa = (vma->vm_pgoff << PAGE_SHIFT) + (hva - vma->vm_start);
 -
 -#ifndef __PAGETABLE_PMD_FOLDED
 -	if ((hva & (PUD_SIZE - 1)) == (pa & (PUD_SIZE - 1)) &&
 -	    ALIGN_DOWN(hva, PUD_SIZE) >= vma->vm_start &&
 -	    ALIGN(hva, PUD_SIZE) <= vma->vm_end)
 -		return PUD_SHIFT;
 -#endif
 -
 -	if ((hva & (PMD_SIZE - 1)) == (pa & (PMD_SIZE - 1)) &&
 -	    ALIGN_DOWN(hva, PMD_SIZE) >= vma->vm_start &&
 -	    ALIGN(hva, PMD_SIZE) <= vma->vm_end)
 -		return PMD_SHIFT;
 -
  	return PAGE_SHIFT;
  }
  
@@@ -1618,9 -1563,9 +1618,9 @@@ static void *get_mmu_memcache(struct kv
  		return &vcpu->arch.pkvm_memcache;
  }
  
 -static int topup_mmu_memcache(struct kvm_vcpu *vcpu, void *memcache)
 +static int topup_mmu_memcache(struct kvm_s2_mmu *mmu, void *memcache)
  {
 -	int min_pages = kvm_mmu_cache_min_pages(vcpu->arch.hw_mmu);
 +	int min_pages = kvm_mmu_cache_min_pages(mmu);
  
  	if (!is_protected_kvm_enabled())
  		return kvm_mmu_topup_memory_cache(memcache, min_pages);
@@@ -1661,85 -1606,54 +1661,85 @@@ static enum kvm_pgtable_prot adjust_nes
  	return prot;
  }
  
 -struct kvm_s2_fault_desc {
 -	struct kvm_vcpu		*vcpu;
 -	phys_addr_t		fault_ipa;
 -	struct kvm_s2_trans	*nested;
 -	struct kvm_memory_slot	*memslot;
 -	unsigned long		hva;
 -};
 +static bool kvm_s2_fault_is_perm(const struct kvm_s2_fault_desc *s2fd)
 +{
 +	return esr_fsc_is_permission_fault(s2fd->esr);
 +}
  
 -static int gmem_abort(const struct kvm_s2_fault_desc *s2fd)
 +static bool kvm_s2_fault_is_exec(const struct kvm_s2_fault_desc *s2fd)
 +{
 +	return esr_abt_is_exec_fault(s2fd->esr);
 +}
 +
 +static bool kvm_s2_fault_is_write(const struct kvm_s2_fault_desc *s2fd)
 +{
 +	return esr_abt_is_write_fault(s2fd->esr);
 +}
 +
 +static u64 kvm_s2_perm_fault_granule(const struct kvm_s2_fault_desc *s2fd)
 +{
 +	u64 level;
 +
 +	if (!kvm_s2_fault_is_perm(s2fd))
 +		return 0;
 +	level = s2fd->esr & ESR_ELx_FSC_LEVEL;
 +	return BIT(ARM64_HW_PGTABLE_LEVEL_SHIFT(level));
 +}
 +
 +static int kvm_s2_fault_get_vma_info(const struct kvm_s2_fault_desc *s2fd,
 +				     struct kvm_s2_fault_vma_info *s2vi);
 +
 +static gfn_t get_canonical_gfn(const struct kvm_s2_fault_desc *s2fd,
 +			       const struct kvm_s2_fault_vma_info *s2vi);
 +
 +static int gmem_abort(const struct kvm_s2_fault_desc *s2fd,
 +		      struct kvm_s2_fault_result *result)
  {
  	bool write_fault, exec_fault;
 -	bool perm_fault = kvm_vcpu_trap_is_permission_fault(s2fd->vcpu);
 +	bool perm_fault = kvm_s2_fault_is_perm(s2fd);
  	enum kvm_pgtable_walk_flags flags = KVM_PGTABLE_WALK_SHARED;
  	enum kvm_pgtable_prot prot = KVM_PGTABLE_PROT_R;
 -	struct kvm_pgtable *pgt = s2fd->vcpu->arch.hw_mmu->pgt;
 -	unsigned long mmu_seq;
 +	struct kvm_pgtable *pgt = s2fd->mmu->pgt;
 +	struct kvm_guest_s2_mapping *mapping = NULL;
 +	struct kvm_s2_fault_vma_info s2vi = {};
  	struct kvm *kvm = s2fd->vcpu->kvm;
  	void *memcache = NULL;
 -	kvm_pfn_t pfn;
 -	gfn_t gfn;
 +	gfn_t canonical_gfn;
  	int ret;
  
  	if (!perm_fault) {
  		memcache = get_mmu_memcache(s2fd->vcpu);
 -		ret = topup_mmu_memcache(s2fd->vcpu, memcache);
 +		ret = topup_mmu_memcache(s2fd->mmu, memcache);
  		if (ret)
  			return ret;
 +		if (kvm_is_nested_s2_mmu(kvm, pgt->mmu)) {
 +			mapping = kmalloc_obj(struct kvm_guest_s2_mapping, GFP_KERNEL_ACCOUNT);
 +			if (!mapping)
 +				return -ENOMEM;
 +		}
  	}
  
 -	if (s2fd->nested)
 -		gfn = kvm_s2_trans_output(s2fd->nested) >> PAGE_SHIFT;
 -	else
 -		gfn = s2fd->fault_ipa >> PAGE_SHIFT;
 +	s2vi.vma_pagesize = PAGE_SIZE;
 +	s2vi.gfn = ALIGN_DOWN(s2fd->fault_ipa, s2vi.vma_pagesize) >> PAGE_SHIFT;
 +	canonical_gfn = get_canonical_gfn(s2fd, &s2vi);
  
 -	write_fault = kvm_is_write_fault(s2fd->vcpu);
 -	exec_fault = kvm_vcpu_trap_is_exec_fault(s2fd->vcpu);
 +	write_fault = kvm_s2_fault_is_write(s2fd);
 +	exec_fault = kvm_s2_fault_is_exec(s2fd);
  
  	VM_WARN_ON_ONCE(write_fault && exec_fault);
  
 -	mmu_seq = kvm->mmu_invalidate_seq;
 +	s2vi.mmu_seq = kvm->mmu_invalidate_seq;
  	/* Pairs with the smp_wmb() in kvm_mmu_invalidate_end(). */
  	smp_rmb();
  
- 	ret = kvm_gmem_get_pfn(kvm, s2fd->memslot, canonical_gfn, &s2vi.pfn, &s2vi.page, NULL);
 -	ret = kvm_gmem_get_pfn(kvm, s2fd->memslot, gfn, &pfn, NULL);
++	ret = kvm_gmem_get_pfn(kvm, s2fd->memslot, canonical_gfn, &s2vi.pfn, NULL);
  	if (ret) {
 -		kvm_prepare_memory_fault_exit(s2fd->vcpu, s2fd->fault_ipa, PAGE_SIZE,
 -					      write_fault, exec_fault, false);
 -		return ret;
 +		/* If result is non-NULL this is a synthetic fault. */
 +		if (!result)
 +			kvm_prepare_memory_fault_exit(s2fd->vcpu, gfn_to_gpa(canonical_gfn),
 +						      s2vi.vma_pagesize, write_fault, exec_fault,
 +						      false);
 +		kfree(mapping);
  	}
  
  	if (!(s2fd->memslot->flags & KVM_MEM_READONLY))
@@@ -1755,7 -1669,7 +1755,7 @@@
  		prot = adjust_nested_exec_perms(kvm, s2fd->nested, prot);
  
  	kvm_fault_lock(kvm);
 -	if (mmu_invalidate_retry(kvm, mmu_seq)) {
 +	if (mmu_invalidate_retry(kvm, s2vi.mmu_seq)) {
  		ret = -EAGAIN;
  		goto out_unlock;
  	}
@@@ -1766,68 -1680,60 +1766,67 @@@
  		 * PTE, which will be preserved.
  		 */
  		prot &= ~KVM_NV_GUEST_MAP_SZ;
 -		ret = KVM_PGT_FN(kvm_pgtable_stage2_relax_perms)(pgt, s2fd->fault_ipa,
 +		ret = KVM_PGT_FN(kvm_pgtable_stage2_relax_perms)(pgt, gfn_to_gpa(s2vi.gfn),
  								 prot, flags);
  	} else {
 -		ret = KVM_PGT_FN(kvm_pgtable_stage2_map)(pgt, s2fd->fault_ipa, PAGE_SIZE,
 -							 __pfn_to_phys(pfn), prot,
 +		ret = KVM_PGT_FN(kvm_pgtable_stage2_map)(pgt, gfn_to_gpa(s2vi.gfn),
 +							 s2vi.vma_pagesize,
 +							 __pfn_to_phys(s2vi.pfn), prot,
  							 memcache, flags);
 +		/*
 +		 * -EAGAIN from kvm_pgtable_stage2_map() can install mappings.
 +		 * We don't know which subrange is installed, track the whole
 +		 * thing.
 +		 */
 +		if ((ret == 0 || ret == -EAGAIN) && kvm_is_nested_s2_mmu(kvm, pgt->mmu)) {
 +			kvm_record_guest_s2_mapping(pgt->mmu, canonical_gfn << PAGE_SHIFT,
 +						    s2fd->fault_ipa, s2vi.vma_pagesize, mapping);
 +			mapping = NULL;
 +		}
  	}
  
  out_unlock:
- 	kvm_release_faultin_page(kvm, s2vi.page, !!ret, prot & KVM_PGTABLE_PROT_W);
  	kvm_fault_unlock(kvm);
 +	kfree(mapping);
  
  	if ((prot & KVM_PGTABLE_PROT_W) && !ret)
 -		mark_page_dirty_in_slot(kvm, s2fd->memslot, gfn);
 +		mark_page_dirty_in_slot(kvm, s2fd->memslot, canonical_gfn);
  
 -	return ret != -EAGAIN ? ret : 0;
 +	if (ret == -EAGAIN)
 +		return result ? ret : 0;
 +
 +	if (result && !ret)
 +		result->mapping_size = s2vi.vma_pagesize;
 +
 +	return ret;
  }
  
 -struct kvm_s2_fault_vma_info {
 -	unsigned long	mmu_seq;
 -	long		vma_pagesize;
 -	vm_flags_t	vm_flags;
 -	unsigned long	max_map_size;
 -	struct page	*page;
 -	kvm_pfn_t	pfn;
 -	gfn_t		gfn;
 -	bool		device;
 -	bool		mte_allowed;
 -	bool		is_vma_cacheable;
 -	bool		map_writable;
 -	bool		map_non_cacheable;
 -};
 -
  static int pkvm_mem_abort(const struct kvm_s2_fault_desc *s2fd)
  {
  	unsigned int flags = FOLL_HWPOISON | FOLL_LONGTERM | FOLL_WRITE;
 +	struct kvm_s2_fault_vma_info s2vi = {};
  	struct kvm_vcpu *vcpu = s2fd->vcpu;
 -	struct kvm_pgtable *pgt = vcpu->arch.hw_mmu->pgt;
 +	struct kvm_pgtable *pgt = s2fd->mmu->pgt;
  	struct mm_struct *mm = current->mm;
  	struct kvm *kvm = vcpu->kvm;
  	void *hyp_memcache;
 -	struct page *page;
  	int ret;
  
  	hyp_memcache = get_mmu_memcache(vcpu);
 -	ret = topup_mmu_memcache(vcpu, hyp_memcache);
 +	ret = topup_mmu_memcache(s2fd->mmu, hyp_memcache);
  	if (ret)
  		return -ENOMEM;
  
 +	ret = kvm_s2_fault_get_vma_info(s2fd, &s2vi);
 +	if (ret)
 +		return ret;
 +
  	ret = account_locked_vm(mm, 1, true);
  	if (ret)
  		return ret;
  
  	mmap_read_lock(mm);
 -	ret = pin_user_pages(s2fd->hva, 1, flags, &page);
 +	ret = pin_user_pages(s2fd->hva, 1, flags, &s2vi.page);
  	mmap_read_unlock(mm);
  
  	if (ret == -EHWPOISON) {
@@@ -1837,7 -1743,7 +1836,7 @@@
  	} else if (ret != 1) {
  		ret = -EFAULT;
  		goto dec_account;
 -	} else if (!folio_test_swapbacked(page_folio(page))) {
 +	} else if (!folio_test_swapbacked(page_folio(s2vi.page))) {
  		/*
  		 * We really can't deal with page-cache pages returned by GUP
  		 * because (a) we may trigger writeback of a page for which we
@@@ -1857,8 -1763,8 +1856,8 @@@
  	}
  
  	write_lock(&kvm->mmu_lock);
 -	ret = pkvm_pgtable_stage2_map(pgt, s2fd->fault_ipa, PAGE_SIZE,
 -				      page_to_phys(page), KVM_PGTABLE_PROT_RWX,
 +	ret = pkvm_pgtable_stage2_map(pgt, gfn_to_gpa(s2vi.gfn), PAGE_SIZE,
 +				      page_to_phys(s2vi.page), KVM_PGTABLE_PROT_RWX,
  				      hyp_memcache, 0);
  	write_unlock(&kvm->mmu_lock);
  	if (ret) {
@@@ -1869,7 -1775,7 +1868,7 @@@
  
  	return 0;
  unpin:
 -	unpin_user_pages(&page, 1);
 +	unpin_user_page(s2vi.page);
  dec_account:
  	account_locked_vm(mm, 1, false);
  	return ret;
@@@ -1886,13 -1792,13 +1885,13 @@@ static short kvm_s2_resolve_vma_size(co
  		vma_shift = PAGE_SHIFT;
  	} else {
  		s2vi->max_map_size = PUD_SIZE;
 -		vma_shift = get_vma_page_shift(vma, s2fd->hva);
 +		vma_shift = get_vma_page_shift(vma);
  	}
  
  	switch (vma_shift) {
  #ifndef __PAGETABLE_PMD_FOLDED
  	case PUD_SHIFT:
 -		if (fault_supports_stage2_huge_mapping(s2fd->memslot, s2fd->hva, PUD_SIZE))
 +		if (fault_supports_stage2_huge_mapping(s2fd, PUD_SIZE))
  			break;
  		fallthrough;
  #endif
@@@ -1900,7 -1806,7 +1899,7 @@@
  		vma_shift = PMD_SHIFT;
  		fallthrough;
  	case PMD_SHIFT:
 -		if (fault_supports_stage2_huge_mapping(s2fd->memslot, s2fd->hva, PMD_SIZE))
 +		if (fault_supports_stage2_huge_mapping(s2fd, PMD_SIZE))
  			break;
  		fallthrough;
  	case CONT_PTE_SHIFT:
@@@ -1941,6 -1847,11 +1940,6 @@@
  	return vma_shift;
  }
  
 -static bool kvm_s2_fault_is_perm(const struct kvm_s2_fault_desc *s2fd)
 -{
 -	return kvm_vcpu_trap_is_permission_fault(s2fd->vcpu);
 -}
 -
  static int kvm_s2_fault_get_vma_info(const struct kvm_s2_fault_desc *s2fd,
  				     struct kvm_s2_fault_vma_info *s2vi)
  {
@@@ -2006,11 -1917,13 +2005,11 @@@ static int kvm_s2_fault_pin_pfn(const s
  		return ret;
  
  	s2vi->pfn = __kvm_faultin_pfn(s2fd->memslot, get_canonical_gfn(s2fd, s2vi),
 -				      kvm_is_write_fault(s2fd->vcpu) ? FOLL_WRITE : 0,
 +				      kvm_s2_fault_is_write(s2fd) ? FOLL_WRITE : 0,
  				      &s2vi->map_writable, &s2vi->page);
  	if (unlikely(is_error_noslot_pfn(s2vi->pfn))) {
 -		if (s2vi->pfn == KVM_PFN_ERR_HWPOISON) {
 -			kvm_send_hwpoison_signal(s2fd->hva, __ffs(s2vi->vma_pagesize));
 -			return 0;
 -		}
 +		if (s2vi->pfn == KVM_PFN_ERR_HWPOISON)
 +			return -EHWPOISON;
  		return -EFAULT;
  	}
  
@@@ -2037,6 -1950,16 +2036,6 @@@
  				return -EFAULT;
  			}
  		} else {
 -			/*
 -			 * If the page was identified as device early by looking at
 -			 * the VMA flags, vma_pagesize is already representing the
 -			 * largest quantity we can map.  If instead it was mapped
 -			 * via __kvm_faultin_pfn(), vma_pagesize is set to PAGE_SIZE
 -			 * and must not be upgraded.
 -			 *
 -			 * In both cases, we don't let transparent_hugepage_adjust()
 -			 * change things at the last minute.
 -			 */
  			s2vi->map_non_cacheable = true;
  		}
  
@@@ -2052,7 -1975,7 +2051,7 @@@ static int kvm_s2_fault_compute_prot(co
  {
  	struct kvm *kvm = s2fd->vcpu->kvm;
  
 -	if (kvm_vcpu_trap_is_exec_fault(s2fd->vcpu) && s2vi->map_non_cacheable)
 +	if (kvm_s2_fault_is_exec(s2fd) && s2vi->map_non_cacheable)
  		return -ENOEXEC;
  
  	/*
@@@ -2061,7 -1984,7 +2060,7 @@@
  	 * and trigger the exception here. Since the memslot is valid, inject
  	 * the fault back to the guest.
  	 */
 -	if (esr_fsc_is_excl_atomic_fault(kvm_vcpu_get_esr(s2fd->vcpu))) {
 +	if (esr_fsc_is_excl_atomic_fault(s2fd->esr)) {
  		kvm_inject_dabt_excl_atomic(s2fd->vcpu, kvm_vcpu_get_hfar(s2fd->vcpu));
  		return 1;
  	}
@@@ -2070,13 -1993,13 +2069,13 @@@
  
  	if (s2vi->map_writable && (s2vi->device ||
  				   !memslot_is_logging(s2fd->memslot) ||
 -				   kvm_is_write_fault(s2fd->vcpu)))
 +				   kvm_s2_fault_is_write(s2fd)))
  		*prot |= KVM_PGTABLE_PROT_W;
  
  	if (s2fd->nested)
  		*prot = adjust_nested_fault_perms(s2fd->nested, *prot);
  
 -	if (kvm_vcpu_trap_is_exec_fault(s2fd->vcpu))
 +	if (kvm_s2_fault_is_exec(s2fd))
  		*prot |= KVM_PGTABLE_PROT_X;
  
  	if (s2vi->map_non_cacheable)
@@@ -2100,14 -2023,11 +2099,14 @@@
  static int kvm_s2_fault_map(const struct kvm_s2_fault_desc *s2fd,
  			    const struct kvm_s2_fault_vma_info *s2vi,
  			    enum kvm_pgtable_prot prot,
 -			    void *memcache)
 +			    void *memcache,
 +			    struct kvm_s2_fault_result *result)
  {
  	enum kvm_pgtable_walk_flags flags = KVM_PGTABLE_WALK_SHARED;
 +	struct kvm_guest_s2_mapping *mapping = NULL;
  	bool writable = prot & KVM_PGTABLE_PROT_W;
  	struct kvm *kvm = s2fd->vcpu->kvm;
 +	phys_addr_t canonical_ipa;
  	struct kvm_pgtable *pgt;
  	long perm_fault_granule;
  	long mapping_size;
@@@ -2115,43 -2035,35 +2114,43 @@@
  	gfn_t gfn;
  	int ret;
  
 +	if (kvm_is_nested_s2_mmu(kvm, s2fd->mmu)) {
 +		mapping = kmalloc_obj(struct kvm_guest_s2_mapping,
 +				      GFP_KERNEL_ACCOUNT);
 +		if (!mapping) {
 +			kvm_release_page_unused(s2vi->page);
 +			return -ENOMEM;
 +		}
 +	}
 +
  	kvm_fault_lock(kvm);
 -	pgt = s2fd->vcpu->arch.hw_mmu->pgt;
 +	pgt = s2fd->mmu->pgt;
  	ret = -EAGAIN;
  	if (mmu_invalidate_retry(kvm, s2vi->mmu_seq))
  		goto out_unlock;
  
 -	perm_fault_granule = (kvm_s2_fault_is_perm(s2fd) ?
 -			      kvm_vcpu_trap_get_perm_fault_granule(s2fd->vcpu) : 0);
 +	perm_fault_granule = kvm_s2_perm_fault_granule(s2fd);
  	mapping_size = s2vi->vma_pagesize;
  	pfn = s2vi->pfn;
  	gfn = s2vi->gfn;
 +	canonical_ipa = gfn_to_gpa(get_canonical_gfn(s2fd, s2vi));
  
  	/*
  	 * If we are not forced to use page mapping, check if we are
 -	 * backed by a THP and thus use block mapping if possible.
 +	 * backed by a huge stage-1 mapping and thus use block mapping if
 +	 * possible.
  	 */
 -	if (mapping_size == PAGE_SIZE &&
 -	    !(s2vi->max_map_size == PAGE_SIZE || s2vi->map_non_cacheable)) {
 +	if (mapping_size == PAGE_SIZE && s2vi->max_map_size != PAGE_SIZE) {
  		if (perm_fault_granule > PAGE_SIZE) {
  			mapping_size = perm_fault_granule;
  		} else {
 -			mapping_size = transparent_hugepage_adjust(kvm, s2fd->memslot,
 -								   s2fd->hva, &pfn,
 -								   &gfn);
 +			mapping_size = transparent_hugepage_adjust(s2fd, &pfn, &gfn);
  			if (mapping_size < 0) {
  				ret = mapping_size;
  				goto out_unlock;
  			}
  		}
 +		canonical_ipa = ALIGN_DOWN(canonical_ipa, mapping_size);
  	}
  
  	if (!perm_fault_granule && !s2vi->map_non_cacheable && kvm_has_mte(kvm))
@@@ -2174,45 -2086,31 +2173,45 @@@
  		ret = KVM_PGT_FN(kvm_pgtable_stage2_map)(pgt, gfn_to_gpa(gfn), mapping_size,
  							 __pfn_to_phys(pfn), prot,
  							 memcache, flags);
 +		/*
 +		 * -EAGAIN from kvm_pgtable_stage2_map() can install mappings.
 +		 * We don't know which subrange is installed, track the whole
 +		 * thing.
 +		 */
 +		if ((ret == 0 || ret == -EAGAIN) && kvm_is_nested_s2_mmu(kvm, pgt->mmu)) {
 +			kvm_record_guest_s2_mapping(pgt->mmu, canonical_ipa,
 +						    gfn_to_gpa(gfn), mapping_size, mapping);
 +			mapping = NULL;
 +		}
  	}
  
  out_unlock:
  	kvm_release_faultin_page(kvm, s2vi->page, !!ret, writable);
  	kvm_fault_unlock(kvm);
 +	kfree(mapping);
  
  	/*
  	 * Mark the page dirty only if the fault is handled successfully,
  	 * making sure we adjust the canonical IPA if the mapping size has
  	 * been updated (via a THP upgrade, for example).
  	 */
 -	if (writable && !ret) {
 -		phys_addr_t ipa = gfn_to_gpa(get_canonical_gfn(s2fd, s2vi));
 -		ipa &= ~(mapping_size - 1);
 -		mark_page_dirty_in_slot(kvm, s2fd->memslot, gpa_to_gfn(ipa));
 -	}
 +	if (writable && !ret)
 +		mark_page_dirty_in_slot(kvm, s2fd->memslot,
 +					gpa_to_gfn(canonical_ipa));
  
 -	if (ret != -EAGAIN)
 -		return ret;
 -	return 0;
 +	if (ret == -EAGAIN)
 +		return result ? ret : 0;
 +
 +	if (result && !ret)
 +		result->mapping_size = mapping_size;
 +
 +	return ret;
  }
  
 -static int user_mem_abort(const struct kvm_s2_fault_desc *s2fd)
 +static int user_mem_abort(const struct kvm_s2_fault_desc *s2fd,
 +			  struct kvm_s2_fault_result *result)
  {
 -	bool perm_fault = kvm_vcpu_trap_is_permission_fault(s2fd->vcpu);
 +	bool perm_fault = kvm_s2_fault_is_perm(s2fd);
  	struct kvm_s2_fault_vma_info s2vi = {};
  	enum kvm_pgtable_prot prot;
  	void *memcache;
@@@ -2230,19 -2128,16 +2229,19 @@@
  	memcache = get_mmu_memcache(s2fd->vcpu);
  	if (!perm_fault || memslot_is_logging(s2fd->memslot) ||
  	    is_protected_kvm_enabled()) {
 -		ret = topup_mmu_memcache(s2fd->vcpu, memcache);
 +		ret = topup_mmu_memcache(s2fd->mmu, memcache);
  		if (ret)
  			return ret;
  	}
  
 -	/*
 -	 * Let's check if we will get back a huge page backed by hugetlbfs, or
 -	 * get block mapping for device MMIO region.
 -	 */
  	ret = kvm_s2_fault_pin_pfn(s2fd, &s2vi);
 +	if (ret == -EHWPOISON) {
 +		/* If result is specified, let the caller handle this. */
 +		if (result)
 +			return -EHWPOISON;
 +		kvm_send_hwpoison_signal(s2fd->hva, __ffs(s2vi.vma_pagesize));
 +		return 0;
 +	}
  	if (ret != 1)
  		return ret;
  
@@@ -2252,7 -2147,7 +2251,7 @@@
  		return ret;
  	}
  
 -	return kvm_s2_fault_map(s2fd, &s2vi, prot, memcache);
 +	return kvm_s2_fault_map(s2fd, &s2vi, prot, memcache, result);
  }
  
  /* Resolve the access fault by making the page young again. */
@@@ -2362,8 -2257,7 +2361,8 @@@ int kvm_handle_guest_sea(struct kvm_vcp
  int kvm_handle_guest_abort(struct kvm_vcpu *vcpu)
  {
  	struct kvm_s2_trans nested_trans, *nested = NULL;
 -	unsigned long esr;
 +	unsigned long esr = kvm_vcpu_get_esr(vcpu);
 +	struct kvm_s2_mmu *mmu = vcpu->arch.hw_mmu;
  	phys_addr_t fault_ipa; /* The address we faulted on */
  	phys_addr_t ipa; /* Always the IPA in the L1 guest phys space */
  	struct kvm_memory_slot *memslot;
@@@ -2372,9 -2266,11 +2371,9 @@@
  	gfn_t gfn;
  	int ret, idx;
  
 -	if (kvm_vcpu_abt_issea(vcpu))
 +	if (esr_abt_is_sea(esr))
  		return kvm_handle_guest_sea(vcpu);
  
 -	esr = kvm_vcpu_get_esr(vcpu);
 -
  	/*
  	 * The fault IPA should be reliable at this point as we're not dealing
  	 * with an SEA.
@@@ -2383,7 -2279,7 +2382,7 @@@
  	if (KVM_BUG_ON(ipa == INVALID_GPA, vcpu->kvm))
  		return -EFAULT;
  
 -	is_iabt = kvm_vcpu_trap_is_iabt(vcpu);
 +	is_iabt = esr_trap_is_iabt(esr);
  
  	if (esr_fsc_is_translation_fault(esr)) {
  		/* Beyond sanitised PARange (which is the IPA limit) */
@@@ -2393,14 -2289,14 +2392,14 @@@
  		}
  
  		/* Falls between the IPA range and the PARange? */
 -		if (fault_ipa >= BIT_ULL(VTCR_EL2_IPA(vcpu->arch.hw_mmu->vtcr))) {
 +		if (fault_ipa >= BIT_ULL(VTCR_EL2_IPA(mmu->vtcr))) {
  			fault_ipa |= FAR_TO_FIPA_OFFSET(kvm_vcpu_get_hfar(vcpu));
  
  			return kvm_inject_sea(vcpu, is_iabt, fault_ipa);
  		}
  	}
  
 -	trace_kvm_guest_fault(*vcpu_pc(vcpu), kvm_vcpu_get_esr(vcpu),
 +	trace_kvm_guest_fault(*vcpu_pc(vcpu), esr,
  			      kvm_vcpu_get_hfar(vcpu), fault_ipa);
  
  	/* Check the stage-2 fault is trans. fault or write fault */
@@@ -2408,10 -2304,10 +2407,10 @@@
  	    !esr_fsc_is_permission_fault(esr) &&
  	    !esr_fsc_is_access_flag_fault(esr) &&
  	    !esr_fsc_is_excl_atomic_fault(esr)) {
 -		kvm_err("Unsupported FSC: EC=%#x xFSC=%#lx ESR_EL2=%#lx\n",
 -			kvm_vcpu_trap_get_class(vcpu),
 -			(unsigned long)kvm_vcpu_trap_get_fault(vcpu),
 -			(unsigned long)kvm_vcpu_get_esr(vcpu));
 +		kvm_err("Unsupported FSC: EC=%#lx xFSC=%#lx ESR_EL2=%#lx\n",
 +			ESR_ELx_EC(esr),
 +			(unsigned long)(esr & ESR_ELx_FSC),
 +			(unsigned long)esr);
  		return -EFAULT;
  	}
  
@@@ -2430,8 -2326,8 +2429,8 @@@
  	 * nothing to walk and we treat it as a 1:1 before going through the
  	 * canonical translation.
  	 */
 -	if (kvm_is_nested_s2_mmu(vcpu->kvm,vcpu->arch.hw_mmu) &&
 -	    vcpu->arch.hw_mmu->nested_stage2_enabled) {
 +	if (kvm_is_nested_s2_mmu(vcpu->kvm, mmu) &&
 +	    mmu->nested_stage2_enabled) {
  		u32 esr;
  
  		ret = kvm_walk_nested_s2(vcpu, fault_ipa, &nested_trans);
@@@ -2460,7 -2356,7 +2459,7 @@@
  	gfn = ipa >> PAGE_SHIFT;
  	memslot = gfn_to_memslot(vcpu->kvm, gfn);
  	hva = gfn_to_hva_memslot_prot(memslot, gfn, &writable);
 -	write_fault = kvm_is_write_fault(vcpu);
 +	write_fault = esr_abt_is_write_fault(esr);
  	if (kvm_is_error_hva(hva) || (write_fault && !writable)) {
  		/*
  		 * The guest has put either its instructions or its page-tables
@@@ -2473,7 -2369,7 +2472,7 @@@
  			goto out;
  		}
  
 -		if (kvm_vcpu_abt_iss1tw(vcpu)) {
 +		if (esr_abt_is_s1ptw(esr)) {
  			ret = kvm_inject_sea_dabt(vcpu, kvm_vcpu_get_hfar(vcpu));
  			goto out_unlock;
  		}
@@@ -2488,7 -2384,7 +2487,7 @@@
  		 * So let's assume that the guest is just being
  		 * cautious, and skip the instruction.
  		 */
 -		if (kvm_is_error_hva(hva) && kvm_vcpu_dabt_is_cm(vcpu)) {
 +		if (kvm_is_error_hva(hva) && esr_dabt_is_cm(esr)) {
  			kvm_incr_pc(vcpu);
  			ret = 1;
  			goto out_unlock;
@@@ -2506,7 -2402,7 +2505,7 @@@
  	}
  
  	/* Userspace should not be able to register out-of-bounds IPAs */
 -	VM_BUG_ON(ipa >= kvm_phys_size(vcpu->arch.hw_mmu));
 +	VM_BUG_ON(ipa >= kvm_phys_size(mmu));
  
  	if (esr_fsc_is_access_flag_fault(esr)) {
  		handle_access_fault(vcpu, fault_ipa);
@@@ -2520,20 -2416,19 +2519,20 @@@
  		.nested		= nested,
  		.memslot	= memslot,
  		.hva		= hva,
 +		.esr		= esr,
 +		.mmu		= mmu,
  	};
  
  	if (kvm_vm_is_protected(vcpu->kvm)) {
  		ret = pkvm_mem_abort(&s2fd);
  	} else {
 -		VM_WARN_ON_ONCE(kvm_vcpu_trap_is_permission_fault(vcpu) &&
 -				!write_fault &&
 -				!kvm_vcpu_trap_is_exec_fault(vcpu));
 +		VM_WARN_ON_ONCE(kvm_s2_fault_is_perm(&s2fd) && !write_fault &&
 +				!kvm_s2_fault_is_exec(&s2fd));
  
  		if (kvm_slot_has_gmem(memslot))
 -			ret = gmem_abort(&s2fd);
 +			ret = gmem_abort(&s2fd, NULL);
  		else
 -			ret = user_mem_abort(&s2fd);
 +			ret = user_mem_abort(&s2fd, NULL);
  	}
  
  	if (ret == 0)
@@@ -2548,16 -2443,14 +2547,16 @@@ out_unlock
  
  bool kvm_unmap_gfn_range(struct kvm *kvm, struct kvm_gfn_range *range)
  {
 +	gpa_t gpa = range->start << PAGE_SHIFT;
 +	size_t size = (range->end - range->start) << PAGE_SHIFT;
 +	bool may_block = range->may_block;
 +
  	if (!kvm->arch.mmu.pgt || kvm_vm_is_protected(kvm))
  		return false;
  
 -	__unmap_stage2_range(&kvm->arch.mmu, range->start << PAGE_SHIFT,
 -			     (range->end - range->start) << PAGE_SHIFT,
 -			     range->may_block);
 +	__unmap_stage2_range(&kvm->arch.mmu, gpa, size, may_block);
 +	kvm_nested_unmap_cipa_range(kvm, gpa, size, may_block);
  
 -	kvm_nested_s2_unmap(kvm, range->may_block);
  	return false;
  }
  
@@@ -2839,7 -2732,7 +2838,7 @@@ void kvm_arch_flush_shadow_memslot(stru
  
  	write_lock(&kvm->mmu_lock);
  	kvm_stage2_unmap_range(&kvm->arch.mmu, gpa, size, true);
 -	kvm_nested_s2_unmap(kvm, true);
 +	kvm_nested_unmap_cipa_range(kvm, gpa, size, true);
  	write_unlock(&kvm->mmu_lock);
  }
  
@@@ -2910,146 -2803,3 +2909,146 @@@ void kvm_toggle_cache(struct kvm_vcpu *
  
  	trace_kvm_toggle_cache(*vcpu_pc(vcpu), was_enabled, now_enabled);
  }
 +
 +/*
 + * Try to walk to the specified GPA in canonical mmu - if unmapped returns 0, if
 + * mapped returns the granule size, otherwise returns an error.
 + */
 +static long kvm_walk_s2(struct kvm_pgtable *pgt,
 +			gpa_t gpa, s8 *level)
 +{
 +	struct kvm *kvm = kvm_s2_mmu_to_kvm(pgt->mmu);
 +	kvm_pte_t pte;
 +	long ret;
 +
 +	guard(read_lock)(&kvm->mmu_lock);
 +
 +	ret = kvm_pgtable_get_leaf(pgt, gpa, &pte, level,
 +				   KVM_PGTABLE_WALK_SHARED);
 +	if (ret)
 +		return ret;
 +	/* Unpopulated, must fault. */
 +	if (!kvm_pte_valid(pte))
 +		return 0;
 +	return kvm_granule_size(*level);
 +}
 +
 +/* Synthesised data abort at specified page table level. */
 +#define PRE_FAULT_ESR(level)				\
 +	 ((ESR_ELx_EC_DABT_LOW << ESR_ELx_EC_SHIFT) |	\
 +	  ESR_ELx_IL | ESR_ELx_FSC_FAULT_L(level))
 +
 +/* Retrieve either a read-only or a read/write hva. */
 +static hva_t gfn_to_hva_memslot_read(struct kvm_memory_slot *slot, gfn_t gfn)
 +{
 +	return gfn_to_hva_memslot_prot(slot, gfn, /*writable=*/NULL);
 +}
 +
 +static long __pre_fault_s2(struct kvm_s2_mmu *mmu, struct kvm_vcpu *vcpu,
 +			   gpa_t gpa, struct kvm_memory_slot *memslot, s8 level)
 +{
 +	const bool is_gmem = kvm_slot_has_gmem(memslot);
 +	const gfn_t gfn = gpa_to_gfn(gpa);
 +	const hva_t hva = is_gmem ? 0 : gfn_to_hva_memslot_read(memslot, gfn);
 +	const struct kvm_s2_fault_desc s2fd = {
 +		.vcpu		= vcpu,
 +		.fault_ipa	= gpa,
 +		.nested		= NULL,
 +		.memslot	= memslot,
 +		.hva		= hva,
 +		.esr		= PRE_FAULT_ESR(level),
 +		.mmu		= mmu,
 +	};
 +	struct kvm_s2_fault_result result = {};
 +	long ret;
 +
 +	if (kvm_is_error_hva(hva))
 +		return -EFAULT;
 +
 +	if (is_gmem)
 +		ret = gmem_abort(&s2fd, &result);
 +	else
 +		ret = user_mem_abort(&s2fd, &result);
 +	if (IS_ERR_VALUE(ret))
 +		return ret;
 +	return result.mapping_size;
 +}
 +
 +static long pre_fault_s2(struct kvm_s2_mmu *mmu, struct kvm_vcpu *vcpu,
 +			 gpa_t gpa, struct kvm_memory_slot *memslot)
 +{
 +	s8 level;
 +	long ret;
 +
 +	/* Try a walk first. */
 +	ret = kvm_walk_s2(mmu->pgt, gpa, &level);
 +	if (ret)
 +		return ret;
 +	/* OK, have to fault page in. */
 +	return __pre_fault_s2(mmu, vcpu, gpa, memslot, level);
 +}
 +
 +static unsigned long
 +pre_fault_bytes_consumed(gpa_t gpa, unsigned long granule_size,
 +			 unsigned long bytes_remaining)
 +{
 +	/* Granules are always a power-of-2. */
 +	const unsigned long granule_bytes_remaining =
 +		granule_size - (gpa % granule_size);
 +
 +	return min(granule_bytes_remaining, bytes_remaining);
 +}
 +
 +/* If you lose the race this many times, time to give up. */
 +#define MAX_PRE_FAULT_RETRIES 3
 +
 +int kvm_arch_pre_fault_allowed(struct kvm_vcpu *vcpu)
 +{
 +	if (is_protected_kvm_enabled())
 +		return -EOPNOTSUPP;
 +	if (!kvm_vcpu_initialized(vcpu))
 +		return -ENOEXEC;
 +
 +	return 0;
 +}
 +
 +/**
 + * kvm_arch_vcpu_pre_fault_memory - pre-fault stage-2 page tables for the
 + * specified GPA.
 + * @vcpu:	The VCPU pointer
 + * @range:	{gpa, size, flags} tuple
 + *
 + * The mapping performed is always best-effort - faulting in is necessarily
 + * racey. The ranges faulted in are canonical, nested page tables are ignored.
 + *
 + * @range->gpa specifies the GPA to pre-fault, @range->size specifies how many
 + * bytes remain to be pre-faulted and @range->flags is reserved and must be 0.
 + *
 + * Returns: the number of bytes the pre-fault consumed, or an error.
 + */
 +long kvm_arch_vcpu_pre_fault_memory(struct kvm_vcpu *vcpu,
 +				    struct kvm_pre_fault_memory *range)
 +{
 +	struct kvm *kvm = vcpu->kvm;
 +	const u64 bytes_remaining = range->size;
 +	struct kvm_s2_mmu *mmu = &kvm->arch.mmu; /* Canonical. */
 +	struct kvm_memory_slot *memslot;
 +	const gpa_t gpa = range->gpa;
 +	int num_retries = 0;
 +	long ret;
 +
 +	memslot = gfn_to_memslot(kvm, gpa_to_gfn(gpa));
 +	if (!memslot)
 +		return -ENOENT;
 +	/* SRCU must be released for progress and only userland can do that. */
 +	if (memslot->flags & KVM_MEMSLOT_INVALID)
 +		return -EAGAIN;
 +
 +	do {
 +		ret = pre_fault_s2(mmu, vcpu, gpa, memslot);
 +	} while (ret == -EAGAIN && num_retries++ < MAX_PRE_FAULT_RETRIES);
 +
 +	if (IS_ERR_VALUE(ret))
 +		return ret;
 +	return pre_fault_bytes_consumed(gpa, ret, bytes_remaining);
 +}

[-- Attachment #2: signature.asc --]
[-- Type: application/pgp-signature, Size: 488 bytes --]

^ permalink raw reply	[flat|nested] 4+ messages in thread

* linux-next: manual merge of the kvm-x86 tree with the kvm-arm tree
@ 2026-01-26 21:43 Mark Brown
  0 siblings, 0 replies; 4+ messages in thread
From: Mark Brown @ 2026-01-26 21:43 UTC (permalink / raw)
  To: Sean Christopherson
  Cc: Linux Kernel Mailing List, Linux Next Mailing List, Marc Zyngier,
	Melody Wang, Michael Roth, Will Deacon, Yicong Yang, Zhou Wang

[-- Attachment #1: Type: text/plain, Size: 1278 bytes --]

Hi all,

Today's linux-next merge of the kvm-x86 tree got a conflict in:

  include/uapi/linux/kvm.h

between commit:

  f174a9ffcd48d ("KVM: arm64: Add exit to userspace on {LD,ST}64B* outside of memslots")

from the kvm-arm tree and commit:

  fa9893fadbc24 ("KVM: Introduce KVM_EXIT_SNP_REQ_CERTS for SNP certificate-fetching")

from the kvm-x86 tree.

I fixed it up (see below) and can carry the fix as necessary. This
is now fixed as far as linux-next is concerned, but any non trivial
conflicts should be mentioned to your upstream maintainer when your tree
is submitted for merging.  You may also want to consider cooperating
with the maintainer of the conflicting tree to minimise any particularly
complex conflicts.

diff --cc include/uapi/linux/kvm.h
index 88cca0e22ecea,9b96a6e73a53f..0000000000000
--- a/include/uapi/linux/kvm.h
+++ b/include/uapi/linux/kvm.h
@@@ -180,7 -190,7 +190,8 @@@ struct kvm_exit_snp_req_certs 
  #define KVM_EXIT_MEMORY_FAULT     39
  #define KVM_EXIT_TDX              40
  #define KVM_EXIT_ARM_SEA          41
 -#define KVM_EXIT_SNP_REQ_CERTS    42
 +#define KVM_EXIT_ARM_LDST64B      42
++#define KVM_EXIT_SNP_REQ_CERTS    43
  
  /* For KVM_EXIT_INTERNAL_ERROR */
  /* Emulate instruction failed. */

[-- Attachment #2: signature.asc --]
[-- Type: application/pgp-signature, Size: 488 bytes --]

^ permalink raw reply	[flat|nested] 4+ messages in thread

* linux-next: manual merge of the kvm-x86 tree with the kvm-arm tree
@ 2024-04-30  4:28 Stephen Rothwell
  0 siblings, 0 replies; 4+ messages in thread
From: Stephen Rothwell @ 2024-04-30  4:28 UTC (permalink / raw)
  To: Sean Christopherson, Christoffer Dall, Marc Zyngier
  Cc: Linux Kernel Mailing List, Linux Next Mailing List, Oliver Upton

[-- Attachment #1: Type: text/plain, Size: 1318 bytes --]

Hi all,

Today's linux-next merge of the kvm-x86 tree got a conflict in:

  tools/testing/selftests/kvm/aarch64/psci_test.c

between commit:

  c3c369b508d9 ("KVM: selftests: Use MPIDR_HWID_BITMASK from cputype.h")

from the kvm-arm tree and commit:

  730cfa45b5f4 ("KVM: selftests: Define _GNU_SOURCE for all selftests code")

from the kvm-x86 tree.

I fixed it up (see below) and can carry the fix as necessary. This
is now fixed as far as linux-next is concerned, but any non trivial
conflicts should be mentioned to your upstream maintainer when your tree
is submitted for merging.  You may also want to consider cooperating
with the maintainer of the conflicting tree to minimise any particularly
complex conflicts.

-- 
Cheers,
Stephen Rothwell

diff --cc tools/testing/selftests/kvm/aarch64/psci_test.c
index 9fa3578d47d5,1c8c6f0c1ca3..000000000000
--- a/tools/testing/selftests/kvm/aarch64/psci_test.c
+++ b/tools/testing/selftests/kvm/aarch64/psci_test.c
@@@ -10,12 -10,7 +10,9 @@@
   *  - A test for KVM's handling of PSCI SYSTEM_SUSPEND and the associated
   *    KVM_SYSTEM_EVENT_SUSPEND UAPI.
   */
- 
- #define _GNU_SOURCE
- 
 +#include <linux/kernel.h>
  #include <linux/psci.h>
 +#include <asm/cputype.h>
  
  #include "kvm_util.h"
  #include "processor.h"

[-- Attachment #2: OpenPGP digital signature --]
[-- Type: application/pgp-signature, Size: 488 bytes --]

^ permalink raw reply	[flat|nested] 4+ messages in thread

* linux-next: manual merge of the kvm-x86 tree with the kvm-arm tree
@ 2023-10-30  2:05 Stephen Rothwell
  0 siblings, 0 replies; 4+ messages in thread
From: Stephen Rothwell @ 2023-10-30  2:05 UTC (permalink / raw)
  To: Sean Christopherson, Christoffer Dall, Marc Zyngier
  Cc: Ackerley Tng, Chao Peng, Isaku Yamahata, Jing Zhang,
	Kirill A. Shutemov, Linux Kernel Mailing List,
	Linux Next Mailing List, Michael Roth, Oliver Upton,
	Paolo Bonzini, Yu Zhang

[-- Attachment #1: Type: text/plain, Size: 9232 bytes --]

Hi all,

Today's linux-next merge of the kvm-x86 tree got a conflict in:

  Documentation/virt/kvm/api.rst

between commits:

  6656cda0f3b2 ("KVM: arm64: Document KVM_ARM_GET_REG_WRITABLE_MASKS")
  dafa493dd01d ("KVM: arm64: Document vCPU feature selection UAPIs")

from the kvm-arm tree and commits:

  8e555bf388af ("KVM: Introduce KVM_SET_USER_MEMORY_REGION2")
  e82df88abb18 ("KVM: Introduce per-page memory attributes")
  fcbef1e5e5d2 ("KVM: Add KVM_CREATE_GUEST_MEMFD ioctl() for guest-specific backing memory")
  a2e4643a589a ("KVM: Add transparent hugepage support for dedicated guest memory")

from the kvm-x86 tree.

I fixed it up (see below) and can carry the fix as necessary. This
is now fixed as far as linux-next is concerned, but any non trivial
conflicts should be mentioned to your upstream maintainer when your tree
is submitted for merging.  You may also want to consider cooperating
with the maintainer of the conflicting tree to minimise any particularly
complex conflicts.

-- 
Cheers,
Stephen Rothwell

diff --cc Documentation/virt/kvm/api.rst
index d75c9a7a7193,205b0c3020c2..000000000000
--- a/Documentation/virt/kvm/api.rst
+++ b/Documentation/virt/kvm/api.rst
@@@ -6124,56 -6107,137 +6161,187 @@@ writes to the CNTVCT_EL0 and CNTPCT_EL
  interface. No error will be returned, but the resulting offset will not be
  applied.
  
 -4.139 KVM_SET_USER_MEMORY_REGION2
 +.. _KVM_ARM_GET_REG_WRITABLE_MASKS:
 +
 +4.139 KVM_ARM_GET_REG_WRITABLE_MASKS
 +-------------------------------------------
 +
 +:Capability: KVM_CAP_ARM_SUPPORTED_REG_MASK_RANGES
 +:Architectures: arm64
 +:Type: vm ioctl
 +:Parameters: struct reg_mask_range (in/out)
 +:Returns: 0 on success, < 0 on error
 +
 +
 +::
 +
 +        #define KVM_ARM_FEATURE_ID_RANGE	0
 +        #define KVM_ARM_FEATURE_ID_RANGE_SIZE	(3 * 8 * 8)
 +
 +        struct reg_mask_range {
 +                __u64 addr;             /* Pointer to mask array */
 +                __u32 range;            /* Requested range */
 +                __u32 reserved[13];
 +        };
 +
 +This ioctl copies the writable masks for a selected range of registers to
 +userspace.
 +
 +The ``addr`` field is a pointer to the destination array where KVM copies
 +the writable masks.
 +
 +The ``range`` field indicates the requested range of registers.
 +``KVM_CHECK_EXTENSION`` for the ``KVM_CAP_ARM_SUPPORTED_REG_MASK_RANGES``
 +capability returns the supported ranges, expressed as a set of flags. Each
 +flag's bit index represents a possible value for the ``range`` field.
 +All other values are reserved for future use and KVM may return an error.
 +
 +The ``reserved[13]`` array is reserved for future use and should be 0, or
 +KVM may return an error.
 +
 +KVM_ARM_FEATURE_ID_RANGE (0)
 +^^^^^^^^^^^^^^^^^^^^^^^^^^^^
 +
 +The Feature ID range is defined as the AArch64 System register space with
 +op0==3, op1=={0, 1, 3}, CRn==0, CRm=={0-7}, op2=={0-7}.
 +
 +The mask returned array pointed to by ``addr`` is indexed by the macro
 +``ARM64_FEATURE_ID_RANGE_IDX(op0, op1, crn, crm, op2)``, allowing userspace
 +to know what fields can be changed for the system register described by
 +``op0, op1, crn, crm, op2``. KVM rejects ID register values that describe a
 +superset of the features supported by the system.
 +
++4.140 KVM_SET_USER_MEMORY_REGION2
+ ---------------------------------
+ 
+ :Capability: KVM_CAP_USER_MEMORY2
+ :Architectures: all
+ :Type: vm ioctl
+ :Parameters: struct kvm_userspace_memory_region2 (in)
+ :Returns: 0 on success, -1 on error
+ 
+ KVM_SET_USER_MEMORY_REGION2 is an extension to KVM_SET_USER_MEMORY_REGION that
+ allows mapping guest_memfd memory into a guest.  All fields shared with
+ KVM_SET_USER_MEMORY_REGION identically.  Userspace can set KVM_MEM_PRIVATE in
+ flags to have KVM bind the memory region to a given guest_memfd range of
+ [guest_memfd_offset, guest_memfd_offset + memory_size].  The target guest_memfd
+ must point at a file created via KVM_CREATE_GUEST_MEMFD on the current VM, and
+ the target range must not be bound to any other memory region.  All standard
+ bounds checks apply (use common sense).
+ 
+ ::
+ 
+   struct kvm_userspace_memory_region2 {
+ 	__u32 slot;
+ 	__u32 flags;
+ 	__u64 guest_phys_addr;
+ 	__u64 memory_size; /* bytes */
+ 	__u64 userspace_addr; /* start of the userspace allocated memory */
+   __u64 guest_memfd_offset;
+ 	__u32 guest_memfd;
+ 	__u32 pad1;
+ 	__u64 pad2[14];
+   };
+ 
+ A KVM_MEM_PRIVATE region _must_ have a valid guest_memfd (private memory) and
+ userspace_addr (shared memory).  However, "valid" for userspace_addr simply
+ means that the address itself must be a legal userspace address.  The backing
+ mapping for userspace_addr is not required to be valid/populated at the time of
+ KVM_SET_USER_MEMORY_REGION2, e.g. shared memory can be lazily mapped/allocated
+ on-demand.
+ 
+ When mapping a gfn into the guest, KVM selects shared vs. private, i.e consumes
+ userspace_addr vs. guest_memfd, based on the gfn's KVM_MEMORY_ATTRIBUTE_PRIVATE
+ state.  At VM creation time, all memory is shared, i.e. the PRIVATE attribute
+ is '0' for all gfns.  Userspace can control whether memory is shared/private by
+ toggling KVM_MEMORY_ATTRIBUTE_PRIVATE via KVM_SET_MEMORY_ATTRIBUTES as needed.
+ 
 -4.140 KVM_SET_MEMORY_ATTRIBUTES
++4.141 KVM_SET_MEMORY_ATTRIBUTES
+ -------------------------------
+ 
+ :Capability: KVM_CAP_MEMORY_ATTRIBUTES
+ :Architectures: x86
+ :Type: vm ioctl
+ :Parameters: struct kvm_memory_attributes(in)
+ :Returns: 0 on success, <0 on error
+ 
+ KVM_SET_MEMORY_ATTRIBUTES allows userspace to set memory attributes for a range
+ of guest physical memory.
+ 
+ ::
+ 
+   struct kvm_memory_attributes {
+ 	__u64 address;
+ 	__u64 size;
+ 	__u64 attributes;
+ 	__u64 flags;
+   };
+ 
+   #define KVM_MEMORY_ATTRIBUTE_PRIVATE           (1ULL << 3)
+ 
+ The address and size must be page aligned.  The supported attributes can be
+ retrieved via ioctl(KVM_CHECK_EXTENSION) on KVM_CAP_MEMORY_ATTRIBUTES.  If
+ executed on a VM, KVM_CAP_MEMORY_ATTRIBUTES precisely returns the attributes
+ supported by that VM.  If executed at system scope, KVM_CAP_MEMORY_ATTRIBUTES
+ returns all attributes supported by KVM.  The only attribute defined at this
+ time is KVM_MEMORY_ATTRIBUTE_PRIVATE, which marks the associated gfn as being
+ guest private memory.
+ 
+ Note, there is no "get" API.  Userspace is responsible for explicitly tracking
+ the state of a gfn/page as needed.
+ 
+ The "flags" field is reserved for future extensions and must be '0'.
+ 
 -4.141 KVM_CREATE_GUEST_MEMFD
++4.142 KVM_CREATE_GUEST_MEMFD
+ ----------------------------
+ 
+ :Capability: KVM_CAP_GUEST_MEMFD
+ :Architectures: none
+ :Type: vm ioctl
+ :Parameters: struct struct kvm_create_guest_memfd(in)
+ :Returns: 0 on success, <0 on error
+ 
+ KVM_CREATE_GUEST_MEMFD creates an anonymous file and returns a file descriptor
+ that refers to it.  guest_memfd files are roughly analogous to files created
+ via memfd_create(), e.g. guest_memfd files live in RAM, have volatile storage,
+ and are automatically released when the last reference is dropped.  Unlike
+ "regular" memfd_create() files, guest_memfd files are bound to their owning
+ virtual machine (see below), cannot be mapped, read, or written by userspace,
+ and cannot be resized  (guest_memfd files do however support PUNCH_HOLE).
+ 
+ ::
+ 
+   struct kvm_create_guest_memfd {
+ 	__u64 size;
+ 	__u64 flags;
+ 	__u64 reserved[6];
+   };
+ 
+   #define KVM_GUEST_MEMFD_ALLOW_HUGEPAGE         (1ULL << 0)
+ 
+ Conceptually, the inode backing a guest_memfd file represents physical memory,
+ i.e. is coupled to the virtual machine as a thing, not to a "struct kvm".  The
+ file itself, which is bound to a "struct kvm", is that instance's view of the
+ underlying memory, e.g. effectively provides the translation of guest addresses
+ to host memory.  This allows for use cases where multiple KVM structures are
+ used to manage a single virtual machine, e.g. when performing intrahost
+ migration of a virtual machine.
+ 
+ KVM currently only supports mapping guest_memfd via KVM_SET_USER_MEMORY_REGION2,
+ and more specifically via the guest_memfd and guest_memfd_offset fields in
+ "struct kvm_userspace_memory_region2", where guest_memfd_offset is the offset
+ into the guest_memfd instance.  For a given guest_memfd file, there can be at
+ most one mapping per page, i.e. binding multiple memory regions to a single
+ guest_memfd range is not allowed (any number of memory regions can be bound to
+ a single guest_memfd file, but the bound ranges must not overlap).
+ 
+ If KVM_GUEST_MEMFD_ALLOW_HUGEPAGE is set in flags, KVM will attempt to allocate
+ and map hugepages for the guest_memfd file.  This is currently best effort.  If
+ KVM_GUEST_MEMFD_ALLOW_HUGEPAGE is set, the size must be aligned to the maximum
+ transparent hugepage size supported by the kernel
+ 
+ See KVM_SET_USER_MEMORY_REGION2 for additional details.
+ 
  5. The kvm_run structure
  ========================
  

[-- Attachment #2: OpenPGP digital signature --]
[-- Type: application/pgp-signature, Size: 488 bytes --]

^ permalink raw reply	[flat|nested] 4+ messages in thread

end of thread, other threads:[~2026-10-02 12:54 UTC | newest]

Thread overview: 4+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-10-02 12:54 linux-next: manual merge of the kvm-x86 tree with the kvm-arm tree Mark Brown
  -- strict thread matches above, loose matches on Subject: below --
2026-01-26 21:43 Mark Brown
2024-04-30  4:28 Stephen Rothwell
2023-10-30  2:05 Stephen Rothwell

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®