From: James Houghton <jthoughton@google.com>
To: Will Deacon <will@kernel.org>,
Catalin Marinas <catalin.marinas@arm.com>,
Muchun Song <muchun.song@linux.dev>,
Oscar Salvador <osalvador@suse.de>,
Andrew Morton <akpm@linux-foundation.org>
Cc: Nikos Nikoleris <nikos.nikoleris@arm.com>,
Linu Cherian <linu.cherian@arm.com>,
Mark Rutland <mark.rutland@arm.com>,
David Hildenbrand <david@kernel.org>,
Ryan Roberts <ryan.roberts@arm.com>,
Nanyong Sun <sunnanyong@huawei.com>, Yu Zhao <yuzhao@google.com>,
Frank van der Linden <fvdl@google.com>,
David Rientjes <rientjes@google.com>,
James Houghton <jthoughton@google.com>,
linux-kernel@vger.kernel.org,
linux-arm-kernel@lists.infradead.org, linux-mm@kvack.org
Subject: [PATCH v2 05/20] hugetlb_vmemmap: Use try_update_vmemmap_pte to update in-use PTEs
Date: Sat, 3 Oct 2026 00:21:08 +0000 [thread overview]
Message-ID: <20261003002123.505555-6-jthoughton@google.com> (raw)
In-Reply-To: <20261003002123.505555-1-jthoughton@google.com>
set_pte_at() cannot be used to replace in-use vmemmap PTEs on arm64, so
replace it with a more specific routine, try_update_vmemmap_pte().
try_update_vmemmap_pte() is used for modifying page table entries such
that there is a guarantee that no fault will be taken.
For the generic implementation, please note one difference: set_pte()
does not invoke page_table_check_ptes_set(), but set_pte_at(), the call
we are about to replace, does. However, this is a functional no-op
because page_table_check_ptes_set() does nothing for init_mm PTEs, which
vmemmap PTEs are.
Signed-off-by: James Houghton <jthoughton@google.com>
---
arch/loongarch/include/asm/pgtable.h | 2 +
arch/riscv/include/asm/pgtable.h | 2 +
arch/x86/include/asm/pgtable.h | 2 +
include/linux/pgtable.h | 22 +++++++++++
mm/hugetlb_vmemmap.c | 56 +++++++++++++++++++---------
5 files changed, 67 insertions(+), 17 deletions(-)
diff --git a/arch/loongarch/include/asm/pgtable.h b/arch/loongarch/include/asm/pgtable.h
index f87603135131..123eda3f6035 100644
--- a/arch/loongarch/include/asm/pgtable.h
+++ b/arch/loongarch/include/asm/pgtable.h
@@ -633,6 +633,8 @@ static inline long pmd_protnone(pmd_t pmd)
#define pmd_leaf(pmd) ((pmd_val(pmd) & _PAGE_HUGE) != 0)
#define pud_leaf(pud) ((pud_val(pud) & _PAGE_HUGE) != 0)
+#define ARCH_WANTS_GENERIC_POPULATE_VMEMMAP_PTE
+
/*
* We provide our own get_unmapped area to cope with the virtual aliasing
* constraints placed on us by the cache architecture.
diff --git a/arch/riscv/include/asm/pgtable.h b/arch/riscv/include/asm/pgtable.h
index 4c8fc6845503..6cfae7720882 100644
--- a/arch/riscv/include/asm/pgtable.h
+++ b/arch/riscv/include/asm/pgtable.h
@@ -1165,6 +1165,8 @@ static inline pud_t pud_modify(pud_t pud, pgprot_t newprot)
#endif /* CONFIG_TRANSPARENT_HUGEPAGE */
+#define ARCH_WANTS_GENERIC_POPULATE_VMEMMAP_PTE
+
/*
* Encode/decode swap entries and swap PTEs. Swap PTEs are all PTEs that
* are !pte_none() && !pte_present().
diff --git a/arch/x86/include/asm/pgtable.h b/arch/x86/include/asm/pgtable.h
index ef0252a09c28..f60c09cdf1ec 100644
--- a/arch/x86/include/asm/pgtable.h
+++ b/arch/x86/include/asm/pgtable.h
@@ -1368,6 +1368,8 @@ static inline pmd_t pmdp_establish(struct vm_area_struct *vma,
}
#endif
+#define ARCH_WANTS_GENERIC_POPULATE_VMEMMAP_PTE
+
#ifdef CONFIG_HAVE_ARCH_TRANSPARENT_HUGEPAGE_PUD
static inline pud_t pudp_establish(struct vm_area_struct *vma,
unsigned long address, pud_t *pudp, pud_t pud)
diff --git a/include/linux/pgtable.h b/include/linux/pgtable.h
index 780fe849ff8b..59ab7ca93548 100644
--- a/include/linux/pgtable.h
+++ b/include/linux/pgtable.h
@@ -457,6 +457,28 @@ static inline void set_ptes(struct mm_struct *mm, unsigned long addr,
#endif
#define set_pte_at(mm, addr, ptep, pte) set_ptes(mm, addr, ptep, pte, 1)
+#ifdef ARCH_WANTS_GENERIC_POPULATE_VMEMMAP_PTE
+/*
+ * try_update_vmemmap_pte - Remap PTEs used by the vmemmap.
+ * @addr: Base address of the remapped PTE.
+ * @ptep: Page table pointer to be overwritten.
+ * @pte: Page table entry to write.
+ *
+ * This function is only to be used to update PTEs that map the vmemmap. The
+ * only valid transitions supported by this function are: leaf-level
+ * (PAGE_SIZE), valid-to-valid. The pfn and prot bits may be changed.
+ *
+ * Implementations of this function must ensure that, while the update is taking
+ * place, CPUs will not fault on the remapped virtual address.
+ */
+static inline int try_update_vmemmap_pte(unsigned long addr, pte_t *ptep,
+ pte_t pte)
+{
+ set_pte(ptep, pte);
+ return 0;
+}
+#endif
+
#ifndef __HAVE_ARCH_PTEP_SET_ACCESS_FLAGS
extern int ptep_set_access_flags(struct vm_area_struct *vma,
unsigned long address, pte_t *ptep,
diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c
index 90db4d069ff6..3fdb1e4ce1a1 100644
--- a/mm/hugetlb_vmemmap.c
+++ b/mm/hugetlb_vmemmap.c
@@ -33,7 +33,7 @@
* operations.
*/
struct vmemmap_remap_walk {
- void (*remap_pte)(pte_t *pte, unsigned long addr,
+ int (*remap_pte)(pte_t *pte, unsigned long addr,
struct vmemmap_remap_walk *walk);
unsigned long nr_walked;
@@ -140,11 +140,13 @@ static int vmemmap_pte_entry(pte_t *pte, unsigned long addr,
unsigned long next, struct mm_walk *walk)
{
struct vmemmap_remap_walk *vmemmap_walk = walk->private;
+ int ret = 0;
- vmemmap_walk->remap_pte(pte, addr, vmemmap_walk);
- vmemmap_walk->nr_walked++;
+ ret = vmemmap_walk->remap_pte(pte, addr, vmemmap_walk);
+ if (!ret)
+ vmemmap_walk->nr_walked++;
- return 0;
+ return ret;
}
static const struct mm_walk_ops vmemmap_remap_ops = {
@@ -196,18 +198,20 @@ static void free_vmemmap_page_list(struct list_head *list)
free_vmemmap_page(page);
}
-static void vmemmap_remap_pte(pte_t *pte, unsigned long addr,
- struct vmemmap_remap_walk *walk)
+static int vmemmap_remap_pte(pte_t *pte, unsigned long addr,
+ struct vmemmap_remap_walk *walk)
{
struct page *page = pte_page(ptep_get(pte));
pte_t entry;
+ bool head;
+ int ret;
+
+ head = walk->nr_walked == 0 && walk->vmemmap_head;
/* Remapping the head page requires r/w */
- if (unlikely(walk->nr_walked == 0 && walk->vmemmap_head)) {
+ if (unlikely(head)) {
VM_WARN_ON_ONCE(!PageHead((const struct page *)addr));
- list_del(&walk->vmemmap_head->lru);
-
/*
* Makes sure that preceding stores to the page contents from
* vmemmap_remap_free() become visible before the set_pte_at()
@@ -226,17 +230,30 @@ static void vmemmap_remap_pte(pte_t *pte, unsigned long addr,
entry = mk_pte(walk->vmemmap_tail, PAGE_KERNEL_RO);
}
+ ret = try_update_vmemmap_pte(addr, pte, entry);
+ if (ret)
+ return ret;
+
+ /* We successfully overwrote the vmemmap PTE, so we can free
+ * the vmemmap page that was just unmapped, and if we mapped
+ * the new head page, remove it from the list so that it
+ * doesn't get freed later.
+ */
list_add(&page->lru, walk->vmemmap_pages);
- set_pte_at(&init_mm, addr, pte, entry);
+ if (head)
+ list_del(&walk->vmemmap_head->lru);
+
+ return 0;
}
-static void vmemmap_restore_pte(pte_t *pte, unsigned long addr,
- struct vmemmap_remap_walk *walk)
+static int vmemmap_restore_pte(pte_t *pte, unsigned long addr,
+ struct vmemmap_remap_walk *walk)
{
struct page *src = pte_page(ptep_get(pte)), *dst;
+ int ret;
if (WARN_ON_ONCE(!walk->vmemmap_tail))
- return;
+ return -EINVAL;
/*
* When restoring a partially-HVOed page, keep the copied head page
@@ -244,20 +261,25 @@ static void vmemmap_restore_pte(pte_t *pte, unsigned long addr,
* page.
*/
if (walk->vmemmap_tail != src)
- return;
+ return 0;
VM_WARN_ON_ONCE(PageHead((const struct page *)addr));
dst = list_first_entry(walk->vmemmap_pages, struct page, lru);
- list_del(&dst->lru);
copy_page(page_to_virt(dst), page_to_virt(src));
/*
* Makes sure that preceding stores to the page contents become visible
- * before the set_pte_at() write.
+ * before the try_update_vmemmap_pte() write.
*/
smp_wmb();
- set_pte_at(&init_mm, addr, pte, mk_pte(dst, PAGE_KERNEL));
+
+ ret = try_update_vmemmap_pte(addr, pte, mk_pte(dst, PAGE_KERNEL));
+ if (ret)
+ return ret;
+
+ list_del(&dst->lru);
+ return 0;
}
/**
--
2.56.0.rc1.315.gc6ed9934b7-goog
next prev parent reply other threads:[~2026-10-03 0:21 UTC|newest]
Thread overview: 21+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-03 0:21 [PATCH v2 00/20] Another attempt at HVO support on arm64 James Houghton
2026-10-03 0:21 ` [PATCH v2 01/20] hugetlb: Don't restore vmemmap of non-HVOed folios on bulk restore error James Houghton
2026-10-03 0:21 ` [PATCH v2 02/20] arm64/pgtable: Clear AF with LDCLR on supported systems James Houghton
2026-10-03 0:21 ` [PATCH v2 03/20] hugetlb_vmemmap: Always flush TLB if needed upon PTE remapping James Houghton
2026-10-03 0:21 ` [PATCH v2 04/20] hugetlb_vmemmap: Leave pages partially HVOed upon restore failure James Houghton
2026-10-03 0:21 ` James Houghton [this message]
2026-10-03 0:21 ` [PATCH v2 06/20] hugetlb_vmemmap: Allow architectures to dynamically disallow HVO James Houghton
2026-10-03 0:21 ` [PATCH v2 07/20] hugetlb_vmemmap: Disable HVO sysctl if arch doesn't support HVO James Houghton
2026-10-03 0:21 ` [PATCH v2 08/20] hugetlb: Fully initialize tail struct pages of non-pre-HVOed bootmem folios James Houghton
2026-10-03 0:21 ` [PATCH v2 09/20] hugetlb_vmemmap: Allow architectures to make HVO enablement boot-time only James Houghton
2026-10-03 0:21 ` [PATCH v2 10/20] hugetlb_vmemmap: Expose whether HVO is enabled to architecture code James Houghton
2026-10-03 0:21 ` [PATCH v2 11/20] arm64: Add bbm_through_af capability James Houghton
2026-10-03 0:21 ` [PATCH v2 12/20] arm64: Implement try_update_vmemmap_pte using the AF trick James Houghton
2026-10-03 0:21 ` [PATCH v2 13/20] arm64: Support hugetlb vmemmap optimization James Houghton
2026-10-03 0:21 ` [PATCH v2 14/20] hugetlb_vmemmap: Add fault injection for in-place vmemmap PTE updates James Houghton
2026-10-03 0:21 ` [PATCH v2 15/20] selftests/mm: Add HugeTLB vmemmap optimization stress test James Houghton
2026-10-03 0:21 ` [PATCH v2 16/20 DO-NOT-MERGE] hugetlb_vmemmap: Use try_populate_vmemmap_pmd for replacing in-use PMDs James Houghton
2026-10-03 0:21 ` [PATCH v2 17/20 DO-NOT-MERGE] arm64: Implement try_populate_vmemmap_pmd using AF trick James Houghton
2026-10-03 0:21 ` [PATCH v2 18/20 DO-NOT-MERGE] arm64: Drop BBML3 requirement for HVO James Houghton
2026-10-03 0:21 ` [PATCH v2 19/20 DO-NOT-MERGE] hugetlb_vmemmap: Add fault injection for in-place vmemmap PMD splits James Houghton
2026-10-03 0:21 ` [PATCH v2 20/20 DO-NOT-MERGE] selftests/mm: Add HVO pmd-split fault injection tests James Houghton
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261003002123.505555-6-jthoughton@google.com \
--to=jthoughton@google.com \
--cc=akpm@linux-foundation.org \
--cc=catalin.marinas@arm.com \
--cc=david@kernel.org \
--cc=fvdl@google.com \
--cc=linu.cherian@arm.com \
--cc=linux-arm-kernel@lists.infradead.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-mm@kvack.org \
--cc=mark.rutland@arm.com \
--cc=muchun.song@linux.dev \
--cc=nikos.nikoleris@arm.com \
--cc=osalvador@suse.de \
--cc=rientjes@google.com \
--cc=ryan.roberts@arm.com \
--cc=sunnanyong@huawei.com \
--cc=will@kernel.org \
--cc=yuzhao@google.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®