From: Kiryl Shutsemau <kirill@shutemov.name>
To: Andrew Morton <akpm@linux-foundation.org>,
David Hildenbrand <david@kernel.org>,
Lorenzo Stoakes <ljs@kernel.org>
Cc: linux-mm@kvack.org, linux-kernel@vger.kernel.org,
kernel-team@meta.com, Zi Yan <ziy@nvidia.com>,
Baolin Wang <baolin.wang@linux.alibaba.com>,
"Liam R . Howlett" <liam@infradead.org>,
Nico Pache <nico.pache@linux.dev>,
Ryan Roberts <ryan.roberts@arm.com>, Dev Jain <dev.jain@arm.com>,
Barry Song <baohua@kernel.org>, Lance Yang <lance.yang@linux.dev>,
Usama Arif <usama.arif@linux.dev>,
Vlastimil Babka <vbabka@kernel.org>, Jann Horn <jannh@google.com>,
"Kiryl Shutsemau (Meta)" <kas@kernel.org>
Subject: [PATCH 10/12] mm/collapse: work out the orders a VMA allows once per VMA
Date: Fri, 4 Sep 2026 16:10:24 +0100 [thread overview]
Message-ID: <3ce85bb56f2bc60f91bf4e9645f247460f3c3f1e.1788533997.git.kas@kernel.org> (raw)
In-Reply-To: <cover.1788533997.git.kas@kernel.org>
From: "Kiryl Shutsemau (Meta)" <kas@kernel.org>
The scan asked collapse_possible_orders() for every PTE table, for an
answer that is a property of the VMA. Both callers walk a VMA a table at
a time, so let them work it out once and pass the mask in. It is only
good while the lock that produced it is held, so madvise_collapse() takes
it again after every collapse.
The mask is then sampled once per VMA rather than once per table. A thp
enabled knob written during a walk takes effect one VMA later, and cannot
widen a collapse: hugepage_vma_revalidate() tests the order again under
the lock the collapse retakes.
Assisted-by: Claude-Code:claude-opus-5
Signed-off-by: Kiryl Shutsemau (Meta) <kas@kernel.org>
---
mm/khugepaged.c | 32 ++++++++++++++++++--------------
1 file changed, 18 insertions(+), 14 deletions(-)
diff --git a/mm/khugepaged.c b/mm/khugepaged.c
index 40fcdd4f2712..f862abb1dbbd 100644
--- a/mm/khugepaged.c
+++ b/mm/khugepaged.c
@@ -1551,12 +1551,12 @@ static enum scan_result mthp_collapse(struct mm_struct *mm,
}
static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma,
- unsigned long start_addr, struct collapse_control *cc)
+ unsigned long start_addr, struct collapse_control *cc,
+ unsigned long enabled_orders)
{
const unsigned int max_ptes_shared = collapse_max_ptes_shared(cc, HPAGE_PMD_ORDER);
const unsigned int max_ptes_swap = collapse_max_ptes_swap(cc, HPAGE_PMD_ORDER);
unsigned int max_ptes_none = collapse_max_ptes_none(cc, vma, HPAGE_PMD_ORDER);
- enum tva_type tva_flags = cc->policy.tva_type;
struct mm_struct *mm = vma->vm_mm;
pmd_t *pmd;
pte_t *pte, *_pte, pteval;
@@ -1567,7 +1567,6 @@ static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma,
struct folio *folio = NULL;
unsigned long failed_pfn = -1;
unsigned long addr;
- unsigned long enabled_orders;
spinlock_t *ptl;
int node = NUMA_NO_NODE, unmapped = 0;
@@ -1581,8 +1580,6 @@ static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma,
collapse_scan_reset(cc);
- enabled_orders = collapse_possible_orders(vma, vma->vm_flags, tva_flags);
-
/*
* If PMD is the only enabled order, enforce max_ptes_none, otherwise
* scan all pages to populate the bitmap for mTHP collapse. The bitmap
@@ -2773,7 +2770,8 @@ static void collapse_control_release(struct collapse_control *cc)
}
static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma,
- unsigned long addr, struct collapse_control *cc)
+ unsigned long addr, struct collapse_control *cc,
+ unsigned long orders)
{
mmap_assert_locked(vma->vm_mm);
/* Whatever the last scan found has to have been run by now */
@@ -2783,7 +2781,7 @@ static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma,
}
if (vma_is_anonymous(vma))
- return collapse_scan_anon_pmd(vma, addr, cc);
+ return collapse_scan_anon_pmd(vma, addr, cc, orders);
/*
* A file collapse works on the page cache and never sees a VMA, so take
@@ -2877,15 +2875,17 @@ static void collapse_scan_mm_slot(unsigned int progress_max,
vma_iter_init(&vmi, mm, khugepaged_scan.address);
for_each_vma(vmi, vma) {
- unsigned long hstart, hend;
+ unsigned long hstart, hend, orders;
cond_resched();
if (unlikely(collapse_test_exit_or_disable(mm))) {
cc->progress++;
break;
}
- if (!collapse_possible_orders(vma, vma->vm_flags,
- TVA_KHUGEPAGED)) {
+ /* One mask for the whole VMA */
+ orders = collapse_possible_orders(vma, vma->vm_flags,
+ cc->policy.tva_type);
+ if (!orders) {
cc->progress++;
continue;
}
@@ -2914,7 +2914,7 @@ static void collapse_scan_mm_slot(unsigned int progress_max,
/* move to next address */
khugepaged_scan.address += HPAGE_PMD_SIZE;
- *result = collapse_scan_pmd(vma, addr, cc);
+ *result = collapse_scan_pmd(vma, addr, cc, orders);
/* Nothing to collapse here, and the lock is still ours */
if (*result != SCAN_SUCCEED) {
if (cc->progress >= progress_max)
@@ -3198,14 +3198,16 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start,
{
struct collapse_control *cc;
struct mm_struct *mm = vma->vm_mm;
- unsigned long hstart, hend, addr;
+ unsigned long hstart, hend, addr, orders;
enum scan_result last_fail = SCAN_FAIL;
int thps = 0;
BUG_ON(vma->vm_start > start);
BUG_ON(vma->vm_end < end);
- if (!collapse_possible_orders(vma, vma->vm_flags, TVA_FORCED_COLLAPSE))
+ orders = collapse_possible_orders(vma, vma->vm_flags,
+ TVA_FORCED_COLLAPSE);
+ if (!orders)
return -EINVAL;
hstart = ALIGN(start, HPAGE_PMD_SIZE);
@@ -3243,9 +3245,11 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start,
}
vma = found;
hend = min(hend, vma->vm_end & HPAGE_PMD_MASK);
+ orders = collapse_possible_orders(vma, vma->vm_flags,
+ cc->policy.tva_type);
}
- result = collapse_scan_pmd(vma, addr, cc);
+ result = collapse_scan_pmd(vma, addr, cc, orders);
/* Nothing to collapse here, and the lock is still ours */
if (result != SCAN_SUCCEED)
goto tally;
--
2.54.0
next prev parent reply other threads:[~2026-09-04 15:10 UTC|newest]
Thread overview: 20+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-04 15:10 [PATCH 00/12] mm/collapse: separate a collapse from its callers Kiryl Shutsemau
2026-09-04 15:10 ` [PATCH 01/12] mm/khugepaged: drop redundant mm_struct pin in madvise_collapse() Kiryl Shutsemau
2026-09-04 15:58 ` Zi Yan
2026-09-04 15:10 ` [PATCH 02/12] mm/khugepaged: count collapses where khugepaged makes them Kiryl Shutsemau
2026-09-05 2:25 ` Zi Yan
2026-09-04 15:10 ` [PATCH 03/12] mm/khugepaged: rename mthp_present_ptes bitmap to eligible_ptes Kiryl Shutsemau
2026-09-05 2:28 ` Zi Yan
2026-09-04 15:10 ` [PATCH 04/12] mm/collapse: add collapse.h for the collapse interface Kiryl Shutsemau
2026-09-05 2:36 ` Zi Yan
2026-09-04 15:10 ` [PATCH 05/12] mm/collapse: state what a collapse may do in the policy Kiryl Shutsemau
2026-09-05 2:44 ` Zi Yan
2026-09-04 15:10 ` [PATCH 06/12] mm/collapse: drop the collapse_possible() wrapper Kiryl Shutsemau
2026-09-05 2:45 ` Zi Yan
2026-09-04 15:10 ` [PATCH 07/12] mm/collapse: name the per-table scan reset for what it resets Kiryl Shutsemau
2026-09-05 18:05 ` Zi Yan
2026-09-04 15:10 ` [PATCH 08/12] mm/collapse: separate scanning a PTE table from collapsing it Kiryl Shutsemau
2026-09-04 15:10 ` [PATCH 09/12] mm/collapse: open-code collapse_single_pmd() in its two callers Kiryl Shutsemau
2026-09-04 15:10 ` Kiryl Shutsemau [this message]
2026-09-04 15:10 ` [PATCH 11/12] mm/collapse: declare the collapse interface in collapse.h Kiryl Shutsemau
2026-09-04 15:10 ` [PATCH 12/12] mm/collapse: implement MADV_COLLAPSE in madvise.c Kiryl Shutsemau
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=3ce85bb56f2bc60f91bf4e9645f247460f3c3f1e.1788533997.git.kas@kernel.org \
--to=kirill@shutemov.name \
--cc=akpm@linux-foundation.org \
--cc=baohua@kernel.org \
--cc=baolin.wang@linux.alibaba.com \
--cc=david@kernel.org \
--cc=dev.jain@arm.com \
--cc=jannh@google.com \
--cc=kas@kernel.org \
--cc=kernel-team@meta.com \
--cc=lance.yang@linux.dev \
--cc=liam@infradead.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-mm@kvack.org \
--cc=ljs@kernel.org \
--cc=nico.pache@linux.dev \
--cc=ryan.roberts@arm.com \
--cc=usama.arif@linux.dev \
--cc=vbabka@kernel.org \
--cc=ziy@nvidia.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®