From: Usama Arif <usama.arif@linux.dev>
To: Matthew Wilcox <willy@infradead.org>
Cc: Andrew Morton <akpm@linux-foundation.org>,
david@kernel.org, chrisl@kernel.org, kasong@tencent.com,
ljs@kernel.org, ziy@nvidia.com, linux-mm@kvack.org,
ying.huang@linux.alibaba.com, Baoquan He <baoquan.he@linux.dev>,
youngjun.park@lge.com, hannes@cmpxchg.org, riel@surriel.com,
shakeel.butt@linux.dev, alex@ghiti.fr, kas@kernel.org,
baohua@kernel.org, dev.jain@arm.com,
baolin.wang@linux.alibaba.com, Nico Pache <nico.pache@linux.dev>,
"Liam R. Howlett" <liam@infradead.org>,
ryan.roberts@arm.com, Vlastimil Babka <vbabka@kernel.org>,
lance.yang@linux.dev, linux-kernel@vger.kernel.org,
nphamcs@gmail.com, shikemeng@huaweicloud.com, yosry@kernel.org,
kernel-team@meta.com
Subject: Re: [PATCH v5 10/11] mm: install PMD swap entries on swap-out
Date: Wed, 29 Jul 2026 17:30:49 +0100 [thread overview]
Message-ID: <26f6af3b-5967-4501-b1e6-cf6969b8bb39@linux.dev> (raw)
In-Reply-To: <amJoe_iRHEaMdo9i@casper.infradead.org>
On 23/07/2026 20:16, Matthew Wilcox wrote:
> On Wed, Jul 22, 2026 at 08:19:41AM -0700, Usama Arif wrote:
>> +++ b/mm/rmap.c
>> @@ -2282,6 +2282,25 @@ static bool try_to_unmap_one(struct folio *folio, struct vm_area_struct *vma,
>> goto walk_abort;
>> }
>>
>> +#ifdef CONFIG_THP_SWAP
>> + /*
>> + * If the folio is in the swap cache and we're not
>> + * asked to split, install a PMD-level swap entry.
>> + */
>> + if (!(flags & TTU_SPLIT_HUGE_PMD) &&
>> + folio_test_anon(folio) &&
>> + folio_test_swapcache(folio)) {
>> + if (set_pmd_swap_entry(&pvmw, folio))
>> + goto walk_abort;
>> +
>> + add_mm_counter(mm, MM_ANONPAGES,
>> + -HPAGE_PMD_NR);
>> + add_mm_counter(mm, MM_SWAPENTS,
>> + HPAGE_PMD_NR);
>> + goto walk_done;
>> + }
>> +#endif
>> +
>> if (flags & TTU_SPLIT_HUGE_PMD) {
>> /*
>> * We temporarily have to drop the PTL and
>
> This makes me sad. It feels like we're bolting more complexity onto
> try_to_unmap() instead of removing TTU_SPLIT_HUGE_PMD entirely.
>
> Ideally we'd just cope with folios of whatever size (including
> PMD and PUD mapped folios) and only handle hugetlb weirdness when we
> absolutely have to. Similarly, we could split PUDs to PMDs and
> PMDs to PTEs when we need to, without the caller specifying
> TTU_SPLIT_anything.
>
> Is there a reason we can't do that?
Hi Matthew,
So I have been looking into this over the past few days, and I can't
come up with a way to do this without having another rmap walk, which
is expensive.
I tried to prototype this (code at the end of the series).
Removing the flag required an additional rmap walk (split_folio_pmd_mappings below),
MMU notifier handling and different fallback behavior on architectures without
PMD softleaf support (see #ifdef CONFIG_ARCH_HAS_PMD_SOFTLEAVES below).
It makes the code more complicated and expensive.
It also changes the behavior of every try_to_unmap() and try_to_migrate()
caller, while this series only needs reclaim to preserve a PMD when it can
install a PMD swap entry. I would therefore prefer to keep this series scoped
to PMD swap entries and handle removal of TTU_SPLIT_HUGE_PMD independently
with the broader rmap, migration, architecture, and THP-splitting review
and testing it requires.
Untested prototype below:
diff --git a/include/linux/rmap.h b/include/linux/rmap.h
index a174758f7777..0e0172c0c852 100644
--- a/include/linux/rmap.h
+++ b/include/linux/rmap.h
@@ -93,7 +93,6 @@ struct anon_vma_chain {
enum ttu_flags {
TTU_USE_SHARED_ZEROPAGE = 0x2, /* for unused pages of large folios */
- TTU_SPLIT_HUGE_PMD = 0x4, /* split huge PMD if any */
TTU_IGNORE_MLOCK = 0x8, /* ignore mlock */
TTU_SYNC = 0x10, /* avoid racy checks with PVMW_SYNC */
TTU_HWPOISON = 0x20, /* do convert pte to hwpoison entry */
diff --git a/mm/huge_memory.c b/mm/huge_memory.c
index 6f7acae5a5fd..1012cb625452 100644
--- a/mm/huge_memory.c
+++ b/mm/huge_memory.c
@@ -3905,6 +3905,38 @@ void vma_adjust_trans_huge(struct vm_area_struct *vma,
split_huge_pmd_if_needed(next, end);
}
+static bool split_folio_pmd_mapping(struct folio *folio,
+ struct vm_area_struct *vma,
+ unsigned long address, void *arg)
+{
+ DEFINE_FOLIO_VMA_WALK(pvmw, folio, vma, address, PVMW_SYNC);
+ struct mmu_notifier_range range;
+
+ range.end = vma_address_end(&pvmw);
+ mmu_notifier_range_init(&range, MMU_NOTIFY_CLEAR, 0, vma->vm_mm,
+ address, range.end);
+ mmu_notifier_invalidate_range_start(&range);
+
+ while (page_vma_mapped_walk(&pvmw)) {
+ if (pvmw.pte)
+ continue;
+ split_huge_pmd_locked(vma, pvmw.address, pvmw.pmd, false);
+ page_vma_mapped_walk_restart(&pvmw);
+ }
+
+ mmu_notifier_invalidate_range_end(&range);
+ return true;
+}
+
+static void split_folio_pmd_mappings(struct folio *folio)
+{
+ struct rmap_walk_control rwc = {
+ .rmap_one = split_folio_pmd_mapping,
+ };
+
+ rmap_walk_locked(folio, &rwc);
+}
+
static void unmap_folio(struct folio *folio)
{
enum ttu_flags ttu_flags = TTU_RMAP_LOCKED | TTU_SYNC |
@@ -3912,18 +3944,23 @@ static void unmap_folio(struct folio *folio)
VM_BUG_ON_FOLIO(!folio_test_large(folio), folio);
- if (folio_test_pmd_mappable(folio))
- ttu_flags |= TTU_SPLIT_HUGE_PMD;
-
/*
* Anon pages need migration entries to preserve them, but file
* pages can simply be left unmapped, then faulted back on demand.
* If that is ever changed (perhaps for mlock), update remap_page().
*/
- if (folio_test_anon(folio))
+ if (folio_test_anon(folio)) {
+ if (folio_test_pmd_mappable(folio))
+ split_folio_pmd_mappings(folio);
try_to_migrate(folio, ttu_flags);
- else
+ } else {
try_to_unmap(folio, ttu_flags | TTU_IGNORE_MLOCK);
+ }
try_to_unmap_flush();
}
diff --git a/mm/rmap.c b/mm/rmap.c
index b7ead3e9f064..f77f380eb03c 100644
--- a/mm/rmap.c
+++ b/mm/rmap.c
@@ -2282,17 +2282,13 @@ static bool try_to_unmap_one(struct folio *folio, struct vm_area_struct *vma,
goto walk_abort;
}
- if (flags & TTU_SPLIT_HUGE_PMD) {
- /*
- * We temporarily have to drop the PTL and
- * restart so we can process the PTE-mapped THP.
- */
- split_huge_pmd_locked(vma, pvmw.address,
- pvmw.pmd, false);
- flags &= ~TTU_SPLIT_HUGE_PMD;
- page_vma_mapped_walk_restart(&pvmw);
- continue;
- }
+ /*
+ * No PMD-level unmap operation handled this mapping, so
+ * split it and retry at PTE granularity.
+ */
+ split_huge_pmd_locked(vma, pvmw.address, pvmw.pmd, false);
+ page_vma_mapped_walk_restart(&pvmw);
+ continue;
}
/* Unexpected PMD-mapped THP? */
@@ -2459,8 +2455,6 @@ void try_to_unmap(struct folio *folio, enum ttu_flags flags)
/*
* @arg: enum ttu_flags will be passed to this argument.
*
- * If TTU_SPLIT_HUGE_PMD is specified any PMD mappings will be split into PTEs
- * containing migration entries.
*/
static bool try_to_migrate_one(struct folio *folio, struct vm_area_struct *vma,
unsigned long address, void *arg)
@@ -2509,24 +2503,11 @@ static bool try_to_migrate_one(struct folio *folio, struct vm_area_struct *vma,
mmu_notifier_invalidate_range_start(&range);
while (page_vma_mapped_walk(&pvmw)) {
- /* PMD-mapped THP migration entry */
+ /* PMD-mapped THP */
if (!pvmw.pte) {
__maybe_unused unsigned long pfn;
__maybe_unused pmd_t pmdval;
- if (flags & TTU_SPLIT_HUGE_PMD) {
- /*
- * split_huge_pmd_locked() might leave the
- * folio mapped through PTEs. Retry the walk
- * so we can detect this scenario and properly
- * abort the walk.
- */
- split_huge_pmd_locked(vma, pvmw.address,
- pvmw.pmd, true);
- flags &= ~TTU_SPLIT_HUGE_PMD;
- page_vma_mapped_walk_restart(&pvmw);
- continue;
- }
#ifdef CONFIG_ARCH_HAS_PMD_SOFTLEAVES
pmdval = pmdp_get(pvmw.pmd);
if (likely(pmd_present(pmdval)))
@@ -2545,6 +2526,11 @@ static bool try_to_migrate_one(struct folio *folio, struct vm_area_struct *vma,
break;
}
continue;
+#else
+ /* Fall back to PTE migration entries on this architecture. */
+ split_huge_pmd_locked(vma, pvmw.address, pvmw.pmd, true);
+ page_vma_mapped_walk_restart(&pvmw);
+ continue;
#endif
}
@@ -2799,11 +2785,11 @@ void try_to_migrate(struct folio *folio, enum ttu_flags flags)
};
/*
- * Migration always ignores mlock and only supports TTU_RMAP_LOCKED and
- * TTU_SPLIT_HUGE_PMD, TTU_SYNC, and TTU_BATCH_FLUSH flags.
+ * Migration always ignores mlock and only supports TTU_RMAP_LOCKED,
+ * TTU_SYNC, and TTU_BATCH_FLUSH flags.
*/
- if (WARN_ON_ONCE(flags & ~(TTU_RMAP_LOCKED | TTU_SPLIT_HUGE_PMD |
- TTU_SYNC | TTU_BATCH_FLUSH)))
+ if (WARN_ON_ONCE(flags & ~(TTU_RMAP_LOCKED | TTU_SYNC |
+ TTU_BATCH_FLUSH)))
return;
if (folio_is_zone_device(folio) &&
diff --git a/mm/truncate.c b/mm/truncate.c
index b58ba940be47..12e1fea5cf50 100644
--- a/mm/truncate.c
+++ b/mm/truncate.c
@@ -182,7 +182,6 @@ static int folio_split_or_unmap(struct folio *folio, struct page *split_at,
{
enum ttu_flags ttu_flags =
TTU_SYNC |
- TTU_SPLIT_HUGE_PMD |
TTU_IGNORE_MLOCK;
int ret;
diff --git a/mm/vmscan.c b/mm/vmscan.c
index 9b8ee902f972..90751b384a5d 100644
--- a/mm/vmscan.c
+++ b/mm/vmscan.c
@@ -1333,8 +1333,6 @@ static unsigned int shrink_folio_list(struct list_head *folio_list,
enum ttu_flags flags = TTU_BATCH_FLUSH;
bool was_swapbacked = folio_test_swapbacked(folio);
- if (folio_test_pmd_mappable(folio))
- flags |= TTU_SPLIT_HUGE_PMD;
/*
* Without TTU_SYNC, try_to_unmap will only begin to
* hold PTL from the first present PTE within a large
--
2.53.0-Meta
next prev parent reply other threads:[~2026-07-29 16:31 UTC|newest]
Thread overview: 29+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-07-22 15:19 [PATCH v5 00/11] mm: PMD-level swap entries for anonymous THPs Usama Arif
2026-07-22 15:19 ` [PATCH v5 01/11] mm: add PMD swap entry detection support Usama Arif
2026-07-24 6:10 ` Dev Jain
2026-07-24 10:00 ` Usama Arif
2026-07-22 15:19 ` [PATCH v5 02/11] mm: add PMD swap entry splitting support Usama Arif
2026-07-24 6:52 ` Dev Jain
2026-07-24 10:05 ` Usama Arif
2026-07-22 15:19 ` [PATCH v5 03/11] mm: handle PMD swap entries in fork path Usama Arif
2026-07-22 15:19 ` [PATCH v5 04/11] mm: zswap: add range lookup for large-folio swapin Usama Arif
2026-07-23 0:01 ` Yosry Ahmed
2026-07-23 12:45 ` Usama Arif
2026-07-23 16:42 ` Yosry Ahmed
2026-07-23 17:15 ` Nhat Pham
2026-07-24 9:59 ` Usama Arif
2026-07-27 16:24 ` Nhat Pham
2026-07-29 13:52 ` Usama Arif
2026-07-22 15:19 ` [PATCH v5 05/11] mm: swap in PMD swap entries as whole THPs during swapoff Usama Arif
2026-07-22 15:19 ` [PATCH v5 06/11] mm: handle PMD swap entries in non-present PMD walkers Usama Arif
2026-07-22 15:19 ` [PATCH v5 07/11] mm: handle PMD swap entries in MADV_WILLNEED Usama Arif
2026-07-22 15:19 ` [PATCH v5 08/11] mm: handle PMD swap entries in UFFDIO_MOVE Usama Arif
2026-07-22 15:19 ` [PATCH v5 09/11] mm: handle PMD swap entry faults on swap-in Usama Arif
2026-07-22 15:19 ` [PATCH v5 10/11] mm: install PMD swap entries on swap-out Usama Arif
2026-07-23 19:16 ` Matthew Wilcox
2026-07-24 10:27 ` Usama Arif
2026-07-29 16:30 ` Usama Arif [this message]
2026-08-06 2:29 ` Luiz Capitulino
2026-08-07 10:03 ` Usama Arif
2026-07-22 15:19 ` [PATCH v5 11/11] selftests/mm: add PMD swap entry tests Usama Arif
2026-08-06 2:29 ` Luiz Capitulino
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=26f6af3b-5967-4501-b1e6-cf6969b8bb39@linux.dev \
--to=usama.arif@linux.dev \
--cc=akpm@linux-foundation.org \
--cc=alex@ghiti.fr \
--cc=baohua@kernel.org \
--cc=baolin.wang@linux.alibaba.com \
--cc=baoquan.he@linux.dev \
--cc=chrisl@kernel.org \
--cc=david@kernel.org \
--cc=dev.jain@arm.com \
--cc=hannes@cmpxchg.org \
--cc=kas@kernel.org \
--cc=kasong@tencent.com \
--cc=kernel-team@meta.com \
--cc=lance.yang@linux.dev \
--cc=liam@infradead.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-mm@kvack.org \
--cc=ljs@kernel.org \
--cc=nico.pache@linux.dev \
--cc=nphamcs@gmail.com \
--cc=riel@surriel.com \
--cc=ryan.roberts@arm.com \
--cc=shakeel.butt@linux.dev \
--cc=shikemeng@huaweicloud.com \
--cc=vbabka@kernel.org \
--cc=willy@infradead.org \
--cc=ying.huang@linux.alibaba.com \
--cc=yosry@kernel.org \
--cc=youngjun.park@lge.com \
--cc=ziy@nvidia.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
Powered by JetHome