mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Usama Arif <usama.arif@linux.dev>
To: Andrew Morton <akpm@linux-foundation.org>,
	david@kernel.org, chrisl@kernel.org, kasong@tencent.com,
	ljs@kernel.org, ziy@nvidia.com, linux-mm@kvack.org
Cc: ying.huang@linux.alibaba.com, Baoquan He <baoquan.he@linux.dev>,
	willy@infradead.org, youngjun.park@lge.com, hannes@cmpxchg.org,
	riel@surriel.com, shakeel.butt@linux.dev, alex@ghiti.fr,
	kas@kernel.org, baohua@kernel.org, dev.jain@arm.com,
	baolin.wang@linux.alibaba.com, Nico Pache <nico.pache@linux.dev>,
	Liam R. Howlett <liam@infradead.org>,
	ryan.roberts@arm.com, Vlastimil Babka <vbabka@kernel.org>,
	lance.yang@linux.dev, linux-kernel@vger.kernel.org,
	nphamcs@gmail.com, shikemeng@huaweicloud.com, yosry@kernel.org,
	qi.zheng@linux.dev, luizcap@redhat.com, kernel-team@meta.com,
	Usama Arif <usama.arif@linux.dev>
Subject: [PATCH v8 12/30] mm/swap: allow duplicating a range of swap entries
Date: Fri,  2 Oct 2026 02:52:26 -0700	[thread overview]
Message-ID: <20261002095503.3585565-13-usama.arif@linux.dev> (raw)
In-Reply-To: <20261002095503.3585565-1-usama.arif@linux.dev>

swap_dup_entry_direct() duplicates one slot at a time. The PMD swap entry
fork path needs HPAGE_PMD_NR of them, and doing that one slot at a time
would take and drop the cluster lock HPAGE_PMD_NR times.

Give it an @nr argument and rename it swap_dup_entries_direct(), and do the
same for swap_retry_table_alloc(), whose GFP_KERNEL retry has to cover the
same range - the caller does not know which slot in it overflowed. Keep the
old single-slot names as inline wrappers so existing callers are untouched.

Unlike the put side, @nr is handed straight to the per-cluster helper, so
the range has to sit inside one cluster. That holds for the only caller
passing nr > 1: a PMD swap entry only exists under CONFIG_THP_SWAP, where
SWAPFILE_CLUSTER == HPAGE_PMD_NR, and a PMD-order folio's slots are only
ever allocated at a cluster head (see alloc_swap_scan_cluster()), so the
range is exactly one cluster. Reject a crossing range with -EINVAL so a
future caller cannot walk off the end of the swap table.

No functional change intended.

Signed-off-by: Usama Arif <usama.arif@linux.dev>
---
 include/linux/swap.h | 12 +++++++++-
 mm/swap.h            | 13 +++++++++-
 mm/swapfile.c        | 56 +++++++++++++++++++++++++++++++++-----------
 3 files changed, 65 insertions(+), 16 deletions(-)

diff --git a/include/linux/swap.h b/include/linux/swap.h
index 43155e122b5c3..73930bb7ee5e0 100644
--- a/include/linux/swap.h
+++ b/include/linux/swap.h
@@ -414,9 +414,14 @@ sector_t swap_folio_sector(struct folio *folio);
  * All entries must be allocated by folio_alloc_swap(). And they must have
  * a swap count > 1. See comments of folio_*_swap helpers for more info.
  */
-int swap_dup_entry_direct(swp_entry_t entry);
+int swap_dup_entries_direct(swp_entry_t entry, int nr);
 void swap_put_entries_direct(swp_entry_t entry, int nr);
 
+static inline int swap_dup_entry_direct(swp_entry_t entry)
+{
+	return swap_dup_entries_direct(entry, 1);
+}
+
 /*
  * folio_free_swap tries to free the swap entries pinned by a swap cache
  * folio, it has to be here to be called by other components.
@@ -458,6 +463,11 @@ static inline void free_swap_cache(struct folio *folio)
 {
 }
 
+static inline int swap_dup_entries_direct(swp_entry_t ent, int nr)
+{
+	return 0;
+}
+
 static inline int swap_dup_entry_direct(swp_entry_t ent)
 {
 	return 0;
diff --git a/mm/swap.h b/mm/swap.h
index b3b54c28929a1..26ff22d63edca 100644
--- a/mm/swap.h
+++ b/mm/swap.h
@@ -222,7 +222,12 @@ static inline void swap_cluster_unlock_irq(struct swap_cluster_info *ci)
 	spin_unlock_irq(&ci->lock);
 }
 
-extern int swap_retry_table_alloc(swp_entry_t entry, gfp_t gfp);
+int swap_retry_table_alloc_nr(swp_entry_t entry, unsigned int nr, gfp_t gfp);
+
+static inline int swap_retry_table_alloc(swp_entry_t entry, gfp_t gfp)
+{
+	return swap_retry_table_alloc_nr(entry, 1, gfp);
+}
 
 /*
  * Below are the core routines for doing swap for a folio.
@@ -428,6 +433,12 @@ static inline int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio)
 	return 0;
 }
 
+static inline int swap_retry_table_alloc_nr(swp_entry_t entry, unsigned int nr,
+					    gfp_t gfp)
+{
+	return -EINVAL;
+}
+
 static inline int swap_retry_table_alloc(swp_entry_t entry, gfp_t gfp)
 {
 	return -EINVAL;
diff --git a/mm/swapfile.c b/mm/swapfile.c
index 280dd906eb187..f9cfdd4600647 100644
--- a/mm/swapfile.c
+++ b/mm/swapfile.c
@@ -1468,11 +1468,16 @@ static bool swap_sync_discard(void)
 
 static int swap_extend_table_alloc(struct swap_info_struct *si,
 				   struct swap_cluster_info *ci,
-				   unsigned int ci_off, gfp_t gfp)
+				   unsigned int ci_off, unsigned int nr,
+				   gfp_t gfp)
 {
 	int count;
+	unsigned int i;
 	void *table;
 
+	/* The range must not run past the end of @ci's swap table. */
+	VM_WARN_ON_ONCE(ci_off + nr > SWAPFILE_CLUSTER);
+
 	table = kzalloc(sizeof(ci->extend_table[0]) * SWAPFILE_CLUSTER, gfp);
 	if (!table)
 		return -ENOMEM;
@@ -1486,15 +1491,21 @@ static int swap_extend_table_alloc(struct swap_info_struct *si,
 	 */
 	if (!cluster_table_is_alloced(ci))
 		goto out_free;
-	count = swp_tb_get_count(__swap_table_get(ci, ci_off));
-	if (count < (SWP_TB_COUNT_MAX - 1))
-		goto out_free;
 	if (ci->extend_table)
 		goto out_free;
-
-	ci->extend_table = table;
-	spin_unlock(&ci->lock);
-	return 0;
+	/*
+	 * The caller may not know which slot in [ci_off, ci_off + nr) hit
+	 * SWP_TB_COUNT_MAX - 1. Confirm at least one slot in the range still
+	 * needs the extend table before committing the allocation.
+	 */
+	for (i = 0; i < nr; i++) {
+		count = swp_tb_get_count(__swap_table_get(ci, ci_off + i));
+		if (count >= (SWP_TB_COUNT_MAX - 1)) {
+			ci->extend_table = table;
+			spin_unlock(&ci->lock);
+			return 0;
+		}
+	}
 
 out_free:
 	spin_unlock(&ci->lock);
@@ -1502,19 +1513,23 @@ static int swap_extend_table_alloc(struct swap_info_struct *si,
 	return 0;
 }
 
-int swap_retry_table_alloc(swp_entry_t entry, gfp_t gfp)
+int swap_retry_table_alloc_nr(swp_entry_t entry, unsigned int nr, gfp_t gfp)
 {
 	int ret;
 	struct swap_info_struct *si;
 	struct swap_cluster_info *ci;
 	unsigned long offset = swp_offset(entry);
 
+	if (WARN_ON_ONCE(swp_cluster_offset(entry) + nr > SWAPFILE_CLUSTER))
+		return -EINVAL;
+
 	si = get_swap_device(entry);
 	if (IS_ERR_OR_NULL(si))
 		return 0;
 
 	ci = __swap_offset_to_cluster(si, offset);
-	ret = swap_extend_table_alloc(si, ci, swp_cluster_offset(entry), gfp);
+	ret = swap_extend_table_alloc(si, ci, swp_cluster_offset(entry), nr,
+				      gfp);
 
 	put_swap_device(si);
 	return ret;
@@ -1690,6 +1705,9 @@ static int __swap_cluster_dup_entry(struct swap_cluster_info *ci,
  * @offset: start offset of slots.
  * @nr: number of slots.
  *
+ * The range [offset, offset + nr) must not cross a cluster boundary; the
+ * caller is responsible for splitting a range that can.
+ *
  * Context: The specified slots must be pinned by existing swap count or swap
  * cache reference, so they won't be released until this helper returns.
  * Return: 0 on success. -ENOMEM if the swap count maxed out (SWP_TB_COUNT_MAX)
@@ -1704,6 +1722,7 @@ static int swap_dup_entries_cluster(struct swap_info_struct *si,
 
 	ci_start = offset % SWAPFILE_CLUSTER;
 	ci_end = ci_start + nr;
+	VM_WARN_ON_ONCE(ci_end > SWAPFILE_CLUSTER);
 	ci_off = ci_start;
 	ci = swap_cluster_lock(si, offset);
 restart:
@@ -1712,7 +1731,8 @@ static int swap_dup_entries_cluster(struct swap_info_struct *si,
 		if (unlikely(err)) {
 			if (err == -ENOMEM) {
 				spin_unlock(&ci->lock);
-				err = swap_extend_table_alloc(si, ci, ci_off, GFP_ATOMIC);
+				err = swap_extend_table_alloc(si, ci, ci_off, 1,
+							      GFP_ATOMIC);
 				spin_lock(&ci->lock);
 				if (!err)
 					goto restart;
@@ -1723,6 +1743,7 @@ static int swap_dup_entries_cluster(struct swap_info_struct *si,
 	swap_cluster_unlock(ci);
 	return 0;
 failed:
+	/* The caller's page-table or swap-cache reference pins every slot. */
 	while (ci_off-- > ci_start)
 		__swap_cluster_put_entry(ci, ci_off);
 	swap_cluster_unlock(ci);
@@ -3966,8 +3987,9 @@ void si_swapinfo(struct sysinfo *val)
 }
 
 /*
- * swap_dup_entry_direct() - Increase reference count of a swap entry by one.
+ * swap_dup_entries_direct() - Increase reference count of swap entries by one.
  * @entry: first swap entry from which we want to increase the refcount.
+ * @nr: number of contiguous swap entries to duplicate.
  *
  * Returns 0 for success, or -ENOMEM if the extend table is required
  * but could not be atomically allocated.  Returns -EINVAL if the swap
@@ -3978,8 +4000,11 @@ void si_swapinfo(struct sysinfo *val)
  * owner. e.g., locking the PTL of a PTE containing the entry being increased.
  * Also the swap entry must have a count >= 1. Otherwise folio_dup_swap should
  * be used.
+ *
+ * Unlike swap_put_entries_direct(), the whole range [entry, entry + nr) must
+ * lie within one swap cluster; a crossing range is rejected with -EINVAL.
  */
-int swap_dup_entry_direct(swp_entry_t entry)
+int swap_dup_entries_direct(swp_entry_t entry, int nr)
 {
 	struct swap_info_struct *si;
 
@@ -3989,6 +4014,9 @@ int swap_dup_entry_direct(swp_entry_t entry)
 		return -EINVAL;
 	}
 
+	if (WARN_ON_ONCE(swp_cluster_offset(entry) + nr > SWAPFILE_CLUSTER))
+		return -EINVAL;
+
 	/*
 	 * The caller must be increasing the swap count from a direct
 	 * reference of the swap slot (e.g. a swap entry in page table).
@@ -3996,7 +4024,7 @@ int swap_dup_entry_direct(swp_entry_t entry)
 	 */
 	VM_WARN_ON_ONCE(!swap_entry_swapped(si, entry));
 
-	return swap_dup_entries_cluster(si, swp_offset(entry), 1);
+	return swap_dup_entries_cluster(si, swp_offset(entry), nr);
 }
 
 #if defined(CONFIG_MEMCG) && defined(CONFIG_BLK_CGROUP)
-- 
2.53.0-Meta


  parent reply	other threads:[~2026-10-02  9:56 UTC|newest]

Thread overview: 34+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-02  9:52 [PATCH v8 00/30] mm: PMD-level swap entries for anonymous THPs Usama Arif
2026-10-02  9:52 ` [PATCH v8 01/30] mm: rename pmd_to_softleaf_folio() to pmd_softleaf_to_folio() Usama Arif
2026-10-02  9:52 ` [PATCH v8 02/30] arm64: mm: add PMD swap-exclusive helpers Usama Arif
2026-10-02  9:52 ` [PATCH v8 03/30] loongarch: " Usama Arif
2026-10-02 14:17   ` Huacai Chen
2026-10-02  9:52 ` [PATCH v8 04/30] powerpc: " Usama Arif
2026-10-02  9:52 ` [PATCH v8 05/30] riscv: " Usama Arif
2026-10-02  9:52 ` [PATCH v8 06/30] s390: " Usama Arif
2026-10-02  9:52 ` [PATCH v8 07/30] x86: " Usama Arif
2026-10-02  9:52 ` [PATCH v8 08/30] mm: recognize PMD swap entries in the softleaf layer Usama Arif
2026-10-02  9:52 ` [PATCH v8 09/30] mm/debug_vm_pgtable: test PMD swap-exclusive helpers Usama Arif
2026-10-02  9:52 ` [PATCH v8 10/30] mm: make PMD migration-entry splitting explicit Usama Arif
2026-10-02  9:52 ` [PATCH v8 11/30] mm: split PMD swap entries into PTE swap entries Usama Arif
2026-10-02  9:52 ` Usama Arif [this message]
2026-10-02  9:52 ` [PATCH v8 13/30] mm: handle PMD swap entries in fork path Usama Arif
2026-10-02  9:52 ` [PATCH v8 14/30] mm: zswap: reject high-order swap cache allocations backed by zswap Usama Arif
2026-10-02  9:52 ` [PATCH v8 15/30] mm: swap in PMD swap entries as whole THPs during swapoff Usama Arif
2026-10-02  9:52 ` [PATCH v8 16/30] fs/proc: account PMD swap entries in smaps Usama Arif
2026-10-02  9:52 ` [PATCH v8 17/30] mm: handle soft-dirty and uffd-wp on PMD swap entries Usama Arif
2026-10-02  9:52 ` [PATCH v8 18/30] mm/hmm: fault PMD swap entries on demand Usama Arif
2026-10-02  9:52 ` [PATCH v8 19/30] mm: free PMD swap entries in zap_huge_pmd() Usama Arif
2026-10-02  9:52 ` [PATCH v8 20/30] mm/madvise: free PMD swap entries with MADV_FREE Usama Arif
2026-10-02  9:52 ` [PATCH v8 21/30] mm/madvise: skip PMD swap entries for MADV_COLD and MADV_PAGEOUT Usama Arif
2026-10-02  9:52 ` [PATCH v8 22/30] mm/madvise: keep PMD swap entries whole for MADV_GUARD_INSTALL/REMOVE Usama Arif
2026-10-02  9:52 ` [PATCH v8 23/30] mm/mincore: report PMD swap-cache residency Usama Arif
2026-10-02  9:52 ` [PATCH v8 24/30] mm/khugepaged: treat PMD swap entries as mapped THPs Usama Arif
2026-10-02  9:52 ` [PATCH v8 25/30] mm: handle PMD swap entries in MADV_WILLNEED Usama Arif
2026-10-02  9:52 ` [PATCH v8 26/30] mm: handle PMD swap entries in UFFDIO_MOVE Usama Arif
2026-10-02  9:52 ` [PATCH v8 27/30] mm: don't PTE-batch a swap-in over a hardware-poisoned subpage Usama Arif
2026-10-02  9:52 ` [PATCH v8 28/30] mm: handle PMD swap entry faults on swap-in Usama Arif
2026-10-02  9:52 ` [PATCH v8 29/30] mm: install PMD swap entries on swap-out Usama Arif
2026-10-02  9:52 ` [PATCH v8 30/30] selftests/mm: add PMD swap entry tests Usama Arif
2026-10-02 14:28 ` [PATCH v8 00/30] mm: PMD-level swap entries for anonymous THPs David Hildenbrand (Arm)
2026-10-02 15:13   ` Zi Yan

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261002095503.3585565-13-usama.arif@linux.dev \
    --to=usama.arif@linux.dev \
    --cc=akpm@linux-foundation.org \
    --cc=alex@ghiti.fr \
    --cc=baohua@kernel.org \
    --cc=baolin.wang@linux.alibaba.com \
    --cc=baoquan.he@linux.dev \
    --cc=chrisl@kernel.org \
    --cc=david@kernel.org \
    --cc=dev.jain@arm.com \
    --cc=hannes@cmpxchg.org \
    --cc=kas@kernel.org \
    --cc=kasong@tencent.com \
    --cc=kernel-team@meta.com \
    --cc=lance.yang@linux.dev \
    --cc=liam@infradead.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-mm@kvack.org \
    --cc=ljs@kernel.org \
    --cc=luizcap@redhat.com \
    --cc=nico.pache@linux.dev \
    --cc=nphamcs@gmail.com \
    --cc=qi.zheng@linux.dev \
    --cc=riel@surriel.com \
    --cc=ryan.roberts@arm.com \
    --cc=shakeel.butt@linux.dev \
    --cc=shikemeng@huaweicloud.com \
    --cc=vbabka@kernel.org \
    --cc=willy@infradead.org \
    --cc=ying.huang@linux.alibaba.com \
    --cc=yosry@kernel.org \
    --cc=youngjun.park@lge.com \
    --cc=ziy@nvidia.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®