mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
* [PATCH] mm/migrate: look up consecutive pages together in do_pages_stat_array()
@ 2026-10-01 15:10 Qiliang Yuan
  2026-10-01 15:37 ` Zi Yan
  0 siblings, 1 reply; 2+ messages in thread
From: Qiliang Yuan @ 2026-10-01 15:10 UTC (permalink / raw)
  To: Andrew Morton, David Hildenbrand, Zi Yan, Matthew Brost,
	Joshua Hahn, Byungchul Park, Gregory Price, Ying Huang,
	Alistair Popple
  Cc: linux-mm, linux-kernel, Qiliang Yuan

move_pages() with a NULL node list reports the node of each page. RDMA
and KV-cache transfer engines use it to find where large registered
buffers live, querying every 4K page of buffers that span hundreds of
gigabytes.

do_pages_stat_array() looks up the VMA and walks the page tables from
the top for every address, taking and dropping the PTE lock each time.
That costs about 105 ns per page, so a 16 GiB buffer takes 440 ms.

Callers almost always pass consecutive addresses. Reuse the VMA while
the address stays inside it. After folio_walk_start() locks a PTE table
or a PMD-mapped folio, answer the following addresses under the same
PMD from that table or folio before releasing the lock. Other
addresses still take the existing path.

On 7.3-rc5 in a 16-vCPU VM, querying every page of a populated buffer:

                       before       after
  4K pages, per page   105 ns       22.7 ns
  THP, per page        92 ns        13.1 ns
  16 GiB, 4K pages     439 ms       95 ms

Signed-off-by: Qiliang Yuan <odys.yuan@gmail.com>
---
 mm/migrate.c | 83 ++++++++++++++++++++++++++++++++++++++++++++----------------
 1 file changed, 61 insertions(+), 22 deletions(-)

diff --git a/mm/migrate.c b/mm/migrate.c
index 15b45832bcfa7..5f470dc7dc746 100644
--- a/mm/migrate.c
+++ b/mm/migrate.c
@@ -2451,44 +2451,83 @@ static int do_pages_move(struct mm_struct *mm, nodemask_t task_nodes,
 	return err;
 }
 
+static int folio_stat(struct folio *folio)
+{
+	if (is_zero_folio(folio) || is_huge_zero_folio(folio))
+		return -EFAULT;
+	if (folio_is_zone_device(folio))
+		return -ENOENT;
+	return folio_nid(folio);
+}
+
+/* Look up a PTE in a locked page table the way folio_walk_start() does. */
+static int pte_stat(struct vm_area_struct *vma, unsigned long addr,
+		    pte_t *ptep)
+{
+	pte_t pte = ptep_get(ptep);
+	struct page *page;
+
+	if (!pte_present(pte))
+		return -ENOENT;
+	page = vm_normal_page(vma, addr, pte);
+	if (page)
+		return folio_stat(page_folio(page));
+	if (is_zero_pfn(pte_pfn(pte)))
+		return -EFAULT;
+	return -ENOENT;
+}
+
 /*
  * Determine the nodes of an array of pages and store it in an array of status.
  */
 static void do_pages_stat_array(struct mm_struct *mm, unsigned long nr_pages,
 				const void __user **pages, int *status)
 {
-	unsigned long i;
+	struct vm_area_struct *vma = NULL;
+	unsigned long i = 0;
 
 	mmap_read_lock(mm);
 
-	for (i = 0; i < nr_pages; i++) {
-		unsigned long addr = (unsigned long)(*pages);
-		struct vm_area_struct *vma;
+	while (i < nr_pages) {
+		unsigned long addr = (unsigned long)pages[i];
+		unsigned long next, end;
 		struct folio_walk fw;
 		struct folio *folio;
-		int err = -EFAULT;
+		pte_t *ptep;
 
-		vma = vma_lookup(mm, addr);
-		if (!vma)
-			goto set_status;
+		if (!vma || addr < vma->vm_start || addr >= vma->vm_end)
+			vma = vma_lookup(mm, addr);
+		if (!vma) {
+			status[i++] = -EFAULT;
+			continue;
+		}
 
 		folio = folio_walk_start(&fw, vma, addr, FW_ZEROPAGE);
-		if (folio) {
-			if (is_zero_folio(folio) || is_huge_zero_folio(folio))
-				err = -EFAULT;
-			else if (folio_is_zone_device(folio))
-				err = -ENOENT;
-			else
-				err = folio_nid(folio);
-			folio_walk_end(&fw, vma);
-		} else {
-			err = -ENOENT;
+		if (!folio) {
+			status[i++] = -ENOENT;
+			continue;
 		}
-set_status:
-		*status = err;
+		status[i++] = folio_stat(folio);
 
-		pages++;
-		status++;
+		/*
+		 * Callers usually pass consecutive pages. Answer those that
+		 * fall under the entry or page table we already hold locked
+		 * instead of walking the page tables again for each of them.
+		 */
+		end = pmd_addr_end(addr, vma->vm_end);
+		ptep = fw.ptep;
+		for (next = addr + PAGE_SIZE; i < nr_pages && next < end;
+		     next += PAGE_SIZE, i++) {
+			if ((unsigned long)pages[i] != next)
+				break;
+			if (fw.level == FW_LEVEL_PMD)
+				status[i] = status[i - 1];
+			else if (fw.level == FW_LEVEL_PTE)
+				status[i] = pte_stat(vma, next, ++ptep);
+			else
+				break;
+		}
+		folio_walk_end(&fw, vma);
 	}
 
 	mmap_read_unlock(mm);

---
base-commit: 551c722f40809618230001baccf219193e22fc5a
change-id: 20261001-bug-mm-move-pages-stat-batch-f62a29ea5867

Best regards,
-- 
Qiliang Yuan <odys.yuan@gmail.com>


^ permalink raw reply	[flat|nested] 2+ messages in thread

* Re: [PATCH] mm/migrate: look up consecutive pages together in do_pages_stat_array()
  2026-10-01 15:10 [PATCH] mm/migrate: look up consecutive pages together in do_pages_stat_array() Qiliang Yuan
@ 2026-10-01 15:37 ` Zi Yan
  0 siblings, 0 replies; 2+ messages in thread
From: Zi Yan @ 2026-10-01 15:37 UTC (permalink / raw)
  To: Qiliang Yuan
  Cc: Andrew Morton, David Hildenbrand, Matthew Brost, Joshua Hahn,
	Byungchul Park, Gregory Price, Ying Huang, Alistair Popple,
	linux-mm, linux-kernel

On 1 Oct 2026, at 11:10, Qiliang Yuan wrote:

> move_pages() with a NULL node list reports the node of each page. RDMA
> and KV-cache transfer engines use it to find where large registered
> buffers live, querying every 4K page of buffers that span hundreds of
> gigabytes.
>
> do_pages_stat_array() looks up the VMA and walks the page tables from
> the top for every address, taking and dropping the PTE lock each time.
> That costs about 105 ns per page, so a 16 GiB buffer takes 440 ms.
>
> Callers almost always pass consecutive addresses. Reuse the VMA while
> the address stays inside it. After folio_walk_start() locks a PTE table
> or a PMD-mapped folio, answer the following addresses under the same
> PMD from that table or folio before releasing the lock. Other
> addresses still take the existing path.

Instead of adding customized batched walk code, it might be better to use
walk_page_range() + pmd_entry callback. See mincore_pte_range() for
reference.

BTW, DO_PAGES_STAT_CHUNK_NR is limited at 16, so the batching benefit
comes from 16 pages instead of 512 pages in one PMD? I wonder if that
limit could be raised and how much more benefit we will see.

>
> On 7.3-rc5 in a 16-vCPU VM, querying every page of a populated buffer:
>
>                        before       after
>   4K pages, per page   105 ns       22.7 ns
>   THP, per page        92 ns        13.1 ns
>   16 GiB, 4K pages     439 ms       95 ms
>
> Signed-off-by: Qiliang Yuan <odys.yuan@gmail.com>
> ---
>  mm/migrate.c | 83 ++++++++++++++++++++++++++++++++++++++++++++----------------
>  1 file changed, 61 insertions(+), 22 deletions(-)
>
> diff --git a/mm/migrate.c b/mm/migrate.c
> index 15b45832bcfa7..5f470dc7dc746 100644
> --- a/mm/migrate.c
> +++ b/mm/migrate.c
> @@ -2451,44 +2451,83 @@ static int do_pages_move(struct mm_struct *mm, nodemask_t task_nodes,
>  	return err;
>  }
>
> +static int folio_stat(struct folio *folio)
> +{
> +	if (is_zero_folio(folio) || is_huge_zero_folio(folio))
> +		return -EFAULT;
> +	if (folio_is_zone_device(folio))
> +		return -ENOENT;
> +	return folio_nid(folio);
> +}
> +
> +/* Look up a PTE in a locked page table the way folio_walk_start() does. */
> +static int pte_stat(struct vm_area_struct *vma, unsigned long addr,
> +		    pte_t *ptep)
> +{
> +	pte_t pte = ptep_get(ptep);
> +	struct page *page;
> +
> +	if (!pte_present(pte))
> +		return -ENOENT;
> +	page = vm_normal_page(vma, addr, pte);
> +	if (page)
> +		return folio_stat(page_folio(page));
> +	if (is_zero_pfn(pte_pfn(pte)))
> +		return -EFAULT;
> +	return -ENOENT;
> +}
> +
>  /*
>   * Determine the nodes of an array of pages and store it in an array of status.
>   */
>  static void do_pages_stat_array(struct mm_struct *mm, unsigned long nr_pages,
>  				const void __user **pages, int *status)
>  {
> -	unsigned long i;
> +	struct vm_area_struct *vma = NULL;
> +	unsigned long i = 0;
>
>  	mmap_read_lock(mm);
>
> -	for (i = 0; i < nr_pages; i++) {
> -		unsigned long addr = (unsigned long)(*pages);
> -		struct vm_area_struct *vma;
> +	while (i < nr_pages) {
> +		unsigned long addr = (unsigned long)pages[i];
> +		unsigned long next, end;
>  		struct folio_walk fw;
>  		struct folio *folio;
> -		int err = -EFAULT;
> +		pte_t *ptep;
>
> -		vma = vma_lookup(mm, addr);
> -		if (!vma)
> -			goto set_status;
> +		if (!vma || addr < vma->vm_start || addr >= vma->vm_end)
> +			vma = vma_lookup(mm, addr);
> +		if (!vma) {
> +			status[i++] = -EFAULT;
> +			continue;
> +		}
>
>  		folio = folio_walk_start(&fw, vma, addr, FW_ZEROPAGE);
> -		if (folio) {
> -			if (is_zero_folio(folio) || is_huge_zero_folio(folio))
> -				err = -EFAULT;
> -			else if (folio_is_zone_device(folio))
> -				err = -ENOENT;
> -			else
> -				err = folio_nid(folio);
> -			folio_walk_end(&fw, vma);
> -		} else {
> -			err = -ENOENT;
> +		if (!folio) {
> +			status[i++] = -ENOENT;
> +			continue;
>  		}
> -set_status:
> -		*status = err;
> +		status[i++] = folio_stat(folio);
>
> -		pages++;
> -		status++;
> +		/*
> +		 * Callers usually pass consecutive pages. Answer those that
> +		 * fall under the entry or page table we already hold locked
> +		 * instead of walking the page tables again for each of them.
> +		 */
> +		end = pmd_addr_end(addr, vma->vm_end);
> +		ptep = fw.ptep;
> +		for (next = addr + PAGE_SIZE; i < nr_pages && next < end;
> +		     next += PAGE_SIZE, i++) {
> +			if ((unsigned long)pages[i] != next)
> +				break;
> +			if (fw.level == FW_LEVEL_PMD)
> +				status[i] = status[i - 1];
> +			else if (fw.level == FW_LEVEL_PTE)
> +				status[i] = pte_stat(vma, next, ++ptep);
> +			else
> +				break;
> +		}
> +		folio_walk_end(&fw, vma);
>  	}
>
>  	mmap_read_unlock(mm);
>
> ---
> base-commit: 551c722f40809618230001baccf219193e22fc5a
> change-id: 20261001-bug-mm-move-pages-stat-batch-f62a29ea5867
>
> Best regards,
> -- 
> Qiliang Yuan <odys.yuan@gmail.com>


Best Regards,
Yan, Zi

^ permalink raw reply	[flat|nested] 2+ messages in thread

end of thread, other threads:[~2026-10-01 15:38 UTC | newest]

Thread overview: 2+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-10-01 15:10 [PATCH] mm/migrate: look up consecutive pages together in do_pages_stat_array() Qiliang Yuan
2026-10-01 15:37 ` Zi Yan

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®