mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
* [PATCH 1/2] mbind: fix verify_pages pte_page
@ 2005-06-06 19:48 Hugh Dickins
  2005-06-06 19:49 ` [PATCH 2/2] mbind: check_range use standard ptwalk Hugh Dickins
  2005-06-07 14:38 ` [PATCH 1/2] mbind: fix verify_pages pte_page Andi Kleen
  0 siblings, 2 replies; 4+ messages in thread
From: Hugh Dickins @ 2005-06-06 19:48 UTC (permalink / raw)
  To: Andrew Morton; +Cc: Andi Kleen, linux-kernel

Strict mbind's check that pages already mapped are on right node has been
using pte_page without checking if pfn_valid, and without page_table_lock
to prevent spurious failures when try_to_unmap_one intervenes between the
pte_present and the pte_page.

Signed-off-by: Hugh Dickins <hugh@veritas.com>
---

 mm/mempolicy.c |   19 ++++++++++++++-----
 1 files changed, 14 insertions(+), 5 deletions(-)

--- 2.6.12-rc6/mm/mempolicy.c	2005-05-25 18:09:21.000000000 +0100
+++ linux/mm/mempolicy.c	2005-06-04 20:41:55.000000000 +0100
@@ -242,6 +242,9 @@ static int
 verify_pages(struct mm_struct *mm,
 	     unsigned long addr, unsigned long end, unsigned long *nodes)
 {
+	int err = 0;
+
+	spin_lock(&mm->page_table_lock);
 	while (addr < end) {
 		struct page *p;
 		pte_t *pte;
@@ -268,17 +271,23 @@ verify_pages(struct mm_struct *mm,
 		}
 		p = NULL;
 		pte = pte_offset_map(pmd, addr);
-		if (pte_present(*pte))
-			p = pte_page(*pte);
+		if (pte_present(*pte)) {
+			unsigned long pfn = pte_pfn(*pte);
+			if (pfn_valid(pfn))
+				p = pfn_to_page(pfn);
+		}
 		pte_unmap(pte);
 		if (p) {
 			unsigned nid = page_to_nid(p);
-			if (!test_bit(nid, nodes))
-				return -EIO;
+			if (!test_bit(nid, nodes)) {
+				err = -EIO;
+				break;
+			}
 		}
 		addr += PAGE_SIZE;
 	}
-	return 0;
+	spin_unlock(&mm->page_table_lock);
+	return err;
 }
 
 /* Step 1: check the range */

^ permalink raw reply	[flat|nested] 4+ messages in thread

* [PATCH 2/2] mbind: check_range use standard ptwalk
  2005-06-06 19:48 [PATCH 1/2] mbind: fix verify_pages pte_page Hugh Dickins
@ 2005-06-06 19:49 ` Hugh Dickins
  2005-06-07 14:39   ` Andi Kleen
  2005-06-07 14:38 ` [PATCH 1/2] mbind: fix verify_pages pte_page Andi Kleen
  1 sibling, 1 reply; 4+ messages in thread
From: Hugh Dickins @ 2005-06-06 19:49 UTC (permalink / raw)
  To: Andrew Morton; +Cc: Andi Kleen, linux-kernel

Strict mbind's check for currently mapped pages being on node has been
using a slow loop which re-evaluates pgd, pud, pmd, pte for each entry:
replace that by a standard four-level page table walk like others in mm.
Since mmap_sem is held for writing, page_table_lock can be taken at the
inner level to limit latency.

Signed-off-by: Hugh Dickins <hugh@veritas.com>
---

 mm/mempolicy.c |  115 ++++++++++++++++++++++++++++++++++-----------------------
 1 files changed, 70 insertions(+), 45 deletions(-)

--- 2.6.12-rc6+/mm/mempolicy.c	2005-06-04 20:41:55.000000000 +0100
+++ linux/mm/mempolicy.c	2005-06-04 20:42:00.000000000 +0100
@@ -238,56 +238,81 @@ static struct mempolicy *mpol_new(int mo
 }
 
 /* Ensure all existing pages follow the policy. */
-static int
-verify_pages(struct mm_struct *mm,
-	     unsigned long addr, unsigned long end, unsigned long *nodes)
+static int check_pte_range(struct mm_struct *mm, pmd_t *pmd,
+		unsigned long addr, unsigned long end, unsigned long *nodes)
 {
-	int err = 0;
+	pte_t *orig_pte;
+	pte_t *pte;
 
 	spin_lock(&mm->page_table_lock);
-	while (addr < end) {
-		struct page *p;
-		pte_t *pte;
-		pmd_t *pmd;
-		pud_t *pud;
-		pgd_t *pgd;
-		pgd = pgd_offset(mm, addr);
-		if (pgd_none(*pgd)) {
-			unsigned long next = (addr + PGDIR_SIZE) & PGDIR_MASK;
-			if (next > addr)
-				break;
-			addr = next;
-			continue;
-		}
-		pud = pud_offset(pgd, addr);
-		if (pud_none(*pud)) {
-			addr = (addr + PUD_SIZE) & PUD_MASK;
+	orig_pte = pte = pte_offset_map(pmd, addr);
+	do {
+		unsigned long pfn;
+		unsigned int nid;
+
+		if (!pte_present(*pte))
 			continue;
-		}
-		pmd = pmd_offset(pud, addr);
-		if (pmd_none(*pmd)) {
-			addr = (addr + PMD_SIZE) & PMD_MASK;
+		pfn = pte_pfn(*pte);
+		if (!pfn_valid(pfn))
 			continue;
-		}
-		p = NULL;
-		pte = pte_offset_map(pmd, addr);
-		if (pte_present(*pte)) {
-			unsigned long pfn = pte_pfn(*pte);
-			if (pfn_valid(pfn))
-				p = pfn_to_page(pfn);
-		}
-		pte_unmap(pte);
-		if (p) {
-			unsigned nid = page_to_nid(p);
-			if (!test_bit(nid, nodes)) {
-				err = -EIO;
-				break;
-			}
-		}
-		addr += PAGE_SIZE;
-	}
+		nid = pfn_to_nid(pfn);
+		if (!test_bit(nid, nodes))
+			break;
+	} while (pte++, addr += PAGE_SIZE, addr != end);
+	pte_unmap(orig_pte);
 	spin_unlock(&mm->page_table_lock);
-	return err;
+	return addr != end;
+}
+
+static inline int check_pmd_range(struct mm_struct *mm, pud_t *pud,
+		unsigned long addr, unsigned long end, unsigned long *nodes)
+{
+	pmd_t *pmd;
+	unsigned long next;
+
+	pmd = pmd_offset(pud, addr);
+	do {
+		next = pmd_addr_end(addr, end);
+		if (pmd_none_or_clear_bad(pmd))
+			continue;
+		if (check_pte_range(mm, pmd, addr, next, nodes))
+			return -EIO;
+	} while (pmd++, addr = next, addr != end);
+	return 0;
+}
+
+static inline int check_pud_range(struct mm_struct *mm, pgd_t *pgd,
+		unsigned long addr, unsigned long end, unsigned long *nodes)
+{
+	pud_t *pud;
+	unsigned long next;
+
+	pud = pud_offset(pgd, addr);
+	do {
+		next = pud_addr_end(addr, end);
+		if (pud_none_or_clear_bad(pud))
+			continue;
+		if (check_pmd_range(mm, pud, addr, next, nodes))
+			return -EIO;
+	} while (pud++, addr = next, addr != end);
+	return 0;
+}
+
+static inline int check_pgd_range(struct mm_struct *mm,
+		unsigned long addr, unsigned long end, unsigned long *nodes)
+{
+	pgd_t *pgd;
+	unsigned long next;
+
+	pgd = pgd_offset(mm, addr);
+	do {
+		next = pgd_addr_end(addr, end);
+		if (pgd_none_or_clear_bad(pgd))
+			continue;
+		if (check_pud_range(mm, pgd, addr, next, nodes))
+			return -EIO;
+	} while (pgd++, addr = next, addr != end);
+	return 0;
 }
 
 /* Step 1: check the range */
@@ -308,7 +333,7 @@ check_range(struct mm_struct *mm, unsign
 		if (prev && prev->vm_end < vma->vm_start)
 			return ERR_PTR(-EFAULT);
 		if ((flags & MPOL_MF_STRICT) && !is_vm_hugetlb_page(vma)) {
-			err = verify_pages(vma->vm_mm,
+			err = check_pgd_range(vma->vm_mm,
 					   vma->vm_start, vma->vm_end, nodes);
 			if (err) {
 				first = ERR_PTR(err);

^ permalink raw reply	[flat|nested] 4+ messages in thread

* Re: [PATCH 1/2] mbind: fix verify_pages pte_page
  2005-06-06 19:48 [PATCH 1/2] mbind: fix verify_pages pte_page Hugh Dickins
  2005-06-06 19:49 ` [PATCH 2/2] mbind: check_range use standard ptwalk Hugh Dickins
@ 2005-06-07 14:38 ` Andi Kleen
  1 sibling, 0 replies; 4+ messages in thread
From: Andi Kleen @ 2005-06-07 14:38 UTC (permalink / raw)
  To: Hugh Dickins; +Cc: Andrew Morton, Andi Kleen, linux-kernel

On Mon, Jun 06, 2005 at 08:48:27PM +0100, Hugh Dickins wrote:
> Strict mbind's check that pages already mapped are on right node has been
> using pte_page without checking if pfn_valid, and without page_table_lock
> to prevent spurious failures when try_to_unmap_one intervenes between the
> pte_present and the pte_page.

Thanks. Looks good.

-Andi

^ permalink raw reply	[flat|nested] 4+ messages in thread

* Re: [PATCH 2/2] mbind: check_range use standard ptwalk
  2005-06-06 19:49 ` [PATCH 2/2] mbind: check_range use standard ptwalk Hugh Dickins
@ 2005-06-07 14:39   ` Andi Kleen
  0 siblings, 0 replies; 4+ messages in thread
From: Andi Kleen @ 2005-06-07 14:39 UTC (permalink / raw)
  To: Hugh Dickins; +Cc: Andrew Morton, Andi Kleen, linux-kernel

On Mon, Jun 06, 2005 at 08:49:38PM +0100, Hugh Dickins wrote:
> Strict mbind's check for currently mapped pages being on node has been
> using a slow loop which re-evaluates pgd, pud, pmd, pte for each entry:
> replace that by a standard four-level page table walk like others in mm.
> Since mmap_sem is held for writing, page_table_lock can be taken at the
> inner level to limit latency.

I doubt you will be able to benchmark a difference since the inner loop
should be normally memory bound anyways, giving the CPU plenty
of time to do some cache accesses. But ok.

-Andi


^ permalink raw reply	[flat|nested] 4+ messages in thread

end of thread, other threads:[~2005-06-07 14:39 UTC | newest]

Thread overview: 4+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2005-06-06 19:48 [PATCH 1/2] mbind: fix verify_pages pte_page Hugh Dickins
2005-06-06 19:49 ` [PATCH 2/2] mbind: check_range use standard ptwalk Hugh Dickins
2005-06-07 14:39   ` Andi Kleen
2005-06-07 14:38 ` [PATCH 1/2] mbind: fix verify_pages pte_page Andi Kleen

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®