* Make 2.5.17 TLB even more friendlier
@ 2002-05-21 6:07 David S. Miller
2002-05-21 12:18 ` Paul Mackerras
0 siblings, 1 reply; 10+ messages in thread
From: David S. Miller @ 2002-05-21 6:07 UTC (permalink / raw)
To: torvalds; +Cc: linux-kernel
How about this one? Linus, you can pull it from:
master.kernel.org:/home/davem/BK/tlb-2.5
if you think it is fine. [ And no, before someone asks, I do
put provide my BK trees anywhere publicly. I don't do it because
it would mean I would have to maintain every tree twice, and I
push often enough to Linus and Marcelo that it does not matter. ]
The idea is to give pte_free_tlb() a way to know "which" pte page
is being killed. We need to know that to flush the virtual PTEs
used to make TLB misses go fast on sparc64.
I verified that if you don't make reference to the pte_page_nr
or pmd_page_nr values, GCC optimizes it completely away.
The next part is allowing for a "full_mm_flush" state such that
tlb_flush() can make decisions based upon that. Then we move
the tlb_{start,end}_vma() invocations one level up. So for
the exit_mmap case it is:
tlb_gather_mmu(mm, 1);
flush_cache_mm(mm);
for each vma {
...
unmap_page_range(...);
}
clear_page_tables();
tlb_finish_mmu();
and for munmap-like operations it is:
tlb_gather_mmu(mm, 0);
for each vma {
..
tlb_start_vma(tlb, vma, start, end);
unmap_page_range(vma, start, end);
tlb_end_vma(tlb, vma, start, end);
}
tlb_finish_mmu();
So on Sparc64 we can make tlb_{start,end}_vma() do the
cache/tlb range flushes. Then tlb_flush will do
a flush_tlb_mm if we are not doing a full flush.
Finally, flush_tlb_pgtables is no longer needed and also
buggy, so we kill it off.
Makes sense?
--- ./include/asm-generic/tlb.h.~1~ Mon May 20 22:26:06 2002
+++ ./include/asm-generic/tlb.h Mon May 20 22:28:46 2002
@@ -22,7 +22,7 @@
*/
#ifdef CONFIG_SMP
#define FREE_PTE_NR 507
- #define tlb_fast_mode(tlb) ((tlb)->nr == ~0UL)
+ #define tlb_fast_mode(tlb) ((tlb)->nr == ~0U)
#else
#define FREE_PTE_NR 1
#define tlb_fast_mode(tlb) 1
@@ -35,7 +35,8 @@
*/
typedef struct free_pte_ctx {
struct mm_struct *mm;
- unsigned long nr; /* set to ~0UL means fast mode */
+ unsigned int nr; /* set to ~0U means fast mode */
+ unsigned int full_mm_flush;
unsigned long freed;
struct page * pages[FREE_PTE_NR];
} mmu_gather_t;
@@ -46,15 +47,18 @@ extern mmu_gather_t mmu_gathers[NR_CPUS]
/* tlb_gather_mmu
* Return a pointer to an initialized mmu_gather_t.
*/
-static inline mmu_gather_t *tlb_gather_mmu(struct mm_struct *mm)
+static inline mmu_gather_t *tlb_gather_mmu(struct mm_struct *mm, int full_mm_flush)
{
mmu_gather_t *tlb = &mmu_gathers[smp_processor_id()];
tlb->mm = mm;
- tlb->freed = 0;
/* Use fast mode if only one CPU is online */
- tlb->nr = smp_num_cpus > 1 ? 0UL : ~0UL;
+ tlb->nr = smp_num_cpus > 1 ? 0U : ~0U;
+
+ tlb->full_mm_flush = full_mm_flush;
+ tlb->freed = 0;
+
return tlb;
}
--- ./include/asm-i386/tlb.h.~1~ Mon May 20 22:32:02 2002
+++ ./include/asm-i386/tlb.h Mon May 20 22:32:11 2002
@@ -5,8 +5,8 @@
* x86 doesn't need any special per-pte or
* per-vma handling..
*/
-#define tlb_start_vma(tlb, vma) do { } while (0)
-#define tlb_end_vma(tlb, vma) do { } while (0)
+#define tlb_start_vma(tlb, vma, start, end) do { } while (0)
+#define tlb_end_vma(tlb, vma, start, end) do { } while (0)
#define tlb_remove_tlb_entry(tlb, pte, address) do { } while (0)
/*
--- ./include/asm-i386/pgalloc.h.~1~ Mon May 20 22:35:48 2002
+++ ./include/asm-i386/pgalloc.h Mon May 20 22:36:03 2002
@@ -36,7 +36,7 @@ static inline void pte_free(struct page
}
-#define pte_free_tlb(tlb,pte) tlb_remove_page((tlb),(pte))
+#define pte_free_tlb(tlb,pte,pte_page_nr) tlb_remove_page((tlb),(pte))
/*
* allocating and freeing a pmd is trivial: the 1-entry pmd is
@@ -46,7 +46,7 @@ static inline void pte_free(struct page
#define pmd_alloc_one(mm, addr) ({ BUG(); ((pmd_t *)2); })
#define pmd_free(x) do { } while (0)
-#define pmd_free_tlb(tlb,x) do { } while (0)
+#define pmd_free_tlb(tlb,x,y) do { } while (0)
#define pgd_populate(mm, pmd, pte) BUG()
#define check_pgt_cache() do { } while (0)
--- ./include/asm-sparc64/tlb.h.~1~ Mon May 20 22:22:43 2002
+++ ./include/asm-sparc64/tlb.h Mon May 20 22:38:08 2002
@@ -1 +1,36 @@
+#ifndef _SPARC64_TLB_H
+#define _SPARC64_TLB_H
+
+#define tlb_flush(tlb) \
+do { if ((tlb)->full_mm_flush) \
+ flush_tlb_mm((tlb)->mm);\
+} while (0)
+
+#define tlb_start_vma(tlb, vma, start, end) \
+ flush_cache_range(vma, start, end)
+#define tlb_end_vma(tlb, vma, start, end) \
+ flush_tlb_range(vma, start, end)
+
+#define tlb_remove_tlb_entry(tlb, pte, address) do { } while (0)
+
#include <asm-generic/tlb.h>
+
+#define pmd_free_tlb(tlb, pmd, pmd_page_nr) pmd_free(pmd)
+
+static __inline__ void pte_free_tlb(mmu_gather_t *tlb, struct page *pte,
+ unsigned long pte_page_nr)
+{
+ pte_free(pte);
+
+ if (!tlb->full_mm_flush) {
+ unsigned long vpte_addr;
+
+ vpte_addr = (tlb_type == spitfire ?
+ VPTE_BASE_SPITFIRE :
+ VPTE_BASE_CHEETAH);
+ vpte_addr += (pte_page_nr << PAGE_SHIFT);
+ flush_tlb_vpte(tlb->mm, vpte_addr);
+ }
+}
+
+#endif /* _SPARC64_TLB_H */
--- ./include/asm-sparc64/tlbflush.h.~1~ Mon May 20 22:34:25 2002
+++ ./include/asm-sparc64/tlbflush.h Mon May 20 22:46:47 2002
@@ -43,6 +43,13 @@ do { struct mm_struct *__mm = (vma)->vm_
SECONDARY_CONTEXT); \
} while(0)
+#define flush_tlb_vpte(mm, addr) \
+do { struct mm_struct *__mm = (mm); \
+ if (CTX_VALID(__mm->context)) \
+ __flush_tlb_page(CTX_HWBITS(__mm->context), (addr)&PAGE_MASK, \
+ SECONDARY_CONTEXT); \
+} while(0)
+
#else /* CONFIG_SMP */
extern void smp_flush_tlb_all(void);
@@ -61,33 +68,9 @@ extern void smp_flush_tlb_page(struct mm
smp_flush_tlb_kernel_range(start, end)
#define flush_tlb_page(vma, page) \
smp_flush_tlb_page((vma)->vm_mm, page)
+#define flush_tlb_vpte(mm, addr) \
+ smp_flush_tlb_page((mm), addr)
#endif /* ! CONFIG_SMP */
-
-static __inline__ void flush_tlb_pgtables(struct mm_struct *mm, unsigned long start,
- unsigned long end)
-{
- /* Note the signed type. */
- long s = start, e = end, vpte_base;
- if (s > e)
- /* Nobody should call us with start below VM hole and end above.
- See if it is really true. */
- BUG();
-#if 0
- /* Currently free_pgtables guarantees this. */
- s &= PMD_MASK;
- e = (e + PMD_SIZE - 1) & PMD_MASK;
-#endif
- vpte_base = (tlb_type == spitfire ?
- VPTE_BASE_SPITFIRE :
- VPTE_BASE_CHEETAH);
- {
- struct vm_area_struct vma;
- vma.vm_mm = mm;
- flush_tlb_range(&vma,
- vpte_base + (s >> (PAGE_SHIFT - 3)),
- vpte_base + (e >> (PAGE_SHIFT - 3)));
- }
-}
#endif /* _SPARC64_TLBFLUSH_H */
--- ./mm/mmap.c.~1~ Mon May 20 22:26:27 2002
+++ ./mm/mmap.c Mon May 20 22:35:11 2002
@@ -785,10 +785,8 @@ no_mmaps:
*/
start_index = pgd_index(first);
end_index = pgd_index(last);
- if (end_index > start_index) {
+ if (end_index > start_index)
clear_page_tables(tlb, start_index, end_index - start_index);
- flush_tlb_pgtables(mm, first & PGDIR_MASK, last & PGDIR_MASK);
- }
}
/* Normal function to fix up a mapping
@@ -846,7 +844,7 @@ static void unmap_region(struct mm_struc
{
mmu_gather_t *tlb;
- tlb = tlb_gather_mmu(mm);
+ tlb = tlb_gather_mmu(mm, 0);
do {
unsigned long from, to;
@@ -854,7 +852,9 @@ static void unmap_region(struct mm_struc
from = start < mpnt->vm_start ? mpnt->vm_start : start;
to = end > mpnt->vm_end ? mpnt->vm_end : end;
+ tlb_start_vma(tlb, mpnt, from, to);
unmap_page_range(tlb, mpnt, from, to);
+ tlb_end_vma(tlb, mpnt, from, to);
} while ((mpnt = mpnt->vm_next) != NULL);
free_pgtables(tlb, prev, start, end);
@@ -1103,7 +1103,7 @@ void exit_mmap(struct mm_struct * mm)
release_segments(mm);
spin_lock(&mm->page_table_lock);
- tlb = tlb_gather_mmu(mm);
+ tlb = tlb_gather_mmu(mm, 1);
flush_cache_mm(mm);
mpnt = mm->mmap;
--- ./mm/memory.c.~1~ Mon May 20 22:26:27 2002
+++ ./mm/memory.c Mon May 20 22:37:41 2002
@@ -75,7 +75,8 @@ mem_map_t * mem_map;
* Note: this doesn't free the actual pages themselves. That
* has been handled earlier when unmapping all the memory regions.
*/
-static inline void free_one_pmd(mmu_gather_t *tlb, pmd_t * dir)
+static inline void free_one_pmd(mmu_gather_t *tlb, pmd_t * dir,
+ unsigned long pte_page_nr)
{
struct page *pte;
@@ -88,28 +89,32 @@ static inline void free_one_pmd(mmu_gath
}
pte = pmd_page(*dir);
pmd_clear(dir);
- pte_free_tlb(tlb, pte);
+ pte_free_tlb(tlb, pte, pte_page_nr);
}
-static inline void free_one_pgd(mmu_gather_t *tlb, pgd_t * dir)
+static inline unsigned long free_one_pgd(mmu_gather_t *tlb, pgd_t * dir,
+ unsigned long pte_page_nr)
{
int j;
pmd_t * pmd;
if (pgd_none(*dir))
- return;
+ goto out;
if (pgd_bad(*dir)) {
pgd_ERROR(*dir);
pgd_clear(dir);
- return;
+ goto out;
}
pmd = pmd_offset(dir, 0);
pgd_clear(dir);
for (j = 0; j < PTRS_PER_PMD ; j++) {
prefetchw(pmd+j+(PREFETCH_STRIDE/16));
- free_one_pmd(tlb, pmd+j);
+ free_one_pmd(tlb, pmd+j, pte_page_nr+j);
}
- pmd_free_tlb(tlb, pmd);
+ pmd_free_tlb(tlb, pmd, (dir - tlb->mm->pgd));
+
+out:
+ return pte_page_nr + PTRS_PER_PMD;
}
/*
@@ -121,10 +126,12 @@ static inline void free_one_pgd(mmu_gath
void clear_page_tables(mmu_gather_t *tlb, unsigned long first, int nr)
{
pgd_t * page_dir = tlb->mm->pgd;
+ unsigned long pte_page_nr;
page_dir += first;
+ pte_page_nr = first * PTRS_PER_PMD;
do {
- free_one_pgd(tlb, page_dir);
+ pte_page_nr = free_one_pgd(tlb, page_dir, pte_page_nr);
page_dir++;
} while (--nr);
@@ -396,13 +403,11 @@ void unmap_page_range(mmu_gather_t *tlb,
if (address >= end)
BUG();
dir = pgd_offset(vma->vm_mm, address);
- tlb_start_vma(tlb, vma);
do {
zap_pmd_range(tlb, dir, address, end - address);
address = (address + PGDIR_SIZE) & PGDIR_MASK;
dir++;
} while (address && (address < end));
- tlb_end_vma(tlb, vma);
}
/*
@@ -429,8 +434,10 @@ void zap_page_range(struct vm_area_struc
spin_lock(&mm->page_table_lock);
flush_cache_range(vma, address, end);
- tlb = tlb_gather_mmu(mm);
+ tlb = tlb_gather_mmu(mm, 0);
+ tlb_start_vma(tlb, vma, address, end);
unmap_page_range(tlb, vma, address, end);
+ tlb_end_vma(tlb, vma, address, end);
tlb_finish_mmu(tlb, start, end);
spin_unlock(&mm->page_table_lock);
}
^ permalink raw reply [flat|nested] 10+ messages in thread* Re: Make 2.5.17 TLB even more friendlier
2002-05-21 6:07 Make 2.5.17 TLB even more friendlier David S. Miller
@ 2002-05-21 12:18 ` Paul Mackerras
2002-05-21 12:10 ` David S. Miller
0 siblings, 1 reply; 10+ messages in thread
From: Paul Mackerras @ 2002-05-21 12:18 UTC (permalink / raw)
To: David S. Miller; +Cc: torvalds, linux-kernel
David S. Miller writes:
> The next part is allowing for a "full_mm_flush" state such that
> tlb_flush() can make decisions based upon that. Then we move
> the tlb_{start,end}_vma() invocations one level up. So for
> the exit_mmap case it is:
>
> tlb_gather_mmu(mm, 1);
>
> flush_cache_mm(mm);
>
> for each vma {
> ...
> unmap_page_range(...);
> }
>
> clear_page_tables();
> tlb_finish_mmu();
I still need tlb_end_vma in this case - or at least I need a hook
that gets called after all the tlb_remove_tlb_entry calls are done but
before clear_page_tables is called. If I had that hook (called in
both the exit_mmap and unmap cases) then I would not need the
tlb_start/end_vma hooks.
Regards,
Paul.
^ permalink raw reply [flat|nested] 10+ messages in thread* Re: Make 2.5.17 TLB even more friendlier
2002-05-21 12:18 ` Paul Mackerras
@ 2002-05-21 12:10 ` David S. Miller
2002-05-21 12:53 ` Paul Mackerras
0 siblings, 1 reply; 10+ messages in thread
From: David S. Miller @ 2002-05-21 12:10 UTC (permalink / raw)
To: paulus; +Cc: torvalds, linux-kernel
From: Paul Mackerras <paulus@samba.org>
Date: Tue, 21 May 2002 22:18:42 +1000 (EST)
I still need tlb_end_vma in this case - or at least I need a hook
that gets called after all the tlb_remove_tlb_entry calls are done but
before clear_page_tables is called. If I had that hook (called in
both the exit_mmap and unmap cases) then I would not need the
tlb_start/end_vma hooks.
You get called via pte_free_tlb() and pmd_free_tlb() for every
operation performed by clear_page_tables(). The PTEs themselves are
all cleared out at the point that clear_page_tables, so you can't
possibly need the PTE contents. I am assuming therefore you just
need to get at the linkage, and those two pte/pmd hooks give you
that.
^ permalink raw reply [flat|nested] 10+ messages in thread
* Re: Make 2.5.17 TLB even more friendlier
2002-05-21 12:10 ` David S. Miller
@ 2002-05-21 12:53 ` Paul Mackerras
2002-05-21 18:42 ` Linus Torvalds
0 siblings, 1 reply; 10+ messages in thread
From: Paul Mackerras @ 2002-05-21 12:53 UTC (permalink / raw)
To: David S. Miller; +Cc: torvalds, linux-kernel
David S. Miller writes:
> You get called via pte_free_tlb() and pmd_free_tlb() for every
> operation performed by clear_page_tables(). The PTEs themselves are
> all cleared out at the point that clear_page_tables, so you can't
> possibly need the PTE contents. I am assuming therefore you just
> need to get at the linkage, and those two pte/pmd hooks give you
> that.
I have a bit in the PTE that tells me if there is an MMU hash table
entry for the virtual address represented by the PTE. This bit is not
affected by set_pte or ptep_get_and_clear etc. and it is not part of
the swap-entry fields of the PTE. Thus I need to have the PTE page
still around at the point where I flush stuff from the MMU hash
table, even though all the PTEs in it have been cleared, so that I can
avoid searching the hash table for PTEs for virtual addresses for
which I have not put a PTE in the hash table.
Now I could of course try to come up with some other way to store this
information, such as a bitmap associated with the mmu context. The
bitmap would be 96kB in size for a 3GB userspace, though. It could be
allocated on a page-by-page basis, I guess. But all the schemes I
have thought of so far end up being more complex than using a special
bit in the PTE.
Regards,
Paul.
^ permalink raw reply [flat|nested] 10+ messages in thread
* Re: Make 2.5.17 TLB even more friendlier
2002-05-21 12:53 ` Paul Mackerras
@ 2002-05-21 18:42 ` Linus Torvalds
2002-05-21 23:05 ` Paul Mackerras
0 siblings, 1 reply; 10+ messages in thread
From: Linus Torvalds @ 2002-05-21 18:42 UTC (permalink / raw)
To: Paul Mackerras; +Cc: David S. Miller, linux-kernel
On Tue, 21 May 2002, Paul Mackerras wrote:
>
> I have a bit in the PTE that tells me if there is an MMU hash table
> entry for the virtual address represented by the PTE.
This is exactly why 2.5.17 has the "tlb_remove_pte_entry()", and why it is
passed down the pte that we just cleared out - so that architectures can
hide details like this in the page tables (the other use is to hide things
like "iTBL vs dTLB" etc).
Sp why don't you just make "tlb_remove_pte_entry()" look at your bit, and
then if that bit is set you try to remove the entry from the hash chains
at that point?
You _have_ to do it this way, in fact, since reserved pages and other
"VM-invisible" pages aren't even passed down to "tlb_remove_page()"
(because they have no freeing logic, and they have no impact on RSS).
Linus
^ permalink raw reply [flat|nested] 10+ messages in thread
* Re: Make 2.5.17 TLB even more friendlier
2002-05-21 18:42 ` Linus Torvalds
@ 2002-05-21 23:05 ` Paul Mackerras
2002-05-21 23:13 ` Linus Torvalds
0 siblings, 1 reply; 10+ messages in thread
From: Paul Mackerras @ 2002-05-21 23:05 UTC (permalink / raw)
To: Linus Torvalds; +Cc: David S. Miller, linux-kernel
Linus Torvalds writes:
> Sp why don't you just make "tlb_remove_pte_entry()" look at your bit, and
> then if that bit is set you try to remove the entry from the hash chains
> at that point?
Simply the desire to batch up the hash table searches for efficiency,
particularly on SMP where we have to take a spinlock before touching
the MMU hash table. This will be an even bigger win on ppc64 on
partitioned machines where we have to call the hypervisor to make
changes to the MMU hash table.
I'm thinking now that I would be better off doing the batching by
storing a list of virtual addresses needing flushing in the
mmu_gather_t, instead of relying on getting at the special bits in the
PTEs at some time after the tlb_remove_tlb_entry call.
Regards,
Paul.
^ permalink raw reply [flat|nested] 10+ messages in thread
* Re: Make 2.5.17 TLB even more friendlier
2002-05-21 23:05 ` Paul Mackerras
@ 2002-05-21 23:13 ` Linus Torvalds
2002-05-22 3:57 ` Paul Mackerras
2002-05-22 8:40 ` Paul Mackerras
0 siblings, 2 replies; 10+ messages in thread
From: Linus Torvalds @ 2002-05-21 23:13 UTC (permalink / raw)
To: Paul Mackerras; +Cc: David S. Miller, linux-kernel
On Wed, 22 May 2002, Paul Mackerras wrote:
> I'm thinking now that I would be better off doing the batching by
> storing a list of virtual addresses needing flushing in the
> mmu_gather_t, instead of relying on getting at the special bits in the
> PTEs at some time after the tlb_remove_tlb_entry call.
Well, you can combine that with something like
static inline void tlb_remove_tlb_entry(tlb, pte, address)
{
if (pte_tlb_hash(pte)) {
if (tlb->start_addr == NOSTART)
tlb->start_addr = address;
tlb->end_addr = address+PAGE_SIZE;
}
}
and then have the tlb_end_vma() do something like
/* No pages mapped? */
if (tlb->start_addr == NOADDR)
return;
pte_remove_range(vma, tlb->start_addr, tlb->end_addr)
tlb->start_addr = NOADDR;
which will bunch them up on a vma granularity (if there is any reason to
do that), while still retaining the optimization that if a VMA was mostly
unmapped you wouldn't need to do a lot of hash table searching because of
the start/end thing.
Linus
^ permalink raw reply [flat|nested] 10+ messages in thread* Re: Make 2.5.17 TLB even more friendlier
2002-05-21 23:13 ` Linus Torvalds
@ 2002-05-22 3:57 ` Paul Mackerras
2002-05-22 4:04 ` Linus Torvalds
2002-05-22 8:40 ` Paul Mackerras
1 sibling, 1 reply; 10+ messages in thread
From: Paul Mackerras @ 2002-05-22 3:57 UTC (permalink / raw)
To: Linus Torvalds; +Cc: David S. Miller, linux-kernel
It seems to me that there is a race in this code in zap_pte_range,
because there is a gap between when we read the pte and when we clear
it:
for (offset=0; offset < size; ptep++, offset += PAGE_SIZE) {
pte_t pte = *ptep;
if (pte_none(pte))
continue;
if (pte_present(pte)) {
unsigned long pfn = pte_pfn(pte);
pte_clear(ptep);
Isn't it possible that another cpu could set the dirty bit in the pte
between the "pte = *ptep" and the "pte_clear(ptep)"? In my case
another cpu could also set the "has hash-table entry" bit.
Shouldn't we do this as "pte = ptep_get_and_clear(ptep)", at least in
the case where we are unmapping stuff?
Paul.
^ permalink raw reply [flat|nested] 10+ messages in thread* Re: Make 2.5.17 TLB even more friendlier
2002-05-22 3:57 ` Paul Mackerras
@ 2002-05-22 4:04 ` Linus Torvalds
0 siblings, 0 replies; 10+ messages in thread
From: Linus Torvalds @ 2002-05-22 4:04 UTC (permalink / raw)
To: Paul Mackerras; +Cc: David S. Miller, linux-kernel
On Wed, 22 May 2002, Paul Mackerras wrote:
>
> It seems to me that there is a race in this code in zap_pte_range,
> because there is a gap between when we read the pte and when we clear
> it:
Yes and no.
There is a race, and yes, another thread might mark it dirty.
However, I've not decided whether we care about it yet. I think we _do_
care, for people doing strange things with their own internal VM
management using mmap/munmap of shared mappings, but on the other hand it
_is_ fairly expensive to do a "ptep_get_and_clear()".
> Shouldn't we do this as "pte = ptep_get_and_clear(ptep)", at least in
> the case where we are unmapping stuff?
Yeah, I want to do it, but I also would really want to avoid the overhead
for the exit case. Which is another reason I'd like to have exit() not use
zap_page_range() at all.
But I'll make that change now, so that we don't lose it. We should just
remember to not do it if we split up exit.
Linus
^ permalink raw reply [flat|nested] 10+ messages in thread
* Re: Make 2.5.17 TLB even more friendlier
2002-05-21 23:13 ` Linus Torvalds
2002-05-22 3:57 ` Paul Mackerras
@ 2002-05-22 8:40 ` Paul Mackerras
1 sibling, 0 replies; 10+ messages in thread
From: Paul Mackerras @ 2002-05-22 8:40 UTC (permalink / raw)
To: Linus Torvalds; +Cc: David S. Miller, linux-kernel
Here are the bits I currently have for doing the TLB flushing on PPC.
The declaration I have for tlb_flush is a bit ugly but AFAICS the only
alternative is to make it a #define. I wanted to make it a static
inline but I can't do that before the #include <asm-generic/tlb.h>
(since we don't have the mmu_gather_t declaration at that point) and
it is no use afterwards (since it has already been referenced).
Regards,
Paul.
diff -urN linux-2.5/include/asm-generic/tlb.h pmac-2.5/include/asm-generic/tlb.h
--- linux-2.5/include/asm-generic/tlb.h Tue May 21 15:27:47 2002
+++ pmac-2.5/include/asm-generic/tlb.h Wed May 22 14:10:54 2002
@@ -37,6 +37,8 @@
struct mm_struct *mm;
unsigned long nr; /* set to ~0UL means fast mode */
unsigned long freed;
+ unsigned long start; /* virtual address range, */
+ unsigned long end; /* used on PPC */
struct page * pages[FREE_PTE_NR];
} mmu_gather_t;
@@ -55,6 +57,7 @@
/* Use fast mode if only one CPU is online */
tlb->nr = smp_num_cpus > 1 ? 0UL : ~0UL;
+ tlb_init_tlb(tlb);
return tlb;
}
@@ -88,11 +91,10 @@
tlb_flush_mmu(tlb, start, end);
}
-
-/* void tlb_remove_page(mmu_gather_t *tlb, pte_t *ptep, unsigned long addr)
- * Must perform the equivalent to __free_pte(pte_get_and_clear(ptep)), while
- * handling the additional races in SMP caused by other CPUs caching valid
- * mappings in their TLBs.
+/* void tlb_remove_page(mmu_gather_t *tlb, struct page *page)
+ * This should free the page given after flushing any reference
+ * to it from the TLB. This should be done no later than the
+ * next call to tlb_finish_mmu for this tlb.
*/
static inline void tlb_remove_page(mmu_gather_t *tlb, struct page *page)
{
diff -urN linux-2.5/include/asm-i386/tlb.h pmac-2.5/include/asm-i386/tlb.h
--- linux-2.5/include/asm-i386/tlb.h Tue May 21 15:27:47 2002
+++ pmac-2.5/include/asm-i386/tlb.h Wed May 22 14:11:05 2002
@@ -5,6 +5,7 @@
* x86 doesn't need any special per-pte or
* per-vma handling..
*/
+#define tlb_init_tlb(tlb) do { } while (0)
#define tlb_start_vma(tlb, vma) do { } while (0)
#define tlb_end_vma(tlb, vma) do { } while (0)
#define tlb_remove_tlb_entry(tlb, pte, address) do { } while (0)
diff -urN linux-2.5/include/asm-ppc/tlb.h pmac-2.5/include/asm-ppc/tlb.h
--- linux-2.5/include/asm-ppc/tlb.h Tue Feb 5 18:40:23 2002
+++ pmac-2.5/include/asm-ppc/tlb.h Wed May 22 15:06:57 2002
@@ -1,4 +1,85 @@
/*
- * BK Id: SCCS/s.tlb.h 1.5 05/17/01 18:14:26 cort
+ * TLB shootdown specifics for PPC
+ *
+ * Copyright (C) 2002 Paul Mackerras, IBM Corp.
+ *
+ * This program is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU General Public License
+ * as published by the Free Software Foundation; either version
+ * 2 of the License, or (at your option) any later version.
*/
+#ifndef _PPC_TLB_H
+#define _PPC_TLB_H
+
+#include <linux/config.h>
+#include <asm/pgtable.h>
+#include <asm/tlbflush.h>
+#include <asm/page.h>
+#include <asm/mmu.h>
+
+#ifdef CONFIG_PPC_STD_MMU
+/* Classic PPC with hash-table based MMU... */
+
+#define _NO_ADDR (~0UL)
+#define tlb_init_tlb(tlb) ((tlb)->start = _NO_ADDR)
+
+struct free_pte_ctx; /* same as mmu_gather_t */
+extern void tlb_flush(struct free_pte_ctx *tlb);
+
+#else
+/* Embedded PPC with software-loaded TLB, very simple... */
+
+#define tlb_init_tlb(tlb) do { } while (0)
+#define tlb_start_vma(tlb, vma) do { } while (0)
+#define tlb_end_vma(tlb, vma) do { } while (0)
+#define tlb_remove_tlb_entry(tlb, pte, address) do { } while (0)
+#define tlb_flush(tlb) flush_tlb_mm((tlb)->mm)
+
+#endif /* CONFIG_PPC_STD_MMU */
+
+/* Get the generic bits... */
#include <asm-generic/tlb.h>
+
+#ifdef CONFIG_PPC_STD_MMU
+
+/* Nothing needed here in fact... */
+#define tlb_start_vma(tlb, vma) do { } while (0)
+
+/*
+ * flush_tlb_mm_range looks at the pte pages for the range of addresses
+ * in order to check the _PAGE_HASHPTE bit. Thus we can't defer
+ * the tlb_flush_mmu call to tlb_finish_mmu time, since by then the
+ * pointers to the pte pages in the pgdir have been zeroed.
+ * Instead we do the tlb_flush_mmu here. In future we could possibly
+ * do something cleverer, like keeping our own pointer(s) to the pte
+ * page(s) that we are interested in.
+ */
+static inline void tlb_end_vma(mmu_gather_t *tlb, struct vm_area_struct *vma)
+{
+ if (tlb->start != _NO_ADDR)
+ tlb_flush(tlb);
+}
+
+static inline void tlb_remove_tlb_entry(mmu_gather_t *tlb, pte_t pte,
+ unsigned long address)
+{
+ if (pte_val(pte) & _PAGE_HASHPTE) {
+ if (tlb->start == _NO_ADDR) {
+ tlb->start = address;
+ } else if (address - tlb->end > 32 * PAGE_SIZE) {
+ /*
+ * If there is a big gap in the range of addresses
+ * needing to be flushed, it is better to do two
+ * separate calls to flush_tlb_mm_range rather than
+ * a single call with a lot of ptes that it will
+ * have to skip over in the middle of the range.
+ */
+ tlb_flush(tlb);
+ tlb->start = address;
+ }
+ tlb->end = address + PAGE_SIZE;
+ }
+}
+#endif /* CONFIG_PPC_STD_MMU */
+
+#endif /* __PPC_TLB_H */
diff -urN linux-2.5/arch/ppc/mm/tlb.c pmac-2.5/arch/ppc/mm/tlb.c
--- linux-2.5/arch/ppc/mm/tlb.c Mon Apr 15 09:48:49 2002
+++ pmac-2.5/arch/ppc/mm/tlb.c Wed May 22 15:08:06 2002
@@ -31,6 +31,8 @@
#include <linux/mm.h>
#include <linux/init.h>
#include <linux/highmem.h>
+#include <asm/tlbflush.h>
+#include <asm/tlb.h>
#include "mmu_decl.h"
@@ -59,7 +61,24 @@
#define FINISH_FLUSH do { } while (0)
#endif
-static void flush_range(struct mm_struct *mm, unsigned long start,
+void tlb_flush(mmu_gather_t *tlb)
+{
+ if (Hash != 0) {
+ if (tlb->start != _NO_ADDR) {
+ flush_tlb_mm_range(tlb->mm, tlb->start, tlb->end);
+ tlb->start = _NO_ADDR;
+ }
+ } else {
+ /*
+ * 603 needs to flush the whole TLB here; we will
+ * have tlb->start = _NO_ADDR since none of its PTEs
+ * can be in the hash table.
+ */
+ _tlbia();
+ }
+}
+
+void flush_tlb_mm_range(struct mm_struct *mm, unsigned long start,
unsigned long end)
{
pmd_t *pmd;
@@ -110,7 +129,7 @@
*/
printk(KERN_ERR "flush_tlb_all called from %p\n",
__builtin_return_address(0));
- flush_range(&init_mm, TASK_SIZE, ~0UL);
+ flush_tlb_mm_range(&init_mm, TASK_SIZE, ~0UL);
FINISH_FLUSH;
}
@@ -119,7 +138,7 @@
*/
void flush_tlb_kernel_range(unsigned long start, unsigned long end)
{
- flush_range(&init_mm, start, end);
+ flush_tlb_mm_range(&init_mm, start, end);
FINISH_FLUSH;
}
@@ -130,18 +149,15 @@
*/
void flush_tlb_mm(struct mm_struct *mm)
{
+ struct vm_area_struct *mp;
+
if (Hash == 0) {
_tlbia();
return;
}
- if (mm->map_count) {
- struct vm_area_struct *mp;
- for (mp = mm->mmap; mp != NULL; mp = mp->vm_next)
- flush_range(mp->vm_mm, mp->vm_start, mp->vm_end);
- } else {
- flush_range(mm, 0, TASK_SIZE);
- }
+ for (mp = mm->mmap; mp != NULL; mp = mp->vm_next)
+ flush_tlb_mm_range(mp->vm_mm, mp->vm_start, mp->vm_end);
FINISH_FLUSH;
}
@@ -170,6 +186,6 @@
void flush_tlb_range(struct vm_area_struct *vma, unsigned long start,
unsigned long end)
{
- flush_range(vma->vm_mm, start, end);
+ flush_tlb_mm_range(vma->vm_mm, start, end);
FINISH_FLUSH;
}
^ permalink raw reply [flat|nested] 10+ messages in thread
end of thread, other threads:[~2002-05-22 8:42 UTC | newest]
Thread overview: 10+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2002-05-21 6:07 Make 2.5.17 TLB even more friendlier David S. Miller
2002-05-21 12:18 ` Paul Mackerras
2002-05-21 12:10 ` David S. Miller
2002-05-21 12:53 ` Paul Mackerras
2002-05-21 18:42 ` Linus Torvalds
2002-05-21 23:05 ` Paul Mackerras
2002-05-21 23:13 ` Linus Torvalds
2002-05-22 3:57 ` Paul Mackerras
2002-05-22 4:04 ` Linus Torvalds
2002-05-22 8:40 ` Paul Mackerras
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®