* [PATCH v4 1/7] proc/task_mmu: remove unnecessary helpers
2026-09-11 19:41 [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock Suren Baghdasaryan
@ 2026-09-11 19:41 ` Suren Baghdasaryan
2026-09-11 19:41 ` [PATCH v4 2/7] proc/task_mmu: remove unnecessary inlines in function definitions Suren Baghdasaryan
` (6 subsequent siblings)
7 siblings, 0 replies; 17+ messages in thread
From: Suren Baghdasaryan @ 2026-09-11 19:41 UTC (permalink / raw)
To: akpm
Cc: liam, ljs, vbabka, david, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel, surenb,
Usama Arif
When per-vma locks were behind a config option, a number of helper
functions were needed to simplify the locking code. Now that these
locks are universally available, we can do a little cleanup.
Remove lock_vma_range(), unlock_vma_range(), query_vma_setup(),
query_vma_teardown() helpers.
No functional change intended.
Signed-off-by: Suren Baghdasaryan <surenb@google.com>
Reviewed-by: Liam R. Howlett (Oracle) <liam@infradead.org>
Reviewed-by: Lorenzo Stoakes (ARM) <ljs@kernel.org>
Acked-by: Usama Arif <usama.arif@linux.dev>
Acked-by: David Hildenbrand (Arm) <david@kernel.org>
---
fs/proc/task_mmu.c | 67 ++++++++++++----------------------------------
1 file changed, 17 insertions(+), 50 deletions(-)
diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c
index e671b4fd8ded..2f500d639db5 100644
--- a/fs/proc/task_mmu.c
+++ b/fs/proc/task_mmu.c
@@ -160,25 +160,6 @@ static void unlock_ctx_vma(struct proc_maps_locking_ctx *lock_ctx)
}
}
-static inline bool lock_vma_range(struct seq_file *m,
- struct proc_maps_locking_ctx *lock_ctx)
-{
- rcu_read_lock();
- reset_lock_ctx(lock_ctx);
-
- return true;
-}
-
-static inline void unlock_vma_range(struct proc_maps_locking_ctx *lock_ctx)
-{
- if (lock_ctx->mmap_locked) {
- unlock_ctx_mm(lock_ctx);
- } else {
- unlock_ctx_vma(lock_ctx);
- rcu_read_unlock();
- }
-}
-
static struct vm_area_struct *get_next_vma(struct proc_maps_private *priv,
loff_t last_pos)
{
@@ -286,13 +267,8 @@ static void *m_start(struct seq_file *m, loff_t *ppos)
return NULL;
}
- if (!lock_vma_range(m, lock_ctx)) {
- mmput(mm);
- put_task_struct(priv->task);
- priv->task = NULL;
- return ERR_PTR(-EINTR);
- }
-
+ rcu_read_lock();
+ reset_lock_ctx(lock_ctx);
/*
* Reset current position if last_addr was set before
* and it's not a sentinel.
@@ -325,7 +301,12 @@ static void m_stop(struct seq_file *m, void *v)
return;
release_task_mempolicy(priv);
- unlock_vma_range(&priv->lock_ctx);
+ if (priv->lock_ctx.mmap_locked) {
+ unlock_ctx_mm(&priv->lock_ctx);
+ } else {
+ unlock_ctx_vma(&priv->lock_ctx);
+ rcu_read_unlock();
+ }
mmput(mm);
put_task_struct(priv->task);
priv->task = NULL;
@@ -518,21 +499,6 @@ static int pid_maps_open(struct inode *inode, struct file *file)
PROCMAP_QUERY_VMA_FLAGS \
)
-static int query_vma_setup(struct proc_maps_locking_ctx *lock_ctx)
-{
- reset_lock_ctx(lock_ctx);
-
- return 0;
-}
-
-static void query_vma_teardown(struct proc_maps_locking_ctx *lock_ctx)
-{
- if (lock_ctx->mmap_locked)
- unlock_ctx_mm(lock_ctx);
- else
- unlock_ctx_vma(lock_ctx);
-}
-
static struct vm_area_struct *query_vma_find_by_addr(struct proc_maps_locking_ctx *lock_ctx,
unsigned long addr)
{
@@ -653,12 +619,7 @@ static int do_procmap_query(struct mm_struct *mm, void __user *uarg)
if (!mm || !mmget_not_zero(mm))
return -ESRCH;
- err = query_vma_setup(&lock_ctx);
- if (err) {
- mmput(mm);
- return err;
- }
-
+ reset_lock_ctx(&lock_ctx);
vma = query_matching_vma(&lock_ctx, karg.query_addr, karg.query_flags);
if (IS_ERR(vma)) {
err = PTR_ERR(vma);
@@ -732,7 +693,10 @@ static int do_procmap_query(struct mm_struct *mm, void __user *uarg)
vm_file = get_file(vma->vm_file);
/* unlock vma or mmap_lock, and put mm_struct before copying data to user */
- query_vma_teardown(&lock_ctx);
+ if (lock_ctx.mmap_locked)
+ unlock_ctx_mm(&lock_ctx);
+ else
+ unlock_ctx_vma(&lock_ctx);
mmput(mm);
if (karg.build_id_size) {
@@ -773,7 +737,10 @@ static int do_procmap_query(struct mm_struct *mm, void __user *uarg)
return 0;
out:
- query_vma_teardown(&lock_ctx);
+ if (lock_ctx.mmap_locked)
+ unlock_ctx_mm(&lock_ctx);
+ else
+ unlock_ctx_vma(&lock_ctx);
mmput(mm);
out_file:
if (vm_file)
--
2.55.0.1007.g17ff1f9808-goog
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v4 2/7] proc/task_mmu: remove unnecessary inlines in function definitions
2026-09-11 19:41 [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock Suren Baghdasaryan
2026-09-11 19:41 ` [PATCH v4 1/7] proc/task_mmu: remove unnecessary helpers Suren Baghdasaryan
@ 2026-09-11 19:41 ` Suren Baghdasaryan
2026-09-11 19:41 ` [PATCH v4 3/7] proc/task_mmu: clarify shmem mapping walk conditions in smap_gather_stats() Suren Baghdasaryan
` (5 subsequent siblings)
7 siblings, 0 replies; 17+ messages in thread
From: Suren Baghdasaryan @ 2026-09-11 19:41 UTC (permalink / raw)
To: akpm
Cc: liam, ljs, vbabka, david, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel, surenb,
Usama Arif
It was pointed out in the previous reviews of this code that many
functions are specified as inline, which is unnecessary as the compile
can make that decision by itself. Cleanup these definitions.
No change in the resulting binary file size with gcc v15.2.0.
No functional change intended.
Signed-off-by: Suren Baghdasaryan <surenb@google.com>
Reviewed-by: Liam R. Howlett (Oracle) <liam@infradead.org>
Reviewed-by: Lorenzo Stoakes (ARM) <ljs@kernel.org>
Acked-by: Usama Arif <usama.arif@linux.dev>
Acked-by: David Hildenbrand (Arm) <david@kernel.org>
---
fs/proc/task_mmu.c | 30 +++++++++++++++---------------
1 file changed, 15 insertions(+), 15 deletions(-)
diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c
index 2f500d639db5..cfc7af1b551d 100644
--- a/fs/proc/task_mmu.c
+++ b/fs/proc/task_mmu.c
@@ -130,7 +130,7 @@ static void release_task_mempolicy(struct proc_maps_private *priv)
}
#endif
-static inline int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx)
+static int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx)
{
int ret = mmap_read_lock_killable(lock_ctx->mm);
@@ -140,7 +140,7 @@ static inline int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx)
return ret;
}
-static inline void unlock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx)
+static void unlock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx)
{
mmap_read_unlock(lock_ctx->mm);
lock_ctx->mmap_locked = false;
@@ -177,8 +177,8 @@ static struct vm_area_struct *get_next_vma(struct proc_maps_private *priv,
return vma;
}
-static inline bool fallback_to_mmap_lock(struct proc_maps_private *priv,
- loff_t pos)
+static bool fallback_to_mmap_lock(struct proc_maps_private *priv,
+ loff_t pos)
{
struct proc_maps_locking_ctx *lock_ctx = &priv->lock_ctx;
@@ -194,7 +194,7 @@ static inline bool fallback_to_mmap_lock(struct proc_maps_private *priv,
return true;
}
-static inline void drop_rcu(struct proc_maps_private *priv)
+static void drop_rcu(struct proc_maps_private *priv)
{
if (priv->lock_ctx.mmap_locked)
return;
@@ -202,7 +202,7 @@ static inline void drop_rcu(struct proc_maps_private *priv)
rcu_read_unlock();
}
-static inline void reacquire_rcu(struct proc_maps_private *priv)
+static void reacquire_rcu(struct proc_maps_private *priv)
{
if (priv->lock_ctx.mmap_locked)
return;
@@ -1230,7 +1230,7 @@ static const struct mm_walk_ops smaps_shmem_walk_vma_lock_ops = {
.walk_lock = PGWALK_VMA_RDLOCK_VERIFY,
};
-static inline const struct mm_walk_ops *
+static const struct mm_walk_ops *
get_smaps_walk_ops(struct proc_maps_private *priv)
{
if (priv->lock_ctx.mmap_locked)
@@ -1238,7 +1238,7 @@ get_smaps_walk_ops(struct proc_maps_private *priv)
return &smaps_walk_vma_lock_ops;
}
-static inline const struct mm_walk_ops *
+static const struct mm_walk_ops *
get_smaps_shmem_walk_ops(struct proc_maps_private *priv)
{
if (priv->lock_ctx.mmap_locked)
@@ -1572,7 +1572,7 @@ struct clear_refs_private {
enum clear_refs_types type;
};
-static inline bool pte_is_pinned(struct vm_area_struct *vma, unsigned long addr, pte_t pte)
+static bool pte_is_pinned(struct vm_area_struct *vma, unsigned long addr, pte_t pte)
{
struct folio *folio;
@@ -1588,8 +1588,8 @@ static inline bool pte_is_pinned(struct vm_area_struct *vma, unsigned long addr,
return folio_maybe_dma_pinned(folio);
}
-static inline void clear_soft_dirty(struct vm_area_struct *vma,
- unsigned long addr, pte_t *pte)
+static void clear_soft_dirty(struct vm_area_struct *vma, unsigned long addr,
+ pte_t *pte)
{
if (!pgtable_supports_soft_dirty())
return;
@@ -1620,7 +1620,7 @@ static inline void clear_soft_dirty(struct vm_area_struct *vma,
}
#if defined(CONFIG_TRANSPARENT_HUGEPAGE)
-static inline void clear_soft_dirty_pmd(struct vm_area_struct *vma,
+static void clear_soft_dirty_pmd(struct vm_area_struct *vma,
unsigned long addr, pmd_t *pmdp)
{
pmd_t old, pmd = *pmdp;
@@ -1646,7 +1646,7 @@ static inline void clear_soft_dirty_pmd(struct vm_area_struct *vma,
}
}
#else
-static inline void clear_soft_dirty_pmd(struct vm_area_struct *vma,
+static void clear_soft_dirty_pmd(struct vm_area_struct *vma,
unsigned long addr, pmd_t *pmdp)
{
}
@@ -1846,7 +1846,7 @@ struct pagemapread {
#define PM_END_OF_BUFFER 1
-static inline pagemap_entry_t make_pme(u64 frame, u64 flags)
+static pagemap_entry_t make_pme(u64 frame, u64 flags)
{
return (pagemap_entry_t) { .pme = (frame & PM_PFRAME_MASK) | flags };
}
@@ -3388,7 +3388,7 @@ static const struct mm_walk_ops show_numa_vma_lock_ops = {
.walk_lock = PGWALK_VMA_RDLOCK_VERIFY,
};
-static inline const struct mm_walk_ops *
+static const struct mm_walk_ops *
get_show_numa_ops(struct proc_maps_private *priv)
{
if (priv->lock_ctx.mmap_locked)
--
2.55.0.1007.g17ff1f9808-goog
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v4 3/7] proc/task_mmu: clarify shmem mapping walk conditions in smap_gather_stats()
2026-09-11 19:41 [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock Suren Baghdasaryan
2026-09-11 19:41 ` [PATCH v4 1/7] proc/task_mmu: remove unnecessary helpers Suren Baghdasaryan
2026-09-11 19:41 ` [PATCH v4 2/7] proc/task_mmu: remove unnecessary inlines in function definitions Suren Baghdasaryan
@ 2026-09-11 19:41 ` Suren Baghdasaryan
2026-09-11 19:44 ` Lorenzo Stoakes (ARM)
2026-09-11 21:00 ` David Hildenbrand (Arm)
2026-09-11 19:41 ` [PATCH v4 4/7] proc/task_mmu: remove special-casing of smap_gather_stats() start parameter Suren Baghdasaryan
` (4 subsequent siblings)
7 siblings, 2 replies; 17+ messages in thread
From: Suren Baghdasaryan @ 2026-09-11 19:41 UTC (permalink / raw)
To: akpm
Cc: liam, ljs, vbabka, david, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel, surenb
smap_gather_stats() optimizes stats gathering by skipping the walk for
shmem mappings in certain conditions. Update the comment to clarify
these conditions and use vma_is_cow_mapping() for CoW identification
instead of open-coding it.
No functional change intended.
Suggested by: David Hildenbrand (Arm) <david@kernel.org>
Signed-off-by: Suren Baghdasaryan <surenb@google.com>
---
fs/proc/task_mmu.c | 23 +++++++++--------------
1 file changed, 9 insertions(+), 14 deletions(-)
diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c
index cfc7af1b551d..0e53eb065a3e 100644
--- a/fs/proc/task_mmu.c
+++ b/fs/proc/task_mmu.c
@@ -1270,23 +1270,18 @@ static void smap_gather_stats(struct proc_maps_private *priv,
if (vma->vm_file && shmem_mapping(vma->vm_file->f_mapping)) {
/*
- * For shared or readonly shmem mappings we know that all
- * swapped out pages belong to the shmem object, and we can
- * obtain the swap value much more efficiently. For private
- * writable mappings, we might have COW pages that are
- * not affected by the parent swapped out pages of the shmem
- * object, so we have to distinguish them during the page walk.
- * Unless we know that the shmem object (or the part mapped by
- * our VMA) has no swapped out pages at all.
+ * CoW mappings might map anon folios that do not belong to
+ * shmem. Perform a less efficient page table walk in this
+ * situation, unless we know that the shmem object (or the
+ * part mapped by our VMA) has no swapped out pages at all.
*/
- unsigned long shmem_swapped = shmem_swap_usage(vma);
+ const unsigned long shmem_swapped = shmem_swap_usage(vma);
+ const bool is_cow = vma_is_cow_mapping(vma);
- if (!start && (!shmem_swapped || (vma->vm_flags & VM_SHARED) ||
- !(vma->vm_flags & VM_WRITE))) {
- mss->swap += shmem_swapped;
- } else {
+ if (start || (shmem_swapped && is_cow))
ops = get_smaps_shmem_walk_ops(priv);
- }
+ else
+ mss->swap += shmem_swapped;
}
if (!start)
--
2.55.0.1007.g17ff1f9808-goog
^ permalink raw reply [flat|nested] 17+ messages in thread* Re: [PATCH v4 3/7] proc/task_mmu: clarify shmem mapping walk conditions in smap_gather_stats()
2026-09-11 19:41 ` [PATCH v4 3/7] proc/task_mmu: clarify shmem mapping walk conditions in smap_gather_stats() Suren Baghdasaryan
@ 2026-09-11 19:44 ` Lorenzo Stoakes (ARM)
2026-09-11 21:00 ` David Hildenbrand (Arm)
1 sibling, 0 replies; 17+ messages in thread
From: Lorenzo Stoakes (ARM) @ 2026-09-11 19:44 UTC (permalink / raw)
To: Suren Baghdasaryan
Cc: akpm, liam, vbabka, david, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel
On Fri, Sep 11, 2026 at 12:41:41PM -0700, Suren Baghdasaryan wrote:
> smap_gather_stats() optimizes stats gathering by skipping the walk for
> shmem mappings in certain conditions. Update the comment to clarify
> these conditions and use vma_is_cow_mapping() for CoW identification
> instead of open-coding it.
>
> No functional change intended.
>
> Suggested by: David Hildenbrand (Arm) <david@kernel.org>
> Signed-off-by: Suren Baghdasaryan <surenb@google.com>
LGTM so:
Reviewed-by: Lorenzo Stoakes (ARM) <ljs@kernel.org>
> ---
> fs/proc/task_mmu.c | 23 +++++++++--------------
> 1 file changed, 9 insertions(+), 14 deletions(-)
>
> diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c
> index cfc7af1b551d..0e53eb065a3e 100644
> --- a/fs/proc/task_mmu.c
> +++ b/fs/proc/task_mmu.c
> @@ -1270,23 +1270,18 @@ static void smap_gather_stats(struct proc_maps_private *priv,
>
> if (vma->vm_file && shmem_mapping(vma->vm_file->f_mapping)) {
> /*
> - * For shared or readonly shmem mappings we know that all
> - * swapped out pages belong to the shmem object, and we can
> - * obtain the swap value much more efficiently. For private
> - * writable mappings, we might have COW pages that are
> - * not affected by the parent swapped out pages of the shmem
> - * object, so we have to distinguish them during the page walk.
> - * Unless we know that the shmem object (or the part mapped by
> - * our VMA) has no swapped out pages at all.
> + * CoW mappings might map anon folios that do not belong to
> + * shmem. Perform a less efficient page table walk in this
> + * situation, unless we know that the shmem object (or the
> + * part mapped by our VMA) has no swapped out pages at all.
> */
> - unsigned long shmem_swapped = shmem_swap_usage(vma);
> + const unsigned long shmem_swapped = shmem_swap_usage(vma);
> + const bool is_cow = vma_is_cow_mapping(vma);
>
> - if (!start && (!shmem_swapped || (vma->vm_flags & VM_SHARED) ||
> - !(vma->vm_flags & VM_WRITE))) {
> - mss->swap += shmem_swapped;
> - } else {
> + if (start || (shmem_swapped && is_cow))
> ops = get_smaps_shmem_walk_ops(priv);
> - }
> + else
> + mss->swap += shmem_swapped;
> }
>
> if (!start)
> --
> 2.55.0.1007.g17ff1f9808-goog
>
--
Cheers, Lorenzo
^ permalink raw reply [flat|nested] 17+ messages in thread* Re: [PATCH v4 3/7] proc/task_mmu: clarify shmem mapping walk conditions in smap_gather_stats()
2026-09-11 19:41 ` [PATCH v4 3/7] proc/task_mmu: clarify shmem mapping walk conditions in smap_gather_stats() Suren Baghdasaryan
2026-09-11 19:44 ` Lorenzo Stoakes (ARM)
@ 2026-09-11 21:00 ` David Hildenbrand (Arm)
2026-09-11 21:31 ` Suren Baghdasaryan
1 sibling, 1 reply; 17+ messages in thread
From: David Hildenbrand (Arm) @ 2026-09-11 21:00 UTC (permalink / raw)
To: Suren Baghdasaryan, akpm
Cc: liam, ljs, vbabka, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel
On 9/11/26 21:41, Suren Baghdasaryan wrote:
> smap_gather_stats() optimizes stats gathering by skipping the walk for
> shmem mappings in certain conditions. Update the comment to clarify
> these conditions and use vma_is_cow_mapping() for CoW identification
> instead of open-coding it.
>
> No functional change intended.
>
> Suggested by: David Hildenbrand (Arm) <david@kernel.org>
> Signed-off-by: Suren Baghdasaryan <surenb@google.com>
> ---
Acked-by: David Hildenbrand (Arm) <david@kernel.org>
--
Cheers,
David
^ permalink raw reply [flat|nested] 17+ messages in thread
* Re: [PATCH v4 3/7] proc/task_mmu: clarify shmem mapping walk conditions in smap_gather_stats()
2026-09-11 21:00 ` David Hildenbrand (Arm)
@ 2026-09-11 21:31 ` Suren Baghdasaryan
0 siblings, 0 replies; 17+ messages in thread
From: Suren Baghdasaryan @ 2026-09-11 21:31 UTC (permalink / raw)
To: David Hildenbrand (Arm)
Cc: akpm, liam, ljs, vbabka, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel
On Fri, Sep 11, 2026 at 2:00 PM David Hildenbrand (Arm)
<david@kernel.org> wrote:
>
> On 9/11/26 21:41, Suren Baghdasaryan wrote:
> > smap_gather_stats() optimizes stats gathering by skipping the walk for
> > shmem mappings in certain conditions. Update the comment to clarify
> > these conditions and use vma_is_cow_mapping() for CoW identification
> > instead of open-coding it.
> >
> > No functional change intended.
> >
> > Suggested by: David Hildenbrand (Arm) <david@kernel.org>
> > Signed-off-by: Suren Baghdasaryan <surenb@google.com>
> > ---
>
> Acked-by: David Hildenbrand (Arm) <david@kernel.org>
Thanks!
>
> --
> Cheers,
>
> David
^ permalink raw reply [flat|nested] 17+ messages in thread
* [PATCH v4 4/7] proc/task_mmu: remove special-casing of smap_gather_stats() start parameter
2026-09-11 19:41 [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock Suren Baghdasaryan
` (2 preceding siblings ...)
2026-09-11 19:41 ` [PATCH v4 3/7] proc/task_mmu: clarify shmem mapping walk conditions in smap_gather_stats() Suren Baghdasaryan
@ 2026-09-11 19:41 ` Suren Baghdasaryan
2026-09-11 19:45 ` Lorenzo Stoakes (ARM)
2026-09-14 10:38 ` David Hildenbrand (Arm)
2026-09-11 19:41 ` [PATCH v4 5/7] proc/task_mmu: change proc_get_vma() to stop returning gate VMA at the end Suren Baghdasaryan
` (3 subsequent siblings)
7 siblings, 2 replies; 17+ messages in thread
From: Suren Baghdasaryan @ 2026-09-11 19:41 UTC (permalink / raw)
To: akpm
Cc: liam, ljs, vbabka, david, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel, surenb
smap_gather_stats() interprets its start parameter to mean vma->vm_start
when it's set to 0. Eliminate this special interpretation and provide two
separate functions for a partial and complete VMA walk.
Since smap_gather_stats() operates within a single VMA, we can replace
walk_page_vma()/walk_page_range() calls with walk_page_range_vma()
which is simpler and also can be called while holding per-VMA lock.
No functional change intended.
Suggested by: Lorenzo Stoakes <ljs@kernel.org>
Signed-off-by: Suren Baghdasaryan <surenb@google.com>
---
fs/proc/task_mmu.c | 52 +++++++++++++++++++++++++++++++---------------
1 file changed, 35 insertions(+), 17 deletions(-)
diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c
index 0e53eb065a3e..a6026ffd07f1 100644
--- a/fs/proc/task_mmu.c
+++ b/fs/proc/task_mmu.c
@@ -1246,20 +1246,26 @@ get_smaps_shmem_walk_ops(struct proc_maps_private *priv)
return &smaps_shmem_walk_vma_lock_ops;
}
-/*
- * Gather mem stats from @vma with the indicated beginning
- * address @start, and keep them in @mss.
+/**
+ * smap_gather_stats_range() - Gather mem stats from a portion of the @vma.
+ * @priv: proc maps private state.
+ * @vma: The VMA to gather stats for.
+ * @mss: The accumulated stats.
+ * @start: The address from which to start.
*
- * Use vm_start of @vma as the beginning address if @start is 0.
+ * This gathers stats for the portion of the VMA starting at the @start
+ * address.
*/
-static void smap_gather_stats(struct proc_maps_private *priv,
- struct vm_area_struct *vma,
- struct mem_size_stats *mss, unsigned long start)
+static void smap_gather_stats_range(struct proc_maps_private *priv,
+ struct vm_area_struct *vma,
+ struct mem_size_stats *mss,
+ unsigned long start)
{
const struct mm_walk_ops *ops = get_smaps_walk_ops(priv);
+ const bool is_partial = start > vma->vm_start;
/* Invalid start */
- if (start >= vma->vm_end)
+ if (start < vma->vm_start || start >= vma->vm_end)
return;
if (vma == get_gate_vma(priv->lock_ctx.mm))
@@ -1278,20 +1284,31 @@ static void smap_gather_stats(struct proc_maps_private *priv,
const unsigned long shmem_swapped = shmem_swap_usage(vma);
const bool is_cow = vma_is_cow_mapping(vma);
- if (start || (shmem_swapped && is_cow))
+ if (is_partial || (shmem_swapped && is_cow))
ops = get_smaps_shmem_walk_ops(priv);
else
mss->swap += shmem_swapped;
}
- if (!start)
- walk_page_vma(vma, ops, mss);
- else
- walk_page_range(vma->vm_mm, start, vma->vm_end, ops, mss);
+ walk_page_range_vma(vma, start, vma->vm_end, ops, mss);
reacquire_rcu(priv);
}
+/**
+ * smap_gather_stats() - Gather mem stats from the entire @vma.
+ * @priv: proc maps private state.
+ * @vma: The VMA to gather stats for.
+ * @mss: The accumulated stats.
+ *
+ * This gathers stats for the whole of the VMA.
+ */
+static void smap_gather_stats(struct proc_maps_private *priv,
+ struct vm_area_struct *vma, struct mem_size_stats *mss)
+{
+ smap_gather_stats_range(priv, vma, mss, vma->vm_start);
+}
+
#define SEQ_PUT_DEC(str, val) \
seq_put_decimal_ull_width(m, str, (val) >> 10, 8)
@@ -1342,7 +1359,7 @@ static int show_smap(struct seq_file *m, void *v)
struct vm_area_struct *vma = v;
struct mem_size_stats mss = {};
- smap_gather_stats(priv, vma, &mss, 0);
+ smap_gather_stats(priv, vma, &mss);
show_map_vma(m, vma);
@@ -1395,7 +1412,7 @@ static int show_smaps_rollup(struct seq_file *m, void *v)
vma_start = vma->vm_start;
do {
- smap_gather_stats(priv, vma, &mss, 0);
+ smap_gather_stats(priv, vma, &mss);
last_vma_end = vma->vm_end;
/*
@@ -1454,14 +1471,15 @@ static int show_smaps_rollup(struct seq_file *m, void *v)
/* Case 1 and 2 above */
if (vma->vm_start >= last_vma_end) {
- smap_gather_stats(priv, vma, &mss, 0);
+ smap_gather_stats(priv, vma, &mss);
last_vma_end = vma->vm_end;
continue;
}
/* Case 4 above */
if (vma->vm_end > last_vma_end) {
- smap_gather_stats(priv, vma, &mss, last_vma_end);
+ smap_gather_stats_range(priv, vma, &mss,
+ last_vma_end);
last_vma_end = vma->vm_end;
}
}
--
2.55.0.1007.g17ff1f9808-goog
^ permalink raw reply [flat|nested] 17+ messages in thread* Re: [PATCH v4 4/7] proc/task_mmu: remove special-casing of smap_gather_stats() start parameter
2026-09-11 19:41 ` [PATCH v4 4/7] proc/task_mmu: remove special-casing of smap_gather_stats() start parameter Suren Baghdasaryan
@ 2026-09-11 19:45 ` Lorenzo Stoakes (ARM)
2026-09-11 19:47 ` Suren Baghdasaryan
2026-09-14 10:38 ` David Hildenbrand (Arm)
1 sibling, 1 reply; 17+ messages in thread
From: Lorenzo Stoakes (ARM) @ 2026-09-11 19:45 UTC (permalink / raw)
To: Suren Baghdasaryan
Cc: akpm, liam, vbabka, david, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel
On Fri, Sep 11, 2026 at 12:41:42PM -0700, Suren Baghdasaryan wrote:
> smap_gather_stats() interprets its start parameter to mean vma->vm_start
> when it's set to 0. Eliminate this special interpretation and provide two
> separate functions for a partial and complete VMA walk.
>
> Since smap_gather_stats() operates within a single VMA, we can replace
> walk_page_vma()/walk_page_range() calls with walk_page_range_vma()
> which is simpler and also can be called while holding per-VMA lock.
>
> No functional change intended.
>
> Suggested by: Lorenzo Stoakes <ljs@kernel.org>
> Signed-off-by: Suren Baghdasaryan <surenb@google.com>
LGTM so:
Reviewed-by: Lorenzo Stoakes (ARM) <ljs@kernel.org>
> ---
> fs/proc/task_mmu.c | 52 +++++++++++++++++++++++++++++++---------------
> 1 file changed, 35 insertions(+), 17 deletions(-)
>
> diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c
> index 0e53eb065a3e..a6026ffd07f1 100644
> --- a/fs/proc/task_mmu.c
> +++ b/fs/proc/task_mmu.c
> @@ -1246,20 +1246,26 @@ get_smaps_shmem_walk_ops(struct proc_maps_private *priv)
> return &smaps_shmem_walk_vma_lock_ops;
> }
>
> -/*
> - * Gather mem stats from @vma with the indicated beginning
> - * address @start, and keep them in @mss.
> +/**
> + * smap_gather_stats_range() - Gather mem stats from a portion of the @vma.
> + * @priv: proc maps private state.
> + * @vma: The VMA to gather stats for.
> + * @mss: The accumulated stats.
> + * @start: The address from which to start.
> *
> - * Use vm_start of @vma as the beginning address if @start is 0.
> + * This gathers stats for the portion of the VMA starting at the @start
> + * address.
> */
> -static void smap_gather_stats(struct proc_maps_private *priv,
> - struct vm_area_struct *vma,
> - struct mem_size_stats *mss, unsigned long start)
> +static void smap_gather_stats_range(struct proc_maps_private *priv,
> + struct vm_area_struct *vma,
> + struct mem_size_stats *mss,
> + unsigned long start)
> {
> const struct mm_walk_ops *ops = get_smaps_walk_ops(priv);
> + const bool is_partial = start > vma->vm_start;
>
> /* Invalid start */
> - if (start >= vma->vm_end)
> + if (start < vma->vm_start || start >= vma->vm_end)
> return;
>
> if (vma == get_gate_vma(priv->lock_ctx.mm))
> @@ -1278,20 +1284,31 @@ static void smap_gather_stats(struct proc_maps_private *priv,
> const unsigned long shmem_swapped = shmem_swap_usage(vma);
> const bool is_cow = vma_is_cow_mapping(vma);
>
> - if (start || (shmem_swapped && is_cow))
> + if (is_partial || (shmem_swapped && is_cow))
> ops = get_smaps_shmem_walk_ops(priv);
> else
> mss->swap += shmem_swapped;
> }
>
> - if (!start)
> - walk_page_vma(vma, ops, mss);
> - else
> - walk_page_range(vma->vm_mm, start, vma->vm_end, ops, mss);
> + walk_page_range_vma(vma, start, vma->vm_end, ops, mss);
>
> reacquire_rcu(priv);
> }
>
> +/**
> + * smap_gather_stats() - Gather mem stats from the entire @vma.
> + * @priv: proc maps private state.
> + * @vma: The VMA to gather stats for.
> + * @mss: The accumulated stats.
> + *
> + * This gathers stats for the whole of the VMA.
> + */
> +static void smap_gather_stats(struct proc_maps_private *priv,
> + struct vm_area_struct *vma, struct mem_size_stats *mss)
> +{
> + smap_gather_stats_range(priv, vma, mss, vma->vm_start);
> +}
> +
> #define SEQ_PUT_DEC(str, val) \
> seq_put_decimal_ull_width(m, str, (val) >> 10, 8)
>
> @@ -1342,7 +1359,7 @@ static int show_smap(struct seq_file *m, void *v)
> struct vm_area_struct *vma = v;
> struct mem_size_stats mss = {};
>
> - smap_gather_stats(priv, vma, &mss, 0);
> + smap_gather_stats(priv, vma, &mss);
>
> show_map_vma(m, vma);
>
> @@ -1395,7 +1412,7 @@ static int show_smaps_rollup(struct seq_file *m, void *v)
>
> vma_start = vma->vm_start;
> do {
> - smap_gather_stats(priv, vma, &mss, 0);
> + smap_gather_stats(priv, vma, &mss);
> last_vma_end = vma->vm_end;
>
> /*
> @@ -1454,14 +1471,15 @@ static int show_smaps_rollup(struct seq_file *m, void *v)
>
> /* Case 1 and 2 above */
> if (vma->vm_start >= last_vma_end) {
> - smap_gather_stats(priv, vma, &mss, 0);
> + smap_gather_stats(priv, vma, &mss);
> last_vma_end = vma->vm_end;
> continue;
> }
>
> /* Case 4 above */
> if (vma->vm_end > last_vma_end) {
> - smap_gather_stats(priv, vma, &mss, last_vma_end);
> + smap_gather_stats_range(priv, vma, &mss,
> + last_vma_end);
> last_vma_end = vma->vm_end;
> }
> }
> --
> 2.55.0.1007.g17ff1f9808-goog
>
--
Cheers, Lorenzo
^ permalink raw reply [flat|nested] 17+ messages in thread* Re: [PATCH v4 4/7] proc/task_mmu: remove special-casing of smap_gather_stats() start parameter
2026-09-11 19:45 ` Lorenzo Stoakes (ARM)
@ 2026-09-11 19:47 ` Suren Baghdasaryan
0 siblings, 0 replies; 17+ messages in thread
From: Suren Baghdasaryan @ 2026-09-11 19:47 UTC (permalink / raw)
To: Lorenzo Stoakes (ARM)
Cc: akpm, liam, vbabka, david, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel
On Fri, Sep 11, 2026 at 12:45 PM Lorenzo Stoakes (ARM) <ljs@kernel.org> wrote:
>
> On Fri, Sep 11, 2026 at 12:41:42PM -0700, Suren Baghdasaryan wrote:
> > smap_gather_stats() interprets its start parameter to mean vma->vm_start
> > when it's set to 0. Eliminate this special interpretation and provide two
> > separate functions for a partial and complete VMA walk.
> >
> > Since smap_gather_stats() operates within a single VMA, we can replace
> > walk_page_vma()/walk_page_range() calls with walk_page_range_vma()
> > which is simpler and also can be called while holding per-VMA lock.
> >
> > No functional change intended.
> >
> > Suggested by: Lorenzo Stoakes <ljs@kernel.org>
> > Signed-off-by: Suren Baghdasaryan <surenb@google.com>
>
> LGTM so:
>
> Reviewed-by: Lorenzo Stoakes (ARM) <ljs@kernel.org>
Thanks!
>
> > ---
> > fs/proc/task_mmu.c | 52 +++++++++++++++++++++++++++++++---------------
> > 1 file changed, 35 insertions(+), 17 deletions(-)
> >
> > diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c
> > index 0e53eb065a3e..a6026ffd07f1 100644
> > --- a/fs/proc/task_mmu.c
> > +++ b/fs/proc/task_mmu.c
> > @@ -1246,20 +1246,26 @@ get_smaps_shmem_walk_ops(struct proc_maps_private *priv)
> > return &smaps_shmem_walk_vma_lock_ops;
> > }
> >
> > -/*
> > - * Gather mem stats from @vma with the indicated beginning
> > - * address @start, and keep them in @mss.
> > +/**
> > + * smap_gather_stats_range() - Gather mem stats from a portion of the @vma.
> > + * @priv: proc maps private state.
> > + * @vma: The VMA to gather stats for.
> > + * @mss: The accumulated stats.
> > + * @start: The address from which to start.
> > *
> > - * Use vm_start of @vma as the beginning address if @start is 0.
> > + * This gathers stats for the portion of the VMA starting at the @start
> > + * address.
> > */
> > -static void smap_gather_stats(struct proc_maps_private *priv,
> > - struct vm_area_struct *vma,
> > - struct mem_size_stats *mss, unsigned long start)
> > +static void smap_gather_stats_range(struct proc_maps_private *priv,
> > + struct vm_area_struct *vma,
> > + struct mem_size_stats *mss,
> > + unsigned long start)
> > {
> > const struct mm_walk_ops *ops = get_smaps_walk_ops(priv);
> > + const bool is_partial = start > vma->vm_start;
> >
> > /* Invalid start */
> > - if (start >= vma->vm_end)
> > + if (start < vma->vm_start || start >= vma->vm_end)
> > return;
> >
> > if (vma == get_gate_vma(priv->lock_ctx.mm))
> > @@ -1278,20 +1284,31 @@ static void smap_gather_stats(struct proc_maps_private *priv,
> > const unsigned long shmem_swapped = shmem_swap_usage(vma);
> > const bool is_cow = vma_is_cow_mapping(vma);
> >
> > - if (start || (shmem_swapped && is_cow))
> > + if (is_partial || (shmem_swapped && is_cow))
> > ops = get_smaps_shmem_walk_ops(priv);
> > else
> > mss->swap += shmem_swapped;
> > }
> >
> > - if (!start)
> > - walk_page_vma(vma, ops, mss);
> > - else
> > - walk_page_range(vma->vm_mm, start, vma->vm_end, ops, mss);
> > + walk_page_range_vma(vma, start, vma->vm_end, ops, mss);
> >
> > reacquire_rcu(priv);
> > }
> >
> > +/**
> > + * smap_gather_stats() - Gather mem stats from the entire @vma.
> > + * @priv: proc maps private state.
> > + * @vma: The VMA to gather stats for.
> > + * @mss: The accumulated stats.
> > + *
> > + * This gathers stats for the whole of the VMA.
> > + */
> > +static void smap_gather_stats(struct proc_maps_private *priv,
> > + struct vm_area_struct *vma, struct mem_size_stats *mss)
> > +{
> > + smap_gather_stats_range(priv, vma, mss, vma->vm_start);
> > +}
> > +
> > #define SEQ_PUT_DEC(str, val) \
> > seq_put_decimal_ull_width(m, str, (val) >> 10, 8)
> >
> > @@ -1342,7 +1359,7 @@ static int show_smap(struct seq_file *m, void *v)
> > struct vm_area_struct *vma = v;
> > struct mem_size_stats mss = {};
> >
> > - smap_gather_stats(priv, vma, &mss, 0);
> > + smap_gather_stats(priv, vma, &mss);
> >
> > show_map_vma(m, vma);
> >
> > @@ -1395,7 +1412,7 @@ static int show_smaps_rollup(struct seq_file *m, void *v)
> >
> > vma_start = vma->vm_start;
> > do {
> > - smap_gather_stats(priv, vma, &mss, 0);
> > + smap_gather_stats(priv, vma, &mss);
> > last_vma_end = vma->vm_end;
> >
> > /*
> > @@ -1454,14 +1471,15 @@ static int show_smaps_rollup(struct seq_file *m, void *v)
> >
> > /* Case 1 and 2 above */
> > if (vma->vm_start >= last_vma_end) {
> > - smap_gather_stats(priv, vma, &mss, 0);
> > + smap_gather_stats(priv, vma, &mss);
> > last_vma_end = vma->vm_end;
> > continue;
> > }
> >
> > /* Case 4 above */
> > if (vma->vm_end > last_vma_end) {
> > - smap_gather_stats(priv, vma, &mss, last_vma_end);
> > + smap_gather_stats_range(priv, vma, &mss,
> > + last_vma_end);
> > last_vma_end = vma->vm_end;
> > }
> > }
> > --
> > 2.55.0.1007.g17ff1f9808-goog
> >
>
> --
> Cheers, Lorenzo
^ permalink raw reply [flat|nested] 17+ messages in thread
* Re: [PATCH v4 4/7] proc/task_mmu: remove special-casing of smap_gather_stats() start parameter
2026-09-11 19:41 ` [PATCH v4 4/7] proc/task_mmu: remove special-casing of smap_gather_stats() start parameter Suren Baghdasaryan
2026-09-11 19:45 ` Lorenzo Stoakes (ARM)
@ 2026-09-14 10:38 ` David Hildenbrand (Arm)
1 sibling, 0 replies; 17+ messages in thread
From: David Hildenbrand (Arm) @ 2026-09-14 10:38 UTC (permalink / raw)
To: Suren Baghdasaryan, akpm
Cc: liam, ljs, vbabka, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel
On 9/11/26 21:41, Suren Baghdasaryan wrote:
> smap_gather_stats() interprets its start parameter to mean vma->vm_start
> when it's set to 0. Eliminate this special interpretation and provide two
> separate functions for a partial and complete VMA walk.
>
> Since smap_gather_stats() operates within a single VMA, we can replace
> walk_page_vma()/walk_page_range() calls with walk_page_range_vma()
> which is simpler and also can be called while holding per-VMA lock.
>
> No functional change intended.
>
> Suggested by: Lorenzo Stoakes <ljs@kernel.org>
> Signed-off-by: Suren Baghdasaryan <surenb@google.com>
> ---
Acked-by: David Hildenbrand (Arm) <david@kernel.org>
--
Cheers,
David
^ permalink raw reply [flat|nested] 17+ messages in thread
* [PATCH v4 5/7] proc/task_mmu: change proc_get_vma() to stop returning gate VMA at the end
2026-09-11 19:41 [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock Suren Baghdasaryan
` (3 preceding siblings ...)
2026-09-11 19:41 ` [PATCH v4 4/7] proc/task_mmu: remove special-casing of smap_gather_stats() start parameter Suren Baghdasaryan
@ 2026-09-11 19:41 ` Suren Baghdasaryan
2026-09-11 19:41 ` [PATCH v4 6/7] proc/task_mmu: read proc/pid/smaps_rollup under per-vma lock Suren Baghdasaryan
` (2 subsequent siblings)
7 siblings, 0 replies; 17+ messages in thread
From: Suren Baghdasaryan @ 2026-09-11 19:41 UTC (permalink / raw)
To: akpm
Cc: liam, ljs, vbabka, david, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel, surenb
proc_get_vma() returning gate VMA at the end is desirable for the its
current m_start/m_next callers, as they need to report a gate VMA at the
end of the address space. This behavior is very specific to these callers
and makes proc_get_vma() hard to use for other purposes.
Move this usage-specific behavior into the callers themselves so that
proc_get_vma() returns either a valid VMA, an error or a NULL when no
more VMAs are available. This makes it more generic, simpler and usable
in the later patches.
Signed-off-by: Suren Baghdasaryan <surenb@google.com>
Reviewed-by: Lorenzo Stoakes (ARM) <ljs@kernel.org>
Acked-by: David Hildenbrand (Arm) <david@kernel.org>
---
fs/proc/task_mmu.c | 28 +++++++++++++++++++++++-----
1 file changed, 23 insertions(+), 5 deletions(-)
diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c
index a6026ffd07f1..aef7ce659889 100644
--- a/fs/proc/task_mmu.c
+++ b/fs/proc/task_mmu.c
@@ -236,9 +236,6 @@ static struct vm_area_struct *proc_get_vma(struct seq_file *m, loff_t *ppos)
* found the extended vma with the same vm_start.
*/
*ppos = vma->vm_end;
- } else {
- *ppos = SENTINEL_VMA_GATE;
- vma = get_gate_vma(priv->lock_ctx.mm);
}
return vma;
@@ -248,6 +245,7 @@ static void *m_start(struct seq_file *m, loff_t *ppos)
{
struct proc_maps_private *priv = m->private;
struct proc_maps_locking_ctx *lock_ctx;
+ struct vm_area_struct *vma;
loff_t last_addr = *ppos;
struct mm_struct *mm;
@@ -277,19 +275,39 @@ static void *m_start(struct seq_file *m, loff_t *ppos)
*ppos = last_addr = priv->last_pos;
vma_iter_init(&priv->iter, mm, (unsigned long)last_addr);
hold_task_mempolicy(priv);
+ /*
+ * If seq_file had to flush its collected data right after m_next() set
+ * position to SENTINEL_VMA_GATE, m_start() will get that sentinel and
+ * should return gate_vma without calling proc_get_vma().
+ */
if (last_addr == SENTINEL_VMA_GATE)
return get_gate_vma(mm);
- return proc_get_vma(m, ppos);
+ vma = proc_get_vma(m, ppos);
+ if (vma)
+ return vma;
+
+ /* Return gate VMA at the end */
+ *ppos = SENTINEL_VMA_GATE;
+ return get_gate_vma(mm);
}
static void *m_next(struct seq_file *m, void *v, loff_t *ppos)
{
+ struct proc_maps_private *priv = m->private;
+ struct vm_area_struct *vma;
+
if (*ppos == SENTINEL_VMA_GATE) {
*ppos = SENTINEL_VMA_END;
return NULL;
}
- return proc_get_vma(m, ppos);
+ vma = proc_get_vma(m, ppos);
+ if (vma)
+ return vma;
+
+ /* Return gate VMA at the end */
+ *ppos = SENTINEL_VMA_GATE;
+ return get_gate_vma(priv->lock_ctx.mm);
}
static void m_stop(struct seq_file *m, void *v)
--
2.55.0.1007.g17ff1f9808-goog
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v4 6/7] proc/task_mmu: read proc/pid/smaps_rollup under per-vma lock
2026-09-11 19:41 [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock Suren Baghdasaryan
` (4 preceding siblings ...)
2026-09-11 19:41 ` [PATCH v4 5/7] proc/task_mmu: change proc_get_vma() to stop returning gate VMA at the end Suren Baghdasaryan
@ 2026-09-11 19:41 ` Suren Baghdasaryan
2026-09-11 19:41 ` [PATCH v4 7/7] selftests/proc: add /proc/pid/smaps_rollup tearing tests Suren Baghdasaryan
2026-09-12 7:24 ` [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock Andrew Morton
7 siblings, 0 replies; 17+ messages in thread
From: Suren Baghdasaryan @ 2026-09-11 19:41 UTC (permalink / raw)
To: akpm
Cc: liam, ljs, vbabka, david, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel, surenb
proc/pid/smaps_rollup can be read using the combination of RCU and
VMA read locks, similar to proc/pid/{maps|smaps|numa_maps}. RCU is
required to safely traverse the VMA tree and VMA lock stabilizes the
VMA being processed and the pagetable walk.
Note that we have to keep the logic to drop mmap_lock on contention
because even when using per-VMA locks we might have to fall back to
holding the mmap_lock.
Running Paul's contention benchmark [1] shows considerable improvement
both in median and in the worst case latencies:
Execution command: run-proc-vs-map.sh --nsamples 20 --rawdata -- \
--busyduration 2 --procfile smaps_rollup
Baseline:
Median Minimum Maximum
0.174 0.161 2.553
0.174 0.164 2.663
0.174 0.165 2.664
0.174 0.166 2.679
0.174 0.167 2.691
0.174 0.168 2.704
0.174 0.169 2.729
0.174 0.172 2.741
0.174 0.174 2.745
0.174 0.174 2.755
0.174 0.175 2.790
0.174 0.177 2.809
0.174 0.179 3.096
0.174 0.183 3.144
0.174 0.184 3.158
0.174 0.185 3.175
0.174 0.185 4.568
0.174 0.198 4.821
0.174 0.214 5.143
0.174 0.251 5.220
Patched:
Median Minimum Maximum
0.007 0.007 1.952
0.007 0.007 1.955
0.007 0.007 1.955
0.007 0.007 1.955
0.007 0.007 1.957
0.007 0.007 1.969
0.007 0.007 2.065
0.007 0.007 2.075
0.007 0.007 2.146
0.007 0.007 2.195
0.007 0.007 2.223
0.007 0.007 2.259
0.007 0.007 2.488
0.007 0.007 2.562
0.007 0.007 2.599
0.007 0.007 2.697
0.007 0.007 3.030
0.007 0.007 3.075
0.007 0.007 3.145
0.007 0.007 3.225
Remove now unused lock_ctx_mm() and move unlock_ctx_vma() next to
unlock_ctx_mm() as they are logically related.
Remove a long comment about 4 cases that we handle when dropping the
mmap lock in the middle of VMA walk due to contention. The first 3
cases explained there are handled naturally and only case 4 needs to
be handled in a special way, which is done in smap_gather_stats() by
gathering stats from the portion of the VMA that has not yet been
processed.
For posterity, moving this comment here:
After dropping the lock, there are four cases to
consider. See the following example for explanation.
+------+------+-----------+
| VMA1 | VMA2 | VMA3 |
+------+------+-----------+
| | | |
4k 8k 16k 400k
Suppose we drop the lock after reading VMA2 due to
contention, then we get:
last_vma_end = 16k
1) VMA2 is freed, but VMA3 exists:
vma_next(vmi) will return VMA3.
In this case, just continue from VMA3.
2) VMA2 still exists:
vma_next(vmi) will return VMA3.
In this case, just continue from VMA3.
3) No more VMAs can be found:
vma_next(vmi) will return NULL.
No more things to do, just break.
4) (last_vma_end - 1) is the middle of a vma (VMA'):
vma_next(vmi) will return VMA' whose range
contains last_vma_end.
Iterate VMA' from last_vma_end.
[1] https://github.com/paulmckrcu/proc-mmap_sem-test
Signed-off-by: Suren Baghdasaryan <surenb@google.com>
Reviewed-by: Lorenzo Stoakes (ARM) <ljs@kernel.org>
---
fs/proc/task_mmu.c | 157 ++++++++++++++++++---------------------------
1 file changed, 63 insertions(+), 94 deletions(-)
diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c
index aef7ce659889..24425e230895 100644
--- a/fs/proc/task_mmu.c
+++ b/fs/proc/task_mmu.c
@@ -130,28 +130,12 @@ static void release_task_mempolicy(struct proc_maps_private *priv)
}
#endif
-static int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx)
-{
- int ret = mmap_read_lock_killable(lock_ctx->mm);
-
- if (!ret)
- lock_ctx->mmap_locked = true;
-
- return ret;
-}
-
static void unlock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx)
{
mmap_read_unlock(lock_ctx->mm);
lock_ctx->mmap_locked = false;
}
-static void reset_lock_ctx(struct proc_maps_locking_ctx *lock_ctx)
-{
- lock_ctx->locked_vma = NULL;
- lock_ctx->mmap_locked = false;
-}
-
static void unlock_ctx_vma(struct proc_maps_locking_ctx *lock_ctx)
{
if (lock_ctx->locked_vma) {
@@ -160,6 +144,12 @@ static void unlock_ctx_vma(struct proc_maps_locking_ctx *lock_ctx)
}
}
+static void reset_lock_ctx(struct proc_maps_locking_ctx *lock_ctx)
+{
+ lock_ctx->locked_vma = NULL;
+ lock_ctx->mmap_locked = false;
+}
+
static struct vm_area_struct *get_next_vma(struct proc_maps_private *priv,
loff_t last_pos)
{
@@ -1402,12 +1392,14 @@ static int show_smap(struct seq_file *m, void *v)
static int show_smaps_rollup(struct seq_file *m, void *v)
{
struct proc_maps_private *priv = m->private;
+ struct proc_maps_locking_ctx *lock_ctx = &priv->lock_ctx;
+ struct mm_struct *mm = lock_ctx->mm;
struct mem_size_stats mss = {};
- struct mm_struct *mm = priv->lock_ctx.mm;
+ unsigned long last_vma_end = 0;
+ unsigned long vma_start = 0;
struct vm_area_struct *vma;
- unsigned long vma_start = 0, last_vma_end = 0;
+ loff_t pos = 0;
int ret = 0;
- VMA_ITERATOR(vmi, mm, 0);
priv->task = get_proc_task(priv->inode);
if (!priv->task)
@@ -1418,90 +1410,63 @@ static int show_smaps_rollup(struct seq_file *m, void *v)
goto out_put_task;
}
- ret = lock_ctx_mm(&priv->lock_ctx);
- if (ret)
- goto out_put_mm;
-
hold_task_mempolicy(priv);
- vma = vma_next(&vmi);
+ rcu_read_lock();
+ reset_lock_ctx(lock_ctx);
+ vma_iter_init(&priv->iter, mm, 0);
+ vma = proc_get_vma(m, &pos);
if (unlikely(!vma))
goto empty_set;
- vma_start = vma->vm_start;
- do {
- smap_gather_stats(priv, vma, &mss);
+ if (!IS_ERR(vma))
+ vma_start = vma->vm_start;
+
+ while (vma) {
+ if (IS_ERR(vma)) {
+ ret = PTR_ERR(vma);
+ goto out_unlock;
+ }
+
+ if (vma->vm_start < last_vma_end) {
+ /*
+ * After retaking the lock, already reported VMA grew
+ * or got merged with the next one and we found it
+ * again. Gather stats for the remaining portion by
+ * starting at last_vma_end.
+ */
+ smap_gather_stats_range(priv, vma, &mss, last_vma_end);
+ } else {
+ /* Found next unreported VMA, start from its beginning */
+ smap_gather_stats(priv, vma, &mss);
+ }
last_vma_end = vma->vm_end;
/*
- * Release mmap_lock temporarily if someone wants to
- * access it for write request.
+ * If the VMA lock is not taken, we hold the often contended
+ * mmap lock. This can happen if we had to fall back to the
+ * mmap lock.
+ *
+ * To relieve pressure, check if it is indeed contended, then
+ * temporarily release it.
*/
- if (mmap_lock_is_contended(mm)) {
- vma_iter_invalidate(&vmi);
- unlock_ctx_mm(&priv->lock_ctx);
- ret = lock_ctx_mm(&priv->lock_ctx);
- if (ret) {
- release_task_mempolicy(priv);
- goto out_put_mm;
- }
-
+ if (lock_ctx->mmap_locked &&
+ mmap_lock_is_contended(lock_ctx->mm)) {
+ unlock_ctx_mm(lock_ctx);
/*
- * After dropping the lock, there are four cases to
- * consider. See the following example for explanation.
- *
- * +------+------+-----------+
- * | VMA1 | VMA2 | VMA3 |
- * +------+------+-----------+
- * | | | |
- * 4k 8k 16k 400k
- *
- * Suppose we drop the lock after reading VMA2 due to
- * contention, then we get:
- *
- * last_vma_end = 16k
- *
- * 1) VMA2 is freed, but VMA3 exists:
- *
- * vma_next(vmi) will return VMA3.
- * In this case, just continue from VMA3.
- *
- * 2) VMA2 still exists:
- *
- * vma_next(vmi) will return VMA3.
- * In this case, just continue from VMA3.
- *
- * 3) No more VMAs can be found:
- *
- * vma_next(vmi) will return NULL.
- * No more things to do, just break.
- *
- * 4) (last_vma_end - 1) is the middle of a vma (VMA'):
- *
- * vma_next(vmi) will return VMA' whose range
- * contains last_vma_end.
- * Iterate VMA' from last_vma_end.
+ * Even though we previously fell back to mmap lock,
+ * we try taking VMA lock for the next VMA, since it
+ * might not be under modification. In the worst case
+ * we will fall back to mmap lock again.
*/
- vma = vma_next(&vmi);
- /* Case 3 above */
- if (!vma)
- break;
-
- /* Case 1 and 2 above */
- if (vma->vm_start >= last_vma_end) {
- smap_gather_stats(priv, vma, &mss);
- last_vma_end = vma->vm_end;
- continue;
- }
-
- /* Case 4 above */
- if (vma->vm_end > last_vma_end) {
- smap_gather_stats_range(priv, vma, &mss,
- last_vma_end);
- last_vma_end = vma->vm_end;
- }
+ rcu_read_lock();
+ reset_lock_ctx(lock_ctx);
+ /* Resume from the last position. */
+ pos = last_vma_end;
+ vma_iter_init(&priv->iter, mm, pos);
}
- } for_each_vma(vmi, vma);
+ vma = proc_get_vma(m, &pos);
+ }
empty_set:
show_vma_header_prefix(m, vma_start, last_vma_end, 0, 0, 0, 0);
@@ -1510,10 +1475,14 @@ static int show_smaps_rollup(struct seq_file *m, void *v)
__show_smap(m, &mss, true);
+out_unlock:
+ if (lock_ctx->mmap_locked) {
+ unlock_ctx_mm(lock_ctx);
+ } else {
+ unlock_ctx_vma(lock_ctx);
+ rcu_read_unlock();
+ }
release_task_mempolicy(priv);
- unlock_ctx_mm(&priv->lock_ctx);
-
-out_put_mm:
mmput(mm);
out_put_task:
put_task_struct(priv->task);
--
2.55.0.1007.g17ff1f9808-goog
^ permalink raw reply [flat|nested] 17+ messages in thread* [PATCH v4 7/7] selftests/proc: add /proc/pid/smaps_rollup tearing tests
2026-09-11 19:41 [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock Suren Baghdasaryan
` (5 preceding siblings ...)
2026-09-11 19:41 ` [PATCH v4 6/7] proc/task_mmu: read proc/pid/smaps_rollup under per-vma lock Suren Baghdasaryan
@ 2026-09-11 19:41 ` Suren Baghdasaryan
2026-09-12 7:24 ` [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock Andrew Morton
7 siblings, 0 replies; 17+ messages in thread
From: Suren Baghdasaryan @ 2026-09-11 19:41 UTC (permalink / raw)
To: akpm
Cc: liam, ljs, vbabka, david, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel, surenb
During tearing tests, smaps_rollup Pss* metrics should stay constant.
Extend /proc/pid/smaps tearing tests to also check for smaps_rollup
consistency.
Signed-off-by: Suren Baghdasaryan <surenb@google.com>
Acked-by: Lorenzo Stoakes (ARM) <ljs@kernel.org>
---
tools/testing/selftests/proc/proc-maps-race.c | 186 +++++++++++++++++-
1 file changed, 181 insertions(+), 5 deletions(-)
diff --git a/tools/testing/selftests/proc/proc-maps-race.c b/tools/testing/selftests/proc/proc-maps-race.c
index 415eccb70468..bf4c5073f6fc 100644
--- a/tools/testing/selftests/proc/proc-maps-race.c
+++ b/tools/testing/selftests/proc/proc-maps-race.c
@@ -80,6 +80,61 @@ enum maps_file {
struct vma_modifier_info;
+enum smaps_rollup_stat {
+ Rss,
+ Pss,
+ Pss_Dirty,
+ Pss_Anon,
+ Pss_File,
+ Pss_Shmem,
+ Shared_Clean,
+ Shared_Dirty,
+ Private_Clean,
+ Private_Dirty,
+ Referenced,
+ Anonymous,
+ KSM,
+ LazyFree,
+ AnonHugePages,
+ ShmemPmdMapped,
+ FilePmdMapped,
+ Shared_Hugetlb,
+ Private_Hugetlb,
+ Swap,
+ SwapPss,
+ Locked,
+ RollupFieldCount
+};
+
+static const char *smaps_rollup_stat_names[RollupFieldCount] = {
+ "Rss",
+ "Pss",
+ "Pss_Dirty",
+ "Pss_Anon",
+ "Pss_File",
+ "Pss_Shmem",
+ "Shared_Clean",
+ "Shared_Dirty",
+ "Private_Clean",
+ "Private_Dirty",
+ "Referenced",
+ "Anonymous",
+ "KSM",
+ "LazyFree",
+ "AnonHugePages",
+ "ShmemPmdMapped",
+ "FilePmdMapped",
+ "Shared_Hugetlb",
+ "Private_Hugetlb",
+ "Swap",
+ "SwapPss",
+ "Locked",
+};
+
+struct smaps_rollup_stats {
+ unsigned long values[RollupFieldCount];
+};
+
FIXTURE(proc_maps_race)
{
struct vma_modifier_info *mod_info;
@@ -91,6 +146,7 @@ FIXTURE(proc_maps_race)
enum maps_file maps_file;
int shared_mem_size;
int skip_pages;
+ int rollup_fd;
int page_size;
int vma_count;
bool verbose;
@@ -132,12 +188,12 @@ struct vma_modifier_info {
void *child_mapped_addr[];
};
-static bool read_page(FIXTURE_DATA(proc_maps_race) *self,
+static bool read_page(FIXTURE_DATA(proc_maps_race) *self, int fd,
struct page_content *page)
{
ssize_t bytes_read;
- bytes_read = read(self->maps_fd, page->data, self->page_size);
+ bytes_read = read(fd, page->data, self->page_size);
if (bytes_read <= 0)
return false;
@@ -175,7 +231,7 @@ static int locate_containing_page(FIXTURE_DATA(proc_maps_race) *self,
char *curr_pos;
char *end_pos;
- if (!read_page(self, &self->page1))
+ if (!read_page(self, self->maps_fd, &self->page1))
return -1;
curr_pos = self->page1.data;
@@ -205,10 +261,11 @@ static bool read_two_pages(FIXTURE_DATA(proc_maps_race) *self)
return false;
for (int i = 0; i < self->skip_pages; i++)
- if (!read_page(self, &self->page1))
+ if (!read_page(self, self->maps_fd, &self->page1))
return false;
- return read_page(self, &self->page1) && read_page(self, &self->page2);
+ return read_page(self, self->maps_fd, &self->page1) &&
+ read_page(self, self->maps_fd, &self->page2);
}
static void copy_line(const char *line_start, const char *line_end,
@@ -317,6 +374,61 @@ static bool read_boundary_lines(FIXTURE_DATA(proc_maps_race) *self,
&first_line->end_addr) == 2;
}
+static bool parse_smaps_rollup(FIXTURE_DATA(proc_maps_race) *self,
+ struct smaps_rollup_stats *stats)
+{
+ unsigned int dev_maj, dev_min, inode;
+ unsigned long start, end, offs;
+ unsigned long value;
+ char name[32], perm[5];
+ char *curr_pos;
+ char *end_pos;
+ char *line_end;
+
+ if (lseek(self->rollup_fd, 0, SEEK_SET) < 0)
+ return false;
+
+ if (!read_page(self, self->rollup_fd, &self->page1))
+ return false;
+
+ curr_pos = self->page1.data;
+ end_pos = self->page1.data + self->page1.size;
+
+ line_end = strchr(curr_pos, '\n');
+ if (!line_end)
+ return false;
+
+ if (sscanf(curr_pos, "%lx-%lx %4s %lx %u:%u %u %31s",
+ &start, &end, perm, &offs, &dev_maj, &dev_min, &inode, name) != 8)
+ return false;
+
+ if (strcmp(name, "[rollup]"))
+ return false;
+
+ for (int stat = 0; stat < ARRAY_SIZE(smaps_rollup_stat_names); stat++) {
+ int len;
+
+ curr_pos = line_end + 1;
+ if (curr_pos >= end_pos)
+ return false;
+
+ line_end = strchr(curr_pos, '\n');
+ if (!line_end)
+ return false;
+
+ if (sscanf(curr_pos, "%31s %lu kB", name, &value) != 2)
+ return false;
+
+ len = strlen(name);
+ if (name[len - 1] != ':' || strncmp(name, smaps_rollup_stat_names[stat], len - 1))
+ return false;
+
+ stats->values[stat] = value;
+ }
+
+ return true;
+}
+
/* Thread synchronization routines */
static void wait_for_state(struct vma_modifier_info *mod_info, enum test_state state)
{
@@ -397,6 +509,40 @@ static bool print_boundaries_on(bool condition, const char *title,
return condition;
}
+static void print_smaps_rollup_stats(const char *title, FIXTURE_DATA(proc_maps_race) *self,
+ struct smaps_rollup_stats *stats)
+{
+ printf("%s", title);
+ for (int stat = 0; stat < ARRAY_SIZE(smaps_rollup_stat_names); stat++)
+ printf("%64s %lu kB\n", smaps_rollup_stat_names[stat], stats->values[stat]);
+}
+
+static bool cmp_smaps_rollup_stat(struct smaps_rollup_stats *s1,
+ struct smaps_rollup_stats *s2, enum smaps_rollup_stat stat)
+{
+ return s1->values[stat] == s2->values[stat];
+}
+
+static bool compare_smaps_rollup(FIXTURE_DATA(proc_maps_race) *self,
+ struct smaps_rollup_stats *expected,
+ struct smaps_rollup_stats *actual)
+{
+ /*
+ * Clean/dirty metrics might change but Pss-related ones
+ * should stay constant.
+ */
+ if (cmp_smaps_rollup_stat(expected, actual, Pss) &&
+ cmp_smaps_rollup_stat(expected, actual, Pss_Anon) &&
+ cmp_smaps_rollup_stat(expected, actual, Pss_File) &&
+ cmp_smaps_rollup_stat(expected, actual, Pss_Shmem))
+ return true;
+
+ print_smaps_rollup_stats("Expected stats:", self, expected);
+ print_smaps_rollup_stats("Actual stats:", self, actual);
+
+ return false;
+}
+
static void report_test_start(const char *name, bool verbose)
{
if (verbose)
@@ -572,6 +718,7 @@ FIXTURE_SETUP(proc_maps_race)
unsigned long first_map_addr;
unsigned long last_map_addr;
unsigned long duration_sec;
+ char rollup_fname[32];
char fname[32];
self->page_size = (unsigned long)sysconf(_SC_PAGESIZE);
@@ -649,6 +796,9 @@ FIXTURE_SETUP(proc_maps_race)
break;
case SMAPS:
sprintf(fname, "/proc/%d/smaps", self->pid);
+ sprintf(rollup_fname, "/proc/%d/smaps_rollup", self->pid);
+ self->rollup_fd = open(rollup_fname, O_RDONLY);
+ ASSERT_NE(self->rollup_fd, -1);
break;
default:
ksft_exit_fail();
@@ -711,6 +861,8 @@ FIXTURE_TEARDOWN(proc_maps_race)
for (int i = 0; i < self->vma_count; i++)
munmap(self->mod_info->child_mapped_addr[i], self->page_size);
close(self->maps_fd);
+ if (self->maps_file == SMAPS)
+ close(self->rollup_fd);
waitpid(self->pid, &status, 0);
munmap(self->mod_info, self->shared_mem_size);
}
@@ -723,6 +875,7 @@ TEST_F(proc_maps_race, test_maps_tearing_from_split)
struct line_content split_first_line;
struct line_content restored_last_line;
struct line_content restored_first_line;
+ struct smaps_rollup_stats orig_stats;
wait_for_state(mod_info, SETUP_READY);
@@ -736,6 +889,8 @@ TEST_F(proc_maps_race, test_maps_tearing_from_split)
report_test_start("Tearing from split", self->verbose);
ASSERT_TRUE(capture_mod_pattern(self, &split_last_line, &split_first_line,
&restored_last_line, &restored_first_line));
+ if (self->maps_file == SMAPS)
+ ASSERT_TRUE(parse_smaps_rollup(self, &orig_stats));
/* Now start concurrent modifications for self->duration_sec */
signal_state(mod_info, TEST_READY);
@@ -799,6 +954,11 @@ TEST_F(proc_maps_race, test_maps_tearing_from_split)
vma_end == self->last_line.end_addr) ||
(vma_start == split_first_line.start_addr &&
vma_end == split_first_line.end_addr));
+ } else {
+ struct smaps_rollup_stats stats;
+
+ ASSERT_TRUE(parse_smaps_rollup(self, &stats));
+ ASSERT_TRUE(compare_smaps_rollup(self, &orig_stats, &stats));
}
clock_gettime(CLOCK_MONOTONIC_COARSE, &end_ts);
end_test_iteration(&end_ts, self->verbose);
@@ -817,6 +977,7 @@ TEST_F(proc_maps_race, test_maps_tearing_from_resize)
struct line_content shrunk_first_line;
struct line_content restored_last_line;
struct line_content restored_first_line;
+ struct smaps_rollup_stats orig_stats;
wait_for_state(mod_info, SETUP_READY);
@@ -830,6 +991,8 @@ TEST_F(proc_maps_race, test_maps_tearing_from_resize)
report_test_start("Tearing from resize", self->verbose);
ASSERT_TRUE(capture_mod_pattern(self, &shrunk_last_line, &shrunk_first_line,
&restored_last_line, &restored_first_line));
+ if (self->maps_file == SMAPS)
+ ASSERT_TRUE(parse_smaps_rollup(self, &orig_stats));
/* Now start concurrent modifications for self->duration_sec */
signal_state(mod_info, TEST_READY);
@@ -880,6 +1043,11 @@ TEST_F(proc_maps_race, test_maps_tearing_from_resize)
ASSERT_TRUE(vma_start == self->last_line.start_addr &&
(vma_end - vma_start == self->page_size * 3 ||
vma_end - vma_start == self->page_size));
+ } else {
+ struct smaps_rollup_stats stats;
+
+ ASSERT_TRUE(parse_smaps_rollup(self, &stats));
+ ASSERT_TRUE(compare_smaps_rollup(self, &orig_stats, &stats));
}
clock_gettime(CLOCK_MONOTONIC_COARSE, &end_ts);
end_test_iteration(&end_ts, self->verbose);
@@ -898,6 +1066,7 @@ TEST_F(proc_maps_race, test_maps_tearing_from_remap)
struct line_content remapped_first_line;
struct line_content restored_last_line;
struct line_content restored_first_line;
+ struct smaps_rollup_stats orig_stats;
wait_for_state(mod_info, SETUP_READY);
@@ -911,6 +1080,8 @@ TEST_F(proc_maps_race, test_maps_tearing_from_remap)
report_test_start("Tearing from remap", self->verbose);
ASSERT_TRUE(capture_mod_pattern(self, &remapped_last_line, &remapped_first_line,
&restored_last_line, &restored_first_line));
+ if (self->maps_file == SMAPS)
+ ASSERT_TRUE(parse_smaps_rollup(self, &orig_stats));
/* Now start concurrent modifications for self->duration_sec */
signal_state(mod_info, TEST_READY);
@@ -963,6 +1134,11 @@ TEST_F(proc_maps_race, test_maps_tearing_from_remap)
vma_end - vma_start == self->page_size * 3) ||
(vma_start == self->last_line.start_addr + self->page_size &&
vma_end - vma_start == self->page_size));
+ } else {
+ struct smaps_rollup_stats stats;
+
+ ASSERT_TRUE(parse_smaps_rollup(self, &stats));
+ ASSERT_TRUE(compare_smaps_rollup(self, &orig_stats, &stats));
}
clock_gettime(CLOCK_MONOTONIC_COARSE, &end_ts);
end_test_iteration(&end_ts, self->verbose);
--
2.55.0.1007.g17ff1f9808-goog
^ permalink raw reply [flat|nested] 17+ messages in thread* Re: [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock
2026-09-11 19:41 [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock Suren Baghdasaryan
` (6 preceding siblings ...)
2026-09-11 19:41 ` [PATCH v4 7/7] selftests/proc: add /proc/pid/smaps_rollup tearing tests Suren Baghdasaryan
@ 2026-09-12 7:24 ` Andrew Morton
2026-09-12 18:55 ` Paul E. McKenney
2026-09-13 19:26 ` Suren Baghdasaryan
7 siblings, 2 replies; 17+ messages in thread
From: Andrew Morton @ 2026-09-12 7:24 UTC (permalink / raw)
To: Suren Baghdasaryan
Cc: liam, ljs, vbabka, david, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel
On Fri, 11 Sep 2026 12:41:38 -0700 Suren Baghdasaryan <surenb@google.com> wrote:
> proc/pid/smaps_rollup can be read using the combination of RCU and
> VMA read locks, similar to proc/pid/{maps|smaps|numa_maps}. RCU is
> required to safely traverse the VMA tree and VMA lock stabilizes the
> VMA being processed and the pagetable walk.
> Note that we have to keep the logic to drop mmap_lock on contention
> because even when using per-VMA locks we might have to fall back to
> holding the mmap_lock.
Nice. Queued, thanks.
The speedups described in [6/7] are significant, although I don't know
how representative Paul's tests are. Probably not very.
So I don't know how much improvement our users will be seeing from
these changes?
And I don't know how much usage smaps_rollup gets in the real world?
As I've no doubt you know, Sashiko is saying things. About these
patches and about the current code. Its pre-existing
clear_soft_dirty_pmd() driveby find looks significant.
https://sashiko.dev/#/patchset/20260911194145.1781926-1-surenb@google.com
^ permalink raw reply [flat|nested] 17+ messages in thread* Re: [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock
2026-09-12 7:24 ` [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock Andrew Morton
@ 2026-09-12 18:55 ` Paul E. McKenney
2026-09-13 19:26 ` Suren Baghdasaryan
1 sibling, 0 replies; 17+ messages in thread
From: Paul E. McKenney @ 2026-09-12 18:55 UTC (permalink / raw)
To: Andrew Morton
Cc: Suren Baghdasaryan, liam, ljs, vbabka, david, willy, jannh,
pfalcato, xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel
On Sat, Sep 12, 2026 at 12:24:36AM -0700, Andrew Morton wrote:
> On Fri, 11 Sep 2026 12:41:38 -0700 Suren Baghdasaryan <surenb@google.com> wrote:
>
> > proc/pid/smaps_rollup can be read using the combination of RCU and
> > VMA read locks, similar to proc/pid/{maps|smaps|numa_maps}. RCU is
> > required to safely traverse the VMA tree and VMA lock stabilizes the
> > VMA being processed and the pagetable walk.
> > Note that we have to keep the logic to drop mmap_lock on contention
> > because even when using per-VMA locks we might have to fall back to
> > holding the mmap_lock.
>
> Nice. Queued, thanks.
>
> The speedups described in [6/7] are significant, although I don't know
> how representative Paul's tests are. Probably not very.
>
> So I don't know how much improvement our users will be seeing from
> these changes?
>
> And I don't know how much usage smaps_rollup gets in the real world?
It is used heavily in some applications for monitoring. This should allow
us to more tightly fence our monitoring software, leaving more CPU for
the application.
So thank you all!!!
Thanx, Paul
> As I've no doubt you know, Sashiko is saying things. About these
> patches and about the current code. Its pre-existing
> clear_soft_dirty_pmd() driveby find looks significant.
>
> https://sashiko.dev/#/patchset/20260911194145.1781926-1-surenb@google.com
^ permalink raw reply [flat|nested] 17+ messages in thread* Re: [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock
2026-09-12 7:24 ` [PATCH v4 0/7] read proc/pid/smaps_rollup under per-vma lock Andrew Morton
2026-09-12 18:55 ` Paul E. McKenney
@ 2026-09-13 19:26 ` Suren Baghdasaryan
1 sibling, 0 replies; 17+ messages in thread
From: Suren Baghdasaryan @ 2026-09-13 19:26 UTC (permalink / raw)
To: Andrew Morton
Cc: liam, ljs, vbabka, david, willy, jannh, paulmck, pfalcato,
xueyuan.chen21, linux-mm, linux-kernel, linux-fsdevel
On Sat, Sep 12, 2026 at 12:24 AM Andrew Morton
<akpm@linux-foundation.org> wrote:
>
> On Fri, 11 Sep 2026 12:41:38 -0700 Suren Baghdasaryan <surenb@google.com> wrote:
>
> > proc/pid/smaps_rollup can be read using the combination of RCU and
> > VMA read locks, similar to proc/pid/{maps|smaps|numa_maps}. RCU is
> > required to safely traverse the VMA tree and VMA lock stabilizes the
> > VMA being processed and the pagetable walk.
> > Note that we have to keep the logic to drop mmap_lock on contention
> > because even when using per-VMA locks we might have to fall back to
> > holding the mmap_lock.
>
> Nice. Queued, thanks.
>
> The speedups described in [6/7] are significant, although I don't know
> how representative Paul's tests are. Probably not very.
>
> So I don't know how much improvement our users will be seeing from
> these changes?
>
> And I don't know how much usage smaps_rollup gets in the real world?
>
>
> As I've no doubt you know, Sashiko is saying things. About these
> patches and about the current code. Its pre-existing
> clear_soft_dirty_pmd() driveby find looks significant.
>
> https://sashiko.dev/#/patchset/20260911194145.1781926-1-surenb@google.com
Thanks! I'll investigate the preexisting ones and will try to fix the
ones that look real.
^ permalink raw reply [flat|nested] 17+ messages in thread