* [PATCH v2 1/8] mm/page_owner: Add PID filtering support
2026-09-03 4:18 [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering Zhen Ni
@ 2026-09-03 4:18 ` Zhen Ni
2026-09-03 4:18 ` [PATCH v2 2/8] mm/page_owner: Add TGID " Zhen Ni
` (8 subsequent siblings)
9 siblings, 0 replies; 20+ messages in thread
From: Zhen Ni @ 2026-09-03 4:18 UTC (permalink / raw)
To: Andrew Morton
Cc: David Hildenbrand, Lorenzo Stoakes, Liam R . Howlett,
Vlastimil Babka, Mike Rapoport, Suren Baghdasaryan, Michal Hocko,
Jonathan Corbet, Shuah Khan, Randy Dunlap, Brendan Jackman,
Johannes Weiner, Zi Yan, linux-mm, linux-doc, linux-kernel,
Zhen Ni
Add PID filtering support. Users can filter page_owner output by process
IDs using the "pid=<pid_list>" command format. The filter supports up to
16 PIDs specified as a comma-separated list. PIDs are stored in sorted
order for efficient binary search matching during page owner iteration.
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
Changes in v2:
- Drop the kstrdup() copy in parse_pid_t_list(); parse the token in
place. This also fixes a leak on success and a kfree() of an advanced
pointer on error in v1.
- Use cmp_int() in cmp_pid_t() to avoid overflow on subtraction.
- Reject pids exceeding PID_MAX_LIMIT.
v1: https://lore.kernel.org/linux-mm/20260828031339.1270699-2-zhen.ni@easystack.cn/
---
mm/page_owner.c | 75 +++++++++++++++++++++++++++++++++++++++++++++++--
1 file changed, 73 insertions(+), 2 deletions(-)
diff --git a/mm/page_owner.c b/mm/page_owner.c
index fbbda7ba914b..0ef081e443f9 100644
--- a/mm/page_owner.c
+++ b/mm/page_owner.c
@@ -12,6 +12,8 @@
#include <linux/seq_file.h>
#include <linux/memcontrol.h>
#include <linux/sched/clock.h>
+#include <linux/bsearch.h>
+#include <linux/sort.h>
#include "page_alloc.h"
@@ -66,12 +68,24 @@ static const char * const page_owner_print_mode_strings[] = {
[PAGE_OWNER_PRINT_STACK_HANDLE] = "stack_handle",
};
+/* PID_MAX_LIMIT = 4,194,304 (7 decimal digits) */
+#define PID_MAX_DIGITS 7
+#define MAX_FILTER_PIDS 16
+
struct page_owner_filter_state {
enum page_owner_print_mode print_mode;
- nodemask_t nid_filter;
bool nid_filter_enabled;
+ bool proc_filter_enabled;
+ nodemask_t nid_filter;
+ int pid_count;
+ pid_t pid_list[MAX_FILTER_PIDS];
};
+static int cmp_pid_t(const void *a, const void *b)
+{
+ return cmp_int(*(pid_t *)a, *(pid_t *)b);
+}
+
static bool page_owner_enabled __initdata;
DEFINE_STATIC_KEY_FALSE(page_owner_inited);
@@ -820,6 +834,19 @@ read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
goto ext_put_continue;
}
+ if (state->proc_filter_enabled) {
+ bool proc_match = false;
+
+ proc_match = bsearch(&page_owner->pid,
+ state->pid_list,
+ state->pid_count,
+ sizeof(pid_t),
+ cmp_pid_t) != NULL;
+
+ if (!proc_match)
+ goto ext_put_continue;
+ }
+
/* Record the next PFN to read in the file offset */
*ppos = pfn + 1;
@@ -927,6 +954,7 @@ static int page_owner_open(struct inode *inode, struct file *file)
state->print_mode = PAGE_OWNER_PRINT_STACK;
nodes_clear(state->nid_filter);
state->nid_filter_enabled = false;
+ state->proc_filter_enabled = false;
file->private_data = state;
return 0;
}
@@ -937,6 +965,27 @@ static int page_owner_release(struct inode *inode, struct file *file)
return 0;
}
+static int parse_pid_t_list(char *str, pid_t *list, int *count)
+{
+ char *token;
+ int i = 0;
+
+ while ((token = strsep(&str, ",")) != NULL) {
+ unsigned int pid;
+
+ if (*token == '\0')
+ continue;
+ if (i >= MAX_FILTER_PIDS)
+ return -E2BIG;
+ if (kstrtouint(token, 10, &pid) != 0 || pid > PID_MAX_LIMIT)
+ return -EINVAL;
+ list[i++] = (pid_t)pid;
+ }
+
+ *count = i;
+ return 0;
+}
+
static ssize_t page_owner_write(struct file *file,
const char __user *buf,
size_t count, loff_t *ppos)
@@ -949,6 +998,9 @@ static ssize_t page_owner_write(struct file *file,
enum page_owner_print_mode new_print_mode;
nodemask_t new_nid_filter;
bool new_nid_filter_enabled;
+ bool new_proc_filter_enabled;
+ pid_t new_pid_list[MAX_FILTER_PIDS];
+ int new_pid_count = 0;
/*
* Maximum input length for filter commands:
@@ -956,8 +1008,10 @@ static ssize_t page_owner_write(struct file *file,
* with sufficient buffer
* - 6 * MAX_NUMNODES: worst case for nid list
* Worst case per node: ",NNNNN" (comma + 5-digit node number) = 6 bytes
+ * - For list filters: (digit+comma) * count + prefix
*/
- if (count > 32 + 6 * MAX_NUMNODES)
+ if (count > 32 + 6 * MAX_NUMNODES +
+ (PID_MAX_DIGITS + 1) * MAX_FILTER_PIDS + 4)
return -EINVAL;
kbuf = memdup_user_nul(buf, count);
@@ -969,6 +1023,11 @@ static ssize_t page_owner_write(struct file *file,
new_print_mode = state->print_mode;
new_nid_filter = state->nid_filter;
new_nid_filter_enabled = state->nid_filter_enabled;
+ new_proc_filter_enabled = state->proc_filter_enabled;
+ if (state->pid_count > 0) {
+ memcpy(new_pid_list, state->pid_list, sizeof(state->pid_list));
+ new_pid_count = state->pid_count;
+ }
while ((token = strsep(&kbuf, " \t\n")) != NULL) {
if (*token == '\0')
@@ -1000,6 +1059,10 @@ static ssize_t page_owner_write(struct file *file,
}
new_nid_filter_enabled = true;
+ } else if (!strncmp(token, "pid=", 4)) {
+ ret = parse_pid_t_list(token + 4, new_pid_list, &new_pid_count);
+ if (ret < 0)
+ goto out_free;
} else {
ret = -EINVAL;
goto out_free;
@@ -1010,6 +1073,14 @@ static ssize_t page_owner_write(struct file *file,
state->print_mode = new_print_mode;
state->nid_filter = new_nid_filter;
state->nid_filter_enabled = new_nid_filter_enabled;
+ state->proc_filter_enabled = new_pid_count > 0;
+ if (new_pid_count > 0) {
+ memcpy(state->pid_list, new_pid_list, sizeof(state->pid_list));
+ state->pid_count = new_pid_count;
+ }
+ if (state->pid_count > 1)
+ sort(state->pid_list, state->pid_count, sizeof(pid_t),
+ cmp_pid_t, NULL);
ret = count;
--
2.20.1
^ permalink raw reply [flat|nested] 20+ messages in thread* [PATCH v2 2/8] mm/page_owner: Add TGID filtering support
2026-09-03 4:18 [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering Zhen Ni
2026-09-03 4:18 ` [PATCH v2 1/8] mm/page_owner: Add PID filtering support Zhen Ni
@ 2026-09-03 4:18 ` Zhen Ni
2026-09-03 4:18 ` [PATCH v2 3/8] mm/page_owner: Add COMM filtering with wildcard support Zhen Ni
` (7 subsequent siblings)
9 siblings, 0 replies; 20+ messages in thread
From: Zhen Ni @ 2026-09-03 4:18 UTC (permalink / raw)
To: Andrew Morton
Cc: David Hildenbrand, Lorenzo Stoakes, Liam R . Howlett,
Vlastimil Babka, Mike Rapoport, Suren Baghdasaryan, Michal Hocko,
Jonathan Corbet, Shuah Khan, Randy Dunlap, Brendan Jackman,
Johannes Weiner, Zi Yan, linux-mm, linux-doc, linux-kernel,
Zhen Ni
Extend filter to support thread group ID (TGID) filtering alongside
PID filtering. Reuses existing PID parsing and binary search
infrastructure with separate TGID list and count fields.
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
Changes in v2:
- No change.
v1: https://lore.kernel.org/linux-mm/20260828031339.1270699-3-zhen.ni@easystack.cn/
---
mm/page_owner.c | 43 ++++++++++++++++++++++++++++++++++++-------
1 file changed, 36 insertions(+), 7 deletions(-)
diff --git a/mm/page_owner.c b/mm/page_owner.c
index 0ef081e443f9..e046e61eb98a 100644
--- a/mm/page_owner.c
+++ b/mm/page_owner.c
@@ -71,6 +71,7 @@ static const char * const page_owner_print_mode_strings[] = {
/* PID_MAX_LIMIT = 4,194,304 (7 decimal digits) */
#define PID_MAX_DIGITS 7
#define MAX_FILTER_PIDS 16
+#define MAX_FILTER_TGIDS 16
struct page_owner_filter_state {
enum page_owner_print_mode print_mode;
@@ -78,7 +79,9 @@ struct page_owner_filter_state {
bool proc_filter_enabled;
nodemask_t nid_filter;
int pid_count;
+ int tgid_count;
pid_t pid_list[MAX_FILTER_PIDS];
+ pid_t tgid_list[MAX_FILTER_TGIDS];
};
static int cmp_pid_t(const void *a, const void *b)
@@ -837,11 +840,19 @@ read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
if (state->proc_filter_enabled) {
bool proc_match = false;
- proc_match = bsearch(&page_owner->pid,
- state->pid_list,
- state->pid_count,
- sizeof(pid_t),
- cmp_pid_t) != NULL;
+ if (state->pid_count > 0)
+ proc_match = bsearch(&page_owner->pid,
+ state->pid_list,
+ state->pid_count,
+ sizeof(pid_t),
+ cmp_pid_t) != NULL;
+
+ if (!proc_match && state->tgid_count > 0)
+ proc_match = bsearch(&page_owner->tgid,
+ state->tgid_list,
+ state->tgid_count,
+ sizeof(pid_t),
+ cmp_pid_t) != NULL;
if (!proc_match)
goto ext_put_continue;
@@ -1000,7 +1011,9 @@ static ssize_t page_owner_write(struct file *file,
bool new_nid_filter_enabled;
bool new_proc_filter_enabled;
pid_t new_pid_list[MAX_FILTER_PIDS];
+ pid_t new_tgid_list[MAX_FILTER_TGIDS];
int new_pid_count = 0;
+ int new_tgid_count = 0;
/*
* Maximum input length for filter commands:
@@ -1011,7 +1024,8 @@ static ssize_t page_owner_write(struct file *file,
* - For list filters: (digit+comma) * count + prefix
*/
if (count > 32 + 6 * MAX_NUMNODES +
- (PID_MAX_DIGITS + 1) * MAX_FILTER_PIDS + 4)
+ (PID_MAX_DIGITS + 1) * MAX_FILTER_PIDS + 4 +
+ (PID_MAX_DIGITS + 1) * MAX_FILTER_TGIDS + 5)
return -EINVAL;
kbuf = memdup_user_nul(buf, count);
@@ -1028,6 +1042,10 @@ static ssize_t page_owner_write(struct file *file,
memcpy(new_pid_list, state->pid_list, sizeof(state->pid_list));
new_pid_count = state->pid_count;
}
+ if (state->tgid_count > 0) {
+ memcpy(new_tgid_list, state->tgid_list, sizeof(state->tgid_list));
+ new_tgid_count = state->tgid_count;
+ }
while ((token = strsep(&kbuf, " \t\n")) != NULL) {
if (*token == '\0')
@@ -1063,6 +1081,10 @@ static ssize_t page_owner_write(struct file *file,
ret = parse_pid_t_list(token + 4, new_pid_list, &new_pid_count);
if (ret < 0)
goto out_free;
+ } else if (!strncmp(token, "tgid=", 5)) {
+ ret = parse_pid_t_list(token + 5, new_tgid_list, &new_tgid_count);
+ if (ret < 0)
+ goto out_free;
} else {
ret = -EINVAL;
goto out_free;
@@ -1073,7 +1095,7 @@ static ssize_t page_owner_write(struct file *file,
state->print_mode = new_print_mode;
state->nid_filter = new_nid_filter;
state->nid_filter_enabled = new_nid_filter_enabled;
- state->proc_filter_enabled = new_pid_count > 0;
+ state->proc_filter_enabled = new_pid_count > 0 || new_tgid_count > 0;
if (new_pid_count > 0) {
memcpy(state->pid_list, new_pid_list, sizeof(state->pid_list));
state->pid_count = new_pid_count;
@@ -1081,6 +1103,13 @@ static ssize_t page_owner_write(struct file *file,
if (state->pid_count > 1)
sort(state->pid_list, state->pid_count, sizeof(pid_t),
cmp_pid_t, NULL);
+ if (new_tgid_count > 0) {
+ memcpy(state->tgid_list, new_tgid_list, sizeof(state->tgid_list));
+ state->tgid_count = new_tgid_count;
+ }
+ if (state->tgid_count > 1)
+ sort(state->tgid_list, state->tgid_count, sizeof(pid_t),
+ cmp_pid_t, NULL);
ret = count;
--
2.20.1
^ permalink raw reply [flat|nested] 20+ messages in thread* [PATCH v2 3/8] mm/page_owner: Add COMM filtering with wildcard support
2026-09-03 4:18 [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering Zhen Ni
2026-09-03 4:18 ` [PATCH v2 1/8] mm/page_owner: Add PID filtering support Zhen Ni
2026-09-03 4:18 ` [PATCH v2 2/8] mm/page_owner: Add TGID " Zhen Ni
@ 2026-09-03 4:18 ` Zhen Ni
2026-09-03 4:18 ` [PATCH v2 4/8] mm/page_owner: Refactor memcg handling for cgroup filter support Zhen Ni
` (6 subsequent siblings)
9 siblings, 0 replies; 20+ messages in thread
From: Zhen Ni @ 2026-09-03 4:18 UTC (permalink / raw)
To: Andrew Morton
Cc: David Hildenbrand, Lorenzo Stoakes, Liam R . Howlett,
Vlastimil Babka, Mike Rapoport, Suren Baghdasaryan, Michal Hocko,
Jonathan Corbet, Shuah Khan, Randy Dunlap, Brendan Jackman,
Johannes Weiner, Zi Yan, linux-mm, linux-doc, linux-kernel,
Zhen Ni
Add process name (COMM) filtering to page_owner with glob-style wildcard
pattern matching support. Users can now filter page_owner output by
process names using flexible patterns.
Supported wildcards:
* : matches any sequence of characters
? : matches any single character
[abc]: matches any character in the set
[a-z]: matches any character in the range
Examples:
comm="python*" : matches python, python3, python3.9, etc.
comm="*sh" : matches bash, zsh, dash, etc.
Also select GLOB from PAGE_OWNER. glob_match() is provided by
lib/glob.c, which is only built when CONFIG_GLOB is set, and
CONFIG_GLOB is a hidden option without a prompt. A config that enables
PAGE_OWNER but has no other GLOB selector would fail to link with an
undefined reference to glob_match().
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
Changes in v2:
- Select GLOB from PAGE_OWNER so that glob_match() is always linked in
when the feature is enabled.
- Drop the kstrdup() copy in parse_comm_list(); parse the token in
place, matching patch 1.
v1: https://lore.kernel.org/linux-mm/20260828031339.1270699-4-zhen.ni@easystack.cn/
---
mm/Kconfig.debug | 1 +
mm/page_owner.c | 71 ++++++++++++++++++++++++++++++++++++++++++++++--
2 files changed, 69 insertions(+), 3 deletions(-)
diff --git a/mm/Kconfig.debug b/mm/Kconfig.debug
index 5737a504efbb..c94c30a60cd5 100644
--- a/mm/Kconfig.debug
+++ b/mm/Kconfig.debug
@@ -106,6 +106,7 @@ config PAGE_OWNER
bool "Track page owner"
depends on DEBUG_KERNEL && STACKTRACE_SUPPORT
select DEBUG_FS
+ select GLOB
select STACKTRACE
select STACKDEPOT
select PAGE_EXTENSION
diff --git a/mm/page_owner.c b/mm/page_owner.c
index e046e61eb98a..b8319bc3a368 100644
--- a/mm/page_owner.c
+++ b/mm/page_owner.c
@@ -14,6 +14,7 @@
#include <linux/sched/clock.h>
#include <linux/bsearch.h>
#include <linux/sort.h>
+#include <linux/glob.h>
#include "page_alloc.h"
@@ -72,6 +73,7 @@ static const char * const page_owner_print_mode_strings[] = {
#define PID_MAX_DIGITS 7
#define MAX_FILTER_PIDS 16
#define MAX_FILTER_TGIDS 16
+#define MAX_FILTER_COMMS 8
struct page_owner_filter_state {
enum page_owner_print_mode print_mode;
@@ -80,8 +82,10 @@ struct page_owner_filter_state {
nodemask_t nid_filter;
int pid_count;
int tgid_count;
+ int comm_count;
pid_t pid_list[MAX_FILTER_PIDS];
pid_t tgid_list[MAX_FILTER_TGIDS];
+ char comm_list[MAX_FILTER_COMMS][TASK_COMM_LEN];
};
static int cmp_pid_t(const void *a, const void *b)
@@ -854,6 +858,20 @@ read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
sizeof(pid_t),
cmp_pid_t) != NULL;
+ if (!proc_match && state->comm_count > 0) {
+ bool comm_match = false;
+ int i;
+
+ for (i = 0; i < state->comm_count; i++) {
+ if (glob_match(state->comm_list[i],
+ page_owner->comm)) {
+ comm_match = true;
+ break;
+ }
+ }
+ proc_match = comm_match;
+ }
+
if (!proc_match)
goto ext_put_continue;
}
@@ -997,6 +1015,27 @@ static int parse_pid_t_list(char *str, pid_t *list, int *count)
return 0;
}
+static int parse_comm_list(char *str, char (*list)[TASK_COMM_LEN], int *count)
+{
+ char *token;
+ int i = 0;
+
+ while ((token = strsep(&str, ",")) != NULL) {
+ token = strstrip(token);
+ if (*token == '\0')
+ continue;
+ if (i >= MAX_FILTER_COMMS)
+ return -E2BIG;
+ strscpy(list[i++], token, TASK_COMM_LEN);
+ }
+
+ if (i == 0)
+ return -EINVAL;
+
+ *count = i;
+ return 0;
+}
+
static ssize_t page_owner_write(struct file *file,
const char __user *buf,
size_t count, loff_t *ppos)
@@ -1012,8 +1051,10 @@ static ssize_t page_owner_write(struct file *file,
bool new_proc_filter_enabled;
pid_t new_pid_list[MAX_FILTER_PIDS];
pid_t new_tgid_list[MAX_FILTER_TGIDS];
+ char (*new_comm_list)[TASK_COMM_LEN] = NULL;
int new_pid_count = 0;
int new_tgid_count = 0;
+ int new_comm_count = 0;
/*
* Maximum input length for filter commands:
@@ -1025,12 +1066,19 @@ static ssize_t page_owner_write(struct file *file,
*/
if (count > 32 + 6 * MAX_NUMNODES +
(PID_MAX_DIGITS + 1) * MAX_FILTER_PIDS + 4 +
- (PID_MAX_DIGITS + 1) * MAX_FILTER_TGIDS + 5)
+ (PID_MAX_DIGITS + 1) * MAX_FILTER_TGIDS + 5 +
+ TASK_COMM_LEN * MAX_FILTER_COMMS + 5)
return -EINVAL;
+ new_comm_list = kmalloc_array(MAX_FILTER_COMMS, TASK_COMM_LEN, GFP_KERNEL);
+ if (!new_comm_list)
+ return -ENOMEM;
+
kbuf = memdup_user_nul(buf, count);
- if (IS_ERR(kbuf))
+ if (IS_ERR(kbuf)) {
+ kfree(new_comm_list);
return PTR_ERR(kbuf);
+ }
orig = kbuf;
@@ -1046,6 +1094,11 @@ static ssize_t page_owner_write(struct file *file,
memcpy(new_tgid_list, state->tgid_list, sizeof(state->tgid_list));
new_tgid_count = state->tgid_count;
}
+ if (state->comm_count > 0) {
+ memcpy(new_comm_list, state->comm_list,
+ state->comm_count * TASK_COMM_LEN);
+ new_comm_count = state->comm_count;
+ }
while ((token = strsep(&kbuf, " \t\n")) != NULL) {
if (*token == '\0')
@@ -1085,6 +1138,10 @@ static ssize_t page_owner_write(struct file *file,
ret = parse_pid_t_list(token + 5, new_tgid_list, &new_tgid_count);
if (ret < 0)
goto out_free;
+ } else if (!strncmp(token, "comm=", 5)) {
+ ret = parse_comm_list(token + 5, new_comm_list, &new_comm_count);
+ if (ret < 0)
+ goto out_free;
} else {
ret = -EINVAL;
goto out_free;
@@ -1095,7 +1152,9 @@ static ssize_t page_owner_write(struct file *file,
state->print_mode = new_print_mode;
state->nid_filter = new_nid_filter;
state->nid_filter_enabled = new_nid_filter_enabled;
- state->proc_filter_enabled = new_pid_count > 0 || new_tgid_count > 0;
+ state->proc_filter_enabled = new_pid_count > 0 ||
+ new_tgid_count > 0 ||
+ new_comm_count > 0;
if (new_pid_count > 0) {
memcpy(state->pid_list, new_pid_list, sizeof(state->pid_list));
state->pid_count = new_pid_count;
@@ -1110,10 +1169,16 @@ static ssize_t page_owner_write(struct file *file,
if (state->tgid_count > 1)
sort(state->tgid_list, state->tgid_count, sizeof(pid_t),
cmp_pid_t, NULL);
+ if (new_comm_count > 0) {
+ memcpy(state->comm_list, new_comm_list,
+ new_comm_count * TASK_COMM_LEN);
+ state->comm_count = new_comm_count;
+ }
ret = count;
out_free:
+ kfree(new_comm_list);
kfree(orig);
return ret;
}
--
2.20.1
^ permalink raw reply [flat|nested] 20+ messages in thread* [PATCH v2 4/8] mm/page_owner: Refactor memcg handling for cgroup filter support
2026-09-03 4:18 [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering Zhen Ni
` (2 preceding siblings ...)
2026-09-03 4:18 ` [PATCH v2 3/8] mm/page_owner: Add COMM filtering with wildcard support Zhen Ni
@ 2026-09-03 4:18 ` Zhen Ni
2026-09-03 4:18 ` [PATCH v2 5/8] mm/page_owner: Add memcg " Zhen Ni
` (5 subsequent siblings)
9 siblings, 0 replies; 20+ messages in thread
From: Zhen Ni @ 2026-09-03 4:18 UTC (permalink / raw)
To: Andrew Morton
Cc: David Hildenbrand, Lorenzo Stoakes, Liam R . Howlett,
Vlastimil Babka, Mike Rapoport, Suren Baghdasaryan, Michal Hocko,
Jonathan Corbet, Shuah Khan, Randy Dunlap, Brendan Jackman,
Johannes Weiner, Zi Yan, linux-mm, linux-doc, linux-kernel,
Zhen Ni
Extract memcg information retrieval from printing logic to prepare for
cgroup filtering support. Introduce struct memcg_info to hold cgroup
data that can be reused for both display output and filtering
decisions.
No functional change - output behavior unchanged.
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
Changes in v2:
- No change.
v1: https://lore.kernel.org/linux-mm/20260828031339.1270699-5-zhen.ni@easystack.cn/
---
mm/page_owner.c | 62 +++++++++++++++++++++++++++++++++++--------------
1 file changed, 44 insertions(+), 18 deletions(-)
diff --git a/mm/page_owner.c b/mm/page_owner.c
index b8319bc3a368..ca9dd8ed9f77 100644
--- a/mm/page_owner.c
+++ b/mm/page_owner.c
@@ -69,6 +69,13 @@ static const char * const page_owner_print_mode_strings[] = {
[PAGE_OWNER_PRINT_STACK_HANDLE] = "stack_handle",
};
+struct memcg_info {
+ char name[80];
+ bool is_slab;
+ bool is_objcg;
+ bool is_online;
+};
+
/* PID_MAX_LIMIT = 4,194,304 (7 decimal digits) */
#define PID_MAX_DIGITS 7
#define MAX_FILTER_PIDS 16
@@ -573,16 +580,13 @@ void pagetypeinfo_showmixedcount_print(struct seq_file *m,
#ifdef CONFIG_MEMCG
/*
- * Looking for memcg information and print it out
+ * Get memcg information from page
*/
-static inline int print_page_owner_memcg(char *kbuf, size_t count, int ret,
- struct page *page)
+static void get_page_memcg_info(struct page *page, struct memcg_info *info)
{
unsigned long memcg_data;
struct obj_cgroup *objcg;
struct mem_cgroup *memcg;
- bool online;
- char name[80];
rcu_read_lock();
memcg_data = READ_ONCE(page->memcg_data);
@@ -590,8 +594,7 @@ static inline int print_page_owner_memcg(char *kbuf, size_t count, int ret,
goto out_unlock;
if (memcg_data & MEMCG_DATA_OBJEXTS) {
- ret += scnprintf(kbuf + ret, count - ret,
- "Slab cache page\n");
+ info->is_slab = true;
goto out_unlock;
}
@@ -600,21 +603,38 @@ static inline int print_page_owner_memcg(char *kbuf, size_t count, int ret,
if (!memcg)
goto out_unlock;
- online = css_is_online(&memcg->css);
- cgroup_name(memcg->css.cgroup, name, sizeof(name));
- ret += scnprintf(kbuf + ret, count - ret,
- "Charged %sto %smemcg %s\n",
- (memcg_data & MEMCG_DATA_KMEM) ? "(via objcg) " : "",
- online ? "" : "offline ",
- name);
+ info->is_objcg = (memcg_data & MEMCG_DATA_KMEM) != 0;
+ info->is_online = css_is_online(&memcg->css);
+ cgroup_name(memcg->css.cgroup, info->name, sizeof(info->name));
out_unlock:
rcu_read_unlock();
+}
+
+/*
+ * Print memcg information from memcg_info
+ */
+static inline int print_page_owner_memcg(char *kbuf, size_t count, int ret,
+ const struct memcg_info *info)
+{
+ if (!info)
+ return ret;
+
+ if (info->is_slab)
+ ret += scnprintf(kbuf + ret, count - ret,
+ "Slab cache page\n");
+
+ if (info->name[0])
+ ret += scnprintf(kbuf + ret, count - ret,
+ "Charged %sto %smemcg %s\n",
+ info->is_objcg ? "(via objcg) " : "",
+ info->is_online ? "" : "offline ",
+ info->name);
return ret;
}
#else
static inline int print_page_owner_memcg(char *kbuf, size_t count, int ret,
- struct page *page)
+ const struct memcg_info *info)
{
return ret;
}
@@ -624,7 +644,8 @@ static ssize_t
print_page_owner(char __user *buf, size_t count, unsigned long pfn,
struct page *page, struct page_owner *page_owner,
depot_stack_handle_t handle,
- struct page_owner_filter_state *state)
+ struct page_owner_filter_state *state,
+ const struct memcg_info *memcg_info)
{
int ret, pageblock_mt, page_mt;
char *kbuf;
@@ -674,7 +695,7 @@ print_page_owner(char __user *buf, size_t count, unsigned long pfn,
migrate_reason_names[page_owner->last_migrate_reason]);
}
- ret = print_page_owner_memcg(kbuf, count, ret, page);
+ ret = print_page_owner_memcg(kbuf, count, ret, memcg_info);
ret += snprintf(kbuf + ret, count - ret, "\n");
if (ret >= count)
@@ -777,6 +798,7 @@ read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
* user through copy_to_user() or GFP_KERNEL allocations.
*/
struct page_owner page_owner_tmp;
+ struct memcg_info memcg_info = {};
/*
* If the new page is in a new MAX_ORDER_NR_PAGES area,
@@ -876,13 +898,17 @@ read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
goto ext_put_continue;
}
+#ifdef CONFIG_MEMCG
+ get_page_memcg_info(page, &memcg_info);
+#endif
+
/* Record the next PFN to read in the file offset */
*ppos = pfn + 1;
page_owner_tmp = *page_owner;
page_ext_put(page_ext);
return print_page_owner(buf, count, pfn, page,
- &page_owner_tmp, handle, state);
+ &page_owner_tmp, handle, state, &memcg_info);
ext_put_continue:
page_ext_put(page_ext);
cond_resched();
--
2.20.1
^ permalink raw reply [flat|nested] 20+ messages in thread* [PATCH v2 5/8] mm/page_owner: Add memcg filter support
2026-09-03 4:18 [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering Zhen Ni
` (3 preceding siblings ...)
2026-09-03 4:18 ` [PATCH v2 4/8] mm/page_owner: Refactor memcg handling for cgroup filter support Zhen Ni
@ 2026-09-03 4:18 ` Zhen Ni
2026-09-03 4:18 ` [PATCH v2 6/8] tools/mm: Add PID/TGID/COMM filtering support to page_owner_filter Zhen Ni
` (4 subsequent siblings)
9 siblings, 0 replies; 20+ messages in thread
From: Zhen Ni @ 2026-09-03 4:18 UTC (permalink / raw)
To: Andrew Morton
Cc: David Hildenbrand, Lorenzo Stoakes, Liam R . Howlett,
Vlastimil Babka, Mike Rapoport, Suren Baghdasaryan, Michal Hocko,
Jonathan Corbet, Shuah Khan, Randy Dunlap, Brendan Jackman,
Johannes Weiner, Zi Yan, linux-mm, linux-doc, linux-kernel,
Zhen Ni
Add memory cgroup filtering to page_owner to allow filtering pages by
their memcg path. This helps debug memory usage patterns for specific
cgroups. Users can now filter page_owner output to show only pages
belonging to a particular memory cgroup.
Collect cgroup path in memcg_info using cgroup_path() and store the
filter state in page_owner_filter_state. When the user sets memcg filter
via "memcg=<path>" command, compare each page's cgroup path against
the specified path and skip non-matching pages using strcmp.
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
Changes in v2:
- Allocate the cgroup path buffer once per read() outside the loop
instead of per page inside get_page_memcg_info(); a GFP_KERNEL
allocation must not sleep inside the page_ext RCU read-side critical
section.
- Guard the memcg= parsing branch with CONFIG_MEMCG
v1: https://lore.kernel.org/linux-mm/20260828031339.1270699-6-zhen.ni@easystack.cn/
---
mm/page_owner.c | 81 ++++++++++++++++++++++++++++++++++++++++++++++---
1 file changed, 76 insertions(+), 5 deletions(-)
diff --git a/mm/page_owner.c b/mm/page_owner.c
index ca9dd8ed9f77..0915bcf46963 100644
--- a/mm/page_owner.c
+++ b/mm/page_owner.c
@@ -71,6 +71,7 @@ static const char * const page_owner_print_mode_strings[] = {
struct memcg_info {
char name[80];
+ char *path;
bool is_slab;
bool is_objcg;
bool is_online;
@@ -86,6 +87,7 @@ struct page_owner_filter_state {
enum page_owner_print_mode print_mode;
bool nid_filter_enabled;
bool proc_filter_enabled;
+ bool memcg_filter_enabled;
nodemask_t nid_filter;
int pid_count;
int tgid_count;
@@ -93,6 +95,7 @@ struct page_owner_filter_state {
pid_t pid_list[MAX_FILTER_PIDS];
pid_t tgid_list[MAX_FILTER_TGIDS];
char comm_list[MAX_FILTER_COMMS][TASK_COMM_LEN];
+ char *memcg_path;
};
static int cmp_pid_t(const void *a, const void *b)
@@ -582,7 +585,8 @@ void pagetypeinfo_showmixedcount_print(struct seq_file *m,
/*
* Get memcg information from page
*/
-static void get_page_memcg_info(struct page *page, struct memcg_info *info)
+static void get_page_memcg_info(struct page *page, struct memcg_info *info,
+ char *path_buf)
{
unsigned long memcg_data;
struct obj_cgroup *objcg;
@@ -606,6 +610,10 @@ static void get_page_memcg_info(struct page *page, struct memcg_info *info)
info->is_objcg = (memcg_data & MEMCG_DATA_KMEM) != 0;
info->is_online = css_is_online(&memcg->css);
cgroup_name(memcg->css.cgroup, info->name, sizeof(info->name));
+ if (path_buf) {
+ info->path = path_buf;
+ cgroup_path(memcg->css.cgroup, info->path, PATH_MAX);
+ }
out_unlock:
rcu_read_unlock();
}
@@ -775,11 +783,23 @@ read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
struct page_ext *page_ext;
struct page_owner *page_owner;
depot_stack_handle_t handle;
+ char *memcg_path_buf = NULL;
struct page_owner_filter_state *state = file->private_data;
+ ssize_t ret;
if (!static_branch_unlikely(&page_owner_inited))
return -EINVAL;
+ /*
+ * Allocate outside the loop, as GFP_KERNEL allocations may not
+ * sleep inside the page_ext RCU read-side critical section.
+ */
+ if (state->memcg_filter_enabled) {
+ memcg_path_buf = kmalloc(PATH_MAX, GFP_KERNEL);
+ if (!memcg_path_buf)
+ return -ENOMEM;
+ }
+
page = NULL;
if (*ppos == 0)
pfn = min_low_pfn;
@@ -899,7 +919,11 @@ read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
}
#ifdef CONFIG_MEMCG
- get_page_memcg_info(page, &memcg_info);
+ get_page_memcg_info(page, &memcg_info, memcg_path_buf);
+ if (state->memcg_filter_enabled)
+ if (!memcg_info.path ||
+ strcmp(memcg_info.path, state->memcg_path) != 0)
+ goto ext_put_continue;
#endif
/* Record the next PFN to read in the file offset */
@@ -907,13 +931,16 @@ read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
page_owner_tmp = *page_owner;
page_ext_put(page_ext);
- return print_page_owner(buf, count, pfn, page,
+ ret = print_page_owner(buf, count, pfn, page,
&page_owner_tmp, handle, state, &memcg_info);
+ kfree(memcg_path_buf);
+ return ret;
ext_put_continue:
page_ext_put(page_ext);
cond_resched();
}
+ kfree(memcg_path_buf);
return 0;
}
@@ -1016,7 +1043,10 @@ static int page_owner_open(struct inode *inode, struct file *file)
static int page_owner_release(struct inode *inode, struct file *file)
{
- kfree(file->private_data);
+ struct page_owner_filter_state *state = file->private_data;
+
+ kfree(state->memcg_path);
+ kfree(state);
return 0;
}
@@ -1075,9 +1105,11 @@ static ssize_t page_owner_write(struct file *file,
nodemask_t new_nid_filter;
bool new_nid_filter_enabled;
bool new_proc_filter_enabled;
+ bool new_memcg_filter_enabled;
pid_t new_pid_list[MAX_FILTER_PIDS];
pid_t new_tgid_list[MAX_FILTER_TGIDS];
char (*new_comm_list)[TASK_COMM_LEN] = NULL;
+ char *new_memcg_path;
int new_pid_count = 0;
int new_tgid_count = 0;
int new_comm_count = 0;
@@ -1093,16 +1125,24 @@ static ssize_t page_owner_write(struct file *file,
if (count > 32 + 6 * MAX_NUMNODES +
(PID_MAX_DIGITS + 1) * MAX_FILTER_PIDS + 4 +
(PID_MAX_DIGITS + 1) * MAX_FILTER_TGIDS + 5 +
- TASK_COMM_LEN * MAX_FILTER_COMMS + 5)
+ TASK_COMM_LEN * MAX_FILTER_COMMS + 5 +
+ PATH_MAX + 6)
return -EINVAL;
new_comm_list = kmalloc_array(MAX_FILTER_COMMS, TASK_COMM_LEN, GFP_KERNEL);
if (!new_comm_list)
return -ENOMEM;
+ new_memcg_path = kmalloc(PATH_MAX, GFP_KERNEL);
+ if (!new_memcg_path) {
+ kfree(new_comm_list);
+ return -ENOMEM;
+ }
+
kbuf = memdup_user_nul(buf, count);
if (IS_ERR(kbuf)) {
kfree(new_comm_list);
+ kfree(new_memcg_path);
return PTR_ERR(kbuf);
}
@@ -1125,6 +1165,9 @@ static ssize_t page_owner_write(struct file *file,
state->comm_count * TASK_COMM_LEN);
new_comm_count = state->comm_count;
}
+ new_memcg_filter_enabled = state->memcg_filter_enabled;
+ if (state->memcg_filter_enabled && state->memcg_path)
+ strscpy(new_memcg_path, state->memcg_path, PATH_MAX);
while ((token = strsep(&kbuf, " \t\n")) != NULL) {
if (*token == '\0')
@@ -1168,12 +1211,35 @@ static ssize_t page_owner_write(struct file *file,
ret = parse_comm_list(token + 5, new_comm_list, &new_comm_count);
if (ret < 0)
goto out_free;
+#ifdef CONFIG_MEMCG
+ } else if (!strncmp(token, "memcg=", 6)) {
+ if (token[6] == '\0') {
+ ret = -EINVAL;
+ goto out_free;
+ }
+ ret = strscpy(new_memcg_path, token + 6, PATH_MAX);
+ if (ret < 0)
+ goto out_free;
+ new_memcg_filter_enabled = true;
+#endif
} else {
ret = -EINVAL;
goto out_free;
}
}
+ if (new_memcg_filter_enabled) {
+ if (!state->memcg_path) {
+ state->memcg_path = kzalloc(PATH_MAX, GFP_KERNEL);
+ if (!state->memcg_path) {
+ ret = -ENOMEM;
+ goto out_free;
+ }
+ } else {
+ memset(state->memcg_path, 0, PATH_MAX);
+ }
+ }
+
/* Commit all filter changes */
state->print_mode = new_print_mode;
state->nid_filter = new_nid_filter;
@@ -1200,11 +1266,16 @@ static ssize_t page_owner_write(struct file *file,
new_comm_count * TASK_COMM_LEN);
state->comm_count = new_comm_count;
}
+ if (new_memcg_filter_enabled) {
+ strscpy(state->memcg_path, new_memcg_path, PATH_MAX);
+ state->memcg_filter_enabled = true;
+ }
ret = count;
out_free:
kfree(new_comm_list);
+ kfree(new_memcg_path);
kfree(orig);
return ret;
}
--
2.20.1
^ permalink raw reply [flat|nested] 20+ messages in thread* [PATCH v2 6/8] tools/mm: Add PID/TGID/COMM filtering support to page_owner_filter
2026-09-03 4:18 [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering Zhen Ni
` (4 preceding siblings ...)
2026-09-03 4:18 ` [PATCH v2 5/8] mm/page_owner: Add memcg " Zhen Ni
@ 2026-09-03 4:18 ` Zhen Ni
2026-09-03 4:18 ` [PATCH v2 7/8] tools/mm: Add memory cgroup " Zhen Ni
` (3 subsequent siblings)
9 siblings, 0 replies; 20+ messages in thread
From: Zhen Ni @ 2026-09-03 4:18 UTC (permalink / raw)
To: Andrew Morton
Cc: David Hildenbrand, Lorenzo Stoakes, Liam R . Howlett,
Vlastimil Babka, Mike Rapoport, Suren Baghdasaryan, Michal Hocko,
Jonathan Corbet, Shuah Khan, Randy Dunlap, Brendan Jackman,
Johannes Weiner, Zi Yan, linux-mm, linux-doc, linux-kernel,
Zhen Ni
Add command-line options for filtering by process ID (PID), thread group
ID (TGID), and process name (COMM) to the page_owner_filter userspace tool.
New options:
-p, --pid PID_LIST : Process IDs (comma-separated, max 16)
-t, --tgid TGID_LIST : Thread Group IDs (comma-separated, max 16)
-c, --comm COMM_LIST : Process names (comma-separated, max 8)
Supports wildcards: * ? [a-z]
Usage examples:
page_owner_filter -p 1234,5678
page_owner_filter -c "python*"
page_owner_filter -n 0 -c kworker* -o output.txt
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
Changes in v2:
- Print error messages for empty -p/-t and -c arguments.
v1: https://lore.kernel.org/linux-mm/20260828031339.1270699-7-zhen.ni@easystack.cn/
---
tools/mm/page_owner_filter.c | 172 ++++++++++++++++++++++++++++++-----
1 file changed, 151 insertions(+), 21 deletions(-)
diff --git a/tools/mm/page_owner_filter.c b/tools/mm/page_owner_filter.c
index 1d1f0a38678a..516c2d109a6a 100644
--- a/tools/mm/page_owner_filter.c
+++ b/tools/mm/page_owner_filter.c
@@ -21,22 +21,24 @@
#include <signal.h>
#define MAX_CMD_LEN 512
+#define TASK_COMM_LEN 16
static void usage(const char *prog)
{
fprintf(stderr, "Usage: %s [OPTIONS]\n", prog);
fprintf(stderr, "\nOptions:\n");
- fprintf(stderr, " -m, --mode MODE : print_mode (stack, handle, or stack_handle)\n");
- fprintf(stderr, " -n, --nid NID_LIST : NUMA node IDs (comma-separated or ranges)\n");
- fprintf(stderr, " -o, --output FILE : output file (default: stdout)\n");
- fprintf(stderr, " -h, --help : show this help message\n");
+ fprintf(stderr, " -m, --mode MODE : print_mode (stack, handle, stack_handle)\n");
+ fprintf(stderr, " -n, --nid NID_LIST : NUMA nodes (comma-separated or ranges)\n");
+ fprintf(stderr, " -p, --pid PID_LIST : Process IDs (comma-separated, max 16)\n");
+ fprintf(stderr, " -t, --tgid TGID_LIST : Thread Group IDs (comma-separated, max 16)\n");
+ fprintf(stderr, " -c, --comm COMM_LIST : Process names (comma-separated, max 8)\n");
+ fprintf(stderr, " Supports wildcards: * ? [a-z]\n");
+ fprintf(stderr, " -o, --output FILE : output file (default: stdout)\n");
+ fprintf(stderr, " -h, --help : show this help message\n");
fprintf(stderr, "\nExamples:\n");
- fprintf(stderr, " %s -m stack\n", prog);
- fprintf(stderr, " %s -m handle\n", prog);
- fprintf(stderr, " %s -m stack_handle\n", prog);
- fprintf(stderr, " %s -m stack -o output.txt\n", prog);
- fprintf(stderr, " %s -n 0,1,2\n", prog);
- fprintf(stderr, " %s -m stack -n 0\n", prog);
+ fprintf(stderr, " %s -m handle -o output.txt\n", prog);
+ fprintf(stderr, " %s -n 0,1 -c bash\n", prog);
+ fprintf(stderr, " %s -c \"python*\" -t 1\n", prog);
}
static int validate_mode(const char *mode)
@@ -132,6 +134,97 @@ static int validate_nid_list(const char *nid_list)
return 0;
}
+static int validate_pid_list(const char *pid_list)
+{
+ const char *p;
+ int count = 0;
+
+ if (!pid_list || strlen(pid_list) == 0) {
+ fprintf(stderr, "Error: Empty pid/tgid list\n");
+ return -1;
+ }
+
+ for (p = pid_list; *p; p++) {
+ if (*p == ',') {
+ count++;
+ continue;
+ }
+ if (!isdigit((unsigned char)*p)) {
+ fprintf(stderr,
+ "Error: Invalid character '%c' in pid_list (only digits allowed)\n",
+ *p);
+ return -1;
+ }
+ }
+
+ if (++count > 16) {
+ fprintf(stderr, "Error: Too many PIDs (max 16)\n");
+ return -1;
+ }
+
+ return 0;
+}
+
+static int validate_tgid_list(const char *tgid_list)
+{
+ return validate_pid_list(tgid_list);
+}
+
+static int validate_comm_list(const char *comm_list)
+{
+ const char *p;
+ const char *comm_start;
+ int count = 0;
+ int comm_len = 0;
+
+ if (!comm_list || strlen(comm_list) == 0) {
+ fprintf(stderr, "Error: Empty comm list\n");
+ return -1;
+ }
+
+ comm_start = comm_list;
+ for (p = comm_list; *p; p++) {
+ if (*p == ',') {
+ /* Check COMM length before separator */
+ if (comm_len == 0) {
+ fprintf(stderr, "Error: Empty COMM in list\n");
+ return -1;
+ }
+ if (comm_len >= TASK_COMM_LEN) {
+ fprintf(stderr,
+ "Error: COMM too long (max %d chars)\n",
+ TASK_COMM_LEN - 1);
+ fprintf(stderr, " Near: %.15s...\n", comm_start);
+ return -1;
+ }
+ count++;
+ comm_len = 0;
+ comm_start = p + 1;
+ continue;
+ }
+ comm_len++;
+ }
+
+ /* Check last COMM */
+ if (comm_len == 0) {
+ fprintf(stderr, "Error: Empty COMM at end of list\n");
+ return -1;
+ }
+ if (comm_len >= TASK_COMM_LEN) {
+ fprintf(stderr, "Error: COMM too long (max %d chars)\n",
+ TASK_COMM_LEN - 1);
+ fprintf(stderr, " Near: %.15s...\n", comm_start);
+ return -1;
+ }
+
+ if (++count > 8) {
+ fprintf(stderr, "Error: Too many COMMs (max 8)\n");
+ return -1;
+ }
+
+ return 0;
+}
+
int main(int argc, char *argv[])
{
const char *output_file = NULL;
@@ -148,6 +241,9 @@ int main(int argc, char *argv[])
static struct option long_options[] = {
{"mode", required_argument, 0, 'm'},
{"nid", required_argument, 0, 'n'},
+ {"pid", required_argument, 0, 'p'},
+ {"tgid", required_argument, 0, 't'},
+ {"comm", required_argument, 0, 'c'},
{"output", required_argument, 0, 'o'},
{"help", no_argument, 0, 'h'},
{0, 0, 0, 0}
@@ -174,7 +270,7 @@ int main(int argc, char *argv[])
return 1;
}
- while ((opt = getopt_long(argc, argv, "m:n:o:h", long_options, NULL)) != -1) {
+ while ((opt = getopt_long(argc, argv, "m:n:p:t:c:o:h", long_options, NULL)) != -1) {
int len;
switch (opt) {
@@ -206,6 +302,48 @@ int main(int argc, char *argv[])
cmd_len += len;
break;
}
+ case 'p': {
+ const char *pid_list = optarg;
+
+ if (validate_pid_list(pid_list) < 0)
+ return 1;
+ len = snprintf(filter_cmd + cmd_len, MAX_CMD_LEN - cmd_len,
+ "%spid=%s", cmd_len > 0 ? " " : "", pid_list);
+ if (len < 0 || cmd_len + len >= MAX_CMD_LEN) {
+ fprintf(stderr, "Error: Command too long\n");
+ return 1;
+ }
+ cmd_len += len;
+ break;
+ }
+ case 't': {
+ const char *tgid_list = optarg;
+
+ if (validate_tgid_list(tgid_list) < 0)
+ return 1;
+ len = snprintf(filter_cmd + cmd_len, MAX_CMD_LEN - cmd_len,
+ "%stgid=%s", cmd_len > 0 ? " " : "", tgid_list);
+ if (len < 0 || cmd_len + len >= MAX_CMD_LEN) {
+ fprintf(stderr, "Error: Command too long\n");
+ return 1;
+ }
+ cmd_len += len;
+ break;
+ }
+ case 'c': {
+ const char *comm_list = optarg;
+
+ if (validate_comm_list(comm_list) < 0)
+ return 1;
+ len = snprintf(filter_cmd + cmd_len, MAX_CMD_LEN - cmd_len,
+ "%scomm=%s", cmd_len > 0 ? " " : "", comm_list);
+ if (len < 0 || cmd_len + len >= MAX_CMD_LEN) {
+ fprintf(stderr, "Error: Command too long\n");
+ return 1;
+ }
+ cmd_len += len;
+ break;
+ }
case 'o':
output_file = optarg;
break;
@@ -220,7 +358,7 @@ int main(int argc, char *argv[])
/* At least one filter must be specified */
if (cmd_len == 0) {
- fprintf(stderr, "Error: At least one filter (-m or -n) must be specified\n\n");
+ fprintf(stderr, "Error: At least one filter must be specified\n\n");
usage(argv[0]);
return 1;
}
@@ -255,15 +393,7 @@ int main(int argc, char *argv[])
ret = write(fd, filter_cmd, strlen(filter_cmd));
if (ret < 0) {
- if (errno == EINVAL) {
- fprintf(stderr, "Error: Kernel rejected the filter command.\n");
- fprintf(stderr, "Possible causes:\n");
- fprintf(stderr, " - Kernel does not support per-fd filtering\n");
- fprintf(stderr, " - NUMA node has no memory\n");
- fprintf(stderr, " - Unknown reason\n");
- } else {
- perror("write filter command");
- }
+ perror("write filter command");
goto out;
}
--
2.20.1
^ permalink raw reply [flat|nested] 20+ messages in thread* [PATCH v2 7/8] tools/mm: Add memory cgroup filtering support to page_owner_filter
2026-09-03 4:18 [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering Zhen Ni
` (5 preceding siblings ...)
2026-09-03 4:18 ` [PATCH v2 6/8] tools/mm: Add PID/TGID/COMM filtering support to page_owner_filter Zhen Ni
@ 2026-09-03 4:18 ` Zhen Ni
2026-09-03 4:18 ` [PATCH v2 8/8] Documentation: page_owner: Document PID/TGID/COMM and cgroup filters Zhen Ni
` (2 subsequent siblings)
9 siblings, 0 replies; 20+ messages in thread
From: Zhen Ni @ 2026-09-03 4:18 UTC (permalink / raw)
To: Andrew Morton
Cc: David Hildenbrand, Lorenzo Stoakes, Liam R . Howlett,
Vlastimil Babka, Mike Rapoport, Suren Baghdasaryan, Michal Hocko,
Jonathan Corbet, Shuah Khan, Randy Dunlap, Brendan Jackman,
Johannes Weiner, Zi Yan, linux-mm, linux-doc, linux-kernel,
Zhen Ni
Add filtering capability for page_owner to allow filtering by
memory cgroup path.
Filter page_owner output by cgroup path:
./page_owner_filter -g /
./page_owner_filter -g /user.slice
./page_owner_filter -g /user.slice -c systemd
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
Changes in v2:
- Print an error message for an empty -g argument.
v1: https://lore.kernel.org/linux-mm/20260828031339.1270699-8-zhen.ni@easystack.cn/
---
tools/mm/page_owner_filter.c | 58 ++++++++++++++++++++++++++++++++++--
1 file changed, 55 insertions(+), 3 deletions(-)
diff --git a/tools/mm/page_owner_filter.c b/tools/mm/page_owner_filter.c
index 516c2d109a6a..8aea7df23eef 100644
--- a/tools/mm/page_owner_filter.c
+++ b/tools/mm/page_owner_filter.c
@@ -20,7 +20,7 @@
#include <getopt.h>
#include <signal.h>
-#define MAX_CMD_LEN 512
+#define MAX_CMD_LEN 2048
#define TASK_COMM_LEN 16
static void usage(const char *prog)
@@ -33,12 +33,13 @@ static void usage(const char *prog)
fprintf(stderr, " -t, --tgid TGID_LIST : Thread Group IDs (comma-separated, max 16)\n");
fprintf(stderr, " -c, --comm COMM_LIST : Process names (comma-separated, max 8)\n");
fprintf(stderr, " Supports wildcards: * ? [a-z]\n");
+ fprintf(stderr, " -g, --cgroup PATH : Memory cgroup path\n");
fprintf(stderr, " -o, --output FILE : output file (default: stdout)\n");
fprintf(stderr, " -h, --help : show this help message\n");
fprintf(stderr, "\nExamples:\n");
fprintf(stderr, " %s -m handle -o output.txt\n", prog);
fprintf(stderr, " %s -n 0,1 -c bash\n", prog);
- fprintf(stderr, " %s -c \"python*\" -t 1\n", prog);
+ fprintf(stderr, " %s -c \"python*\" -g user.slice\n", prog);
}
static int validate_mode(const char *mode)
@@ -225,6 +226,40 @@ static int validate_comm_list(const char *comm_list)
return 0;
}
+static int validate_cgroup_path(const char *path)
+{
+ char cgroup_path[512];
+ const char *input_path = path;
+ int is_cgroup_v2 = 0;
+
+ if (!path || strlen(path) == 0) {
+ fprintf(stderr, "Error: Empty cgroup path\n");
+ return -1;
+ }
+
+ if (path[0] == '/')
+ input_path++;
+
+ /* Check if v1 memory controller exists */
+ if (access("/sys/fs/cgroup/memory", F_OK) != 0)
+ is_cgroup_v2 = 1;
+
+ if (is_cgroup_v2)
+ snprintf(cgroup_path, sizeof(cgroup_path),
+ "/sys/fs/cgroup/%s/memory.stat", input_path);
+ else
+ snprintf(cgroup_path, sizeof(cgroup_path),
+ "/sys/fs/cgroup/memory/%s/memory.stat", input_path);
+
+ if (access(cgroup_path, F_OK) != 0) {
+ fprintf(stderr, "Error: Cgroup path '%s': not found or no memory controller\n",
+ path);
+ return -1;
+ }
+
+ return 0;
+}
+
int main(int argc, char *argv[])
{
const char *output_file = NULL;
@@ -244,6 +279,7 @@ int main(int argc, char *argv[])
{"pid", required_argument, 0, 'p'},
{"tgid", required_argument, 0, 't'},
{"comm", required_argument, 0, 'c'},
+ {"cgroup", required_argument, 0, 'g'},
{"output", required_argument, 0, 'o'},
{"help", no_argument, 0, 'h'},
{0, 0, 0, 0}
@@ -270,7 +306,7 @@ int main(int argc, char *argv[])
return 1;
}
- while ((opt = getopt_long(argc, argv, "m:n:p:t:c:o:h", long_options, NULL)) != -1) {
+ while ((opt = getopt_long(argc, argv, "m:n:p:t:c:g:o:h", long_options, NULL)) != -1) {
int len;
switch (opt) {
@@ -344,6 +380,22 @@ int main(int argc, char *argv[])
cmd_len += len;
break;
}
+ case 'g': {
+ const char *cgroup_path = optarg;
+
+ if (validate_cgroup_path(cgroup_path) < 0)
+ return 1;
+ const char *path = (cgroup_path[0] == '/') ? cgroup_path + 1 : cgroup_path;
+
+ len = snprintf(filter_cmd + cmd_len, MAX_CMD_LEN - cmd_len,
+ "%smemcg=/%s", cmd_len > 0 ? " " : "", path);
+ if (len < 0 || cmd_len + len >= MAX_CMD_LEN) {
+ fprintf(stderr, "Error: Command too long\n");
+ return 1;
+ }
+ cmd_len += len;
+ break;
+ }
case 'o':
output_file = optarg;
break;
--
2.20.1
^ permalink raw reply [flat|nested] 20+ messages in thread* [PATCH v2 8/8] Documentation: page_owner: Document PID/TGID/COMM and cgroup filters
2026-09-03 4:18 [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering Zhen Ni
` (6 preceding siblings ...)
2026-09-03 4:18 ` [PATCH v2 7/8] tools/mm: Add memory cgroup " Zhen Ni
@ 2026-09-03 4:18 ` Zhen Ni
[not found] ` <20260902221225.228fb4b18e115ba55b29fe29@linux-foundation.org>
2026-09-04 8:25 ` Vlastimil Babka (SUSE)
9 siblings, 0 replies; 20+ messages in thread
From: Zhen Ni @ 2026-09-03 4:18 UTC (permalink / raw)
To: Andrew Morton
Cc: David Hildenbrand, Lorenzo Stoakes, Liam R . Howlett,
Vlastimil Babka, Mike Rapoport, Suren Baghdasaryan, Michal Hocko,
Jonathan Corbet, Shuah Khan, Randy Dunlap, Brendan Jackman,
Johannes Weiner, Zi Yan, linux-mm, linux-doc, linux-kernel,
Zhen Ni
Update page_owner.rst to document process and cgroup filtering
support in page_owner_filter tool. Add usage examples for PID, TGID,
COMM (with wildcard support), and cgroup filters along with their
respective limits.
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
Changes in v2:
- Quote the wildcard pattern in the -c example.
v1: https://lore.kernel.org/linux-mm/20260828031339.1270699-9-zhen.ni@easystack.cn/
---
Documentation/mm/page_owner.rst | 25 ++++++++++++++++++++++++-
1 file changed, 24 insertions(+), 1 deletion(-)
diff --git a/Documentation/mm/page_owner.rst b/Documentation/mm/page_owner.rst
index a6bd3fe6423a..493f38db0677 100644
--- a/Documentation/mm/page_owner.rst
+++ b/Documentation/mm/page_owner.rst
@@ -283,7 +283,7 @@ page_owner supports filtering output at the kernel level before reading,
which reduces the amount of data that needs to be processed in userspace.
The page_owner_filter tool provides a convenient interface for this filtering
-capability. It supports two types of filters:
+capability. It supports the following types of filters:
1. **print_mode filter**: Control what information is printed for each page
- ``stack``: Print full stack traces (default, compatible with existing usage)
@@ -300,6 +300,16 @@ capability. It supports two types of filters:
- Ranges: ``-n 0-3``
- Mixed format: ``-n 0,2-3,5``
+3. **Process filters**: Filter pages by process identifiers
+ - Filter by process ID: ``-p PID_LIST`` (comma-separated, max 16)
+ - Filter by thread group ID: ``-t TGID_LIST`` (comma-separated, max 16)
+ - Filter by task command name: ``-c COMM_LIST`` (comma-separated, max 8)
+ - Name matching supports wildcards: ``*``, ``?``, ``[a-z]``
+
+4. **Cgroup (memcg) filter**: Filter pages by memory cgroup
+ - Filter by cgroup path: ``-g CGROUP``
+ - Useful for containerized environments and multi-tenant systems
+
Usage examples::
# Filter by print mode
@@ -310,9 +320,19 @@ Usage examples::
./page_owner_filter -n 0
./page_owner_filter -n 0-3
+ # Filter by process
+ ./page_owner_filter -p 1234
+ ./page_owner_filter -t 1,2,3
+ ./page_owner_filter -c 'python*'
+
+ # Filter by cgroup
+ ./page_owner_filter -g system.slice
+ ./page_owner_filter -g kubepods/besteffort/pod123
+
# Combined filters
./page_owner_filter -m stack -n 0,1,2
./page_owner_filter -m handle -n 0,2-3
+ ./page_owner_filter -g user.slice -c bash -n 0
# Save to file
./page_owner_filter -m handle -o filtered_output.txt
@@ -323,6 +343,9 @@ reduce output size by ~66% (84MB vs 244MB) and improve read performance by ~4.4x
compared to full stack output.
The NUMA node filter is useful for NUMA-aware memory allocation analysis and debugging.
+Process filters help isolate memory allocations for specific processes or tasks.
+The cgroup filter is essential for containerized environments where you need to
+analyze memory usage per container or service.
Behind the scenes, page_owner_filter opens /sys/kernel/debug/page_owner and
writes filter commands before reading the filtered output. The filtering uses
--
2.20.1
^ permalink raw reply [flat|nested] 20+ messages in thread[parent not found: <20260902221225.228fb4b18e115ba55b29fe29@linux-foundation.org>]
* Re: [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering
[not found] ` <20260902221225.228fb4b18e115ba55b29fe29@linux-foundation.org>
@ 2026-09-03 12:00 ` zhen.ni
0 siblings, 0 replies; 20+ messages in thread
From: zhen.ni @ 2026-09-03 12:00 UTC (permalink / raw)
To: Andrew Morton
Cc: David Hildenbrand, Lorenzo Stoakes, Liam R . Howlett,
Vlastimil Babka, Mike Rapoport, Suren Baghdasaryan, Michal Hocko,
Jonathan Corbet, Shuah Khan, Randy Dunlap, Brendan Jackman,
Johannes Weiner, Zi Yan, linux-mm, linux-doc, linux-kernel,
Zhen Ni
在 2026/9/3 13:12, Andrew Morton 写道:
> On Thu, 3 Sep 2026 12:18:11 +0800 Zhen Ni <zhen.ni@easystack.cn> wrote:
>
>> This patch series adds process and memory cgroup filtering support to
>> page_owner.
>
> Cool. Is any of this useful?
> https://sashiko.dev/#/patchset/20260903041819.1776630-1-zhen.ni@easystack.cn
>
>
Hi Andrew,
All findings that v2 fixed are confirmed gone. Replies to the new
findings below, quoted verbatim, in report order.
> Will this comparison safely handle concurrent updates to the PID?
>
> Since cmp_int() is a macro that evaluates the arguments twice:
> ((a) > (b)) - ((a) < (b))
>
> And bsearch() below receives a pointer to page_owner->pid which can
be updated
> concurrently without locks, could the dereferenced pointer change
between the
> two evaluations?
>
> If the value changes from being less than the target to greater than the
> target, both comparisons evaluate to false. This would cause cmp_int() to
> incorrectly return 0 and falsely match unrelated processes.
[Shared-fd concurrency] Will not fix.
Reproducing it needs a concurrent writer updating page records while the
same fd is being read, which is outside the designed configure-then-read
usage. Whether to add spinlock protection for shared-fd read/write on a
non-production-grade debugging interface was already discussed at
length in the earlier series "mm/page_owner: add per-fd filter
infrastructure for print_mode and NUMA filtering"; the conclusion there
was that the locking cost does not match the benefit, shared-fd
read/write is not a recommended usage, and the recommendation is to
simply use the tool.
> Does this array need to be zero-initialized?
Will not fix.
Nothing reads past the NUL terminator or beyond the valid entries.
> Is new_proc_filter_enabled ever read after being initialized here?
>
> It seems this unused variable masks a logic bug below where
state->pid_count
> fails to clear when the filter is disabled.
The variable itself is indeed unused - the commit path recomputes the
flag from new_pid_count/new_tgid_count/new_comm_count directly, so the
staging copy can be dropped. Will remove it in v3.
[No clear-filter operation] But it masks no logic bug. There is no
clear operation by design. To start over, close the fd and open a
fresh one; the page_owner_filter tool already works this way. This
also matches the semantics of the original filter introduction (mode
and nid have always worked this way). A clear operation would be
over-design: the recommended page_owner_filter tool never needs it.
If one run does not produce the wanted result, the simplest fix is to
adjust the arguments and run it again, not to fight the already-set
filters on an open fd.
> When a user writes a filter with fewer than MAX_FILTER_PIDS, does this
> unconditional copy of sizeof(state->pid_list) capture uninitialized
kernel
> stack memory from new_pid_list into the heap-allocated state object?
Will not fix. The uninitialized tail is never read: bsearch is bounded
by state->pid_count.
> If the filter is explicitly disabled (new_pid_count == 0), this block is
> skipped and state->pid_count is not reset to 0.
>
> Will this cause the filter to be unintentionally re-enabled if a user
> subsequently writes a different configuration like "mode=stack"?
>
> The old state->pid_count would be copied to new_pid_count during
> initialization, causing state->proc_filter_enabled to be incorrectly set
> back to true.
[No clear-filter operation] (see above).
> Will this allocate the new_tgid_list array on the stack without
> initialization?
Will not fix. bsearch on the tgid list is bounded by
state->tgid_count, so the uninitialized tail is never consumed.
> If a user tries to clear a PID filter by writing "pid=" while a TGID
> filter is still active, new_pid_count will be 0 but
> state->proc_filter_enabled remains true.
>
> Will state->pid_count fail to update to 0 since the update block requires
> new_pid_count > 0, resulting in the page_owner output continuing to
filter
> on stale criteria?
[No clear-filter operation] (see above).
> When the filter is initially empty (state->tgid_count == 0),
> new_tgid_list remains mostly uninitialized if only one element is set
> (for example, writing "tgid=123").
>
> Does this unconditional copy of sizeof(state->tgid_list) copy
uninitialized
> stack memory into the heap-allocated state, potentially triggering a
KMSAN
> uninit-value warning?
Will not fix. Same as the pid list.
> Similar to the PID filter above, if a user attempts to clear an active
> TGID filter while a PID filter is active, does this new_tgid_count > 0
> check prevent state->tgid_count from being zeroed and retain the stale
> data?
[No clear-filter operation] (see above).
> Can this lockless read of state->comm_list observe unterminated strings?
> Without guaranteed NUL padding, this could cause an out-of-bounds
access in
> glob_match() if data races occur.
[Shared-fd concurrency] Will not fix.
> Does this leave the trailing padding of the 16-byte slot uninitialized?
> Since only up to the string length is initialized, the remainder of the
> buffer may contain garbage.
Will not fix. glob_match() stops at the NUL written by strscpy(), so
the padding past it is never read.
> Does returning -EINVAL for an empty list prevent users from clearing the
> filter? If a user writes comm= to the /sys/kernel/debug/page_owner file,
> the parser aborts. Other filters allow clearing their state by
providing an
> empty list.
[No clear-filter operation] (see above).
> Does allocating this array without zero-initialization leave trailing
bytes
> as uninitialized memory? Replacing kmalloc_array() with kcalloc() (or
using
> kzalloc()) would prevent uninitialized heap memory from being used.
Will not fix. No consumer of the list reads past strscpy()'s NUL
terminator, so uninitialized bytes are unreachable.
> Can this lockless update copy uninitialized heap memory into the shared
> state structure? This can trigger KMSAN use-of-uninitialized-memory
warnings.
Will not fix. Same reasoning as above; the relocated bytes are never
read.
> Even if parse_comm_list() were modified to successfully return with
> new_comm_count == 0, would skipping the update here permanently leave
the old
> filter active? state->comm_count is never reset to 0 in this function,
> which breaks filter management on an open file descriptor.
[No clear-filter operation] (see above).
> Can this lead to a NULL pointer dereference in strcmp() on weakly-ordered
> architectures?
>
> A thread calling pwrite() on the page_owner file could be updating
the filter
> state while another thread concurrently calls pread(). Since
page_owner_write()
> updates the state without memory barriers:
>
> page_owner_write():
> strscpy(state->memcg_path, new_memcg_path, PATH_MAX);
> state->memcg_filter_enabled = true;
>
> Could the CPU commit the true flag to memory before the state->memcg_path
> pointer is allocated or written? If the reader observes
> state->memcg_filter_enabled == true but state->memcg_path is still NULL,
> strcmp() will crash.
[Shared-fd concurrency], but with a distinction: a torn read that
mis-filters one or two pages is acceptable, a crash is not - so this
one is worth fixing without any locking. The fix is to snapshot the
pointer once:
char *filter_path = READ_ONCE(state->memcg_path);
if (!memcg_info.path || !filter_path ||
strcmp(memcg_info.path, filter_path) != 0)
goto ext_put_continue;
Will include this in v3.
> Is there a risk of a memory leak here if two threads write to the
page_owner
> file concurrently?
>
> VFS pwrite() does not hold f_pos_lock, so multiple threads could observe
> !state->memcg_path simultaneously, perform the kzalloc(), and
sequentially
> overwrite the state->memcg_path pointer. Would this irrevocably lose the
> first allocation?
[Shared-fd concurrency] (see above).
> Since this filter accepts wildcard patterns like -c "[a-zA-Z]*worker*",
> should the length check allow for patterns that exceed TASK_COMM_LEN?
>
> A valid glob pattern string can easily exceed the 15-character limit of
> a process name because character classes and wildcards take up more
space
> than the literal characters they match.
>
> By applying TASK_COMM_LEN to the pattern string here, does this prevent
> the use of complex but valid glob patterns?
Will not fix. The kernel itself truncates comm to TASK_COMM_LEN at
source - complex over-long wildcards are a corner case.
"[a-zA-Z]*worker*" works as "[a-z]*worker*" or simply "*worker*".
> Similarly, does applying the TASK_COMM_LEN limit here incorrectly reject
> valid wildcard strings for the last entry in the list?
Will not fix, same as above.
> Does this limit prevent the use of maximum-length cgroup paths?
>
> Kernel path limits are typically 4096 bytes. Since the command buffer
here is
> limited to 2048 bytes, deeply nested but valid cgroup paths may fail to
> process.
[Tool buffer sizes]
Will not fix. The kernel limit is indeed PATH_MAX (4096), but that is
an edge case - realistic cgroup paths are around 100 bytes, and the
buffer is already a 512-byte allowance, 5x the common case. The
MAX_CMD_LEN 2048 is likewise an estimate, not a design limit: a
precise worst-case sum of every parameter's maximum size would not be
worth the complexity, and the generous budget is more than enough in
practice.
> Could this 512-byte array silently truncate long but valid cgroup paths?
>
> When snprintf is used later in this function, paths longer than this
buffer
> will be truncated. This will cause the access check to operate on an
> incomplete path, failing validation and presenting a confusing error
to the
> user instead of correctly validating the path.
[Tool buffer sizes] (see above).
> Is this heuristic reliable for detecting cgroup v2?
>
> On a cgroup v2 system, "memory" is a completely valid name for a
cgroup. If
> an administrator or runtime creates a cgroup named "memory" at the
root, this
> directory will exist.
>
> The tool would then incorrectly assume cgroup v1 semantics, which
would cause
> valid v2 paths to be falsely rejected during validation.
"/sys/fs/cgroup/memory" is indeed potentially not robust enough, but
here our purpose is not to determine whether the cgroup is v1 or
v2—rather, it is to check whether the memory controller itself is v1 or
v2. Other mixed v1/v2 scenarios are not our concern, so we can directly
check, similar to the code below.
snprintf(cgroup_path, sizeof(cgroup_path),
"/sys/fs/cgroup/%s/memory.stat", input_path);
if (access(cgroup_path, F_OK) == 0)
return 0;
snprintf(cgroup_path, sizeof(cgroup_path),
"/sys/fs/cgroup/memory/%s/memory.stat", input_path);
if (access(cgroup_path, F_OK) == 0)
return 0;
fprintf(stderr, "Error: Cgroup path '%s': "
"not found or no memory controller\n", path);
return -1;
> Will this fail to validate the root cgroup on v2 systems?
>
> If a user tries to filter by the root cgroup by passing "-g /",
input_path
> becomes an empty string. This snprintf call will then construct
> "/sys/fs/cgroup//memory.stat".
>
> Since cgroup v2 does not expose memory.stat at the root hierarchy
level, the
> access check will fail, incorrectly rejecting the valid root filter and
> preventing analysis of root-level allocations.
Not a bug, and the probe above covers it: for "-g /" the v2 candidate
is "/sys/fs/cgroup//memory.stat", path resolution collapses the double
slash, access() succeeds, and "-g /" passes on both cgroup v1 and v2
(cover letter, section 2.4).
If the above solution is acceptable, I will send the corresponding v3
version.
Thanks,
Zhen Ni
^ permalink raw reply [flat|nested] 20+ messages in thread
* Re: [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering
2026-09-03 4:18 [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering Zhen Ni
` (8 preceding siblings ...)
[not found] ` <20260902221225.228fb4b18e115ba55b29fe29@linux-foundation.org>
@ 2026-09-04 8:25 ` Vlastimil Babka (SUSE)
2026-09-07 4:08 ` zhen.ni
9 siblings, 1 reply; 20+ messages in thread
From: Vlastimil Babka (SUSE) @ 2026-09-04 8:25 UTC (permalink / raw)
To: Zhen Ni, Andrew Morton
Cc: David Hildenbrand, Lorenzo Stoakes, Liam R . Howlett,
Mike Rapoport, Suren Baghdasaryan, Michal Hocko, Jonathan Corbet,
Shuah Khan, Randy Dunlap, Brendan Jackman, Johannes Weiner,
Zi Yan, linux-mm, linux-doc, linux-kernel, Steven Rostedt
On 9/3/26 06:18, Zhen Ni wrote:
> This patch series adds process and memory cgroup filtering support to
> page_owner. Following the previous series that introduced print_mode and
> NUMA node filters:
> https://lore.kernel.org/linux-mm/20260707115411.1714314-1-zhen.ni@easystack.cn/
>
> This series adds filtering capabilities to page_owner, allowing users to
> filter output by specific processes and memory cgroups. Users can now
> filter page_owner output by PID, TGID, COMM (with wildcard support), and
> memory cgroup path. This makes page_owner debugging more focused and
> efficient for tracking memory allocations in specific contexts.
I wonder about the usefulness of all the new filters. In my experience
page_owner is useful to find a kernel memory leak code, and for that the
stacktraces are most useful. Dealing with things like pid/tgid/comm/cgroups
sounds more like your aim is to profile and optimize particular userspace to
use less kernel memory? In that case, isn't it rather the area of memory
allocation profiling (or maybe tracing with bpf), not page_owner?
Moreover, tracing or bpf can already do such kind of filtering and AFAIK
ftrace filters for tracepoints are nice and generic, while this is adding a
bunch of custom parsing and filtering. So that makes me somewhat sceptical.
> Targeted filtering provides significant performance benefits on large memory
> servers by reducing both execution time and output size. By filtering at the
> kernel level before reading, only relevant page allocations are processed,
> dramatically reducing the amount of data that needs to be handled in userspace.
>
> This series extends page_owner filtering capabilities with:
> - PID filtering
> - TGID filtering
> - COMM filtering with wildcard support
> - Cgroup (memcg) filtering for containerized environments
>
> The series is organized as follows:
>
> Patches 1-3: Add PID, TGID, and COMM filtering support to page_owner
> - Support filtering by process ID
> - Support filtering by thread group ID
> - Support filtering by command name with wildcards
>
> Patch 4: Refactor memcg handling to prepare for cgroup filter support
> Patch 5: Add memory cgroup filtering support
>
> Patches 6-7: Update page_owner_filter tool with corresponding features
>
> Patch 8: Update documentation
>
> These filters are particularly useful for:
> - Debugging memory leaks in specific processes
> - Analyzing memory usage in containerized environments
> - Isolating allocations from specific services or applications
>
^ permalink raw reply [flat|nested] 20+ messages in thread* Re: [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering
2026-09-04 8:25 ` Vlastimil Babka (SUSE)
@ 2026-09-07 4:08 ` zhen.ni
2026-09-07 15:21 ` Vlastimil Babka (SUSE)
0 siblings, 1 reply; 20+ messages in thread
From: zhen.ni @ 2026-09-07 4:08 UTC (permalink / raw)
To: Vlastimil Babka (SUSE), Andrew Morton
Cc: David Hildenbrand, Lorenzo Stoakes, Liam R . Howlett,
Mike Rapoport, Suren Baghdasaryan, Michal Hocko, Jonathan Corbet,
Shuah Khan, Randy Dunlap, Brendan Jackman, Johannes Weiner,
Zi Yan, linux-mm, linux-doc, linux-kernel, Steven Rostedt
在 2026/9/4 16:25, Vlastimil Babka (SUSE) 写道:
> On 9/3/26 06:18, Zhen Ni wrote:
>> This patch series adds process and memory cgroup filtering support to
>> page_owner. Following the previous series that introduced print_mode and
>> NUMA node filters:
>> https://lore.kernel.org/linux-mm/20260707115411.1714314-1-zhen.ni@easystack.cn/
>>
>> This series adds filtering capabilities to page_owner, allowing users to
>> filter output by specific processes and memory cgroups. Users can now
>> filter page_owner output by PID, TGID, COMM (with wildcard support), and
>> memory cgroup path. This makes page_owner debugging more focused and
>> efficient for tracking memory allocations in specific contexts.
>
> I wonder about the usefulness of all the new filters. In my experience
> page_owner is useful to find a kernel memory leak code, and for that the
> stacktraces are most useful. Dealing with things like pid/tgid/comm/cgroups
> sounds more like your aim is to profile and optimize particular userspace to
> use less kernel memory? In that case, isn't it rather the area of memory
> allocation profiling (or maybe tracing with bpf), not page_owner?
>
> Moreover, tracing or bpf can already do such kind of filtering and AFAIK
> ftrace filters for tracepoints are nice and generic, while this is adding a
> bunch of custom parsing and filtering. So that makes me somewhat sceptical.
>
Thanks for the review, and the scepticism is fair - let me first
clarify where I agree with you, then explain the niche I think these
filters fill.
In memory usage source analysis and memory leak analysis, I believe
page_owner has its own unique niche:
1. Nearly all historical allocation records are queryable. Dynamic
tracing tools (bpf, ftrace) cannot do this.
2. The full allocation stack is recorded via stackdepot. Memory
allocation profiling as a code-tagging technique cannot do this.
3. Zero extra usage cost (works as long as page_owner is enabled) and
a low barrier to entry (one echo line versus writing a bpf
program).
On the pain points that motivated the series. On production machines
with large memory configurations (e.g., 250GB+):
1. Collecting page_owner information takes minutes to tens of minutes.
2. The output is several gigabytes to over 10GB.
That makes the raw output nearly unreadable and forces post-processing
with tools/mm/page_owner_sort.c, adding further workload. The root
causes are:
1. The PFN scan itself - unavoidable, it is the price of page_owner's
core function of covering every page.
2. Printing every stack for every page - this dominates the cost and
is avoidable. stackdepot already deduplicates stacks and keeps a
refcount per unique stack; page_owner then re-prints the same stack
once per page, and page_owner_sort deduplicates it all over again
in userspace.
The filters target exactly this waste: they keep page_owner focused on
the user's area of interest instead of paying the full print cost.
On the overlap with dynamic tracing and allocation profiling: the
features do look similar, but the usage scenarios differ. page_owner
is not enabled by default on production systems, so most developers
rightly reach for the lighter-weight tools first - dynamic tracing or
allocation profiling. But when page_owner is already enabled, or the
lighter-weight tools cannot solve (or cannot conveniently solve) the
problem and enabling page_owner is an option, the advantages above
kick in: the historical snapshot is filterable in place, and the
filtered output is small enough to read directly.
So I see the filters as completing page_owner for the scenarios where
it is the right tool, rather than competing with the profiling and
tracing tooling.
>> Targeted filtering provides significant performance benefits on large memory
>> servers by reducing both execution time and output size. By filtering at the
>> kernel level before reading, only relevant page allocations are processed,
>> dramatically reducing the amount of data that needs to be handled in userspace.
>>
Thanks,
Zhen Ni
^ permalink raw reply [flat|nested] 20+ messages in thread* Re: [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering
2026-09-07 4:08 ` zhen.ni
@ 2026-09-07 15:21 ` Vlastimil Babka (SUSE)
2026-09-08 2:41 ` zhen.ni
0 siblings, 1 reply; 20+ messages in thread
From: Vlastimil Babka (SUSE) @ 2026-09-07 15:21 UTC (permalink / raw)
To: zhen.ni, Andrew Morton
Cc: David Hildenbrand, Lorenzo Stoakes, Liam R . Howlett,
Mike Rapoport, Suren Baghdasaryan, Michal Hocko, Jonathan Corbet,
Shuah Khan, Randy Dunlap, Brendan Jackman, Johannes Weiner,
Zi Yan, linux-mm, linux-doc, linux-kernel, Steven Rostedt
On 9/7/26 06:08, zhen.ni wrote:
>
>
> 在 2026/9/4 16:25, Vlastimil Babka (SUSE) 写道:
>> On 9/3/26 06:18, Zhen Ni wrote:
>>> This patch series adds process and memory cgroup filtering support to
>>> page_owner. Following the previous series that introduced print_mode and
>>> NUMA node filters:
>>> https://lore.kernel.org/linux-mm/20260707115411.1714314-1-zhen.ni@easystack.cn/
>>>
>>> This series adds filtering capabilities to page_owner, allowing users to
>>> filter output by specific processes and memory cgroups. Users can now
>>> filter page_owner output by PID, TGID, COMM (with wildcard support), and
>>> memory cgroup path. This makes page_owner debugging more focused and
>>> efficient for tracking memory allocations in specific contexts.
>>
>> I wonder about the usefulness of all the new filters. In my experience
>> page_owner is useful to find a kernel memory leak code, and for that the
>> stacktraces are most useful. Dealing with things like pid/tgid/comm/cgroups
>> sounds more like your aim is to profile and optimize particular userspace to
>> use less kernel memory? In that case, isn't it rather the area of memory
>> allocation profiling (or maybe tracing with bpf), not page_owner?
>>
>> Moreover, tracing or bpf can already do such kind of filtering and AFAIK
>> ftrace filters for tracepoints are nice and generic, while this is adding a
>> bunch of custom parsing and filtering. So that makes me somewhat sceptical.
>>
>
> Thanks for the review, and the scepticism is fair - let me first
> clarify where I agree with you, then explain the niche I think these
> filters fill.
Sorry but your whole reply reads like a LLM slop. It took a lot of mental
effor to actually try and engage with it.
> In memory usage source analysis and memory leak analysis, I believe
> page_owner has its own unique niche:
>
> 1. Nearly all historical allocation records are queryable. Dynamic
> tracing tools (bpf, ftrace) cannot do this.
This seems totally the opposite. page_owner gives you the current allocation
snapshot. Tracing can provide the full historical record of allocation and
freeing, which can be also postprocessed for snapshots.
> 2. The full allocation stack is recorded via stackdepot. Memory
> allocation profiling as a code-tagging technique cannot do this.
I think there's some extension (or plan for it), Suren would know better.
> 3. Zero extra usage cost (works as long as page_owner is enabled) and
> a low barrier to entry (one echo line versus writing a bpf
> program).
>
> On the pain points that motivated the series. On production machines
> with large memory configurations (e.g., 250GB+):
>
> 1. Collecting page_owner information takes minutes to tens of minutes.
> 2. The output is several gigabytes to over 10GB.
>
> That makes the raw output nearly unreadable and forces post-processing
> with tools/mm/page_owner_sort.c, adding further workload. The root
> causes are:
the "workload" is just cpu time though
> 1. The PFN scan itself - unavoidable, it is the price of page_owner's
> core function of covering every page.
> 2. Printing every stack for every page - this dominates the cost and
> is avoidable. stackdepot already deduplicates stacks and keeps a
> refcount per unique stack; page_owner then re-prints the same stack
> once per page, and page_owner_sort deduplicates it all over again
> in userspace.
Since it's avoidable, why mention it at all? Oh I know, LLM slop.
> The filters target exactly this waste: they keep page_owner focused on
> the user's area of interest instead of paying the full print cost.
>
> On the overlap with dynamic tracing and allocation profiling: the
> features do look similar, but the usage scenarios differ. page_owner
> is not enabled by default on production systems, so most developers
> rightly reach for the lighter-weight tools first - dynamic tracing or
> allocation profiling.
Good! That's an argument against.
> But when page_owner is already enabled, or the
> lighter-weight tools cannot solve (or cannot conveniently solve) the
Adding custom filtering code to kernel vs user convenience is a trade-off.
> problem and enabling page_owner is an option, the advantages above
> kick in: the historical snapshot is filterable in place, and the
> filtered output is small enough to read directly.
>
> So I see the filters as completing page_owner for the scenarios where
> it is the right tool, rather than competing with the profiling and
> tracing tooling.
I'm not convinced it's worth it.
>>> Targeted filtering provides significant performance benefits on large memory
>>> servers by reducing both execution time and output size. By filtering at the
>>> kernel level before reading, only relevant page allocations are processed,
>>> dramatically reducing the amount of data that needs to be handled in userspace.
>>>
>
> Thanks,
> Zhen Ni
^ permalink raw reply [flat|nested] 20+ messages in thread
* Re: [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering
2026-09-07 15:21 ` Vlastimil Babka (SUSE)
@ 2026-09-08 2:41 ` zhen.ni
2026-09-08 6:24 ` Weijie Yuan
0 siblings, 1 reply; 20+ messages in thread
From: zhen.ni @ 2026-09-08 2:41 UTC (permalink / raw)
To: Vlastimil Babka (SUSE), Andrew Morton
Cc: David Hildenbrand, Lorenzo Stoakes, Liam R . Howlett,
Mike Rapoport, Suren Baghdasaryan, Michal Hocko, Jonathan Corbet,
Shuah Khan, Randy Dunlap, Brendan Jackman, Johannes Weiner,
Zi Yan, linux-mm, linux-doc, linux-kernel, Steven Rostedt
在 2026/9/7 23:21, Vlastimil Babka (SUSE) 写道:
> On 9/7/26 06:08, zhen.ni wrote:
>>
>>
>> 在 2026/9/4 16:25, Vlastimil Babka (SUSE) 写道:
>>> On 9/3/26 06:18, Zhen Ni wrote:
>>>> This patch series adds process and memory cgroup filtering support to
>>>> page_owner. Following the previous series that introduced print_mode and
>>>> NUMA node filters:
>>>> https://lore.kernel.org/linux-mm/20260707115411.1714314-1-zhen.ni@easystack.cn/
>>>>
>>>> This series adds filtering capabilities to page_owner, allowing users to
>>>> filter output by specific processes and memory cgroups. Users can now
>>>> filter page_owner output by PID, TGID, COMM (with wildcard support), and
>>>> memory cgroup path. This makes page_owner debugging more focused and
>>>> efficient for tracking memory allocations in specific contexts.
>>>
>>> I wonder about the usefulness of all the new filters. In my experience
>>> page_owner is useful to find a kernel memory leak code, and for that the
>>> stacktraces are most useful. Dealing with things like pid/tgid/comm/cgroups
>>> sounds more like your aim is to profile and optimize particular userspace to
>>> use less kernel memory? In that case, isn't it rather the area of memory
>>> allocation profiling (or maybe tracing with bpf), not page_owner?
>>>
>>> Moreover, tracing or bpf can already do such kind of filtering and AFAIK
>>> ftrace filters for tracepoints are nice and generic, while this is adding a
>>> bunch of custom parsing and filtering. So that makes me somewhat sceptical.
>>>
>>
>> Thanks for the review, and the scepticism is fair - let me first
>> clarify where I agree with you, then explain the niche I think these
>> filters fill.
>
> Sorry but your whole reply reads like a LLM slop. It took a lot of mental
> effor to actually try and engage with it.
>
This reply was written by me, and its viewpoints are not simply copied
directly from an LLM. The LLM only assists with the wording.
>> In memory usage source analysis and memory leak analysis, I believe
>> page_owner has its own unique niche:
>>
>> 1. Nearly all historical allocation records are queryable. Dynamic
>> tracing tools (bpf, ftrace) cannot do this.
>
> This seems totally the opposite. page_owner gives you the current allocation
> snapshot. Tracing can provide the full historical record of allocation and
> freeing, which can be also postprocessed for snapshots.
>
Here you completely misunderstood my meaning, perhaps I did not express
it clearly.
The biggest difference between page_owner and dynamic tracing tools is
that page_owner traverses all currently allocated pages using PFN;
whereas dynamic tracing tools can only view pages after the observation
point is established. The pages collected by page_owner include those
from relatively early in the kernel — as long as they are still
occupying memory, they can be detected. This is what I mean by
"historical allocation records."
>> 2. The full allocation stack is recorded via stackdepot. Memory
>> allocation profiling as a code-tagging technique cannot do this.
>
> I think there's some extension (or plan for it), Suren would know better.
>
I have previously researched the code and documentation of Memory
allocation profiling, so this viewpoint is not directly copied from an
LLM. The reason I think it cannot record stacks is: its design goal is
to be lightweight and usable in production environments, therefore it
chose the code-tagging approach. If it recorded stacks, it would already
deviate from its original design goal.
>> 3. Zero extra usage cost (works as long as page_owner is enabled) and
>> a low barrier to entry (one echo line versus writing a bpf
>> program).
>>
>> On the pain points that motivated the series. On production machines
>> with large memory configurations (e.g., 250GB+):
>>
>> 1. Collecting page_owner information takes minutes to tens of minutes.
>> 2. The output is several gigabytes to over 10GB.
>>
>> That makes the raw output nearly unreadable and forces post-processing
>> with tools/mm/page_owner_sort.c, adding further workload. The root
>> causes are:
>
> the "workload" is just cpu time though
This workload is not just CPU — printing the stack takes up the vast
majority of it.
>
>> 1. The PFN scan itself - unavoidable, it is the price of page_owner's
>> core function of covering every page.
>> 2. Printing every stack for every page - this dominates the cost and
>> is avoidable. stackdepot already deduplicates stacks and keeps a
>> refcount per unique stack; page_owner then re-prints the same stack
>> once per page, and page_owner_sort deduplicates it all over again
>> in userspace.
>
> Since it's avoidable, why mention it at all? Oh I know, LLM slop.
>
I feel like you haven't really understood what I'm trying to express.
>> The filters target exactly this waste: they keep page_owner focused on
>> the user's area of interest instead of paying the full print cost.
>>
>> On the overlap with dynamic tracing and allocation profiling: the
>> features do look similar, but the usage scenarios differ. page_owner
>> is not enabled by default on production systems, so most developers
>> rightly reach for the lighter-weight tools first - dynamic tracing or
>> allocation profiling.
>
> Good! That's an argument against.
>
>> But when page_owner is already enabled, or the
>> lighter-weight tools cannot solve (or cannot conveniently solve) the
>
> Adding custom filtering code to kernel vs user convenience is a trade-off.
>
>> problem and enabling page_owner is an option, the advantages above
>> kick in: the historical snapshot is filterable in place, and the
>> filtered output is small enough to read directly.
>>
>> So I see the filters as completing page_owner for the scenarios where
>> it is the right tool, rather than competing with the profiling and
>> tracing tooling.
>
> I'm not convinced it's worth it.
You may keep your opinion. I think you haven't carefully read my reply,
or you simply assumed this is just an automated reply from an LLM.
Perhaps you haven't actually used page_owner on a server with very large
memory to investigate memory usage or leak issues (for example, the NIC
ring buffer usage problem cannot be solved with dynamic tracing tools).
Once you've used it, you'll know how much you need a filter.
Thanks,
Zhen
^ permalink raw reply [flat|nested] 20+ messages in thread
* Re: [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering
2026-09-08 2:41 ` zhen.ni
@ 2026-09-08 6:24 ` Weijie Yuan
2026-09-08 6:40 ` zhen.ni
0 siblings, 1 reply; 20+ messages in thread
From: Weijie Yuan @ 2026-09-08 6:24 UTC (permalink / raw)
To: zhen.ni
Cc: Vlastimil Babka (SUSE),
Andrew Morton, David Hildenbrand, Lorenzo Stoakes,
Liam R . Howlett, Mike Rapoport, Suren Baghdasaryan,
Michal Hocko, Jonathan Corbet, Shuah Khan, Randy Dunlap,
Brendan Jackman, Johannes Weiner, Zi Yan, linux-mm, linux-doc,
linux-kernel, Steven Rostedt
From an outsider..
On Tue, Sep 08, 2026 at 10:41:23AM +0800, zhen.ni wrote:
> >
> > Sorry but your whole reply reads like a LLM slop. It took a lot of
> > mental effor to actually try and engage with it.
> >
>
> This reply was written by me, and its viewpoints are not simply copied
> directly from an LLM. The LLM only assists with the wording.
So that's probably the reason. Your previous replies was indeed very
much like the output of an LLM. Even if the idea is your own, after
being "polished" by LLM, it's hard to read. I know English is not our
first language, and it's not that easy to speak like a native speaker.
But may I ask you did you read it before sending them out? I really
found this bunch of overly formal content a little painful to read.
So I suspect the LLM may have taken too much liberty in polishing your
text, to the point that your replies no longer read like something a
person would naturally write.
The "Summary and Plan" in your reply in v1, both in its formatting and
wording, looks quite similar to LLM-generated text to me. So I can
understand why Lorenzo asked whether you had used an LLM. Come on, we
all know what LLM writing looks like.
That said, I do not know exactly how you used the LLM on your original
text, so please forgive me if I have this wrong.
Thanks.
^ permalink raw reply [flat|nested] 20+ messages in thread
* Re: [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering
2026-09-08 6:24 ` Weijie Yuan
@ 2026-09-08 6:40 ` zhen.ni
2026-09-08 7:04 ` Weijie Yuan
0 siblings, 1 reply; 20+ messages in thread
From: zhen.ni @ 2026-09-08 6:40 UTC (permalink / raw)
To: Weijie Yuan
Cc: Vlastimil Babka (SUSE),
Andrew Morton, David Hildenbrand, Lorenzo Stoakes,
Liam R . Howlett, Mike Rapoport, Suren Baghdasaryan,
Michal Hocko, Jonathan Corbet, Shuah Khan, Randy Dunlap,
Brendan Jackman, Johannes Weiner, Zi Yan, linux-mm, linux-doc,
linux-kernel, Steven Rostedt
在 2026/9/8 14:24, Weijie Yuan 写道:
> From an outsider..
>
> On Tue, Sep 08, 2026 at 10:41:23AM +0800, zhen.ni wrote:
>>>
>>> Sorry but your whole reply reads like a LLM slop. It took a lot of
>>> mental effor to actually try and engage with it.
>>>
>>
>> This reply was written by me, and its viewpoints are not simply copied
>> directly from an LLM. The LLM only assists with the wording.
>
> So that's probably the reason. Your previous replies was indeed very
> much like the output of an LLM. Even if the idea is your own, after
> being "polished" by LLM, it's hard to read. I know English is not our
> first language, and it's not that easy to speak like a native speaker.
> But may I ask you did you read it before sending them out? I really
> found this bunch of overly formal content a little painful to read.
>
> So I suspect the LLM may have taken too much liberty in polishing your
> text, to the point that your replies no longer read like something a
> person would naturally write.
>
> The "Summary and Plan" in your reply in v1, both in its formatting and
> wording, looks quite similar to LLM-generated text to me. So I can
> understand why Lorenzo asked whether you had used an LLM. Come on, we
> all know what LLM writing looks like.
>
> That said, I do not know exactly how you used the LLM on your original
> text, so please forgive me if I have this wrong.
>
> Thanks.
>
>
I really can't distinguish the LLM flavor that clearly in English. I'm
sorry for causing some trouble.
My normal reply process is to first describe my ideas, then have the LLM
check for logic and grammar issues, and do a round of polishing. I check
the final draft once more to see if it has deviated from my ideas, and
make corresponding revisions.
Thanks,
Zhen
^ permalink raw reply [flat|nested] 20+ messages in thread
* Re: [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering
2026-09-08 6:40 ` zhen.ni
@ 2026-09-08 7:04 ` Weijie Yuan
2026-09-08 9:13 ` Lorenzo Stoakes (ARM)
0 siblings, 1 reply; 20+ messages in thread
From: Weijie Yuan @ 2026-09-08 7:04 UTC (permalink / raw)
To: zhen.ni
Cc: Vlastimil Babka (SUSE),
Andrew Morton, David Hildenbrand, Lorenzo Stoakes,
Liam R . Howlett, Mike Rapoport, Suren Baghdasaryan,
Michal Hocko, Jonathan Corbet, Shuah Khan, Randy Dunlap,
Brendan Jackman, Johannes Weiner, Zi Yan, linux-mm, linux-doc,
linux-kernel, Steven Rostedt
On Tue, Sep 08, 2026 at 02:40:10PM +0800, zhen.ni wrote:
>
>
> 在 2026/9/8 14:24, Weijie Yuan 写道:
> > From an outsider..
> >
> > On Tue, Sep 08, 2026 at 10:41:23AM +0800, zhen.ni wrote:
> > > >
> > > > Sorry but your whole reply reads like a LLM slop. It took a lot of
> > > > mental effor to actually try and engage with it.
> > > >
> > >
> > > This reply was written by me, and its viewpoints are not simply copied
> > > directly from an LLM. The LLM only assists with the wording.
> >
> > So that's probably the reason. Your previous replies was indeed very
> > much like the output of an LLM. Even if the idea is your own, after
> > being "polished" by LLM, it's hard to read. I know English is not our
> > first language, and it's not that easy to speak like a native speaker.
> > But may I ask you did you read it before sending them out? I really
> > found this bunch of overly formal content a little painful to read.
> >
> > So I suspect the LLM may have taken too much liberty in polishing your
> > text, to the point that your replies no longer read like something a
> > person would naturally write.
> >
> > The "Summary and Plan" in your reply in v1, both in its formatting and
> > wording, looks quite similar to LLM-generated text to me. So I can
> > understand why Lorenzo asked whether you had used an LLM. Come on, we
> > all know what LLM writing looks like.
> >
> > That said, I do not know exactly how you used the LLM on your original
> > text, so please forgive me if I have this wrong.
> >
> > Thanks.
> >
> I really can't distinguish the LLM flavor that clearly in English.
Yes, sometimes I can not as well. But I guess the native ones can.
So.. ;-)
> I'm sorry for causing some trouble.
That's fine. No need to say sorry.
I know and understand that we non-native speakers would rather not make
mistakes over these little language detail issues. But over-polishing
can end up having the opposite effect. Keeping some of your natural
writing style may actually be what the community prefers, especially
these AI days.
> My normal reply process is to first describe my ideas, then have the
> LLM check for logic and grammar issues, and do a round of polishing. I
> check the final draft once more to see if it has deviated from my
> ideas, and make corresponding revisions.
Sigh, so it is hard to know where to draw the line. Write more and learn
more about "human-like" writing, perhaps. :)
Thanks!
^ permalink raw reply [flat|nested] 20+ messages in thread
* Re: [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering
2026-09-08 7:04 ` Weijie Yuan
@ 2026-09-08 9:13 ` Lorenzo Stoakes (ARM)
2026-09-08 11:18 ` zhen.ni
2026-09-08 17:37 ` Weijie Yuan
0 siblings, 2 replies; 20+ messages in thread
From: Lorenzo Stoakes (ARM) @ 2026-09-08 9:13 UTC (permalink / raw)
To: Weijie Yuan
Cc: zhen.ni, Vlastimil Babka (SUSE),
Andrew Morton, David Hildenbrand, Liam R . Howlett,
Mike Rapoport, Suren Baghdasaryan, Michal Hocko, Jonathan Corbet,
Shuah Khan, Randy Dunlap, Brendan Jackman, Johannes Weiner,
Zi Yan, linux-mm, linux-doc, linux-kernel, Steven Rostedt
On Tue, Sep 08, 2026 at 03:04:02PM +0800, Weijie Yuan wrote:
> On Tue, Sep 08, 2026 at 02:40:10PM +0800, zhen.ni wrote:
> >
> >
> > 在 2026/9/8 14:24, Weijie Yuan 写道:
> > > From an outsider..
> > >
> > > On Tue, Sep 08, 2026 at 10:41:23AM +0800, zhen.ni wrote:
> > > > >
> > > > > Sorry but your whole reply reads like a LLM slop. It took a lot of
> > > > > mental effor to actually try and engage with it.
> > > > >
> > > >
> > > > This reply was written by me, and its viewpoints are not simply copied
> > > > directly from an LLM. The LLM only assists with the wording.
> > >
> > > So that's probably the reason. Your previous replies was indeed very
> > > much like the output of an LLM. Even if the idea is your own, after
> > > being "polished" by LLM, it's hard to read. I know English is not our
> > > first language, and it's not that easy to speak like a native speaker.
> > > But may I ask you did you read it before sending them out? I really
> > > found this bunch of overly formal content a little painful to read.
> > >
> > > So I suspect the LLM may have taken too much liberty in polishing your
> > > text, to the point that your replies no longer read like something a
> > > person would naturally write.
> > >
> > > The "Summary and Plan" in your reply in v1, both in its formatting and
> > > wording, looks quite similar to LLM-generated text to me. So I can
> > > understand why Lorenzo asked whether you had used an LLM. Come on, we
> > > all know what LLM writing looks like.
> > >
> > > That said, I do not know exactly how you used the LLM on your original
> > > text, so please forgive me if I have this wrong.
> > >
> > > Thanks.
> > >
> > I really can't distinguish the LLM flavor that clearly in English.
>
> Yes, sometimes I can not as well. But I guess the native ones can.
> So.. ;-)
>
> > I'm sorry for causing some trouble.
>
> That's fine. No need to say sorry.
>
> I know and understand that we non-native speakers would rather not make
> mistakes over these little language detail issues. But over-polishing
> can end up having the opposite effect. Keeping some of your natural
> writing style may actually be what the community prefers, especially
> these AI days.
>
> > My normal reply process is to first describe my ideas, then have the
> > LLM check for logic and grammar issues, and do a round of polishing. I
> > check the final draft once more to see if it has deviated from my
> > ideas, and make corresponding revisions.
>
> Sigh, so it is hard to know where to draw the line. Write more and learn
> more about "human-like" writing, perhaps. :)
>
> Thanks!
Thanks Weijie appreciate your input :) and it's good to get a perspective from a
non-native speaker on this!
Generally I empathise with LLM usage for helping non-native speakers with
English, that's a great use of it, so I don't object to it _in general_ BUT as
Weijie points out there's better ways of using and worse ways of using it.
You have to ensure that your meaning is transmitted properly without it
ultimately becoming essentially a conversation between a reviewer and an LLM.
So the technial discussion you are engaging in MUST be your own.
As for the 'summary' emails - please don't send them at all.
I feel like often they're used to generate a new prompt for the LLM and it
really ends up being 'workslopping', that is, making reviewers do more work
while the LLM takes care of things for you.
And that crosses the line really from 'aid to English' into it being a problem.
Instead, ENGAGE IN CONVERSATION with the reviewer, in line.
So if a review says:
"Please make this function do X, Y, Z".
Don't put something in a summary email or anything like that. REPLY to them,
quoting the request and respond. Like:
> Please make this function do X, Y, Z.
OK makes sense about X, Y is a bit tricker because of ... and Z is
impossible because ....
For instance.
That way it is human-to-human interaction at all times, with maybe the LLM
helping with translation along the way.
As for the review - Vlastimil has raised legitimate technical points so I would
engage with those directly.
It's very reasonable for him to conclude an LLM was used for more than
translation, certainly the rather ludicrious 'document the world in the commit
message' approach looks inhuman.
So engage on the technical points without summary emails, and please put
documentation in a documentation file :)
Also I think it's reasonable now for you to use an:
Assisted-by: LLM
Tag on this series.
--
Cheers, Lorenzo
^ permalink raw reply [flat|nested] 20+ messages in thread
* Re: [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering
2026-09-08 9:13 ` Lorenzo Stoakes (ARM)
@ 2026-09-08 11:18 ` zhen.ni
2026-09-08 17:37 ` Weijie Yuan
1 sibling, 0 replies; 20+ messages in thread
From: zhen.ni @ 2026-09-08 11:18 UTC (permalink / raw)
To: Lorenzo Stoakes (ARM), Weijie Yuan
Cc: Vlastimil Babka (SUSE),
Andrew Morton, David Hildenbrand, Liam R . Howlett,
Mike Rapoport, Suren Baghdasaryan, Michal Hocko, Jonathan Corbet,
Shuah Khan, Randy Dunlap, Brendan Jackman, Johannes Weiner,
Zi Yan, linux-mm, linux-doc, linux-kernel, Steven Rostedt
在 2026/9/8 17:13, Lorenzo Stoakes (ARM) 写道:
> On Tue, Sep 08, 2026 at 03:04:02PM +0800, Weijie Yuan wrote:
>> On Tue, Sep 08, 2026 at 02:40:10PM +0800, zhen.ni wrote:
>>>
>>>
>>> 在 2026/9/8 14:24, Weijie Yuan 写道:
>>>> From an outsider..
>>>>
>>>> On Tue, Sep 08, 2026 at 10:41:23AM +0800, zhen.ni wrote:
>>>>>>
>>>>>> Sorry but your whole reply reads like a LLM slop. It took a lot of
>>>>>> mental effor to actually try and engage with it.
>>>>>>
>>>>>
>>>>> This reply was written by me, and its viewpoints are not simply copied
>>>>> directly from an LLM. The LLM only assists with the wording.
>>>>
>>>> So that's probably the reason. Your previous replies was indeed very
>>>> much like the output of an LLM. Even if the idea is your own, after
>>>> being "polished" by LLM, it's hard to read. I know English is not our
>>>> first language, and it's not that easy to speak like a native speaker.
>>>> But may I ask you did you read it before sending them out? I really
>>>> found this bunch of overly formal content a little painful to read.
>>>>
>>>> So I suspect the LLM may have taken too much liberty in polishing your
>>>> text, to the point that your replies no longer read like something a
>>>> person would naturally write.
>>>>
>>>> The "Summary and Plan" in your reply in v1, both in its formatting and
>>>> wording, looks quite similar to LLM-generated text to me. So I can
>>>> understand why Lorenzo asked whether you had used an LLM. Come on, we
>>>> all know what LLM writing looks like.
>>>>
>>>> That said, I do not know exactly how you used the LLM on your original
>>>> text, so please forgive me if I have this wrong.
>>>>
>>>> Thanks.
>>>>
>>> I really can't distinguish the LLM flavor that clearly in English.
>>
>> Yes, sometimes I can not as well. But I guess the native ones can.
>> So.. ;-)
>>
>>> I'm sorry for causing some trouble.
>>
>> That's fine. No need to say sorry.
>>
>> I know and understand that we non-native speakers would rather not make
>> mistakes over these little language detail issues. But over-polishing
>> can end up having the opposite effect. Keeping some of your natural
>> writing style may actually be what the community prefers, especially
>> these AI days.
>>
>>> My normal reply process is to first describe my ideas, then have the
>>> LLM check for logic and grammar issues, and do a round of polishing. I
>>> check the final draft once more to see if it has deviated from my
>>> ideas, and make corresponding revisions.
>>
>> Sigh, so it is hard to know where to draw the line. Write more and learn
>> more about "human-like" writing, perhaps. :)
>>
>> Thanks!
>
> Thanks Weijie appreciate your input :) and it's good to get a perspective from a
> non-native speaker on this!
>
> Generally I empathise with LLM usage for helping non-native speakers with
> English, that's a great use of it, so I don't object to it _in general_ BUT as
> Weijie points out there's better ways of using and worse ways of using it.
>
> You have to ensure that your meaning is transmitted properly without it
> ultimately becoming essentially a conversation between a reviewer and an LLM.
>
> So the technial discussion you are engaging in MUST be your own.
>
> As for the 'summary' emails - please don't send them at all.
>
> I feel like often they're used to generate a new prompt for the LLM and it
> really ends up being 'workslopping', that is, making reviewers do more work
> while the LLM takes care of things for you.
>
> And that crosses the line really from 'aid to English' into it being a problem.
>
> Instead, ENGAGE IN CONVERSATION with the reviewer, in line.
>
> So if a review says:
>
> "Please make this function do X, Y, Z".
>
> Don't put something in a summary email or anything like that. REPLY to them,
> quoting the request and respond. Like:
>
> > Please make this function do X, Y, Z.
>
> OK makes sense about X, Y is a bit tricker because of ... and Z is
> impossible because ....
>
> For instance.
>
> That way it is human-to-human interaction at all times, with maybe the LLM
> helping with translation along the way.
>
Noted with thanks.
> As for the review - Vlastimil has raised legitimate technical points so I would
> engage with those directly.
>
> It's very reasonable for him to conclude an LLM was used for more than
> translation, certainly the rather ludicrious 'document the world in the commit
> message' approach looks inhuman.
I'm not sure which patch you are referring to, as its commit message
appears to be unusually long (seemingly stuffed with massive LLM-
generated information).
For my 8 patches, the commit messages are not generated by an LLM, and
they are written very concisely.
If you are referring to the cover letter being long, I would like to
briefly explain. The cover letter itself isn't that long; the extra
length comes from the test programs and test results I attached. I felt
it was necessary to let the reviewers know exactly what tests I
performed and how effective they were, so as to ease the review burden.
However, since this information cannot be placed within the individual
patches, I had to put it in the cover letter for now. Please note that
this content will not be included in the final commit messages later on.
>
> So engage on the technical points without summary emails, and please put
> documentation in a documentation file :)
>
> Also I think it's reasonable now for you to use an:
>
> Assisted-by: LLM
>
> Tag on this series.
>
I resorted to an LLM only to structure the language in my reply email —
purely for efficiency, to help expedite the review process. For the
patch series itself, however, I had ample time to prepare, and both the
code and the documentation were written without any LLM involvement.
Therefore, I believe the Tag in question is not really suitable here.
Thanks,
Zhen
^ permalink raw reply [flat|nested] 20+ messages in thread
* Re: [PATCH v2 0/8] mm/page_owner: Add PID/TGID/COMM and cgroup filtering
2026-09-08 9:13 ` Lorenzo Stoakes (ARM)
2026-09-08 11:18 ` zhen.ni
@ 2026-09-08 17:37 ` Weijie Yuan
1 sibling, 0 replies; 20+ messages in thread
From: Weijie Yuan @ 2026-09-08 17:37 UTC (permalink / raw)
To: Lorenzo Stoakes (ARM)
Cc: zhen.ni, Vlastimil Babka (SUSE),
Andrew Morton, David Hildenbrand, Liam R . Howlett,
Mike Rapoport, Suren Baghdasaryan, Michal Hocko, Jonathan Corbet,
Shuah Khan, Randy Dunlap, Brendan Jackman, Johannes Weiner,
Zi Yan, linux-mm, linux-doc, linux-kernel, Steven Rostedt
On Tue, Sep 08, 2026 at 10:13:22AM +0100, Lorenzo Stoakes (ARM) wrote:
> On Tue, Sep 08, 2026 at 03:04:02PM +0800, Weijie Yuan wrote:
> > On Tue, Sep 08, 2026 at 02:40:10PM +0800, zhen.ni wrote:
> > >
> > >
> > > 在 2026/9/8 14:24, Weijie Yuan 写道:
> > > > From an outsider..
> > > >
> > > > On Tue, Sep 08, 2026 at 10:41:23AM +0800, zhen.ni wrote:
> > > > > >
> > > > > > Sorry but your whole reply reads like a LLM slop. It took a lot of
> > > > > > mental effor to actually try and engage with it.
> > > > > >
> > > > >
> > > > > This reply was written by me, and its viewpoints are not simply copied
> > > > > directly from an LLM. The LLM only assists with the wording.
> > > >
> > > > So that's probably the reason. Your previous replies was indeed very
> > > > much like the output of an LLM. Even if the idea is your own, after
> > > > being "polished" by LLM, it's hard to read. I know English is not our
> > > > first language, and it's not that easy to speak like a native speaker.
> > > > But may I ask you did you read it before sending them out? I really
> > > > found this bunch of overly formal content a little painful to read.
> > > >
> > > > So I suspect the LLM may have taken too much liberty in polishing your
> > > > text, to the point that your replies no longer read like something a
> > > > person would naturally write.
> > > >
> > > > The "Summary and Plan" in your reply in v1, both in its formatting and
> > > > wording, looks quite similar to LLM-generated text to me. So I can
> > > > understand why Lorenzo asked whether you had used an LLM. Come on, we
> > > > all know what LLM writing looks like.
> > > >
> > > > That said, I do not know exactly how you used the LLM on your original
> > > > text, so please forgive me if I have this wrong.
> > > >
> > > > Thanks.
> > > >
> > > I really can't distinguish the LLM flavor that clearly in English.
> >
> > Yes, sometimes I can not as well. But I guess the native ones can.
> > So.. ;-)
> >
> > > I'm sorry for causing some trouble.
> >
> > That's fine. No need to say sorry.
> >
> > I know and understand that we non-native speakers would rather not make
> > mistakes over these little language detail issues. But over-polishing
> > can end up having the opposite effect. Keeping some of your natural
> > writing style may actually be what the community prefers, especially
> > these AI days.
> >
> > > My normal reply process is to first describe my ideas, then have the
> > > LLM check for logic and grammar issues, and do a round of polishing. I
> > > check the final draft once more to see if it has deviated from my
> > > ideas, and make corresponding revisions.
> >
> > Sigh, so it is hard to know where to draw the line. Write more and learn
> > more about "human-like" writing, perhaps. :)
> >
> > Thanks!
>
> Thanks Weijie appreciate your input :) and it's good to get a perspective from a
> non-native speaker on this!
Thanks for your kind words!
> Generally I empathise with LLM usage for helping non-native speakers with
> English, that's a great use of it, so I don't object to it _in general_ BUT as
> Weijie points out there's better ways of using and worse ways of using it.
Honestly, my English isn't that great either. Sometimes, I find I cannot
express myself very well in English. Then, to avoid unnecessary
misunderstandings caused by my clumsy wording, I may ask an LLM how to
phrase something more naturally in English. That way, we don't end up
wasting everyone's time over wording issues.
So my suggestion would be to use an LLM only to polish small bits of
text, say a few words or a sentence, rather than feeding it a large
chunk at once. I've found that when I give agents too much text, it
tends to polish too aggressively and start expanding things.
(Only for expression issues)
It may be a bit like sending a patch series: if the series is small and
the changes are limited, it is usually fairly easy for reviewers to
follow. But once a series grows to a dozen or twenty patches, there is
much more information to process, and reviewers are more likely to miss
something. I actually made that mistake myself recently.
So once it gives you an expanded version, and especially if you're not a
native speaker and aren't very sensitive to the nuances of English, it's
easy to think, "Well, it says a bit more, but it also sounds more
complete and precise."
But I suspect that native speakers may naturally pick up on something
different: the expanded text can start to feel less like something a
real person would write. It can also become verbose enough that the
actual point gets buried, which becomes hard to catch for reviewers.
> You have to ensure that your meaning is transmitted properly without it
> ultimately becoming essentially a conversation between a reviewer and an LLM.
>
> So the technial discussion you are engaging in MUST be your own.
Pessimistically speaking, perhaps this is just becoming a new normal,
and a new problem, for open source projects: reviewers spend a lot of
time carefully writing out their feedback, only for all of it to end up
as prompts for the contributor's agent.
That perhaps makes Linus's point about trust even more relevant.
And..
> As for the 'summary' emails - please don't send them at all.
>
> I feel like often they're used to generate a new prompt for the LLM and it
> really ends up being 'workslopping', that is, making reviewers do more work
> while the LLM takes care of things for you.
>
> And that crosses the line really from 'aid to English' into it being a problem.
>
> Instead, ENGAGE IN CONVERSATION with the reviewer, in line.
>
> So if a review says:
>
> "Please make this function do X, Y, Z".
>
> Don't put something in a summary email or anything like that. REPLY to them,
> quoting the request and respond. Like:
>
> > Please make this function do X, Y, Z.
>
> OK makes sense about X, Y is a bit tricker because of ... and Z is
> impossible because ....
>
> For instance.
>
> That way it is human-to-human interaction at all times, with maybe the LLM
> helping with translation along the way.
I totally agree with the above. Moreover, I've noticed that some authors
of patches carrying an Assisted-by tag seem to do this quite often
lately: After reviewers reply to them, the next version will often
contain rather stiff, overly "summarized" descriptions of the
discussion.
Another example is when a patch changes just one line of documentation,
yet the commit message still goes out of its way to say something like,
"This is a documentation-only change and does not affect code
functionality." It is hard not to associate that with a typical LLM
habit: always trying to give some explicit reassurance or confirmation
that nothing else was affected. (I don't know if this is a requirement
for some subsystem, or if these LLM users are also non-English speakers)
I may be broadening the discussion a bit here (sorry for taking it
somewhat off topic), but I really like this passage from the Git
project's documentation:
| It's good manners to reply to each comment in the mailing list
| discussion instead of letting the next version of your patch be your
| only response. Tell the reviewer whether you plan to make the
| suggested change, keep the original, or pursue a different approach.
| This way reviewers can respond to your reasoning before you spend time
| preparing a version they may not agree with, and later do not need to
| inspect your v2 to figure out whether you implemented their comment or
| not.
(Okay, I admit I wrote part of that. ;-))
Of course, that is Git's culture, and it may not necessarily apply to
the kernel. Still, I think Patrick made a good point that this: "... it
encourages more social interactions between contributors."
I haven't been in this community for very long, but I've run into this a
few times recently as well: contributors will often just send a new
version without replying to comments on the previous one. Sometimes they
do put a response to you below the three-dash line in the new version,
though. (And sometimes I suspect that even the commentary part is also
LLM-generated.)
Submitting-patches already says something related to this, and I
happened to be wondering recently whether that part could be improved
and made a little more explicit. But that would also benefit from input
from people here with much more experience than I have.
Perhaps I could send an RFC and see what the experienced ones think.
Thanks.
^ permalink raw reply [flat|nested] 20+ messages in thread