* [PATCH 2/6] mm/page_owner: add bpf_iter target "page_owner"
2026-10-09 11:23 [PATCH 0/6] mm/page_owner: add bpf_iter and kfuncs Zhen Ni
@ 2026-10-09 11:23 ` Zhen Ni
2026-10-09 11:23 ` [PATCH 3/6] mm/page_owner: add open-coded page_owner iterator Zhen Ni
` (3 subsequent siblings)
4 siblings, 0 replies; 7+ messages in thread
From: Zhen Ni @ 2026-10-09 11:23 UTC (permalink / raw)
To: ast, daniel, andrii, eddyz87, memxor, martin.lau, song,
yonghong.song, jolsa, emil, ihor.solodrai, akpm, vbabka, surenb,
mhocko, brendan.jackman, hannes, ziy, shuah
Cc: bpf, linux-mm, linux-kselftest, linux-kernel, Zhen Ni
Register a bpf_iter target "page_owner" that walks the pages eligible
for page_owner output, with the same eligibility rules as the
/sys/kernel/debug/page_owner read path.
For each eligible page the attached BPF program receives the pfn, the
page and the page_owner record, and can freely filter and format output.
The po_snap record in the scan cursor is taken inside the page_ext
RCU window and handed to the BPF program as a copy, so no RCU window
is needed at consumption time. Torn reads remain possible if the
record is concurrently updated while it is copied, but this is the
same best-effort behavior as the read path has always had.
The target is registered from pageowner_init() when both
CONFIG_PAGE_OWNER and CONFIG_BPF_SYSCALL are enabled; a registration
failure only disables the bpf_iter target.
Userspace tools such as bpftrace can attach to this target like any
other bpf_iter. For example, to count pages owned by init (pid 1):
bpftrace -e 'iter:page_owner / ctx->po->pid == 1 / { @ = count(); }'
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
mm/Makefile | 3 ++
mm/page_owner.c | 33 +++---------
mm/page_owner.h | 48 ++++++++++++++++++
mm/page_owner_iter.c | 116 +++++++++++++++++++++++++++++++++++++++++++
4 files changed, 174 insertions(+), 26 deletions(-)
create mode 100644 mm/page_owner.h
create mode 100644 mm/page_owner_iter.c
diff --git a/mm/Makefile b/mm/Makefile
index e7245cb88c66..d7307b5a964c 100644
--- a/mm/Makefile
+++ b/mm/Makefile
@@ -115,6 +115,9 @@ obj-$(CONFIG_DEBUG_KMEMLEAK) += kmemleak.o
obj-$(CONFIG_DEBUG_RODATA_TEST) += rodata_test.o
obj-$(CONFIG_DEBUG_VM_PGTABLE) += debug_vm_pgtable.o
obj-$(CONFIG_PAGE_OWNER) += page_owner.o
+ifdef CONFIG_BPF_SYSCALL
+obj-$(CONFIG_PAGE_OWNER) += page_owner_iter.o
+endif
obj-$(CONFIG_MEMORY_ISOLATION) += page_isolation.o
obj-$(CONFIG_ZSMALLOC) += zsmalloc.o
obj-$(CONFIG_GENERIC_EARLY_IOREMAP) += early_ioremap.o
diff --git a/mm/page_owner.c b/mm/page_owner.c
index bafe474f3360..eb420803a434 100644
--- a/mm/page_owner.c
+++ b/mm/page_owner.c
@@ -14,6 +14,7 @@
#include <linux/sched/clock.h>
#include "page_alloc.h"
+#include "page_owner.h"
/*
* TODO: teach PAGE_OWNER_STACK_DEPTH (__dump_page_owner and save_stack)
@@ -21,31 +22,6 @@
*/
#define PAGE_OWNER_STACK_DEPTH (16)
-struct page_owner {
- unsigned short order;
- short last_migrate_reason;
- gfp_t gfp_mask;
- depot_stack_handle_t handle;
- depot_stack_handle_t free_handle;
- u64 ts_nsec;
- u64 free_ts_nsec;
- char comm[TASK_COMM_LEN];
- pid_t pid;
- pid_t tgid;
- pid_t free_pid;
- pid_t free_tgid;
-};
-
-/*
- * Cursor and per-page result of page_owner_next_eligible().
- */
-struct page_owner_scan {
- /* resume point */
- unsigned long pfn;
- struct page *page;
- struct page_owner po_snap;
-};
-
struct stack {
struct stack_record *stack_record;
struct stack *next;
@@ -741,7 +717,7 @@ void __dump_page_owner(const struct page *page)
* must advance scan->pfn past the hit before calling again to resume
* the scan.
*/
-static bool page_owner_next_eligible(struct page_owner_scan *scan)
+bool page_owner_next_eligible(struct page_owner_scan *scan)
{
unsigned long pfn = scan->pfn;
@@ -1196,6 +1172,11 @@ static int __init pageowner_init(void)
&stack_fops);
debugfs_create_file("count_threshold", 0600, dir, NULL,
&threshold_fops);
+
+#ifdef CONFIG_BPF_SYSCALL
+ if (page_owner_iter_register())
+ pr_warn("page_owner: failed to register bpf_iter target\n");
+#endif
return 0;
}
late_initcall(pageowner_init)
diff --git a/mm/page_owner.h b/mm/page_owner.h
new file mode 100644
index 000000000000..fda83be69275
--- /dev/null
+++ b/mm/page_owner.h
@@ -0,0 +1,48 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * mm/ internal page_owner and page_owner_iter declarations
+ */
+
+#ifndef __MM_PAGE_OWNER_H
+#define __MM_PAGE_OWNER_H
+
+#include <linux/page_ext.h>
+#include <linux/sched.h>
+#include <linux/stackdepot.h>
+
+/*
+ * mm/page_owner.c
+ */
+struct page_owner {
+ unsigned short order;
+ short last_migrate_reason;
+ gfp_t gfp_mask;
+ depot_stack_handle_t handle;
+ depot_stack_handle_t free_handle;
+ u64 ts_nsec;
+ u64 free_ts_nsec;
+ char comm[TASK_COMM_LEN];
+ pid_t pid;
+ pid_t tgid;
+ pid_t free_pid;
+ pid_t free_tgid;
+};
+
+/*
+ * Cursor and per-page result of page_owner_next_eligible().
+ */
+struct page_owner_scan {
+ /* resume point */
+ unsigned long pfn;
+ struct page *page;
+ struct page_owner po_snap;
+};
+
+bool page_owner_next_eligible(struct page_owner_scan *scan);
+
+/*
+ * mm/page_owner_iter.c
+ */
+int page_owner_iter_register(void);
+
+#endif /* __MM_PAGE_OWNER_H */
diff --git a/mm/page_owner_iter.c b/mm/page_owner_iter.c
new file mode 100644
index 000000000000..33dc9dcd59de
--- /dev/null
+++ b/mm/page_owner_iter.c
@@ -0,0 +1,116 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <linux/bpf.h>
+#include <linux/btf_ids.h>
+#include <linux/memblock.h>
+#include <linux/mm.h>
+#include <linux/seq_file.h>
+
+#include "page_owner.h"
+
+struct bpf_iter__page_owner {
+ __bpf_md_ptr(struct bpf_iter_meta *, meta);
+ __bpf_md_ptr(struct page *, page);
+ __bpf_md_ptr(struct page_owner *, po);
+ u64 pfn;
+};
+
+DEFINE_BPF_ITER_FUNC(page_owner, struct bpf_iter_meta *meta,
+ struct page *page, struct page_owner *po, u64 pfn)
+
+static void *page_owner_seq_start(struct seq_file *seq, loff_t *posp)
+{
+ struct page_owner_scan *scan = seq->private;
+
+ if (*posp == 0)
+ scan->pfn = min_low_pfn;
+ else
+ scan->pfn++;
+
+ if (!page_owner_next_eligible(scan))
+ return NULL;
+ return scan->page;
+}
+
+static void *page_owner_seq_next(struct seq_file *seq, void *v, loff_t *posp)
+{
+ struct page_owner_scan *scan = seq->private;
+
+ ++*posp;
+ scan->pfn++;
+ if (!page_owner_next_eligible(scan))
+ return NULL;
+
+ return scan->page;
+}
+
+static void page_owner_seq_stop(struct seq_file *seq, void *v)
+{
+}
+
+static int page_owner_prog_seq_show(struct bpf_prog *prog,
+ struct bpf_iter_meta *meta, void *v)
+{
+ struct page_owner_scan *scan = meta->seq->private;
+ struct bpf_iter__page_owner ctx;
+
+ /* @po: hit-time snapshot, no RCU window needed. */
+ ctx.meta = meta;
+ ctx.page = (struct page *)v;
+ ctx.po = &scan->po_snap;
+ ctx.pfn = scan->pfn;
+
+ return bpf_iter_run_prog(prog, &ctx);
+}
+
+static int page_owner_seq_show(struct seq_file *seq, void *v)
+{
+ struct bpf_iter_meta meta;
+ struct bpf_prog *prog;
+
+ meta.seq = seq;
+ prog = bpf_iter_get_info(&meta, false);
+ if (!prog)
+ return 0;
+
+ return page_owner_prog_seq_show(prog, &meta, v);
+}
+
+static const struct seq_operations page_owner_iter_seq_ops = {
+ .start = page_owner_seq_start,
+ .next = page_owner_seq_next,
+ .stop = page_owner_seq_stop,
+ .show = page_owner_seq_show,
+};
+
+static const struct bpf_iter_seq_info page_owner_iter_seq_info = {
+ .seq_ops = &page_owner_iter_seq_ops,
+ .init_seq_private = NULL,
+ .fini_seq_private = NULL,
+ .seq_priv_size = sizeof(struct page_owner_scan),
+};
+
+static struct bpf_iter_reg page_owner_iter_reg_info = {
+ .target = "page_owner",
+ .ctx_arg_info_size = 2,
+ .ctx_arg_info = {
+ { offsetof(struct bpf_iter__page_owner, page),
+ PTR_TO_BTF_ID_OR_NULL },
+ { offsetof(struct bpf_iter__page_owner, po),
+ PTR_TO_BTF_ID_OR_NULL },
+ },
+ .seq_info = &page_owner_iter_seq_info,
+};
+
+BTF_ID_LIST(page_owner_btf_ids)
+BTF_ID(struct, page)
+BTF_ID(struct, page_owner)
+
+int page_owner_iter_register(void)
+{
+ page_owner_iter_reg_info.ctx_arg_info[0].btf_id =
+ page_owner_btf_ids[0];
+ page_owner_iter_reg_info.ctx_arg_info[1].btf_id =
+ page_owner_btf_ids[1];
+
+ return bpf_iter_reg_target(&page_owner_iter_reg_info);
+}
--
2.20.1
^ permalink raw reply [flat|nested] 7+ messages in thread* [PATCH 3/6] mm/page_owner: add open-coded page_owner iterator
2026-10-09 11:23 [PATCH 0/6] mm/page_owner: add bpf_iter and kfuncs Zhen Ni
2026-10-09 11:23 ` [PATCH 2/6] mm/page_owner: add bpf_iter target "page_owner" Zhen Ni
@ 2026-10-09 11:23 ` Zhen Ni
2026-10-09 11:23 ` [PATCH 4/6] mm/page_owner: add bpf_page_owner_get_nid() kfunc Zhen Ni
` (2 subsequent siblings)
4 siblings, 0 replies; 7+ messages in thread
From: Zhen Ni @ 2026-10-09 11:23 UTC (permalink / raw)
To: ast, daniel, andrii, eddyz87, memxor, martin.lau, song,
yonghong.song, jolsa, emil, ihor.solodrai, akpm, vbabka, surenb,
mhocko, brendan.jackman, hannes, ziy, shuah
Cc: bpf, linux-mm, linux-kselftest, linux-kernel, Zhen Ni
Add an open-coded bpf_iter over the pages eligible for page_owner
output, reusing page_owner_next_eligible().
A sleepable BPF program drives it with
bpf_iter_page_owner_new()/next()/destroy(); each next() returns the
page_owner_scan. po_snap is copied inside the page_ext RCU window, so
the program needs no RCU section when it reads it.
Also add bpf_page_owner_stack_snprint(), which renders the stack trace
for a stack_depot handle into a caller-provided buffer.
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
kernel/bpf/helpers.c | 6 +++++
mm/page_owner_iter.c | 52 ++++++++++++++++++++++++++++++++++++++++++++
2 files changed, 58 insertions(+)
diff --git a/kernel/bpf/helpers.c b/kernel/bpf/helpers.c
index 712dca5a2c5b..23a2d2d71f4d 100644
--- a/kernel/bpf/helpers.c
+++ b/kernel/bpf/helpers.c
@@ -4946,6 +4946,12 @@ BTF_ID_FLAGS(func, bpf_iter_dmabuf_new, KF_ITER_NEW | KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_iter_dmabuf_next, KF_ITER_NEXT | KF_RET_NULL | KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_iter_dmabuf_destroy, KF_ITER_DESTROY | KF_SLEEPABLE)
#endif
+#ifdef CONFIG_PAGE_OWNER
+BTF_ID_FLAGS(func, bpf_iter_page_owner_new, KF_ITER_NEW | KF_SLEEPABLE)
+BTF_ID_FLAGS(func, bpf_iter_page_owner_next, KF_ITER_NEXT | KF_RET_NULL | KF_SLEEPABLE)
+BTF_ID_FLAGS(func, bpf_iter_page_owner_destroy, KF_ITER_DESTROY | KF_SLEEPABLE)
+BTF_ID_FLAGS(func, bpf_page_owner_stack_snprint, KF_SLEEPABLE)
+#endif
BTF_ID_FLAGS(func, __bpf_trap)
BTF_ID_FLAGS(func, bpf_strcmp, KF_PERFMON);
BTF_ID_FLAGS(func, bpf_strcasecmp, KF_PERFMON);
diff --git a/mm/page_owner_iter.c b/mm/page_owner_iter.c
index 33dc9dcd59de..80693d417407 100644
--- a/mm/page_owner_iter.c
+++ b/mm/page_owner_iter.c
@@ -4,6 +4,7 @@
#include <linux/memblock.h>
#include <linux/mm.h>
#include <linux/seq_file.h>
+#include <linux/stackdepot.h>
#include "page_owner.h"
@@ -114,3 +115,54 @@ int page_owner_iter_register(void)
return bpf_iter_reg_target(&page_owner_iter_reg_info);
}
+
+/* opaque to BPF programs */
+struct bpf_iter_page_owner {
+ __u64 __opaque[16];
+} __attribute__((aligned(8)));
+
+struct bpf_iter_page_owner_kern {
+ struct page_owner_scan scan;
+} __attribute__((aligned(8)));
+
+__bpf_kfunc_start_defs();
+
+__bpf_kfunc int bpf_iter_page_owner_new(struct bpf_iter_page_owner *it)
+{
+ struct bpf_iter_page_owner_kern *kit = (void *)it;
+
+ BUILD_BUG_ON(sizeof(*kit) > sizeof(*it));
+ BUILD_BUG_ON(__alignof__(*kit) != __alignof__(*it));
+
+ kit->scan.pfn = min_low_pfn;
+ kit->scan.page = NULL;
+ return 0;
+}
+
+__bpf_kfunc struct page_owner_scan *
+bpf_iter_page_owner_next(struct bpf_iter_page_owner *it)
+{
+ struct bpf_iter_page_owner_kern *kit = (void *)it;
+
+ cond_resched();
+
+ if (kit->scan.page)
+ kit->scan.pfn++;
+
+ if (!page_owner_next_eligible(&kit->scan))
+ return NULL;
+
+ return &kit->scan;
+}
+
+__bpf_kfunc void bpf_iter_page_owner_destroy(struct bpf_iter_page_owner *it)
+{
+}
+
+__bpf_kfunc int bpf_page_owner_stack_snprint(depot_stack_handle_t handle,
+ char *buf, u32 buf_size)
+{
+ return stack_depot_snprint(handle, buf, buf_size, 0);
+}
+
+__bpf_kfunc_end_defs();
--
2.20.1
^ permalink raw reply [flat|nested] 7+ messages in thread* [PATCH 4/6] mm/page_owner: add bpf_page_owner_get_nid() kfunc
2026-10-09 11:23 [PATCH 0/6] mm/page_owner: add bpf_iter and kfuncs Zhen Ni
2026-10-09 11:23 ` [PATCH 2/6] mm/page_owner: add bpf_iter target "page_owner" Zhen Ni
2026-10-09 11:23 ` [PATCH 3/6] mm/page_owner: add open-coded page_owner iterator Zhen Ni
@ 2026-10-09 11:23 ` Zhen Ni
2026-10-09 11:23 ` [PATCH 5/6] mm/page_owner: add bpf_page_owner_get_memcg_info() kfunc Zhen Ni
2026-10-09 11:23 ` [PATCH 6/6] selftests/bpf: add tests for the page_owner bpf_iter target Zhen Ni
4 siblings, 0 replies; 7+ messages in thread
From: Zhen Ni @ 2026-10-09 11:23 UTC (permalink / raw)
To: ast, daniel, andrii, eddyz87, memxor, martin.lau, song,
yonghong.song, jolsa, emil, ihor.solodrai, akpm, vbabka, surenb,
mhocko, brendan.jackman, hannes, ziy, shuah
Cc: bpf, linux-mm, linux-kselftest, linux-kernel, Zhen Ni
Extract the NUMA node lookup from read_page_owner() into a shared
page_owner_get_nid() helper, and expose it to BPF programs as
bpf_page_owner_get_nid().
The nid bitfield layout is build-time configuration, so BPF programs
cannot accurately derive it from page->flags.
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
kernel/bpf/helpers.c | 1 +
mm/page_owner.c | 26 +++++++++++++++-----------
mm/page_owner.h | 1 +
mm/page_owner_iter.c | 5 +++++
4 files changed, 22 insertions(+), 11 deletions(-)
diff --git a/kernel/bpf/helpers.c b/kernel/bpf/helpers.c
index 23a2d2d71f4d..3e8afd19a1f4 100644
--- a/kernel/bpf/helpers.c
+++ b/kernel/bpf/helpers.c
@@ -4951,6 +4951,7 @@ BTF_ID_FLAGS(func, bpf_iter_page_owner_new, KF_ITER_NEW | KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_iter_page_owner_next, KF_ITER_NEXT | KF_RET_NULL | KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_iter_page_owner_destroy, KF_ITER_DESTROY | KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_page_owner_stack_snprint, KF_SLEEPABLE)
+BTF_ID_FLAGS(func, bpf_page_owner_get_nid, KF_SLEEPABLE)
#endif
BTF_ID_FLAGS(func, __bpf_trap)
BTF_ID_FLAGS(func, bpf_strcmp, KF_PERFMON);
diff --git a/mm/page_owner.c b/mm/page_owner.c
index eb420803a434..43a722678acc 100644
--- a/mm/page_owner.c
+++ b/mm/page_owner.c
@@ -711,6 +711,19 @@ void __dump_page_owner(const struct page *page)
page_ext_put(page_ext);
}
+/*
+ * NUMA node of @page, bypassing PF_POISONED_CHECK() in page_to_nid() to
+ * avoid VM_BUG_ON on poisoned pages; -1 for poisoned pages.
+ */
+int page_owner_get_nid(struct page *page)
+{
+ memdesc_flags_t page_flags = READ_ONCE(page->flags);
+
+ if (page_flags.f == PAGE_POISON_PATTERN)
+ return -1;
+ return memdesc_nid(&page_flags);
+}
+
/*
* Advance @scan to the next page eligible for page_owner output.
* On a hit, fill scan->pfn/page/po_snap and return true; the caller
@@ -810,18 +823,9 @@ read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
/* Find an allocated page */
while (page_owner_next_eligible(&scan)) {
if (state->nid_filter_enabled) {
- int nid;
- memdesc_flags_t page_flags =
- READ_ONCE(scan.page->flags);
+ int nid = page_owner_get_nid(scan.page);
- /*
- * Bypass PF_POISONED_CHECK() in page_to_nid() to avoid
- * VM_BUG_ON when accessing poisoned pages.
- */
- if (page_flags.f == PAGE_POISON_PATTERN)
- goto skip_continue;
- nid = memdesc_nid(&page_flags);
- if (!node_isset(nid, state->nid_filter))
+ if (nid < 0 || !node_isset(nid, state->nid_filter))
goto skip_continue;
}
diff --git a/mm/page_owner.h b/mm/page_owner.h
index fda83be69275..c57859d18dc3 100644
--- a/mm/page_owner.h
+++ b/mm/page_owner.h
@@ -39,6 +39,7 @@ struct page_owner_scan {
};
bool page_owner_next_eligible(struct page_owner_scan *scan);
+int page_owner_get_nid(struct page *page);
/*
* mm/page_owner_iter.c
diff --git a/mm/page_owner_iter.c b/mm/page_owner_iter.c
index 80693d417407..907517a29144 100644
--- a/mm/page_owner_iter.c
+++ b/mm/page_owner_iter.c
@@ -165,4 +165,9 @@ __bpf_kfunc int bpf_page_owner_stack_snprint(depot_stack_handle_t handle,
return stack_depot_snprint(handle, buf, buf_size, 0);
}
+__bpf_kfunc int bpf_page_owner_get_nid(struct page_owner_scan *scan)
+{
+ return page_owner_get_nid(scan->page);
+}
+
__bpf_kfunc_end_defs();
--
2.20.1
^ permalink raw reply [flat|nested] 7+ messages in thread* [PATCH 5/6] mm/page_owner: add bpf_page_owner_get_memcg_info() kfunc
2026-10-09 11:23 [PATCH 0/6] mm/page_owner: add bpf_iter and kfuncs Zhen Ni
` (2 preceding siblings ...)
2026-10-09 11:23 ` [PATCH 4/6] mm/page_owner: add bpf_page_owner_get_nid() kfunc Zhen Ni
@ 2026-10-09 11:23 ` Zhen Ni
2026-10-09 11:23 ` [PATCH 6/6] selftests/bpf: add tests for the page_owner bpf_iter target Zhen Ni
4 siblings, 0 replies; 7+ messages in thread
From: Zhen Ni @ 2026-10-09 11:23 UTC (permalink / raw)
To: ast, daniel, andrii, eddyz87, memxor, martin.lau, song,
yonghong.song, jolsa, emil, ihor.solodrai, akpm, vbabka, surenb,
mhocko, brendan.jackman, hannes, ziy, shuah
Cc: bpf, linux-mm, linux-kselftest, linux-kernel, Zhen Ni
Expose per-page memcg information to BPF programs, so that a bpf_iter
over page_owner can aggregate and filter allocations by memcg.
Attributing an allocation to a specific cgroup needs its full cgroup
path, not just the memcg name.
Split memcg info collection out of print_page_owner_memcg() into
get_page_memcg_info(), which fills struct memcg_info and, when given
a caller-provided buffer, writes the full cgroup path via
cgroup_path(). print_page_owner_memcg() now consumes the struct;
this introduces no functional change to the debugfs output.
The helper takes a single READ_ONCE snapshot within one RCU read-side
critical section, so the flags and cgroup path reflect a consistent
view of the page. cgroup_path() returns the path relative to the
cgroup hierarchy root, so memcg attribution works on both cgroup v1
and v2.
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
kernel/bpf/helpers.c | 1 +
mm/page_owner.c | 85 ++++++++++++++++++++++++++++++++++----------
mm/page_owner.h | 9 +++++
mm/page_owner_iter.c | 7 ++++
4 files changed, 83 insertions(+), 19 deletions(-)
diff --git a/kernel/bpf/helpers.c b/kernel/bpf/helpers.c
index 3e8afd19a1f4..b07a4dc46b58 100644
--- a/kernel/bpf/helpers.c
+++ b/kernel/bpf/helpers.c
@@ -4952,6 +4952,7 @@ BTF_ID_FLAGS(func, bpf_iter_page_owner_next, KF_ITER_NEXT | KF_RET_NULL | KF_SLE
BTF_ID_FLAGS(func, bpf_iter_page_owner_destroy, KF_ITER_DESTROY | KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_page_owner_stack_snprint, KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_page_owner_get_nid, KF_SLEEPABLE)
+BTF_ID_FLAGS(func, bpf_page_owner_get_memcg_info, KF_SLEEPABLE)
#endif
BTF_ID_FLAGS(func, __bpf_trap)
BTF_ID_FLAGS(func, bpf_strcmp, KF_PERFMON);
diff --git a/mm/page_owner.c b/mm/page_owner.c
index 43a722678acc..ad503422b0c4 100644
--- a/mm/page_owner.c
+++ b/mm/page_owner.c
@@ -538,16 +538,22 @@ void pagetypeinfo_showmixedcount_print(struct seq_file *m,
#ifdef CONFIG_MEMCG
/*
- * Looking for memcg information and print it out
+ * Collect memcg information of @page into @info; when @buf and a
+ * nonzero @size are given, also copy the full cgroup path of the
+ * charged memcg into @buf.
+ *
+ * Return: 0 if @page is charged to an attributable memcg; -ENODATA
+ * otherwise.
*/
-static inline int print_page_owner_memcg(char *kbuf, size_t count, int ret,
- struct page *page)
+int get_page_memcg_info(struct page *page, struct memcg_info *info,
+ char *buf, int size)
{
unsigned long memcg_data;
struct obj_cgroup *objcg;
struct mem_cgroup *memcg;
- bool online;
- char name[80];
+ int ret = -ENODATA;
+
+ *info = (struct memcg_info){};
rcu_read_lock();
memcg_data = READ_ONCE(page->memcg_data);
@@ -555,31 +561,68 @@ static inline int print_page_owner_memcg(char *kbuf, size_t count, int ret,
goto out_unlock;
if (memcg_data & MEMCG_DATA_OBJEXTS) {
- ret += scnprintf(kbuf + ret, count - ret,
- "Slab cache page\n");
+ info->is_slab = true;
goto out_unlock;
}
objcg = (void *)(memcg_data & ~OBJEXTS_FLAGS_MASK);
- memcg = objcg ? obj_cgroup_memcg(objcg) : NULL;
+ if (!objcg)
+ goto out_unlock;
+
+ memcg = obj_cgroup_memcg(objcg);
if (!memcg)
goto out_unlock;
- online = css_is_online(&memcg->css);
- cgroup_name(memcg->css.cgroup, name, sizeof(name));
- ret += scnprintf(kbuf + ret, count - ret,
- "Charged %sto %smemcg %s\n",
- (memcg_data & MEMCG_DATA_KMEM) ? "(via objcg) " : "",
- online ? "" : "offline ",
- name);
+ info->is_kmem = (memcg_data & MEMCG_DATA_KMEM) != 0;
+ info->is_online = css_is_online(&memcg->css);
+ cgroup_name(memcg->css.cgroup, info->name, sizeof(info->name));
+
+ if (buf && size)
+ cgroup_path(memcg->css.cgroup, buf, size);
+ ret = 0;
out_unlock:
rcu_read_unlock();
+ if (ret < 0 && buf && size)
+ buf[0] = '\0';
+
+ return ret;
+}
+
+/*
+ * Print memcg information from memcg_info
+ */
+static inline int print_page_owner_memcg(char *kbuf, size_t count, int ret,
+ const struct memcg_info *info)
+{
+ if (!info)
+ return ret;
+
+ if (info->is_slab)
+ ret += scnprintf(kbuf + ret, count - ret,
+ "Slab cache page\n");
+
+ if (info->name[0])
+ ret += scnprintf(kbuf + ret, count - ret,
+ "Charged %sto %smemcg %s\n",
+ info->is_kmem ? "(via objcg) " : "",
+ info->is_online ? "" : "offline ",
+ info->name);
+
return ret;
}
#else
+int get_page_memcg_info(struct page *page, struct memcg_info *info,
+ char *buf, int size)
+{
+ *info = (struct memcg_info){};
+ if (buf && size)
+ buf[0] = '\0';
+ return -EOPNOTSUPP;
+}
+
static inline int print_page_owner_memcg(char *kbuf, size_t count, int ret,
- struct page *page)
+ const struct memcg_info *info)
{
return ret;
}
@@ -589,7 +632,8 @@ static ssize_t
print_page_owner(char __user *buf, size_t count, unsigned long pfn,
struct page *page, struct page_owner *page_owner,
depot_stack_handle_t handle,
- struct page_owner_filter_state *state)
+ struct page_owner_filter_state *state,
+ const struct memcg_info *memcg_info)
{
int ret, pageblock_mt, page_mt;
char *kbuf;
@@ -639,7 +683,7 @@ print_page_owner(char __user *buf, size_t count, unsigned long pfn,
migrate_reason_names[page_owner->last_migrate_reason]);
}
- ret = print_page_owner_memcg(kbuf, count, ret, page);
+ ret = print_page_owner_memcg(kbuf, count, ret, memcg_info);
ret += snprintf(kbuf + ret, count - ret, "\n");
if (ret >= count)
@@ -810,6 +854,7 @@ static ssize_t
read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
{
struct page_owner_scan scan;
+ struct memcg_info memcg_info;
struct page_owner_filter_state *state = file->private_data;
if (!static_branch_unlikely(&page_owner_inited))
@@ -832,8 +877,10 @@ read_page_owner(struct file *file, char __user *buf, size_t count, loff_t *ppos)
/* Record the next PFN to read in the file offset */
*ppos = scan.pfn + 1;
+ get_page_memcg_info(scan.page, &memcg_info, NULL, 0);
return print_page_owner(buf, count, scan.pfn, scan.page,
- &scan.po_snap, scan.po_snap.handle, state);
+ &scan.po_snap, scan.po_snap.handle, state,
+ &memcg_info);
skip_continue:
scan.pfn++;
cond_resched();
diff --git a/mm/page_owner.h b/mm/page_owner.h
index c57859d18dc3..e74fc664d9b1 100644
--- a/mm/page_owner.h
+++ b/mm/page_owner.h
@@ -38,8 +38,17 @@ struct page_owner_scan {
struct page_owner po_snap;
};
+struct memcg_info {
+ char name[80];
+ bool is_slab;
+ bool is_kmem;
+ bool is_online;
+};
+
bool page_owner_next_eligible(struct page_owner_scan *scan);
int page_owner_get_nid(struct page *page);
+int get_page_memcg_info(struct page *page, struct memcg_info *info,
+ char *buf, int size);
/*
* mm/page_owner_iter.c
diff --git a/mm/page_owner_iter.c b/mm/page_owner_iter.c
index 907517a29144..e7a8ca1e90e8 100644
--- a/mm/page_owner_iter.c
+++ b/mm/page_owner_iter.c
@@ -170,4 +170,11 @@ __bpf_kfunc int bpf_page_owner_get_nid(struct page_owner_scan *scan)
return page_owner_get_nid(scan->page);
}
+__bpf_kfunc int bpf_page_owner_get_memcg_info(struct page_owner_scan *scan,
+ struct memcg_info *info,
+ char *buf, u32 size)
+{
+ return get_page_memcg_info(scan->page, info, buf, size);
+}
+
__bpf_kfunc_end_defs();
--
2.20.1
^ permalink raw reply [flat|nested] 7+ messages in thread* [PATCH 6/6] selftests/bpf: add tests for the page_owner bpf_iter target
2026-10-09 11:23 [PATCH 0/6] mm/page_owner: add bpf_iter and kfuncs Zhen Ni
` (3 preceding siblings ...)
2026-10-09 11:23 ` [PATCH 5/6] mm/page_owner: add bpf_page_owner_get_memcg_info() kfunc Zhen Ni
@ 2026-10-09 11:23 ` Zhen Ni
2026-10-09 13:48 ` bot+bpf-ci
4 siblings, 1 reply; 7+ messages in thread
From: Zhen Ni @ 2026-10-09 11:23 UTC (permalink / raw)
To: ast, daniel, andrii, eddyz87, memxor, martin.lau, song,
yonghong.song, jolsa, emil, ihor.solodrai, akpm, vbabka, surenb,
mhocko, brendan.jackman, hannes, ziy, shuah
Cc: bpf, linux-mm, linux-kselftest, linux-kernel, Zhen Ni
Add a sleepable BPF program and a userspace runner covering the seq_file
and open-coded page_owner iterators and the get_nid, stack_snprint and
get_memcg_info kfuncs. The test checks:
- seq_file walk, unfiltered and pid-filtered: records read back from
the iterator fd match the lines the program printed.
- open-coded iterator parity with the seq path.
- the kfuncs' contracts: valid nid per page, a non-empty stack
trace, and memcg classification.
Skipped when debugfs page_owner is unavailable.
Signed-off-by: Zhen Ni <zhen.ni@easystack.cn>
---
.../bpf/prog_tests/page_owner_iter.c | 173 ++++++++++++++++++
.../selftests/bpf/progs/bpf_iter_page_owner.c | 161 ++++++++++++++++
2 files changed, 334 insertions(+)
create mode 100644 tools/testing/selftests/bpf/prog_tests/page_owner_iter.c
create mode 100644 tools/testing/selftests/bpf/progs/bpf_iter_page_owner.c
diff --git a/tools/testing/selftests/bpf/prog_tests/page_owner_iter.c b/tools/testing/selftests/bpf/prog_tests/page_owner_iter.c
new file mode 100644
index 000000000000..a621712272bc
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/page_owner_iter.c
@@ -0,0 +1,173 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (c) 2026 */
+
+#include <test_progs.h>
+#include "bpf_iter_page_owner.skel.h"
+
+#define DEBUGFS_PAGE_OWNER "/sys/kernel/debug/page_owner"
+
+static bool page_owner_available(void)
+{
+ return access(DEBUGFS_PAGE_OWNER, R_OK) == 0;
+}
+
+/* Count link_fd output lines; -1 on error. */
+static long iter_record_count(int link_fd)
+{
+ char *line = NULL;
+ size_t len = 0;
+ long count = 0;
+ int iter_fd, err;
+ FILE *f;
+
+ iter_fd = bpf_iter_create(link_fd);
+ if (!ASSERT_GE(iter_fd, 0, "bpf_iter_create"))
+ return -1;
+
+ f = fdopen(iter_fd, "r");
+ if (!f) {
+ close(iter_fd);
+ return -1;
+ }
+ while (getline(&line, &len, f) > 0)
+ count++;
+ err = ferror(f);
+ free(line);
+ fclose(f);
+ return err ? -1 : count;
+}
+
+/* Validate bpf_page_owner_get_memcg_info() results from the open-coded run. */
+static void check_memcg_skel(struct bpf_iter_page_owner *skel)
+{
+ fprintf(stderr,
+ "memcg: seen=%u charged=%u online=%u nodata=%u notsup=%u bad=%u first='%s' name='%s' flags=%u\n",
+ skel->bss->memcg_seen, skel->bss->memcg_charged,
+ skel->bss->memcg_online, skel->bss->memcg_nodata,
+ skel->bss->memcg_notsup, skel->bss->memcg_bad,
+ skel->bss->memcg_first_path, skel->bss->memcg_first_name,
+ skel->bss->memcg_first_flags);
+
+ if (!ASSERT_GT(skel->bss->memcg_seen, 0, "memcg_seen_gt_0"))
+ return;
+
+ /* all -EOPNOTSUPP: CONFIG_MEMCG is off */
+ if (skel->bss->memcg_notsup == skel->bss->memcg_seen) {
+ ASSERT_EQ(skel->bss->memcg_bad, 0, "memcg_stub_invariants");
+ return;
+ }
+
+ if (!ASSERT_EQ(skel->bss->memcg_bad, 0, "memcg_kfunc_invariants"))
+ return;
+
+ /* Every call is either charged or nodata */
+ if (!ASSERT_EQ(skel->bss->memcg_charged + skel->bss->memcg_nodata,
+ skel->bss->memcg_seen, "memcg_ret_classified"))
+ return;
+
+ if (!skel->bss->memcg_first_saved)
+ return;
+
+ if (!ASSERT_EQ(skel->bss->memcg_first_path[0], '/',
+ "memcg_first_path_abs"))
+ return;
+
+ if (skel->bss->memcg_first_name[0])
+ ASSERT_OK_PTR(strstr(skel->bss->memcg_first_path,
+ skel->bss->memcg_first_name),
+ "memcg_name_in_path");
+}
+
+void serial_test_page_owner_iter(void)
+{
+ struct bpf_iter_page_owner *skel = NULL;
+ LIBBPF_OPTS(bpf_test_run_opts, opts);
+ long iter_count, unfiltered_count, oc_diff, pid;
+ unsigned int printed_before;
+ int link_fd, prog_fd;
+
+ if (!page_owner_available()) {
+ test__skip();
+ return;
+ }
+
+ skel = bpf_iter_page_owner__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "bpf_iter_page_owner__open_and_load"))
+ return;
+
+ /* --- seq file iterator: unfiltered run --- */
+ skel->bss->filter_en = 0;
+
+ if (!ASSERT_OK(bpf_iter_page_owner__attach(skel), "skel_attach"))
+ goto destroy;
+
+ link_fd = bpf_link__fd(skel->links.dump_page_owner);
+ iter_count = iter_record_count(link_fd);
+ if (iter_count < 0)
+ goto destroy;
+ unfiltered_count = iter_count;
+
+ fprintf(stderr, "iter_count=%ld printed_lines=%u total_seen=%u\n",
+ iter_count, skel->bss->printed_lines, skel->bss->total_seen);
+
+ ASSERT_GT(iter_count, 0, "unfiltered_lines_gt_0");
+
+ ASSERT_EQ((long)skel->bss->printed_lines, iter_count,
+ "lines_match_printed");
+
+ /* unfiltered: every visited page is printed, so the two must match */
+ ASSERT_EQ(skel->bss->total_seen, skel->bss->printed_lines,
+ "unfiltered_all_printed");
+
+ /* --- seq file iterator: pid-filtered run --- */
+ pid = getpid();
+ skel->bss->filter_en = 1;
+ skel->bss->filter_pid = (pid_t)pid;
+
+ /* printed_lines is cumulative across runs: compare the delta */
+ printed_before = skel->bss->printed_lines;
+ iter_count = iter_record_count(link_fd);
+ if (iter_count < 0)
+ goto destroy;
+
+ fprintf(stderr,
+ "pid=%ld iter_count=%ld printed_delta=%u\n",
+ pid, iter_count, skel->bss->printed_lines - printed_before);
+
+ ASSERT_GT(iter_count, 0, "filtered_lines_gt_0");
+ ASSERT_EQ(skel->bss->printed_lines - printed_before, iter_count,
+ "filtered_printed_match");
+
+ /* --- open-coded iterator parity --- */
+ prog_fd = bpf_program__fd(skel->progs.scan_open_coded);
+
+ ASSERT_OK(bpf_prog_test_run_opts(prog_fd, &opts), "prog_test_run");
+ fprintf(stderr,
+ "open_coded_seen=%u nid_seen=%u first_nid=%d "
+ "stack_bytes=%u stack=[\n%s]\n",
+ skel->bss->open_coded_seen, skel->bss->nid_seen,
+ skel->data->first_nid, skel->bss->stack_bytes,
+ skel->bss->stack_text);
+
+ /* same eligibility as the seq path: counts must be close */
+ ASSERT_GT(skel->bss->open_coded_seen, 0, "oc_seen_gt_0");
+ oc_diff = (long)skel->bss->open_coded_seen - unfiltered_count;
+ if (oc_diff < 0)
+ oc_diff = -oc_diff;
+ ASSERT_LE(oc_diff, unfiltered_count / 100 + 64,
+ "oc_seen_close_to_iter");
+
+ /* the nid kfunc returned a valid node for every page */
+ ASSERT_EQ(skel->bss->nid_seen, skel->bss->open_coded_seen,
+ "nid_valid_all");
+ ASSERT_GE(skel->data->first_nid, 0, "first_nid_valid");
+
+ ASSERT_GT(skel->bss->stack_bytes, 0, "stack_snprint_ok");
+ ASSERT_OK_PTR(strstr(skel->bss->stack_text, "+0x"), "stack_frames");
+ ASSERT_OK_PTR(strchr(skel->bss->stack_text, '\n'), "stack_lines");
+
+ check_memcg_skel(skel);
+
+destroy:
+ bpf_iter_page_owner__destroy(skel);
+}
diff --git a/tools/testing/selftests/bpf/progs/bpf_iter_page_owner.c b/tools/testing/selftests/bpf/progs/bpf_iter_page_owner.c
new file mode 100644
index 000000000000..fada6a314a15
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/bpf_iter_page_owner.c
@@ -0,0 +1,161 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (c) 2026 */
+
+#include <vmlinux.h>
+#include <bpf/bpf_helpers.h>
+
+char _license[] SEC("license") = "GPL";
+
+__u32 filter_en;
+__u32 filter_pid;
+__u32 total_seen;
+__u32 printed_lines;
+
+__u32 open_coded_seen;
+__u32 nid_seen;
+__s32 first_nid = -1;
+char stack_text[2048];
+__u32 stack_bytes;
+
+__u32 memcg_seen;
+__u32 memcg_charged;
+__u32 memcg_online;
+__u32 memcg_nodata;
+__u32 memcg_notsup;
+__u32 memcg_bad;
+__u32 memcg_first_saved;
+__u32 memcg_first_flags; /* bit0 is_online, bit1 is_kmem */
+__u32 memcg_first_path_len;
+char memcg_pathbuf[4096];
+char memcg_first_path[4096];
+char memcg_first_name[80];
+
+#define MEMCG_ENODATA 61
+#define MEMCG_EOPNOTSUPP 95
+
+#define MEMCG_F_ONLINE (1 << 0)
+#define MEMCG_F_OBJCG (1 << 1)
+
+extern int bpf_iter_page_owner_new(struct bpf_iter_page_owner *it) __ksym;
+extern struct page_owner_scan *
+bpf_iter_page_owner_next(struct bpf_iter_page_owner *it) __ksym;
+extern void bpf_iter_page_owner_destroy(struct bpf_iter_page_owner *it) __ksym;
+extern int bpf_page_owner_stack_snprint(depot_stack_handle_t handle,
+ char *buf, u32 buf_size) __ksym;
+extern int bpf_page_owner_get_nid(struct page_owner_scan *scan) __ksym;
+extern int bpf_page_owner_get_memcg_info(struct page_owner_scan *scan,
+ struct memcg_info *info,
+ char *buf, u32 size) __ksym;
+
+static __u32 copy_str(char *dst, const char *src, const __u32 max)
+{
+ __u32 i = 0;
+
+ while (i + 1 < max && src[i]) {
+ dst[i] = src[i];
+ i++;
+ }
+ dst[i] = '\0';
+ return i;
+}
+
+/*
+ * Exercise bpf_page_owner_get_memcg_info() until 32 pages charged to an
+ * ONLINE memcg have been seen, checking its per-return contract:
+ * - 0: buf holds an absolute cgroup path;
+ * - -ENODATA: buf was NUL-terminated;
+ * - -EOPNOTSUPP: counted (CONFIG_MEMCG is off);
+ *
+ * The first online charged sample is saved for userspace to print and
+ * cross-check (name vs basename of path).
+ */
+static void check_memcg(struct page_owner_scan *scan)
+{
+ static struct memcg_info info;
+ int err;
+
+ err = bpf_page_owner_get_memcg_info(scan, &info, memcg_pathbuf,
+ sizeof(memcg_pathbuf));
+ memcg_seen++;
+
+ if (err == 0) {
+ memcg_charged++;
+
+ if (memcg_pathbuf[0] != '/') {
+ memcg_bad++;
+ return;
+ }
+
+ if (!info.is_online)
+ return;
+ memcg_online++;
+
+ /* keep the first online sample for userspace */
+ if (!memcg_first_saved) {
+ memcg_first_path_len = copy_str(memcg_first_path,
+ memcg_pathbuf,
+ sizeof(memcg_first_path));
+ copy_str(memcg_first_name, info.name,
+ sizeof(memcg_first_name));
+ memcg_first_flags = MEMCG_F_ONLINE |
+ (info.is_kmem ? MEMCG_F_OBJCG : 0);
+ memcg_first_saved = 1;
+ }
+ } else if (err == -MEMCG_ENODATA) {
+ memcg_nodata++;
+ if (memcg_pathbuf[0] != '\0')
+ memcg_bad++;
+ } else if (err == -MEMCG_EOPNOTSUPP) {
+ memcg_notsup++;
+ if (memcg_pathbuf[0] != '\0')
+ memcg_bad++;
+ } else {
+ memcg_bad++;
+ }
+}
+
+SEC("syscall")
+int scan_open_coded(const void *ctx)
+{
+ struct bpf_iter_page_owner it;
+ struct page_owner_scan *scan;
+
+ bpf_iter_page_owner_new(&it);
+
+ while ((scan = bpf_iter_page_owner_next(&it))) {
+ open_coded_seen++;
+ if (bpf_page_owner_get_nid(scan) >= 0)
+ nid_seen++;
+ if (memcg_online < 32)
+ check_memcg(scan);
+ /* grab the stack of the 100th page for userspace to check */
+ if (open_coded_seen == 100) {
+ first_nid = bpf_page_owner_get_nid(scan);
+ if (scan->po_snap.handle)
+ stack_bytes = bpf_page_owner_stack_snprint(
+ scan->po_snap.handle,
+ stack_text, sizeof(stack_text));
+ }
+ }
+ bpf_iter_page_owner_destroy(&it);
+ return 0;
+}
+
+SEC("iter/page_owner")
+int dump_page_owner(struct bpf_iter__page_owner *ctx)
+{
+ struct page_owner *po = ctx->po;
+
+ if (!po)
+ return 0;
+
+ total_seen++;
+
+ if (filter_en && po->pid != (pid_t)filter_pid)
+ return 0;
+
+ BPF_SEQ_PRINTF(ctx->meta->seq, "pfn %llu pid %d comm %s\n",
+ ctx->pfn, po->pid, po->comm);
+ printed_lines++;
+ return 0;
+}
--
2.20.1
^ permalink raw reply [flat|nested] 7+ messages in thread* Re: [PATCH 6/6] selftests/bpf: add tests for the page_owner bpf_iter target
2026-10-09 11:23 ` [PATCH 6/6] selftests/bpf: add tests for the page_owner bpf_iter target Zhen Ni
@ 2026-10-09 13:48 ` bot+bpf-ci
0 siblings, 0 replies; 7+ messages in thread
From: bot+bpf-ci @ 2026-10-09 13:48 UTC (permalink / raw)
To: zhen.ni, ast, daniel, andrii, eddyz87, memxor, martin.lau, song,
yonghong.song, jolsa, emil, ihor.solodrai, akpm, vbabka, surenb,
mhocko, brendan.jackman, hannes, ziy, shuah
Cc: bpf, linux-mm, linux-kselftest, linux-kernel, zhen.ni, ast,
andrii, daniel, martin.lau, eddyz87, yonghong.song, mason,
ihor.solodrai
[-- Attachment #1: Type: text/plain, Size: 6857 bytes --]
> selftests/bpf: add tests for the page_owner bpf_iter target
TL;DR:
- Selftest BPF program may fail to build, breaking test_progs, if
CONFIG_PAGE_OWNER types are missing from vmlinux.h
- Iterator read errors (e.g. -EAGAIN) silently end the test as a pass
- Skip gives no reason
> diff --git a/tools/testing/selftests/bpf/prog_tests/page_owner_iter.c b/tools/testing/selftests/bpf/prog_tests/page_owner_iter.c
> new file mode 100644
> index 000000000000..a621712272bc
> --- /dev/null
> +++ b/tools/testing/selftests/bpf/prog_tests/page_owner_iter.c
> @@ -0,0 +1,173 @@
[ ... ]
> +/* Count link_fd output lines; -1 on error. */
> +static long iter_record_count(int link_fd)
> +{
[ ... ]
> + f = fdopen(iter_fd, "r");
> + if (!f) {
> + close(iter_fd);
> + return -1;
> + }
> + while (getline(&line, &len, f) > 0)
> + count++;
> + err = ferror(f);
> + free(line);
> + fclose(f);
> + return err ? -1 : count;
> +}
[ ... ]
> +void serial_test_page_owner_iter(void)
> +{
[ ... ]
> + if (!page_owner_available()) {
> + test__skip();
> + return;
> + }
A subsystem pattern flags this as potentially concerning: a test that skips
silently, with no message.
test__skip() takes no message and nothing is printed before it, so the
result is a bare skip. The page_owner file only exists with
CONFIG_PAGE_OWNER=y, page_owner=on on the kernel command line, and debugfs
mounted, so someone reading the output cannot tell which of these was
missing.
Other prog_tests print the reason before skipping, for example in
bpf_iter.c:
printf("%s:SKIP:no __builtin_btf_type_id\n", __func__);
test__skip();
Could this print why it skipped, for example that page_owner debugfs is
unavailable and needs CONFIG_PAGE_OWNER plus page_owner=on?
That said, test_progs still reports the test as skipped in its summary, and
many existing prog_tests call test__skip() without a message, so only the
reason is missing here.
> +
> + skel = bpf_iter_page_owner__open_and_load();
> + if (!ASSERT_OK_PTR(skel, "bpf_iter_page_owner__open_and_load"))
> + return;
> +
> + /* --- seq file iterator: unfiltered run --- */
> + skel->bss->filter_en = 0;
> +
> + if (!ASSERT_OK(bpf_iter_page_owner__attach(skel), "skel_attach"))
> + goto destroy;
> +
> + link_fd = bpf_link__fd(skel->links.dump_page_owner);
> + iter_count = iter_record_count(link_fd);
> + if (iter_count < 0)
> + goto destroy;
If reading the iterator fails, can this test still report success without
running any of its checks?
iter_record_count() only records an assertion failure when bpf_iter_create()
fails. When fdopen() fails, or when getline() stops on a read error so that
ferror(f) is set, it quietly returns -1.
The two callers in serial_test_page_owner_iter(), the unfiltered pass here
and the pid-filtered pass below, then do:
if (iter_count < 0)
goto destroy;
Neither calls ASSERT_*() or test__fail(), so test_progs marks the test as
OK and skips the unfiltered, filtered, open-coded and memcg checks.
read() on a BPF iterator fd can fail with real errors from bpf_seq_read():
an ERR_PTR returned by the target's start()/next(), -E2BIG, -ENOMEM, or
-EAGAIN.
The -EAGAIN case looks like a likely trigger in the pid-filtered pass.
There dump_page_owner() returns 0 without printing for every page that does
not belong to the test's pid, and bpf_seq_read() returns -EAGAIN once
num_objs reaches MAX_ITER_OBJECTS (1000000) while offs == 0:
if (num_objs >= MAX_ITER_OBJECTS) {
if (offs == 0) {
err = -EAGAIN;
...
So on a machine where more than about 1M consecutive iterated pages do not
belong to the test process, the filtered pass would end in a silent pass.
Should these error paths fail the test, for example with ASSERT_OK_PTR() on
the FILE * and an assertion on ferror()? Should -EAGAIN be retried rather
than treated as an error?
[ ... ]
> diff --git a/tools/testing/selftests/bpf/progs/bpf_iter_page_owner.c b/tools/testing/selftests/bpf/progs/bpf_iter_page_owner.c
> new file mode 100644
> index 000000000000..fada6a314a15
> --- /dev/null
> +++ b/tools/testing/selftests/bpf/progs/bpf_iter_page_owner.c
> @@ -0,0 +1,161 @@
> +// SPDX-License-Identifier: GPL-2.0
> +/* Copyright (c) 2026 */
> +
> +#include <vmlinux.h>
> +#include <bpf/bpf_helpers.h>
[ ... ]
> +SEC("syscall")
> +int scan_open_coded(const void *ctx)
> +{
> + struct bpf_iter_page_owner it;
> + struct page_owner_scan *scan;
Does this program build against the kernels the BPF selftests are normally
built and run on?
The program needs several types from vmlinux.h: struct bpf_iter_page_owner
(declared as a stack variable here), struct page_owner_scan, struct
memcg_info, struct bpf_iter__page_owner, and struct page_owner (dereferenced
as po->pid and po->comm in dump_page_owner()).
struct page_owner is defined only in mm/page_owner.c, which is built only
when CONFIG_PAGE_OWNER=y (obj-$(CONFIG_PAGE_OWNER) += page_owner.o in
mm/Makefile). The new iterator and kfunc types live in the page_owner code
added by the earlier patches in this series, so they are in BTF only when
CONFIG_PAGE_OWNER is enabled too.
CONFIG_PAGE_OWNER is default n (mm/Kconfig.debug), and none of
tools/testing/selftests/bpf/config, config.vm or config.<arch> enables it.
This patch does not touch those files either.
The selftests build compiles every progs/*.c unconditionally:
Makefile.skel:
BPF_SRCS := $(notdir $(wildcard progs/*.c))
BPF_OBJS := $(patsubst %.c,$(RDIR)/%.bpf.o,$(BPF_SRCS))
Makefile:
$(OUTPUT)/test_progs: $(RUNNER_PREREQS) $(BPF_OBJS) $(ALL_SKELS) FORCE
vmlinux.h is generated from the VMLINUX_BTF of the kernel under test. On a
kernel built from the selftests config, this declaration fails with an
incomplete-type error:
struct bpf_iter_page_owner it;
and ctx->po / po->pid fail the same way. That would fail the whole
test_progs build, not just this test.
In the tree this patch was checked against, the earlier page_owner patches
are not present either, so the file cannot compile there at all.
The runtime check in page_owner_available() cannot help, because the
failure happens at build time.
Should CONFIG_PAGE_OWNER=y (plus its DEBUG_KERNEL dependency) be added to
tools/testing/selftests/bpf/config? Or should the program use local type
definitions with core relocations (___local flavors plus
bpf_core_type_exists()) so it still compiles when the types are missing?
Also, pageowner_init() creates /sys/kernel/debug/page_owner only when booted
with page_owner=on. Without that parameter the test will always skip on
test machines, even after the config is fixed.
---
AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
CI run summary: https://github.com/kernel-patches/bpf/actions/runs/37933988642
^ permalink raw reply [flat|nested] 7+ messages in thread