* [RFC PATCH v4 1/9] mm/damon/paddr: remove page_fault access check primitive
2026-10-05 8:46 [RFC PATCH v4 0/9] mm/damon: hardware-sampled access reports Ravi Jonnalagadda
@ 2026-10-05 8:46 ` Ravi Jonnalagadda
2026-10-05 8:46 ` [RFC PATCH v4 2/9] mm/damon/core: replace the access report buffer with per-context rings Ravi Jonnalagadda
` (8 subsequent siblings)
9 siblings, 0 replies; 11+ messages in thread
From: Ravi Jonnalagadda @ 2026-10-05 8:46 UTC (permalink / raw)
To: SJ Park, Andrew Morton
Cc: damon, linux-mm, linux-kernel, Gregory Price, David Rientjes,
Wei Xu, Jonathan Corbet, Bijan Tabatabai, Ajay Joshi,
Honggyu Kim, Yunjeong Mun, Akinobu Mita, Lian Wang, Kunwu Chan,
Ravi Jonnalagadda, Jonathan Cameron
The page_fault access check primitive reports faults through
damon_report_page_fault() into the global, mutex-protected
damon_access_reports[] buffer. A later patch in this series replaces that
buffer with per-context rings fed only by perf-event probes, which leaves
the page-fault producer with nowhere to report. Remove the producer here
so no report is silently dropped.
Remove damon_pa_prepare_access_checks_faults() and its supporting
damon_pa_change_protection()/damon_pa_change_protection_one() helpers,
and the page_fault dispatch in damon_pa_prepare_access_checks(). Remove
damon_report_page_fault() and its sole caller, do_damon_page(), along
with the two page-fault-handler dispatch sites in mm/memory.c that
selected it over the ordinary NUMA-hinting fault path.
Writing Y to the sysfs page_fault file now returns -EOPNOTSUPP, and core
validation rejects a context with page_fault set, so a configuration
cannot select a primitive that no longer reports and have every region
read as cold. With page_fault gone, page_table is the only access check
primitive left, so a context must enable it. This leaves MM_CP_DAMON
with no user that sets it; its check in mm/mprotect.c is left for a
separate cleanup.
Signed-off-by: Ravi Jonnalagadda <ravis.opensrc@gmail.com>
---
include/linux/damon.h | 12 +----------
mm/damon/core.c | 24 +++------------------
mm/damon/paddr.c | 57 -------------------------------------------------
mm/damon/sysfs-sample.c | 7 +++---
mm/memory.c | 53 ---------------------------------------------
5 files changed, 8 insertions(+), 145 deletions(-)
diff --git a/include/linux/damon.h b/include/linux/damon.h
index 3d0c05df3258..7940840b4da9 100644
--- a/include/linux/damon.h
+++ b/include/linux/damon.h
@@ -930,7 +930,7 @@ struct damon_attrs {
* struct damon_primitives_enabled - Enablement of access sampling primitives.
*
* @page_table: Page table Accessed bits scanning.
- * @page_fault: Page faults monitoring.
+ * @page_fault: Page faults monitoring. Not supported; must be false.
*
* Read &struct damon_sample_control for more details.
*/
@@ -1299,13 +1299,6 @@ int damon_call(struct damon_ctx *ctx, struct damon_call_control *control);
int damos_walk(struct damon_ctx *ctx, struct damos_walk_control *control);
void damon_report_access(struct damon_access_report *report);
-#ifdef CONFIG_MMU
-void damon_report_page_fault(struct vm_fault *vmf, bool huge_pmd);
-#else
-static inline void damon_report_page_fault(struct vm_fault *vmf, bool huge_pmd)
-{
-}
-#endif
int damon_set_region_system_rams_default(struct damon_target *t,
unsigned long *start, unsigned long *end,
@@ -1323,9 +1316,6 @@ unsigned long damon_alloced_bytes(void);
static inline void damon_report_access(struct damon_access_report *report)
{
}
-static inline void damon_report_page_fault(struct vm_fault *vmf, bool huge_pmd)
-{
-}
#endif /* CONFIG_DAMON */
diff --git a/mm/damon/core.c b/mm/damon/core.c
index ddf9c08aa6c1..3638e2054030 100644
--- a/mm/damon/core.c
+++ b/mm/damon/core.c
@@ -2087,7 +2087,9 @@ static int damon_commit_sample_filters(struct damon_sample_control *dst,
static bool damon_primitives_enabled_invalid(
struct damon_primitives_enabled *config)
{
- return config->page_table == config->page_fault;
+ if (config->page_fault)
+ return true;
+ return !config->page_table;
}
static int damon_commit_sample_control(
@@ -2544,26 +2546,6 @@ void damon_report_access(struct damon_access_report *report)
mutex_unlock(&damon_access_reports_lock);
}
-#ifdef CONFIG_MMU
-void damon_report_page_fault(struct vm_fault *vmf, bool huge_pmd)
-{
- struct damon_access_report access_report = {
- .vaddr = vmf->address,
- .size = 1, /* todo: set appripriately */
- .cpu = smp_processor_id(),
- .tid = task_pid_vnr(current),
- .is_write = vmf->flags & FAULT_FLAG_WRITE,
- };
-
- if (huge_pmd)
- access_report.paddr = PFN_PHYS(pmd_pfn(vmf->orig_pmd));
- else
- access_report.paddr = PFN_PHYS(pte_pfn(vmf->orig_pte));
-
- damon_report_access(&access_report);
-}
-#endif
-
/*
* Reset the aggregated monitoring results ('nr_accesses' of each region).
*/
diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c
index f4fa7c231e55..65a5b3269d1d 100644
--- a/mm/damon/paddr.c
+++ b/mm/damon/paddr.c
@@ -67,67 +67,10 @@ static void damon_pa_prepare_access_checks_abit(struct damon_ctx *ctx)
}
}
-static bool damon_pa_change_protection_one(struct folio *folio,
- struct vm_area_struct *vma, unsigned long addr, void *arg)
-{
- /* todo: batch or remove tlb flushing */
- struct mmu_gather tlb;
-
- if (!vma_is_accessible(vma))
- return true;
-
- tlb_gather_mmu(&tlb, vma->vm_mm);
-
- change_protection(&tlb, vma, addr, addr + PAGE_SIZE, MM_CP_DAMON);
-
- tlb_finish_mmu(&tlb);
- return true;
-}
-
-static void damon_pa_change_protection(unsigned long paddr)
-{
- struct folio *folio = damon_get_folio(PHYS_PFN(paddr));
- struct rmap_walk_control rwc = {
- .rmap_one = damon_pa_change_protection_one,
- .anon_lock = folio_lock_anon_vma_read,
- };
- bool need_lock;
-
- if (!folio)
- return;
- if (!folio_mapped(folio) || !folio_raw_mapping(folio))
- return;
-
- need_lock = !folio_test_anon(folio) || folio_test_ksm(folio);
- if (need_lock && !folio_trylock(folio))
- return;
-
- rmap_walk(folio, &rwc);
-
- if (need_lock)
- folio_unlock(folio);
-}
-
-static void damon_pa_prepare_access_checks_faults(struct damon_ctx *ctx)
-{
- struct damon_target *t;
- struct damon_region *r;
-
- damon_for_each_target(t, ctx) {
- damon_for_each_region(r, t) {
- r->sampling_addr = damon_rand(ctx, r->ar.start,
- r->ar.end);
- damon_pa_change_protection(r->sampling_addr);
- }
- }
-}
-
static void damon_pa_prepare_access_checks(struct damon_ctx *ctx)
{
if (ctx->sample_control.primitives_enabled.page_table)
damon_pa_prepare_access_checks_abit(ctx);
- if (ctx->sample_control.primitives_enabled.page_fault)
- damon_pa_prepare_access_checks_faults(ctx);
}
static bool damon_pa_young(phys_addr_t paddr)
diff --git a/mm/damon/sysfs-sample.c b/mm/damon/sysfs-sample.c
index ffc9c8545547..27f35cbb509f 100644
--- a/mm/damon/sysfs-sample.c
+++ b/mm/damon/sysfs-sample.c
@@ -421,7 +421,9 @@ static ssize_t page_fault_store(struct kobject *kobj,
if (err)
return err;
- primitives->page_fault = enable;
+ if (enable)
+ return -EOPNOTSUPP;
+ primitives->page_fault = false;
return count;
}
@@ -590,8 +592,7 @@ int damon_sysfs_set_sample_control(
{
control->primitives_enabled.page_table =
sysfs_sample->primitives->page_table;
- control->primitives_enabled.page_fault =
- sysfs_sample->primitives->page_fault;
+ control->primitives_enabled.page_fault = false;
return damon_sysfs_set_sample_filters(control,
sysfs_sample->filters);
diff --git a/mm/memory.c b/mm/memory.c
index 41278e32dde6..44034d5b32ab 100644
--- a/mm/memory.c
+++ b/mm/memory.c
@@ -6565,54 +6565,6 @@ static void fix_spurious_fault(struct vm_fault *vmf,
}
}
-/*
- * NOTE: This is only poc purpose "hack" that will not be upstreamed as is.
- * More discussions between all stakeholders including maintainers of MM core,
- * NUMA balancing, and DAMON should be made to make this upstreamable.
- * (https://lore.kernel.org/20251128193947.80866-1-sj@kernel.org)
- *
- * This function is called from page fault handler, for page faults on
- * P{TE,MD}-protected but vma-accessible pages. DAMON is making the fake
- * protection for access sampling purpose. This function simply clear the
- * protection and report this access to DAMON, by calling
- * damon_report_page_fault().
- *
- * The protection clear code is copied from NUMA fault handling code for PTE.
- * Again, this is only poc purpose "hack" to show what information DAMON want
- * from page fault events, rather than an upstream-aimed version.
- */
-static vm_fault_t do_damon_page(struct vm_fault *vmf, bool huge_pmd)
-{
- struct vm_area_struct *vma = vmf->vma;
- struct folio *folio;
- pte_t pte, old_pte;
- bool writable = false, ignore_writable = false;
- bool pte_write_upgrade = vma_wants_manual_pte_write_upgrade(vma);
-
- spin_lock(vmf->ptl);
- old_pte = ptep_get(vmf->pte);
- if (unlikely(!pte_same(old_pte, vmf->orig_pte))) {
- pte_unmap_unlock(vmf->pte, vmf->ptl);
- return 0;
- }
- pte = pte_modify(old_pte, vma->vm_page_prot);
- writable = pte_write(pte);
- if (!writable && pte_write_upgrade &&
- can_change_pte_writable(vma, vmf->address, pte))
- writable = true;
- folio = vm_normal_folio(vma, vmf->address, pte);
- if (folio && folio_test_large(folio))
- numa_rebuild_large_mapping(vmf, vma, folio, pte,
- ignore_writable, pte_write_upgrade);
- else
- numa_rebuild_single_mapping(vmf, vma, vmf->address, vmf->pte,
- writable);
- pte_unmap_unlock(vmf->pte, vmf->ptl);
-
- damon_report_page_fault(vmf, huge_pmd);
- return 0;
-}
-
/*
* These routines also need to handle stuff like marking pages dirty
* and/or accessed for architectures that don't do it in hardware (most
@@ -6686,8 +6638,6 @@ static vm_fault_t handle_pte_fault(struct vm_fault *vmf)
*/
if (userfaultfd_pte_rwp(vmf->vma, vmf->orig_pte))
return do_uffd_rwp(vmf);
- if (sysctl_numa_balancing_mode == NUMA_BALANCING_DISABLED)
- return do_damon_page(vmf, false);
return do_numa_page(vmf);
}
@@ -6806,9 +6756,6 @@ static vm_fault_t __handle_mm_fault(struct vm_area_struct *vma,
if (pmd_protnone(vmf.orig_pmd) && vma_is_accessible(vma)) {
if (userfaultfd_huge_pmd_rwp(vma, vmf.orig_pmd))
return do_huge_pmd_uffd_rwp(&vmf);
- if (sysctl_numa_balancing_mode ==
- NUMA_BALANCING_DISABLED)
- return do_damon_page(&vmf, true);
return do_huge_pmd_numa_page(&vmf);
}
--
Git-157)
^ permalink raw reply [flat|nested] 11+ messages in thread* [RFC PATCH v4 2/9] mm/damon/core: replace the access report buffer with per-context rings
2026-10-05 8:46 [RFC PATCH v4 0/9] mm/damon: hardware-sampled access reports Ravi Jonnalagadda
2026-10-05 8:46 ` [RFC PATCH v4 1/9] mm/damon/paddr: remove page_fault access check primitive Ravi Jonnalagadda
@ 2026-10-05 8:46 ` Ravi Jonnalagadda
2026-10-05 8:46 ` [RFC PATCH v4 3/9] mm/damon: add perf-event overflow handler feeding the report ring Ravi Jonnalagadda
` (7 subsequent siblings)
9 siblings, 0 replies; 11+ messages in thread
From: Ravi Jonnalagadda @ 2026-10-05 8:46 UTC (permalink / raw)
To: SJ Park, Andrew Morton
Cc: damon, linux-mm, linux-kernel, Gregory Price, David Rientjes,
Wei Xu, Jonathan Corbet, Bijan Tabatabai, Ajay Joshi,
Honggyu Kim, Yunjeong Mun, Akinobu Mita, Lian Wang, Kunwu Chan,
Ravi Jonnalagadda, Jonathan Cameron
damon_report_access() queues reports into the global damon_access_reports[]
buffer under a mutex, and kdamond applies them from there. A perf-event
overflow handler runs in NMI context and cannot take that mutex, and the
buffer has no other producer.
Replace the buffer with a per-context, per-CPU SPSC report ring
(ctx->perf_rings) that an NMI-context producer can publish into, and a
kdamond drain that credits regions from each ring's pending reports.
NMI safety: the producer uses a busy counter to drop re-entrant reports on
the same CPU, publishes with smp_wmb() before advancing the head, and sets
a pending-CPU bitmask with smp_mb__before_atomic() so the consumer catches
any report published between the bit-clear and the READ_ONCE(head). The
consumer publishes the tail with smp_store_release() after reading the
entries, and the producer reads it with smp_load_acquire(), so a slot is
not reused while it is still being read. A source allocates the ring
before it arms its first perf event for the context, and
damon_destroy_ctx() frees it after the events are released, so no
in-flight NMI can reach freed storage. Per-CPU counters count the
reports dropped because a ring was full or its busy guard was held, the
reports drained into a region and those that matched none; the
damon_get_*() accessors read them for the kunit tests.
The drain matches each report to a region by binary search over a
per-target region snapshot built in ar.start order, the region-list
invariant damon_credit_report_bsearch() relies on. Pid targets are
matched by the thread group id of the target's task, so a target named
by any of its threads matches, compared as the global pid number, since
a producer may run in any pid namespace. Each drained report adds one
to the region's probe_hits[] slot for its probe, at most once per
sampling interval, the same bound a probe hit has for the samples DAMON
takes itself; that keeps probe_hits[] within the samples per
aggregation, as the code that splits, merges and rescales it assumes.
A physical address is matched in the context's address unit.
The reports are data attribute samples, so the drain does not change
nr_accesses, which stays with the access check primitives. That leaves
the access_reported flag and kdamond_apply_zero_access_report(), which
served only the page fault primitive's reports, with no user; remove
them.
Signed-off-by: Ravi Jonnalagadda <ravis.opensrc@gmail.com>
---
include/linux/damon.h | 102 ++++++++-
mm/damon/core.c | 591 ++++++++++++++++++++++++++++++++++++++++++--------
2 files changed, 605 insertions(+), 88 deletions(-)
diff --git a/include/linux/damon.h b/include/linux/damon.h
index 7940840b4da9..e6d2e9d0936e 100644
--- a/include/linux/damon.h
+++ b/include/linux/damon.h
@@ -17,9 +17,19 @@
#define DAMON_MIN_REGION_SZ PAGE_SIZE
/* Maximum number of monitoring probes. */
#define DAMON_MAX_PROBES (4)
+/*
+ * Sentinel value for damon_access_report.probe_idx: 0 means no probe
+ * attribution (matches zero-init of struct damon_access_report on the stack).
+ * Perf-event probe indices start at 1.
+ */
+#define DAMON_PROBE_IDX_NONE 0
/* Max priority score for DAMON-based operation schemes */
#define DAMOS_MAX_SCORE (99)
+/* Per-CPU SPSC ring: size must be a power of two. */
+#define DAMON_REPORT_RING_SIZE 256
+#define DAMON_REPORT_RING_MASK (DAMON_REPORT_RING_SIZE - 1)
+
/**
* struct damon_addr_range - Represents an address region of [@start, @end).
* @start: Start address of the region (inclusive).
@@ -74,7 +84,9 @@ struct damon_region {
/* for age calculation. */
unsigned int last_nr_accesses;
unsigned char last_probe_hits[DAMON_MAX_PROBES];
- bool access_reported;
+ /* probes already credited in sampling interval probes_reported_sis */
+ unsigned long probes_reported_sis;
+ unsigned char probes_reported;
};
/**
@@ -110,7 +122,18 @@ struct damon_target {
* @size: The size of the accessed address range.
* @cpu: The id of the CPU that made the access.
* @tid: The task id of the task that made the access.
+ * @tgid: The thread group id of the task that made the access, in
+ * the initial pid namespace. A monitoring target created
+ * for a process is matched against it.
* @is_write: Whether the access is write.
+ * @probe_idx: 1-based index of the reporting probe; the drain credits
+ * probe_hits[@probe_idx - 1]. Set by the reporting
+ * source, so the drain needs no list walk.
+ * 0 is reserved (no probe attribution; matches zero-init);
+ * perf-event probe indices start at 1.
+ * @ctx: Context whose report ring receives the report; set by
+ * the reporting source. A report with no context is
+ * dropped by damon_report_access().
*
* Any DAMON API callers that notified access events can report the information
* to DAMON using damon_report_access(). This struct contains the reporting
@@ -122,11 +145,49 @@ struct damon_access_report {
unsigned long size;
unsigned int cpu;
pid_t tid;
+ pid_t tgid;
bool is_write;
+ int probe_idx;
+ struct damon_ctx *ctx;
/* private: */
unsigned long report_jiffies; /* when this report is made */
};
+/**
+ * struct damon_report_ring - Per-CPU SPSC ring for NMI-safe access reports.
+ *
+ * @head: Write index; updated by the NMI producer.
+ * @tail: Read index; updated by the kdamond consumer.
+ * @entries: Ring buffer entries.
+ *
+ * One ring per CPU, per context with a perf-event probe. The producer
+ * (NMI overflow handler) writes to @head; the consumer (kdamond) reads
+ * from @tail. Both indices are unsigned and wrap modulo
+ * DAMON_REPORT_RING_SIZE.
+ */
+struct damon_report_ring {
+ unsigned int head; /* written by producer (NMI) */
+ unsigned int tail /* written by consumer (kdamond) */
+ ____cacheline_aligned_in_smp;
+ struct damon_access_report entries[DAMON_REPORT_RING_SIZE]
+ ____cacheline_aligned_in_smp;
+};
+
+/*
+ * struct damon_target_lookup - Cached, sorted region snapshot for one target.
+ * @regions: Array of region pointers, sorted by ar.start (address order).
+ * @nr_regions: Number of entries in @regions.
+ * @tgid: Thread group id of a pid target's task; 0 if it has none.
+ *
+ * Built once per drain by damon_build_target_lookup() so the ring drain can
+ * binary-search a target's regions instead of walking the list.
+ */
+struct damon_target_lookup {
+ struct damon_region **regions;
+ unsigned int nr_regions;
+ pid_t tgid;
+};
+
/**
* enum damos_action - Represents an action of a Data Access Monitoring-based
* Operation Scheme.
@@ -871,9 +932,12 @@ struct damon_filter {
* struct damon_probe - Data region attribute probe.
*
* @weight: Relative priority of the attribute for this probe.
+ * @event_driven: Whether the probe's hits arrive through the report ring
+ * drain rather than the apply_probes callback.
*/
struct damon_probe {
unsigned int weight;
+ bool event_driven;
/* private: */
/* Preparation actions to apply to each probing memory. */
struct list_head preps;
@@ -1091,6 +1155,29 @@ struct damon_ctx {
/* @rnd_state: Per-ctx PRNG state for damon_rand(). */
struct rnd_state rnd_state;
+
+ /* Reusable drain-loop snapshot buffer (avoids per-tick kmalloc). */
+ struct {
+ struct damon_target_lookup *lookups;
+ unsigned int nr_lookups;
+ struct damon_region **region_buf;
+ unsigned int region_buf_cap;
+ } drain_snapshot;
+
+ /*
+ * Per-context perf-event report ring. A perf overflow handler carries
+ * a pointer to the ctx that armed it (damon_access_report.ctx), so its
+ * reports route to this ring and two perf-driven ctxs never share one.
+ *
+ * A source allocates it with damon_ctx_alloc_perf_ring() before it
+ * arms the first perf event of the ctx, and damon_destroy_ctx() frees
+ * it after the events are released, so no in-flight NMI can reach
+ * freed storage.
+ * While perf_rings is NULL, every perf report for the ctx is dropped.
+ */
+ struct damon_report_ring __percpu *perf_rings;
+ int __percpu *perf_ring_busy;
+ cpumask_t perf_pending;
};
/* Get a random number in [@l, @r) using @ctx's lockless PRNG. */
@@ -1209,6 +1296,7 @@ void damon_destroy_filter(struct damon_filter *f);
struct damon_probe *damon_new_probe(void);
void damon_add_probe(struct damon_ctx *ctx, struct damon_probe *probe);
+bool damon_has_event_driven_probes(struct damon_ctx *ctx);
struct damon_region *damon_new_region(unsigned long start, unsigned long end);
unsigned int damon_nr_accesses_mvsum(struct damon_region *r,
@@ -1298,13 +1386,20 @@ int damon_kdamond_pid(struct damon_ctx *ctx);
int damon_call(struct damon_ctx *ctx, struct damon_call_control *control);
int damos_walk(struct damon_ctx *ctx, struct damos_walk_control *control);
-void damon_report_access(struct damon_access_report *report);
+bool damon_report_access(struct damon_access_report *report);
+int damon_ctx_alloc_perf_ring(struct damon_ctx *ctx);
int damon_set_region_system_rams_default(struct damon_target *t,
unsigned long *start, unsigned long *end,
unsigned long addr_unit,
unsigned long min_region_sz);
+unsigned long damon_get_report_overflow(void);
+unsigned long damon_get_report_ring_full(void);
+unsigned long damon_get_report_busy_drop(void);
+unsigned long damon_get_samples_drained(void);
+unsigned long damon_get_samples_no_region(void);
+
#ifdef CONFIG_ACMA
unsigned long damon_alloced_bytes(void);
@@ -1313,8 +1408,9 @@ unsigned long damon_alloced_bytes(void);
#else /* CONFIG_DAMON */
-static inline void damon_report_access(struct damon_access_report *report)
+static inline bool damon_report_access(struct damon_access_report *report)
{
+ return false;
}
#endif /* CONFIG_DAMON */
diff --git a/mm/damon/core.c b/mm/damon/core.c
index 3638e2054030..8f78700afedd 100644
--- a/mm/damon/core.c
+++ b/mm/damon/core.c
@@ -22,7 +22,66 @@
#define CREATE_TRACE_POINTS
#include <trace/events/damon.h>
-#define DAMON_ACCESS_REPORTS_CAP 1000
+/*
+ * Reports are fed to DAMON through a per-context, per-CPU SPSC ring
+ * (ctx->perf_rings). A report carries its owning ctx, so each context drains
+ * only its own ring.
+ */
+/*
+ * Report drops are counted per reason. A ring-full drop means the consumer
+ * did not keep up with the producer; a busy-guard drop means an NMI nested on
+ * top of a same-CPU producer.
+ */
+static DEFINE_PER_CPU(unsigned long, damon_report_ring_full_perf);
+static DEFINE_PER_CPU(unsigned long, damon_report_busy_drop_perf);
+static DEFINE_PER_CPU(unsigned long, damon_samples_drained);
+static DEFINE_PER_CPU(unsigned long, damon_samples_no_region);
+
+unsigned long damon_get_report_ring_full(void)
+{
+ unsigned long sum = 0;
+ int cpu;
+
+ for_each_possible_cpu(cpu)
+ sum += per_cpu(damon_report_ring_full_perf, cpu);
+ return sum;
+}
+
+unsigned long damon_get_report_busy_drop(void)
+{
+ unsigned long sum = 0;
+ int cpu;
+
+ for_each_possible_cpu(cpu)
+ sum += per_cpu(damon_report_busy_drop_perf, cpu);
+ return sum;
+}
+
+/* Reports dropped for either reason. */
+unsigned long damon_get_report_overflow(void)
+{
+ return damon_get_report_ring_full() + damon_get_report_busy_drop();
+}
+
+unsigned long damon_get_samples_drained(void)
+{
+ unsigned long sum = 0;
+ int cpu;
+
+ for_each_possible_cpu(cpu)
+ sum += per_cpu(damon_samples_drained, cpu);
+ return sum;
+}
+
+unsigned long damon_get_samples_no_region(void)
+{
+ unsigned long sum = 0;
+ int cpu;
+
+ for_each_possible_cpu(cpu)
+ sum += per_cpu(damon_samples_no_region, cpu);
+ return sum;
+}
static DEFINE_MUTEX(damon_lock);
static int nr_running_ctxs;
@@ -33,11 +92,6 @@ static struct damon_operations damon_registered_ops[NR_DAMON_OPS];
static struct kmem_cache *damon_region_cache __ro_after_init;
-static DEFINE_MUTEX(damon_access_reports_lock);
-static struct damon_access_report damon_access_reports[
- DAMON_ACCESS_REPORTS_CAP];
-static int damon_access_reports_len;
-
/* Should be called under damon_ops_lock with id smaller than NR_DAMON_OPS */
static bool __damon_is_registered_ops(enum damon_ops_id id)
{
@@ -288,6 +342,34 @@ static bool damon_has_probe_weights(struct damon_ctx *c)
return false;
}
+/**
+ * damon_has_event_driven_probes() - return true if @ctx has any event-driven
+ * probes registered.
+ * @ctx: the DAMON context whose probes are inspected.
+ *
+ * Event-driven probes (e.g. perf-event IBS/PEBS) populate probe_hits[] via
+ * the SPSC ring drain rather than the apply_probes vtable. A context drains
+ * its perf report ring only when this returns true.
+ *
+ * Return: true if @ctx has an event-driven probe.
+ */
+bool damon_has_event_driven_probes(struct damon_ctx *ctx)
+{
+ struct damon_probe *p;
+
+ damon_for_each_probe(p, ctx) {
+ if (p->event_driven)
+ return true;
+ }
+ return false;
+}
+
+/* Does @ctx drain a perf report ring? */
+static bool damon_drains_ring_perf(struct damon_ctx *ctx)
+{
+ return damon_has_event_driven_probes(ctx);
+}
+
/*
* damon_mvsum() - Returns pseudo moving sum value for a time window.
* @current_nr: The value of the current aggregation window.
@@ -414,7 +496,8 @@ struct damon_region *damon_new_region(unsigned long start, unsigned long end)
region->age = 0;
region->last_nr_accesses = 0;
- region->access_reported = false;
+ region->probes_reported_sis = 0;
+ region->probes_reported = 0;
return region;
}
@@ -1025,6 +1108,38 @@ struct damon_ctx *damon_new_ctx(void)
return ctx;
}
+/*
+ * Lazily allocate the per-ctx perf report ring. Call before arming any perf
+ * event that reports into @ctx, so an overflow can never observe a half-built
+ * ring. Idempotent: a ctx with several perf probes allocates once. The ring
+ * is freed in damon_destroy_ctx().
+ */
+int damon_ctx_alloc_perf_ring(struct damon_ctx *ctx)
+{
+ if (ctx->perf_rings)
+ return 0; /* already allocated for an earlier probe */
+ ctx->perf_rings = alloc_percpu(struct damon_report_ring);
+ if (!ctx->perf_rings)
+ return -ENOMEM;
+ ctx->perf_ring_busy = alloc_percpu(int);
+ if (!ctx->perf_ring_busy) {
+ free_percpu(ctx->perf_rings);
+ ctx->perf_rings = NULL;
+ return -ENOMEM;
+ }
+ cpumask_clear(&ctx->perf_pending);
+ return 0;
+}
+
+/* Free the per-ctx perf ring. Caller must ensure no perf event is armed. */
+static void damon_ctx_free_perf_ring(struct damon_ctx *ctx)
+{
+ free_percpu(ctx->perf_rings);
+ ctx->perf_rings = NULL;
+ free_percpu(ctx->perf_ring_busy);
+ ctx->perf_ring_busy = NULL;
+}
+
static void damon_destroy_targets(struct damon_ctx *ctx)
{
struct damon_target *t, *next_t;
@@ -1050,6 +1165,12 @@ void damon_destroy_ctx(struct damon_ctx *ctx)
damon_for_each_sample_filter_safe(f, next_f, &ctx->sample_control)
damon_destroy_sample_filter(f, &ctx->sample_control);
+ /* No-op if never allocated. */
+ damon_ctx_free_perf_ring(ctx);
+
+ /* Free the reusable ring-drain region snapshot buffers. */
+ kfree(ctx->drain_snapshot.lookups);
+ kfree(ctx->drain_snapshot.region_buf);
kfree(ctx);
}
@@ -2521,29 +2642,102 @@ int damos_walk(struct damon_ctx *ctx, struct damos_walk_control *control)
* damon_report_access() - Report identified access events to DAMON.
* @report: The reporting access information.
*
- * Report access events to DAMON.
+ * Report access events to DAMON via a per-context per-CPU SPSC lockless ring
+ * (ctx->perf_rings). Producer is the local CPU (typically NMI from a
+ * hardware-sampling backend); consumer is the kdamond drain in
+ * kdamond_check_reported_accesses().
+ *
+ * The destination ring is selected by this_cpu_ptr(), i.e. by the CPU calling
+ * this function, not by @report->cpu, which is sample metadata used by the
+ * drain-side filter. The two coincide for a sample delivered by an interrupt
+ * on the CPU that produced it.
*
- * Context: May sleep.
+ * A backend whose PMU writes a record stream into a memory buffer instead of
+ * raising a per-sample interrupt, or one reading a device counter table, must
+ * therefore decode CPU N's buffer on CPU N -- for example by queueing per-CPU
+ * work with queue_work_on() -- rather than calling this function in a loop
+ * from one thread. A single-thread loop puts every report in that thread's
+ * ring, which caps machine-wide capacity at DAMON_REPORT_RING_SIZE - 1
+ * reports per drain regardless of the number of producing CPUs.
*
- * NOTE: we may be able to implement this as a lockless queue, and allow any
- * context. As the overhead is unknown, and region-based DAMON logics would
- * guarantee the reports would be not made that frequently, let's start with
- * this simple implementation.
+ * Context: any (NMI-safe). An NMI nesting on top of a process-context
+ * producer on the same CPU would otherwise stomp the same entries[head]
+ * slot; the busy guard detects and drops in that case.
+ *
+ * If the ring is full, the sample is dropped and the per-CPU ring-full
+ * counter incremented; a busy-guard drop increments the busy-drop counter.
+ *
+ * Return: true if the report was queued, false if it was dropped. A producer
+ * holding a single report may ignore this. A producer decoding a batch out
+ * of a hardware buffer should stop on false and leave the remainder in that
+ * buffer for the next round, since a report released from the buffer but not
+ * queued here is not delivered.
*/
-void damon_report_access(struct damon_access_report *report)
+bool damon_report_access(struct damon_access_report *report)
{
- struct damon_access_report *dst;
+ /*
+ * Only perf-event reports (probe_idx >= 1) have a ring to feed. A
+ * probe_idx == DAMON_PROBE_IDX_NONE report has nowhere to go and is
+ * dropped here rather than at each caller.
+ */
+ struct damon_report_ring *ring;
+ cpumask_t *pending;
+ int __percpu *busy_pcpu;
+ unsigned int head, next;
+ int busy;
+ bool queued = false;
+ struct damon_ctx *pctx = report->ctx;
+
+ if (report->probe_idx == DAMON_PROBE_IDX_NONE)
+ return false;
- /* silently fail for races */
- if (!mutex_trylock(&damon_access_reports_lock))
- return;
- dst = &damon_access_reports[damon_access_reports_len++];
- /* just drop all existing reports in favor of simplicity. */
- if (damon_access_reports_len == DAMON_ACCESS_REPORTS_CAP)
- damon_access_reports_len = 0;
- *dst = *report;
- dst->report_jiffies = jiffies;
- mutex_unlock(&damon_access_reports_lock);
+ /*
+ * A perf report must carry its owning ctx (set by the overflow handler)
+ * and that ctx must have allocated its per-ctx perf ring; drop the
+ * report otherwise.
+ */
+ if (!pctx || !pctx->perf_rings || !pctx->perf_ring_busy)
+ return false;
+
+ /* Pin to a CPU so the SPSC invariant holds for preemptible callers. */
+ preempt_disable();
+ busy_pcpu = pctx->perf_ring_busy;
+ busy = this_cpu_inc_return(*busy_pcpu);
+ if (busy != 1) {
+ /* NMI nested on a process-context producer; drop. */
+ this_cpu_inc(damon_report_busy_drop_perf);
+ goto out;
+ }
+
+ ring = this_cpu_ptr(pctx->perf_rings);
+ pending = &pctx->perf_pending;
+ head = ring->head;
+ next = (head + 1) & DAMON_REPORT_RING_MASK;
+
+ /* pairs with the consumer's smp_store_release() of tail */
+ if (next == smp_load_acquire(&ring->tail)) {
+ this_cpu_inc(damon_report_ring_full_perf);
+ goto out;
+ }
+
+ ring->entries[head] = *report;
+ ring->entries[head].report_jiffies = jiffies;
+ smp_wmb(); /* publish entry before head advance */
+ WRITE_ONCE(ring->head, next);
+ /*
+ * Order the head advance before publishing the pending bit so
+ * that the consumer, on observing the bit, is also guaranteed
+ * to observe the new head. cpumask_set_cpu / set_bit are
+ * documented as unordered RMW (atomic_bitops.txt), hence the
+ * explicit barrier.
+ */
+ smp_mb__before_atomic();
+ cpumask_set_cpu(smp_processor_id(), pending);
+ queued = true;
+out:
+ this_cpu_dec(*busy_pcpu);
+ preempt_enable();
+ return queued;
}
/*
@@ -4247,73 +4441,299 @@ static bool damon_sample_filter_out(struct damon_access_report *report,
return !filter->allow;
}
-static void kdamond_apply_access_report(struct damon_access_report *report,
- struct damon_target *t, struct damon_ctx *ctx)
+/*
+ * Build a snapshot of the ctx's targets and their region arrays for use by
+ * the ring drain loop. The snapshot buffer is reused across ticks, grown via
+ * krealloc only when a new high water mark is reached.
+ *
+ * Only the kdamond mutates the target list (other threads go through
+ * damon_call()), so the list cannot change between the two passes, even while
+ * krealloc_array() sleeps. Regions within a target are kept address-sorted by
+ * DAMON, so the snapshot arrays are directly binary-searchable.
+ */
+static struct damon_target_lookup *damon_build_target_lookup(
+ struct damon_ctx *ctx, unsigned int *nr_targets_out)
{
- struct damon_region *r;
- unsigned long addr;
+ struct damon_target *t;
+ struct damon_target_lookup *tbl;
+ unsigned int nr_targets = 0, total_regions = 0, ti = 0, ri = 0;
- if (damon_sample_filter_out(report, &ctx->sample_control))
- return;
- if (damon_target_has_pid(ctx))
- addr = report->vaddr;
- else
- addr = report->paddr;
+ damon_for_each_target(t, ctx) {
+ nr_targets++;
+ total_regions += damon_nr_regions(t);
+ }
+ *nr_targets_out = nr_targets;
+
+ if (nr_targets > ctx->drain_snapshot.nr_lookups) {
+ tbl = krealloc_array(ctx->drain_snapshot.lookups,
+ nr_targets, sizeof(*tbl), GFP_KERNEL);
+ if (!tbl)
+ return NULL;
+ ctx->drain_snapshot.lookups = tbl;
+ ctx->drain_snapshot.nr_lookups = nr_targets;
+ }
+ tbl = ctx->drain_snapshot.lookups;
- /* todo: make search faster, e.g., binary search? */
- damon_for_each_region(r, t) {
- if (addr < r->ar.start)
- continue;
- if (r->ar.end < addr + report->size)
- continue;
- if (!r->access_reported)
- damon_update_region_access_rate(r, true);
- r->access_reported = true;
+ if (total_regions > ctx->drain_snapshot.region_buf_cap) {
+ struct damon_region **buf;
+
+ buf = krealloc_array(ctx->drain_snapshot.region_buf,
+ total_regions, sizeof(*buf), GFP_KERNEL);
+ if (!buf)
+ return NULL;
+ ctx->drain_snapshot.region_buf = buf;
+ ctx->drain_snapshot.region_buf_cap = total_regions;
}
+
+ /*
+ * DAMON maintains each target's region list sorted by ar.start.
+ * damon_credit_report_bsearch() binary-searches by address, so the
+ * snapshot built here must preserve that order. If the region-list
+ * ordering invariant ever changes, this builder must sort explicitly.
+ */
+ damon_for_each_target(t, ctx) {
+ struct damon_region *r;
+
+ tbl[ti].regions = &ctx->drain_snapshot.region_buf[ri];
+ tbl[ti].nr_regions = damon_nr_regions(t);
+ tbl[ti].tgid = 0;
+ if (damon_target_has_pid(ctx)) {
+ struct task_struct *task;
+
+ /* a target may be named by any of its threads */
+ rcu_read_lock();
+ task = pid_task(t->pid, PIDTYPE_PID);
+ if (task)
+ tbl[ti].tgid = task_tgid_nr(task);
+ rcu_read_unlock();
+ }
+ damon_for_each_region(r, t)
+ ctx->drain_snapshot.region_buf[ri++] = r;
+ ti++;
+ }
+
+ return tbl;
}
-static unsigned int kdamond_apply_zero_access_report(struct damon_ctx *ctx)
-{
+/*
+ * Binary-search a sorted region snapshot for the region containing @addr and,
+ * on a hit, add the report to the region's hits of probe @pidx, at most once
+ * per sampling interval @sis. Returns true if a region was found (straddling
+ * reports that spill past the region end are rejected).
+ */
+static bool damon_credit_report_bsearch(struct damon_region **regions,
+ unsigned int nr_regions, unsigned long addr,
+ unsigned long size, int pidx, unsigned long sis)
+{
+ struct damon_region *r = NULL;
+ int left = 0, right = (int)nr_regions - 1, mid;
+
+ while (left <= right) {
+ /* Avoid (left + right) overflow at large nr_regions. */
+ mid = left + (right - left) / 2;
+ if (addr < regions[mid]->ar.start) {
+ right = mid - 1;
+ } else if (addr >= regions[mid]->ar.end) {
+ left = mid + 1;
+ } else {
+ r = regions[mid];
+ break;
+ }
+ }
+ if (!r)
+ return false;
+ /* Reject reports straddling a region boundary. */
+ if (addr + size > r->ar.end)
+ return false;
+
+ /*
+ * pidx is always >= 1 here: __kdamond_drain_ring rejects
+ * DAMON_PROBE_IDX_NONE (0) entries before calling this. Ring
+ * probe_idx is 1-based, but probe_hits[] storage is 0-based to match
+ * all readers (wsum, mvsum, update, aggregate reset, merge).
+ * Convert here: probe_hits[pidx - 1].
+ */
+ if (r->probes_reported_sis != sis) {
+ r->probes_reported_sis = sis;
+ r->probes_reported = 0;
+ }
+ /* at most one hit per interval, as DAMON's own sampling */
+ if (!(r->probes_reported & BIT(pidx - 1))) {
+ r->probe_hits[pidx - 1]++;
+ r->probes_reported |= BIT(pidx - 1);
+ }
+ return true;
+}
+
+/*
+ * __kdamond_drain_ring - drain a per-CPU SPSC ring into region probe_hits.
+ * @ctx: draining context (owns @ring for this run).
+ * @tbl: pre-built sorted per-target region snapshot (shared).
+ * @ring_pcpu: the per-CPU ring base (ctx's perf ring).
+ * @pending: the matching pending cpumask.
+ *
+ * Drops stale reports and reports rejected by the context's sample filters,
+ * matches the rest to a region by address and, for pid targets, by thread
+ * group id, and adds the report to the region's probe hits, at most once per
+ * probe per sampling interval. Reports are data attribute samples, so they do
+ * not change the region's nr_accesses.
+ *
+ * Each ring entry carries its own probe_idx (set by the overflow handler), so
+ * no list walk is needed to resolve the probe_hits[] slot.
+ *
+ * A per-target sorted region snapshot is built once per drain (by the caller)
+ * so each entry is matched to its region via O(log R) binary search rather
+ * than a linear damon_for_each_region() walk. Iterates the ring's pending
+ * cpumask to drain only CPUs with published reports.
+ */
+static void __kdamond_drain_ring(struct damon_ctx *ctx,
+ struct damon_target_lookup *tbl,
+ struct damon_report_ring __percpu *ring_pcpu,
+ cpumask_t *pending)
+{
+ int cpu;
+ struct damon_report_ring *ring;
+ unsigned int tail, head;
+ struct damon_access_report *entry;
struct damon_target *t;
- struct damon_region *r;
- unsigned int max_nr_accesses = 0;
+ unsigned long match_addr, match_size;
+ bool found;
+ unsigned int ti;
- damon_for_each_target(t, ctx) {
- damon_for_each_region(r, t) {
- if (r->access_reported)
- r->access_reported = false;
- else
- damon_update_region_access_rate(r, false);
- max_nr_accesses = max(max_nr_accesses, r->nr_accesses);
+ /*
+ * Unified paddr/vaddr drain. The address space of the monitoring
+ * target selects which address of the report is matched: contexts
+ * whose targets carry a pid are monitoring a virtual address space and
+ * match report->vaddr, the others match report->paddr.
+ *
+ * For pid-target contexts, filter by thread group id so entries from
+ * unrelated processes are not credited to the wrong target.
+ * damon_target_has_pid(ctx) gates this filter: it is false for paddr
+ * ops, whose targets have no pid, so paddr crediting is unfiltered.
+ */
+ for_each_cpu(cpu, pending) {
+ ring = per_cpu_ptr(ring_pcpu, cpu);
+ cpumask_clear_cpu(cpu, pending);
+ /*
+ * Pair with the producer's smp_mb__before_atomic() between
+ * the head publish and cpumask_set_cpu(): order the bit clear
+ * before the head read so a producer publishing between the
+ * clear and the READ_ONCE(head) is observed via the bit it
+ * re-sets, not lost as a stale-head drain.
+ */
+ smp_mb__after_atomic();
+ head = READ_ONCE(ring->head);
+ smp_rmb(); /* pair with smp_wmb in producer */
+ tail = ring->tail;
+
+ while (tail != head) {
+ unsigned long stale_before;
+ int pidx;
+
+ entry = &ring->entries[tail];
+ /*
+ * Entries older than one sampling interval are from
+ * an earlier interval and are dropped.
+ */
+ stale_before = jiffies -
+ usecs_to_jiffies(ctx->attrs.sample_interval);
+ if (time_before(entry->report_jiffies, stale_before))
+ goto next;
+ pidx = entry->probe_idx;
+ /*
+ * Every entry in this ring is a perf-event report
+ * (probe_idx >= 1); damon_report_access() drops any
+ * DAMON_PROBE_IDX_NONE report before it reaches a ring.
+ * Reject only out-of-range indices (> DAMON_MAX_PROBES)
+ * and, defensively, any non-positive value.
+ */
+ if (pidx <= 0 || pidx > DAMON_MAX_PROBES)
+ goto next;
+
+ /* Drop reports rejected by the ctx sample filters. */
+ if (damon_sample_filter_out(entry,
+ &ctx->sample_control))
+ goto next;
+
+ /*
+ * Select the address that matches the address space
+ * the targets of this context are monitoring. A
+ * report that carries no address for that space
+ * cannot be credited.
+ */
+ if (damon_target_has_pid(ctx)) {
+ match_addr = entry->vaddr;
+ match_size = entry->size;
+ } else {
+ /* regions are in addr_unit units */
+ match_addr = entry->paddr / ctx->addr_unit;
+ match_size = max(entry->size / ctx->addr_unit,
+ 1UL);
+ }
+ if (!match_addr)
+ goto next;
+
+ found = false;
+ ti = 0;
+ damon_for_each_target(t, ctx) {
+ /* pid targets: match the tgid of a live task */
+ if (damon_target_has_pid(ctx) &&
+ (!tbl[ti].tgid ||
+ tbl[ti].tgid != entry->tgid)) {
+ ti++;
+ continue;
+ }
+ if (damon_credit_report_bsearch(tbl[ti].regions,
+ tbl[ti].nr_regions, match_addr,
+ match_size, pidx,
+ ctx->passed_sample_intervals)) {
+ this_cpu_inc(damon_samples_drained);
+ found = true;
+ break;
+ }
+ ti++;
+ }
+ if (!found)
+ this_cpu_inc(damon_samples_no_region);
+next:
+ tail = (tail + 1) & DAMON_REPORT_RING_MASK;
}
+ /* finish reading entries before the producer reuses them */
+ smp_store_release(&ring->tail, tail);
}
- return max_nr_accesses;
}
-static unsigned int kdamond_check_reported_accesses(struct damon_ctx *ctx)
+/*
+ * kdamond_check_reported_accesses - drain the per-ctx perf report ring this
+ * ctx feeds. Called from kdamond main loop after each sampling interval.
+ *
+ * Each context's per-CPU perf ring (ctx->perf_rings) holds event-driven
+ * probe (probe_idx >= 1) reports. A ctx drains it when it has event-driven
+ * probes registered.
+ *
+ * The per-target sorted region snapshot is built once per drain.
+ */
+static void kdamond_check_reported_accesses(struct damon_ctx *ctx)
{
- int i;
- struct damon_access_report *report;
- struct damon_target *t;
-
- /* currently damon_access_report supports only physical address */
- if (damon_target_has_pid(ctx))
- return 0;
+ struct damon_target_lookup *tbl;
+ unsigned int nr_targets = 0;
- mutex_lock(&damon_access_reports_lock);
- for (i = 0; i < damon_access_reports_len; i++) {
- report = &damon_access_reports[i];
- if (time_before(report->report_jiffies,
- jiffies -
- usecs_to_jiffies(
- ctx->attrs.sample_interval)))
- continue;
- damon_for_each_target(t, ctx)
- kdamond_apply_access_report(report, t, ctx);
+ /*
+ * Build the sorted region snapshot once for this drain. If the alloc
+ * fails, skip this drain.
+ */
+ tbl = damon_build_target_lookup(ctx, &nr_targets);
+ if (!nr_targets)
+ return;
+ if (!tbl) {
+ pr_warn_ratelimited(
+ "damon: target-lookup alloc failed; ring drain skipped this tick\n");
+ return;
}
- mutex_unlock(&damon_access_reports_lock);
- /* For nr_accesses_bp, absence of access should also be reported. */
- return kdamond_apply_zero_access_report(ctx);
+
+ if (damon_drains_ring_perf(ctx))
+ __kdamond_drain_ring(ctx, tbl, ctx->perf_rings,
+ &ctx->perf_pending);
}
/*
@@ -4370,14 +4790,15 @@ static int kdamond_fn(void *data)
kdamond_usleep(sample_interval);
ctx->passed_sample_intervals++;
- if (!access_check_disabled) {
- /* todo: make these non-exclusive */
- if (ctx->sample_control.primitives_enabled.page_fault)
- max_merge_score =
- kdamond_check_reported_accesses(ctx);
- else if (ctx->ops.check_accesses)
- max_merge_score = ctx->ops.check_accesses(ctx);
- }
+ /*
+ * Perf-event probes feed damon_report_access() into the per-ctx
+ * ring; drain it here.
+ */
+ if (damon_drains_ring_perf(ctx))
+ kdamond_check_reported_accesses(ctx);
+
+ if (!access_check_disabled && ctx->ops.check_accesses)
+ max_merge_score = ctx->ops.check_accesses(ctx);
if (ctx->ops.apply_probes) {
if (time_after_eq(ctx->passed_sample_intervals,
--
Git-157)
^ permalink raw reply [flat|nested] 11+ messages in thread* [RFC PATCH v4 3/9] mm/damon: add perf-event overflow handler feeding the report ring
2026-10-05 8:46 [RFC PATCH v4 0/9] mm/damon: hardware-sampled access reports Ravi Jonnalagadda
2026-10-05 8:46 ` [RFC PATCH v4 1/9] mm/damon/paddr: remove page_fault access check primitive Ravi Jonnalagadda
2026-10-05 8:46 ` [RFC PATCH v4 2/9] mm/damon/core: replace the access report buffer with per-context rings Ravi Jonnalagadda
@ 2026-10-05 8:46 ` Ravi Jonnalagadda
2026-10-05 8:46 ` [RFC PATCH v4 4/9] mm/damon/ops-common: use probe-weighted score when probe weights are set Ravi Jonnalagadda
` (6 subsequent siblings)
9 siblings, 0 replies; 11+ messages in thread
From: Ravi Jonnalagadda @ 2026-10-05 8:46 UTC (permalink / raw)
To: SJ Park, Andrew Morton
Cc: damon, linux-mm, linux-kernel, Gregory Price, David Rientjes,
Wei Xu, Jonathan Corbet, Bijan Tabatabai, Ajay Joshi,
Honggyu Kim, Yunjeong Mun, Akinobu Mita, Lian Wang, Kunwu Chan,
Ravi Jonnalagadda, Jonathan Cameron
Add mm/damon/perf_source.c, an ops-agnostic perf-event source that turns
PMU overflow samples into DAMON access reports. The overflow handler runs
in NMI context and reports through damon_report_access(), so it works with
either paddr or vaddr ops.
Each armed event carries its owning context, set when the probe is set up,
and the handler routes every report to that context's perf ring. Setup
allocates the ring before arming the first counter, so the handler always
finds one. Teardown clears the event's context with a release store
before releasing the counters, and the handler reads it with a matching
acquire load, so an overflow racing teardown drops its sample instead of
reporting into a context that is going away.
The handler sets whichever address fields the PMU provides: paddr when
PERF_SAMPLE_PHYS_ADDR is set, vaddr when PERF_SAMPLE_ADDR is set, plus the
CPU it runs on. Both addresses are page aligned, matching the PAGE_SIZE
report size, so a sample that lands in the last page of a region is not
rejected by the drain as straddling the region end. The drain matches on
whichever address the context's targets use. The handler reports both
the thread id and the thread group id, because a VA-only PMU samples
whichever thread of a process happened to access memory while the drain
matches a target created for that process. The thread group id is the
global one, which is what the drain compares the target's pid with.
Per-CPU perf events are armed and released through cpuhp callbacks, so a
CPU coming online while a probe is armed gets a counter and a CPU going
offline releases its own. A CPU on which the counter cannot be created
is left unsampled with a warning, since failing the online callback
would block the CPU from coming online.
A probe with event_driven set has its hits credited through the
report-ring drain rather than the apply_probes vtable. Both ops sets
honour that flag in their probe vtables: an event-driven probe has no
software prep action, and is skipped in the apply_probes loop, so the
drain is the only thing that credits its probe_hits[]. When no probe of
the context is sampled by DAMON itself, the apply_probes vtable also
skips the per-region folio lookup or page table walk.
damon_perf_probe_setup() computes probe_idx by walking ctx->probes,
assigning 1-based indices because 0 is the zero-init sentinel
DAMON_PROBE_IDX_NONE, and the overflow handler warns once if it sees 0.
A refcounted per-PMU owner tracks which context owns each PMU type: a
second context claiming the same type gets -EBUSY, and ownership is
released when the last probe for that type is torn down.
Co-developed-by: Akinobu Mita <akinobu.mita@gmail.com>
Signed-off-by: Akinobu Mita <akinobu.mita@gmail.com>
Signed-off-by: Ravi Jonnalagadda <ravis.opensrc@gmail.com>
---
include/linux/damon.h | 1 +
mm/damon/Kconfig | 19 +++
mm/damon/Makefile | 1 +
mm/damon/core.c | 18 +++
mm/damon/paddr.c | 15 +-
mm/damon/perf_source.c | 378 +++++++++++++++++++++++++++++++++++++++++++++++++
mm/damon/perf_source.h | 25 ++++
mm/damon/vaddr.c | 15 +-
8 files changed, 469 insertions(+), 3 deletions(-)
diff --git a/include/linux/damon.h b/include/linux/damon.h
index e6d2e9d0936e..7c5a416e1eff 100644
--- a/include/linux/damon.h
+++ b/include/linux/damon.h
@@ -1297,6 +1297,7 @@ void damon_destroy_filter(struct damon_filter *f);
struct damon_probe *damon_new_probe(void);
void damon_add_probe(struct damon_ctx *ctx, struct damon_probe *probe);
bool damon_has_event_driven_probes(struct damon_ctx *ctx);
+bool damon_has_sampling_probes(struct damon_ctx *ctx);
struct damon_region *damon_new_region(unsigned long start, unsigned long end);
unsigned int damon_nr_accesses_mvsum(struct damon_region *r,
diff --git a/mm/damon/Kconfig b/mm/damon/Kconfig
index c7b6f3125e79..c78e819372ab 100644
--- a/mm/damon/Kconfig
+++ b/mm/damon/Kconfig
@@ -64,6 +64,25 @@ config DAMON_PADDR
This builds the default data access monitoring operations for DAMON
that works for the physical address space.
+config DAMON_PERF_SOURCE
+ bool "perf-event source for DAMON probes"
+ depends on DAMON && PERF_EVENTS
+ help
+ Provides a PMU-agnostic perf-event overflow handler that feeds
+ physical-address and virtual-address access reports into DAMON via
+ damon_report_access(). Needed for DAMON probes backed by an
+ address-sampling PMU.
+
+ The overflow handler is NMI-safe: it writes to a per-CPU SPSC
+ ring rather than taking any lock. The PMU is selected at probe
+ creation time via perf_event_attr (e.g. AMD IBS Op, Intel PEBS).
+
+ A given PMU type may be driven by at most one DAMON context at a
+ time; damon_perf_probe_setup() returns -EBUSY if another context
+ already owns that PMU type. Distinct contexts may each drive
+ different PMUs concurrently, each draining its own per-context
+ report ring.
+
config DAMON_VADDR_KUNIT_TEST
bool "Test for DAMON operations" if !KUNIT_ALL_TESTS
depends on DAMON_VADDR && KUNIT=y
diff --git a/mm/damon/Makefile b/mm/damon/Makefile
index 22494754f41e..1abf6f2a5133 100644
--- a/mm/damon/Makefile
+++ b/mm/damon/Makefile
@@ -9,3 +9,4 @@ obj-$(CONFIG_DAMON_RECLAIM) += modules-common.o reclaim.o
obj-$(CONFIG_DAMON_LRU_SORT) += modules-common.o lru_sort.o
obj-$(CONFIG_DAMON_STAT) += modules-common.o stat.o
obj-$(CONFIG_DAMON_ACMA) += modules-common.o acma.o
+obj-$(CONFIG_DAMON_PERF_SOURCE) += perf_source.o
diff --git a/mm/damon/core.c b/mm/damon/core.c
index 8f78700afedd..a15dc7e9125a 100644
--- a/mm/damon/core.c
+++ b/mm/damon/core.c
@@ -364,6 +364,24 @@ bool damon_has_event_driven_probes(struct damon_ctx *ctx)
return false;
}
+/**
+ * damon_has_sampling_probes() - return true if @ctx has a probe that DAMON
+ * samples itself, that is, one that is not event-driven.
+ * @ctx: the DAMON context whose probes are inspected.
+ *
+ * Return: true if @ctx has a probe that is not event-driven.
+ */
+bool damon_has_sampling_probes(struct damon_ctx *ctx)
+{
+ struct damon_probe *p;
+
+ damon_for_each_probe(p, ctx) {
+ if (!p->event_driven)
+ return true;
+ }
+ return false;
+}
+
/* Does @ctx drain a perf report ring? */
static bool damon_drains_ring_perf(struct damon_ctx *ctx)
{
diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c
index 65a5b3269d1d..baaa3af917be 100644
--- a/mm/damon/paddr.c
+++ b/mm/damon/paddr.c
@@ -118,6 +118,10 @@ static void damon_pa_prep_probes_region(struct damon_region *r,
{
struct damon_prep *p;
+ /* event-driven probes have no software prep */
+ if (probe->event_driven)
+ return;
+
damon_for_each_prep(p, probe) {
switch (p->action) {
case DAMON_PREP_SET_PGIDLE:
@@ -193,6 +197,7 @@ static unsigned int damon_pa_apply_probes(struct damon_ctx *ctx,
struct damon_region *r;
struct damon_probe *p;
unsigned int max_wsum = 0;
+ bool sampling = damon_has_sampling_probes(ctx);
damon_for_each_target(t, ctx) {
damon_for_each_region(r, t) {
@@ -203,16 +208,24 @@ static unsigned int damon_pa_apply_probes(struct damon_ctx *ctx,
if (set_samples)
r->sampling_addr = damon_rand(ctx, r->ar.start,
r->ar.end);
+ if (!sampling)
+ goto wsum;
pa = damon_pa_phys_addr(r->sampling_addr,
ctx->addr_unit);
folio = damon_get_folio(PHYS_PFN(pa));
damon_for_each_probe(p, ctx) {
- if (damon_pa_filter_pass(folio, p))
+ /*
+ * Event-driven probes are credited by the ring
+ * drain; only sampling-based probes are here.
+ */
+ if (!p->event_driven &&
+ damon_pa_filter_pass(folio, p))
r->probe_hits[i]++;
i++;
}
if (folio)
folio_put(folio);
+wsum:
if (return_max_wsum)
max_wsum = max(damon_probe_hits_wsum(r, false,
false, ctx), max_wsum);
diff --git a/mm/damon/perf_source.c b/mm/damon/perf_source.c
new file mode 100644
index 000000000000..65bbfe850e70
--- /dev/null
+++ b/mm/damon/perf_source.c
@@ -0,0 +1,378 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * DAMON perf-event source
+ *
+ * Provides a PMU-agnostic NMI-safe overflow handler that feeds physical- and
+ * virtual-address access reports into DAMON via damon_report_access().
+ * The PMU is selected at probe creation time via perf_event_attr.
+ */
+
+#include <linux/cpuhotplug.h>
+#include <linux/damon.h>
+#include <linux/init.h>
+#include <linux/perf_event.h>
+#include <linux/slab.h>
+#include "perf_source.h"
+
+/* PMU event attribute for perf-event probe configuration */
+struct damon_perf_event_attr {
+ u32 type;
+ u64 config;
+ u64 config1;
+ u64 config2;
+ bool sample_phys_addr;
+ bool sample_weight_struct;
+ bool exclude_kernel;
+ bool exclude_hv;
+ bool freq;
+ u64 sample_freq;
+ u64 sample_period;
+ u32 wakeup_events;
+ u32 precise_ip;
+};
+
+struct damon_perf_probe_event {
+ struct damon_perf_event_attr attr;
+ void *priv; /* struct damon_perf_probe_state * */
+ struct hlist_node hlist_node;
+ int probe_idx; /* index into probe_hits[]; set at registration */
+ struct damon_ctx *ctx; /* owning ctx; NULLed at teardown */
+};
+
+struct damon_perf_probe_state {
+ struct perf_event * __percpu *event;
+};
+
+static void damon_perf_overflow(struct perf_event *perf_event,
+ struct perf_sample_data *data,
+ struct pt_regs *regs)
+{
+ struct damon_perf_probe_event *event =
+ perf_event->overflow_handler_context;
+ int probe_idx;
+ struct damon_ctx *ctx;
+ struct damon_access_report report = {
+ .size = PAGE_SIZE,
+ .cpu = smp_processor_id(),
+ };
+
+ /*
+ * Teardown NULLs event->ctx (with a release barrier) before releasing
+ * the per-CPU perf events, so an in-flight overflow racing the
+ * disable/release observes the torn-down state and drops the sample
+ * instead of reporting into a freed ctx. Pairs with the
+ * smp_store_release(&event->ctx, NULL) in damon_perf_probe_teardown().
+ */
+ ctx = smp_load_acquire(&event->ctx);
+ if (!ctx)
+ return;
+ probe_idx = event->probe_idx;
+ report.probe_idx = probe_idx;
+ report.ctx = ctx; /* route to this ctx's per-ctx perf ring */
+
+ /* probe_idx 0 is the zero-init sentinel; a valid index must be >= 1 */
+ if (WARN_ONCE(probe_idx == 0,
+ "damon-perf: overflow handler called with probe_idx=0\n"))
+ return;
+
+ if (!data)
+ return;
+
+ /*
+ * Populate whichever address fields the PMU provides; which ones it
+ * provides depends on the PMU and on the sample_type requested.
+ * Gate on sample_flags rather than testing for zero: the flag is
+ * the authoritative indicator of field validity.
+ */
+ if (data->sample_flags & PERF_SAMPLE_PHYS_ADDR)
+ report.paddr = data->phys_addr & PAGE_MASK;
+ if (data->sample_flags & PERF_SAMPLE_ADDR)
+ report.vaddr = data->addr & PAGE_MASK;
+
+ if (!report.paddr && !report.vaddr)
+ return;
+
+ if (data->sample_flags & PERF_SAMPLE_DATA_SRC)
+ report.is_write = !!(data->data_src.mem_op & PERF_MEM_OP_STORE);
+ report.tid = task_pid_vnr(current);
+ /* global id: the drain compares it with the target task's tgid */
+ report.tgid = task_tgid_nr(current);
+ damon_report_access(&report);
+}
+
+static enum cpuhp_state damon_perf_cpuhp_state;
+
+/*
+ * Per-PMU exclusivity: each PMU type may be owned by at most one damon_ctx.
+ * Multiple probes from the same ctx sharing a PMU type are allowed; a second
+ * ctx attempting to grab a PMU type already owned returns -EBUSY.
+ */
+struct damon_pmu_owner {
+ struct list_head node;
+ u32 pmu_type;
+ atomic_long_t owner_ctx;
+ atomic_t refcount;
+};
+
+static LIST_HEAD(damon_pmu_owner_list);
+static DEFINE_SPINLOCK(damon_pmu_owner_lock);
+
+static void damon_perf_event_init_attr(struct damon_perf_probe_event *event,
+ struct perf_event_attr *attr)
+{
+ u64 stype = PERF_SAMPLE_TIME | PERF_SAMPLE_PERIOD | PERF_SAMPLE_ADDR;
+
+ if (event->attr.sample_phys_addr)
+ stype |= PERF_SAMPLE_PHYS_ADDR;
+ if (event->attr.sample_weight_struct)
+ stype |= PERF_SAMPLE_WEIGHT_STRUCT;
+ stype |= PERF_SAMPLE_DATA_SRC;
+
+ *attr = (struct perf_event_attr) {
+ .size = sizeof(*attr),
+ .type = event->attr.type,
+ .config = event->attr.config,
+ .config1 = event->attr.config1,
+ .config2 = event->attr.config2,
+ .freq = event->attr.freq,
+ .sample_type = stype,
+ .precise_ip = event->attr.precise_ip,
+ .pinned = 1,
+ /*
+ * Created disabled, and enabled by the caller once the counter
+ * is fully set up.
+ */
+ .disabled = 1,
+ .wakeup_events = event->attr.wakeup_events,
+ .exclude_kernel = event->attr.exclude_kernel,
+ .exclude_hv = event->attr.exclude_hv,
+ };
+ if (event->attr.freq)
+ attr->sample_freq = event->attr.sample_freq;
+ else
+ attr->sample_period = event->attr.sample_period;
+}
+
+static int damon_perf_cpu_online(unsigned int cpu, struct hlist_node *node)
+{
+ struct damon_perf_probe_event *event = hlist_entry(node,
+ struct damon_perf_probe_event, hlist_node);
+ struct damon_perf_probe_state *perf = event->priv;
+ struct perf_event_attr attr;
+ struct perf_event *perf_event;
+
+ if (!perf)
+ return 0;
+
+ damon_perf_event_init_attr(event, &attr);
+
+ perf_event = perf_event_create_kernel_counter(&attr, cpu, NULL,
+ damon_perf_overflow,
+ event);
+ if (IS_ERR(perf_event)) {
+ pr_warn_ratelimited("damon-perf: cpu %u event create failed: %ld\n",
+ cpu, PTR_ERR(perf_event));
+ return 0;
+ }
+ per_cpu(*perf->event, cpu) = perf_event;
+
+ perf_event_enable(perf_event);
+ return 0;
+}
+
+static int damon_perf_cpu_offline(unsigned int cpu, struct hlist_node *node)
+{
+ struct damon_perf_probe_event *event = hlist_entry(node,
+ struct damon_perf_probe_event, hlist_node);
+ struct damon_perf_probe_state *perf = event->priv;
+ struct perf_event *perf_event;
+
+ if (!perf)
+ return 0;
+
+ perf_event = per_cpu(*perf->event, cpu);
+ if (perf_event) {
+ perf_event_disable(perf_event);
+ perf_event_release_kernel(perf_event);
+ per_cpu(*perf->event, cpu) = NULL;
+ }
+ return 0;
+}
+
+/**
+ * damon_perf_probe_setup - arm perf_events for a DAMON probe.
+ * @ctx: DAMON context that owns the probe.
+ * @probe: the damon_probe being armed; its list position in ctx->probes
+ * determines the probe_idx stored in ring entries.
+ * @event: perf event descriptor (caller fills .attr fields)
+ *
+ * Computes probe_idx by walking ctx->probes so the caller does not need
+ * to track it externally.
+ *
+ * Return: 0 on success, negative errno on failure.
+ */
+int damon_perf_probe_setup(struct damon_ctx *ctx,
+ struct damon_probe *probe,
+ struct damon_perf_probe_event *event)
+{
+ struct damon_perf_probe_state *perf;
+ struct damon_pmu_owner *owner, *found = NULL;
+ struct damon_probe *p;
+ int idx = 0;
+ int err = -ENOMEM;
+
+ /*
+ * Per-PMU exclusivity: find or create an owner slot for this PMU type.
+ * Multiple probes from the same ctx sharing a PMU type are allowed;
+ * a second ctx attempting the same PMU type returns -EBUSY.
+ */
+ spin_lock(&damon_pmu_owner_lock);
+ list_for_each_entry(owner, &damon_pmu_owner_list, node) {
+ if (owner->pmu_type == event->attr.type) {
+ long cur = atomic_long_read(&owner->owner_ctx);
+
+ if (cur != 0L && cur != (long)ctx) {
+ spin_unlock(&damon_pmu_owner_lock);
+ return -EBUSY;
+ }
+ atomic_long_set(&owner->owner_ctx, (long)ctx);
+ atomic_inc(&owner->refcount);
+ found = owner;
+ break;
+ }
+ }
+ if (!found) {
+ /*
+ * GFP_ATOMIC: this allocation runs while holding
+ * damon_pmu_owner_lock (a spinlock), so it must not sleep.
+ */
+ owner = kzalloc_obj(*owner, GFP_ATOMIC);
+ if (!owner) {
+ spin_unlock(&damon_pmu_owner_lock);
+ return -ENOMEM;
+ }
+ owner->pmu_type = event->attr.type;
+ atomic_long_set(&owner->owner_ctx, (long)ctx);
+ atomic_set(&owner->refcount, 1);
+ list_add(&owner->node, &damon_pmu_owner_list);
+ found = owner;
+ }
+ spin_unlock(&damon_pmu_owner_lock);
+
+ /* Compute probe_idx by walking ctx->probes list */
+ damon_for_each_probe(p, ctx) {
+ if (p == probe)
+ break;
+ idx++;
+ }
+ if (idx >= DAMON_MAX_PROBES) {
+ err = -ENOSPC;
+ goto release_owner;
+ }
+ event->probe_idx = idx + 1; /* 1-based; 0 is reserved sentinel */
+ event->ctx = ctx; /* route overflow reports to this ctx */
+
+ /*
+ * Allocate the ctx's per-ctx perf report ring before arming any event,
+ * so the overflow handler always finds a ready ring. Idempotent across
+ * a ctx's multiple probes.
+ */
+ err = damon_ctx_alloc_perf_ring(ctx);
+ if (err)
+ goto release_owner;
+
+ err = -ENOMEM;
+ perf = kzalloc_obj(*perf, GFP_KERNEL);
+ if (!perf)
+ goto release_owner;
+
+ perf->event = alloc_percpu(typeof(*perf->event));
+ if (!perf->event)
+ goto free_perf;
+
+ event->priv = perf;
+ INIT_HLIST_NODE(&event->hlist_node);
+
+ err = cpuhp_state_add_instance(damon_perf_cpuhp_state,
+ &event->hlist_node);
+ if (err)
+ goto free_event;
+
+ return 0;
+
+free_event:
+ free_percpu(perf->event);
+free_perf:
+ kfree(perf);
+ event->priv = NULL;
+release_owner:
+ spin_lock(&damon_pmu_owner_lock);
+ if (atomic_dec_and_test(&found->refcount)) {
+ list_del(&found->node);
+ spin_unlock(&damon_pmu_owner_lock);
+ kfree(found);
+ } else {
+ spin_unlock(&damon_pmu_owner_lock);
+ }
+ return err;
+}
+
+/**
+ * damon_perf_probe_teardown - disarm perf_events.
+ * @ctx: DAMON context that owns the probe (used to release per-PMU ownership).
+ * @event: perf event descriptor previously passed to damon_perf_probe_setup()
+ */
+void damon_perf_probe_teardown(struct damon_ctx *ctx,
+ struct damon_perf_probe_event *event)
+{
+ struct damon_perf_probe_state *perf = event->priv;
+ struct damon_pmu_owner *owner, *tmp;
+
+ if (!perf)
+ return;
+
+ /* Stop in-flight overflows from reporting; see damon_perf_overflow(). */
+ smp_store_release(&event->ctx, NULL);
+ cpuhp_state_remove_instance(damon_perf_cpuhp_state,
+ &event->hlist_node);
+ free_percpu(perf->event);
+ kfree(perf);
+ event->priv = NULL;
+
+ /*
+ * Release per-PMU ownership when the last probe for this
+ * ctx/PMU-type pair is torn down.
+ */
+ spin_lock(&damon_pmu_owner_lock);
+ list_for_each_entry_safe(owner, tmp, &damon_pmu_owner_list, node) {
+ if (owner->pmu_type == event->attr.type &&
+ atomic_long_read(&owner->owner_ctx) == (long)ctx) {
+ /*
+ * Free under the lock so a concurrent same-PMU teardown
+ * cannot observe and free the same owner.
+ */
+ if (atomic_dec_and_test(&owner->refcount)) {
+ list_del(&owner->node);
+ kfree(owner);
+ }
+ break;
+ }
+ }
+ spin_unlock(&damon_pmu_owner_lock);
+}
+
+static int __init damon_perf_source_init(void)
+{
+ int ret;
+
+ ret = cpuhp_setup_state_multi(CPUHP_AP_ONLINE_DYN,
+ "mm/damon/perf_source:online",
+ damon_perf_cpu_online,
+ damon_perf_cpu_offline);
+ if (ret < 0)
+ return ret;
+ damon_perf_cpuhp_state = ret;
+ return 0;
+}
+
+device_initcall(damon_perf_source_init);
diff --git a/mm/damon/perf_source.h b/mm/damon/perf_source.h
new file mode 100644
index 000000000000..8dcb128936ef
--- /dev/null
+++ b/mm/damon/perf_source.h
@@ -0,0 +1,25 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * DAMON perf-event source - public interface
+ *
+ * Callers that register perf-event probes include this header.
+ */
+
+#ifndef _DAMON_PERF_SOURCE_H
+#define _DAMON_PERF_SOURCE_H
+
+#ifdef CONFIG_DAMON_PERF_SOURCE
+
+#include <linux/damon.h>
+#include <linux/perf_event.h>
+
+struct damon_perf_probe_event;
+
+int damon_perf_probe_setup(struct damon_ctx *ctx,
+ struct damon_probe *probe,
+ struct damon_perf_probe_event *event);
+void damon_perf_probe_teardown(struct damon_ctx *ctx,
+ struct damon_perf_probe_event *event);
+
+#endif /* CONFIG_DAMON_PERF_SOURCE */
+#endif /* _DAMON_PERF_SOURCE_H */
diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c
index b549496ea8e2..78a5409fb1a6 100644
--- a/mm/damon/vaddr.c
+++ b/mm/damon/vaddr.c
@@ -490,6 +490,10 @@ static void damon_va_prep_probe_region(struct damon_ctx *ctx,
{
struct damon_prep *p;
+ /* event-driven probes have no software prep */
+ if (probe->event_driven)
+ return;
+
damon_for_each_prep(p, probe) {
switch (p->action) {
case DAMON_PREP_SET_PGIDLE:
@@ -594,7 +598,12 @@ static void damon_va_probe_folio(struct damon_ctx *ctx,
int i = 0;
damon_for_each_probe(probe, ctx) {
- if (damon_va_filter_pass(folio, probe, pte, pmd, mm,
+ /*
+ * Event-driven probes are credited by the ring drain; only
+ * sampling-based probes are here.
+ */
+ if (!probe->event_driven &&
+ damon_va_filter_pass(folio, probe, pte, pmd, mm,
r->sampling_addr))
r->probe_hits[i]++;
i++;
@@ -703,6 +712,7 @@ static unsigned int damon_va_apply_probes(struct damon_ctx *ctx,
struct mm_struct *mm;
struct damon_region *r;
unsigned int max_wsum = 0;
+ bool sampling = damon_has_sampling_probes(ctx);
damon_for_each_target(t, ctx) {
mm = damon_get_mm(t);
@@ -710,7 +720,8 @@ static unsigned int damon_va_apply_probes(struct damon_ctx *ctx,
if (set_samples)
r->sampling_addr = damon_rand(ctx, r->ar.start,
r->ar.end);
- __damon_va_apply_probes(ctx, mm, r);
+ if (sampling)
+ __damon_va_apply_probes(ctx, mm, r);
if (return_max_wsum)
max_wsum = max(damon_probe_hits_wsum(r, false,
false, ctx), max_wsum);
--
Git-157)
^ permalink raw reply [flat|nested] 11+ messages in thread* [RFC PATCH v4 4/9] mm/damon/ops-common: use probe-weighted score when probe weights are set
2026-10-05 8:46 [RFC PATCH v4 0/9] mm/damon: hardware-sampled access reports Ravi Jonnalagadda
` (2 preceding siblings ...)
2026-10-05 8:46 ` [RFC PATCH v4 3/9] mm/damon: add perf-event overflow handler feeding the report ring Ravi Jonnalagadda
@ 2026-10-05 8:46 ` Ravi Jonnalagadda
2026-10-05 8:46 ` [RFC PATCH v4 5/9] mm/damon: add perf_event prep type, core lifecycle, and PMU arm/disarm Ravi Jonnalagadda
` (5 subsequent siblings)
9 siblings, 0 replies; 11+ messages in thread
From: Ravi Jonnalagadda @ 2026-10-05 8:46 UTC (permalink / raw)
To: SJ Park, Andrew Morton
Cc: damon, linux-mm, linux-kernel, Gregory Price, David Rientjes,
Wei Xu, Jonathan Corbet, Bijan Tabatabai, Ajay Joshi,
Honggyu Kim, Yunjeong Mun, Akinobu Mita, Lian Wang, Kunwu Chan,
Ravi Jonnalagadda, Jonathan Cameron
When probe weights are configured (damon_has_probe_weights()), use
damon_probe_hits_wsum() for the frequency subscore in damon_hot_score()
instead of damon_nr_accesses_mvsum(). This routes the probe hits of a
perf-event probe into the DAMOS priority score.
Make damon_has_probe_weights() non-static and declare it in damon.h so
ops-common.c can call it.
Scale the weighted-hit sum in 64 bits and clamp the result to
DAMON_MAX_SUBSCORE, so a large sum cannot overflow the subscore range.
Signed-off-by: Ravi Jonnalagadda <ravis.opensrc@gmail.com>
---
include/linux/damon.h | 1 +
mm/damon/core.c | 2 +-
mm/damon/ops-common.c | 19 ++++++++++++++++---
3 files changed, 18 insertions(+), 4 deletions(-)
diff --git a/include/linux/damon.h b/include/linux/damon.h
index 7c5a416e1eff..b0895ef477ed 100644
--- a/include/linux/damon.h
+++ b/include/linux/damon.h
@@ -1306,6 +1306,7 @@ unsigned char damon_probe_hits_mvsum(int probe_idx, struct damon_region *r,
struct damon_ctx *ctx);
unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, bool mv,
struct damon_ctx *ctx);
+bool damon_has_probe_weights(struct damon_ctx *c);
int damon_set_regions(struct damon_target *t, struct damon_addr_range *ranges,
unsigned int nr_ranges, unsigned long min_region_sz);
diff --git a/mm/damon/core.c b/mm/damon/core.c
index a15dc7e9125a..9fc536238c2f 100644
--- a/mm/damon/core.c
+++ b/mm/damon/core.c
@@ -331,7 +331,7 @@ static struct damon_probe *damon_nth_probe(int n, struct damon_ctx *ctx)
return NULL;
}
-static bool damon_has_probe_weights(struct damon_ctx *c)
+bool damon_has_probe_weights(struct damon_ctx *c)
{
struct damon_probe *p;
diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c
index 2f3bf86a221b..3c5cd745f4ba 100644
--- a/mm/damon/ops-common.c
+++ b/mm/damon/ops-common.c
@@ -172,9 +172,22 @@ int damon_hot_score(struct damon_ctx *c, struct damon_region *r,
unsigned int age_weight = s->quota.weight_age;
int hotness;
- freq_subscore = mult_frac(damon_nr_accesses_mvsum(r, c),
- DAMON_MAX_SUBSCORE,
- damon_nr_samples_per_aggr(&c->attrs));
+ if (damon_has_probe_weights(c)) {
+ unsigned int wsum = damon_probe_hits_wsum(r, false, true, c);
+ u64 subscore = div_u64((u64)wsum * DAMON_MAX_SUBSCORE,
+ damon_nr_samples_per_aggr(&c->attrs));
+
+ /*
+ * Score by the weighted probe hits. Clamp to
+ * DAMON_MAX_SUBSCORE so a large weighted-hit sum cannot
+ * overflow the subscore range.
+ */
+ freq_subscore = min_t(u64, subscore, DAMON_MAX_SUBSCORE);
+ } else {
+ freq_subscore = mult_frac(damon_nr_accesses_mvsum(r, c),
+ DAMON_MAX_SUBSCORE,
+ damon_nr_samples_per_aggr(&c->attrs));
+ }
age_in_sec = div_u64((u64)r->age * c->attrs.aggr_interval,
USEC_PER_SEC);
--
Git-157)
^ permalink raw reply [flat|nested] 11+ messages in thread* [RFC PATCH v4 5/9] mm/damon: add perf_event prep type, core lifecycle, and PMU arm/disarm
2026-10-05 8:46 [RFC PATCH v4 0/9] mm/damon: hardware-sampled access reports Ravi Jonnalagadda
` (3 preceding siblings ...)
2026-10-05 8:46 ` [RFC PATCH v4 4/9] mm/damon/ops-common: use probe-weighted score when probe weights are set Ravi Jonnalagadda
@ 2026-10-05 8:46 ` Ravi Jonnalagadda
2026-10-05 8:46 ` [RFC PATCH v4 6/9] mm/damon/sysfs: expose perf_event prep attributes Ravi Jonnalagadda
` (4 subsequent siblings)
9 siblings, 0 replies; 11+ messages in thread
From: Ravi Jonnalagadda @ 2026-10-05 8:46 UTC (permalink / raw)
To: SJ Park, Andrew Morton
Cc: damon, linux-mm, linux-kernel, Gregory Price, David Rientjes,
Wei Xu, Jonathan Corbet, Bijan Tabatabai, Ajay Joshi,
Honggyu Kim, Yunjeong Mun, Akinobu Mita, Lian Wang, Kunwu Chan,
Ravi Jonnalagadda, Jonathan Cameron
The prep interface exposes a single preparation action, set_pgidle, which
arms the page-idle scan used by the software access check. There is no
way for user space to configure a probe whose access information comes
from a hardware PMU sampling facility, even though the perf-event source
(CONFIG_DAMON_PERF_SOURCE) already provides the NMI-safe overflow
handler, the report ring, and the probe setup/teardown lifecycle to
consume such samples.
Add a DAMON_PREP_PERF_EVENT prep action carrying a subset of
perf_event_attr (type, config, config1, config2, sampling period or
frequency, wakeup_events, precise_ip, sample_phys_addr,
sample_weight_struct, exclude_kernel, exclude_hv), and wire it through
include/linux/damon.h, mm/damon/perf_source.h and mm/damon/core.c. A
probe built with a perf_event prep is to be event driven, which the
sysfs patch that follows sets: DAMON does not walk it in the
apply_probes vtable; instead the PMU samples memory accesses and feeds
region hit counters through the report ring.
The prep also carries how many counters the PMU needs. A per-CPU PMU is
armed by adding a cpuhp instance, which opens one kernel counter on every
online CPU. A system-wide PMU is a single hardware unit served by exactly
one counter, so a single_instance flag opens one kernel counter pinned to
a fixed online CPU and bypasses the cpuhp fan-out. The pin names an
online CPU because a kernel counter with no task must be bound to one.
The PMU is only armed when building the context that will actually run. A
param_ctx built for a commit carries the attributes but defers arming,
because arming a throwaway context would collide with the running
context's per-PMU ownership, and a commit_live flag distinguishes the
dry-run validation pass from the real commit. The commit path hands the
armed event between the running probe and the committed source probe,
keeping the running event untouched on the common weight-only commit and
re-arming only when the perf attributes change or the probe's list
position no longer matches the position its running event's probe_idx
was computed for; a probe reorder with unchanged attributes still needs
a re-arm, or the event keeps crediting probe_hits[] at its old index.
The armed event is released both when monitoring is turned off, from
kdamond_fn()'s exit path after damon_destroy_targets(), and when the
context is destroyed. Turning monitoring back on through sysfs builds
a fresh context, which arms a new event from the probe attributes. The
teardown barrier the overflow handler pairs with, a release store of
event->ctx = NULL, covers the single-instance path as well as the per-CPU
one. damon_perf_probe_teardown() also frees the event descriptor, which
the probe owns from the time the descriptor is attached to it.
__damon_commit_ctx() and damon_commit_probes() gain the commit_live flag
described above, false for the validation and test-context passes and
true for the real commit; its one existing kunit call site in
mm/damon/tests/core-kunit.h is updated to pass false. Wiring these
attributes through sysfs is a separate, following commit.
Co-developed-by: Akinobu Mita <akinobu.mita@gmail.com>
Signed-off-by: Akinobu Mita <akinobu.mita@gmail.com>
Signed-off-by: Ravi Jonnalagadda <ravis.opensrc@gmail.com>
---
include/linux/damon.h | 29 ++++++++
mm/damon/core.c | 154 ++++++++++++++++++++++++++++++++++++++++---
mm/damon/perf_source.c | 157 +++++++++++++++++++++++++++++---------------
mm/damon/perf_source.h | 32 ++++++++-
mm/damon/tests/core-kunit.h | 2 +-
5 files changed, 310 insertions(+), 64 deletions(-)
diff --git a/include/linux/damon.h b/include/linux/damon.h
index b0895ef477ed..f84cdc583880 100644
--- a/include/linux/damon.h
+++ b/include/linux/damon.h
@@ -868,18 +868,44 @@ struct damon_intervals_goal {
* enum damon_prep_action - DAMON probing preparation action.
*
* @DAMON_PREP_SET_PGIDLE: Set the probing memory as idle page.
+ * @DAMON_PREP_PERF_EVENT: Back the probe with a perf event.
*/
enum damon_prep_action {
DAMON_PREP_SET_PGIDLE,
+ DAMON_PREP_PERF_EVENT,
};
/**
* struct damon_prep - DAMON probing preparation request.
*
* @action: Action to do to the probing memory for the preparation.
+ * @perf: perf_event_attr subset selecting the PMU and sampling
+ * parameters. Only valid when @action is DAMON_PREP_PERF_EVENT.
+ *
+ * A DAMON_PREP_PERF_EVENT prep turns the containing &struct damon_probe into
+ * an event-driven probe: a kernel perf event (e.g. AMD IBS Op, Intel PEBS)
+ * samples memory accesses and feeds them into the probe hit counters via the
+ * report ring. The @perf fields are copied into a perf_event_attr when the
+ * kdamond is turned on, or when a commit changes them.
*/
struct damon_prep {
enum damon_prep_action action;
+ struct {
+ u32 type;
+ u64 config;
+ u64 config1;
+ u64 config2;
+ u64 sample_period;
+ u64 sample_freq;
+ u32 wakeup_events;
+ u32 precise_ip;
+ bool sample_phys_addr;
+ bool sample_weight_struct;
+ bool exclude_kernel;
+ bool exclude_hv;
+ bool freq;
+ bool single_instance;
+ } perf;
/* private: */
/* siblings list. */
struct list_head list;
@@ -934,10 +960,13 @@ struct damon_filter {
* @weight: Relative priority of the attribute for this probe.
* @event_driven: Whether the probe's hits arrive through the report ring
* drain rather than the apply_probes callback.
+ * @perf_priv: Perf-event state of an event-driven probe, released by
+ * damon_perf_probe_teardown().
*/
struct damon_probe {
unsigned int weight;
bool event_driven;
+ void *perf_priv;
/* private: */
/* Preparation actions to apply to each probing memory. */
struct list_head preps;
diff --git a/mm/damon/core.c b/mm/damon/core.c
index 9fc536238c2f..c2acae6e19ec 100644
--- a/mm/damon/core.c
+++ b/mm/damon/core.c
@@ -18,6 +18,7 @@
/* for damon_get_folio() used by node eligible memory metrics */
#include "ops-common.h"
+#include "perf_source.h"
#define CREATE_TRACE_POINTS
#include <trace/events/damon.h>
@@ -285,6 +286,13 @@ struct damon_probe *damon_new_probe(void)
if (!p)
return NULL;
p->weight = 0;
+ p->event_driven = false;
+ /*
+ * Must be NULL: damon_destroy_ctx() and damon_commit_probes() call
+ * damon_perf_probe_teardown() for a probe whose perf_priv is set, so a
+ * probe destroyed before it is ever armed must not carry garbage there.
+ */
+ p->perf_priv = NULL;
INIT_LIST_HEAD(&p->preps);
INIT_LIST_HEAD(&p->filters);
INIT_LIST_HEAD(&p->list);
@@ -1177,13 +1185,29 @@ void damon_destroy_ctx(struct damon_ctx *ctx)
damon_for_each_scheme_safe(s, next_s, ctx)
damon_destroy_scheme(s);
- damon_for_each_probe_safe(p, next_p, ctx)
+ damon_for_each_probe_safe(p, next_p, ctx) {
+#ifdef CONFIG_DAMON_PERF_SOURCE
+ /*
+ * Release the PMU counters before freeing the probe.
+ * damon_perf_probe_teardown() owns and frees the event
+ * descriptor.
+ */
+ if (p->perf_priv) {
+ damon_perf_probe_teardown(ctx, p->perf_priv);
+ p->perf_priv = NULL;
+ }
+#endif
damon_destroy_probe(p);
+ }
damon_for_each_sample_filter_safe(f, next_f, &ctx->sample_control)
damon_destroy_sample_filter(f, &ctx->sample_control);
- /* No-op if never allocated. */
+ /*
+ * The probe loop above has released every perf event that reports into
+ * the ring, so no overflow handler can still reach it. No-op if never
+ * allocated.
+ */
damon_ctx_free_perf_ring(ctx);
/* Free the reusable ring-drain region snapshot buffers. */
@@ -2016,6 +2040,7 @@ static int damon_commit_targets(
static void damon_commit_prep(struct damon_prep *dst, struct damon_prep *src)
{
dst->action = src->action;
+ dst->perf = src->perf;
}
static int damon_commit_preps(struct damon_probe *dst, struct damon_probe *src)
@@ -2110,7 +2135,73 @@ static int damon_commit_filters(struct damon_probe *dst,
return 0;
}
-static int damon_commit_probes(struct damon_ctx *dst, struct damon_ctx *src)
+#ifdef CONFIG_DAMON_PERF_SOURCE
+/*
+ * Hand off the armed perf event between a running probe (@dst, in the live
+ * @ctx) and the committed source probe (@src, from a discarded param_ctx that
+ * was built without arming). Keeps the running event untouched when the perf
+ * attributes are unchanged AND @dst is still at the list position (@dst_idx)
+ * its running event's probe_idx was computed for (the common weight-only
+ * commit); otherwise tears down the old event and re-arms from @src's
+ * carried attributes. The position check matters because a probe reorder
+ * with unchanged attrs would otherwise keep crediting probe_hits[] at the
+ * probe's old index instead of its new one.
+ *
+ * @src->perf_priv carries the attributes @src was built with, unarmed;
+ * ownership of that descriptor moves to @dst here.
+ */
+static int damon_commit_perf_probe(struct damon_ctx *ctx,
+ struct damon_probe *dst, struct damon_probe *src, int dst_idx)
+{
+ struct damon_perf_probe_event *dst_ev = dst->perf_priv;
+ struct damon_perf_probe_event *src_ev = src->perf_priv;
+ int err;
+
+ if (!src_ev) {
+ /* Source has no perf probe: tear down any running event. */
+ if (dst_ev) {
+ damon_perf_probe_teardown(ctx, dst_ev);
+ dst->perf_priv = NULL;
+ }
+ return 0;
+ }
+
+ /*
+ * Already armed with identical attrs and still at the list position
+ * its probe_idx was computed for: keep the running event. A probe
+ * reorder with unchanged attrs must still be re-armed below, or
+ * the event keeps crediting the probe_hits[] slot for its
+ * old position instead of its new one.
+ */
+ if (dst_ev && dst_ev->priv &&
+ dst_ev->probe_idx == dst_idx + 1 &&
+ !memcmp(&dst_ev->attr, &src_ev->attr, sizeof(dst_ev->attr)))
+ return 0;
+
+ /* Attrs or position changed, or dst not armed: re-arm from src. */
+ if (dst_ev) {
+ damon_perf_probe_teardown(ctx, dst_ev);
+ dst->perf_priv = NULL;
+ }
+ err = damon_perf_probe_setup(ctx, dst, src_ev);
+ if (err) {
+ /* a failed commit stops the kdamond, releasing the probes */
+ return err;
+ }
+ /* Ownership of src_ev moves to dst; the param_ctx must not free it. */
+ dst->perf_priv = src_ev;
+ src->perf_priv = NULL;
+ return 0;
+}
+#endif /* CONFIG_DAMON_PERF_SOURCE */
+
+/*
+ * @commit_live is false for the dry-run validation pass (a commit into a
+ * throwaway test_ctx) and true for the real commit into the running ctx.
+ * The PMU may only be armed or disarmed on the real commit.
+ */
+static int damon_commit_probes(struct damon_ctx *dst, struct damon_ctx *src,
+ bool commit_live)
{
struct damon_probe *dst_probe, *next, *src_probe, *new_probe;
int i = 0, j = 0, err;
@@ -2119,13 +2210,29 @@ static int damon_commit_probes(struct damon_ctx *dst, struct damon_ctx *src)
src_probe = damon_nth_probe(i++, src);
if (src_probe) {
dst_probe->weight = src_probe->weight;
+ dst_probe->event_driven = src_probe->event_driven;
err = damon_commit_preps(dst_probe, src_probe);
if (err)
return err;
err = damon_commit_filters(dst_probe, src_probe);
if (err)
return err;
+#ifdef CONFIG_DAMON_PERF_SOURCE
+ if (commit_live) {
+ err = damon_commit_perf_probe(dst, dst_probe,
+ src_probe, i - 1);
+ if (err)
+ return err;
+ }
+#endif
} else {
+#ifdef CONFIG_DAMON_PERF_SOURCE
+ if (commit_live && dst_probe->perf_priv) {
+ damon_perf_probe_teardown(dst,
+ dst_probe->perf_priv);
+ dst_probe->perf_priv = NULL;
+ }
+#endif
damon_destroy_probe(dst_probe);
}
}
@@ -2139,12 +2246,23 @@ static int damon_commit_probes(struct damon_ctx *dst, struct damon_ctx *src)
return -ENOMEM;
damon_add_probe(dst, new_probe);
new_probe->weight = src_probe->weight;
+ new_probe->event_driven = src_probe->event_driven;
err = damon_commit_preps(new_probe, src_probe);
if (err)
return err;
err = damon_commit_filters(new_probe, src_probe);
if (err)
return err;
+#ifdef CONFIG_DAMON_PERF_SOURCE
+ if (commit_live && src_probe->perf_priv) {
+ err = damon_perf_probe_setup(dst, new_probe,
+ src_probe->perf_priv);
+ if (err)
+ return err;
+ new_probe->perf_priv = src_probe->perf_priv;
+ src_probe->perf_priv = NULL;
+ }
+#endif
}
return 0;
}
@@ -2242,7 +2360,8 @@ static int damon_commit_sample_control(
return damon_commit_sample_filters(dst, src);
}
-static int __damon_commit_ctx(struct damon_ctx *dst, struct damon_ctx *src)
+static int __damon_commit_ctx(struct damon_ctx *dst, struct damon_ctx *src,
+ bool commit_live)
{
int err;
struct damos *scheme;
@@ -2290,7 +2409,7 @@ static int __damon_commit_ctx(struct damon_ctx *dst, struct damon_ctx *src)
}
dst->pause = src->pause;
dst->ops = src->ops;
- err = damon_commit_probes(dst, src);
+ err = damon_commit_probes(dst, src, commit_live);
if (err)
return err;
err = damon_commit_sample_control(&dst->sample_control,
@@ -2312,7 +2431,7 @@ static struct damon_ctx *damon_new_test_ctx(struct damon_ctx *dst)
test_ctx = damon_new_ctx();
if (!test_ctx)
return NULL;
- err = __damon_commit_ctx(test_ctx, dst);
+ err = __damon_commit_ctx(test_ctx, dst, false);
if (err) {
damon_destroy_ctx(test_ctx);
return NULL;
@@ -2341,10 +2460,10 @@ int damon_commit_ctx(struct damon_ctx *dst, struct damon_ctx *src)
test_ctx = damon_new_test_ctx(dst);
if (!test_ctx)
return -ENOMEM;
- err = __damon_commit_ctx(test_ctx, src);
+ err = __damon_commit_ctx(test_ctx, src, false);
if (err)
goto out;
- err = __damon_commit_ctx(dst, src);
+ err = __damon_commit_ctx(dst, src, true);
out:
damon_destroy_ctx(test_ctx);
return err;
@@ -2465,7 +2584,7 @@ int damon_start(struct damon_ctx **ctxs, int nr_ctxs, bool exclusive)
if (!test_ctx)
return -ENOMEM;
- err = __damon_commit_ctx(test_ctx, ctxs[i]);
+ err = __damon_commit_ctx(test_ctx, ctxs[i], false);
damon_destroy_ctx(test_ctx);
if (err)
return err;
@@ -4913,6 +5032,23 @@ static int kdamond_fn(void *data)
done:
damon_destroy_targets(ctx);
+#ifdef CONFIG_DAMON_PERF_SOURCE
+ /*
+ * Release perf-event probes here so a stopped kdamond leaves no event
+ * firing overflows into its report ring, and holds no PMU ownership.
+ */
+ {
+ struct damon_probe *p, *next_p;
+
+ damon_for_each_probe_safe(p, next_p, ctx) {
+ if (p->perf_priv) {
+ damon_perf_probe_teardown(ctx, p->perf_priv);
+ p->perf_priv = NULL;
+ }
+ }
+ }
+#endif
+
kfree(ctx->regions_score_histogram);
mutex_lock(&ctx->call_controls_lock);
ctx->call_controls_obsolete = true;
diff --git a/mm/damon/perf_source.c b/mm/damon/perf_source.c
index 65bbfe850e70..4baa8aebcfa8 100644
--- a/mm/damon/perf_source.c
+++ b/mm/damon/perf_source.c
@@ -14,33 +14,17 @@
#include <linux/slab.h>
#include "perf_source.h"
-/* PMU event attribute for perf-event probe configuration */
-struct damon_perf_event_attr {
- u32 type;
- u64 config;
- u64 config1;
- u64 config2;
- bool sample_phys_addr;
- bool sample_weight_struct;
- bool exclude_kernel;
- bool exclude_hv;
- bool freq;
- u64 sample_freq;
- u64 sample_period;
- u32 wakeup_events;
- u32 precise_ip;
-};
-
-struct damon_perf_probe_event {
- struct damon_perf_event_attr attr;
- void *priv; /* struct damon_perf_probe_state * */
- struct hlist_node hlist_node;
- int probe_idx; /* index into probe_hits[]; set at registration */
- struct damon_ctx *ctx; /* owning ctx; NULLed at teardown */
-};
+/*
+ * struct damon_perf_event_attr and struct damon_perf_probe_event are defined
+ * in perf_source.h so that the sysfs configuration surface can build a probe
+ * event descriptor before handing it to damon_perf_probe_setup().
+ */
struct damon_perf_probe_state {
+ /* per-CPU probes (PEBS/IBS) */
struct perf_event * __percpu *event;
+ /* single-instance probes (system-wide PMU) */
+ struct perf_event *single_event;
};
static void damon_perf_overflow(struct perf_event *perf_event,
@@ -104,8 +88,7 @@ static enum cpuhp_state damon_perf_cpuhp_state;
/*
* Per-PMU exclusivity: each PMU type may be owned by at most one damon_ctx.
- * Multiple probes from the same ctx sharing a PMU type are allowed; a second
- * ctx attempting to grab a PMU type already owned returns -EBUSY.
+ * Multiple probes from the same ctx sharing a PMU type are allowed.
*/
struct damon_pmu_owner {
struct list_head node;
@@ -224,7 +207,8 @@ int damon_perf_probe_setup(struct damon_ctx *ctx,
/*
* Per-PMU exclusivity: find or create an owner slot for this PMU type.
* Multiple probes from the same ctx sharing a PMU type are allowed;
- * a second ctx attempting the same PMU type returns -EBUSY.
+ * a second ctx attempting the same PMU type returns -EBUSY. A commit
+ * that changes the perf attributes re-arms through this function too.
*/
spin_lock(&damon_pmu_owner_lock);
list_for_each_entry(owner, &damon_pmu_owner_list, node) {
@@ -285,12 +269,51 @@ int damon_perf_probe_setup(struct damon_ctx *ctx,
perf = kzalloc_obj(*perf, GFP_KERNEL);
if (!perf)
goto release_owner;
+ event->priv = perf;
+
+ /*
+ * A system-wide PMU is a single hardware unit rather than a per-CPU
+ * counter, so it needs exactly one counter: the cpuhp fan-out below
+ * would run one redundant sampler per CPU against the one device and
+ * corrupt its shared state. Pin that counter to a fixed online CPU
+ * and bypass cpuhp.
+ *
+ * A kernel counter with no task must name a CPU. If that CPU goes
+ * offline the counter stops and is not migrated.
+ */
+ if (event->attr.single_instance) {
+ struct perf_event_attr attr;
+ int cpu = cpumask_first(cpu_online_mask);
+
+ damon_perf_event_init_attr(event, &attr);
+ /*
+ * Pass @event as the overflow context, as the per-CPU path
+ * (damon_perf_cpu_online()) does: damon_perf_overflow() reads
+ * event->ctx via smp_load_acquire() for the teardown barrier.
+ */
+ perf->single_event = perf_event_create_kernel_counter(&attr,
+ cpu, NULL, damon_perf_overflow, event);
+ if (IS_ERR(perf->single_event)) {
+ err = PTR_ERR(perf->single_event);
+ perf->single_event = NULL;
+ pr_warn("damon-perf: single-instance event create failed: %d\n",
+ err);
+ goto free_perf;
+ }
+ perf_event_enable(perf->single_event);
+ /*
+ * Ownership is already held via the per-PMU owner->refcount
+ * acquired at the top of setup; the single-instance path shares
+ * that slot, so no separate refcount is taken here. Teardown
+ * releases it through the same owner list as the per-CPU path.
+ */
+ return 0;
+ }
perf->event = alloc_percpu(typeof(*perf->event));
if (!perf->event)
goto free_perf;
- event->priv = perf;
INIT_HLIST_NODE(&event->hlist_node);
err = cpuhp_state_add_instance(damon_perf_cpuhp_state,
@@ -326,39 +349,67 @@ void damon_perf_probe_teardown(struct damon_ctx *ctx,
struct damon_perf_probe_event *event)
{
struct damon_perf_probe_state *perf = event->priv;
- struct damon_pmu_owner *owner, *tmp;
- if (!perf)
- return;
+ if (perf) {
+ struct damon_pmu_owner *owner, *tmp;
- /* Stop in-flight overflows from reporting; see damon_perf_overflow(). */
- smp_store_release(&event->ctx, NULL);
- cpuhp_state_remove_instance(damon_perf_cpuhp_state,
- &event->hlist_node);
- free_percpu(perf->event);
- kfree(perf);
- event->priv = NULL;
+ /*
+ * Signal in-flight NMI overflow handlers to drop samples
+ * before tearing down the perf events and freeing their
+ * backing state. Pairs with smp_load_acquire(&event->ctx)
+ * in damon_perf_overflow(). This applies to both the per-CPU
+ * and single-instance paths: a single-instance counter can
+ * also deliver an overflow racing teardown.
+ */
+ smp_store_release(&event->ctx, NULL);
- /*
- * Release per-PMU ownership when the last probe for this
- * ctx/PMU-type pair is torn down.
- */
- spin_lock(&damon_pmu_owner_lock);
- list_for_each_entry_safe(owner, tmp, &damon_pmu_owner_list, node) {
- if (owner->pmu_type == event->attr.type &&
- atomic_long_read(&owner->owner_ctx) == (long)ctx) {
+ if (perf->single_event) {
/*
- * Free under the lock so a concurrent same-PMU teardown
- * cannot observe and free the same owner.
+ * Single-instance probe: no cpuhp instance was added,
+ * so release the one counter. disable() also quiesces
+ * any pending overflow before the release.
*/
- if (atomic_dec_and_test(&owner->refcount)) {
- list_del(&owner->node);
- kfree(owner);
+ perf_event_disable(perf->single_event);
+ perf_event_release_kernel(perf->single_event);
+ perf->single_event = NULL;
+ } else {
+ /*
+ * cpuhp_state_remove_instance() disables+releases each
+ * CPU's perf event; once it returns no new overflow can
+ * be delivered for this event.
+ */
+ cpuhp_state_remove_instance(damon_perf_cpuhp_state,
+ &event->hlist_node);
+ free_percpu(perf->event);
+ }
+ kfree(perf);
+ event->priv = NULL;
+
+ /*
+ * Release per-PMU ownership when the last probe for this
+ * ctx/PMU-type pair is torn down.
+ */
+ spin_lock(&damon_pmu_owner_lock);
+ list_for_each_entry_safe(owner, tmp, &damon_pmu_owner_list,
+ node) {
+ if (owner->pmu_type == event->attr.type &&
+ atomic_long_read(&owner->owner_ctx) == (long)ctx) {
+ /*
+ * Free under the lock so a concurrent same-PMU
+ * teardown cannot observe and free the same
+ * owner.
+ */
+ if (atomic_dec_and_test(&owner->refcount)) {
+ list_del(&owner->node);
+ kfree(owner);
+ }
+ break;
}
- break;
}
+ spin_unlock(&damon_pmu_owner_lock);
}
- spin_unlock(&damon_pmu_owner_lock);
+ /* teardown owns the event allocation */
+ kfree(event);
}
static int __init damon_perf_source_init(void)
diff --git a/mm/damon/perf_source.h b/mm/damon/perf_source.h
index 8dcb128936ef..12ca61be7243 100644
--- a/mm/damon/perf_source.h
+++ b/mm/damon/perf_source.h
@@ -13,7 +13,37 @@
#include <linux/damon.h>
#include <linux/perf_event.h>
-struct damon_perf_probe_event;
+/*
+ * PMU event attributes for a perf-event probe. A subset of perf_event_attr
+ * chosen at probe creation time to select the PMU (AMD IBS, Intel PEBS, ...)
+ * and its sampling parameters.
+ */
+struct damon_perf_event_attr {
+ u32 type;
+ u64 config;
+ u64 config1;
+ u64 config2;
+ bool sample_phys_addr;
+ bool sample_weight_struct;
+ bool exclude_kernel;
+ bool exclude_hv;
+ bool freq;
+ /* system-wide PMU: open one counter, not one per CPU */
+ bool single_instance;
+ u64 sample_freq;
+ u64 sample_period;
+ u32 wakeup_events;
+ u32 precise_ip;
+};
+
+struct damon_perf_probe_event {
+ struct damon_perf_event_attr attr;
+ struct damon_ctx *ctx; /* owning ctx for ring routing; set at setup */
+ void *priv; /* struct damon_perf_probe_state * */
+ struct hlist_node hlist_node;
+ /* 1-based probe index for probe_hits[]; set at registration */
+ int probe_idx;
+};
int damon_perf_probe_setup(struct damon_ctx *ctx,
struct damon_probe *probe,
diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h
index 3fbb4e4e36fa..43432ae5f68f 100644
--- a/mm/damon/tests/core-kunit.h
+++ b/mm/damon/tests/core-kunit.h
@@ -1590,7 +1590,7 @@ static void damon_test_commit_probes_for(struct kunit *test,
kunit_skip(test, "src alloc fail");
}
- err = damon_commit_probes(dst, src);
+ err = damon_commit_probes(dst, src, false);
KUNIT_EXPECT_EQ(test, err, 0);
if (err)
goto out;
--
Git-157)
^ permalink raw reply [flat|nested] 11+ messages in thread* [RFC PATCH v4 6/9] mm/damon/sysfs: expose perf_event prep attributes
2026-10-05 8:46 [RFC PATCH v4 0/9] mm/damon: hardware-sampled access reports Ravi Jonnalagadda
` (4 preceding siblings ...)
2026-10-05 8:46 ` [RFC PATCH v4 5/9] mm/damon: add perf_event prep type, core lifecycle, and PMU arm/disarm Ravi Jonnalagadda
@ 2026-10-05 8:46 ` Ravi Jonnalagadda
2026-10-05 8:46 ` [RFC PATCH v4 7/9] mm/damon/tests/drain-kunit: kunit for report rings and ring drain Ravi Jonnalagadda
` (3 subsequent siblings)
9 siblings, 0 replies; 11+ messages in thread
From: Ravi Jonnalagadda @ 2026-10-05 8:46 UTC (permalink / raw)
To: SJ Park, Andrew Morton
Cc: damon, linux-mm, linux-kernel, Gregory Price, David Rientjes,
Wei Xu, Jonathan Corbet, Bijan Tabatabai, Ajay Joshi,
Honggyu Kim, Yunjeong Mun, Akinobu Mita, Lian Wang, Kunwu Chan,
Ravi Jonnalagadda, Jonathan Cameron
Wire the DAMON_PREP_PERF_EVENT prep type added to core.c and
perf_source.h into sysfs, so user space can configure a PMU-backed
probe entirely through the preps/N/ directory: type, config, config1,
config2, freq, sample_period, sample_freq, wakeup_events, precise_ip,
sample_phys_addr, sample_weight_struct, exclude_kernel, exclude_hv and
single_instance.
For example:
echo perf_event > .../probes/0/preps/0/prep_action
echo 8 > .../probes/0/preps/0/type
echo 0x100 > .../probes/0/preps/0/sample_period
echo 1 > .../probes/0/preps/0/sample_phys_addr
damon_sysfs_set_perf_probe() builds the damon_perf_probe_event from the
prep's stored attributes and, when arming, calls
damon_perf_probe_setup() to open the PMU counter(s) immediately;
deferred (non-arming) builds carry the attributes for a later commit to
arm, per the commit_live distinction in the previous commit.
damon_sysfs_set_preps() validates a DAMON_PREP_PERF_EVENT prep before it
can reach the PMU. freq selects which one of sample_freq and
sample_period carries the sampling rate, so the selected one must be set
and the other zero, since the counter would otherwise be armed with a
zero period and take no samples. precise_ip is a 2-bit bitfield in
perf_event_attr, so a value above 3 is rejected. A second
DAMON_PREP_PERF_EVENT prep on the same probe is rejected, since a probe
is backed by one PMU counter.
A probe with a perf_event prep cannot have probe filters, since the
drain does not apply them to its samples. With CONFIG_DAMON_PERF_SOURCE
disabled, a perf_event prep is rejected with -EOPNOTSUPP. The prep
attribute stores take damon_sysfs_lock with mutex_trylock(), as the
other DAMON sysfs stores do, and return -EBUSY while it is held, so a
commit reads a consistent set of values.
Co-developed-by: Akinobu Mita <akinobu.mita@gmail.com>
Signed-off-by: Akinobu Mita <akinobu.mita@gmail.com>
Signed-off-by: Ravi Jonnalagadda <ravis.opensrc@gmail.com>
---
mm/damon/sysfs.c | 304 ++++++++++++++++++++++++++++++++++++++++++++++++++++---
1 file changed, 290 insertions(+), 14 deletions(-)
diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c
index 80e6fc8004e5..3fc40dd540a0 100644
--- a/mm/damon/sysfs.c
+++ b/mm/damon/sysfs.c
@@ -7,6 +7,7 @@
#include <linux/slab.h>
#include "sysfs-common.h"
+#include "perf_source.h"
/*
* init region directory
@@ -756,6 +757,21 @@ static const struct kobj_type damon_sysfs_intervals_ktype = {
struct damon_sysfs_prep {
struct kobject kobj;
enum damon_prep_action action;
+ /* perf_event_attr subset; valid when action == DAMON_PREP_PERF_EVENT */
+ u32 perf_type;
+ u64 config;
+ u64 config1;
+ u64 config2;
+ u64 sample_period;
+ u64 sample_freq;
+ u32 wakeup_events;
+ u32 precise_ip;
+ bool sample_phys_addr;
+ bool sample_weight_struct;
+ bool exclude_kernel;
+ bool exclude_hv;
+ bool freq;
+ bool single_instance;
};
static struct damon_sysfs_prep *damon_sysfs_prep_alloc(void)
@@ -780,6 +796,10 @@ damon_sysfs_prep_action_names[] = {
.action = DAMON_PREP_SET_PGIDLE,
.name = "set_pgidle",
},
+ {
+ .action = DAMON_PREP_PERF_EVENT,
+ .name = "perf_event",
+ },
};
static ssize_t prep_action_show(struct kobject *kobj,
@@ -834,8 +854,132 @@ static void damon_sysfs_prep_release(struct kobject *kobj)
static struct kobj_attribute damon_sysfs_prep_prep_action_attr =
__ATTR_RW_MODE(prep_action, 0600);
+/*
+ * perf_event configuration attributes. These mirror a subset of
+ * perf_event_attr and are only meaningful when prep_action is "perf_event".
+ * They select the PMU (via type/config) and its sampling parameters, and are
+ * copied into the perf-event probe when DAMON is turned on or the inputs are
+ * committed.
+ *
+ * The sysfs file names stay bare (type, config, ...) while the backing C
+ * symbols are prefixed to avoid clashing with identically named attributes
+ * elsewhere in this file.
+ */
+#define DAMON_SYSFS_PREP_PERF_U32(name, field) \
+static ssize_t damon_sysfs_prep_##name##_show(struct kobject *kobj, \
+ struct kobj_attribute *attr, char *buf) \
+{ \
+ struct damon_sysfs_prep *prep = container_of(kobj, \
+ struct damon_sysfs_prep, kobj); \
+ return sysfs_emit(buf, "%u\n", prep->field); \
+} \
+static ssize_t damon_sysfs_prep_##name##_store(struct kobject *kobj, \
+ struct kobj_attribute *attr, const char *buf, \
+ size_t count) \
+{ \
+ struct damon_sysfs_prep *prep = container_of(kobj, \
+ struct damon_sysfs_prep, kobj); \
+ u32 v; \
+ int err = kstrtou32(buf, 0, &v); \
+ if (err) \
+ return err; \
+ if (!mutex_trylock(&damon_sysfs_lock)) \
+ return -EBUSY; \
+ prep->field = v; \
+ mutex_unlock(&damon_sysfs_lock); \
+ return count; \
+} \
+static struct kobj_attribute damon_sysfs_prep_##name##_attr = __ATTR(name, \
+ 0600, damon_sysfs_prep_##name##_show, \
+ damon_sysfs_prep_##name##_store)
+
+#define DAMON_SYSFS_PREP_PERF_U64(name, field) \
+static ssize_t damon_sysfs_prep_##name##_show(struct kobject *kobj, \
+ struct kobj_attribute *attr, char *buf) \
+{ \
+ struct damon_sysfs_prep *prep = container_of(kobj, \
+ struct damon_sysfs_prep, kobj); \
+ return sysfs_emit(buf, "%llu\n", prep->field); \
+} \
+static ssize_t damon_sysfs_prep_##name##_store(struct kobject *kobj, \
+ struct kobj_attribute *attr, const char *buf, \
+ size_t count) \
+{ \
+ struct damon_sysfs_prep *prep = container_of(kobj, \
+ struct damon_sysfs_prep, kobj); \
+ u64 v; \
+ int err = kstrtou64(buf, 0, &v); \
+ if (err) \
+ return err; \
+ if (!mutex_trylock(&damon_sysfs_lock)) \
+ return -EBUSY; \
+ prep->field = v; \
+ mutex_unlock(&damon_sysfs_lock); \
+ return count; \
+} \
+static struct kobj_attribute damon_sysfs_prep_##name##_attr = __ATTR(name, \
+ 0600, damon_sysfs_prep_##name##_show, \
+ damon_sysfs_prep_##name##_store)
+
+#define DAMON_SYSFS_PREP_PERF_BOOL(name, field) \
+static ssize_t damon_sysfs_prep_##name##_show(struct kobject *kobj, \
+ struct kobj_attribute *attr, char *buf) \
+{ \
+ struct damon_sysfs_prep *prep = container_of(kobj, \
+ struct damon_sysfs_prep, kobj); \
+ return sysfs_emit(buf, "%u\n", prep->field); \
+} \
+static ssize_t damon_sysfs_prep_##name##_store(struct kobject *kobj, \
+ struct kobj_attribute *attr, const char *buf, \
+ size_t count) \
+{ \
+ struct damon_sysfs_prep *prep = container_of(kobj, \
+ struct damon_sysfs_prep, kobj); \
+ bool v; \
+ int err = kstrtobool(buf, &v); \
+ if (err) \
+ return err; \
+ if (!mutex_trylock(&damon_sysfs_lock)) \
+ return -EBUSY; \
+ prep->field = v; \
+ mutex_unlock(&damon_sysfs_lock); \
+ return count; \
+} \
+static struct kobj_attribute damon_sysfs_prep_##name##_attr = __ATTR(name, \
+ 0600, damon_sysfs_prep_##name##_show, \
+ damon_sysfs_prep_##name##_store)
+
+DAMON_SYSFS_PREP_PERF_U32(type, perf_type);
+DAMON_SYSFS_PREP_PERF_U64(config, config);
+DAMON_SYSFS_PREP_PERF_U64(config1, config1);
+DAMON_SYSFS_PREP_PERF_U64(config2, config2);
+DAMON_SYSFS_PREP_PERF_U64(sample_period, sample_period);
+DAMON_SYSFS_PREP_PERF_U64(sample_freq, sample_freq);
+DAMON_SYSFS_PREP_PERF_U32(wakeup_events, wakeup_events);
+DAMON_SYSFS_PREP_PERF_U32(precise_ip, precise_ip);
+DAMON_SYSFS_PREP_PERF_BOOL(sample_phys_addr, sample_phys_addr);
+DAMON_SYSFS_PREP_PERF_BOOL(sample_weight_struct, sample_weight_struct);
+DAMON_SYSFS_PREP_PERF_BOOL(exclude_kernel, exclude_kernel);
+DAMON_SYSFS_PREP_PERF_BOOL(exclude_hv, exclude_hv);
+DAMON_SYSFS_PREP_PERF_BOOL(freq, freq);
+DAMON_SYSFS_PREP_PERF_BOOL(single_instance, single_instance);
+
static struct attribute *damon_sysfs_prep_attrs[] = {
&damon_sysfs_prep_prep_action_attr.attr,
+ &damon_sysfs_prep_type_attr.attr,
+ &damon_sysfs_prep_config_attr.attr,
+ &damon_sysfs_prep_config1_attr.attr,
+ &damon_sysfs_prep_config2_attr.attr,
+ &damon_sysfs_prep_sample_period_attr.attr,
+ &damon_sysfs_prep_sample_freq_attr.attr,
+ &damon_sysfs_prep_wakeup_events_attr.attr,
+ &damon_sysfs_prep_precise_ip_attr.attr,
+ &damon_sysfs_prep_sample_phys_addr_attr.attr,
+ &damon_sysfs_prep_sample_weight_struct_attr.attr,
+ &damon_sysfs_prep_exclude_kernel_attr.attr,
+ &damon_sysfs_prep_exclude_hv_attr.attr,
+ &damon_sysfs_prep_freq_attr.attr,
+ &damon_sysfs_prep_single_instance_attr.attr,
NULL,
};
ATTRIBUTE_GROUPS(damon_sysfs_prep);
@@ -2261,14 +2405,60 @@ static int damon_sysfs_set_preps(struct damon_probe *probe,
struct damon_sysfs_preps *sys_preps)
{
int i;
+ bool seen_perf_prep = false;
for (i = 0; i < sys_preps->nr; i++) {
struct damon_sysfs_prep *sys_prep = sys_preps->preps_arr[i];
struct damon_prep *prep;
+#ifndef CONFIG_DAMON_PERF_SOURCE
+ if (sys_prep->action == DAMON_PREP_PERF_EVENT)
+ return -EOPNOTSUPP;
+#endif
+
+ /* freq selects which of sample_freq and sample_period is set */
+ if (sys_prep->action == DAMON_PREP_PERF_EVENT &&
+ (sys_prep->freq ?
+ !sys_prep->sample_freq || sys_prep->sample_period :
+ !sys_prep->sample_period || sys_prep->sample_freq))
+ return -EINVAL;
+ /* precise_ip is a 2-bit bitfield in perf_event_attr */
+ if (sys_prep->action == DAMON_PREP_PERF_EVENT &&
+ sys_prep->precise_ip > 3)
+ return -EINVAL;
+ /* At most one perf-event prep per probe. */
+ if (sys_prep->action == DAMON_PREP_PERF_EVENT && seen_perf_prep)
+ return -EINVAL;
+ if (sys_prep->action == DAMON_PREP_PERF_EVENT)
+ seen_perf_prep = true;
+
prep = damon_new_prep(sys_prep->action);
if (!prep)
return -ENOMEM;
+ if (sys_prep->action == DAMON_PREP_PERF_EVENT) {
+ /*
+ * damon_new_prep() does not zero prep->perf; clear it
+ * so any field not assigned below starts from a known
+ * zero rather than kmalloc garbage.
+ */
+ memset(&prep->perf, 0, sizeof(prep->perf));
+ prep->perf.type = sys_prep->perf_type;
+ prep->perf.config = sys_prep->config;
+ prep->perf.config1 = sys_prep->config1;
+ prep->perf.config2 = sys_prep->config2;
+ prep->perf.sample_period = sys_prep->sample_period;
+ prep->perf.sample_freq = sys_prep->sample_freq;
+ prep->perf.wakeup_events = sys_prep->wakeup_events;
+ prep->perf.precise_ip = sys_prep->precise_ip;
+ prep->perf.sample_phys_addr =
+ sys_prep->sample_phys_addr;
+ prep->perf.sample_weight_struct =
+ sys_prep->sample_weight_struct;
+ prep->perf.exclude_kernel = sys_prep->exclude_kernel;
+ prep->perf.exclude_hv = sys_prep->exclude_hv;
+ prep->perf.freq = sys_prep->freq;
+ prep->perf.single_instance = sys_prep->single_instance;
+ }
damon_add_prep(probe, prep);
}
return 0;
@@ -2308,8 +2498,81 @@ static int damon_sysfs_set_filters(struct damon_probe *probe,
return 0;
}
-static int damon_sysfs_set_probe(struct damon_probe *probe,
- struct damon_sysfs_probe *sys_probe)
+#ifdef CONFIG_DAMON_PERF_SOURCE
+/*
+ * Build a perf-event probe descriptor from the probe's DAMON_PREP_PERF_EVENT
+ * prep and attach it to @probe. The descriptor is always carried (so the ring
+ * drain and the commit hand-off recognise the probe as event-driven), but the
+ * PMU counters are only armed when @arm is set.
+ *
+ * @arm is true when building the context that will actually run (turn-on
+ * path); it is false when building a param_ctx for a commit, which is
+ * discarded after validation. Arming a param_ctx would collide with the
+ * running context's perf-probe ownership and return -EBUSY, so the commit path
+ * defers arming to damon_commit_perf_probe().
+ */
+static int damon_sysfs_set_perf_probe(struct damon_ctx *ctx,
+ struct damon_probe *probe, bool arm)
+{
+ struct damon_prep *prep;
+
+ damon_for_each_prep(prep, probe) {
+ struct damon_perf_probe_event *event;
+ int err;
+
+ if (prep->action != DAMON_PREP_PERF_EVENT)
+ continue;
+
+ event = kzalloc_obj(*event, GFP_KERNEL);
+ if (!event)
+ return -ENOMEM;
+ event->attr.type = prep->perf.type;
+ event->attr.config = prep->perf.config;
+ event->attr.config1 = prep->perf.config1;
+ event->attr.config2 = prep->perf.config2;
+ event->attr.sample_period = prep->perf.sample_period;
+ event->attr.sample_freq = prep->perf.sample_freq;
+ event->attr.wakeup_events = prep->perf.wakeup_events;
+ event->attr.precise_ip = prep->perf.precise_ip;
+ event->attr.sample_phys_addr = prep->perf.sample_phys_addr;
+ event->attr.sample_weight_struct =
+ prep->perf.sample_weight_struct;
+ event->attr.exclude_kernel = prep->perf.exclude_kernel;
+ event->attr.exclude_hv = prep->perf.exclude_hv;
+ event->attr.freq = prep->perf.freq;
+ event->attr.single_instance = prep->perf.single_instance;
+
+ probe->perf_priv = event;
+ probe->event_driven = true;
+ if (arm) {
+ err = damon_perf_probe_setup(ctx, probe, event);
+ if (err) {
+ probe->perf_priv = NULL;
+ kfree(event);
+ return err;
+ }
+ }
+ /* At most one perf-event prep per probe. */
+ break;
+ }
+ return 0;
+}
+#endif /* CONFIG_DAMON_PERF_SOURCE */
+
+static bool damon_sysfs_probe_has_perf_prep(struct damon_probe *probe)
+{
+ struct damon_prep *prep;
+
+ damon_for_each_prep(prep, probe) {
+ if (prep->action == DAMON_PREP_PERF_EVENT)
+ return true;
+ }
+ return false;
+}
+
+static int damon_sysfs_set_probe(struct damon_ctx *ctx,
+ struct damon_probe *probe,
+ struct damon_sysfs_probe *sys_probe, bool arm)
{
struct damon_sysfs_filters *sys_filters;
struct damon_sysfs_preps *sys_preps;
@@ -2322,13 +2585,25 @@ static int damon_sysfs_set_probe(struct damon_probe *probe,
return err;
}
sys_filters = sys_probe->filters;
- if (!sys_filters)
- return 0;
- return damon_sysfs_set_filters(probe, sys_filters);
+ /* the drain does not apply probe filters to perf_event samples */
+ if (sys_filters && sys_filters->nr &&
+ damon_sysfs_probe_has_perf_prep(probe))
+ return -EINVAL;
+ if (sys_filters) {
+ err = damon_sysfs_set_filters(probe, sys_filters);
+ if (err)
+ return err;
+ }
+#ifdef CONFIG_DAMON_PERF_SOURCE
+ err = damon_sysfs_set_perf_probe(ctx, probe, arm);
+ if (err)
+ return err;
+#endif
+ return 0;
}
static int damon_sysfs_set_probes(struct damon_ctx *ctx,
- struct damon_sysfs_probes *sys_probes)
+ struct damon_sysfs_probes *sys_probes, bool arm)
{
int i, err;
@@ -2342,7 +2617,7 @@ static int damon_sysfs_set_probes(struct damon_ctx *ctx,
damon_add_probe(ctx, p);
sys_probe = sys_probes->probes_arr[i];
p->weight = sys_probe->weight;
- err = damon_sysfs_set_probe(p, sys_probe);
+ err = damon_sysfs_set_probe(ctx, p, sys_probe, arm);
if (err)
return err;
}
@@ -2443,7 +2718,7 @@ static inline bool damon_sysfs_kdamond_running(
}
static int damon_sysfs_apply_inputs(struct damon_ctx *ctx,
- struct damon_sysfs_context *sys_ctx)
+ struct damon_sysfs_context *sys_ctx, bool arm)
{
enum damon_ops_id ops_id;
int err;
@@ -2461,7 +2736,7 @@ static int damon_sysfs_apply_inputs(struct damon_ctx *ctx,
err = damon_sysfs_set_attrs(ctx, sys_ctx->attrs);
if (err)
return err;
- err = damon_sysfs_set_probes(ctx, sys_ctx->attrs->probes);
+ err = damon_sysfs_set_probes(ctx, sys_ctx->attrs->probes, arm);
if (err)
return err;
err = damon_sysfs_set_sample_control(&ctx->sample_control,
@@ -2475,7 +2750,7 @@ static int damon_sysfs_apply_inputs(struct damon_ctx *ctx,
}
static struct damon_ctx *damon_sysfs_build_ctx(
- struct damon_sysfs_context *sys_ctx);
+ struct damon_sysfs_context *sys_ctx, bool arm);
/*
* damon_sysfs_commit_input() - Commit user inputs to a running kdamond.
@@ -2495,7 +2770,8 @@ static int damon_sysfs_commit_input(void *data)
if (kdamond->contexts->nr != 1)
return -EINVAL;
- param_ctx = damon_sysfs_build_ctx(kdamond->contexts->contexts_arr[0]);
+ param_ctx = damon_sysfs_build_ctx(kdamond->contexts->contexts_arr[0],
+ false);
if (IS_ERR(param_ctx))
return PTR_ERR(param_ctx);
err = damon_commit_ctx(kdamond->damon_ctx, param_ctx);
@@ -2553,7 +2829,7 @@ static int damon_sysfs_upd_tuned_intervals(void *data)
}
static struct damon_ctx *damon_sysfs_build_ctx(
- struct damon_sysfs_context *sys_ctx)
+ struct damon_sysfs_context *sys_ctx, bool arm)
{
struct damon_ctx *ctx = damon_new_ctx();
int err;
@@ -2561,7 +2837,7 @@ static struct damon_ctx *damon_sysfs_build_ctx(
if (!ctx)
return ERR_PTR(-ENOMEM);
- err = damon_sysfs_apply_inputs(ctx, sys_ctx);
+ err = damon_sysfs_apply_inputs(ctx, sys_ctx, arm);
if (err) {
damon_destroy_ctx(ctx);
return ERR_PTR(err);
@@ -2613,7 +2889,7 @@ static int damon_sysfs_turn_damon_on(struct damon_sysfs_kdamond *kdamond)
if (!repeat_call_control)
return -ENOMEM;
- ctx = damon_sysfs_build_ctx(kdamond->contexts->contexts_arr[0]);
+ ctx = damon_sysfs_build_ctx(kdamond->contexts->contexts_arr[0], true);
if (IS_ERR(ctx)) {
kfree(repeat_call_control);
return PTR_ERR(ctx);
--
Git-157)
^ permalink raw reply [flat|nested] 11+ messages in thread* [RFC PATCH v4 7/9] mm/damon/tests/drain-kunit: kunit for report rings and ring drain
2026-10-05 8:46 [RFC PATCH v4 0/9] mm/damon: hardware-sampled access reports Ravi Jonnalagadda
` (5 preceding siblings ...)
2026-10-05 8:46 ` [RFC PATCH v4 6/9] mm/damon/sysfs: expose perf_event prep attributes Ravi Jonnalagadda
@ 2026-10-05 8:46 ` Ravi Jonnalagadda
2026-10-05 8:46 ` [RFC PATCH v4 8/9] mm/damon/core: cap the region merge threshold per target Ravi Jonnalagadda
` (2 subsequent siblings)
9 siblings, 0 replies; 11+ messages in thread
From: Ravi Jonnalagadda @ 2026-10-05 8:46 UTC (permalink / raw)
To: SJ Park, Andrew Morton
Cc: damon, linux-mm, linux-kernel, Gregory Price, David Rientjes,
Wei Xu, Jonathan Corbet, Bijan Tabatabai, Ajay Joshi,
Honggyu Kim, Yunjeong Mun, Akinobu Mita, Lian Wang, Kunwu Chan,
Ravi Jonnalagadda, Jonathan Cameron
Add kunit coverage for the per-context perf report rings and the drain
that credits regions from them. Wire the suites into core.c (included
after the drain and its counters are defined) so CONFIG_DAMON_KUNIT_TEST=y
builds them.
perf-kunit.h covers the ring itself: reports injected and accepted,
overflow safety on wrap, and a report that carries no owning context
being dropped.
drain-kunit.h covers the drain and the damon_report_access() producer:
- a vaddr report credited to the region holding its address, and not
credited when its thread group id does not match the target
- a paddr report credited with no thread-group filtering
- a perf report credited to the probe hits of its probe index
- reports adding at most one probe hit per sampling interval and none
changing nr_accesses, with and without a probe weight set
- per-context isolation: a report reaches only its own context's ring
- the producer's return value, and a full ring counted in the ring-full
counter and freed again by a drain
- the address-space match: a report whose virtual address falls inside a
region of a physical-address target, and whose physical address falls
outside it, is not credited
Signed-off-by: Ravi Jonnalagadda <ravis.opensrc@gmail.com>
---
mm/damon/core.c | 2 +
mm/damon/tests/drain-kunit.h | 786 +++++++++++++++++++++++++++++++++++++++++++
mm/damon/tests/perf-kunit.h | 133 ++++++++
3 files changed, 921 insertions(+)
diff --git a/mm/damon/core.c b/mm/damon/core.c
index c2acae6e19ec..e3c29d7cff23 100644
--- a/mm/damon/core.c
+++ b/mm/damon/core.c
@@ -5208,3 +5208,5 @@ struct damon_region *damon_search(unsigned long addr, struct pid *pid)
subsys_initcall(damon_init);
#include "tests/core-kunit.h"
+#include "tests/drain-kunit.h"
+#include "tests/perf-kunit.h"
diff --git a/mm/damon/tests/drain-kunit.h b/mm/damon/tests/drain-kunit.h
new file mode 100644
index 000000000000..d6e576724421
--- /dev/null
+++ b/mm/damon/tests/drain-kunit.h
@@ -0,0 +1,786 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * DAMON kunit tests for the unified paddr/vaddr report drain path.
+ *
+ * Included at the bottom of core.c (after kdamond_check_reported_accesses
+ * is defined) so the static function is visible.
+ */
+
+#ifdef CONFIG_DAMON_KUNIT_TEST
+
+#ifndef _DAMON_DRAIN_KUNIT_H
+#define _DAMON_DRAIN_KUNIT_H
+
+#include <kunit/test.h>
+#include <linux/damon.h>
+
+/*
+ * Reports are dispatched by probe_idx: probe_idx == DAMON_PROBE_IDX_NONE (0)
+ * has no ring to feed and is dropped by damon_report_access(); probe_idx >= 1
+ * lands in the owning context's per-context perf ring (ctx->perf_rings). The
+ * drain dispatcher kdamond_check_reported_accesses() drains the perf ring only
+ * for a ctx that has event-driven probes.
+ *
+ * Attach a dummy event-driven probe AND allocate the ctx's per-ctx perf ring
+ * so the ctx both drains the perf ring and has ring storage for injected
+ * probe_idx>=1 reports. In a live run damon_perf_probe_setup() allocates the
+ * ring; kunit has no real perf event, so it allocates directly. Returns
+ * 0/-ENOMEM.
+ */
+static int damon_test_attach_perf_probe(struct damon_ctx *ctx)
+{
+ struct damon_probe *p = damon_new_probe();
+ int err;
+
+ if (!p)
+ return -ENOMEM;
+ p->event_driven = true;
+ damon_add_probe(ctx, p);
+
+ err = damon_ctx_alloc_perf_ring(ctx);
+ if (err)
+ return err;
+ return 0;
+}
+
+/*
+ * Mark @ctx as monitoring the physical address space.
+ *
+ * The drain matches a report against the address space of the context, which
+ * damon_target_has_pid() derives from ctx->ops.id, so a context whose targets
+ * carry no pid needs the paddr id for its reports to be matched by paddr.
+ * Only the id is set: the drain reads no other operations field, and these
+ * tests call it directly rather than through a kdamond.
+ */
+static void damon_test_set_paddr_ctx(struct damon_ctx *ctx)
+{
+ ctx->ops.id = DAMON_OPS_PADDR;
+}
+
+/*
+ * Test A: vaddr entry with a matching thread group id drains correctly.
+ *
+ * Create a vaddr ctx with target pid=current, region [0x1000, 0x2000).
+ * Inject entry: paddr=0, vaddr=0x1500, tgid=current tgid, probe_idx=1, ctx=ctx.
+ * After drain: probe_hits[0]==1 (probe_idx 1 stored 0-based), samples_drained
+ * increments.
+ */
+static void damon_test_unified_vaddr_match(struct kunit *test)
+{
+ struct damon_ctx *ctx;
+ struct damon_target *t;
+ struct damon_region *r;
+ struct damon_access_report rep = {
+ .paddr = 0,
+ .vaddr = 0x1500,
+ .probe_idx = 1,
+ .size = PAGE_SIZE,
+ };
+ unsigned long before, after;
+ int hits;
+
+ ctx = damon_new_ctx();
+ if (!ctx)
+ kunit_skip(test, "ctx alloc failed");
+ if (damon_test_attach_perf_probe(ctx)) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "perf probe alloc failed");
+ }
+
+ t = damon_new_target();
+ if (!t) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "target alloc failed");
+ }
+ t->pid = get_pid(task_tgid(current));
+ if (!t->pid) {
+ damon_free_target(t);
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "pid alloc failed");
+ }
+ rep.tgid = task_tgid_nr(current);
+ rep.ctx = ctx; /* route to this ctx's per-ctx perf ring */
+
+ /*
+ * Region must fully contain the report [vaddr, vaddr + size): a report
+ * straddling the region end is rejected by the drain. With
+ * vaddr=0x1500 and size=PAGE_SIZE the region must reach >= 0x2500.
+ */
+ r = damon_new_region(0x1000, 0x3000);
+ if (!r) {
+ put_pid(t->pid);
+ damon_free_target(t);
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "region alloc failed");
+ }
+ damon_add_region(r, t);
+ damon_add_target(ctx, t);
+
+ rep.report_jiffies = jiffies;
+ before = damon_get_samples_drained();
+ damon_report_access(&rep);
+ kdamond_check_reported_accesses(ctx);
+ after = damon_get_samples_drained();
+
+ hits = 0;
+ damon_for_each_region(r, t)
+ hits += r->probe_hits[0];
+
+ KUNIT_EXPECT_EQ(test, hits, 1);
+ KUNIT_EXPECT_GT(test, after, before);
+
+ put_pid(t->pid);
+ damon_destroy_ctx(ctx);
+}
+
+/*
+ * Test B: a vaddr entry whose thread group id matches no target is dropped.
+ *
+ * Same setup but inject with a thread group id no target carries.
+ * probe_hits[0]==0 (probe_idx 1 stored 0-based), samples_no_region increments.
+ */
+static void damon_test_unified_vaddr_tgid_mismatch(struct kunit *test)
+{
+ struct damon_ctx *ctx;
+ struct damon_target *t;
+ struct damon_region *r;
+ struct damon_access_report rep = {
+ .paddr = 0,
+ .vaddr = 0x1500,
+ .tgid = 9999, /* matches no target */
+ .probe_idx = 1,
+ .size = PAGE_SIZE,
+ };
+ unsigned long before, after;
+ int hits;
+
+ ctx = damon_new_ctx();
+ if (!ctx)
+ kunit_skip(test, "ctx alloc failed");
+ if (damon_test_attach_perf_probe(ctx)) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "perf probe alloc failed");
+ }
+
+ t = damon_new_target();
+ if (!t) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "target alloc failed");
+ }
+ t->pid = get_pid(task_tgid(current));
+ if (!t->pid) {
+ damon_free_target(t);
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "pid alloc failed");
+ }
+ rep.ctx = ctx;
+
+ /* wide enough for the report: the id mismatch is the sole reject */
+ r = damon_new_region(0x1000, 0x3000);
+ if (!r) {
+ put_pid(t->pid);
+ damon_free_target(t);
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "region alloc failed");
+ }
+ damon_add_region(r, t);
+ damon_add_target(ctx, t);
+
+ rep.report_jiffies = jiffies;
+ before = damon_get_samples_no_region();
+ damon_report_access(&rep);
+ kdamond_check_reported_accesses(ctx);
+ after = damon_get_samples_no_region();
+
+ hits = 0;
+ damon_for_each_region(r, t)
+ hits += r->probe_hits[0];
+
+ KUNIT_EXPECT_EQ(test, hits, 0);
+ KUNIT_EXPECT_GT(test, after, before);
+
+ put_pid(t->pid);
+ damon_destroy_ctx(ctx);
+}
+
+/*
+ * Test C: paddr entry drains correctly (no id filter for paddr ops).
+ *
+ * Create a paddr ctx (no pid), region [0x10000, 0x20000).
+ * Inject: paddr=0x15000, vaddr=0, probe_idx=1, ctx=ctx.
+ * After drain: probe_hits[0]==1 (probe_idx 1 stored 0-based).
+ */
+static void damon_test_unified_paddr_no_regression(struct kunit *test)
+{
+ struct damon_ctx *ctx;
+ struct damon_target *t;
+ struct damon_region *r;
+ struct damon_access_report rep = {
+ .paddr = 0x15000,
+ .vaddr = 0,
+ .tid = 0,
+ .probe_idx = 1,
+ .size = PAGE_SIZE,
+ };
+ unsigned long before, after;
+ int hits;
+
+ ctx = damon_new_ctx();
+ if (!ctx)
+ kunit_skip(test, "ctx alloc failed");
+ if (damon_test_attach_perf_probe(ctx)) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "perf probe alloc failed");
+ }
+
+ t = damon_new_target();
+ if (!t) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "target alloc failed");
+ }
+ t->pid = NULL; /* paddr target: no pid */
+ damon_test_set_paddr_ctx(ctx);
+ rep.ctx = ctx;
+
+ r = damon_new_region(0x10000, 0x20000);
+ if (!r) {
+ damon_free_target(t);
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "region alloc failed");
+ }
+ damon_add_region(r, t);
+ damon_add_target(ctx, t);
+
+ rep.report_jiffies = jiffies;
+ before = damon_get_samples_drained();
+ damon_report_access(&rep);
+ kdamond_check_reported_accesses(ctx);
+ after = damon_get_samples_drained();
+
+ hits = 0;
+ damon_for_each_region(r, t)
+ hits += r->probe_hits[0];
+
+ KUNIT_EXPECT_EQ(test, hits, 1);
+ KUNIT_EXPECT_GT(test, after, before);
+
+ damon_destroy_ctx(ctx);
+}
+
+/* Reports add one probe hit per interval and never change nr_accesses. */
+static void damon_test_report_hits_only(struct kunit *test)
+{
+ struct damon_ctx *ctx;
+ struct damon_target *t;
+ struct damon_region *r;
+ struct damon_access_report rep = {
+ .paddr = 0x15000,
+ .probe_idx = 1,
+ .size = PAGE_SIZE,
+ };
+
+ ctx = damon_new_ctx();
+ if (!ctx)
+ kunit_skip(test, "ctx alloc failed");
+ if (damon_test_attach_perf_probe(ctx)) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "perf probe alloc failed");
+ }
+
+ t = damon_new_target();
+ if (!t) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "target alloc failed");
+ }
+ t->pid = NULL;
+ damon_test_set_paddr_ctx(ctx);
+ rep.ctx = ctx;
+
+ r = damon_new_region(0x10000, 0x20000);
+ if (!r) {
+ damon_free_target(t);
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "region alloc failed");
+ }
+ damon_add_region(r, t);
+ damon_add_target(ctx, t);
+
+ rep.report_jiffies = jiffies;
+ preempt_disable();
+ damon_report_access(&rep);
+ damon_report_access(&rep);
+ preempt_enable();
+ kdamond_check_reported_accesses(ctx);
+ KUNIT_EXPECT_EQ(test, r->nr_accesses, 0u);
+ KUNIT_EXPECT_EQ(test, r->probe_hits[0], 1);
+
+ /* a report drained in the next interval adds another hit */
+ ctx->passed_sample_intervals++;
+ rep.report_jiffies = jiffies;
+ preempt_disable();
+ damon_report_access(&rep);
+ preempt_enable();
+ kdamond_check_reported_accesses(ctx);
+ KUNIT_EXPECT_EQ(test, r->nr_accesses, 0u);
+ KUNIT_EXPECT_EQ(test, r->probe_hits[0], 2);
+
+ damon_destroy_ctx(ctx);
+}
+
+/* The same holds with a probe weight set (data attributes-only mode). */
+static void damon_test_report_hits_only_weighted(struct kunit *test)
+{
+ struct damon_ctx *ctx;
+ struct damon_target *t;
+ struct damon_region *r;
+ struct damon_probe *p;
+ struct damon_access_report rep = {
+ .paddr = 0x15000,
+ .probe_idx = 1,
+ .size = PAGE_SIZE,
+ };
+
+ ctx = damon_new_ctx();
+ if (!ctx)
+ kunit_skip(test, "ctx alloc failed");
+ if (damon_test_attach_perf_probe(ctx)) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "perf probe alloc failed");
+ }
+ damon_for_each_probe(p, ctx)
+ p->weight = 1;
+
+ t = damon_new_target();
+ if (!t) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "target alloc failed");
+ }
+ t->pid = NULL;
+ damon_test_set_paddr_ctx(ctx);
+ rep.ctx = ctx;
+
+ r = damon_new_region(0x10000, 0x20000);
+ if (!r) {
+ damon_free_target(t);
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "region alloc failed");
+ }
+ damon_add_region(r, t);
+ damon_add_target(ctx, t);
+
+ preempt_disable();
+ damon_report_access(&rep);
+ damon_report_access(&rep);
+ preempt_enable();
+ kdamond_check_reported_accesses(ctx);
+ KUNIT_EXPECT_EQ(test, r->nr_accesses, 0u);
+ KUNIT_EXPECT_EQ(test, r->probe_hits[0], 1);
+
+ ctx->passed_sample_intervals++;
+ preempt_disable();
+ damon_report_access(&rep);
+ preempt_enable();
+ kdamond_check_reported_accesses(ctx);
+ KUNIT_EXPECT_EQ(test, r->nr_accesses, 0u);
+ KUNIT_EXPECT_EQ(test, r->probe_hits[0], 2);
+
+ damon_destroy_ctx(ctx);
+}
+
+static void damon_test_ring1_perf_credit(struct kunit *test)
+{
+ struct damon_ctx *ctx;
+ struct damon_target *t;
+ struct damon_region *r;
+ struct damon_access_report rep = {
+ .paddr = 0x15000,
+ .vaddr = 0,
+ .tid = 0,
+ .probe_idx = 1, /* -> per-ctx perf ring */
+ .size = PAGE_SIZE,
+ };
+ unsigned long before, after;
+ int hits;
+
+ ctx = damon_new_ctx();
+ if (!ctx)
+ kunit_skip(test, "ctx alloc failed");
+ if (damon_test_attach_perf_probe(ctx)) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "perf probe alloc failed");
+ }
+
+ t = damon_new_target();
+ if (!t) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "target alloc failed");
+ }
+ t->pid = NULL;
+ damon_test_set_paddr_ctx(ctx);
+ rep.ctx = ctx;
+
+ r = damon_new_region(0x10000, 0x20000);
+ if (!r) {
+ damon_free_target(t);
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "region alloc failed");
+ }
+ damon_add_region(r, t);
+ damon_add_target(ctx, t);
+
+ rep.report_jiffies = jiffies;
+ before = damon_get_samples_drained();
+ damon_report_access(&rep);
+ kdamond_check_reported_accesses(ctx);
+ after = damon_get_samples_drained();
+
+ hits = 0;
+ damon_for_each_region(r, t)
+ hits += r->probe_hits[0];
+
+ KUNIT_EXPECT_EQ(test, hits, 1);
+ KUNIT_EXPECT_GT(test, after, before);
+
+ damon_destroy_ctx(ctx);
+}
+
+/*
+ * Test per-context perf ring isolation.
+ *
+ * Two independent perf ctxs (each with its own event-driven probe and its own
+ * per-ctx perf ring) each receive one probe_idx=1 report tagged with their
+ * respective ctx. Each ctx must credit exactly its own report and see nothing
+ * from the other: perf reports route to the owning ctx's ring, and two
+ * perf-driven ctxs coexist without sharing a ring.
+ */
+static void damon_test_perf_per_ctx_isolation(struct kunit *test)
+{
+ struct damon_ctx *ctx_a, *ctx_b;
+ struct damon_target *ta, *tb;
+ struct damon_region *ra, *rb;
+ struct damon_access_report rep_a = {
+ .paddr = 0x15000, .probe_idx = 1, .size = PAGE_SIZE,
+ };
+ struct damon_access_report rep_b = {
+ .paddr = 0x35000, .probe_idx = 1, .size = PAGE_SIZE,
+ };
+ int hits_a, hits_b;
+
+ ctx_a = damon_new_ctx();
+ ctx_b = damon_new_ctx();
+ if (!ctx_a || !ctx_b) {
+ if (ctx_a)
+ damon_destroy_ctx(ctx_a);
+ if (ctx_b)
+ damon_destroy_ctx(ctx_b);
+ kunit_skip(test, "ctx alloc failed");
+ }
+ if (damon_test_attach_perf_probe(ctx_a) ||
+ damon_test_attach_perf_probe(ctx_b)) {
+ damon_destroy_ctx(ctx_a);
+ damon_destroy_ctx(ctx_b);
+ kunit_skip(test, "perf probe alloc failed");
+ }
+
+ ta = damon_new_target();
+ tb = damon_new_target();
+ if (!ta || !tb) {
+ if (ta)
+ damon_free_target(ta);
+ if (tb)
+ damon_free_target(tb);
+ damon_destroy_ctx(ctx_a);
+ damon_destroy_ctx(ctx_b);
+ kunit_skip(test, "target alloc failed");
+ }
+ ta->pid = NULL;
+ tb->pid = NULL;
+ damon_test_set_paddr_ctx(ctx_a);
+ damon_test_set_paddr_ctx(ctx_b);
+
+ ra = damon_new_region(0x10000, 0x20000); /* holds rep_a paddr */
+ rb = damon_new_region(0x30000, 0x40000); /* holds rep_b paddr */
+ if (!ra || !rb) {
+ if (ra)
+ damon_free_region(ra);
+ if (rb)
+ damon_free_region(rb);
+ damon_free_target(ta);
+ damon_free_target(tb);
+ damon_destroy_ctx(ctx_a);
+ damon_destroy_ctx(ctx_b);
+ kunit_skip(test, "region alloc failed");
+ }
+ damon_add_region(ra, ta);
+ damon_add_target(ctx_a, ta);
+ damon_add_region(rb, tb);
+ damon_add_target(ctx_b, tb);
+
+ /* Each report is tagged with its owning ctx. */
+ rep_a.ctx = ctx_a;
+ rep_b.ctx = ctx_b;
+ rep_a.report_jiffies = jiffies;
+ rep_b.report_jiffies = jiffies;
+
+ /*
+ * Report into both ctx rings, then drain each ctx. ctx_a must credit
+ * only rep_a; ctx_b must credit only rep_b -- no cross-talk.
+ */
+ damon_report_access(&rep_a);
+ damon_report_access(&rep_b);
+ kdamond_check_reported_accesses(ctx_a);
+ kdamond_check_reported_accesses(ctx_b);
+
+ hits_a = 0;
+ damon_for_each_region(ra, ta)
+ hits_a += ra->probe_hits[0];
+ hits_b = 0;
+ damon_for_each_region(rb, tb)
+ hits_b += rb->probe_hits[0];
+
+ /* each ctx credited its own report only */
+ KUNIT_EXPECT_EQ(test, hits_a, 1);
+ KUNIT_EXPECT_EQ(test, hits_b, 1);
+
+ damon_destroy_ctx(ctx_a);
+ damon_destroy_ctx(ctx_b);
+}
+
+/*
+ * Test the queued/dropped return value, and that a ring-full drop is counted
+ * as ring-full rather than busy-guard.
+ *
+ * A per-context perf ring is private to its ctx, so a freshly created ctx
+ * starts with an empty ring nobody else writes to and the counts are exact.
+ * Preemption is held across the loop so every report targets the same CPU's
+ * ring, per the SPSC invariant damon_report_access() documents.
+ *
+ * A ring holds DAMON_REPORT_RING_SIZE - 1 entries (one slot is kept empty to
+ * distinguish full from empty), so exactly that many reports are queued and
+ * every one after that is dropped.
+ */
+static void damon_test_report_return_value(struct kunit *test)
+{
+ struct damon_ctx *ctx;
+ struct damon_access_report rep = {
+ .paddr = 0x15000, .probe_idx = 1, .size = PAGE_SIZE,
+ };
+ unsigned long full_before, busy_before;
+ unsigned int queued = 0, dropped = 0;
+ int i;
+
+ ctx = damon_new_ctx();
+ if (!ctx)
+ kunit_skip(test, "ctx alloc failed");
+ if (damon_test_attach_perf_probe(ctx)) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "perf probe alloc failed");
+ }
+ rep.ctx = ctx;
+
+ preempt_disable();
+ full_before = damon_get_report_ring_full();
+ busy_before = damon_get_report_busy_drop();
+
+ /* One past capacity, so the last iteration must be a drop. */
+ for (i = 0; i < DAMON_REPORT_RING_SIZE; i++) {
+ if (damon_report_access(&rep))
+ queued++;
+ else
+ dropped++;
+ }
+ preempt_enable();
+
+ KUNIT_EXPECT_EQ(test, queued, (unsigned int)DAMON_REPORT_RING_SIZE - 1);
+ KUNIT_EXPECT_EQ(test, dropped, 1u);
+ /* No NMI nests here, so the drop must be the full ring. */
+ KUNIT_EXPECT_GT(test, damon_get_report_ring_full(), full_before);
+ KUNIT_EXPECT_EQ(test, damon_get_report_busy_drop(), busy_before);
+
+ damon_destroy_ctx(ctx);
+}
+
+/*
+ * Test that draining restores capacity: fill the ring, drain it via the
+ * dispatcher, then report again and expect the report to be queued.
+ */
+static void damon_test_report_drain_restores_capacity(struct kunit *test)
+{
+ struct damon_ctx *ctx;
+ struct damon_target *t;
+ struct damon_region *r;
+ struct damon_access_report rep = {
+ .paddr = 0x15000, .probe_idx = 1, .size = PAGE_SIZE,
+ };
+ bool queued;
+ int i;
+
+ ctx = damon_new_ctx();
+ if (!ctx)
+ kunit_skip(test, "ctx alloc failed");
+ if (damon_test_attach_perf_probe(ctx)) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "perf probe alloc failed");
+ }
+
+ t = damon_new_target();
+ if (!t) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "target alloc failed");
+ }
+ t->pid = NULL;
+ damon_test_set_paddr_ctx(ctx);
+ rep.ctx = ctx;
+
+ r = damon_new_region(0x10000, 0x20000);
+ if (!r) {
+ damon_free_target(t);
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "region alloc failed");
+ }
+ damon_add_region(r, t);
+ damon_add_target(ctx, t);
+
+ /* Fill the ring: the last report is dropped. */
+ preempt_disable();
+ for (i = 0; i < DAMON_REPORT_RING_SIZE; i++)
+ damon_report_access(&rep);
+ queued = damon_report_access(&rep);
+ preempt_enable();
+ KUNIT_EXPECT_FALSE(test, queued);
+
+ kdamond_check_reported_accesses(ctx);
+
+ /* Capacity is back. */
+ KUNIT_EXPECT_TRUE(test, damon_report_access(&rep));
+
+ damon_destroy_ctx(ctx);
+}
+
+/*
+ * Test that a report is matched by the address space of the target rather than
+ * by which address it carries.
+ *
+ * Create a paddr ctx with an event-driven probe, region [0x10000, 0x20000).
+ * Inject a report whose vaddr falls inside that region and whose paddr falls
+ * outside it. A paddr target matches the paddr, so the report finds no
+ * region and is counted as such.
+ */
+static void damon_test_report_addr_space_keyed(struct kunit *test)
+{
+ struct damon_ctx *ctx;
+ struct damon_target *t;
+ struct damon_region *r;
+ struct damon_access_report rep = {
+ .paddr = 0x95000, /* outside the region */
+ .vaddr = 0x15000, /* inside the region */
+ .tid = 0,
+ .probe_idx = 1,
+ .size = PAGE_SIZE,
+ };
+ unsigned long before, after;
+
+ ctx = damon_new_ctx();
+ if (!ctx)
+ kunit_skip(test, "ctx alloc failed");
+ if (damon_test_attach_perf_probe(ctx)) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "perf probe alloc failed");
+ }
+ rep.ctx = ctx; /* route to this ctx's per-ctx perf ring */
+
+ t = damon_new_target();
+ if (!t) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "target alloc failed");
+ }
+ t->pid = NULL; /* paddr target: no pid */
+ damon_test_set_paddr_ctx(ctx);
+
+ r = damon_new_region(0x10000, 0x20000);
+ if (!r) {
+ damon_free_target(t);
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "region alloc failed");
+ }
+ damon_add_region(r, t);
+ damon_add_target(ctx, t);
+
+ rep.report_jiffies = jiffies;
+ before = damon_get_samples_no_region();
+ damon_report_access(&rep);
+ kdamond_check_reported_accesses(ctx);
+ after = damon_get_samples_no_region();
+
+ /* The vaddr was not used to match a paddr target. */
+ KUNIT_EXPECT_GT(test, after, before);
+ damon_for_each_region(r, t)
+ KUNIT_EXPECT_EQ(test, r->probe_hits[0], 0);
+
+ damon_destroy_ctx(ctx);
+}
+
+/*
+ * Test that filling a perf ring beyond capacity increments the ring-full
+ * counter. The existing damon_test_report_drain_restores_capacity verifies
+ * capacity returns after a drain; this verifies the counter side of the
+ * same overflow.
+ */
+static void damon_test_ring_full_counter_increments(struct kunit *test)
+{
+ struct damon_ctx *ctx;
+ struct damon_access_report rep = {
+ .paddr = 0x15000, .probe_idx = 1, .size = PAGE_SIZE,
+ };
+ unsigned long full_before;
+ int i;
+
+ ctx = damon_new_ctx();
+ if (!ctx)
+ kunit_skip(test, "ctx alloc failed");
+ if (damon_test_attach_perf_probe(ctx)) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "perf probe alloc failed");
+ }
+ damon_test_set_paddr_ctx(ctx);
+ rep.ctx = ctx;
+
+ full_before = damon_get_report_ring_full();
+
+ /* Fill the ring to capacity, then push one more to trigger overflow. */
+ preempt_disable();
+ for (i = 0; i < DAMON_REPORT_RING_SIZE; i++)
+ damon_report_access(&rep);
+ damon_report_access(&rep); /* this one overflows */
+ preempt_enable();
+
+ KUNIT_EXPECT_GT(test, damon_get_report_ring_full(), full_before);
+
+ damon_destroy_ctx(ctx);
+}
+
+static struct kunit_case damon_drain_test_cases[] = {
+ KUNIT_CASE(damon_test_unified_vaddr_match),
+ KUNIT_CASE(damon_test_unified_vaddr_tgid_mismatch),
+ KUNIT_CASE(damon_test_unified_paddr_no_regression),
+ KUNIT_CASE(damon_test_report_hits_only),
+ KUNIT_CASE(damon_test_report_hits_only_weighted),
+ KUNIT_CASE(damon_test_ring1_perf_credit),
+ KUNIT_CASE(damon_test_perf_per_ctx_isolation),
+ KUNIT_CASE(damon_test_report_return_value),
+ KUNIT_CASE(damon_test_report_drain_restores_capacity),
+ KUNIT_CASE(damon_test_report_addr_space_keyed),
+ KUNIT_CASE(damon_test_ring_full_counter_increments),
+ {}
+};
+
+static struct kunit_suite damon_drain_test_suite = {
+ .name = "damon_drain",
+ .test_cases = damon_drain_test_cases,
+};
+
+kunit_test_suite(damon_drain_test_suite);
+
+#endif /* _DAMON_DRAIN_KUNIT_H */
+
+#endif /* CONFIG_DAMON_KUNIT_TEST */
diff --git a/mm/damon/tests/perf-kunit.h b/mm/damon/tests/perf-kunit.h
new file mode 100644
index 000000000000..de4472b54224
--- /dev/null
+++ b/mm/damon/tests/perf-kunit.h
@@ -0,0 +1,133 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * DAMON kunit tests for the per-context perf report ring.
+ *
+ * Included at the bottom of core.c, after tests/drain-kunit.h, whose
+ * damon_test_attach_perf_probe() helper these tests reuse.
+ */
+
+#ifdef CONFIG_DAMON_KUNIT_TEST
+
+#ifndef _DAMON_PERF_KUNIT_H
+#define _DAMON_PERF_KUNIT_H
+
+#include <kunit/test.h>
+#include <linux/damon.h>
+
+/*
+ * A report with probe_idx >= 1 is enqueued into the ring of the context named
+ * by report->ctx, so these tests build a context with an allocated perf ring
+ * and point the injected reports at it. A freshly allocated ring is empty,
+ * which makes the accepted and rejected counts below exact.
+ */
+
+/*
+ * Test A: perf ring basic write
+ *
+ * Inject reports into a context's perf ring via damon_report_access() and
+ * verify each one is accepted.
+ */
+static void damon_test_perf_ring_basic(struct kunit *test)
+{
+ struct damon_ctx *ctx;
+ struct damon_access_report report = {
+ .paddr = 0x1000, .size = PAGE_SIZE, .probe_idx = 1,
+ };
+ int i;
+
+ ctx = damon_new_ctx();
+ if (!ctx)
+ kunit_skip(test, "ctx alloc failed");
+ if (damon_test_attach_perf_probe(ctx)) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "perf ring alloc failed");
+ }
+ report.ctx = ctx;
+
+ for (i = 0; i < 3; i++)
+ KUNIT_EXPECT_TRUE(test, damon_report_access(&report));
+
+ damon_destroy_ctx(ctx);
+}
+
+/*
+ * Test B: ring overflow is reported and does not corrupt head/tail
+ *
+ * Fill a context's perf ring to capacity and verify that further writes are
+ * refused and counted rather than overwriting live entries. The ring holds
+ * DAMON_REPORT_RING_SIZE - 1 entries, one slot being reserved to distinguish
+ * full from empty, so the last two of the writes below must be refused.
+ */
+static void damon_test_perf_ring_overflow_safety(struct kunit *test)
+{
+ struct damon_ctx *ctx;
+ struct damon_access_report report = {
+ .paddr = 0x3000, .size = PAGE_SIZE, .probe_idx = 1,
+ };
+ unsigned long overflow_before, overflow_after;
+ int queued = 0, refused = 0;
+ int i;
+
+ ctx = damon_new_ctx();
+ if (!ctx)
+ kunit_skip(test, "ctx alloc failed");
+ if (damon_test_attach_perf_probe(ctx)) {
+ damon_destroy_ctx(ctx);
+ kunit_skip(test, "perf ring alloc failed");
+ }
+ report.ctx = ctx;
+
+ /* Pinned so every write lands in the same CPU's ring. */
+ preempt_disable();
+ overflow_before = damon_get_report_overflow();
+
+ for (i = 0; i < DAMON_REPORT_RING_SIZE + 1; i++) {
+ if (damon_report_access(&report))
+ queued++;
+ else
+ refused++;
+ }
+
+ overflow_after = damon_get_report_overflow();
+ preempt_enable();
+
+ KUNIT_EXPECT_EQ(test, queued, DAMON_REPORT_RING_SIZE - 1);
+ KUNIT_EXPECT_EQ(test, refused, 2);
+ KUNIT_EXPECT_GT(test, overflow_after, overflow_before);
+
+ damon_destroy_ctx(ctx);
+}
+
+/*
+ * Test C: a perf report without an owning context is refused
+ *
+ * A report with probe_idx >= 1 but no ctx cannot be routed to a ring. Verify
+ * it is refused instead of dereferenced.
+ */
+static void damon_test_perf_report_requires_ctx(struct kunit *test)
+{
+ struct damon_access_report report = {
+ .paddr = 0x5000, .size = PAGE_SIZE, .probe_idx = 1,
+ .ctx = NULL,
+ };
+
+ KUNIT_EXPECT_FALSE(test, damon_report_access(&report));
+}
+
+static struct kunit_case damon_perf_test_cases[] = {
+ KUNIT_CASE(damon_test_perf_ring_basic),
+ KUNIT_CASE(damon_test_perf_ring_overflow_safety),
+ KUNIT_CASE(damon_test_perf_report_requires_ctx),
+ {}
+};
+
+static struct kunit_suite damon_perf_test_suite = {
+ .name = "damon_perf",
+ .test_cases = damon_perf_test_cases,
+};
+
+kunit_test_suite(damon_perf_test_suite);
+
+#endif /* _DAMON_PERF_KUNIT_H */
+
+#endif /* CONFIG_DAMON_KUNIT_TEST */
--
Git-157)
^ permalink raw reply [flat|nested] 11+ messages in thread* [RFC PATCH v4 8/9] mm/damon/core: cap the region merge threshold per target
2026-10-05 8:46 [RFC PATCH v4 0/9] mm/damon: hardware-sampled access reports Ravi Jonnalagadda
` (6 preceding siblings ...)
2026-10-05 8:46 ` [RFC PATCH v4 7/9] mm/damon/tests/drain-kunit: kunit for report rings and ring drain Ravi Jonnalagadda
@ 2026-10-05 8:46 ` Ravi Jonnalagadda
2026-10-05 8:46 ` [RFC PATCH v4 9/9] mm/damon/core: apply probe_hits_wsum filters to node_eligible_mem_bp Ravi Jonnalagadda
2026-10-05 9:24 ` [RFC PATCH v4 0/9] mm/damon: hardware-sampled access reports SJ Park
9 siblings, 0 replies; 11+ messages in thread
From: Ravi Jonnalagadda @ 2026-10-05 8:46 UTC (permalink / raw)
To: SJ Park, Andrew Morton
Cc: damon, linux-mm, linux-kernel, Gregory Price, David Rientjes,
Wei Xu, Jonathan Corbet, Bijan Tabatabai, Ajay Joshi,
Honggyu Kim, Yunjeong Mun, Akinobu Mita, Lian Wang, Kunwu Chan,
Ravi Jonnalagadda, Jonathan Cameron
kdamond_merge_regions() applies one context-wide threshold, the highest
merge score across all targets divided by 10, to every target. When a
high-traffic target and a low-traffic target share a context, the
high-traffic target sets a threshold permissive enough to erase the
hot/cold boundary in the low-traffic one: a region with nr_accesses=1
next to one with nr_accesses=0 merges because 1 <= threshold, even though
that difference is the only signal the cold scheme has to admit it.
On the regular merge pass, cap the threshold for each target by a tenth of
that target's own maximum merge score. The score is the one
damon_merge_regions_of() compares, so a probe-weighted context is capped
by its weighted hit sums.
The passes that follow the regular one exist only to bring the region
count under max_nr_regions by escalating the threshold. They keep using
the escalated threshold uncapped, so that bound still holds.
In a probe-weighted context the merge score is a weighted hit sum, which
can exceed max_thres, the samples-per-aggregation ceiling the escalating
passes stop at, when the probe weights sum above 1. Raise max_thres to
ten times the initial threshold there, so those passes still run.
Signed-off-by: Ravi Jonnalagadda <ravis.opensrc@gmail.com>
---
mm/damon/core.c | 30 +++++++++++++++++++++++++++++-
1 file changed, 29 insertions(+), 1 deletion(-)
diff --git a/mm/damon/core.c b/mm/damon/core.c
index e3c29d7cff23..7c0df07f0a90 100644
--- a/mm/damon/core.c
+++ b/mm/damon/core.c
@@ -4167,12 +4167,40 @@ static noinline_for_stack void kdamond_merge_regions(struct damon_ctx *c,
unsigned int nr_regions;
unsigned int max_thres;
bool count_age = true;
+ bool use_probe_hits = damon_has_probe_weights(c);
max_thres = damon_nr_samples_per_aggr(&c->attrs);
+ /* weighted scores can exceed max_thres; threshold is max score / 10 */
+ if (use_probe_hits)
+ max_thres = max(threshold * 10, max_thres);
while (true) {
nr_regions = 0;
damon_for_each_target(t, c) {
- damon_merge_regions_of(t, threshold, sz_limit, c,
+ struct damon_region *r;
+ unsigned int t_max = 0, t_thres = threshold;
+
+ /*
+ * On the regular pass, cap the threshold at a tenth of
+ * this target's maximum merge score. A high-traffic
+ * target in the same context must not set a threshold
+ * so permissive that a low-traffic target's hot/cold
+ * boundary merges away before the cold scheme can act
+ * on it. The score is the one damon_merge_regions_of()
+ * compares, so probe-weighted contexts are capped by
+ * their weighted hit sums.
+ *
+ * The passes that follow exist only to bring the region
+ * count under max_nr_regions, so they use the escalated
+ * threshold as is.
+ */
+ if (count_age) {
+ damon_for_each_region(r, t)
+ t_max = max(t_max,
+ damon_merge_score(r, false, c,
+ use_probe_hits));
+ t_thres = min(threshold, t_max / 10);
+ }
+ damon_merge_regions_of(t, t_thres, sz_limit, c,
count_age);
nr_regions += damon_nr_regions(t);
}
--
Git-157)
^ permalink raw reply [flat|nested] 11+ messages in thread* [RFC PATCH v4 9/9] mm/damon/core: apply probe_hits_wsum filters to node_eligible_mem_bp
2026-10-05 8:46 [RFC PATCH v4 0/9] mm/damon: hardware-sampled access reports Ravi Jonnalagadda
` (7 preceding siblings ...)
2026-10-05 8:46 ` [RFC PATCH v4 8/9] mm/damon/core: cap the region merge threshold per target Ravi Jonnalagadda
@ 2026-10-05 8:46 ` Ravi Jonnalagadda
2026-10-05 9:24 ` [RFC PATCH v4 0/9] mm/damon: hardware-sampled access reports SJ Park
9 siblings, 0 replies; 11+ messages in thread
From: Ravi Jonnalagadda @ 2026-10-05 8:46 UTC (permalink / raw)
To: SJ Park, Andrew Morton
Cc: damon, linux-mm, linux-kernel, Gregory Price, David Rientjes,
Wei Xu, Jonathan Corbet, Bijan Tabatabai, Ajay Joshi,
Honggyu Kim, Yunjeong Mun, Akinobu Mita, Lian Wang, Kunwu Chan,
Ravi Jonnalagadda, Jonathan Cameron
node_eligible_mem_bp is the share of a scheme's eligible memory that is
on a given node, and memory is eligible when its region matches the
scheme's access pattern. In data attributes-only monitoring nr_accesses
is not updated, so the access pattern cannot tell which regions the
scheme is for; its probe_hits_wsum filters do.
Count a region as eligible only if it also passes the scheme's
probe_hits_wsum filters, decided as the scheme's core filters are: the
first matching one decides, and if none matches, the region passes
unless the last of them is an allow filter and the scheme has no ops
filters. The address and target filters stay out of the metric, as
before, since they scope where the scheme acts rather than which memory
it is for.
Signed-off-by: Ravi Jonnalagadda <ravis.opensrc@gmail.com>
---
mm/damon/core.c | 34 +++++++++++++++++++++++++++++++---
1 file changed, 31 insertions(+), 3 deletions(-)
diff --git a/mm/damon/core.c b/mm/damon/core.c
index 7c0df07f0a90..bae4569f0ab3 100644
--- a/mm/damon/core.c
+++ b/mm/damon/core.c
@@ -3609,6 +3609,31 @@ static unsigned long damos_get_node_memcg_used_bp(
}
#ifdef CONFIG_DAMON_PADDR
+/*
+ * Whether @r passes the probe_hits_wsum filters of @s, decided as the core
+ * filters of a scheme are, with the other filter types left out: the first
+ * matching filter decides, and if none matches, @r passes unless the last of
+ * them is an allow filter and @s has no ops filters. These filters select
+ * regions by their data attributes, so they are part of what makes memory
+ * eligible for the scheme, unlike its address and target filters.
+ */
+static bool damos_probe_filters_pass(struct damon_ctx *c,
+ struct damon_target *t, struct damon_region *r, struct damos *s)
+{
+ struct damos_filter *filter;
+ bool pass = true;
+
+ damos_for_each_core_filter(filter, s) {
+ if (filter->type != DAMOS_FILTER_TYPE_PROBE_HITS_WSUM)
+ continue;
+ if (damos_filter_match(c, t, r, filter, c->min_region_sz))
+ return filter->allow;
+ pass = !filter->allow;
+ }
+ /* as damos_set_filters_default_reject(): ops filters decide the rest */
+ return pass || !list_empty(&s->ops_filters);
+}
+
/*
* damos_calc_eligible_bytes() - Calculate raw eligible bytes per node.
* @c: The DAMON context.
@@ -3616,9 +3641,10 @@ static unsigned long damos_get_node_memcg_used_bp(
* @nid: The target NUMA node id.
* @total: Output for total eligible bytes across all nodes.
*
- * Iterates through each folio in eligible regions to accurately determine
- * which node the memory resides on. Returns eligible bytes on the specified
- * node and sets *total to the sum across all nodes.
+ * A region is eligible if it matches the access pattern of @s and passes its
+ * probe_hits_wsum filters. Iterates through each folio in eligible regions to
+ * accurately determine which node the memory resides on. Returns eligible bytes
+ * on the specified node and sets *total to the sum across all nodes.
*
* Note: This function requires damon_get_folio() from ops-common.c, which is
* only available when CONFIG_DAMON_PADDR is enabled. It also requires the
@@ -3638,6 +3664,8 @@ static phys_addr_t damos_calc_eligible_bytes(struct damon_ctx *c,
if (!__damos_valid_target(r, s, c))
continue;
+ if (!damos_probe_filters_pass(c, t, r, s))
+ continue;
/* Convert from core address units to physical bytes */
addr = (phys_addr_t)r->ar.start * c->addr_unit;
--
Git-157)
^ permalink raw reply [flat|nested] 11+ messages in thread* Re: [RFC PATCH v4 0/9] mm/damon: hardware-sampled access reports
2026-10-05 8:46 [RFC PATCH v4 0/9] mm/damon: hardware-sampled access reports Ravi Jonnalagadda
` (8 preceding siblings ...)
2026-10-05 8:46 ` [RFC PATCH v4 9/9] mm/damon/core: apply probe_hits_wsum filters to node_eligible_mem_bp Ravi Jonnalagadda
@ 2026-10-05 9:24 ` SJ Park
9 siblings, 0 replies; 11+ messages in thread
From: SJ Park @ 2026-10-05 9:24 UTC (permalink / raw)
To: Ravi Jonnalagadda
Cc: SJ Park, Andrew Morton, damon, linux-mm, linux-kernel,
Gregory Price, David Rientjes, Wei Xu, Jonathan Corbet,
Bijan Tabatabai, Ajay Joshi, Honggyu Kim, Yunjeong Mun,
Akinobu Mita, Lian Wang, Kunwu Chan, Jonathan Cameron
Hi Ravi,
On Mon, 05 Oct 2026 01:46:44 -0700 Ravi Jonnalagadda <ravis.opensrc@gmail.com> wrote:
> This series lets DAMON monitor a hardware-sampled data attribute -- the
> addresses a PMU saw accessed -- through the data attributes monitoring
> (probe) interface, and run DAMOS schemes on it in data attributes-only
> mode. Patches 3, 5, and 6 are co-developed with Akinobu Mita, building
> on his earlier perf-event proposal [3].
Thank you for this series! I'm in travel, so it would take time to review this
in depth, though. Hopefully I will get get some bandwidth starting from next
week. Feel free to send new versions meanwhile, if you think those are needed.
I will review the latest version when I get the bandwidth.
Thanks,
SJ
[...]
^ permalink raw reply [flat|nested] 11+ messages in thread