* [PATCH for-7.5 1/5] memcg: move memory.high enforcement out of try_charge_memcg()
2026-10-10 21:09 [PATCH for-7.5 0/5] memcg: clean up memory.high enforcement Shakeel Butt
@ 2026-10-10 21:09 ` Shakeel Butt
2026-10-10 21:09 ` [PATCH for-7.5 2/5] memcg: use high_work for kernel threads Shakeel Butt
` (3 subsequent siblings)
4 siblings, 0 replies; 6+ messages in thread
From: Shakeel Butt @ 2026-10-10 21:09 UTC (permalink / raw)
To: Andrew Morton
Cc: Johannes Weiner, Michal Hocko, Roman Gushchin, Muchun Song,
Tejun Heo, Meta kernel team, cgroups, linux-mm, linux-kernel
try_charge_memcg() does a lot. It charges the page counters, reclaims
when a limit is hit, may call the OOM killer, and at the end it checks
and enforces memory.high. Move the memory.high part into a new function,
memcg_enforce_high(), so that try_charge_memcg() is shorter and the
memory.high code is easier to read and change.
The code and its comments move as they are. The only difference is that
the new function gets allow_spinning from gfp_mask itself.
No functional change.
---
mm/memcontrol.c | 125 ++++++++++++++++++++++++++----------------------
1 file changed, 69 insertions(+), 56 deletions(-)
diff --git a/mm/memcontrol.c b/mm/memcontrol.c
index aad0498a7bd6..20c5dfe1a894 100644
--- a/mm/memcontrol.c
+++ b/mm/memcontrol.c
@@ -2696,6 +2696,74 @@ void __mem_cgroup_handle_over_high(gfp_t gfp_mask)
css_put(&memcg->css);
}
+/*
+ * Called after @nr_pages were charged to @memcg. If @memcg or one of its
+ * ancestors is over memory.high or memory.swap.high, start reclaim or slow
+ * down the task.
+ */
+static void memcg_enforce_high(struct mem_cgroup *memcg, unsigned int nr_pages,
+ gfp_t gfp_mask)
+{
+ bool allow_spinning = gfpflags_allow_spinning(gfp_mask);
+
+ /*
+ * If the hierarchy is above the normal consumption range, schedule
+ * reclaim on returning to userland. We can perform reclaim here
+ * if __GFP_RECLAIM but let's always punt for simplicity and so that
+ * GFP_KERNEL can consistently be used during reclaim. @memcg is
+ * not recorded as it most likely matches current's and won't
+ * change in the meantime. As high limit is checked again before
+ * reclaim, the cost of mismatch is negligible.
+ */
+ do {
+ bool mem_high, swap_high;
+
+ mem_high = page_counter_read(&memcg->memory) >
+ READ_ONCE(memcg->memory.high);
+ swap_high = page_counter_read(&memcg->swap) >
+ READ_ONCE(memcg->swap.high);
+
+ /* Don't bother a random interrupted task */
+ if (!in_task()) {
+ if (mem_high) {
+ if (allow_spinning)
+ schedule_work(&memcg->high_work);
+ else
+ irq_work_queue(&memcg->high_irq_work);
+ break;
+ }
+ continue;
+ }
+
+ if (mem_high || swap_high) {
+ /*
+ * The allocating tasks in this cgroup will need to do
+ * reclaim or be throttled to prevent further growth
+ * of the memory or swap footprints.
+ *
+ * Target some best-effort fairness between the tasks,
+ * and distribute reclaim work and delay penalties
+ * based on how much each task is actually allocating.
+ */
+ current->memcg_nr_pages_over_high += nr_pages;
+ set_notify_resume(current);
+ break;
+ }
+ } while ((memcg = parent_mem_cgroup(memcg)));
+
+ /*
+ * Reclaim is set up above to be called from the userland
+ * return path. But also attempt synchronous reclaim to avoid
+ * excessive overrun while the task is still inside the
+ * kernel. If this is successful, the return path will see it
+ * when it rechecks the overage and simply bail out.
+ */
+ if (current->memcg_nr_pages_over_high > MEMCG_CHARGE_BATCH &&
+ !(current->flags & PF_MEMALLOC) &&
+ gfpflags_allow_blocking(gfp_mask))
+ __mem_cgroup_handle_over_high(gfp_mask);
+}
+
static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask,
unsigned int nr_pages)
{
@@ -2853,62 +2921,7 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask,
if (batch > nr_pages)
refill_stock(memcg, batch - nr_pages);
- /*
- * If the hierarchy is above the normal consumption range, schedule
- * reclaim on returning to userland. We can perform reclaim here
- * if __GFP_RECLAIM but let's always punt for simplicity and so that
- * GFP_KERNEL can consistently be used during reclaim. @memcg is
- * not recorded as it most likely matches current's and won't
- * change in the meantime. As high limit is checked again before
- * reclaim, the cost of mismatch is negligible.
- */
- do {
- bool mem_high, swap_high;
-
- mem_high = page_counter_read(&memcg->memory) >
- READ_ONCE(memcg->memory.high);
- swap_high = page_counter_read(&memcg->swap) >
- READ_ONCE(memcg->swap.high);
-
- /* Don't bother a random interrupted task */
- if (!in_task()) {
- if (mem_high) {
- if (allow_spinning)
- schedule_work(&memcg->high_work);
- else
- irq_work_queue(&memcg->high_irq_work);
- break;
- }
- continue;
- }
-
- if (mem_high || swap_high) {
- /*
- * The allocating tasks in this cgroup will need to do
- * reclaim or be throttled to prevent further growth
- * of the memory or swap footprints.
- *
- * Target some best-effort fairness between the tasks,
- * and distribute reclaim work and delay penalties
- * based on how much each task is actually allocating.
- */
- current->memcg_nr_pages_over_high += batch;
- set_notify_resume(current);
- break;
- }
- } while ((memcg = parent_mem_cgroup(memcg)));
-
- /*
- * Reclaim is set up above to be called from the userland
- * return path. But also attempt synchronous reclaim to avoid
- * excessive overrun while the task is still inside the
- * kernel. If this is successful, the return path will see it
- * when it rechecks the overage and simply bail out.
- */
- if (current->memcg_nr_pages_over_high > MEMCG_CHARGE_BATCH &&
- !(current->flags & PF_MEMALLOC) &&
- gfpflags_allow_blocking(gfp_mask))
- __mem_cgroup_handle_over_high(gfp_mask);
+ memcg_enforce_high(memcg, batch, gfp_mask);
return ret;
}
--
2.53.0-Meta
^ permalink raw reply [flat|nested] 6+ messages in thread* [PATCH for-7.5 2/5] memcg: use high_work for kernel threads
2026-10-10 21:09 [PATCH for-7.5 0/5] memcg: clean up memory.high enforcement Shakeel Butt
2026-10-10 21:09 ` [PATCH for-7.5 1/5] memcg: move memory.high enforcement out of try_charge_memcg() Shakeel Butt
@ 2026-10-10 21:09 ` Shakeel Butt
2026-10-10 21:09 ` [PATCH for-7.5 3/5] memcg: use high_work for remote charges Shakeel Butt
` (2 subsequent siblings)
4 siblings, 0 replies; 6+ messages in thread
From: Shakeel Butt @ 2026-10-10 21:09 UTC (permalink / raw)
To: Andrew Morton
Cc: Johannes Weiner, Michal Hocko, Roman Gushchin, Muchun Song,
Tejun Heo, Meta kernel team, cgroups, linux-mm, linux-kernel
Kernel threads never return to userspace, so deferred memory.high
reclaim does not run. Reclaim and throttling during a charge can also
delay work they do for other tasks.
Treat kernel threads like interrupts. Queue high_work for the first
cgroup over memory.high and skip reclaim and throttling in the charging
thread.
User workers created by create_io_thread() and vhost_task_create() keep
their current behavior. Vhost kernel threads use the new path, even when
they borrow the owner's mm.
The worker requests a fixed batch of pages each time. A fast allocator
can keep usage well above memory.high. Kernel threads also skip
memory.swap.high throttling and its high-event reporting. The memory.max
and memory.swap.max checks stay the same.
Suggested-by: Tejun Heo <tj@kernel.org>
Signed-off-by: Shakeel Butt <shakeel.butt@linux.dev>
---
mm/memcontrol.c | 13 +++++++++++--
1 file changed, 11 insertions(+), 2 deletions(-)
diff --git a/mm/memcontrol.c b/mm/memcontrol.c
index 20c5dfe1a894..8116fce6ab96 100644
--- a/mm/memcontrol.c
+++ b/mm/memcontrol.c
@@ -2705,6 +2705,7 @@ static void memcg_enforce_high(struct mem_cgroup *memcg, unsigned int nr_pages,
gfp_t gfp_mask)
{
bool allow_spinning = gfpflags_allow_spinning(gfp_mask);
+ bool use_worker = !in_task() || (current->flags & PF_KTHREAD);
/*
* If the hierarchy is above the normal consumption range, schedule
@@ -2723,8 +2724,12 @@ static void memcg_enforce_high(struct mem_cgroup *memcg, unsigned int nr_pages,
swap_high = page_counter_read(&memcg->swap) >
READ_ONCE(memcg->swap.high);
- /* Don't bother a random interrupted task */
- if (!in_task()) {
+ /*
+ * Don't make an unrelated interrupted task handle this charge.
+ * Kernel threads may be working for other tasks, so let a worker
+ * reclaim instead.
+ */
+ if (use_worker) {
if (mem_high) {
if (allow_spinning)
schedule_work(&memcg->high_work);
@@ -2751,6 +2756,10 @@ static void memcg_enforce_high(struct mem_cgroup *memcg, unsigned int nr_pages,
}
} while ((memcg = parent_mem_cgroup(memcg)));
+ /* Interrupts and kernel threads stop here. */
+ if (use_worker)
+ return;
+
/*
* Reclaim is set up above to be called from the userland
* return path. But also attempt synchronous reclaim to avoid
--
2.53.0-Meta
^ permalink raw reply [flat|nested] 6+ messages in thread* [PATCH for-7.5 3/5] memcg: use high_work for remote charges
2026-10-10 21:09 [PATCH for-7.5 0/5] memcg: clean up memory.high enforcement Shakeel Butt
2026-10-10 21:09 ` [PATCH for-7.5 1/5] memcg: move memory.high enforcement out of try_charge_memcg() Shakeel Butt
2026-10-10 21:09 ` [PATCH for-7.5 2/5] memcg: use high_work for kernel threads Shakeel Butt
@ 2026-10-10 21:09 ` Shakeel Butt
2026-10-10 21:09 ` [PATCH for-7.5 4/5] memcg: avoid queuing high_work from reclaim Shakeel Butt
2026-10-10 21:09 ` [PATCH for-7.5 5/5] memcg: simplify high limit handling after a charge Shakeel Butt
4 siblings, 0 replies; 6+ messages in thread
From: Shakeel Butt @ 2026-10-10 21:09 UTC (permalink / raw)
To: Andrew Morton
Cc: Johannes Weiner, Michal Hocko, Roman Gushchin, Muchun Song,
Tejun Heo, Meta kernel team, cgroups, linux-mm, linux-kernel
Some charges go to a cgroup outside the current mm's cgroup and its
ancestors. The high handler uses current->mm, so it cannot handle these
overages. Filesystem notifications and remote page faults can take
this path.
Use mm_match_cgroup() before checking the high limits. If it does not
match, use high_work instead of recording task overage or doing
synchronous reclaim and throttling. Charges to the same cgroup or an
ancestor keep their current handling. Tasks without an mm also use the
worker.
Like kernel threads, remote chargers skip memory.swap.high throttling
and its high-event reporting, and a fast allocator can outrun the worker.
The memory.max and memory.swap.max checks stay the same.
Suggested-by: Tejun Heo <tj@kernel.org>
Signed-off-by: Shakeel Butt <shakeel.butt@linux.dev>
---
mm/memcontrol.c | 6 ++++--
1 file changed, 4 insertions(+), 2 deletions(-)
diff --git a/mm/memcontrol.c b/mm/memcontrol.c
index 8116fce6ab96..3456960cbe68 100644
--- a/mm/memcontrol.c
+++ b/mm/memcontrol.c
@@ -2705,7 +2705,8 @@ static void memcg_enforce_high(struct mem_cgroup *memcg, unsigned int nr_pages,
gfp_t gfp_mask)
{
bool allow_spinning = gfpflags_allow_spinning(gfp_mask);
- bool use_worker = !in_task() || (current->flags & PF_KTHREAD);
+ bool use_worker = !in_task() || (current->flags & PF_KTHREAD) ||
+ !current->mm || !mm_match_cgroup(current->mm, memcg);
/*
* If the hierarchy is above the normal consumption range, schedule
@@ -2728,6 +2729,7 @@ static void memcg_enforce_high(struct mem_cgroup *memcg, unsigned int nr_pages,
* Don't make an unrelated interrupted task handle this charge.
* Kernel threads may be working for other tasks, so let a worker
* reclaim instead.
+ * Remote charges need reclaim from their charged hierarchy.
*/
if (use_worker) {
if (mem_high) {
@@ -2756,7 +2758,7 @@ static void memcg_enforce_high(struct mem_cgroup *memcg, unsigned int nr_pages,
}
} while ((memcg = parent_mem_cgroup(memcg)));
- /* Interrupts and kernel threads stop here. */
+ /* Interrupts, kernel threads, and remote chargers stop here. */
if (use_worker)
return;
--
2.53.0-Meta
^ permalink raw reply [flat|nested] 6+ messages in thread* [PATCH for-7.5 4/5] memcg: avoid queuing high_work from reclaim
2026-10-10 21:09 [PATCH for-7.5 0/5] memcg: clean up memory.high enforcement Shakeel Butt
` (2 preceding siblings ...)
2026-10-10 21:09 ` [PATCH for-7.5 3/5] memcg: use high_work for remote charges Shakeel Butt
@ 2026-10-10 21:09 ` Shakeel Butt
2026-10-10 21:09 ` [PATCH for-7.5 5/5] memcg: simplify high limit handling after a charge Shakeel Butt
4 siblings, 0 replies; 6+ messages in thread
From: Shakeel Butt @ 2026-10-10 21:09 UTC (permalink / raw)
To: Andrew Morton
Cc: Johannes Weiner, Michal Hocko, Roman Gushchin, Muchun Song,
Tejun Heo, Meta kernel team, cgroups, linux-mm, linux-kernel
Kernel threads and remote chargers use high_work when memory.high is
exceeded. Their allocations during reclaim can also charge memory,
for example when zswap stores a page.
Without a check, the high_work worker can queue itself from its own
reclaim. The pending bit is cleared before high_work runs, so each run
can queue another run.
Skip memory.high enforcement for charges from task context with
PF_MEMALLOC set when they would use high_work. This covers kernel
threads and remote charges. The memory is still charged. Interrupt
charges keep their current path.
In a VM with incompressible zswap data, the worker queued itself 163 times
without this check.
Signed-off-by: Shakeel Butt <shakeel.butt@linux.dev>
---
mm/memcontrol.c | 8 ++++++++
1 file changed, 8 insertions(+)
diff --git a/mm/memcontrol.c b/mm/memcontrol.c
index 3456960cbe68..d1e022e6572f 100644
--- a/mm/memcontrol.c
+++ b/mm/memcontrol.c
@@ -2708,6 +2708,14 @@ static void memcg_enforce_high(struct mem_cgroup *memcg, unsigned int nr_pages,
bool use_worker = !in_task() || (current->flags & PF_KTHREAD) ||
!current->mm || !mm_match_cgroup(current->mm, memcg);
+ /*
+ * Reclaim can charge memory, for example when zswap stores a page.
+ * Don't queue high_work from a reclaiming task, or the
+ * high_work worker could keep queuing itself.
+ */
+ if (in_task() && use_worker && (current->flags & PF_MEMALLOC))
+ return;
+
/*
* If the hierarchy is above the normal consumption range, schedule
* reclaim on returning to userland. We can perform reclaim here
--
2.53.0-Meta
^ permalink raw reply [flat|nested] 6+ messages in thread* [PATCH for-7.5 5/5] memcg: simplify high limit handling after a charge
2026-10-10 21:09 [PATCH for-7.5 0/5] memcg: clean up memory.high enforcement Shakeel Butt
` (3 preceding siblings ...)
2026-10-10 21:09 ` [PATCH for-7.5 4/5] memcg: avoid queuing high_work from reclaim Shakeel Butt
@ 2026-10-10 21:09 ` Shakeel Butt
4 siblings, 0 replies; 6+ messages in thread
From: Shakeel Butt @ 2026-10-10 21:09 UTC (permalink / raw)
To: Andrew Morton
Cc: Johannes Weiner, Michal Hocko, Roman Gushchin, Muchun Song,
Tejun Heo, Meta kernel team, cgroups, linux-mm, linux-kernel
Separate the hierarchy scan from high-limit enforcement. Find the first
cgroup over the applicable limit, then queue high_work or record the
task's overage after the scan. Worker contexts check only memory.high;
local charges also check memory.swap.high.
Read in_task() once and clarify the enforcement comments. Keep the
synchronous pending-overage check independent of the scan result so a
blocking charge still handles overage left by earlier charges.
No functional change intended.
Signed-off-by: Shakeel Butt <shakeel.butt@linux.dev>
---
mm/memcontrol.c | 100 +++++++++++++++++++++---------------------------
1 file changed, 43 insertions(+), 57 deletions(-)
diff --git a/mm/memcontrol.c b/mm/memcontrol.c
index d1e022e6572f..2a3742036b45 100644
--- a/mm/memcontrol.c
+++ b/mm/memcontrol.c
@@ -2602,9 +2602,8 @@ static unsigned long calculate_high_delay(unsigned int nr_pages,
}
/*
- * Reclaims memory over the high limit. Called directly from
- * try_charge() (context permitting), as well as from the userland
- * return path where reclaim is always able to block.
+ * Handle pending high overage during a blocking charge or on return
+ * to userspace.
*/
void __mem_cgroup_handle_over_high(gfp_t gfp_mask)
{
@@ -2704,78 +2703,65 @@ void __mem_cgroup_handle_over_high(gfp_t gfp_mask)
static void memcg_enforce_high(struct mem_cgroup *memcg, unsigned int nr_pages,
gfp_t gfp_mask)
{
+ struct mem_cgroup *iter;
bool allow_spinning = gfpflags_allow_spinning(gfp_mask);
- bool use_worker = !in_task() || (current->flags & PF_KTHREAD) ||
- !current->mm || !mm_match_cgroup(current->mm, memcg);
+ bool task_context = in_task();
+ bool kthread = task_context && (current->flags & PF_KTHREAD);
+ bool use_high_work = !task_context || kthread || !current->mm ||
+ !mm_match_cgroup(current->mm, memcg);
/*
* Reclaim can charge memory, for example when zswap stores a page.
* Don't queue high_work from a reclaiming task, or the
* high_work worker could keep queuing itself.
*/
- if (in_task() && use_worker && (current->flags & PF_MEMALLOC))
+ if (task_context && use_high_work && (current->flags & PF_MEMALLOC))
return;
/*
- * If the hierarchy is above the normal consumption range, schedule
- * reclaim on returning to userland. We can perform reclaim here
- * if __GFP_RECLAIM but let's always punt for simplicity and so that
- * GFP_KERNEL can consistently be used during reclaim. @memcg is
- * not recorded as it most likely matches current's and won't
- * change in the meantime. As high limit is checked again before
- * reclaim, the cost of mismatch is negligible.
+ * Check the charged cgroup and its parents. Local charges normally defer
+ * reclaim until return to userspace so it can use GFP_KERNEL. Large
+ * pending overages may be handled synchronously below. The handler looks
+ * up current->mm and rechecks the limits.
*/
- do {
+ for (iter = memcg; iter; iter = parent_mem_cgroup(iter)) {
bool mem_high, swap_high;
- mem_high = page_counter_read(&memcg->memory) >
- READ_ONCE(memcg->memory.high);
- swap_high = page_counter_read(&memcg->swap) >
- READ_ONCE(memcg->swap.high);
-
- /*
- * Don't make an unrelated interrupted task handle this charge.
- * Kernel threads may be working for other tasks, so let a worker
- * reclaim instead.
- * Remote charges need reclaim from their charged hierarchy.
- */
- if (use_worker) {
- if (mem_high) {
- if (allow_spinning)
- schedule_work(&memcg->high_work);
- else
- irq_work_queue(&memcg->high_irq_work);
- break;
- }
- continue;
- }
-
- if (mem_high || swap_high) {
- /*
- * The allocating tasks in this cgroup will need to do
- * reclaim or be throttled to prevent further growth
- * of the memory or swap footprints.
- *
- * Target some best-effort fairness between the tasks,
- * and distribute reclaim work and delay penalties
- * based on how much each task is actually allocating.
- */
- current->memcg_nr_pages_over_high += nr_pages;
- set_notify_resume(current);
+ mem_high = page_counter_read(&iter->memory) >
+ READ_ONCE(iter->memory.high);
+ swap_high = !use_high_work &&
+ (page_counter_read(&iter->swap) > READ_ONCE(iter->swap.high));
+ if (mem_high || swap_high)
break;
- }
- } while ((memcg = parent_mem_cgroup(memcg)));
+ }
- /* Interrupts, kernel threads, and remote chargers stop here. */
- if (use_worker)
+ /*
+ * Use a worker for interrupts, kernel threads, and tasks whose mm
+ * does not match the charged hierarchy.
+ */
+ if (use_high_work) {
+ if (!iter)
+ return;
+ if (allow_spinning)
+ schedule_work(&iter->high_work);
+ else
+ irq_work_queue(&iter->high_irq_work);
return;
+ }
+
+ if (iter) {
+ /*
+ * Count the full batch so reclaim work and delay penalties scale
+ * with the pages charged by this task.
+ */
+ current->memcg_nr_pages_over_high += nr_pages;
+ set_notify_resume(current);
+ }
/*
- * Reclaim is set up above to be called from the userland
- * return path. But also attempt synchronous reclaim to avoid
- * excessive overrun while the task is still inside the
- * kernel. If this is successful, the return path will see it
- * when it rechecks the overage and simply bail out.
+ * Earlier charges may have left high overage pending even if this charge
+ * found no new overage. Handle it now if this charge can block, to limit
+ * overrun while the task remains in the kernel.
*/
if (current->memcg_nr_pages_over_high > MEMCG_CHARGE_BATCH &&
!(current->flags & PF_MEMALLOC) &&
--
2.53.0-Meta
^ permalink raw reply [flat|nested] 6+ messages in thread