mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: "Jose A. Perez de Azpillaga" <azpijr@gmail.com>
To: Andrew Morton <akpm@linux-foundation.org>,
	David Hildenbrand <david@kernel.org>,
	Shuah Khan <shuah@kernel.org>
Cc: "Jose A. Perez de Azpillaga" <azpijr@gmail.com>,
	Lorenzo Stoakes <ljs@kernel.org>,
	"Liam R. Howlett" <liam@infradead.org>,
	Vlastimil Babka <vbabka@kernel.org>,
	Mike Rapoport <rppt@kernel.org>,
	Suren Baghdasaryan <surenb@google.com>,
	Michal Hocko <mhocko@suse.com>,
	linux-mm@kvack.org, linux-kselftest@vger.kernel.org,
	linux-kernel@vger.kernel.org
Subject: [PATCH] selftests/mm: check MREMAP_DONTUNMAP mlock accounting
Date: Mon, 28 Sep 2026 13:10:42 +0200	[thread overview]
Message-ID: <20260928111055.482136-1-azpijr@gmail.com> (raw)

MREMAP_DONTUNMAP keeps the source VMA in place, but clears its mlock flags
for the whole VMA while setting them on the destination VMA.  Two cases
leak mm->locked_vm as a result:

 - an unfaulted mlock-on-fault VMA moved behind itself self-merges, so the
   single resulting VMA loses the flags without the accounting being
   dropped;

 - a partial mremap() moves only part of the range, leaving the pages which
   are not moved accounted as locked in a VMA whose flags were cleared.

Add three cases to the MREMAP_DONTUNMAP selftest which mlock() the source
VMA, with and without MLOCK_ONFAULT, perform the operation and check that
VmLck comes back to zero once everything is unmapped.  Each case runs in
its own process, so it starts from a clean mm with VmLck at zero and a
failure cannot propagate to the cases which follow.

Verified on x86_64: the three cases fail on v7.3-rc5 and pass on
mm-unstable with the fixes from the "mm/mremap: fix two issues with
MREMAP_DONTUNMAP" series applied.

Signed-off-by: Jose A. Perez de Azpillaga <azpijr@gmail.com>
---
 tools/testing/selftests/mm/mremap_dontunmap.c | 273 +++++++++++++++++-
 1 file changed, 272 insertions(+), 1 deletion(-)

diff --git a/tools/testing/selftests/mm/mremap_dontunmap.c b/tools/testing/selftests/mm/mremap_dontunmap.c
index 96ba537facf7..2ec58b6cdc9f 100644
--- a/tools/testing/selftests/mm/mremap_dontunmap.c
+++ b/tools/testing/selftests/mm/mremap_dontunmap.c
@@ -7,6 +7,8 @@
  */
 #define _GNU_SOURCE
 #include <sys/mman.h>
+#include <sys/syscall.h>
+#include <sys/wait.h>
 #include <linux/mman.h>
 #include <errno.h>
 #include <stdio.h>
@@ -37,6 +39,61 @@ static void dump_maps(void)
 		}								\
 	} while (0)

+/*
+ * Same as mlock2.h's, plus an ENOSYS fallback for libc headers without
+ * __NR_mlock2.  It is not taken from the header because that also defines
+ * seek_to_smaps_entry(), which nothing here uses and which then warns
+ * (-Wunused-function).
+ */
+static int mlock2_(void *start, size_t len, int flags)
+{
+#ifdef __NR_mlock2
+	return syscall(__NR_mlock2, start, len, flags);
+#else
+	errno = ENOSYS;
+	return -1;
+#endif
+}
+
+/*
+ * Locked memory size in kB, as reported by /proc/self/status, which is
+ * mm->locked_vm accounted in kB.  Used to check that the mlock() accounting
+ * balances across a MREMAP_DONTUNMAP operation.
+ *
+ * Returns LOCKED_VM_UNKNOWN if it cannot be read: the callers run in a child
+ * whose exit status is the test result, so this must not exit the process or
+ * print anything the TAP output parser would act on.
+ */
+#define LOCKED_VM_UNKNOWN	((unsigned long)-1)
+
+static unsigned long get_proc_locked_vm_size(void)
+{
+	unsigned long lock_size;
+	char *line = NULL;
+	size_t size = 0;
+	FILE *f;
+
+	f = fopen("/proc/self/status", "r");
+	if (!f) {
+		fprintf(stderr, "cannot open /proc/self/status: %s\n",
+			strerror(errno));
+		return LOCKED_VM_UNKNOWN;
+	}
+
+	while (getline(&line, &size, f) != -1) {
+		if (sscanf(line, "VmLck:\t%8lu kB", &lock_size) == 1) {
+			free(line);
+			fclose(f);
+			return lock_size;
+		}
+	}
+
+	free(line);
+	fclose(f);
+	fprintf(stderr, "cannot parse VmLck in /proc/self/status\n");
+	return LOCKED_VM_UNKNOWN;
+}
+
 // Try a simple operation for to "test" for kernel support this prevents
 // reporting tests as failed when it's run on an older kernel.
 static int kernel_support_for_mremap_dontunmap()
@@ -335,6 +392,216 @@ static void mremap_dontunmap_partial_mapping_overwrite(void)
 	ksft_test_result_pass("%s\n", __func__);
 }

+/*
+ * Child exit codes for the accounting cases: any other exit code, and any
+ * signal, is reported as an error rather than mistaken for a leak.
+ */
+#define CASE_LEAK		2
+#define CASE_SKIP		77
+#define CASE_SKIP_ENOSYS	78
+#define CASE_SETUP_ERROR	79
+
+/* Report a setup failure from a case: the child's exit status carries it back. */
+static int case_failed(const char *where, const char *what)
+{
+	fprintf(stderr, "%s: %s: %s\n", where, what, strerror(errno));
+	return CASE_SETUP_ERROR;
+}
+
+/* Report a check which failed for a reason errno does not describe. */
+static int case_unexpected(const char *where, const char *what)
+{
+	fprintf(stderr, "%s: unexpected %s\n", where, what);
+	return CASE_SETUP_ERROR;
+}
+
+/*
+ * Only EPERM/ENOMEM are expected with a small RLIMIT_MEMLOCK, and ENOSYS means
+ * the kernel has no mlock2(); anything else is a genuine setup failure.
+ */
+static int lock_failed(const char *where, const char *call)
+{
+	if (errno == EPERM || errno == ENOMEM)
+		return CASE_SKIP;
+	if (errno == ENOSYS)
+		return CASE_SKIP_ENOSYS;
+
+	return case_failed(where, call);
+}
+
+/*
+ * Run one accounting case in a child, so that it starts with a clean mm and an
+ * empty VmLck, and report its outcome.
+ */
+static void run_locked_case(const char *label, int (*fn)(void))
+{
+	int status;
+	pid_t pid;
+
+	/* do not let the child flush a copy of our TAP output */
+	fflush(NULL);
+
+	pid = fork();
+	if (pid < 0) {
+		ksft_test_result_error("%s: fork: %s\n", label,
+				       strerror(errno));
+		return;
+	}
+	if (!pid)
+		_exit(fn());
+
+	if (waitpid(pid, &status, 0) == -1) {
+		ksft_test_result_error("%s: waitpid: %s\n", label,
+				       strerror(errno));
+		return;
+	}
+
+	if (WIFSIGNALED(status)) {
+		ksft_test_result_error("%s: killed by signal %d\n", label,
+				       WTERMSIG(status));
+		return;
+	}
+
+	if (!WIFEXITED(status)) {
+		ksft_test_result_error("%s: child did not exit\n", label);
+		return;
+	}
+
+	switch (WEXITSTATUS(status)) {
+	case 0:
+		ksft_test_result_pass("%s: locked memory released\n", label);
+		break;
+	case CASE_LEAK:
+		ksft_test_result_fail("%s: locked memory leaked\n", label);
+		break;
+	case CASE_SKIP:
+		ksft_test_result_skip("%s: mlock not permitted\n", label);
+		break;
+	case CASE_SKIP_ENOSYS:
+		ksft_test_result_skip("%s: mlock2 not supported\n", label);
+		break;
+	default:
+		ksft_test_result_error("%s: child exited with %d (see stderr)\n",
+				       label, WEXITSTATUS(status));
+		break;
+	}
+}
+
+/*
+ * An unfaulted mlock-on-fault VMA moved behind itself: the source and
+ * destination VMAs are adjacent and mergeable, and merging them clears the
+ * mlock flags of the single resulting VMA, leaking mm->locked_vm.
+ */
+static int case_mlock_onfault_self_merge(void)
+{
+	unsigned long locked;
+	void *source, *dest, *reserve;
+
+	/*
+	 * Two adjacent pages: the source VMA goes in the first, the
+	 * destination in the second, so the two are mergeable.
+	 */
+	reserve = mmap(NULL, 2 * page_size, PROT_NONE,
+		       MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+	if (reserve == MAP_FAILED)
+		return case_failed(__func__, "mmap reserve");
+	if (munmap(reserve, 2 * page_size) == -1)
+		return case_failed(__func__, "munmap reserve");
+
+	source = mmap(reserve, page_size, PROT_READ | PROT_WRITE,
+		      MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED, -1, 0);
+	if (source != reserve)
+		return case_unexpected(__func__, "source address");
+
+	/* Locked on fault, but deliberately left unfaulted. */
+	if (mlock2_(source, page_size, MLOCK_ONFAULT))
+		return lock_failed(__func__, "mlock2");
+
+	dest = mremap(source, page_size, page_size,
+		      MREMAP_DONTUNMAP | MREMAP_MAYMOVE | MREMAP_FIXED,
+		      source + page_size);
+	if (dest == MAP_FAILED)
+		return case_failed(__func__, "mremap");
+
+	if (munmap(dest, page_size) == -1)
+		return case_failed(__func__, "munmap destination");
+	if (munmap(source, page_size) == -1)
+		return case_failed(__func__, "munmap source");
+
+	locked = get_proc_locked_vm_size();
+	if (locked == LOCKED_VM_UNKNOWN)
+		return CASE_SETUP_ERROR;
+
+	return locked ? CASE_LEAK : 0;
+}
+
+/*
+ * A partial MREMAP_DONTUNMAP of a locked VMA: all but the last page is moved,
+ * leaving the source VMA mapped.  Both the moved pages and the VMA left
+ * behind must give up their mlock accounting.
+ *
+ * The destination goes into a window with a guard page on either side, so that
+ * it is not adjacent to and cannot merge with the source VMA: this case must
+ * exercise the partial-copy accounting on its own.  The window's hole is
+ * smaller than the source mapping, so the source cannot land in it.
+ */
+static int case_locked_partial(int onfault)
+{
+	unsigned long num_pages = 3;
+	unsigned long span = (num_pages - 1) * page_size;
+	unsigned long locked;
+	void *source, *guard, *dest, *moved;
+
+	guard = mmap(NULL, span + 2 * page_size, PROT_NONE,
+		     MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+	if (guard == MAP_FAILED)
+		return case_failed(__func__, "mmap guard");
+	dest = guard + page_size;
+	if (munmap(dest, span) == -1)
+		return case_failed(__func__, "munmap destination window");
+
+	source = mmap(NULL, num_pages * page_size, PROT_READ | PROT_WRITE,
+		      MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+	if (source == MAP_FAILED)
+		return case_failed(__func__, "mmap");
+
+	if (onfault) {
+		if (mlock2_(source, num_pages * page_size, MLOCK_ONFAULT))
+			return lock_failed(__func__, "mlock2");
+	} else if (mlock(source, num_pages * page_size)) {
+		return lock_failed(__func__, "mlock");
+	}
+
+	/* Move all but the last page, leaving the source partially mapped. */
+	moved = mremap(source, span, span,
+		       MREMAP_DONTUNMAP | MREMAP_MAYMOVE | MREMAP_FIXED, dest);
+	if (moved == MAP_FAILED)
+		return case_failed(__func__, "mremap");
+	if (moved != dest)
+		return case_unexpected(__func__, "destination address");
+
+	if (munmap(dest, span) == -1)
+		return case_failed(__func__, "munmap destination");
+	if (munmap(source, num_pages * page_size) == -1)
+		return case_failed(__func__, "munmap source");
+
+	locked = get_proc_locked_vm_size();
+	if (locked == LOCKED_VM_UNKNOWN)
+		return CASE_SETUP_ERROR;
+
+	return locked ? CASE_LEAK : 0;
+}
+
+static int case_locked_partial_mlock(void)
+{
+	return case_locked_partial(0);
+}
+
+static int case_locked_partial_onfault(void)
+{
+	return case_locked_partial(1);
+}
+
 int main(void)
 {
 	ksft_print_header();
@@ -348,7 +615,7 @@ int main(void)
 		ksft_finished();
 	}

-	ksft_set_plan(5);
+	ksft_set_plan(8);

 	// Keep a page sized buffer around for when we need it.
 	page_buffer =
@@ -361,6 +628,10 @@ int main(void)
 	mremap_dontunmap_simple_fixed();
 	mremap_dontunmap_partial_mapping();
 	mremap_dontunmap_partial_mapping_overwrite();
+	run_locked_case("mlock-onfault self-merge",
+			case_mlock_onfault_self_merge);
+	run_locked_case("mlock partial", case_locked_partial_mlock);
+	run_locked_case("mlock2 onfault partial", case_locked_partial_onfault);

 	BUG_ON(munmap(page_buffer, page_size) == -1,
 	       "unable to unmap page buffer");
--
2.55.0


             reply	other threads:[~2026-09-28 11:11 UTC|newest]

Thread overview: 3+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-28 11:10 Jose A. Perez de Azpillaga [this message]
2026-09-28 12:08 ` David Hildenbrand (Arm)
2026-09-28 12:17   ` Lorenzo Stoakes (ARM)

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260928111055.482136-1-azpijr@gmail.com \
    --to=azpijr@gmail.com \
    --cc=akpm@linux-foundation.org \
    --cc=david@kernel.org \
    --cc=liam@infradead.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-kselftest@vger.kernel.org \
    --cc=linux-mm@kvack.org \
    --cc=ljs@kernel.org \
    --cc=mhocko@suse.com \
    --cc=rppt@kernel.org \
    --cc=shuah@kernel.org \
    --cc=surenb@google.com \
    --cc=vbabka@kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®