mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Christian Brauner <brauner@kernel.org>
To: linux-fsdevel@vger.kernel.org
Cc: Alexander Viro <viro@zeniv.linux.org.uk>, Jan Kara <jack@suse.cz>,
	 linux-kernel@vger.kernel.org, Jeff Layton <jlayton@kernel.org>,
	 Jann Horn <jannh@google.com>, Neil Brown <neil@brown.name>,
	 Amir Goldstein <amir73il@gmail.com>,
	 "Christian Brauner (Amutable)" <brauner@kernel.org>
Subject: [PATCH 20/21] selftests/filesystems: add a helper that holds a readdir in a page fault
Date: Fri, 02 Oct 2026 15:52:51 +0200	[thread overview]
Message-ID: <20261002-work-mount-fixes-4-v1-20-dd44b89d44ce@kernel.org> (raw)
In-Reply-To: <20261002-work-mount-fixes-4-v1-0-dd44b89d44ce@kernel.org>

A getdents64() whose buffer is a page registered with userfaultfd sits
in handle_userfault() with whatever iterate_dir() took before it copied
the entries. Add readdir_hold.h for tests that want to know what that
blocks: it opens the userfaultfd before the test enters a user
namespace, the fault happens in the kernel and needs CAP_SYS_PTRACE in
the initial one, starts the readdir in a thread, waits until that
thread is stuck, queues a create behind it and waits for it to settle,
then probes a lookup with a watchdog and says whether it came back.
The page is released afterwards so that everything drains on a kernel
that still holds the lock across the copy.

The two users follow.

Signed-off-by: Christian Brauner (Amutable) <brauner@kernel.org>
---
 tools/testing/selftests/filesystems/readdir_hold.h | 224 +++++++++++++++++++++
 1 file changed, 224 insertions(+)

diff --git a/tools/testing/selftests/filesystems/readdir_hold.h b/tools/testing/selftests/filesystems/readdir_hold.h
new file mode 100644
index 000000000000..57eee1bef470
--- /dev/null
+++ b/tools/testing/selftests/filesystems/readdir_hold.h
@@ -0,0 +1,224 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * Hold a readdir of a directory in the page fault of its buffer and see
+ * whether a create and a lookup in that directory wait for it. For a
+ * directory that is permanently empty they must not.
+ */
+#ifndef __SELFTESTS_READDIR_HOLD_H
+#define __SELFTESTS_READDIR_HOLD_H
+
+#include <errno.h>
+#include <fcntl.h>
+#include <poll.h>
+#include <pthread.h>
+#include <stdbool.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <linux/userfaultfd.h>
+#include <sys/ioctl.h>
+#include <sys/mman.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+
+#define HOLD_FAULT_MS	5000	/* for the reader to reach the fault */
+#define HOLD_QUEUE_MS	2000	/* for the create to queue up behind it */
+#define HOLD_LOOKUP_MS	5000	/* for the lookup to come back */
+
+struct readdir_hold {
+	int uffd;
+	int taskdir;		/* /proc/self/task, from before any namespace change */
+	long page_size;
+	char *page;		/* faults until released */
+	int dfd;
+	pid_t creator_tid;
+	int created;
+	int done[2];		/* the finder writes a byte when it is back */
+};
+
+/*
+ * The fault happens in the kernel, so the userfaultfd needs CAP_SYS_PTRACE
+ * in the initial user namespace or vm.unprivileged_userfaultfd. Call this
+ * before entering a user namespace.
+ */
+static inline int readdir_hold_init(struct readdir_hold *h)
+{
+	struct uffdio_api api = { .api = UFFD_API };
+
+	h->page = MAP_FAILED;
+	h->page_size = sysconf(_SC_PAGESIZE);
+	h->taskdir = open("/proc/self/task", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+	h->uffd = syscall(__NR_userfaultfd, O_CLOEXEC | O_NONBLOCK);
+	if (h->uffd < 0 || h->taskdir < 0 || ioctl(h->uffd, UFFDIO_API, &api)) {
+		if (h->uffd >= 0)
+			close(h->uffd);
+		if (h->taskdir >= 0)
+			close(h->taskdir);
+		h->uffd = h->taskdir = -1;
+		return -1;
+	}
+	return 0;
+}
+
+static inline void readdir_hold_destroy(struct readdir_hold *h)
+{
+	if (h->uffd >= 0)
+		close(h->uffd);
+	if (h->taskdir >= 0)
+		close(h->taskdir);
+	h->uffd = h->taskdir = -1;
+}
+
+static inline void *readdir_hold_reader(void *arg)
+{
+	struct readdir_hold *h = arg;
+
+	/* the first byte written to the buffer faults until released */
+	syscall(__NR_getdents64, h->dfd, h->page, h->page_size);
+	return NULL;
+}
+
+static inline void *readdir_hold_creator(void *arg)
+{
+	struct readdir_hold *h = arg;
+
+	h->creator_tid = syscall(__NR_gettid);
+	/* takes the directory lock exclusive before it fails */
+	mkdirat(h->dfd, "x", 0755);
+	__atomic_store_n(&h->created, 1, __ATOMIC_RELEASE);
+	return NULL;
+}
+
+static inline void *readdir_hold_finder(void *arg)
+{
+	struct readdir_hold *h = arg;
+	int fd;
+
+	/* a lookup that misses the dcache takes the lock shared */
+	fd = openat(h->dfd, "no_such_name", O_RDONLY | O_CLOEXEC);
+	if (fd >= 0)
+		close(fd);
+	if (write(h->done[1], "x", 1) != 1)
+		perror("readdir_hold: finder");
+	return NULL;
+}
+
+/* the creator is back, or waits in the kernel for the lock */
+static inline bool readdir_hold_creator_settled(struct readdir_hold *h)
+{
+	char path[32], buf[256], *p;
+	ssize_t n;
+	int fd;
+
+	if (__atomic_load_n(&h->created, __ATOMIC_ACQUIRE))
+		return true;
+	if (!h->creator_tid)
+		return false;
+	snprintf(path, sizeof(path), "%d/stat", h->creator_tid);
+	fd = openat(h->taskdir, path, O_RDONLY | O_CLOEXEC);
+	if (fd < 0)
+		return false;
+	n = read(fd, buf, sizeof(buf) - 1);
+	close(fd);
+	if (n <= 0)
+		return false;
+	buf[n] = 0;
+	/* "pid (comm) state ..." */
+	p = strrchr(buf, ')');
+	return p && p[1] == ' ' && p[2] == 'D';
+}
+
+static inline bool readdir_hold_faulted(struct readdir_hold *h)
+{
+	struct pollfd pfd = { .fd = h->uffd, .events = POLLIN };
+	struct uffd_msg msg;
+
+	if (poll(&pfd, 1, HOLD_FAULT_MS) != 1)
+		return false;
+	if (read(h->uffd, &msg, sizeof(msg)) != sizeof(msg))
+		return false;
+	return msg.event == UFFD_EVENT_PAGEFAULT;
+}
+
+/* let the reader go on */
+static inline void readdir_hold_release(struct readdir_hold *h)
+{
+	struct uffdio_copy cp = {
+		.dst = (unsigned long)h->page,
+		.len = h->page_size,
+	};
+	void *zero;
+
+	zero = calloc(1, h->page_size);
+	if (!zero)
+		return;
+	cp.src = (unsigned long)zero;
+	if (ioctl(h->uffd, UFFDIO_COPY, &cp) && errno != EEXIST)
+		perror("readdir_hold: UFFDIO_COPY");
+	free(zero);
+}
+
+/*
+ * A readdir of @dfd that sticks in the fault of its buffer, then a create
+ * and a lookup in @dfd. Whether the lookup came back while the readdir was
+ * still stuck goes to @stalled. Returns -1 when that could not be found
+ * out.
+ */
+static inline int readdir_hold_check(struct readdir_hold *h, int dfd,
+				     bool *stalled)
+{
+	pthread_t reader, creator, finder;
+	struct uffdio_register reg = {};
+	struct pollfd pfd;
+	int ret = -1, i;
+
+	h->dfd = dfd;
+	h->created = 0;
+	h->creator_tid = 0;
+	h->page = mmap(NULL, h->page_size, PROT_READ | PROT_WRITE,
+		       MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+	if (h->page == MAP_FAILED)
+		return -1;
+	reg.range.start = (unsigned long)h->page;
+	reg.range.len = h->page_size;
+	reg.mode = UFFDIO_REGISTER_MODE_MISSING;
+	if (ioctl(h->uffd, UFFDIO_REGISTER, &reg) || pipe2(h->done, O_CLOEXEC))
+		goto out_page;
+
+	if (pthread_create(&reader, NULL, readdir_hold_reader, h))
+		goto out_pipe;
+	if (!readdir_hold_faulted(h))
+		goto out_reader;
+	if (pthread_create(&creator, NULL, readdir_hold_creator, h))
+		goto out_reader;
+	for (i = 0; i < HOLD_QUEUE_MS / 10 && !readdir_hold_creator_settled(h); i++)
+		usleep(10000);
+	if (pthread_create(&finder, NULL, readdir_hold_finder, h))
+		goto out_creator;
+
+	pfd.fd = h->done[0];
+	pfd.events = POLLIN;
+	*stalled = poll(&pfd, 1, HOLD_LOOKUP_MS) != 1;
+	ret = 0;
+
+	readdir_hold_release(h);
+	pthread_join(finder, NULL);
+out_creator:
+	if (ret)
+		readdir_hold_release(h);
+	pthread_join(creator, NULL);
+out_reader:
+	if (ret)
+		readdir_hold_release(h);
+	pthread_join(reader, NULL);
+out_pipe:
+	close(h->done[0]);
+	close(h->done[1]);
+out_page:
+	munmap(h->page, h->page_size);
+	h->page = MAP_FAILED;
+	return ret;
+}
+
+#endif /* __SELFTESTS_READDIR_HOLD_H */

-- 
2.53.0


  parent reply	other threads:[~2026-10-02 13:54 UTC|newest]

Thread overview: 24+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-02 13:52 [PATCH 00/21] mount: more bugfixes, trapped in the Black Lodge edition Christian Brauner
2026-10-02 13:52 ` [PATCH 01/21] namespace: unhash a dentry before detaching the mounts on it Christian Brauner
2026-10-02 13:52 ` [PATCH 02/21] namei: don't reveal overmounted entries in refwalk Christian Brauner
2026-10-02 13:52 ` [PATCH 03/21] fcntl: refuse F_SET_RW_HINT on an immutable inode Christian Brauner
2026-10-02 13:52 ` [PATCH 04/21] selftests/filesystems: check that an immutable inode takes no write hint Christian Brauner
2026-10-02 13:52 ` [PATCH 05/21] namespace: refuse an automount below a mount that is in no namespace Christian Brauner
2026-10-02 13:52 ` [PATCH 06/21] namespace: handle mount locking for automounts correctly Christian Brauner
2026-10-02 13:52 ` [PATCH 07/21] nullfs: don't update the access time Christian Brauner
2026-10-02 13:52 ` [PATCH 08/21] namespace: never expire a locked mount Christian Brauner
2026-10-02 13:52 ` [PATCH 09/21] namespace: keep the lock on a mount that a propagated copy is moved beneath Christian Brauner
2026-10-02 13:52 ` [PATCH 10/21] selftests/filesystems: check that a lock lands on the right mount and stays Christian Brauner
2026-10-02 13:52 ` [PATCH 11/21] selftests/filesystems: check the atime of the empty mount namespace root Christian Brauner
2026-10-02 13:52 ` [PATCH 12/21] selftests/filesystems: check that an automount below an overlay layer is refused Christian Brauner
2026-10-02 13:52 ` [PATCH 13/21] fhandle: decide the subtree check under mount_lock Christian Brauner
2026-10-02 13:52 ` [PATCH 14/21] namespace: keep the private nullfs instance in knullfs Christian Brauner
2026-10-02 13:52 ` [PATCH 15/21] namespace: nothing is mounted on or written through knullfs Christian Brauner
2026-10-02 13:52 ` [PATCH 16/21] fsnotify: let a filesystem refuse marks on its objects Christian Brauner
2026-10-02 14:26   ` Amir Goldstein
2026-10-02 13:52 ` [PATCH 17/21] nullfs: refuse file locks Christian Brauner
2026-10-02 13:52 ` [PATCH 18/21] nullfs: refuse leases and delegations Christian Brauner
2026-10-03  8:20   ` Jeff Layton
2026-10-02 13:52 ` [PATCH 19/21] readdir: take no inode lock on an immutable directory Christian Brauner
2026-10-02 13:52 ` Christian Brauner [this message]
2026-10-02 13:52 ` [PATCH 21/21] selftests/filesystems: check that reading the root of an empty mount namespace stalls nobody Christian Brauner

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261002-work-mount-fixes-4-v1-20-dd44b89d44ce@kernel.org \
    --to=brauner@kernel.org \
    --cc=amir73il@gmail.com \
    --cc=jack@suse.cz \
    --cc=jannh@google.com \
    --cc=jlayton@kernel.org \
    --cc=linux-fsdevel@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=neil@brown.name \
    --cc=viro@zeniv.linux.org.uk \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®