From: Christian Brauner <brauner@kernel.org>
To: linux-fsdevel@vger.kernel.org
Cc: Alexander Viro <viro@zeniv.linux.org.uk>, Jan Kara <jack@suse.cz>,
linux-kernel@vger.kernel.org, Jeff Layton <jlayton@kernel.org>,
Jann Horn <jannh@google.com>, Neil Brown <neil@brown.name>,
Amir Goldstein <amir73il@gmail.com>,
"Christian Brauner (Amutable)" <brauner@kernel.org>
Subject: [PATCH 20/21] selftests/filesystems: add a helper that holds a readdir in a page fault
Date: Fri, 02 Oct 2026 15:52:51 +0200 [thread overview]
Message-ID: <20261002-work-mount-fixes-4-v1-20-dd44b89d44ce@kernel.org> (raw)
In-Reply-To: <20261002-work-mount-fixes-4-v1-0-dd44b89d44ce@kernel.org>
A getdents64() whose buffer is a page registered with userfaultfd sits
in handle_userfault() with whatever iterate_dir() took before it copied
the entries. Add readdir_hold.h for tests that want to know what that
blocks: it opens the userfaultfd before the test enters a user
namespace, the fault happens in the kernel and needs CAP_SYS_PTRACE in
the initial one, starts the readdir in a thread, waits until that
thread is stuck, queues a create behind it and waits for it to settle,
then probes a lookup with a watchdog and says whether it came back.
The page is released afterwards so that everything drains on a kernel
that still holds the lock across the copy.
The two users follow.
Signed-off-by: Christian Brauner (Amutable) <brauner@kernel.org>
---
tools/testing/selftests/filesystems/readdir_hold.h | 224 +++++++++++++++++++++
1 file changed, 224 insertions(+)
diff --git a/tools/testing/selftests/filesystems/readdir_hold.h b/tools/testing/selftests/filesystems/readdir_hold.h
new file mode 100644
index 000000000000..57eee1bef470
--- /dev/null
+++ b/tools/testing/selftests/filesystems/readdir_hold.h
@@ -0,0 +1,224 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * Hold a readdir of a directory in the page fault of its buffer and see
+ * whether a create and a lookup in that directory wait for it. For a
+ * directory that is permanently empty they must not.
+ */
+#ifndef __SELFTESTS_READDIR_HOLD_H
+#define __SELFTESTS_READDIR_HOLD_H
+
+#include <errno.h>
+#include <fcntl.h>
+#include <poll.h>
+#include <pthread.h>
+#include <stdbool.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <linux/userfaultfd.h>
+#include <sys/ioctl.h>
+#include <sys/mman.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+
+#define HOLD_FAULT_MS 5000 /* for the reader to reach the fault */
+#define HOLD_QUEUE_MS 2000 /* for the create to queue up behind it */
+#define HOLD_LOOKUP_MS 5000 /* for the lookup to come back */
+
+struct readdir_hold {
+ int uffd;
+ int taskdir; /* /proc/self/task, from before any namespace change */
+ long page_size;
+ char *page; /* faults until released */
+ int dfd;
+ pid_t creator_tid;
+ int created;
+ int done[2]; /* the finder writes a byte when it is back */
+};
+
+/*
+ * The fault happens in the kernel, so the userfaultfd needs CAP_SYS_PTRACE
+ * in the initial user namespace or vm.unprivileged_userfaultfd. Call this
+ * before entering a user namespace.
+ */
+static inline int readdir_hold_init(struct readdir_hold *h)
+{
+ struct uffdio_api api = { .api = UFFD_API };
+
+ h->page = MAP_FAILED;
+ h->page_size = sysconf(_SC_PAGESIZE);
+ h->taskdir = open("/proc/self/task", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ h->uffd = syscall(__NR_userfaultfd, O_CLOEXEC | O_NONBLOCK);
+ if (h->uffd < 0 || h->taskdir < 0 || ioctl(h->uffd, UFFDIO_API, &api)) {
+ if (h->uffd >= 0)
+ close(h->uffd);
+ if (h->taskdir >= 0)
+ close(h->taskdir);
+ h->uffd = h->taskdir = -1;
+ return -1;
+ }
+ return 0;
+}
+
+static inline void readdir_hold_destroy(struct readdir_hold *h)
+{
+ if (h->uffd >= 0)
+ close(h->uffd);
+ if (h->taskdir >= 0)
+ close(h->taskdir);
+ h->uffd = h->taskdir = -1;
+}
+
+static inline void *readdir_hold_reader(void *arg)
+{
+ struct readdir_hold *h = arg;
+
+ /* the first byte written to the buffer faults until released */
+ syscall(__NR_getdents64, h->dfd, h->page, h->page_size);
+ return NULL;
+}
+
+static inline void *readdir_hold_creator(void *arg)
+{
+ struct readdir_hold *h = arg;
+
+ h->creator_tid = syscall(__NR_gettid);
+ /* takes the directory lock exclusive before it fails */
+ mkdirat(h->dfd, "x", 0755);
+ __atomic_store_n(&h->created, 1, __ATOMIC_RELEASE);
+ return NULL;
+}
+
+static inline void *readdir_hold_finder(void *arg)
+{
+ struct readdir_hold *h = arg;
+ int fd;
+
+ /* a lookup that misses the dcache takes the lock shared */
+ fd = openat(h->dfd, "no_such_name", O_RDONLY | O_CLOEXEC);
+ if (fd >= 0)
+ close(fd);
+ if (write(h->done[1], "x", 1) != 1)
+ perror("readdir_hold: finder");
+ return NULL;
+}
+
+/* the creator is back, or waits in the kernel for the lock */
+static inline bool readdir_hold_creator_settled(struct readdir_hold *h)
+{
+ char path[32], buf[256], *p;
+ ssize_t n;
+ int fd;
+
+ if (__atomic_load_n(&h->created, __ATOMIC_ACQUIRE))
+ return true;
+ if (!h->creator_tid)
+ return false;
+ snprintf(path, sizeof(path), "%d/stat", h->creator_tid);
+ fd = openat(h->taskdir, path, O_RDONLY | O_CLOEXEC);
+ if (fd < 0)
+ return false;
+ n = read(fd, buf, sizeof(buf) - 1);
+ close(fd);
+ if (n <= 0)
+ return false;
+ buf[n] = 0;
+ /* "pid (comm) state ..." */
+ p = strrchr(buf, ')');
+ return p && p[1] == ' ' && p[2] == 'D';
+}
+
+static inline bool readdir_hold_faulted(struct readdir_hold *h)
+{
+ struct pollfd pfd = { .fd = h->uffd, .events = POLLIN };
+ struct uffd_msg msg;
+
+ if (poll(&pfd, 1, HOLD_FAULT_MS) != 1)
+ return false;
+ if (read(h->uffd, &msg, sizeof(msg)) != sizeof(msg))
+ return false;
+ return msg.event == UFFD_EVENT_PAGEFAULT;
+}
+
+/* let the reader go on */
+static inline void readdir_hold_release(struct readdir_hold *h)
+{
+ struct uffdio_copy cp = {
+ .dst = (unsigned long)h->page,
+ .len = h->page_size,
+ };
+ void *zero;
+
+ zero = calloc(1, h->page_size);
+ if (!zero)
+ return;
+ cp.src = (unsigned long)zero;
+ if (ioctl(h->uffd, UFFDIO_COPY, &cp) && errno != EEXIST)
+ perror("readdir_hold: UFFDIO_COPY");
+ free(zero);
+}
+
+/*
+ * A readdir of @dfd that sticks in the fault of its buffer, then a create
+ * and a lookup in @dfd. Whether the lookup came back while the readdir was
+ * still stuck goes to @stalled. Returns -1 when that could not be found
+ * out.
+ */
+static inline int readdir_hold_check(struct readdir_hold *h, int dfd,
+ bool *stalled)
+{
+ pthread_t reader, creator, finder;
+ struct uffdio_register reg = {};
+ struct pollfd pfd;
+ int ret = -1, i;
+
+ h->dfd = dfd;
+ h->created = 0;
+ h->creator_tid = 0;
+ h->page = mmap(NULL, h->page_size, PROT_READ | PROT_WRITE,
+ MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
+ if (h->page == MAP_FAILED)
+ return -1;
+ reg.range.start = (unsigned long)h->page;
+ reg.range.len = h->page_size;
+ reg.mode = UFFDIO_REGISTER_MODE_MISSING;
+ if (ioctl(h->uffd, UFFDIO_REGISTER, ®) || pipe2(h->done, O_CLOEXEC))
+ goto out_page;
+
+ if (pthread_create(&reader, NULL, readdir_hold_reader, h))
+ goto out_pipe;
+ if (!readdir_hold_faulted(h))
+ goto out_reader;
+ if (pthread_create(&creator, NULL, readdir_hold_creator, h))
+ goto out_reader;
+ for (i = 0; i < HOLD_QUEUE_MS / 10 && !readdir_hold_creator_settled(h); i++)
+ usleep(10000);
+ if (pthread_create(&finder, NULL, readdir_hold_finder, h))
+ goto out_creator;
+
+ pfd.fd = h->done[0];
+ pfd.events = POLLIN;
+ *stalled = poll(&pfd, 1, HOLD_LOOKUP_MS) != 1;
+ ret = 0;
+
+ readdir_hold_release(h);
+ pthread_join(finder, NULL);
+out_creator:
+ if (ret)
+ readdir_hold_release(h);
+ pthread_join(creator, NULL);
+out_reader:
+ if (ret)
+ readdir_hold_release(h);
+ pthread_join(reader, NULL);
+out_pipe:
+ close(h->done[0]);
+ close(h->done[1]);
+out_page:
+ munmap(h->page, h->page_size);
+ h->page = MAP_FAILED;
+ return ret;
+}
+
+#endif /* __SELFTESTS_READDIR_HOLD_H */
--
2.53.0
next prev parent reply other threads:[~2026-10-02 13:54 UTC|newest]
Thread overview: 24+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-02 13:52 [PATCH 00/21] mount: more bugfixes, trapped in the Black Lodge edition Christian Brauner
2026-10-02 13:52 ` [PATCH 01/21] namespace: unhash a dentry before detaching the mounts on it Christian Brauner
2026-10-02 13:52 ` [PATCH 02/21] namei: don't reveal overmounted entries in refwalk Christian Brauner
2026-10-02 13:52 ` [PATCH 03/21] fcntl: refuse F_SET_RW_HINT on an immutable inode Christian Brauner
2026-10-02 13:52 ` [PATCH 04/21] selftests/filesystems: check that an immutable inode takes no write hint Christian Brauner
2026-10-02 13:52 ` [PATCH 05/21] namespace: refuse an automount below a mount that is in no namespace Christian Brauner
2026-10-02 13:52 ` [PATCH 06/21] namespace: handle mount locking for automounts correctly Christian Brauner
2026-10-02 13:52 ` [PATCH 07/21] nullfs: don't update the access time Christian Brauner
2026-10-02 13:52 ` [PATCH 08/21] namespace: never expire a locked mount Christian Brauner
2026-10-02 13:52 ` [PATCH 09/21] namespace: keep the lock on a mount that a propagated copy is moved beneath Christian Brauner
2026-10-02 13:52 ` [PATCH 10/21] selftests/filesystems: check that a lock lands on the right mount and stays Christian Brauner
2026-10-02 13:52 ` [PATCH 11/21] selftests/filesystems: check the atime of the empty mount namespace root Christian Brauner
2026-10-02 13:52 ` [PATCH 12/21] selftests/filesystems: check that an automount below an overlay layer is refused Christian Brauner
2026-10-02 13:52 ` [PATCH 13/21] fhandle: decide the subtree check under mount_lock Christian Brauner
2026-10-02 13:52 ` [PATCH 14/21] namespace: keep the private nullfs instance in knullfs Christian Brauner
2026-10-02 13:52 ` [PATCH 15/21] namespace: nothing is mounted on or written through knullfs Christian Brauner
2026-10-02 13:52 ` [PATCH 16/21] fsnotify: let a filesystem refuse marks on its objects Christian Brauner
2026-10-02 14:26 ` Amir Goldstein
2026-10-02 13:52 ` [PATCH 17/21] nullfs: refuse file locks Christian Brauner
2026-10-02 13:52 ` [PATCH 18/21] nullfs: refuse leases and delegations Christian Brauner
2026-10-03 8:20 ` Jeff Layton
2026-10-02 13:52 ` [PATCH 19/21] readdir: take no inode lock on an immutable directory Christian Brauner
2026-10-02 13:52 ` Christian Brauner [this message]
2026-10-02 13:52 ` [PATCH 21/21] selftests/filesystems: check that reading the root of an empty mount namespace stalls nobody Christian Brauner
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261002-work-mount-fixes-4-v1-20-dd44b89d44ce@kernel.org \
--to=brauner@kernel.org \
--cc=amir73il@gmail.com \
--cc=jack@suse.cz \
--cc=jannh@google.com \
--cc=jlayton@kernel.org \
--cc=linux-fsdevel@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=neil@brown.name \
--cc=viro@zeniv.linux.org.uk \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®