From: Christian Brauner <brauner@kernel.org>
To: linux-fsdevel@vger.kernel.org
Cc: Alexander Viro <viro@zeniv.linux.org.uk>, Jan Kara <jack@suse.cz>,
linux-kernel@vger.kernel.org, Jeff Layton <jlayton@kernel.org>,
Jann Horn <jannh@google.com>, Neil Brown <neil@brown.name>,
Amir Goldstein <amir73il@gmail.com>,
"Christian Brauner (Amutable)" <brauner@kernel.org>
Subject: [PATCH 10/21] selftests/filesystems: check that a lock lands on the right mount and stays
Date: Fri, 02 Oct 2026 15:52:41 +0200 [thread overview]
Message-ID: <20261002-work-mount-fixes-4-v1-10-dd44b89d44ce@kernel.org> (raw)
In-Reply-To: <20261002-work-mount-fixes-4-v1-0-dd44b89d44ce@kernel.org>
Check MNT_LOCKED:
- A child in a user namespace of its own triggers the tracefs automount
through a directory descriptor on the host's debugfs mount. The copy
in the child's namespace has to be locked and the host's mount unlocked.
- A child moves a bind of that automount beneath a locked covering mount,
unmounts the covering mount and asks for the umount of an unlocked
ancestor. This must not expire the mount which holds the lock now.
- The host mounts and unmounts on a directory that a locked mount covers
in a user namespace further down the propagation chain. That cover has
to be locked afterwards as it was before.
Signed-off-by: Christian Brauner (Amutable) <brauner@kernel.org>
---
.../filesystems/umount_propagation/Makefile | 2 +-
.../umount_propagation/locked_mount_test.c | 432 +++++++++++++++++++++
2 files changed, 433 insertions(+), 1 deletion(-)
diff --git a/tools/testing/selftests/filesystems/umount_propagation/Makefile b/tools/testing/selftests/filesystems/umount_propagation/Makefile
index eb85612abf8d..31f783a92b23 100644
--- a/tools/testing/selftests/filesystems/umount_propagation/Makefile
+++ b/tools/testing/selftests/filesystems/umount_propagation/Makefile
@@ -1,5 +1,5 @@
# SPDX-License-Identifier: GPL-2.0
-TEST_GEN_PROGS := umount_propagation_test shrink_submounts_test
+TEST_GEN_PROGS := umount_propagation_test shrink_submounts_test locked_mount_test
CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES)
diff --git a/tools/testing/selftests/filesystems/umount_propagation/locked_mount_test.c b/tools/testing/selftests/filesystems/umount_propagation/locked_mount_test.c
new file mode 100644
index 000000000000..0799b2ba545d
--- /dev/null
+++ b/tools/testing/selftests/filesystems/umount_propagation/locked_mount_test.c
@@ -0,0 +1,432 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * MNT_LOCKED keeps the owner of a user namespace from revealing what a
+ * mount covers. The lock has to be set on the copy in that namespace and
+ * only there, it has to survive the expiry of a mount placed beneath the
+ * locked one, and it has to survive the propagated umount of such a copy.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/mount.h>
+#include <sys/prctl.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+#include <sys/wait.h>
+
+#include "../../kselftest_harness.h"
+
+#ifndef MOVE_MOUNT_F_EMPTY_PATH
+#define MOVE_MOUNT_F_EMPTY_PATH 0x00000004
+#endif
+#ifndef MOVE_MOUNT_BENEATH
+#define MOVE_MOUNT_BENEATH 0x00000200
+#endif
+
+#define DIR_LEN 64
+#define PATH_LEN 192
+#define FLAGS (MS_NOSUID | MS_NODEV | MS_NOEXEC)
+
+/* exit codes of the children, each test says what they mean */
+enum {
+ CHILD_OK,
+ CHILD_SETUP,
+ CHILD_STEP1,
+ CHILD_STEP2,
+ CHILD_STEP3,
+ CHILD_STEP4,
+};
+
+static int write_file(const char *path, const char *s)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CLOEXEC);
+ if (fd >= 0) {
+ n = write(fd, s, strlen(s));
+ close(fd);
+ }
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+static int create_file(const char *path, const char *s)
+{
+ ssize_t n = -1;
+ int fd;
+
+ fd = open(path, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC, 0644);
+ if (fd >= 0) {
+ n = write(fd, s, strlen(s));
+ close(fd);
+ }
+ return n == (ssize_t)strlen(s) ? 0 : -1;
+}
+
+/* Root in a new user namespace with a copy of the mount namespace. */
+static int enter_userns(void)
+{
+ uid_t uid = getuid();
+ gid_t gid = getgid();
+ char map[32];
+
+ prctl(PR_SET_DUMPABLE, 1);
+ if (unshare(CLONE_NEWUSER | CLONE_NEWNS))
+ return -1;
+ if (write_file("/proc/self/setgroups", "deny") && errno != ENOENT)
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", uid);
+ if (write_file("/proc/self/uid_map", map))
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", gid);
+ if (write_file("/proc/self/gid_map", map))
+ return -1;
+ return setgid(0) || setuid(0) ? -1 : 0;
+}
+
+static int wait_child(pid_t pid)
+{
+ int status;
+
+ if (waitpid(pid, &status, 0) != pid || !WIFEXITED(status))
+ return -1;
+ return WEXITSTATUS(status);
+}
+
+static int wait_byte(int fd)
+{
+ char c;
+
+ return read(fd, &c, 1) == 1 ? 0 : -1;
+}
+
+static int send_byte(int fd)
+{
+ return write(fd, "x", 1) == 1 ? 0 : -1;
+}
+
+/* A read of @path fails: the file is covered. */
+static bool covered(const char *path)
+{
+ int fd = open(path, O_RDONLY | O_CLOEXEC);
+
+ if (fd >= 0)
+ close(fd);
+ return fd < 0;
+}
+
+FIXTURE(locked_mount) {
+ char base[DIR_LEN];
+ bool tracing; /* debugfs with the tracefs automount is there */
+};
+
+FIXTURE_SETUP(locked_mount)
+{
+ char p[PATH_LEN];
+ struct stat st;
+
+ if (geteuid())
+ SKIP(return, "test requires root");
+
+ snprintf(self->base, sizeof(self->base), "/tmp/locked_mount.XXXXXX");
+ ASSERT_NE(mkdtemp(self->base), NULL);
+ ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+ ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0);
+ ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, "mode=0755"), 0);
+
+ /* an automount that a user can name: the tracefs below debugfs */
+ snprintf(p, sizeof(p), "%s/dbg", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ self->tracing = !mount("debugfs", p, "debugfs", FLAGS, NULL);
+ if (self->tracing) {
+ /* the cases trigger it themselves */
+ snprintf(p, sizeof(p), "%s/dbg/tracing", self->base);
+ self->tracing = !fstatat(AT_FDCWD, p, &st, AT_NO_AUTOMOUNT) &&
+ S_ISDIR(st.st_mode);
+ }
+}
+
+FIXTURE_TEARDOWN(locked_mount)
+{
+ umount2(self->base, MNT_DETACH);
+ rmdir(self->base);
+}
+
+/*
+ * The child keeps a directory descriptor on the host's debugfs mount, moves
+ * to a user namespace of its own and triggers the automount through the
+ * descriptor. The mount goes below the host's mount and propagates into the
+ * child's copy. The child's copy has to be locked, the host's not.
+ */
+static int automount_child(const char *base, int dfd, int to_host, int from_host)
+{
+ char p[PATH_LEN];
+ struct stat st;
+
+ if (enter_userns())
+ return CHILD_SETUP;
+ if (fstatat(dfd, "tracing/.", &st, 0))
+ return CHILD_STEP1;
+ if (send_byte(to_host) || wait_byte(from_host))
+ return CHILD_SETUP;
+ /* the flags of the copy are locked: EPERM */
+ snprintf(p, sizeof(p), "%s/dbg/tracing", base);
+ if (!mount(NULL, p, NULL, MS_REMOUNT | MS_BIND, NULL) || errno != EPERM)
+ return CHILD_STEP2;
+ return CHILD_OK;
+}
+
+TEST_F(locked_mount, automount_locked_in_the_triggering_namespace)
+{
+ int to_host[2], from_host[2], dfd, ret;
+ char p[PATH_LEN];
+ pid_t pid;
+
+ if (!self->tracing)
+ SKIP(return, "test requires debugfs with the tracefs automount");
+
+ snprintf(p, sizeof(p), "%s/dbg", self->base);
+ ASSERT_EQ(mount(NULL, p, NULL, MS_SHARED, NULL), 0);
+ dfd = open(p, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ ASSERT_GE(dfd, 0);
+ ASSERT_EQ(pipe(to_host), 0);
+ ASSERT_EQ(pipe(from_host), 0);
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0)
+ _exit(automount_child(self->base, dfd, to_host[1], from_host[0]));
+ close(dfd);
+ ASSERT_EQ(wait_byte(to_host[0]), 0);
+
+ /* the host's own mount isn't locked: the flags can go */
+ snprintf(p, sizeof(p), "%s/dbg/tracing", self->base);
+ EXPECT_EQ(mount(NULL, p, NULL, MS_REMOUNT | MS_BIND, NULL), 0);
+
+ ASSERT_EQ(send_byte(from_host[1]), 0);
+ ret = wait_child(pid);
+ TH_LOG("child exit code %d", ret);
+ EXPECT_EQ(ret, CHILD_OK);
+}
+
+/*
+ * A copy of a tree with a locked cover. The child puts a shrinkable mount
+ * beneath the cover, which hands the lock down, unmounts the cover and then
+ * asks for the umount of an unlocked ancestor. That expires the shrinkable
+ * mount on a kernel that doesn't look at the lock, and the covered
+ * directory is bare.
+ */
+static int expiry_child(const char *base)
+{
+ char srv[PATH_LEN], shr[PATH_LEN], x[PATH_LEN], c[PATH_LEN];
+ char hidden[PATH_LEN], secret[PATH_LEN];
+
+ snprintf(srv, sizeof(srv), "%s/srv", base);
+ snprintf(shr, sizeof(shr), "%s/shr", base);
+ snprintf(x, sizeof(x), "%s/x", base);
+ snprintf(c, sizeof(c), "%s/c", base);
+ snprintf(hidden, sizeof(hidden), "%s/x/hidden", base);
+ snprintf(secret, sizeof(secret), "%s/x/hidden/secret", base);
+
+ if (enter_userns())
+ return CHILD_SETUP;
+ if (mount(srv, x, NULL, MS_BIND | MS_REC, NULL) ||
+ mount(shr, c, NULL, MS_BIND, NULL))
+ return CHILD_SETUP;
+ /* the cover is locked */
+ if (!umount2(hidden, 0) || errno != EINVAL)
+ return CHILD_STEP1;
+ if (syscall(__NR_move_mount, AT_FDCWD, c, AT_FDCWD, hidden, MOVE_MOUNT_BENEATH))
+ return CHILD_STEP2;
+ /* the lock moved down, the cover may go */
+ if (umount2(hidden, 0))
+ return CHILD_STEP3;
+ /* the holder of the lock may not, in any way */
+ if (!umount2(hidden, 0) || errno != EINVAL)
+ return CHILD_STEP4;
+ if (chdir(x))
+ return CHILD_SETUP;
+ umount2(x, 0);
+ return covered(secret) ? CHILD_OK : CHILD_STEP4;
+}
+
+TEST_F(locked_mount, expiry_leaves_a_locked_mount_alone)
+{
+ char p[PATH_LEN], q[PATH_LEN];
+ pid_t pid;
+ int ret;
+
+ if (!self->tracing)
+ SKIP(return, "test requires debugfs with the tracefs automount");
+
+ snprintf(p, sizeof(p), "%s/dbg/tracing", self->base);
+ snprintf(q, sizeof(q), "%s/shr", self->base);
+ ASSERT_EQ(mkdir(q, 0755), 0);
+ /* a bind of an automount is shrinkable as well */
+ ASSERT_EQ(mount(p, q, NULL, MS_BIND, NULL), 0);
+ snprintf(p, sizeof(p), "%s/srv", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, "mode=0755"), 0);
+ snprintf(p, sizeof(p), "%s/srv/hidden", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ snprintf(p, sizeof(p), "%s/srv/hidden/secret", self->base);
+ ASSERT_EQ(create_file(p, "covered-by-root\n"), 0);
+ snprintf(p, sizeof(p), "%s/srv/hidden", self->base);
+ ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, "mode=0755"), 0);
+ snprintf(p, sizeof(p), "%s/x", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ snprintf(p, sizeof(p), "%s/c", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0)
+ _exit(expiry_child(self->base));
+ ret = wait_child(pid);
+ TH_LOG("child exit code %d", ret);
+ EXPECT_EQ(ret, CHILD_OK);
+}
+
+/*
+ * The copy of the mount tree that a user namespace gets at its creation has
+ * the automounts in it locked unless they are on an expiry list. A bind of
+ * such a tree inside the namespace copies the lock to the automount but not
+ * to the root of the bind, so the child may ask for the umount of that root.
+ * That must not expire the locked automount below it: the root is busy, the
+ * umount fails and the automount has to be there afterwards.
+ */
+static int copied_tree_child(const char *base)
+{
+ char dbg[PATH_LEN], x[PATH_LEN], tracing[PATH_LEN];
+ struct stat root, st;
+
+ snprintf(dbg, sizeof(dbg), "%s/dbg", base);
+ snprintf(x, sizeof(x), "%s/x", base);
+ snprintf(tracing, sizeof(tracing), "%s/x/tracing", base);
+
+ if (enter_userns())
+ return CHILD_SETUP;
+ if (mount(dbg, x, NULL, MS_BIND | MS_REC, NULL))
+ return CHILD_SETUP;
+ /* the copy of the automount carries the lock */
+ if (!umount2(tracing, 0) || errno != EINVAL)
+ return CHILD_STEP1;
+ /* the root of the bind doesn't; keep it busy */
+ if (chdir(x))
+ return CHILD_SETUP;
+ if (!umount2(x, 0) || errno != EBUSY)
+ return CHILD_STEP2;
+ /* the automount below it must not have gone */
+ if (stat(x, &root) || fstatat(AT_FDCWD, tracing, &st, AT_NO_AUTOMOUNT))
+ return CHILD_SETUP;
+ return st.st_dev != root.st_dev ? CHILD_OK : CHILD_STEP3;
+}
+
+TEST_F(locked_mount, umount_of_a_bind_leaves_a_locked_automount_alone)
+{
+ char p[PATH_LEN];
+ struct stat st;
+ pid_t pid;
+ int ret;
+
+ if (!self->tracing)
+ SKIP(return, "test requires debugfs with the tracefs automount");
+
+ /* the copy the child gets has to contain the automount */
+ snprintf(p, sizeof(p), "%s/dbg/tracing/.", self->base);
+ ASSERT_EQ(stat(p, &st), 0);
+ snprintf(p, sizeof(p), "%s/x", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0)
+ _exit(copied_tree_child(self->base));
+ ret = wait_child(pid);
+ TH_LOG("child exit code %d", ret);
+ EXPECT_EQ(ret, CHILD_OK);
+}
+
+/*
+ * Z is a user namespace with a copy of a tree in which P/d is covered by a
+ * locked mount. The host mounts X on P/d, which propagates beneath Z's
+ * cover, and unmounts it again. Z's cover has to be locked afterwards as
+ * it was before.
+ */
+static int propagation_child(const char *base, int to_host, int from_host)
+{
+ char d[PATH_LEN], secret[PATH_LEN];
+
+ snprintf(d, sizeof(d), "%s/P/d", base);
+ snprintf(secret, sizeof(secret), "%s/P/d/secret", base);
+
+ if (enter_userns())
+ return CHILD_SETUP;
+ /* the cover is locked */
+ if (!umount2(d, 0) || errno != EINVAL)
+ return CHILD_STEP1;
+ if (send_byte(to_host) || wait_byte(from_host))
+ return CHILD_SETUP;
+ /* X came and went beneath it: still locked */
+ if (!umount2(d, 0) || errno != EINVAL)
+ return CHILD_STEP2;
+ return covered(secret) ? CHILD_OK : CHILD_STEP3;
+}
+
+TEST_F(locked_mount, propagated_copy_keeps_the_cover_locked)
+{
+ int to_host[2], from_host[2], ret;
+ char p[PATH_LEN];
+ pid_t pid;
+
+ snprintf(p, sizeof(p), "%s/P", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, "mode=0755"), 0);
+ ASSERT_EQ(mount(NULL, p, NULL, MS_SHARED, NULL), 0);
+ snprintf(p, sizeof(p), "%s/P/d", self->base);
+ ASSERT_EQ(mkdir(p, 0755), 0);
+ snprintf(p, sizeof(p), "%s/P/d/secret", self->base);
+ ASSERT_EQ(create_file(p, "covered-by-root\n"), 0);
+ ASSERT_EQ(pipe(to_host), 0);
+ ASSERT_EQ(pipe(from_host), 0);
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ /* M: a manager's namespace that covers P/d for Z */
+ pid_t z;
+
+ if (unshare(CLONE_NEWNS))
+ _exit(CHILD_SETUP);
+ snprintf(p, sizeof(p), "%s/P", self->base);
+ if (mount(NULL, p, NULL, MS_SLAVE, NULL))
+ _exit(CHILD_SETUP);
+ snprintf(p, sizeof(p), "%s/P/d", self->base);
+ if (mount("tmpfs", p, "tmpfs", 0, "mode=0755"))
+ _exit(CHILD_SETUP);
+ z = fork();
+ if (z < 0)
+ _exit(CHILD_SETUP);
+ if (z == 0)
+ _exit(propagation_child(self->base, to_host[1], from_host[0]));
+ _exit(wait_child(z));
+ }
+ ASSERT_EQ(wait_byte(to_host[0]), 0);
+
+ /* the host mounts on P/d and unmounts again; both propagate */
+ snprintf(p, sizeof(p), "%s/P/d", self->base);
+ ASSERT_EQ(mount("tmpfs", p, "tmpfs", 0, "mode=0755"), 0);
+ ASSERT_EQ(umount2(p, 0), 0);
+
+ ASSERT_EQ(send_byte(from_host[1]), 0);
+ ret = wait_child(pid);
+ TH_LOG("child exit code %d", ret);
+ EXPECT_EQ(ret, CHILD_OK);
+}
+
+TEST_HARNESS_MAIN
--
2.53.0
next prev parent reply other threads:[~2026-10-02 13:54 UTC|newest]
Thread overview: 24+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-02 13:52 [PATCH 00/21] mount: more bugfixes, trapped in the Black Lodge edition Christian Brauner
2026-10-02 13:52 ` [PATCH 01/21] namespace: unhash a dentry before detaching the mounts on it Christian Brauner
2026-10-02 13:52 ` [PATCH 02/21] namei: don't reveal overmounted entries in refwalk Christian Brauner
2026-10-02 13:52 ` [PATCH 03/21] fcntl: refuse F_SET_RW_HINT on an immutable inode Christian Brauner
2026-10-02 13:52 ` [PATCH 04/21] selftests/filesystems: check that an immutable inode takes no write hint Christian Brauner
2026-10-02 13:52 ` [PATCH 05/21] namespace: refuse an automount below a mount that is in no namespace Christian Brauner
2026-10-02 13:52 ` [PATCH 06/21] namespace: handle mount locking for automounts correctly Christian Brauner
2026-10-02 13:52 ` [PATCH 07/21] nullfs: don't update the access time Christian Brauner
2026-10-02 13:52 ` [PATCH 08/21] namespace: never expire a locked mount Christian Brauner
2026-10-02 13:52 ` [PATCH 09/21] namespace: keep the lock on a mount that a propagated copy is moved beneath Christian Brauner
2026-10-02 13:52 ` Christian Brauner [this message]
2026-10-02 13:52 ` [PATCH 11/21] selftests/filesystems: check the atime of the empty mount namespace root Christian Brauner
2026-10-02 13:52 ` [PATCH 12/21] selftests/filesystems: check that an automount below an overlay layer is refused Christian Brauner
2026-10-02 13:52 ` [PATCH 13/21] fhandle: decide the subtree check under mount_lock Christian Brauner
2026-10-02 13:52 ` [PATCH 14/21] namespace: keep the private nullfs instance in knullfs Christian Brauner
2026-10-02 13:52 ` [PATCH 15/21] namespace: nothing is mounted on or written through knullfs Christian Brauner
2026-10-02 13:52 ` [PATCH 16/21] fsnotify: let a filesystem refuse marks on its objects Christian Brauner
2026-10-02 14:26 ` Amir Goldstein
2026-10-02 13:52 ` [PATCH 17/21] nullfs: refuse file locks Christian Brauner
2026-10-02 13:52 ` [PATCH 18/21] nullfs: refuse leases and delegations Christian Brauner
2026-10-03 8:20 ` Jeff Layton
2026-10-02 13:52 ` [PATCH 19/21] readdir: take no inode lock on an immutable directory Christian Brauner
2026-10-02 13:52 ` [PATCH 20/21] selftests/filesystems: add a helper that holds a readdir in a page fault Christian Brauner
2026-10-02 13:52 ` [PATCH 21/21] selftests/filesystems: check that reading the root of an empty mount namespace stalls nobody Christian Brauner
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261002-work-mount-fixes-4-v1-10-dd44b89d44ce@kernel.org \
--to=brauner@kernel.org \
--cc=amir73il@gmail.com \
--cc=jack@suse.cz \
--cc=jannh@google.com \
--cc=jlayton@kernel.org \
--cc=linux-fsdevel@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=neil@brown.name \
--cc=viro@zeniv.linux.org.uk \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®