* [PATCH v4 1/8] landlock: Rename quiet_masks to quiet_access
2026-10-02 12:43 [PATCH v4 0/8] Landlock: Namespace and capability control Mickaël Salaün
@ 2026-10-02 12:43 ` Mickaël Salaün
2026-10-02 12:43 ` [PATCH v4 2/8] landlock: Wrap per-layer access masks in struct layer_config Mickaël Salaün
` (6 subsequent siblings)
7 siblings, 0 replies; 9+ messages in thread
From: Mickaël Salaün @ 2026-10-02 12:43 UTC (permalink / raw)
To: Christian Brauner, Günther Noack, Paul Moore, Serge E . Hallyn
Cc: Mickaël Salaün, Daniel Durning, Jonathan Corbet,
Justin Suess, Lennart Poettering, Mikhail Ivanov,
Nicolas Bouchinet, Shervin Oloumi, Tingmao Wang, kernel-team,
linux-fsdevel, linux-kernel, linux-security-module
The base stores the fs/net/scope quiet bitmasks in a struct access_masks
field named quiet_masks. A following commit adds the sibling struct
permission_masks field quiet_permission for capabilities and namespaces.
quiet_masks drops the access category and keeps the type half ("masks"),
while quiet_permission keeps the category and drops the type half, so
the two read as an inconsistent pair.
Rename the base field to quiet_access, giving the symmetric,
category-named pair quiet_access/quiet_permission that also matches the
UAPI handled_access_*/quiet_access_* naming. This is a
no-functional-change rename.
Cc: Günther Noack <gnoack@google.com>
Cc: Tingmao Wang <m@maowtm.org>
Reviewed-by: Tingmao Wang <m@maowtm.org>
Signed-off-by: Mickaël Salaün <mic@digikod.net>
---
Changes since v3:
https://patch.msgid.link/20260726161400.3010511-5-mic@digikod.net
- Adapt the rename to the split ruleset/domain and logging code.
- Add Reviewed-by: Tingmao Wang.
Changes since v2:
- New patch.
---
security/landlock/domain.c | 2 +-
security/landlock/domain.h | 4 ++--
security/landlock/log.c | 6 +++---
security/landlock/ruleset.h | 4 ++--
security/landlock/syscalls.c | 10 +++++-----
5 files changed, 13 insertions(+), 13 deletions(-)
diff --git a/security/landlock/domain.c b/security/landlock/domain.c
index 4031b581be07..c66663f8cd8b 100644
--- a/security/landlock/domain.c
+++ b/security/landlock/domain.c
@@ -479,7 +479,7 @@ landlock_merge_ruleset(struct landlock_domain *const parent,
return ERR_PTR(err);
#ifdef CONFIG_SECURITY_LANDLOCK_LOG
- new_dom->hierarchy->quiet_masks = ruleset->quiet_masks;
+ new_dom->hierarchy->quiet_access = ruleset->quiet_access;
#endif /* CONFIG_SECURITY_LANDLOCK_LOG */
return no_free_ptr(new_dom);
diff --git a/security/landlock/domain.h b/security/landlock/domain.h
index caa3d19d2c43..03f24382537c 100644
--- a/security/landlock/domain.h
+++ b/security/landlock/domain.h
@@ -130,10 +130,10 @@ struct landlock_hierarchy {
*/
log_new_exec : 1;
/**
- * @quiet_masks: Bitmasks of access that should be quieted (i.e. not
+ * @quiet_access: Bitmasks of access that should be quieted (i.e. not
* logged) if the related object is marked as quiet.
*/
- struct access_masks quiet_masks;
+ struct access_masks quiet_access;
#endif /* CONFIG_SECURITY_LANDLOCK_LOG */
};
diff --git a/security/landlock/log.c b/security/landlock/log.c
index a8578a6f2ce9..3f9ae1bacb23 100644
--- a/security/landlock/log.c
+++ b/security/landlock/log.c
@@ -436,7 +436,7 @@ is_denial_quieted(const struct landlock_request *const request,
if (object_quiet_flag) {
const access_mask_t quiet_mask =
pick_access_mask_for_request_type(
- request->type, youngest_denied->quiet_masks);
+ request->type, youngest_denied->quiet_access);
return (quiet_mask & missing) == missing;
}
@@ -447,10 +447,10 @@ is_denial_quieted(const struct landlock_request *const request,
*/
switch (request->type) {
case LANDLOCK_REQUEST_SCOPE_SIGNAL:
- return !!(youngest_denied->quiet_masks.scope &
+ return !!(youngest_denied->quiet_access.scope &
LANDLOCK_SCOPE_SIGNAL);
case LANDLOCK_REQUEST_SCOPE_ABSTRACT_UNIX_SOCKET:
- return !!(youngest_denied->quiet_masks.scope &
+ return !!(youngest_denied->quiet_access.scope &
LANDLOCK_SCOPE_ABSTRACT_UNIX_SOCKET);
/*
* Leave LANDLOCK_REQUEST_PTRACE and LANDLOCK_REQUEST_FS_CHANGE_TOPOLOGY
diff --git a/security/landlock/ruleset.h b/security/landlock/ruleset.h
index cf77f1806a95..8ebdd4fa098f 100644
--- a/security/landlock/ruleset.h
+++ b/security/landlock/ruleset.h
@@ -182,11 +182,11 @@ struct landlock_ruleset {
#endif /* CONFIG_TRACEPOINTS */
/**
- * @quiet_masks: Stores the quiet flags for an unmerged ruleset. For a
+ * @quiet_access: Stores the quiet flags for an unmerged ruleset. For a
* merged domain, this is stored in each layer's struct
* landlock_hierarchy instead.
*/
- struct access_masks quiet_masks;
+ struct access_masks quiet_access;
/**
* @handled_masks: Contains the subset of filesystem and network actions
* that are handled by this ruleset.
diff --git a/security/landlock/syscalls.c b/security/landlock/syscalls.c
index 1d02d57f4c48..17d8e9b7b0c6 100644
--- a/security/landlock/syscalls.c
+++ b/security/landlock/syscalls.c
@@ -280,9 +280,9 @@ SYSCALL_DEFINE3(landlock_create_ruleset,
if (IS_ERR(ruleset))
return PTR_ERR(ruleset);
- ruleset->quiet_masks.fs = ruleset_attr.quiet_access_fs;
- ruleset->quiet_masks.net = ruleset_attr.quiet_access_net;
- ruleset->quiet_masks.scope = ruleset_attr.quiet_scoped;
+ ruleset->quiet_access.fs = ruleset_attr.quiet_access_fs;
+ ruleset->quiet_access.net = ruleset_attr.quiet_access_net;
+ ruleset->quiet_access.scope = ruleset_attr.quiet_scoped;
/*
* Emits before anon_inode_getfd() installs the file descriptor, while
@@ -382,7 +382,7 @@ static int add_rule_path_beneath(struct landlock_ruleset *const ruleset,
return -EINVAL;
/* Checks for useless quiet flag. */
- if (flags & LANDLOCK_ADD_RULE_QUIET && !ruleset->quiet_masks.fs)
+ if (flags & LANDLOCK_ADD_RULE_QUIET && !ruleset->quiet_access.fs)
return -EINVAL;
/* Gets and checks the new rule. */
@@ -423,7 +423,7 @@ static int add_rule_net_port(struct landlock_ruleset *ruleset,
return -EINVAL;
/* Checks for useless quiet flag. */
- if (flags & LANDLOCK_ADD_RULE_QUIET && !ruleset->quiet_masks.net)
+ if (flags & LANDLOCK_ADD_RULE_QUIET && !ruleset->quiet_access.net)
return -EINVAL;
/* Denies inserting a rule with port greater than 65535. */
--
2.55.0
^ permalink raw reply [flat|nested] 9+ messages in thread* [PATCH v4 2/8] landlock: Wrap per-layer access masks in struct layer_config
2026-10-02 12:43 [PATCH v4 0/8] Landlock: Namespace and capability control Mickaël Salaün
2026-10-02 12:43 ` [PATCH v4 1/8] landlock: Rename quiet_masks to quiet_access Mickaël Salaün
@ 2026-10-02 12:43 ` Mickaël Salaün
2026-10-02 12:43 ` [PATCH v4 3/8] landlock: Enforce namespace use restrictions Mickaël Salaün
` (5 subsequent siblings)
7 siblings, 0 replies; 9+ messages in thread
From: Mickaël Salaün @ 2026-10-02 12:43 UTC (permalink / raw)
To: Christian Brauner, Günther Noack, Paul Moore, Serge E . Hallyn
Cc: Mickaël Salaün, Daniel Durning, Jonathan Corbet,
Justin Suess, Lennart Poettering, Mikhail Ivanov,
Nicolas Bouchinet, Shervin Oloumi, Tingmao Wang, kernel-team,
linux-fsdevel, linux-kernel, linux-security-module
The domain layer FAM stores struct access_masks values directly, while a
ruleset stores the equivalent single mutable value. Per-category
permissions need additional per-layer data beyond the handled-access
bitfields.
Introduce struct layer_config as the common value type. Keeping the
handled bitfields in its .handled member leaves struct access_masks as a
lightweight parameter type for functions that only need those bitfields,
while the complete layer can be snapshotted with one assignment.
At this point struct layer_config only wraps the four-byte access_masks,
so it does not grow the per-domain allocation: the maximum 16-entry FAM
remains 64 bytes.
No functional change.
Cc: Günther Noack <gnoack@google.com>
Reviewed-by: Günther Noack <gnoack@google.com>
Reviewed-by: Tingmao Wang <m@maowtm.org>
Signed-off-by: Mickaël Salaün <mic@digikod.net>
---
Changes since v3:
https://patch.msgid.link/20260726161400.3010511-6-mic@digikod.net
- Adapt the wrapper to the ruleset/domain split by using one
layer_config value for the ruleset singleton and the domain FAM.
- Generalize the struct access_masks comment because the type also
represents quiet and request masks.
Changes since v2:
https://patch.msgid.link/20260527181127.879771-4-mic@digikod.net
- The rebase required adopting the released check-time struct
layer_masks and the new per-ruleset quiet_access (added by the base's
merged quiet feature) and updating the struct landlock_ruleset @layers
kdoc; struct layer_config already existed in v2.
Changes since v1:
https://patch.msgid.link/20260312100444.2609563-5-mic@digikod.net
- Add Reviewed-by: Tingmao Wang.
- Address Günther Noack's review nits:
- Clarify that _LANDLOCK_ACCESS_FS_INITIALLY_DENIED is ORed with
the .handled field of all ruleset->layers[] entries (not the
entries themselves).
- Rename landlock_upgrade_handled_access_masks() to
landlock_upgrade_handled_layer_config() to match the parameter
type.
- Rewrap the @layers kdoc in struct landlock_ruleset.
- Rename struct layer_rights to struct layer_config: "config" is
the more general term for per-layer state.
- Add Reviewed-by: Günther Noack.
---
include/trace/events/landlock.h | 6 +++---
security/landlock/access.h | 32 +++++++++++++++++++++++---------
security/landlock/cred.h | 2 +-
security/landlock/domain.c | 10 +++++-----
security/landlock/domain.h | 30 ++++++++++++++----------------
security/landlock/fs.c | 2 +-
security/landlock/net.c | 2 +-
security/landlock/ruleset.c | 6 +++---
security/landlock/ruleset.h | 5 ++---
security/landlock/syscalls.c | 4 ++--
10 files changed, 55 insertions(+), 44 deletions(-)
diff --git a/include/trace/events/landlock.h b/include/trace/events/landlock.h
index 3a43638c9bc2..215d08c1e03d 100644
--- a/include/trace/events/landlock.h
+++ b/include/trace/events/landlock.h
@@ -353,9 +353,9 @@ TRACE_EVENT(landlock_create_ruleset,
TP_fast_assign(
__entry->ruleset_id = ruleset->id;
__entry->ruleset_version = ruleset->version;
- __entry->handled_fs = ruleset->handled_masks.fs;
- __entry->handled_net = ruleset->handled_masks.net;
- __entry->scoped = ruleset->handled_masks.scope;
+ __entry->handled_fs = ruleset->layer.handled.fs;
+ __entry->handled_net = ruleset->layer.handled.net;
+ __entry->scoped = ruleset->layer.handled.scope;
),
TP_printk("ruleset=%llx.%llu handled_fs=%s handled_net=%s scoped=%s",
diff --git a/security/landlock/access.h b/security/landlock/access.h
index f843835851d0..1b1dede27925 100644
--- a/security/landlock/access.h
+++ b/security/landlock/access.h
@@ -19,9 +19,9 @@
/*
* All access rights that are denied by default whether they are handled or not
- * by a ruleset/layer. This must be ORed with all domain->handled_masks[]
- * entries when we need to get the absolute handled access masks, see
- * landlock_upgrade_handled_access_masks().
+ * by a ruleset/layer. This must be ORed with the .handled field of all
+ * domain->layers[] entries when we need to get the absolute handled access
+ * masks, see landlock_upgrade_handled_layer_config().
*/
/* clang-format off */
#define _LANDLOCK_ACCESS_FS_INITIALLY_DENIED ( \
@@ -45,7 +45,7 @@ static_assert(BITS_PER_TYPE(access_mask_t) >= LANDLOCK_NUM_SCOPE);
/* Makes sure for_each_set_bit() and for_each_clear_bit() calls are OK. */
static_assert(sizeof(unsigned long) >= sizeof(access_mask_t));
-/* Ruleset access masks. */
+/* Access masks (bitfields only). */
struct access_masks {
access_mask_t fs : LANDLOCK_NUM_ACCESS_FS;
access_mask_t net : LANDLOCK_NUM_ACCESS_NET;
@@ -61,6 +61,20 @@ union access_masks_all {
static_assert(sizeof(typeof_member(union access_masks_all, masks)) ==
sizeof(typeof_member(union access_masks_all, all)));
+/**
+ * struct layer_config - Per-layer access configuration
+ *
+ * A ruleset stores one mutable layer and a domain stores a flexible array of
+ * immutable layers.
+ */
+struct layer_config {
+ /**
+ * @handled: Bitmask of access rights handled (i.e. restricted) by this
+ * layer.
+ */
+ struct access_masks handled;
+};
+
#define _LANDLOCK_LAYER_MASK_PADDING \
(BITS_PER_TYPE(access_mask_t) - LANDLOCK_NUM_ACCESS_MAX - \
IS_ENABLED(CONFIG_SECURITY_LANDLOCK_LOG))
@@ -131,17 +145,17 @@ static_assert(BITS_PER_TYPE(deny_masks_t) >=
static_assert(HWEIGHT(LANDLOCK_MAX_NUM_LAYERS) == 1);
/* Upgrades with all initially denied by default access rights. */
-static inline struct access_masks
-landlock_upgrade_handled_access_masks(struct access_masks access_masks)
+static inline struct layer_config
+landlock_upgrade_handled_layer_config(struct layer_config layer_config)
{
/*
* All access rights that are denied by default whether they are
* explicitly handled or not.
*/
- if (access_masks.fs)
- access_masks.fs |= _LANDLOCK_ACCESS_FS_INITIALLY_DENIED;
+ if (layer_config.handled.fs)
+ layer_config.handled.fs |= _LANDLOCK_ACCESS_FS_INITIALLY_DENIED;
- return access_masks;
+ return layer_config;
}
/* Checks the subset relation between access masks. */
diff --git a/security/landlock/cred.h b/security/landlock/cred.h
index a5ff9957949a..1c6a4838f38b 100644
--- a/security/landlock/cred.h
+++ b/security/landlock/cred.h
@@ -138,7 +138,7 @@ landlock_get_applicable_subject(const struct cred *const cred,
for (layer_level = domain->num_layers - 1; layer_level >= 0;
layer_level--) {
union access_masks_all layer = {
- .masks = domain->handled_masks[layer_level],
+ .masks = domain->layers[layer_level].handled,
};
if (layer.all & masks_all.all) {
diff --git a/security/landlock/domain.c b/security/landlock/domain.c
index c66663f8cd8b..636abcd75aff 100644
--- a/security/landlock/domain.c
+++ b/security/landlock/domain.c
@@ -50,7 +50,7 @@ static struct landlock_domain *create_domain(const u32 num_layers)
struct landlock_domain *new_domain;
build_check_domain();
- new_domain = kzalloc_flex(*new_domain, handled_masks, num_layers,
+ new_domain = kzalloc_flex(*new_domain, layers, num_layers,
GFP_KERNEL_ACCOUNT);
if (!new_domain)
return ERR_PTR(-ENOMEM);
@@ -328,8 +328,8 @@ static int merge_ruleset(struct landlock_domain *const dst,
if (WARN_ON_ONCE(dst->num_layers < 1))
return -EINVAL;
- dst->handled_masks[dst->num_layers - 1] =
- landlock_upgrade_handled_access_masks(src->handled_masks);
+ dst->layers[dst->num_layers - 1] =
+ landlock_upgrade_handled_layer_config(src->layer);
/* Merges the @src inode tree. */
err = merge_tree(dst, src, LANDLOCK_KEY_INODE);
@@ -404,8 +404,8 @@ static int inherit_ruleset(struct landlock_domain *const parent,
/*
* Copies the parent layer stack and leaves a space for the new layer.
*/
- memcpy(child->handled_masks, parent->handled_masks,
- flex_array_size(parent, handled_masks, parent->num_layers));
+ memcpy(child->layers, parent->layers,
+ flex_array_size(parent, layers, parent->num_layers));
if (WARN_ON_ONCE(!parent->hierarchy))
return -EINVAL;
diff --git a/security/landlock/domain.h b/security/landlock/domain.h
index 03f24382537c..bdec27c3bc7f 100644
--- a/security/landlock/domain.h
+++ b/security/landlock/domain.h
@@ -217,7 +217,7 @@ struct landlock_domain {
* @work_free: Enables to free a domain within a lockless
* section. This is only used by landlock_put_domain_deferred()
* when @usage reaches zero. The fields @usage, @num_layers and
- * @handled_masks are then unused.
+ * @layers are then unused.
*/
struct work_struct work_free;
struct {
@@ -233,18 +233,16 @@ struct landlock_domain {
*/
u32 num_layers;
/**
- * @handled_masks: Contains the subset of filesystem and
- * network actions that are restricted by a domain. A
- * domain saves all layers of merged rulesets in a stack
- * (FAM), starting from the first layer to the last one.
- * These layers are used when merging rulesets, for user
- * space backward compatibility (i.e. future-proof), and
- * to properly handle merged rulesets without
- * overlapping access rights. These layers are set once
- * and never changed for the lifetime of the domain.
+ * @layers: Per-layer access configuration. A domain
+ * saves all layers of merged rulesets in a stack (FAM),
+ * starting from the first layer to the last one. These
+ * layers are used when merging rulesets, for user space
+ * backward compatibility (i.e. future-proof), and to
+ * properly handle merged rulesets without overlapping
+ * access rights. These layers are set once and never
+ * changed for the lifetime of the domain.
*/
- struct access_masks
- handled_masks[] __counted_by(num_layers);
+ struct layer_config layers[] __counted_by(num_layers);
};
};
};
@@ -254,7 +252,7 @@ landlock_get_fs_access_mask(const struct landlock_domain *const domain,
const u16 layer_level)
{
/* Handles all initially denied by default access rights. */
- return domain->handled_masks[layer_level].fs |
+ return domain->layers[layer_level].handled.fs |
_LANDLOCK_ACCESS_FS_INITIALLY_DENIED;
}
@@ -262,14 +260,14 @@ static inline access_mask_t
landlock_get_net_access_mask(const struct landlock_domain *const domain,
const u16 layer_level)
{
- return domain->handled_masks[layer_level].net;
+ return domain->layers[layer_level].handled.net;
}
static inline access_mask_t
landlock_get_scope_mask(const struct landlock_domain *const domain,
const u16 layer_level)
{
- return domain->handled_masks[layer_level].scope;
+ return domain->layers[layer_level].handled.scope;
}
/**
@@ -288,7 +286,7 @@ landlock_union_access_masks(const struct landlock_domain *const domain)
for (layer_level = 0; layer_level < domain->num_layers; layer_level++) {
union access_masks_all layer = {
- .masks = domain->handled_masks[layer_level],
+ .masks = domain->layers[layer_level].handled,
};
matches.all |= layer.all;
diff --git a/security/landlock/fs.c b/security/landlock/fs.c
index cab43892ec2f..a8fa8f77e775 100644
--- a/security/landlock/fs.c
+++ b/security/landlock/fs.c
@@ -342,7 +342,7 @@ int landlock_append_fs_rule(struct landlock_ruleset *const ruleset,
/* Transforms relative access rights to absolute ones. */
access_rights |= LANDLOCK_MASK_ACCESS_FS &
- ~(ruleset->handled_masks.fs |
+ ~(ruleset->layer.handled.fs |
_LANDLOCK_ACCESS_FS_INITIALLY_DENIED);
id.key.object = get_inode_object(d_backing_inode(path->dentry));
if (IS_ERR(id.key.object))
diff --git a/security/landlock/net.c b/security/landlock/net.c
index 6fe0dbde3b78..8fd73cf4bd15 100644
--- a/security/landlock/net.c
+++ b/security/landlock/net.c
@@ -35,7 +35,7 @@ int landlock_append_net_rule(struct landlock_ruleset *const ruleset,
BUILD_BUG_ON(sizeof(port) > sizeof(id.key.data));
/* Transforms relative access rights to absolute ones. */
- access_rights |= LANDLOCK_MASK_ACCESS_NET & ~ruleset->handled_masks.net;
+ access_rights |= LANDLOCK_MASK_ACCESS_NET & ~ruleset->layer.handled.net;
mutex_lock(&ruleset->lock);
err = landlock_insert_rule(ruleset, id, access_rights, flags);
diff --git a/security/landlock/ruleset.c b/security/landlock/ruleset.c
index a5d135d085cb..edf9396deac6 100644
--- a/security/landlock/ruleset.c
+++ b/security/landlock/ruleset.c
@@ -64,20 +64,20 @@ landlock_create_ruleset(const access_mask_t fs_access_mask,
LANDLOCK_MASK_ACCESS_FS;
WARN_ON_ONCE(fs_access_mask != mask);
- new_ruleset->handled_masks.fs |= mask;
+ new_ruleset->layer.handled.fs |= mask;
}
if (net_access_mask) {
const access_mask_t mask = net_access_mask &
LANDLOCK_MASK_ACCESS_NET;
WARN_ON_ONCE(net_access_mask != mask);
- new_ruleset->handled_masks.net |= mask;
+ new_ruleset->layer.handled.net |= mask;
}
if (scope_mask) {
const access_mask_t mask = scope_mask & LANDLOCK_MASK_SCOPE;
WARN_ON_ONCE(scope_mask != mask);
- new_ruleset->handled_masks.scope |= mask;
+ new_ruleset->layer.handled.scope |= mask;
}
return new_ruleset;
}
diff --git a/security/landlock/ruleset.h b/security/landlock/ruleset.h
index 8ebdd4fa098f..424055a7af86 100644
--- a/security/landlock/ruleset.h
+++ b/security/landlock/ruleset.h
@@ -188,10 +188,9 @@ struct landlock_ruleset {
*/
struct access_masks quiet_access;
/**
- * @handled_masks: Contains the subset of filesystem and network actions
- * that are handled by this ruleset.
+ * @layer: Access configuration for this ruleset's single mutable layer.
*/
- struct access_masks handled_masks;
+ struct layer_config layer;
};
struct landlock_ruleset *
diff --git a/security/landlock/syscalls.c b/security/landlock/syscalls.c
index 17d8e9b7b0c6..ec616d198184 100644
--- a/security/landlock/syscalls.c
+++ b/security/landlock/syscalls.c
@@ -377,7 +377,7 @@ static int add_rule_path_beneath(struct landlock_ruleset *const ruleset,
return -ENOMSG;
/* Checks that allowed_access matches the @ruleset constraints. */
- mask = ruleset->handled_masks.fs;
+ mask = ruleset->layer.handled.fs;
if ((path_beneath_attr.allowed_access | mask) != mask)
return -EINVAL;
@@ -418,7 +418,7 @@ static int add_rule_net_port(struct landlock_ruleset *ruleset,
return -ENOMSG;
/* Checks that allowed_access matches the @ruleset constraints. */
- mask = ruleset->handled_masks.net;
+ mask = ruleset->layer.handled.net;
if ((net_port_attr.allowed_access | mask) != mask)
return -EINVAL;
--
2.55.0
^ permalink raw reply [flat|nested] 9+ messages in thread* [PATCH v4 3/8] landlock: Enforce namespace use restrictions
2026-10-02 12:43 [PATCH v4 0/8] Landlock: Namespace and capability control Mickaël Salaün
2026-10-02 12:43 ` [PATCH v4 1/8] landlock: Rename quiet_masks to quiet_access Mickaël Salaün
2026-10-02 12:43 ` [PATCH v4 2/8] landlock: Wrap per-layer access masks in struct layer_config Mickaël Salaün
@ 2026-10-02 12:43 ` Mickaël Salaün
2026-10-02 12:43 ` [PATCH v4 4/8] landlock: Enforce capability restrictions Mickaël Salaün
` (4 subsequent siblings)
7 siblings, 0 replies; 9+ messages in thread
From: Mickaël Salaün @ 2026-10-02 12:43 UTC (permalink / raw)
To: Christian Brauner, Günther Noack, Paul Moore, Serge E . Hallyn
Cc: Mickaël Salaün, Daniel Durning, Jonathan Corbet,
Justin Suess, Lennart Poettering, Mikhail Ivanov,
Nicolas Bouchinet, Shervin Oloumi, Tingmao Wang, kernel-team,
linux-fsdevel, linux-kernel, linux-security-module
Add Landlock enforcement for namespace use through the LSM
namespace_init and namespace_install hooks. This lets a sandboxed
process restrict which namespace types it can create, join, or acquire a
file descriptor for, using LANDLOCK_PERMISSION_NAMESPACE_USE and
per-type rules.
Introduce the handled_permissions field in struct landlock_ruleset_attr:
each permission gates every use of a kernel-defined category (CLONE_NEW*
namespace types, CAP_* capabilities) with complete deny-by-default
coverage, so unknown member values need no validation, being denied
until a rule allows them. This UAPI extension advances the Landlock ABI
from 11 to 12.
There is no domain-ancestry bypass and no namespace-creator tracking,
only a flat per-layer allowed-types bitmask: hook_namespace_init()
covers creation through clone(2), unshare(2), open_tree(2) and
fsmount(2), and hook_namespace_install() covers setns(2). Each
permission hook maps its member mask explicitly in the shared permission
walker, which warns and denies if a future caller omits that mapping.
These categorical denials return -EPERM, matching Landlock's scope and
mount-topology denials rather than its object-access -EACCES convention.
struct permission_masks is forced to eight bytes with __packed and
__aligned because it must eventually hold both the eight namespace types
and all capabilities, and because m68k GCC otherwise packs its u64
bitfields at byte granularity, making the structure smaller than
sizeof(u64). Carrying it grows struct layer_config from 4 to 16 bytes,
so the largest 16-entry domain FAM grows from 64 to 256 bytes.
The permissions selector keeps the rule attribute self-describing:
- User space can mask it against handled_permissions, as it masks
allowed_access against handled_access_fs, so a program built against
newer headers still adds the rules an older kernel supports.
- It tells apart two attributes that share the same three-u64 layout.
Unknown member bits are ignored, so without it a mismatched rule type
would apply a capability mask as namespace types.
- It leaves room for a rule to carry several permissions, or for another
rule type to key the same permission differently.
The rule also carries a quiet_namespace_types bitmask that suppresses
audit submission for denied members without granting them. A sandbox
knowingly running a caller that probes a type it will never be granted
would otherwise flood the audit log and drown the surprising denials
that matter. Quiet is per-member rather than a coarse per-category
ruleset bit, so a sandbox can silence CLONE_NEWNET while still auditing
CLONE_NEWUTS; making it the complement of the allowed set would be
broad, could not audit a member that is neither allowed nor explicitly
quieted, and would auto-hide members added by future kernels. It is a
per-rule bitmask rather than the LANDLOCK_ADD_RULE_QUIET flag and the
ruleset quiet_access_* masks, which suit the unbounded rb-tree objects
of filesystem and network rules, so that flag is rejected here. Only
the youngest denying layer's quiet mask decides, so a parent cannot
silence a denial made by a deeper layer.
Trace the handled permission mask, every successful namespace rule, and
namespace denials. Successful effective no-ops advance the version, and
quiet denials remain trace-visible with logged=0. The denial event
takes the blockers argument of the filesystem and network events and
records the same blockers_type and blockers_access fields, so one filter
expression spans the mask-bearing denial events. Handled and rule
permission masks may contain permissions from multiple domains in one
field, so their names are domain-qualified, such as namespace.use. A
blocker is already interpreted in the domain supplied by the audit
prefix or denial event, so it retains the bare action name use, matching
existing access blockers such as read_file.
User namespace creation does not require capabilities, so Landlock can
restrict it directly. Non-user namespace types require CAP_SYS_ADMIN
before the Landlock check is reached; when the capability permission
added by the next commit is also handled, both must allow the operation.
Cc: Christian Brauner <brauner@kernel.org>
Cc: Günther Noack <gnoack@google.com>
Cc: Paul Moore <paul@paul-moore.com>
Cc: Serge E. Hallyn <serge@hallyn.com>
Signed-off-by: Mickaël Salaün <mic@digikod.net>
---
Changes since v3:
https://patch.msgid.link/20260726161400.3010511-8-mic@digikod.net
- Adapt to the merged namespace/audit prerequisites and the
ruleset/domain split, snapshotting the allowed and quiet masks under
the source ruleset lock.
- Suppress and assert the expected warnings in the invalid-input KUnit
tests.
- Map namespace types with a lookup table, and drop __attribute_const__
from the conversion helpers, which warn on invalid input.
- Advance the Landlock ABI to 12.
- Spell out perm as permission in the UAPI constant and members, the
internal masks and helpers, and the trace fields, and name the audit
blocker namespace.use after its domain instead of perm.namespace_use.
- Justify the permissions selector by what it does today: user space can
mask it against handled_permissions, and it tells apart two attributes
that share one layout, whose unknown member bits are ignored.
- Trace handled permissions, successful namespace rules, and denials.
Successful effective no-ops advance the version, while quiet denials
remain trace-visible with logged=0.
- Tell programs to omit empty rules.
- Complete the permission-rule errors and explain fail-closed dispatch
and the -EPERM convention.
- Drop Reviewed-by: Tingmao Wang: this version folds the namespace trace
events and ABI 12 into this patch, which her v3 review did not cover;
a fresh review is welcome.
Changes since v2:
https://patch.msgid.link/20260527181127.879771-5-mic@digikod.net
- Rename the namespace rule attribute fields (allowed_perm to perm,
namespace_types to allowed_namespace_types) and add a
quiet_namespace_types bitmask that suppresses the audit records of
specific denied namespace types, together with the shared per-layer
quiet member mask read in landlock_log_denial(); the rule attribute
grows from 16 to 24 bytes.
- Copy the accumulated quiet_perm mask into the domain hierarchy in
merge_ruleset(), under the ruleset merge lock and atomically with the
allowed mask (no separate lock).
- Dropped Reviewed-by: Günther Noack and Tingmao Wang, as this version
adds the quiet member mask described above, which their v1 review did
not cover. Fresh review welcome.
Changes since v1:
https://patch.msgid.link/20260312100444.2609563-6-mic@digikod.net
- Add __packed __aligned(sizeof(u64)) to struct perm_masks to fix
static_assert failure on m68k, where GCC packs bitfields at byte
granularity.
- Use ns_id instead of inum in namespace audit records.
- Add WARN_ON_ONCE guards for invalid perm_bit or request_value in
landlock_perm_is_denied(), denying with the youngest layer on
invalid input (suggested by Tingmao Wang).
- Fix double backtick in landlock_perm_is_denied() kernel-doc.
- Add Reviewed-by: Tingmao Wang.
- Mention commit 935a04923ad2 ("nsproxy: Add FOR_EACH_NS_TYPE()
X-macro and CLONE_NS_ALL") as a dependency in the body and add
Depends-on: trailer.
- Rename internal struct perm_rules to perm_masks to parallel the
sibling access_masks in struct layer_config.
- Document the allowed_perm design rationale (extensibility for
future sub-permissions, type discriminant safeguard).
- Rename LANDLOCK_PERM_NAMESPACE_ENTER to LANDLOCK_PERM_NAMESPACE_USE
and audit blocker perm.namespace_enter to perm.namespace_use for
semantic accuracy. The verb _ENTER fits setns/unshare/clone
(caller becomes namespace member) but misleads for open_tree and
fsmount (caller holds an fd reference, does not enter). _USE
covers both cases and mirrors LANDLOCK_PERM_CAPABILITY_USE.
Update the commit title accordingly.
- Replace "chokepoint"/"gateway" prose in @handled_perm kdoc and the
Permission flags DOC block with the per-category framing.
- Expand the LANDLOCK_PERM_NAMESPACE_USE kdoc to enumerate creation
(unshare/clone/clone3), joining (setns), and fd-reference
(open_tree/fsmount) paths.
- Rewrite the commit body to drop chokepoint/gateway terminology in
favour of per-category framing, matching the doc rewrite.
- Rename struct layer_rights to struct layer_config (companion
change to the introducing commit).
- Surface the empty-check semantics in the
landlock_namespace_attr.namespace_types kdoc: a rule that sets only
bits unknown to the running kernel succeeds but has no runtime
effect.
- Cascade the LSM hook rename namespace_alloc -> namespace_init
(LSM_HOOK_INIT registration and local handler hook_namespace_alloc ->
hook_namespace_init), companion change to the introducing commit.
- Rename the static helper landlock_check_ns_type() to check_ns_type():
the landlock_ prefix is reserved for non-static symbols exported via
headers; file-static helpers follow the prefix-free convention used
in security/landlock/.
- Add Reviewed-by: Günther Noack.
---
include/linux/landlock.h | 26 +-
include/trace/events/landlock.h | 131 +++++++++-
include/uapi/linux/landlock.h | 73 ++++++
security/landlock/Makefile | 3 +-
security/landlock/access.h | 29 ++-
security/landlock/audit.c | 28 +-
security/landlock/domain.c | 1 +
security/landlock/domain.h | 61 +++++
security/landlock/limits.h | 7 +
security/landlock/log.c | 37 ++-
security/landlock/log.h | 9 +-
security/landlock/ns.c | 241 ++++++++++++++++++
security/landlock/ns.h | 18 ++
security/landlock/ruleset.c | 21 +-
security/landlock/ruleset.h | 24 +-
security/landlock/setup.c | 2 +
security/landlock/syscalls.c | 114 ++++++++-
security/landlock/trace.c | 15 +-
tools/testing/selftests/landlock/base_test.c | 2 +-
tools/testing/selftests/landlock/trace.h | 3 +-
tools/testing/selftests/landlock/trace_test.c | 8 +-
21 files changed, 800 insertions(+), 53 deletions(-)
create mode 100644 security/landlock/ns.c
create mode 100644 security/landlock/ns.h
diff --git a/include/linux/landlock.h b/include/linux/landlock.h
index 004cbd0b9298..d288e3b6756f 100644
--- a/include/linux/landlock.h
+++ b/include/linux/landlock.h
@@ -13,14 +13,16 @@
#include <uapi/linux/landlock.h>
/*
- * Access-right and scope names, shared between the audit records (get_blocker()
- * in security/landlock/audit.c) and the trace events
+ * Access-right, scope, and permission names, shared between the audit records
+ * (get_blocker() in security/landlock/audit.c) and the trace events
* (include/trace/events/landlock.h). A consumer defines
* _LANDLOCK_NAME_ENTRY(mask, name) before expanding a list and undefines it
* afterwards: audit maps each entry to a "[bit] = name" slot for O(1) lookup,
* the trace events map it to a __print_flags() { mask, name } pair. The bit
* value lives only in the LANDLOCK_* UAPI constant each entry references.
- * Names are unprefixed; audit prepends the "fs."/"net."/"scope." category.
+ * Access-right and scope names are unprefixed; audit prepends the
+ * "fs."/"net."/"scope." category. Permission entries carry an action and a
+ * domain for the qualified and bare views below.
*/
#define _LANDLOCK_ACCESS_FS_NAMES \
_LANDLOCK_NAME_ENTRY(LANDLOCK_ACCESS_FS_EXECUTE, "execute"), \
@@ -53,4 +55,22 @@
"abstract_unix_socket"), \
_LANDLOCK_NAME_ENTRY(LANDLOCK_SCOPE_SIGNAL, "signal")
+#define _LANDLOCK_PERMISSION_NAMESPACE_NAME "namespace"
+
+#define _LANDLOCK_PERMISSION_LIST(entry) \
+ entry(LANDLOCK_PERMISSION_NAMESPACE_USE, "use", \
+ _LANDLOCK_PERMISSION_NAMESPACE_NAME)
+
+#define _LANDLOCK_PERMISSION_QUALIFIED_ENTRY(mask, action, domain) \
+ _LANDLOCK_NAME_ENTRY(mask, domain "." action)
+
+#define _LANDLOCK_PERMISSION_BARE_ENTRY(mask, action, ...) \
+ _LANDLOCK_NAME_ENTRY(mask, action)
+
+#define _LANDLOCK_PERMISSION_NAMES \
+ _LANDLOCK_PERMISSION_LIST(_LANDLOCK_PERMISSION_QUALIFIED_ENTRY)
+
+#define _LANDLOCK_PERMISSION_BLOCKER_NAMES \
+ _LANDLOCK_PERMISSION_LIST(_LANDLOCK_PERMISSION_BARE_ENTRY)
+
#endif /* _LINUX_LANDLOCK_H */
diff --git a/include/trace/events/landlock.h b/include/trace/events/landlock.h
index 215d08c1e03d..d5d08f751a53 100644
--- a/include/trace/events/landlock.h
+++ b/include/trace/events/landlock.h
@@ -36,6 +36,7 @@ static_assert(sizeof(access_mask_t) <= sizeof(u64));
TRACE_DEFINE_ENUM(LANDLOCK_REQUEST_FS_CHANGE_TOPOLOGY);
TRACE_DEFINE_ENUM(LANDLOCK_REQUEST_FS_ACCESS);
TRACE_DEFINE_ENUM(LANDLOCK_REQUEST_NET_ACCESS);
+TRACE_DEFINE_ENUM(LANDLOCK_REQUEST_NAMESPACE);
#ifdef CREATE_TRACE_POINTS
@@ -281,8 +282,8 @@ static inline const char *__trace_landlock_print_layers(
* stateless ftrace filter can select the denials the domain submits to
* audit with logged==1, without reconstructing it from the per-execution
* log flags. Denial events order their fields as domain, same_exec,
- * logged, then blockers (deny_access events only), then the type-specific
- * object fields, then any variable-length field.
+ * logged, then the blockers verdict input, then the
+ * type-specific object fields, then any variable-length field.
*
* Relational referents
* ~~~~~~~~~~~~~~~~~~~~~
@@ -302,9 +303,9 @@ static inline const char *__trace_landlock_print_layers(
* Blocker fields
* ~~~~~~~~~~~~~~
*
- * The filesystem and network blocker arguments identify the request type
- * and carry its final missing access subset when applicable. The type
- * determines how to interpret the access value.
+ * Blocker arguments identify the request type and carry its final missing
+ * access subset when applicable. The type determines how to interpret the
+ * access value.
*/
/*
@@ -343,11 +344,12 @@ TRACE_EVENT(landlock_create_ruleset,
TP_ARGS(ruleset),
TP_STRUCT__entry(
- __field( u64, ruleset_id )
- __field( u64, ruleset_version )
- __field( access_mask_t, handled_fs )
- __field( access_mask_t, handled_net )
- __field( access_mask_t, scoped )
+ __field( u64, ruleset_id )
+ __field( u64, ruleset_version )
+ __field( access_mask_t, handled_fs )
+ __field( access_mask_t, handled_net )
+ __field( access_mask_t, scoped )
+ __field( access_mask_t, handled_permissions )
),
TP_fast_assign(
@@ -356,13 +358,17 @@ TRACE_EVENT(landlock_create_ruleset,
__entry->handled_fs = ruleset->layer.handled.fs;
__entry->handled_net = ruleset->layer.handled.net;
__entry->scoped = ruleset->layer.handled.scope;
+ __entry->handled_permissions =
+ ruleset->layer.handled.permissions;
),
- TP_printk("ruleset=%llx.%llu handled_fs=%s handled_net=%s scoped=%s",
+ TP_printk("ruleset=%llx.%llu handled_fs=%s handled_net=%s scoped=%s handled_permissions=%s",
__entry->ruleset_id, __entry->ruleset_version,
__print_flags(__entry->handled_fs, "|", _LANDLOCK_ACCESS_FS_NAMES),
__print_flags(__entry->handled_net, "|", _LANDLOCK_ACCESS_NET_NAMES),
- __print_flags(__entry->scoped, "|", _LANDLOCK_SCOPE_NAMES))
+ __print_flags(__entry->scoped, "|", _LANDLOCK_SCOPE_NAMES),
+ __print_flags(__entry->handled_permissions, "|",
+ _LANDLOCK_PERMISSION_NAMES))
);
/**
@@ -496,6 +502,56 @@ TRACE_EVENT(landlock_add_rule_net_port,
__entry->port)
);
+/**
+ * landlock_add_rule_namespace - Namespace rule added to a ruleset
+ *
+ * @ruleset: Source ruleset (never NULL).
+ * @flags: Complete validated landlock_add_rule_flags value supplied by this
+ * successful call, not the rule's accumulated quiet state.
+ * @permissions: Validated permission mask from the rule attribute.
+ * @allowed_namespace_types: Effective known namespace types allowed by this
+ * call, using CLONE_NEW* values.
+ * @quiet_namespace_types: Effective known namespace types quieted by this
+ * call, using CLONE_NEW* values.
+ *
+ * Emitted by sys_landlock_add_rule() under the modified ruleset's lock, so
+ * the reported ruleset is a stable snapshot that no concurrent writer can
+ * change.
+ */
+TRACE_EVENT(landlock_add_rule_namespace,
+
+ TP_PROTO(const struct landlock_ruleset *ruleset, u32 flags,
+ u64 permissions, u64 allowed_namespace_types,
+ u64 quiet_namespace_types),
+
+ TP_ARGS(ruleset, flags, permissions, allowed_namespace_types,
+ quiet_namespace_types),
+
+ TP_STRUCT__entry(
+ __field( u64, ruleset_id )
+ __field( u64, ruleset_version )
+ __field( access_mask_t, permissions )
+ __field( u64, allowed_namespace_types )
+ __field( u64, quiet_namespace_types )
+ ),
+
+ TP_fast_assign(
+ lockdep_assert_held(&ruleset->lock);
+ __entry->ruleset_id = ruleset->id;
+ __entry->ruleset_version = ruleset->version;
+ __entry->permissions = permissions;
+ __entry->allowed_namespace_types = allowed_namespace_types;
+ __entry->quiet_namespace_types = quiet_namespace_types;
+ ),
+
+ TP_printk("ruleset=%llx.%llu permissions=%s allowed_namespace_types=0x%llx quiet_namespace_types=0x%llx",
+ __entry->ruleset_id, __entry->ruleset_version,
+ __print_flags(__entry->permissions, "|",
+ _LANDLOCK_PERMISSION_NAMES),
+ __entry->allowed_namespace_types,
+ __entry->quiet_namespace_types)
+);
+
/**
* landlock_create_domain - New domain created
*
@@ -873,6 +929,57 @@ TRACE_EVENT(landlock_deny_access_net,
__entry->port)
);
+/**
+ * landlock_deny_permission_namespace - Namespace use denied
+ *
+ * @hierarchy: Denying domain's hierarchy node (never NULL); its id is the
+ * domain field.
+ * @same_exec: Whether the current task entered the denying domain itself.
+ * @logged: Whether this denial was selected for audit logging.
+ * @blockers: Request type and final missing permission subset (never NULL).
+ * @namespace_type: CLONE_NEW* namespace type that was denied.
+ * @namespace_id: Namespace ID, or 0 when creation was denied.
+ *
+ * Emitted when a Landlock domain denies namespace use.
+ */
+TRACE_EVENT(landlock_deny_permission_namespace,
+
+ TP_PROTO(const struct landlock_hierarchy *hierarchy, bool same_exec,
+ bool logged, const struct landlock_blockers *blockers,
+ u32 namespace_type, u64 namespace_id),
+
+ TP_ARGS(hierarchy, same_exec, logged, blockers, namespace_type,
+ namespace_id),
+
+ TP_STRUCT__entry(
+ __field( u64, domain_id )
+ __field( bool, same_exec )
+ __field( bool, logged )
+ __field( enum landlock_request_type, blockers_type )
+ __field( access_mask_t, blockers_access )
+ __field( u32, namespace_type )
+ __field( u64, namespace_id )
+ ),
+
+ TP_fast_assign(
+ __entry->domain_id = hierarchy->id;
+ __entry->same_exec = same_exec;
+ __entry->logged = logged;
+ __entry->blockers_type = blockers->type;
+ __entry->blockers_access = blockers->access;
+ __entry->namespace_type = namespace_type;
+ __entry->namespace_id = namespace_id;
+ ),
+
+ TP_printk("domain=%llx same_exec=%d logged=%d blockers=%s namespace_type=0x%x namespace_id=%llu",
+ __entry->domain_id, __entry->same_exec, __entry->logged,
+ __entry->blockers_type == LANDLOCK_REQUEST_NAMESPACE ?
+ __print_flags(__entry->blockers_access, "|",
+ _LANDLOCK_PERMISSION_BLOCKER_NAMES) :
+ "unknown",
+ __entry->namespace_type, __entry->namespace_id)
+);
+
/**
* landlock_deny_ptrace - Ptrace access denied by a Landlock domain
*
diff --git a/include/uapi/linux/landlock.h b/include/uapi/linux/landlock.h
index cceda3b3b961..bb8ec589ddac 100644
--- a/include/uapi/linux/landlock.h
+++ b/include/uapi/linux/landlock.h
@@ -78,6 +78,11 @@ struct landlock_ruleset_attr {
* @quiet_scoped: Bitmask of scoped actions which should not be logged.
*/
__u64 quiet_scoped;
+ /**
+ * @handled_permissions: Bitmask of handled permissions (cf. `Permission
+ * flags`_).
+ */
+ __u64 handled_permissions;
};
/**
@@ -228,6 +233,10 @@ enum landlock_rule_type {
* landlock_net_port_attr .
*/
LANDLOCK_RULE_NET_PORT,
+ /**
+ * @LANDLOCK_RULE_NAMESPACE: Type of a &struct landlock_namespace_attr .
+ */
+ LANDLOCK_RULE_NAMESPACE,
};
/**
@@ -281,6 +290,41 @@ struct landlock_net_port_attr {
__u64 port;
};
+/**
+ * struct landlock_namespace_attr - Namespace type definition
+ *
+ * Argument of sys_landlock_add_rule() with %LANDLOCK_RULE_NAMESPACE.
+ */
+struct landlock_namespace_attr {
+ /**
+ * @permissions: Must be set to %LANDLOCK_PERMISSION_NAMESPACE_USE.
+ */
+ __u64 permissions;
+ /**
+ * @allowed_namespace_types: Bitmask of namespace types (``CLONE_NEW*``
+ * flags) to allow under this rule. Unknown bits are silently ignored
+ * for forward compatibility.
+ */
+ __u64 allowed_namespace_types;
+ /**
+ * @quiet_namespace_types: Bitmask of namespace types (``CLONE_NEW*``
+ * flags) whose denial by this layer should not be submitted to audit,
+ * even when landlock_restrict_self() enables audit logging. Only audit
+ * records attributed to this layer are suppressed; denial tracepoints
+ * still fire (see `Permission flags`_). Bits also set in
+ * @allowed_namespace_types have no effect, since an allowed type is
+ * never denied. Unknown bits are silently ignored.
+ *
+ * At least one of @allowed_namespace_types or @quiet_namespace_types
+ * must be non-zero, otherwise the call returns ``-ENOMSG``. The
+ * non-zero check runs on the raw input before unknown-bit masking, so a
+ * rule that sets only bits unknown to the running kernel succeeds but
+ * has no runtime effect. Programs should omit this rule when they
+ * neither allow nor quiet a namespace type.
+ */
+ __u64 quiet_namespace_types;
+};
+
/**
* DOC: fs_access
*
@@ -507,4 +551,33 @@ struct landlock_net_port_attr {
#define LANDLOCK_SCOPE_SIGNAL (1ULL << 1)
/* clang-format on*/
+/**
+ * DOC: permission
+ *
+ * Permission flags
+ * ~~~~~~~~~~~~~~~~
+ *
+ * These flags restrict the use of members of a category, each member being
+ * identified by a constant from another kernel subsystem (e.g. CLONE_NEW*
+ * namespace types, CAP_* capabilities). A flag covers every kernel path that
+ * uses a member of its category, and members that no rule explicitly allows are
+ * denied. Values unknown to the running kernel are silently accepted for
+ * forward compatibility and stay denied by default. See
+ * Documentation/security/landlock.rst for design details.
+ *
+ * When a ruleset handles multiple permissions whose operations overlap (e.g. a
+ * non-user namespace needs both its namespace type and CAP_SYS_ADMIN), the
+ * operation is allowed only if each handled permission independently allows it.
+ * See Documentation/userspace-api/landlock.rst.
+ *
+ * - %LANDLOCK_PERMISSION_NAMESPACE_USE: Restrict the use of specific namespace
+ * types: creation (:manpage:`unshare(2)`, :manpage:`clone(2)`,
+ * :manpage:`clone3(2)`), joining (:manpage:`setns(2)`), and acquiring an fd
+ * reference (:manpage:`open_tree(2)`, :manpage:`fsmount(2)`). A process in a
+ * Landlock domain that handles this permission is denied from using namespace
+ * types that are not explicitly allowed by a %LANDLOCK_RULE_NAMESPACE rule.
+ * Support added in Landlock ABI version 12.
+ */
+#define LANDLOCK_PERMISSION_NAMESPACE_USE (1ULL << 0)
+
#endif /* _UAPI_LINUX_LANDLOCK_H */
diff --git a/security/landlock/Makefile b/security/landlock/Makefile
index 2711f4876939..e88ca842c782 100644
--- a/security/landlock/Makefile
+++ b/security/landlock/Makefile
@@ -9,7 +9,8 @@ landlock-y := \
task.o \
fs.o \
tsync.o \
- domain.o
+ domain.o \
+ ns.o
landlock-$(CONFIG_INET) += net.o
diff --git a/security/landlock/access.h b/security/landlock/access.h
index 1b1dede27925..e3dbaa29a29f 100644
--- a/security/landlock/access.h
+++ b/security/landlock/access.h
@@ -42,14 +42,17 @@ static_assert(BITS_PER_TYPE(access_mask_t) >= LANDLOCK_NUM_ACCESS_FS);
static_assert(BITS_PER_TYPE(access_mask_t) >= LANDLOCK_NUM_ACCESS_NET);
/* Makes sure all scoped rights can be stored. */
static_assert(BITS_PER_TYPE(access_mask_t) >= LANDLOCK_NUM_SCOPE);
+/* Makes sure all permissions can be stored. */
+static_assert(BITS_PER_TYPE(access_mask_t) >= LANDLOCK_NUM_PERMISSION);
/* Makes sure for_each_set_bit() and for_each_clear_bit() calls are OK. */
static_assert(sizeof(unsigned long) >= sizeof(access_mask_t));
-/* Access masks (bitfields only). */
+/* Access and permission masks (bitfields only). */
struct access_masks {
access_mask_t fs : LANDLOCK_NUM_ACCESS_FS;
access_mask_t net : LANDLOCK_NUM_ACCESS_NET;
access_mask_t scope : LANDLOCK_NUM_SCOPE;
+ access_mask_t permissions : LANDLOCK_NUM_PERMISSION;
} __packed __aligned(sizeof(u32));
union access_masks_all {
@@ -61,16 +64,34 @@ union access_masks_all {
static_assert(sizeof(typeof_member(union access_masks_all, masks)) ==
sizeof(typeof_member(union access_masks_all, all)));
+/**
+ * struct permission_masks - Per-permission member bitmasks
+ */
+struct permission_masks {
+ /**
+ * @ns_types: Namespace type member mask, indexed in FOR_EACH_NS_TYPE()
+ * order.
+ */
+ u64 ns_types : LANDLOCK_NUM_NAMESPACE_TYPE;
+} __packed __aligned(sizeof(u64));
+
+static_assert(sizeof(struct permission_masks) == sizeof(u64));
+
/**
* struct layer_config - Per-layer access configuration
*
* A ruleset stores one mutable layer and a domain stores a flexible array of
- * immutable layers.
+ * immutable layers. Unlike filesystem and network access rights, namespace
+ * types use a flat bitmask because their keyspace is small and bounded.
*/
struct layer_config {
/**
- * @handled: Bitmask of access rights handled (i.e. restricted) by this
- * layer.
+ * @allowed: Members allowed by each handled permission.
+ */
+ struct permission_masks allowed;
+ /**
+ * @handled: Bitmask of access rights and permissions handled (i.e.
+ * restricted) by this layer.
*/
struct access_masks handled;
};
diff --git a/security/landlock/audit.c b/security/landlock/audit.c
index e02963834e48..5386f1411ba6 100644
--- a/security/landlock/audit.c
+++ b/security/landlock/audit.c
@@ -21,10 +21,10 @@
#include "log.h"
/*
- * Access-right and scope names are built from the lists shared with the trace
- * events (see <linux/landlock.h>). The designated initializer places each name
- * at its bit index, so the lookup stays O(1) and does not depend on the entry
- * order. log_blockers() adds the "fs."/"net."/"scope." category prefix.
+ * Access-right, scope, and permission names are built from the lists shared
+ * with the trace events (see <linux/landlock.h>). The designated initializer
+ * places each name at its bit index, so the lookup stays O(1) and does not
+ * depend on the entry order. log_blockers() adds the related category prefix.
*/
#define _LANDLOCK_NAME_ENTRY(mask, name) [BIT_INDEX(mask)] = name
@@ -40,6 +40,12 @@ static const char *const scope_strings[] = { _LANDLOCK_SCOPE_NAMES };
static_assert(ARRAY_SIZE(scope_strings) == LANDLOCK_NUM_SCOPE);
+static const char *const permission_strings[] = {
+ _LANDLOCK_PERMISSION_BLOCKER_NAMES
+};
+
+static_assert(ARRAY_SIZE(permission_strings) == LANDLOCK_NUM_PERMISSION);
+
#undef _LANDLOCK_NAME_ENTRY
static __attribute_const__ const char *
@@ -73,6 +79,11 @@ get_blocker(const enum landlock_request_type type,
case LANDLOCK_REQUEST_SCOPE_SIGNAL:
WARN_ON_ONCE(access_bit != -1);
return scope_strings[BIT_INDEX(LANDLOCK_SCOPE_SIGNAL)];
+
+ case LANDLOCK_REQUEST_NAMESPACE:
+ if (WARN_ON_ONCE(access_bit >= ARRAY_SIZE(permission_strings)))
+ return "unknown";
+ return permission_strings[access_bit];
}
WARN_ON_ONCE(1);
@@ -82,8 +93,8 @@ get_blocker(const enum landlock_request_type type,
/*
* Returns the audit category prefix prepended to the unprefixed blocker name
* returned by get_blocker() (filesystem and network access rights,
- * change_topology, and scopes). The ptrace blocker is standalone and carries
- * its full name in get_blocker(), so it uses no prefix.
+ * change_topology, scopes, and permissions). The ptrace blocker is standalone:
+ * its full name comes from get_blocker(), so it uses no prefix.
*/
static __attribute_const__ const char *
blocker_prefix(const enum landlock_request_type type)
@@ -102,6 +113,9 @@ blocker_prefix(const enum landlock_request_type type)
case LANDLOCK_REQUEST_SCOPE_ABSTRACT_UNIX_SOCKET:
case LANDLOCK_REQUEST_SCOPE_SIGNAL:
return "scope.";
+
+ case LANDLOCK_REQUEST_NAMESPACE:
+ return _LANDLOCK_PERMISSION_NAMESPACE_NAME ".";
}
WARN_ON_ONCE(1);
@@ -163,7 +177,7 @@ static void log_domain(struct landlock_hierarchy *const hierarchy)
*
* @request: Detail of the user space request.
* @youngest_denied: The youngest hierarchy node that denied the access.
- * @missing: The set of denied access rights.
+ * @missing: The set of denied access rights or permissions.
* @logged: Whether the denial is selected for logging, as computed by
* landlock_log_denial() (domain policy and quiet rules).
*
diff --git a/security/landlock/domain.c b/security/landlock/domain.c
index 636abcd75aff..4c0b1f0e1db5 100644
--- a/security/landlock/domain.c
+++ b/security/landlock/domain.c
@@ -480,6 +480,7 @@ landlock_merge_ruleset(struct landlock_domain *const parent,
#ifdef CONFIG_SECURITY_LANDLOCK_LOG
new_dom->hierarchy->quiet_access = ruleset->quiet_access;
+ new_dom->hierarchy->quiet_permission = ruleset->quiet_permission;
#endif /* CONFIG_SECURITY_LANDLOCK_LOG */
return no_free_ptr(new_dom);
diff --git a/security/landlock/domain.h b/security/landlock/domain.h
index bdec27c3bc7f..ab3c97c4455a 100644
--- a/security/landlock/domain.h
+++ b/security/landlock/domain.h
@@ -134,6 +134,11 @@ struct landlock_hierarchy {
* logged) if the related object is marked as quiet.
*/
struct access_masks quiet_access;
+ /**
+ * @quiet_permission: Per-member quiet bitmasks for permission types in
+ * this layer.
+ */
+ struct permission_masks quiet_permission;
#endif /* CONFIG_SECURITY_LANDLOCK_LOG */
};
@@ -322,4 +327,60 @@ static inline void landlock_get_domain(struct landlock_domain *const domain)
refcount_inc(&domain->usage);
}
+/**
+ * landlock_permission_is_denied - Check if a permission request is denied
+ *
+ * @domain: The enforced domain.
+ * @permission_bit: The LANDLOCK_PERMISSION_* flag to check. Must have
+ * exactly one bit set.
+ * @request_value: Compact bitmask to look for (e.g. the result of
+ * landlock_ns_type_to_bit()). Must have exactly one bit set.
+ *
+ * Iterate from the youngest layer to the oldest. For each layer that handles
+ * @permission_bit, check whether @request_value is present in the layer's
+ * allowed bitmask. Return on the first (youngest) denying layer.
+ *
+ * Return: The youngest denying layer + 1, or 0 if allowed.
+ */
+static inline size_t
+landlock_permission_is_denied(const struct landlock_domain *const domain,
+ const access_mask_t permission_bit,
+ const u64 request_value)
+{
+ ssize_t layer;
+
+ BUILD_BUG_ON(sizeof(permission_bit) > sizeof(u32));
+
+ if (WARN_ON_ONCE(hweight32(permission_bit) != 1) ||
+ WARN_ON_ONCE(hweight64(request_value) != 1))
+ return domain->num_layers;
+
+ for (layer = domain->num_layers - 1; layer >= 0; layer--) {
+ u64 allowed;
+
+ if (!(domain->layers[layer].handled.permissions &
+ permission_bit))
+ continue;
+
+ /*
+ * Current callers pass only permission types with an explicit
+ * case below. The default catches a new caller missing its
+ * member mask.
+ */
+ switch (permission_bit) {
+ case LANDLOCK_PERMISSION_NAMESPACE_USE:
+ allowed = domain->layers[layer].allowed.ns_types;
+ break;
+ default:
+ WARN_ONCE(1, "Unknown permission %u\n",
+ (unsigned int)permission_bit);
+ return layer + 1;
+ }
+
+ if (!(allowed & request_value))
+ return layer + 1;
+ }
+ return 0;
+}
+
#endif /* _SECURITY_LANDLOCK_DOMAIN_H */
diff --git a/security/landlock/limits.h b/security/landlock/limits.h
index 1a7c5fb8f6fd..767d57799f60 100644
--- a/security/landlock/limits.h
+++ b/security/landlock/limits.h
@@ -12,6 +12,7 @@
#include <linux/bitops.h>
#include <linux/limits.h>
+#include <linux/ns/ns_common_types.h>
#include <uapi/linux/landlock.h>
/* clang-format off */
@@ -31,6 +32,12 @@
#define LANDLOCK_MASK_SCOPE ((LANDLOCK_LAST_SCOPE << 1) - 1)
#define LANDLOCK_NUM_SCOPE __const_hweight64(LANDLOCK_MASK_SCOPE)
+#define LANDLOCK_LAST_PERMISSION LANDLOCK_PERMISSION_NAMESPACE_USE
+#define LANDLOCK_MASK_PERMISSION ((LANDLOCK_LAST_PERMISSION << 1) - 1)
+#define LANDLOCK_NUM_PERMISSION __const_hweight64(LANDLOCK_MASK_PERMISSION)
+
+#define LANDLOCK_NUM_NAMESPACE_TYPE __const_hweight64((u64)CLONE_NS_ALL)
+
#define LANDLOCK_NUM_ACCESS_MAX \
MAX(MAX(LANDLOCK_NUM_ACCESS_FS, LANDLOCK_NUM_ACCESS_NET), LANDLOCK_NUM_SCOPE)
diff --git a/security/landlock/log.c b/security/landlock/log.c
index 3f9ae1bacb23..8cb1dfd0d4ae 100644
--- a/security/landlock/log.c
+++ b/security/landlock/log.c
@@ -17,6 +17,7 @@
#include "domain.h"
#include "limits.h"
#include "log.h"
+#include "ns.h"
#include "ruleset.h"
#include "trace.h"
@@ -382,6 +383,31 @@ static bool is_valid_request(const struct landlock_request *const request)
if (WARN_ON_ONCE(!(!!request->layer_plus_one ^ !!request->access)))
return false;
+ if (WARN_ON_ONCE(request->access && request->permission))
+ return false;
+
+ switch (request->type) {
+ case LANDLOCK_REQUEST_NAMESPACE:
+ if (WARN_ON_ONCE(request->permission !=
+ LANDLOCK_PERMISSION_NAMESPACE_USE) ||
+ WARN_ON_ONCE(request->audit.type != LSM_AUDIT_DATA_NS))
+ return false;
+ break;
+ case LANDLOCK_REQUEST_PTRACE:
+ case LANDLOCK_REQUEST_FS_CHANGE_TOPOLOGY:
+ case LANDLOCK_REQUEST_FS_ACCESS:
+ case LANDLOCK_REQUEST_NET_ACCESS:
+ case LANDLOCK_REQUEST_SCOPE_ABSTRACT_UNIX_SOCKET:
+ case LANDLOCK_REQUEST_SCOPE_SIGNAL:
+ if (WARN_ON_ONCE(request->permission))
+ return false;
+ break;
+ default:
+ WARN_ONCE(1, "Unknown Landlock request type %d\n",
+ request->type);
+ return false;
+ }
+
if (request->access) {
if (WARN_ON_ONCE(!(!!request->layer_masks ^
!!request->all_existing_optional_access)))
@@ -442,8 +468,8 @@ is_denial_quieted(const struct landlock_request *const request,
}
/*
- * Either the object is not quiet, or this is a scope request. We check
- * request->type to distinguish between the two cases.
+ * Per-object quieting did not apply. Check request->type for scope and
+ * permission quieting; ptrace and topology requests are never quiet.
*/
switch (request->type) {
case LANDLOCK_REQUEST_SCOPE_SIGNAL:
@@ -452,6 +478,9 @@ is_denial_quieted(const struct landlock_request *const request,
case LANDLOCK_REQUEST_SCOPE_ABSTRACT_UNIX_SOCKET:
return !!(youngest_denied->quiet_access.scope &
LANDLOCK_SCOPE_ABSTRACT_UNIX_SOCKET);
+ case LANDLOCK_REQUEST_NAMESPACE:
+ return !!(youngest_denied->quiet_permission.ns_types &
+ landlock_ns_type_to_bit(request->audit.u.ns.ns_type));
/*
* Leave LANDLOCK_REQUEST_PTRACE and LANDLOCK_REQUEST_FS_CHANGE_TOPOLOGY
* unhandled for now - they are never quiet.
@@ -508,8 +537,8 @@ void landlock_log_denial(const struct landlock_cred_security *const subject,
if (!is_valid_request(request))
return;
- missing = request->access;
- if (missing) {
+ missing = request->access ? request->access : request->permission;
+ if (request->access) {
/* Gets the nearest domain that denies the request. */
if (request->layer_masks) {
youngest_layer = get_denied_layer(subject->domain,
diff --git a/security/landlock/log.h b/security/landlock/log.h
index faa30e26e42a..a12956cac40e 100644
--- a/security/landlock/log.h
+++ b/security/landlock/log.h
@@ -25,9 +25,11 @@ enum landlock_request_type {
LANDLOCK_REQUEST_NET_ACCESS,
LANDLOCK_REQUEST_SCOPE_ABSTRACT_UNIX_SOCKET,
LANDLOCK_REQUEST_SCOPE_SIGNAL,
+ LANDLOCK_REQUEST_NAMESPACE,
};
struct landlock_blockers {
+ /* Blocking access rights or permissions, as selected by @type. */
access_mask_t access;
enum landlock_request_type type;
};
@@ -58,8 +60,13 @@ struct landlock_signal_trace {
* CONFIG_SECURITY_LANDLOCK_LOG is not set.
*/
struct landlock_request {
- /* Mandatory fields. */
+ /* Mandatory request type. */
enum landlock_request_type type;
+
+ /* Required field for permission requests. */
+ access_mask_t permission;
+
+ /* Mandatory audit context. */
struct common_audit_data audit;
/**
diff --git a/security/landlock/ns.c b/security/landlock/ns.c
new file mode 100644
index 000000000000..cd44c927f83a
--- /dev/null
+++ b/security/landlock/ns.c
@@ -0,0 +1,241 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * Landlock - Namespace hooks
+ *
+ * Copyright © 2026 Cloudflare, Inc.
+ */
+
+#include <linux/bitops.h>
+#include <linux/bug.h>
+#include <linux/lsm_audit.h>
+#include <linux/lsm_hooks.h>
+#include <linux/ns/ns_common_types.h>
+#include <linux/ns_common.h>
+#include <linux/nsproxy.h>
+#include <uapi/linux/landlock.h>
+
+#include "cred.h"
+#include "domain.h"
+#include "limits.h"
+#include "log.h"
+#include "ns.h"
+#include "ruleset.h"
+#include "setup.h"
+
+#define _LANDLOCK_NS_TYPE(type, flag) flag,
+
+static const u32 landlock_namespace_types[] = {
+ /* clang-format off */
+ FOR_EACH_NS_TYPE(_LANDLOCK_NS_TYPE)
+ /* clang-format on */
+};
+
+#undef _LANDLOCK_NS_TYPE
+
+static_assert(ARRAY_SIZE(landlock_namespace_types) ==
+ LANDLOCK_NUM_NAMESPACE_TYPE);
+
+/* Ensures the audit ns_id field can hold ns_common.ns_id without truncation. */
+static_assert(sizeof(((struct common_audit_data *)NULL)->u.ns.ns_id) >=
+ sizeof(((struct ns_common *)NULL)->ns_id));
+
+/**
+ * landlock_ns_type_to_bit - Convert a namespace type to a compact bitmask
+ *
+ * @ns_type: Namespace type (``CLONE_NEW*``).
+ *
+ * Return: The compact bit for @ns_type, or 0 if @ns_type is invalid (with a
+ * warning).
+ */
+u64 landlock_ns_type_to_bit(const u32 ns_type)
+{
+ size_t i;
+
+ for (i = 0; i < ARRAY_SIZE(landlock_namespace_types); i++) {
+ if (landlock_namespace_types[i] == ns_type)
+ return BIT_ULL(i);
+ }
+
+ WARN_ONCE(1, "Unknown namespace type 0x%x\n", ns_type);
+ return 0;
+}
+
+/**
+ * landlock_ns_types_to_bits - Convert namespace types to a compact bitmask
+ *
+ * @ns_types: Bitmask of namespace types (``CLONE_NEW*``).
+ *
+ * Return: The compact bits for all known @ns_types. Warns if unknown bits are
+ * present (callers must pre-mask user input).
+ */
+u64 landlock_ns_types_to_bits(const u64 ns_types)
+{
+ u64 bits = 0;
+ size_t i;
+
+ /* Callers pre-mask (CLONE_NS_ALL); the WARN guards future callers. */
+ WARN_ON_ONCE(ns_types & ~(u64)CLONE_NS_ALL);
+ for (i = 0; i < ARRAY_SIZE(landlock_namespace_types); i++) {
+ if (ns_types & landlock_namespace_types[i])
+ bits |= BIT_ULL(i);
+ }
+ return bits;
+}
+
+static const struct access_masks ns_permission = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+};
+
+/**
+ * check_ns_type - Check namespace use permission
+ *
+ * @ns: The namespace being allocated or installed.
+ *
+ * Shared check for namespace_init (creation via clone(2), unshare(2),
+ * open_tree(2), or fsmount(2)) and namespace_install (use via setns(2)): denies
+ * when the namespace type is not in the domain's allowed set. At allocation
+ * time @ns->ns_id is still zero and is logged as such.
+ *
+ * Return: 0 if allowed, -EPERM if denied.
+ */
+static int check_ns_type(struct ns_common *const ns)
+{
+ const struct landlock_cred_security *subject;
+ size_t denied_layer;
+
+ subject = landlock_get_applicable_subject(current_cred(), ns_permission,
+ NULL);
+ if (!subject)
+ return 0;
+
+ denied_layer = landlock_permission_is_denied(
+ subject->domain, LANDLOCK_PERMISSION_NAMESPACE_USE,
+ landlock_ns_type_to_bit(ns->ns_type));
+ if (!denied_layer)
+ return 0;
+
+ landlock_log_denial(subject,
+ &(struct landlock_request){
+ .type = LANDLOCK_REQUEST_NAMESPACE,
+ .audit.type = LSM_AUDIT_DATA_NS,
+ .permission =
+ LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .audit.u.ns.ns_type = ns->ns_type,
+ .audit.u.ns.ns_id = ns->ns_id,
+ .layer_plus_one = denied_layer,
+ });
+ return -EPERM;
+}
+
+static int hook_namespace_init(struct ns_common *const ns)
+{
+ return check_ns_type(ns);
+}
+
+static int hook_namespace_install(const struct nsset *const nsset,
+ struct ns_common *const ns)
+{
+ return check_ns_type(ns);
+}
+
+static struct security_hook_list landlock_hooks[] __ro_after_init = {
+ LSM_HOOK_INIT(namespace_init, hook_namespace_init),
+ LSM_HOOK_INIT(namespace_install, hook_namespace_install),
+};
+
+__init void landlock_add_ns_hooks(void)
+{
+ security_add_hooks(landlock_hooks, ARRAY_SIZE(landlock_hooks),
+ &landlock_lsmid);
+}
+
+#ifdef CONFIG_SECURITY_LANDLOCK_KUNIT_TEST
+
+#include <kunit/test.h>
+
+static void test_ns_type_to_bit(struct kunit *const test)
+{
+ u64 seen = 0;
+ size_t i;
+
+ for (i = 0; i < ARRAY_SIZE(landlock_namespace_types); i++) {
+ const u64 bit =
+ landlock_ns_type_to_bit(landlock_namespace_types[i]);
+
+ KUNIT_EXPECT_NE(test, 0ULL, bit);
+ KUNIT_EXPECT_EQ(test, 0ULL, seen & bit);
+ seen |= bit;
+ }
+
+ KUNIT_EXPECT_EQ(test, GENMASK_ULL(LANDLOCK_NUM_NAMESPACE_TYPE - 1, 0),
+ seen);
+}
+
+static void test_ns_type_to_bit_unknown(struct kunit *const test)
+{
+ if (!IS_ENABLED(CONFIG_BUG))
+ kunit_skip(test, "requires CONFIG_BUG");
+
+ /* clang-format off */
+ kunit_warning_suppress(test) {
+ /* clang-format on */
+ KUNIT_EXPECT_EQ(test, 0ULL,
+ landlock_ns_type_to_bit(CLONE_THREAD));
+ KUNIT_EXPECT_SUPPRESSED_WARNING_COUNT(test, 1);
+ }
+}
+
+static void test_ns_types_to_bits_all(struct kunit *const test)
+{
+ KUNIT_EXPECT_EQ(test, GENMASK_ULL(LANDLOCK_NUM_NAMESPACE_TYPE - 1, 0),
+ landlock_ns_types_to_bits(CLONE_NS_ALL));
+}
+
+static void test_ns_types_to_bits_single(struct kunit *const test)
+{
+ size_t i;
+
+ for (i = 0; i < ARRAY_SIZE(landlock_namespace_types); i++)
+ KUNIT_EXPECT_EQ(
+ test,
+ landlock_ns_type_to_bit(landlock_namespace_types[i]),
+ landlock_ns_types_to_bits(landlock_namespace_types[i]));
+}
+
+static void test_ns_types_to_bits_unknown(struct kunit *const test)
+{
+ if (!IS_ENABLED(CONFIG_BUG))
+ kunit_skip(test, "requires CONFIG_BUG");
+
+ /* clang-format off */
+ kunit_warning_suppress(test) {
+ /* clang-format on */
+ KUNIT_EXPECT_EQ(test, 0ULL,
+ landlock_ns_types_to_bits(CLONE_THREAD));
+ KUNIT_EXPECT_SUPPRESSED_WARNING_COUNT(test, 1);
+ }
+}
+
+static void test_ns_types_to_bits_zero(struct kunit *const test)
+{
+ KUNIT_EXPECT_EQ(test, 0ULL, landlock_ns_types_to_bits(0));
+}
+
+static struct kunit_case test_cases[] = {
+ KUNIT_CASE(test_ns_type_to_bit),
+ KUNIT_CASE(test_ns_type_to_bit_unknown),
+ KUNIT_CASE(test_ns_types_to_bits_all),
+ KUNIT_CASE(test_ns_types_to_bits_single),
+ KUNIT_CASE(test_ns_types_to_bits_unknown),
+ KUNIT_CASE(test_ns_types_to_bits_zero),
+ {}
+};
+
+static struct kunit_suite test_suite = {
+ .name = "landlock_ns",
+ .test_cases = test_cases,
+};
+
+kunit_test_suite(test_suite);
+
+#endif /* CONFIG_SECURITY_LANDLOCK_KUNIT_TEST */
diff --git a/security/landlock/ns.h b/security/landlock/ns.h
new file mode 100644
index 000000000000..46d3bd17479d
--- /dev/null
+++ b/security/landlock/ns.h
@@ -0,0 +1,18 @@
+/* SPDX-License-Identifier: GPL-2.0-only */
+/*
+ * Landlock - Namespace hooks
+ *
+ * Copyright © 2026 Cloudflare, Inc.
+ */
+
+#ifndef _SECURITY_LANDLOCK_NS_H
+#define _SECURITY_LANDLOCK_NS_H
+
+#include <linux/types.h>
+
+u64 landlock_ns_type_to_bit(u32 ns_type);
+u64 landlock_ns_types_to_bits(u64 ns_types);
+
+__init void landlock_add_ns_hooks(void);
+
+#endif /* _SECURITY_LANDLOCK_NS_H */
diff --git a/security/landlock/ruleset.c b/security/landlock/ruleset.c
index edf9396deac6..11776b138a56 100644
--- a/security/landlock/ruleset.c
+++ b/security/landlock/ruleset.c
@@ -31,15 +31,15 @@
#include <trace/events/landlock.h>
-struct landlock_ruleset *
-landlock_create_ruleset(const access_mask_t fs_access_mask,
- const access_mask_t net_access_mask,
- const access_mask_t scope_mask)
+struct landlock_ruleset *landlock_create_ruleset(
+ const access_mask_t fs_access_mask, const access_mask_t net_access_mask,
+ const access_mask_t scope_mask, const access_mask_t permission_mask)
{
struct landlock_ruleset *new_ruleset;
/* Informs about useless ruleset. */
- if (!fs_access_mask && !net_access_mask && !scope_mask)
+ if (!fs_access_mask && !net_access_mask && !scope_mask &&
+ !permission_mask)
return ERR_PTR(-ENOMSG);
new_ruleset = kzalloc_obj(*new_ruleset, GFP_KERNEL_ACCOUNT);
@@ -66,6 +66,7 @@ landlock_create_ruleset(const access_mask_t fs_access_mask,
WARN_ON_ONCE(fs_access_mask != mask);
new_ruleset->layer.handled.fs |= mask;
}
+
if (net_access_mask) {
const access_mask_t mask = net_access_mask &
LANDLOCK_MASK_ACCESS_NET;
@@ -73,12 +74,22 @@ landlock_create_ruleset(const access_mask_t fs_access_mask,
WARN_ON_ONCE(net_access_mask != mask);
new_ruleset->layer.handled.net |= mask;
}
+
if (scope_mask) {
const access_mask_t mask = scope_mask & LANDLOCK_MASK_SCOPE;
WARN_ON_ONCE(scope_mask != mask);
new_ruleset->layer.handled.scope |= mask;
}
+
+ if (permission_mask) {
+ const access_mask_t mask = permission_mask &
+ LANDLOCK_MASK_PERMISSION;
+
+ WARN_ON_ONCE(permission_mask != mask);
+ new_ruleset->layer.handled.permissions |= mask;
+ }
+
return new_ruleset;
}
diff --git a/security/landlock/ruleset.h b/security/landlock/ruleset.h
index 424055a7af86..3a38d7079dc2 100644
--- a/security/landlock/ruleset.h
+++ b/security/landlock/ruleset.h
@@ -184,19 +184,27 @@ struct landlock_ruleset {
/**
* @quiet_access: Stores the quiet flags for an unmerged ruleset. For a
* merged domain, this is stored in each layer's struct
- * landlock_hierarchy instead.
+ * landlock_hierarchy instead. Its permissions member is unused because
+ * permission quieting is per member rather than per permission.
*/
struct access_masks quiet_access;
+#ifdef CONFIG_SECURITY_LANDLOCK_LOG
+ /**
+ * @quiet_permission: Per-member quiet bitmasks for permission types in
+ * this ruleset. A denied member whose bit is set here is not submitted
+ * to audit when this layer denies it.
+ */
+ struct permission_masks quiet_permission;
+#endif /* CONFIG_SECURITY_LANDLOCK_LOG */
/**
* @layer: Access configuration for this ruleset's single mutable layer.
*/
struct layer_config layer;
};
-struct landlock_ruleset *
-landlock_create_ruleset(const access_mask_t access_mask_fs,
- const access_mask_t access_mask_net,
- const access_mask_t scope_mask);
+struct landlock_ruleset *landlock_create_ruleset(
+ const access_mask_t access_mask_fs, const access_mask_t access_mask_net,
+ const access_mask_t scope_mask, const access_mask_t permission_mask);
void landlock_put_ruleset(struct landlock_ruleset *const ruleset);
@@ -247,4 +255,10 @@ static inline void landlock_get_ruleset(struct landlock_ruleset *const ruleset)
refcount_inc(&ruleset->usage);
}
+static inline access_mask_t
+landlock_get_permission_mask(const struct landlock_ruleset *const ruleset)
+{
+ return ruleset->layer.handled.permissions;
+}
+
#endif /* _SECURITY_LANDLOCK_RULESET_H */
diff --git a/security/landlock/setup.c b/security/landlock/setup.c
index 47dac1736f10..a7ed776b41b4 100644
--- a/security/landlock/setup.c
+++ b/security/landlock/setup.c
@@ -17,6 +17,7 @@
#include "fs.h"
#include "id.h"
#include "net.h"
+#include "ns.h"
#include "setup.h"
#include "task.h"
@@ -68,6 +69,7 @@ static int __init landlock_init(void)
landlock_add_task_hooks();
landlock_add_fs_hooks();
landlock_add_net_hooks();
+ landlock_add_ns_hooks();
landlock_init_id();
landlock_initialized = true;
pr_info("Up and running.\n");
diff --git a/security/landlock/syscalls.c b/security/landlock/syscalls.c
index ec616d198184..c37edcc478bc 100644
--- a/security/landlock/syscalls.c
+++ b/security/landlock/syscalls.c
@@ -20,6 +20,7 @@
#include <linux/fs.h>
#include <linux/limits.h>
#include <linux/mount.h>
+#include <linux/ns/ns_common_types.h>
#include <linux/path.h>
#include <linux/sched.h>
#include <linux/sched/signal.h>
@@ -35,6 +36,7 @@
#include "fs.h"
#include "limits.h"
#include "net.h"
+#include "ns.h"
#include "ruleset.h"
#include "setup.h"
#include "tsync.h"
@@ -98,7 +100,9 @@ static void build_check_abi(void)
struct landlock_ruleset_attr ruleset_attr;
struct landlock_path_beneath_attr path_beneath_attr;
struct landlock_net_port_attr net_port_attr;
+ struct landlock_namespace_attr namespace_attr;
size_t ruleset_size, path_beneath_size, net_port_size;
+ size_t namespace_size;
/*
* For each user space ABI structures, first checks that there is no
@@ -111,8 +115,9 @@ static void build_check_abi(void)
ruleset_size += sizeof(ruleset_attr.quiet_access_fs);
ruleset_size += sizeof(ruleset_attr.quiet_access_net);
ruleset_size += sizeof(ruleset_attr.quiet_scoped);
+ ruleset_size += sizeof(ruleset_attr.handled_permissions);
BUILD_BUG_ON(sizeof(ruleset_attr) != ruleset_size);
- BUILD_BUG_ON(sizeof(ruleset_attr) != 48);
+ BUILD_BUG_ON(sizeof(ruleset_attr) != 56);
path_beneath_size = sizeof(path_beneath_attr.allowed_access);
path_beneath_size += sizeof(path_beneath_attr.parent_fd);
@@ -123,6 +128,12 @@ static void build_check_abi(void)
net_port_size += sizeof(net_port_attr.port);
BUILD_BUG_ON(sizeof(net_port_attr) != net_port_size);
BUILD_BUG_ON(sizeof(net_port_attr) != 16);
+
+ namespace_size = sizeof(namespace_attr.permissions);
+ namespace_size += sizeof(namespace_attr.allowed_namespace_types);
+ namespace_size += sizeof(namespace_attr.quiet_namespace_types);
+ BUILD_BUG_ON(sizeof(namespace_attr) != namespace_size);
+ BUILD_BUG_ON(sizeof(namespace_attr) != 24);
}
/* Ruleset handling */
@@ -172,7 +183,7 @@ static const struct file_operations ruleset_fops = {
* If the change involves a fix that requires userspace awareness, also update
* the errata documentation in Documentation/userspace-api/landlock.rst .
*/
-const int landlock_abi_version = 11;
+const int landlock_abi_version = 12;
/**
* sys_landlock_create_ruleset - Create a new ruleset
@@ -197,14 +208,13 @@ const int landlock_abi_version = 11;
* returned errors are:
*
* - %EOPNOTSUPP: Landlock is supported by the kernel but disabled at boot time;
- * - %EINVAL: unknown @flags, or unknown access, or unknown scope, or too small
- * @size;
+ * - %EINVAL: unknown @flags, access, scope, or permission, or too small @size;
* - %EINVAL: quiet_access_fs, quiet_access_net, or quiet_scoped is not a
* subset of the corresponding handled_access_fs, handled_access_net, or
* scoped;
* - %E2BIG: @attr or @size inconsistencies;
* - %EFAULT: @attr or @size inconsistencies;
- * - %ENOMSG: empty &landlock_ruleset_attr.handled_access_fs.
+ * - %ENOMSG: all handled access, scope, and permission fields are empty.
*
* .. kernel-doc:: include/uapi/linux/landlock.h
* :identifiers: landlock_create_ruleset_flags
@@ -273,10 +283,16 @@ SYSCALL_DEFINE3(landlock_create_ruleset,
ruleset_attr.scoped)
return -EINVAL;
+ /* Checks permission content (and 32-bits cast). */
+ if ((ruleset_attr.handled_permissions | LANDLOCK_MASK_PERMISSION) !=
+ LANDLOCK_MASK_PERMISSION)
+ return -EINVAL;
+
/* Checks arguments and transforms to kernel struct. */
ruleset = landlock_create_ruleset(ruleset_attr.handled_access_fs,
ruleset_attr.handled_access_net,
- ruleset_attr.scoped);
+ ruleset_attr.scoped,
+ ruleset_attr.handled_permissions);
if (IS_ERR(ruleset))
return PTR_ERR(ruleset);
@@ -435,13 +451,90 @@ static int add_rule_net_port(struct landlock_ruleset *ruleset,
net_port_attr.allowed_access, flags);
}
+static int add_rule_namespace(struct landlock_ruleset *const ruleset,
+ const void __user *const rule_attr,
+ const u32 flags)
+{
+ struct landlock_namespace_attr ns_attr;
+ access_mask_t mask;
+ u64 allowed_types, quiet_types;
+ int ret;
+
+ /*
+ * Namespace rules support no add-rule flags. In particular,
+ * LANDLOCK_ADD_RULE_QUIET is filesystem/network only.
+ */
+ if (flags)
+ return -EINVAL;
+
+ /* Copies raw user space buffer. */
+ ret = copy_from_user(&ns_attr, rule_attr, sizeof(ns_attr));
+ if (ret)
+ return -EFAULT;
+
+ /* Informs about useless rule: empty permissions. */
+ if (!ns_attr.permissions)
+ return -ENOMSG;
+
+ /*
+ * The permissions selector must match
+ * LANDLOCK_PERMISSION_NAMESPACE_USE. The valid set is a single bit
+ * today, so this is an exact match now; the check broadens to a subset
+ * test once another supported permission is added.
+ */
+ if (ns_attr.permissions != LANDLOCK_PERMISSION_NAMESPACE_USE)
+ return -EINVAL;
+
+ /*
+ * Checks that permissions match the ruleset constraints. This also
+ * makes quieting require the category to be handled.
+ */
+ mask = landlock_get_permission_mask(ruleset);
+ if (!(mask & LANDLOCK_PERMISSION_NAMESPACE_USE))
+ return -EINVAL;
+
+ /*
+ * Informs about useless rule: neither allows nor quiets anything. A
+ * quiet-only rule (empty allowed set) is legal.
+ */
+ if (!ns_attr.allowed_namespace_types && !ns_attr.quiet_namespace_types)
+ return -ENOMSG;
+
+ /*
+ * Stores only the namespace types this kernel knows about. Unknown
+ * bits are silently accepted for forward compatibility: user space
+ * compiled against newer headers can pass new CLONE_NEW* flags without
+ * getting EINVAL on older kernels. Unknown bits have no effect because
+ * no hook checks them. The quiet bitmask suppresses logging of denials
+ * attributed to this layer; see landlock_log_denial().
+ */
+ allowed_types = ns_attr.allowed_namespace_types & CLONE_NS_ALL;
+ quiet_types = ns_attr.quiet_namespace_types & CLONE_NS_ALL;
+
+ mutex_lock(&ruleset->lock);
+ ruleset->layer.allowed.ns_types |=
+ landlock_ns_types_to_bits(allowed_types);
+#ifdef CONFIG_SECURITY_LANDLOCK_LOG
+ ruleset->quiet_permission.ns_types |=
+ landlock_ns_types_to_bits(quiet_types);
+#endif /* CONFIG_SECURITY_LANDLOCK_LOG */
+#ifdef CONFIG_TRACEPOINTS
+ ruleset->version++;
+#endif /* CONFIG_TRACEPOINTS */
+ trace_landlock_add_rule_namespace(ruleset, flags, ns_attr.permissions,
+ allowed_types, quiet_types);
+ mutex_unlock(&ruleset->lock);
+ return 0;
+}
+
/**
* sys_landlock_add_rule - Add a new rule to a ruleset
*
* @ruleset_fd: File descriptor tied to the ruleset that should be extended
* with the new rule.
* @rule_type: Identify the structure type pointed to by @rule_attr:
- * %LANDLOCK_RULE_PATH_BENEATH or %LANDLOCK_RULE_NET_PORT.
+ * %LANDLOCK_RULE_PATH_BENEATH, %LANDLOCK_RULE_NET_PORT, or
+ * %LANDLOCK_RULE_NAMESPACE.
* @rule_attr: Pointer to a rule (matching the @rule_type).
* @flags: Must be 0 or %LANDLOCK_ADD_RULE_QUIET.
*
@@ -458,11 +551,16 @@ static int add_rule_net_port(struct landlock_ruleset *ruleset,
* &landlock_path_beneath_attr.allowed_access or
* &landlock_net_port_attr.allowed_access is not a subset of the ruleset
* handled accesses)
+ * - %EINVAL: A nonzero &landlock_namespace_attr.permissions is not
+ * %LANDLOCK_PERMISSION_NAMESPACE_USE or is not handled by the ruleset;
* - %EINVAL: &landlock_net_port_attr.port is greater than 65535;
* - %EINVAL: LANDLOCK_ADD_RULE_QUIET is passed but the ruleset has no
* quiet access bits set for the corresponding rule type.
* - %ENOMSG: Empty accesses (e.g. &landlock_path_beneath_attr.allowed_access is
* 0) and no flags;
+ * - %ENOMSG: &landlock_namespace_attr.permissions is 0, or both
+ * &landlock_namespace_attr.allowed_namespace_types and
+ * &landlock_namespace_attr.quiet_namespace_types are 0;
* - %EBADF: @ruleset_fd is not a file descriptor for the current thread, or a
* member of @rule_attr is not a file descriptor as expected;
* - %EBADFD: @ruleset_fd is not a ruleset file descriptor, or a member of
@@ -495,6 +593,8 @@ SYSCALL_DEFINE4(landlock_add_rule, const int, ruleset_fd,
return add_rule_path_beneath(ruleset, rule_attr, flags);
case LANDLOCK_RULE_NET_PORT:
return add_rule_net_port(ruleset, rule_attr, flags);
+ case LANDLOCK_RULE_NAMESPACE:
+ return add_rule_namespace(ruleset, rule_attr, flags);
default:
return -EINVAL;
}
diff --git a/security/landlock/trace.c b/security/landlock/trace.c
index 225dbf37bab0..300afcc82220 100644
--- a/security/landlock/trace.c
+++ b/security/landlock/trace.c
@@ -62,7 +62,7 @@ void landlock_trace_free_domain(const struct landlock_hierarchy *const hierarchy
*
* @request: Detail of the user space request.
* @youngest_denied: The youngest hierarchy node that denied the access.
- * @missing: The final missing access subset, when applicable.
+ * @missing: The final missing access or permission subset, when applicable.
* @same_exec: Whether the policy subject is the same executable that called
* landlock_restrict_self() for the denying domain, as computed
* by landlock_log_denial().
@@ -81,6 +81,19 @@ void landlock_trace_denial(
const access_mask_t missing, const bool same_exec, const bool logged)
{
switch (request->type) {
+ case LANDLOCK_REQUEST_NAMESPACE:
+ if (trace_landlock_deny_permission_namespace_enabled()) {
+ const struct landlock_blockers blockers = {
+ .access = missing,
+ .type = request->type,
+ };
+
+ trace_landlock_deny_permission_namespace(
+ youngest_denied, same_exec, logged, &blockers,
+ request->audit.u.ns.ns_type,
+ request->audit.u.ns.ns_id);
+ }
+ break;
case LANDLOCK_REQUEST_FS_ACCESS:
case LANDLOCK_REQUEST_FS_CHANGE_TOPOLOGY:
if (trace_landlock_deny_access_fs_enabled()) {
diff --git a/tools/testing/selftests/landlock/base_test.c b/tools/testing/selftests/landlock/base_test.c
index d20ab8f0862c..58fe322d8637 100644
--- a/tools/testing/selftests/landlock/base_test.c
+++ b/tools/testing/selftests/landlock/base_test.c
@@ -76,7 +76,7 @@ TEST(abi_version)
const struct landlock_ruleset_attr ruleset_attr = {
.handled_access_fs = LANDLOCK_ACCESS_FS_READ_FILE,
};
- ASSERT_EQ(11, landlock_create_ruleset(NULL, 0,
+ ASSERT_EQ(12, landlock_create_ruleset(NULL, 0,
LANDLOCK_CREATE_RULESET_VERSION));
ASSERT_EQ(-1, landlock_create_ruleset(&ruleset_attr, 0,
diff --git a/tools/testing/selftests/landlock/trace.h b/tools/testing/selftests/landlock/trace.h
index 2ec863362173..fe2d1a0d8de3 100644
--- a/tools/testing/selftests/landlock/trace.h
+++ b/tools/testing/selftests/landlock/trace.h
@@ -101,7 +101,8 @@
"ruleset=[0-9a-f]\\+\\.[0-9]\\+ " \
"handled_fs=[a-z_|]* " \
"handled_net=[a-z_|]* " \
- "scoped=[a-z_|]*$"
+ "scoped=[a-z_|]* " \
+ "handled_permissions=[a-z._|]*$"
#define REGEX_CREATE_DOMAIN(task) \
TRACE_PREFIX(task) \
diff --git a/tools/testing/selftests/landlock/trace_test.c b/tools/testing/selftests/landlock/trace_test.c
index f9b293a9dd56..cb49ce8dbeda 100644
--- a/tools/testing/selftests/landlock/trace_test.c
+++ b/tools/testing/selftests/landlock/trace_test.c
@@ -148,7 +148,7 @@ TEST_F(trace, no_trace_when_disabled)
/*
* Verifies that landlock_create_ruleset emits a trace event with the correct
- * handled access masks.
+ * handled access and permission masks.
*/
TEST_F(trace, create_ruleset)
{
@@ -186,6 +186,12 @@ TEST_F(trace, create_ruleset)
"handled_net", field, sizeof(field)));
EXPECT_STREQ("bind_tcp", field);
+ /* Verify that no permission is handled. */
+ EXPECT_EQ(0, tracefs_extract_field(
+ buf, REGEX_CREATE_RULESET(TRACE_TASK),
+ "handled_permissions", field, sizeof(field)));
+ EXPECT_STREQ("", field);
+
/* Verify version is 0 at creation (no rules added yet). */
EXPECT_EQ(0,
tracefs_extract_field(buf, REGEX_CREATE_RULESET(TRACE_TASK),
--
2.55.0
^ permalink raw reply [flat|nested] 9+ messages in thread* [PATCH v4 4/8] landlock: Enforce capability restrictions
2026-10-02 12:43 [PATCH v4 0/8] Landlock: Namespace and capability control Mickaël Salaün
` (2 preceding siblings ...)
2026-10-02 12:43 ` [PATCH v4 3/8] landlock: Enforce namespace use restrictions Mickaël Salaün
@ 2026-10-02 12:43 ` Mickaël Salaün
2026-10-02 12:43 ` [PATCH v4 5/8] selftests/landlock: Add namespace restriction tests Mickaël Salaün
` (3 subsequent siblings)
7 siblings, 0 replies; 9+ messages in thread
From: Mickaël Salaün @ 2026-10-02 12:43 UTC (permalink / raw)
To: Christian Brauner, Günther Noack, Paul Moore, Serge E . Hallyn
Cc: Mickaël Salaün, Daniel Durning, Jonathan Corbet,
Justin Suess, Lennart Poettering, Mikhail Ivanov,
Nicolas Bouchinet, Shervin Oloumi, Tingmao Wang, kernel-team,
linux-fsdevel, linux-kernel, linux-security-module
Add Landlock enforcement for capability use via the LSM capable hook.
This lets a sandboxed process restrict which Linux capabilities it can
exercise, using LANDLOCK_PERMISSION_CAPABILITY_USE and per-capability
rules.
The check is a flat per-layer allowed-capabilities test, with no
domain-ancestry bypass, no cross-namespace discriminant, and no
dependency on the target user namespace. These categorical denials
return -EPERM, like the namespace permission introduced by the previous
commit, and they mirror its per-capability allowed and quiet masks, so
LANDLOCK_ADD_RULE_QUIET stays rejected for this rule type. Successful
capability rules and denials are traced as the previous commit
describes.
Enforce only at capability exercise time rather than modifying the
credential's capability sets. Modifying them would give capget(2) an
accurate view of usable capabilities, but no LSM other than commoncap
does so; Landlock follows that convention. A sandboxed process inside a
user namespace therefore sees all capabilities via capget(2) but
receives -EPERM when attempting to use one that is denied.
An earlier design bypassed this hook for namespace management through a
domain-ancestry comparison with the namespace creator's stored Landlock
domain. That made a Landlock decision depend on kernel namespace state,
and the ns != cred->user_ns heuristic did not accurately identify
namespace-management operations. The flat, namespace-agnostic check
instead enforces capability and namespace restrictions independently;
creating a non-user namespace requires an allowed CAP_SYS_ADMIN even
when combined with CLONE_NEWUSER in one unshare().
Cc: Christian Brauner <brauner@kernel.org>
Cc: Günther Noack <gnoack@google.com>
Cc: Paul Moore <paul@paul-moore.com>
Cc: Serge E. Hallyn <serge@hallyn.com>
Signed-off-by: Mickaël Salaün <mic@digikod.net>
---
Changes since v3:
https://patch.msgid.link/20260726161400.3010511-9-mic@digikod.net
- Adapt the capability mask to the ruleset/domain split.
- Spell out perm as permission in the UAPI constant and members, the
internal masks and helpers, and the trace fields, and name the audit
blocker capability.use after its domain instead of
perm.capability_use.
- Suppress and assert the expected warnings in the invalid-input KUnit
tests.
- Drop __attribute_const__ from the conversion helpers, which warn on
invalid input.
- Trace successful capability rules and denials. Successful effective
no-ops advance the version, while quiet denials remain trace-visible
with logged=0.
- Tell programs to omit empty rules, complete the rule errors, and
explain target-user-namespace independence and the -EPERM convention.
- Drop Reviewed-by: Tingmao Wang: this version folds the capability
trace events into this patch, which her v3 review did not cover; a
fresh review is welcome.
Changes since v2:
https://patch.msgid.link/20260527181127.879771-6-mic@digikod.net
- Rename the capability rule attribute fields (allowed_perm to perm,
capabilities to allowed_capabilities) and add a quiet_capabilities
bitmask that suppresses the audit records of specific denied
capabilities, consuming the shared per-layer quiet member mask in
landlock_log_denial(); the rule attribute grows from 16 to 24 bytes.
- Dropped Reviewed-by: Günther Noack and Tingmao Wang, as this version
adds the quiet member mask described above, which their v1 review did
not cover. Fresh review welcome.
Changes since v1:
https://patch.msgid.link/20260312100444.2609563-7-mic@digikod.net
- Add Reviewed-by: Tingmao Wang.
- Rename internal struct perm_rules to perm_masks (companion change
to the preceding commit).
- Rename LANDLOCK_PERM_NAMESPACE_ENTER references to
LANDLOCK_PERM_NAMESPACE_USE (companion change to the introducing
commit).
- Rename struct layer_rights to struct layer_config (companion
change to the introducing commit).
- Clarify in the commit body and hook_capable() kdoc that commoncap
(not Landlock) is registered with LSM_ORDER_FIRST.
- Surface the empty-check semantics in the
landlock_capability_attr.capabilities kdoc: a rule that sets only
bits unknown to the running kernel (above CAP_LAST_CAP) succeeds
but has no runtime effect.
- Add explicit static_assert that LANDLOCK_NUM_PERM_CAP +
LANDLOCK_NUM_PERM_NS fits in a u64, complementing the existing
implicit sizeof guard on struct perm_masks.
- Add Reviewed-by: Günther Noack.
---
include/linux/landlock.h | 5 +-
include/trace/events/landlock.h | 96 +++++++++++++++++++
include/uapi/linux/landlock.h | 49 ++++++++++
security/landlock/Makefile | 3 +-
security/landlock/access.h | 10 +-
security/landlock/audit.c | 4 +
security/landlock/cap.c | 163 ++++++++++++++++++++++++++++++++
security/landlock/cap.h | 48 ++++++++++
security/landlock/domain.h | 3 +
security/landlock/limits.h | 4 +-
security/landlock/log.c | 10 ++
security/landlock/log.h | 1 +
security/landlock/setup.c | 2 +
security/landlock/syscalls.c | 95 ++++++++++++++++++-
security/landlock/trace.c | 12 +++
15 files changed, 498 insertions(+), 7 deletions(-)
create mode 100644 security/landlock/cap.c
create mode 100644 security/landlock/cap.h
diff --git a/include/linux/landlock.h b/include/linux/landlock.h
index d288e3b6756f..0ee4880988ba 100644
--- a/include/linux/landlock.h
+++ b/include/linux/landlock.h
@@ -56,10 +56,13 @@
_LANDLOCK_NAME_ENTRY(LANDLOCK_SCOPE_SIGNAL, "signal")
#define _LANDLOCK_PERMISSION_NAMESPACE_NAME "namespace"
+#define _LANDLOCK_PERMISSION_CAPABILITY_NAME "capability"
#define _LANDLOCK_PERMISSION_LIST(entry) \
entry(LANDLOCK_PERMISSION_NAMESPACE_USE, "use", \
- _LANDLOCK_PERMISSION_NAMESPACE_NAME)
+ _LANDLOCK_PERMISSION_NAMESPACE_NAME), \
+ entry(LANDLOCK_PERMISSION_CAPABILITY_USE, "use", \
+ _LANDLOCK_PERMISSION_CAPABILITY_NAME)
#define _LANDLOCK_PERMISSION_QUALIFIED_ENTRY(mask, action, domain) \
_LANDLOCK_NAME_ENTRY(mask, domain "." action)
diff --git a/include/trace/events/landlock.h b/include/trace/events/landlock.h
index d5d08f751a53..c13475d5180a 100644
--- a/include/trace/events/landlock.h
+++ b/include/trace/events/landlock.h
@@ -37,6 +37,7 @@ TRACE_DEFINE_ENUM(LANDLOCK_REQUEST_FS_CHANGE_TOPOLOGY);
TRACE_DEFINE_ENUM(LANDLOCK_REQUEST_FS_ACCESS);
TRACE_DEFINE_ENUM(LANDLOCK_REQUEST_NET_ACCESS);
TRACE_DEFINE_ENUM(LANDLOCK_REQUEST_NAMESPACE);
+TRACE_DEFINE_ENUM(LANDLOCK_REQUEST_CAPABILITY);
#ifdef CREATE_TRACE_POINTS
@@ -552,6 +553,53 @@ TRACE_EVENT(landlock_add_rule_namespace,
__entry->quiet_namespace_types)
);
+/**
+ * landlock_add_rule_capability - Capability rule added to a ruleset
+ *
+ * @ruleset: Source ruleset (never NULL).
+ * @flags: Complete validated landlock_add_rule_flags value supplied by this
+ * successful call, not the rule's accumulated quiet state.
+ * @permissions: Validated permission mask from the rule attribute.
+ * @allowed_capabilities: Effective known capabilities allowed by this call.
+ * @quiet_capabilities: Effective known capabilities quieted by this call.
+ *
+ * Emitted by sys_landlock_add_rule() under the modified ruleset's lock, so
+ * the reported ruleset is a stable snapshot that no concurrent writer can
+ * change.
+ */
+TRACE_EVENT(landlock_add_rule_capability,
+
+ TP_PROTO(const struct landlock_ruleset *ruleset, u32 flags,
+ u64 permissions, u64 allowed_capabilities,
+ u64 quiet_capabilities),
+
+ TP_ARGS(ruleset, flags, permissions, allowed_capabilities,
+ quiet_capabilities),
+
+ TP_STRUCT__entry(
+ __field( u64, ruleset_id )
+ __field( u64, ruleset_version )
+ __field( access_mask_t, permissions )
+ __field( u64, allowed_capabilities )
+ __field( u64, quiet_capabilities )
+ ),
+
+ TP_fast_assign(
+ lockdep_assert_held(&ruleset->lock);
+ __entry->ruleset_id = ruleset->id;
+ __entry->ruleset_version = ruleset->version;
+ __entry->permissions = permissions;
+ __entry->allowed_capabilities = allowed_capabilities;
+ __entry->quiet_capabilities = quiet_capabilities;
+ ),
+
+ TP_printk("ruleset=%llx.%llu permissions=%s allowed_capabilities=0x%llx quiet_capabilities=0x%llx",
+ __entry->ruleset_id, __entry->ruleset_version,
+ __print_flags(__entry->permissions, "|",
+ _LANDLOCK_PERMISSION_NAMES),
+ __entry->allowed_capabilities, __entry->quiet_capabilities)
+);
+
/**
* landlock_create_domain - New domain created
*
@@ -980,6 +1028,54 @@ TRACE_EVENT(landlock_deny_permission_namespace,
__entry->namespace_type, __entry->namespace_id)
);
+/**
+ * landlock_deny_permission_capability - Capability use denied
+ *
+ * @hierarchy: Denying domain's hierarchy node (never NULL); its id is the
+ * domain field.
+ * @same_exec: Whether the current task entered the denying domain itself.
+ * @logged: Whether this denial was selected for audit logging.
+ * @blockers: Request type and final missing permission subset (never NULL).
+ * @capability: CAP_* number that was denied.
+ *
+ * Emitted when a Landlock domain denies capability use, except for checks
+ * made with CAP_OPT_NOAUDIT, which are denied without emitting this event.
+ */
+TRACE_EVENT(landlock_deny_permission_capability,
+
+ TP_PROTO(const struct landlock_hierarchy *hierarchy, bool same_exec,
+ bool logged, const struct landlock_blockers *blockers,
+ int capability),
+
+ TP_ARGS(hierarchy, same_exec, logged, blockers, capability),
+
+ TP_STRUCT__entry(
+ __field( u64, domain_id )
+ __field( bool, same_exec )
+ __field( bool, logged )
+ __field( enum landlock_request_type, blockers_type )
+ __field( access_mask_t, blockers_access )
+ __field( int, capability )
+ ),
+
+ TP_fast_assign(
+ __entry->domain_id = hierarchy->id;
+ __entry->same_exec = same_exec;
+ __entry->logged = logged;
+ __entry->blockers_type = blockers->type;
+ __entry->blockers_access = blockers->access;
+ __entry->capability = capability;
+ ),
+
+ TP_printk("domain=%llx same_exec=%d logged=%d blockers=%s capability=%d",
+ __entry->domain_id, __entry->same_exec, __entry->logged,
+ __entry->blockers_type == LANDLOCK_REQUEST_CAPABILITY ?
+ __print_flags(__entry->blockers_access, "|",
+ _LANDLOCK_PERMISSION_BLOCKER_NAMES) :
+ "unknown",
+ __entry->capability)
+);
+
/**
* landlock_deny_ptrace - Ptrace access denied by a Landlock domain
*
diff --git a/include/uapi/linux/landlock.h b/include/uapi/linux/landlock.h
index bb8ec589ddac..3128e58d4b78 100644
--- a/include/uapi/linux/landlock.h
+++ b/include/uapi/linux/landlock.h
@@ -237,6 +237,11 @@ enum landlock_rule_type {
* @LANDLOCK_RULE_NAMESPACE: Type of a &struct landlock_namespace_attr .
*/
LANDLOCK_RULE_NAMESPACE,
+ /**
+ * @LANDLOCK_RULE_CAPABILITY: Type of a &struct
+ * landlock_capability_attr .
+ */
+ LANDLOCK_RULE_CAPABILITY,
};
/**
@@ -325,6 +330,42 @@ struct landlock_namespace_attr {
__u64 quiet_namespace_types;
};
+/**
+ * struct landlock_capability_attr - Capability definition
+ *
+ * Argument of sys_landlock_add_rule() with %LANDLOCK_RULE_CAPABILITY.
+ */
+struct landlock_capability_attr {
+ /**
+ * @permissions: Must be set to %LANDLOCK_PERMISSION_CAPABILITY_USE.
+ */
+ __u64 permissions;
+ /**
+ * @allowed_capabilities: Bitmask of capabilities (``1ULL << CAP_*``) to
+ * allow under this rule. Bits above ``CAP_LAST_CAP`` are silently
+ * ignored for forward compatibility.
+ */
+ __u64 allowed_capabilities;
+ /**
+ * @quiet_capabilities: Bitmask of capabilities (``1ULL << CAP_*``)
+ * whose denial by this layer should not be submitted to audit, even if
+ * audit logging would normally take place per landlock_restrict_self()
+ * flags. Only audit records attributed to this layer are suppressed;
+ * denial tracepoints still fire (see `Permission flags`_). Bits also
+ * set in @allowed_capabilities have no effect, since an allowed
+ * capability is never denied. Bits above ``CAP_LAST_CAP`` are silently
+ * ignored.
+ *
+ * At least one of @allowed_capabilities or @quiet_capabilities must be
+ * non-zero, otherwise the call returns ``-ENOMSG``. The non-zero check
+ * runs on the raw input before unknown-bit masking, so a rule that sets
+ * only bits unknown to the running kernel (above ``CAP_LAST_CAP``)
+ * succeeds but has no runtime effect. Programs should omit this rule
+ * when they neither allow nor quiet a capability.
+ */
+ __u64 quiet_capabilities;
+};
+
/**
* DOC: fs_access
*
@@ -577,7 +618,15 @@ struct landlock_namespace_attr {
* Landlock domain that handles this permission is denied from using namespace
* types that are not explicitly allowed by a %LANDLOCK_RULE_NAMESPACE rule.
* Support added in Landlock ABI version 12.
+ * - %LANDLOCK_PERMISSION_CAPABILITY_USE: Restrict the use of specific Linux
+ * capabilities. A process in a Landlock domain that handles this permission
+ * is denied from exercising capabilities that are not explicitly allowed by a
+ * %LANDLOCK_RULE_CAPABILITY rule. This hook is purely restrictive: it can
+ * deny capabilities that the kernel would otherwise grant, but it can never
+ * grant capabilities that the kernel already denied. Support added in
+ * Landlock ABI version 12.
*/
#define LANDLOCK_PERMISSION_NAMESPACE_USE (1ULL << 0)
+#define LANDLOCK_PERMISSION_CAPABILITY_USE (1ULL << 1)
#endif /* _UAPI_LINUX_LANDLOCK_H */
diff --git a/security/landlock/Makefile b/security/landlock/Makefile
index e88ca842c782..97cf668db165 100644
--- a/security/landlock/Makefile
+++ b/security/landlock/Makefile
@@ -10,7 +10,8 @@ landlock-y := \
fs.o \
tsync.o \
domain.o \
- ns.o
+ ns.o \
+ cap.o
landlock-$(CONFIG_INET) += net.o
diff --git a/security/landlock/access.h b/security/landlock/access.h
index e3dbaa29a29f..f9eaa2f53686 100644
--- a/security/landlock/access.h
+++ b/security/landlock/access.h
@@ -73,16 +73,24 @@ struct permission_masks {
* order.
*/
u64 ns_types : LANDLOCK_NUM_NAMESPACE_TYPE;
+ /**
+ * @caps: Capability member mask, indexed by CAP_* values.
+ */
+ u64 caps : LANDLOCK_NUM_CAPABILITY;
} __packed __aligned(sizeof(u64));
static_assert(sizeof(struct permission_masks) == sizeof(u64));
+/* All permission_masks bitfields must fit in a single u64. */
+static_assert(LANDLOCK_NUM_CAPABILITY + LANDLOCK_NUM_NAMESPACE_TYPE <=
+ BITS_PER_TYPE(u64));
/**
* struct layer_config - Per-layer access configuration
*
* A ruleset stores one mutable layer and a domain stores a flexible array of
* immutable layers. Unlike filesystem and network access rights, namespace
- * types use a flat bitmask because their keyspace is small and bounded.
+ * types and capabilities use flat bitmasks because their keyspaces are small
+ * and bounded.
*/
struct layer_config {
/**
diff --git a/security/landlock/audit.c b/security/landlock/audit.c
index 5386f1411ba6..be49ef47b4ea 100644
--- a/security/landlock/audit.c
+++ b/security/landlock/audit.c
@@ -81,6 +81,7 @@ get_blocker(const enum landlock_request_type type,
return scope_strings[BIT_INDEX(LANDLOCK_SCOPE_SIGNAL)];
case LANDLOCK_REQUEST_NAMESPACE:
+ case LANDLOCK_REQUEST_CAPABILITY:
if (WARN_ON_ONCE(access_bit >= ARRAY_SIZE(permission_strings)))
return "unknown";
return permission_strings[access_bit];
@@ -116,6 +117,9 @@ blocker_prefix(const enum landlock_request_type type)
case LANDLOCK_REQUEST_NAMESPACE:
return _LANDLOCK_PERMISSION_NAMESPACE_NAME ".";
+
+ case LANDLOCK_REQUEST_CAPABILITY:
+ return _LANDLOCK_PERMISSION_CAPABILITY_NAME ".";
}
WARN_ON_ONCE(1);
diff --git a/security/landlock/cap.c b/security/landlock/cap.c
new file mode 100644
index 000000000000..5b41589ac24e
--- /dev/null
+++ b/security/landlock/cap.c
@@ -0,0 +1,163 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * Landlock - Capability hooks
+ *
+ * Copyright © 2026 Cloudflare, Inc.
+ */
+
+#include <linux/capability.h>
+#include <linux/cred.h>
+#include <linux/lsm_audit.h>
+#include <linux/lsm_hooks.h>
+#include <uapi/linux/landlock.h>
+
+#include "cap.h"
+#include "cred.h"
+#include "domain.h"
+#include "limits.h"
+#include "log.h"
+#include "ruleset.h"
+#include "setup.h"
+
+static const struct access_masks cap_permission = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+};
+
+/**
+ * hook_capable - Deny capability use for Landlock-sandboxed processes
+ *
+ * @cred: Credentials being checked.
+ * @ns: Target user namespace (intentionally ignored; policy depends on the
+ * acting credentials and @cap, not namespace ownership).
+ * @cap: Capability number (CAP_*).
+ * @opts: Capability check options. CAP_OPT_NOAUDIT skips the denial record:
+ * no audit record, no trace event, and no denial count.
+ *
+ * Pure bitmask check: denies the capability if it is not in the layer's allowed
+ * set. This hook is purely restrictive: commoncap is registered with
+ * LSM_ORDER_FIRST so cap_capable() always runs first, which means Landlock can
+ * deny capabilities that commoncap would allow, but never grant capabilities
+ * that commoncap denied.
+ *
+ * Return: 0 if allowed, -EPERM if capability use is denied.
+ */
+static int hook_capable(const struct cred *cred, struct user_namespace *ns,
+ int cap, unsigned int opts)
+{
+ const struct landlock_cred_security *subject;
+ size_t denied_layer;
+
+ subject = landlock_get_applicable_subject(cred, cap_permission, NULL);
+ if (!subject)
+ return 0;
+
+ denied_layer = landlock_permission_is_denied(
+ subject->domain, LANDLOCK_PERMISSION_CAPABILITY_USE,
+ landlock_cap_to_bit(cap));
+ if (!denied_layer)
+ return 0;
+
+ if (!(opts & CAP_OPT_NOAUDIT))
+ landlock_log_denial(
+ subject,
+ &(struct landlock_request){
+ .type = LANDLOCK_REQUEST_CAPABILITY,
+ .audit.type = LSM_AUDIT_DATA_CAP,
+ .audit.u.cap = cap,
+ .permission =
+ LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .layer_plus_one = denied_layer,
+ });
+
+ return -EPERM;
+}
+
+static struct security_hook_list landlock_hooks[] __ro_after_init = {
+ LSM_HOOK_INIT(capable, hook_capable),
+};
+
+__init void landlock_add_cap_hooks(void)
+{
+ security_add_hooks(landlock_hooks, ARRAY_SIZE(landlock_hooks),
+ &landlock_lsmid);
+}
+
+#ifdef CONFIG_SECURITY_LANDLOCK_KUNIT_TEST
+
+#include <kunit/test.h>
+
+static void test_cap_to_bit(struct kunit *const test)
+{
+ KUNIT_EXPECT_EQ(test, BIT_ULL(0), landlock_cap_to_bit(0));
+ KUNIT_EXPECT_EQ(test, BIT_ULL(CAP_NET_RAW),
+ landlock_cap_to_bit(CAP_NET_RAW));
+ KUNIT_EXPECT_EQ(test, BIT_ULL(CAP_SYS_ADMIN),
+ landlock_cap_to_bit(CAP_SYS_ADMIN));
+ KUNIT_EXPECT_EQ(test, BIT_ULL(CAP_LAST_CAP),
+ landlock_cap_to_bit(CAP_LAST_CAP));
+}
+
+static void test_cap_to_bit_invalid(struct kunit *const test)
+{
+ if (!IS_ENABLED(CONFIG_BUG))
+ kunit_skip(test, "requires CONFIG_BUG");
+
+ /* clang-format off */
+ kunit_warning_suppress(test) {
+ /* clang-format on */
+ KUNIT_EXPECT_EQ(test, 0ULL, landlock_cap_to_bit(-1));
+ KUNIT_EXPECT_SUPPRESSED_WARNING_COUNT(test, 1);
+ KUNIT_EXPECT_EQ(test, 0ULL,
+ landlock_cap_to_bit(CAP_LAST_CAP + 1));
+ /* WARN_ON_ONCE() only reports the first invalid input. */
+ KUNIT_EXPECT_SUPPRESSED_WARNING_COUNT(test, 1);
+ }
+}
+
+static void test_caps_to_bits_valid(struct kunit *const test)
+{
+ KUNIT_EXPECT_EQ(test, (u64)CAP_VALID_MASK,
+ landlock_caps_to_bits(CAP_VALID_MASK));
+ KUNIT_EXPECT_EQ(test, BIT_ULL(CAP_NET_RAW),
+ landlock_caps_to_bits(BIT_ULL(CAP_NET_RAW)));
+}
+
+static void test_caps_to_bits_unknown(struct kunit *const test)
+{
+ if (!IS_ENABLED(CONFIG_BUG))
+ kunit_skip(test, "requires CONFIG_BUG");
+
+ /* clang-format off */
+ kunit_warning_suppress(test) {
+ /* clang-format on */
+ KUNIT_EXPECT_EQ(
+ test, 0ULL,
+ landlock_caps_to_bits(BIT_ULL(CAP_LAST_CAP + 1)));
+ KUNIT_EXPECT_SUPPRESSED_WARNING_COUNT(test, 1);
+ }
+}
+
+static void test_caps_to_bits_zero(struct kunit *const test)
+{
+ KUNIT_EXPECT_EQ(test, 0ULL, landlock_caps_to_bits(0));
+}
+
+static struct kunit_case test_cases[] = {
+ /* clang-format off */
+ KUNIT_CASE(test_cap_to_bit),
+ KUNIT_CASE(test_cap_to_bit_invalid),
+ KUNIT_CASE(test_caps_to_bits_valid),
+ KUNIT_CASE(test_caps_to_bits_unknown),
+ KUNIT_CASE(test_caps_to_bits_zero),
+ {}
+ /* clang-format on */
+};
+
+static struct kunit_suite test_suite = {
+ .name = "landlock_cap",
+ .test_cases = test_cases,
+};
+
+kunit_test_suite(test_suite);
+
+#endif /* CONFIG_SECURITY_LANDLOCK_KUNIT_TEST */
diff --git a/security/landlock/cap.h b/security/landlock/cap.h
new file mode 100644
index 000000000000..11208cbf15d2
--- /dev/null
+++ b/security/landlock/cap.h
@@ -0,0 +1,48 @@
+/* SPDX-License-Identifier: GPL-2.0-only */
+/*
+ * Landlock - Capability hooks
+ *
+ * Copyright © 2026 Cloudflare, Inc.
+ */
+
+#ifndef _SECURITY_LANDLOCK_CAP_H
+#define _SECURITY_LANDLOCK_CAP_H
+
+#include <linux/bitops.h>
+#include <linux/bug.h>
+#include <linux/capability.h>
+#include <linux/types.h>
+
+/**
+ * landlock_cap_to_bit - Convert a capability number to a compact bitmask
+ *
+ * @cap: Capability number (CAP_*).
+ *
+ * Return: BIT_ULL(@cap), or 0 if @cap is invalid (with a WARN).
+ */
+static inline u64 landlock_cap_to_bit(const int cap)
+{
+ if (WARN_ON_ONCE(!cap_valid(cap)))
+ return 0;
+
+ return BIT_ULL(cap);
+}
+
+/**
+ * landlock_caps_to_bits - Validate and mask a capability bitmask
+ *
+ * @capabilities: Bitmask of capabilities (e.g. from user space).
+ *
+ * Return: @capabilities masked to known capabilities. Warns if unknown bits
+ * are present (callers must pre-mask for user input).
+ */
+static inline u64 landlock_caps_to_bits(const u64 capabilities)
+{
+ /* Callers pre-mask (CAP_VALID_MASK); the WARN guards future callers. */
+ WARN_ON_ONCE(capabilities & ~CAP_VALID_MASK);
+ return capabilities & CAP_VALID_MASK;
+}
+
+__init void landlock_add_cap_hooks(void);
+
+#endif /* _SECURITY_LANDLOCK_CAP_H */
diff --git a/security/landlock/domain.h b/security/landlock/domain.h
index ab3c97c4455a..60e8eb2d6ac2 100644
--- a/security/landlock/domain.h
+++ b/security/landlock/domain.h
@@ -371,6 +371,9 @@ landlock_permission_is_denied(const struct landlock_domain *const domain,
case LANDLOCK_PERMISSION_NAMESPACE_USE:
allowed = domain->layers[layer].allowed.ns_types;
break;
+ case LANDLOCK_PERMISSION_CAPABILITY_USE:
+ allowed = domain->layers[layer].allowed.caps;
+ break;
default:
WARN_ONCE(1, "Unknown permission %u\n",
(unsigned int)permission_bit);
diff --git a/security/landlock/limits.h b/security/landlock/limits.h
index 767d57799f60..ddd300e10689 100644
--- a/security/landlock/limits.h
+++ b/security/landlock/limits.h
@@ -11,6 +11,7 @@
#define _SECURITY_LANDLOCK_LIMITS_H
#include <linux/bitops.h>
+#include <linux/capability.h>
#include <linux/limits.h>
#include <linux/ns/ns_common_types.h>
#include <uapi/linux/landlock.h>
@@ -32,11 +33,12 @@
#define LANDLOCK_MASK_SCOPE ((LANDLOCK_LAST_SCOPE << 1) - 1)
#define LANDLOCK_NUM_SCOPE __const_hweight64(LANDLOCK_MASK_SCOPE)
-#define LANDLOCK_LAST_PERMISSION LANDLOCK_PERMISSION_NAMESPACE_USE
+#define LANDLOCK_LAST_PERMISSION LANDLOCK_PERMISSION_CAPABILITY_USE
#define LANDLOCK_MASK_PERMISSION ((LANDLOCK_LAST_PERMISSION << 1) - 1)
#define LANDLOCK_NUM_PERMISSION __const_hweight64(LANDLOCK_MASK_PERMISSION)
#define LANDLOCK_NUM_NAMESPACE_TYPE __const_hweight64((u64)CLONE_NS_ALL)
+#define LANDLOCK_NUM_CAPABILITY (CAP_LAST_CAP + 1)
#define LANDLOCK_NUM_ACCESS_MAX \
MAX(MAX(LANDLOCK_NUM_ACCESS_FS, LANDLOCK_NUM_ACCESS_NET), LANDLOCK_NUM_SCOPE)
diff --git a/security/landlock/log.c b/security/landlock/log.c
index 8cb1dfd0d4ae..43efbb95725c 100644
--- a/security/landlock/log.c
+++ b/security/landlock/log.c
@@ -12,6 +12,7 @@
#include "access.h"
#include "audit.h"
+#include "cap.h"
#include "common.h"
#include "cred.h"
#include "domain.h"
@@ -393,6 +394,12 @@ static bool is_valid_request(const struct landlock_request *const request)
WARN_ON_ONCE(request->audit.type != LSM_AUDIT_DATA_NS))
return false;
break;
+ case LANDLOCK_REQUEST_CAPABILITY:
+ if (WARN_ON_ONCE(request->permission !=
+ LANDLOCK_PERMISSION_CAPABILITY_USE) ||
+ WARN_ON_ONCE(request->audit.type != LSM_AUDIT_DATA_CAP))
+ return false;
+ break;
case LANDLOCK_REQUEST_PTRACE:
case LANDLOCK_REQUEST_FS_CHANGE_TOPOLOGY:
case LANDLOCK_REQUEST_FS_ACCESS:
@@ -481,6 +488,9 @@ is_denial_quieted(const struct landlock_request *const request,
case LANDLOCK_REQUEST_NAMESPACE:
return !!(youngest_denied->quiet_permission.ns_types &
landlock_ns_type_to_bit(request->audit.u.ns.ns_type));
+ case LANDLOCK_REQUEST_CAPABILITY:
+ return !!(youngest_denied->quiet_permission.caps &
+ landlock_cap_to_bit(request->audit.u.cap));
/*
* Leave LANDLOCK_REQUEST_PTRACE and LANDLOCK_REQUEST_FS_CHANGE_TOPOLOGY
* unhandled for now - they are never quiet.
diff --git a/security/landlock/log.h b/security/landlock/log.h
index a12956cac40e..596feeeb7aa6 100644
--- a/security/landlock/log.h
+++ b/security/landlock/log.h
@@ -26,6 +26,7 @@ enum landlock_request_type {
LANDLOCK_REQUEST_SCOPE_ABSTRACT_UNIX_SOCKET,
LANDLOCK_REQUEST_SCOPE_SIGNAL,
LANDLOCK_REQUEST_NAMESPACE,
+ LANDLOCK_REQUEST_CAPABILITY,
};
struct landlock_blockers {
diff --git a/security/landlock/setup.c b/security/landlock/setup.c
index a7ed776b41b4..971419d663bb 100644
--- a/security/landlock/setup.c
+++ b/security/landlock/setup.c
@@ -11,6 +11,7 @@
#include <linux/lsm_hooks.h>
#include <uapi/linux/lsm.h>
+#include "cap.h"
#include "common.h"
#include "cred.h"
#include "errata.h"
@@ -70,6 +71,7 @@ static int __init landlock_init(void)
landlock_add_fs_hooks();
landlock_add_net_hooks();
landlock_add_ns_hooks();
+ landlock_add_cap_hooks();
landlock_init_id();
landlock_initialized = true;
pr_info("Up and running.\n");
diff --git a/security/landlock/syscalls.c b/security/landlock/syscalls.c
index c37edcc478bc..fb5af318b69b 100644
--- a/security/landlock/syscalls.c
+++ b/security/landlock/syscalls.c
@@ -31,6 +31,7 @@
#include <linux/uaccess.h>
#include <uapi/linux/landlock.h>
+#include "cap.h"
#include "cred.h"
#include "domain.h"
#include "fs.h"
@@ -101,8 +102,9 @@ static void build_check_abi(void)
struct landlock_path_beneath_attr path_beneath_attr;
struct landlock_net_port_attr net_port_attr;
struct landlock_namespace_attr namespace_attr;
+ struct landlock_capability_attr capability_attr;
size_t ruleset_size, path_beneath_size, net_port_size;
- size_t namespace_size;
+ size_t namespace_size, capability_size;
/*
* For each user space ABI structures, first checks that there is no
@@ -134,6 +136,12 @@ static void build_check_abi(void)
namespace_size += sizeof(namespace_attr.quiet_namespace_types);
BUILD_BUG_ON(sizeof(namespace_attr) != namespace_size);
BUILD_BUG_ON(sizeof(namespace_attr) != 24);
+
+ capability_size = sizeof(capability_attr.permissions);
+ capability_size += sizeof(capability_attr.allowed_capabilities);
+ capability_size += sizeof(capability_attr.quiet_capabilities);
+ BUILD_BUG_ON(sizeof(capability_attr) != capability_size);
+ BUILD_BUG_ON(sizeof(capability_attr) != 24);
}
/* Ruleset handling */
@@ -527,14 +535,88 @@ static int add_rule_namespace(struct landlock_ruleset *const ruleset,
return 0;
}
+static int add_rule_capability(struct landlock_ruleset *const ruleset,
+ const void __user *const rule_attr,
+ const u32 flags)
+{
+ struct landlock_capability_attr cap_attr;
+ access_mask_t mask;
+ u64 allowed_caps, quiet_caps;
+ int ret;
+
+ /*
+ * Capability rules support no add-rule flags. In particular,
+ * LANDLOCK_ADD_RULE_QUIET is filesystem/network only.
+ */
+ if (flags)
+ return -EINVAL;
+
+ /* Copies raw user space buffer. */
+ ret = copy_from_user(&cap_attr, rule_attr, sizeof(cap_attr));
+ if (ret)
+ return -EFAULT;
+
+ /* Informs about useless rule: empty permissions. */
+ if (!cap_attr.permissions)
+ return -ENOMSG;
+
+ /*
+ * The permissions selector must match
+ * LANDLOCK_PERMISSION_CAPABILITY_USE. The valid set is a single bit
+ * today, so this is an exact match now; the check broadens to a subset
+ * test once another supported permission is added.
+ */
+ if (cap_attr.permissions != LANDLOCK_PERMISSION_CAPABILITY_USE)
+ return -EINVAL;
+
+ /*
+ * Checks that permissions match the ruleset constraints. This also
+ * makes quieting require the category to be handled.
+ */
+ mask = landlock_get_permission_mask(ruleset);
+ if (!(mask & LANDLOCK_PERMISSION_CAPABILITY_USE))
+ return -EINVAL;
+
+ /*
+ * Informs about useless rule: neither allows nor quiets anything. A
+ * quiet-only rule (empty allowed set) is legal.
+ */
+ if (!cap_attr.allowed_capabilities && !cap_attr.quiet_capabilities)
+ return -ENOMSG;
+
+ /*
+ * Stores only the capabilities this kernel knows about. Unknown bits
+ * are silently accepted for forward compatibility: user space compiled
+ * against newer headers can pass new CAP_* bits without getting EINVAL
+ * on older kernels. Unknown bits have no effect because no hook checks
+ * them. The quiet bitmask suppresses logging of denials attributed to
+ * this layer; see landlock_log_denial().
+ */
+ allowed_caps = cap_attr.allowed_capabilities & CAP_VALID_MASK;
+ quiet_caps = cap_attr.quiet_capabilities & CAP_VALID_MASK;
+
+ mutex_lock(&ruleset->lock);
+ ruleset->layer.allowed.caps |= landlock_caps_to_bits(allowed_caps);
+#ifdef CONFIG_SECURITY_LANDLOCK_LOG
+ ruleset->quiet_permission.caps |= landlock_caps_to_bits(quiet_caps);
+#endif /* CONFIG_SECURITY_LANDLOCK_LOG */
+#ifdef CONFIG_TRACEPOINTS
+ ruleset->version++;
+#endif /* CONFIG_TRACEPOINTS */
+ trace_landlock_add_rule_capability(ruleset, flags, cap_attr.permissions,
+ allowed_caps, quiet_caps);
+ mutex_unlock(&ruleset->lock);
+ return 0;
+}
+
/**
* sys_landlock_add_rule - Add a new rule to a ruleset
*
* @ruleset_fd: File descriptor tied to the ruleset that should be extended
* with the new rule.
* @rule_type: Identify the structure type pointed to by @rule_attr:
- * %LANDLOCK_RULE_PATH_BENEATH, %LANDLOCK_RULE_NET_PORT, or
- * %LANDLOCK_RULE_NAMESPACE.
+ * %LANDLOCK_RULE_PATH_BENEATH, %LANDLOCK_RULE_NET_PORT,
+ * %LANDLOCK_RULE_NAMESPACE, or %LANDLOCK_RULE_CAPABILITY.
* @rule_attr: Pointer to a rule (matching the @rule_type).
* @flags: Must be 0 or %LANDLOCK_ADD_RULE_QUIET.
*
@@ -553,6 +635,8 @@ static int add_rule_namespace(struct landlock_ruleset *const ruleset,
* handled accesses)
* - %EINVAL: A nonzero &landlock_namespace_attr.permissions is not
* %LANDLOCK_PERMISSION_NAMESPACE_USE or is not handled by the ruleset;
+ * - %EINVAL: A nonzero &landlock_capability_attr.permissions is not
+ * %LANDLOCK_PERMISSION_CAPABILITY_USE or is not handled by the ruleset;
* - %EINVAL: &landlock_net_port_attr.port is greater than 65535;
* - %EINVAL: LANDLOCK_ADD_RULE_QUIET is passed but the ruleset has no
* quiet access bits set for the corresponding rule type.
@@ -561,6 +645,9 @@ static int add_rule_namespace(struct landlock_ruleset *const ruleset,
* - %ENOMSG: &landlock_namespace_attr.permissions is 0, or both
* &landlock_namespace_attr.allowed_namespace_types and
* &landlock_namespace_attr.quiet_namespace_types are 0;
+ * - %ENOMSG: &landlock_capability_attr.permissions is 0, or both
+ * &landlock_capability_attr.allowed_capabilities and
+ * &landlock_capability_attr.quiet_capabilities are 0;
* - %EBADF: @ruleset_fd is not a file descriptor for the current thread, or a
* member of @rule_attr is not a file descriptor as expected;
* - %EBADFD: @ruleset_fd is not a ruleset file descriptor, or a member of
@@ -595,6 +682,8 @@ SYSCALL_DEFINE4(landlock_add_rule, const int, ruleset_fd,
return add_rule_net_port(ruleset, rule_attr, flags);
case LANDLOCK_RULE_NAMESPACE:
return add_rule_namespace(ruleset, rule_attr, flags);
+ case LANDLOCK_RULE_CAPABILITY:
+ return add_rule_capability(ruleset, rule_attr, flags);
default:
return -EINVAL;
}
diff --git a/security/landlock/trace.c b/security/landlock/trace.c
index 300afcc82220..995c6b948e16 100644
--- a/security/landlock/trace.c
+++ b/security/landlock/trace.c
@@ -94,6 +94,18 @@ void landlock_trace_denial(
request->audit.u.ns.ns_id);
}
break;
+ case LANDLOCK_REQUEST_CAPABILITY:
+ if (trace_landlock_deny_permission_capability_enabled()) {
+ const struct landlock_blockers blockers = {
+ .access = missing,
+ .type = request->type,
+ };
+
+ trace_landlock_deny_permission_capability(
+ youngest_denied, same_exec, logged, &blockers,
+ request->audit.u.cap);
+ }
+ break;
case LANDLOCK_REQUEST_FS_ACCESS:
case LANDLOCK_REQUEST_FS_CHANGE_TOPOLOGY:
if (trace_landlock_deny_access_fs_enabled()) {
--
2.55.0
^ permalink raw reply [flat|nested] 9+ messages in thread* [PATCH v4 5/8] selftests/landlock: Add namespace restriction tests
2026-10-02 12:43 [PATCH v4 0/8] Landlock: Namespace and capability control Mickaël Salaün
` (3 preceding siblings ...)
2026-10-02 12:43 ` [PATCH v4 4/8] landlock: Enforce capability restrictions Mickaël Salaün
@ 2026-10-02 12:43 ` Mickaël Salaün
2026-10-02 12:43 ` [PATCH v4 6/8] selftests/landlock: Add capability " Mickaël Salaün
` (2 subsequent siblings)
7 siblings, 0 replies; 9+ messages in thread
From: Mickaël Salaün @ 2026-10-02 12:43 UTC (permalink / raw)
To: Christian Brauner, Günther Noack, Paul Moore, Serge E . Hallyn
Cc: Mickaël Salaün, Daniel Durning, Jonathan Corbet,
Justin Suess, Lennart Poettering, Mikhail Ivanov,
Nicolas Bouchinet, Shervin Oloumi, Tingmao Wang, kernel-team,
linux-fsdevel, linux-kernel, linux-security-module
Add tests covering LANDLOCK_PERMISSION_NAMESPACE_USE (namespace creation
via unshare/clone and namespace entry via setns) and its interaction
with LANDLOCK_PERMISSION_CAPABILITY_USE.
Rule validation covers the forward-compatibility contract: unknown
CLONE_NEW* bits, including holes, upper-range bits and bit 63, are
accepted at rule-add time yet have no runtime effect, because
deny-by-default still applies once the domain is enforced. Empty masks,
invalid permission selectors and non-zero flags are rejected.
Creation tests use FIXTURE_VARIANT to exercise all eight namespace types
across allowed/denied and privileged/unprivileged combinations, so every
type is shown to reach security_namespace_init(). Stacking tests cover
the three per-layer allow/deny combinations that exercise distinct
walker paths, and one domain handles both permissions at once.
Entry tests subject setns to the same type-based check through
security_namespace_install(), across processes, and require both
permissions to allow a non-user namespace.
The four open_tree(2) and fsmount(2) variants, anonymous and not, all
create a mount namespace: they converge in security_namespace_init()
with CLONE_NEWNS through different alloc_mnt_ns() paths. A
CLONE_NEWUSER | CLONE_NEWUTS unshare under partial-allow rules shows
that creation is sequential, so a denial on either type rolls the whole
syscall back with EPERM.
Audit tests cover allowed and denied creation and entry, per-member
quiet masks and the youngest-denying-layer rule. Trace tests cover
handled permissions, successful and rejected rules, version history,
namespace IDs, and quiet denials, including repeated and unknown-only
effective no-ops.
Cc: Christian Brauner <brauner@kernel.org>
Cc: Günther Noack <gnoack@google.com>
Cc: Paul Moore <paul@paul-moore.com>
Cc: Serge E. Hallyn <serge@hallyn.com>
Signed-off-by: Mickaël Salaün <mic@digikod.net>
---
Changes since v3:
https://patch.msgid.link/20260726161400.3010511-10-mic@digikod.net
- Cover the EFAULT result for an invalid namespace rule pointer.
- Drop CAP_AUDIT_CONTROL after cleaning up the audit fixture.
- Add trace tests for handled permissions, successful and rejected
rules, namespace IDs, and quiet denials, including repeated and
unknown-only effective no-ops.
- Make the successful clone3 assertion fatal before passing its PID to
waitpid().
Changes since v2:
https://patch.msgid.link/20260527181127.879771-7-mic@digikod.net
- Rebased for ABI 11 and the renamed namespace rule attribute (perm
plus allowed_namespace_types/quiet_namespace_types).
- Add per-member quiet audit tests for the namespace permission:
create_quieted, quiet_is_per_member, allow_and_quiet,
allow_and_quiet_same_member (quiet on an allowed member is inert),
quiet_youngest_layer_wins, and quiet_parent_only_denying_layer, plus a
quiet-only-rule leg in add_rule_bad_attr and the add_ns_rule_full
helper.
- Add a positive both-caps mount setns leg (two_perm_mnt_setns_allowed)
and the both-perms inverse test two_perm_ns_denied (the capability is
allowed but the namespace type is not, so the namespace hook denies).
- Add two_perm_combined_unshare_denied and
two_perm_combined_unshare_allowed: combining CLONE_NEWUSER with
another type in a single unshare(2) does not exempt the non-user
namespace from the capability check, so it still requires an allowed
CAP_SYS_ADMIN when the domain handles LANDLOCK_PERM_CAPABILITY_USE.
- Add add_rule_unknown_no_runtime_effect_setns, the install-hook mirror
of the creation-hook unknown-bit test.
- Add ns_audit.setns_quieted (a quieted setns denial is still EPERM but
unlogged) and ns_audit.quiet_unknown_bit_no_effect (an unknown quiet
bit does not suppress a known denial).
- Trim ns_proc_open to two representative types (user and mnt): opening
/proc/self/ns/* never reaches a Landlock hook and the path is
namespace-type-agnostic, so the other per-type variants added no
coverage.
Changes since v1:
https://patch.msgid.link/20260312100444.2609563-9-mic@digikod.net
- Update audit test patterns: namespace_inum replaced by namespace_id.
- Fix user_denied.setns expected error: security_namespace_install()
now runs before userns_install(), so Landlock returns EPERM before
userns_install() returns EINVAL.
- Rename LANDLOCK_PERM_NAMESPACE_ENTER references to
LANDLOCK_PERM_NAMESPACE_USE (companion change to the introducing
commit).
- Add ns_proc_open fixture covering all 8 namespace types: verify that
open("/proc/self/ns/<type>", O_RDONLY) does not trigger Landlock
denial under LANDLOCK_PERM_NAMESPACE_USE. Defensive boundary
documentation that the procfs ns/<type> open path is outside the
per-category permission's scope; catches future regressions if a hook
is misplaced.
- Add ns_mount_fd fixture covering open_tree(OPEN_TREE_CLONE),
open_tree(OPEN_TREE_NAMESPACE), fsmount(FSMOUNT_CLOEXEC), and
fsmount(FSMOUNT_NAMESPACE) with denied/allowed/unsandboxed
variants. All four converge in security_namespace_init() with
CLONE_NEWNS but exercise different code paths through alloc_mnt_ns().
- Add ns_create_multi_flag fixture covering a combined unshare with
CLONE_NEWUSER | CLONE_NEWUTS under partial-allow rules: documents
that any Landlock denial in the sequential namespace creation
rolls back the entire syscall with EPERM.
- Reshape setns_cross_process variants: rename allowed to
sandboxed_allowed and add an unsandboxed variant, so the fixture
now covers denied, sandboxed_allowed, and unsandboxed.
- Add sys_open_tree, sys_fsopen, sys_fsconfig, sys_fsmount syscall
wrappers in wrappers.h, and update fs_test.c to use sys_open_tree
instead of its local open_tree wrapper.
- Rename comments referencing security_namespace_alloc() and
hook_namespace_alloc() to security_namespace_init() and
hook_namespace_init() (companion change to the LSM hook rename in
the introducing commit).
- Document that setns_cross_process exercises only CLONE_NEWUTS:
the same enforcement applies to every namespace type via the
unified hook_namespace_install() helper.
- Add add_rule_unknown_no_runtime_effect: assert that a rule
listing only unknown namespace bits is accepted at rule-add time
but has no runtime effect, so an actual CLONE_NEW* operation is
still denied by deny-by-default once the domain is enforced.
- Add ns_stacking parent_denies variant covering the inverse
direction of stacking: layer 1 denies, layer 2 allows, operation
still denied. Completes the per-layer walker direction coverage.
---
tools/testing/selftests/landlock/common.h | 23 +
tools/testing/selftests/landlock/config | 5 +
tools/testing/selftests/landlock/fs_test.c | 13 +-
tools/testing/selftests/landlock/ns_test.c | 2643 +++++++++++++++++++
tools/testing/selftests/landlock/trace.h | 25 +
tools/testing/selftests/landlock/wrappers.h | 29 +
6 files changed, 2728 insertions(+), 10 deletions(-)
create mode 100644 tools/testing/selftests/landlock/ns_test.c
diff --git a/tools/testing/selftests/landlock/common.h b/tools/testing/selftests/landlock/common.h
index c5124de68a51..25aa4777cc33 100644
--- a/tools/testing/selftests/landlock/common.h
+++ b/tools/testing/selftests/landlock/common.h
@@ -130,6 +130,29 @@ static void __maybe_unused clear_ambient_cap(
EXPECT_EQ(0, cap_get_ambient(cap));
}
+/*
+ * Returns true if the current process is in the initial user namespace.
+ * Compares the readlink targets of /proc/self/ns/user and /proc/1/ns/user.
+ */
+static bool __maybe_unused is_in_init_user_ns(void)
+{
+ char self_buf[64], init_buf[64];
+ ssize_t self_len, init_len;
+
+ self_len = readlink("/proc/self/ns/user", self_buf, sizeof(self_buf));
+ if (self_len <= 0 || self_len >= (ssize_t)sizeof(self_buf))
+ return false;
+
+ init_len = readlink("/proc/1/ns/user", init_buf, sizeof(init_buf));
+ if (init_len <= 0 || init_len >= (ssize_t)sizeof(init_buf))
+ return false;
+
+ if (self_len != init_len)
+ return false;
+
+ return memcmp(self_buf, init_buf, self_len) == 0;
+}
+
/* Receives an FD from a UNIX socket. Returns the received FD, or -errno. */
static int __maybe_unused recv_fd(int usock)
{
diff --git a/tools/testing/selftests/landlock/config b/tools/testing/selftests/landlock/config
index d86321936fd8..5568d7a4b2d7 100644
--- a/tools/testing/selftests/landlock/config
+++ b/tools/testing/selftests/landlock/config
@@ -5,6 +5,7 @@ CONFIG_CGROUP_SCHED=y
CONFIG_ENABLE_DEFAULT_TRACERS=y
CONFIG_FTRACE=y
CONFIG_INET=y
+CONFIG_IPC_NS=y
CONFIG_IPV6=y
CONFIG_KEYS=y
CONFIG_MPTCP=y
@@ -12,10 +13,14 @@ CONFIG_MPTCP_IPV6=y
CONFIG_NET=y
CONFIG_NET_NS=y
CONFIG_OVERLAY_FS=y
+CONFIG_PID_NS=y
CONFIG_PROC_FS=y
CONFIG_SECURITY=y
CONFIG_SECURITY_LANDLOCK=y
CONFIG_SHMEM=y
CONFIG_SYSFS=y
+CONFIG_TIME_NS=y
CONFIG_TMPFS=y
CONFIG_TMPFS_XATTR=y
+CONFIG_USER_NS=y
+CONFIG_UTS_NS=y
diff --git a/tools/testing/selftests/landlock/fs_test.c b/tools/testing/selftests/landlock/fs_test.c
index 6e979cef884d..7e25a6f6073c 100644
--- a/tools/testing/selftests/landlock/fs_test.c
+++ b/tools/testing/selftests/landlock/fs_test.c
@@ -57,13 +57,6 @@ int renameat2(int olddirfd, const char *oldpath, int newdirfd,
}
#endif
-#ifndef open_tree
-int open_tree(int dfd, const char *filename, unsigned int flags)
-{
- return syscall(__NR_open_tree, dfd, filename, flags);
-}
-#endif
-
static int sys_execveat(int dirfd, const char *pathname, char *const argv[],
char *const envp[], int flags)
{
@@ -2628,9 +2621,9 @@ TEST_F_FORK(layout1, refer_mount_root_deny)
/* Creates a mount object from a non-mount point. */
set_cap(_metadata, CAP_SYS_ADMIN);
- root_fd =
- open_tree(AT_FDCWD, dir_s1d1,
- AT_EMPTY_PATH | OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ root_fd = sys_open_tree(AT_FDCWD, dir_s1d1,
+ AT_EMPTY_PATH | OPEN_TREE_CLONE |
+ OPEN_TREE_CLOEXEC);
clear_cap(_metadata, CAP_SYS_ADMIN);
ASSERT_LE(0, root_fd);
diff --git a/tools/testing/selftests/landlock/ns_test.c b/tools/testing/selftests/landlock/ns_test.c
new file mode 100644
index 000000000000..6faafc110844
--- /dev/null
+++ b/tools/testing/selftests/landlock/ns_test.c
@@ -0,0 +1,2643 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Landlock tests - Namespace restriction
+ *
+ * Copyright © 2026 Cloudflare, Inc.
+ */
+
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <linux/capability.h>
+#include <linux/landlock.h>
+#include <linux/mount.h>
+#include <linux/nsfs.h>
+#include <sched.h>
+#include <stdio.h>
+#include <sys/ioctl.h>
+#include <sys/prctl.h>
+#include <sys/wait.h>
+#include <syscall.h>
+#include <unistd.h>
+
+#include "audit.h"
+#include "common.h"
+#include "trace.h"
+
+#define TRACE_TASK "ns_test"
+
+/*
+ * Max length for /proc/self/ns/<name> paths (longest: "/proc/self/ns/cgroup").
+ */
+#define NS_PROC_PATH_MAX 32
+
+static int create_ns_ruleset(void)
+{
+ const struct landlock_ruleset_attr attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ };
+
+ return landlock_create_ruleset(&attr, sizeof(attr), 0);
+}
+
+static int add_ns_rule(int ruleset_fd, __u64 ns_type)
+{
+ const struct landlock_namespace_attr attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .allowed_namespace_types = ns_type,
+ };
+
+ return landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE, &attr, 0);
+}
+
+static int add_ns_rule_full(int ruleset_fd, __u64 allowed, __u64 quiet)
+{
+ const struct landlock_namespace_attr attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .allowed_namespace_types = allowed,
+ .quiet_namespace_types = quiet,
+ };
+
+ return landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE, &attr, 0);
+}
+
+/*
+ * Returns the /proc/self/NS entry name for a given CLONE_NEW* type, or NULL if
+ * unknown. Used to check kernel support without side effects.
+ */
+static const char *ns_proc_name(__u64 ns_type)
+{
+ switch (ns_type) {
+ case CLONE_NEWNS:
+ return "mnt";
+ case CLONE_NEWCGROUP:
+ return "cgroup";
+ case CLONE_NEWUTS:
+ return "uts";
+ case CLONE_NEWIPC:
+ return "ipc";
+ case CLONE_NEWUSER:
+ return "user";
+ case CLONE_NEWPID:
+ return "pid";
+ case CLONE_NEWNET:
+ return "net";
+ case CLONE_NEWTIME:
+ return "time";
+ default:
+ return NULL;
+ }
+}
+
+static bool ns_is_supported(__u64 ns_type, char *proc_path, size_t size)
+{
+ const char *ns_name;
+
+ ns_name = ns_proc_name(ns_type);
+ if (!ns_name)
+ return false;
+
+ snprintf(proc_path, size, "/proc/self/ns/%s", ns_name);
+ return access(proc_path, F_OK) == 0;
+}
+
+/* Rule validation tests */
+
+TEST(add_rule_bad_attr)
+{
+ const struct landlock_ruleset_attr cap_only_attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ };
+ int ruleset_fd;
+ struct landlock_namespace_attr attr = {};
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+
+ /* Invalid rule pointer returns EFAULT. */
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ NULL, 0));
+ ASSERT_EQ(EFAULT, errno);
+
+ /* Empty permissions selector returns ENOMSG. */
+ attr.permissions = 0;
+ attr.allowed_namespace_types = CLONE_NEWUTS;
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+ ASSERT_EQ(ENOMSG, errno);
+
+ /* Valid namespace selector plus an extra unhandled selector bit. */
+ attr.permissions = LANDLOCK_PERMISSION_NAMESPACE_USE |
+ LANDLOCK_PERMISSION_CAPABILITY_USE;
+ attr.allowed_namespace_types = CLONE_NEWUTS;
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+ ASSERT_EQ(EINVAL, errno);
+
+ /* permissions with wrong type. */
+ attr.permissions = LANDLOCK_PERMISSION_CAPABILITY_USE;
+ attr.allowed_namespace_types = CLONE_NEWUTS;
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+ ASSERT_EQ(EINVAL, errno);
+
+ /*
+ * Unknown namespace bits (e.g. bit 63) are silently accepted for
+ * forward compatibility. Only known CLONE_NEW* bits are stored.
+ */
+ attr.permissions = LANDLOCK_PERMISSION_NAMESPACE_USE;
+ attr.allowed_namespace_types = 1ULL << 63;
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+
+ /* Useless rule: neither allowed nor quiet types set returns ENOMSG. */
+ attr.permissions = LANDLOCK_PERMISSION_NAMESPACE_USE;
+ attr.allowed_namespace_types = 0;
+ attr.quiet_namespace_types = 0;
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+ ASSERT_EQ(ENOMSG, errno);
+
+ /* Quiet-only rule (empty allowed set) is legal. */
+ attr.permissions = LANDLOCK_PERMISSION_NAMESPACE_USE;
+ attr.allowed_namespace_types = 0;
+ attr.quiet_namespace_types = CLONE_NEWUTS;
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+
+ /* Allow and quiet in the same rule. */
+ attr.permissions = LANDLOCK_PERMISSION_NAMESPACE_USE;
+ attr.allowed_namespace_types = CLONE_NEWNET;
+ attr.quiet_namespace_types = CLONE_NEWUTS;
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+ attr.quiet_namespace_types = 0;
+
+ /*
+ * Bit 1 is not a CLONE_NEW* value but is silently accepted for forward
+ * compatibility (no hole rejection).
+ */
+ attr.permissions = LANDLOCK_PERMISSION_NAMESPACE_USE;
+ attr.allowed_namespace_types = (1ULL << 1);
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+
+ /* Multi-bit values are valid (bitmask allows multiple types). */
+ attr.permissions = LANDLOCK_PERMISSION_NAMESPACE_USE;
+ attr.allowed_namespace_types = CLONE_NEWUTS | CLONE_NEWNET;
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+
+ /*
+ * LANDLOCK_ADD_RULE_QUIET (and any other flag) is filesystem/network
+ * only and must be rejected for namespace rules, even when every attr
+ * field is otherwise valid.
+ */
+ attr.permissions = LANDLOCK_PERMISSION_NAMESPACE_USE;
+ attr.allowed_namespace_types = CLONE_NEWUTS;
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, LANDLOCK_ADD_RULE_QUIET));
+ ASSERT_EQ(EINVAL, errno);
+
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /*
+ * Ruleset handles LANDLOCK_PERMISSION_CAPABILITY_USE but not
+ * LANDLOCK_PERMISSION_NAMESPACE_USE: adding a namespace rule must be
+ * rejected.
+ */
+ ruleset_fd = landlock_create_ruleset(&cap_only_attr,
+ sizeof(cap_only_attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ attr.permissions = LANDLOCK_PERMISSION_NAMESPACE_USE;
+ attr.allowed_namespace_types = CLONE_NEWUTS;
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+ ASSERT_EQ(EINVAL, errno);
+ EXPECT_EQ(0, close(ruleset_fd));
+}
+
+/*
+ * Unknown namespace types in the upper range are silently accepted (allow-list:
+ * they have no effect since the kernel never checks them).
+ */
+TEST(add_rule_unknown)
+{
+ int ruleset_fd;
+ struct landlock_namespace_attr attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ };
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+
+ /*
+ * Bit 31 is in the lower 32 bits but not a CLONE_NEW* value. Silently
+ * accepted for forward compatibility (no hole rejection).
+ */
+ attr.allowed_namespace_types = 1ULL << 31;
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+
+ /* Bit 32 is in the unknown upper range: silently accepted. */
+ attr.allowed_namespace_types = 1ULL << 32;
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+
+ EXPECT_EQ(0, close(ruleset_fd));
+}
+
+/*
+ * A rule that lists only namespace bits unknown to the running kernel is
+ * accepted by landlock_add_rule() but has no runtime effect: once the domain is
+ * enforced, any actual CLONE_NEW* operation is still denied by the per-category
+ * deny-by-default behaviour. This documents the forward-compatibility
+ * contract: unknown bits are silently accepted so the same policy can be loaded
+ * across kernels, but they never grant a permission that the running kernel
+ * knows nothing about.
+ */
+TEST(add_rule_unknown_no_runtime_effect)
+{
+ const struct landlock_ruleset_attr ruleset_attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ };
+ struct landlock_namespace_attr attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ /* Only unknown bits: bit 31 (in lower 32) and bit 32. */
+ .allowed_namespace_types = (1ULL << 31) | (1ULL << 32),
+ };
+ int ruleset_fd;
+
+ disable_caps(_metadata);
+
+ ruleset_fd =
+ landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+
+ /*
+ * Baseline: absent the domain, creating CLONE_NEWUTS succeeds. This
+ * ensures the EPERM below is attributable to Landlock rather than to a
+ * missing capability, and fails if the hook always allows.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWUTS));
+
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /*
+ * CLONE_NEWUTS is a real, known CLONE_NEW* type but was not authorised
+ * by the rule above; deny-by-default applies.
+ */
+ EXPECT_EQ(-1, unshare(CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/*
+ * The install-hook mirror of add_rule_unknown_no_runtime_effect: a rule listing
+ * only unknown namespace bits has no runtime effect on setns(2) either. Once
+ * enforced, entering a known namespace type is still denied by the per-category
+ * deny-by-default behaviour.
+ */
+TEST(add_rule_unknown_no_runtime_effect_setns)
+{
+ const struct landlock_ruleset_attr ruleset_attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ };
+ struct landlock_namespace_attr attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ /* Only unknown bits: bit 31 (in lower 32) and bit 32. */
+ .allowed_namespace_types = (1ULL << 31) | (1ULL << 32),
+ };
+ int ruleset_fd, ns_fd;
+
+ disable_caps(_metadata);
+
+ /* Open the NS FD before enforcing the domain. */
+ ns_fd = open("/proc/self/ns/uts", O_RDONLY);
+ ASSERT_LE(0, ns_fd);
+
+ ruleset_fd =
+ landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &attr, 0));
+
+ /*
+ * Baseline: absent the domain, entering the process's own UTS namespace
+ * succeeds, so the later EPERM is attributable to Landlock rather than
+ * to a missing capability, and fails if the hook always allows.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, setns(ns_fd, CLONE_NEWUTS));
+
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /*
+ * CLONE_NEWUTS is a real, known CLONE_NEW* type but was not authorised
+ * by the rule above; deny-by-default applies at the install hook.
+ */
+ EXPECT_EQ(-1, setns(ns_fd, CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, close(ns_fd));
+}
+
+/* Namespace creation tests (variant-based positive/negative) */
+
+/* clang-format off */
+FIXTURE(ns_create) {
+ /* clang-format on */
+ char proc_path[NS_PROC_PATH_MAX];
+};
+
+FIXTURE_VARIANT(ns_create)
+{
+ const __u64 namespace_types;
+ const bool is_sandboxed;
+ const bool has_rule;
+ const bool drop_all_caps;
+ const int expected_result;
+};
+
+/*
+ * Unsandboxed baseline: no Landlock domain is enforced. User namespace creation
+ * should succeed without any restriction.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, user_unsandboxed) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWUSER,
+ .is_sandboxed = false,
+ .has_rule = false,
+ .drop_all_caps = false,
+ .expected_result = 0,
+};
+
+/*
+ * User namespace creation denied: handled by Landlock but no rule allows
+ * CLONE_NEWUSER.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, user_denied) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWUSER,
+ .is_sandboxed = true,
+ .has_rule = false,
+ .drop_all_caps = false,
+ .expected_result = EPERM,
+};
+
+/* User namespace creation allowed: Landlock rule permits CLONE_NEWUSER. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, user_allowed) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWUSER,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = false,
+ .expected_result = 0,
+};
+
+/*
+ * User namespace creation while unprivileged: the process has no capabilities
+ * but unshare(CLONE_NEWUSER) is an unprivileged operation so it still succeeds.
+ * The Landlock rule allows it. For setns, the capability check (CAP_SYS_ADMIN)
+ * fails first since the process has no capabilities, yielding EPERM.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, user_unprivileged) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWUSER,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = true,
+ .expected_result = 0,
+};
+
+/*
+ * Unsandboxed baseline for non-user namespace: no Landlock domain, process has
+ * CAP_SYS_ADMIN. UTS creation should succeed.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, uts_unsandboxed) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWUTS,
+ .is_sandboxed = false,
+ .has_rule = false,
+ .drop_all_caps = false,
+ .expected_result = 0,
+};
+
+/*
+ * Non-user namespace denied: process has CAP_SYS_ADMIN (passes ns_capable), but
+ * Landlock denies (no rule).
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, uts_denied) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWUTS,
+ .is_sandboxed = true,
+ .has_rule = false,
+ .drop_all_caps = false,
+ .expected_result = EPERM,
+};
+
+/*
+ * Non-user namespace allowed: process has CAP_SYS_ADMIN and Landlock rule
+ * permits CLONE_NEWUTS.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, uts_allowed) {
+ .namespace_types = CLONE_NEWUTS,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = false,
+ .expected_result = 0,
+};
+/* clang-format on */
+
+/*
+ * Unprivileged namespace creation: process lacks CAP_SYS_ADMIN, so the kernel
+ * denies creation regardless of Landlock rules. Landlock cannot authorize what
+ * the kernel denied (LSM hooks are restriction-only). The rule is present to
+ * verify Landlock does not change the error code.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, uts_unprivileged) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWUTS,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = true,
+ .expected_result = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, ipc_denied) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWIPC,
+ .is_sandboxed = true,
+ .has_rule = false,
+ .drop_all_caps = false,
+ .expected_result = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, ipc_allowed) {
+ .namespace_types = CLONE_NEWIPC,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = false,
+ .expected_result = 0,
+};
+/* clang-format on */
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, ipc_unprivileged) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWIPC,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = true,
+ .expected_result = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, mnt_denied) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWNS,
+ .is_sandboxed = true,
+ .has_rule = false,
+ .drop_all_caps = false,
+ .expected_result = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, mnt_allowed) {
+ .namespace_types = CLONE_NEWNS,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = false,
+ .expected_result = 0,
+};
+/* clang-format on */
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, mnt_unprivileged) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWNS,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = true,
+ .expected_result = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, cgroup_denied) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWCGROUP,
+ .is_sandboxed = true,
+ .has_rule = false,
+ .drop_all_caps = false,
+ .expected_result = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, cgroup_allowed) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWCGROUP,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = false,
+ .expected_result = 0,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, cgroup_unprivileged) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWCGROUP,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = true,
+ .expected_result = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, pid_denied) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWPID,
+ .is_sandboxed = true,
+ .has_rule = false,
+ .drop_all_caps = false,
+ .expected_result = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, pid_allowed) {
+ .namespace_types = CLONE_NEWPID,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = false,
+ .expected_result = 0,
+};
+/* clang-format on */
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, pid_unprivileged) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWPID,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = true,
+ .expected_result = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, net_denied) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWNET,
+ .is_sandboxed = true,
+ .has_rule = false,
+ .drop_all_caps = false,
+ .expected_result = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, net_allowed) {
+ .namespace_types = CLONE_NEWNET,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = false,
+ .expected_result = 0,
+};
+/* clang-format on */
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, net_unprivileged) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWNET,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = true,
+ .expected_result = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, time_denied) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWTIME,
+ .is_sandboxed = true,
+ .has_rule = false,
+ .drop_all_caps = false,
+ .expected_result = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, time_allowed) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWTIME,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = false,
+ .expected_result = 0,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create, time_unprivileged) {
+ /* clang-format on */
+ .namespace_types = CLONE_NEWTIME,
+ .is_sandboxed = true,
+ .has_rule = true,
+ .drop_all_caps = true,
+ .expected_result = EPERM,
+};
+
+FIXTURE_SETUP(ns_create)
+{
+ ASSERT_TRUE(ns_is_supported(variant->namespace_types, self->proc_path,
+ sizeof(self->proc_path)))
+ {
+ TH_LOG("Namespace type 0x%llx not supported",
+ (unsigned long long)variant->namespace_types);
+ }
+
+ if (variant->drop_all_caps)
+ drop_caps(_metadata);
+ else
+ disable_caps(_metadata);
+}
+
+FIXTURE_TEARDOWN(ns_create)
+{
+}
+
+TEST_F(ns_create, unshare)
+{
+ int ruleset_fd, err;
+
+ if (variant->is_sandboxed) {
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+
+ if (variant->has_rule)
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd,
+ variant->namespace_types));
+
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+ }
+
+ /*
+ * Non-user namespaces need CAP_SYS_ADMIN for the privileged path. User
+ * namespaces and unprivileged tests skip this.
+ */
+ if (!variant->drop_all_caps &&
+ variant->namespace_types != CLONE_NEWUSER)
+ set_cap(_metadata, CAP_SYS_ADMIN);
+
+ err = unshare(variant->namespace_types);
+ if (variant->expected_result) {
+ EXPECT_EQ(-1, err);
+ EXPECT_EQ(variant->expected_result, errno);
+ } else {
+ EXPECT_EQ(0, err);
+ }
+
+ if (!variant->drop_all_caps &&
+ variant->namespace_types != CLONE_NEWUSER)
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/*
+ * clone3 exercises a different kernel entry point than unshare: it goes through
+ * kernel_clone() -> copy_process() -> copy_namespaces() ->
+ * create_new_namespaces(). Both paths converge at __ns_common_init() ->
+ * security_namespace_init(), but the entry point and argument handling differ.
+ */
+TEST_F(ns_create, clone3)
+{
+ int ruleset_fd, status;
+ pid_t pid;
+ struct clone_args args = {};
+
+ if (variant->is_sandboxed) {
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+
+ if (variant->has_rule)
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd,
+ variant->namespace_types));
+
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+ }
+
+ if (!variant->drop_all_caps &&
+ variant->namespace_types != CLONE_NEWUSER)
+ set_cap(_metadata, CAP_SYS_ADMIN);
+
+ args.flags = variant->namespace_types;
+ args.exit_signal = SIGCHLD;
+ pid = sys_clone3(&args, sizeof(args));
+ if (pid == 0)
+ _exit(EXIT_SUCCESS);
+
+ if (variant->expected_result) {
+ EXPECT_EQ(-1, pid);
+ EXPECT_EQ(variant->expected_result, errno);
+ } else {
+ ASSERT_LE(0, pid);
+ ASSERT_EQ(pid, waitpid(pid, &status, 0));
+ ASSERT_EQ(1, WIFEXITED(status));
+ ASSERT_EQ(EXIT_SUCCESS, WEXITSTATUS(status));
+ }
+
+ if (!variant->drop_all_caps &&
+ variant->namespace_types != CLONE_NEWUSER)
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/*
+ * setns exercises the namespace install path: validate_ns() ->
+ * security_namespace_install() -> hook_namespace_install(). This is a
+ * different LSM hook than creation, so it must be tested separately for each
+ * type.
+ *
+ * Mount namespace setns requires both CAP_SYS_ADMIN and CAP_SYS_CHROOT (checked
+ * by mntns_install), so the allowed variant sets both.
+ */
+TEST_F(ns_create, setns)
+{
+ int ruleset_fd, ns_fd, err, expected;
+
+ /*
+ * setns into the process's own user NS returns EINVAL from
+ * userns_install() (rejects re-entry), but when Landlock denies the
+ * operation, security_namespace_install() returns EPERM before
+ * userns_install() runs.
+ */
+ if (variant->namespace_types == CLONE_NEWUSER &&
+ !variant->expected_result) {
+ expected = EINVAL;
+ } else {
+ expected = variant->expected_result;
+ }
+
+ /* Open the NS FD before enforcing the domain. */
+ ns_fd = open(self->proc_path, O_RDONLY);
+ ASSERT_LE(0, ns_fd);
+
+ if (variant->is_sandboxed) {
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+
+ if (variant->has_rule)
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd,
+ variant->namespace_types));
+
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+ }
+
+ if (!variant->drop_all_caps) {
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ /*
+ * mntns_install() requires CAP_SYS_CHROOT in addition to
+ * CAP_SYS_ADMIN.
+ */
+ if (variant->namespace_types == CLONE_NEWNS)
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ }
+
+ err = setns(ns_fd, variant->namespace_types);
+ if (expected) {
+ EXPECT_EQ(-1, err);
+ EXPECT_EQ(expected, errno);
+ } else {
+ EXPECT_EQ(0, err);
+ }
+
+ if (!variant->drop_all_caps) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ if (variant->namespace_types == CLONE_NEWNS)
+ clear_cap(_metadata, CAP_SYS_CHROOT);
+ }
+
+ EXPECT_EQ(0, close(ns_fd));
+}
+
+/* Additional namespace creation tests */
+
+/*
+ * When LANDLOCK_PERMISSION_NAMESPACE_USE is not handled by any domain,
+ * namespace creation must produce the same result as without Landlock. Unlike
+ * the unsandboxed variants of ns_create (which have no domain at all), this
+ * test verifies that a domain handling only FS access does not interfere with
+ * namespace operations.
+ */
+TEST(ns_create_unhandled)
+{
+ const struct landlock_ruleset_attr attr = {
+ .handled_access_fs = LANDLOCK_ACCESS_FS_READ_FILE,
+ };
+ int ruleset_fd;
+
+ disable_caps(_metadata);
+
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /* User namespace creation should still work (unhandled). */
+ EXPECT_EQ(0, unshare(CLONE_NEWUSER));
+}
+
+/*
+ * Layer stacking: both layers must allow CLONE_NEWUSER for the operation to
+ * succeed. Variants exercise the three combinations of per-layer allow/deny
+ * that exercise distinct semantics; the (deny, deny) combination is omitted
+ * because it is covered by every other "deny" test in this file.
+ */
+/* clang-format off */
+FIXTURE(ns_stacking) {};
+/* clang-format on */
+
+FIXTURE_VARIANT(ns_stacking)
+{
+ bool first_layer_allows;
+ bool second_layer_allows;
+};
+
+/* Layer 1 allows, layer 2 denies -> child denies. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_stacking, deny) {
+ /* clang-format on */
+ .first_layer_allows = true,
+ .second_layer_allows = false,
+};
+
+/* Both layers allow CLONE_NEWUSER -> operation succeeds. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_stacking, allow) {
+ /* clang-format on */
+ .first_layer_allows = true,
+ .second_layer_allows = true,
+};
+
+/*
+ * Layer 1 denies, layer 2 allows -> still denied: a child layer cannot grant
+ * what an ancestor layer withheld. This complements the
+ * parent-allows/child-denies variant above; together they verify the walker
+ * checks both layers and accepts only the (allow, allow) cell.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_stacking, parent_denies) {
+ /* clang-format on */
+ .first_layer_allows = false,
+ .second_layer_allows = true,
+};
+
+FIXTURE_SETUP(ns_stacking)
+{
+ disable_caps(_metadata);
+}
+
+FIXTURE_TEARDOWN(ns_stacking)
+{
+}
+
+/*
+ * Verify that any layer can deny an operation: enforcement requires all layers
+ * to allow. Variants exercise the three combinations that exercise distinct
+ * walker paths (allow/deny, allow/allow, deny/allow); only allow/allow lets the
+ * operation through.
+ */
+TEST_F(ns_stacking, two_layers)
+{
+ int ruleset_fd;
+ const bool expect_success = variant->first_layer_allows &&
+ variant->second_layer_allows;
+
+ /* First layer: allow or deny depending on variant. */
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ if (variant->first_layer_allows)
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd, CLONE_NEWUSER));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /* Second layer: allow or deny depending on variant. */
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ if (variant->second_layer_allows)
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd, CLONE_NEWUSER));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ if (expect_success) {
+ EXPECT_EQ(0, unshare(CLONE_NEWUSER));
+ } else {
+ EXPECT_EQ(-1, unshare(CLONE_NEWUSER));
+ EXPECT_EQ(EPERM, errno);
+ }
+}
+
+/*
+ * Combined capability and namespace permissions in a single domain. Verifies
+ * that both permission types can coexist and are enforced independently.
+ */
+TEST(combined_cap_ns)
+{
+ const struct landlock_ruleset_attr attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_CAPABILITY_USE |
+ LANDLOCK_PERMISSION_NAMESPACE_USE,
+ };
+ const struct landlock_capability_attr cap_attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .allowed_capabilities = (1ULL << CAP_SYS_ADMIN),
+ };
+ const struct landlock_namespace_attr ns_attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .allowed_namespace_types = CLONE_NEWUSER,
+ };
+ int ruleset_fd;
+
+ /* Isolate hostname changes from other tests. */
+ ASSERT_EQ(0, unshare(CLONE_NEWUTS));
+
+ disable_caps(_metadata);
+
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &cap_attr, 0));
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &ns_attr, 0));
+
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /* CAP_SYS_ADMIN use allowed by capability rule. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, sethostname("test", 4));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /* CAP_SYS_CHROOT denied (not in allowed capability rules). */
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ EXPECT_EQ(-1, chroot("/"));
+ EXPECT_EQ(EPERM, errno);
+
+ /*
+ * UTS namespace creation denied by Landlock (not in allowed namespace
+ * rules). CAP_SYS_ADMIN is needed for the kernel's ns_capable() check
+ * to pass, so that Landlock's hook is actually reached.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, unshare(CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /* User namespace creation allowed by namespace rule. */
+ EXPECT_EQ(0, unshare(CLONE_NEWUSER));
+}
+
+/*
+ * Partial allow: one namespace type is allowed, another is denied. Verifies
+ * that rules are per-type.
+ */
+TEST(ns_create_partial)
+{
+ int ruleset_fd;
+
+ disable_caps(_metadata);
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+
+ /* Only allow UTS namespace creation. */
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd, CLONE_NEWUTS));
+
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /* UTS namespace should be allowed. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, unshare(CLONE_NEWUTS));
+
+ /* User namespace should be denied (no rule). */
+ EXPECT_EQ(-1, unshare(CLONE_NEWUSER));
+ EXPECT_EQ(EPERM, errno);
+}
+
+/*
+ * open_tree(2) and fsmount(2) acquire a file descriptor referring to a
+ * newly-created mount namespace. Both call paths funnel into
+ * security_namespace_init() with CLONE_NEWNS, gated by
+ * LANDLOCK_PERMISSION_NAMESPACE_USE. Without coverage here, regressions in
+ * those paths would slip past the suite.
+ */
+/* clang-format off */
+FIXTURE(ns_mount_fd) {};
+/* clang-format on */
+
+FIXTURE_VARIANT(ns_mount_fd)
+{
+ bool sandboxed;
+ bool has_rule;
+ int expected_errno;
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_mount_fd, denied) {
+ /* clang-format on */
+ .sandboxed = true,
+ .has_rule = false,
+ .expected_errno = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_mount_fd, allowed) {
+ /* clang-format on */
+ .sandboxed = true,
+ .has_rule = true,
+ .expected_errno = 0,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_mount_fd, unsandboxed) {
+ /* clang-format on */
+ .sandboxed = false,
+ .has_rule = false,
+ .expected_errno = 0,
+};
+
+FIXTURE_SETUP(ns_mount_fd)
+{
+}
+
+FIXTURE_TEARDOWN(ns_mount_fd)
+{
+}
+
+/*
+ * open_tree(OPEN_TREE_CLONE) creates an anonymous mount namespace to hold the
+ * cloned mount tree. hook_namespace_init() fires with CLONE_NEWNS.
+ */
+TEST_F(ns_mount_fd, open_tree_clone)
+{
+ int ruleset_fd, fd;
+
+ disable_caps(_metadata);
+
+ if (variant->sandboxed) {
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ if (variant->has_rule)
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd, CLONE_NEWNS));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+ }
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ fd = sys_open_tree(AT_FDCWD, "/",
+ OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC | AT_RECURSIVE);
+ if (variant->expected_errno) {
+ EXPECT_EQ(-1, fd);
+ EXPECT_EQ(variant->expected_errno, errno);
+ } else {
+ ASSERT_LE(0, fd);
+ EXPECT_EQ(0, close(fd));
+ }
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/*
+ * open_tree(OPEN_TREE_NAMESPACE) clones the mount tree into a new
+ * (non-anonymous) mount namespace. Same hook (CLONE_NEWNS) but a different
+ * code path inside fs/namespace.c (open_new_namespace -> alloc_mnt_ns).
+ * OPEN_TREE_NAMESPACE and OPEN_TREE_CLONE are mutually exclusive.
+ */
+TEST_F(ns_mount_fd, open_tree_namespace)
+{
+ int ruleset_fd, fd;
+
+ disable_caps(_metadata);
+
+ if (variant->sandboxed) {
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ if (variant->has_rule)
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd, CLONE_NEWNS));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+ }
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ fd = sys_open_tree(AT_FDCWD, "/",
+ OPEN_TREE_NAMESPACE | OPEN_TREE_CLOEXEC |
+ AT_RECURSIVE);
+ if (variant->expected_errno) {
+ EXPECT_EQ(-1, fd);
+ EXPECT_EQ(variant->expected_errno, errno);
+ } else {
+ ASSERT_LE(0, fd);
+ EXPECT_EQ(0, close(fd));
+ }
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/*
+ * fsmount(2) without FSMOUNT_NAMESPACE creates an anonymous mount namespace to
+ * attach the new superblock. hook_namespace_init() fires with CLONE_NEWNS.
+ * The fs context (fsopen + fsconfig) is set up before sandboxing because
+ * Landlock here only handles the namespace permission.
+ */
+TEST_F(ns_mount_fd, fsmount_default)
+{
+ int ruleset_fd, fs_fd, mnt_fd;
+
+ disable_caps(_metadata);
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ fs_fd = sys_fsopen("tmpfs", 0);
+ ASSERT_LE(0, fs_fd);
+ ASSERT_EQ(0, sys_fsconfig(fs_fd, FSCONFIG_CMD_CREATE, NULL, NULL, 0));
+
+ if (variant->sandboxed) {
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ if (variant->has_rule)
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd, CLONE_NEWNS));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+ }
+
+ mnt_fd = sys_fsmount(fs_fd, FSMOUNT_CLOEXEC, 0);
+ if (variant->expected_errno) {
+ EXPECT_EQ(-1, mnt_fd);
+ EXPECT_EQ(variant->expected_errno, errno);
+ } else {
+ ASSERT_LE(0, mnt_fd);
+ EXPECT_EQ(0, close(mnt_fd));
+ }
+ EXPECT_EQ(0, close(fs_fd));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/*
+ * fsmount(2) with FSMOUNT_NAMESPACE creates a (non-anonymous) mount namespace
+ * for the new mount. Same hook as the default path, different code path.
+ */
+TEST_F(ns_mount_fd, fsmount_namespace)
+{
+ int ruleset_fd, fs_fd, mnt_fd;
+
+ disable_caps(_metadata);
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ fs_fd = sys_fsopen("tmpfs", 0);
+ ASSERT_LE(0, fs_fd);
+ ASSERT_EQ(0, sys_fsconfig(fs_fd, FSCONFIG_CMD_CREATE, NULL, NULL, 0));
+
+ if (variant->sandboxed) {
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ if (variant->has_rule)
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd, CLONE_NEWNS));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+ }
+
+ mnt_fd = sys_fsmount(fs_fd, FSMOUNT_CLOEXEC | FSMOUNT_NAMESPACE, 0);
+ if (variant->expected_errno) {
+ EXPECT_EQ(-1, mnt_fd);
+ EXPECT_EQ(variant->expected_errno, errno);
+ } else {
+ ASSERT_LE(0, mnt_fd);
+ EXPECT_EQ(0, close(mnt_fd));
+ }
+ EXPECT_EQ(0, close(fs_fd));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/*
+ * unshare(2) with multiple CLONE_NEW* flags: when
+ * LANDLOCK_PERMISSION_NAMESPACE_USE denies any of the requested types, the
+ * entire syscall fails with EPERM. This documents the kernel's atomic
+ * behavior: create_new_namespaces(), called by copy_namespaces(), creates
+ * namespaces sequentially, and the first Landlock denial rolls back the whole
+ * operation. Mixing CLONE_NEWUSER (no capability check) with another
+ * CLONE_NEW* type is the typical container-runtime bootstrap pattern.
+ */
+/* clang-format off */
+FIXTURE(ns_create_multi_flag) {};
+/* clang-format on */
+
+FIXTURE_VARIANT(ns_create_multi_flag)
+{
+ __u64 allowed_types;
+ int expected_errno;
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create_multi_flag, partial_denied) {
+ /* clang-format on */
+ /* User namespace allowed; UTS namespace denied. */
+ .allowed_types = CLONE_NEWUSER,
+ .expected_errno = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_create_multi_flag, both_allowed) {
+ /* clang-format on */
+ .allowed_types = CLONE_NEWUSER | CLONE_NEWUTS,
+ .expected_errno = 0,
+};
+
+FIXTURE_SETUP(ns_create_multi_flag)
+{
+}
+
+FIXTURE_TEARDOWN(ns_create_multi_flag)
+{
+}
+
+TEST_F(ns_create_multi_flag, unshare)
+{
+ int ruleset_fd, status, err;
+ pid_t child;
+
+ disable_caps(_metadata);
+
+ /* Run unshare(2) in a child to avoid polluting the test process. */
+ child = fork();
+ ASSERT_LE(0, child);
+
+ if (child == 0) {
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd, variant->allowed_types));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ err = unshare(CLONE_NEWUSER | CLONE_NEWUTS);
+ if (variant->expected_errno) {
+ EXPECT_EQ(-1, err);
+ EXPECT_EQ(variant->expected_errno, errno);
+ } else {
+ EXPECT_EQ(0, err);
+ }
+ _exit(_metadata->exit_code);
+ }
+
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ ASSERT_EQ(1, WIFEXITED(status));
+ ASSERT_EQ(EXIT_SUCCESS, WEXITSTATUS(status));
+}
+
+/* clang-format off */
+FIXTURE(setns_cross_process) {};
+/* clang-format on */
+
+FIXTURE_VARIANT(setns_cross_process)
+{
+ bool is_sandboxed;
+ bool has_rule;
+ int expected_setns;
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(setns_cross_process, denied) {
+ /* clang-format on */
+ .is_sandboxed = true,
+ .has_rule = false,
+ .expected_setns = EPERM,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(setns_cross_process, sandboxed_allowed) {
+ /* clang-format on */
+ .is_sandboxed = true,
+ .has_rule = true,
+ .expected_setns = 0,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(setns_cross_process, unsandboxed) {
+ /* clang-format on */
+ .is_sandboxed = false,
+ .has_rule = false,
+ .expected_setns = 0,
+};
+
+FIXTURE_SETUP(setns_cross_process)
+{
+}
+
+FIXTURE_TEARDOWN(setns_cross_process)
+{
+}
+
+/*
+ * setns into a child's UTS namespace: when sandboxed with
+ * LANDLOCK_PERMISSION_NAMESPACE_USE denying UTS, the rule-based check applies
+ * regardless of which process created the namespace. This fixture exercises
+ * only CLONE_NEWUTS; the same enforcement applies to every namespace type (see
+ * hook_namespace_install() in security/landlock/ns.c), so per-type variants
+ * would not exercise different code paths.
+ */
+TEST_F(setns_cross_process, setns)
+{
+ int ruleset_fd, ns_fd, status;
+ pid_t child;
+ int pipe_parent[2], pipe_child[2];
+ char buf, path[64];
+
+ disable_caps(_metadata);
+
+ /*
+ * Enable dumpable so the parent can read /proc/<child>/ns/uts. Without
+ * this, ptrace access checks (PTRACE_MODE_READ) prevent opening another
+ * process's namespace entries.
+ */
+ ASSERT_EQ(0, prctl(PR_SET_DUMPABLE, 1, 0, 0, 0));
+
+ ASSERT_EQ(0, pipe2(pipe_parent, O_CLOEXEC));
+ ASSERT_EQ(0, pipe2(pipe_child, O_CLOEXEC));
+
+ child = fork();
+ ASSERT_LE(0, child);
+
+ if (child == 0) {
+ EXPECT_EQ(0, close(pipe_parent[1]));
+ EXPECT_EQ(0, close(pipe_child[0]));
+
+ /* Child: create a UTS namespace. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWUTS));
+
+ drop_caps(_metadata);
+ ASSERT_EQ(0, prctl(PR_SET_DUMPABLE, 1, 0, 0, 0));
+
+ /* Signal parent that the namespace is ready. */
+ ASSERT_EQ(1, write(pipe_child[1], ".", 1));
+
+ /* Wait for parent to finish testing. */
+ ASSERT_EQ(1, read(pipe_parent[0], &buf, 1));
+ _exit(_metadata->exit_code);
+ }
+
+ EXPECT_EQ(0, close(pipe_parent[0]));
+ EXPECT_EQ(0, close(pipe_child[1]));
+
+ /* Wait for child namespace. */
+ ASSERT_EQ(1, read(pipe_child[0], &buf, 1));
+ EXPECT_EQ(0, close(pipe_child[0]));
+
+ /* Open the child's NS FD BEFORE creating the domain. */
+ snprintf(path, sizeof(path), "/proc/%d/ns/uts", child);
+ ns_fd = open(path, O_RDONLY);
+ ASSERT_LE(0, ns_fd);
+
+ if (variant->is_sandboxed) {
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ if (variant->has_rule)
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd, CLONE_NEWUTS));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+ }
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ if (variant->expected_setns) {
+ EXPECT_EQ(-1, setns(ns_fd, CLONE_NEWUTS));
+ EXPECT_EQ(variant->expected_setns, errno);
+ } else {
+ EXPECT_EQ(0, setns(ns_fd, CLONE_NEWUTS));
+ }
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, close(ns_fd));
+
+ /* Release child. */
+ ASSERT_EQ(1, write(pipe_parent[1], ".", 1));
+ EXPECT_EQ(0, close(pipe_parent[1]));
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ ASSERT_EQ(1, WIFEXITED(status));
+ ASSERT_EQ(EXIT_SUCCESS, WEXITSTATUS(status));
+}
+
+/*
+ * Verify that LANDLOCK_PERMISSION_NAMESPACE_USE and
+ * LANDLOCK_PERMISSION_CAPABILITY_USE apply simultaneously: creating/entering a
+ * non-user namespace requires both the namespace type to be allowed AND
+ * CAP_SYS_ADMIN to be allowed. User namespace creation is the exception (no
+ * capable() call from the kernel).
+ */
+TEST(setns_and_create)
+{
+ int ruleset_fd, ns_fd;
+ const struct landlock_ruleset_attr attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_NAMESPACE_USE |
+ LANDLOCK_PERMISSION_CAPABILITY_USE,
+ };
+ const struct landlock_namespace_attr ns_attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .allowed_namespace_types = CLONE_NEWUTS,
+ };
+ const struct landlock_capability_attr cap_attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .allowed_capabilities = (1ULL << CAP_SYS_ADMIN),
+ };
+
+ disable_caps(_metadata);
+
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &ns_attr, 0));
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &cap_attr, 0));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /* UTS unshare: allowed by NS rule + CAP_SYS_ADMIN allowed. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWUTS));
+
+ /* IPC unshare: denied by NS rule (type not allowed). */
+ EXPECT_EQ(-1, unshare(CLONE_NEWIPC));
+ EXPECT_EQ(EPERM, errno);
+
+ /* setns into current UTS: allowed by NS rule. */
+ ns_fd = open("/proc/self/ns/uts", O_RDONLY);
+ ASSERT_LE(0, ns_fd);
+ EXPECT_EQ(0, setns(ns_fd, CLONE_NEWUTS));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, close(ns_fd));
+
+ /*
+ * User namespace creation: only LANDLOCK_PERMISSION_NAMESPACE_USE
+ * needed (no capable() call from the kernel for user NS). Denied
+ * because CLONE_NEWUSER is not in the allowed namespace types.
+ */
+ EXPECT_EQ(-1, unshare(CLONE_NEWUSER));
+ EXPECT_EQ(EPERM, errno);
+}
+
+/*
+ * Verify that LANDLOCK_PERMISSION_CAPABILITY_USE can deny the CAP_SYS_ADMIN
+ * check that the kernel performs before the Landlock namespace hook is reached.
+ * The NS type is allowed but the required capability is not, so the operation
+ * fails on the capability check.
+ *
+ * User namespace creation is the exception: no capable() call, so the operation
+ * succeeds with just LANDLOCK_PERMISSION_NAMESPACE_USE.
+ */
+TEST(two_permissions_cap_denied)
+{
+ const struct landlock_ruleset_attr attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_NAMESPACE_USE |
+ LANDLOCK_PERMISSION_CAPABILITY_USE,
+ };
+ const struct landlock_namespace_attr ns_attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .allowed_namespace_types = CLONE_NEWUTS | CLONE_NEWUSER,
+ };
+ /* CAP_SYS_ADMIN is NOT allowed. */
+ const struct landlock_capability_attr cap_attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .allowed_capabilities = (1ULL << CAP_SYS_CHROOT),
+ };
+ int ruleset_fd, ns_fd;
+
+ disable_caps(_metadata);
+
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &ns_attr, 0));
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &cap_attr, 0));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /*
+ * UTS creation: the process holds CAP_SYS_ADMIN but Landlock denies it
+ * (not in the cap rule), so the kernel's ns_capable(CAP_SYS_ADMIN) gate
+ * fails before the namespace hook is reached.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, unshare(CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+
+ /*
+ * setns into the current UTS namespace: the type is allowed but
+ * CAP_SYS_ADMIN is denied, so the kernel's ns_capable(CAP_SYS_ADMIN)
+ * gate fails before the namespace hook is reached (the setns capability
+ * leg, complementing the unshare/create leg above).
+ */
+ ns_fd = open("/proc/self/ns/uts", O_RDONLY);
+ ASSERT_LE(0, ns_fd);
+ EXPECT_EQ(-1, setns(ns_fd, CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ EXPECT_EQ(0, close(ns_fd));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /*
+ * User NS creation: no capable() call from the kernel, so only
+ * LANDLOCK_PERMISSION_NAMESPACE_USE applies. CLONE_NEWUSER is in the
+ * allowed set, so this succeeds.
+ */
+ EXPECT_EQ(0, unshare(CLONE_NEWUSER));
+}
+
+/*
+ * Combining CLONE_NEWUSER with another namespace type in a single unshare(2)
+ * call does not exempt the non-user namespace from the capability check: the
+ * kernel checks ns_capable(new_user_ns, CAP_SYS_ADMIN) for it, and the
+ * namespace-agnostic capability hook denies it when the domain handles
+ * LANDLOCK_PERMISSION_CAPABILITY_USE without allowing CAP_SYS_ADMIN. Both
+ * namespace types are allowed here, yet the combined call fails atomically,
+ * exactly like the equivalent two-call sequence.
+ */
+TEST(two_permissions_combined_unshare_denied)
+{
+ const struct landlock_ruleset_attr attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_NAMESPACE_USE |
+ LANDLOCK_PERMISSION_CAPABILITY_USE,
+ };
+ /* Both namespace types are allowed; no capability is allowed. */
+ const struct landlock_namespace_attr ns_attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .allowed_namespace_types = CLONE_NEWUSER | CLONE_NEWUTS,
+ };
+ int ruleset_fd;
+
+ disable_caps(_metadata);
+
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &ns_attr, 0));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /*
+ * CAP_SYS_ADMIN is held so the kernel's commoncap check passes (via
+ * ownership of the new user namespace) and the denial is attributable
+ * to Landlock. The UTS leg needs CAP_SYS_ADMIN, which the capability
+ * hook denies, so the atomic unshare(2) fails and no namespace is
+ * created.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, unshare(CLONE_NEWUSER | CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /* The user namespace alone has no capability check, so it succeeds. */
+ EXPECT_EQ(0, unshare(CLONE_NEWUSER));
+}
+
+/*
+ * Positive complement to two_permissions_combined_unshare_denied: allowing
+ * CAP_SYS_ADMIN lets the same combined unshare(CLONE_NEWUSER | CLONE_NEWUTS)
+ * succeed, proving the denial above comes from the missing capability allowance
+ * and not the namespace rule. Run in a child because a successful unshare(2)
+ * mutates the caller.
+ */
+TEST(two_permissions_combined_unshare_allowed)
+{
+ const struct landlock_ruleset_attr attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_NAMESPACE_USE |
+ LANDLOCK_PERMISSION_CAPABILITY_USE,
+ };
+ const struct landlock_namespace_attr ns_attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .allowed_namespace_types = CLONE_NEWUSER | CLONE_NEWUTS,
+ };
+ const struct landlock_capability_attr cap_attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .allowed_capabilities = (1ULL << CAP_SYS_ADMIN),
+ };
+ int ruleset_fd, status;
+ pid_t child;
+
+ disable_caps(_metadata);
+
+ child = fork();
+ ASSERT_LE(0, child);
+ if (child == 0) {
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0,
+ landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &ns_attr, 0));
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd,
+ LANDLOCK_RULE_CAPABILITY,
+ &cap_attr, 0));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, unshare(CLONE_NEWUSER | CLONE_NEWUTS));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ _exit(_metadata->exit_code);
+ }
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ ASSERT_EQ(1, WIFEXITED(status));
+ ASSERT_EQ(EXIT_SUCCESS, WEXITSTATUS(status));
+}
+
+/*
+ * Inverse of two_permissions_cap_denied: the required capability is allowed but
+ * the namespace type is not, so the Landlock namespace hook denies the
+ * operation (the capability leg passes). A positive control (an allowed type
+ * creates successfully with the same allowed capability) proves CAP_SYS_ADMIN
+ * is genuinely exercisable, so the UTS denial is attributable to the namespace
+ * hook rather than the capability hook.
+ */
+TEST(two_permissions_ns_denied)
+{
+ const struct landlock_ruleset_attr attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_NAMESPACE_USE |
+ LANDLOCK_PERMISSION_CAPABILITY_USE,
+ };
+ /* CLONE_NEWUTS is NOT allowed; CLONE_NEWNET is. */
+ const struct landlock_namespace_attr ns_attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .allowed_namespace_types = CLONE_NEWNET,
+ };
+ const struct landlock_capability_attr cap_attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .allowed_capabilities = (1ULL << CAP_SYS_ADMIN),
+ };
+ int ruleset_fd;
+
+ disable_caps(_metadata);
+
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &ns_attr, 0));
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &cap_attr, 0));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /*
+ * Positive control: CLONE_NEWNET is allowed and CAP_SYS_ADMIN is
+ * allowed, so the kernel's ns_capable() gate passes and the namespace
+ * hook allows it.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, unshare(CLONE_NEWNET));
+
+ /*
+ * CAP_SYS_ADMIN is allowed (ns_capable passes), but CLONE_NEWUTS is not
+ * in the allowed namespace types, so the Landlock namespace hook denies
+ * it.
+ */
+ EXPECT_EQ(-1, unshare(CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/*
+ * Mount namespace setns is unique: the kernel checks both CAP_SYS_ADMIN and
+ * CAP_SYS_CHROOT in mntns_install(). Verify that allowing CAP_SYS_ADMIN alone
+ * is not sufficient.
+ */
+TEST(two_permissions_mnt_setns)
+{
+ const struct landlock_ruleset_attr attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_NAMESPACE_USE |
+ LANDLOCK_PERMISSION_CAPABILITY_USE,
+ };
+ const struct landlock_namespace_attr ns_attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .allowed_namespace_types = CLONE_NEWNS,
+ };
+ const struct landlock_capability_attr cap_admin = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .allowed_capabilities = (1ULL << CAP_SYS_ADMIN),
+ };
+ const struct landlock_capability_attr cap_admin_chroot = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .allowed_capabilities = (1ULL << CAP_SYS_ADMIN) |
+ (1ULL << CAP_SYS_CHROOT),
+ };
+ int ruleset_fd, ns_fd;
+
+ disable_caps(_metadata);
+
+ /* Layer 1: allow mount NS + CAP_SYS_ADMIN only (no CAP_SYS_CHROOT). */
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &ns_attr, 0));
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &cap_admin, 0));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ ns_fd = open("/proc/self/ns/mnt", O_RDONLY);
+ ASSERT_LE(0, ns_fd);
+
+ /*
+ * Fails: mntns_install() checks CAP_SYS_ADMIN (allowed) then
+ * CAP_SYS_CHROOT (denied by LANDLOCK_PERMISSION_CAPABILITY_USE).
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ EXPECT_EQ(-1, setns(ns_fd, CLONE_NEWNS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ clear_cap(_metadata, CAP_SYS_CHROOT);
+
+ /* Layer 2: also allows CAP_SYS_CHROOT. */
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &ns_attr, 0));
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &cap_admin_chroot, 0));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /*
+ * Still fails: layer 1 still denies CAP_SYS_CHROOT. Landlock layer
+ * stacking means the most restrictive layer wins.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ EXPECT_EQ(-1, setns(ns_fd, CLONE_NEWNS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ clear_cap(_metadata, CAP_SYS_CHROOT);
+ EXPECT_EQ(0, close(ns_fd));
+}
+
+/*
+ * Positive complement to two_permissions_mnt_setns: a single layer that allows
+ * the mount namespace type AND both CAP_SYS_ADMIN and CAP_SYS_CHROOT lets
+ * setns(CLONE_NEWNS) into the process's own mount namespace succeed. Proves
+ * the denial legs above are caused by the withheld CAP_SYS_CHROOT, not an
+ * unconditional block of mount setns.
+ */
+TEST(two_permissions_mnt_setns_allowed)
+{
+ const struct landlock_ruleset_attr attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_NAMESPACE_USE |
+ LANDLOCK_PERMISSION_CAPABILITY_USE,
+ };
+ const struct landlock_namespace_attr ns_attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .allowed_namespace_types = CLONE_NEWNS,
+ };
+ const struct landlock_capability_attr cap_attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .allowed_capabilities = (1ULL << CAP_SYS_ADMIN) |
+ (1ULL << CAP_SYS_CHROOT),
+ };
+ int ruleset_fd, ns_fd;
+
+ disable_caps(_metadata);
+
+ /* Open the NS FD before enforcing the domain. */
+ ns_fd = open("/proc/self/ns/mnt", O_RDONLY);
+ ASSERT_LE(0, ns_fd);
+
+ ruleset_fd = landlock_create_ruleset(&attr, sizeof(attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &ns_attr, 0));
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &cap_attr, 0));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /*
+ * mntns_install() checks CAP_SYS_ADMIN then CAP_SYS_CHROOT (both
+ * allowed by the rule) and the mount NS type is allowed, so setns into
+ * the process's own mount namespace succeeds.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ EXPECT_EQ(0, setns(ns_fd, CLONE_NEWNS));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ clear_cap(_metadata, CAP_SYS_CHROOT);
+ EXPECT_EQ(0, close(ns_fd));
+}
+
+/*
+ * setns() through a namespace fd inherited across fork() is still governed by
+ * the child's own Landlock domain: the child enforces a domain denying
+ * CLONE_NEWUTS, then joining the inherited /proc/self/ns/uts fd is denied.
+ */
+TEST(setns_inherited_fd)
+{
+ int ns_fd, status;
+ pid_t child;
+
+ disable_caps(_metadata);
+
+ /* Open the UTS ns fd before fork() so the child inherits it. */
+ ns_fd = open("/proc/self/ns/uts", O_RDONLY);
+ ASSERT_LE(0, ns_fd);
+
+ child = fork();
+ ASSERT_LE(0, child);
+ if (child == 0) {
+ int ruleset_fd;
+
+ /*
+ * Baseline: joining the inherited fd works before the domain,
+ * so the later EPERM is attributable to Landlock.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, setns(ns_fd, CLONE_NEWUTS));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ /* No rule allows UTS. */
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, setns(ns_fd, CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ _exit(_metadata->exit_code);
+ return;
+ }
+
+ EXPECT_EQ(0, close(ns_fd));
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ if (WIFSIGNALED(status) || !WIFEXITED(status) ||
+ WEXITSTATUS(status) != EXIT_SUCCESS)
+ _metadata->exit_code = KSFT_FAIL;
+}
+
+/* Audit tests */
+
+static int matches_log_ns_create(int audit_fd, __u64 ns_type)
+{
+ static const char log_template[] = REGEX_LANDLOCK_PREFIX
+ " blockers=namespace\\.use namespace_type=0x%x namespace_id=0$";
+ char log_match[sizeof(log_template) + 10];
+ int log_match_len;
+
+ log_match_len = snprintf(log_match, sizeof(log_match), log_template,
+ (unsigned int)ns_type);
+ if (log_match_len >= sizeof(log_match))
+ return -E2BIG;
+
+ return audit_match_record(audit_fd, AUDIT_LANDLOCK_ACCESS, log_match,
+ NULL);
+}
+
+static int matches_log_ns_setns(int audit_fd, __u64 ns_type, __u64 ns_id)
+{
+ static const char log_template[] = REGEX_LANDLOCK_PREFIX
+ " blockers=namespace\\.use namespace_type=0x%x namespace_id=%llu$";
+ char log_match[sizeof(log_template) + 32];
+ int log_match_len;
+
+ log_match_len = snprintf(log_match, sizeof(log_match), log_template,
+ (unsigned int)ns_type,
+ (unsigned long long)ns_id);
+ if (log_match_len >= sizeof(log_match))
+ return -E2BIG;
+
+ return audit_match_record(audit_fd, AUDIT_LANDLOCK_ACCESS, log_match,
+ NULL);
+}
+
+FIXTURE(ns_audit)
+{
+ struct audit_filter audit_filter;
+ int audit_fd;
+};
+
+FIXTURE_SETUP(ns_audit)
+{
+ ASSERT_TRUE(is_in_init_user_ns());
+
+ disable_caps(_metadata);
+
+ set_cap(_metadata, CAP_AUDIT_CONTROL);
+ self->audit_fd = audit_init_with_exe_filter(&self->audit_filter);
+ EXPECT_LE(0, self->audit_fd);
+ clear_cap(_metadata, CAP_AUDIT_CONTROL);
+}
+
+FIXTURE_TEARDOWN(ns_audit)
+{
+ set_cap(_metadata, CAP_AUDIT_CONTROL);
+ EXPECT_EQ(0, audit_cleanup(self->audit_fd, &self->audit_filter));
+ clear_cap(_metadata, CAP_AUDIT_CONTROL);
+}
+
+/*
+ * Verifies that a denied namespace creation produces the expected audit record
+ * with the namespace.use blocker string and namespace_type.
+ */
+TEST_F(ns_audit, create_denied)
+{
+ struct audit_records records;
+ int ruleset_fd;
+ __u64 domain_id;
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, unshare(CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ EXPECT_EQ(0, matches_log_ns_create(self->audit_fd, CLONE_NEWUTS));
+
+ /*
+ * The domain allocation record is emitted in the same event as the
+ * first denial; anchor its status=allocated and enforcing pid. Both
+ * access and domain records are now consumed, so none remain.
+ */
+ EXPECT_EQ(0, matches_log_domain_allocated(self->audit_fd, getpid(),
+ &domain_id));
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+}
+
+TEST_F(ns_audit, create_allowed)
+{
+ struct audit_records records;
+ int ruleset_fd;
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd, CLONE_NEWUTS));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, unshare(CLONE_NEWUTS));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /* No records: allowed operations never trigger audit logging. */
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+}
+
+TEST_F(ns_audit, setns_allowed)
+{
+ struct audit_records records;
+ int ruleset_fd, ns_fd;
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd, CLONE_NEWUTS));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ ns_fd = open("/proc/self/ns/uts", O_RDONLY);
+ ASSERT_LE(0, ns_fd);
+
+ /* Allowed: should succeed with no audit record. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, setns(ns_fd, CLONE_NEWUTS));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, close(ns_fd));
+
+ /* No records: allowed setns never triggers audit logging. */
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+}
+
+TEST_F(ns_audit, setns_denied)
+{
+ struct audit_records records;
+ int ruleset_fd, ns_fd;
+ __u64 ns_id, domain_id;
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ /* No rule allows UTS -> denied. */
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ ns_fd = open("/proc/self/ns/uts", O_RDONLY);
+ ASSERT_LE(0, ns_fd);
+
+ /* The audit record logs the target namespace's exact ns_id. */
+ ASSERT_EQ(0, ioctl(ns_fd, NS_GET_ID, &ns_id));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, setns(ns_fd, CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, close(ns_fd));
+
+ /* Verify the audit record for setns denial, anchored to the ns_id. */
+ EXPECT_EQ(0, matches_log_ns_setns(self->audit_fd, CLONE_NEWUTS, ns_id));
+
+ /*
+ * The domain allocation record is emitted in the same event as the
+ * first denial; anchor its status=allocated and enforcing pid. Both
+ * access and domain records are now consumed, so none remain.
+ */
+ EXPECT_EQ(0, matches_log_domain_allocated(self->audit_fd, getpid(),
+ &domain_id));
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+}
+
+/*
+ * A quieted denial is still denied (EPERM); only its audit record is
+ * suppressed. Quiet-only rule (empty allowed set).
+ */
+TEST_F(ns_audit, create_quieted)
+{
+ struct audit_records records;
+ int ruleset_fd;
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_ns_rule_full(ruleset_fd, 0, CLONE_NEWUTS));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, unshare(CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /* The denial is suppressed: no access record. */
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+}
+
+/*
+ * A quieted setns denial is still denied (EPERM); only its audit record is
+ * suppressed. The install-hook mirror of create_quieted.
+ */
+TEST_F(ns_audit, setns_quieted)
+{
+ struct audit_records records;
+ int ruleset_fd, ns_fd;
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_ns_rule_full(ruleset_fd, 0, CLONE_NEWUTS));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ ns_fd = open("/proc/self/ns/uts", O_RDONLY);
+ ASSERT_LE(0, ns_fd);
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, setns(ns_fd, CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, close(ns_fd));
+
+ /* The denial is suppressed: no access record. */
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+}
+
+/*
+ * Quiet is per-member: quieting one namespace type does not silence the denial
+ * of another type denied by the same layer.
+ */
+TEST_F(ns_audit, quiet_is_per_member)
+{
+ struct audit_records records;
+ int ruleset_fd;
+ __u64 domain_id;
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ /*
+ * Quiet CLONE_NEWNET only; CLONE_NEWUTS is neither allowed nor quiet.
+ */
+ ASSERT_EQ(0, add_ns_rule_full(ruleset_fd, 0, CLONE_NEWNET));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, unshare(CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /* The unquieted CLONE_NEWUTS denial is still logged. */
+ EXPECT_EQ(0, matches_log_ns_create(self->audit_fd, CLONE_NEWUTS));
+ EXPECT_EQ(0, matches_log_domain_allocated(self->audit_fd, getpid(),
+ &domain_id));
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+}
+
+/*
+ * Quieting an unknown namespace bit has no effect: a rule that quiets only a
+ * bit the running kernel does not know still logs the denial of a known type.
+ * Mirrors the forward-compatibility contract for the quiet mask.
+ */
+TEST_F(ns_audit, quiet_unknown_bit_no_effect)
+{
+ struct audit_records records;
+ int ruleset_fd;
+ __u64 domain_id;
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ /* Quiet only an unknown bit (bit 31 is not a CLONE_NEW* value). */
+ ASSERT_EQ(0, add_ns_rule_full(ruleset_fd, 0, (1ULL << 31)));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, unshare(CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /* The unknown quiet bit does not suppress the known denial. */
+ EXPECT_EQ(0, matches_log_ns_create(self->audit_fd, CLONE_NEWUTS));
+ EXPECT_EQ(0, matches_log_domain_allocated(self->audit_fd, getpid(),
+ &domain_id));
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+}
+
+/* A single rule can both allow one member and quiet the denial of another. */
+TEST_F(ns_audit, allow_and_quiet)
+{
+ struct audit_records records;
+ int ruleset_fd;
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_ns_rule_full(ruleset_fd, CLONE_NEWNET, CLONE_NEWUTS));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /* Allowed member: succeeds. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, unshare(CLONE_NEWNET));
+ /* Quieted member: denied but not logged. */
+ EXPECT_EQ(-1, unshare(CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+}
+
+/*
+ * Quieting a member that is also allowed is inert: an allowed member is never
+ * denied, so its quiet bit has no effect.
+ */
+TEST_F(ns_audit, allow_and_quiet_same_member)
+{
+ struct audit_records records;
+ int ruleset_fd;
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ /* Same member both allowed and quieted. */
+ ASSERT_EQ(0, add_ns_rule_full(ruleset_fd, CLONE_NEWNET, CLONE_NEWNET));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /* The member is allowed (quiet is inert), so it succeeds. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, unshare(CLONE_NEWNET));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /* Nothing denied, nothing logged. */
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+}
+
+/*
+ * Only the youngest denying layer decides quieting: a parent that quiets a
+ * member cannot silence a denial by a deeper layer that handles but does not
+ * quiet it.
+ */
+TEST_F(ns_audit, quiet_youngest_layer_wins)
+{
+ struct audit_records records;
+ int ruleset_fd;
+
+ /* Parent layer: handles namespaces and quiets CLONE_NEWUTS. */
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_ns_rule_full(ruleset_fd, 0, CLONE_NEWUTS));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /* Child layer: handles namespaces but does not quiet CLONE_NEWUTS. */
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, unshare(CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /* The child layer denies and does not quiet: logged. */
+ EXPECT_EQ(0, matches_log_ns_create(self->audit_fd, CLONE_NEWUTS));
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+}
+
+/*
+ * A parent layer that quiets and denies a member suppresses its own denial even
+ * when it is not the youngest layer: a child layer allows the member, so the
+ * parent is the youngest (and only) denying layer and its stored quiet mask
+ * fires. Complements quiet_youngest_layer_wins.
+ */
+TEST_F(ns_audit, quiet_parent_only_denying_layer)
+{
+ struct audit_records records;
+ int ruleset_fd;
+
+ /* Parent layer: quiet-only, so it denies CLONE_NEWUTS. */
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_ns_rule_full(ruleset_fd, 0, CLONE_NEWUTS));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /* Child layer: allows CLONE_NEWUTS, so only the parent denies it. */
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd, CLONE_NEWUTS));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, unshare(CLONE_NEWUTS));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /* The parent quiets and is the youngest denying layer: suppressed. */
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+}
+
+/* Trace tests */
+
+/* clang-format off */
+FIXTURE(ns_trace) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(ns_trace)
+{
+ int ret;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+
+ ret = tracefs_fixture_setup();
+ if (ret) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_NAMESPACE_ENABLE,
+ true));
+ ASSERT_EQ(0, tracefs_enable_event(
+ TRACEFS_DENY_PERMISSION_NAMESPACE_ENABLE, true));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(ns_trace)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, false);
+ tracefs_enable_event(TRACEFS_ADD_RULE_NAMESPACE_ENABLE, false);
+ tracefs_enable_event(TRACEFS_DENY_PERMISSION_NAMESPACE_ENABLE, false);
+ tracefs_fixture_teardown();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/*
+ * Check the permission fields and version history, including two rejected calls
+ * and two successful calls that do not extend the effective policy: a repeated
+ * rule and an unknown-only rule.
+ */
+TEST_F(ns_trace, rule_events)
+{
+ const char *const version_1 =
+ REGEX_ADD_RULE_NAMESPACE_VERSION(TRACE_TASK, "1");
+ const char *const version_2 =
+ REGEX_ADD_RULE_NAMESPACE_VERSION(TRACE_TASK, "2");
+ const char *const version_3 =
+ REGEX_ADD_RULE_NAMESPACE_VERSION(TRACE_TASK, "3");
+ const struct landlock_namespace_attr valid_attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .allowed_namespace_types = CLONE_NEWUTS,
+ .quiet_namespace_types = CLONE_NEWNET,
+ };
+ const struct landlock_namespace_attr invalid_attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .allowed_namespace_types = CLONE_NEWUTS,
+ };
+ char expected[32], field[64];
+ char *buf;
+ int ruleset_fd;
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_ns_rule_full(ruleset_fd, CLONE_NEWUTS, CLONE_NEWNET));
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &invalid_attr, 0));
+ ASSERT_EQ(EINVAL, errno);
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &valid_attr, LANDLOCK_ADD_RULE_QUIET));
+ ASSERT_EQ(EINVAL, errno);
+ ASSERT_EQ(0, add_ns_rule_full(ruleset_fd, CLONE_NEWUTS, CLONE_NEWNET));
+ ASSERT_EQ(0, add_ns_rule(ruleset_fd, 1ULL << 63));
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ ASSERT_EQ(1,
+ tracefs_count_matches(buf, REGEX_CREATE_RULESET(TRACE_TASK)));
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_CREATE_RULESET(TRACE_TASK),
+ "handled_permissions", field, sizeof(field)));
+ EXPECT_STREQ("namespace.use", field);
+
+ ASSERT_EQ(3,
+ tracefs_count_matches(buf, "landlock_add_rule_namespace: "));
+ ASSERT_EQ(3, tracefs_count_matches(
+ buf, REGEX_ADD_RULE_NAMESPACE(TRACE_TASK)));
+ EXPECT_EQ(1, tracefs_count_matches(buf, version_1));
+ EXPECT_EQ(1, tracefs_count_matches(buf, version_2));
+ EXPECT_EQ(1, tracefs_count_matches(buf, version_3));
+
+ ASSERT_EQ(0, tracefs_extract_field(buf, version_1, "permissions", field,
+ sizeof(field)));
+ EXPECT_STREQ("namespace.use", field);
+ ASSERT_EQ(0, tracefs_extract_field(buf, version_1,
+ "allowed_namespace_types", field,
+ sizeof(field)));
+ /* Trace fields use raw CLONE_NEW* values, not internal bit indexes. */
+ snprintf(expected, sizeof(expected), "0x%llx",
+ (unsigned long long)CLONE_NEWUTS);
+ EXPECT_STREQ(expected, field);
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, version_1, "quiet_namespace_types",
+ field, sizeof(field)));
+ snprintf(expected, sizeof(expected), "0x%llx",
+ (unsigned long long)CLONE_NEWNET);
+ EXPECT_STREQ(expected, field);
+
+ /* The rejected calls emit no event and do not consume version 2. */
+ ASSERT_EQ(0, tracefs_extract_field(buf, version_2, "permissions", field,
+ sizeof(field)));
+ EXPECT_STREQ("namespace.use", field);
+ ASSERT_EQ(0, tracefs_extract_field(buf, version_2,
+ "allowed_namespace_types", field,
+ sizeof(field)));
+ snprintf(expected, sizeof(expected), "0x%llx",
+ (unsigned long long)CLONE_NEWUTS);
+ EXPECT_STREQ(expected, field);
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, version_2, "quiet_namespace_types",
+ field, sizeof(field)));
+ snprintf(expected, sizeof(expected), "0x%llx",
+ (unsigned long long)CLONE_NEWNET);
+ EXPECT_STREQ(expected, field);
+
+ /* The unknown-only call succeeds but contributes no effective bits. */
+ ASSERT_EQ(0, tracefs_extract_field(buf, version_3, "permissions", field,
+ sizeof(field)));
+ EXPECT_STREQ("namespace.use", field);
+ ASSERT_EQ(0, tracefs_extract_field(buf, version_3,
+ "allowed_namespace_types", field,
+ sizeof(field)));
+ EXPECT_STREQ("0x0", field);
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, version_3, "quiet_namespace_types",
+ field, sizeof(field)));
+ EXPECT_STREQ("0x0", field);
+
+ free(buf);
+}
+
+static void test_ns_trace_denial(struct __test_metadata *const _metadata,
+ const bool use_setns, const bool quiet)
+{
+ char expected[32], field[64];
+ int ruleset_fd, ns_fd = -1, ret;
+ __u64 expected_ns_id = 0;
+ char *buf;
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ if (use_setns) {
+ ns_fd = open("/proc/self/ns/uts", O_RDONLY | O_CLOEXEC);
+ ASSERT_LE(0, ns_fd);
+ ASSERT_EQ(0, ioctl(ns_fd, NS_GET_ID, &expected_ns_id));
+ }
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ if (quiet)
+ ASSERT_EQ(0, add_ns_rule_full(ruleset_fd, 0, CLONE_NEWUTS));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ if (use_setns)
+ ret = setns(ns_fd, CLONE_NEWUTS);
+ else
+ ret = unshare(CLONE_NEWUTS);
+ EXPECT_EQ(-1, ret);
+ EXPECT_EQ(EPERM, errno);
+
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ if (ns_fd >= 0)
+ EXPECT_EQ(0, close(ns_fd));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+ ASSERT_EQ(1, tracefs_count_matches(
+ buf, REGEX_DENY_PERMISSION_NAMESPACE(TRACE_TASK)))
+ {
+ TH_LOG("Expected one namespace denial event\n%s", buf);
+ }
+
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PERMISSION_NAMESPACE(TRACE_TASK),
+ "domain", field, sizeof(field)));
+ EXPECT_STRNE("0", field);
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PERMISSION_NAMESPACE(TRACE_TASK),
+ "same_exec", field, sizeof(field)));
+ EXPECT_STREQ("1", field);
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PERMISSION_NAMESPACE(TRACE_TASK),
+ "logged", field, sizeof(field)));
+ EXPECT_STREQ(quiet ? "0" : "1", field);
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PERMISSION_NAMESPACE(TRACE_TASK),
+ "blockers", field, sizeof(field)));
+ EXPECT_STREQ("use", field);
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PERMISSION_NAMESPACE(TRACE_TASK),
+ "namespace_type", field, sizeof(field)));
+ snprintf(expected, sizeof(expected), "0x%llx",
+ (unsigned long long)CLONE_NEWUTS);
+ EXPECT_STREQ(expected, field);
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PERMISSION_NAMESPACE(TRACE_TASK),
+ "namespace_id", field, sizeof(field)));
+ EXPECT_EQ(use_setns ? expected_ns_id : 0, strtoull(field, NULL, 10));
+
+ free(buf);
+}
+
+TEST_F(ns_trace, deny_create)
+{
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+ test_ns_trace_denial(_metadata, false, false);
+}
+
+TEST_F(ns_trace, deny_create_quiet)
+{
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+ test_ns_trace_denial(_metadata, false, true);
+}
+
+TEST_F(ns_trace, deny_setns)
+{
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+ test_ns_trace_denial(_metadata, true, false);
+}
+
+/* clang-format off */
+FIXTURE(ns_proc_open) {
+ /* clang-format on */
+ struct audit_filter audit_filter;
+ int audit_fd;
+};
+
+FIXTURE_VARIANT(ns_proc_open)
+{
+ __u64 ns_type;
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_proc_open, mnt) {
+ /* clang-format on */
+ .ns_type = CLONE_NEWNS,
+};
+
+/* clang-format off */
+FIXTURE_VARIANT_ADD(ns_proc_open, user) {
+ /* clang-format on */
+ .ns_type = CLONE_NEWUSER,
+};
+
+FIXTURE_SETUP(ns_proc_open)
+{
+ ASSERT_TRUE(is_in_init_user_ns());
+
+ disable_caps(_metadata);
+
+ set_cap(_metadata, CAP_AUDIT_CONTROL);
+ self->audit_fd = audit_init_with_exe_filter(&self->audit_filter);
+ EXPECT_LE(0, self->audit_fd);
+ clear_cap(_metadata, CAP_AUDIT_CONTROL);
+}
+
+FIXTURE_TEARDOWN(ns_proc_open)
+{
+ set_cap(_metadata, CAP_AUDIT_CONTROL);
+ EXPECT_EQ(0, audit_cleanup(self->audit_fd, &self->audit_filter));
+ clear_cap(_metadata, CAP_AUDIT_CONTROL);
+}
+
+/*
+ * Opening /proc/self/ns/<type> only acquires a procfs reference, not membership
+ * or an acquired fd of the kind LANDLOCK_PERMISSION_NAMESPACE_USE gates.
+ * Verify the open is unrestricted even when the permission is handled with no
+ * rules.
+ */
+TEST_F(ns_proc_open, open_unrestricted)
+{
+ char proc_path[NS_PROC_PATH_MAX];
+ struct audit_records records;
+ int ruleset_fd, fd;
+
+ ASSERT_TRUE(
+ ns_is_supported(variant->ns_type, proc_path, sizeof(proc_path)))
+ {
+ TH_LOG("Namespace type 0x%llx not supported",
+ (unsigned long long)variant->ns_type);
+ }
+
+ ruleset_fd = create_ns_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ fd = open(proc_path, O_RDONLY);
+ ASSERT_LE(0, fd)
+ {
+ TH_LOG("open(%s) failed: %s", proc_path, strerror(errno));
+ }
+
+ /* No Landlock denial. */
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+
+ EXPECT_EQ(0, close(fd));
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/landlock/trace.h b/tools/testing/selftests/landlock/trace.h
index fe2d1a0d8de3..568c13d8eee4 100644
--- a/tools/testing/selftests/landlock/trace.h
+++ b/tools/testing/selftests/landlock/trace.h
@@ -31,6 +31,8 @@
TRACEFS_LANDLOCK_DIR "/landlock_add_rule_path_beneath/enable"
#define TRACEFS_ADD_RULE_NET_PORT_ENABLE \
TRACEFS_LANDLOCK_DIR "/landlock_add_rule_net_port/enable"
+#define TRACEFS_ADD_RULE_NAMESPACE_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_add_rule_namespace/enable"
#define TRACEFS_CHECK_RULE_FS_ENABLE \
TRACEFS_LANDLOCK_DIR "/landlock_check_rule_inode/enable"
#define TRACEFS_CHECK_RULE_NET_ENABLE \
@@ -39,6 +41,8 @@
TRACEFS_LANDLOCK_DIR "/landlock_deny_access_fs/enable"
#define TRACEFS_DENY_ACCESS_NET_ENABLE \
TRACEFS_LANDLOCK_DIR "/landlock_deny_access_net/enable"
+#define TRACEFS_DENY_PERMISSION_NAMESPACE_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_deny_permission_namespace/enable"
#define TRACEFS_DENY_PTRACE_ENABLE \
TRACEFS_LANDLOCK_DIR "/landlock_deny_ptrace/enable"
#define TRACEFS_DENY_SCOPE_SIGNAL_ENABLE \
@@ -95,6 +99,17 @@
"access_rights=[a-z_|]* " \
"port=[0-9]\\+$"
+#define REGEX_ADD_RULE_NAMESPACE_VERSION(task, version) \
+ TRACE_PREFIX(task) \
+ "landlock_add_rule_namespace: " \
+ "ruleset=[0-9a-f]\\+\\." version " " \
+ "permissions=[a-z._|]* " \
+ "allowed_namespace_types=0x[0-9a-f]\\+ " \
+ "quiet_namespace_types=0x[0-9a-f]\\+$"
+
+#define REGEX_ADD_RULE_NAMESPACE(task) \
+ REGEX_ADD_RULE_NAMESPACE_VERSION(task, "[0-9]\\+")
+
#define REGEX_CREATE_RULESET(task) \
TRACE_PREFIX(task) \
"landlock_create_ruleset: " \
@@ -148,6 +163,16 @@
"blockers=[a-z_|]* " \
"port=-\\?[0-9]\\+$"
+#define REGEX_DENY_PERMISSION_NAMESPACE(task) \
+ TRACE_PREFIX(task) \
+ "landlock_deny_permission_namespace: " \
+ "domain=[0-9a-f]\\+ " \
+ "same_exec=[01] " \
+ "logged=[01] " \
+ "blockers=[a-z_|]* " \
+ "namespace_type=0x[0-9a-f]\\+ " \
+ "namespace_id=[0-9]\\+$"
+
#define REGEX_DENY_PTRACE(task) \
TRACE_PREFIX(task) \
"landlock_deny_ptrace: " \
diff --git a/tools/testing/selftests/landlock/wrappers.h b/tools/testing/selftests/landlock/wrappers.h
index 65548323e45d..e6fe46b7c2cc 100644
--- a/tools/testing/selftests/landlock/wrappers.h
+++ b/tools/testing/selftests/landlock/wrappers.h
@@ -9,6 +9,7 @@
#define _GNU_SOURCE
#include <linux/landlock.h>
+#include <linux/sched.h>
#include <sys/syscall.h>
#include <sys/types.h>
#include <unistd.h>
@@ -45,3 +46,31 @@ static inline pid_t sys_gettid(void)
{
return syscall(__NR_gettid);
}
+
+static inline pid_t sys_clone3(struct clone_args *args, size_t size)
+{
+ return syscall(__NR_clone3, args, size);
+}
+
+static inline int sys_open_tree(int dfd, const char *filename,
+ unsigned int flags)
+{
+ return syscall(__NR_open_tree, dfd, filename, flags);
+}
+
+static inline int sys_fsopen(const char *fsname, unsigned int flags)
+{
+ return syscall(__NR_fsopen, fsname, flags);
+}
+
+static inline int sys_fsconfig(int fs_fd, unsigned int cmd, const char *key,
+ const void *value, int aux)
+{
+ return syscall(__NR_fsconfig, fs_fd, cmd, key, value, aux);
+}
+
+static inline int sys_fsmount(int fs_fd, unsigned int flags,
+ unsigned int attr_flags)
+{
+ return syscall(__NR_fsmount, fs_fd, flags, attr_flags);
+}
--
2.55.0
^ permalink raw reply [flat|nested] 9+ messages in thread* [PATCH v4 6/8] selftests/landlock: Add capability restriction tests
2026-10-02 12:43 [PATCH v4 0/8] Landlock: Namespace and capability control Mickaël Salaün
` (4 preceding siblings ...)
2026-10-02 12:43 ` [PATCH v4 5/8] selftests/landlock: Add namespace restriction tests Mickaël Salaün
@ 2026-10-02 12:43 ` Mickaël Salaün
2026-10-02 12:44 ` [PATCH v4 7/8] samples/landlock: Add capability and namespace restriction support Mickaël Salaün
2026-10-02 12:44 ` [PATCH v4 8/8] landlock: Add documentation for capability and namespace restrictions Mickaël Salaün
7 siblings, 0 replies; 9+ messages in thread
From: Mickaël Salaün @ 2026-10-02 12:43 UTC (permalink / raw)
To: Christian Brauner, Günther Noack, Paul Moore, Serge E . Hallyn
Cc: Mickaël Salaün, Daniel Durning, Jonathan Corbet,
Justin Suess, Lennart Poettering, Mikhail Ivanov,
Nicolas Bouchinet, Shervin Oloumi, Tingmao Wang, kernel-team,
linux-fsdevel, linux-kernel, linux-security-module
Add tests to exercise LANDLOCK_PERMISSION_CAPABILITY_USE enforcement. A
sandboxed process is denied a handled capability when no rule grants it,
and an explicit rule restores it. Unknown capability values above
CAP_LAST_CAP are accepted at rule-add time yet have no runtime effect,
because deny-by-default still applies once the domain is enforced.
Stacking tests cover the per-layer allow/deny combinations, including a
layer that does not handle the permission, and invalid rule attributes
return the expected errors.
Three tests exercise non-standard capability contexts: enforcing a
domain via CAP_SYS_ADMIN authorization without no_new_privs, restricting
capabilities gained through the kernel's user namespace ownership bypass
(cap_capable_helper), and checking CAP_SYS_ADMIN against a descendant
user namespace, with an unsandboxed positive control, to confirm the
decision is independent of the target.
Audit tests verify denied and allowed capabilities, per-member quiet
interactions, and CAP_OPT_NOAUDIT suppression. Trace tests cover
accepted, rejected, and no-op rule adds, and normal, quiet, and
CAP_OPT_NOAUDIT denials.
Test coverage for security/landlock is 91.8% of 2864 lines according to
LLVM 22.
Cc: Christian Brauner <brauner@kernel.org>
Cc: Günther Noack <gnoack@google.com>
Cc: Paul Moore <paul@paul-moore.com>
Cc: Serge E. Hallyn <serge@hallyn.com>
Signed-off-by: Mickaël Salaün <mic@digikod.net>
---
Changes since v3:
https://patch.msgid.link/20260726161400.3010511-11-mic@digikod.net
- Cover the EFAULT result for an invalid capability rule pointer.
- Drop CAP_AUDIT_CONTROL after cleaning up the audit fixture.
- Isolate cap_stacking and cap_audit in private UTS namespaces before
their successful sethostname() calls.
- Add trace tests for successful and rejected rules, effective no-ops,
and normal, quiet, and CAP_OPT_NOAUDIT denials.
- Extend audit coverage for normal-denial controls, same-member
allow/quiet overlap, and youngest-denier behavior.
- Test target-user-namespace independence with an unsandboxed positive
control.
Changes since v2:
https://patch.msgid.link/20260527181127.879771-8-mic@digikod.net
- Rebased for ABI 11 and the renamed capability rule attribute (perm
plus allowed_capabilities/quiet_capabilities).
- Add per-member quiet audit tests for the capability permission:
cap_audit.quieted, plus quiet-only and allow+quiet legs in
add_rule_bad_attr and the add_cap_rule_full helper.
- add_rule_bad_attr: drop the dead quiet_capabilities reset after the
allow+quiet rule and reset it explicitly where the following rule
needs an allow-only attr, clarifying intent.
- Strengthen the audit tests with positive controls: cap_audit.quieted
and cap_audit.noaudit_probe_not_logged now also assert that a normal
(non-quiet, non-noaudit) capability denial in the same domain IS
logged, proving suppression is not global. cap_audit.quieted also
asserts the quieted capability is still denied (EPERM), not merely
unlogged.
- Add cap_audit.quiet_unknown_bit_no_effect (an unknown quiet bit does
not suppress a known capability denial).
Changes since v1:
https://patch.msgid.link/20260312100444.2609563-8-mic@digikod.net
- Reflow comments after check-linux.sh comment fixes.
- Rename LANDLOCK_PERM_NAMESPACE_ENTER references to
LANDLOCK_PERM_NAMESPACE_USE and bump the abi_version expectation
to 11 (companion changes to the introducing commit).
- Add add_rule_unknown_no_runtime_effect: assert that a rule listing
only unknown capability bits is accepted at rule-add time but has
no runtime effect, so an actual CAP_* exercise (sethostname with
CAP_SYS_ADMIN) is still denied by deny-by-default once the domain
is enforced.
- Add cap_stacking parent_denies variant covering the inverse
direction of stacking: layer 1 denies CAP_SYS_ADMIN, layer 2
allows, capability still denied. Completes the per-layer walker
direction coverage.
- Assert records.domain == 0 in cap_audit.allowed so the test also
checks that no domain-allocation record is emitted when nothing
is denied.
---
tools/testing/selftests/landlock/base_test.c | 19 +
tools/testing/selftests/landlock/cap_test.c | 1364 ++++++++++++++++++
tools/testing/selftests/landlock/trace.h | 24 +
3 files changed, 1407 insertions(+)
create mode 100644 tools/testing/selftests/landlock/cap_test.c
diff --git a/tools/testing/selftests/landlock/base_test.c b/tools/testing/selftests/landlock/base_test.c
index 58fe322d8637..89c0d96607de 100644
--- a/tools/testing/selftests/landlock/base_test.c
+++ b/tools/testing/selftests/landlock/base_test.c
@@ -142,6 +142,25 @@ TEST(errata)
ASSERT_EQ(EINVAL, errno);
}
+#define PERMISSION_LAST LANDLOCK_PERMISSION_CAPABILITY_USE
+
+TEST(ruleset_with_unknown_permission)
+{
+ __u64 permission_mask;
+
+ for (permission_mask = 1ULL << 63; permission_mask != PERMISSION_LAST;
+ permission_mask >>= 1) {
+ struct landlock_ruleset_attr ruleset_attr = {
+ .handled_permissions = permission_mask,
+ };
+
+ /* Unknown handled_permissions values must be rejected. */
+ ASSERT_EQ(-1, landlock_create_ruleset(&ruleset_attr,
+ sizeof(ruleset_attr), 0));
+ ASSERT_EQ(EINVAL, errno);
+ }
+}
+
/* Tests ordering of syscall argument checks. */
TEST(create_ruleset_checks_ordering)
{
diff --git a/tools/testing/selftests/landlock/cap_test.c b/tools/testing/selftests/landlock/cap_test.c
new file mode 100644
index 000000000000..8673934aac11
--- /dev/null
+++ b/tools/testing/selftests/landlock/cap_test.c
@@ -0,0 +1,1364 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Landlock tests - Capability restriction
+ *
+ * Copyright © 2026 Cloudflare, Inc.
+ */
+
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <linux/capability.h>
+#include <linux/landlock.h>
+#include <sched.h>
+#include <stdio.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/wait.h>
+#include <sys/xattr.h>
+#include <unistd.h>
+
+#include "audit.h"
+#include "common.h"
+#include "trace.h"
+
+#define TRACE_TASK "cap_test"
+
+static bool list_has_xattr(const char *list, ssize_t len, const char *name)
+{
+ const char *p;
+
+ for (p = list; p < list + len; p += strlen(p) + 1)
+ if (strcmp(p, name) == 0)
+ return true;
+ return false;
+}
+
+static int create_cap_ruleset(void)
+{
+ const struct landlock_ruleset_attr attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ };
+
+ return landlock_create_ruleset(&attr, sizeof(attr), 0);
+}
+
+static int add_cap_rule(int ruleset_fd, __u64 cap)
+{
+ const struct landlock_capability_attr attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .allowed_capabilities = (1ULL << cap),
+ };
+
+ return landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY, &attr,
+ 0);
+}
+
+static int add_cap_rule_full(int ruleset_fd, __u64 allowed, __u64 quiet)
+{
+ const struct landlock_capability_attr attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .allowed_capabilities = allowed,
+ .quiet_capabilities = quiet,
+ };
+
+ return landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY, &attr,
+ 0);
+}
+
+TEST(add_rule_bad_attr)
+{
+ const struct landlock_ruleset_attr ns_only_attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ };
+ int ruleset_fd;
+ struct landlock_capability_attr attr = {};
+
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+
+ /* Invalid rule pointer returns EFAULT. */
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ NULL, 0));
+ ASSERT_EQ(EFAULT, errno);
+
+ /* Empty permissions selector returns ENOMSG. */
+ attr.permissions = 0;
+ attr.allowed_capabilities = (1ULL << CAP_NET_RAW);
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &attr, 0));
+ ASSERT_EQ(ENOMSG, errno);
+
+ /* Useless rule: neither allowed nor quiet capabilities set. */
+ attr.permissions = LANDLOCK_PERMISSION_CAPABILITY_USE;
+ attr.allowed_capabilities = 0;
+ attr.quiet_capabilities = 0;
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &attr, 0));
+ ASSERT_EQ(ENOMSG, errno);
+
+ /* Quiet-only rule (empty allowed set) is legal. */
+ attr.permissions = LANDLOCK_PERMISSION_CAPABILITY_USE;
+ attr.allowed_capabilities = 0;
+ attr.quiet_capabilities = (1ULL << CAP_NET_RAW);
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &attr, 0));
+
+ /* Allow and quiet different members in the same rule. */
+ attr.permissions = LANDLOCK_PERMISSION_CAPABILITY_USE;
+ attr.allowed_capabilities = (1ULL << CAP_SYS_ADMIN);
+ attr.quiet_capabilities = (1ULL << CAP_NET_RAW);
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &attr, 0));
+
+ /* Valid capability selector plus an extra unhandled selector bit. */
+ attr.permissions = LANDLOCK_PERMISSION_CAPABILITY_USE |
+ LANDLOCK_PERMISSION_NAMESPACE_USE;
+ attr.allowed_capabilities = (1ULL << CAP_NET_RAW);
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &attr, 0));
+ ASSERT_EQ(EINVAL, errno);
+
+ /* permissions with wrong type. */
+ attr.permissions = LANDLOCK_PERMISSION_NAMESPACE_USE;
+ attr.allowed_capabilities = (1ULL << CAP_NET_RAW);
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &attr, 0));
+ ASSERT_EQ(EINVAL, errno);
+
+ /*
+ * Unknown capability bits (e.g. bit 63) are silently accepted for
+ * forward compatibility. Only known bits are stored.
+ */
+ attr.permissions = LANDLOCK_PERMISSION_CAPABILITY_USE;
+ attr.allowed_capabilities = 1ULL << 63;
+ attr.quiet_capabilities = 0;
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &attr, 0));
+
+ /*
+ * LANDLOCK_ADD_RULE_QUIET (and any other flag) is filesystem/network
+ * only and must be rejected for capability rules, even when every attr
+ * field is otherwise valid.
+ */
+ attr.permissions = LANDLOCK_PERMISSION_CAPABILITY_USE;
+ attr.allowed_capabilities = (1ULL << CAP_NET_RAW);
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &attr, LANDLOCK_ADD_RULE_QUIET));
+ ASSERT_EQ(EINVAL, errno);
+
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /*
+ * Ruleset handles LANDLOCK_PERMISSION_NAMESPACE_USE but not
+ * LANDLOCK_PERMISSION_CAPABILITY_USE: adding a capability rule must be
+ * rejected.
+ */
+ ruleset_fd =
+ landlock_create_ruleset(&ns_only_attr, sizeof(ns_only_attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+ attr.permissions = LANDLOCK_PERMISSION_CAPABILITY_USE;
+ attr.allowed_capabilities = (1ULL << CAP_NET_RAW);
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &attr, 0));
+ ASSERT_EQ(EINVAL, errno);
+ EXPECT_EQ(0, close(ruleset_fd));
+}
+
+/*
+ * Unknown capability values above CAP_LAST_CAP are silently accepted
+ * (allow-list: they have no effect since the kernel never checks them).
+ */
+TEST(add_rule_unknown)
+{
+ int ruleset_fd;
+ struct landlock_capability_attr attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ };
+
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+
+ /* Just above CAP_LAST_CAP should succeed. */
+ attr.allowed_capabilities = (1ULL << (CAP_LAST_CAP + 1));
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &attr, 0));
+
+ /* High values (below bit 63) should succeed. */
+ attr.allowed_capabilities = (1ULL << 62);
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &attr, 0));
+
+ EXPECT_EQ(0, close(ruleset_fd));
+}
+
+/*
+ * A rule that lists only capability bits unknown to the running kernel is
+ * accepted by landlock_add_rule() but has no runtime effect: once the domain is
+ * enforced, any actual CAP_* capability is still denied by the per-category
+ * deny-by-default behaviour. This documents the forward-compatibility
+ * contract: unknown bits are silently accepted so the same policy can be loaded
+ * across kernels, but they never grant a capability that the running kernel
+ * knows nothing about.
+ */
+TEST(add_rule_unknown_no_runtime_effect)
+{
+ const struct landlock_ruleset_attr ruleset_attr = {
+ .handled_permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ };
+ struct landlock_capability_attr attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ /* Only unknown bits above CAP_LAST_CAP. */
+ .allowed_capabilities = (1ULL << (CAP_LAST_CAP + 1)) |
+ (1ULL << 62),
+ };
+ int ruleset_fd;
+
+ disable_caps(_metadata);
+
+ ruleset_fd =
+ landlock_create_ruleset(&ruleset_attr, sizeof(ruleset_attr), 0);
+ ASSERT_LE(0, ruleset_fd);
+
+ ASSERT_EQ(0, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &attr, 0));
+
+ /*
+ * Isolate hostname changes and establish a baseline: with CAP_SYS_ADMIN
+ * and absent the domain, sethostname(2) succeeds. This ensures the
+ * EPERM below is attributable to Landlock rather than to a missing
+ * capability, and fails if the hook always allows.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWUTS));
+ ASSERT_EQ(0, sethostname("baseline", 8));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /*
+ * CAP_SYS_ADMIN is a real, known capability but was not authorised by
+ * the rule above; deny-by-default applies. sethostname(2) requires
+ * CAP_SYS_ADMIN.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, sethostname("test", 4));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/* clang-format off */
+FIXTURE(cap_enforce) {};
+/* clang-format on */
+
+FIXTURE_VARIANT(cap_enforce)
+{
+ const bool is_sandboxed;
+ const bool handle_caps;
+ const __u64 allowed_cap;
+ const int expected_sysadmin;
+ const int expected_chroot;
+};
+
+/*
+ * Unsandboxed baseline: no Landlock domain is enforced. Both capabilities
+ * should work normally.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(cap_enforce, unsandboxed) {
+ .is_sandboxed = false,
+ .handle_caps = false,
+ .allowed_cap = 0,
+ .expected_sysadmin = 0,
+ .expected_chroot = 0,
+};
+/* clang-format on */
+
+/*
+ * Denied: capabilities are handled but no rule allows them. All capability
+ * checks must be denied by Landlock even if the capability is effective.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(cap_enforce, denied) {
+ .is_sandboxed = true,
+ .handle_caps = true,
+ .allowed_cap = 0,
+ .expected_sysadmin = EPERM,
+ .expected_chroot = EPERM,
+};
+/* clang-format on */
+
+/*
+ * Allowed: CAP_SYS_ADMIN is allowed by rule, CAP_SYS_CHROOT is not. Only the
+ * explicitly allowed capability should succeed.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(cap_enforce, allowed) {
+ .is_sandboxed = true,
+ .handle_caps = true,
+ .allowed_cap = CAP_SYS_ADMIN,
+ .expected_sysadmin = 0,
+ .expected_chroot = EPERM,
+};
+/* clang-format on */
+
+/*
+ * Unhandled: the ruleset does not handle LANDLOCK_PERMISSION_CAPABILITY_USE at
+ * all (only handles FS access). Both capabilities should work since the domain
+ * does not restrict them.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(cap_enforce, unhandled) {
+ .is_sandboxed = true,
+ .handle_caps = false,
+ .allowed_cap = 0,
+ .expected_sysadmin = 0,
+ .expected_chroot = 0,
+};
+/* clang-format on */
+
+FIXTURE_SETUP(cap_enforce)
+{
+ disable_caps(_metadata);
+}
+
+FIXTURE_TEARDOWN(cap_enforce)
+{
+}
+
+/*
+ * Capability enforcement: tests the four fundamental enforcement scenarios
+ * (unsandboxed baseline, denied, allowed, unhandled) using two independent
+ * capability checks (sethostname for CAP_SYS_ADMIN, chroot for CAP_SYS_CHROOT).
+ */
+TEST_F(cap_enforce, use)
+{
+ int ruleset_fd;
+
+ /* Isolate hostname changes from other tests. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWUTS));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ if (variant->is_sandboxed) {
+ if (variant->handle_caps) {
+ ruleset_fd = create_cap_ruleset();
+ } else {
+ const struct landlock_ruleset_attr attr = {
+ .handled_access_fs =
+ LANDLOCK_ACCESS_FS_READ_FILE,
+ };
+
+ ruleset_fd =
+ landlock_create_ruleset(&attr, sizeof(attr), 0);
+ }
+ ASSERT_LE(0, ruleset_fd);
+
+ if (variant->allowed_cap)
+ ASSERT_EQ(0, add_cap_rule(ruleset_fd,
+ variant->allowed_cap));
+
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+ }
+
+ /* Test CAP_SYS_ADMIN via sethostname. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ if (variant->expected_sysadmin) {
+ EXPECT_EQ(-1, sethostname("test", 4));
+ EXPECT_EQ(variant->expected_sysadmin, errno);
+ } else {
+ EXPECT_EQ(0, sethostname("test", 4));
+ }
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /* Test CAP_SYS_CHROOT via chroot. */
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ if (variant->expected_chroot) {
+ EXPECT_EQ(-1, chroot("/"));
+ EXPECT_EQ(variant->expected_chroot, errno);
+ } else {
+ EXPECT_EQ(0, chroot("/"));
+ }
+}
+
+/*
+ * Layer stacking: both layers must allow CAP_SYS_ADMIN for the capability to be
+ * exercisable. Variants cover the three per-layer combinations that exercise
+ * distinct walker paths (allow/deny, allow/allow, deny/allow), an unsandboxed
+ * baseline, and a mixed-layer case where one layer does not handle
+ * LANDLOCK_PERMISSION_CAPABILITY_USE at all.
+ */
+/* clang-format off */
+FIXTURE(cap_stacking) {};
+/* clang-format on */
+
+FIXTURE_VARIANT(cap_stacking)
+{
+ const bool is_sandboxed;
+ const bool first_layer_allows;
+ const bool second_layer_allows;
+ const bool second_layer_is_fs_only;
+ const int expected_sysadmin;
+ const int expected_chroot;
+};
+
+/*
+ * Unsandboxed baseline: no Landlock layers are stacked. Both capabilities
+ * should work normally.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(cap_stacking, unsandboxed) {
+ .is_sandboxed = false,
+ .first_layer_allows = false,
+ .second_layer_allows = false,
+ .expected_sysadmin = 0,
+ .expected_chroot = 0,
+};
+/* clang-format on */
+
+/* Layer 1 allows CAP_SYS_ADMIN, layer 2 denies -> denied. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(cap_stacking, deny) {
+ .is_sandboxed = true,
+ .first_layer_allows = true,
+ .second_layer_allows = false,
+ .expected_sysadmin = EPERM,
+ .expected_chroot = EPERM,
+};
+/* clang-format on */
+
+/* Both layers allow CAP_SYS_ADMIN -> sysadmin succeeds, chroot still denied. */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(cap_stacking, allow) {
+ .is_sandboxed = true,
+ .first_layer_allows = true,
+ .second_layer_allows = true,
+ .expected_sysadmin = 0,
+ .expected_chroot = EPERM,
+};
+/* clang-format on */
+
+/*
+ * Layer 1 denies CAP_SYS_ADMIN, layer 2 allows -> still denied: a child layer
+ * cannot grant what an ancestor layer withheld. Complements the
+ * parent-allows/child-denies variant; together they verify the walker checks
+ * both layers and accepts only the (allow, allow) cell.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(cap_stacking, parent_denies) {
+ .is_sandboxed = true,
+ .first_layer_allows = false,
+ .second_layer_allows = true,
+ .expected_sysadmin = EPERM,
+ .expected_chroot = EPERM,
+};
+/* clang-format on */
+
+/*
+ * Mixed layers: first layer handles LANDLOCK_PERMISSION_CAPABILITY_USE (denies
+ * all caps), second layer is FS-only (does not handle it). The permission
+ * walker iterates from youngest (layer 1) to oldest (layer 0) and must skip the
+ * FS-only layer to find the denying layer beneath.
+ */
+/* clang-format off */
+FIXTURE_VARIANT_ADD(cap_stacking, mixed_layers) {
+ /* clang-format on */
+ .is_sandboxed = true,
+ .first_layer_allows = false,
+ .second_layer_is_fs_only = true,
+ .expected_sysadmin = EPERM,
+ .expected_chroot = EPERM,
+};
+
+FIXTURE_SETUP(cap_stacking)
+{
+ disable_caps(_metadata);
+
+ /* Isolate every successful sethostname() from the other tests. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWUTS));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(cap_stacking)
+{
+}
+
+TEST_F(cap_stacking, two_layers)
+{
+ int ruleset_fd;
+
+ if (variant->is_sandboxed) {
+ /*
+ * First layer: handles LANDLOCK_PERMISSION_CAPABILITY_USE; rule
+ * added per variant.
+ */
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ if (variant->first_layer_allows)
+ ASSERT_EQ(0, add_cap_rule(ruleset_fd, CAP_SYS_ADMIN));
+
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ if (variant->second_layer_is_fs_only) {
+ /*
+ * Second layer: FS-only (does not handle
+ * LANDLOCK_PERMISSION_CAPABILITY_USE). The permission
+ * walker must skip this layer.
+ */
+ const struct landlock_ruleset_attr fs_attr = {
+ .handled_access_fs =
+ LANDLOCK_ACCESS_FS_READ_FILE,
+ };
+
+ ruleset_fd = landlock_create_ruleset(
+ &fs_attr, sizeof(fs_attr), 0);
+ } else {
+ /* Second layer: cap allow or deny. */
+ ruleset_fd = create_cap_ruleset();
+ if (variant->second_layer_allows)
+ ASSERT_EQ(0, add_cap_rule(ruleset_fd,
+ CAP_SYS_ADMIN));
+ }
+ ASSERT_LE(0, ruleset_fd);
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+ }
+
+ /* Test CAP_SYS_ADMIN via sethostname. */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ if (variant->expected_sysadmin) {
+ EXPECT_EQ(-1, sethostname("test", 4));
+ EXPECT_EQ(variant->expected_sysadmin, errno);
+ } else {
+ EXPECT_EQ(0, sethostname("test", 4));
+ }
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /* Test CAP_SYS_CHROOT via chroot. */
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ if (variant->expected_chroot) {
+ EXPECT_EQ(-1, chroot("/"));
+ EXPECT_EQ(variant->expected_chroot, errno);
+ } else {
+ EXPECT_EQ(0, chroot("/"));
+ }
+ clear_cap(_metadata, CAP_SYS_CHROOT);
+}
+
+/*
+ * Verify that LANDLOCK_PERMISSION_CAPABILITY_USE enforces when the domain is
+ * applied without no_new_privs, using CAP_SYS_ADMIN for
+ * landlock_restrict_self() authorization instead. Privileged processes (e.g.
+ * container managers) can sandbox themselves this way.
+ */
+TEST(cap_without_nnp)
+{
+ int ruleset_fd;
+
+ disable_caps(_metadata);
+
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+
+ /* Allow CAP_SYS_CHROOT but not CAP_SYS_ADMIN. */
+ ASSERT_EQ(0, add_cap_rule(ruleset_fd, CAP_SYS_CHROOT));
+
+ /*
+ * Enforce WITHOUT NNP: landlock_restrict_self() succeeds when the
+ * caller has CAP_SYS_ADMIN (checked before the new domain takes
+ * effect).
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, landlock_restrict_self(ruleset_fd, 0));
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /*
+ * CAP_SYS_ADMIN is still in effective set but Landlock denies it:
+ * cap_capable() returns 0, then hook_capable() returns -EPERM.
+ */
+ EXPECT_EQ(-1, sethostname("test", 4));
+ EXPECT_EQ(EPERM, errno);
+
+ /* CAP_SYS_CHROOT is allowed by the rule. */
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ EXPECT_EQ(0, chroot("/"));
+}
+
+/*
+ * Verify that capabilities gained through user namespace ownership are still
+ * restricted by LANDLOCK_PERMISSION_CAPABILITY_USE. When a process creates a
+ * user namespace, the kernel grants CAP_FULL_SET in the new namespace via
+ * cap_capable_helper()'s ownership bypass. Landlock's hook_capable() must
+ * still deny capabilities not in the allowed set, ensuring that user namespace
+ * creation cannot be used to escape capability restrictions.
+ */
+TEST(cap_userns_ownership_bypass)
+{
+ pid_t child;
+ int status;
+
+ child = fork();
+ ASSERT_LE(0, child);
+ if (child == 0) {
+ int ruleset_fd;
+
+ disable_caps(_metadata);
+
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+
+ /* Allow CAP_SYS_ADMIN only. */
+ ASSERT_EQ(0, add_cap_rule(ruleset_fd, CAP_SYS_ADMIN));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /*
+ * Create a user namespace. This is unprivileged and does not
+ * require capabilities. LANDLOCK_PERMISSION_NAMESPACE_USE is
+ * not handled so namespace creation is unrestricted.
+ */
+ ASSERT_EQ(0, unshare(CLONE_NEWUSER));
+
+ /*
+ * After unshare(CLONE_NEWUSER), the kernel set cap_effective =
+ * CAP_FULL_SET in the new namespace. Create a UTS namespace
+ * (requires CAP_SYS_ADMIN in the new user NS). Landlock allows
+ * CAP_SYS_ADMIN.
+ */
+ ASSERT_EQ(0, unshare(CLONE_NEWUTS))
+ {
+ TH_LOG("unshare(CLONE_NEWUTS): %s", strerror(errno));
+ }
+
+ /*
+ * sethostname checks against uts_ns->user_ns, which is now the
+ * new user NS. CAP_SYS_ADMIN is allowed.
+ */
+ EXPECT_EQ(0, sethostname("test", 4));
+
+ /*
+ * chroot checks against current_user_ns(), which is the new
+ * user NS. The process has CAP_SYS_CHROOT in cap_effective
+ * (from user NS creation), so cap_capable() returns 0. But
+ * Landlock denies because no rule allows CAP_SYS_CHROOT.
+ */
+ EXPECT_EQ(-1, chroot("/"));
+ EXPECT_EQ(EPERM, errno);
+
+ _exit(_metadata->exit_code);
+ return;
+ }
+
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ if (WIFSIGNALED(status) || !WIFEXITED(status) ||
+ WEXITSTATUS(status) != EXIT_SUCCESS)
+ _metadata->exit_code = KSFT_FAIL;
+}
+
+/*
+ * Verify that capability restrictions do not depend on which user namespace the
+ * kernel checks. An actor in the initial user namespace can enter a descendant
+ * user namespace with CAP_SYS_ADMIN. With the same Linux capability state,
+ * handling capability use without an allow rule must deny that target-namespace
+ * check even though the target differs from current_user_ns().
+ */
+TEST(cap_target_userns_independent)
+{
+ int sockets[2], status, target_userns_fd;
+ pid_t actor, target;
+
+ disable_caps(_metadata);
+ ASSERT_EQ(0,
+ socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, sockets));
+
+ target = fork();
+ ASSERT_LE(0, target);
+ if (target == 0) {
+ int fd;
+
+ close(sockets[0]);
+ if (unshare(CLONE_NEWUSER))
+ _exit(255);
+ fd = open("/proc/self/ns/user", O_RDONLY | O_CLOEXEC);
+ if (fd < 0 || send_fd(sockets[1], fd))
+ _exit(255);
+ close(fd);
+ close(sockets[1]);
+ _exit(EXIT_SUCCESS);
+ }
+
+ close(sockets[1]);
+ target_userns_fd = recv_fd(sockets[0]);
+ ASSERT_LE(0, target_userns_fd);
+ EXPECT_EQ(0, close(sockets[0]));
+ ASSERT_EQ(target, waitpid(target, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ ASSERT_EQ(EXIT_SUCCESS, WEXITSTATUS(status));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ for (int sandboxed = 0; sandboxed <= 1; sandboxed++) {
+ actor = fork();
+ ASSERT_LE(0, actor);
+ if (actor == 0) {
+ int ret, saved_errno;
+
+ if (sandboxed) {
+ int ruleset_fd = create_cap_ruleset();
+
+ if (ruleset_fd < 0 ||
+ prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0) ||
+ landlock_restrict_self(ruleset_fd, 0) ||
+ close(ruleset_fd))
+ _exit(255);
+ }
+
+ ret = setns(target_userns_fd, CLONE_NEWUSER);
+ saved_errno = errno;
+ _exit(ret ? saved_errno : EXIT_SUCCESS);
+ }
+
+ ASSERT_EQ(actor, waitpid(actor, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ EXPECT_EQ(sandboxed ? EPERM : EXIT_SUCCESS,
+ WEXITSTATUS(status));
+ }
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, close(target_userns_fd));
+}
+
+/* Audit tests */
+
+static int matches_log_cap(int audit_fd, int cap_number)
+{
+ static const char log_template[] = REGEX_LANDLOCK_PREFIX
+ " blockers=capability\\.use capability=%d $";
+ char log_match[sizeof(log_template) + 10];
+ int log_match_len;
+
+ log_match_len = snprintf(log_match, sizeof(log_match), log_template,
+ cap_number);
+ if (log_match_len >= sizeof(log_match))
+ return -E2BIG;
+
+ return audit_match_record(audit_fd, AUDIT_LANDLOCK_ACCESS, log_match,
+ NULL);
+}
+
+FIXTURE(cap_audit)
+{
+ struct audit_filter audit_filter;
+ int audit_fd;
+};
+
+FIXTURE_SETUP(cap_audit)
+{
+ ASSERT_TRUE(is_in_init_user_ns());
+
+ disable_caps(_metadata);
+
+ /* Isolate the allowed test's successful sethostname(). */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWUTS));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ set_cap(_metadata, CAP_AUDIT_CONTROL);
+ self->audit_fd = audit_init_with_exe_filter(&self->audit_filter);
+ EXPECT_LE(0, self->audit_fd);
+ clear_cap(_metadata, CAP_AUDIT_CONTROL);
+}
+
+FIXTURE_TEARDOWN(cap_audit)
+{
+ set_cap(_metadata, CAP_AUDIT_CONTROL);
+ EXPECT_EQ(0, audit_cleanup(self->audit_fd, &self->audit_filter));
+ clear_cap(_metadata, CAP_AUDIT_CONTROL);
+}
+
+/*
+ * Verifies that a denied capability produces the expected audit record with the
+ * correct capability number and blocker string.
+ */
+TEST_F(cap_audit, denied)
+{
+ struct audit_records records;
+ int ruleset_fd;
+ __u64 domain_id;
+
+ /* Baseline: chroot works before Landlock. */
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ ASSERT_EQ(0, chroot("/"));
+ clear_cap(_metadata, CAP_SYS_CHROOT);
+
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ /* Allow CAP_AUDIT_CONTROL for child-side audit cleanup. */
+ ASSERT_EQ(0, add_cap_rule(ruleset_fd, CAP_AUDIT_CONTROL));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /* Deny CAP_SYS_CHROOT (no allow rule). */
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ EXPECT_EQ(-1, chroot("/"));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_CHROOT);
+
+ EXPECT_EQ(0, matches_log_cap(self->audit_fd, CAP_SYS_CHROOT));
+
+ /*
+ * The domain allocation record is emitted in the same event as the
+ * first denial; anchor its status=allocated and enforcing pid. Both
+ * access and domain records are now consumed, so none remain.
+ */
+ EXPECT_EQ(0, matches_log_domain_allocated(self->audit_fd, getpid(),
+ &domain_id));
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+}
+
+TEST_F(cap_audit, allowed)
+{
+ struct audit_records records;
+ int ruleset_fd;
+
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_cap_rule(ruleset_fd, CAP_SYS_ADMIN));
+ /* Allow CAP_AUDIT_CONTROL for child-side audit cleanup. */
+ ASSERT_EQ(0, add_cap_rule(ruleset_fd, CAP_AUDIT_CONTROL));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, sethostname("test", 4));
+
+ /* No records: allowed operations never trigger audit logging. */
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+}
+
+/* An allowed capability remains allowed when the same rule also quiets it. */
+TEST_F(cap_audit, allow_and_quiet_same_member)
+{
+ struct audit_records records;
+ int ruleset_fd;
+
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_cap_rule_full(ruleset_fd,
+ (1ULL << CAP_SYS_ADMIN) |
+ (1ULL << CAP_AUDIT_CONTROL),
+ 1ULL << CAP_SYS_ADMIN));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(0, sethostname("test", 4));
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /*
+ * Quiet is inert for the allowed member: nothing is denied or logged.
+ */
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+}
+
+/*
+ * A quieted capability denial is still denied (EPERM); only its audit record is
+ * suppressed. Quiet-only rule (empty allowed set).
+ */
+TEST_F(cap_audit, quieted)
+{
+ struct audit_records records;
+ int ruleset_fd;
+ __u64 domain_id;
+
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ /* Allow CAP_AUDIT_CONTROL for child-side audit cleanup. */
+ ASSERT_EQ(0, add_cap_rule(ruleset_fd, CAP_AUDIT_CONTROL));
+ ASSERT_EQ(0,
+ add_cap_rule_full(ruleset_fd, 0, (1ULL << CAP_SYS_CHROOT)));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ EXPECT_EQ(-1, chroot("/"));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_CHROOT);
+
+ /*
+ * Positive control: CAP_SYS_ADMIN is denied but not quieted, so its
+ * denial IS logged. This proves quieting is per-member, not global.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ EXPECT_EQ(-1, sethostname("test", 4));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ /*
+ * The quieted CAP_SYS_CHROOT denial is suppressed; the CAP_SYS_ADMIN
+ * denial is logged.
+ */
+ EXPECT_EQ(0, matches_log_cap(self->audit_fd, CAP_SYS_ADMIN));
+ EXPECT_EQ(0, matches_log_domain_allocated(self->audit_fd, getpid(),
+ &domain_id));
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+}
+
+/*
+ * Only the youngest denying layer decides quieting: a parent cannot silence a
+ * denial by a deeper layer that handles but does not quiet the capability.
+ */
+TEST_F(cap_audit, quiet_youngest_layer_wins)
+{
+ struct audit_records records;
+ int ruleset_fd;
+ __u64 domain_id;
+
+ /* Baseline: commoncap allows chroot before Landlock. */
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ ASSERT_EQ(0, chroot("/"));
+ clear_cap(_metadata, CAP_SYS_CHROOT);
+
+ /* Parent layer: denies and quiets CAP_SYS_CHROOT. */
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_cap_rule_full(ruleset_fd, 1ULL << CAP_AUDIT_CONTROL,
+ 1ULL << CAP_SYS_CHROOT));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /* Child layer: denies CAP_SYS_CHROOT without quieting it. */
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_cap_rule(ruleset_fd, CAP_AUDIT_CONTROL));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ EXPECT_EQ(-1, chroot("/"));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_CHROOT);
+
+ /*
+ * The child is the youngest denier and does not quiet: log the denial.
+ */
+ EXPECT_EQ(0, matches_log_cap(self->audit_fd, CAP_SYS_CHROOT));
+ EXPECT_EQ(0, matches_log_domain_allocated(self->audit_fd, getpid(),
+ &domain_id));
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+}
+
+/*
+ * Quieting an unknown capability bit has no effect: a rule that quiets only a
+ * bit above CAP_LAST_CAP still logs the denial of a known capability.
+ */
+TEST_F(cap_audit, quiet_unknown_bit_no_effect)
+{
+ struct audit_records records;
+ int ruleset_fd;
+ __u64 domain_id;
+
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_cap_rule(ruleset_fd, CAP_AUDIT_CONTROL));
+ /* Quiet only an unknown bit (bit 63 is above CAP_LAST_CAP). */
+ ASSERT_EQ(0, add_cap_rule_full(ruleset_fd, 0, (1ULL << 63)));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ EXPECT_EQ(-1, chroot("/"));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_CHROOT);
+
+ /* The unknown quiet bit does not suppress the known denial. */
+ EXPECT_EQ(0, matches_log_cap(self->audit_fd, CAP_SYS_CHROOT));
+ EXPECT_EQ(0, matches_log_domain_allocated(self->audit_fd, getpid(),
+ &domain_id));
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+}
+
+/*
+ * A capability check made with CAP_OPT_NOAUDIT is denied by Landlock but not
+ * logged: hook_capable() emits no Landlock audit record when the caller passes
+ * CAP_OPT_NOAUDIT, so the denial is silent here even though CAP_SYS_ADMIN is
+ * neither allowed nor quieted. The probe is the
+ * ns_capable_noaudit(CAP_SYS_ADMIN) call in simple_xattr_list(), reached
+ * through tmpfs, which decides whether trusted.* xattrs are visible to
+ * listxattr(2).
+ */
+TEST_F(cap_audit, noaudit_probe_not_logged)
+{
+ struct audit_records records;
+ int ruleset_fd, fd;
+ __u64 domain_id;
+ char list[256];
+ ssize_t len;
+
+ /*
+ * Private tmpfs holding a trusted.* xattr so its listxattr(2)
+ * visibility depends only on the CAP_SYS_ADMIN noaudit probe.
+ */
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+ ASSERT_EQ(0, mount("test", "/tmp", "tmpfs", 0, NULL));
+ fd = open("/tmp/f", O_CREAT | O_RDWR | O_CLOEXEC, 0600);
+ ASSERT_LE(0, fd);
+ ASSERT_EQ(0, fsetxattr(fd, "trusted.landlock_test", "x", 1, 0));
+
+ /* Baseline: with CAP_SYS_ADMIN, the trusted xattr is listed. */
+ len = flistxattr(fd, list, sizeof(list));
+ ASSERT_LE(0, len);
+ ASSERT_TRUE(list_has_xattr(list, len, "trusted.landlock_test"));
+
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ /* Allow CAP_AUDIT_CONTROL for child-side audit cleanup. */
+ ASSERT_EQ(0, add_cap_rule(ruleset_fd, CAP_AUDIT_CONTROL));
+ /* CAP_SYS_ADMIN is neither allowed nor quieted. */
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ /*
+ * CAP_SYS_ADMIN is still effective, but Landlock denies the noaudit
+ * probe, so the trusted.* xattr is filtered out of the listing.
+ */
+ len = flistxattr(fd, list, sizeof(list));
+ ASSERT_LE(0, len);
+ EXPECT_FALSE(list_has_xattr(list, len, "trusted.landlock_test"));
+ EXPECT_EQ(0, close(fd));
+
+ /* The denied noaudit probe leaves no audit record nor denial. */
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+
+ /*
+ * Positive control: a normal (audited) CAP_SYS_ADMIN denial in the same
+ * domain IS logged. This proves the empty record set above is caused
+ * by CAP_OPT_NOAUDIT, not by an always-allow or always-silent bug.
+ */
+ EXPECT_EQ(-1, sethostname("test", 4));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+
+ EXPECT_EQ(0, matches_log_cap(self->audit_fd, CAP_SYS_ADMIN));
+ EXPECT_EQ(0, matches_log_domain_allocated(self->audit_fd, getpid(),
+ &domain_id));
+ EXPECT_EQ(0, audit_count_records(self->audit_fd, &records));
+ EXPECT_EQ(0, records.access);
+ EXPECT_EQ(0, records.domain);
+}
+
+/* Trace tests */
+
+/* clang-format off */
+FIXTURE(cap_trace) {
+ /* clang-format on */
+ int tracefs_ok;
+};
+
+FIXTURE_SETUP(cap_trace)
+{
+ int ret;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ ASSERT_EQ(0, unshare(CLONE_NEWNS));
+ ASSERT_EQ(0, mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL));
+
+ ret = tracefs_fixture_setup();
+ if (ret) {
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ self->tracefs_ok = 0;
+ SKIP(return, "tracefs not available");
+ }
+ self->tracefs_ok = 1;
+
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, true));
+ ASSERT_EQ(0, tracefs_enable_event(TRACEFS_ADD_RULE_CAPABILITY_ENABLE,
+ true));
+ ASSERT_EQ(0, tracefs_enable_event(
+ TRACEFS_DENY_PERMISSION_CAPABILITY_ENABLE, true));
+ ASSERT_EQ(0, tracefs_clear());
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+FIXTURE_TEARDOWN(cap_trace)
+{
+ if (!self->tracefs_ok)
+ return;
+
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ tracefs_enable_event(TRACEFS_CREATE_RULESET_ENABLE, false);
+ tracefs_enable_event(TRACEFS_ADD_RULE_CAPABILITY_ENABLE, false);
+ tracefs_enable_event(TRACEFS_DENY_PERMISSION_CAPABILITY_ENABLE, false);
+ tracefs_fixture_teardown();
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+}
+
+/*
+ * Check the permission fields and version history, including two rejected calls
+ * and two successful calls that do not extend the effective policy: a repeated
+ * rule and an unknown-only rule.
+ */
+TEST_F(cap_trace, rule_events)
+{
+ const char *const version_1 =
+ REGEX_ADD_RULE_CAPABILITY_VERSION(TRACE_TASK, "1");
+ const char *const version_2 =
+ REGEX_ADD_RULE_CAPABILITY_VERSION(TRACE_TASK, "2");
+ const char *const version_3 =
+ REGEX_ADD_RULE_CAPABILITY_VERSION(TRACE_TASK, "3");
+ const struct landlock_capability_attr valid_attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .allowed_capabilities = 1ULL << CAP_SYS_ADMIN,
+ .quiet_capabilities = 1ULL << CAP_SYS_CHROOT,
+ };
+ const struct landlock_capability_attr invalid_attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .allowed_capabilities = 1ULL << CAP_SYS_ADMIN,
+ };
+ char expected[32], field[64];
+ char *buf;
+ int ruleset_fd;
+
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ ASSERT_EQ(0, add_cap_rule_full(ruleset_fd, 1ULL << CAP_SYS_ADMIN,
+ 1ULL << CAP_SYS_CHROOT));
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &invalid_attr, 0));
+ ASSERT_EQ(EINVAL, errno);
+ ASSERT_EQ(-1, landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &valid_attr, LANDLOCK_ADD_RULE_QUIET));
+ ASSERT_EQ(EINVAL, errno);
+ ASSERT_EQ(0, add_cap_rule_full(ruleset_fd, 1ULL << CAP_SYS_ADMIN,
+ 1ULL << CAP_SYS_CHROOT));
+ ASSERT_EQ(0, add_cap_rule(ruleset_fd, 63));
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+
+ ASSERT_EQ(1,
+ tracefs_count_matches(buf, REGEX_CREATE_RULESET(TRACE_TASK)));
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_CREATE_RULESET(TRACE_TASK),
+ "handled_permissions", field, sizeof(field)));
+ EXPECT_STREQ("capability.use", field);
+
+ ASSERT_EQ(3,
+ tracefs_count_matches(buf, "landlock_add_rule_capability: "));
+ ASSERT_EQ(3, tracefs_count_matches(
+ buf, REGEX_ADD_RULE_CAPABILITY(TRACE_TASK)));
+ EXPECT_EQ(1, tracefs_count_matches(buf, version_1));
+ EXPECT_EQ(1, tracefs_count_matches(buf, version_2));
+ EXPECT_EQ(1, tracefs_count_matches(buf, version_3));
+
+ ASSERT_EQ(0, tracefs_extract_field(buf, version_1, "permissions", field,
+ sizeof(field)));
+ EXPECT_STREQ("capability.use", field);
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, version_1, "allowed_capabilities",
+ field, sizeof(field)));
+ snprintf(expected, sizeof(expected), "0x%llx", 1ULL << CAP_SYS_ADMIN);
+ EXPECT_STREQ(expected, field);
+ ASSERT_EQ(0, tracefs_extract_field(buf, version_1, "quiet_capabilities",
+ field, sizeof(field)));
+ snprintf(expected, sizeof(expected), "0x%llx", 1ULL << CAP_SYS_CHROOT);
+ EXPECT_STREQ(expected, field);
+
+ /* The rejected calls emit no event and do not consume version 2. */
+ ASSERT_EQ(0, tracefs_extract_field(buf, version_2, "permissions", field,
+ sizeof(field)));
+ EXPECT_STREQ("capability.use", field);
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, version_2, "allowed_capabilities",
+ field, sizeof(field)));
+ snprintf(expected, sizeof(expected), "0x%llx", 1ULL << CAP_SYS_ADMIN);
+ EXPECT_STREQ(expected, field);
+ ASSERT_EQ(0, tracefs_extract_field(buf, version_2, "quiet_capabilities",
+ field, sizeof(field)));
+ snprintf(expected, sizeof(expected), "0x%llx", 1ULL << CAP_SYS_CHROOT);
+ EXPECT_STREQ(expected, field);
+
+ /* The unknown-only call succeeds but contributes no effective bits. */
+ ASSERT_EQ(0, tracefs_extract_field(buf, version_3, "permissions", field,
+ sizeof(field)));
+ EXPECT_STREQ("capability.use", field);
+ ASSERT_EQ(0,
+ tracefs_extract_field(buf, version_3, "allowed_capabilities",
+ field, sizeof(field)));
+ EXPECT_STREQ("0x0", field);
+ ASSERT_EQ(0, tracefs_extract_field(buf, version_3, "quiet_capabilities",
+ field, sizeof(field)));
+ EXPECT_STREQ("0x0", field);
+
+ free(buf);
+}
+
+enum cap_trace_denial_kind {
+ CAP_TRACE_DENY,
+ CAP_TRACE_DENY_QUIET,
+ CAP_TRACE_DENY_NOAUDIT,
+};
+
+static void exercise_cap_trace_denial(struct __test_metadata *const _metadata,
+ const enum cap_trace_denial_kind kind)
+{
+ int ruleset_fd, fd = -1;
+ char list[256];
+ ssize_t len;
+
+ disable_caps(_metadata);
+
+ if (kind == CAP_TRACE_DENY_NOAUDIT) {
+ set_cap(_metadata, CAP_SYS_ADMIN);
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ ASSERT_EQ(0, mount("test", "/tmp", "tmpfs", 0, NULL));
+ fd = open("/tmp/f", O_CREAT | O_RDWR | O_CLOEXEC, 0600);
+ ASSERT_LE(0, fd);
+ ASSERT_EQ(0, fsetxattr(fd, "trusted.landlock_test", "x", 1, 0));
+
+ /* Baseline: CAP_SYS_ADMIN exposes the trusted xattr. */
+ len = flistxattr(fd, list, sizeof(list));
+ ASSERT_LE(0, len);
+ ASSERT_TRUE(list_has_xattr(list, len, "trusted.landlock_test"));
+ } else {
+ set_cap(_metadata, CAP_SYS_CHROOT);
+ }
+
+ ruleset_fd = create_cap_ruleset();
+ ASSERT_LE(0, ruleset_fd);
+ if (kind == CAP_TRACE_DENY_QUIET)
+ ASSERT_EQ(0, add_cap_rule_full(ruleset_fd, 0,
+ 1ULL << CAP_SYS_CHROOT));
+ enforce_ruleset(_metadata, ruleset_fd);
+ EXPECT_EQ(0, close(ruleset_fd));
+
+ if (kind == CAP_TRACE_DENY_NOAUDIT) {
+ /* Probe CAP_SYS_ADMIN with CAP_OPT_NOAUDIT. */
+ len = flistxattr(fd, list, sizeof(list));
+ ASSERT_LE(0, len);
+ EXPECT_FALSE(
+ list_has_xattr(list, len, "trusted.landlock_test"));
+ EXPECT_EQ(0, close(fd));
+
+ /* Control: an ordinary denial in the same domain is traced. */
+ EXPECT_EQ(-1, chroot("/"));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_ADMIN);
+ clear_cap(_metadata, CAP_SYS_CHROOT);
+ } else {
+ EXPECT_EQ(-1, chroot("/"));
+ EXPECT_EQ(EPERM, errno);
+ clear_cap(_metadata, CAP_SYS_CHROOT);
+ }
+}
+
+static void test_cap_trace_denial(struct __test_metadata *const _metadata,
+ const enum cap_trace_denial_kind kind,
+ const int expected_capability,
+ const char *const expected_logged)
+{
+ char expected[16], field[64];
+ int status;
+ char *buf;
+ pid_t child;
+
+ ASSERT_EQ(0, tracefs_clear_buf());
+
+ child = fork();
+ ASSERT_LE(0, child);
+ if (child == 0) {
+ exercise_cap_trace_denial(_metadata, kind);
+ _exit(_metadata->exit_code);
+ }
+
+ ASSERT_EQ(child, waitpid(child, &status, 0));
+ ASSERT_TRUE(WIFEXITED(status));
+ ASSERT_EQ(EXIT_SUCCESS, WEXITSTATUS(status));
+
+ buf = tracefs_read_buf();
+ ASSERT_NE(NULL, buf);
+ ASSERT_EQ(1, tracefs_count_matches(
+ buf, REGEX_DENY_PERMISSION_CAPABILITY(TRACE_TASK)))
+ {
+ TH_LOG("Expected one capability denial event\n%s", buf);
+ }
+
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PERMISSION_CAPABILITY(TRACE_TASK),
+ "domain", field, sizeof(field)));
+ EXPECT_STRNE("0", field);
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PERMISSION_CAPABILITY(TRACE_TASK),
+ "same_exec", field, sizeof(field)));
+ EXPECT_STREQ("1", field);
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PERMISSION_CAPABILITY(TRACE_TASK),
+ "logged", field, sizeof(field)));
+ EXPECT_STREQ(expected_logged, field);
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PERMISSION_CAPABILITY(TRACE_TASK),
+ "blockers", field, sizeof(field)));
+ EXPECT_STREQ("use", field);
+ ASSERT_EQ(0, tracefs_extract_field(
+ buf, REGEX_DENY_PERMISSION_CAPABILITY(TRACE_TASK),
+ "capability", field, sizeof(field)));
+ snprintf(expected, sizeof(expected), "%d", expected_capability);
+ EXPECT_STREQ(expected, field);
+
+ free(buf);
+}
+
+TEST_F(cap_trace, deny)
+{
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+ test_cap_trace_denial(_metadata, CAP_TRACE_DENY, CAP_SYS_CHROOT, "1");
+}
+
+TEST_F(cap_trace, deny_quiet)
+{
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+ test_cap_trace_denial(_metadata, CAP_TRACE_DENY_QUIET, CAP_SYS_CHROOT,
+ "0");
+}
+
+/*
+ * A CAP_OPT_NOAUDIT probe is denied without any record. The audit counterpart
+ * checks the missing audit record; this one checks that the CAP_SYS_ADMIN probe
+ * emits no trace event, while the CAP_SYS_CHROOT denial that follows it in the
+ * same domain still does.
+ */
+TEST_F(cap_trace, noaudit_probe_not_traced)
+{
+ if (!self->tracefs_ok)
+ SKIP(return, "tracefs not available");
+ test_cap_trace_denial(_metadata, CAP_TRACE_DENY_NOAUDIT, CAP_SYS_CHROOT,
+ "1");
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/landlock/trace.h b/tools/testing/selftests/landlock/trace.h
index 568c13d8eee4..36e8f16171b0 100644
--- a/tools/testing/selftests/landlock/trace.h
+++ b/tools/testing/selftests/landlock/trace.h
@@ -33,6 +33,8 @@
TRACEFS_LANDLOCK_DIR "/landlock_add_rule_net_port/enable"
#define TRACEFS_ADD_RULE_NAMESPACE_ENABLE \
TRACEFS_LANDLOCK_DIR "/landlock_add_rule_namespace/enable"
+#define TRACEFS_ADD_RULE_CAPABILITY_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_add_rule_capability/enable"
#define TRACEFS_CHECK_RULE_FS_ENABLE \
TRACEFS_LANDLOCK_DIR "/landlock_check_rule_inode/enable"
#define TRACEFS_CHECK_RULE_NET_ENABLE \
@@ -43,6 +45,8 @@
TRACEFS_LANDLOCK_DIR "/landlock_deny_access_net/enable"
#define TRACEFS_DENY_PERMISSION_NAMESPACE_ENABLE \
TRACEFS_LANDLOCK_DIR "/landlock_deny_permission_namespace/enable"
+#define TRACEFS_DENY_PERMISSION_CAPABILITY_ENABLE \
+ TRACEFS_LANDLOCK_DIR "/landlock_deny_permission_capability/enable"
#define TRACEFS_DENY_PTRACE_ENABLE \
TRACEFS_LANDLOCK_DIR "/landlock_deny_ptrace/enable"
#define TRACEFS_DENY_SCOPE_SIGNAL_ENABLE \
@@ -110,6 +114,17 @@
#define REGEX_ADD_RULE_NAMESPACE(task) \
REGEX_ADD_RULE_NAMESPACE_VERSION(task, "[0-9]\\+")
+#define REGEX_ADD_RULE_CAPABILITY_VERSION(task, version) \
+ TRACE_PREFIX(task) \
+ "landlock_add_rule_capability: " \
+ "ruleset=[0-9a-f]\\+\\." version " " \
+ "permissions=[a-z._|]* " \
+ "allowed_capabilities=0x[0-9a-f]\\+ " \
+ "quiet_capabilities=0x[0-9a-f]\\+$"
+
+#define REGEX_ADD_RULE_CAPABILITY(task) \
+ REGEX_ADD_RULE_CAPABILITY_VERSION(task, "[0-9]\\+")
+
#define REGEX_CREATE_RULESET(task) \
TRACE_PREFIX(task) \
"landlock_create_ruleset: " \
@@ -173,6 +188,15 @@
"namespace_type=0x[0-9a-f]\\+ " \
"namespace_id=[0-9]\\+$"
+#define REGEX_DENY_PERMISSION_CAPABILITY(task) \
+ TRACE_PREFIX(task) \
+ "landlock_deny_permission_capability: " \
+ "domain=[0-9a-f]\\+ " \
+ "same_exec=[01] " \
+ "logged=[01] " \
+ "blockers=[a-z_|]* " \
+ "capability=[0-9]\\+$"
+
#define REGEX_DENY_PTRACE(task) \
TRACE_PREFIX(task) \
"landlock_deny_ptrace: " \
--
2.55.0
^ permalink raw reply [flat|nested] 9+ messages in thread* [PATCH v4 7/8] samples/landlock: Add capability and namespace restriction support
2026-10-02 12:43 [PATCH v4 0/8] Landlock: Namespace and capability control Mickaël Salaün
` (5 preceding siblings ...)
2026-10-02 12:43 ` [PATCH v4 6/8] selftests/landlock: Add capability " Mickaël Salaün
@ 2026-10-02 12:44 ` Mickaël Salaün
2026-10-02 12:44 ` [PATCH v4 8/8] landlock: Add documentation for capability and namespace restrictions Mickaël Salaün
7 siblings, 0 replies; 9+ messages in thread
From: Mickaël Salaün @ 2026-10-02 12:44 UTC (permalink / raw)
To: Christian Brauner, Günther Noack, Paul Moore, Serge E . Hallyn
Cc: Mickaël Salaün, Daniel Durning, Jonathan Corbet,
Justin Suess, Lennart Poettering, Mikhail Ivanov,
Nicolas Bouchinet, Shervin Oloumi, Tingmao Wang, kernel-team,
linux-fsdevel, linux-kernel, linux-security-module
Extend the sandboxer sample to demonstrate the new Landlock capability
and namespace restriction features. LL_CAP takes a colon-delimited list
of allowed capabilities, parsed with cap_from_name(3) from libcap so
names and numeric strings are both accepted. LL_NS takes a
colon-delimited list of allowed namespace types by short name. Add
best-effort degradation for older kernels that predate the
LANDLOCK_PERMISSION_* features.
Allow creating user and UTS namespaces but deny network namespaces, as
an unprivileged user. The first command succeeds and sets the hostname
inside the new UTS namespace; the second is denied because the network
namespace type is not allowed:
LL_FS_RO=/ LL_FS_RW=/proc LL_NS="user:uts" \
./sandboxer /bin/sh -c \
"unshare --user --uts --map-root-user hostname sandbox \
&& ! unshare --user --net true"
Allow only user namespace creation and CAP_SYS_CHROOT, denying all other
capabilities and namespace types. An unprivileged process creates a
user namespace, which requires no capability, and calls chroot inside it
using the CAP_SYS_CHROOT granted within that namespace:
LL_FS_RO=/ LL_FS_RW="" LL_NS="user" LL_CAP="cap_sys_chroot" \
./sandboxer /bin/sh -c \
"unshare --user --keep-caps chroot / true"
Allow user namespace creation but deny network namespaces, and quiet the
network-namespace denials with LL_NS_QUIET so the denied creation is not
audit logged. The negated second command makes the whole line succeed:
LL_FS_RO=/ LL_FS_RW=/proc LL_NS="user" LL_NS_QUIET="net" \
./sandboxer /bin/sh -c \
"unshare --user --map-root-user true && ! unshare --user --net true"
Cc: Christian Brauner <brauner@kernel.org>
Cc: Günther Noack <gnoack@google.com>
Cc: Paul Moore <paul@paul-moore.com>
Cc: Serge E. Hallyn <serge@hallyn.com>
Cc: Tingmao Wang <m@maowtm.org>
Signed-off-by: Mickaël Salaün <mic@digikod.net>
---
Changes since v3:
https://patch.msgid.link/20260726161400.3010511-12-mic@digikod.net
- Bump the sample's latest supported ABI to 12.
- Gate namespace and capability permissions and rule calls on ABI 12.
Consume unsupported settings before executing the sandboxed command.
- Reject capability numbers that cannot fit in the UAPI's 64-bit mask.
- Expand the Kconfig help from filesystem-only wording to the complete
sandboxer policy and state the libcap development-file dependency.
- Clarify that quiet capability and namespace members suppress audit
logging, and that quieting alone still restricts the permission.
Changes since v2:
https://patch.msgid.link/20260527181127.879771-9-mic@digikod.net
- Rebased for ABI 11: bump LANDLOCK_ABI_LAST to 11 and extend the ABI
back-compat fall-through with a new case stripping the LANDLOCK_PERM_*
handled bits for ABI < 11.
- Adopt the renamed capability and namespace rule attributes: the rule
bodies use perm plus allowed_capabilities / allowed_namespace_types
(matching the per-member quiet restructuring in the enforcement
patches).
- Add LL_CAP_QUIET and LL_NS_QUIET to quiet capability and
namespace-type denials (per-member, merged into the allowed rule).
Changes since v1:
https://patch.msgid.link/20260312100444.2609563-11-mic@digikod.net
- Rename LANDLOCK_PERM_NAMESPACE_ENTER references to
LANDLOCK_PERM_NAMESPACE_USE (companion change to the introducing
commit).
- Replace handled_perm = 0 with a per-bit mask in the ABI compat
fall-through, mirroring the doc example so future ABI extensions
adding new LANDLOCK_PERM_* bits do not get stripped.
- Parse LL_CAP values with cap_from_name(3) from libcap so users
can pass capability names (e.g. "cap_sys_chroot") in addition to
numbers. cap_from_name accepts both: the canonical name lookup
is case-insensitive, and a numeric-string fallback maps "18" to
CAP_SYS_CHROOT identically to the previous numeric-only path.
Drop the BITS_PER_TYPE workaround and the manual numeric bound
check (cap_from_name does the right thing in both cases). Link
the sandboxer against libcap by adding userldlibs += -lcap in
samples/landlock/Makefile. Update help text and example command
to show capability names (suggested by Günther Noack).
- Rename the LL_CAPS env var to LL_CAP for consistency with the
singular form of all other sandboxer env vars (LL_NS, LL_FS_RO,
LL_FS_RW, LL_TCP_BIND, LL_TCP_CONNECT, LL_SCOPED, LL_FORCE_LOG).
Internal symbols renamed accordingly: ENV_CAPS_NAME -> ENV_CAP_NAME,
populate_ruleset_caps() -> populate_ruleset_cap().
- Tingmao Wang's v1 Reviewed-by is not carried forward to v2: the
cap_from_name() / libcap migration is a material implementation
change requested by Günther Noack that was not part of his
review. Cc'd instead.
---
samples/Kconfig | 6 +-
samples/landlock/Makefile | 1 +
samples/landlock/sandboxer.c | 230 ++++++++++++++++++++++++++++++++++-
3 files changed, 232 insertions(+), 5 deletions(-)
diff --git a/samples/Kconfig b/samples/Kconfig
index a75e8e78330d..b18efc19b85d 100644
--- a/samples/Kconfig
+++ b/samples/Kconfig
@@ -166,8 +166,10 @@ config SAMPLE_LANDLOCK
bool "Landlock example"
depends on CC_CAN_LINK && HEADERS_INSTALL
help
- Build a simple Landlock sandbox manager able to start a process
- restricted by a user-defined filesystem access control policy.
+ Build a Landlock sandbox manager able to start a process restricted
+ by user-defined filesystem, network, scope, namespace, and capability
+ policies. This sample requires the libcap development headers and
+ library.
config SAMPLE_PIDFD
bool "pidfd sample"
diff --git a/samples/landlock/Makefile b/samples/landlock/Makefile
index 5d601e51c2eb..b30239c8a281 100644
--- a/samples/landlock/Makefile
+++ b/samples/landlock/Makefile
@@ -3,6 +3,7 @@
userprogs-always-y := sandboxer
userccflags += -I usr/include
+userldlibs += -lcap
.PHONY: all clean
diff --git a/samples/landlock/sandboxer.c b/samples/landlock/sandboxer.c
index 030583273f3f..4a86ae6d4552 100644
--- a/samples/landlock/sandboxer.c
+++ b/samples/landlock/sandboxer.c
@@ -12,17 +12,20 @@
#include <arpa/inet.h>
#include <errno.h>
#include <fcntl.h>
+#include <limits.h>
#include <linux/landlock.h>
#include <linux/socket.h>
+#include <sched.h>
+#include <stdbool.h>
#include <stddef.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
+#include <sys/capability.h>
#include <sys/prctl.h>
#include <sys/stat.h>
#include <sys/syscall.h>
#include <unistd.h>
-#include <stdbool.h>
#if defined(__GLIBC__)
#include <linux/prctl.h>
@@ -62,6 +65,10 @@ static inline int landlock_restrict_self(const int ruleset_fd,
#define ENV_TCP_BIND_NAME "LL_TCP_BIND"
#define ENV_TCP_CONNECT_NAME "LL_TCP_CONNECT"
#define ENV_NET_QUIET_NAME "LL_NET_QUIET"
+#define ENV_NS_NAME "LL_NS"
+#define ENV_NS_QUIET_NAME "LL_NS_QUIET"
+#define ENV_CAP_NAME "LL_CAP"
+#define ENV_CAP_QUIET_NAME "LL_CAP_QUIET"
#define ENV_SCOPED_NAME "LL_SCOPED"
#define ENV_QUIET_ACCESS_NAME "LL_QUIET_ACCESS"
#define ENV_FORCE_LOG_NAME "LL_FORCE_LOG"
@@ -69,6 +76,8 @@ static inline int landlock_restrict_self(const int ruleset_fd,
#define ENV_UDP_CONNECT_SEND_NAME "LL_UDP_CONNECT_SEND"
#define ENV_DELIMITER ":"
+#define ARRAY_SIZE(array) (sizeof(array) / sizeof((array)[0]))
+
static int str2num(const char *numstr, __u64 *num_dst)
{
char *endptr = NULL;
@@ -232,6 +241,166 @@ static int populate_ruleset_net(const char *const env_var, const int ruleset_fd,
return ret;
}
+static __u64 str2ns(const char *const name)
+{
+ static const struct {
+ const char *name;
+ __u64 value;
+ } ns_map[] = {
+ /* clang-format off */
+ { "cgroup", CLONE_NEWCGROUP },
+ { "ipc", CLONE_NEWIPC },
+ { "mnt", CLONE_NEWNS },
+ { "net", CLONE_NEWNET },
+ { "pid", CLONE_NEWPID },
+ { "time", CLONE_NEWTIME },
+ { "user", CLONE_NEWUSER },
+ { "uts", CLONE_NEWUTS },
+ /* clang-format on */
+ };
+ size_t i;
+
+ for (i = 0; i < ARRAY_SIZE(ns_map); i++) {
+ if (strcmp(name, ns_map[i].name) == 0)
+ return ns_map[i].value;
+ }
+ return 0;
+}
+
+/*
+ * Parses a colon-delimited list of namespace type names into a bitmask.
+ * Returns 0 on success (mask 0 when the variable is unset or empty), or 1 on a
+ * parse error.
+ */
+static int parse_ns_list(const char *const env_var, __u64 *const mask)
+{
+ int ret = 1;
+ char *env_ns_name, *env_ns_name_next, *strns;
+
+ *mask = 0;
+ env_ns_name = getenv(env_var);
+ if (!env_ns_name)
+ return 0;
+ env_ns_name = strdup(env_ns_name);
+ unsetenv(env_var);
+
+ env_ns_name_next = env_ns_name;
+ while ((strns = strsep(&env_ns_name_next, ENV_DELIMITER))) {
+ __u64 ns_type;
+
+ if (strcmp(strns, "") == 0)
+ continue;
+
+ ns_type = str2ns(strns);
+ if (!ns_type) {
+ fprintf(stderr, "Unknown namespace type \"%s\"\n",
+ strns);
+ goto out_free_name;
+ }
+ *mask |= ns_type;
+ }
+ ret = 0;
+
+out_free_name:
+ free(env_ns_name);
+ return ret;
+}
+
+static int populate_ruleset_ns(const char *const allowed_env,
+ const char *const quiet_env,
+ const int ruleset_fd)
+{
+ struct landlock_namespace_attr ns_attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ };
+
+ if (parse_ns_list(allowed_env, &ns_attr.allowed_namespace_types))
+ return 1;
+ if (parse_ns_list(quiet_env, &ns_attr.quiet_namespace_types))
+ return 1;
+
+ if (!ns_attr.allowed_namespace_types && !ns_attr.quiet_namespace_types)
+ return 0;
+
+ if (landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE, &ns_attr,
+ 0)) {
+ fprintf(stderr,
+ "Failed to update the ruleset with namespace types: %s\n",
+ strerror(errno));
+ return 1;
+ }
+ return 0;
+}
+
+/*
+ * Parses a colon-delimited list of capability names into a bitmask. Returns 0
+ * on success (mask 0 when the variable is unset or empty), or 1 on a parse
+ * error.
+ */
+static int parse_cap_list(const char *const env_var, __u64 *const mask)
+{
+ int ret = 1;
+ char *env_cap_name, *env_cap_name_next, *strcap;
+
+ *mask = 0;
+ env_cap_name = getenv(env_var);
+ if (!env_cap_name)
+ return 0;
+ env_cap_name = strdup(env_cap_name);
+ unsetenv(env_var);
+
+ env_cap_name_next = env_cap_name;
+ while ((strcap = strsep(&env_cap_name_next, ENV_DELIMITER))) {
+ cap_value_t cap;
+
+ if (strcmp(strcap, "") == 0)
+ continue;
+
+ if (cap_from_name(strcap, &cap)) {
+ fprintf(stderr, "Failed to parse capability \"%s\"\n",
+ strcap);
+ goto out_free_name;
+ }
+ if ((unsigned int)cap >= sizeof(*mask) * CHAR_BIT) {
+ fprintf(stderr, "Capability \"%s\" is out of range\n",
+ strcap);
+ goto out_free_name;
+ }
+ *mask |= 1ULL << cap;
+ }
+ ret = 0;
+
+out_free_name:
+ free(env_cap_name);
+ return ret;
+}
+
+static int populate_ruleset_cap(const char *const allowed_env,
+ const char *const quiet_env,
+ const int ruleset_fd)
+{
+ struct landlock_capability_attr cap_attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ };
+
+ if (parse_cap_list(allowed_env, &cap_attr.allowed_capabilities))
+ return 1;
+ if (parse_cap_list(quiet_env, &cap_attr.quiet_capabilities))
+ return 1;
+
+ if (!cap_attr.allowed_capabilities && !cap_attr.quiet_capabilities)
+ return 0;
+
+ if (landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY, &cap_attr,
+ 0)) {
+ fprintf(stderr,
+ "Failed to update the ruleset with capabilities: %s\n",
+ strerror(errno));
+ return 1;
+ }
+ return 0;
+}
+
/* Returns true on error, false otherwise. */
static bool check_ruleset_scope(const char *const env_var,
struct landlock_ruleset_attr *ruleset_attr)
@@ -369,7 +538,7 @@ static int add_quiet_access(const char *const env_var,
return 0;
}
-#define LANDLOCK_ABI_LAST 11
+#define LANDLOCK_ABI_LAST 12
#define XSTR(s) #s
#define STR(s) XSTR(s)
@@ -397,6 +566,22 @@ static const char help[] =
"* " ENV_UDP_CONNECT_SEND_NAME ": remote UDP ports allowed to connect "
"or send to (client: use as destination port / server: receive only from it)\n"
"(caution: sending requires being able to bind to a local source port)\n"
+ "* " ENV_NS_NAME ": namespace types allowed to use\n"
+ " (cgroup, ipc, mnt, net, pid, time, user, uts)\n"
+ "* " ENV_NS_QUIET_NAME
+ ": namespace types whose denial should not be audit logged\n"
+ " (same value format as " ENV_NS_NAME
+ "; quieting an allowed member is inert,\n"
+ " as an allowed member is never denied; setting it alone still\n"
+ " restricts namespace use, denying every type)\n"
+ "* " ENV_CAP_NAME ": capabilities allowed to use, as names or numbers\n"
+ " (e.g. cap_net_bind_service, cap_sys_admin, 18)\n"
+ "* " ENV_CAP_QUIET_NAME
+ ": capabilities whose denial should not be audit logged\n"
+ " (same value format as " ENV_CAP_NAME
+ "; quieting an allowed member is inert,\n"
+ " as an allowed member is never denied; setting it alone still\n"
+ " restricts capability use, denying every capability)\n"
"* " ENV_SCOPED_NAME ": actions denied on the outside of the landlock domain\n"
" - \"a\" to restrict opening abstract unix sockets\n"
" - \"s\" to restrict sending signals\n"
@@ -423,6 +608,8 @@ static const char help[] =
ENV_TCP_BIND_NAME "=\"9418\" "
ENV_TCP_CONNECT_NAME "=\"80:443\" "
ENV_UDP_CONNECT_SEND_NAME "=\"53\" "
+ ENV_NS_NAME "=\"user:uts:net\" "
+ ENV_CAP_NAME "=\"cap_sys_admin\" "
ENV_SCOPED_NAME "=\"a:s\" "
"%1$s bash -i\n"
"\n"
@@ -451,6 +638,8 @@ int main(const int argc, char *const argv[], char *const *const envp)
.quiet_access_fs = 0,
.quiet_access_net = 0,
.quiet_scoped = 0,
+ .handled_permissions = LANDLOCK_PERMISSION_NAMESPACE_USE |
+ LANDLOCK_PERMISSION_CAPABILITY_USE,
};
bool quiet_supported = true;
int supported_restrict_flags = LANDLOCK_RESTRICT_SELF_LOG_NEW_EXEC_ON |
@@ -552,7 +741,12 @@ int main(const int argc, char *const argv[], char *const *const envp)
supported_restrict_flags &=
~LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS;
set_restrict_flags &= ~LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS;
-
+ __attribute__((fallthrough));
+ case 11:
+ /* Removes LANDLOCK_PERMISSION_* for ABI < 12 */
+ ruleset_attr.handled_permissions &=
+ ~(LANDLOCK_PERMISSION_NAMESPACE_USE |
+ LANDLOCK_PERMISSION_CAPABILITY_USE);
/* Must be printed for any ABI < LANDLOCK_ABI_LAST. */
fprintf(stderr,
"Hint: You should update the running kernel "
@@ -597,6 +791,26 @@ int main(const int argc, char *const argv[], char *const *const envp)
~LANDLOCK_ACCESS_NET_CONNECT_SEND_UDP;
}
+ /* Removes namespace handling if not set by a user. */
+ if (!getenv(ENV_NS_NAME) && !getenv(ENV_NS_QUIET_NAME))
+ ruleset_attr.handled_permissions &=
+ ~LANDLOCK_PERMISSION_NAMESPACE_USE;
+ if (!(ruleset_attr.handled_permissions &
+ LANDLOCK_PERMISSION_NAMESPACE_USE)) {
+ unsetenv(ENV_NS_NAME);
+ unsetenv(ENV_NS_QUIET_NAME);
+ }
+
+ /* Removes capability handling if not set by a user. */
+ if (!getenv(ENV_CAP_NAME) && !getenv(ENV_CAP_QUIET_NAME))
+ ruleset_attr.handled_permissions &=
+ ~LANDLOCK_PERMISSION_CAPABILITY_USE;
+ if (!(ruleset_attr.handled_permissions &
+ LANDLOCK_PERMISSION_CAPABILITY_USE)) {
+ unsetenv(ENV_CAP_NAME);
+ unsetenv(ENV_CAP_QUIET_NAME);
+ }
+
if (check_ruleset_scope(ENV_SCOPED_NAME, &ruleset_attr))
return 1;
@@ -680,6 +894,16 @@ int main(const int argc, char *const argv[], char *const *const envp)
}
}
+ if ((ruleset_attr.handled_permissions &
+ LANDLOCK_PERMISSION_NAMESPACE_USE) &&
+ populate_ruleset_ns(ENV_NS_NAME, ENV_NS_QUIET_NAME, ruleset_fd))
+ goto err_close_ruleset;
+
+ if ((ruleset_attr.handled_permissions &
+ LANDLOCK_PERMISSION_CAPABILITY_USE) &&
+ populate_ruleset_cap(ENV_CAP_NAME, ENV_CAP_QUIET_NAME, ruleset_fd))
+ goto err_close_ruleset;
+
if (!(set_restrict_flags & LANDLOCK_RESTRICT_SELF_NO_NEW_PRIVS) &&
prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0)) {
perror("Failed to restrict privileges");
--
2.55.0
^ permalink raw reply [flat|nested] 9+ messages in thread* [PATCH v4 8/8] landlock: Add documentation for capability and namespace restrictions
2026-10-02 12:43 [PATCH v4 0/8] Landlock: Namespace and capability control Mickaël Salaün
` (6 preceding siblings ...)
2026-10-02 12:44 ` [PATCH v4 7/8] samples/landlock: Add capability and namespace restriction support Mickaël Salaün
@ 2026-10-02 12:44 ` Mickaël Salaün
7 siblings, 0 replies; 9+ messages in thread
From: Mickaël Salaün @ 2026-10-02 12:44 UTC (permalink / raw)
To: Christian Brauner, Günther Noack, Paul Moore, Serge E . Hallyn
Cc: Mickaël Salaün, Daniel Durning, Jonathan Corbet,
Justin Suess, Lennart Poettering, Mikhail Ivanov,
Nicolas Bouchinet, Shervin Oloumi, Tingmao Wang, kernel-team,
linux-fsdevel, linux-kernel, linux-security-module
Document the two new Landlock permission categories, introduced with ABI
12, in the userspace API guide, admin guide, trace-event guide, UAPI
comments, and kernel security documentation.
The kernel security documentation generalizes the existing per-object
access-rights framing into three restriction models (handled_access_*,
handled_permissions, scoped) and states criteria for choosing among
them, so future permission or scope additions have a documented place to
land.
Cc: Christian Brauner <brauner@kernel.org>
Cc: Günther Noack <gnoack@google.com>
Cc: Paul Moore <paul@paul-moore.com>
Cc: Serge E. Hallyn <serge@hallyn.com>
Signed-off-by: Mickaël Salaün <mic@digikod.net>
---
Changes since v3:
https://patch.msgid.link/20260726161400.3010511-13-mic@digikod.net
- Document the new permissions as ABI 12.
- Name the documented audit blockers namespace.use and capability.use
after their domains.
- Document handled_permissions, the four permission rule and denial
events, effective no-ops, raw member values, namespace IDs, and
audit-suppressed denials; add handled_permissions to the basic trace
example.
- Describe the namespace creation and entry paths, and state that quiet
denials stay trace-visible.
- Describe the quiet ruleset fields as suppressing audit records rather
than logs, now that denials are also trace events.
Changes since v2:
https://patch.msgid.link/20260527181127.879771-10-mic@digikod.net
- Fix the CAP_BPF forward-compatibility example in
Documentation/security/landlock.rst: rules allow-list, so describe a
rule *allowing* CAP_SYS_ADMIN continuing to allow operations now
gated by CAP_SYS_ADMIN || CAP_BPF (was worded as a rule "denying";
suggested by Günther Noack).
- Drop the "category" jargon from the Capability rules and Namespace
rules summaries in Documentation/userspace-api/landlock.rst and add
capabilities(7) / namespaces(7) cross-references there, in the rule
examples, and in the ABI restriction sections (suggested by Günther
Noack).
- Reword LANDLOCK_PERM_NAMESPACE_USE as gating "acquisition of access
to namespaces" (was "acquisition of namespace associations") and
"allow using" (was "allow entering") specific namespace types, to
match the _USE semantics covering creation, entry, and fd-reference
(suggested by Günther Noack).
- Document the per-member audit quieting for the new permissions in the
Capability and namespace restrictions section: the quiet_capabilities
and quiet_namespace_types rule fields suppress the audit records of
specific denied members (independently of the allowed members),
require the permission to be handled, are member-granular, and should
quiet only known-expected members, never a blanket set.
- Documentation/security/landlock.rst prose polish (suggested by
Günther Noack):
hyphenate "user-namespace-based"; reword the design-philosophy
operation-set sentence; add an introductory paragraph to "Ruleset
restriction models" (the ruleset denies by default, rules allow-list
exceptions); split the per-object and per-category sections into
shorter paragraphs and lead the per-category one with "Each
LANDLOCK_PERM_* flag maps to its own rule type"; reframe the
per-object new-operation/new-object-type text around object type
(was "rule type") and tighten it.
- State the combined deny-by-default explicitly in
Documentation/userspace-api/landlock.rst: when a domain handles both
permissions, an operation is allowed only if each independently
allows it (e.g. unshare(CLONE_NEWNET) needs both CLONE_NEWNET and
CAP_SYS_ADMIN allowed).
- Note that the open_tree(OPEN_TREE_CLONE) / move_mount anonymous-mount
bypass path requires CAP_SYS_ADMIN (typically obtained by first
creating a user namespace).
- Add a concrete downstream-operations example to the per-category
model in Documentation/security/landlock.rst (denying CAP_SYS_ADMIN
blocks all admin operations such as mounts).
- Document that creating a non-user namespace always requires
CAP_SYS_ADMIN, so when the domain handles LANDLOCK_PERM_CAPABILITY_USE
a rule allowing CAP_SYS_ADMIN is needed even when CLONE_NEWUSER is
combined in the same unshare() call; only user namespace creation
needs no capability rule.
Changes since v1:
https://patch.msgid.link/20260312100444.2609563-12-mic@digikod.net
The userspace API and security guides were revamped to match the v2
permission model: the previous chokepoints/gateways prose is replaced
with the per-object (handled_access_*) versus per-category
(handled_perm) framing, and a new Design philosophy section in the
security guide states Landlock's principle (data, processes, kernel
resources).
- Rename namespace_inum to namespace_id in audit field documentation
to match the renamed audit field.
- Rename LANDLOCK_PERM_NAMESPACE_ENTER references to
LANDLOCK_PERM_NAMESPACE_USE (companion change to the introducing
commit), and enumerate the seven kernel paths it gates in the
userspace API guide (membership via unshare/clone/clone3/setns; fd
reference via open_tree/fsmount).
- Clarify that LANDLOCK_PERM_NAMESPACE_USE gates *acquisition* of
namespace associations only (namespaces the process is already a
member of when the domain is enforced are implicitly allowed) and
that LANDLOCK_PERM_CAPABILITY_USE gates every exercise of a
capability after the domain is enforced, regardless of how the
capability was obtained.
- Document the rationale for accepting (rather than rejecting)
unknown category member values in rule bodies: rejection would tie
Landlock policy semantics to the running kernel's category-member
set, making cross-kernel policies brittle. Acceptance is fail-safe
in both directions and lets a policy activate as written when a
value becomes real on a future kernel.
- Replace handled_perm = 0 with a per-bit mask in the userspace API
guide's ABI compat fall-through, so future ABI extensions adding
new LANDLOCK_PERM_* bits do not get stripped on the path that
drops the v10 bits.
- Add a bridging sentence in the per-category permissions section
of Documentation/security/landlock.rst contrasting per-category
permissions with per-object access rights: per-category gates the
prerequisite operation itself rather than restricting specific
operations on a single resource instance (suggested by Günther
Noack).
- Disambiguate the orthogonality invariant in
Documentation/security/landlock.rst from the UAPI scoped field
("all new scoped features" -> "all Landlock access controls";
suggested by Justin Suess).
- Add an introductory paragraph in
Documentation/userspace-api/landlock.rst contrasting
LANDLOCK_PERM_CAPABILITY_USE with PR_SET_NO_NEW_PRIVS: NNP is the
broader mechanism that blocks privilege acquisition via execve(2),
while CAPABILITY_USE restricts the exercise of capabilities the
process already holds (including those gained via CLONE_NEWUSER,
which NNP does not block); sandboxes typically set both
(suggested by Justin Suess).
- Disambiguate "category": object-side uses "object type" / "resource
kind"; "category" stays for the per-category permissions model.
---
Documentation/admin-guide/LSM/landlock.rst | 46 ++--
Documentation/security/landlock.rst | 178 ++++++++++++++-
Documentation/trace/events-landlock.rst | 66 ++++--
Documentation/userspace-api/landlock.rst | 243 +++++++++++++++++++--
include/trace/events/landlock.h | 11 +-
include/uapi/linux/landlock.h | 17 +-
6 files changed, 503 insertions(+), 58 deletions(-)
diff --git a/Documentation/admin-guide/LSM/landlock.rst b/Documentation/admin-guide/LSM/landlock.rst
index 2998552bb730..718e0d23ffb5 100644
--- a/Documentation/admin-guide/LSM/landlock.rst
+++ b/Documentation/admin-guide/LSM/landlock.rst
@@ -7,7 +7,7 @@ Landlock: system-wide management
================================
:Author: Mickaël Salaün
-:Date: August 2026
+:Date: October 2026
Landlock can leverage the audit framework to log events.
@@ -20,10 +20,12 @@ Audit
Denied access requests are logged by default for a sandboxed program if `audit`
is enabled. This default behavior can be changed with the
sys_landlock_restrict_self() flags (cf.
-Documentation/userspace-api/landlock.rst), or suppressed on a per-object
-basis by using ``LANDLOCK_ADD_RULE_QUIET`` (ABI 10+). Landlock logs can
-also be masked thanks to audit rules. Landlock can generate 2 audit
-record types.
+Documentation/userspace-api/landlock.rst), suppressed on a per-object
+basis by using ``LANDLOCK_ADD_RULE_QUIET`` (ABI 10+), or suppressed for
+specific capability or namespace members with the corresponding permission
+rule's ``quiet_*`` field (ABI 12+). Quiet rules only suppress audit submission;
+they do not suppress denial trace events. Landlock logs can also be masked
+thanks to audit rules. Landlock can generate two audit record types.
Record types
------------
@@ -65,14 +67,28 @@ AUDIT_LANDLOCK_ACCESS
- scope.abstract_unix_socket - Abstract UNIX socket connection denied
- scope.signal - Signal sending denied
+ **namespace.*** - Namespace restrictions (ABI 12+):
+ - namespace.use - Namespace use was denied (creation via
+ :manpage:`unshare(2)`, :manpage:`clone(2)`, :manpage:`clone3(2)`,
+ :manpage:`open_tree(2)`, or :manpage:`fsmount(2)`, or joining via
+ :manpage:`setns(2)`);
+ ``namespace_type`` indicates the type (hex ``CLONE_NEW*`` bitmask),
+ and ``namespace_id`` identifies the target namespace for
+ :manpage:`setns(2)` operations (zero for creation)
+
+ **capability.*** - Capability restrictions (ABI 12+):
+ - capability.use - Capability use was denied;
+ ``capability`` indicates the capability number
+
Multiple blockers can appear in a single event (comma-separated) when
multiple access rights are missing. For example, creating a regular file
in a directory that lacks both ``make_reg`` and ``refer`` rights would show
``blockers=fs.make_reg,fs.refer``.
- The object identification fields (path, dev, ino for filesystem; opid,
- ocomm for signals) depend on the type of access being blocked and provide
- context about what resource was involved in the denial.
+ The object identification fields depend on the type of access being blocked:
+ ``path``, ``dev``, ``ino`` for filesystem; ``opid``, ``ocomm`` for signals;
+ ``namespace_type`` and ``namespace_id`` for namespace operations;
+ ``capability`` for capability use.
AUDIT_LANDLOCK_DOMAIN
@@ -178,8 +194,9 @@ If you get spammed with audit logs related to Landlock, this is either an
attack attempt or a bug in the security policy. We can put in place some
filters to limit noise with two complementary ways:
-- with sys_landlock_restrict_self()'s flags, or
- ``LANDLOCK_ADD_RULE_QUIET`` (ABI 10+) if we can fix the sandboxed
+- with sys_landlock_restrict_self()'s flags,
+ ``LANDLOCK_ADD_RULE_QUIET`` (ABI 10+), or the permission rules'
+ per-member ``quiet_*`` fields (ABI 12+) if we can fix the sandboxed
programs,
- or with audit rules (see :manpage:`auditctl(8)`).
@@ -243,8 +260,10 @@ with these exceptions:
- **NOAUDIT hooks**: Some LSM hooks suppress logging for speculative
permission probes (e.g., reading ``/proc/<pid>/status`` uses
- ``PTRACE_MODE_NOAUDIT``). When NOAUDIT is set, neither audit records
- nor trace events are emitted, and the denial is not counted in
+ ``PTRACE_MODE_NOAUDIT``, and :manpage:`listxattr(2)` probes
+ ``CAP_SYS_ADMIN`` with ``CAP_OPT_NOAUDIT`` to decide whether to list
+ ``trusted.*`` extended attributes). When NOAUDIT is set, neither audit
+ records nor trace events are emitted, and the denial is not counted in
``denials``. The denial is still enforced. This avoids performance
overhead and noise from speculative probes that test permissions
without performing an actual access.
@@ -269,7 +288,8 @@ Observability security considerations
Both audit records and trace events expose information about all
Landlock-sandboxed processes on the system, including filesystem paths
-being accessed, network ports, and process identities. System
+being accessed, network ports, capability numbers, namespace types and IDs,
+and process identities. System
administrators must ensure that access to audit logs (controlled by the
audit subsystem configuration) and to trace events (requiring
``CAP_SYS_ADMIN`` or ``CAP_BPF`` + ``CAP_PERFMON``) is restricted to
diff --git a/Documentation/security/landlock.rst b/Documentation/security/landlock.rst
index 2d6e1076484e..a5ac758ff299 100644
--- a/Documentation/security/landlock.rst
+++ b/Documentation/security/landlock.rst
@@ -8,7 +8,7 @@ Landlock LSM: kernel documentation
==================================
:Author: Mickaël Salaün
-:Date: August 2026
+:Date: October 2026
Landlock's goal is to create scoped access-control (i.e. sandboxing). To
harden a whole system, this feature should be available to any process,
@@ -130,6 +130,167 @@ The reasoning is:
restrictions, because access within the same scope is already
allowed based on ``LANDLOCK_ACCESS_FS_RESOLVE_UNIX``.
+Composability with user namespaces
+----------------------------------
+
+Landlock domain-based scoping and the kernel's user-namespace-based capability
+scoping enforce isolation over independent hierarchies. Landlock checks domain
+ancestry; the kernel's ``ns_capable()`` checks user namespace ancestry. These
+hierarchies are orthogonal: Landlock enforcement is deterministic with respect
+to its own configuration, regardless of namespace or capability state, and vice
+versa. This orthogonality is a design invariant that must hold for all Landlock
+access controls.
+
+Design philosophy
+-----------------
+
+Landlock's goal is to restrict a sandboxed process's access to three kinds of
+resources: data (files, sockets, pipes), other processes (signals, ptrace), and
+kernel-internal resources whose use widens the kernel attack surface
+(capabilities, namespace types). Each access right or permission gates one or
+more operations that grant such access; restricting the operations is how
+Landlock restricts the underlying access.
+
+When designing a new access control, identify the protected resource kind
+first (data, processes, or kernel-internal resources). The operations to
+restrict follow from the protected resource, by identifying which kernel code
+paths grant access to the resource and at which place in the code the access to
+the resource can be gated. Do not design a permission around
+"restrict the unshare(2) syscall" or similar mechanism-centric framings; design
+it around "restrict the process from acquiring access to namespace types" (the
+protected resource), letting the operation set follow.
+
+Ruleset restriction models
+--------------------------
+
+Landlock provides three restriction models that differ in how rules identify the
+resource being restricted.
+
+In general, the ``struct landlock_ruleset_attr`` specifies the operations to be
+denied by default under the enforced policy. The *rules* added to the ruleset
+define the exceptions to these restrictions, allow-listing specific conditions
+under which these operations are still permitted.
+
+Per-object access rights (``handled_access_*``)
+~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+
+Per-object access rights control operations on a specific resource instance,
+identified in the rule key by a value drawn from an open-ended space: a file
+hierarchy referenced by ``parent_fd``, or a network port identified by its
+16-bit number.
+
+Each ``handled_access_*`` field declares a set of access rights, operations
+which are to be denied by default once the ruleset is enforced.
+
+The rule body declares which of the multiple distinct operations on that object
+instance are allowed (open, read, write, truncate; bind, connect).
+
+Operations are grouped by object type in the respective ``handled_access_*``
+field: a new operation on an existing type extends that field, and a new object
+type gets its own ``handled_access_*`` field.
+
+Per-category permissions (``handled_permissions``)
+~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+
+Per-category permissions control the process's exercise of category members,
+where the category is a small kernel-defined enumeration (a Linux capability
+number ``CAP_*``, a namespace type ``CLONE_NEW*``). Unlike per-object access
+rights, which restrict specific operations on a single resource instance,
+per-category permissions gate the prerequisite operation itself (exercising a
+capability, acquiring access to a namespace), so gating it transitively covers a
+broad set of downstream operations (e.g. denying ``CAP_SYS_ADMIN`` blocks all
+admin operations such as mounts).
+
+These category members are the LSM-level access-control objects (the entities
+the process is authorized against) even though they are enum values rather than
+externally-instantiated kernel data structures. Per-category permissions apply
+where the controlled operation collapses to "may the process use this category
+member at all" (use a capability; acquire access to a namespace), so the rule
+body lists which category members the process may exercise.
+
+Each ``LANDLOCK_PERMISSION_*`` flag maps to its own rule type and covers every
+kernel path that exercises a member. When a ruleset handles a permission, all
+uses of category members are denied unless explicitly allowed by a rule.
+
+Logging of a denied member can be suppressed at the same granularity as the
+restriction itself: per rule and per member. A rule names the members whose
+denial should not be audited; suppression only affects the audit record, never
+the denial or its trace event, which reports ``logged=0``. Suppression only
+applies when the layer owning the rule is the one that denied the member.
+
+See Documentation/userspace-api/landlock.rst for the concrete syscall paths
+covered by each permission.
+
+The category enum is owned by the corresponding kernel subsystem (capabilities,
+namespaces, etc.). Userspace policy authors query category member availability
+via the relevant non-Landlock interfaces:
+
+* For capabilities: ``<linux/capability.h>``,
+ ``/proc/sys/kernel/cap_last_cap``, ``prctl(PR_CAPBSET_READ)``.
+* For namespaces: ``<linux/sched.h>``, ``/proc/$$/ns/*``,
+ :manpage:`unshare(2)` runtime probe.
+
+The Landlock ABI version does not encode this availability; ABI versioning
+describes which Landlock features (rule types, access rights, scopes,
+permissions) the kernel implements, not which category members the kernel knows
+about.
+
+Forward compatibility for new category members follows a simple rule set:
+
+* New members in future kernels are automatically denied: rules whitelist
+ specific values, and a member not in any rule is denied.
+* Kernel-side compatibility for split categories is handled by the owning
+ subsystem (e.g., when ``CAP_BPF`` was split from ``CAP_SYS_ADMIN``, either
+ capability became sufficient for the affected operations, so a rule allowing
+ ``CAP_SYS_ADMIN`` continues to allow operations now gated by
+ ``CAP_SYS_ADMIN || CAP_BPF``).
+* Unknown values in the rule body are silently accepted rather than rejected.
+ Rejecting them would tie Landlock policy semantics to the running kernel's
+ category-member set: a rule built against future headers would fail to load
+ on older kernels, forcing policy authors to know each kernel's enumeration.
+ Acceptance is fail-safe in both directions: a rule referring to a value the
+ running kernel does not yet know has no effect (deny-by-default still applies
+ to that operation), and a rule written against future headers loads
+ identically across kernels so the same policy keeps the same restrictions.
+ When a value becomes real on a future kernel, the policy activates as written
+ by the author.
+* In contrast, unknown ``LANDLOCK_PERMISSION_*`` flags in
+ ``handled_permissions`` are rejected (``-EINVAL``), since Landlock owns that
+ bit space.
+
+Cross-domain scopes (``scoped``)
+~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+
+Scopes restrict **cross-domain interactions** categorically, without rules.
+Setting a scope flag (e.g. ``LANDLOCK_SCOPE_SIGNAL``) denies the operation to
+targets outside the Landlock domain or its children. Like per-category
+permissions, scopes provide complete coverage of the controlled operation.
+
+Choosing a model for a new feature
+~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
+
+* If the new feature controls operations on resource objects supplied by the
+ sandbox author, extend or add a per-object access right
+ (``handled_access_*``).
+* If the new feature controls a per-category operation gated by an enum (a
+ Linux capability, a namespace type, a socket family, etc.), use a
+ per-category permission (``handled_permissions``). When several such enums
+ could classify the operation, prefer the enum the originating subsystem
+ already
+ uses for capability/access checks (e.g. ``CAP_*`` for ``capable()`` hooks,
+ ``CLONE_NEW*`` for namespace hooks).
+* When an operation is gated by multiple kernel-defined enums (a classic
+ example being ``CAP_SYS_ADMIN`` plus a ``CLONE_NEW*`` flag for non-user
+ namespace creation), define one per-category permission per enum dimension.
+ Sandbox authors handle each dimension's permission in
+ ``handled_permissions`` and add rules for each; the kernel enforces each
+ dimension at its own LSM hook. ``LANDLOCK_PERMISSION_NAMESPACE_USE`` and
+ ``LANDLOCK_PERMISSION_CAPABILITY_USE`` follow this pattern.
+* If the new feature restricts a categorical cross-domain interaction with no
+ per-target granularity, use a cross-domain scope (``scoped``).
+* For all three models, confirm a single LSM hook (or small set of related
+ hooks) covers every kernel path that exercises the operation.
+
Tests
=====
@@ -151,6 +312,18 @@ Filesystem
.. kernel-doc:: security/landlock/fs.h
:identifiers:
+Namespace
+---------
+
+.. kernel-doc:: security/landlock/ns.h
+ :identifiers:
+
+Capability
+----------
+
+.. kernel-doc:: security/landlock/cap.h
+ :identifiers:
+
Process credential
------------------
@@ -198,7 +371,8 @@ whether or not the kernel is built with audit support, so a
tracepoints-only build reports the selection audit would make. A quiet
rule (``LANDLOCK_ADD_RULE_QUIET`` with the access in the ``quiet_*``
fields of ``struct landlock_ruleset_attr``) suppresses logging by
-setting ``logged=0`` the same way.
+setting ``logged=0`` the same way, as do the per-member ``quiet_*``
+masks of capability and namespace rules.
See Documentation/admin-guide/LSM/landlock.rst for audit record format,
tracepoint usage, and filtering examples.
diff --git a/Documentation/trace/events-landlock.rst b/Documentation/trace/events-landlock.rst
index 304a8d06273e..9ccf2769a2d5 100644
--- a/Documentation/trace/events-landlock.rst
+++ b/Documentation/trace/events-landlock.rst
@@ -6,7 +6,7 @@ Landlock Trace Events
=====================
:Author: Mickaël Salaün
-:Date: September 2026
+:Date: October 2026
Landlock emits trace events for sandbox lifecycle operations and access
denials. These events can be consumed by ftrace (for human-readable
@@ -35,6 +35,8 @@ Landlock trace events are organized in four categories:
- ``landlock_add_rule_net_port``: a network port rule is added to a ruleset
- ``landlock_create_domain``: a new domain is created from a ruleset
- ``landlock_enforce_domain``: a domain is enforced on a thread
+- ``landlock_add_rule_namespace``: a namespace rule is added to a ruleset
+- ``landlock_add_rule_capability``: a capability rule is added to a ruleset
**Denial events** are emitted when an access is denied:
@@ -44,6 +46,8 @@ Landlock trace events are organized in four categories:
- ``landlock_deny_scope_signal``: signal delivery denied
- ``landlock_deny_scope_abstract_unix_socket``: abstract unix socket
access denied
+- ``landlock_deny_permission_namespace``: namespace use denied
+- ``landlock_deny_permission_capability``: capability use denied
**Rule evaluation events** are emitted during rule matching:
@@ -84,7 +88,7 @@ Landlock, not from regular file permissions::
$ LC_ALL=C LL_FS_RO=/usr:/lib:/lib64:/bin:/etc/ld.so.cache LL_FS_RW=/tmp \
./sandboxer cat /etc/passwd
$ cat /sys/kernel/tracing/trace_pipe
- cat-127 [...] landlock_create_ruleset: ruleset=195cc6b76.0 handled_fs=execute|write_file|read_file|read_dir|remove_dir|remove_file|make_char|make_dir|make_reg|make_sock|make_fifo|make_block|make_sym|refer|truncate|ioctl_dev|resolve_unix handled_net= scoped=
+ cat-127 [...] landlock_create_ruleset: ruleset=195cc6b76.0 handled_fs=execute|write_file|read_file|read_dir|remove_dir|remove_file|make_char|make_dir|make_reg|make_sock|make_fifo|make_block|make_sym|refer|truncate|ioctl_dev|resolve_unix handled_net= scoped= handled_permissions=
cat-127 [...] landlock_create_domain: domain=195cc6b7c parent=0 ruleset=195cc6b76.6
cat-127 [...] landlock_enforce_domain: domain=195cc6b7c complete=1 process_wide=1 no_new_privs=1
cat-127 [...] landlock_deny_access_fs: domain=195cc6b7c same_exec=0 logged=0 blockers=read_file dev=0:17 ino=5901179 path=/etc/passwd
@@ -119,10 +123,13 @@ in some field formats:
Audit uses string ``dev="<s_id>"``. Numeric format is more precise
for machine parsing.
-- **Denied access field**: The ``deny_access_fs`` and ``deny_access_net``
- tracepoints use the ``blockers=`` field name (same as audit). Both
- render the blocked access rights as names: audit prefixes the category
- and separates with commas (e.g., ``blockers=fs.read_file``), while the
+- **Blocker field**: Denial tracepoints that carry a blocker use the
+ ``blockers=`` field name (same as audit). The ``deny_access_fs`` and
+ ``deny_access_net`` tracepoints render the blocked access rights, the
+ ``deny_permission_namespace`` and ``deny_permission_capability``
+ tracepoints render the blocked permission, and a ``change_topology``
+ denial renders its request type. Audit prefixes the category and
+ separates with commas (e.g., ``blockers=fs.read_file``), while the
tracepoints omit the category (carried by the event name) and separate
with ``|`` (e.g., ``blockers=read_file``). Scope and ptrace
tracepoints omit ``blockers`` because the event name identifies the
@@ -160,10 +167,34 @@ Ruleset versioning
==================
Syscall events include a ruleset version (``ruleset=<hex_id>.<version>``)
-that tracks the number of rules added to the ruleset. The version is
-incremented on each ``landlock_add_rule()`` call and frozen at
-``landlock_restrict_self()`` time. This enables trace consumers to
-correlate a domain with the exact set of rules it was created from.
+that tracks the number of successful add-rule calls on the ruleset. The
+version is incremented on each successful ``landlock_add_rule()`` call and
+frozen at ``landlock_restrict_self()`` time. Namespace and capability calls
+increment it even when their effective known masks are empty or already
+present, so the event stream preserves every successful call. This enables
+trace consumers to correlate a domain with the exact rule history from which
+it was created.
+
+Permission rule and denial events
+=================================
+
+``landlock_create_ruleset`` reports handled permissions as symbolic names in
+``handled_permissions=``. ``landlock_add_rule_namespace`` and
+``landlock_add_rule_capability`` report ``permissions=`` plus the effective
+known allowed and quiet member masks. Capability masks and the raw
+``CLONE_NEW*`` namespace masks are printed in hexadecimal. Unknown-only input
+therefore appears as zero while still advancing the ruleset version, and adding
+a member already present still emits an event.
+
+``landlock_deny_permission_namespace`` reports the denied raw
+``CLONE_NEW*`` value. ``namespace_id`` is zero for namespace creation and is
+the exact target namespace ID for :manpage:`setns(2)`.
+``landlock_deny_permission_capability`` reports the denied ``CAP_*`` number.
+Like every denial event, both include the denying domain, ``same_exec``, and
+``logged``.
+
+Per-member quiet masks suppress audit submission but never these trace events,
+which report ``logged=0``.
Domain enforcement
==================
@@ -290,8 +321,8 @@ per-domain state in BPF maps:
``domain=`` key (join to the ``create_domain`` recorded in step 1),
building the per-domain thread set; filter ``complete==1`` for a
one-event-per-operation summary.
-3. On ``landlock_deny_access_*``: look up the domain, decide whether
- to count, alert, or ignore the denial based on custom policy.
+3. On ``landlock_deny_*``: look up the domain, decide whether to count,
+ alert, or ignore the denial based on custom policy.
4. On ``landlock_free_domain``: clean up the per-domain state, log
final statistics.
@@ -304,11 +335,12 @@ reconcile incomplete state.
Audit filtering equivalence
===========================
-The ``logged`` field reflects the domain's log policy but not the global
-``audit_enabled`` toggle, so it does not change when audit is turned on
-or off. When audit is enabled, ``logged==1`` selects the denials the
-domain submits to audit (audit-side rate-limiting and exclude rules may
-still drop some), so a stateless ftrace filter can select them::
+The ``logged`` field reflects the domain's log policy and per-request quiet
+selection, but not the global ``audit_enabled`` toggle, so it does not
+change when audit is turned on or off. When audit is enabled, ``logged==1``
+selects the denials the domain submits to audit (audit-side rate-limiting and
+exclude rules may still drop some), so a stateless ftrace filter can select
+them::
# Show only denials that audit would also log:
echo 'logged==1' > \
diff --git a/Documentation/userspace-api/landlock.rst b/Documentation/userspace-api/landlock.rst
index 84cb7bf6b3ed..3d45b956b640 100644
--- a/Documentation/userspace-api/landlock.rst
+++ b/Documentation/userspace-api/landlock.rst
@@ -8,7 +8,7 @@ Landlock: unprivileged access control
=====================================
:Author: Mickaël Salaün
-:Date: August 2026
+:Date: October 2026
The goal of Landlock is to enable restriction of ambient rights (e.g. global
filesystem or network access) for a set of processes. Because Landlock
@@ -29,21 +29,30 @@ If Landlock is not currently supported, we need to
Landlock rules
==============
-A Landlock rule describes an action on an object which the process intends to
-perform. A set of rules is aggregated in a ruleset, which can then restrict
-the thread enforcing it, and its future children.
+A Landlock rule describes the actions a process is allowed to perform on a
+specific resource. A set of rules is aggregated in a ruleset, which can then
+restrict the thread enforcing it, and its future children.
-The two existing types of rules are:
+The existing types of rules are:
Filesystem rules
- For these rules, the object is a file hierarchy,
- and the related filesystem actions are defined with
- `filesystem access rights`.
+ The rule key is a file hierarchy, and the actions it allows are
+ defined with `filesystem access rights`.
Network rules (since ABI v4 for TCP and v10 for UDP)
For these rules, the object is a TCP or UDP port,
and the related actions are defined with `network access rights`.
+Capability rules (since ABI v12)
+ The rule body lists which Linux capabilities (see
+ :manpage:`capabilities(7)`) the process may exercise; the action is
+ defined with `permission flags`.
+
+Namespace rules (since ABI v12)
+ The rule body lists which namespace types (see
+ :manpage:`namespaces(7)`) the process may use; the action is defined
+ with `permission flags`.
+
Defining and enforcing a security policy
----------------------------------------
@@ -87,6 +96,9 @@ to be explicit about the denied-by-default access rights.
.scoped =
LANDLOCK_SCOPE_ABSTRACT_UNIX_SOCKET |
LANDLOCK_SCOPE_SIGNAL,
+ .handled_permissions =
+ LANDLOCK_PERMISSION_NAMESPACE_USE |
+ LANDLOCK_PERMISSION_CAPABILITY_USE,
};
Because we may not know which kernel version an application will be executed
@@ -140,6 +152,13 @@ version, and only use the available subset of access rights:
ruleset_attr.handled_access_net &=
~(LANDLOCK_ACCESS_NET_BIND_UDP |
LANDLOCK_ACCESS_NET_CONNECT_SEND_UDP);
+ __attribute__((fallthrough));
+ case 10:
+ case 11:
+ /* Removes LANDLOCK_PERMISSION_* for ABI < 12 */
+ ruleset_attr.handled_permissions &=
+ ~(LANDLOCK_PERMISSION_NAMESPACE_USE |
+ LANDLOCK_PERMISSION_CAPABILITY_USE);
}
This enables the creation of an inclusive ruleset that will contain our rules.
@@ -242,6 +261,56 @@ the program explicitly called :manpage:`bind(2)` on port 0.
err = landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NET_PORT,
&udp_bind, 0);
+Capability and namespace rules use a different attribute layout:
+``permissions`` identifies the permission category (a single
+``LANDLOCK_PERMISSION_*`` flag) and a type-specific value field carries the
+bitmask to allow within it. See `Capability and namespace restrictions`_ for
+the model.
+
+For capability access-control, we can add rules that allow specific
+capabilities (see :manpage:`capabilities(7)` for the list of ``CAP_*``
+values). For instance, to allow ``CAP_SYS_CHROOT`` (so the sandboxed
+process can call :manpage:`chroot(2)` inside a user namespace):
+
+.. code-block:: c
+
+ struct landlock_capability_attr cap_attr = {
+ .permissions = LANDLOCK_PERMISSION_CAPABILITY_USE,
+ .allowed_capabilities = (1ULL << CAP_SYS_CHROOT),
+ };
+
+ cap_attr.permissions &= ruleset_attr.handled_permissions;
+ if (cap_attr.permissions)
+ err = landlock_add_rule(ruleset_fd, LANDLOCK_RULE_CAPABILITY,
+ &cap_attr, 0);
+
+For namespace access-control, we can add rules that allow using specific
+namespace types (see :manpage:`namespaces(7)` for the list of ``CLONE_NEW*``
+values): creating them via :manpage:`unshare(2)` / :manpage:`clone(2)` /
+:manpage:`clone3(2)`, joining them via :manpage:`setns(2)`, or acquiring an fd
+reference via :manpage:`open_tree(2)` / :manpage:`fsmount(2)`. For instance,
+to allow creating user namespaces (which grants all capabilities inside the new
+namespace):
+
+.. code-block:: c
+
+ struct landlock_namespace_attr ns_attr = {
+ .permissions = LANDLOCK_PERMISSION_NAMESPACE_USE,
+ .allowed_namespace_types = CLONE_NEWUSER,
+ };
+
+ ns_attr.permissions &= ruleset_attr.handled_permissions;
+ if (ns_attr.permissions)
+ err = landlock_add_rule(ruleset_fd, LANDLOCK_RULE_NAMESPACE,
+ &ns_attr, 0);
+
+Together, these two rules allow an unprivileged process to create a user
+namespace and call :manpage:`chroot(2)` inside it, while denying all other
+capabilities and namespace types. User namespace creation is the one operation
+that does not require ``CAP_SYS_ADMIN``, so no capability rule is needed for it.
+See `Capability and namespace restrictions`_ for details on capability
+requirements.
+
When passing a non-zero ``flags`` argument to ``landlock_restrict_self()``, a
similar backwards compatibility check is needed for the restrict flags
(see sys_landlock_restrict_self() documentation for available flags):
@@ -442,9 +511,139 @@ The operations which can be scoped are:
A :manpage:`sendto(2)` on a socket which was previously connected will not
be restricted. This works for both datagram and stream sockets.
-IPC scoping does not support exceptions via :manpage:`landlock_add_rule(2)`.
-If an operation is scoped within a domain, no rules can be added to allow access
-to resources or processes outside of the scope.
+Scoping does not support exceptions via :manpage:`landlock_add_rule(2)`. If an
+operation is scoped within a domain, no rules can be added to allow access to
+resources or processes outside of the scope.
+
+Capability and namespace restrictions
+-------------------------------------
+
+``handled_permissions`` declares per-category permissions: each permission
+selects which members of a kernel-defined category (CAP_* capabilities,
+CLONE_NEW* namespace types) the process may use. Unlike per-object access
+rights (``handled_access_*``) or cross-domain scopes (``scoped``), per-category
+permissions constrain the sandboxed process's own use of these enums; members
+not allowed by a rule are denied by default.
+
+``LANDLOCK_PERMISSION_NAMESPACE_USE`` gates *acquisition of access* to
+namespaces: creation via :manpage:`unshare(2)` / :manpage:`clone(2)`
+/ :manpage:`clone3(2)`, entry via :manpage:`setns(2)`, and fd-reference
+acquisition via :manpage:`open_tree(2)` / :manpage:`fsmount(2)`. Namespaces
+the process is already a member of when the domain is enforced are implicitly
+allowed (the process could not continue running otherwise); rules describe
+which new namespace types the process may acquire.
+``LANDLOCK_PERMISSION_CAPABILITY_USE`` gates every exercise of a capability
+after the domain is enforced, regardless
+of how the capability was obtained (inherited credentials, ``CLONE_NEWUSER``
+grant, ``setuid``/file-cap-bearing :manpage:`execve(2)`, etc.). Configuring
+both together restricts what privileges are available *and* the namespaces in
+which they take effect, which matters because user namespace creation has no
+capability check and grants all capabilities within the new namespace: gating
+only one of the two leaves a kernel attack-surface widening path open. When a
+domain handles both permissions, an operation is allowed only if each
+independently allows it: :manpage:`unshare(2)` with ``CLONE_NEWNET`` then
+requires both a rule allowing ``CLONE_NEWNET`` and a rule allowing
+``CAP_SYS_ADMIN``.
+
+``LANDLOCK_PERMISSION_CAPABILITY_USE`` complements :manpage:`prctl(2)`
+``PR_SET_NO_NEW_PRIVS`` but does not replace it. ``PR_SET_NO_NEW_PRIVS``
+prevents privilege *acquisition* via :manpage:`execve(2)` (setuid, file
+capability xattrs, privilege-elevating LSM transitions) and is a prerequisite
+for unprivileged Landlock self-sandboxing.
+``LANDLOCK_PERMISSION_CAPABILITY_USE`` restricts *exercise* of capabilities the
+process already holds, including those
+gained via ``CLONE_NEWUSER`` which ``PR_SET_NO_NEW_PRIVS`` does not block.
+Sandboxes typically set both.
+
+Rules are added with ``LANDLOCK_RULE_CAPABILITY`` and &struct
+landlock_capability_attr (each rule lists ``CAP_*`` values to allow), and with
+``LANDLOCK_RULE_NAMESPACE`` and &struct landlock_namespace_attr (each rule
+lists ``CLONE_NEW*`` flags to allow). Landlock is purely restrictive: it can
+only deny what the traditional check would have allowed, never grant additional
+privileges.
+
+Denials of specific capabilities or namespace types can be excluded from audit
+logging per rule, by listing them in the ``quiet_capabilities`` or
+``quiet_namespace_types`` field of the rule (independently of the allowed
+members). Quieting requires the permission to be handled, is member-granular,
+and only suppresses the audit record of a denial attributed to the rule's own
+layer; it never changes the denial itself or suppresses its trace event, which
+reports ``logged=0``. Sandboxes should quiet only specific members known to be
+requested but expected to be denied (for example for compatibility with older
+callers), never a blanket set: a blanket set follows the running kernel's known
+members and would hide surprising or future denials from audit.
+
+Rule bodies silently accept values unknown to the current kernel (capabilities
+above ``CAP_LAST_CAP``, unrecognised ``CLONE_NEW*`` bits): they have no runtime
+effect, so a rule compiled against future kernel headers loads without error on
+older kernels. Future kernels gain new members denied by default until a rule
+explicitly allows them.
+
+The single ``LANDLOCK_PERMISSION_NAMESPACE_USE`` bit gates every kernel path
+that grants the calling process access to a namespace of the controlled types,
+whether by becoming a member of the namespace or by holding a file descriptor
+that references it. The covered syscall paths are:
+
+* :manpage:`unshare(2)` with ``CLONE_NEW*``: the caller becomes a member of a
+ newly-created namespace.
+* :manpage:`clone(2)` (or :manpage:`clone3(2)`) with ``CLONE_NEW*``: the
+ child becomes a member of a newly-created namespace.
+* :manpage:`setns(2)`: the caller becomes a member of an existing namespace
+ referenced by file descriptor.
+* :manpage:`open_tree(2)` with ``OPEN_TREE_NAMESPACE``: the caller obtains a
+ file descriptor referring to a newly-created mount namespace.
+* :manpage:`open_tree(2)` with ``OPEN_TREE_CLONE``: the caller obtains a file
+ descriptor referring to a newly-created anonymous mount namespace.
+* :manpage:`fsmount(2)` with ``FSMOUNT_NAMESPACE``: the caller obtains a file
+ descriptor referring to a newly-created mount namespace.
+* :manpage:`fsmount(2)` (default): the caller obtains a file descriptor
+ referring to a newly-created anonymous mount namespace.
+
+Anonymous mount namespaces (created by ``open_tree(OPEN_TREE_CLONE)`` and the
+default :manpage:`fsmount(2)`) are intentionally covered by the bit even though
+the calling process does not become a member of them. Without this coverage, a
+sandboxed process could combine ``open_tree(OPEN_TREE_CLONE)`` with
+:manpage:`move_mount(2)` to graft mounts from a freshly-allocated mount
+namespace into its current namespace, bypassing a ``CLONE_NEWNS`` restriction
+(these mount operations require ``CAP_SYS_ADMIN``, which a sandboxed process
+typically obtains by first creating a user namespace).
+
+In practice, unprivileged processes first create a user namespace (which
+requires no capability and grants all capabilities within it), then use those
+capabilities to create other namespace types. All non-user namespace types
+require ``CAP_SYS_ADMIN`` for both creation and :manpage:`setns(2)` entry; mount
+namespace entry additionally requires ``CAP_SYS_CHROOT``. For
+:manpage:`setns(2)`, capabilities are checked relative to the target namespace,
+so a process in an ancestor user namespace naturally satisfies them; this
+includes joining user namespaces, which requires ``CAP_SYS_ADMIN``. When
+``LANDLOCK_PERMISSION_CAPABILITY_USE`` is also handled, each of these
+capabilities must be explicitly allowed by a rule.
+
+Creating a non-user namespace always requires ``CAP_SYS_ADMIN``, and
+``LANDLOCK_PERMISSION_CAPABILITY_USE`` (when handled) gates that check
+regardless of which user namespace it targets. This is true whether the
+namespace is created
+on its own, combined with ``CLONE_NEWUSER`` in a single :manpage:`unshare(2)`
+call, or in a separate :manpage:`unshare(2)` call after a user namespace: in
+all cases the kernel checks ``CAP_SYS_ADMIN`` against the owning user
+namespace, and Landlock denies it unless a rule allows
+``CAP_SYS_ADMIN``. Combining ``CLONE_NEWUSER`` with another ``CLONE_NEW*``
+flag therefore does not avoid the capability check; only user namespace
+creation itself needs no capability rule.
+
+When creating child user namespaces, it is recommended to also create a
+dedicated Landlock domain with restrictions relevant to each namespace context.
+
+Note that ``LANDLOCK_PERMISSION_CAPABILITY_USE`` restricts the *use* of
+capabilities, not their presence in the process's credential. Capability sets
+can change
+after a domain is enforced through user namespace entry or :manpage:`capset(2)`;
+privileged sandboxes that did not set ``PR_SET_NO_NEW_PRIVS`` may also gain
+capabilities through :manpage:`execve(2)` of binaries with file capabilities.
+In all cases, :manpage:`capget(2)` will report the credential's capability sets,
+but any denied capability will fail with ``EPERM`` when exercised. Do not rely
+on :manpage:`capget(2)` to determine whether the policy permits a given
+capability; only the actual operation will return ``EPERM`` upon denial.
Truncating files
----------------
@@ -610,7 +809,7 @@ Access rights
-------------
.. kernel-doc:: include/uapi/linux/landlock.h
- :identifiers: fs_access net_access scope
+ :identifiers: fs_access net_access scope permission
Creating a new ruleset
----------------------
@@ -629,7 +828,8 @@ Extending a ruleset
.. kernel-doc:: include/uapi/linux/landlock.h
:identifiers: landlock_rule_type landlock_path_beneath_attr
- landlock_net_port_attr
+ landlock_net_port_attr landlock_capability_attr
+ landlock_namespace_attr
Enforcing a ruleset
-------------------
@@ -834,6 +1034,23 @@ with ``LANDLOCK_RESTRICT_SELF_TSYNC``, no_new_privs is set on all threads
of the process. As explained in the tutorial above, leaving no_new_privs
unset is risky even when Landlock does not require it.
+Capability (ABI < 12)
+---------------------
+
+Starting with the Landlock ABI version 12, it is possible to restrict
+:manpage:`capabilities(7)` with the new ``LANDLOCK_PERMISSION_CAPABILITY_USE``
+permission flag and ``LANDLOCK_RULE_CAPABILITY`` rule type.
+
+Namespace (ABI < 12)
+--------------------
+
+Starting with the Landlock ABI version 12, it is possible to restrict namespace
+use (see :manpage:`namespaces(7)`) across creation (:manpage:`unshare(2)`,
+:manpage:`clone(2)`, :manpage:`clone3(2)`), entry (:manpage:`setns(2)`), and
+fd-reference acquisition (:manpage:`open_tree(2)`, :manpage:`fsmount(2)`) with
+the new ``LANDLOCK_PERMISSION_NAMESPACE_USE`` permission flag and
+``LANDLOCK_RULE_NAMESPACE`` rule type.
+
.. _kernel_support:
Kernel support
diff --git a/include/trace/events/landlock.h b/include/trace/events/landlock.h
index c13475d5180a..b11f98d0a736 100644
--- a/include/trace/events/landlock.h
+++ b/include/trace/events/landlock.h
@@ -278,11 +278,12 @@ static inline const char *__trace_landlock_print_layers(
* Every denial event shares three fields. domain is the ID of the
* innermost domain that blocked the access. same_exec tells whether the
* current task is the same executable that entered that domain. logged is
- * the domain's audit-logging decision for this denial (its log_status is
- * enabled and the per-execution flag selected by same_exec is set); a
- * stateless ftrace filter can select the denials the domain submits to
- * audit with logged==1, without reconstructing it from the per-execution
- * log flags. Denial events order their fields as domain, same_exec,
+ * the complete audit-submission selection for this denial: the domain's
+ * log_status and per-execution flag, per-object or per-member quiet rules,
+ * and request-specific no-audit options. This excludes global audit state
+ * and audit-side filters. A stateless ftrace filter can select the denials
+ * the domain submits to audit with logged==1, without reconstructing those
+ * inputs. Denial events order their fields as domain, same_exec,
* logged, then the blockers verdict input, then the
* type-specific object fields, then any variable-length field.
*
diff --git a/include/uapi/linux/landlock.h b/include/uapi/linux/landlock.h
index 3128e58d4b78..24510e5d8479 100644
--- a/include/uapi/linux/landlock.h
+++ b/include/uapi/linux/landlock.h
@@ -33,14 +33,14 @@
* (and that they have tested with a kernel that supported them all).
*
* @quiet_access_fs and @quiet_access_net are bitmasks of actions for which a
- * denial by this layer will not trigger a log if the corresponding object (or
- * its children, for filesystem rules) is marked with the "quiet" bit via
- * %LANDLOCK_ADD_RULE_QUIET, even if logging would normally take place per
+ * denial by this layer will not trigger an audit record if the corresponding
+ * object (or its children, for filesystem rules) is marked with the "quiet" bit
+ * via %LANDLOCK_ADD_RULE_QUIET, even if logging would normally take place per
* landlock_restrict_self() flags. @quiet_scoped is similar, except that it
* does not require marking any objects as quiet - if the ruleset is created
* with any bits set in @quiet_scoped, then denial of such scoped resources will
- * not trigger any log. These 3 fields are available since Landlock ABI version
- * 10.
+ * not trigger an audit record. These three fields are available since Landlock
+ * ABI version 10.
*
* @quiet_access_fs, @quiet_access_net and @quiet_scoped must be a subset of
* @handled_access_fs, @handled_access_net and @scoped respectively.
@@ -66,16 +66,17 @@ struct landlock_ruleset_attr {
__u64 scoped;
/**
* @quiet_access_fs: Bitmask of filesystem actions which should not be
- * logged if per-object quiet flag is set.
+ * submitted to audit if the per-object quiet flag is set.
*/
__u64 quiet_access_fs;
/**
* @quiet_access_net: Bitmask of network actions which should not be
- * logged if per-object quiet flag is set.
+ * submitted to audit if the per-object quiet flag is set.
*/
__u64 quiet_access_net;
/**
- * @quiet_scoped: Bitmask of scoped actions which should not be logged.
+ * @quiet_scoped: Bitmask of scoped actions which should not be
+ * submitted to audit.
*/
__u64 quiet_scoped;
/**
--
2.55.0
^ permalink raw reply [flat|nested] 9+ messages in thread