* [PATCH v4 0/3] io_uring: add fremovexattr and flistxattr support
@ 2026-09-18 9:00 Aditya Prakash Srivastava
2026-09-18 9:00 ` [PATCH v4 1/3] tools: sync io_uring.h UAPI header Aditya Prakash Srivastava
` (2 more replies)
0 siblings, 3 replies; 5+ messages in thread
From: Aditya Prakash Srivastava @ 2026-09-18 9:00 UTC (permalink / raw)
To: Jens Axboe, Christian Brauner, Gabriel Krisman Bertazi
Cc: Alexander Viro, Jan Kara, io-uring, linux-fsdevel, linux-kernel,
Aditya Prakash Srivastava
This series completes io_uring's FD-based xattr operations by adding
support for:
- IORING_OP_FREMOVEXATTR
- IORING_OP_FLISTXATTR
This allows asynchronous xattr removal and listing.
New test cases have been added to the liburing test suite
(test/xattr.c) and verified to pass. A separate userspace patch
implementing the matching prep helpers, sanitizers, and test cases
has been updated and submitted in parallel.
- Patch 1 syncs the tools/ directory's io_uring.h UAPI header to
align it with kernel mainline prior to introducing the new opcodes.
- Patch 2 makes the necessary VFS-layer list/remove helpers
non-static and declares them in fs/internal.h.
- Patch 3 implements the io_uring operational support (opcodes,
opdefs, preparation, and issue handlers) and invokes these helpers.
Changes since v3:
- Split the tools/ UAPI header sync of pre-existing mainline opcodes
into a separate precursor patch (Patch 1) (Suggested by Gabriel).
- Robustify io_fremovexattr_prep by initializing kname and kvalue
to NULL to avoid potential garbage-pointer crashes in the
io_xattr_cleanup path (Reported by Gabriel).
- Add strict validation to both fremovexattr and flistxattr to
zero-check and reject all unused SQE fields (Reported by Gabriel).
- Reset ix->ctx.kname to NULL on error inside io_fremovexattr_prep.
Changes since v2:
- Revert unnecessary formatting changes to filename_listxattr in
Patch 1 (now Patch 2).
Changes since v1:
- Omit path-based opcodes to prioritize optimal FD-based variants.
- Limit exported VFS helpers to only file_listxattr and
file_removexattr.
- Rewrite standalone test program into a standard liburing testcase.
Aditya Prakash Srivastava (3):
tools: sync io_uring.h UAPI header
fs: make file_listxattr and file_removexattr helpers non-static
io_uring: add fremovexattr and flistxattr support
fs/internal.h | 2 +
fs/xattr.c | 3 +-
include/uapi/linux/io_uring.h | 2 +
io_uring/opdef.c | 18 ++
io_uring/xattr.c | 83 +++++++
io_uring/xattr.h | 6 +
tools/include/uapi/linux/io_uring.h | 342 ++++++++++++++++++++++++++--
7 files changed, 438 insertions(+), 18 deletions(-)
--
2.47.3
^ permalink raw reply [flat|nested] 5+ messages in thread
* [PATCH v4 1/3] tools: sync io_uring.h UAPI header
2026-09-18 9:00 [PATCH v4 0/3] io_uring: add fremovexattr and flistxattr support Aditya Prakash Srivastava
@ 2026-09-18 9:00 ` Aditya Prakash Srivastava
2026-09-18 16:49 ` Caleb Sander Mateos
2026-09-18 9:00 ` [PATCH v4 2/3] fs: make file_listxattr and file_removexattr helpers non-static Aditya Prakash Srivastava
2026-09-18 9:00 ` [PATCH v4 3/3] io_uring: add fremovexattr and flistxattr support Aditya Prakash Srivastava
2 siblings, 1 reply; 5+ messages in thread
From: Aditya Prakash Srivastava @ 2026-09-18 9:00 UTC (permalink / raw)
To: Jens Axboe, Christian Brauner, Gabriel Krisman Bertazi
Cc: Alexander Viro, Jan Kara, io-uring, linux-fsdevel, linux-kernel,
Aditya Prakash Srivastava
Sync tools/include/uapi/linux/io_uring.h with the main kernel
include/uapi/linux/io_uring.h UAPI header to ensure they are
perfectly aligned.
Additionally, correct a minor space-before-tab checkpatch.pl warning
in the comments of both headers (no functional change).
Suggested-by: Gabriel Krisman Bertazi <krisman@suse.de>
Signed-off-by: Aditya Prakash Srivastava <aditya.ansh182@gmail.com>
---
include/uapi/linux/io_uring.h | 2 +-
tools/include/uapi/linux/io_uring.h | 340 ++++++++++++++++++++++++++--
2 files changed, 325 insertions(+), 17 deletions(-)
diff --git a/include/uapi/linux/io_uring.h b/include/uapi/linux/io_uring.h
index 909fb7aea638..1f300f6b85d7 100644
--- a/include/uapi/linux/io_uring.h
+++ b/include/uapi/linux/io_uring.h
@@ -426,7 +426,7 @@ enum io_uring_op {
* IORING_RECVSEND_BUNDLE Used with IOSQE_BUFFER_SELECT. If set, send or
* recv will grab as many buffers from the buffer
* group ID given and send them all. The completion
- * result will be the number of buffers send, with
+ * result will be the number of buffers send, with
* the starting buffer ID in cqe->flags as per
* usual for provided buffer usage. The buffers
* will be contiguous from the starting buffer ID.
diff --git a/tools/include/uapi/linux/io_uring.h b/tools/include/uapi/linux/io_uring.h
index f1c16f817742..1f300f6b85d7 100644
--- a/tools/include/uapi/linux/io_uring.h
+++ b/tools/include/uapi/linux/io_uring.h
@@ -10,6 +10,8 @@
#include <linux/fs.h>
#include <linux/types.h>
+#include <linux/io_uring/zcrx.h>
+
/*
* this file is shared with liburing and that has to autodetect
* if linux/time_types.h is available or not, it can
@@ -50,7 +52,7 @@ struct io_uring_sqe {
};
__u32 len; /* buffer size or number of iovecs */
union {
- __kernel_rwf_t rw_flags;
+ __u32 rw_flags;
__u32 fsync_flags;
__u16 poll_events; /* compatibility */
__u32 poll32_events; /* word-reversed for BE */
@@ -71,6 +73,9 @@ struct io_uring_sqe {
__u32 uring_cmd_flags;
__u32 waitid_flags;
__u32 futex_flags;
+ __u32 install_fd_flags;
+ __u32 nop_flags;
+ __u32 pipe_flags;
};
__u64 user_data; /* data to be passed back at completion time */
/* pack this to avoid bogus arm OABI complaints */
@@ -85,17 +90,26 @@ struct io_uring_sqe {
union {
__s32 splice_fd_in;
__u32 file_index;
+ __u32 zcrx_ifq_idx;
__u32 optlen;
struct {
__u16 addr_len;
__u16 __pad3[1];
};
+ struct {
+ __u8 write_stream;
+ __u8 __pad4[3];
+ };
};
union {
struct {
__u64 addr3;
__u64 __pad2[1];
};
+ struct {
+ __u64 attr_ptr; /* pointer to attribute information */
+ __u64 attr_type_mask; /* bit mask of attributes */
+ };
__u64 optval;
/*
* If the ring is initialized with IORING_SETUP_SQE128, then
@@ -105,6 +119,18 @@ struct io_uring_sqe {
};
};
+/* sqe->attr_type_mask flags */
+#define IORING_RW_ATTR_FLAG_PI (1U << 0)
+/* PI attribute information */
+struct io_uring_attr_pi {
+ __u16 flags;
+ __u16 app_tag;
+ __u32 len;
+ __u64 addr;
+ __u64 seed;
+ __u64 rsvd;
+};
+
/*
* If sqe->file_index is set to this for opcodes that instantiate a new
* direct descriptor (like openat/openat2/accept), then io_uring will allocate
@@ -114,7 +140,7 @@ struct io_uring_sqe {
*/
#define IORING_FILE_INDEX_ALLOC (~0U)
-enum {
+enum io_uring_sqe_flags_bit {
IOSQE_FIXED_FILE_BIT,
IOSQE_IO_DRAIN_BIT,
IOSQE_IO_LINK_BIT,
@@ -164,7 +190,8 @@ enum {
/*
* If COOP_TASKRUN is set, get notified if task work is available for
* running and a kernel transition would be needed to run it. This sets
- * IORING_SQ_TASKRUN in the sq ring flags. Not valid with COOP_TASKRUN.
+ * IORING_SQ_TASKRUN in the sq ring flags. Not valid without COOP_TASKRUN
+ * or DEFER_TASKRUN.
*/
#define IORING_SETUP_TASKRUN_FLAG (1U << 9)
#define IORING_SETUP_SQE128 (1U << 10) /* SQEs are 128 byte */
@@ -198,6 +225,33 @@ enum {
*/
#define IORING_SETUP_NO_SQARRAY (1U << 16)
+/* Use hybrid poll in iopoll process */
+#define IORING_SETUP_HYBRID_IOPOLL (1U << 17)
+
+/*
+ * Allow both 16b and 32b CQEs. If a 32b CQE is posted, it will have
+ * IORING_CQE_F_32 set in cqe->flags.
+ */
+#define IORING_SETUP_CQE_MIXED (1U << 18)
+
+/*
+ * Allow both 64b and 128b SQEs. If a 128b SQE is posted, it will have
+ * a 128b opcode.
+ */
+#define IORING_SETUP_SQE_MIXED (1U << 19)
+
+/*
+ * When set, io_uring ignores SQ head and tail and fetches SQEs to submit
+ * starting from index 0 instead from the index stored in the head pointer.
+ * IOW, the user should place all SQE at the beginning of the SQ memory
+ * before issuing a submission syscall.
+ *
+ * It requires IORING_SETUP_NO_SQARRAY and is incompatible with
+ * IORING_SETUP_SQPOLL. The user must also never change the SQ head and tail
+ * values and keep it set to 0. Any other value is undefined behaviour.
+ */
+#define IORING_SETUP_SQ_REWIND (1U << 20)
+
enum io_uring_op {
IORING_OP_NOP,
IORING_OP_READV,
@@ -253,6 +307,17 @@ enum io_uring_op {
IORING_OP_FUTEX_WAIT,
IORING_OP_FUTEX_WAKE,
IORING_OP_FUTEX_WAITV,
+ IORING_OP_FIXED_FD_INSTALL,
+ IORING_OP_FTRUNCATE,
+ IORING_OP_BIND,
+ IORING_OP_LISTEN,
+ IORING_OP_RECV_ZC,
+ IORING_OP_EPOLL_WAIT,
+ IORING_OP_READV_FIXED,
+ IORING_OP_WRITEV_FIXED,
+ IORING_OP_PIPE,
+ IORING_OP_NOP128,
+ IORING_OP_URING_CMD128,
/* this goes last, obviously */
IORING_OP_LAST,
@@ -262,9 +327,13 @@ enum io_uring_op {
* sqe->uring_cmd_flags top 8bits aren't available for userspace
* IORING_URING_CMD_FIXED use registered buffer; pass this flag
* along with setting sqe->buf_index.
+ * IORING_URING_CMD_MULTISHOT must be used with buffer select, like other
+ * multishot commands. Not compatible with
+ * IORING_URING_CMD_FIXED, for now.
*/
#define IORING_URING_CMD_FIXED (1U << 0)
-#define IORING_URING_CMD_MASK IORING_URING_CMD_FIXED
+#define IORING_URING_CMD_MULTISHOT (1U << 1)
+#define IORING_URING_CMD_MASK (IORING_URING_CMD_FIXED | IORING_URING_CMD_MULTISHOT)
/*
@@ -274,6 +343,10 @@ enum io_uring_op {
/*
* sqe->timeout_flags
+ *
+ * IORING_TIMEOUT_IMMEDIATE_ARG: If set, sqe->addr stores the timeout
+ * value in nanoseconds instead of
+ * pointing to a timespec.
*/
#define IORING_TIMEOUT_ABS (1U << 0)
#define IORING_TIMEOUT_UPDATE (1U << 1)
@@ -282,6 +355,7 @@ enum io_uring_op {
#define IORING_LINK_TIMEOUT_UPDATE (1U << 4)
#define IORING_TIMEOUT_ETIME_SUCCESS (1U << 5)
#define IORING_TIMEOUT_MULTISHOT (1U << 6)
+#define IORING_TIMEOUT_IMMEDIATE_ARG (1U << 7)
#define IORING_TIMEOUT_CLOCK_MASK (IORING_TIMEOUT_BOOTTIME | IORING_TIMEOUT_REALTIME)
#define IORING_TIMEOUT_UPDATE_MASK (IORING_TIMEOUT_UPDATE | IORING_LINK_TIMEOUT_UPDATE)
/*
@@ -348,11 +422,24 @@ enum io_uring_op {
* 0 is reported if zerocopy was actually possible.
* IORING_NOTIF_USAGE_ZC_COPIED if data was copied
* (at least partially).
+ *
+ * IORING_RECVSEND_BUNDLE Used with IOSQE_BUFFER_SELECT. If set, send or
+ * recv will grab as many buffers from the buffer
+ * group ID given and send them all. The completion
+ * result will be the number of buffers send, with
+ * the starting buffer ID in cqe->flags as per
+ * usual for provided buffer usage. The buffers
+ * will be contiguous from the starting buffer ID.
+ *
+ * IORING_SEND_VECTORIZED If set, SEND[_ZC] will take a pointer to a io_vec
+ * to allow vectorized send operations.
*/
#define IORING_RECVSEND_POLL_FIRST (1U << 0)
#define IORING_RECV_MULTISHOT (1U << 1)
#define IORING_RECVSEND_FIXED_BUF (1U << 2)
#define IORING_SEND_ZC_REPORT_USAGE (1U << 3)
+#define IORING_RECVSEND_BUNDLE (1U << 4)
+#define IORING_SEND_VECTORIZED (1U << 5)
/*
* cqe.res for IORING_CQE_F_NOTIF if
@@ -367,11 +454,13 @@ enum io_uring_op {
* accept flags stored in sqe->ioprio
*/
#define IORING_ACCEPT_MULTISHOT (1U << 0)
+#define IORING_ACCEPT_DONTWAIT (1U << 1)
+#define IORING_ACCEPT_POLL_FIRST (1U << 2)
/*
* IORING_OP_MSG_RING command types, stored in sqe->addr
*/
-enum {
+enum io_uring_msg_ring_flags {
IORING_MSG_DATA, /* pass sqe->len as 'res' and off as user_data */
IORING_MSG_SEND_FD, /* send a registered fd to another ring */
};
@@ -386,11 +475,30 @@ enum {
/* Pass through the flags from sqe->file_index to cqe->flags */
#define IORING_MSG_RING_FLAGS_PASS (1U << 1)
+/*
+ * IORING_OP_FIXED_FD_INSTALL flags (sqe->install_fd_flags)
+ *
+ * IORING_FIXED_FD_NO_CLOEXEC Don't mark the fd as O_CLOEXEC
+ */
+#define IORING_FIXED_FD_NO_CLOEXEC (1U << 0)
+
+/*
+ * IORING_OP_NOP flags (sqe->nop_flags)
+ *
+ * IORING_NOP_INJECT_RESULT Inject result from sqe->result
+ */
+#define IORING_NOP_INJECT_RESULT (1U << 0)
+#define IORING_NOP_FILE (1U << 1)
+#define IORING_NOP_FIXED_FILE (1U << 2)
+#define IORING_NOP_FIXED_BUFFER (1U << 3)
+#define IORING_NOP_TW (1U << 4)
+#define IORING_NOP_CQE32 (1U << 5)
+
/*
* IO completion data structure (Completion Queue Entry)
*/
struct io_uring_cqe {
- __u64 user_data; /* sqe->data submission passed back */
+ __u64 user_data; /* sqe->user_data value passed back */
__s32 res; /* result code for this event */
__u32 flags;
@@ -409,15 +517,33 @@ struct io_uring_cqe {
* IORING_CQE_F_SOCK_NONEMPTY If set, more data to read after socket recv
* IORING_CQE_F_NOTIF Set for notification CQEs. Can be used to distinct
* them from sends.
+ * IORING_CQE_F_BUF_MORE If set, the buffer ID set in the completion will get
+ * more completions. In other words, the buffer is being
+ * partially consumed, and will be used by the kernel for
+ * more completions. This is only set for buffers used via
+ * the incremental buffer consumption, as provided by
+ * a ring buffer setup with IOU_PBUF_RING_INC. For any
+ * other provided buffer type, all completions with a
+ * buffer passed back is automatically returned to the
+ * application.
+ * IORING_CQE_F_SKIP If set, then the application/liburing must ignore this
+ * CQE. It's only purpose is to fill a gap in the ring,
+ * if a large CQE is attempted posted when the ring has
+ * just a single small CQE worth of space left before
+ * wrapping.
+ * IORING_CQE_F_32 If set, this is a 32b/big-cqe posting. Use with rings
+ * setup in a mixed CQE mode, where both 16b and 32b
+ * CQEs may be posted to the CQ ring.
*/
#define IORING_CQE_F_BUFFER (1U << 0)
#define IORING_CQE_F_MORE (1U << 1)
#define IORING_CQE_F_SOCK_NONEMPTY (1U << 2)
#define IORING_CQE_F_NOTIF (1U << 3)
+#define IORING_CQE_F_BUF_MORE (1U << 4)
+#define IORING_CQE_F_SKIP (1U << 5)
+#define IORING_CQE_F_32 (1U << 15)
-enum {
- IORING_CQE_BUFFER_SHIFT = 16,
-};
+#define IORING_CQE_BUFFER_SHIFT 16
/*
* Magic offsets for the application to mmap the data it needs
@@ -478,6 +604,9 @@ struct io_cqring_offsets {
#define IORING_ENTER_SQ_WAIT (1U << 2)
#define IORING_ENTER_EXT_ARG (1U << 3)
#define IORING_ENTER_REGISTERED_RING (1U << 4)
+#define IORING_ENTER_ABS_TIMER (1U << 5)
+#define IORING_ENTER_EXT_ARG_REG (1U << 6)
+#define IORING_ENTER_NO_IOWAIT (1U << 7)
/*
* Passed in for io_uring_setup(2). Copied back with updated info on success
@@ -512,11 +641,15 @@ struct io_uring_params {
#define IORING_FEAT_CQE_SKIP (1U << 11)
#define IORING_FEAT_LINKED_FILE (1U << 12)
#define IORING_FEAT_REG_REG_RING (1U << 13)
+#define IORING_FEAT_RECVSEND_BUNDLE (1U << 14)
+#define IORING_FEAT_MIN_TIMEOUT (1U << 15)
+#define IORING_FEAT_RW_ATTR (1U << 16)
+#define IORING_FEAT_NO_IOWAIT (1U << 17)
/*
* io_uring_register(2) opcodes and arguments
*/
-enum {
+enum io_uring_register_op {
IORING_REGISTER_BUFFERS = 0,
IORING_UNREGISTER_BUFFERS = 1,
IORING_REGISTER_FILES = 2,
@@ -558,6 +691,38 @@ enum {
/* register a range of fixed file slots for automatic slot allocation */
IORING_REGISTER_FILE_ALLOC_RANGE = 25,
+ /* return status information for a buffer group */
+ IORING_REGISTER_PBUF_STATUS = 26,
+
+ /* set/clear busy poll settings */
+ IORING_REGISTER_NAPI = 27,
+ IORING_UNREGISTER_NAPI = 28,
+
+ IORING_REGISTER_CLOCK = 29,
+
+ /* clone registered buffers from source ring to current ring */
+ IORING_REGISTER_CLONE_BUFFERS = 30,
+
+ /* send MSG_RING without having a ring */
+ IORING_REGISTER_SEND_MSG_RING = 31,
+
+ /* register a netdev hw rx queue for zerocopy */
+ IORING_REGISTER_ZCRX_IFQ = 32,
+
+ /* resize CQ ring */
+ IORING_REGISTER_RESIZE_RINGS = 33,
+
+ IORING_REGISTER_MEM_REGION = 34,
+
+ /* query various aspects of io_uring, see linux/io_uring/query.h */
+ IORING_REGISTER_QUERY = 35,
+
+ /* auxiliary zcrx configuration, see enum zcrx_ctrl_op */
+ IORING_REGISTER_ZCRX_CTRL = 36,
+
+ /* register bpf filtering programs */
+ IORING_REGISTER_BPF_FILTER = 37,
+
/* this goes last */
IORING_REGISTER_LAST,
@@ -566,7 +731,7 @@ enum {
};
/* io-wq worker categories */
-enum {
+enum io_wq_type {
IO_WQ_BOUND,
IO_WQ_UNBOUND,
};
@@ -578,6 +743,31 @@ struct io_uring_files_update {
__aligned_u64 /* __s32 * */ fds;
};
+enum {
+ /* initialise with user provided memory pointed by user_addr */
+ IORING_MEM_REGION_TYPE_USER = 1,
+};
+
+struct io_uring_region_desc {
+ __u64 user_addr;
+ __u64 size;
+ __u32 flags;
+ __u32 id;
+ __u64 mmap_offset;
+ __u64 __resv[4];
+};
+
+enum {
+ /* expose the region as registered wait arguments */
+ IORING_MEM_REGION_REG_WAIT_ARG = 1,
+};
+
+struct io_uring_mem_region_reg {
+ __u64 region_uptr; /* struct io_uring_region_desc * */
+ __u64 flags;
+ __u64 __resv[2];
+};
+
/*
* Register a fully sparse file space, rather than pass in an array of all
* -1 file descriptors.
@@ -638,6 +828,32 @@ struct io_uring_restriction {
__u32 resv2[3];
};
+struct io_uring_task_restriction {
+ __u16 flags;
+ __u16 nr_res;
+ __u32 resv[3];
+ __DECLARE_FLEX_ARRAY(struct io_uring_restriction, restrictions);
+};
+
+struct io_uring_clock_register {
+ __u32 clockid;
+ __u32 __resv[3];
+};
+
+enum {
+ IORING_REGISTER_SRC_REGISTERED = (1U << 0),
+ IORING_REGISTER_DST_REPLACE = (1U << 1),
+};
+
+struct io_uring_clone_buffers {
+ __u32 src_fd;
+ __u32 flags;
+ __u32 src_off;
+ __u32 dst_off;
+ __u32 nr;
+ __u32 pad[3];
+};
+
struct io_uring_buf {
__u64 addr;
__u32 len;
@@ -670,9 +886,17 @@ struct io_uring_buf_ring {
* mmap(2) with the offset set as:
* IORING_OFF_PBUF_RING | (bgid << IORING_OFF_PBUF_SHIFT)
* to get a virtual mapping for the ring.
+ * IOU_PBUF_RING_INC: If set, buffers consumed from this buffer ring can be
+ * consumed incrementally. Normally one (or more) buffers
+ * are fully consumed. With incremental consumptions, it's
+ * feasible to register big ranges of buffers, and each
+ * use of it will consume only as much as it needs. This
+ * requires that both the kernel and application keep
+ * track of where the current read/recv index is at.
*/
-enum {
+enum io_uring_register_pbuf_ring_flags {
IOU_PBUF_RING_MMAP = 1,
+ IOU_PBUF_RING_INC = 2,
};
/* argument for IORING_(UN)REGISTER_PBUF_RING */
@@ -681,13 +905,57 @@ struct io_uring_buf_reg {
__u32 ring_entries;
__u16 bgid;
__u16 flags;
- __u64 resv[3];
+ __u32 min_left;
+ __u32 resv[5];
+};
+
+/* argument for IORING_REGISTER_PBUF_STATUS */
+struct io_uring_buf_status {
+ __u32 buf_group; /* input */
+ __u32 head; /* output */
+ __u32 resv[8];
+};
+
+enum io_uring_napi_op {
+ /* register/ungister backward compatible opcode */
+ IO_URING_NAPI_REGISTER_OP = 0,
+
+ /* opcodes to update napi_list when static tracking is used */
+ IO_URING_NAPI_STATIC_ADD_ID = 1,
+ IO_URING_NAPI_STATIC_DEL_ID = 2
+};
+
+enum io_uring_napi_tracking_strategy {
+ /* value must be 0 for backward compatibility */
+ IO_URING_NAPI_TRACKING_DYNAMIC = 0,
+ IO_URING_NAPI_TRACKING_STATIC = 1,
+ IO_URING_NAPI_TRACKING_INACTIVE = 255
+};
+
+/* argument for IORING_(UN)REGISTER_NAPI */
+struct io_uring_napi {
+ __u32 busy_poll_to;
+ __u8 prefer_busy_poll;
+
+ /* a io_uring_napi_op value */
+ __u8 opcode;
+ __u8 pad[2];
+
+ /*
+ * for IO_URING_NAPI_REGISTER_OP, it is a
+ * io_uring_napi_tracking_strategy value.
+ *
+ * for IO_URING_NAPI_STATIC_ADD_ID/IO_URING_NAPI_STATIC_DEL_ID
+ * it is the napi id to add/del from napi_list.
+ */
+ __u32 op_param;
+ __u32 resv;
};
/*
* io_uring_restriction->opcode values
*/
-enum {
+enum io_uring_register_restriction_op {
/* Allow an io_uring_register(2) opcode */
IORING_RESTRICTION_REGISTER_OP = 0,
@@ -703,10 +971,33 @@ enum {
IORING_RESTRICTION_LAST
};
+enum {
+ IORING_REG_WAIT_TS = (1U << 0),
+};
+
+/*
+ * Argument for io_uring_enter(2) with
+ * IORING_GETEVENTS | IORING_ENTER_EXT_ARG_REG set, where the actual argument
+ * is an index into a previously registered fixed wait region described by
+ * the below structure.
+ */
+struct io_uring_reg_wait {
+ struct __kernel_timespec ts;
+ __u32 min_wait_usec;
+ __u32 flags;
+ __u64 sigmask;
+ __u32 sigmask_sz;
+ __u32 pad[3];
+ __u64 pad2[2];
+};
+
+/*
+ * Argument for io_uring_enter(2) with IORING_GETEVENTS | IORING_ENTER_EXT_ARG
+ */
struct io_uring_getevents_arg {
__u64 sigmask;
__u32 sigmask_sz;
- __u32 pad;
+ __u32 min_wait_usec;
__u64 ts;
};
@@ -743,11 +1034,28 @@ struct io_uring_recvmsg_out {
/*
* Argument for IORING_OP_URING_CMD when file is a socket
*/
-enum {
+enum io_uring_socket_op {
SOCKET_URING_OP_SIOCINQ = 0,
SOCKET_URING_OP_SIOCOUTQ,
SOCKET_URING_OP_GETSOCKOPT,
SOCKET_URING_OP_SETSOCKOPT,
+ SOCKET_URING_OP_TX_TIMESTAMP,
+ SOCKET_URING_OP_GETSOCKNAME,
+};
+
+/*
+ * SOCKET_URING_OP_TX_TIMESTAMP definitions
+ */
+
+#define IORING_TIMESTAMP_HW_SHIFT 16
+/* The cqe->flags bit from which the timestamp type is stored */
+#define IORING_TIMESTAMP_TYPE_SHIFT (IORING_TIMESTAMP_HW_SHIFT + 1)
+/* The cqe->flags flag signifying whether it's a hardware timestamp */
+#define IORING_CQE_F_TSTAMP_HW ((__u32)1 << IORING_TIMESTAMP_HW_SHIFT)
+
+struct io_timespec {
+ __u64 tv_sec;
+ __u64 tv_nsec;
};
#ifdef __cplusplus
--
2.47.3
^ permalink raw reply [flat|nested] 5+ messages in thread
* [PATCH v4 2/3] fs: make file_listxattr and file_removexattr helpers non-static
2026-09-18 9:00 [PATCH v4 0/3] io_uring: add fremovexattr and flistxattr support Aditya Prakash Srivastava
2026-09-18 9:00 ` [PATCH v4 1/3] tools: sync io_uring.h UAPI header Aditya Prakash Srivastava
@ 2026-09-18 9:00 ` Aditya Prakash Srivastava
2026-09-18 9:00 ` [PATCH v4 3/3] io_uring: add fremovexattr and flistxattr support Aditya Prakash Srivastava
2 siblings, 0 replies; 5+ messages in thread
From: Aditya Prakash Srivastava @ 2026-09-18 9:00 UTC (permalink / raw)
To: Jens Axboe, Christian Brauner, Gabriel Krisman Bertazi
Cc: Alexander Viro, Jan Kara, io-uring, linux-fsdevel, linux-kernel,
Aditya Prakash Srivastava
In preparation for adding IORING_OP_FREMOVEXATTR and
IORING_OP_FLISTXATTR support in io_uring, we need to invoke the VFS-layer
helpers from within io_uring.
Make the following helpers non-static and declare them in fs/internal.h:
- file_listxattr()
- file_removexattr()
No functional change is introduced.
Reviewed-by: Jan Kara <jack@suse.cz>
Signed-off-by: Aditya Prakash Srivastava <aditya.ansh182@gmail.com>
---
fs/internal.h | 2 ++
fs/xattr.c | 3 +--
2 files changed, 3 insertions(+), 2 deletions(-)
diff --git a/fs/internal.h b/fs/internal.h
index c658c8a5ebd5..ff7344ac8cef 100644
--- a/fs/internal.h
+++ b/fs/internal.h
@@ -298,6 +298,8 @@ int filename_setxattr(int dfd, struct filename *filename,
unsigned int lookup_flags, struct kernel_xattr_ctx *ctx);
int setxattr_copy(const char __user *name, struct kernel_xattr_ctx *ctx);
int import_xattr_name(struct xattr_name *kname, const char __user *name);
+ssize_t file_listxattr(struct file *f, char __user *list, size_t size);
+int file_removexattr(struct file *f, struct xattr_name *kname);
int may_write_xattr(struct mnt_idmap *idmap, struct inode *inode);
diff --git a/fs/xattr.c b/fs/xattr.c
index d58979115200..31b8a5eeec1e 100644
--- a/fs/xattr.c
+++ b/fs/xattr.c
@@ -953,7 +953,6 @@ listxattr(struct dentry *d, char __user *list, size_t size)
return error;
}
-static
ssize_t file_listxattr(struct file *f, char __user *list, size_t size)
{
audit_file(f);
@@ -1036,7 +1035,7 @@ removexattr(struct mnt_idmap *idmap, struct dentry *d, const char *name)
return vfs_removexattr(idmap, d, name);
}
-static int file_removexattr(struct file *f, struct xattr_name *kname)
+int file_removexattr(struct file *f, struct xattr_name *kname)
{
int error = mnt_want_write_file(f);
--
2.47.3
^ permalink raw reply [flat|nested] 5+ messages in thread
* [PATCH v4 3/3] io_uring: add fremovexattr and flistxattr support
2026-09-18 9:00 [PATCH v4 0/3] io_uring: add fremovexattr and flistxattr support Aditya Prakash Srivastava
2026-09-18 9:00 ` [PATCH v4 1/3] tools: sync io_uring.h UAPI header Aditya Prakash Srivastava
2026-09-18 9:00 ` [PATCH v4 2/3] fs: make file_listxattr and file_removexattr helpers non-static Aditya Prakash Srivastava
@ 2026-09-18 9:00 ` Aditya Prakash Srivastava
2 siblings, 0 replies; 5+ messages in thread
From: Aditya Prakash Srivastava @ 2026-09-18 9:00 UTC (permalink / raw)
To: Jens Axboe, Christian Brauner, Gabriel Krisman Bertazi
Cc: Alexander Viro, Jan Kara, io-uring, linux-fsdevel, linux-kernel,
Aditya Prakash Srivastava
Add support for IORING_OP_FREMOVEXATTR and IORING_OP_FLISTXATTR. This
enables xattr listing and removal operations to be executed in an
asynchronous fashion.
Signed-off-by: Aditya Prakash Srivastava <aditya.ansh182@gmail.com>
---
include/uapi/linux/io_uring.h | 2 +
io_uring/opdef.c | 18 +++++++
io_uring/xattr.c | 83 +++++++++++++++++++++++++++++
io_uring/xattr.h | 6 +++
tools/include/uapi/linux/io_uring.h | 2 +
5 files changed, 111 insertions(+)
diff --git a/include/uapi/linux/io_uring.h b/include/uapi/linux/io_uring.h
index 1f300f6b85d7..45de6dd6e3c4 100644
--- a/include/uapi/linux/io_uring.h
+++ b/include/uapi/linux/io_uring.h
@@ -318,6 +318,8 @@ enum io_uring_op {
IORING_OP_PIPE,
IORING_OP_NOP128,
IORING_OP_URING_CMD128,
+ IORING_OP_FREMOVEXATTR,
+ IORING_OP_FLISTXATTR,
/* this goes last, obviously */
IORING_OP_LAST,
diff --git a/io_uring/opdef.c b/io_uring/opdef.c
index cf3aa2242cd7..14a511f7a9df 100644
--- a/io_uring/opdef.c
+++ b/io_uring/opdef.c
@@ -591,6 +591,16 @@ const struct io_issue_def io_issue_defs[] = {
.prep = io_uring_cmd_prep,
.issue = io_uring_cmd,
},
+ [IORING_OP_FREMOVEXATTR] = {
+ .needs_file = 1,
+ .prep = io_fremovexattr_prep,
+ .issue = io_fremovexattr,
+ },
+ [IORING_OP_FLISTXATTR] = {
+ .needs_file = 1,
+ .prep = io_flistxattr_prep,
+ .issue = io_flistxattr,
+ },
};
const struct io_cold_def io_cold_defs[] = {
@@ -849,6 +859,14 @@ const struct io_cold_def io_cold_defs[] = {
.sqe_copy = io_uring_cmd_sqe_copy,
.cleanup = io_uring_cmd_cleanup,
},
+ [IORING_OP_FREMOVEXATTR] = {
+ .name = "FREMOVEXATTR",
+ .cleanup = io_xattr_cleanup,
+ },
+ [IORING_OP_FLISTXATTR] = {
+ .name = "FLISTXATTR",
+ .cleanup = io_xattr_cleanup,
+ },
};
const char *io_uring_get_opcode(u8 opcode)
diff --git a/io_uring/xattr.c b/io_uring/xattr.c
index 5303df3f247f..f2d6b1904058 100644
--- a/io_uring/xattr.c
+++ b/io_uring/xattr.c
@@ -195,3 +195,86 @@ int io_setxattr(struct io_kiocb *req, unsigned int issue_flags)
io_xattr_finish(req, ret);
return IOU_COMPLETE;
}
+
+int io_fremovexattr_prep(struct io_kiocb *req, const struct io_uring_sqe *sqe)
+{
+ struct io_xattr *ix = io_kiocb_to_cmd(req, struct io_xattr);
+ const char __user *name;
+ int ret;
+
+ INIT_DELAYED_FILENAME(&ix->filename);
+ ix->ctx.kname = NULL;
+ ix->ctx.kvalue = NULL;
+
+ if (READ_ONCE(sqe->off) || READ_ONCE(sqe->addr2) || READ_ONCE(sqe->len) ||
+ READ_ONCE(sqe->xattr_flags) || READ_ONCE(sqe->buf_index) ||
+ READ_ONCE(sqe->splice_fd_in) || READ_ONCE(sqe->addr3) ||
+ READ_ONCE(sqe->__pad2[0]))
+ return -EINVAL;
+
+ name = u64_to_user_ptr(READ_ONCE(sqe->addr));
+
+ ix->ctx.kname = kmalloc_obj(*ix->ctx.kname);
+ if (!ix->ctx.kname)
+ return -ENOMEM;
+
+ ret = import_xattr_name(ix->ctx.kname, name);
+ if (ret) {
+ kfree(ix->ctx.kname);
+ ix->ctx.kname = NULL;
+ return ret;
+ }
+
+ req->flags |= REQ_F_NEED_CLEANUP;
+ req->flags |= REQ_F_FORCE_ASYNC;
+ return 0;
+}
+
+int io_fremovexattr(struct io_kiocb *req, unsigned int issue_flags)
+{
+ struct io_xattr *ix = io_kiocb_to_cmd(req, struct io_xattr);
+ int ret;
+
+ WARN_ON_ONCE(issue_flags & IO_URING_F_NONBLOCK);
+
+ ret = file_removexattr(req->file, ix->ctx.kname);
+ io_xattr_finish(req, ret);
+ return IOU_COMPLETE;
+}
+
+int io_flistxattr_prep(struct io_kiocb *req, const struct io_uring_sqe *sqe)
+{
+ struct io_xattr *ix = io_kiocb_to_cmd(req, struct io_xattr);
+
+ INIT_DELAYED_FILENAME(&ix->filename);
+ ix->ctx.kname = NULL;
+ ix->ctx.kvalue = NULL;
+
+ if (READ_ONCE(sqe->addr) || READ_ONCE(sqe->off) ||
+ READ_ONCE(sqe->buf_index) || READ_ONCE(sqe->splice_fd_in) ||
+ READ_ONCE(sqe->addr3) || READ_ONCE(sqe->__pad2[0]))
+ return -EINVAL;
+
+ ix->ctx.value = u64_to_user_ptr(READ_ONCE(sqe->addr2));
+ ix->ctx.size = READ_ONCE(sqe->len);
+ ix->ctx.flags = READ_ONCE(sqe->xattr_flags);
+
+ if (ix->ctx.flags)
+ return -EINVAL;
+
+ req->flags |= REQ_F_NEED_CLEANUP;
+ req->flags |= REQ_F_FORCE_ASYNC;
+ return 0;
+}
+
+int io_flistxattr(struct io_kiocb *req, unsigned int issue_flags)
+{
+ struct io_xattr *ix = io_kiocb_to_cmd(req, struct io_xattr);
+ int ret;
+
+ WARN_ON_ONCE(issue_flags & IO_URING_F_NONBLOCK);
+
+ ret = file_listxattr(req->file, ix->ctx.value, ix->ctx.size);
+ io_xattr_finish(req, ret);
+ return IOU_COMPLETE;
+}
diff --git a/io_uring/xattr.h b/io_uring/xattr.h
index 9b459d2ae90c..d2487b49a5d2 100644
--- a/io_uring/xattr.h
+++ b/io_uring/xattr.h
@@ -13,3 +13,9 @@ int io_fgetxattr(struct io_kiocb *req, unsigned int issue_flags);
int io_getxattr_prep(struct io_kiocb *req, const struct io_uring_sqe *sqe);
int io_getxattr(struct io_kiocb *req, unsigned int issue_flags);
+
+int io_fremovexattr_prep(struct io_kiocb *req, const struct io_uring_sqe *sqe);
+int io_fremovexattr(struct io_kiocb *req, unsigned int issue_flags);
+
+int io_flistxattr_prep(struct io_kiocb *req, const struct io_uring_sqe *sqe);
+int io_flistxattr(struct io_kiocb *req, unsigned int issue_flags);
diff --git a/tools/include/uapi/linux/io_uring.h b/tools/include/uapi/linux/io_uring.h
index 1f300f6b85d7..45de6dd6e3c4 100644
--- a/tools/include/uapi/linux/io_uring.h
+++ b/tools/include/uapi/linux/io_uring.h
@@ -318,6 +318,8 @@ enum io_uring_op {
IORING_OP_PIPE,
IORING_OP_NOP128,
IORING_OP_URING_CMD128,
+ IORING_OP_FREMOVEXATTR,
+ IORING_OP_FLISTXATTR,
/* this goes last, obviously */
IORING_OP_LAST,
--
2.47.3
^ permalink raw reply [flat|nested] 5+ messages in thread
* Re: [PATCH v4 1/3] tools: sync io_uring.h UAPI header
2026-09-18 9:00 ` [PATCH v4 1/3] tools: sync io_uring.h UAPI header Aditya Prakash Srivastava
@ 2026-09-18 16:49 ` Caleb Sander Mateos
0 siblings, 0 replies; 5+ messages in thread
From: Caleb Sander Mateos @ 2026-09-18 16:49 UTC (permalink / raw)
To: Aditya Prakash Srivastava
Cc: Jens Axboe, Christian Brauner, Gabriel Krisman Bertazi,
Alexander Viro, Jan Kara, io-uring, linux-fsdevel, linux-kernel
On Fri, Sep 18, 2026 at 2:16 AM Aditya Prakash Srivastava
<aditya.ansh182@gmail.com> wrote:
>
> Sync tools/include/uapi/linux/io_uring.h with the main kernel
> include/uapi/linux/io_uring.h UAPI header to ensure they are
> perfectly aligned.
>
> Additionally, correct a minor space-before-tab checkpatch.pl warning
> in the comments of both headers (no functional change).
>
> Suggested-by: Gabriel Krisman Bertazi <krisman@suse.de>
> Signed-off-by: Aditya Prakash Srivastava <aditya.ansh182@gmail.com>
> ---
> include/uapi/linux/io_uring.h | 2 +-
> tools/include/uapi/linux/io_uring.h | 340 ++++++++++++++++++++++++++--
> 2 files changed, 325 insertions(+), 17 deletions(-)
>
> diff --git a/include/uapi/linux/io_uring.h b/include/uapi/linux/io_uring.h
> index 909fb7aea638..1f300f6b85d7 100644
> --- a/include/uapi/linux/io_uring.h
> +++ b/include/uapi/linux/io_uring.h
> @@ -426,7 +426,7 @@ enum io_uring_op {
> * IORING_RECVSEND_BUNDLE Used with IOSQE_BUFFER_SELECT. If set, send or
> * recv will grab as many buffers from the buffer
> * group ID given and send them all. The completion
> - * result will be the number of buffers send, with
> + * result will be the number of buffers send, with
Can you drop the tab character and just use a space between the words?
Best,
Caleb
> * the starting buffer ID in cqe->flags as per
> * usual for provided buffer usage. The buffers
> * will be contiguous from the starting buffer ID.
> diff --git a/tools/include/uapi/linux/io_uring.h b/tools/include/uapi/linux/io_uring.h
> index f1c16f817742..1f300f6b85d7 100644
> --- a/tools/include/uapi/linux/io_uring.h
> +++ b/tools/include/uapi/linux/io_uring.h
> @@ -10,6 +10,8 @@
>
> #include <linux/fs.h>
> #include <linux/types.h>
> +#include <linux/io_uring/zcrx.h>
> +
> /*
> * this file is shared with liburing and that has to autodetect
> * if linux/time_types.h is available or not, it can
> @@ -50,7 +52,7 @@ struct io_uring_sqe {
> };
> __u32 len; /* buffer size or number of iovecs */
> union {
> - __kernel_rwf_t rw_flags;
> + __u32 rw_flags;
> __u32 fsync_flags;
> __u16 poll_events; /* compatibility */
> __u32 poll32_events; /* word-reversed for BE */
> @@ -71,6 +73,9 @@ struct io_uring_sqe {
> __u32 uring_cmd_flags;
> __u32 waitid_flags;
> __u32 futex_flags;
> + __u32 install_fd_flags;
> + __u32 nop_flags;
> + __u32 pipe_flags;
> };
> __u64 user_data; /* data to be passed back at completion time */
> /* pack this to avoid bogus arm OABI complaints */
> @@ -85,17 +90,26 @@ struct io_uring_sqe {
> union {
> __s32 splice_fd_in;
> __u32 file_index;
> + __u32 zcrx_ifq_idx;
> __u32 optlen;
> struct {
> __u16 addr_len;
> __u16 __pad3[1];
> };
> + struct {
> + __u8 write_stream;
> + __u8 __pad4[3];
> + };
> };
> union {
> struct {
> __u64 addr3;
> __u64 __pad2[1];
> };
> + struct {
> + __u64 attr_ptr; /* pointer to attribute information */
> + __u64 attr_type_mask; /* bit mask of attributes */
> + };
> __u64 optval;
> /*
> * If the ring is initialized with IORING_SETUP_SQE128, then
> @@ -105,6 +119,18 @@ struct io_uring_sqe {
> };
> };
>
> +/* sqe->attr_type_mask flags */
> +#define IORING_RW_ATTR_FLAG_PI (1U << 0)
> +/* PI attribute information */
> +struct io_uring_attr_pi {
> + __u16 flags;
> + __u16 app_tag;
> + __u32 len;
> + __u64 addr;
> + __u64 seed;
> + __u64 rsvd;
> +};
> +
> /*
> * If sqe->file_index is set to this for opcodes that instantiate a new
> * direct descriptor (like openat/openat2/accept), then io_uring will allocate
> @@ -114,7 +140,7 @@ struct io_uring_sqe {
> */
> #define IORING_FILE_INDEX_ALLOC (~0U)
>
> -enum {
> +enum io_uring_sqe_flags_bit {
> IOSQE_FIXED_FILE_BIT,
> IOSQE_IO_DRAIN_BIT,
> IOSQE_IO_LINK_BIT,
> @@ -164,7 +190,8 @@ enum {
> /*
> * If COOP_TASKRUN is set, get notified if task work is available for
> * running and a kernel transition would be needed to run it. This sets
> - * IORING_SQ_TASKRUN in the sq ring flags. Not valid with COOP_TASKRUN.
> + * IORING_SQ_TASKRUN in the sq ring flags. Not valid without COOP_TASKRUN
> + * or DEFER_TASKRUN.
> */
> #define IORING_SETUP_TASKRUN_FLAG (1U << 9)
> #define IORING_SETUP_SQE128 (1U << 10) /* SQEs are 128 byte */
> @@ -198,6 +225,33 @@ enum {
> */
> #define IORING_SETUP_NO_SQARRAY (1U << 16)
>
> +/* Use hybrid poll in iopoll process */
> +#define IORING_SETUP_HYBRID_IOPOLL (1U << 17)
> +
> +/*
> + * Allow both 16b and 32b CQEs. If a 32b CQE is posted, it will have
> + * IORING_CQE_F_32 set in cqe->flags.
> + */
> +#define IORING_SETUP_CQE_MIXED (1U << 18)
> +
> +/*
> + * Allow both 64b and 128b SQEs. If a 128b SQE is posted, it will have
> + * a 128b opcode.
> + */
> +#define IORING_SETUP_SQE_MIXED (1U << 19)
> +
> +/*
> + * When set, io_uring ignores SQ head and tail and fetches SQEs to submit
> + * starting from index 0 instead from the index stored in the head pointer.
> + * IOW, the user should place all SQE at the beginning of the SQ memory
> + * before issuing a submission syscall.
> + *
> + * It requires IORING_SETUP_NO_SQARRAY and is incompatible with
> + * IORING_SETUP_SQPOLL. The user must also never change the SQ head and tail
> + * values and keep it set to 0. Any other value is undefined behaviour.
> + */
> +#define IORING_SETUP_SQ_REWIND (1U << 20)
> +
> enum io_uring_op {
> IORING_OP_NOP,
> IORING_OP_READV,
> @@ -253,6 +307,17 @@ enum io_uring_op {
> IORING_OP_FUTEX_WAIT,
> IORING_OP_FUTEX_WAKE,
> IORING_OP_FUTEX_WAITV,
> + IORING_OP_FIXED_FD_INSTALL,
> + IORING_OP_FTRUNCATE,
> + IORING_OP_BIND,
> + IORING_OP_LISTEN,
> + IORING_OP_RECV_ZC,
> + IORING_OP_EPOLL_WAIT,
> + IORING_OP_READV_FIXED,
> + IORING_OP_WRITEV_FIXED,
> + IORING_OP_PIPE,
> + IORING_OP_NOP128,
> + IORING_OP_URING_CMD128,
>
> /* this goes last, obviously */
> IORING_OP_LAST,
> @@ -262,9 +327,13 @@ enum io_uring_op {
> * sqe->uring_cmd_flags top 8bits aren't available for userspace
> * IORING_URING_CMD_FIXED use registered buffer; pass this flag
> * along with setting sqe->buf_index.
> + * IORING_URING_CMD_MULTISHOT must be used with buffer select, like other
> + * multishot commands. Not compatible with
> + * IORING_URING_CMD_FIXED, for now.
> */
> #define IORING_URING_CMD_FIXED (1U << 0)
> -#define IORING_URING_CMD_MASK IORING_URING_CMD_FIXED
> +#define IORING_URING_CMD_MULTISHOT (1U << 1)
> +#define IORING_URING_CMD_MASK (IORING_URING_CMD_FIXED | IORING_URING_CMD_MULTISHOT)
>
>
> /*
> @@ -274,6 +343,10 @@ enum io_uring_op {
>
> /*
> * sqe->timeout_flags
> + *
> + * IORING_TIMEOUT_IMMEDIATE_ARG: If set, sqe->addr stores the timeout
> + * value in nanoseconds instead of
> + * pointing to a timespec.
> */
> #define IORING_TIMEOUT_ABS (1U << 0)
> #define IORING_TIMEOUT_UPDATE (1U << 1)
> @@ -282,6 +355,7 @@ enum io_uring_op {
> #define IORING_LINK_TIMEOUT_UPDATE (1U << 4)
> #define IORING_TIMEOUT_ETIME_SUCCESS (1U << 5)
> #define IORING_TIMEOUT_MULTISHOT (1U << 6)
> +#define IORING_TIMEOUT_IMMEDIATE_ARG (1U << 7)
> #define IORING_TIMEOUT_CLOCK_MASK (IORING_TIMEOUT_BOOTTIME | IORING_TIMEOUT_REALTIME)
> #define IORING_TIMEOUT_UPDATE_MASK (IORING_TIMEOUT_UPDATE | IORING_LINK_TIMEOUT_UPDATE)
> /*
> @@ -348,11 +422,24 @@ enum io_uring_op {
> * 0 is reported if zerocopy was actually possible.
> * IORING_NOTIF_USAGE_ZC_COPIED if data was copied
> * (at least partially).
> + *
> + * IORING_RECVSEND_BUNDLE Used with IOSQE_BUFFER_SELECT. If set, send or
> + * recv will grab as many buffers from the buffer
> + * group ID given and send them all. The completion
> + * result will be the number of buffers send, with
> + * the starting buffer ID in cqe->flags as per
> + * usual for provided buffer usage. The buffers
> + * will be contiguous from the starting buffer ID.
> + *
> + * IORING_SEND_VECTORIZED If set, SEND[_ZC] will take a pointer to a io_vec
> + * to allow vectorized send operations.
> */
> #define IORING_RECVSEND_POLL_FIRST (1U << 0)
> #define IORING_RECV_MULTISHOT (1U << 1)
> #define IORING_RECVSEND_FIXED_BUF (1U << 2)
> #define IORING_SEND_ZC_REPORT_USAGE (1U << 3)
> +#define IORING_RECVSEND_BUNDLE (1U << 4)
> +#define IORING_SEND_VECTORIZED (1U << 5)
>
> /*
> * cqe.res for IORING_CQE_F_NOTIF if
> @@ -367,11 +454,13 @@ enum io_uring_op {
> * accept flags stored in sqe->ioprio
> */
> #define IORING_ACCEPT_MULTISHOT (1U << 0)
> +#define IORING_ACCEPT_DONTWAIT (1U << 1)
> +#define IORING_ACCEPT_POLL_FIRST (1U << 2)
>
> /*
> * IORING_OP_MSG_RING command types, stored in sqe->addr
> */
> -enum {
> +enum io_uring_msg_ring_flags {
> IORING_MSG_DATA, /* pass sqe->len as 'res' and off as user_data */
> IORING_MSG_SEND_FD, /* send a registered fd to another ring */
> };
> @@ -386,11 +475,30 @@ enum {
> /* Pass through the flags from sqe->file_index to cqe->flags */
> #define IORING_MSG_RING_FLAGS_PASS (1U << 1)
>
> +/*
> + * IORING_OP_FIXED_FD_INSTALL flags (sqe->install_fd_flags)
> + *
> + * IORING_FIXED_FD_NO_CLOEXEC Don't mark the fd as O_CLOEXEC
> + */
> +#define IORING_FIXED_FD_NO_CLOEXEC (1U << 0)
> +
> +/*
> + * IORING_OP_NOP flags (sqe->nop_flags)
> + *
> + * IORING_NOP_INJECT_RESULT Inject result from sqe->result
> + */
> +#define IORING_NOP_INJECT_RESULT (1U << 0)
> +#define IORING_NOP_FILE (1U << 1)
> +#define IORING_NOP_FIXED_FILE (1U << 2)
> +#define IORING_NOP_FIXED_BUFFER (1U << 3)
> +#define IORING_NOP_TW (1U << 4)
> +#define IORING_NOP_CQE32 (1U << 5)
> +
> /*
> * IO completion data structure (Completion Queue Entry)
> */
> struct io_uring_cqe {
> - __u64 user_data; /* sqe->data submission passed back */
> + __u64 user_data; /* sqe->user_data value passed back */
> __s32 res; /* result code for this event */
> __u32 flags;
>
> @@ -409,15 +517,33 @@ struct io_uring_cqe {
> * IORING_CQE_F_SOCK_NONEMPTY If set, more data to read after socket recv
> * IORING_CQE_F_NOTIF Set for notification CQEs. Can be used to distinct
> * them from sends.
> + * IORING_CQE_F_BUF_MORE If set, the buffer ID set in the completion will get
> + * more completions. In other words, the buffer is being
> + * partially consumed, and will be used by the kernel for
> + * more completions. This is only set for buffers used via
> + * the incremental buffer consumption, as provided by
> + * a ring buffer setup with IOU_PBUF_RING_INC. For any
> + * other provided buffer type, all completions with a
> + * buffer passed back is automatically returned to the
> + * application.
> + * IORING_CQE_F_SKIP If set, then the application/liburing must ignore this
> + * CQE. It's only purpose is to fill a gap in the ring,
> + * if a large CQE is attempted posted when the ring has
> + * just a single small CQE worth of space left before
> + * wrapping.
> + * IORING_CQE_F_32 If set, this is a 32b/big-cqe posting. Use with rings
> + * setup in a mixed CQE mode, where both 16b and 32b
> + * CQEs may be posted to the CQ ring.
> */
> #define IORING_CQE_F_BUFFER (1U << 0)
> #define IORING_CQE_F_MORE (1U << 1)
> #define IORING_CQE_F_SOCK_NONEMPTY (1U << 2)
> #define IORING_CQE_F_NOTIF (1U << 3)
> +#define IORING_CQE_F_BUF_MORE (1U << 4)
> +#define IORING_CQE_F_SKIP (1U << 5)
> +#define IORING_CQE_F_32 (1U << 15)
>
> -enum {
> - IORING_CQE_BUFFER_SHIFT = 16,
> -};
> +#define IORING_CQE_BUFFER_SHIFT 16
>
> /*
> * Magic offsets for the application to mmap the data it needs
> @@ -478,6 +604,9 @@ struct io_cqring_offsets {
> #define IORING_ENTER_SQ_WAIT (1U << 2)
> #define IORING_ENTER_EXT_ARG (1U << 3)
> #define IORING_ENTER_REGISTERED_RING (1U << 4)
> +#define IORING_ENTER_ABS_TIMER (1U << 5)
> +#define IORING_ENTER_EXT_ARG_REG (1U << 6)
> +#define IORING_ENTER_NO_IOWAIT (1U << 7)
>
> /*
> * Passed in for io_uring_setup(2). Copied back with updated info on success
> @@ -512,11 +641,15 @@ struct io_uring_params {
> #define IORING_FEAT_CQE_SKIP (1U << 11)
> #define IORING_FEAT_LINKED_FILE (1U << 12)
> #define IORING_FEAT_REG_REG_RING (1U << 13)
> +#define IORING_FEAT_RECVSEND_BUNDLE (1U << 14)
> +#define IORING_FEAT_MIN_TIMEOUT (1U << 15)
> +#define IORING_FEAT_RW_ATTR (1U << 16)
> +#define IORING_FEAT_NO_IOWAIT (1U << 17)
>
> /*
> * io_uring_register(2) opcodes and arguments
> */
> -enum {
> +enum io_uring_register_op {
> IORING_REGISTER_BUFFERS = 0,
> IORING_UNREGISTER_BUFFERS = 1,
> IORING_REGISTER_FILES = 2,
> @@ -558,6 +691,38 @@ enum {
> /* register a range of fixed file slots for automatic slot allocation */
> IORING_REGISTER_FILE_ALLOC_RANGE = 25,
>
> + /* return status information for a buffer group */
> + IORING_REGISTER_PBUF_STATUS = 26,
> +
> + /* set/clear busy poll settings */
> + IORING_REGISTER_NAPI = 27,
> + IORING_UNREGISTER_NAPI = 28,
> +
> + IORING_REGISTER_CLOCK = 29,
> +
> + /* clone registered buffers from source ring to current ring */
> + IORING_REGISTER_CLONE_BUFFERS = 30,
> +
> + /* send MSG_RING without having a ring */
> + IORING_REGISTER_SEND_MSG_RING = 31,
> +
> + /* register a netdev hw rx queue for zerocopy */
> + IORING_REGISTER_ZCRX_IFQ = 32,
> +
> + /* resize CQ ring */
> + IORING_REGISTER_RESIZE_RINGS = 33,
> +
> + IORING_REGISTER_MEM_REGION = 34,
> +
> + /* query various aspects of io_uring, see linux/io_uring/query.h */
> + IORING_REGISTER_QUERY = 35,
> +
> + /* auxiliary zcrx configuration, see enum zcrx_ctrl_op */
> + IORING_REGISTER_ZCRX_CTRL = 36,
> +
> + /* register bpf filtering programs */
> + IORING_REGISTER_BPF_FILTER = 37,
> +
> /* this goes last */
> IORING_REGISTER_LAST,
>
> @@ -566,7 +731,7 @@ enum {
> };
>
> /* io-wq worker categories */
> -enum {
> +enum io_wq_type {
> IO_WQ_BOUND,
> IO_WQ_UNBOUND,
> };
> @@ -578,6 +743,31 @@ struct io_uring_files_update {
> __aligned_u64 /* __s32 * */ fds;
> };
>
> +enum {
> + /* initialise with user provided memory pointed by user_addr */
> + IORING_MEM_REGION_TYPE_USER = 1,
> +};
> +
> +struct io_uring_region_desc {
> + __u64 user_addr;
> + __u64 size;
> + __u32 flags;
> + __u32 id;
> + __u64 mmap_offset;
> + __u64 __resv[4];
> +};
> +
> +enum {
> + /* expose the region as registered wait arguments */
> + IORING_MEM_REGION_REG_WAIT_ARG = 1,
> +};
> +
> +struct io_uring_mem_region_reg {
> + __u64 region_uptr; /* struct io_uring_region_desc * */
> + __u64 flags;
> + __u64 __resv[2];
> +};
> +
> /*
> * Register a fully sparse file space, rather than pass in an array of all
> * -1 file descriptors.
> @@ -638,6 +828,32 @@ struct io_uring_restriction {
> __u32 resv2[3];
> };
>
> +struct io_uring_task_restriction {
> + __u16 flags;
> + __u16 nr_res;
> + __u32 resv[3];
> + __DECLARE_FLEX_ARRAY(struct io_uring_restriction, restrictions);
> +};
> +
> +struct io_uring_clock_register {
> + __u32 clockid;
> + __u32 __resv[3];
> +};
> +
> +enum {
> + IORING_REGISTER_SRC_REGISTERED = (1U << 0),
> + IORING_REGISTER_DST_REPLACE = (1U << 1),
> +};
> +
> +struct io_uring_clone_buffers {
> + __u32 src_fd;
> + __u32 flags;
> + __u32 src_off;
> + __u32 dst_off;
> + __u32 nr;
> + __u32 pad[3];
> +};
> +
> struct io_uring_buf {
> __u64 addr;
> __u32 len;
> @@ -670,9 +886,17 @@ struct io_uring_buf_ring {
> * mmap(2) with the offset set as:
> * IORING_OFF_PBUF_RING | (bgid << IORING_OFF_PBUF_SHIFT)
> * to get a virtual mapping for the ring.
> + * IOU_PBUF_RING_INC: If set, buffers consumed from this buffer ring can be
> + * consumed incrementally. Normally one (or more) buffers
> + * are fully consumed. With incremental consumptions, it's
> + * feasible to register big ranges of buffers, and each
> + * use of it will consume only as much as it needs. This
> + * requires that both the kernel and application keep
> + * track of where the current read/recv index is at.
> */
> -enum {
> +enum io_uring_register_pbuf_ring_flags {
> IOU_PBUF_RING_MMAP = 1,
> + IOU_PBUF_RING_INC = 2,
> };
>
> /* argument for IORING_(UN)REGISTER_PBUF_RING */
> @@ -681,13 +905,57 @@ struct io_uring_buf_reg {
> __u32 ring_entries;
> __u16 bgid;
> __u16 flags;
> - __u64 resv[3];
> + __u32 min_left;
> + __u32 resv[5];
> +};
> +
> +/* argument for IORING_REGISTER_PBUF_STATUS */
> +struct io_uring_buf_status {
> + __u32 buf_group; /* input */
> + __u32 head; /* output */
> + __u32 resv[8];
> +};
> +
> +enum io_uring_napi_op {
> + /* register/ungister backward compatible opcode */
> + IO_URING_NAPI_REGISTER_OP = 0,
> +
> + /* opcodes to update napi_list when static tracking is used */
> + IO_URING_NAPI_STATIC_ADD_ID = 1,
> + IO_URING_NAPI_STATIC_DEL_ID = 2
> +};
> +
> +enum io_uring_napi_tracking_strategy {
> + /* value must be 0 for backward compatibility */
> + IO_URING_NAPI_TRACKING_DYNAMIC = 0,
> + IO_URING_NAPI_TRACKING_STATIC = 1,
> + IO_URING_NAPI_TRACKING_INACTIVE = 255
> +};
> +
> +/* argument for IORING_(UN)REGISTER_NAPI */
> +struct io_uring_napi {
> + __u32 busy_poll_to;
> + __u8 prefer_busy_poll;
> +
> + /* a io_uring_napi_op value */
> + __u8 opcode;
> + __u8 pad[2];
> +
> + /*
> + * for IO_URING_NAPI_REGISTER_OP, it is a
> + * io_uring_napi_tracking_strategy value.
> + *
> + * for IO_URING_NAPI_STATIC_ADD_ID/IO_URING_NAPI_STATIC_DEL_ID
> + * it is the napi id to add/del from napi_list.
> + */
> + __u32 op_param;
> + __u32 resv;
> };
>
> /*
> * io_uring_restriction->opcode values
> */
> -enum {
> +enum io_uring_register_restriction_op {
> /* Allow an io_uring_register(2) opcode */
> IORING_RESTRICTION_REGISTER_OP = 0,
>
> @@ -703,10 +971,33 @@ enum {
> IORING_RESTRICTION_LAST
> };
>
> +enum {
> + IORING_REG_WAIT_TS = (1U << 0),
> +};
> +
> +/*
> + * Argument for io_uring_enter(2) with
> + * IORING_GETEVENTS | IORING_ENTER_EXT_ARG_REG set, where the actual argument
> + * is an index into a previously registered fixed wait region described by
> + * the below structure.
> + */
> +struct io_uring_reg_wait {
> + struct __kernel_timespec ts;
> + __u32 min_wait_usec;
> + __u32 flags;
> + __u64 sigmask;
> + __u32 sigmask_sz;
> + __u32 pad[3];
> + __u64 pad2[2];
> +};
> +
> +/*
> + * Argument for io_uring_enter(2) with IORING_GETEVENTS | IORING_ENTER_EXT_ARG
> + */
> struct io_uring_getevents_arg {
> __u64 sigmask;
> __u32 sigmask_sz;
> - __u32 pad;
> + __u32 min_wait_usec;
> __u64 ts;
> };
>
> @@ -743,11 +1034,28 @@ struct io_uring_recvmsg_out {
> /*
> * Argument for IORING_OP_URING_CMD when file is a socket
> */
> -enum {
> +enum io_uring_socket_op {
> SOCKET_URING_OP_SIOCINQ = 0,
> SOCKET_URING_OP_SIOCOUTQ,
> SOCKET_URING_OP_GETSOCKOPT,
> SOCKET_URING_OP_SETSOCKOPT,
> + SOCKET_URING_OP_TX_TIMESTAMP,
> + SOCKET_URING_OP_GETSOCKNAME,
> +};
> +
> +/*
> + * SOCKET_URING_OP_TX_TIMESTAMP definitions
> + */
> +
> +#define IORING_TIMESTAMP_HW_SHIFT 16
> +/* The cqe->flags bit from which the timestamp type is stored */
> +#define IORING_TIMESTAMP_TYPE_SHIFT (IORING_TIMESTAMP_HW_SHIFT + 1)
> +/* The cqe->flags flag signifying whether it's a hardware timestamp */
> +#define IORING_CQE_F_TSTAMP_HW ((__u32)1 << IORING_TIMESTAMP_HW_SHIFT)
> +
> +struct io_timespec {
> + __u64 tv_sec;
> + __u64 tv_nsec;
> };
>
> #ifdef __cplusplus
> --
> 2.47.3
>
>
^ permalink raw reply [flat|nested] 5+ messages in thread
end of thread, other threads:[~2026-09-18 16:50 UTC | newest]
Thread overview: 5+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-09-18 9:00 [PATCH v4 0/3] io_uring: add fremovexattr and flistxattr support Aditya Prakash Srivastava
2026-09-18 9:00 ` [PATCH v4 1/3] tools: sync io_uring.h UAPI header Aditya Prakash Srivastava
2026-09-18 16:49 ` Caleb Sander Mateos
2026-09-18 9:00 ` [PATCH v4 2/3] fs: make file_listxattr and file_removexattr helpers non-static Aditya Prakash Srivastava
2026-09-18 9:00 ` [PATCH v4 3/3] io_uring: add fremovexattr and flistxattr support Aditya Prakash Srivastava
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®