* [PATCH v3 1/2] RDMA/rxe: copy send WQE to kernel buffer before processing
2026-10-07 21:27 [PATCH v3 0/2] RDMA/rxe: Fix TOCTOU races on mmap'd send queue Tristan Madani
@ 2026-10-07 21:27 ` Tristan Madani
2026-10-07 21:27 ` [PATCH v3 2/2] RDMA/rxe: copy send WQE to kernel buffer in completer path Tristan Madani
1 sibling, 0 replies; 3+ messages in thread
From: Tristan Madani @ 2026-10-07 21:27 UTC (permalink / raw)
To: Zhu Yanjun, Jason Gunthorpe, Leon Romanovsky
Cc: linux-rdma, linux-kernel, stable, Moni Shoua, Tristan Madani
From: Tristan Madani <tristan@talencesecurity.com>
The rxe send queue is mapped into userspace via mmap. The requester
processes Work Queue Entries (WQEs) directly from this shared buffer
without first copying them to kernel memory. Userspace can modify WQE
fields (num_sge, sge_offset, SGE entries) between kernel reads,
leading to inconsistent state in copy_data().
This is the send-path counterpart to the receive-path fixes:
- commit 22b8fbded65b8 ("RDMA/rxe: Fix TOCTOU heap overflow in
get_srq_wqe")
- commit d6ab440240a04 ("RDMA/rxe: Copy WQE to local buffer in
non-SRQ receive path")
Fix by copying the send WQE to a kernel-private buffer in req_next_wqe()
before processing. The local copy is reused across multi-packet sends
to preserve DMA progress state (cur_sge, sge_offset, resid). Field
updates are written back to the shared queue using WRITE_ONCE() for
individual fields and smp_store_release() for state transitions, so
the completer and userspace observe consistent values without the
tearing risk of bulk memcpy().
The copy is invalidated when the WQE index advances (last packet sent,
error, local ops, UD oversized) or when a retry resets WQE state.
Fixes: 8700e3e7c485 ("Soft RoCE driver")
Cc: stable@vger.kernel.org
Signed-off-by: Tristan Madani <tristan@talencesecurity.com>
---
drivers/infiniband/sw/rxe/rxe_req.c | 59 +++++++++++++++++++++++++--
drivers/infiniband/sw/rxe/rxe_verbs.h | 6 +++
2 files changed, 61 insertions(+), 4 deletions(-)
diff --git a/drivers/infiniband/sw/rxe/rxe_req.c b/drivers/infiniband/sw/rxe/rxe_req.c
index 24f5c044363f7..c72de69630e58 100644
--- a/drivers/infiniband/sw/rxe/rxe_req.c
+++ b/drivers/infiniband/sw/rxe/rxe_req.c
@@ -161,6 +161,26 @@ static void req_check_sq_drain_done(struct rxe_qp *qp)
spin_unlock_irqrestore(&qp->state_lock, flags);
}
+/* Write back requester WQE fields to shared memory using targeted
+ * stores so the completer and userspace observe consistent state.
+ */
+static void rxe_req_writeback_wqe(struct rxe_qp *qp)
+{
+ struct rxe_send_wqe *shared = qp->req.shared_wqe;
+ struct rxe_send_wqe *local = &qp->req.send_wqe.wqe;
+
+ if (!qp->req.send_wqe_valid || !shared)
+ return;
+
+ WRITE_ONCE(shared->status, local->status);
+ WRITE_ONCE(shared->first_psn, local->first_psn);
+ WRITE_ONCE(shared->last_psn, local->last_psn);
+ WRITE_ONCE(shared->mask, local->mask);
+ WRITE_ONCE(shared->has_rd_atomic, local->has_rd_atomic);
+ /* State must be last so the completer sees prior updates */
+ smp_store_release(&shared->state, local->state);
+}
+
static struct rxe_send_wqe *__req_next_wqe(struct rxe_qp *qp)
{
struct rxe_queue *q = qp->sq.queue;
@@ -178,6 +198,8 @@ static struct rxe_send_wqe *req_next_wqe(struct rxe_qp *qp)
{
struct rxe_send_wqe *wqe;
unsigned long flags;
+ unsigned int num_sge;
+ size_t copy_size;
req_check_sq_drain_done(qp);
@@ -193,6 +215,24 @@ static struct rxe_send_wqe *req_next_wqe(struct rxe_qp *qp)
}
spin_unlock_irqrestore(&qp->state_lock, flags);
+ /* Reuse the existing kernel-private copy if still valid */
+ if (qp->req.send_wqe_valid && qp->req.shared_wqe == wqe)
+ return &qp->req.send_wqe.wqe;
+
+ /* Copy WQE from userspace-mapped shared queue to kernel-private
+ * buffer to prevent TOCTOU races on DMA state fields.
+ */
+ num_sge = wqe->dma.num_sge;
+ if (unlikely(num_sge > qp->sq.max_sge)) {
+ rxe_dbg_qp(qp, "invalid num_sge in send WQE\n");
+ return NULL;
+ }
+ copy_size = sizeof(*wqe) + num_sge * sizeof(struct rxe_sge);
+ memcpy(&qp->req.send_wqe.wqe, wqe, copy_size);
+ qp->req.shared_wqe = wqe;
+ qp->req.send_wqe_valid = true;
+
+ wqe = &qp->req.send_wqe.wqe;
wqe->mask = wr_opcode_mask(wqe->wr.opcode, qp);
return wqe;
}
@@ -582,9 +622,13 @@ static void update_state(struct rxe_qp *qp, struct rxe_pkt_info *pkt)
{
qp->req.opcode = pkt->opcode;
- if (pkt->mask & RXE_END_MASK)
+ rxe_req_writeback_wqe(qp);
+
+ if (pkt->mask & RXE_END_MASK) {
qp->req.wqe_index = queue_next_index(qp->sq.queue,
qp->req.wqe_index);
+ qp->req.send_wqe_valid = false;
+ }
qp->need_req_skb = 0;
@@ -634,7 +678,9 @@ static int rxe_do_local_ops(struct rxe_qp *qp, struct rxe_send_wqe *wqe)
wqe->state = wqe_state_done;
wqe->status = IB_WC_SUCCESS;
+ rxe_req_writeback_wqe(qp);
qp->req.wqe_index = queue_next_index(qp->sq.queue, qp->req.wqe_index);
+ qp->req.send_wqe_valid = false;
return 0;
}
@@ -695,6 +741,7 @@ int rxe_requester(struct rxe_qp *qp)
if (unlikely(qp->req.need_retry && !qp->req.wait_for_rnr_timer)) {
req_retry(qp);
qp->req.need_retry = 0;
+ qp->req.send_wqe_valid = false;
}
wqe = req_next_wqe(qp);
@@ -772,10 +819,12 @@ int rxe_requester(struct rxe_qp *qp)
wqe->last_psn = qp->req.psn;
qp->req.psn = (qp->req.psn + 1) & BTH_PSN_MASK;
qp->req.opcode = IB_OPCODE_UD_SEND_ONLY;
- qp->req.wqe_index = queue_next_index(qp->sq.queue,
- qp->req.wqe_index);
wqe->state = wqe_state_done;
wqe->status = IB_WC_SUCCESS;
+ rxe_req_writeback_wqe(qp);
+ qp->req.wqe_index = queue_next_index(qp->sq.queue,
+ qp->req.wqe_index);
+ qp->req.send_wqe_valid = false;
goto done;
}
payload = mtu;
@@ -839,8 +888,10 @@ int rxe_requester(struct rxe_qp *qp)
goto out;
err:
/* update wqe_index for each wqe completion */
- qp->req.wqe_index = queue_next_index(qp->sq.queue, qp->req.wqe_index);
wqe->state = wqe_state_error;
+ rxe_req_writeback_wqe(qp);
+ qp->req.wqe_index = queue_next_index(qp->sq.queue, qp->req.wqe_index);
+ qp->req.send_wqe_valid = false;
rxe_qp_error(qp);
exit:
ret = -EAGAIN;
diff --git a/drivers/infiniband/sw/rxe/rxe_verbs.h b/drivers/infiniband/sw/rxe/rxe_verbs.h
index 0f5ffd94643f9..a22dfc6e5ae3c 100644
--- a/drivers/infiniband/sw/rxe/rxe_verbs.h
+++ b/drivers/infiniband/sw/rxe/rxe_verbs.h
@@ -114,6 +114,12 @@ struct rxe_req_info {
int wait_for_rnr_timer;
int noack_pkts;
int again;
+ struct rxe_send_wqe *shared_wqe;
+ bool send_wqe_valid;
+ struct {
+ struct rxe_send_wqe wqe;
+ struct ib_sge sge[RXE_MAX_SGE];
+ } send_wqe;
};
struct rxe_comp_info {
--
2.47.3
^ permalink raw reply [flat|nested] 3+ messages in thread* [PATCH v3 2/2] RDMA/rxe: copy send WQE to kernel buffer in completer path
2026-10-07 21:27 [PATCH v3 0/2] RDMA/rxe: Fix TOCTOU races on mmap'd send queue Tristan Madani
2026-10-07 21:27 ` [PATCH v3 1/2] RDMA/rxe: copy send WQE to kernel buffer before processing Tristan Madani
@ 2026-10-07 21:27 ` Tristan Madani
1 sibling, 0 replies; 3+ messages in thread
From: Tristan Madani @ 2026-10-07 21:27 UTC (permalink / raw)
To: Zhu Yanjun, Jason Gunthorpe, Leon Romanovsky
Cc: linux-rdma, linux-kernel, stable, Moni Shoua, Tristan Madani
From: Tristan Madani <tristan@talencesecurity.com>
The rxe completer processes send Work Queue Entries (WQEs) directly
from the mmap'd shared send queue without copying them to kernel memory.
Userspace can modify WQE fields (num_sge, opcode, PSN values, DMA state)
while the completer is processing them, leading to inconsistent
decisions in check_psn() and check_ack(), and potential out-of-bounds
access in copy_data() via a corrupted dma.num_sge.
This is the completer-path counterpart to the previous commit which
fixed the requester path.
Fix by copying the WQE to a kernel-private buffer in get_wqe() before
processing. The local copy is reused across multi-packet operations
(e.g. RDMA READ responses) when the WQE state is unchanged, preserving
DMA progress across completer invocations. The copy is refreshed when
the state changes (via smp_load_acquire pairing with the requester
smp_store_release) to ensure PSN and other fields are consistent.
Field updates (status, has_rd_atomic) are written back to the shared
queue using WRITE_ONCE() before advancing the consumer pointer, so
userspace observes consistent completion values.
Fixes: 8700e3e7c485 ("Soft RoCE driver")
Cc: stable@vger.kernel.org
Signed-off-by: Tristan Madani <tristan@talencesecurity.com>
---
drivers/infiniband/sw/rxe/rxe_comp.c | 54 ++++++++++++++++++++++++++-
drivers/infiniband/sw/rxe/rxe_verbs.h | 6 +++
2 files changed, 58 insertions(+), 2 deletions(-)
diff --git a/drivers/infiniband/sw/rxe/rxe_comp.c b/drivers/infiniband/sw/rxe/rxe_comp.c
index 1390e861bd1d7..f57c4396c28cb 100644
--- a/drivers/infiniband/sw/rxe/rxe_comp.c
+++ b/drivers/infiniband/sw/rxe/rxe_comp.c
@@ -88,6 +88,18 @@ static inline unsigned long rnrnak_jiffies(u8 timeout)
usecs_to_jiffies(rnrnak_usec[timeout]), 1);
}
+static void rxe_comp_writeback_wqe(struct rxe_qp *qp)
+{
+ struct rxe_send_wqe *shared = qp->comp.shared_wqe;
+ struct rxe_send_wqe *local = &qp->comp.comp_wqe.wqe;
+
+ if (!qp->comp.comp_wqe_valid || !shared)
+ return;
+
+ WRITE_ONCE(shared->status, local->status);
+ WRITE_ONCE(shared->has_rd_atomic, local->has_rd_atomic);
+}
+
static enum ib_wc_opcode wr_to_wc_opcode(enum ib_wr_opcode opcode)
{
switch (opcode) {
@@ -142,17 +154,52 @@ static inline enum comp_state get_wqe(struct rxe_qp *qp,
struct rxe_send_wqe **wqe_p)
{
struct rxe_send_wqe *wqe;
+ u32 state;
+ unsigned int num_sge;
/* we come here whether or not we found a response packet to see if
* there are any posted WQEs
*/
wqe = queue_head(qp->sq.queue, QUEUE_TYPE_FROM_CLIENT);
- *wqe_p = wqe;
/* no WQE or requester has not started it yet */
- if (!wqe || wqe->state == wqe_state_posted)
+ if (!wqe) {
+ *wqe_p = NULL;
return pkt ? COMPST_DONE : COMPST_EXIT;
+ }
+ /* Pairs with smp_store_release() in rxe_req_writeback_wqe() */
+ state = smp_load_acquire(&wqe->state);
+ if (state == wqe_state_posted) {
+ *wqe_p = wqe;
+ return pkt ? COMPST_DONE : COMPST_EXIT;
+ }
+
+ /* Reuse existing local copy if still processing same WQE
+ * with unchanged state (preserves DMA progress for multi-packet ops)
+ */
+ if (qp->comp.comp_wqe_valid && qp->comp.shared_wqe == wqe &&
+ qp->comp.comp_wqe.wqe.state == state) {
+ wqe = &qp->comp.comp_wqe.wqe;
+ *wqe_p = wqe;
+ goto check_state;
+ }
+
+ /* Copy shared WQE to kernel-private buffer */
+ num_sge = wqe->dma.num_sge;
+ if (unlikely(num_sge > RXE_MAX_SGE))
+ num_sge = RXE_MAX_SGE;
+
+ qp->comp.shared_wqe = wqe;
+ memcpy(&qp->comp.comp_wqe.wqe, wqe,
+ sizeof(*wqe) + num_sge * sizeof(struct rxe_sge));
+ qp->comp.comp_wqe_valid = true;
+ qp->comp.comp_wqe.wqe.dma.num_sge = num_sge;
+
+ wqe = &qp->comp.comp_wqe.wqe;
+ *wqe_p = wqe;
+
+check_state:
/* WQE does not require an ack */
if (wqe->state == wqe_state_done)
return COMPST_COMP_WQE;
@@ -454,6 +501,9 @@ static void do_complete(struct rxe_qp *qp, struct rxe_send_wqe *wqe)
if (post)
make_send_cqe(qp, wqe, &cqe);
+ rxe_comp_writeback_wqe(qp);
+ qp->comp.comp_wqe_valid = false;
+
queue_advance_consumer(qp->sq.queue, QUEUE_TYPE_FROM_CLIENT);
if (post)
diff --git a/drivers/infiniband/sw/rxe/rxe_verbs.h b/drivers/infiniband/sw/rxe/rxe_verbs.h
index a22dfc6e5ae3c..6205dbc29dff6 100644
--- a/drivers/infiniband/sw/rxe/rxe_verbs.h
+++ b/drivers/infiniband/sw/rxe/rxe_verbs.h
@@ -130,6 +130,12 @@ struct rxe_comp_info {
int started_retry;
u32 retry_cnt;
u32 rnr_retry;
+ struct rxe_send_wqe *shared_wqe;
+ bool comp_wqe_valid;
+ struct {
+ struct rxe_send_wqe wqe;
+ struct ib_sge sge[RXE_MAX_SGE];
+ } comp_wqe;
};
/* responder states */
--
2.47.3
^ permalink raw reply [flat|nested] 3+ messages in thread