* [PATCH v2 1/4] RDMA/rxe: Take MR reference under MW lock
2026-09-28 17:17 [PATCH v2 0/4] RDMA/rxe: Fix MW/MR lifetime races Dongliang Qin
@ 2026-09-28 17:17 ` Dongliang Qin
2026-09-28 17:17 ` [PATCH v2 2/4] RDMA/rxe: Reserve MR state during MW binding Dongliang Qin
` (2 subsequent siblings)
3 siblings, 0 replies; 5+ messages in thread
From: Dongliang Qin @ 2026-09-28 17:17 UTC (permalink / raw)
To: Zhu Yanjun, Jason Gunthorpe, Leon Romanovsky
Cc: Dongliang Qin, linux-rdma, linux-kernel, Bob Pearson, stable
The responder validates an MW, drops it, then reads mw->mr and takes
the MR reference. A bind or invalidate can swap the MR in this window,
so the responder may acquire a zero reference or use an MR that has
already been freed.
Move the MW lookup, validation, and MR reference acquisition into
rxe_mw_get_mr(), and perform all of them while holding mw->lock.
Fixes: cdd0b85675ae ("RDMA/rxe: Implement memory access through MWs")
Cc: stable@vger.kernel.org
Signed-off-by: Dongliang Qin <cccccccccccc777777@gmail.com>
---
drivers/infiniband/sw/rxe/rxe_loc.h | 3 ++-
drivers/infiniband/sw/rxe/rxe_mw.c | 34 +++++++++++++++++--------
drivers/infiniband/sw/rxe/rxe_resp.c | 38 +++-------------------------
3 files changed, 28 insertions(+), 47 deletions(-)
diff --git a/drivers/infiniband/sw/rxe/rxe_loc.h b/drivers/infiniband/sw/rxe/rxe_loc.h
index 64d636bf80fd2..2cbec92566f70 100644
--- a/drivers/infiniband/sw/rxe/rxe_loc.h
+++ b/drivers/infiniband/sw/rxe/rxe_loc.h
@@ -86,7 +86,8 @@ int rxe_alloc_mw(struct ib_mw *ibmw, struct ib_udata *udata);
int rxe_dealloc_mw(struct ib_mw *ibmw);
int rxe_bind_mw(struct rxe_qp *qp, struct rxe_send_wqe *wqe);
int rxe_invalidate_mw(struct rxe_qp *qp, u32 rkey);
-struct rxe_mw *rxe_lookup_mw(struct rxe_qp *qp, int access, u32 rkey);
+struct rxe_mr *rxe_mw_get_mr(struct rxe_qp *qp, int access, u32 rkey,
+ u64 *offset);
void rxe_mw_cleanup(struct rxe_pool_elem *elem);
/* rxe_net.c */
diff --git a/drivers/infiniband/sw/rxe/rxe_mw.c b/drivers/infiniband/sw/rxe/rxe_mw.c
index bddb7a2578313..04f795adacf53 100644
--- a/drivers/infiniband/sw/rxe/rxe_mw.c
+++ b/drivers/infiniband/sw/rxe/rxe_mw.c
@@ -291,26 +291,38 @@ int rxe_invalidate_mw(struct rxe_qp *qp, u32 rkey)
return ret;
}
-struct rxe_mw *rxe_lookup_mw(struct rxe_qp *qp, int access, u32 rkey)
+struct rxe_mr *rxe_mw_get_mr(struct rxe_qp *qp, int access, u32 rkey,
+ u64 *offset)
{
struct rxe_dev *rxe = to_rdev(qp->ibqp.device);
- struct rxe_pd *pd = to_rpd(qp->ibqp.pd);
+ struct rxe_mr *mr = NULL;
struct rxe_mw *mw;
- int index = rkey >> 8;
- mw = rxe_pool_get_index(&rxe->mw_pool, index);
+ mw = rxe_pool_get_index(&rxe->mw_pool, rkey >> 8);
if (!mw)
return NULL;
- if (unlikely((mw->rkey != rkey) || rxe_mw_pd(mw) != pd ||
- (mw->ibmw.type == IB_MW_TYPE_2 && mw->qp != qp) ||
- (mw->length == 0) || ((access & mw->access) != access) ||
- mw->state != RXE_MW_STATE_VALID)) {
- rxe_put(mw);
- return NULL;
+ spin_lock_bh(&mw->lock);
+
+ if (mw->rkey == rkey && rxe_mw_pd(mw) == to_rpd(qp->ibqp.pd) &&
+ (mw->ibmw.type != IB_MW_TYPE_2 || mw->qp == qp) &&
+ mw->length != 0 && mw->state == RXE_MW_STATE_VALID &&
+ (access & mw->access) == access) {
+ mr = mw->mr;
+ if (mr && mr->state == RXE_MR_STATE_VALID && rxe_get(mr)) {
+ if (offset)
+ *offset = (mw->access & IB_ZERO_BASED) ?
+ mw->addr : 0;
+ } else {
+ mr = NULL;
+ }
}
- return mw;
+ spin_unlock_bh(&mw->lock);
+
+ rxe_put(mw);
+
+ return mr;
}
void rxe_mw_cleanup(struct rxe_pool_elem *elem)
diff --git a/drivers/infiniband/sw/rxe/rxe_resp.c b/drivers/infiniband/sw/rxe/rxe_resp.c
index 02b16e2b49b8f..f5a957026fe87 100644
--- a/drivers/infiniband/sw/rxe/rxe_resp.c
+++ b/drivers/infiniband/sw/rxe/rxe_resp.c
@@ -468,7 +468,6 @@ static enum resp_states check_rkey(struct rxe_qp *qp,
struct rxe_pkt_info *pkt)
{
struct rxe_mr *mr = NULL;
- struct rxe_mw *mw = NULL;
u64 va;
u32 rkey;
u32 resid;
@@ -520,26 +519,12 @@ static enum resp_states check_rkey(struct rxe_qp *qp,
pktlen = payload_size(pkt);
if (rkey_is_mw(rkey)) {
- mw = rxe_lookup_mw(qp, access, rkey);
- if (!mw) {
- rxe_dbg_qp(qp, "no MW matches rkey %#x\n", rkey);
- state = get_rkey_violation_state(pkt);
- goto err;
- }
-
- mr = mw->mr;
+ mr = rxe_mw_get_mr(qp, access, rkey, &qp->resp.offset);
if (!mr) {
- rxe_dbg_qp(qp, "MW doesn't have an MR\n");
+ rxe_dbg_qp(qp, "no MW/MR matches rkey %#x\n", rkey);
state = get_rkey_violation_state(pkt);
goto err;
}
-
- if (mw->access & IB_ZERO_BASED)
- qp->resp.offset = mw->addr;
-
- rxe_get(mr);
- rxe_put(mw);
- mw = NULL;
} else {
mr = lookup_mr(qp->pd, access, rkey, RXE_LOOKUP_REMOTE);
if (!mr) {
@@ -605,8 +590,6 @@ static enum resp_states check_rkey(struct rxe_qp *qp,
qp->resp.mr = NULL;
if (mr)
rxe_put(mr);
- if (mw)
- rxe_put(mw);
return state;
}
@@ -894,24 +877,9 @@ static struct rxe_mr *rxe_recheck_mr(struct rxe_qp *qp, u32 rkey)
{
struct rxe_dev *rxe = to_rdev(qp->ibqp.device);
struct rxe_mr *mr;
- struct rxe_mw *mw;
if (rkey_is_mw(rkey)) {
- mw = rxe_pool_get_index(&rxe->mw_pool, rkey >> 8);
- if (!mw)
- return NULL;
-
- mr = mw->mr;
- if (mw->rkey != rkey || mw->state != RXE_MW_STATE_VALID ||
- !mr || mr->state != RXE_MR_STATE_VALID) {
- rxe_put(mw);
- return NULL;
- }
-
- rxe_get(mr);
- rxe_put(mw);
-
- return mr;
+ return rxe_mw_get_mr(qp, IB_ACCESS_REMOTE_READ, rkey, NULL);
}
mr = rxe_pool_get_index(&rxe->mr_pool, rkey >> 8);
--
2.43.0
^ permalink raw reply [flat|nested] 5+ messages in thread* [PATCH v2 2/4] RDMA/rxe: Reserve MR state during MW binding
2026-09-28 17:17 [PATCH v2 0/4] RDMA/rxe: Fix MW/MR lifetime races Dongliang Qin
2026-09-28 17:17 ` [PATCH v2 1/4] RDMA/rxe: Take MR reference under MW lock Dongliang Qin
@ 2026-09-28 17:17 ` Dongliang Qin
2026-09-28 17:17 ` [PATCH v2 3/4] RDMA/rxe: Invalidate MWs on QP destroy Dongliang Qin
2026-09-28 17:17 ` [PATCH v2 4/4] RDMA/rxe: Do not force cleanup on pool timeout Dongliang Qin
3 siblings, 0 replies; 5+ messages in thread
From: Dongliang Qin @ 2026-09-28 17:17 UTC (permalink / raw)
To: Zhu Yanjun, Jason Gunthorpe, Leon Romanovsky
Cc: Dongliang Qin, linux-rdma, linux-kernel, Bob Pearson, stable
MR invalidation, fast registration, reregistration, and deregistration
only test num_mw before changing MR state. A concurrent type-2 MW bind
can increment num_mw after that test and leave an MW bound to an MR
that is being destroyed.
Reserve num_mw == -1 while the MR state is changing, and allow binds
only while num_mw is nonnegative. This keeps the MW binding count and
MR lifetime synchronized without adding a new object.
An unprivileged user with access to the device can use this race to corrupt
kernel memory and escalate privileges.
A concurrent bind and deregistration reproducer made KASAN report:
BUG: KASAN: slab-use-after-free in rxe_mr_copy+0xd5/0x3d0
Read of size 1 at addr ff11000105d217b0 by task kworker/u16:3/44
Workqueue: rxe_wq do_work
Call Trace:
rxe_mr_copy+0xd5/0x3d0
rxe_receiver+0x1b50/0x3ed0
do_work+0xbb/0x250
process_one_work+0x412/0x780
worker_thread+0x341/0x5b0
kthread+0x1b8/0x210
Freed by task 1151:
kasan_save_stack+0x33/0x60
kfree+0x17c/0x450
rxe_mr_cleanup+0x3c/0x90
__rxe_cleanup+0x145/0x1e0
rxe_dereg_mr+0x4e/0x110
ib_dereg_mr_user+0xa4/0x1a0
ib_uverbs_ioctl+0x12f/0x1c0
Fixes: 32a577b4c3a9 ("RDMA/rxe: Add support for bind MW work requests")
Cc: stable@vger.kernel.org
Signed-off-by: Dongliang Qin <cccccccccccc777777@gmail.com>
---
drivers/infiniband/sw/rxe/rxe_loc.h | 4 ++
drivers/infiniband/sw/rxe/rxe_mr.c | 70 +++++++++++++++++++++++++--
drivers/infiniband/sw/rxe/rxe_mw.c | 44 +++++++++--------
drivers/infiniband/sw/rxe/rxe_verbs.c | 13 ++++-
4 files changed, 105 insertions(+), 26 deletions(-)
diff --git a/drivers/infiniband/sw/rxe/rxe_loc.h b/drivers/infiniband/sw/rxe/rxe_loc.h
index 2cbec92566f70..5e95e5c5e32d6 100644
--- a/drivers/infiniband/sw/rxe/rxe_loc.h
+++ b/drivers/infiniband/sw/rxe/rxe_loc.h
@@ -72,6 +72,10 @@ enum resp_states rxe_mr_do_atomic_op(struct rxe_mr *mr, u64 iova, int opcode,
enum resp_states rxe_mr_do_atomic_write(struct rxe_mr *mr, u64 iova, u64 value);
struct rxe_mr *lookup_mr(struct rxe_pd *pd, int access, u32 key,
enum rxe_mr_lookup_type type);
+int rxe_mr_get_mw(struct rxe_mr *mr);
+void rxe_mr_put_mw(struct rxe_mr *mr);
+bool rxe_mr_reserve_mw_state(struct rxe_mr *mr);
+void rxe_mr_release_mw_state(struct rxe_mr *mr);
int mr_check_range(struct rxe_mr *mr, u64 iova, size_t length);
int advance_dma_data(struct rxe_dma_info *dma, unsigned int length);
int rxe_invalidate_mr(struct rxe_qp *qp, u32 key);
diff --git a/drivers/infiniband/sw/rxe/rxe_mr.c b/drivers/infiniband/sw/rxe/rxe_mr.c
index 71d9ea4772890..4dc5b7405832f 100644
--- a/drivers/infiniband/sw/rxe/rxe_mr.c
+++ b/drivers/infiniband/sw/rxe/rxe_mr.c
@@ -721,6 +721,53 @@ struct rxe_mr *lookup_mr(struct rxe_pd *pd, int access, u32 key,
return mr;
}
+/*
+ * num_mw counts the MWs bound to an MR. The value -1 is reserved while
+ * the MR is changing state so that a new MW binding cannot race with
+ * invalidation, fast registration, or deregistration.
+ */
+int rxe_mr_get_mw(struct rxe_mr *mr)
+{
+ int old;
+
+ if (!rxe_get(mr))
+ return 0;
+
+ old = atomic_read(&mr->num_mw);
+ do {
+ if (old < 0)
+ goto err_put;
+ } while (!atomic_try_cmpxchg(&mr->num_mw, &old, old + 1));
+
+ if (mr->state != RXE_MR_STATE_VALID) {
+ atomic_dec_return(&mr->num_mw);
+ rxe_put(mr);
+ return 0;
+ }
+
+ return 1;
+
+err_put:
+ rxe_put(mr);
+ return 0;
+}
+
+void rxe_mr_put_mw(struct rxe_mr *mr)
+{
+ atomic_dec_return(&mr->num_mw);
+ rxe_put(mr);
+}
+
+bool rxe_mr_reserve_mw_state(struct rxe_mr *mr)
+{
+ return atomic_cmpxchg(&mr->num_mw, 0, -1) == 0;
+}
+
+void rxe_mr_release_mw_state(struct rxe_mr *mr)
+{
+ atomic_xchg(&mr->num_mw, 0);
+}
+
int rxe_invalidate_mr(struct rxe_qp *qp, u32 key)
{
struct rxe_dev *rxe = to_rdev(qp->ibqp.device);
@@ -743,7 +790,7 @@ int rxe_invalidate_mr(struct rxe_qp *qp, u32 key)
goto err_drop_ref;
}
- if (atomic_read(&mr->num_mw) > 0) {
+ if (!rxe_mr_reserve_mw_state(mr)) {
rxe_dbg_mr(mr, "Attempt to invalidate an MR while bound to MWs\n");
ret = -EINVAL;
goto err_drop_ref;
@@ -752,11 +799,16 @@ int rxe_invalidate_mr(struct rxe_qp *qp, u32 key)
if (unlikely(mr->ibmr.type != IB_MR_TYPE_MEM_REG)) {
rxe_dbg_mr(mr, "Type (%d) is wrong\n", mr->ibmr.type);
ret = -EINVAL;
- goto err_drop_ref;
+ goto err_release;
}
mr->state = RXE_MR_STATE_FREE;
+ rxe_mr_release_mw_state(mr);
ret = 0;
+ goto err_drop_ref;
+
+err_release:
+ rxe_mr_release_mw_state(mr);
err_drop_ref:
rxe_put(mr);
@@ -777,23 +829,26 @@ int rxe_reg_fast_mr(struct rxe_qp *qp, struct rxe_send_wqe *wqe)
u32 key = wqe->wr.wr.reg.key;
u32 access = wqe->wr.wr.reg.access;
+ if (!rxe_mr_reserve_mw_state(mr))
+ return -EINVAL;
+
/* user can only register MR in free state */
if (unlikely(mr->state != RXE_MR_STATE_FREE)) {
rxe_dbg_mr(mr, "mr->lkey = 0x%x not free\n", mr->lkey);
- return -EINVAL;
+ goto err_release;
}
/* user can only register mr with qp in same protection domain */
if (unlikely(qp->ibqp.pd != mr->ibmr.pd)) {
rxe_dbg_mr(mr, "qp->pd and mr->pd don't match\n");
- return -EINVAL;
+ goto err_release;
}
/* user is only allowed to change key portion of l/rkey */
if (unlikely((mr->lkey & ~0xff) != (key & ~0xff))) {
rxe_dbg_mr(mr, "key = 0x%x has wrong index mr->lkey = 0x%x\n",
key, mr->lkey);
- return -EINVAL;
+ goto err_release;
}
mr->access = access;
@@ -801,8 +856,13 @@ int rxe_reg_fast_mr(struct rxe_qp *qp, struct rxe_send_wqe *wqe)
mr->rkey = key;
mr->ibmr.iova = wqe->wr.wr.reg.mr->iova;
mr->state = RXE_MR_STATE_VALID;
+ rxe_mr_release_mw_state(mr);
return 0;
+
+err_release:
+ rxe_mr_release_mw_state(mr);
+ return -EINVAL;
}
void rxe_mr_cleanup(struct rxe_pool_elem *elem)
diff --git a/drivers/infiniband/sw/rxe/rxe_mw.c b/drivers/infiniband/sw/rxe/rxe_mw.c
index 04f795adacf53..82e9fef89b6c3 100644
--- a/drivers/infiniband/sw/rxe/rxe_mw.c
+++ b/drivers/infiniband/sw/rxe/rxe_mw.c
@@ -136,33 +136,39 @@ static int rxe_check_bind_mw(struct rxe_qp *qp, struct rxe_send_wqe *wqe,
return 0;
}
-static void rxe_do_bind_mw(struct rxe_qp *qp, struct rxe_send_wqe *wqe,
- struct rxe_mw *mw, struct rxe_mr *mr, int access)
+static int rxe_do_bind_mw(struct rxe_qp *qp, struct rxe_send_wqe *wqe,
+ struct rxe_mw *mw, struct rxe_mr *mr, int access)
{
u32 key = wqe->wr.wr.mw.rkey & 0xff;
- mw->rkey = (mw->rkey & ~0xff) | key;
- mw->access = access;
- mw->state = RXE_MW_STATE_VALID;
- mw->addr = wqe->wr.wr.mw.addr;
- mw->length = wqe->wr.wr.mw.length;
+ if (mw->ibmw.type == IB_MW_TYPE_2 && !rxe_get(qp))
+ return -EINVAL;
+
+ if (mr && !rxe_mr_get_mw(mr)) {
+ if (mw->ibmw.type == IB_MW_TYPE_2)
+ rxe_put(qp);
+ return -EINVAL;
+ }
if (mw->mr) {
- rxe_put(mw->mr);
- atomic_dec(&mw->mr->num_mw);
+ rxe_mr_put_mw(mw->mr);
mw->mr = NULL;
}
- if (mw->length) {
+ if (mr)
mw->mr = mr;
- atomic_inc(&mr->num_mw);
- rxe_get(mr);
- }
+
+ mw->rkey = (mw->rkey & ~0xff) | key;
+ mw->access = access;
+ mw->state = RXE_MW_STATE_VALID;
+ mw->addr = wqe->wr.wr.mw.addr;
+ mw->length = wqe->wr.wr.mw.length;
if (mw->ibmw.type == IB_MW_TYPE_2) {
- rxe_get(qp);
mw->qp = qp;
}
+
+ return 0;
}
int rxe_bind_mw(struct rxe_qp *qp, struct rxe_send_wqe *wqe)
@@ -213,7 +219,9 @@ int rxe_bind_mw(struct rxe_qp *qp, struct rxe_send_wqe *wqe)
if (ret)
goto err_unlock;
- rxe_do_bind_mw(qp, wqe, mw, mr, access);
+ ret = rxe_do_bind_mw(qp, wqe, mw, mr, access);
+ if (ret)
+ goto err_unlock;
err_unlock:
spin_unlock_bh(&mw->lock);
err_drop_mr:
@@ -250,8 +258,7 @@ static void rxe_do_invalidate_mw(struct rxe_mw *mw)
/* valid type 2 MW will always have an MR pointer */
mr = mw->mr;
mw->mr = NULL;
- atomic_dec(&mr->num_mw);
- rxe_put(mr);
+ rxe_mr_put_mw(mr);
mw->access = 0;
mw->addr = 0;
@@ -336,8 +343,7 @@ void rxe_mw_cleanup(struct rxe_pool_elem *elem)
struct rxe_mr *mr = mw->mr;
mw->mr = NULL;
- atomic_dec(&mr->num_mw);
- rxe_put(mr);
+ rxe_mr_put_mw(mr);
}
if (mw->qp) {
diff --git a/drivers/infiniband/sw/rxe/rxe_verbs.c b/drivers/infiniband/sw/rxe/rxe_verbs.c
index 3864284522ebc..8553c8402c619 100644
--- a/drivers/infiniband/sw/rxe/rxe_verbs.c
+++ b/drivers/infiniband/sw/rxe/rxe_verbs.c
@@ -1337,6 +1337,11 @@ static struct ib_mr *rxe_rereg_user_mr(struct ib_mr *ibmr, int flags,
return ERR_PTR(-EOPNOTSUPP);
}
+ if (!rxe_mr_reserve_mw_state(mr)) {
+ rxe_err_mr(mr, "mr has mws bound\n");
+ return ERR_PTR(-EINVAL);
+ }
+
if (flags & IB_MR_REREG_PD) {
rxe_put(old_pd);
rxe_get(pd);
@@ -1346,6 +1351,8 @@ static struct ib_mr *rxe_rereg_user_mr(struct ib_mr *ibmr, int flags,
if (flags & IB_MR_REREG_ACCESS)
mr->access = access;
+ rxe_mr_release_mw_state(mr);
+
return NULL;
}
@@ -1401,13 +1408,15 @@ static int rxe_dereg_mr(struct ib_mr *ibmr, struct ib_udata *udata)
struct rxe_mr *mr = to_rmr(ibmr);
int err, cleanup_err;
- /* See IBA 10.6.7.2.6 */
- if (atomic_read(&mr->num_mw) > 0) {
+ /* See IBA 10.6.7.2.6. Leave num_mw set to -1 for destruction. */
+ if (!rxe_mr_reserve_mw_state(mr)) {
err = -EINVAL;
rxe_dbg_mr(mr, "mr has mw's bound\n");
goto err_out;
}
+ mr->state = RXE_MR_STATE_INVALID;
+
cleanup_err = rxe_cleanup(mr);
if (cleanup_err)
rxe_err_mr(mr, "cleanup failed, err = %d\n", cleanup_err);
--
2.43.0
^ permalink raw reply [flat|nested] 5+ messages in thread* [PATCH v2 3/4] RDMA/rxe: Invalidate MWs on QP destroy
2026-09-28 17:17 [PATCH v2 0/4] RDMA/rxe: Fix MW/MR lifetime races Dongliang Qin
2026-09-28 17:17 ` [PATCH v2 1/4] RDMA/rxe: Take MR reference under MW lock Dongliang Qin
2026-09-28 17:17 ` [PATCH v2 2/4] RDMA/rxe: Reserve MR state during MW binding Dongliang Qin
@ 2026-09-28 17:17 ` Dongliang Qin
2026-09-28 17:17 ` [PATCH v2 4/4] RDMA/rxe: Do not force cleanup on pool timeout Dongliang Qin
3 siblings, 0 replies; 5+ messages in thread
From: Dongliang Qin @ 2026-09-28 17:17 UTC (permalink / raw)
To: Zhu Yanjun, Jason Gunthorpe, Leon Romanovsky
Cc: Dongliang Qin, linux-rdma, linux-kernel, Bob Pearson, stable
A type-2 MW holds a reference to the QP that bound it. If the QP is
destroyed while an MW is still bound, the MW keeps the QP alive and the
responder can retain a stale QP association.
Scanning the global MW pool to find these MWs is unsafe: pool entries are
struct rxe_pool_elem pointers, and concurrent deallocation can invalidate
the object being visited. It is also unnecessary because only MWs bound to
the QP being destroyed are relevant.
Track type-2 MWs on a per-QP list. Clear qp->valid and stop the send task
before invalidating the list so a late bind cannot attach a new MW after
invalidation. Yield between MWs to avoid spending a long time in a
non-preemptible loop when a QP has many bound MWs.
Fixes: 32a577b4c3a9 ("RDMA/rxe: Add support for bind MW work requests")
Cc: stable@vger.kernel.org
Signed-off-by: Dongliang Qin <cccccccccccc777777@gmail.com>
---
Changes in v2:
- Track type-2 MWs bound to a QP in a per-QP list instead of scanning
the global MW pool.
- Fix the XArray iterator type mismatch by no longer treating pool
elements as struct rxe_mw pointers.
- Clear qp->valid and stop the send task before invalidating MWs so a
late bind cannot attach a new MW after invalidation.
- Yield between MW invalidations to avoid long non-preemptible loops.
v1: https://lore.kernel.org/linux-rdma/20260928155351.3222978-4-cccccccccccc777777@gmail.com/
drivers/infiniband/sw/rxe/rxe_loc.h | 1 +
drivers/infiniband/sw/rxe/rxe_mw.c | 44 +++++++++++++++++++++++++--
drivers/infiniband/sw/rxe/rxe_qp.c | 2 ++
drivers/infiniband/sw/rxe/rxe_verbs.c | 8 +++++
drivers/infiniband/sw/rxe/rxe_verbs.h | 3 ++
5 files changed, 55 insertions(+), 3 deletions(-)
diff --git a/drivers/infiniband/sw/rxe/rxe_loc.h b/drivers/infiniband/sw/rxe/rxe_loc.h
index 5e95e5c5e32d6..5c648a5ef34b1 100644
--- a/drivers/infiniband/sw/rxe/rxe_loc.h
+++ b/drivers/infiniband/sw/rxe/rxe_loc.h
@@ -90,6 +90,7 @@ int rxe_alloc_mw(struct ib_mw *ibmw, struct ib_udata *udata);
int rxe_dealloc_mw(struct ib_mw *ibmw);
int rxe_bind_mw(struct rxe_qp *qp, struct rxe_send_wqe *wqe);
int rxe_invalidate_mw(struct rxe_qp *qp, u32 rkey);
+void rxe_invalidate_mws(struct rxe_qp *qp);
struct rxe_mr *rxe_mw_get_mr(struct rxe_qp *qp, int access, u32 rkey,
u64 *offset);
void rxe_mw_cleanup(struct rxe_pool_elem *elem);
diff --git a/drivers/infiniband/sw/rxe/rxe_mw.c b/drivers/infiniband/sw/rxe/rxe_mw.c
index 82e9fef89b6c3..678f251c8ded3 100644
--- a/drivers/infiniband/sw/rxe/rxe_mw.c
+++ b/drivers/infiniband/sw/rxe/rxe_mw.c
@@ -166,6 +166,9 @@ static int rxe_do_bind_mw(struct rxe_qp *qp, struct rxe_send_wqe *wqe,
if (mw->ibmw.type == IB_MW_TYPE_2) {
mw->qp = qp;
+ spin_lock(&qp->mw_lock);
+ list_add(&mw->qp_list, &qp->mw_list);
+ spin_unlock(&qp->mw_lock);
}
return 0;
@@ -247,11 +250,13 @@ static int rxe_check_invalidate_mw(struct rxe_qp *qp, struct rxe_mw *mw)
static void rxe_do_invalidate_mw(struct rxe_mw *mw)
{
- struct rxe_qp *qp;
struct rxe_mr *mr;
+ struct rxe_qp *qp = mw->qp;
+
+ spin_lock(&qp->mw_lock);
+ list_del_init(&mw->qp_list);
+ spin_unlock(&qp->mw_lock);
- /* valid type 2 MW will always have a QP pointer */
- qp = mw->qp;
mw->qp = NULL;
rxe_put(qp);
@@ -298,6 +303,35 @@ int rxe_invalidate_mw(struct rxe_qp *qp, u32 rkey)
return ret;
}
+void rxe_invalidate_mws(struct rxe_qp *qp)
+{
+ struct rxe_mw *mw;
+
+ for (;;) {
+ spin_lock(&qp->mw_lock);
+ if (list_empty(&qp->mw_list)) {
+ spin_unlock(&qp->mw_lock);
+ return;
+ }
+
+ mw = list_first_entry(&qp->mw_list, struct rxe_mw, qp_list);
+ if (!rxe_get(mw)) {
+ list_del_init(&mw->qp_list);
+ spin_unlock(&qp->mw_lock);
+ continue;
+ }
+ spin_unlock(&qp->mw_lock);
+
+ spin_lock_bh(&mw->lock);
+ if (mw->qp == qp)
+ rxe_do_invalidate_mw(mw);
+ spin_unlock_bh(&mw->lock);
+
+ rxe_put(mw);
+ cond_resched();
+ }
+}
+
struct rxe_mr *rxe_mw_get_mr(struct rxe_qp *qp, int access, u32 rkey,
u64 *offset)
{
@@ -349,6 +383,10 @@ void rxe_mw_cleanup(struct rxe_pool_elem *elem)
if (mw->qp) {
struct rxe_qp *qp = mw->qp;
+ spin_lock(&qp->mw_lock);
+ list_del_init(&mw->qp_list);
+ spin_unlock(&qp->mw_lock);
+
mw->qp = NULL;
rxe_put(qp);
}
diff --git a/drivers/infiniband/sw/rxe/rxe_qp.c b/drivers/infiniband/sw/rxe/rxe_qp.c
index 311f285d78a6b..132d224a278b2 100644
--- a/drivers/infiniband/sw/rxe/rxe_qp.c
+++ b/drivers/infiniband/sw/rxe/rxe_qp.c
@@ -220,6 +220,8 @@ static void rxe_qp_init_misc(struct rxe_dev *rxe, struct rxe_qp *qp,
}
spin_lock_init(&qp->state_lock);
+ spin_lock_init(&qp->mw_lock);
+ INIT_LIST_HEAD(&qp->mw_list);
spin_lock_init(&qp->sq.sq_lock);
spin_lock_init(&qp->rq.producer_lock);
diff --git a/drivers/infiniband/sw/rxe/rxe_verbs.c b/drivers/infiniband/sw/rxe/rxe_verbs.c
index 8553c8402c619..1f45e8bc53da5 100644
--- a/drivers/infiniband/sw/rxe/rxe_verbs.c
+++ b/drivers/infiniband/sw/rxe/rxe_verbs.c
@@ -650,6 +650,7 @@ static int rxe_query_qp(struct ib_qp *ibqp, struct ib_qp_attr *attr,
static int rxe_destroy_qp(struct ib_qp *ibqp, struct ib_udata *udata)
{
struct rxe_qp *qp = to_rqp(ibqp);
+ unsigned long flags;
int err;
err = rxe_qp_chk_destroy(qp);
@@ -658,6 +659,13 @@ static int rxe_destroy_qp(struct ib_qp *ibqp, struct ib_udata *udata)
goto err_out;
}
+ spin_lock_irqsave(&qp->state_lock, flags);
+ qp->valid = 0;
+ spin_unlock_irqrestore(&qp->state_lock, flags);
+
+ rxe_disable_task(&qp->send_task);
+ rxe_invalidate_mws(qp);
+
err = rxe_cleanup(qp);
if (err)
rxe_err_qp(qp, "cleanup failed, err = %d\n", err);
diff --git a/drivers/infiniband/sw/rxe/rxe_verbs.h b/drivers/infiniband/sw/rxe/rxe_verbs.h
index 0f5ffd94643f9..d6585581e9c33 100644
--- a/drivers/infiniband/sw/rxe/rxe_verbs.h
+++ b/drivers/infiniband/sw/rxe/rxe_verbs.h
@@ -288,6 +288,8 @@ struct rxe_qp {
struct timer_list rnr_nak_timer;
spinlock_t state_lock; /* guard requester and completer */
+ spinlock_t mw_lock;
+ struct list_head mw_list;
struct execute_work cleanup_work;
};
@@ -387,6 +389,7 @@ struct rxe_mw {
int access;
u64 addr;
u64 length;
+ struct list_head qp_list;
};
struct rxe_mcg {
--
2.43.0
^ permalink raw reply [flat|nested] 5+ messages in thread* [PATCH v2 4/4] RDMA/rxe: Do not force cleanup on pool timeout
2026-09-28 17:17 [PATCH v2 0/4] RDMA/rxe: Fix MW/MR lifetime races Dongliang Qin
` (2 preceding siblings ...)
2026-09-28 17:17 ` [PATCH v2 3/4] RDMA/rxe: Invalidate MWs on QP destroy Dongliang Qin
@ 2026-09-28 17:17 ` Dongliang Qin
3 siblings, 0 replies; 5+ messages in thread
From: Dongliang Qin @ 2026-09-28 17:17 UTC (permalink / raw)
To: Zhu Yanjun, Jason Gunthorpe, Leon Romanovsky
Cc: Dongliang Qin, linux-rdma, linux-kernel, Bob Pearson, stable
The sleepable pool cleanup path waits up to 50 seconds for the last
reference and then continues cleanup anyway. If a reference is still
held, that turns a lifetime bug into a use-after-free.
Wait for completion instead. A stuck object is easier to diagnose than
a stale pointer; the existing non-sleepable path is unchanged.
Fixes: 215d0a755e1b ("RDMA/rxe: Stop lookup of partially built objects")
Cc: stable@vger.kernel.org
Signed-off-by: Dongliang Qin <cccccccccccc777777@gmail.com>
---
drivers/infiniband/sw/rxe/rxe_pool.c | 14 ++------------
1 file changed, 2 insertions(+), 12 deletions(-)
diff --git a/drivers/infiniband/sw/rxe/rxe_pool.c b/drivers/infiniband/sw/rxe/rxe_pool.c
index d9cb682fd71f8..d5ff5d453f5f8 100644
--- a/drivers/infiniband/sw/rxe/rxe_pool.c
+++ b/drivers/infiniband/sw/rxe/rxe_pool.c
@@ -178,7 +178,7 @@ int __rxe_cleanup(struct rxe_pool_elem *elem, bool sleepable)
{
struct rxe_pool *pool = elem->pool;
struct xarray *xa = &pool->xa;
- int ret, err = 0;
+ int err = 0;
void *xa_ret;
if (sleepable)
@@ -201,17 +201,7 @@ int __rxe_cleanup(struct rxe_pool_elem *elem, bool sleepable)
* return to rdma-core
*/
if (sleepable) {
- if (!completion_done(&elem->complete)) {
- ret = wait_for_completion_timeout(&elem->complete,
- msecs_to_jiffies(50000));
-
- /* Shouldn't happen. There are still references to
- * the object but, rather than deadlock, free the
- * object or pass back to rdma-core.
- */
- if (WARN_ON(!ret))
- err = -ETIMEDOUT;
- }
+ wait_for_completion(&elem->complete);
} else {
unsigned long until = jiffies + RXE_POOL_TIMEOUT;
--
2.43.0
^ permalink raw reply [flat|nested] 5+ messages in thread