mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Dongliang Qin <cccccccccccc777777@gmail.com>
To: Zhu Yanjun <zyjzyj2000@gmail.com>, Jason Gunthorpe <jgg@ziepe.ca>,
	Leon Romanovsky <leon@kernel.org>
Cc: Dongliang Qin <cccccccccccc777777@gmail.com>,
	linux-rdma@vger.kernel.org, linux-kernel@vger.kernel.org,
	Bob Pearson <rpearsonhpe@gmail.com>,
	stable@vger.kernel.org
Subject: [PATCH v2 3/4] RDMA/rxe: Invalidate MWs on QP destroy
Date: Tue, 29 Sep 2026 01:17:55 +0800	[thread overview]
Message-ID: <20260928171756.3254016-4-cccccccccccc777777@gmail.com> (raw)
In-Reply-To: <20260928171756.3254016-1-cccccccccccc777777@gmail.com>

A type-2 MW holds a reference to the QP that bound it. If the QP is
destroyed while an MW is still bound, the MW keeps the QP alive and the
responder can retain a stale QP association.

Scanning the global MW pool to find these MWs is unsafe: pool entries are
struct rxe_pool_elem pointers, and concurrent deallocation can invalidate
the object being visited. It is also unnecessary because only MWs bound to
the QP being destroyed are relevant.

Track type-2 MWs on a per-QP list. Clear qp->valid and stop the send task
before invalidating the list so a late bind cannot attach a new MW after
invalidation. Yield between MWs to avoid spending a long time in a
non-preemptible loop when a QP has many bound MWs.

Fixes: 32a577b4c3a9 ("RDMA/rxe: Add support for bind MW work requests")

Cc: stable@vger.kernel.org
Signed-off-by: Dongliang Qin <cccccccccccc777777@gmail.com>
---
Changes in v2:

- Track type-2 MWs bound to a QP in a per-QP list instead of scanning
  the global MW pool.
- Fix the XArray iterator type mismatch by no longer treating pool
  elements as struct rxe_mw pointers.
- Clear qp->valid and stop the send task before invalidating MWs so a
  late bind cannot attach a new MW after invalidation.
- Yield between MW invalidations to avoid long non-preemptible loops.

v1: https://lore.kernel.org/linux-rdma/20260928155351.3222978-4-cccccccccccc777777@gmail.com/

drivers/infiniband/sw/rxe/rxe_loc.h   |  1 +
drivers/infiniband/sw/rxe/rxe_mw.c    | 44 +++++++++++++++++++++++++--
 drivers/infiniband/sw/rxe/rxe_qp.c    |  2 ++
 drivers/infiniband/sw/rxe/rxe_verbs.c |  8 +++++
 drivers/infiniband/sw/rxe/rxe_verbs.h |  3 ++
 5 files changed, 55 insertions(+), 3 deletions(-)

diff --git a/drivers/infiniband/sw/rxe/rxe_loc.h b/drivers/infiniband/sw/rxe/rxe_loc.h
index 5e95e5c5e32d6..5c648a5ef34b1 100644
--- a/drivers/infiniband/sw/rxe/rxe_loc.h
+++ b/drivers/infiniband/sw/rxe/rxe_loc.h
@@ -90,6 +90,7 @@ int rxe_alloc_mw(struct ib_mw *ibmw, struct ib_udata *udata);
 int rxe_dealloc_mw(struct ib_mw *ibmw);
 int rxe_bind_mw(struct rxe_qp *qp, struct rxe_send_wqe *wqe);
 int rxe_invalidate_mw(struct rxe_qp *qp, u32 rkey);
+void rxe_invalidate_mws(struct rxe_qp *qp);
 struct rxe_mr *rxe_mw_get_mr(struct rxe_qp *qp, int access, u32 rkey,
 			     u64 *offset);
 void rxe_mw_cleanup(struct rxe_pool_elem *elem);
diff --git a/drivers/infiniband/sw/rxe/rxe_mw.c b/drivers/infiniband/sw/rxe/rxe_mw.c
index 82e9fef89b6c3..678f251c8ded3 100644
--- a/drivers/infiniband/sw/rxe/rxe_mw.c
+++ b/drivers/infiniband/sw/rxe/rxe_mw.c
@@ -166,6 +166,9 @@ static int rxe_do_bind_mw(struct rxe_qp *qp, struct rxe_send_wqe *wqe,
 
 	if (mw->ibmw.type == IB_MW_TYPE_2) {
 		mw->qp = qp;
+		spin_lock(&qp->mw_lock);
+		list_add(&mw->qp_list, &qp->mw_list);
+		spin_unlock(&qp->mw_lock);
 	}
 
 	return 0;
@@ -247,11 +250,13 @@ static int rxe_check_invalidate_mw(struct rxe_qp *qp, struct rxe_mw *mw)
 
 static void rxe_do_invalidate_mw(struct rxe_mw *mw)
 {
-	struct rxe_qp *qp;
 	struct rxe_mr *mr;
+	struct rxe_qp *qp = mw->qp;
+
+	spin_lock(&qp->mw_lock);
+	list_del_init(&mw->qp_list);
+	spin_unlock(&qp->mw_lock);
 
-	/* valid type 2 MW will always have a QP pointer */
-	qp = mw->qp;
 	mw->qp = NULL;
 	rxe_put(qp);
 
@@ -298,6 +303,35 @@ int rxe_invalidate_mw(struct rxe_qp *qp, u32 rkey)
 	return ret;
 }
 
+void rxe_invalidate_mws(struct rxe_qp *qp)
+{
+	struct rxe_mw *mw;
+
+	for (;;) {
+		spin_lock(&qp->mw_lock);
+		if (list_empty(&qp->mw_list)) {
+			spin_unlock(&qp->mw_lock);
+			return;
+		}
+
+		mw = list_first_entry(&qp->mw_list, struct rxe_mw, qp_list);
+		if (!rxe_get(mw)) {
+			list_del_init(&mw->qp_list);
+			spin_unlock(&qp->mw_lock);
+			continue;
+		}
+		spin_unlock(&qp->mw_lock);
+
+		spin_lock_bh(&mw->lock);
+		if (mw->qp == qp)
+			rxe_do_invalidate_mw(mw);
+		spin_unlock_bh(&mw->lock);
+
+		rxe_put(mw);
+		cond_resched();
+	}
+}
+
 struct rxe_mr *rxe_mw_get_mr(struct rxe_qp *qp, int access, u32 rkey,
 			     u64 *offset)
 {
@@ -349,6 +383,10 @@ void rxe_mw_cleanup(struct rxe_pool_elem *elem)
 	if (mw->qp) {
 		struct rxe_qp *qp = mw->qp;
 
+		spin_lock(&qp->mw_lock);
+		list_del_init(&mw->qp_list);
+		spin_unlock(&qp->mw_lock);
+
 		mw->qp = NULL;
 		rxe_put(qp);
 	}
diff --git a/drivers/infiniband/sw/rxe/rxe_qp.c b/drivers/infiniband/sw/rxe/rxe_qp.c
index 311f285d78a6b..132d224a278b2 100644
--- a/drivers/infiniband/sw/rxe/rxe_qp.c
+++ b/drivers/infiniband/sw/rxe/rxe_qp.c
@@ -220,6 +220,8 @@ static void rxe_qp_init_misc(struct rxe_dev *rxe, struct rxe_qp *qp,
 	}
 
 	spin_lock_init(&qp->state_lock);
+	spin_lock_init(&qp->mw_lock);
+	INIT_LIST_HEAD(&qp->mw_list);
 
 	spin_lock_init(&qp->sq.sq_lock);
 	spin_lock_init(&qp->rq.producer_lock);
diff --git a/drivers/infiniband/sw/rxe/rxe_verbs.c b/drivers/infiniband/sw/rxe/rxe_verbs.c
index 8553c8402c619..1f45e8bc53da5 100644
--- a/drivers/infiniband/sw/rxe/rxe_verbs.c
+++ b/drivers/infiniband/sw/rxe/rxe_verbs.c
@@ -650,6 +650,7 @@ static int rxe_query_qp(struct ib_qp *ibqp, struct ib_qp_attr *attr,
 static int rxe_destroy_qp(struct ib_qp *ibqp, struct ib_udata *udata)
 {
 	struct rxe_qp *qp = to_rqp(ibqp);
+	unsigned long flags;
 	int err;
 
 	err = rxe_qp_chk_destroy(qp);
@@ -658,6 +659,13 @@ static int rxe_destroy_qp(struct ib_qp *ibqp, struct ib_udata *udata)
 		goto err_out;
 	}
 
+	spin_lock_irqsave(&qp->state_lock, flags);
+	qp->valid = 0;
+	spin_unlock_irqrestore(&qp->state_lock, flags);
+
+	rxe_disable_task(&qp->send_task);
+	rxe_invalidate_mws(qp);
+
 	err = rxe_cleanup(qp);
 	if (err)
 		rxe_err_qp(qp, "cleanup failed, err = %d\n", err);
diff --git a/drivers/infiniband/sw/rxe/rxe_verbs.h b/drivers/infiniband/sw/rxe/rxe_verbs.h
index 0f5ffd94643f9..d6585581e9c33 100644
--- a/drivers/infiniband/sw/rxe/rxe_verbs.h
+++ b/drivers/infiniband/sw/rxe/rxe_verbs.h
@@ -288,6 +288,8 @@ struct rxe_qp {
 	struct timer_list rnr_nak_timer;
 
 	spinlock_t		state_lock; /* guard requester and completer */
+	spinlock_t		mw_lock;
+	struct list_head	mw_list;
 
 	struct execute_work	cleanup_work;
 };
@@ -387,6 +389,7 @@ struct rxe_mw {
 	int			access;
 	u64			addr;
 	u64			length;
+	struct list_head	qp_list;
 };
 
 struct rxe_mcg {
-- 
2.43.0

  parent reply	other threads:[~2026-09-28 17:18 UTC|newest]

Thread overview: 5+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-28 17:17 [PATCH v2 0/4] RDMA/rxe: Fix MW/MR lifetime races Dongliang Qin
2026-09-28 17:17 ` [PATCH v2 1/4] RDMA/rxe: Take MR reference under MW lock Dongliang Qin
2026-09-28 17:17 ` [PATCH v2 2/4] RDMA/rxe: Reserve MR state during MW binding Dongliang Qin
2026-09-28 17:17 ` Dongliang Qin [this message]
2026-09-28 17:17 ` [PATCH v2 4/4] RDMA/rxe: Do not force cleanup on pool timeout Dongliang Qin

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260928171756.3254016-4-cccccccccccc777777@gmail.com \
    --to=cccccccccccc777777@gmail.com \
    --cc=jgg@ziepe.ca \
    --cc=leon@kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-rdma@vger.kernel.org \
    --cc=rpearsonhpe@gmail.com \
    --cc=stable@vger.kernel.org \
    --cc=zyjzyj2000@gmail.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®