From: Xiubo Li via B4 Relay <devnull+xiubo.li.clyso.com@kernel.org>
To: Ilya Dryomov <idryomov@gmail.com>,
Alex Markuze <amarkuze@redhat.com>,
Viacheslav Dubeyko <slava@dubeyko.com>
Cc: ceph-devel@vger.kernel.org, linux-kernel@vger.kernel.org,
Xiubo Li <xiubo.li@clyso.com>
Subject: [PATCH v6 3/5] ceph: add wait_list_lock for wait-list serialization
Date: Sat, 29 Aug 2026 04:35:03 -0700 [thread overview]
Message-ID: <20260829-ceph-mdsc-mutex-optimization-v6-3-466936ccbd9d@clyso.com> (raw)
In-Reply-To: <20260829-ceph-mdsc-mutex-optimization-v6-0-466936ccbd9d@clyso.com>
From: Xiubo Li <xiubo.li@clyso.com>
The per-MDS session wait list and the global waiting-for-map list
are currently serialized by mdsc->mutex, even though the list
operations themselves don't need the mutex's broader protection.
Introduce a dedicated spinlock to guard these lists so that
waking and kicking waiters can run outside the mutex.
Reviewed-by: Viacheslav Dubeyko <slava@dubeyko.com>
Signed-off-by: Xiubo Li <xiubo.li@clyso.com>
---
fs/ceph/mds_client.c | 18 +++++++++++++++++-
fs/ceph/mds_client.h | 3 +++
2 files changed, 20 insertions(+), 1 deletion(-)
diff --git a/fs/ceph/mds_client.c b/fs/ceph/mds_client.c
index fdaf6f56ecd3..5983ae6e3085 100644
--- a/fs/ceph/mds_client.c
+++ b/fs/ceph/mds_client.c
@@ -3689,7 +3689,9 @@ static void __do_request(struct ceph_mds_client *mdsc,
doutc(cl, "no mdsmap, waiting for map\n");
trace_ceph_mdsc_suspend_request(mdsc, session, req,
ceph_mdsc_suspend_reason_no_mdsmap);
+ spin_lock(&mdsc->wait_list_lock);
list_add(&req->r_wait, &mdsc->waiting_for_map);
+ spin_unlock(&mdsc->wait_list_lock);
return;
}
if (!(mdsc->fsc->mount_options->flags &
@@ -3712,7 +3714,9 @@ static void __do_request(struct ceph_mds_client *mdsc,
doutc(cl, "no mds or not active, waiting for map\n");
trace_ceph_mdsc_suspend_request(mdsc, session, req,
ceph_mdsc_suspend_reason_no_active_mds);
+ spin_lock(&mdsc->wait_list_lock);
list_add(&req->r_wait, &mdsc->waiting_for_map);
+ spin_unlock(&mdsc->wait_list_lock);
return;
}
@@ -3760,9 +3764,12 @@ static void __do_request(struct ceph_mds_client *mdsc,
if (ceph_test_mount_opt(mdsc->fsc, CLEANRECOVER)) {
trace_ceph_mdsc_suspend_request(mdsc, session, req,
ceph_mdsc_suspend_reason_rejected);
+ spin_lock(&mdsc->wait_list_lock);
list_add(&req->r_wait, &mdsc->waiting_for_map);
- } else
+ spin_unlock(&mdsc->wait_list_lock);
+ } else {
err = -EACCES;
+ }
goto out_session;
}
@@ -3777,7 +3784,9 @@ static void __do_request(struct ceph_mds_client *mdsc,
}
trace_ceph_mdsc_suspend_request(mdsc, session, req,
ceph_mdsc_suspend_reason_session);
+ spin_lock(&mdsc->wait_list_lock);
list_add(&req->r_wait, &session->s_waiting);
+ spin_unlock(&mdsc->wait_list_lock);
goto out_session;
}
@@ -3870,7 +3879,9 @@ static void __wake_requests(struct ceph_mds_client *mdsc,
struct ceph_mds_request *req;
LIST_HEAD(tmp_list);
+ spin_lock(&mdsc->wait_list_lock);
list_splice_init(head, &tmp_list);
+ spin_unlock(&mdsc->wait_list_lock);
while (!list_empty(&tmp_list)) {
req = list_entry(tmp_list.next,
@@ -3903,7 +3914,9 @@ static void kick_requests(struct ceph_mds_client *mdsc, int mds)
if (req->r_session &&
req->r_session->s_mds == mds) {
doutc(cl, " kicking tid %llu\n", req->r_tid);
+ spin_lock(&mdsc->wait_list_lock);
list_del_init(&req->r_wait);
+ spin_unlock(&mdsc->wait_list_lock);
trace_ceph_mdsc_resume_request(mdsc, req);
__do_request(mdsc, req);
}
@@ -6365,6 +6378,7 @@ int ceph_mdsc_init(struct ceph_fs_client *fsc)
mdsc->snap_realms = RB_ROOT;
INIT_LIST_HEAD(&mdsc->snap_empty);
spin_lock_init(&mdsc->snap_empty_lock);
+ spin_lock_init(&mdsc->wait_list_lock);
xa_init(&mdsc->request_tree);
INIT_DELAYED_WORK(&mdsc->delayed_work, delayed_work);
mdsc->last_renew_caps = jiffies;
@@ -6447,7 +6461,9 @@ static void wait_requests(struct ceph_mds_client *mdsc)
mutex_lock(&mdsc->mutex);
while ((req = __get_oldest_req(mdsc))) {
doutc(cl, "timed out on tid %llu\n", req->r_tid);
+ spin_lock(&mdsc->wait_list_lock);
list_del_init(&req->r_wait);
+ spin_unlock(&mdsc->wait_list_lock);
__unregister_request(mdsc, req);
}
}
diff --git a/fs/ceph/mds_client.h b/fs/ceph/mds_client.h
index baba5dcf0def..0ea144ba4c7d 100644
--- a/fs/ceph/mds_client.h
+++ b/fs/ceph/mds_client.h
@@ -531,6 +531,9 @@ struct ceph_mds_client {
struct list_head snap_empty;
int num_snap_realms;
spinlock_t snap_empty_lock; /* protect snap_empty */
+ spinlock_t wait_list_lock; /* protect waiting_for_map
+ * and s_waiting lists
+ */
u64 last_tid; /* most recent mds request */
u64 oldest_tid; /* oldest incomplete mds request,
--
2.53.0
next prev parent reply other threads:[~2026-08-29 11:35 UTC|newest]
Thread overview: 6+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-29 11:35 [PATCH v6 0/5] ceph: reduce mdsc->mutex contention in the cephfs kclient Xiubo Li via B4 Relay
2026-08-29 11:35 ` [PATCH v6 1/5] ceph: use READ_ONCE/WRITE_ONCE for oldest_tid Xiubo Li via B4 Relay
2026-08-29 11:35 ` [PATCH v6 2/5] ceph: replace the request_tree rbtree with an xarray keyed by r_tid Xiubo Li via B4 Relay
2026-08-29 11:35 ` Xiubo Li via B4 Relay [this message]
2026-08-29 11:35 ` [PATCH v6 4/5] ceph: move mdsc->mutex into __do_request() Xiubo Li via B4 Relay
2026-08-29 11:35 ` [PATCH v6 5/5] ceph: narrow mdsc->mutex scope in replay_unsafe_requests Xiubo Li via B4 Relay
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260829-ceph-mdsc-mutex-optimization-v6-3-466936ccbd9d@clyso.com \
--to=devnull+xiubo.li.clyso.com@kernel.org \
--cc=amarkuze@redhat.com \
--cc=ceph-devel@vger.kernel.org \
--cc=idryomov@gmail.com \
--cc=linux-kernel@vger.kernel.org \
--cc=slava@dubeyko.com \
--cc=xiubo.li@clyso.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®