From: Lei Chen <lei.chen@smartx.com>
To: Jens Axboe <axboe@kernel.dk>
Cc: linux-block@vger.kernel.org, linux-kernel@vger.kernel.org,
Lei Chen <lei.chen@smartx.com>
Subject: [PATCH v1 2/4] blk-mq: manage driver tags at the tag set level
Date: Fri, 9 Oct 2026 11:01:09 +0800 [thread overview]
Message-ID: <20261009030111.57784-3-lei.chen@smartx.com> (raw)
In-Reply-To: <20261009030111.57784-1-lei.chen@smartx.com>
blk_mq_map_swqueue() allocates and frees driver tags while initializing
an individual request queue, even though those tags are shared by all
queues using the tag set. Which hardware queues need tags is determined
by the tag set's CPU maps, so manage their lifetime at that level.
Move tag allocation and reclamation out of blk_mq_map_swqueue(). Reclaim
unmapped tags before returning a newly allocated tag set. When updating
the hardware queue count, allocate missing tags after rebuilding the CPU
maps and before rebuilding hardware contexts. Free unused tags only after
all request queues have been remapped, while they are still frozen, since
old hardware contexts may still reference them during teardown. Serialize
these runtime changes with update_nr_hwq_lock.
Track hardware queue indices referenced by the CPU maps in a bitmap to
avoid rescanning every CPU and map for each tag being reclaimed. Keep
queue 0's tags for allocation failure fallback and rebuild the bitmap
when fallback changes the mappings.
This leaves request queue initialization to associate software and
hardware contexts without changing the shared driver tags or CPU maps.
Signed-off-by: Lei Chen <lei.chen@smartx.com>
---
block/blk-mq.c | 129 ++++++++++++++++++++++++++++++++---------
include/linux/blk-mq.h | 5 ++
2 files changed, 108 insertions(+), 26 deletions(-)
diff --git a/block/blk-mq.c b/block/blk-mq.c
index 59452b855afe..12e28d779e2c 100644
--- a/block/blk-mq.c
+++ b/block/blk-mq.c
@@ -9,6 +9,7 @@
#include <linux/module.h>
#include <linux/backing-dev.h>
#include <linux/bio.h>
+#include <linux/bitmap.h>
#include <linux/blkdev.h>
#include <linux/blk-integrity.h>
#include <linux/kmemleak.h>
@@ -4159,9 +4160,74 @@ static void __blk_mq_free_map_and_rqs(struct blk_mq_tag_set *set,
set->tags[hctx_idx] = NULL;
}
+static void blk_mq_update_mapped_tags_bitmap(struct blk_mq_tag_set *set)
+{
+ unsigned int i, cpu;
+
+ bitmap_zero(set->mapped_tags_bitmap, set->nr_hw_queues);
+ if (set->nr_hw_queues == 1) {
+ __set_bit(0, set->mapped_tags_bitmap);
+ return;
+ }
+
+ for (i = 0; i < set->nr_maps; i++) {
+ struct blk_mq_queue_map *map = &set->map[i];
+
+ if (!map->nr_queues)
+ continue;
+ for_each_possible_cpu(cpu)
+ __set_bit(map->mq_map[cpu], set->mapped_tags_bitmap);
+ }
+}
+
+static void blk_mq_alloc_mapped_tags(struct blk_mq_tag_set *set)
+{
+ unsigned int cpu, i;
+ bool remapped = false;
+
+ lockdep_assert_held_write(&set->update_nr_hwq_lock);
+
+ for_each_possible_cpu(cpu) {
+ for (i = 0; i < set->nr_maps; i++) {
+ struct blk_mq_queue_map *map = &set->map[i];
+ unsigned int hctx_idx;
+
+ if (!map->nr_queues)
+ continue;
+ hctx_idx = map->mq_map[cpu];
+ if (!set->tags[hctx_idx] &&
+ !__blk_mq_alloc_map_and_rqs(set, hctx_idx)) {
+ /* Queue 0 always has tags available for fallback. */
+ map->mq_map[cpu] = 0;
+ remapped = true;
+ }
+ }
+ }
+ if (remapped)
+ blk_mq_update_mapped_tags_bitmap(set);
+}
+
+static bool blk_mq_tagset_tags_mapped(struct blk_mq_tag_set *set,
+ unsigned int hctx_idx)
+{
+ return test_bit(hctx_idx, set->mapped_tags_bitmap);
+}
+
+/* The queue maps must be stable and the unused tags no longer in use. */
+static void blk_mq_free_unmapped_tags(struct blk_mq_tag_set *set)
+{
+ unsigned int i;
+
+ /* Keep queue 0 as a fallback if a later tag allocation fails. */
+ for (i = 1; i < set->nr_hw_queues; i++) {
+ if (set->tags[i] && !blk_mq_tagset_tags_mapped(set, i))
+ __blk_mq_free_map_and_rqs(set, i);
+ }
+}
+
static void blk_mq_map_swqueue(struct request_queue *q)
{
- unsigned int j, hctx_idx;
+ unsigned int j;
unsigned long i;
struct blk_mq_hw_ctx *hctx;
struct blk_mq_ctx *ctx;
@@ -4183,19 +4249,6 @@ static void blk_mq_map_swqueue(struct request_queue *q)
HCTX_TYPE_DEFAULT, i);
continue;
}
- hctx_idx = set->map[j].mq_map[i];
- /* unmapped hw queue can be remapped after CPU topo changed */
- if (!set->tags[hctx_idx] &&
- !__blk_mq_alloc_map_and_rqs(set, hctx_idx)) {
- /*
- * If tags initialization fail for some hctx,
- * that hctx won't be brought online. In this
- * case, remap the current ctx to hctx[0] which
- * is guaranteed to always have tags allocated
- */
- set->map[j].mq_map[i] = 0;
- }
-
hctx = blk_mq_map_queue_type(q, j, i);
ctx->hctxs[j] = hctx;
/*
@@ -4226,18 +4279,8 @@ static void blk_mq_map_swqueue(struct request_queue *q)
queue_for_each_hw_ctx(q, hctx, i) {
int cpu;
- /*
- * If no software queues are mapped to this hardware queue,
- * disable it and free the request entries.
- */
+ /* Disable hardware queues with no mapped software queues. */
if (!hctx->nr_ctx) {
- /* Never unmap queue 0. We need it as a
- * fallback in case of a new remap fails
- * allocation
- */
- if (i)
- __blk_mq_free_map_and_rqs(set, i);
-
hctx->tags = NULL;
continue;
}
@@ -4778,12 +4821,18 @@ static void blk_mq_update_queue_map(struct blk_mq_tag_set *set)
BUG_ON(set->nr_maps > 1);
blk_mq_map_queues(&set->map[HCTX_TYPE_DEFAULT]);
}
+ blk_mq_update_mapped_tags_bitmap(set);
}
+/*
+ * On successful growth, also install a larger mapped_tags_bitmap. The caller
+ * rebuilds its contents when updating the queue maps under update_nr_hwq_lock.
+ */
static struct blk_mq_tags **blk_mq_prealloc_tag_set_tags(
struct blk_mq_tag_set *set,
int new_nr_hw_queues)
{
+ unsigned long *new_mapped_tags_bitmap;
struct blk_mq_tags **new_tags;
int i;
@@ -4795,6 +4844,13 @@ static struct blk_mq_tags **blk_mq_prealloc_tag_set_tags(
if (!new_tags)
return ERR_PTR(-ENOMEM);
+ new_mapped_tags_bitmap = bitmap_zalloc_node(new_nr_hw_queues, GFP_KERNEL,
+ set->numa_node);
+ if (!new_mapped_tags_bitmap) {
+ kfree(new_tags);
+ return ERR_PTR(-ENOMEM);
+ }
+
if (set->tags)
memcpy(new_tags, set->tags, set->nr_hw_queues *
sizeof(*set->tags));
@@ -4811,12 +4867,15 @@ static struct blk_mq_tags **blk_mq_prealloc_tag_set_tags(
cond_resched();
}
+ bitmap_free(set->mapped_tags_bitmap);
+ set->mapped_tags_bitmap = new_mapped_tags_bitmap;
return new_tags;
out_unwind:
while (--i >= set->nr_hw_queues) {
if (!blk_mq_is_shared_tags(set->flags))
blk_mq_free_map_and_rqs(set, new_tags[i], i);
}
+ bitmap_free(new_mapped_tags_bitmap);
kfree(new_tags);
return ERR_PTR(-ENOMEM);
}
@@ -4880,9 +4939,17 @@ int blk_mq_alloc_tag_set(struct blk_mq_tag_set *set)
if (ret)
goto out_free_srcu;
}
+
+ set->mapped_tags_bitmap = bitmap_zalloc_node(set->nr_hw_queues, GFP_KERNEL,
+ set->numa_node);
+ if (!set->mapped_tags_bitmap) {
+ ret = -ENOMEM;
+ goto out_cleanup_srcu;
+ }
+
ret = init_srcu_struct(&set->tags_srcu);
if (ret)
- goto out_cleanup_srcu;
+ goto out_free_mapped_tags_bitmap;
init_rwsem(&set->update_nr_hwq_lock);
@@ -4907,6 +4974,7 @@ int blk_mq_alloc_tag_set(struct blk_mq_tag_set *set)
ret = blk_mq_alloc_set_map_and_rqs(set);
if (ret)
goto out_free_mq_map;
+ blk_mq_free_unmapped_tags(set);
mutex_init(&set->tag_list_lock);
INIT_LIST_HEAD(&set->tag_list);
@@ -4922,6 +4990,9 @@ int blk_mq_alloc_tag_set(struct blk_mq_tag_set *set)
set->tags = NULL;
out_cleanup_tags_srcu:
cleanup_srcu_struct(&set->tags_srcu);
+out_free_mapped_tags_bitmap:
+ bitmap_free(set->mapped_tags_bitmap);
+ set->mapped_tags_bitmap = NULL;
out_cleanup_srcu:
if (set->flags & BLK_MQ_F_BLOCKING)
cleanup_srcu_struct(set->srcu);
@@ -4968,6 +5039,9 @@ void blk_mq_free_tag_set(struct blk_mq_tag_set *set)
kfree(set->tags);
set->tags = NULL;
+ bitmap_free(set->mapped_tags_bitmap);
+ set->mapped_tags_bitmap = NULL;
+
srcu_barrier(&set->tags_srcu);
cleanup_srcu_struct(&set->tags_srcu);
if (set->flags & BLK_MQ_F_BLOCKING) {
@@ -5158,6 +5232,7 @@ static void __blk_mq_update_nr_hw_queues(struct blk_mq_tag_set *set,
fallback:
blk_mq_update_queue_map(set);
+ blk_mq_alloc_mapped_tags(set);
list_for_each_entry(q, &set->tag_list, tag_set_list) {
__blk_mq_realloc_hw_ctxs(set, q);
@@ -5174,6 +5249,8 @@ static void __blk_mq_update_nr_hw_queues(struct blk_mq_tag_set *set,
}
blk_mq_map_swqueue(q);
}
+ /* All queues have stopped using tags excluded by the new maps. */
+ blk_mq_free_unmapped_tags(set);
switch_back:
/* The blk_mq_elv_switch_back unfreezes queue for us. */
list_for_each_entry(q, &set->tag_list, tag_set_list) {
diff --git a/include/linux/blk-mq.h b/include/linux/blk-mq.h
index af878597afb8..3ef989dc4f99 100644
--- a/include/linux/blk-mq.h
+++ b/include/linux/blk-mq.h
@@ -517,6 +517,10 @@ enum hctx_type {
* tag set.
* @tags: Tag sets. One tag set per hardware queue. Has @nr_hw_queues
* elements.
+ * @mapped_tags_bitmap: Bitmap of hardware queue indices referenced by active
+ * CPU maps. Rebuilt before publication or with update_nr_hwq_lock
+ * held for writing. Tags for index 0 are retained as a fallback
+ * even when its bit is clear.
* @shared_tags:
* Shared set of tags. Has @nr_hw_queues elements. If set,
* shared by all @tags.
@@ -545,6 +549,7 @@ struct blk_mq_tag_set {
void *driver_data;
struct blk_mq_tags **tags;
+ unsigned long *mapped_tags_bitmap;
struct blk_mq_tags *shared_tags;
--
2.43.0
next prev parent reply other threads:[~2026-10-09 3:01 UTC|newest]
Thread overview: 5+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-09 3:01 [PATCH v1 0/4] Expose blk-mq tag sets through debugfs Lei Chen
2026-10-09 3:01 ` [PATCH v1 1/4] blk-mq: remove stale comment about non-present CPU mapping Lei Chen
2026-10-09 3:01 ` Lei Chen [this message]
2026-10-09 3:01 ` [PATCH v1 3/4] block: expose blk-mq tag sets through debugfs Lei Chen
2026-10-09 3:01 ` [PATCH v1 4/4] block: link request queue debugfs directories to tag sets Lei Chen
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261009030111.57784-3-lei.chen@smartx.com \
--to=lei.chen@smartx.com \
--cc=axboe@kernel.dk \
--cc=linux-block@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®