* [PATCH net-next v6 1/4] net: mana: prepare HWC ownership for safe reinitialization
2026-10-07 12:53 [PATCH net-next v6 0/4] net: mana: concurrent HWC requests and dynamic queue depth Wei Hu
@ 2026-10-07 12:53 ` Wei Hu
2026-10-07 12:53 ` [PATCH net-next v6 2/4] net: mana: give each HWC message slot its own completion state Wei Hu
` (2 subsequent siblings)
3 siblings, 0 replies; 5+ messages in thread
From: Wei Hu @ 2026-10-07 12:53 UTC (permalink / raw)
To: Long Li, Konstantin Taranov, Jakub Kicinski, David S. Miller,
Paolo Abeni, Eric Dumazet, Andrew Lunn, Jason Gunthorpe,
Leon Romanovsky, Haiyang Zhang, K. Y. Srinivasan, Wei Liu,
Dexuan Cui, Shradha Gupta, Simon Horman, Erni Sri Satya Vennela,
Stephen Hemminger, Shiraz Saleem
Cc: netdev, linux-rdma, linux-hyperv, linux-kernel, Aditya Garg,
Dipayaan Roy, Kees Cook, Kees Cook, Manish Awasthi, Wei Hu
From: Long Li <longli@microsoft.com>
Dynamic HWC queue sizing tears down the bootstrap queues and establishes a
second channel before publishing it. Prepare the existing HWC ownership and
teardown paths so that sequence cannot free or reuse state still reachable
by either the PF or an EQ handler.
The shared-memory aperture carries one request at a time. Serialize the
complete ESTABLISH_HWC and DESTROY_HWC transactions, including possession
polling and response validation. Record setup_active immediately before the
MMIO submission and clear it only after DESTROY_HWC is acknowledged.
Treat an unacknowledged destroy as retained PF ownership. Fence the HWC EQ
and withdraw its CQ from dispatch, but keep all queue memory and wrappers
for a later teardown retry. A new channel creation retries that retained
context before assigning fresh queues to the PF.
Snapshot the bootstrap CQ count and CQ ID after INIT_DONE. Publish the
fixed bound with the allocated table so later init events cannot grow the
bound past the allocation. The EQ handler acquires the table pointer but
does not take a per-event CQ reference; HWC teardown instead fences the
dedicated EQ and waits for the existing IRQ-side RCU readers before freeing
the table.
On acknowledged teardown, destroy the HWC EQ before the CQ and completion
buffer, then release the TX and RX queues. This ordering makes both the
normal reinitialization path and every failure unwind safe.
Link: https://lore.kernel.org/r/20260908035201.402424-2-longli@microsoft.com
Link: https://lore.kernel.org/r/20260908035201.402424-5-longli@microsoft.com
Link: https://lore.kernel.org/r/178910960115.219967.13830871915506436112@kernel.org
Signed-off-by: Long Li <longli@microsoft.com>
Signed-off-by: Wei Hu <weh@microsoft.com>
---
.../net/ethernet/microsoft/mana/gdma_main.c | 25 ++-
.../net/ethernet/microsoft/mana/hw_channel.c | 150 +++++++++++++++---
.../net/ethernet/microsoft/mana/shm_channel.c | 58 ++++++-
include/net/mana/gdma.h | 3 +
include/net/mana/hw_channel.h | 4 +
include/net/mana/shm_channel.h | 10 +-
6 files changed, 217 insertions(+), 33 deletions(-)
diff --git a/drivers/net/ethernet/microsoft/mana/gdma_main.c b/drivers/net/ethernet/microsoft/mana/gdma_main.c
index 8e9bfc1d6a2a..f63e236d4d19 100644
--- a/drivers/net/ethernet/microsoft/mana/gdma_main.c
+++ b/drivers/net/ethernet/microsoft/mana/gdma_main.c
@@ -911,6 +911,7 @@ static void mana_gd_process_eqe(struct gdma_queue *eq)
union gdma_eqe_info eqe_info;
enum gdma_eqe_type type;
struct gdma_event event;
+ struct gdma_queue **cq_table;
struct gdma_queue *cq;
struct gdma_eqe *eqe;
u32 cq_id;
@@ -922,11 +923,16 @@ static void mana_gd_process_eqe(struct gdma_queue *eq)
switch (type) {
case GDMA_EQE_COMPLETION:
cq_id = eqe->details[0] & 0xFFFFFF;
- if (WARN_ON_ONCE(cq_id >= gc->max_num_cqs))
+ /* The IRQ handler's RCU read-side section protects the table
+ * until HWC teardown has fenced its EQ and waited for readers.
+ */
+ cq_table = smp_load_acquire(&gc->cq_table);
+ if (!cq_table || cq_id >= READ_ONCE(gc->max_num_cqs))
break;
- cq = gc->cq_table[cq_id];
- if (WARN_ON_ONCE(!cq || cq->type != GDMA_CQ || cq->id != cq_id))
+ cq = READ_ONCE(cq_table[cq_id]);
+ if (!cq || WARN_ON_ONCE(cq->type != GDMA_CQ ||
+ cq->id != cq_id))
break;
if (cq->cq.callback)
@@ -1150,23 +1156,32 @@ int mana_gd_test_eq(struct gdma_context *gc, struct gdma_queue *eq)
return err;
}
-static void mana_gd_destroy_eq(struct gdma_context *gc, bool flush_evenets,
+static void mana_gd_destroy_eq(struct gdma_context *gc, bool flush_events,
struct gdma_queue *queue)
{
int err;
- if (flush_evenets) {
+ if (queue->eq.msix_index == INVALID_PCI_MSIX_INDEX)
+ return;
+
+ if (flush_events) {
err = mana_gd_test_eq(gc, queue);
if (err && mana_need_log(gc, err))
dev_warn(gc->dev, "Failed to flush EQ: %d\n", err);
}
mana_gd_deregister_irq(queue);
+ queue->eq.msix_index = INVALID_PCI_MSIX_INDEX;
if (queue->eq.disable_needed)
mana_gd_disable_queue(queue);
}
+void mana_gd_fence_eq(struct gdma_context *gc, struct gdma_queue *queue)
+{
+ mana_gd_destroy_eq(gc, false, queue);
+}
+
static int mana_gd_create_eq(struct gdma_dev *gd,
const struct gdma_queue_spec *spec,
bool create_hwq, struct gdma_queue *queue)
diff --git a/drivers/net/ethernet/microsoft/mana/hw_channel.c b/drivers/net/ethernet/microsoft/mana/hw_channel.c
index 3bca4b683134..b1d972968f0d 100644
--- a/drivers/net/ethernet/microsoft/mana/hw_channel.c
+++ b/drivers/net/ethernet/microsoft/mana/hw_channel.c
@@ -134,7 +134,7 @@ static void mana_hwc_init_event_handler(void *ctx, struct gdma_queue *q_self,
switch (type) {
case HWC_INIT_DATA_CQID:
- hwc->cq->gdma_cq->id = val;
+ WRITE_ONCE(hwc->hwc_init_cq_id, val);
break;
case HWC_INIT_DATA_RQID:
@@ -158,7 +158,11 @@ static void mana_hwc_init_event_handler(void *ctx, struct gdma_queue *q_self,
break;
case HWC_INIT_DATA_MAX_NUM_CQS:
- gd->gdma_context->max_num_cqs = val;
+ /* Store only; establish_channel() commits it to
+ * max_num_cqs once, so a later event cannot grow the
+ * bound past the allocation. Pairs with its READ_ONCE().
+ */
+ WRITE_ONCE(hwc->hwc_init_max_num_cqs, val);
break;
case HWC_INIT_DATA_PDID:
@@ -382,16 +386,49 @@ static void mana_hwc_comp_event(void *ctx, struct gdma_queue *q_self)
mana_gd_ring_cq(q_self, SET_ARM_BIT);
}
-static void mana_hwc_destroy_cq(struct gdma_context *gc, struct hwc_cq *hwc_cq)
+static int mana_hwc_publish_cq(struct gdma_context *gc,
+ struct gdma_queue *cq)
{
- kfree(hwc_cq->comp_buf);
+ struct gdma_queue **cq_table = READ_ONCE(gc->cq_table);
+ u32 id = READ_ONCE(cq->id);
- if (hwc_cq->gdma_cq)
- mana_gd_destroy_queue(gc, hwc_cq->gdma_cq);
+ if (!cq_table || id >= READ_ONCE(gc->max_num_cqs) ||
+ READ_ONCE(cq_table[id]))
+ return -EINVAL;
+
+ WRITE_ONCE(cq_table[id], cq);
+
+ return 0;
+}
+
+static void mana_hwc_unpublish_cq(struct gdma_context *gc,
+ struct gdma_queue *cq)
+{
+ struct gdma_queue **cq_table = READ_ONCE(gc->cq_table);
+ u32 id;
+
+ if (!cq_table || !cq)
+ return;
+ id = READ_ONCE(cq->id);
+ if (id < READ_ONCE(gc->max_num_cqs) &&
+ READ_ONCE(cq_table[id]) == cq)
+ WRITE_ONCE(cq_table[id], NULL);
+}
+
+static void mana_hwc_destroy_cq(struct gdma_context *gc, struct hwc_cq *hwc_cq)
+{
+ /* Destroy the EQ first: it deregisters the IRQ and drains in-flight
+ * handlers, so none can touch the CQ after it is freed.
+ */
if (hwc_cq->gdma_eq)
mana_gd_destroy_queue(gc, hwc_cq->gdma_eq);
+ /* Safe to free now that the EQ handler is fenced. */
+ if (hwc_cq->gdma_cq)
+ mana_gd_destroy_queue(gc, hwc_cq->gdma_cq);
+
+ kfree(hwc_cq->comp_buf);
kfree(hwc_cq);
}
@@ -674,6 +711,9 @@ static int mana_hwc_establish_channel(struct gdma_context *gc, u16 *q_depth,
struct gdma_queue *sq = hwc->txq->gdma_wq;
struct gdma_queue *eq = hwc->cq->gdma_eq;
struct gdma_queue *cq = hwc->cq->gdma_cq;
+ struct gdma_queue **cq_table;
+ u32 num_cqs;
+ u32 cq_id;
int err;
init_completion(&hwc->hwc_init_eqe_comp);
@@ -683,7 +723,7 @@ static int mana_hwc_establish_channel(struct gdma_context *gc, u16 *q_depth,
cq->mem_info.dma_handle,
rq->mem_info.dma_handle,
sq->mem_info.dma_handle,
- eq->eq.msix_index);
+ eq->eq.msix_index, &hwc->setup_active);
if (err)
return err;
@@ -694,15 +734,45 @@ static int mana_hwc_establish_channel(struct gdma_context *gc, u16 *q_depth,
*max_req_msg_size = hwc->hwc_init_max_req_msg_size;
*max_resp_msg_size = hwc->hwc_init_max_resp_msg_size;
- /* Both were set in mana_hwc_init_event_handler(). */
- if (WARN_ON(cq->id >= gc->max_num_cqs))
+ /* Snapshot the device-reported count and id once, so the same value
+ * sizes, bounds and indexes cq_table even across the sleeping
+ * vcalloc() and a concurrent init event.
+ */
+ num_cqs = READ_ONCE(hwc->hwc_init_max_num_cqs);
+ cq_id = READ_ONCE(hwc->hwc_init_cq_id);
+
+ /* Both operands come from untrusted HWC bootstrap events; a missing
+ * MAX_NUM_CQS leaves num_cqs at 0. Reject rather than WARN_ON() so a
+ * malformed device response cannot panic a panic_on_warn guest.
+ */
+ if (cq_id >= num_cqs) {
+ dev_err_ratelimited(hwc->dev,
+ "HWC: bad CQ id %u >= max %u\n",
+ cq_id, num_cqs);
return -EPROTO;
+ }
- gc->cq_table = vcalloc(gc->max_num_cqs, sizeof(struct gdma_queue *));
- if (!gc->cq_table)
+ /* Init events remain enabled, so commit the validated CQ ID once. */
+ WRITE_ONCE(cq->id, cq_id);
+
+ cq_table = vcalloc(num_cqs, sizeof(*cq_table));
+ if (!cq_table)
return -ENOMEM;
- gc->cq_table[cq->id] = cq;
+ /* Publish the bound and the initialised table together; the release
+ * pairs with smp_load_acquire() in mana_gd_process_eqe().
+ */
+ WRITE_ONCE(gc->max_num_cqs, num_cqs);
+ /* Pairs with smp_load_acquire() in mana_gd_process_eqe(). */
+ smp_store_release(&gc->cq_table, cq_table);
+
+ err = mana_hwc_publish_cq(gc, cq);
+ if (err) {
+ dev_err_ratelimited(hwc->dev,
+ "HWC: failed to publish CQ %u: %d\n",
+ cq_id, err);
+ return err;
+ }
return 0;
}
@@ -759,6 +829,13 @@ int mana_hwc_create_channel(struct gdma_context *gc)
u16 q_depth_max;
int err;
+ /* Retry a retained context before assigning queues to the PF again. */
+ if (gd->driver_data) {
+ mana_hwc_destroy_channel(gc);
+ if (gd->driver_data)
+ return -ETIMEDOUT;
+ }
+
hwc = kzalloc_obj(*hwc);
if (!hwc)
return -ENOMEM;
@@ -808,30 +885,56 @@ int mana_hwc_create_channel(struct gdma_context *gc)
return err;
}
+static void mana_hwc_fence_channel(struct gdma_context *gc,
+ struct hw_channel_context *hwc)
+{
+ if (!hwc->cq)
+ return;
+
+ if (hwc->cq->gdma_eq)
+ mana_gd_fence_eq(gc, hwc->cq->gdma_eq);
+
+ if (hwc->cq->gdma_cq)
+ mana_hwc_unpublish_cq(gc, hwc->cq->gdma_cq);
+}
+
void mana_hwc_destroy_channel(struct gdma_context *gc)
{
struct hw_channel_context *hwc = gc->hwc.driver_data;
+ struct gdma_queue **old_cq_table;
+ int err;
if (!hwc)
return;
- /* gc->max_num_cqs is set in mana_hwc_init_event_handler(). If it's
- * non-zero, the HWC worked and we should tear down the HWC here.
+ /* An unacknowledged destroy leaves the PF's mappings live. Fence
+ * software dispatch, but retain every PF-visible allocation.
*/
- if (gc->max_num_cqs > 0) {
- mana_smc_teardown_hwc(&gc->shm_channel, false);
- gc->max_num_cqs = 0;
+ err = mana_smc_teardown_hwc(&gc->shm_channel, false,
+ &hwc->setup_active);
+ if (err) {
+ dev_err(hwc->dev,
+ "HWC teardown failed: %d, retaining PF-visible resources\n",
+ err);
+ mana_hwc_fence_channel(gc, hwc);
+ return;
}
+ /* Fence the EQ before releasing any state its handlers can reach. */
+ if (hwc->cq)
+ mana_hwc_destroy_cq(hwc->gdma_dev->gdma_context, hwc->cq);
+
+ /* Reset only after mana_hwc_destroy_cq() has cleared the CQ table
+ * slot, so it is not left dangling.
+ */
+ WRITE_ONCE(gc->max_num_cqs, 0);
+
if (hwc->txq)
mana_hwc_destroy_wq(hwc, hwc->txq);
if (hwc->rxq)
mana_hwc_destroy_wq(hwc, hwc->rxq);
- if (hwc->cq)
- mana_hwc_destroy_cq(hwc->gdma_dev->gdma_context, hwc->cq);
-
kfree(hwc->caller_ctx);
hwc->caller_ctx = NULL;
@@ -848,8 +951,11 @@ void mana_hwc_destroy_channel(struct gdma_context *gc)
gc->hwc.driver_data = NULL;
gc->hwc.gdma_context = NULL;
- vfree(gc->cq_table);
- gc->cq_table = NULL;
+ old_cq_table = READ_ONCE(gc->cq_table);
+ /* Stop new table readers before waiting for existing IRQ readers. */
+ smp_store_release(&gc->cq_table, NULL);
+ synchronize_rcu();
+ vfree(old_cq_table);
}
int mana_hwc_send_request(struct hw_channel_context *hwc, u32 req_len,
diff --git a/drivers/net/ethernet/microsoft/mana/shm_channel.c b/drivers/net/ethernet/microsoft/mana/shm_channel.c
index d21b5db06e50..df9d05062dd5 100644
--- a/drivers/net/ethernet/microsoft/mana/shm_channel.c
+++ b/drivers/net/ethernet/microsoft/mana/shm_channel.c
@@ -125,13 +125,24 @@ static int mana_smc_read_response(struct shm_channel *sc, u32 msg_type,
void mana_smc_init(struct shm_channel *sc, struct device *dev,
void __iomem *base)
{
+ /* The first call precedes channel publication. Later calls refresh the
+ * BAR mapping after reset or resume without reinitializing a live lock.
+ */
+ if (!sc->transaction_lock_initialized) {
+ mutex_init(&sc->transaction_lock);
+ sc->transaction_lock_initialized = true;
+ }
+
+ mutex_lock(&sc->transaction_lock);
sc->dev = dev;
sc->base = base;
+ mutex_unlock(&sc->transaction_lock);
}
-int mana_smc_setup_hwc(struct shm_channel *sc, bool reset_vf, u64 eq_addr,
- u64 cq_addr, u64 rq_addr, u64 sq_addr,
- u32 eq_msix_index)
+static int mana_smc_setup_hwc_locked(struct shm_channel *sc, bool reset_vf,
+ u64 eq_addr, u64 cq_addr, u64 rq_addr,
+ u64 sq_addr, u32 eq_msix_index,
+ bool *submitted)
{
union smc_proto_hdr *hdr;
u16 all_addr_h4bits = 0;
@@ -144,6 +155,9 @@ int mana_smc_setup_hwc(struct shm_channel *sc, bool reset_vf, u64 eq_addr,
int err;
int i;
+ lockdep_assert_held(&sc->transaction_lock);
+ *submitted = false;
+
/* Ensure VF already has possession of shared memory */
err = mana_smc_poll_register(sc->base, false);
if (err) {
@@ -229,6 +243,7 @@ int mana_smc_setup_hwc(struct shm_channel *sc, bool reset_vf, u64 eq_addr,
/* Write 256-message buffer to shared memory (final 32-bit write
* triggers HW to set possession bit to PF).
*/
+ *submitted = true;
dword = (u32 *)shm_buf;
for (i = 0; i < SMC_APERTURE_DWORDS; i++)
writel(*dword++, sc->base + i * SMC_BASIC_UNIT);
@@ -248,11 +263,32 @@ int mana_smc_setup_hwc(struct shm_channel *sc, bool reset_vf, u64 eq_addr,
return 0;
}
-int mana_smc_teardown_hwc(struct shm_channel *sc, bool reset_vf)
+int mana_smc_setup_hwc(struct shm_channel *sc, bool reset_vf, u64 eq_addr,
+ u64 cq_addr, u64 rq_addr, u64 sq_addr,
+ u32 eq_msix_index, bool *submitted)
+{
+ int err;
+
+ mutex_lock(&sc->transaction_lock);
+ err = mana_smc_setup_hwc_locked(sc, reset_vf, eq_addr, cq_addr,
+ rq_addr, sq_addr, eq_msix_index,
+ submitted);
+ mutex_unlock(&sc->transaction_lock);
+
+ return err;
+}
+
+static int mana_smc_teardown_hwc_locked(struct shm_channel *sc, bool reset_vf,
+ bool *setup_active)
{
union smc_proto_hdr hdr = {};
int err;
+ lockdep_assert_held(&sc->transaction_lock);
+
+ if (!*setup_active)
+ return 0;
+
/* Ensure already has possession of shared memory */
err = mana_smc_poll_register(sc->base, false);
if (err) {
@@ -283,5 +319,19 @@ int mana_smc_teardown_hwc(struct shm_channel *sc, bool reset_vf)
return err;
}
+ *setup_active = false;
+
return 0;
}
+
+int mana_smc_teardown_hwc(struct shm_channel *sc, bool reset_vf,
+ bool *setup_active)
+{
+ int err;
+
+ mutex_lock(&sc->transaction_lock);
+ err = mana_smc_teardown_hwc_locked(sc, reset_vf, setup_active);
+ mutex_unlock(&sc->transaction_lock);
+
+ return err;
+}
diff --git a/include/net/mana/gdma.h b/include/net/mana/gdma.h
index 308950f9b54b..f8278e3817c8 100644
--- a/include/net/mana/gdma.h
+++ b/include/net/mana/gdma.h
@@ -520,6 +520,9 @@ int mana_gd_create_mana_wq_cq(struct gdma_dev *gd,
void mana_gd_destroy_queue(struct gdma_context *gc, struct gdma_queue *queue);
+/* Remove an EQ from interrupt dispatch without freeing its queue memory. */
+void mana_gd_fence_eq(struct gdma_context *gc, struct gdma_queue *queue);
+
int mana_gd_poll_cq(struct gdma_queue *cq, struct gdma_comp *comp, int num_cqe);
void mana_gd_ring_cq(struct gdma_queue *cq, u8 arm_bit);
diff --git a/include/net/mana/hw_channel.h b/include/net/mana/hw_channel.h
index 1ab2d66c891e..eb1c005bc091 100644
--- a/include/net/mana/hw_channel.h
+++ b/include/net/mana/hw_channel.h
@@ -185,6 +185,8 @@ struct hw_channel_context {
u16 hwc_init_q_depth_max;
u32 hwc_init_max_req_msg_size;
u32 hwc_init_max_resp_msg_size;
+ u32 hwc_init_max_num_cqs;
+ u32 hwc_init_cq_id;
struct completion hwc_init_eqe_comp;
@@ -199,6 +201,8 @@ struct hw_channel_context {
u32 dest_vrcq_id;
u32 hwc_timeout;
+ /* The PF may own the HWC queues while this is true. */
+ bool setup_active;
struct hwc_caller_ctx *caller_ctx;
};
diff --git a/include/net/mana/shm_channel.h b/include/net/mana/shm_channel.h
index dbabcfb95daf..c8daa61fa759 100644
--- a/include/net/mana/shm_channel.h
+++ b/include/net/mana/shm_channel.h
@@ -4,6 +4,8 @@
#ifndef _SHM_CHANNEL_H
#define _SHM_CHANNEL_H
+#include <linux/mutex.h>
+
#define SMC_APERTURE_BITS 256
#define SMC_BASIC_UNIT (sizeof(u32))
#define SMC_APERTURE_DWORDS (SMC_APERTURE_BITS / (SMC_BASIC_UNIT * 8))
@@ -13,6 +15,9 @@
struct shm_channel {
struct device *dev;
void __iomem *base;
+ /* Protects base and complete aperture request/response transactions. */
+ struct mutex transaction_lock;
+ bool transaction_lock_initialized;
};
void mana_smc_init(struct shm_channel *sc, struct device *dev,
@@ -20,8 +25,9 @@ void mana_smc_init(struct shm_channel *sc, struct device *dev,
int mana_smc_setup_hwc(struct shm_channel *sc, bool reset_vf, u64 eq_addr,
u64 cq_addr, u64 rq_addr, u64 sq_addr,
- u32 eq_msix_index);
+ u32 eq_msix_index, bool *submitted);
-int mana_smc_teardown_hwc(struct shm_channel *sc, bool reset_vf);
+int mana_smc_teardown_hwc(struct shm_channel *sc, bool reset_vf,
+ bool *setup_active);
#endif /* _SHM_CHANNEL_H */
--
2.43.0
^ permalink raw reply [flat|nested] 5+ messages in thread* [PATCH net-next v6 2/4] net: mana: give each HWC message slot its own completion state
2026-10-07 12:53 [PATCH net-next v6 0/4] net: mana: concurrent HWC requests and dynamic queue depth Wei Hu
2026-10-07 12:53 ` [PATCH net-next v6 1/4] net: mana: prepare HWC ownership for safe reinitialization Wei Hu
@ 2026-10-07 12:53 ` Wei Hu
2026-10-07 12:53 ` [PATCH net-next v6 3/4] net: mana: support concurrent HWC requests Wei Hu
2026-10-07 12:53 ` [PATCH net-next v6 4/4] net: mana: add dynamic HWC queue depth with reinit path Wei Hu
3 siblings, 0 replies; 5+ messages in thread
From: Wei Hu @ 2026-10-07 12:53 UTC (permalink / raw)
To: Long Li, Konstantin Taranov, Jakub Kicinski, David S. Miller,
Paolo Abeni, Eric Dumazet, Andrew Lunn, Jason Gunthorpe,
Leon Romanovsky, Haiyang Zhang, K. Y. Srinivasan, Wei Liu,
Dexuan Cui, Shradha Gupta, Simon Horman, Erni Sri Satya Vennela,
Stephen Hemminger, Shiraz Saleem
Cc: netdev, linux-rdma, linux-hyperv, linux-kernel, Aditya Garg,
Dipayaan Roy, Kees Cook, Kees Cook, Manish Awasthi, Wei Hu
From: Long Li <longli@microsoft.com>
Concurrent requests require the response handler and the waiting sender to
agree on the lifetime and result state of each message slot.
Protect the caller's output buffer, Linux error and device status with a
per-slot lock. Publish the output buffer only while the sender is waiting
and withdraw it before every return. A late response therefore cannot copy
through a pointer to caller storage whose lifetime has ended.
Reinitialize completion and result state when the slot is acquired. If
the response records its result while the timed wait expires, return that
result instead of reporting a false timeout.
HWC responses contain only the slot ID. After a genuine timeout, prevent
the serialized bootstrap channel from reusing that ID until teardown. This
keeps a stale response from completing a later request before the next
patch replaces the depth-one latch with per-slot sender and response
ownership.
Split response-status handling and request cleanup into helpers so the send
path has structured returns rather than phase-jumping gotos.
Link: https://lore.kernel.org/r/20260908035201.402424-3-longli@microsoft.com
Link: https://lore.kernel.org/r/20260908035201.402424-4-longli@microsoft.com
Signed-off-by: Long Li <longli@microsoft.com>
Signed-off-by: Wei Hu <weh@microsoft.com>
---
.../net/ethernet/microsoft/mana/hw_channel.c | 134 +++++++++++++-----
include/net/mana/hw_channel.h | 7 +-
2 files changed, 106 insertions(+), 35 deletions(-)
diff --git a/drivers/net/ethernet/microsoft/mana/hw_channel.c b/drivers/net/ethernet/microsoft/mana/hw_channel.c
index b1d972968f0d..89b6e30863da 100644
--- a/drivers/net/ethernet/microsoft/mana/hw_channel.c
+++ b/drivers/net/ethernet/microsoft/mana/hw_channel.c
@@ -9,6 +9,7 @@
static int mana_hwc_get_msg_index(struct hw_channel_context *hwc, u16 *msg_id)
{
struct gdma_resource *r = &hwc->inflight_msg_res;
+ struct hwc_caller_ctx *ctx;
unsigned long flags;
u32 index;
@@ -16,9 +17,22 @@ static int mana_hwc_get_msg_index(struct hw_channel_context *hwc, u16 *msg_id)
spin_lock_irqsave(&r->lock, flags);
+ if (hwc->hwc_timed_out) {
+ spin_unlock_irqrestore(&r->lock, flags);
+ up(&hwc->sema);
+ return -ETIMEDOUT;
+ }
+
index = find_first_zero_bit(hwc->inflight_msg_res.map,
hwc->inflight_msg_res.size);
+ ctx = &hwc->caller_ctx[index];
+ reinit_completion(&ctx->comp_event);
+ ctx->output_buf = NULL;
+ ctx->output_buflen = 0;
+ ctx->error = -EINPROGRESS;
+ ctx->status_code = 0;
+
bitmap_set(hwc->inflight_msg_res.map, index, 1);
spin_unlock_irqrestore(&r->lock, flags);
@@ -28,12 +42,15 @@ static int mana_hwc_get_msg_index(struct hw_channel_context *hwc, u16 *msg_id)
return 0;
}
-static void mana_hwc_put_msg_index(struct hw_channel_context *hwc, u16 msg_id)
+static void mana_hwc_put_msg_index(struct hw_channel_context *hwc, u16 msg_id,
+ bool timed_out)
{
struct gdma_resource *r = &hwc->inflight_msg_res;
unsigned long flags;
spin_lock_irqsave(&r->lock, flags);
+ if (timed_out)
+ hwc->hwc_timed_out = true;
bitmap_clear(hwc->inflight_msg_res.map, msg_id, 1);
spin_unlock_irqrestore(&r->lock, flags);
@@ -90,14 +107,21 @@ static void mana_hwc_handle_resp(struct hw_channel_context *hwc, u32 resp_len,
}
ctx = hwc->caller_ctx + msg_id;
- err = mana_hwc_verify_resp_msg(ctx, resp_msg, resp_len);
- if (err)
- goto out;
- ctx->status_code = resp_msg->status;
+ spin_lock(&ctx->lock);
+ if (!ctx->output_buf) {
+ spin_unlock(&ctx->lock);
+ mana_hwc_post_rx_wqe(hwc->rxq, rx_req);
+ return;
+ }
- memcpy(ctx->output_buf, resp_msg, resp_len);
-out:
+ err = mana_hwc_verify_resp_msg(ctx, resp_msg, resp_len);
+ if (!err) {
+ ctx->status_code = resp_msg->status;
+ memcpy(ctx->output_buf, resp_msg, resp_len);
+ }
+
+ ctx->output_buf = NULL;
ctx->error = err;
/* Must post rx wqe before complete(), otherwise the next rx may
@@ -106,6 +130,7 @@ static void mana_hwc_handle_resp(struct hw_channel_context *hwc, u32 resp_len,
mana_hwc_post_rx_wqe(hwc->rxq, rx_req);
complete(&ctx->comp_event);
+ spin_unlock(&ctx->lock);
}
static void mana_hwc_init_event_handler(void *ctx, struct gdma_queue *q_self,
@@ -694,8 +719,10 @@ static int mana_hwc_test_channel(struct hw_channel_context *hwc, u16 q_depth,
if (!ctx)
return -ENOMEM;
- for (i = 0; i < q_depth; ++i)
+ for (i = 0; i < q_depth; ++i) {
init_completion(&ctx[i].comp_event);
+ spin_lock_init(&ctx[i].lock);
+ }
hwc->caller_ctx = ctx;
@@ -958,6 +985,38 @@ void mana_hwc_destroy_channel(struct gdma_context *gc)
vfree(old_cq_table);
}
+static int mana_hwc_response_status(struct hw_channel_context *hwc,
+ u32 command, int error, u32 status)
+{
+ if (error)
+ return error;
+
+ if (!status || status == GDMA_STATUS_MORE_ENTRIES)
+ return 0;
+
+ if (status == GDMA_STATUS_CMD_UNSUPPORTED)
+ return -EOPNOTSUPP;
+
+ if (command != MANA_QUERY_PHY_STAT)
+ dev_err(hwc->dev, "Command 0x%x failed with status: 0x%x\n",
+ command, status);
+
+ return -EPROTO;
+}
+
+static void mana_hwc_finish_request(struct hw_channel_context *hwc,
+ struct hwc_caller_ctx *ctx, u16 msg_id,
+ bool timed_out)
+{
+ unsigned long flags;
+
+ spin_lock_irqsave(&ctx->lock, flags);
+ ctx->output_buf = NULL;
+ spin_unlock_irqrestore(&ctx->lock, flags);
+
+ mana_hwc_put_msg_index(hwc, msg_id, timed_out);
+}
+
int mana_hwc_send_request(struct hw_channel_context *hwc, u32 req_len,
const void *req, u32 resp_len, void *resp)
{
@@ -965,26 +1024,32 @@ int mana_hwc_send_request(struct hw_channel_context *hwc, u32 req_len,
struct hwc_wq *txq = hwc->txq;
struct gdma_req_hdr *req_msg;
struct hwc_caller_ctx *ctx;
+ unsigned long flags;
u32 dest_vrcq;
u32 dest_vrq;
u32 command;
+ u32 status;
u16 msg_id;
int err;
- mana_hwc_get_msg_index(hwc, &msg_id);
+ err = mana_hwc_get_msg_index(hwc, &msg_id);
+ if (err)
+ return err;
tx_wr = &txq->msg_buf->reqs[msg_id];
+ ctx = hwc->caller_ctx + msg_id;
if (req_len > tx_wr->buf_len) {
dev_err(hwc->dev, "HWC: req msg size: %d > %d\n", req_len,
tx_wr->buf_len);
- err = -EINVAL;
- goto out;
+ mana_hwc_finish_request(hwc, ctx, msg_id, false);
+ return -EINVAL;
}
- ctx = hwc->caller_ctx + msg_id;
+ spin_lock_irqsave(&ctx->lock, flags);
ctx->output_buf = resp;
ctx->output_buflen = resp_len;
+ spin_unlock_irqrestore(&ctx->lock, flags);
req_msg = (struct gdma_req_hdr *)tx_wr->buf_va;
if (req)
@@ -1006,11 +1071,23 @@ int mana_hwc_send_request(struct hw_channel_context *hwc, u32 req_len,
err = mana_hwc_post_tx_wqe(txq, tx_wr, dest_vrq, dest_vrcq, false);
if (err) {
dev_err(hwc->dev, "HWC: Failed to post send WQE: %d\n", err);
- goto out;
+ mana_hwc_finish_request(hwc, ctx, msg_id, false);
+ return err;
}
if (!wait_for_completion_timeout(&ctx->comp_event,
- (msecs_to_jiffies(hwc->hwc_timeout)))) {
+ msecs_to_jiffies(hwc->hwc_timeout))) {
+ spin_lock_irqsave(&ctx->lock, flags);
+ ctx->output_buf = NULL;
+ err = ctx->error;
+ status = ctx->status_code;
+ spin_unlock_irqrestore(&ctx->lock, flags);
+
+ if (err != -EINPROGRESS) {
+ mana_hwc_finish_request(hwc, ctx, msg_id, false);
+ return mana_hwc_response_status(hwc, command, err, status);
+ }
+
if (hwc->hwc_timeout != 0)
dev_err(hwc->dev, "Command 0x%x timed out: %u ms\n",
command, hwc->hwc_timeout);
@@ -1019,27 +1096,16 @@ int mana_hwc_send_request(struct hw_channel_context *hwc, u32 req_len,
if (hwc->hwc_timeout > 1)
hwc->hwc_timeout = 1;
- err = -ETIMEDOUT;
- goto out;
+ mana_hwc_finish_request(hwc, ctx, msg_id, true);
+ return -ETIMEDOUT;
}
- if (ctx->error) {
- err = ctx->error;
- goto out;
- }
+ spin_lock_irqsave(&ctx->lock, flags);
+ ctx->output_buf = NULL;
+ err = ctx->error;
+ status = ctx->status_code;
+ spin_unlock_irqrestore(&ctx->lock, flags);
- if (ctx->status_code && ctx->status_code != GDMA_STATUS_MORE_ENTRIES) {
- if (ctx->status_code == GDMA_STATUS_CMD_UNSUPPORTED) {
- err = -EOPNOTSUPP;
- goto out;
- }
- if (command != MANA_QUERY_PHY_STAT)
- dev_err(hwc->dev, "Command 0x%x failed with status: 0x%x\n",
- command, ctx->status_code);
- err = -EPROTO;
- goto out;
- }
-out:
- mana_hwc_put_msg_index(hwc, msg_id);
- return err;
+ mana_hwc_finish_request(hwc, ctx, msg_id, false);
+ return mana_hwc_response_status(hwc, command, err, status);
}
diff --git a/include/net/mana/hw_channel.h b/include/net/mana/hw_channel.h
index eb1c005bc091..e733301f8740 100644
--- a/include/net/mana/hw_channel.h
+++ b/include/net/mana/hw_channel.h
@@ -168,10 +168,12 @@ struct hwc_wq {
struct hwc_caller_ctx {
struct completion comp_event;
+ /* Protects the output buffer and response state from timeout. */
+ spinlock_t lock;
void *output_buf;
u32 output_buflen;
- u32 error; /* Linux error code */
+ int error; /* Linux error code */
u32 status_code;
};
@@ -201,6 +203,9 @@ struct hw_channel_context {
u32 dest_vrcq_id;
u32 hwc_timeout;
+ /* Prevents message ID reuse after a timeout; protected by the map lock. */
+ bool hwc_timed_out;
+
/* The PF may own the HWC queues while this is true. */
bool setup_active;
struct hwc_caller_ctx *caller_ctx;
--
2.43.0
^ permalink raw reply [flat|nested] 5+ messages in thread* [PATCH net-next v6 3/4] net: mana: support concurrent HWC requests
2026-10-07 12:53 [PATCH net-next v6 0/4] net: mana: concurrent HWC requests and dynamic queue depth Wei Hu
2026-10-07 12:53 ` [PATCH net-next v6 1/4] net: mana: prepare HWC ownership for safe reinitialization Wei Hu
2026-10-07 12:53 ` [PATCH net-next v6 2/4] net: mana: give each HWC message slot its own completion state Wei Hu
@ 2026-10-07 12:53 ` Wei Hu
2026-10-07 12:53 ` [PATCH net-next v6 4/4] net: mana: add dynamic HWC queue depth with reinit path Wei Hu
3 siblings, 0 replies; 5+ messages in thread
From: Wei Hu @ 2026-10-07 12:53 UTC (permalink / raw)
To: Long Li, Konstantin Taranov, Jakub Kicinski, David S. Miller,
Paolo Abeni, Eric Dumazet, Andrew Lunn, Jason Gunthorpe,
Leon Romanovsky, Haiyang Zhang, K. Y. Srinivasan, Wei Liu,
Dexuan Cui, Shradha Gupta, Simon Horman, Erni Sri Satya Vennela,
Stephen Hemminger, Shiraz Saleem
Cc: netdev, linux-rdma, linux-hyperv, linux-kernel, Aditya Garg,
Dipayaan Roy, Kees Cook, Kees Cook, Manish Awasthi, Wei Hu
From: Long Li <longli@microsoft.com>
Replace the depth-one timeout latch with per-slot sender and response
ownership. Initialize each slot under the inflight resource lock, publish
its bitmap bit last, and release it only after both sides have finished.
This preserves response-buffer withdrawal and timeout quarantine while
allowing independent message IDs to make progress.
Serialize SQ posting so concurrent senders cannot corrupt the producer
state. A posted request which times out keeps its response reference and
admission permit until its late response arrives or teardown cancels it.
Return -EBUSY when all slots remain occupied. Admission pressure is not a
response timeout and must not make callers start dead-channel recovery.
Synchronize hwc_timeout updates with atomic access helpers. Teardown
cancellation is terminal for a channel instance, firmware and query
updates may replace only a live timeout, and fail-fast reduction may only
lower a live value to one millisecond.
Keep setup private until the queues, admission state, caller contexts and
direct EQ test are ready. Runtime teardown unpublishes the channel, rejects
new admissions, wakes waiters and drains active senders before releasing or
retaining PF-owned mappings.
Split admission, submission, waiting, response status and unwind into
helpers so mana_hwc_send_request() uses structured returns instead of
phase-jumping gotos.
Link: https://lore.kernel.org/r/20260908035201.402424-4-longli@microsoft.com
Link: https://lore.kernel.org/r/20260914175044.2a26bb46@kernel.org
Signed-off-by: Long Li <longli@microsoft.com>
Signed-off-by: Wei Hu <weh@microsoft.com>
---
.../net/ethernet/microsoft/mana/gdma_main.c | 80 ++-
.../net/ethernet/microsoft/mana/hw_channel.c | 578 +++++++++++++-----
include/net/mana/gdma.h | 9 +
include/net/mana/hw_channel.h | 43 +-
4 files changed, 527 insertions(+), 183 deletions(-)
diff --git a/drivers/net/ethernet/microsoft/mana/gdma_main.c b/drivers/net/ethernet/microsoft/mana/gdma_main.c
index f63e236d4d19..bf4c528f8c75 100644
--- a/drivers/net/ethernet/microsoft/mana/gdma_main.c
+++ b/drivers/net/ethernet/microsoft/mana/gdma_main.c
@@ -162,6 +162,8 @@ static int mana_gd_init_registers(struct pci_dev *pdev)
bool mana_need_log(struct gdma_context *gc, int err)
{
struct hw_channel_context *hwc;
+ bool need_log = true;
+ unsigned long flags;
if (err != -ETIMEDOUT)
return true;
@@ -169,11 +171,13 @@ bool mana_need_log(struct gdma_context *gc, int err)
if (!gc)
return true;
+ spin_lock_irqsave(&gc->hwc_lock, flags);
hwc = gc->hwc.driver_data;
- if (hwc && hwc->hwc_timeout == 0)
- return false;
+ if (hwc && !mana_hwc_timeout_read(hwc))
+ need_log = false;
+ spin_unlock_irqrestore(&gc->hwc_lock, flags);
- return true;
+ return need_log;
}
static int mana_gd_query_max_resources(struct pci_dev *pdev)
@@ -317,7 +321,8 @@ static int mana_gd_query_max_resources(struct pci_dev *pdev)
return 0;
}
-static int mana_gd_query_hwc_timeout(struct pci_dev *pdev, u32 *timeout_val)
+static int mana_gd_query_hwc_timeout(struct pci_dev *pdev, u32 timeout_ms,
+ u32 *new_timeout_ms)
{
struct gdma_context *gc = pci_get_drvdata(pdev);
struct gdma_query_hwc_timeout_resp resp = {};
@@ -326,12 +331,12 @@ static int mana_gd_query_hwc_timeout(struct pci_dev *pdev, u32 *timeout_val)
mana_gd_init_req_hdr(&req.hdr, GDMA_QUERY_HWC_TIMEOUT,
sizeof(req), sizeof(resp));
- req.timeout_ms = *timeout_val;
+ req.timeout_ms = timeout_ms;
err = mana_gd_send_request(gc, sizeof(req), &req, sizeof(resp), &resp);
if (err || resp.hdr.status)
return err ? err : -EPROTO;
- *timeout_val = resp.timeout_ms;
+ *new_timeout_ms = resp.timeout_ms;
return 0;
}
@@ -387,9 +392,27 @@ static int mana_gd_detect_devices(struct pci_dev *pdev)
int mana_gd_send_request(struct gdma_context *gc, u32 req_len, const void *req,
u32 resp_len, void *resp)
{
- struct hw_channel_context *hwc = gc->hwc.driver_data;
+ struct hw_channel_context *hwc;
+ unsigned long flags;
+ int err;
+
+ spin_lock_irqsave(&gc->hwc_lock, flags);
+ hwc = gc->hwc.driver_data;
+ if (!hwc) {
+ spin_unlock_irqrestore(&gc->hwc_lock, flags);
+ return -ENODEV;
+ }
+ hwc->active_senders++;
+ spin_unlock_irqrestore(&gc->hwc_lock, flags);
- return mana_hwc_send_request(hwc, req_len, req, resp_len, resp);
+ err = mana_hwc_send_request(hwc, req_len, req, resp_len, resp);
+
+ spin_lock_irqsave(&gc->hwc_lock, flags);
+ if (--hwc->active_senders == 0)
+ wake_up(&gc->hwc_drain_waitq);
+ spin_unlock_irqrestore(&gc->hwc_lock, flags);
+
+ return err;
}
EXPORT_SYMBOL_NS(mana_gd_send_request, "NET_MANA");
@@ -710,6 +733,7 @@ static void mana_serv_reset(struct pci_dev *pdev)
{
struct gdma_context *gc = pci_get_drvdata(pdev);
struct hw_channel_context *hwc;
+ unsigned long flags;
int ret;
if (!gc) {
@@ -719,14 +743,17 @@ static void mana_serv_reset(struct pci_dev *pdev)
return;
}
+ spin_lock_irqsave(&gc->hwc_lock, flags);
hwc = gc->hwc.driver_data;
if (!hwc) {
+ spin_unlock_irqrestore(&gc->hwc_lock, flags);
dev_err(&pdev->dev, "MANA service: no HWC\n");
goto out;
}
/* HWC is not responding in this case, so don't wait */
- hwc->hwc_timeout = 0;
+ mana_hwc_timeout_cancel(hwc);
+ spin_unlock_irqrestore(&gc->hwc_lock, flags);
dev_info(&pdev->dev, "MANA reset cycle start\n");
@@ -1107,7 +1134,9 @@ static void mana_gd_deregister_irq(struct gdma_queue *queue)
synchronize_rcu();
}
-int mana_gd_test_eq(struct gdma_context *gc, struct gdma_queue *eq)
+static int mana_gd_test_eq_request(struct gdma_context *gc,
+ struct hw_channel_context *hwc,
+ struct gdma_queue *eq)
{
struct gdma_generate_test_event_req req = {};
struct gdma_general_resp resp = {};
@@ -1125,7 +1154,12 @@ int mana_gd_test_eq(struct gdma_context *gc, struct gdma_queue *eq)
req.hdr.dev_id = eq->gdma_dev->dev_id;
req.queue_index = eq->id;
- err = mana_gd_send_request(gc, sizeof(req), &req, sizeof(resp), &resp);
+ if (hwc)
+ err = mana_hwc_send_request(hwc, sizeof(req), &req,
+ sizeof(resp), &resp);
+ else
+ err = mana_gd_send_request(gc, sizeof(req), &req,
+ sizeof(resp), &resp);
if (err) {
if (mana_need_log(gc, err))
dev_err(dev, "test_eq failed: %d\n", err);
@@ -1156,6 +1190,17 @@ int mana_gd_test_eq(struct gdma_context *gc, struct gdma_queue *eq)
return err;
}
+int mana_gd_test_eq(struct gdma_context *gc, struct gdma_queue *eq)
+{
+ return mana_gd_test_eq_request(gc, NULL, eq);
+}
+
+int mana_gd_test_hwc_eq(struct hw_channel_context *hwc,
+ struct gdma_queue *eq)
+{
+ return mana_gd_test_eq_request(hwc->gdma_dev->gdma_context, hwc, eq);
+}
+
static void mana_gd_destroy_eq(struct gdma_context *gc, bool flush_events,
struct gdma_queue *queue)
{
@@ -1348,6 +1393,7 @@ static int mana_gd_create_dma_region(struct gdma_dev *gd,
if (gmi->nr_pages == 0 && !MANA_PAGE_ALIGNED(gmi->virt_addr))
return -EINVAL;
+ /* The caller must keep the HWC alive throughout queue creation. */
hwc = gc->hwc.driver_data;
req_msg_size = struct_size(req, page_addr_list, num_page);
if (req_msg_size > hwc->max_req_msg_size)
@@ -1551,9 +1597,12 @@ int mana_gd_verify_vf_version(struct pci_dev *pdev)
struct gdma_verify_ver_resp resp = {};
struct gdma_verify_ver_req req = {};
struct hw_channel_context *hwc;
+ u32 timeout_ms;
int err;
+ /* The setup caller must exclude concurrent HWC teardown. */
hwc = gc->hwc.driver_data;
+
mana_gd_init_req_hdr(&req.hdr, GDMA_VERIFY_VF_DRIVER_VERSION,
sizeof(req), sizeof(resp));
@@ -1589,12 +1638,16 @@ int mana_gd_verify_vf_version(struct pci_dev *pdev)
&gc->pf_cap_flags1);
if (resp.pf_cap_flags1 & GDMA_DRV_CAP_FLAG_1_HWC_TIMEOUT_RECONFIG) {
- err = mana_gd_query_hwc_timeout(pdev, &hwc->hwc_timeout);
+ err = mana_gd_query_hwc_timeout(pdev,
+ mana_hwc_timeout_read(hwc),
+ &timeout_ms);
if (err) {
dev_err(gc->dev, "Failed to set the hwc timeout %d\n", err);
return err;
}
- dev_dbg(gc->dev, "set the hwc timeout to %u\n", hwc->hwc_timeout);
+ mana_hwc_timeout_update(hwc, timeout_ms);
+ dev_dbg(gc->dev, "set the hwc timeout to %u\n",
+ mana_hwc_timeout_read(hwc));
}
return 0;
}
@@ -2547,6 +2600,7 @@ static int mana_gd_probe(struct pci_dev *pdev, const struct pci_device_id *ent)
mutex_init(&gc->eq_test_event_mutex);
mutex_init(&gc->gic_mutex);
+ spin_lock_init(&gc->hwc_lock);
pci_set_drvdata(pdev, gc);
gc->bar0_pa = pci_resource_start(pdev, 0);
gc->bar0_size = pci_resource_len(pdev, 0);
diff --git a/drivers/net/ethernet/microsoft/mana/hw_channel.c b/drivers/net/ethernet/microsoft/mana/hw_channel.c
index 89b6e30863da..9fdd84ccba65 100644
--- a/drivers/net/ethernet/microsoft/mana/hw_channel.c
+++ b/drivers/net/ethernet/microsoft/mana/hw_channel.c
@@ -6,57 +6,122 @@
#include <net/mana/hw_channel.h>
#include <linux/vmalloc.h>
-static int mana_hwc_get_msg_index(struct hw_channel_context *hwc, u16 *msg_id)
+u32 mana_hwc_timeout_read(const struct hw_channel_context *hwc)
+{
+ return READ_ONCE(hwc->hwc_timeout);
+}
+
+void mana_hwc_timeout_update(struct hw_channel_context *hwc, u32 timeout_ms)
+{
+ u32 old_timeout;
+
+ if (!timeout_ms)
+ return;
+
+ old_timeout = mana_hwc_timeout_read(hwc);
+ while (old_timeout &&
+ cmpxchg(&hwc->hwc_timeout, old_timeout, timeout_ms) != old_timeout)
+ old_timeout = mana_hwc_timeout_read(hwc);
+}
+
+void mana_hwc_timeout_cancel(struct hw_channel_context *hwc)
+{
+ xchg(&hwc->hwc_timeout, 0);
+}
+
+static void mana_hwc_timeout_reduce(struct hw_channel_context *hwc)
+{
+ u32 old_timeout = mana_hwc_timeout_read(hwc);
+
+ while (old_timeout > 1 &&
+ cmpxchg(&hwc->hwc_timeout, old_timeout, 1) != old_timeout)
+ old_timeout = mana_hwc_timeout_read(hwc);
+}
+
+static int mana_hwc_get_msg_index(struct hw_channel_context *hwc, void *resp,
+ u32 resp_len,
+ struct hwc_caller_ctx **caller_ctx)
{
struct gdma_resource *r = &hwc->inflight_msg_res;
struct hwc_caller_ctx *ctx;
unsigned long flags;
+ bool channel_up;
+ u32 wait_ms;
u32 index;
- down(&hwc->sema);
+ wait_ms = mana_hwc_timeout_read(hwc);
+ if (down_timeout(&hwc->sema, msecs_to_jiffies(wait_ms))) {
+ spin_lock_irqsave(&r->lock, flags);
+ channel_up = hwc->channel_up;
+ spin_unlock_irqrestore(&r->lock, flags);
- spin_lock_irqsave(&r->lock, flags);
+ /* Slot pressure is not evidence that the HWC stopped responding. */
+ return channel_up ? -EBUSY : -ENODEV;
+ }
- if (hwc->hwc_timed_out) {
+ spin_lock_irqsave(&r->lock, flags);
+ if (!hwc->channel_up) {
spin_unlock_irqrestore(&r->lock, flags);
up(&hwc->sema);
- return -ETIMEDOUT;
+ return -ENODEV;
}
- index = find_first_zero_bit(hwc->inflight_msg_res.map,
- hwc->inflight_msg_res.size);
+ /* The semaphore admits at most r->size holders at a time, so a slot
+ * acquired above always has a free bit waiting for it here.
+ */
+ index = find_first_zero_bit(r->map, r->size);
+ if (WARN_ON_ONCE(index >= r->size)) {
+ spin_unlock_irqrestore(&r->lock, flags);
+ up(&hwc->sema);
+ return -EIO;
+ }
ctx = &hwc->caller_ctx[index];
reinit_completion(&ctx->comp_event);
- ctx->output_buf = NULL;
- ctx->output_buflen = 0;
+ /* Take both references (sender + response handler) before publishing
+ * the slot, so an early response cannot free it under the sender.
+ */
+ refcount_set(&ctx->refcnt, 2);
+ ctx->output_buf = resp;
+ ctx->output_buflen = resp_len;
ctx->error = -EINPROGRESS;
ctx->status_code = 0;
+ ctx->responded = false;
+ ctx->resp_pending = true;
+ ctx->msg_id = index;
- bitmap_set(hwc->inflight_msg_res.map, index, 1);
+ /* The response path takes r->lock before ctx->lock, so publishing the
+ * bitmap last makes every field above visible before it can consume
+ * the response-side reference.
+ */
+ bitmap_set(r->map, index, 1);
spin_unlock_irqrestore(&r->lock, flags);
- *msg_id = index;
+ *caller_ctx = ctx;
return 0;
}
-static void mana_hwc_put_msg_index(struct hw_channel_context *hwc, u16 msg_id,
- bool timed_out)
+static void mana_hwc_put_msg_index(struct hw_channel_context *hwc, u16 msg_id)
{
struct gdma_resource *r = &hwc->inflight_msg_res;
unsigned long flags;
spin_lock_irqsave(&r->lock, flags);
- if (timed_out)
- hwc->hwc_timed_out = true;
- bitmap_clear(hwc->inflight_msg_res.map, msg_id, 1);
+ bitmap_clear(r->map, msg_id, 1);
spin_unlock_irqrestore(&r->lock, flags);
up(&hwc->sema);
}
+static void hwc_ctx_put(struct hw_channel_context *hwc,
+ struct hwc_caller_ctx *ctx)
+{
+ if (refcount_dec_and_test(&ctx->refcnt))
+ mana_hwc_put_msg_index(hwc, ctx->msg_id);
+}
+
static int mana_hwc_verify_resp_msg(const struct hwc_caller_ctx *caller_ctx,
const struct gdma_resp_hdr *resp_msg,
u32 resp_len)
@@ -97,40 +162,54 @@ static void mana_hwc_handle_resp(struct hw_channel_context *hwc, u32 resp_len,
struct hwc_work_request *rx_req, u16 msg_id)
{
const struct gdma_resp_hdr *resp_msg = rx_req->buf_va;
+ struct gdma_resource *r = &hwc->inflight_msg_res;
struct hwc_caller_ctx *ctx;
+ bool release;
int err;
- if (!test_bit(msg_id, hwc->inflight_msg_res.map)) {
+ spin_lock(&r->lock);
+ if (!test_bit(msg_id, r->map)) {
+ spin_unlock(&r->lock);
dev_err(hwc->dev, "hwc_rx: invalid msg_id = %u\n", msg_id);
mana_hwc_post_rx_wqe(hwc->rxq, rx_req);
return;
}
ctx = hwc->caller_ctx + msg_id;
-
spin_lock(&ctx->lock);
- if (!ctx->output_buf) {
+ spin_unlock(&r->lock);
+
+ /* Consume the response-side reference exactly once. This releases a
+ * quarantined slot after its late response arrives.
+ */
+ release = ctx->resp_pending;
+ ctx->resp_pending = false;
+
+ if (ctx->responded) {
spin_unlock(&ctx->lock);
mana_hwc_post_rx_wqe(hwc->rxq, rx_req);
+ if (release)
+ hwc_ctx_put(hwc, ctx);
return;
}
+ ctx->responded = true;
err = mana_hwc_verify_resp_msg(ctx, resp_msg, resp_len);
if (!err) {
ctx->status_code = resp_msg->status;
memcpy(ctx->output_buf, resp_msg, resp_len);
}
-
- ctx->output_buf = NULL;
ctx->error = err;
- /* Must post rx wqe before complete(), otherwise the next rx may
- * hit no_wqe error.
+ /* Post RX WQE before completing; the next response may arrive
+ * immediately and needs a posted buffer.
*/
mana_hwc_post_rx_wqe(hwc->rxq, rx_req);
-
complete(&ctx->comp_event);
spin_unlock(&ctx->lock);
+
+ if (release)
+ hwc_ctx_put(hwc, ctx);
}
static void mana_hwc_init_event_handler(void *ctx, struct gdma_queue *q_self,
@@ -221,7 +300,7 @@ static void mana_hwc_init_event_handler(void *ctx, struct gdma_queue *q_self,
switch (type) {
case HWC_DATA_CFG_HWC_TIMEOUT:
- hwc->hwc_timeout = val;
+ mana_hwc_timeout_update(hwc, val);
break;
case HWC_DATA_HW_LINK_CONNECT:
@@ -622,6 +701,7 @@ static int mana_hwc_create_wq(struct hw_channel_context *hwc,
hwc_wq->gdma_wq = queue;
hwc_wq->queue_depth = q_depth;
hwc_wq->hwc_cq = hwc_cq;
+ spin_lock_init(&hwc_wq->lock);
err = mana_hwc_alloc_dma_buf(hwc, q_depth, max_msg_size,
&hwc_wq->msg_buf);
@@ -639,7 +719,7 @@ static int mana_hwc_create_wq(struct hw_channel_context *hwc,
return err;
}
-static int mana_hwc_post_tx_wqe(const struct hwc_wq *hwc_txq,
+static int mana_hwc_post_tx_wqe(struct hwc_wq *hwc_txq,
struct hwc_work_request *req,
u32 dest_virt_rq_id, u32 dest_virt_rcq_id,
bool dest_pf)
@@ -678,7 +758,10 @@ static int mana_hwc_post_tx_wqe(const struct hwc_wq *hwc_txq,
req->wqe_req.inline_oob_data = tx_oob;
req->wqe_req.client_data_unit = 0;
+ spin_lock(&hwc_txq->lock);
err = mana_gd_post_and_ring(hwc_txq->gdma_wq, &req->wqe_req, NULL);
+ spin_unlock(&hwc_txq->lock);
+
if (err)
dev_err(dev, "Failed to post WQE on HWC SQ: %d\n", err);
return err;
@@ -697,43 +780,55 @@ static int mana_hwc_init_inflight_msg(struct hw_channel_context *hwc,
return err;
}
-static int mana_hwc_test_channel(struct hw_channel_context *hwc, u16 q_depth,
- u32 max_req_msg_size, u32 max_resp_msg_size)
+static int mana_hwc_test_channel(struct hw_channel_context *hwc)
{
- struct gdma_context *gc = hwc->gdma_dev->gdma_context;
struct hwc_wq *hwc_rxq = hwc->rxq;
struct hwc_work_request *req;
struct hwc_caller_ctx *ctx;
+ unsigned long flags;
int err;
int i;
/* Post all WQEs on the RQ */
- for (i = 0; i < q_depth; i++) {
+ for (i = 0; i < hwc->num_inflight_msg; i++) {
req = &hwc_rxq->msg_buf->reqs[i];
err = mana_hwc_post_rx_wqe(hwc_rxq, req);
if (err)
return err;
}
- ctx = kzalloc_objs(*ctx, q_depth);
+ ctx = kzalloc_objs(*ctx, hwc->num_inflight_msg);
if (!ctx)
return -ENOMEM;
- for (i = 0; i < q_depth; ++i) {
+ for (i = 0; i < hwc->num_inflight_msg; ++i) {
init_completion(&ctx[i].comp_event);
spin_lock_init(&ctx[i].lock);
}
hwc->caller_ctx = ctx;
- return mana_gd_test_eq(gc, hwc->cq->gdma_eq);
+ /* Setup owns hwc directly; runtime publication follows this test. */
+ spin_lock_irqsave(&hwc->inflight_msg_res.lock, flags);
+ hwc->channel_up = true;
+ spin_unlock_irqrestore(&hwc->inflight_msg_res.lock, flags);
+
+ err = mana_gd_test_hwc_eq(hwc, hwc->cq->gdma_eq);
+ if (err) {
+ spin_lock_irqsave(&hwc->inflight_msg_res.lock, flags);
+ hwc->channel_up = false;
+ spin_unlock_irqrestore(&hwc->inflight_msg_res.lock, flags);
+ }
+
+ return err;
}
-static int mana_hwc_establish_channel(struct gdma_context *gc, u16 *q_depth,
+static int mana_hwc_establish_channel(struct hw_channel_context *hwc,
+ u16 *q_depth,
u32 *max_req_msg_size,
u32 *max_resp_msg_size)
{
- struct hw_channel_context *hwc = gc->hwc.driver_data;
+ struct gdma_context *gc = hwc->gdma_dev->gdma_context;
struct gdma_queue *rq = hwc->rxq->gdma_wq;
struct gdma_queue *sq = hwc->txq->gdma_wq;
struct gdma_queue *eq = hwc->cq->gdma_eq;
@@ -848,68 +943,108 @@ static int mana_hwc_init_queues(struct hw_channel_context *hwc, u16 q_depth,
return err;
}
-int mana_hwc_create_channel(struct gdma_context *gc)
+static int mana_hwc_publish_channel(struct hw_channel_context *hwc)
+{
+ struct gdma_context *gc = hwc->gdma_dev->gdma_context;
+ unsigned long flags;
+ int err = 0;
+
+ spin_lock_irqsave(&gc->hwc_lock, flags);
+ if (WARN_ON_ONCE(gc->hwc.driver_data))
+ err = -EBUSY;
+ else
+ gc->hwc.driver_data = hwc;
+ spin_unlock_irqrestore(&gc->hwc_lock, flags);
+
+ return err;
+}
+
+static struct hw_channel_context *
+mana_hwc_unpublish_channel(struct gdma_context *gc)
{
- u32 max_req_msg_size, max_resp_msg_size;
- struct gdma_dev *gd = &gc->hwc;
struct hw_channel_context *hwc;
- u16 q_depth_max;
- int err;
+ unsigned long flags;
- /* Retry a retained context before assigning queues to the PF again. */
- if (gd->driver_data) {
- mana_hwc_destroy_channel(gc);
- if (gd->driver_data)
- return -ETIMEDOUT;
- }
+ spin_lock_irqsave(&gc->hwc_lock, flags);
+ hwc = gc->hwc.driver_data;
+ gc->hwc.driver_data = NULL;
+ spin_unlock_irqrestore(&gc->hwc_lock, flags);
- hwc = kzalloc_obj(*hwc);
- if (!hwc)
- return -ENOMEM;
+ return hwc;
+}
- gd->gdma_context = gc;
- gd->driver_data = hwc;
- hwc->gdma_dev = gd;
- hwc->dev = gc->dev;
- hwc->hwc_timeout = HW_CHANNEL_WAIT_RESOURCE_TIMEOUT_MS;
+static void mana_hwc_retain_channel(struct hw_channel_context *hwc)
+{
+ struct gdma_context *gc = hwc->gdma_dev->gdma_context;
+ unsigned long flags;
- /* HWC's instance number is always 0. */
- gd->dev_id.as_uint32 = 0;
- gd->dev_id.type = GDMA_DEVICE_HWC;
+ spin_lock_irqsave(&gc->hwc_lock, flags);
+ if (WARN_ON_ONCE(gc->hwc.driver_data))
+ dev_err(hwc->dev, "HWC retention slot is already occupied\n");
+ else
+ gc->hwc.driver_data = hwc;
+ spin_unlock_irqrestore(&gc->hwc_lock, flags);
+}
- gd->pdid = INVALID_PDID;
- gd->doorbell = INVALID_DOORBELL;
+static bool mana_hwc_senders_drained(struct gdma_context *gc,
+ struct hw_channel_context *hwc)
+{
+ unsigned long flags;
+ bool drained;
- /* mana_hwc_init_queues() only creates the required data structures,
- * and doesn't touch the HWC device.
- */
- err = mana_hwc_init_queues(hwc, HW_CHANNEL_VF_BOOTSTRAP_QUEUE_DEPTH,
- HW_CHANNEL_MAX_REQUEST_SIZE,
- HW_CHANNEL_MAX_RESPONSE_SIZE);
- if (err) {
- dev_err(hwc->dev, "Failed to initialize HWC: %d\n", err);
- goto out;
- }
+ spin_lock_irqsave(&gc->hwc_lock, flags);
+ drained = hwc->active_senders == 0;
+ spin_unlock_irqrestore(&gc->hwc_lock, flags);
- err = mana_hwc_establish_channel(gc, &q_depth_max, &max_req_msg_size,
- &max_resp_msg_size);
- if (err) {
- dev_err(hwc->dev, "Failed to establish HWC: %d\n", err);
- goto out;
+ return drained;
+}
+
+static void mana_hwc_stop_channel(struct hw_channel_context *hwc)
+{
+ struct gdma_resource *r = &hwc->inflight_msg_res;
+ struct gdma_context *gc = hwc->gdma_dev->gdma_context;
+ unsigned long flags;
+ int i;
+
+ if (hwc->num_inflight_msg) {
+ spin_lock_irqsave(&r->lock, flags);
+ hwc->channel_up = false;
+ spin_unlock_irqrestore(&r->lock, flags);
+ /* Wake one admission waiter; each rejected waiter returns the
+ * permit and wakes the next.
+ */
+ up(&hwc->sema);
}
+ mana_hwc_timeout_cancel(hwc);
- err = mana_hwc_test_channel(gc->hwc.driver_data,
- HW_CHANNEL_VF_BOOTSTRAP_QUEUE_DEPTH,
- max_req_msg_size, max_resp_msg_size);
- if (err) {
- dev_err(hwc->dev, "Failed to test HWC: %d\n", err);
- goto out;
+ for (i = 0; hwc->caller_ctx && i < hwc->num_inflight_msg; i++) {
+ struct hwc_caller_ctx *ctx;
+ bool drop_resp_ref;
+
+ spin_lock_irqsave(&r->lock, flags);
+ if (!test_bit(i, r->map)) {
+ spin_unlock_irqrestore(&r->lock, flags);
+ continue;
+ }
+
+ ctx = &hwc->caller_ctx[i];
+ spin_lock(&ctx->lock);
+ spin_unlock(&r->lock);
+
+ if (!ctx->responded)
+ ctx->error = -ENODEV;
+ ctx->output_buf = NULL;
+ drop_resp_ref = ctx->resp_pending;
+ ctx->resp_pending = false;
+ ctx->responded = true;
+ complete(&ctx->comp_event);
+ spin_unlock_irqrestore(&ctx->lock, flags);
+
+ if (drop_resp_ref)
+ hwc_ctx_put(hwc, ctx);
}
- return 0;
-out:
- mana_hwc_destroy_channel(gc);
- return err;
+ wait_event(gc->hwc_drain_waitq, mana_hwc_senders_drained(gc, hwc));
}
static void mana_hwc_fence_channel(struct gdma_context *gc,
@@ -925,14 +1060,13 @@ static void mana_hwc_fence_channel(struct gdma_context *gc,
mana_hwc_unpublish_cq(gc, hwc->cq->gdma_cq);
}
-void mana_hwc_destroy_channel(struct gdma_context *gc)
+static bool mana_hwc_release_channel(struct hw_channel_context *hwc)
{
- struct hw_channel_context *hwc = gc->hwc.driver_data;
+ struct gdma_context *gc = hwc->gdma_dev->gdma_context;
struct gdma_queue **old_cq_table;
int err;
- if (!hwc)
- return;
+ mana_hwc_stop_channel(hwc);
/* An unacknowledged destroy leaves the PF's mappings live. Fence
* software dispatch, but retain every PF-visible allocation.
@@ -944,7 +1078,8 @@ void mana_hwc_destroy_channel(struct gdma_context *gc)
"HWC teardown failed: %d, retaining PF-visible resources\n",
err);
mana_hwc_fence_channel(gc, hwc);
- return;
+ mana_hwc_retain_channel(hwc);
+ return false;
}
/* Fence the EQ before releasing any state its handlers can reach. */
@@ -972,10 +1107,7 @@ void mana_hwc_destroy_channel(struct gdma_context *gc)
hwc->gdma_dev->doorbell = INVALID_DOORBELL;
hwc->gdma_dev->pdid = INVALID_PDID;
- hwc->hwc_timeout = 0;
-
kfree(hwc);
- gc->hwc.driver_data = NULL;
gc->hwc.gdma_context = NULL;
old_cq_table = READ_ONCE(gc->cq_table);
@@ -983,13 +1115,151 @@ void mana_hwc_destroy_channel(struct gdma_context *gc)
smp_store_release(&gc->cq_table, NULL);
synchronize_rcu();
vfree(old_cq_table);
+
+ return true;
}
-static int mana_hwc_response_status(struct hw_channel_context *hwc,
- u32 command, int error, u32 status)
+static int mana_hwc_create_bootstrap_channel(struct hw_channel_context *hwc)
{
- if (error)
- return error;
+ u32 max_req_msg_size, max_resp_msg_size;
+ u16 q_depth_max;
+ int err;
+
+ err = mana_hwc_init_queues(hwc, HW_CHANNEL_VF_BOOTSTRAP_QUEUE_DEPTH,
+ HW_CHANNEL_MAX_REQUEST_SIZE,
+ HW_CHANNEL_MAX_RESPONSE_SIZE);
+ if (err) {
+ dev_err(hwc->dev, "Failed to initialize HWC: %d\n", err);
+ return err;
+ }
+
+ err = mana_hwc_establish_channel(hwc, &q_depth_max, &max_req_msg_size,
+ &max_resp_msg_size);
+ if (err) {
+ dev_err(hwc->dev, "Failed to establish HWC: %d\n", err);
+ return err;
+ }
+
+ err = mana_hwc_test_channel(hwc);
+ if (err)
+ dev_err(hwc->dev, "Failed to test HWC: %d\n", err);
+
+ return err;
+}
+
+int mana_hwc_create_channel(struct gdma_context *gc)
+{
+ struct gdma_dev *gd = &gc->hwc;
+ struct hw_channel_context *hwc;
+ int err;
+
+ /* Retry a retained context before assigning queues to the PF again. */
+ if (gd->driver_data) {
+ mana_hwc_destroy_channel(gc);
+ if (gd->driver_data)
+ return -ETIMEDOUT;
+ }
+
+ hwc = kzalloc_obj(*hwc);
+ if (!hwc)
+ return -ENOMEM;
+
+ gd->gdma_context = gc;
+ hwc->gdma_dev = gd;
+ hwc->dev = gc->dev;
+ WRITE_ONCE(hwc->hwc_timeout, HW_CHANNEL_WAIT_RESOURCE_TIMEOUT_MS);
+ init_waitqueue_head(&gc->hwc_drain_waitq);
+
+ /* HWC's instance number is always 0. */
+ gd->dev_id.as_uint32 = 0;
+ gd->dev_id.type = GDMA_DEVICE_HWC;
+
+ gd->pdid = INVALID_PDID;
+ gd->doorbell = INVALID_DOORBELL;
+
+ err = mana_hwc_create_bootstrap_channel(hwc);
+ if (err) {
+ mana_hwc_release_channel(hwc);
+ return err;
+ }
+
+ err = mana_hwc_publish_channel(hwc);
+ if (err) {
+ mana_hwc_release_channel(hwc);
+ return err;
+ }
+
+ return 0;
+}
+
+void mana_hwc_destroy_channel(struct gdma_context *gc)
+{
+ struct hw_channel_context *hwc;
+
+ hwc = mana_hwc_unpublish_channel(gc);
+ if (!hwc)
+ return;
+
+ mana_hwc_release_channel(hwc);
+}
+
+static void mana_hwc_abort_request(struct hw_channel_context *hwc,
+ struct hwc_caller_ctx *ctx)
+{
+ unsigned long flags;
+ bool drop_resp_ref;
+
+ spin_lock_irqsave(&ctx->lock, flags);
+ ctx->output_buf = NULL;
+ drop_resp_ref = ctx->resp_pending;
+ ctx->resp_pending = false;
+ ctx->responded = true;
+ spin_unlock_irqrestore(&ctx->lock, flags);
+
+ if (drop_resp_ref)
+ hwc_ctx_put(hwc, ctx);
+ hwc_ctx_put(hwc, ctx);
+}
+
+static int mana_hwc_submit_request(struct hw_channel_context *hwc,
+ struct hwc_caller_ctx *ctx,
+ struct hwc_work_request *tx_wr)
+{
+ unsigned long flags;
+ bool drop_resp_ref = false;
+ int err;
+
+ spin_lock_irqsave(&ctx->lock, flags);
+ if (ctx->responded) {
+ err = ctx->error ?: -ENODEV;
+ } else {
+ err = mana_hwc_post_tx_wqe(hwc->txq, tx_wr,
+ hwc->dest_vrq_id,
+ hwc->dest_vrcq_id, false);
+ if (err) {
+ ctx->output_buf = NULL;
+ drop_resp_ref = ctx->resp_pending;
+ ctx->resp_pending = false;
+ ctx->responded = true;
+ }
+ }
+ spin_unlock_irqrestore(&ctx->lock, flags);
+
+ if (!err)
+ return 0;
+
+ if (drop_resp_ref)
+ hwc_ctx_put(hwc, ctx);
+ hwc_ctx_put(hwc, ctx);
+
+ return err;
+}
+
+static int mana_hwc_response_result(struct hw_channel_context *hwc,
+ u32 command, int err, u32 status)
+{
+ if (err)
+ return err;
if (!status || status == GDMA_STATUS_MORE_ENTRIES)
return 0;
@@ -1004,108 +1274,86 @@ static int mana_hwc_response_status(struct hw_channel_context *hwc,
return -EPROTO;
}
-static void mana_hwc_finish_request(struct hw_channel_context *hwc,
- struct hwc_caller_ctx *ctx, u16 msg_id,
- bool timed_out)
+static int mana_hwc_wait_for_response(struct hw_channel_context *hwc,
+ struct hwc_caller_ctx *ctx,
+ u32 command)
{
unsigned long flags;
+ bool abandoned = false;
+ u32 wait_ms;
+ u32 status;
+ int err;
+
+ wait_ms = mana_hwc_timeout_read(hwc);
+ if (wait_for_completion_timeout(&ctx->comp_event,
+ msecs_to_jiffies(wait_ms))) {
+ spin_lock_irqsave(&ctx->lock, flags);
+ ctx->output_buf = NULL;
+ err = ctx->error;
+ status = ctx->status_code;
+ spin_unlock_irqrestore(&ctx->lock, flags);
+ hwc_ctx_put(hwc, ctx);
+
+ return mana_hwc_response_result(hwc, command, err, status);
+ }
spin_lock_irqsave(&ctx->lock, flags);
ctx->output_buf = NULL;
+ err = ctx->error;
+ status = ctx->status_code;
+ if (err == -EINPROGRESS) {
+ ctx->responded = true;
+ abandoned = true;
+ }
spin_unlock_irqrestore(&ctx->lock, flags);
- mana_hwc_put_msg_index(hwc, msg_id, timed_out);
+ if (!abandoned) {
+ hwc_ctx_put(hwc, ctx);
+ return mana_hwc_response_result(hwc, command, err, status);
+ }
+
+ if (wait_ms)
+ dev_err(hwc->dev, "Command 0x%x timed out: %u ms\n",
+ command, wait_ms);
+
+ mana_hwc_timeout_reduce(hwc);
+ hwc_ctx_put(hwc, ctx);
+
+ return -ETIMEDOUT;
}
int mana_hwc_send_request(struct hw_channel_context *hwc, u32 req_len,
const void *req, u32 resp_len, void *resp)
{
struct hwc_work_request *tx_wr;
- struct hwc_wq *txq = hwc->txq;
struct gdma_req_hdr *req_msg;
struct hwc_caller_ctx *ctx;
- unsigned long flags;
- u32 dest_vrcq;
- u32 dest_vrq;
u32 command;
- u32 status;
- u16 msg_id;
int err;
- err = mana_hwc_get_msg_index(hwc, &msg_id);
+ err = mana_hwc_get_msg_index(hwc, resp, resp_len, &ctx);
if (err)
return err;
- tx_wr = &txq->msg_buf->reqs[msg_id];
- ctx = hwc->caller_ctx + msg_id;
-
+ tx_wr = &hwc->txq->msg_buf->reqs[ctx->msg_id];
if (req_len > tx_wr->buf_len) {
dev_err(hwc->dev, "HWC: req msg size: %d > %d\n", req_len,
tx_wr->buf_len);
- mana_hwc_finish_request(hwc, ctx, msg_id, false);
+ mana_hwc_abort_request(hwc, ctx);
return -EINVAL;
}
- spin_lock_irqsave(&ctx->lock, flags);
- ctx->output_buf = resp;
- ctx->output_buflen = resp_len;
- spin_unlock_irqrestore(&ctx->lock, flags);
-
- req_msg = (struct gdma_req_hdr *)tx_wr->buf_va;
+ req_msg = tx_wr->buf_va;
if (req)
memcpy(req_msg, req, req_len);
-
- req_msg->req.hwc_msg_id = msg_id;
+ req_msg->req.hwc_msg_id = ctx->msg_id;
tx_wr->msg_size = req_len;
command = req_msg->req.msg_type;
- /* The hardware reports the HWC destination queues through
- * HWC_INIT_DATA_DEST_RQ_ID and HWC_INIT_DATA_DEST_CQ_ID, and
- * always supplies values that are valid for this function, so no
- * PF-specific handling is needed here.
- */
- dest_vrq = hwc->dest_vrq_id;
- dest_vrcq = hwc->dest_vrcq_id;
-
- err = mana_hwc_post_tx_wqe(txq, tx_wr, dest_vrq, dest_vrcq, false);
- if (err) {
- dev_err(hwc->dev, "HWC: Failed to post send WQE: %d\n", err);
- mana_hwc_finish_request(hwc, ctx, msg_id, false);
+ err = mana_hwc_submit_request(hwc, ctx, tx_wr);
+ if (err)
return err;
- }
-
- if (!wait_for_completion_timeout(&ctx->comp_event,
- msecs_to_jiffies(hwc->hwc_timeout))) {
- spin_lock_irqsave(&ctx->lock, flags);
- ctx->output_buf = NULL;
- err = ctx->error;
- status = ctx->status_code;
- spin_unlock_irqrestore(&ctx->lock, flags);
-
- if (err != -EINPROGRESS) {
- mana_hwc_finish_request(hwc, ctx, msg_id, false);
- return mana_hwc_response_status(hwc, command, err, status);
- }
-
- if (hwc->hwc_timeout != 0)
- dev_err(hwc->dev, "Command 0x%x timed out: %u ms\n",
- command, hwc->hwc_timeout);
-
- /* Reduce further waiting if HWC no response */
- if (hwc->hwc_timeout > 1)
- hwc->hwc_timeout = 1;
-
- mana_hwc_finish_request(hwc, ctx, msg_id, true);
- return -ETIMEDOUT;
- }
-
- spin_lock_irqsave(&ctx->lock, flags);
- ctx->output_buf = NULL;
- err = ctx->error;
- status = ctx->status_code;
- spin_unlock_irqrestore(&ctx->lock, flags);
- mana_hwc_finish_request(hwc, ctx, msg_id, false);
- return mana_hwc_response_status(hwc, command, err, status);
+ return mana_hwc_wait_for_response(hwc, ctx, command);
}
diff --git a/include/net/mana/gdma.h b/include/net/mana/gdma.h
index f8278e3817c8..ff843e0ee839 100644
--- a/include/net/mana/gdma.h
+++ b/include/net/mana/gdma.h
@@ -468,6 +468,15 @@ struct gdma_context {
/* Hardware communication channel (HWC) */
struct gdma_dev hwc;
+ /* Sender drain; the final wakeup runs under hwc_lock. */
+ wait_queue_head_t hwc_drain_waitq;
+
+ /* Protects runtime HWC publication and active sender references.
+ * Setup owns an unpublished HWC directly; timeout updates use atomic
+ * access helpers and do not require this lock.
+ */
+ spinlock_t hwc_lock;
+
/* Azure network adapter */
struct gdma_dev mana;
diff --git a/include/net/mana/hw_channel.h b/include/net/mana/hw_channel.h
index e733301f8740..01568d10b6fd 100644
--- a/include/net/mana/hw_channel.h
+++ b/include/net/mana/hw_channel.h
@@ -164,17 +164,32 @@ struct hwc_wq {
u16 queue_depth;
struct hwc_cq *hwc_cq;
+
+ /* Serializes SQ posting; unused for the RQ. */
+ spinlock_t lock;
};
struct hwc_caller_ctx {
struct completion comp_event;
- /* Protects the output buffer and response state from timeout. */
+
+ /* The slot is initialized while unpublished under inflight_msg_res.lock.
+ * Once its bitmap bit is set, lock protects every field below except
+ * msg_id and refcnt. The sender owns output_buf; the response handler
+ * may write it only while holding lock.
+ */
spinlock_t lock;
void *output_buf;
u32 output_buflen;
-
- int error; /* Linux error code */
+ int error;
u32 status_code;
+ bool responded;
+ bool resp_pending;
+
+ /* Tracks sender + response-handler ownership. The last put releases
+ * the bitmap slot under inflight_msg_res.lock.
+ */
+ refcount_t refcnt;
+ u16 msg_id;
};
struct hw_channel_context {
@@ -196,18 +211,30 @@ struct hw_channel_context {
struct hwc_wq *txq;
struct hwc_cq *cq;
+ /* Admission permits. Timed-out requests retain theirs until a
+ * response or teardown releases the slot.
+ */
struct semaphore sema;
struct gdma_resource inflight_msg_res;
u32 dest_vrq_id;
u32 dest_vrcq_id;
+
+ /* Zero permanently cancels waits for this channel instance. Firmware
+ * and query updates may replace a live nonzero value; fail-fast may
+ * only reduce a live value to one millisecond.
+ */
u32 hwc_timeout;
- /* Prevents message ID reuse after a timeout; protected by the map lock. */
- bool hwc_timed_out;
+ /* Checked after slot acquisition; cleared on teardown to reject sends. */
+ bool channel_up;
/* The PF may own the HWC queues while this is true. */
bool setup_active;
+
+ /* mana_gd_send_request() callers, including waiters; under hwc_lock. */
+ unsigned int active_senders;
+
struct hwc_caller_ctx *caller_ctx;
};
@@ -216,5 +243,11 @@ void mana_hwc_destroy_channel(struct gdma_context *gc);
int mana_hwc_send_request(struct hw_channel_context *hwc, u32 req_len,
const void *req, u32 resp_len, void *resp);
+int mana_gd_test_hwc_eq(struct hw_channel_context *hwc,
+ struct gdma_queue *eq);
+
+u32 mana_hwc_timeout_read(const struct hw_channel_context *hwc);
+void mana_hwc_timeout_update(struct hw_channel_context *hwc, u32 timeout_ms);
+void mana_hwc_timeout_cancel(struct hw_channel_context *hwc);
#endif /* _HW_CHANNEL_H */
--
2.43.0
^ permalink raw reply [flat|nested] 5+ messages in thread* [PATCH net-next v6 4/4] net: mana: add dynamic HWC queue depth with reinit path
2026-10-07 12:53 [PATCH net-next v6 0/4] net: mana: concurrent HWC requests and dynamic queue depth Wei Hu
` (2 preceding siblings ...)
2026-10-07 12:53 ` [PATCH net-next v6 3/4] net: mana: support concurrent HWC requests Wei Hu
@ 2026-10-07 12:53 ` Wei Hu
3 siblings, 0 replies; 5+ messages in thread
From: Wei Hu @ 2026-10-07 12:53 UTC (permalink / raw)
To: Long Li, Konstantin Taranov, Jakub Kicinski, David S. Miller,
Paolo Abeni, Eric Dumazet, Andrew Lunn, Jason Gunthorpe,
Leon Romanovsky, Haiyang Zhang, K. Y. Srinivasan, Wei Liu,
Dexuan Cui, Shradha Gupta, Simon Horman, Erni Sri Satya Vennela,
Stephen Hemminger, Shiraz Saleem
Cc: netdev, linux-rdma, linux-hyperv, linux-kernel, Aditya Garg,
Dipayaan Roy, Kees Cook, Kees Cook, Manish Awasthi, Wei Hu
From: Long Li <longli@microsoft.com>
The HWC starts with depth-one bootstrap queues. Treat the PF-reported depth
as a maximum and rebuild the channel exactly for reports from 2 through
128. Keep the untouched bootstrap queues for depth one and reports above
128, and require a rebuilt channel to repeat the selected depth.
Treat the reported request and response sizes as directional maxima.
Rebuild only when both reports equal the 4096-byte bootstrap allocation,
then enforce the effective report/allocation limit on every request and
expected response.
Keep each candidate channel private through queue creation, establishment
and the direct EQ test. Publish only the final tested channel. If a rebuild
fails after acknowledged teardown, destroy that candidate and establish
fresh bootstrap queues; never reuse mappings already handed to the PF.
Invalidate the previous HWC doorbell before every ESTABLISH_HWC. Accept a
new doorbell only after validating the BAR0 page geometry and reported
index, and suppress MMIO while re-establishment leaves the HWC doorbell
invalid.
Do not advertise the proposed dynamic-depth capability bit. Driver-version
advertisement occurs after HWC establishment and cannot negotiate behavior
needed while the channel is being created.
Use focused build, validation, rebuild and fallback helpers instead of the
goto-heavy creation flow from v5.
Link: https://lore.kernel.org/r/20260908035201.402424-5-longli@microsoft.com
Signed-off-by: Long Li <longli@microsoft.com>
Signed-off-by: Wei Hu <weh@microsoft.com>
---
.../net/ethernet/microsoft/mana/gdma_main.c | 24 +-
.../net/ethernet/microsoft/mana/hw_channel.c | 290 +++++++++++++++---
include/net/mana/gdma.h | 2 +
include/net/mana/hw_channel.h | 10 +-
4 files changed, 266 insertions(+), 60 deletions(-)
diff --git a/drivers/net/ethernet/microsoft/mana/gdma_main.c b/drivers/net/ethernet/microsoft/mana/gdma_main.c
index bf4c528f8c75..571c7f1d5b21 100644
--- a/drivers/net/ethernet/microsoft/mana/gdma_main.c
+++ b/drivers/net/ethernet/microsoft/mana/gdma_main.c
@@ -580,12 +580,26 @@ static int mana_gd_disable_queue(struct gdma_queue *queue)
#define DOORBELL_OFFSET_EQ 0xFF8
#define DOORBELL_OFFSET_DIM 0x820
+bool mana_gd_is_valid_doorbell(struct gdma_context *gc, u32 db_id)
+{
+ if (!gc->db_page_size || gc->db_page_off >= gc->bar0_size)
+ return false;
+
+ return gc->db_page_off +
+ gc->db_page_size * ((u64)db_id + 1) <= gc->bar0_size;
+}
+
static void mana_gd_ring_doorbell(struct gdma_context *gc, u32 db_index,
enum gdma_queue_type q_type, u32 qid,
u32 tail_ptr, u8 num_req)
{
- void __iomem *addr = gc->db_page_base + gc->db_page_size * db_index;
union gdma_doorbell_entry e = {};
+ void __iomem *addr;
+
+ if (unlikely(db_index == INVALID_DOORBELL))
+ return;
+
+ addr = gc->db_page_base + gc->db_page_size * db_index;
switch (q_type) {
case GDMA_EQ:
@@ -1675,13 +1689,7 @@ int mana_gd_register_device(struct gdma_dev *gd)
return err ? err : -EPROTO;
}
- /* Validate that doorbell page for db_id is within the BAR0 region.
- * In mana_gd_ring_doorbell(), the address is calculated as:
- * addr = db_page_base + db_page_size * db_id
- * = (bar0_va + db_page_off) + (db_page_size * db_id)
- * So we need: db_page_off + db_page_size * (db_id + 1) <= bar0_size
- */
- if (gc->db_page_off + gc->db_page_size * ((u64)resp.db_id + 1) > gc->bar0_size) {
+ if (!mana_gd_is_valid_doorbell(gc, resp.db_id)) {
dev_err(gc->dev, "Doorbell ID %u out of range\n", resp.db_id);
return -EPROTO;
}
diff --git a/drivers/net/ethernet/microsoft/mana/hw_channel.c b/drivers/net/ethernet/microsoft/mana/hw_channel.c
index 9fdd84ccba65..c9f7db5a76a5 100644
--- a/drivers/net/ethernet/microsoft/mana/hw_channel.c
+++ b/drivers/net/ethernet/microsoft/mana/hw_channel.c
@@ -227,8 +227,16 @@ static void mana_hwc_init_event_handler(void *ctx, struct gdma_queue *q_self,
switch (event->type) {
case GDMA_EQE_HWC_INIT_EQ_ID_DB:
eq_db.as_uint32 = event->details[0];
+ if (!mana_gd_is_valid_doorbell(gd->gdma_context,
+ eq_db.doorbell)) {
+ dev_err(hwc->dev, "HWC: invalid doorbell %u\n",
+ eq_db.doorbell);
+ break;
+ }
+
hwc->cq->gdma_eq->id = eq_db.eq_id;
gd->doorbell = eq_db.doorbell;
+ hwc->hwc_init_doorbell = true;
break;
case GDMA_EQE_HWC_INIT_DATA:
@@ -250,7 +258,8 @@ static void mana_hwc_init_event_handler(void *ctx, struct gdma_queue *q_self,
break;
case HWC_INIT_DATA_QUEUE_DEPTH:
- hwc->hwc_init_q_depth_max = (u16)val;
+ /* Preserve the full 24-bit report for validation. */
+ hwc->hwc_init_q_depth_max = val;
break;
case HWC_INIT_DATA_MAX_REQUEST:
@@ -617,7 +626,8 @@ static int mana_hwc_alloc_dma_buf(struct hw_channel_context *hwc, u16 q_depth,
dma_buf->num_reqs = q_depth;
- buf_size = MANA_PAGE_ALIGN(q_depth * max_msg_size);
+ /* mana_gd_alloc_memory() requires a power-of-two length. */
+ buf_size = roundup_pow_of_two(MANA_PAGE_ALIGN(q_depth * max_msg_size));
gmi = &dma_buf->mem_info;
err = mana_gd_alloc_memory(gc, buf_size, gmi, false);
@@ -823,10 +833,15 @@ static int mana_hwc_test_channel(struct hw_channel_context *hwc)
return err;
}
-static int mana_hwc_establish_channel(struct hw_channel_context *hwc,
- u16 *q_depth,
- u32 *max_req_msg_size,
- u32 *max_resp_msg_size)
+struct mana_hwc_init_report {
+ u32 queue_depth;
+ u32 max_req_msg_size;
+ u32 max_resp_msg_size;
+};
+
+static int
+mana_hwc_establish_channel(struct hw_channel_context *hwc,
+ struct mana_hwc_init_report *report)
{
struct gdma_context *gc = hwc->gdma_dev->gdma_context;
struct gdma_queue *rq = hwc->rxq->gdma_wq;
@@ -838,6 +853,18 @@ static int mana_hwc_establish_channel(struct hw_channel_context *hwc,
u32 cq_id;
int err;
+ hwc->hwc_init_q_depth_max = 0;
+ hwc->hwc_init_max_req_msg_size = 0;
+ hwc->hwc_init_max_resp_msg_size = 0;
+ hwc->hwc_init_max_num_cqs = 0;
+ hwc->hwc_init_cq_id = 0;
+ hwc->hwc_init_doorbell = false;
+ gc->hwc.pdid = INVALID_PDID;
+ /* Re-establish must not rearm through the previous channel's doorbell. */
+ gc->hwc.doorbell = INVALID_DOORBELL;
+ hwc->dest_vrq_id = 0;
+ hwc->dest_vrcq_id = 0;
+
init_completion(&hwc->hwc_init_eqe_comp);
err = mana_smc_setup_hwc(&gc->shm_channel, false,
@@ -852,9 +879,14 @@ static int mana_hwc_establish_channel(struct hw_channel_context *hwc,
if (!wait_for_completion_timeout(&hwc->hwc_init_eqe_comp, 60 * HZ))
return -ETIMEDOUT;
- *q_depth = hwc->hwc_init_q_depth_max;
- *max_req_msg_size = hwc->hwc_init_max_req_msg_size;
- *max_resp_msg_size = hwc->hwc_init_max_resp_msg_size;
+ if (!hwc->hwc_init_doorbell) {
+ dev_err(hwc->dev, "HWC: missing valid doorbell in init data\n");
+ return -EPROTO;
+ }
+
+ report->queue_depth = hwc->hwc_init_q_depth_max;
+ report->max_req_msg_size = hwc->hwc_init_max_req_msg_size;
+ report->max_resp_msg_size = hwc->hwc_init_max_resp_msg_size;
/* Snapshot the device-reported count and id once, so the same value
* sizes, bounds and indexes cq_table even across the sleeping
@@ -904,6 +936,9 @@ static int mana_hwc_init_queues(struct hw_channel_context *hwc, u16 q_depth,
{
int err;
+ if (q_depth > U16_MAX / 2)
+ return -EINVAL;
+
err = mana_hwc_init_inflight_msg(hwc, q_depth);
if (err)
return err;
@@ -936,6 +971,7 @@ static int mana_hwc_init_queues(struct hw_channel_context *hwc, u16 q_depth,
hwc->num_inflight_msg = q_depth;
hwc->max_req_msg_size = max_req_msg_size;
+ hwc->max_resp_msg_size = max_resp_msg_size;
return 0;
out:
@@ -943,6 +979,44 @@ static int mana_hwc_init_queues(struct hw_channel_context *hwc, u16 q_depth,
return err;
}
+static void mana_hwc_clear_cq_table(struct gdma_context *gc)
+{
+ struct gdma_queue **cq_table;
+
+ cq_table = READ_ONCE(gc->cq_table);
+ WRITE_ONCE(gc->max_num_cqs, 0);
+ /* Stop new table readers before waiting for prior RCU readers. */
+ smp_store_release(&gc->cq_table, NULL);
+ synchronize_rcu();
+ vfree(cq_table);
+}
+
+/* Setup owns an unpublished HWC and has no runtime senders here. */
+static void mana_hwc_destroy_queues(struct hw_channel_context *hwc)
+{
+ struct gdma_context *gc = hwc->gdma_dev->gdma_context;
+
+ if (hwc->cq) {
+ mana_hwc_destroy_cq(gc, hwc->cq);
+ hwc->cq = NULL;
+ }
+ mana_hwc_clear_cq_table(gc);
+
+ if (hwc->txq) {
+ mana_hwc_destroy_wq(hwc, hwc->txq);
+ hwc->txq = NULL;
+ }
+ if (hwc->rxq) {
+ mana_hwc_destroy_wq(hwc, hwc->rxq);
+ hwc->rxq = NULL;
+ }
+
+ kfree(hwc->caller_ctx);
+ hwc->caller_ctx = NULL;
+ mana_gd_free_res_map(&hwc->inflight_msg_res);
+ hwc->num_inflight_msg = 0;
+}
+
static int mana_hwc_publish_channel(struct hw_channel_context *hwc)
{
struct gdma_context *gc = hwc->gdma_dev->gdma_context;
@@ -1060,17 +1134,11 @@ static void mana_hwc_fence_channel(struct gdma_context *gc,
mana_hwc_unpublish_cq(gc, hwc->cq->gdma_cq);
}
-static bool mana_hwc_release_channel(struct hw_channel_context *hwc)
+static int mana_hwc_teardown_queues(struct hw_channel_context *hwc)
{
struct gdma_context *gc = hwc->gdma_dev->gdma_context;
- struct gdma_queue **old_cq_table;
int err;
- mana_hwc_stop_channel(hwc);
-
- /* An unacknowledged destroy leaves the PF's mappings live. Fence
- * software dispatch, but retain every PF-visible allocation.
- */
err = mana_smc_teardown_hwc(&gc->shm_channel, false,
&hwc->setup_active);
if (err) {
@@ -1078,31 +1146,26 @@ static bool mana_hwc_release_channel(struct hw_channel_context *hwc)
"HWC teardown failed: %d, retaining PF-visible resources\n",
err);
mana_hwc_fence_channel(gc, hwc);
- mana_hwc_retain_channel(hwc);
- return false;
+ return err;
}
- /* Fence the EQ before releasing any state its handlers can reach. */
- if (hwc->cq)
- mana_hwc_destroy_cq(hwc->gdma_dev->gdma_context, hwc->cq);
-
- /* Reset only after mana_hwc_destroy_cq() has cleared the CQ table
- * slot, so it is not left dangling.
- */
- WRITE_ONCE(gc->max_num_cqs, 0);
-
- if (hwc->txq)
- mana_hwc_destroy_wq(hwc, hwc->txq);
+ mana_hwc_destroy_queues(hwc);
- if (hwc->rxq)
- mana_hwc_destroy_wq(hwc, hwc->rxq);
+ return 0;
+}
- kfree(hwc->caller_ctx);
- hwc->caller_ctx = NULL;
+static bool mana_hwc_release_channel(struct hw_channel_context *hwc)
+{
+ struct gdma_context *gc = hwc->gdma_dev->gdma_context;
+ int err;
- mana_gd_free_res_map(&hwc->inflight_msg_res);
+ mana_hwc_stop_channel(hwc);
- hwc->num_inflight_msg = 0;
+ err = mana_hwc_teardown_queues(hwc);
+ if (err) {
+ mana_hwc_retain_channel(hwc);
+ return false;
+ }
hwc->gdma_dev->doorbell = INVALID_DOORBELL;
hwc->gdma_dev->pdid = INVALID_PDID;
@@ -1110,36 +1173,150 @@ static bool mana_hwc_release_channel(struct hw_channel_context *hwc)
kfree(hwc);
gc->hwc.gdma_context = NULL;
- old_cq_table = READ_ONCE(gc->cq_table);
- /* Stop new table readers before waiting for existing IRQ readers. */
- smp_store_release(&gc->cq_table, NULL);
- synchronize_rcu();
- vfree(old_cq_table);
-
return true;
}
-static int mana_hwc_create_bootstrap_channel(struct hw_channel_context *hwc)
+static int
+mana_hwc_validate_report(struct hw_channel_context *hwc,
+ const struct mana_hwc_init_report *report,
+ u32 expected_depth, bool check_depth,
+ bool require_exact_msg_sizes)
+{
+ if (!report->queue_depth || !report->max_req_msg_size ||
+ !report->max_resp_msg_size) {
+ dev_err(hwc->dev,
+ "HWC: invalid maxima depth=%u req=%u resp=%u\n",
+ report->queue_depth, report->max_req_msg_size,
+ report->max_resp_msg_size);
+ return -EPROTO;
+ }
+
+ if (require_exact_msg_sizes &&
+ (report->max_req_msg_size != HW_CHANNEL_MAX_REQUEST_SIZE ||
+ report->max_resp_msg_size != HW_CHANNEL_MAX_RESPONSE_SIZE)) {
+ dev_err(hwc->dev,
+ "HWC: rebuilt message maxima req=%u resp=%u, expected %u/%u\n",
+ report->max_req_msg_size, report->max_resp_msg_size,
+ HW_CHANNEL_MAX_REQUEST_SIZE,
+ HW_CHANNEL_MAX_RESPONSE_SIZE);
+ return -EPROTO;
+ }
+
+ if (check_depth && report->queue_depth != expected_depth) {
+ dev_err(hwc->dev, "HWC: rebuilt depth %u, expected %u\n",
+ report->queue_depth, expected_depth);
+ return -EPROTO;
+ }
+
+ return 0;
+}
+
+static int mana_hwc_build_channel(struct hw_channel_context *hwc, u16 q_depth,
+ struct mana_hwc_init_report *report)
{
- u32 max_req_msg_size, max_resp_msg_size;
- u16 q_depth_max;
int err;
- err = mana_hwc_init_queues(hwc, HW_CHANNEL_VF_BOOTSTRAP_QUEUE_DEPTH,
+ err = mana_hwc_init_queues(hwc, q_depth,
HW_CHANNEL_MAX_REQUEST_SIZE,
HW_CHANNEL_MAX_RESPONSE_SIZE);
if (err) {
- dev_err(hwc->dev, "Failed to initialize HWC: %d\n", err);
+ dev_err(hwc->dev, "Failed to initialize HWC depth %u: %d\n",
+ q_depth, err);
return err;
}
- err = mana_hwc_establish_channel(hwc, &q_depth_max, &max_req_msg_size,
- &max_resp_msg_size);
- if (err) {
- dev_err(hwc->dev, "Failed to establish HWC: %d\n", err);
+ err = mana_hwc_establish_channel(hwc, report);
+ if (err)
+ dev_err(hwc->dev, "Failed to establish HWC depth %u: %d\n",
+ q_depth, err);
+
+ return err;
+}
+
+static int
+mana_hwc_restore_bootstrap(struct hw_channel_context *hwc,
+ struct mana_hwc_init_report *report)
+{
+ int err;
+
+ err = mana_hwc_build_channel(hwc,
+ HW_CHANNEL_VF_BOOTSTRAP_QUEUE_DEPTH,
+ report);
+ if (err)
return err;
+
+ return mana_hwc_validate_report(hwc, report, 0, false, false);
+}
+
+static int mana_hwc_rebuild_channel(struct hw_channel_context *hwc,
+ u16 q_depth,
+ struct mana_hwc_init_report *report)
+{
+ int cleanup_err;
+ int err;
+
+ err = mana_hwc_teardown_queues(hwc);
+ if (err)
+ return err;
+
+ err = mana_hwc_build_channel(hwc, q_depth, report);
+ if (!err)
+ err = mana_hwc_validate_report(hwc, report, q_depth, true, true);
+ if (!err)
+ return 0;
+
+ dev_warn(hwc->dev,
+ "HWC depth %u rebuild failed, restoring bootstrap: %d\n",
+ q_depth, err);
+
+ cleanup_err = mana_hwc_teardown_queues(hwc);
+ if (cleanup_err)
+ return cleanup_err;
+
+ return mana_hwc_restore_bootstrap(hwc, report);
+}
+
+static int mana_hwc_prepare_channel(struct hw_channel_context *hwc)
+{
+ struct mana_hwc_init_report report;
+ u16 rebuild_depth;
+ int err;
+
+ err = mana_hwc_build_channel(hwc,
+ HW_CHANNEL_VF_BOOTSTRAP_QUEUE_DEPTH,
+ &report);
+ if (err)
+ return err;
+
+ err = mana_hwc_validate_report(hwc, &report, 0, false, false);
+ if (err)
+ return err;
+
+ if (report.queue_depth > HW_CHANNEL_MAX_QUEUE_DEPTH) {
+ dev_warn(hwc->dev,
+ "HWC depth %u exceeds limit %u, keeping bootstrap\n",
+ report.queue_depth, HW_CHANNEL_MAX_QUEUE_DEPTH);
+ } else if (report.queue_depth >
+ HW_CHANNEL_VF_BOOTSTRAP_QUEUE_DEPTH &&
+ (report.max_req_msg_size != HW_CHANNEL_MAX_REQUEST_SIZE ||
+ report.max_resp_msg_size != HW_CHANNEL_MAX_RESPONSE_SIZE)) {
+ dev_warn(hwc->dev,
+ "HWC maxima req=%u resp=%u differ from rebuild policy %u/%u, keeping bootstrap\n",
+ report.max_req_msg_size, report.max_resp_msg_size,
+ HW_CHANNEL_MAX_REQUEST_SIZE,
+ HW_CHANNEL_MAX_RESPONSE_SIZE);
+ } else if (report.queue_depth >
+ HW_CHANNEL_VF_BOOTSTRAP_QUEUE_DEPTH) {
+ rebuild_depth = report.queue_depth;
+ err = mana_hwc_rebuild_channel(hwc, rebuild_depth, &report);
+ if (err)
+ return err;
}
+ /* MAX_REQUEST sizes the RQ; MAX_RESPONSE sizes the SQ. */
+ hwc->rx_msg_size_limit = report.max_req_msg_size;
+ hwc->tx_msg_size_limit = report.max_resp_msg_size;
+
err = mana_hwc_test_channel(hwc);
if (err)
dev_err(hwc->dev, "Failed to test HWC: %d\n", err);
@@ -1177,7 +1354,7 @@ int mana_hwc_create_channel(struct gdma_context *gc)
gd->pdid = INVALID_PDID;
gd->doorbell = INVALID_DOORBELL;
- err = mana_hwc_create_bootstrap_channel(hwc);
+ err = mana_hwc_prepare_channel(hwc);
if (err) {
mana_hwc_release_channel(hwc);
return err;
@@ -1329,8 +1506,19 @@ int mana_hwc_send_request(struct hw_channel_context *hwc, u32 req_len,
struct gdma_req_hdr *req_msg;
struct hwc_caller_ctx *ctx;
u32 command;
+ u32 req_limit;
+ u32 resp_limit;
int err;
+ req_limit = min(hwc->tx_msg_size_limit, hwc->max_resp_msg_size);
+ resp_limit = min(hwc->rx_msg_size_limit, hwc->max_req_msg_size);
+ if (req_len > req_limit || resp_len > resp_limit) {
+ dev_err(hwc->dev,
+ "HWC: message sizes req=%u/%u resp=%u/%u exceed channel maxima\n",
+ req_len, req_limit, resp_len, resp_limit);
+ return -EMSGSIZE;
+ }
+
err = mana_hwc_get_msg_index(hwc, resp, resp_len, &ctx);
if (err)
return err;
diff --git a/include/net/mana/gdma.h b/include/net/mana/gdma.h
index ff843e0ee839..a624b17ec7a5 100644
--- a/include/net/mana/gdma.h
+++ b/include/net/mana/gdma.h
@@ -534,6 +534,8 @@ void mana_gd_fence_eq(struct gdma_context *gc, struct gdma_queue *queue);
int mana_gd_poll_cq(struct gdma_queue *cq, struct gdma_comp *comp, int num_cqe);
+bool mana_gd_is_valid_doorbell(struct gdma_context *gc, u32 db_id);
+
void mana_gd_ring_cq(struct gdma_queue *cq, u8 arm_bit);
ssize_t mana_gd_read_ring(struct gdma_queue *q, char __user *buf,
diff --git a/include/net/mana/hw_channel.h b/include/net/mana/hw_channel.h
index 01568d10b6fd..896effd1f716 100644
--- a/include/net/mana/hw_channel.h
+++ b/include/net/mana/hw_channel.h
@@ -11,6 +11,9 @@
#define HW_CHANNEL_VF_BOOTSTRAP_QUEUE_DEPTH 1
+/* Largest supported rebuild depth; larger reports retain bootstrap queues. */
+#define HW_CHANNEL_MAX_QUEUE_DEPTH 128
+
#define HWC_INIT_DATA_CQID 1
#define HWC_INIT_DATA_RQID 2
#define HWC_INIT_DATA_SQID 3
@@ -198,12 +201,17 @@ struct hw_channel_context {
u16 num_inflight_msg;
u32 max_req_msg_size;
+ u32 max_resp_msg_size;
+ /* Immutable reported maxima for the final RQ and SQ directions. */
+ u32 rx_msg_size_limit;
+ u32 tx_msg_size_limit;
- u16 hwc_init_q_depth_max;
+ u32 hwc_init_q_depth_max;
u32 hwc_init_max_req_msg_size;
u32 hwc_init_max_resp_msg_size;
u32 hwc_init_max_num_cqs;
u32 hwc_init_cq_id;
+ bool hwc_init_doorbell;
struct completion hwc_init_eqe_comp;
--
2.43.0
^ permalink raw reply [flat|nested] 5+ messages in thread