From: Keith Busch <kbusch@kernel.org>
To: Naman Jain <namjain@linux.microsoft.com>
Cc: linux-nvme@lists.infradead.org, Jens Axboe <axboe@kernel.dk>,
Christoph Hellwig <hch@lst.de>, Sagi Grimberg <sagi@grimberg.me>,
Michael Kelley <mhklinux@outlook.com>,
linux-hyperv@vger.kernel.org, linux-kernel@vger.kernel.org
Subject: Re: [RFC PATCH 3/3] nvme-pci: defer completions when rescheduling is needed
Date: Fri, 9 Oct 2026 09:30:40 -0600 [thread overview]
Message-ID: <askIoNon8UVyNJiJ@kbusch-mbp> (raw)
In-Reply-To: <20261009050556.2817978-4-namjain@linux.microsoft.com>
On Fri, Oct 09, 2026 at 05:05:56AM +0000, Naman Jain wrote:
> static irqreturn_t nvme_irq(int irq, void *data)
> {
> struct nvme_queue *nvmeq = data;
> DEFINE_IO_COMP_BATCH(iob);
>
> + if (unlikely(test_and_clear_bit(NVMEQ_IRQ_POLL_HANDOFF,
> + &nvmeq->flags))) {
> + if (!nvme_cqe_pending(nvmeq))
> + return IRQ_HANDLED;
> + }
> + if (unlikely(test_bit(NVMEQ_IRQ_POLL_PAUSED, &nvmeq->flags) &&
> + (!test_bit(NVMEQ_IRQ_POLL_INIT, &nvmeq->flags) ||
> + test_bit(NVMEQ_IRQ_POLL_MASKED, &nvmeq->flags) ||
> + READ_ONCE(to_pci_dev(nvmeq->dev->dev)->error_state) !=
> + pci_channel_io_normal)))
> + return IRQ_HANDLED;
> +
> + if (unlikely(in_hardirq() &&
> + test_bit(NVMEQ_IRQ_POLL_INIT, &nvmeq->flags) &&
> + !test_bit(NVMEQ_IRQ_POLL_PAUSED, &nvmeq->flags) &&
> + need_resched() && nvme_cqe_pending(nvmeq) &&
> + !test_and_set_bit_lock(NVMEQ_IRQ_POLL, &nvmeq->flags))) {
> + set_bit(NVMEQ_IRQ_POLL_MASKED, &nvmeq->flags);
> + disable_irq_nosync(irq);
> + irq_poll_sched(&nvmeq->iopoll);
> + return IRQ_HANDLED;
> + }
The disable_irq_nosync still does non-posted transactions for msi and
msi-x. NVMe has a shortcut for msi, but there's currently nothing for
msi-x. For that case, I have a prep patch I consulted with Thomas, and
the result is below. I'm delayed sending it out while catching up from
travel, but I think this will be even more efficient.
---
diff --git a/drivers/irqchip/irq-msi-lib.c b/drivers/irqchip/irq-msi-lib.c
index 45e0ed3134ce1..bff622b4839aa 100644
--- a/drivers/irqchip/irq-msi-lib.c
+++ b/drivers/irqchip/irq-msi-lib.c
@@ -126,6 +126,7 @@ bool msi_lib_init_dev_msi_info(struct device *dev, struct irq_domain *domain,
*/
if (info->flags & MSI_FLAG_PCI_MSI_MASK_PARENT) {
chip->irq_mask = irq_chip_mask_parent;
+ chip->irq_mask_nowait = NULL;
chip->irq_unmask = irq_chip_unmask_parent;
}
diff --git a/drivers/pci/msi/irqdomain.c b/drivers/pci/msi/irqdomain.c
index 6e65f0f44112e..92a53c2772659 100644
--- a/drivers/pci/msi/irqdomain.c
+++ b/drivers/pci/msi/irqdomain.c
@@ -162,6 +162,11 @@ static void pci_irq_mask_msix(struct irq_data *data)
pci_msix_mask(irq_data_get_msi_desc(data));
}
+static void pci_irq_mask_nowait_msix(struct irq_data *data)
+{
+ pci_msix_mask_nowait(irq_data_get_msi_desc(data));
+}
+
static void pci_irq_unmask_msix(struct irq_data *data)
{
pci_msix_unmask(irq_data_get_msi_desc(data));
@@ -182,6 +187,7 @@ static const struct msi_domain_template pci_msix_template = {
.irq_startup = pci_irq_startup_msix,
.irq_shutdown = pci_irq_shutdown_msix,
.irq_mask = pci_irq_mask_msix,
+ .irq_mask_nowait = pci_irq_mask_nowait_msix,
.irq_unmask = pci_irq_unmask_msix,
.irq_write_msi_msg = pci_msi_domain_write_msg,
.flags = IRQCHIP_ONESHOT_SAFE,
diff --git a/drivers/pci/msi/msi.h b/drivers/pci/msi/msi.h
index 0b420b319f50f..337ff9d9da0ed 100644
--- a/drivers/pci/msi/msi.h
+++ b/drivers/pci/msi/msi.h
@@ -40,10 +40,16 @@ static inline void pci_msix_write_vector_ctrl(struct msi_desc *desc, u32 ctrl)
writel(ctrl, desc_addr + PCI_MSIX_ENTRY_VECTOR_CTRL);
}
-static inline void pci_msix_mask(struct msi_desc *desc)
+/* Posted write only: the device may still send an interrupt after return */
+static inline void pci_msix_mask_nowait(struct msi_desc *desc)
{
desc->pci.msix_ctrl |= PCI_MSIX_ENTRY_CTRL_MASKBIT;
pci_msix_write_vector_ctrl(desc, desc->pci.msix_ctrl);
+}
+
+static inline void pci_msix_mask(struct msi_desc *desc)
+{
+ pci_msix_mask_nowait(desc);
/* Flush write to device */
readl(desc->pci.mask_base);
}
diff --git a/include/linux/interrupt.h b/include/linux/interrupt.h
index 3bf969ad8fe07..d40c992ddb8a1 100644
--- a/include/linux/interrupt.h
+++ b/include/linux/interrupt.h
@@ -231,6 +231,7 @@ extern void devm_free_irq(struct device *dev, unsigned int irq, void *dev_id);
bool irq_has_action(unsigned int irq);
extern void disable_irq_nosync(unsigned int irq);
+extern void disable_irq_nowait(unsigned int irq);
extern bool disable_hardirq(unsigned int irq);
extern void disable_irq(unsigned int irq);
extern void disable_percpu_irq(unsigned int irq);
diff --git a/include/linux/irq.h b/include/linux/irq.h
index f485369b1b4f7..de095299f20ed 100644
--- a/include/linux/irq.h
+++ b/include/linux/irq.h
@@ -455,6 +455,9 @@ static inline irq_hw_number_t irqd_to_hwirq(struct irq_data *d)
* @irq_disable: disable the interrupt
* @irq_ack: start of a new interrupt
* @irq_mask: mask an interrupt source
+ * @irq_mask_nowait: mask an interrupt source without waiting for the
+ * hardware to acknowledge it (optional, falls back to
+ * @irq_mask)
* @irq_mask_ack: ack and mask an interrupt source
* @irq_unmask: unmask an interrupt source
* @irq_eoi: end of interrupt
@@ -505,6 +508,7 @@ struct irq_chip {
void (*irq_ack)(struct irq_data *data);
void (*irq_mask)(struct irq_data *data);
+ void (*irq_mask_nowait)(struct irq_data *data);
void (*irq_mask_ack)(struct irq_data *data);
void (*irq_unmask)(struct irq_data *data);
void (*irq_eoi)(struct irq_data *data);
diff --git a/kernel/irq/chip.c b/kernel/irq/chip.c
index de754db414d1d..4e76f025b5f68 100644
--- a/kernel/irq/chip.c
+++ b/kernel/irq/chip.c
@@ -399,6 +399,37 @@ void irq_disable(struct irq_desc *desc)
__irq_disable(desc, irq_settings_disable_unlazy(desc));
}
+static void mask_irq_nowait(struct irq_desc *desc)
+{
+ if (irqd_irq_masked(&desc->irq_data))
+ return;
+
+ if (desc->irq_data.chip->irq_mask_nowait) {
+ desc->irq_data.chip->irq_mask_nowait(&desc->irq_data);
+ irq_state_set_masked(desc);
+ } else {
+ mask_irq(desc);
+ }
+}
+
+/**
+ * irq_disable_nowait - Mark interrupt disabled and mask it without waiting
+ * @desc: irq descriptor which should be disabled
+ *
+ * Like irq_disable() with lazy disable turned off, but the hardware may
+ * still deliver an interrupt after this returns. The flow handler deals
+ * with that the same way it does for a lazily disabled interrupt.
+ */
+void irq_disable_nowait(struct irq_desc *desc)
+{
+ if (desc->irq_data.chip->irq_disable) {
+ __irq_disable(desc, true);
+ return;
+ }
+ irq_state_set_disabled(desc);
+ mask_irq_nowait(desc);
+}
+
void irq_percpu_enable(struct irq_desc *desc, unsigned int cpu)
{
if (desc->irq_data.chip->irq_enable)
diff --git a/kernel/irq/internals.h b/kernel/irq/internals.h
index 0ce21dd454047..3ba13932ba831 100644
--- a/kernel/irq/internals.h
+++ b/kernel/irq/internals.h
@@ -98,6 +98,7 @@ extern void irq_startup_managed(struct irq_desc *desc);
extern void irq_shutdown(struct irq_desc *desc);
extern void irq_shutdown_and_deactivate(struct irq_desc *desc);
extern void irq_disable(struct irq_desc *desc);
+extern void irq_disable_nowait(struct irq_desc *desc);
extern void irq_percpu_enable(struct irq_desc *desc, unsigned int cpu);
extern void irq_percpu_disable(struct irq_desc *desc, unsigned int cpu);
extern void mask_irq(struct irq_desc *desc);
diff --git a/kernel/irq/manage.c b/kernel/irq/manage.c
index 2fbff2618a1e2..ecb4d2144b9d5 100644
--- a/kernel/irq/manage.c
+++ b/kernel/irq/manage.c
@@ -705,6 +705,27 @@ void disable_irq_nosync(unsigned int irq)
}
EXPORT_SYMBOL(disable_irq_nosync);
+/**
+ * disable_irq_nowait - mask an irq without waiting for the hardware
+ * @irq: Interrupt to disable
+ *
+ * Like disable_irq_nosync(), but masks the interrupt at the chip right away
+ * instead of lazily, and does not wait for the hardware to acknowledge the
+ * mask. An interrupt that still arrives is held pending and replayed by
+ * enable_irq(). Use this when an extra interrupt is cheaper than flushing
+ * the mask, e.g. switching from interrupts to polling.
+ *
+ * This function may be called from IRQ context.
+ */
+void disable_irq_nowait(unsigned int irq)
+{
+ scoped_irqdesc_get_and_buslock(irq, IRQ_GET_DESC_CHECK_GLOBAL) {
+ if (!scoped_irqdesc->depth++)
+ irq_disable_nowait(scoped_irqdesc);
+ }
+}
+EXPORT_SYMBOL_GPL(disable_irq_nowait);
+
/**
* disable_irq - disable an irq and wait for completion
* @irq: Interrupt to disable
--
next prev parent reply other threads:[~2026-10-09 15:30 UTC|newest]
Thread overview: 11+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-09 5:05 [RFC PATCH 0/3] nvme-pci: yield completions under scheduler pressure Naman Jain
2026-10-09 5:05 ` [RFC PATCH 1/3] nvme-pci: select IRQ_POLL Naman Jain
2026-10-09 5:05 ` [RFC PATCH 2/3] nvme-pci: make completion queue polling softirq-safe Naman Jain
2026-10-09 5:05 ` [RFC PATCH 3/3] nvme-pci: defer completions when rescheduling is needed Naman Jain
2026-10-09 15:30 ` Keith Busch [this message]
2026-10-09 5:54 ` [RFC PATCH 0/3] nvme-pci: yield completions under scheduler pressure Michael Kelley
2026-10-09 6:51 ` Naman Jain
2026-10-09 10:16 ` Naman Jain
2026-10-09 11:27 ` Luigi Rizzo
2026-10-09 14:58 ` Luigi Rizzo
2026-10-09 11:36 ` Fengnan
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=askIoNon8UVyNJiJ@kbusch-mbp \
--to=kbusch@kernel.org \
--cc=axboe@kernel.dk \
--cc=hch@lst.de \
--cc=linux-hyperv@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-nvme@lists.infradead.org \
--cc=mhklinux@outlook.com \
--cc=namjain@linux.microsoft.com \
--cc=sagi@grimberg.me \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®