mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Keith Busch <kbusch@kernel.org>
To: Naman Jain <namjain@linux.microsoft.com>
Cc: linux-nvme@lists.infradead.org, Jens Axboe <axboe@kernel.dk>,
	Christoph Hellwig <hch@lst.de>, Sagi Grimberg <sagi@grimberg.me>,
	Michael Kelley <mhklinux@outlook.com>,
	linux-hyperv@vger.kernel.org, linux-kernel@vger.kernel.org
Subject: Re: [RFC PATCH 3/3] nvme-pci: defer completions when rescheduling is needed
Date: Fri, 9 Oct 2026 09:30:40 -0600	[thread overview]
Message-ID: <askIoNon8UVyNJiJ@kbusch-mbp> (raw)
In-Reply-To: <20261009050556.2817978-4-namjain@linux.microsoft.com>

On Fri, Oct 09, 2026 at 05:05:56AM +0000, Naman Jain wrote:
>  static irqreturn_t nvme_irq(int irq, void *data)
>  {
>  	struct nvme_queue *nvmeq = data;
>  	DEFINE_IO_COMP_BATCH(iob);
>  
> +	if (unlikely(test_and_clear_bit(NVMEQ_IRQ_POLL_HANDOFF,
> +					&nvmeq->flags))) {
> +		if (!nvme_cqe_pending(nvmeq))
> +			return IRQ_HANDLED;
> +	}
> +	if (unlikely(test_bit(NVMEQ_IRQ_POLL_PAUSED, &nvmeq->flags) &&
> +		     (!test_bit(NVMEQ_IRQ_POLL_INIT, &nvmeq->flags) ||
> +		      test_bit(NVMEQ_IRQ_POLL_MASKED, &nvmeq->flags) ||
> +		      READ_ONCE(to_pci_dev(nvmeq->dev->dev)->error_state) !=
> +				pci_channel_io_normal)))
> +		return IRQ_HANDLED;
> +
> +	if (unlikely(in_hardirq() &&
> +		     test_bit(NVMEQ_IRQ_POLL_INIT, &nvmeq->flags) &&
> +		     !test_bit(NVMEQ_IRQ_POLL_PAUSED, &nvmeq->flags) &&
> +		     need_resched() && nvme_cqe_pending(nvmeq) &&
> +		     !test_and_set_bit_lock(NVMEQ_IRQ_POLL, &nvmeq->flags))) {
> +		set_bit(NVMEQ_IRQ_POLL_MASKED, &nvmeq->flags);
> +		disable_irq_nosync(irq);
> +		irq_poll_sched(&nvmeq->iopoll);
> +		return IRQ_HANDLED;
> +	}

The disable_irq_nosync still does non-posted transactions for msi and
msi-x. NVMe has a shortcut for msi, but there's currently nothing for
msi-x. For that case, I have a prep patch I consulted with Thomas, and
the result is below. I'm delayed sending it out while catching up from
travel, but I think this will be even more efficient.

---
diff --git a/drivers/irqchip/irq-msi-lib.c b/drivers/irqchip/irq-msi-lib.c
index 45e0ed3134ce1..bff622b4839aa 100644
--- a/drivers/irqchip/irq-msi-lib.c
+++ b/drivers/irqchip/irq-msi-lib.c
@@ -126,6 +126,7 @@ bool msi_lib_init_dev_msi_info(struct device *dev, struct irq_domain *domain,
 	 */
 	if (info->flags & MSI_FLAG_PCI_MSI_MASK_PARENT) {
 		chip->irq_mask		= irq_chip_mask_parent;
+		chip->irq_mask_nowait	= NULL;
 		chip->irq_unmask	= irq_chip_unmask_parent;
 	}
 
diff --git a/drivers/pci/msi/irqdomain.c b/drivers/pci/msi/irqdomain.c
index 6e65f0f44112e..92a53c2772659 100644
--- a/drivers/pci/msi/irqdomain.c
+++ b/drivers/pci/msi/irqdomain.c
@@ -162,6 +162,11 @@ static void pci_irq_mask_msix(struct irq_data *data)
 	pci_msix_mask(irq_data_get_msi_desc(data));
 }
 
+static void pci_irq_mask_nowait_msix(struct irq_data *data)
+{
+	pci_msix_mask_nowait(irq_data_get_msi_desc(data));
+}
+
 static void pci_irq_unmask_msix(struct irq_data *data)
 {
 	pci_msix_unmask(irq_data_get_msi_desc(data));
@@ -182,6 +187,7 @@ static const struct msi_domain_template pci_msix_template = {
 		.irq_startup		= pci_irq_startup_msix,
 		.irq_shutdown		= pci_irq_shutdown_msix,
 		.irq_mask		= pci_irq_mask_msix,
+		.irq_mask_nowait	= pci_irq_mask_nowait_msix,
 		.irq_unmask		= pci_irq_unmask_msix,
 		.irq_write_msi_msg	= pci_msi_domain_write_msg,
 		.flags			= IRQCHIP_ONESHOT_SAFE,
diff --git a/drivers/pci/msi/msi.h b/drivers/pci/msi/msi.h
index 0b420b319f50f..337ff9d9da0ed 100644
--- a/drivers/pci/msi/msi.h
+++ b/drivers/pci/msi/msi.h
@@ -40,10 +40,16 @@ static inline void pci_msix_write_vector_ctrl(struct msi_desc *desc, u32 ctrl)
 		writel(ctrl, desc_addr + PCI_MSIX_ENTRY_VECTOR_CTRL);
 }
 
-static inline void pci_msix_mask(struct msi_desc *desc)
+/* Posted write only: the device may still send an interrupt after return */
+static inline void pci_msix_mask_nowait(struct msi_desc *desc)
 {
 	desc->pci.msix_ctrl |= PCI_MSIX_ENTRY_CTRL_MASKBIT;
 	pci_msix_write_vector_ctrl(desc, desc->pci.msix_ctrl);
+}
+
+static inline void pci_msix_mask(struct msi_desc *desc)
+{
+	pci_msix_mask_nowait(desc);
 	/* Flush write to device */
 	readl(desc->pci.mask_base);
 }
diff --git a/include/linux/interrupt.h b/include/linux/interrupt.h
index 3bf969ad8fe07..d40c992ddb8a1 100644
--- a/include/linux/interrupt.h
+++ b/include/linux/interrupt.h
@@ -231,6 +231,7 @@ extern void devm_free_irq(struct device *dev, unsigned int irq, void *dev_id);
 
 bool irq_has_action(unsigned int irq);
 extern void disable_irq_nosync(unsigned int irq);
+extern void disable_irq_nowait(unsigned int irq);
 extern bool disable_hardirq(unsigned int irq);
 extern void disable_irq(unsigned int irq);
 extern void disable_percpu_irq(unsigned int irq);
diff --git a/include/linux/irq.h b/include/linux/irq.h
index f485369b1b4f7..de095299f20ed 100644
--- a/include/linux/irq.h
+++ b/include/linux/irq.h
@@ -455,6 +455,9 @@ static inline irq_hw_number_t irqd_to_hwirq(struct irq_data *d)
  * @irq_disable:	disable the interrupt
  * @irq_ack:		start of a new interrupt
  * @irq_mask:		mask an interrupt source
+ * @irq_mask_nowait:	mask an interrupt source without waiting for the
+ *			hardware to acknowledge it (optional, falls back to
+ *			@irq_mask)
  * @irq_mask_ack:	ack and mask an interrupt source
  * @irq_unmask:		unmask an interrupt source
  * @irq_eoi:		end of interrupt
@@ -505,6 +508,7 @@ struct irq_chip {
 
 	void		(*irq_ack)(struct irq_data *data);
 	void		(*irq_mask)(struct irq_data *data);
+	void		(*irq_mask_nowait)(struct irq_data *data);
 	void		(*irq_mask_ack)(struct irq_data *data);
 	void		(*irq_unmask)(struct irq_data *data);
 	void		(*irq_eoi)(struct irq_data *data);
diff --git a/kernel/irq/chip.c b/kernel/irq/chip.c
index de754db414d1d..4e76f025b5f68 100644
--- a/kernel/irq/chip.c
+++ b/kernel/irq/chip.c
@@ -399,6 +399,37 @@ void irq_disable(struct irq_desc *desc)
 	__irq_disable(desc, irq_settings_disable_unlazy(desc));
 }
 
+static void mask_irq_nowait(struct irq_desc *desc)
+{
+	if (irqd_irq_masked(&desc->irq_data))
+		return;
+
+	if (desc->irq_data.chip->irq_mask_nowait) {
+		desc->irq_data.chip->irq_mask_nowait(&desc->irq_data);
+		irq_state_set_masked(desc);
+	} else {
+		mask_irq(desc);
+	}
+}
+
+/**
+ * irq_disable_nowait - Mark interrupt disabled and mask it without waiting
+ * @desc:	irq descriptor which should be disabled
+ *
+ * Like irq_disable() with lazy disable turned off, but the hardware may
+ * still deliver an interrupt after this returns. The flow handler deals
+ * with that the same way it does for a lazily disabled interrupt.
+ */
+void irq_disable_nowait(struct irq_desc *desc)
+{
+	if (desc->irq_data.chip->irq_disable) {
+		__irq_disable(desc, true);
+		return;
+	}
+	irq_state_set_disabled(desc);
+	mask_irq_nowait(desc);
+}
+
 void irq_percpu_enable(struct irq_desc *desc, unsigned int cpu)
 {
 	if (desc->irq_data.chip->irq_enable)
diff --git a/kernel/irq/internals.h b/kernel/irq/internals.h
index 0ce21dd454047..3ba13932ba831 100644
--- a/kernel/irq/internals.h
+++ b/kernel/irq/internals.h
@@ -98,6 +98,7 @@ extern void irq_startup_managed(struct irq_desc *desc);
 extern void irq_shutdown(struct irq_desc *desc);
 extern void irq_shutdown_and_deactivate(struct irq_desc *desc);
 extern void irq_disable(struct irq_desc *desc);
+extern void irq_disable_nowait(struct irq_desc *desc);
 extern void irq_percpu_enable(struct irq_desc *desc, unsigned int cpu);
 extern void irq_percpu_disable(struct irq_desc *desc, unsigned int cpu);
 extern void mask_irq(struct irq_desc *desc);
diff --git a/kernel/irq/manage.c b/kernel/irq/manage.c
index 2fbff2618a1e2..ecb4d2144b9d5 100644
--- a/kernel/irq/manage.c
+++ b/kernel/irq/manage.c
@@ -705,6 +705,27 @@ void disable_irq_nosync(unsigned int irq)
 }
 EXPORT_SYMBOL(disable_irq_nosync);
 
+/**
+ * disable_irq_nowait - mask an irq without waiting for the hardware
+ * @irq: Interrupt to disable
+ *
+ * Like disable_irq_nosync(), but masks the interrupt at the chip right away
+ * instead of lazily, and does not wait for the hardware to acknowledge the
+ * mask. An interrupt that still arrives is held pending and replayed by
+ * enable_irq(). Use this when an extra interrupt is cheaper than flushing
+ * the mask, e.g. switching from interrupts to polling.
+ *
+ * This function may be called from IRQ context.
+ */
+void disable_irq_nowait(unsigned int irq)
+{
+	scoped_irqdesc_get_and_buslock(irq, IRQ_GET_DESC_CHECK_GLOBAL) {
+		if (!scoped_irqdesc->depth++)
+			irq_disable_nowait(scoped_irqdesc);
+	}
+}
+EXPORT_SYMBOL_GPL(disable_irq_nowait);
+
 /**
  * disable_irq - disable an irq and wait for completion
  * @irq: Interrupt to disable
--

  reply	other threads:[~2026-10-09 15:30 UTC|newest]

Thread overview: 11+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-09  5:05 [RFC PATCH 0/3] nvme-pci: yield completions under scheduler pressure Naman Jain
2026-10-09  5:05 ` [RFC PATCH 1/3] nvme-pci: select IRQ_POLL Naman Jain
2026-10-09  5:05 ` [RFC PATCH 2/3] nvme-pci: make completion queue polling softirq-safe Naman Jain
2026-10-09  5:05 ` [RFC PATCH 3/3] nvme-pci: defer completions when rescheduling is needed Naman Jain
2026-10-09 15:30   ` Keith Busch [this message]
2026-10-09  5:54 ` [RFC PATCH 0/3] nvme-pci: yield completions under scheduler pressure Michael Kelley
2026-10-09  6:51   ` Naman Jain
2026-10-09 10:16     ` Naman Jain
2026-10-09 11:27       ` Luigi Rizzo
2026-10-09 14:58         ` Luigi Rizzo
2026-10-09 11:36       ` Fengnan

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=askIoNon8UVyNJiJ@kbusch-mbp \
    --to=kbusch@kernel.org \
    --cc=axboe@kernel.dk \
    --cc=hch@lst.de \
    --cc=linux-hyperv@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-nvme@lists.infradead.org \
    --cc=mhklinux@outlook.com \
    --cc=namjain@linux.microsoft.com \
    --cc=sagi@grimberg.me \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®