mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Shameer Kolothum <skolothumtho@nvidia.com>
To: <kvm@vger.kernel.org>, <linux-pci@vger.kernel.org>,
	<linux-kernel@vger.kernel.org>
Cc: <alex@shazbot.org>, <jgg@ziepe.ca>, <kevin.tian@intel.com>,
	<kbusch@meta.com>, <michal.winiarski@intel.com>,
	<satyanarayana.k.v.p@intel.com>, <sonangp@nvidia.com>,
	<ankita@nvidia.com>, <nathanc@nvidia.com>, <mochs@nvidia.com>,
	<skolothumtho@nvidia.com>
Subject: [RFC PATCH v2 01/16] vfio/pci: Add a device access gate
Date: Tue, 29 Sep 2026 18:32:50 +0100	[thread overview]
Message-ID: <20260929173305.204856-2-skolothumtho@nvidia.com> (raw)
In-Reply-To: <20260929173305.204856-1-skolothumtho@nvidia.com>

PCI error recovery needs to block new device accesses and drain
outstanding accesses before changing device state.

Add an SRCU-based access gate, enabled by pci_recovery_supported.
Serialize open and close state updates with access_lock.

Assisted-by: LLM
Signed-off-by: Shameer Kolothum <skolothumtho@nvidia.com>
---
 drivers/vfio/pci/vfio_pci_priv.h |  3 ++
 include/linux/vfio_pci_core.h    | 19 ++++++++++++
 drivers/vfio/pci/vfio_pci_core.c | 53 ++++++++++++++++++++++++++++++++
 3 files changed, 75 insertions(+)

diff --git a/drivers/vfio/pci/vfio_pci_priv.h b/drivers/vfio/pci/vfio_pci_priv.h
index 4e7162234a2e..623e73f379cc 100644
--- a/drivers/vfio/pci/vfio_pci_priv.h
+++ b/drivers/vfio/pci/vfio_pci_priv.h
@@ -73,6 +73,9 @@ u16 vfio_pci_memory_lock_and_enable(struct vfio_pci_core_device *vdev);
 void vfio_pci_memory_unlock_and_restore(struct vfio_pci_core_device *vdev,
 					u16 cmd);
 
+int vfio_pci_core_access_begin(struct vfio_pci_core_device *vdev);
+void vfio_pci_core_access_end(struct vfio_pci_core_device *vdev, int idx);
+
 #ifdef CONFIG_VFIO_PCI_IGD
 bool vfio_pci_is_intel_display(struct pci_dev *pdev);
 int vfio_pci_igd_init(struct vfio_pci_core_device *vdev);
diff --git a/include/linux/vfio_pci_core.h b/include/linux/vfio_pci_core.h
index 9a1674c152aa..de0993280344 100644
--- a/include/linux/vfio_pci_core.h
+++ b/include/linux/vfio_pci_core.h
@@ -13,6 +13,7 @@
 #include <linux/vfio.h>
 #include <linux/irqbypass.h>
 #include <linux/rcupdate.h>
+#include <linux/srcu.h>
 #include <linux/types.h>
 #include <linux/uuid.h>
 #include <linux/notifier.h>
@@ -129,6 +130,7 @@ struct vfio_pci_core_device {
 	bool			disable_idle_d3:1;
 	bool			nointxmask:1;
 	bool			disable_vga:1;
+	bool			pci_recovery_supported:1;
 	/* Flags modified at runtime - dedicated storage unit */
 	bool			needs_reset;
 	bool			pm_intx_masked;
@@ -148,6 +150,23 @@ struct vfio_pci_core_device {
 	struct vfio_pci_core_device	*sriov_pf_core_dev;
 	struct notifier_block	nb;
 	struct rw_semaphore	memory_lock;
+	/*
+	 * Device accesses take access_srcu and check device_open and
+	 * access_blocked. To block access, set access_blocked and wait for
+	 * existing readers with synchronize_srcu(). New accesses fail with
+	 * -EIO while blocked.
+	 *
+	 * Close clears access_blocked before device_open.
+	 *
+	 * INTx paths use irqlock instead of access_srcu to coordinate with
+	 * recovery. They read pci_recovery_enabled and access_blocked with
+	 * READ_ONCE(), since writers do not always hold irqlock.
+	 */
+	struct srcu_struct	access_srcu;
+	bool			access_blocked;
+	/* Set after open completes, cleared before close tears down state. */
+	bool			device_open;
+	struct mutex		access_lock;	/* gate flag writers */
 	struct list_head	dmabufs;
 };
 
diff --git a/drivers/vfio/pci/vfio_pci_core.c b/drivers/vfio/pci/vfio_pci_core.c
index 6757054e9d87..88a78766c178 100644
--- a/drivers/vfio/pci/vfio_pci_core.c
+++ b/drivers/vfio/pci/vfio_pci_core.c
@@ -662,10 +662,18 @@ int vfio_pci_core_enable(struct vfio_pci_core_device *vdev)
 	if (!vfio_vga_disabled(vdev) && vfio_pci_is_vga(pdev))
 		vdev->has_vga = true;
 
+	if (vdev->pci_recovery_supported) {
+		ret = init_srcu_struct(&vdev->access_srcu);
+		if (ret)
+			goto out_free_config;
+	}
+
 	vfio_pci_core_map_bars(vdev);
 
 	return 0;
 
+out_free_config:
+	vfio_config_free(vdev);
 out_free_zdev:
 	vfio_pci_zdev_close_device(vdev);
 out_free_state:
@@ -812,6 +820,9 @@ void vfio_pci_core_disable(struct vfio_pci_core_device *vdev)
 	/* Put the pm-runtime usage counter acquired during enable */
 	if (!vdev->disable_idle_d3)
 		pm_runtime_put(&pdev->dev);
+
+	if (vdev->pci_recovery_supported)
+		cleanup_srcu_struct(&vdev->access_srcu);
 }
 EXPORT_SYMBOL_GPL(vfio_pci_core_disable);
 
@@ -820,6 +831,13 @@ void vfio_pci_core_close_device(struct vfio_device *core_vdev)
 	struct vfio_pci_core_device *vdev =
 		container_of(core_vdev, struct vfio_pci_core_device, vdev);
 
+	if (vdev->pci_recovery_supported) {
+		scoped_guard(mutex, &vdev->access_lock) {
+			WRITE_ONCE(vdev->access_blocked, false);
+			WRITE_ONCE(vdev->device_open, false);
+		}
+	}
+
 	if (vdev->sriov_pf_core_dev) {
 		mutex_lock(&vdev->sriov_pf_core_dev->vf_token->lock);
 		WARN_ON(!vdev->sriov_pf_core_dev->vf_token->users);
@@ -852,6 +870,12 @@ void vfio_pci_core_finish_enable(struct vfio_pci_core_device *vdev)
 		vdev->sriov_pf_core_dev->vf_token->users++;
 		mutex_unlock(&vdev->sriov_pf_core_dev->vf_token->lock);
 	}
+
+	if (vdev->pci_recovery_supported) {
+		guard(mutex)(&vdev->access_lock);
+		WRITE_ONCE(vdev->access_blocked, false);
+		WRITE_ONCE(vdev->device_open, true);
+	}
 }
 EXPORT_SYMBOL_GPL(vfio_pci_core_finish_enable);
 
@@ -1680,6 +1704,33 @@ static ssize_t vfio_pci_rw(struct vfio_pci_core_device *vdev, char __user *buf,
 	return ret;
 }
 
+/*
+ * Return an SRCU index for access_end(), or -EIO if access is blocked
+ * or the device is closed. Do not wait for recovery.
+ */
+int vfio_pci_core_access_begin(struct vfio_pci_core_device *vdev)
+{
+	int idx;
+
+	if (!vdev->pci_recovery_supported)
+		return 0;
+
+	idx = srcu_read_lock(&vdev->access_srcu);
+	if (unlikely(!READ_ONCE(vdev->device_open) ||
+		     READ_ONCE(vdev->access_blocked))) {
+		srcu_read_unlock(&vdev->access_srcu, idx);
+		return -EIO;
+	}
+
+	return idx;
+}
+
+void vfio_pci_core_access_end(struct vfio_pci_core_device *vdev, int idx)
+{
+	if (vdev->pci_recovery_supported)
+		srcu_read_unlock(&vdev->access_srcu, idx);
+}
+
 ssize_t vfio_pci_core_read(struct vfio_device *core_vdev, char __user *buf,
 		size_t count, loff_t *ppos)
 {
@@ -2196,6 +2247,7 @@ int vfio_pci_core_init_dev(struct vfio_device *core_vdev)
 	if (ret && ret != -EOPNOTSUPP)
 		return ret;
 	INIT_LIST_HEAD(&vdev->dmabufs);
+	mutex_init(&vdev->access_lock);
 	init_rwsem(&vdev->memory_lock);
 	xa_init(&vdev->ctx);
 
@@ -2210,6 +2262,7 @@ void vfio_pci_core_release_dev(struct vfio_device *core_vdev)
 
 	mutex_destroy(&vdev->igate);
 	mutex_destroy(&vdev->ioeventfds_lock);
+	mutex_destroy(&vdev->access_lock);
 	kfree(vdev->region);
 	kfree(vdev->pm_save);
 }
-- 
2.43.0


  reply	other threads:[~2026-09-29 17:34 UTC|newest]

Thread overview: 17+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-29 17:32 [RFC PATCH v2 00/16] vfio/pci: Handle PCI error recovery and report state to userspace Shameer Kolothum
2026-09-29 17:32 ` Shameer Kolothum [this message]
2026-09-29 17:32 ` [RFC PATCH v2 02/16] vfio/pci: Gate config space access Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 03/16] vfio/pci: Buffer ROM reads before copying to userspace Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 04/16] vfio/pci: Gate BAR and ROM access Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 05/16] vfio/pci: Fail BAR faults while access is blocked Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 06/16] vfio/pci: Gate interrupt configuration Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 07/16] vfio/pci: Gate function reset and runtime power management Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 08/16] vfio/pci: Gate device information queries and DMA-BUF export Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 09/16] vfio/pci: Add PCI error recovery state Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 10/16] vfio/pci: Quiesce INTx while access is blocked Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 11/16] vfio/pci: Add INTx recovery start and finish helpers Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 12/16] vfio/pci: Restore device state from slot_reset() Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 13/16] vfio/pci: Complete recovery in resume() Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 14/16] vfio/pci: Block device access during host recovery Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 15/16] vfio/pci: Add VFIO_DEVICE_FEATURE_PCI_ERROR_RECOVERY Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 16/16] vfio/pci: Enable host PCI error recovery for vfio-pci Shameer Kolothum

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260929173305.204856-2-skolothumtho@nvidia.com \
    --to=skolothumtho@nvidia.com \
    --cc=alex@shazbot.org \
    --cc=ankita@nvidia.com \
    --cc=jgg@ziepe.ca \
    --cc=kbusch@meta.com \
    --cc=kevin.tian@intel.com \
    --cc=kvm@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-pci@vger.kernel.org \
    --cc=michal.winiarski@intel.com \
    --cc=mochs@nvidia.com \
    --cc=nathanc@nvidia.com \
    --cc=satyanarayana.k.v.p@intel.com \
    --cc=sonangp@nvidia.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®