From: Shameer Kolothum <skolothumtho@nvidia.com>
To: <kvm@vger.kernel.org>, <linux-pci@vger.kernel.org>,
<linux-kernel@vger.kernel.org>
Cc: <alex@shazbot.org>, <jgg@ziepe.ca>, <kevin.tian@intel.com>,
<kbusch@meta.com>, <michal.winiarski@intel.com>,
<satyanarayana.k.v.p@intel.com>, <sonangp@nvidia.com>,
<ankita@nvidia.com>, <nathanc@nvidia.com>, <mochs@nvidia.com>,
<skolothumtho@nvidia.com>
Subject: [RFC PATCH v2 01/16] vfio/pci: Add a device access gate
Date: Tue, 29 Sep 2026 18:32:50 +0100 [thread overview]
Message-ID: <20260929173305.204856-2-skolothumtho@nvidia.com> (raw)
In-Reply-To: <20260929173305.204856-1-skolothumtho@nvidia.com>
PCI error recovery needs to block new device accesses and drain
outstanding accesses before changing device state.
Add an SRCU-based access gate, enabled by pci_recovery_supported.
Serialize open and close state updates with access_lock.
Assisted-by: LLM
Signed-off-by: Shameer Kolothum <skolothumtho@nvidia.com>
---
drivers/vfio/pci/vfio_pci_priv.h | 3 ++
include/linux/vfio_pci_core.h | 19 ++++++++++++
drivers/vfio/pci/vfio_pci_core.c | 53 ++++++++++++++++++++++++++++++++
3 files changed, 75 insertions(+)
diff --git a/drivers/vfio/pci/vfio_pci_priv.h b/drivers/vfio/pci/vfio_pci_priv.h
index 4e7162234a2e..623e73f379cc 100644
--- a/drivers/vfio/pci/vfio_pci_priv.h
+++ b/drivers/vfio/pci/vfio_pci_priv.h
@@ -73,6 +73,9 @@ u16 vfio_pci_memory_lock_and_enable(struct vfio_pci_core_device *vdev);
void vfio_pci_memory_unlock_and_restore(struct vfio_pci_core_device *vdev,
u16 cmd);
+int vfio_pci_core_access_begin(struct vfio_pci_core_device *vdev);
+void vfio_pci_core_access_end(struct vfio_pci_core_device *vdev, int idx);
+
#ifdef CONFIG_VFIO_PCI_IGD
bool vfio_pci_is_intel_display(struct pci_dev *pdev);
int vfio_pci_igd_init(struct vfio_pci_core_device *vdev);
diff --git a/include/linux/vfio_pci_core.h b/include/linux/vfio_pci_core.h
index 9a1674c152aa..de0993280344 100644
--- a/include/linux/vfio_pci_core.h
+++ b/include/linux/vfio_pci_core.h
@@ -13,6 +13,7 @@
#include <linux/vfio.h>
#include <linux/irqbypass.h>
#include <linux/rcupdate.h>
+#include <linux/srcu.h>
#include <linux/types.h>
#include <linux/uuid.h>
#include <linux/notifier.h>
@@ -129,6 +130,7 @@ struct vfio_pci_core_device {
bool disable_idle_d3:1;
bool nointxmask:1;
bool disable_vga:1;
+ bool pci_recovery_supported:1;
/* Flags modified at runtime - dedicated storage unit */
bool needs_reset;
bool pm_intx_masked;
@@ -148,6 +150,23 @@ struct vfio_pci_core_device {
struct vfio_pci_core_device *sriov_pf_core_dev;
struct notifier_block nb;
struct rw_semaphore memory_lock;
+ /*
+ * Device accesses take access_srcu and check device_open and
+ * access_blocked. To block access, set access_blocked and wait for
+ * existing readers with synchronize_srcu(). New accesses fail with
+ * -EIO while blocked.
+ *
+ * Close clears access_blocked before device_open.
+ *
+ * INTx paths use irqlock instead of access_srcu to coordinate with
+ * recovery. They read pci_recovery_enabled and access_blocked with
+ * READ_ONCE(), since writers do not always hold irqlock.
+ */
+ struct srcu_struct access_srcu;
+ bool access_blocked;
+ /* Set after open completes, cleared before close tears down state. */
+ bool device_open;
+ struct mutex access_lock; /* gate flag writers */
struct list_head dmabufs;
};
diff --git a/drivers/vfio/pci/vfio_pci_core.c b/drivers/vfio/pci/vfio_pci_core.c
index 6757054e9d87..88a78766c178 100644
--- a/drivers/vfio/pci/vfio_pci_core.c
+++ b/drivers/vfio/pci/vfio_pci_core.c
@@ -662,10 +662,18 @@ int vfio_pci_core_enable(struct vfio_pci_core_device *vdev)
if (!vfio_vga_disabled(vdev) && vfio_pci_is_vga(pdev))
vdev->has_vga = true;
+ if (vdev->pci_recovery_supported) {
+ ret = init_srcu_struct(&vdev->access_srcu);
+ if (ret)
+ goto out_free_config;
+ }
+
vfio_pci_core_map_bars(vdev);
return 0;
+out_free_config:
+ vfio_config_free(vdev);
out_free_zdev:
vfio_pci_zdev_close_device(vdev);
out_free_state:
@@ -812,6 +820,9 @@ void vfio_pci_core_disable(struct vfio_pci_core_device *vdev)
/* Put the pm-runtime usage counter acquired during enable */
if (!vdev->disable_idle_d3)
pm_runtime_put(&pdev->dev);
+
+ if (vdev->pci_recovery_supported)
+ cleanup_srcu_struct(&vdev->access_srcu);
}
EXPORT_SYMBOL_GPL(vfio_pci_core_disable);
@@ -820,6 +831,13 @@ void vfio_pci_core_close_device(struct vfio_device *core_vdev)
struct vfio_pci_core_device *vdev =
container_of(core_vdev, struct vfio_pci_core_device, vdev);
+ if (vdev->pci_recovery_supported) {
+ scoped_guard(mutex, &vdev->access_lock) {
+ WRITE_ONCE(vdev->access_blocked, false);
+ WRITE_ONCE(vdev->device_open, false);
+ }
+ }
+
if (vdev->sriov_pf_core_dev) {
mutex_lock(&vdev->sriov_pf_core_dev->vf_token->lock);
WARN_ON(!vdev->sriov_pf_core_dev->vf_token->users);
@@ -852,6 +870,12 @@ void vfio_pci_core_finish_enable(struct vfio_pci_core_device *vdev)
vdev->sriov_pf_core_dev->vf_token->users++;
mutex_unlock(&vdev->sriov_pf_core_dev->vf_token->lock);
}
+
+ if (vdev->pci_recovery_supported) {
+ guard(mutex)(&vdev->access_lock);
+ WRITE_ONCE(vdev->access_blocked, false);
+ WRITE_ONCE(vdev->device_open, true);
+ }
}
EXPORT_SYMBOL_GPL(vfio_pci_core_finish_enable);
@@ -1680,6 +1704,33 @@ static ssize_t vfio_pci_rw(struct vfio_pci_core_device *vdev, char __user *buf,
return ret;
}
+/*
+ * Return an SRCU index for access_end(), or -EIO if access is blocked
+ * or the device is closed. Do not wait for recovery.
+ */
+int vfio_pci_core_access_begin(struct vfio_pci_core_device *vdev)
+{
+ int idx;
+
+ if (!vdev->pci_recovery_supported)
+ return 0;
+
+ idx = srcu_read_lock(&vdev->access_srcu);
+ if (unlikely(!READ_ONCE(vdev->device_open) ||
+ READ_ONCE(vdev->access_blocked))) {
+ srcu_read_unlock(&vdev->access_srcu, idx);
+ return -EIO;
+ }
+
+ return idx;
+}
+
+void vfio_pci_core_access_end(struct vfio_pci_core_device *vdev, int idx)
+{
+ if (vdev->pci_recovery_supported)
+ srcu_read_unlock(&vdev->access_srcu, idx);
+}
+
ssize_t vfio_pci_core_read(struct vfio_device *core_vdev, char __user *buf,
size_t count, loff_t *ppos)
{
@@ -2196,6 +2247,7 @@ int vfio_pci_core_init_dev(struct vfio_device *core_vdev)
if (ret && ret != -EOPNOTSUPP)
return ret;
INIT_LIST_HEAD(&vdev->dmabufs);
+ mutex_init(&vdev->access_lock);
init_rwsem(&vdev->memory_lock);
xa_init(&vdev->ctx);
@@ -2210,6 +2262,7 @@ void vfio_pci_core_release_dev(struct vfio_device *core_vdev)
mutex_destroy(&vdev->igate);
mutex_destroy(&vdev->ioeventfds_lock);
+ mutex_destroy(&vdev->access_lock);
kfree(vdev->region);
kfree(vdev->pm_save);
}
--
2.43.0
next prev parent reply other threads:[~2026-09-29 17:34 UTC|newest]
Thread overview: 17+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-29 17:32 [RFC PATCH v2 00/16] vfio/pci: Handle PCI error recovery and report state to userspace Shameer Kolothum
2026-09-29 17:32 ` Shameer Kolothum [this message]
2026-09-29 17:32 ` [RFC PATCH v2 02/16] vfio/pci: Gate config space access Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 03/16] vfio/pci: Buffer ROM reads before copying to userspace Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 04/16] vfio/pci: Gate BAR and ROM access Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 05/16] vfio/pci: Fail BAR faults while access is blocked Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 06/16] vfio/pci: Gate interrupt configuration Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 07/16] vfio/pci: Gate function reset and runtime power management Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 08/16] vfio/pci: Gate device information queries and DMA-BUF export Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 09/16] vfio/pci: Add PCI error recovery state Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 10/16] vfio/pci: Quiesce INTx while access is blocked Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 11/16] vfio/pci: Add INTx recovery start and finish helpers Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 12/16] vfio/pci: Restore device state from slot_reset() Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 13/16] vfio/pci: Complete recovery in resume() Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 14/16] vfio/pci: Block device access during host recovery Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 15/16] vfio/pci: Add VFIO_DEVICE_FEATURE_PCI_ERROR_RECOVERY Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 16/16] vfio/pci: Enable host PCI error recovery for vfio-pci Shameer Kolothum
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260929173305.204856-2-skolothumtho@nvidia.com \
--to=skolothumtho@nvidia.com \
--cc=alex@shazbot.org \
--cc=ankita@nvidia.com \
--cc=jgg@ziepe.ca \
--cc=kbusch@meta.com \
--cc=kevin.tian@intel.com \
--cc=kvm@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-pci@vger.kernel.org \
--cc=michal.winiarski@intel.com \
--cc=mochs@nvidia.com \
--cc=nathanc@nvidia.com \
--cc=satyanarayana.k.v.p@intel.com \
--cc=sonangp@nvidia.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®