From: Shameer Kolothum <skolothumtho@nvidia.com>
To: <kvm@vger.kernel.org>, <linux-pci@vger.kernel.org>,
<linux-kernel@vger.kernel.org>
Cc: <alex@shazbot.org>, <jgg@ziepe.ca>, <kevin.tian@intel.com>,
<kbusch@meta.com>, <michal.winiarski@intel.com>,
<satyanarayana.k.v.p@intel.com>, <sonangp@nvidia.com>,
<ankita@nvidia.com>, <nathanc@nvidia.com>, <mochs@nvidia.com>,
<skolothumtho@nvidia.com>
Subject: [RFC PATCH v2 14/16] vfio/pci: Block device access during host recovery
Date: Tue, 29 Sep 2026 18:33:03 +0100 [thread overview]
Message-ID: <20260929173305.204856-15-skolothumtho@nvidia.com> (raw)
In-Reply-To: <20260929173305.204856-1-skolothumtho@nvidia.com>
For userspace that enables recovery, extend error_detected() to block
device access and drain existing SRCU readers. Mask INTx, revoke BAR
mappings and DMA-BUFs, and disable bus mastering on a normal channel.
Return CAN_RECOVER for a normal channel, NEED_RESET for a frozen channel
and DISCONNECT for permanent failure. If saving PCI_COMMAND or disabling
bus mastering fails, return NONE and leave access blocked. Preserve the
original command and sequence number across nested events.
Track host recovery across close and reopen, including without userspace
opt-in. Reject open during a recorded transaction or after disconnection.
Clear deferred power requests at close and keep DMA-BUFs revoked while
access is blocked.
Reject hot reset for devices in ongoing recovery, but allow failed
devices so healthy siblings can be reset.
Reject userspace SR-IOV configuration while access is blocked.
Assisted-by: LLM
Signed-off-by: Shameer Kolothum <skolothumtho@nvidia.com>
---
Note: The existing open/close paths are not serialized against the full
host recovery transaction. This series blocks reopening during an
already recorded transaction, but recovery can still start after the
open check, and close can overlap the physical reset between callbacks.
See the cover letter's "Locking and open questions" section.
---
drivers/vfio/pci/vfio_pci.c | 3 +
drivers/vfio/pci/vfio_pci_core.c | 166 ++++++++++++++++++++++++++++-
drivers/vfio/pci/vfio_pci_dmabuf.c | 5 +
3 files changed, 173 insertions(+), 1 deletion(-)
diff --git a/drivers/vfio/pci/vfio_pci.c b/drivers/vfio/pci/vfio_pci.c
index 830369ff878d..4887dd062c77 100644
--- a/drivers/vfio/pci/vfio_pci.c
+++ b/drivers/vfio/pci/vfio_pci.c
@@ -213,6 +213,9 @@ static int vfio_pci_sriov_configure(struct pci_dev *pdev, int nr_virtfn)
if (!enable_sriov)
return -ENOENT;
+ if (vdev->pci_recovery_supported && READ_ONCE(vdev->access_blocked))
+ return -EIO;
+
return vfio_pci_core_sriov_configure(vdev, nr_virtfn);
}
diff --git a/drivers/vfio/pci/vfio_pci_core.c b/drivers/vfio/pci/vfio_pci_core.c
index d6cc34240b25..2ec8022e4a69 100644
--- a/drivers/vfio/pci/vfio_pci_core.c
+++ b/drivers/vfio/pci/vfio_pci_core.c
@@ -617,6 +617,15 @@ int vfio_pci_core_enable(struct vfio_pci_core_device *vdev)
u16 cmd;
u8 msix_pos;
+ if (vdev->pci_recovery_supported) {
+ if (pci_dev_is_disconnected(pdev))
+ return -ENODEV;
+ scoped_guard(mutex, &vdev->access_lock) {
+ if (vdev->pci_recovery_host_active)
+ return -EBUSY;
+ }
+ }
+
if (!vdev->disable_idle_d3) {
ret = pm_runtime_resume_and_get(&pdev->dev);
if (ret < 0)
@@ -860,6 +869,10 @@ void vfio_pci_core_close_device(struct vfio_device *core_vdev)
WRITE_ONCE(vdev->access_blocked, false);
WRITE_ONCE(vdev->device_open, false);
WRITE_ONCE(vdev->pci_recovery_flags, 0);
+ /* Discard pending D0 so it cannot be replayed after reopen. */
+ down_write(&vdev->memory_lock);
+ vdev->power_up_pending = false;
+ up_write(&vdev->memory_lock);
}
}
@@ -1843,6 +1856,15 @@ void vfio_pci_core_access_end(struct vfio_pci_core_device *vdev, int idx)
srcu_read_unlock(&vdev->access_srcu, idx);
}
+/* Reject new accesses and wait for existing SRCU readers. */
+static void vfio_pci_core_block_access(struct vfio_pci_core_device *vdev)
+{
+ lockdep_assert_held(&vdev->access_lock);
+
+ WRITE_ONCE(vdev->access_blocked, true);
+ synchronize_srcu(&vdev->access_srcu);
+}
+
ssize_t vfio_pci_core_read(struct vfio_device *core_vdev, char __user *buf,
size_t count, loff_t *ppos)
{
@@ -2521,19 +2543,147 @@ vfio_pci_signal_recovery_event(struct vfio_pci_core_device *vdev)
rcu_read_unlock();
}
+/*
+ * Save PCI_COMMAND, clear bus mastering and set the requested bits.
+ * Reject an all-ones read from an inaccessible device.
+ */
+static int
+vfio_pci_recovery_save_and_clear_master(struct vfio_pci_core_device *vdev, u16 set)
+{
+ struct pci_dev *pdev = vdev->pdev;
+ u16 command;
+ int ret;
+
+ ret = pci_read_config_word(pdev, PCI_COMMAND, &vdev->pci_recovery_command);
+ if (ret)
+ return ret;
+ if (PCI_POSSIBLE_ERROR(vdev->pci_recovery_command))
+ return -EIO;
+
+ command = (vdev->pci_recovery_command & ~PCI_COMMAND_MASTER) | set;
+ ret = pci_write_config_word(pdev, PCI_COMMAND, command);
+ if (ret)
+ return ret;
+
+ vdev->pci_recovery_command_valid = true;
+ return 0;
+}
+
pci_ers_result_t vfio_pci_core_aer_err_detected(struct pci_dev *pdev,
pci_channel_state_t state)
{
struct vfio_pci_core_device *vdev = dev_get_drvdata(&pdev->dev);
struct vfio_pci_eventfd *eventfd;
+ pci_ers_result_t result = PCI_ERS_RESULT_CAN_RECOVER;
+ unsigned long irq_flags;
+ bool notify_recovery = false;
+ bool nested;
+ u32 flags;
+ int ret = 0;
+
+ if (!vdev->pci_recovery_supported)
+ goto out;
+
+ mutex_lock(&vdev->access_lock);
+ /*
+ * Track the host transaction even when userspace has not enabled
+ * recovery. Reopen must wait for resume() or permanent failure.
+ */
+ vdev->pci_recovery_host_active =
+ state != pci_channel_io_perm_failure;
+ if (!vdev->pci_recovery_enabled)
+ goto out_unlock;
+ /* Keep failed devices blocked until close. */
+ if (vdev->pci_recovery_flags & VFIO_PCI_RECOVERY_FAILED) {
+ result = PCI_ERS_RESULT_NONE;
+ goto out_unlock;
+ }
+
+ if (!vdev->device_open) {
+ result = PCI_ERS_RESULT_NONE;
+ goto out_unlock;
+ }
+
+ notify_recovery = true;
+ vfio_pci_core_block_access(vdev);
+ /*
+ * Preserve the saved command for nested events. The current value
+ * may already have bus mastering disabled by recovery.
+ */
+ nested = vdev->pci_recovery_flags & VFIO_PCI_RECOVERY_IN_PROGRESS;
+ if (!nested)
+ vdev->pci_recovery_command_valid = false;
+ vfio_pci_intx_recovery_start(vdev);
+ vfio_pci_zap_and_down_write_memory_lock(vdev);
+ /*
+ * Serialize PCI_COMMAND updates with the INTx handler, which does
+ * not take access_srcu.
+ */
+ spin_lock_irqsave(&vdev->irqlock, irq_flags);
+ if (state == pci_channel_io_normal && vdev->pci_2_3 && !nested)
+ ret = vfio_pci_recovery_save_and_clear_master(
+ vdev, PCI_COMMAND_INTX_DISABLE);
+ spin_unlock_irqrestore(&vdev->irqlock, irq_flags);
+ vfio_pci_dma_buf_move(vdev, true);
+
+ /*
+ * Start a new sequence unless this event belongs to an active
+ * transaction. Publish the flags in one store to avoid exposing
+ * a transient cleared state to lockless readers.
+ */
+ if (nested) {
+ flags = vdev->pci_recovery_flags;
+ } else {
+ if (++vdev->pci_recovery_sequence == 0)
+ vdev->pci_recovery_sequence++;
+ flags = 0;
+ }
+
+ if (state == pci_channel_io_perm_failure) {
+ WRITE_ONCE(vdev->pci_recovery_flags,
+ (flags | VFIO_PCI_RECOVERY_FAILED) &
+ ~VFIO_PCI_RECOVERY_IN_PROGRESS);
+ vdev->pci_recovery_command_valid = false;
+ result = PCI_ERS_RESULT_DISCONNECT;
+ goto out_memory;
+ }
+
+ if (state == pci_channel_io_frozen) {
+ WRITE_ONCE(vdev->pci_recovery_flags,
+ flags | VFIO_PCI_RECOVERY_IN_PROGRESS |
+ VFIO_PCI_RECOVERY_FROZEN);
+ result = PCI_ERS_RESULT_NEED_RESET;
+ goto out_memory;
+ }
+
+ if (!ret && !vdev->pci_2_3 && !nested)
+ ret = vfio_pci_recovery_save_and_clear_master(vdev, 0);
+
+ if (ret) {
+ flags |= VFIO_PCI_RECOVERY_FAILED;
+ flags &= ~VFIO_PCI_RECOVERY_IN_PROGRESS;
+ result = PCI_ERS_RESULT_NONE;
+ } else {
+ flags |= VFIO_PCI_RECOVERY_IN_PROGRESS;
+ }
+ WRITE_ONCE(vdev->pci_recovery_flags, flags);
+
+out_memory:
+ up_write(&vdev->memory_lock);
+out_unlock:
+ mutex_unlock(&vdev->access_lock);
+
+out:
rcu_read_lock();
eventfd = rcu_dereference(vdev->err_trigger);
if (eventfd)
eventfd_signal(eventfd->ctx);
rcu_read_unlock();
+ if (notify_recovery)
+ vfio_pci_signal_recovery_event(vdev);
- return PCI_ERS_RESULT_CAN_RECOVER;
+ return result;
}
EXPORT_SYMBOL_GPL(vfio_pci_core_aer_err_detected);
@@ -2926,6 +3076,20 @@ static int vfio_pci_dev_set_hot_reset(struct vfio_device_set *dev_set,
break;
}
+ /*
+ * Recovery releases memory_lock between callbacks. Check the
+ * access block after taking the lock as well. Allow failed
+ * devices so they do not prevent resetting healthy siblings.
+ */
+ if (vdev->pci_recovery_supported &&
+ READ_ONCE(vdev->access_blocked) &&
+ !(READ_ONCE(vdev->pci_recovery_flags) &
+ VFIO_PCI_RECOVERY_FAILED)) {
+ up_write(&vdev->memory_lock);
+ ret = -EBUSY;
+ break;
+ }
+
vfio_pci_dma_buf_move(vdev, true);
vfio_pci_zap_bars(vdev);
}
diff --git a/drivers/vfio/pci/vfio_pci_dmabuf.c b/drivers/vfio/pci/vfio_pci_dmabuf.c
index c1a0250af680..69c41c3146f8 100644
--- a/drivers/vfio/pci/vfio_pci_dmabuf.c
+++ b/drivers/vfio/pci/vfio_pci_dmabuf.c
@@ -350,6 +350,11 @@ void vfio_pci_dma_buf_move(struct vfio_pci_core_device *vdev, bool revoked)
lockdep_assert_held_write(&vdev->memory_lock);
+ /* Reset and power-state cleanup must not undo recovery revocation. */
+ if (!revoked && vdev->pci_recovery_supported &&
+ READ_ONCE(vdev->access_blocked))
+ return;
+
list_for_each_entry_safe(priv, tmp, &vdev->dmabufs, dmabufs_elm) {
if (!get_file_active(&priv->dmabuf->file))
continue;
--
2.43.0
next prev parent reply other threads:[~2026-09-29 17:34 UTC|newest]
Thread overview: 17+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-29 17:32 [RFC PATCH v2 00/16] vfio/pci: Handle PCI error recovery and report state to userspace Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 01/16] vfio/pci: Add a device access gate Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 02/16] vfio/pci: Gate config space access Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 03/16] vfio/pci: Buffer ROM reads before copying to userspace Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 04/16] vfio/pci: Gate BAR and ROM access Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 05/16] vfio/pci: Fail BAR faults while access is blocked Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 06/16] vfio/pci: Gate interrupt configuration Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 07/16] vfio/pci: Gate function reset and runtime power management Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 08/16] vfio/pci: Gate device information queries and DMA-BUF export Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 09/16] vfio/pci: Add PCI error recovery state Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 10/16] vfio/pci: Quiesce INTx while access is blocked Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 11/16] vfio/pci: Add INTx recovery start and finish helpers Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 12/16] vfio/pci: Restore device state from slot_reset() Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 13/16] vfio/pci: Complete recovery in resume() Shameer Kolothum
2026-09-29 17:33 ` Shameer Kolothum [this message]
2026-09-29 17:33 ` [RFC PATCH v2 15/16] vfio/pci: Add VFIO_DEVICE_FEATURE_PCI_ERROR_RECOVERY Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 16/16] vfio/pci: Enable host PCI error recovery for vfio-pci Shameer Kolothum
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260929173305.204856-15-skolothumtho@nvidia.com \
--to=skolothumtho@nvidia.com \
--cc=alex@shazbot.org \
--cc=ankita@nvidia.com \
--cc=jgg@ziepe.ca \
--cc=kbusch@meta.com \
--cc=kevin.tian@intel.com \
--cc=kvm@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-pci@vger.kernel.org \
--cc=michal.winiarski@intel.com \
--cc=mochs@nvidia.com \
--cc=nathanc@nvidia.com \
--cc=satyanarayana.k.v.p@intel.com \
--cc=sonangp@nvidia.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®