From: Shameer Kolothum <skolothumtho@nvidia.com>
To: <kvm@vger.kernel.org>, <linux-pci@vger.kernel.org>,
<linux-kernel@vger.kernel.org>
Cc: <alex@shazbot.org>, <jgg@ziepe.ca>, <kevin.tian@intel.com>,
<kbusch@meta.com>, <michal.winiarski@intel.com>,
<satyanarayana.k.v.p@intel.com>, <sonangp@nvidia.com>,
<ankita@nvidia.com>, <nathanc@nvidia.com>, <mochs@nvidia.com>,
<skolothumtho@nvidia.com>
Subject: [RFC PATCH v2 12/16] vfio/pci: Restore device state from slot_reset()
Date: Tue, 29 Sep 2026 18:33:01 +0100 [thread overview]
Message-ID: <20260929173305.204856-13-skolothumtho@nvidia.com> (raw)
In-Reply-To: <20260929173305.204856-1-skolothumtho@nvidia.com>
A host reset clears device configuration. Add slot_reset() to restore
the state saved at open before tearing down interrupts. MSI-X shutdown
requires the restored BARs to access its table.
Load vdev->pci_saved_state before entering D0, since the power transition
may restore BARs. PCI core's saved copy may have been overwritten by a
user reset or D3 entry. Discard the stale pre-reset pm_save and update
the power state without restoring it. Keep access blocked and report
failure if the snapshot is missing or restoration fails.
Keep active INTx masked in the loaded snapshot so restoration cannot
re-enable it before IRQ teardown.
Use pci_set_power_state() as existing recovery callbacks do.
Hold access_lock to exclude close's interrupt teardown. Return NONE on
failure to allow recovery of other devices under the bridge to continue.
Assisted-by: LLM
Signed-off-by: Shameer Kolothum <skolothumtho@nvidia.com>
---
Note: This series adds a VFIO recovery callback that can reacquire
pci_bus_sem through the D0 ASPM update while AER already holds it. This
can deadlock behind a queued writer. Similar callback patterns in other
drivers do not resolve this new VFIO locking issue; topology does not
establish lock ownership. See the cover letter's "Locking and open
questions" section.
---
drivers/vfio/pci/vfio_pci_core.c | 101 +++++++++++++++++++++++++++++++
1 file changed, 101 insertions(+)
diff --git a/drivers/vfio/pci/vfio_pci_core.c b/drivers/vfio/pci/vfio_pci_core.c
index 667c5813f6c7..a7b7499e071c 100644
--- a/drivers/vfio/pci/vfio_pci_core.c
+++ b/drivers/vfio/pci/vfio_pci_core.c
@@ -311,6 +311,15 @@ static void vfio_pci_probe_power_state(struct vfio_pci_core_device *vdev)
vdev->needs_pm_restore = !(pmcsr & PCI_PM_CTRL_NO_SOFT_RESET);
}
+/* Keep recovery's INTx mask when restoring a loaded PCI state snapshot. */
+static void vfio_pci_recovery_mask_saved_intx(struct vfio_pci_core_device *vdev)
+{
+ if (READ_ONCE(vdev->access_blocked) && vdev->pci_2_3 &&
+ READ_ONCE(vdev->irq_type) == VFIO_PCI_INTX_IRQ_INDEX)
+ vdev->pdev->saved_config_space[PCI_COMMAND / 4] |=
+ PCI_COMMAND_INTX_DISABLE;
+}
+
/*
* pci_set_power_state() wrapper handling devices which perform a soft reset on
* D3->D0 transition. Save state prior to D0/1/2->D3, stash it on the vdev,
@@ -2499,6 +2508,18 @@ void vfio_pci_core_unregister_device(struct vfio_pci_core_device *vdev)
}
EXPORT_SYMBOL_GPL(vfio_pci_core_unregister_device);
+static void
+vfio_pci_signal_recovery_event(struct vfio_pci_core_device *vdev)
+{
+ struct vfio_pci_eventfd *eventfd;
+
+ rcu_read_lock();
+ eventfd = rcu_dereference(vdev->pci_recovery_trigger);
+ if (eventfd)
+ eventfd_signal(eventfd->ctx);
+ rcu_read_unlock();
+}
+
pci_ers_result_t vfio_pci_core_aer_err_detected(struct pci_dev *pdev,
pci_channel_state_t state)
{
@@ -2515,6 +2536,85 @@ pci_ers_result_t vfio_pci_core_aer_err_detected(struct pci_dev *pdev,
}
EXPORT_SYMBOL_GPL(vfio_pci_core_aer_err_detected);
+/* Caller holds memory_lock. Discard the pre-reset PM snapshot. */
+static int vfio_pci_recovery_restore_state(struct vfio_pci_core_device *vdev)
+{
+ struct pci_dev *pdev = vdev->pdev;
+ int ret;
+
+ if (!vdev->pci_saved_state)
+ return -ENODATA;
+ ret = pci_load_saved_state(pdev, vdev->pci_saved_state);
+ if (ret)
+ return ret;
+
+ kfree(vdev->pm_save);
+ vdev->pm_save = NULL;
+ ret = pci_set_power_state(pdev, PCI_D0);
+ if (ret)
+ return ret;
+
+ vfio_pci_recovery_mask_saved_intx(vdev);
+ pci_restore_state(pdev);
+ return 0;
+}
+
+static pci_ers_result_t vfio_pci_core_aer_slot_reset(struct pci_dev *pdev)
+{
+ struct vfio_pci_core_device *vdev = dev_get_drvdata(&pdev->dev);
+ int ret = 0;
+
+ mutex_lock(&vdev->access_lock);
+ if (!(vdev->pci_recovery_flags & VFIO_PCI_RECOVERY_IN_PROGRESS) ||
+ !vdev->device_open) {
+ mutex_unlock(&vdev->access_lock);
+ return PCI_ERS_RESULT_NONE;
+ }
+
+ /*
+ * Load the open snapshot before D0, which may restore BARs from
+ * saved state. Restore BARs before IRQ teardown so MSI-X shutdown
+ * can access the table after the host reset.
+ */
+ down_write(&vdev->memory_lock);
+ ret = vfio_pci_recovery_restore_state(vdev);
+ up_write(&vdev->memory_lock);
+ if (ret)
+ goto out_failed;
+
+ /*
+ * Close clears device_open under access_lock before IRQ teardown,
+ * excluding this callback from the teardown path.
+ */
+ mutex_lock(&vdev->igate);
+ if (vdev->irq_type < VFIO_PCI_NUM_IRQS)
+ ret = vfio_pci_set_irqs_ioctl(vdev,
+ VFIO_IRQ_SET_DATA_NONE |
+ VFIO_IRQ_SET_ACTION_TRIGGER,
+ vdev->irq_type, 0, 0, NULL);
+ mutex_unlock(&vdev->igate);
+ if (ret)
+ goto out_failed;
+
+ WRITE_ONCE(vdev->pci_recovery_flags,
+ vdev->pci_recovery_flags | VFIO_PCI_RECOVERY_RESET);
+ mutex_unlock(&vdev->access_lock);
+
+ return PCI_ERS_RESULT_RECOVERED;
+
+out_failed:
+ WRITE_ONCE(vdev->pci_recovery_flags,
+ (vdev->pci_recovery_flags | VFIO_PCI_RECOVERY_FAILED) &
+ ~VFIO_PCI_RECOVERY_IN_PROGRESS);
+ vdev->pci_recovery_command_valid = false;
+ mutex_unlock(&vdev->access_lock);
+ /* Report failure here; resume() skips completed transactions. */
+ vfio_pci_signal_recovery_event(vdev);
+
+ /* Allow recovery of other devices under the bridge to continue. */
+ return PCI_ERS_RESULT_NONE;
+}
+
int vfio_pci_core_sriov_configure(struct vfio_pci_core_device *vdev,
int nr_virtfn)
{
@@ -2587,6 +2687,7 @@ EXPORT_SYMBOL_GPL(vfio_pci_core_sriov_configure);
const struct pci_error_handlers vfio_pci_core_err_handlers = {
.error_detected = vfio_pci_core_aer_err_detected,
+ .slot_reset = vfio_pci_core_aer_slot_reset,
};
EXPORT_SYMBOL_GPL(vfio_pci_core_err_handlers);
--
2.43.0
next prev parent reply other threads:[~2026-09-29 17:34 UTC|newest]
Thread overview: 17+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-29 17:32 [RFC PATCH v2 00/16] vfio/pci: Handle PCI error recovery and report state to userspace Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 01/16] vfio/pci: Add a device access gate Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 02/16] vfio/pci: Gate config space access Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 03/16] vfio/pci: Buffer ROM reads before copying to userspace Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 04/16] vfio/pci: Gate BAR and ROM access Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 05/16] vfio/pci: Fail BAR faults while access is blocked Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 06/16] vfio/pci: Gate interrupt configuration Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 07/16] vfio/pci: Gate function reset and runtime power management Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 08/16] vfio/pci: Gate device information queries and DMA-BUF export Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 09/16] vfio/pci: Add PCI error recovery state Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 10/16] vfio/pci: Quiesce INTx while access is blocked Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 11/16] vfio/pci: Add INTx recovery start and finish helpers Shameer Kolothum
2026-09-29 17:33 ` Shameer Kolothum [this message]
2026-09-29 17:33 ` [RFC PATCH v2 13/16] vfio/pci: Complete recovery in resume() Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 14/16] vfio/pci: Block device access during host recovery Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 15/16] vfio/pci: Add VFIO_DEVICE_FEATURE_PCI_ERROR_RECOVERY Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 16/16] vfio/pci: Enable host PCI error recovery for vfio-pci Shameer Kolothum
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260929173305.204856-13-skolothumtho@nvidia.com \
--to=skolothumtho@nvidia.com \
--cc=alex@shazbot.org \
--cc=ankita@nvidia.com \
--cc=jgg@ziepe.ca \
--cc=kbusch@meta.com \
--cc=kevin.tian@intel.com \
--cc=kvm@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-pci@vger.kernel.org \
--cc=michal.winiarski@intel.com \
--cc=mochs@nvidia.com \
--cc=nathanc@nvidia.com \
--cc=satyanarayana.k.v.p@intel.com \
--cc=sonangp@nvidia.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®