From: Shameer Kolothum <skolothumtho@nvidia.com>
To: <kvm@vger.kernel.org>, <linux-pci@vger.kernel.org>,
<linux-kernel@vger.kernel.org>
Cc: <alex@shazbot.org>, <jgg@ziepe.ca>, <kevin.tian@intel.com>,
<kbusch@meta.com>, <michal.winiarski@intel.com>,
<satyanarayana.k.v.p@intel.com>, <sonangp@nvidia.com>,
<ankita@nvidia.com>, <nathanc@nvidia.com>, <mochs@nvidia.com>,
<skolothumtho@nvidia.com>
Subject: [RFC PATCH v2 15/16] vfio/pci: Add VFIO_DEVICE_FEATURE_PCI_ERROR_RECOVERY
Date: Tue, 29 Sep 2026 18:33:04 +0100 [thread overview]
Message-ID: <20260929173305.204856-16-skolothumtho@nvidia.com> (raw)
In-Reply-To: <20260929173305.204856-1-skolothumtho@nvidia.com>
Add a device feature to enable host PCI error recovery and query its
status. SET registers an eventfd for start and completion notifications;
eventfd -1 disables recovery. GET returns the status flags and sequence
number, including while recovery is active.
Reject SET while access is blocked or recovery has failed. Reset the
per-open status and sequence when updating the eventfd. Existing
VFIO_PCI_ERR_IRQ_INDEX notifications are unchanged.
Assisted-by: LLM
Signed-off-by: Shameer Kolothum <skolothumtho@nvidia.com>
---
include/uapi/linux/vfio.h | 44 +++++++++++++
drivers/vfio/pci/vfio_pci_core.c | 108 +++++++++++++++++++++++++++++++
2 files changed, 152 insertions(+)
diff --git a/include/uapi/linux/vfio.h b/include/uapi/linux/vfio.h
index e41437fa17ad..b755849fd39b 100644
--- a/include/uapi/linux/vfio.h
+++ b/include/uapi/linux/vfio.h
@@ -1555,6 +1555,50 @@ struct vfio_device_feature_zpci_err {
#define VFIO_DEVICE_FEATURE_ZPCI_ERROR 13
+/*
+ * Report host PCI error recovery state for this device.
+ *
+ * VFIO_DEVICE_FEATURE_SET with a valid eventfd enables recovery for this
+ * open and registers notifications for the start and end of each event.
+ * SET with eventfd -1 disables recovery. flags and sequence must be zero.
+ * SET returns -EBUSY during recovery, after recovery failure, or while
+ * device access is blocked, and -ENODEV after close has started.
+ * VFIO_PCI_ERR_IRQ_INDEX notifications are unchanged.
+ *
+ * VFIO_DEVICE_FEATURE_GET returns the current state and eventfd = -1.
+ * GET is allowed during recovery, but may wait for a running callback.
+ *
+ * The sequence number increments for each event and resets to zero when
+ * recovery is enabled. Use it to detect coalesced notifications. Device
+ * accesses return -EIO while IN_PROGRESS is set. Recovery may complete
+ * before userspace reads the notification, so userspace must check the
+ * sequence and status rather than wait to observe IN_PROGRESS.
+ *
+ * CHANNEL_FROZEN records a frozen channel. DEVICE_RESET records a host
+ * reset; userspace must reconfigure interrupts with VFIO_DEVICE_SET_IRQS.
+ * FAILED indicates that recovery failed and access remains blocked until
+ * close and reopen. Status bits persist until the next event or SET.
+ * ENABLED indicates that recovery is enabled.
+ *
+ * Enabling recovery during an existing event does not provide status or
+ * notifications for that event.
+ *
+ * Open returns -EBUSY while a previously recorded host recovery is active,
+ * including events reported while recovery was disabled.
+ */
+struct vfio_device_pci_error_recovery {
+ __u32 flags;
+#define VFIO_PCI_ERROR_RECOVERY_IN_PROGRESS (1U << 0)
+#define VFIO_PCI_ERROR_RECOVERY_CHANNEL_FROZEN (1U << 1)
+#define VFIO_PCI_ERROR_RECOVERY_DEVICE_RESET (1U << 2)
+#define VFIO_PCI_ERROR_RECOVERY_FAILED (1U << 3)
+#define VFIO_PCI_ERROR_RECOVERY_ENABLED (1U << 4)
+ __s32 eventfd;
+ __aligned_u64 sequence;
+};
+
+#define VFIO_DEVICE_FEATURE_PCI_ERROR_RECOVERY 14
+
/* -------- API for Type1 VFIO IOMMU -------- */
/**
diff --git a/drivers/vfio/pci/vfio_pci_core.c b/drivers/vfio/pci/vfio_pci_core.c
index 2ec8022e4a69..314d77868803 100644
--- a/drivers/vfio/pci/vfio_pci_core.c
+++ b/drivers/vfio/pci/vfio_pci_core.c
@@ -1752,6 +1752,111 @@ static int vfio_pci_core_feature_token(struct vfio_pci_core_device *vdev,
return 0;
}
+/* Consumes the eventfd reference on both success and failure. */
+static int vfio_pci_recovery_set(struct vfio_pci_core_device *vdev,
+ struct eventfd_ctx *ctx, bool enable)
+{
+ int ret;
+
+ guard(mutex)(&vdev->access_lock);
+ if (!vdev->device_open) {
+ ret = -ENODEV;
+ goto out_put;
+ }
+ if (vdev->access_blocked) {
+ ret = -EBUSY;
+ goto out_put;
+ }
+ if (!enable &&
+ (vdev->pci_recovery_flags &
+ (VFIO_PCI_RECOVERY_IN_PROGRESS | VFIO_PCI_RECOVERY_FAILED))) {
+ ret = -EBUSY;
+ goto out_put;
+ }
+
+ mutex_lock(&vdev->igate);
+ ret = vfio_pci_eventfd_replace_locked(vdev, &vdev->pci_recovery_trigger,
+ ctx);
+ mutex_unlock(&vdev->igate);
+ if (ret)
+ goto out_put;
+
+ WRITE_ONCE(vdev->pci_recovery_enabled, enable);
+ /*
+ * Reset the per-open status when updating the recovery eventfd.
+ * The access block and host transaction state are unchanged.
+ */
+ WRITE_ONCE(vdev->pci_recovery_flags, 0);
+ vdev->pci_recovery_sequence = 0;
+ vdev->pci_recovery_command_valid = false;
+
+ return 0;
+
+out_put:
+ if (ctx)
+ eventfd_ctx_put(ctx);
+ return ret;
+}
+
+static int
+vfio_pci_core_feature_error_recovery(struct vfio_pci_core_device *vdev, u32 flags,
+ struct vfio_device_pci_error_recovery __user *arg,
+ size_t argsz)
+{
+ struct vfio_device_pci_error_recovery state = { .eventfd = -1 };
+ struct eventfd_ctx *ctx = NULL;
+ bool enable;
+ int ret;
+
+ if (!vdev->pci_recovery_supported)
+ return -ENOTTY;
+
+ ret = vfio_check_feature(flags, argsz,
+ VFIO_DEVICE_FEATURE_GET |
+ VFIO_DEVICE_FEATURE_SET, sizeof(state));
+ if (ret != 1)
+ return ret;
+
+ if (flags & VFIO_DEVICE_FEATURE_GET) {
+ scoped_guard(mutex, &vdev->access_lock) {
+ u32 rflags = vdev->pci_recovery_flags;
+
+ if (vdev->pci_recovery_enabled)
+ state.flags |= VFIO_PCI_ERROR_RECOVERY_ENABLED;
+ if (rflags & VFIO_PCI_RECOVERY_IN_PROGRESS)
+ state.flags |=
+ VFIO_PCI_ERROR_RECOVERY_IN_PROGRESS;
+ if (rflags & VFIO_PCI_RECOVERY_FROZEN)
+ state.flags |=
+ VFIO_PCI_ERROR_RECOVERY_CHANNEL_FROZEN;
+ if (rflags & VFIO_PCI_RECOVERY_RESET)
+ state.flags |=
+ VFIO_PCI_ERROR_RECOVERY_DEVICE_RESET;
+ if (rflags & VFIO_PCI_RECOVERY_FAILED)
+ state.flags |= VFIO_PCI_ERROR_RECOVERY_FAILED;
+ state.sequence = vdev->pci_recovery_sequence;
+ }
+
+ if (copy_to_user(arg, &state, sizeof(state)))
+ return -EFAULT;
+ return 0;
+ }
+
+ if (copy_from_user(&state, arg, sizeof(state)))
+ return -EFAULT;
+ if (state.flags || state.sequence || state.eventfd < -1)
+ return -EINVAL;
+
+ enable = state.eventfd >= 0;
+ if (enable) {
+ ctx = eventfd_ctx_fdget(state.eventfd);
+ if (IS_ERR(ctx))
+ return PTR_ERR(ctx);
+ }
+
+ return vfio_pci_recovery_set(vdev, ctx, enable);
+}
+
int vfio_pci_core_ioctl_feature(struct vfio_device *device, u32 flags,
void __user *arg, size_t argsz)
{
@@ -1772,6 +1877,9 @@ int vfio_pci_core_ioctl_feature(struct vfio_device *device, u32 flags,
return vfio_pci_core_feature_dma_buf(vdev, flags, arg, argsz);
case VFIO_DEVICE_FEATURE_ZPCI_ERROR:
return vfio_pci_zdev_feature_err(device, flags, arg, argsz);
+
+ case VFIO_DEVICE_FEATURE_PCI_ERROR_RECOVERY:
+ return vfio_pci_core_feature_error_recovery(vdev, flags, arg, argsz);
default:
return -ENOTTY;
}
--
2.43.0
next prev parent reply other threads:[~2026-09-29 17:35 UTC|newest]
Thread overview: 17+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-29 17:32 [RFC PATCH v2 00/16] vfio/pci: Handle PCI error recovery and report state to userspace Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 01/16] vfio/pci: Add a device access gate Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 02/16] vfio/pci: Gate config space access Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 03/16] vfio/pci: Buffer ROM reads before copying to userspace Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 04/16] vfio/pci: Gate BAR and ROM access Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 05/16] vfio/pci: Fail BAR faults while access is blocked Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 06/16] vfio/pci: Gate interrupt configuration Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 07/16] vfio/pci: Gate function reset and runtime power management Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 08/16] vfio/pci: Gate device information queries and DMA-BUF export Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 09/16] vfio/pci: Add PCI error recovery state Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 10/16] vfio/pci: Quiesce INTx while access is blocked Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 11/16] vfio/pci: Add INTx recovery start and finish helpers Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 12/16] vfio/pci: Restore device state from slot_reset() Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 13/16] vfio/pci: Complete recovery in resume() Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 14/16] vfio/pci: Block device access during host recovery Shameer Kolothum
2026-09-29 17:33 ` Shameer Kolothum [this message]
2026-09-29 17:33 ` [RFC PATCH v2 16/16] vfio/pci: Enable host PCI error recovery for vfio-pci Shameer Kolothum
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260929173305.204856-16-skolothumtho@nvidia.com \
--to=skolothumtho@nvidia.com \
--cc=alex@shazbot.org \
--cc=ankita@nvidia.com \
--cc=jgg@ziepe.ca \
--cc=kbusch@meta.com \
--cc=kevin.tian@intel.com \
--cc=kvm@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-pci@vger.kernel.org \
--cc=michal.winiarski@intel.com \
--cc=mochs@nvidia.com \
--cc=nathanc@nvidia.com \
--cc=satyanarayana.k.v.p@intel.com \
--cc=sonangp@nvidia.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®