mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Shameer Kolothum <skolothumtho@nvidia.com>
To: <kvm@vger.kernel.org>, <linux-pci@vger.kernel.org>,
	<linux-kernel@vger.kernel.org>
Cc: <alex@shazbot.org>, <jgg@ziepe.ca>, <kevin.tian@intel.com>,
	<kbusch@meta.com>, <michal.winiarski@intel.com>,
	<satyanarayana.k.v.p@intel.com>, <sonangp@nvidia.com>,
	<ankita@nvidia.com>, <nathanc@nvidia.com>, <mochs@nvidia.com>,
	<skolothumtho@nvidia.com>
Subject: [RFC PATCH v2 15/16] vfio/pci: Add VFIO_DEVICE_FEATURE_PCI_ERROR_RECOVERY
Date: Tue, 29 Sep 2026 18:33:04 +0100	[thread overview]
Message-ID: <20260929173305.204856-16-skolothumtho@nvidia.com> (raw)
In-Reply-To: <20260929173305.204856-1-skolothumtho@nvidia.com>

Add a device feature to enable host PCI error recovery and query its
status. SET registers an eventfd for start and completion notifications;
eventfd -1 disables recovery. GET returns the status flags and sequence
number, including while recovery is active.

Reject SET while access is blocked or recovery has failed. Reset the
per-open status and sequence when updating the eventfd. Existing
VFIO_PCI_ERR_IRQ_INDEX notifications are unchanged.

Assisted-by: LLM
Signed-off-by: Shameer Kolothum <skolothumtho@nvidia.com>
---
 include/uapi/linux/vfio.h        |  44 +++++++++++++
 drivers/vfio/pci/vfio_pci_core.c | 108 +++++++++++++++++++++++++++++++
 2 files changed, 152 insertions(+)

diff --git a/include/uapi/linux/vfio.h b/include/uapi/linux/vfio.h
index e41437fa17ad..b755849fd39b 100644
--- a/include/uapi/linux/vfio.h
+++ b/include/uapi/linux/vfio.h
@@ -1555,6 +1555,50 @@ struct vfio_device_feature_zpci_err {
 
 #define VFIO_DEVICE_FEATURE_ZPCI_ERROR 13
 
+/*
+ * Report host PCI error recovery state for this device.
+ *
+ * VFIO_DEVICE_FEATURE_SET with a valid eventfd enables recovery for this
+ * open and registers notifications for the start and end of each event.
+ * SET with eventfd -1 disables recovery. flags and sequence must be zero.
+ * SET returns -EBUSY during recovery, after recovery failure, or while
+ * device access is blocked, and -ENODEV after close has started.
+ * VFIO_PCI_ERR_IRQ_INDEX notifications are unchanged.
+ *
+ * VFIO_DEVICE_FEATURE_GET returns the current state and eventfd = -1.
+ * GET is allowed during recovery, but may wait for a running callback.
+ *
+ * The sequence number increments for each event and resets to zero when
+ * recovery is enabled. Use it to detect coalesced notifications. Device
+ * accesses return -EIO while IN_PROGRESS is set. Recovery may complete
+ * before userspace reads the notification, so userspace must check the
+ * sequence and status rather than wait to observe IN_PROGRESS.
+ *
+ * CHANNEL_FROZEN records a frozen channel. DEVICE_RESET records a host
+ * reset; userspace must reconfigure interrupts with VFIO_DEVICE_SET_IRQS.
+ * FAILED indicates that recovery failed and access remains blocked until
+ * close and reopen. Status bits persist until the next event or SET.
+ * ENABLED indicates that recovery is enabled.
+ *
+ * Enabling recovery during an existing event does not provide status or
+ * notifications for that event.
+ *
+ * Open returns -EBUSY while a previously recorded host recovery is active,
+ * including events reported while recovery was disabled.
+ */
+struct vfio_device_pci_error_recovery {
+	__u32 flags;
+#define VFIO_PCI_ERROR_RECOVERY_IN_PROGRESS	(1U << 0)
+#define VFIO_PCI_ERROR_RECOVERY_CHANNEL_FROZEN	(1U << 1)
+#define VFIO_PCI_ERROR_RECOVERY_DEVICE_RESET	(1U << 2)
+#define VFIO_PCI_ERROR_RECOVERY_FAILED		(1U << 3)
+#define VFIO_PCI_ERROR_RECOVERY_ENABLED		(1U << 4)
+	__s32 eventfd;
+	__aligned_u64 sequence;
+};
+
+#define VFIO_DEVICE_FEATURE_PCI_ERROR_RECOVERY 14
+
 /* -------- API for Type1 VFIO IOMMU -------- */
 
 /**
diff --git a/drivers/vfio/pci/vfio_pci_core.c b/drivers/vfio/pci/vfio_pci_core.c
index 2ec8022e4a69..314d77868803 100644
--- a/drivers/vfio/pci/vfio_pci_core.c
+++ b/drivers/vfio/pci/vfio_pci_core.c
@@ -1752,6 +1752,111 @@ static int vfio_pci_core_feature_token(struct vfio_pci_core_device *vdev,
 	return 0;
 }
 
+/* Consumes the eventfd reference on both success and failure. */
+static int vfio_pci_recovery_set(struct vfio_pci_core_device *vdev,
+				 struct eventfd_ctx *ctx, bool enable)
+{
+	int ret;
+
+	guard(mutex)(&vdev->access_lock);
+	if (!vdev->device_open) {
+		ret = -ENODEV;
+		goto out_put;
+	}
+	if (vdev->access_blocked) {
+		ret = -EBUSY;
+		goto out_put;
+	}
+	if (!enable &&
+	    (vdev->pci_recovery_flags &
+	     (VFIO_PCI_RECOVERY_IN_PROGRESS | VFIO_PCI_RECOVERY_FAILED))) {
+		ret = -EBUSY;
+		goto out_put;
+	}
+
+	mutex_lock(&vdev->igate);
+	ret = vfio_pci_eventfd_replace_locked(vdev, &vdev->pci_recovery_trigger,
+					      ctx);
+	mutex_unlock(&vdev->igate);
+	if (ret)
+		goto out_put;
+
+	WRITE_ONCE(vdev->pci_recovery_enabled, enable);
+	/*
+	 * Reset the per-open status when updating the recovery eventfd.
+	 * The access block and host transaction state are unchanged.
+	 */
+	WRITE_ONCE(vdev->pci_recovery_flags, 0);
+	vdev->pci_recovery_sequence = 0;
+	vdev->pci_recovery_command_valid = false;
+
+	return 0;
+
+out_put:
+	if (ctx)
+		eventfd_ctx_put(ctx);
+	return ret;
+}
+
+static int
+vfio_pci_core_feature_error_recovery(struct vfio_pci_core_device *vdev, u32 flags,
+				     struct vfio_device_pci_error_recovery __user *arg,
+				     size_t argsz)
+{
+	struct vfio_device_pci_error_recovery state = { .eventfd = -1 };
+	struct eventfd_ctx *ctx = NULL;
+	bool enable;
+	int ret;
+
+	if (!vdev->pci_recovery_supported)
+		return -ENOTTY;
+
+	ret = vfio_check_feature(flags, argsz,
+				 VFIO_DEVICE_FEATURE_GET |
+				 VFIO_DEVICE_FEATURE_SET, sizeof(state));
+	if (ret != 1)
+		return ret;
+
+	if (flags & VFIO_DEVICE_FEATURE_GET) {
+		scoped_guard(mutex, &vdev->access_lock) {
+			u32 rflags = vdev->pci_recovery_flags;
+
+			if (vdev->pci_recovery_enabled)
+				state.flags |= VFIO_PCI_ERROR_RECOVERY_ENABLED;
+			if (rflags & VFIO_PCI_RECOVERY_IN_PROGRESS)
+				state.flags |=
+					VFIO_PCI_ERROR_RECOVERY_IN_PROGRESS;
+			if (rflags & VFIO_PCI_RECOVERY_FROZEN)
+				state.flags |=
+					VFIO_PCI_ERROR_RECOVERY_CHANNEL_FROZEN;
+			if (rflags & VFIO_PCI_RECOVERY_RESET)
+				state.flags |=
+					VFIO_PCI_ERROR_RECOVERY_DEVICE_RESET;
+			if (rflags & VFIO_PCI_RECOVERY_FAILED)
+				state.flags |= VFIO_PCI_ERROR_RECOVERY_FAILED;
+			state.sequence = vdev->pci_recovery_sequence;
+		}
+
+		if (copy_to_user(arg, &state, sizeof(state)))
+			return -EFAULT;
+		return 0;
+	}
+
+	if (copy_from_user(&state, arg, sizeof(state)))
+		return -EFAULT;
+	if (state.flags || state.sequence || state.eventfd < -1)
+		return -EINVAL;
+
+	enable = state.eventfd >= 0;
+	if (enable) {
+		ctx = eventfd_ctx_fdget(state.eventfd);
+		if (IS_ERR(ctx))
+			return PTR_ERR(ctx);
+	}
+
+	return vfio_pci_recovery_set(vdev, ctx, enable);
+}
+
 int vfio_pci_core_ioctl_feature(struct vfio_device *device, u32 flags,
 				void __user *arg, size_t argsz)
 {
@@ -1772,6 +1877,9 @@ int vfio_pci_core_ioctl_feature(struct vfio_device *device, u32 flags,
 		return vfio_pci_core_feature_dma_buf(vdev, flags, arg, argsz);
 	case VFIO_DEVICE_FEATURE_ZPCI_ERROR:
 		return vfio_pci_zdev_feature_err(device, flags, arg, argsz);
+
+	case VFIO_DEVICE_FEATURE_PCI_ERROR_RECOVERY:
+		return vfio_pci_core_feature_error_recovery(vdev, flags, arg, argsz);
 	default:
 		return -ENOTTY;
 	}
-- 
2.43.0


  parent reply	other threads:[~2026-09-29 17:35 UTC|newest]

Thread overview: 17+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-29 17:32 [RFC PATCH v2 00/16] vfio/pci: Handle PCI error recovery and report state to userspace Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 01/16] vfio/pci: Add a device access gate Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 02/16] vfio/pci: Gate config space access Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 03/16] vfio/pci: Buffer ROM reads before copying to userspace Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 04/16] vfio/pci: Gate BAR and ROM access Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 05/16] vfio/pci: Fail BAR faults while access is blocked Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 06/16] vfio/pci: Gate interrupt configuration Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 07/16] vfio/pci: Gate function reset and runtime power management Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 08/16] vfio/pci: Gate device information queries and DMA-BUF export Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 09/16] vfio/pci: Add PCI error recovery state Shameer Kolothum
2026-09-29 17:32 ` [RFC PATCH v2 10/16] vfio/pci: Quiesce INTx while access is blocked Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 11/16] vfio/pci: Add INTx recovery start and finish helpers Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 12/16] vfio/pci: Restore device state from slot_reset() Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 13/16] vfio/pci: Complete recovery in resume() Shameer Kolothum
2026-09-29 17:33 ` [RFC PATCH v2 14/16] vfio/pci: Block device access during host recovery Shameer Kolothum
2026-09-29 17:33 ` Shameer Kolothum [this message]
2026-09-29 17:33 ` [RFC PATCH v2 16/16] vfio/pci: Enable host PCI error recovery for vfio-pci Shameer Kolothum

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260929173305.204856-16-skolothumtho@nvidia.com \
    --to=skolothumtho@nvidia.com \
    --cc=alex@shazbot.org \
    --cc=ankita@nvidia.com \
    --cc=jgg@ziepe.ca \
    --cc=kbusch@meta.com \
    --cc=kevin.tian@intel.com \
    --cc=kvm@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-pci@vger.kernel.org \
    --cc=michal.winiarski@intel.com \
    --cc=mochs@nvidia.com \
    --cc=nathanc@nvidia.com \
    --cc=satyanarayana.k.v.p@intel.com \
    --cc=sonangp@nvidia.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®