mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Ankit Soni <Ankit.Soni@amd.com>
To: <iommu@lists.linux.dev>, <joro@8bytes.org>, <will@kernel.org>,
	<jgg@nvidia.com>
Cc: <suravee.suthikulpanit@amd.com>, <vasant.hegde@amd.com>,
	<robin.murphy@arm.com>, <joao.m.martins@oracle.com>,
	<alejandro.j.jimenez@oracle.com>, <pasha.tatashin@soleen.com>,
	<rppt@kernel.org>, <pratyush@kernel.org>, <skhawaja@google.com>,
	<praan@google.com>, <baolu.lu@linux.intel.com>,
	<dwmw2@infradead.org>, <kevin.tian@intel.com>,
	<dmatlack@google.com>, <vipinsh@google.com>,
	<kexec@lists.infradead.org>, <linux-kernel@vger.kernel.org>
Subject: [RFC PATCH 4/7] iommu/amd: preserve IOMMU and device state for live update
Date: Mon, 5 Oct 2026 06:40:14 +0000	[thread overview]
Message-ID: <20261005064018.1558-5-Ankit.Soni@amd.com> (raw)
In-Reply-To: <20261005064018.1558-1-Ankit.Soni@amd.com>

Implement the .preserve and .preserve_device callbacks. Pin the PCI
segment's device table for KHO and record each device's domain ID,
page-table mode and GCR3 tree, so the next kernel can rebuild them.

The device table is shared by every IOMMU in a segment while the
callbacks run once per IOMMU, so the pin is reference counted.

Signed-off-by: Ankit Soni <Ankit.Soni@amd.com>
---
 drivers/iommu/amd/Makefile          |   1 +
 drivers/iommu/amd/amd_iommu.h       |  12 ++
 drivers/iommu/amd/amd_iommu_types.h |  14 ++
 drivers/iommu/amd/iommu.c           |   7 +
 drivers/iommu/amd/liveupdate.c      | 281 ++++++++++++++++++++++++++++
 5 files changed, 315 insertions(+)
 create mode 100644 drivers/iommu/amd/liveupdate.c

diff --git a/drivers/iommu/amd/Makefile b/drivers/iommu/amd/Makefile
index 94b8ef2acb18..227bbe920c26 100644
--- a/drivers/iommu/amd/Makefile
+++ b/drivers/iommu/amd/Makefile
@@ -2,3 +2,4 @@
 obj-y += iommu.o init.o quirks.o ppr.o pasid.o
 obj-$(CONFIG_AMD_IOMMU_IOMMUFD) += iommufd.o nested.o
 obj-$(CONFIG_AMD_IOMMU_DEBUGFS) += debugfs.o
+obj-$(CONFIG_IOMMU_LIVEUPDATE) += liveupdate.o
diff --git a/drivers/iommu/amd/amd_iommu.h b/drivers/iommu/amd/amd_iommu.h
index a2fe804b038b..5cf32e4898dc 100644
--- a/drivers/iommu/amd/amd_iommu.h
+++ b/drivers/iommu/amd/amd_iommu.h
@@ -225,4 +225,16 @@ amd_iommu_make_clear_dte(struct iommu_dev_data *dev_data, struct dev_table_entry
 struct iommu_domain *
 amd_iommu_alloc_domain_nested(struct iommufd_viommu *viommu, u32 flags,
 			      const struct iommu_user_data *user_data);
+
+#ifdef CONFIG_IOMMU_LIVEUPDATE
+/* LIVE UPDATE (drivers/iommu/amd/liveupdate.c) */
+int amd_iommu_preserve(struct iommu_device *iommu_dev,
+		       struct iommu_hw_ser *ser);
+void amd_iommu_unpreserve(struct iommu_device *iommu_dev,
+			  struct iommu_hw_ser *ser);
+int amd_iommu_preserve_device(struct device *dev,
+			      struct iommu_device_ser *device_ser);
+void amd_iommu_unpreserve_device(struct device *dev,
+				 struct iommu_device_ser *device_ser);
+#endif /* CONFIG_IOMMU_LIVEUPDATE */
 #endif /* AMD_IOMMU_H */
diff --git a/drivers/iommu/amd/amd_iommu_types.h b/drivers/iommu/amd/amd_iommu_types.h
index 3dbe20023456..fc9d98dfc6cd 100644
--- a/drivers/iommu/amd/amd_iommu_types.h
+++ b/drivers/iommu/amd/amd_iommu_types.h
@@ -596,6 +596,20 @@ struct amd_iommu_pci_seg {
 	 */
 	struct dev_table_entry *dev_table;
 
+#ifdef CONFIG_IOMMU_LIVEUPDATE
+	/*
+	 * The device table is shared by every IOMMU in this PCI segment, but
+	 * the live-update .preserve callback runs once per IOMMU, and each of
+	 * those IOMMUs is preserved and unpreserved independently by the core.
+	 * KHO preservation is not refcounted, so the references are counted
+	 * here and the shared pages are pinned on the first and unpinned only
+	 * on the last -- otherwise one IOMMU being unpreserved would drop the
+	 * pin from underneath the devices still preserved behind every other
+	 * IOMMU in the segment.
+	 */
+	unsigned int dev_table_preserve_count;
+#endif
+
 	/*
 	 * The rlookup iommu table is used to find the IOMMU which is
 	 * responsible for a specific device. It is indexed by the PCI
diff --git a/drivers/iommu/amd/iommu.c b/drivers/iommu/amd/iommu.c
index a83ce4521f7f..72df97b99589 100644
--- a/drivers/iommu/amd/iommu.c
+++ b/drivers/iommu/amd/iommu.c
@@ -32,6 +32,7 @@
 #include <linux/percpu.h>
 #include <linux/cc_platform.h>
 #include <linux/crash_dump.h>
+#include <linux/iommu-liveupdate.h>
 #include <asm/irq_remapping.h>
 #include <asm/io_apic.h>
 #include <asm/apic.h>
@@ -3221,6 +3222,12 @@ const struct iommu_ops amd_iommu_ops = {
 	.page_response = amd_iommu_page_response,
 	.get_viommu_size = amd_iommufd_get_viommu_size,
 	.viommu_init = amd_iommufd_viommu_init,
+#ifdef CONFIG_IOMMU_LIVEUPDATE
+	.preserve_device	= amd_iommu_preserve_device,
+	.unpreserve_device	= amd_iommu_unpreserve_device,
+	.preserve		= amd_iommu_preserve,
+	.unpreserve		= amd_iommu_unpreserve,
+#endif
 };
 
 #ifdef CONFIG_IRQ_REMAP
diff --git a/drivers/iommu/amd/liveupdate.c b/drivers/iommu/amd/liveupdate.c
new file mode 100644
index 000000000000..096a23bb4e7b
--- /dev/null
+++ b/drivers/iommu/amd/liveupdate.c
@@ -0,0 +1,281 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * AMD IOMMU (AMD-Vi) live update support.
+ *
+ * Copyright (C) 2026 Advanced Micro Devices, Inc.
+ * Author: Ankit Soni <Ankit.Soni@amd.com>
+ */
+
+#include <linux/iommu-liveupdate.h>
+#include <linux/pci.h>
+
+#include "amd_iommu.h"
+#include "../iommu-pages.h"
+
+/* Each 4K GCR3 table level holds 512 u64 entries. */
+#define GCR3_ENTRIES_PER_LEVEL	512
+
+/**
+ * amd_iommu_preserve - Preserve one AMD IOMMU instance for live update
+ * @iommu_dev: Core handle for the IOMMU whose state is being preserved
+ * @ser: Serialized IOMMU-instance record to fill in
+ *
+ * Pins the PCI-segment Device Table. Command, event, PPR and GA buffers are
+ * not preserved: they are drained or quiesced at shutdown and the next kernel
+ * allocates fresh ones, matching Intel phase 1.
+ *
+ * Return: 0 on success, negative errno on failure (any pages pinned by this
+ * call are released before returning).
+ */
+int amd_iommu_preserve(struct iommu_device *iommu_dev, struct iommu_hw_ser *ser)
+{
+	struct amd_iommu *iommu = container_of(iommu_dev, struct amd_iommu, iommu);
+	struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
+	int ret;
+
+	if (!pci_seg->dev_table_preserve_count) {
+		ret = iommu_preserve_pages(pci_seg->dev_table);
+		if (ret)
+			return ret;
+	}
+	pci_seg->dev_table_preserve_count++;
+
+	ser->type			= IOMMU_AMD;
+	ser->token			= iommu->mmio_phys;
+	ser->amd.mmio_phys		= iommu->mmio_phys;
+	ser->amd.dev_table_phys		= __pa(pci_seg->dev_table);
+	ser->amd.dev_table_size		= pci_seg->dev_table_size;
+	ser->amd.pci_seg_id		= pci_seg->id;
+	ser->amd.efr			= iommu->features;
+	ser->amd.efr2			= iommu->features2;
+
+	return 0;
+}
+
+/**
+ * amd_iommu_unpreserve - Release live-update state of one AMD IOMMU instance
+ * @iommu_dev: Core handle for the IOMMU whose state is being released
+ * @ser: Serialized IOMMU-instance record (unused; state is derived from @iommu)
+ *
+ * Reverse of amd_iommu_preserve(): once the last IOMMU of the segment has
+ * dropped its reference, unpins the shared Device Table.
+ */
+void amd_iommu_unpreserve(struct iommu_device *iommu_dev,
+			  struct iommu_hw_ser *ser)
+{
+	struct amd_iommu *iommu = container_of(iommu_dev, struct amd_iommu, iommu);
+	struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
+
+	if (WARN_ON(!pci_seg->dev_table_preserve_count))
+		return;
+
+	if (!--pci_seg->dev_table_preserve_count)
+		iommu_unpreserve_pages(pci_seg->dev_table);
+}
+
+/*
+ * Map the domain's page-table mode onto its handoff wire value. Done with an
+ * explicit switch rather than a cast so that renumbering
+ * enum protection_domain_mode cannot silently change the ABI.
+ */
+static u32 pd_mode_to_ser(enum protection_domain_mode pd_mode)
+{
+	switch (pd_mode) {
+	case PD_MODE_V1:
+		return IOMMU_AMD_SER_PD_MODE_V1;
+	case PD_MODE_V2:
+		return IOMMU_AMD_SER_PD_MODE_V2;
+	default:
+		return IOMMU_AMD_SER_PD_MODE_NONE;
+	}
+}
+
+static void unpreserve_gcr3_level(u64 *tbl, int level)
+{
+	int i;
+
+	if (level > 0) {
+		for (i = 0; i < GCR3_ENTRIES_PER_LEVEL; i++) {
+			u64 *child;
+
+			if (!(tbl[i] & GCR3_VALID))
+				continue;
+
+			child = iommu_phys_to_virt(tbl[i] & PAGE_MASK);
+			unpreserve_gcr3_level(child, level - 1);
+		}
+	}
+
+	iommu_unpreserve_pages(tbl);
+}
+
+static int preserve_gcr3_level(u64 *tbl, int level)
+{
+	u64 *child;
+	int i, ret;
+
+	ret = iommu_preserve_pages(tbl);
+	if (ret)
+		return ret;
+
+	if (level == 0)
+		return 0;
+
+	for (i = 0; i < GCR3_ENTRIES_PER_LEVEL; i++) {
+		if (!(tbl[i] & GCR3_VALID))
+			continue;
+
+		child = iommu_phys_to_virt(tbl[i] & PAGE_MASK);
+		ret = preserve_gcr3_level(child, level - 1);
+		if (ret)
+			goto err_unwind;
+	}
+
+	return 0;
+
+err_unwind:
+	while (--i >= 0) {
+		if (!(tbl[i] & GCR3_VALID))
+			continue;
+
+		child = iommu_phys_to_virt(tbl[i] & PAGE_MASK);
+		unpreserve_gcr3_level(child, level - 1);
+	}
+	iommu_unpreserve_pages(tbl);
+	return ret;
+}
+
+/*
+ * A nested attach programs the DTE from the guest's own descriptor instead of
+ * from dev_data: the DOMID is the viommu's host domain ID for that guest domain,
+ * the GCR3 pointer is the guest's, and dev_data->domain is never assigned, so it
+ * still refers to whatever was attached before -- the nest parent, or nothing at
+ * all (see set_dte_nested()). None of that is recoverable from the state
+ * amd_iommu_preserve_device() serializes, and the core cannot filter it out
+ * either: a device on a nested domain is preserved against the domain of its
+ * paging parent, which is a legitimately preserved domain, so every check up to
+ * this point passes (see find_hwpt_paging()).
+ *
+ * Guest translation that this kernel set up itself always has a host-allocated
+ * GCR3 table behind it, which is what tells the two apart.
+ */
+static bool dev_is_nested_attached(struct iommu_dev_data *dev_data)
+{
+	struct amd_iommu *iommu = get_amd_iommu_from_dev_data(dev_data);
+	struct dev_table_entry *dev_table = get_dev_table(iommu);
+
+	if (!dev_table)
+		return false;
+
+	return (READ_ONCE(dev_table[dev_data->devid].data[0]) & DTE_FLAG_GV) &&
+	       !dev_data->gcr3_info.gcr3_tbl;
+}
+
+/**
+ * amd_iommu_preserve_device - Preserve per-device live-update state
+ * @dev: The device being preserved
+ * @device_ser: Serialized per-device record to fill in
+ *
+ * The device's DTE rides across the kexec inside the preserved Device Table and
+ * keeps being used by the hardware, so everything the DTE still points at
+ * afterwards must be KHO-pinned. For a PASID-capable device that is the GCR3
+ * directory tree; the domain's page tables are pinned by the core when it
+ * preserves the domain. Anything not pinned must instead be dropped from the
+ * DTE, which is what amd_iommu_clear_unpreserved_dtes() does at shutdown.
+ *
+ * Records the domain ID the DTE tags this device with, and the domain's
+ * page-table mode so the next kernel can confirm its own default agrees before
+ * adopting the preserved tables. Both describe the DTE only as long as it was
+ * programmed from dev_data, so a device whose DTE came from a guest descriptor
+ * instead is refused outright; see dev_is_nested_attached().
+ *
+ * The walk runs during the quiesced live-update window, where the device is
+ * owned by its userspace driver and no PASID is being attached or detached, so
+ * the GCR3 tree is stable and no iommu-group lock is taken.
+ *
+ * Return: 0 on success, negative errno otherwise.
+ */
+int amd_iommu_preserve_device(struct device *dev,
+			      struct iommu_device_ser *device_ser)
+{
+	struct gcr3_tbl_info *gcr3_info;
+	struct iommu_dev_data *dev_data;
+	int ret;
+
+	if (!dev_is_pci(dev)) {
+		dev_err(dev, "cannot preserve non-PCI device\n");
+		return -EOPNOTSUPP;
+	}
+
+	dev_data = dev_iommu_priv_get(dev);
+	if (!dev_data)
+		return -EINVAL;
+
+	if (dev_is_nested_attached(dev_data)) {
+		dev_err(dev, "cannot preserve device attached to a nested domain\n");
+		return -EOPNOTSUPP;
+	}
+
+	if (!dev_data->domain)
+		return -EINVAL;
+
+	/* Page-table mode is a domain property, independent of PASID use. */
+	device_ser->amd.pd_mode = pd_mode_to_ser(dev_data->domain->pd_mode);
+
+	gcr3_info = &dev_data->gcr3_info;
+	if (!gcr3_info->gcr3_tbl) {
+		/*
+		 * Non-PASID device: the DTE's DOMID field holds the plain
+		 * protection-domain ID (see amd_iommu_set_dte_v1()), and there
+		 * is no GCR3 tree to pin.
+		 */
+		device_ser->domain_iommu_ser.attachment_id = dev_data->domain->id;
+		device_ser->amd.gcr3_tbl_phys = 0;
+		device_ser->amd.gcr3_glx = 0;
+		return 0;
+	}
+
+	/* PASID device: pin the whole GCR3 directory tree. */
+	ret = preserve_gcr3_level(gcr3_info->gcr3_tbl, gcr3_info->glx);
+	if (ret)
+		return ret;
+
+	/*
+	 * For a GCR3 device the DTE's DOMID field holds gcr3_info->domid, not
+	 * domain->id (see set_dte_gcr3_table()).
+	 */
+	device_ser->domain_iommu_ser.attachment_id = gcr3_info->domid;
+	device_ser->amd.gcr3_tbl_phys = __pa(gcr3_info->gcr3_tbl);
+	device_ser->amd.gcr3_glx = gcr3_info->glx;
+
+	return 0;
+}
+
+/**
+ * amd_iommu_unpreserve_device - Release per-device live-update state
+ * @dev: The device whose state is being released
+ * @device_ser: Serialized per-device record (unused)
+ *
+ * Reverse of amd_iommu_preserve_device(): unpins the GCR3 tree of a
+ * PASID-capable device. The DTE itself lives in the shared Device Table
+ * released by amd_iommu_unpreserve().
+ */
+void amd_iommu_unpreserve_device(struct device *dev,
+				 struct iommu_device_ser *device_ser)
+{
+	struct gcr3_tbl_info *gcr3_info;
+	struct iommu_dev_data *dev_data;
+
+	if (!dev_is_pci(dev))
+		return;
+
+	dev_data = dev_iommu_priv_get(dev);
+	if (!dev_data)
+		return;
+
+	gcr3_info = &dev_data->gcr3_info;
+	if (!gcr3_info->gcr3_tbl)
+		return;
+
+	unpreserve_gcr3_level(gcr3_info->gcr3_tbl, gcr3_info->glx);
+}
-- 
2.43.0


  parent reply	other threads:[~2026-10-05  6:44 UTC|newest]

Thread overview: 8+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-05  6:40 [RFC PATCH 0/7] iommu/amd: Implement live update state preservation Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 1/7] iommu/amd: defer device attach only on a kdump boot Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 2/7] liveupdate: parse the incoming handover tree before late_time_init() Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 3/7] iommu/kho/abi: add AMD IOMMU live-update serialisation structs Ankit Soni
2026-10-05  6:40 ` Ankit Soni [this message]
2026-10-05  6:40 ` [RFC PATCH 5/7] iommu/amd: clear unpreserved DTEs and quiesce logs at live-update shutdown Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 6/7] iommu/amd: restore preserved state on a live-update boot Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 7/7] iommu/amd: reattach preserved devices to their restored domains Ankit Soni

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261005064018.1558-5-Ankit.Soni@amd.com \
    --to=ankit.soni@amd.com \
    --cc=alejandro.j.jimenez@oracle.com \
    --cc=baolu.lu@linux.intel.com \
    --cc=dmatlack@google.com \
    --cc=dwmw2@infradead.org \
    --cc=iommu@lists.linux.dev \
    --cc=jgg@nvidia.com \
    --cc=joao.m.martins@oracle.com \
    --cc=joro@8bytes.org \
    --cc=kevin.tian@intel.com \
    --cc=kexec@lists.infradead.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=pasha.tatashin@soleen.com \
    --cc=praan@google.com \
    --cc=pratyush@kernel.org \
    --cc=robin.murphy@arm.com \
    --cc=rppt@kernel.org \
    --cc=skhawaja@google.com \
    --cc=suravee.suthikulpanit@amd.com \
    --cc=vasant.hegde@amd.com \
    --cc=vipinsh@google.com \
    --cc=will@kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®