mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Ankit Soni <Ankit.Soni@amd.com>
To: <iommu@lists.linux.dev>, <joro@8bytes.org>, <will@kernel.org>,
	<jgg@nvidia.com>
Cc: <suravee.suthikulpanit@amd.com>, <vasant.hegde@amd.com>,
	<robin.murphy@arm.com>, <joao.m.martins@oracle.com>,
	<alejandro.j.jimenez@oracle.com>, <pasha.tatashin@soleen.com>,
	<rppt@kernel.org>, <pratyush@kernel.org>, <skhawaja@google.com>,
	<praan@google.com>, <baolu.lu@linux.intel.com>,
	<dwmw2@infradead.org>, <kevin.tian@intel.com>,
	<dmatlack@google.com>, <vipinsh@google.com>,
	<kexec@lists.infradead.org>, <linux-kernel@vger.kernel.org>
Subject: [RFC PATCH 6/7] iommu/amd: restore preserved state on a live-update boot
Date: Mon, 5 Oct 2026 06:40:16 +0000	[thread overview]
Message-ID: <20261005064018.1558-7-Ankit.Soni@amd.com> (raw)
In-Reply-To: <20261005064018.1558-1-Ankit.Soni@amd.com>

Adopt the preserved device table for the PCI segment instead of
allocating a fresh one, and reserve the domain IDs it already carries so
a domain allocated by this kernel cannot be handed an ID the hardware is
still tagging cache entries with.

Also stop clearing the enable bit on a unit that was handed over
translating. Doing so would drop its devices into untranslated
passthrough while their DMA is still in flight.

Signed-off-by: Ankit Soni <Ankit.Soni@amd.com>
---
 drivers/iommu/amd/amd_iommu.h  |   7 ++
 drivers/iommu/amd/init.c       | 165 ++++++++++++++++++++++++++-------
 drivers/iommu/amd/liveupdate.c |  33 +++++++
 3 files changed, 174 insertions(+), 31 deletions(-)

diff --git a/drivers/iommu/amd/amd_iommu.h b/drivers/iommu/amd/amd_iommu.h
index 4402724bfd06..ac6d7a17eb40 100644
--- a/drivers/iommu/amd/amd_iommu.h
+++ b/drivers/iommu/amd/amd_iommu.h
@@ -237,9 +237,16 @@ int amd_iommu_preserve_device(struct device *dev,
 void amd_iommu_unpreserve_device(struct device *dev,
 				 struct iommu_device_ser *device_ser);
 void amd_iommu_clear_unpreserved_dtes(struct amd_iommu *iommu);
+void amd_iommu_restore_dev_table(struct amd_iommu *iommu,
+				 struct iommu_hw_ser *ser);
 #else
 static inline void amd_iommu_clear_unpreserved_dtes(struct amd_iommu *iommu)
 {
 }
+
+static inline void amd_iommu_restore_dev_table(struct amd_iommu *iommu,
+					       struct iommu_hw_ser *ser)
+{
+}
 #endif /* CONFIG_IOMMU_LIVEUPDATE */
 #endif /* AMD_IOMMU_H */
diff --git a/drivers/iommu/amd/init.c b/drivers/iommu/amd/init.c
index 3b46d46f7143..8aba3b715a02 100644
--- a/drivers/iommu/amd/init.c
+++ b/drivers/iommu/amd/init.c
@@ -33,6 +33,7 @@
 
 #include <linux/crash_dump.h>
 #include <linux/iommu-liveupdate.h>
+#include <linux/kexec_handover.h>
 
 #include "amd_iommu.h"
 #include "../irq_remapping.h"
@@ -1135,14 +1136,42 @@ static void set_dte_bit(struct dev_table_entry *dte, u8 bit)
 	dte->data[i] |= (1UL << _bit);
 }
 
-static bool __reuse_device_table(struct amd_iommu *iommu)
+/*
+ * Reserve the domain IDs the previous kernel programmed into the adopted Device
+ * Table, so this kernel never hands out an ID the hardware still tags cache
+ * entries with. @pci_seg->old_dev_tbl_cpy must already point at that table.
+ */
+static bool reserve_dev_table_domain_ids(struct amd_iommu_pci_seg *pci_seg)
 {
-	struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
 	struct dev_table_entry *old_dev_tbl_entry;
-	u32 lo, hi, old_devtb_size, devid;
-	phys_addr_t old_devtb_phys;
+	u32 devid;
 	u16 dom_id;
 	bool dte_v;
+
+	for (devid = 0; devid <= pci_seg->last_bdf; devid++) {
+		old_dev_tbl_entry = &pci_seg->old_dev_tbl_cpy[devid];
+		dte_v = FIELD_GET(DTE_FLAG_V, old_dev_tbl_entry->data[0]);
+		dom_id = FIELD_GET(DTE_DOMID_MASK, old_dev_tbl_entry->data[1]);
+
+		if (!dte_v || !dom_id)
+			continue;
+		/*
+		 * ID reservation can fail with -ENOSPC when there
+		 * are multiple devices present in the same domain,
+		 * hence check only for -ENOMEM.
+		 */
+		if (amd_iommu_pdom_id_reserve(dom_id, GFP_KERNEL) == -ENOMEM)
+			return false;
+	}
+
+	return true;
+}
+
+static bool __reuse_device_table(struct amd_iommu *iommu)
+{
+	struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
+	u32 lo, hi, old_devtb_size;
+	phys_addr_t old_devtb_phys;
 	u64 entry;
 
 	/* Each IOMMU use separate device table with the same size */
@@ -1178,22 +1207,6 @@ static bool __reuse_device_table(struct amd_iommu *iommu)
 		return false;
 	}
 
-	for (devid = 0; devid <= pci_seg->last_bdf; devid++) {
-		old_dev_tbl_entry = &pci_seg->old_dev_tbl_cpy[devid];
-		dte_v = FIELD_GET(DTE_FLAG_V, old_dev_tbl_entry->data[0]);
-		dom_id = FIELD_GET(DTE_DOMID_MASK, old_dev_tbl_entry->data[1]);
-
-		if (!dte_v || !dom_id)
-			continue;
-		/*
-		 * ID reservation can fail with -ENOSPC when there
-		 * are multiple devices present in the same domain,
-		 * hence check only for -ENOMEM.
-		 */
-		if (amd_iommu_pdom_id_reserve(dom_id, GFP_KERNEL) == -ENOMEM)
-			return false;
-	}
-
 	return true;
 }
 
@@ -1201,27 +1214,69 @@ static bool reuse_device_table(void)
 {
 	struct amd_iommu *iommu;
 	struct amd_iommu_pci_seg *pci_seg;
-
-	if (!amd_iommu_pre_enabled)
-		return false;
-
-	pr_warn("Translation is already enabled - trying to reuse translation structures\n");
+	bool reused = false;
 
 	/*
 	 * All IOMMUs within PCI segment shares common device table.
 	 * Hence reuse device table only once per PCI segment.
 	 */
 	for_each_pci_segment(pci_seg) {
+		struct amd_iommu *preserved_iommu = NULL;
+		struct iommu_hw_ser *ser = NULL;
+
+		/*
+		 * The device table is shared by the whole segment and pinned
+		 * by whichever unit preserved it, so any member holding
+		 * preserved state can describe it for all of them.
+		 */
+		for_each_iommu(iommu) {
+			if (pci_seg->id != iommu->pci_seg->id)
+				continue;
+
+			ser = iommu_get_preserved_data(iommu->mmio_phys,
+						       IOMMU_AMD);
+			if (ser) {
+				preserved_iommu = iommu;
+				break;
+			}
+		}
+
+		if (preserved_iommu) {
+			amd_iommu_restore_dev_table(preserved_iommu, ser);
+			/*
+			 * Fatal for the same reason a failed table adopt is;
+			 * see amd_iommu_restore_dev_table().
+			 */
+			if (!reserve_dev_table_domain_ids(pci_seg))
+				panic("AMD-Vi: IOMMU:%d cannot reserve the preserved domain IDs\n",
+				      preserved_iommu->index);
+			reused = true;
+			continue;
+		}
+
+		/*
+		 * Nothing was handed over for this segment. The kdump path can
+		 * still remap the previous kernel's table, but only when every
+		 * IOMMU came up with translation already enabled.
+		 */
+		if (!amd_iommu_pre_enabled)
+			return false;
+
+		pr_warn("Translation is already enabled - trying to reuse translation structures\n");
+
 		for_each_iommu(iommu) {
 			if (pci_seg->id != iommu->pci_seg->id)
 				continue;
 			if (!__reuse_device_table(iommu))
 				return false;
+			if (!reserve_dev_table_domain_ids(pci_seg))
+				return false;
+			reused = true;
 			break;
 		}
 	}
 
-	return true;
+	return reused;
 }
 
 struct dev_table_entry *amd_iommu_get_ivhd_dte_flags(u16 segid, u16 devid)
@@ -1948,6 +2003,7 @@ static int __init init_iommu_one(struct amd_iommu *iommu, struct ivhd_header *h,
 
 static int __init init_iommu_one_late(struct amd_iommu *iommu)
 {
+	struct iommu_hw_ser *ser;
 	int ret;
 
 	ret = alloc_iommu_buffers(iommu);
@@ -1957,7 +2013,25 @@ static int __init init_iommu_one_late(struct amd_iommu *iommu)
 	iommu->int_enabled = false;
 
 	init_translation_status(iommu);
-	if (translation_pre_enabled(iommu) && !is_kdump_kernel()) {
+
+	/* The MMIO physical base is the handoff token; NULL means a cold boot. */
+	ser = iommu_get_preserved_data(iommu->mmio_phys, IOMMU_AMD);
+
+	if (translation_pre_enabled(iommu) && !is_kdump_kernel() && !ser) {
+		/*
+		 * A handover boot that left this unit translating but produced
+		 * no record for it is not a stale enable. Either the handover
+		 * was never consumable (this kernel booted without
+		 * liveupdate=1) or it was not visible yet. Clearing the enable
+		 * bit drops every device behind this unit into untranslated
+		 * passthrough while its DMA is still in flight, aimed at
+		 * addresses that mean nothing in this kernel's physical map,
+		 * so stop instead of doing it silently.
+		 */
+		if (is_kho_boot())
+			panic("AMD-Vi: IOMMU:%d was handed over translating but no preserved state is available; refusing to disable it under live DMA\n",
+			      iommu->index);
+
 		iommu_disable(iommu);
 		clear_translation_pre_enabled(iommu);
 		pr_warn("Translation was enabled for IOMMU:%d but we are not in kdump mode\n",
@@ -2944,6 +3018,18 @@ static void early_enable_iommus(void)
 		}
 
 		for_each_iommu(iommu) {
+			/*
+			 * Units with no preserved devices come back with
+			 * translation off. Programming their buffers without
+			 * enabling them leaves amd_iommu_flush_all_caches()
+			 * waiting on a completion idle hardware never posts,
+			 * so bring them up the normal way.
+			 */
+			if (!translation_pre_enabled(iommu)) {
+				early_enable_iommu(iommu);
+				continue;
+			}
+
 			iommu_disable_command_buffer(iommu);
 			iommu_disable_event_buffer(iommu);
 			iommu_disable_irtcachedis(iommu);
@@ -3033,12 +3119,25 @@ static void enable_iommus_vapic(void)
 #endif
 }
 
-static void disable_iommus(void)
+static bool iommu_was_handed_over(struct amd_iommu *iommu)
+{
+	struct iommu_hw_ser *ser;
+
+	ser = iommu_get_preserved_data(iommu->mmio_phys, IOMMU_AMD);
+
+	return ser;
+}
+
+static void __disable_iommus(bool keep_handed_over)
 {
 	struct amd_iommu *iommu;
 
-	for_each_iommu(iommu)
+	for_each_iommu(iommu) {
+		if (keep_handed_over && iommu_was_handed_over(iommu))
+			continue;
+
 		iommu_disable(iommu);
+	}
 
 #ifdef CONFIG_IRQ_REMAP
 	if (AMD_IOMMU_GUEST_IR_VAPIC(amd_iommu_guest_ir))
@@ -3046,6 +3145,11 @@ static void disable_iommus(void)
 #endif
 }
 
+static void disable_iommus(void)
+{
+	__disable_iommus(false);
+}
+
 /*
  * Bound the wait for a log engine to report itself idle. An engine only has
  * to finish a write it has already started, which takes microseconds, so this
@@ -3388,9 +3492,8 @@ static int __init early_amd_iommu_init(void)
 			amd_iommu_pgtable = PD_MODE_NONE;
 	}
 
-	/* Disable any previously enabled IOMMUs */
 	if (!is_kdump_kernel() || amd_iommu_disabled)
-		disable_iommus();
+		__disable_iommus(true);
 
 	if (amd_iommu_irq_remap)
 		amd_iommu_irq_remap = check_ioapic_information();
diff --git a/drivers/iommu/amd/liveupdate.c b/drivers/iommu/amd/liveupdate.c
index a3a9ebea5138..2e9a1de6b1ff 100644
--- a/drivers/iommu/amd/liveupdate.c
+++ b/drivers/iommu/amd/liveupdate.c
@@ -362,3 +362,36 @@ void amd_iommu_clear_unpreserved_dtes(struct amd_iommu *iommu)
 
 	amd_iommu_flush_all_caches(iommu);
 }
+
+void amd_iommu_restore_dev_table(struct amd_iommu *iommu,
+				 struct iommu_hw_ser *ser)
+{
+	struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
+
+	if (ser->amd.dev_table_size != pci_seg->dev_table_size)
+		panic("AMD-Vi: IOMMU:%d preserved device table size mismatch (0x%x vs 0x%x)\n",
+		      iommu->index, ser->amd.dev_table_size,
+		      pci_seg->dev_table_size);
+
+	if (ser->amd.pci_seg_id != pci_seg->id)
+		panic("AMD-Vi: IOMMU:%d preserved PCI segment mismatch (%u vs %u)\n",
+		      iommu->index, ser->amd.pci_seg_id, pci_seg->id);
+
+	if (ser->amd.efr != iommu->features ||
+	    ser->amd.efr2 != iommu->features2)
+		panic("AMD-Vi: IOMMU:%d preserved feature registers mismatch (EFR 0x%llx/0x%llx vs 0x%llx/0x%llx)\n",
+		      iommu->index, ser->amd.efr, ser->amd.efr2,
+		      iommu->features, iommu->features2);
+
+	/*
+	 * Reclaim the folio from KHO the first time this segment's table is
+	 * seen, then adopt it. Later IOMMUs in the same segment reference the
+	 * same physical table and must not restore it again.
+	 */
+	if (!ser->amd.restored) {
+		iommu_restore_pages(ser->amd.dev_table_phys);
+		ser->amd.restored = 1;
+	}
+
+	pci_seg->old_dev_tbl_cpy = __va(ser->amd.dev_table_phys);
+}
-- 
2.43.0


  parent reply	other threads:[~2026-10-05  6:43 UTC|newest]

Thread overview: 8+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-05  6:40 [RFC PATCH 0/7] iommu/amd: Implement live update state preservation Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 1/7] iommu/amd: defer device attach only on a kdump boot Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 2/7] liveupdate: parse the incoming handover tree before late_time_init() Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 3/7] iommu/kho/abi: add AMD IOMMU live-update serialisation structs Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 4/7] iommu/amd: preserve IOMMU and device state for live update Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 5/7] iommu/amd: clear unpreserved DTEs and quiesce logs at live-update shutdown Ankit Soni
2026-10-05  6:40 ` Ankit Soni [this message]
2026-10-05  6:40 ` [RFC PATCH 7/7] iommu/amd: reattach preserved devices to their restored domains Ankit Soni

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261005064018.1558-7-Ankit.Soni@amd.com \
    --to=ankit.soni@amd.com \
    --cc=alejandro.j.jimenez@oracle.com \
    --cc=baolu.lu@linux.intel.com \
    --cc=dmatlack@google.com \
    --cc=dwmw2@infradead.org \
    --cc=iommu@lists.linux.dev \
    --cc=jgg@nvidia.com \
    --cc=joao.m.martins@oracle.com \
    --cc=joro@8bytes.org \
    --cc=kevin.tian@intel.com \
    --cc=kexec@lists.infradead.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=pasha.tatashin@soleen.com \
    --cc=praan@google.com \
    --cc=pratyush@kernel.org \
    --cc=robin.murphy@arm.com \
    --cc=rppt@kernel.org \
    --cc=skhawaja@google.com \
    --cc=suravee.suthikulpanit@amd.com \
    --cc=vasant.hegde@amd.com \
    --cc=vipinsh@google.com \
    --cc=will@kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®