From: Ankit Soni <Ankit.Soni@amd.com>
To: <iommu@lists.linux.dev>, <joro@8bytes.org>, <will@kernel.org>,
<jgg@nvidia.com>
Cc: <suravee.suthikulpanit@amd.com>, <vasant.hegde@amd.com>,
<robin.murphy@arm.com>, <joao.m.martins@oracle.com>,
<alejandro.j.jimenez@oracle.com>, <pasha.tatashin@soleen.com>,
<rppt@kernel.org>, <pratyush@kernel.org>, <skhawaja@google.com>,
<praan@google.com>, <baolu.lu@linux.intel.com>,
<dwmw2@infradead.org>, <kevin.tian@intel.com>,
<dmatlack@google.com>, <vipinsh@google.com>,
<kexec@lists.infradead.org>, <linux-kernel@vger.kernel.org>
Subject: [RFC PATCH 6/7] iommu/amd: restore preserved state on a live-update boot
Date: Mon, 5 Oct 2026 06:40:16 +0000 [thread overview]
Message-ID: <20261005064018.1558-7-Ankit.Soni@amd.com> (raw)
In-Reply-To: <20261005064018.1558-1-Ankit.Soni@amd.com>
Adopt the preserved device table for the PCI segment instead of
allocating a fresh one, and reserve the domain IDs it already carries so
a domain allocated by this kernel cannot be handed an ID the hardware is
still tagging cache entries with.
Also stop clearing the enable bit on a unit that was handed over
translating. Doing so would drop its devices into untranslated
passthrough while their DMA is still in flight.
Signed-off-by: Ankit Soni <Ankit.Soni@amd.com>
---
drivers/iommu/amd/amd_iommu.h | 7 ++
drivers/iommu/amd/init.c | 165 ++++++++++++++++++++++++++-------
drivers/iommu/amd/liveupdate.c | 33 +++++++
3 files changed, 174 insertions(+), 31 deletions(-)
diff --git a/drivers/iommu/amd/amd_iommu.h b/drivers/iommu/amd/amd_iommu.h
index 4402724bfd06..ac6d7a17eb40 100644
--- a/drivers/iommu/amd/amd_iommu.h
+++ b/drivers/iommu/amd/amd_iommu.h
@@ -237,9 +237,16 @@ int amd_iommu_preserve_device(struct device *dev,
void amd_iommu_unpreserve_device(struct device *dev,
struct iommu_device_ser *device_ser);
void amd_iommu_clear_unpreserved_dtes(struct amd_iommu *iommu);
+void amd_iommu_restore_dev_table(struct amd_iommu *iommu,
+ struct iommu_hw_ser *ser);
#else
static inline void amd_iommu_clear_unpreserved_dtes(struct amd_iommu *iommu)
{
}
+
+static inline void amd_iommu_restore_dev_table(struct amd_iommu *iommu,
+ struct iommu_hw_ser *ser)
+{
+}
#endif /* CONFIG_IOMMU_LIVEUPDATE */
#endif /* AMD_IOMMU_H */
diff --git a/drivers/iommu/amd/init.c b/drivers/iommu/amd/init.c
index 3b46d46f7143..8aba3b715a02 100644
--- a/drivers/iommu/amd/init.c
+++ b/drivers/iommu/amd/init.c
@@ -33,6 +33,7 @@
#include <linux/crash_dump.h>
#include <linux/iommu-liveupdate.h>
+#include <linux/kexec_handover.h>
#include "amd_iommu.h"
#include "../irq_remapping.h"
@@ -1135,14 +1136,42 @@ static void set_dte_bit(struct dev_table_entry *dte, u8 bit)
dte->data[i] |= (1UL << _bit);
}
-static bool __reuse_device_table(struct amd_iommu *iommu)
+/*
+ * Reserve the domain IDs the previous kernel programmed into the adopted Device
+ * Table, so this kernel never hands out an ID the hardware still tags cache
+ * entries with. @pci_seg->old_dev_tbl_cpy must already point at that table.
+ */
+static bool reserve_dev_table_domain_ids(struct amd_iommu_pci_seg *pci_seg)
{
- struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
struct dev_table_entry *old_dev_tbl_entry;
- u32 lo, hi, old_devtb_size, devid;
- phys_addr_t old_devtb_phys;
+ u32 devid;
u16 dom_id;
bool dte_v;
+
+ for (devid = 0; devid <= pci_seg->last_bdf; devid++) {
+ old_dev_tbl_entry = &pci_seg->old_dev_tbl_cpy[devid];
+ dte_v = FIELD_GET(DTE_FLAG_V, old_dev_tbl_entry->data[0]);
+ dom_id = FIELD_GET(DTE_DOMID_MASK, old_dev_tbl_entry->data[1]);
+
+ if (!dte_v || !dom_id)
+ continue;
+ /*
+ * ID reservation can fail with -ENOSPC when there
+ * are multiple devices present in the same domain,
+ * hence check only for -ENOMEM.
+ */
+ if (amd_iommu_pdom_id_reserve(dom_id, GFP_KERNEL) == -ENOMEM)
+ return false;
+ }
+
+ return true;
+}
+
+static bool __reuse_device_table(struct amd_iommu *iommu)
+{
+ struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
+ u32 lo, hi, old_devtb_size;
+ phys_addr_t old_devtb_phys;
u64 entry;
/* Each IOMMU use separate device table with the same size */
@@ -1178,22 +1207,6 @@ static bool __reuse_device_table(struct amd_iommu *iommu)
return false;
}
- for (devid = 0; devid <= pci_seg->last_bdf; devid++) {
- old_dev_tbl_entry = &pci_seg->old_dev_tbl_cpy[devid];
- dte_v = FIELD_GET(DTE_FLAG_V, old_dev_tbl_entry->data[0]);
- dom_id = FIELD_GET(DTE_DOMID_MASK, old_dev_tbl_entry->data[1]);
-
- if (!dte_v || !dom_id)
- continue;
- /*
- * ID reservation can fail with -ENOSPC when there
- * are multiple devices present in the same domain,
- * hence check only for -ENOMEM.
- */
- if (amd_iommu_pdom_id_reserve(dom_id, GFP_KERNEL) == -ENOMEM)
- return false;
- }
-
return true;
}
@@ -1201,27 +1214,69 @@ static bool reuse_device_table(void)
{
struct amd_iommu *iommu;
struct amd_iommu_pci_seg *pci_seg;
-
- if (!amd_iommu_pre_enabled)
- return false;
-
- pr_warn("Translation is already enabled - trying to reuse translation structures\n");
+ bool reused = false;
/*
* All IOMMUs within PCI segment shares common device table.
* Hence reuse device table only once per PCI segment.
*/
for_each_pci_segment(pci_seg) {
+ struct amd_iommu *preserved_iommu = NULL;
+ struct iommu_hw_ser *ser = NULL;
+
+ /*
+ * The device table is shared by the whole segment and pinned
+ * by whichever unit preserved it, so any member holding
+ * preserved state can describe it for all of them.
+ */
+ for_each_iommu(iommu) {
+ if (pci_seg->id != iommu->pci_seg->id)
+ continue;
+
+ ser = iommu_get_preserved_data(iommu->mmio_phys,
+ IOMMU_AMD);
+ if (ser) {
+ preserved_iommu = iommu;
+ break;
+ }
+ }
+
+ if (preserved_iommu) {
+ amd_iommu_restore_dev_table(preserved_iommu, ser);
+ /*
+ * Fatal for the same reason a failed table adopt is;
+ * see amd_iommu_restore_dev_table().
+ */
+ if (!reserve_dev_table_domain_ids(pci_seg))
+ panic("AMD-Vi: IOMMU:%d cannot reserve the preserved domain IDs\n",
+ preserved_iommu->index);
+ reused = true;
+ continue;
+ }
+
+ /*
+ * Nothing was handed over for this segment. The kdump path can
+ * still remap the previous kernel's table, but only when every
+ * IOMMU came up with translation already enabled.
+ */
+ if (!amd_iommu_pre_enabled)
+ return false;
+
+ pr_warn("Translation is already enabled - trying to reuse translation structures\n");
+
for_each_iommu(iommu) {
if (pci_seg->id != iommu->pci_seg->id)
continue;
if (!__reuse_device_table(iommu))
return false;
+ if (!reserve_dev_table_domain_ids(pci_seg))
+ return false;
+ reused = true;
break;
}
}
- return true;
+ return reused;
}
struct dev_table_entry *amd_iommu_get_ivhd_dte_flags(u16 segid, u16 devid)
@@ -1948,6 +2003,7 @@ static int __init init_iommu_one(struct amd_iommu *iommu, struct ivhd_header *h,
static int __init init_iommu_one_late(struct amd_iommu *iommu)
{
+ struct iommu_hw_ser *ser;
int ret;
ret = alloc_iommu_buffers(iommu);
@@ -1957,7 +2013,25 @@ static int __init init_iommu_one_late(struct amd_iommu *iommu)
iommu->int_enabled = false;
init_translation_status(iommu);
- if (translation_pre_enabled(iommu) && !is_kdump_kernel()) {
+
+ /* The MMIO physical base is the handoff token; NULL means a cold boot. */
+ ser = iommu_get_preserved_data(iommu->mmio_phys, IOMMU_AMD);
+
+ if (translation_pre_enabled(iommu) && !is_kdump_kernel() && !ser) {
+ /*
+ * A handover boot that left this unit translating but produced
+ * no record for it is not a stale enable. Either the handover
+ * was never consumable (this kernel booted without
+ * liveupdate=1) or it was not visible yet. Clearing the enable
+ * bit drops every device behind this unit into untranslated
+ * passthrough while its DMA is still in flight, aimed at
+ * addresses that mean nothing in this kernel's physical map,
+ * so stop instead of doing it silently.
+ */
+ if (is_kho_boot())
+ panic("AMD-Vi: IOMMU:%d was handed over translating but no preserved state is available; refusing to disable it under live DMA\n",
+ iommu->index);
+
iommu_disable(iommu);
clear_translation_pre_enabled(iommu);
pr_warn("Translation was enabled for IOMMU:%d but we are not in kdump mode\n",
@@ -2944,6 +3018,18 @@ static void early_enable_iommus(void)
}
for_each_iommu(iommu) {
+ /*
+ * Units with no preserved devices come back with
+ * translation off. Programming their buffers without
+ * enabling them leaves amd_iommu_flush_all_caches()
+ * waiting on a completion idle hardware never posts,
+ * so bring them up the normal way.
+ */
+ if (!translation_pre_enabled(iommu)) {
+ early_enable_iommu(iommu);
+ continue;
+ }
+
iommu_disable_command_buffer(iommu);
iommu_disable_event_buffer(iommu);
iommu_disable_irtcachedis(iommu);
@@ -3033,12 +3119,25 @@ static void enable_iommus_vapic(void)
#endif
}
-static void disable_iommus(void)
+static bool iommu_was_handed_over(struct amd_iommu *iommu)
+{
+ struct iommu_hw_ser *ser;
+
+ ser = iommu_get_preserved_data(iommu->mmio_phys, IOMMU_AMD);
+
+ return ser;
+}
+
+static void __disable_iommus(bool keep_handed_over)
{
struct amd_iommu *iommu;
- for_each_iommu(iommu)
+ for_each_iommu(iommu) {
+ if (keep_handed_over && iommu_was_handed_over(iommu))
+ continue;
+
iommu_disable(iommu);
+ }
#ifdef CONFIG_IRQ_REMAP
if (AMD_IOMMU_GUEST_IR_VAPIC(amd_iommu_guest_ir))
@@ -3046,6 +3145,11 @@ static void disable_iommus(void)
#endif
}
+static void disable_iommus(void)
+{
+ __disable_iommus(false);
+}
+
/*
* Bound the wait for a log engine to report itself idle. An engine only has
* to finish a write it has already started, which takes microseconds, so this
@@ -3388,9 +3492,8 @@ static int __init early_amd_iommu_init(void)
amd_iommu_pgtable = PD_MODE_NONE;
}
- /* Disable any previously enabled IOMMUs */
if (!is_kdump_kernel() || amd_iommu_disabled)
- disable_iommus();
+ __disable_iommus(true);
if (amd_iommu_irq_remap)
amd_iommu_irq_remap = check_ioapic_information();
diff --git a/drivers/iommu/amd/liveupdate.c b/drivers/iommu/amd/liveupdate.c
index a3a9ebea5138..2e9a1de6b1ff 100644
--- a/drivers/iommu/amd/liveupdate.c
+++ b/drivers/iommu/amd/liveupdate.c
@@ -362,3 +362,36 @@ void amd_iommu_clear_unpreserved_dtes(struct amd_iommu *iommu)
amd_iommu_flush_all_caches(iommu);
}
+
+void amd_iommu_restore_dev_table(struct amd_iommu *iommu,
+ struct iommu_hw_ser *ser)
+{
+ struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
+
+ if (ser->amd.dev_table_size != pci_seg->dev_table_size)
+ panic("AMD-Vi: IOMMU:%d preserved device table size mismatch (0x%x vs 0x%x)\n",
+ iommu->index, ser->amd.dev_table_size,
+ pci_seg->dev_table_size);
+
+ if (ser->amd.pci_seg_id != pci_seg->id)
+ panic("AMD-Vi: IOMMU:%d preserved PCI segment mismatch (%u vs %u)\n",
+ iommu->index, ser->amd.pci_seg_id, pci_seg->id);
+
+ if (ser->amd.efr != iommu->features ||
+ ser->amd.efr2 != iommu->features2)
+ panic("AMD-Vi: IOMMU:%d preserved feature registers mismatch (EFR 0x%llx/0x%llx vs 0x%llx/0x%llx)\n",
+ iommu->index, ser->amd.efr, ser->amd.efr2,
+ iommu->features, iommu->features2);
+
+ /*
+ * Reclaim the folio from KHO the first time this segment's table is
+ * seen, then adopt it. Later IOMMUs in the same segment reference the
+ * same physical table and must not restore it again.
+ */
+ if (!ser->amd.restored) {
+ iommu_restore_pages(ser->amd.dev_table_phys);
+ ser->amd.restored = 1;
+ }
+
+ pci_seg->old_dev_tbl_cpy = __va(ser->amd.dev_table_phys);
+}
--
2.43.0
next prev parent reply other threads:[~2026-10-05 6:43 UTC|newest]
Thread overview: 8+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-05 6:40 [RFC PATCH 0/7] iommu/amd: Implement live update state preservation Ankit Soni
2026-10-05 6:40 ` [RFC PATCH 1/7] iommu/amd: defer device attach only on a kdump boot Ankit Soni
2026-10-05 6:40 ` [RFC PATCH 2/7] liveupdate: parse the incoming handover tree before late_time_init() Ankit Soni
2026-10-05 6:40 ` [RFC PATCH 3/7] iommu/kho/abi: add AMD IOMMU live-update serialisation structs Ankit Soni
2026-10-05 6:40 ` [RFC PATCH 4/7] iommu/amd: preserve IOMMU and device state for live update Ankit Soni
2026-10-05 6:40 ` [RFC PATCH 5/7] iommu/amd: clear unpreserved DTEs and quiesce logs at live-update shutdown Ankit Soni
2026-10-05 6:40 ` Ankit Soni [this message]
2026-10-05 6:40 ` [RFC PATCH 7/7] iommu/amd: reattach preserved devices to their restored domains Ankit Soni
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261005064018.1558-7-Ankit.Soni@amd.com \
--to=ankit.soni@amd.com \
--cc=alejandro.j.jimenez@oracle.com \
--cc=baolu.lu@linux.intel.com \
--cc=dmatlack@google.com \
--cc=dwmw2@infradead.org \
--cc=iommu@lists.linux.dev \
--cc=jgg@nvidia.com \
--cc=joao.m.martins@oracle.com \
--cc=joro@8bytes.org \
--cc=kevin.tian@intel.com \
--cc=kexec@lists.infradead.org \
--cc=linux-kernel@vger.kernel.org \
--cc=pasha.tatashin@soleen.com \
--cc=praan@google.com \
--cc=pratyush@kernel.org \
--cc=robin.murphy@arm.com \
--cc=rppt@kernel.org \
--cc=skhawaja@google.com \
--cc=suravee.suthikulpanit@amd.com \
--cc=vasant.hegde@amd.com \
--cc=vipinsh@google.com \
--cc=will@kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®