mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Ankit Soni <Ankit.Soni@amd.com>
To: <iommu@lists.linux.dev>, <joro@8bytes.org>, <will@kernel.org>,
	<jgg@nvidia.com>
Cc: <suravee.suthikulpanit@amd.com>, <vasant.hegde@amd.com>,
	<robin.murphy@arm.com>, <joao.m.martins@oracle.com>,
	<alejandro.j.jimenez@oracle.com>, <pasha.tatashin@soleen.com>,
	<rppt@kernel.org>, <pratyush@kernel.org>, <skhawaja@google.com>,
	<praan@google.com>, <baolu.lu@linux.intel.com>,
	<dwmw2@infradead.org>, <kevin.tian@intel.com>,
	<dmatlack@google.com>, <vipinsh@google.com>,
	<kexec@lists.infradead.org>, <linux-kernel@vger.kernel.org>
Subject: [RFC PATCH 5/7] iommu/amd: clear unpreserved DTEs and quiesce logs at live-update shutdown
Date: Mon, 5 Oct 2026 06:40:15 +0000	[thread overview]
Message-ID: <20261005064018.1558-6-Ankit.Soni@amd.com> (raw)
In-Reply-To: <20261005064018.1558-1-Ankit.Soni@amd.com>

Keep translation enabled on the units carrying preserved devices, but
block every device that was not preserved, so the next kernel cannot
translate through page tables it did not inherit.

Then stop the command, event, PPR and GA engines and wait for each to
report itself idle. Those buffers are not preserved, so an engine still
running would write into pages the next kernel is free to reuse.

If an engine does not go idle within the timeout, panic rather than
continue. Completing the handover would leave a running DMA engine
writing into pages the next kernel owns, and that corruption is both
silent and impossible to attribute later. Failing the live update is the
lesser harm.

Signed-off-by: Ankit Soni <Ankit.Soni@amd.com>
---
 drivers/iommu/amd/amd_iommu.h  |  5 ++
 drivers/iommu/amd/init.c       | 88 +++++++++++++++++++++++++++++++++-
 drivers/iommu/amd/liveupdate.c | 83 ++++++++++++++++++++++++++++++++
 3 files changed, 175 insertions(+), 1 deletion(-)

diff --git a/drivers/iommu/amd/amd_iommu.h b/drivers/iommu/amd/amd_iommu.h
index 5cf32e4898dc..4402724bfd06 100644
--- a/drivers/iommu/amd/amd_iommu.h
+++ b/drivers/iommu/amd/amd_iommu.h
@@ -236,5 +236,10 @@ int amd_iommu_preserve_device(struct device *dev,
 			      struct iommu_device_ser *device_ser);
 void amd_iommu_unpreserve_device(struct device *dev,
 				 struct iommu_device_ser *device_ser);
+void amd_iommu_clear_unpreserved_dtes(struct amd_iommu *iommu);
+#else
+static inline void amd_iommu_clear_unpreserved_dtes(struct amd_iommu *iommu)
+{
+}
 #endif /* CONFIG_IOMMU_LIVEUPDATE */
 #endif /* AMD_IOMMU_H */
diff --git a/drivers/iommu/amd/init.c b/drivers/iommu/amd/init.c
index 40726dfef273..3b46d46f7143 100644
--- a/drivers/iommu/amd/init.c
+++ b/drivers/iommu/amd/init.c
@@ -32,6 +32,7 @@
 #include <asm/sev.h>
 
 #include <linux/crash_dump.h>
+#include <linux/iommu-liveupdate.h>
 
 #include "amd_iommu.h"
 #include "../irq_remapping.h"
@@ -3045,6 +3046,91 @@ static void disable_iommus(void)
 #endif
 }
 
+/*
+ * Bound the wait for a log engine to report itself idle. An engine only has
+ * to finish a write it has already started, which takes microseconds, so this
+ * is ample. It is deliberately far shorter than MMIO_STATUS_TIMEOUT because
+ * this runs on the live-update shutdown path, where any stall is downtime.
+ */
+#define LU_LOG_QUIESCE_RETRIES	10000		/* x 10us = 100ms */
+
+/*
+ * Clearing a log's enable bit only requests a stop; the engine may still be
+ * completing a write. Each log reports its real state in a separate "running"
+ * status bit, so wait for that to clear before handing over.
+ */
+static void wait_log_stopped(struct amd_iommu *iommu, u32 run_mask,
+			     const char *name)
+{
+	u32 status;
+	int i;
+
+	for (i = 0; i < LU_LOG_QUIESCE_RETRIES; ++i) {
+		status = readl(iommu->mmio_base + MMIO_STATUS_OFFSET);
+		if (!(status & run_mask))
+			return;
+		udelay(10);
+	}
+
+	/*
+	 * The buffer is not preserved, so the next kernel is free to reuse
+	 * these pages. An engine still running here would keep writing into
+	 * them after the kexec, corrupting whatever the next kernel puts
+	 * there. That corruption is silent and unattributable, so refuse the
+	 * handover instead of completing one we cannot prove is safe.
+	 */
+	panic("AMD-Vi: IOMMU:%d %s log still running at handover; refusing to hand over a running DMA engine\n",
+	      iommu->index, name);
+}
+
+/*
+ * Stop the hardware writing event/PPR/GA logs and reading the command
+ * buffer, without turning translation off. Call after DTE cleanup: the
+ * cache flush still needs the command buffer. The next kernel allocates
+ * fresh buffers.
+ */
+static void amd_iommu_quiesce_logs(struct amd_iommu *iommu)
+{
+	/*
+	 * The completion wait at the end of amd_iommu_clear_unpreserved_dtes()
+	 * has already drained the command buffer, so there is nothing left for
+	 * the hardware to read.
+	 */
+	iommu_disable_command_buffer(iommu);
+
+	iommu_feature_disable(iommu, CONTROL_EVT_INT_EN);
+	iommu_disable_event_buffer(iommu);
+	wait_log_stopped(iommu, MMIO_STATUS_EVT_RUN_MASK, "event");
+
+	iommu_feature_disable(iommu, CONTROL_GAINT_EN);
+	iommu_feature_disable(iommu, CONTROL_GALOG_EN);
+	wait_log_stopped(iommu, MMIO_STATUS_GALOG_RUN_MASK, "GA");
+
+	iommu_feature_disable(iommu, CONTROL_PPRINT_EN);
+	iommu_feature_disable(iommu, CONTROL_PPRLOG_EN);
+	iommu_feature_disable(iommu, CONTROL_PPR_EN);
+	wait_log_stopped(iommu, MMIO_STATUS_PPR_RUN_MASK, "PPR");
+}
+
+static void amd_iommu_shutdown(void)
+{
+	struct amd_iommu *iommu;
+
+	for_each_iommu(iommu) {
+		if (iommu_preserved_state(&iommu->iommu)) {
+			amd_iommu_clear_unpreserved_dtes(iommu);
+			amd_iommu_quiesce_logs(iommu);
+		} else {
+			iommu_disable(iommu);
+		}
+	}
+
+#ifdef CONFIG_IRQ_REMAP
+	if (AMD_IOMMU_GUEST_IR_VAPIC(amd_iommu_guest_ir))
+		amd_iommu_irq_ops.capability &= ~(1 << IRQ_POSTING_CAP);
+#endif
+}
+
 /*
  * Suspend/Resume support
  * disable suspend until real resume implemented
@@ -3500,7 +3586,7 @@ static int __init state_next(void)
 		break;
 	case IOMMU_ACPI_FINISHED:
 		early_enable_iommus();
-		x86_platform.iommu_shutdown = disable_iommus;
+		x86_platform.iommu_shutdown = amd_iommu_shutdown;
 		init_state = IOMMU_ENABLED;
 		break;
 	case IOMMU_ENABLED:
diff --git a/drivers/iommu/amd/liveupdate.c b/drivers/iommu/amd/liveupdate.c
index 096a23bb4e7b..a3a9ebea5138 100644
--- a/drivers/iommu/amd/liveupdate.c
+++ b/drivers/iommu/amd/liveupdate.c
@@ -279,3 +279,86 @@ void amd_iommu_unpreserve_device(struct device *dev,
 
 	unpreserve_gcr3_level(gcr3_info->gcr3_tbl, gcr3_info->glx);
 }
+
+/*
+ * Reset one non-preserved device's DTE to the blocked state during live-update
+ * shutdown. Every unpreserved device is reset so the next kernel starts from a
+ * clean slate for it and cannot translate through a domain whose page tables
+ * were not preserved.
+ */
+static int clear_unpreserved_dte(struct device *dev,
+				 struct iommu_device *iommu_dev, void *arg)
+{
+	struct amd_iommu *iommu = container_of(iommu_dev, struct amd_iommu,
+					       iommu);
+	struct dev_table_entry new = {};
+	struct iommu_dev_data *dev_data;
+
+	dev_data = dev_iommu_priv_get(dev);
+	if (!dev_data)
+		return 0;
+
+	if (dev_is_pci(dev) && dev_iommu_preserved_state(dev))
+		return 0;
+
+	amd_iommu_make_clear_dte(dev_data, &new);
+	amd_iommu_update_dte(iommu, dev_data, &new);
+
+	return 0;
+}
+
+static void clear_irq_dtes(struct amd_iommu *iommu)
+{
+	struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
+	struct dev_table_entry *dev_table = get_dev_table(iommu);
+	u32 devid;
+	u64 dte2;
+
+	if (!amd_iommu_irq_remap)
+		return;
+
+	/*
+	 * Interrupt remapping tables are not preserved. Clear the interrupt
+	 * fields on every DTE so none still points at a table the next kernel
+	 * is free to recycle. DTE_DATA2_INTR_MASK is everything in data[2]
+	 * except the guest page-table level, which preserved devices still
+	 * need. The GCR3 pointer lives in data[0] and data[1], so it is not
+	 * affected.
+	 *
+	 * This must run after the clear_unpreserved_dte() pass, because
+	 * write_dte_upper128() deliberately copies DTE_DATA2_INTR_MASK back
+	 * from the old entry. Clearing a DTE therefore keeps its interrupt
+	 * fields, and only this walk removes them.
+	 *
+	 * The walk is by raw devid, so there is no iommu_dev_data to take
+	 * dte_lock on, and looking one up per entry would make this O(n^2).
+	 * Going without the lock is safe only because amd_iommu_shutdown()
+	 * runs from native_machine_shutdown(), after the other CPUs are
+	 * stopped and interrupts are off, so nothing can race these writes.
+	 */
+	for (devid = 0; devid <= pci_seg->last_bdf; devid++) {
+		dte2 = READ_ONCE(dev_table[devid].data[2]);
+		if (!(dte2 & DTE_IRQ_REMAP_ENABLE))
+			continue;
+
+		WRITE_ONCE(dev_table[devid].data[2],
+			   dte2 & ~DTE_DATA2_INTR_MASK);
+	}
+}
+
+/**
+ * amd_iommu_clear_unpreserved_dtes - Quiesce non-preserved devices at shutdown
+ * @iommu: The IOMMU whose device table is being cleaned up
+ */
+void amd_iommu_clear_unpreserved_dtes(struct amd_iommu *iommu)
+{
+	struct iommu_dev_iter iter = {
+		.fn = clear_unpreserved_dte,
+		.iommu = &iommu->iommu,
+	};
+
+	iommu_for_each_dev(&iter);
+	clear_irq_dtes(iommu);
+
+	amd_iommu_flush_all_caches(iommu);
+}
-- 
2.43.0


  parent reply	other threads:[~2026-10-05  6:43 UTC|newest]

Thread overview: 8+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-05  6:40 [RFC PATCH 0/7] iommu/amd: Implement live update state preservation Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 1/7] iommu/amd: defer device attach only on a kdump boot Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 2/7] liveupdate: parse the incoming handover tree before late_time_init() Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 3/7] iommu/kho/abi: add AMD IOMMU live-update serialisation structs Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 4/7] iommu/amd: preserve IOMMU and device state for live update Ankit Soni
2026-10-05  6:40 ` Ankit Soni [this message]
2026-10-05  6:40 ` [RFC PATCH 6/7] iommu/amd: restore preserved state on a live-update boot Ankit Soni
2026-10-05  6:40 ` [RFC PATCH 7/7] iommu/amd: reattach preserved devices to their restored domains Ankit Soni

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261005064018.1558-6-Ankit.Soni@amd.com \
    --to=ankit.soni@amd.com \
    --cc=alejandro.j.jimenez@oracle.com \
    --cc=baolu.lu@linux.intel.com \
    --cc=dmatlack@google.com \
    --cc=dwmw2@infradead.org \
    --cc=iommu@lists.linux.dev \
    --cc=jgg@nvidia.com \
    --cc=joao.m.martins@oracle.com \
    --cc=joro@8bytes.org \
    --cc=kevin.tian@intel.com \
    --cc=kexec@lists.infradead.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=pasha.tatashin@soleen.com \
    --cc=praan@google.com \
    --cc=pratyush@kernel.org \
    --cc=robin.murphy@arm.com \
    --cc=rppt@kernel.org \
    --cc=skhawaja@google.com \
    --cc=suravee.suthikulpanit@amd.com \
    --cc=vasant.hegde@amd.com \
    --cc=vipinsh@google.com \
    --cc=will@kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®