[RFC PATCH 5/7] iommu/amd: clear unpreserved DTEs and quiesce logs at live-update shutdown

From: Ankit Soni

Date: Mon Oct 05 2026 - 02:48:16 EST


Keep translation enabled on the units carrying preserved devices, but
block every device that was not preserved, so the next kernel cannot
translate through page tables it did not inherit.

Then stop the command, event, PPR and GA engines and wait for each to
report itself idle. Those buffers are not preserved, so an engine still
running would write into pages the next kernel is free to reuse.

If an engine does not go idle within the timeout, panic rather than
continue. Completing the handover would leave a running DMA engine
writing into pages the next kernel owns, and that corruption is both
silent and impossible to attribute later. Failing the live update is the
lesser harm.

Signed-off-by: Ankit Soni <Ankit.Soni@xxxxxxx>
---
drivers/iommu/amd/amd_iommu.h | 5 ++
drivers/iommu/amd/init.c | 88 +++++++++++++++++++++++++++++++++-
drivers/iommu/amd/liveupdate.c | 83 ++++++++++++++++++++++++++++++++
3 files changed, 175 insertions(+), 1 deletion(-)

diff --git a/drivers/iommu/amd/amd_iommu.h b/drivers/iommu/amd/amd_iommu.h
index 5cf32e4898dc..4402724bfd06 100644
--- a/drivers/iommu/amd/amd_iommu.h
+++ b/drivers/iommu/amd/amd_iommu.h
@@ -236,5 +236,10 @@ int amd_iommu_preserve_device(struct device *dev,
struct iommu_device_ser *device_ser);
void amd_iommu_unpreserve_device(struct device *dev,
struct iommu_device_ser *device_ser);
+void amd_iommu_clear_unpreserved_dtes(struct amd_iommu *iommu);
+#else
+static inline void amd_iommu_clear_unpreserved_dtes(struct amd_iommu *iommu)
+{
+}
#endif /* CONFIG_IOMMU_LIVEUPDATE */
#endif /* AMD_IOMMU_H */
diff --git a/drivers/iommu/amd/init.c b/drivers/iommu/amd/init.c
index 40726dfef273..3b46d46f7143 100644
--- a/drivers/iommu/amd/init.c
+++ b/drivers/iommu/amd/init.c
@@ -32,6 +32,7 @@
#include <asm/sev.h>

#include <linux/crash_dump.h>
+#include <linux/iommu-liveupdate.h>

#include "amd_iommu.h"
#include "../irq_remapping.h"
@@ -3045,6 +3046,91 @@ static void disable_iommus(void)
#endif
}

+/*
+ * Bound the wait for a log engine to report itself idle. An engine only has
+ * to finish a write it has already started, which takes microseconds, so this
+ * is ample. It is deliberately far shorter than MMIO_STATUS_TIMEOUT because
+ * this runs on the live-update shutdown path, where any stall is downtime.
+ */
+#define LU_LOG_QUIESCE_RETRIES 10000 /* x 10us = 100ms */
+
+/*
+ * Clearing a log's enable bit only requests a stop; the engine may still be
+ * completing a write. Each log reports its real state in a separate "running"
+ * status bit, so wait for that to clear before handing over.
+ */
+static void wait_log_stopped(struct amd_iommu *iommu, u32 run_mask,
+ const char *name)
+{
+ u32 status;
+ int i;
+
+ for (i = 0; i < LU_LOG_QUIESCE_RETRIES; ++i) {
+ status = readl(iommu->mmio_base + MMIO_STATUS_OFFSET);
+ if (!(status & run_mask))
+ return;
+ udelay(10);
+ }
+
+ /*
+ * The buffer is not preserved, so the next kernel is free to reuse
+ * these pages. An engine still running here would keep writing into
+ * them after the kexec, corrupting whatever the next kernel puts
+ * there. That corruption is silent and unattributable, so refuse the
+ * handover instead of completing one we cannot prove is safe.
+ */
+ panic("AMD-Vi: IOMMU:%d %s log still running at handover; refusing to hand over a running DMA engine\n",
+ iommu->index, name);
+}
+
+/*
+ * Stop the hardware writing event/PPR/GA logs and reading the command
+ * buffer, without turning translation off. Call after DTE cleanup: the
+ * cache flush still needs the command buffer. The next kernel allocates
+ * fresh buffers.
+ */
+static void amd_iommu_quiesce_logs(struct amd_iommu *iommu)
+{
+ /*
+ * The completion wait at the end of amd_iommu_clear_unpreserved_dtes()
+ * has already drained the command buffer, so there is nothing left for
+ * the hardware to read.
+ */
+ iommu_disable_command_buffer(iommu);
+
+ iommu_feature_disable(iommu, CONTROL_EVT_INT_EN);
+ iommu_disable_event_buffer(iommu);
+ wait_log_stopped(iommu, MMIO_STATUS_EVT_RUN_MASK, "event");
+
+ iommu_feature_disable(iommu, CONTROL_GAINT_EN);
+ iommu_feature_disable(iommu, CONTROL_GALOG_EN);
+ wait_log_stopped(iommu, MMIO_STATUS_GALOG_RUN_MASK, "GA");
+
+ iommu_feature_disable(iommu, CONTROL_PPRINT_EN);
+ iommu_feature_disable(iommu, CONTROL_PPRLOG_EN);
+ iommu_feature_disable(iommu, CONTROL_PPR_EN);
+ wait_log_stopped(iommu, MMIO_STATUS_PPR_RUN_MASK, "PPR");
+}
+
+static void amd_iommu_shutdown(void)
+{
+ struct amd_iommu *iommu;
+
+ for_each_iommu(iommu) {
+ if (iommu_preserved_state(&iommu->iommu)) {
+ amd_iommu_clear_unpreserved_dtes(iommu);
+ amd_iommu_quiesce_logs(iommu);
+ } else {
+ iommu_disable(iommu);
+ }
+ }
+
+#ifdef CONFIG_IRQ_REMAP
+ if (AMD_IOMMU_GUEST_IR_VAPIC(amd_iommu_guest_ir))
+ amd_iommu_irq_ops.capability &= ~(1 << IRQ_POSTING_CAP);
+#endif
+}
+
/*
* Suspend/Resume support
* disable suspend until real resume implemented
@@ -3500,7 +3586,7 @@ static int __init state_next(void)
break;
case IOMMU_ACPI_FINISHED:
early_enable_iommus();
- x86_platform.iommu_shutdown = disable_iommus;
+ x86_platform.iommu_shutdown = amd_iommu_shutdown;
init_state = IOMMU_ENABLED;
break;
case IOMMU_ENABLED:
diff --git a/drivers/iommu/amd/liveupdate.c b/drivers/iommu/amd/liveupdate.c
index 096a23bb4e7b..a3a9ebea5138 100644
--- a/drivers/iommu/amd/liveupdate.c
+++ b/drivers/iommu/amd/liveupdate.c
@@ -279,3 +279,86 @@ void amd_iommu_unpreserve_device(struct device *dev,

unpreserve_gcr3_level(gcr3_info->gcr3_tbl, gcr3_info->glx);
}
+
+/*
+ * Reset one non-preserved device's DTE to the blocked state during live-update
+ * shutdown. Every unpreserved device is reset so the next kernel starts from a
+ * clean slate for it and cannot translate through a domain whose page tables
+ * were not preserved.
+ */
+static int clear_unpreserved_dte(struct device *dev,
+ struct iommu_device *iommu_dev, void *arg)
+{
+ struct amd_iommu *iommu = container_of(iommu_dev, struct amd_iommu,
+ iommu);
+ struct dev_table_entry new = {};
+ struct iommu_dev_data *dev_data;
+
+ dev_data = dev_iommu_priv_get(dev);
+ if (!dev_data)
+ return 0;
+
+ if (dev_is_pci(dev) && dev_iommu_preserved_state(dev))
+ return 0;
+
+ amd_iommu_make_clear_dte(dev_data, &new);
+ amd_iommu_update_dte(iommu, dev_data, &new);
+
+ return 0;
+}
+
+static void clear_irq_dtes(struct amd_iommu *iommu)
+{
+ struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
+ struct dev_table_entry *dev_table = get_dev_table(iommu);
+ u32 devid;
+ u64 dte2;
+
+ if (!amd_iommu_irq_remap)
+ return;
+
+ /*
+ * Interrupt remapping tables are not preserved. Clear the interrupt
+ * fields on every DTE so none still points at a table the next kernel
+ * is free to recycle. DTE_DATA2_INTR_MASK is everything in data[2]
+ * except the guest page-table level, which preserved devices still
+ * need. The GCR3 pointer lives in data[0] and data[1], so it is not
+ * affected.
+ *
+ * This must run after the clear_unpreserved_dte() pass, because
+ * write_dte_upper128() deliberately copies DTE_DATA2_INTR_MASK back
+ * from the old entry. Clearing a DTE therefore keeps its interrupt
+ * fields, and only this walk removes them.
+ *
+ * The walk is by raw devid, so there is no iommu_dev_data to take
+ * dte_lock on, and looking one up per entry would make this O(n^2).
+ * Going without the lock is safe only because amd_iommu_shutdown()
+ * runs from native_machine_shutdown(), after the other CPUs are
+ * stopped and interrupts are off, so nothing can race these writes.
+ */
+ for (devid = 0; devid <= pci_seg->last_bdf; devid++) {
+ dte2 = READ_ONCE(dev_table[devid].data[2]);
+ if (!(dte2 & DTE_IRQ_REMAP_ENABLE))
+ continue;
+
+ WRITE_ONCE(dev_table[devid].data[2],
+ dte2 & ~DTE_DATA2_INTR_MASK);
+ }
+}
+
+/**
+ * amd_iommu_clear_unpreserved_dtes - Quiesce non-preserved devices at shutdown
+ * @iommu: The IOMMU whose device table is being cleaned up
+ */
+void amd_iommu_clear_unpreserved_dtes(struct amd_iommu *iommu)
+{
+ struct iommu_dev_iter iter = {
+ .fn = clear_unpreserved_dte,
+ .iommu = &iommu->iommu,
+ };
+
+ iommu_for_each_dev(&iter);
+ clear_irq_dtes(iommu);
+
+ amd_iommu_flush_all_caches(iommu);
+}
--
2.43.0