Re: [PATCH v4 12/18] iommu/vt-d: Handle reattach of the restored domain
From: Baolu Lu
Date: Thu Aug 27 2026 - 04:12:28 EST
On 8/8/26 10:27, Samiullah Khawaja wrote:
Reattach the restored domain to the preserved device using restored
domain ID. While reattaching do not setup the context and PASID entries
as those are preserved during liveupdate.
Signed-off-by: Samiullah Khawaja <skhawaja@xxxxxxxxxx>
---
drivers/iommu/intel/iommu.c | 9 ++-
drivers/iommu/intel/iommu.h | 10 +++
drivers/iommu/intel/liveupdate.c | 120 +++++++++++++++++++++++++++++++
3 files changed, 136 insertions(+), 3 deletions(-)
diff --git a/drivers/iommu/intel/iommu.c b/drivers/iommu/intel/iommu.c
index 42d3ff6db281..5370214629f4 100644
--- a/drivers/iommu/intel/iommu.c
+++ b/drivers/iommu/intel/iommu.c
@@ -864,7 +864,7 @@ static bool dev_needs_extra_dtlb_flush(struct pci_dev *pdev)
return true;
}
-static void iommu_enable_pci_ats(struct device_domain_info *info)
+void intel_iommu_enable_pci_ats(struct device_domain_info *info)
{
struct pci_dev *pdev;
@@ -1225,7 +1225,7 @@ domain_context_mapping(struct dmar_domain *domain, struct device *dev)
if (ret)
return ret;
- iommu_enable_pci_ats(info);
+ intel_iommu_enable_pci_ats(info);
return 0;
}
@@ -3164,6 +3164,9 @@ static int intel_iommu_attach_device(struct iommu_domain *domain,
{
int ret;
+ if (dev_iommu_restored_state(dev))
+ return intel_iommu_restore_device(domain, dev);
+
device_block_translation(dev);
ret = paging_domain_compatible(domain, dev);
@@ -3376,7 +3379,7 @@ static void intel_iommu_probe_finalize(struct device *dev)
info->pasid_enabled = 1;
if (sm_supported(iommu) && !dev_is_real_dma_subdevice(dev)) {
- iommu_enable_pci_ats(info);
+ intel_iommu_enable_pci_ats(info);
/* Assign a DEVTLB cache tag to the default domain. */
if (info->ats_enabled && info->domain) {
u16 did = domain_id_iommu(info->domain, iommu);
diff --git a/drivers/iommu/intel/iommu.h b/drivers/iommu/intel/iommu.h
index b33a12528066..1ef3b9309d44 100644
--- a/drivers/iommu/intel/iommu.h
+++ b/drivers/iommu/intel/iommu.h
@@ -1187,6 +1187,8 @@ void domain_detach_iommu(struct dmar_domain *domain, struct intel_iommu *iommu);
void device_block_translation(struct device *dev);
int paging_domain_compatible(struct iommu_domain *domain, struct device *dev);
+void intel_iommu_enable_pci_ats(struct device_domain_info *info);
+
struct dev_pasid_info *
domain_add_dev_pasid(struct iommu_domain *domain,
struct device *dev, ioasid_t pasid);
@@ -1309,6 +1311,8 @@ void intel_iommu_unpreserve(struct iommu_device *iommu,
void clear_unpreserved_context_entries(struct intel_iommu *iommu);
void intel_iommu_liveupdate_restore_root_table(struct intel_iommu *iommu,
struct iommu_hw_ser *iommu_ser);
+int intel_iommu_restore_device(struct iommu_domain *domain,
+ struct device *dev);
#else
static inline int intel_iommu_preserve_device(struct device *dev,
struct iommu_device_ser *device_ser)
@@ -1340,6 +1344,12 @@ static inline void intel_iommu_liveupdate_restore_root_table(struct intel_iommu
struct iommu_hw_ser *iommu_ser)
{
}
+
+static inline int intel_iommu_restore_device(struct iommu_domain *domain,
+ struct device *dev)
+{
+ return -EOPNOTSUPP;
+}
#endif
#ifdef CONFIG_INTEL_IOMMU_SVM
diff --git a/drivers/iommu/intel/liveupdate.c b/drivers/iommu/intel/liveupdate.c
index 480eab2d966b..05dea3893399 100644
--- a/drivers/iommu/intel/liveupdate.c
+++ b/drivers/iommu/intel/liveupdate.c
@@ -12,6 +12,7 @@
#include <linux/iommu-liveupdate.h>
#include <linux/module.h>
#include <linux/pci.h>
+#include <linux/pci-ats.h>
#include "iommu.h"
#include "../iommu-pages.h"
@@ -342,6 +343,125 @@ void intel_iommu_liveupdate_restore_root_table(struct intel_iommu *iommu,
BUG_ON(iommu_for_each_preserved_device(_restore_used_domain_ids, iommu));
}
+static void domain_detach_reattached_iommu(struct dmar_domain *domain,
+ struct intel_iommu *iommu)
+{
+ struct iommu_domain_info *info;
+
+ guard(mutex)(&iommu->did_lock);
+ info = xa_load(&domain->iommu_array, iommu->seq_id);
+ if (--info->refcnt == 0) {
+ xa_erase(&domain->iommu_array, iommu->seq_id);
+ kfree(info);
+ }
+}
+
+static int domain_reattach_iommu(struct dmar_domain *domain,
+ struct intel_iommu *iommu,
+ struct iommu_device_ser *device_ser)
+{
+ struct iommu_domain_info *info, *curr;
+ int restored_did;
+ int ret;
+
+ if (!iommu_domain_restored_state(&domain->domain))
+ return -EINVAL;
+
+ restored_did = device_ser->domain_iommu_ser.attachment_id;
+ if (!ida_exists(&iommu->domain_ida, restored_did))
+ return -EINVAL;
It seems that checking only whether the domain ID is reserved on this
IOMMU may not be sufficient. It would be safer to also verify that:
- device_ser->domain_iommu_ser.iommu_phys matches iommu->reg_phys, and
- device_ser->domain_iommu_ser.domain_phys matches @domain.
?
+
+ info = kzalloc_obj(*info);
+ if (!info)
+ return -ENOMEM;
+
+ guard(mutex)(&iommu->did_lock);
+ curr = xa_load(&domain->iommu_array, iommu->seq_id);
+ if (curr) {
+ curr->refcnt++;
+ kfree(info);
+ return 0;
+ }
+
+ info->refcnt = 1;
+ info->did = restored_did;
+ info->iommu = iommu;
+ curr = xa_cmpxchg(&domain->iommu_array, iommu->seq_id,
+ NULL, info, GFP_KERNEL);
+ if (curr) {
+ ret = xa_err(curr) ? : -EBUSY;
+ goto err_unlock;
+ }
+
+ return 0;
+
+err_unlock:
+ kfree(info);
+ return ret;
+}
+
+/**
+ * intel_iommu_restore_device() - Restore device domain attachment after live update
+ * @domain: Restored domain
+ * @dev: Restored device
+ *
+ * Return: 0 on success, or negative error code.
+ */
+int intel_iommu_restore_device(struct iommu_domain *domain,
+ struct device *dev)
+{
+ struct iommu_device_ser *device_ser = dev_iommu_restored_state(dev);
+ struct device_domain_info *info = dev_iommu_priv_get(dev);
+ struct dmar_domain *dmar_domain = to_dmar_domain(domain);
+ struct intel_iommu *iommu = info->iommu;
+ unsigned long flags;
+ int ret;
+
+ if (!device_ser)
+ return -EINVAL;
+
+ if (dev_is_real_dma_subdevice(dev))
+ return -EOPNOTSUPP;
... or, move above check here to ensure the attachment relationship
between the @device and @domain?
+
+ ret = domain_reattach_iommu(dmar_domain, iommu, device_ser);
+ if (ret)
+ return ret;
+
+ info->domain = dmar_domain;
+ info->domain_attached = true;
+ spin_lock_irqsave(&dmar_domain->lock, flags);
+ list_add(&info->link, &dmar_domain->devices);
+ spin_unlock_irqrestore(&dmar_domain->lock, flags);
+
+ if (!sm_supported(iommu))
+ intel_iommu_enable_pci_ats(info);
Could you please clarify the PCI ATS behavior across live update?
My understanding is that PCI devices may bypass reset during kexec and
are then re-initialized by the new kernel (is that correct?). If so, ATS
state in hardware would depend on its state before kexec in the old
kernel.
In a normal reboot, ATS is expected to be disabled and the device ATC is
empty. But that assumption may not hold for live update. If that is
true, can we still use the same approach to keep hardware ATS state and
info->ats_enabled in sync?
+
+ ret = cache_tag_assign_domain(dmar_domain, dev, IOMMU_NO_PASID);
+ if (ret)
+ goto err;
+
+ ret = iopf_for_domain_set(domain, dev);
+ if (ret)
+ goto err;
+
+ return 0;
+
+err:
+ /*
+ * Detach the restored domain from device and iommu on failure, but keep
+ * the hardware state intact.
+ */
+ info->domain_attached = false;
+ cache_tag_unassign_domain(info->domain, dev, IOMMU_NO_PASID);
+ spin_lock_irqsave(&info->domain->lock, flags);
+ list_del(&info->link);
+ spin_unlock_irqrestore(&info->domain->lock, flags);
+
+ domain_detach_reattached_iommu(info->domain, iommu);
+ info->domain = NULL;
+ return ret;
+}
+
/**
* intel_iommu_preserve_device() - Intel IOMMU callback to preserve device state
* @dev: Target device
Thanks,
baolu