[RFC PATCH 4/7] iommu/amd: preserve IOMMU and device state for live update

Ankit Soni Ankit.Soni at amd.com
Sun Oct 4 23:40:14 PDT 2026


Implement the .preserve and .preserve_device callbacks. Pin the PCI
segment's device table for KHO and record each device's domain ID,
page-table mode and GCR3 tree, so the next kernel can rebuild them.

The device table is shared by every IOMMU in a segment while the
callbacks run once per IOMMU, so the pin is reference counted.

Signed-off-by: Ankit Soni <Ankit.Soni at amd.com>
---
 drivers/iommu/amd/Makefile          |   1 +
 drivers/iommu/amd/amd_iommu.h       |  12 ++
 drivers/iommu/amd/amd_iommu_types.h |  14 ++
 drivers/iommu/amd/iommu.c           |   7 +
 drivers/iommu/amd/liveupdate.c      | 281 ++++++++++++++++++++++++++++
 5 files changed, 315 insertions(+)
 create mode 100644 drivers/iommu/amd/liveupdate.c

diff --git a/drivers/iommu/amd/Makefile b/drivers/iommu/amd/Makefile
index 94b8ef2acb18..227bbe920c26 100644
--- a/drivers/iommu/amd/Makefile
+++ b/drivers/iommu/amd/Makefile
@@ -2,3 +2,4 @@
 obj-y += iommu.o init.o quirks.o ppr.o pasid.o
 obj-$(CONFIG_AMD_IOMMU_IOMMUFD) += iommufd.o nested.o
 obj-$(CONFIG_AMD_IOMMU_DEBUGFS) += debugfs.o
+obj-$(CONFIG_IOMMU_LIVEUPDATE) += liveupdate.o
diff --git a/drivers/iommu/amd/amd_iommu.h b/drivers/iommu/amd/amd_iommu.h
index a2fe804b038b..5cf32e4898dc 100644
--- a/drivers/iommu/amd/amd_iommu.h
+++ b/drivers/iommu/amd/amd_iommu.h
@@ -225,4 +225,16 @@ amd_iommu_make_clear_dte(struct iommu_dev_data *dev_data, struct dev_table_entry
 struct iommu_domain *
 amd_iommu_alloc_domain_nested(struct iommufd_viommu *viommu, u32 flags,
 			      const struct iommu_user_data *user_data);
+
+#ifdef CONFIG_IOMMU_LIVEUPDATE
+/* LIVE UPDATE (drivers/iommu/amd/liveupdate.c) */
+int amd_iommu_preserve(struct iommu_device *iommu_dev,
+		       struct iommu_hw_ser *ser);
+void amd_iommu_unpreserve(struct iommu_device *iommu_dev,
+			  struct iommu_hw_ser *ser);
+int amd_iommu_preserve_device(struct device *dev,
+			      struct iommu_device_ser *device_ser);
+void amd_iommu_unpreserve_device(struct device *dev,
+				 struct iommu_device_ser *device_ser);
+#endif /* CONFIG_IOMMU_LIVEUPDATE */
 #endif /* AMD_IOMMU_H */
diff --git a/drivers/iommu/amd/amd_iommu_types.h b/drivers/iommu/amd/amd_iommu_types.h
index 3dbe20023456..fc9d98dfc6cd 100644
--- a/drivers/iommu/amd/amd_iommu_types.h
+++ b/drivers/iommu/amd/amd_iommu_types.h
@@ -596,6 +596,20 @@ struct amd_iommu_pci_seg {
 	 */
 	struct dev_table_entry *dev_table;
 
+#ifdef CONFIG_IOMMU_LIVEUPDATE
+	/*
+	 * The device table is shared by every IOMMU in this PCI segment, but
+	 * the live-update .preserve callback runs once per IOMMU, and each of
+	 * those IOMMUs is preserved and unpreserved independently by the core.
+	 * KHO preservation is not refcounted, so the references are counted
+	 * here and the shared pages are pinned on the first and unpinned only
+	 * on the last -- otherwise one IOMMU being unpreserved would drop the
+	 * pin from underneath the devices still preserved behind every other
+	 * IOMMU in the segment.
+	 */
+	unsigned int dev_table_preserve_count;
+#endif
+
 	/*
 	 * The rlookup iommu table is used to find the IOMMU which is
 	 * responsible for a specific device. It is indexed by the PCI
diff --git a/drivers/iommu/amd/iommu.c b/drivers/iommu/amd/iommu.c
index a83ce4521f7f..72df97b99589 100644
--- a/drivers/iommu/amd/iommu.c
+++ b/drivers/iommu/amd/iommu.c
@@ -32,6 +32,7 @@
 #include <linux/percpu.h>
 #include <linux/cc_platform.h>
 #include <linux/crash_dump.h>
+#include <linux/iommu-liveupdate.h>
 #include <asm/irq_remapping.h>
 #include <asm/io_apic.h>
 #include <asm/apic.h>
@@ -3221,6 +3222,12 @@ const struct iommu_ops amd_iommu_ops = {
 	.page_response = amd_iommu_page_response,
 	.get_viommu_size = amd_iommufd_get_viommu_size,
 	.viommu_init = amd_iommufd_viommu_init,
+#ifdef CONFIG_IOMMU_LIVEUPDATE
+	.preserve_device	= amd_iommu_preserve_device,
+	.unpreserve_device	= amd_iommu_unpreserve_device,
+	.preserve		= amd_iommu_preserve,
+	.unpreserve		= amd_iommu_unpreserve,
+#endif
 };
 
 #ifdef CONFIG_IRQ_REMAP
diff --git a/drivers/iommu/amd/liveupdate.c b/drivers/iommu/amd/liveupdate.c
new file mode 100644
index 000000000000..096a23bb4e7b
--- /dev/null
+++ b/drivers/iommu/amd/liveupdate.c
@@ -0,0 +1,281 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * AMD IOMMU (AMD-Vi) live update support.
+ *
+ * Copyright (C) 2026 Advanced Micro Devices, Inc.
+ * Author: Ankit Soni <Ankit.Soni at amd.com>
+ */
+
+#include <linux/iommu-liveupdate.h>
+#include <linux/pci.h>
+
+#include "amd_iommu.h"
+#include "../iommu-pages.h"
+
+/* Each 4K GCR3 table level holds 512 u64 entries. */
+#define GCR3_ENTRIES_PER_LEVEL	512
+
+/**
+ * amd_iommu_preserve - Preserve one AMD IOMMU instance for live update
+ * @iommu_dev: Core handle for the IOMMU whose state is being preserved
+ * @ser: Serialized IOMMU-instance record to fill in
+ *
+ * Pins the PCI-segment Device Table. Command, event, PPR and GA buffers are
+ * not preserved: they are drained or quiesced at shutdown and the next kernel
+ * allocates fresh ones, matching Intel phase 1.
+ *
+ * Return: 0 on success, negative errno on failure (any pages pinned by this
+ * call are released before returning).
+ */
+int amd_iommu_preserve(struct iommu_device *iommu_dev, struct iommu_hw_ser *ser)
+{
+	struct amd_iommu *iommu = container_of(iommu_dev, struct amd_iommu, iommu);
+	struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
+	int ret;
+
+	if (!pci_seg->dev_table_preserve_count) {
+		ret = iommu_preserve_pages(pci_seg->dev_table);
+		if (ret)
+			return ret;
+	}
+	pci_seg->dev_table_preserve_count++;
+
+	ser->type			= IOMMU_AMD;
+	ser->token			= iommu->mmio_phys;
+	ser->amd.mmio_phys		= iommu->mmio_phys;
+	ser->amd.dev_table_phys		= __pa(pci_seg->dev_table);
+	ser->amd.dev_table_size		= pci_seg->dev_table_size;
+	ser->amd.pci_seg_id		= pci_seg->id;
+	ser->amd.efr			= iommu->features;
+	ser->amd.efr2			= iommu->features2;
+
+	return 0;
+}
+
+/**
+ * amd_iommu_unpreserve - Release live-update state of one AMD IOMMU instance
+ * @iommu_dev: Core handle for the IOMMU whose state is being released
+ * @ser: Serialized IOMMU-instance record (unused; state is derived from @iommu)
+ *
+ * Reverse of amd_iommu_preserve(): once the last IOMMU of the segment has
+ * dropped its reference, unpins the shared Device Table.
+ */
+void amd_iommu_unpreserve(struct iommu_device *iommu_dev,
+			  struct iommu_hw_ser *ser)
+{
+	struct amd_iommu *iommu = container_of(iommu_dev, struct amd_iommu, iommu);
+	struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
+
+	if (WARN_ON(!pci_seg->dev_table_preserve_count))
+		return;
+
+	if (!--pci_seg->dev_table_preserve_count)
+		iommu_unpreserve_pages(pci_seg->dev_table);
+}
+
+/*
+ * Map the domain's page-table mode onto its handoff wire value. Done with an
+ * explicit switch rather than a cast so that renumbering
+ * enum protection_domain_mode cannot silently change the ABI.
+ */
+static u32 pd_mode_to_ser(enum protection_domain_mode pd_mode)
+{
+	switch (pd_mode) {
+	case PD_MODE_V1:
+		return IOMMU_AMD_SER_PD_MODE_V1;
+	case PD_MODE_V2:
+		return IOMMU_AMD_SER_PD_MODE_V2;
+	default:
+		return IOMMU_AMD_SER_PD_MODE_NONE;
+	}
+}
+
+static void unpreserve_gcr3_level(u64 *tbl, int level)
+{
+	int i;
+
+	if (level > 0) {
+		for (i = 0; i < GCR3_ENTRIES_PER_LEVEL; i++) {
+			u64 *child;
+
+			if (!(tbl[i] & GCR3_VALID))
+				continue;
+
+			child = iommu_phys_to_virt(tbl[i] & PAGE_MASK);
+			unpreserve_gcr3_level(child, level - 1);
+		}
+	}
+
+	iommu_unpreserve_pages(tbl);
+}
+
+static int preserve_gcr3_level(u64 *tbl, int level)
+{
+	u64 *child;
+	int i, ret;
+
+	ret = iommu_preserve_pages(tbl);
+	if (ret)
+		return ret;
+
+	if (level == 0)
+		return 0;
+
+	for (i = 0; i < GCR3_ENTRIES_PER_LEVEL; i++) {
+		if (!(tbl[i] & GCR3_VALID))
+			continue;
+
+		child = iommu_phys_to_virt(tbl[i] & PAGE_MASK);
+		ret = preserve_gcr3_level(child, level - 1);
+		if (ret)
+			goto err_unwind;
+	}
+
+	return 0;
+
+err_unwind:
+	while (--i >= 0) {
+		if (!(tbl[i] & GCR3_VALID))
+			continue;
+
+		child = iommu_phys_to_virt(tbl[i] & PAGE_MASK);
+		unpreserve_gcr3_level(child, level - 1);
+	}
+	iommu_unpreserve_pages(tbl);
+	return ret;
+}
+
+/*
+ * A nested attach programs the DTE from the guest's own descriptor instead of
+ * from dev_data: the DOMID is the viommu's host domain ID for that guest domain,
+ * the GCR3 pointer is the guest's, and dev_data->domain is never assigned, so it
+ * still refers to whatever was attached before -- the nest parent, or nothing at
+ * all (see set_dte_nested()). None of that is recoverable from the state
+ * amd_iommu_preserve_device() serializes, and the core cannot filter it out
+ * either: a device on a nested domain is preserved against the domain of its
+ * paging parent, which is a legitimately preserved domain, so every check up to
+ * this point passes (see find_hwpt_paging()).
+ *
+ * Guest translation that this kernel set up itself always has a host-allocated
+ * GCR3 table behind it, which is what tells the two apart.
+ */
+static bool dev_is_nested_attached(struct iommu_dev_data *dev_data)
+{
+	struct amd_iommu *iommu = get_amd_iommu_from_dev_data(dev_data);
+	struct dev_table_entry *dev_table = get_dev_table(iommu);
+
+	if (!dev_table)
+		return false;
+
+	return (READ_ONCE(dev_table[dev_data->devid].data[0]) & DTE_FLAG_GV) &&
+	       !dev_data->gcr3_info.gcr3_tbl;
+}
+
+/**
+ * amd_iommu_preserve_device - Preserve per-device live-update state
+ * @dev: The device being preserved
+ * @device_ser: Serialized per-device record to fill in
+ *
+ * The device's DTE rides across the kexec inside the preserved Device Table and
+ * keeps being used by the hardware, so everything the DTE still points at
+ * afterwards must be KHO-pinned. For a PASID-capable device that is the GCR3
+ * directory tree; the domain's page tables are pinned by the core when it
+ * preserves the domain. Anything not pinned must instead be dropped from the
+ * DTE, which is what amd_iommu_clear_unpreserved_dtes() does at shutdown.
+ *
+ * Records the domain ID the DTE tags this device with, and the domain's
+ * page-table mode so the next kernel can confirm its own default agrees before
+ * adopting the preserved tables. Both describe the DTE only as long as it was
+ * programmed from dev_data, so a device whose DTE came from a guest descriptor
+ * instead is refused outright; see dev_is_nested_attached().
+ *
+ * The walk runs during the quiesced live-update window, where the device is
+ * owned by its userspace driver and no PASID is being attached or detached, so
+ * the GCR3 tree is stable and no iommu-group lock is taken.
+ *
+ * Return: 0 on success, negative errno otherwise.
+ */
+int amd_iommu_preserve_device(struct device *dev,
+			      struct iommu_device_ser *device_ser)
+{
+	struct gcr3_tbl_info *gcr3_info;
+	struct iommu_dev_data *dev_data;
+	int ret;
+
+	if (!dev_is_pci(dev)) {
+		dev_err(dev, "cannot preserve non-PCI device\n");
+		return -EOPNOTSUPP;
+	}
+
+	dev_data = dev_iommu_priv_get(dev);
+	if (!dev_data)
+		return -EINVAL;
+
+	if (dev_is_nested_attached(dev_data)) {
+		dev_err(dev, "cannot preserve device attached to a nested domain\n");
+		return -EOPNOTSUPP;
+	}
+
+	if (!dev_data->domain)
+		return -EINVAL;
+
+	/* Page-table mode is a domain property, independent of PASID use. */
+	device_ser->amd.pd_mode = pd_mode_to_ser(dev_data->domain->pd_mode);
+
+	gcr3_info = &dev_data->gcr3_info;
+	if (!gcr3_info->gcr3_tbl) {
+		/*
+		 * Non-PASID device: the DTE's DOMID field holds the plain
+		 * protection-domain ID (see amd_iommu_set_dte_v1()), and there
+		 * is no GCR3 tree to pin.
+		 */
+		device_ser->domain_iommu_ser.attachment_id = dev_data->domain->id;
+		device_ser->amd.gcr3_tbl_phys = 0;
+		device_ser->amd.gcr3_glx = 0;
+		return 0;
+	}
+
+	/* PASID device: pin the whole GCR3 directory tree. */
+	ret = preserve_gcr3_level(gcr3_info->gcr3_tbl, gcr3_info->glx);
+	if (ret)
+		return ret;
+
+	/*
+	 * For a GCR3 device the DTE's DOMID field holds gcr3_info->domid, not
+	 * domain->id (see set_dte_gcr3_table()).
+	 */
+	device_ser->domain_iommu_ser.attachment_id = gcr3_info->domid;
+	device_ser->amd.gcr3_tbl_phys = __pa(gcr3_info->gcr3_tbl);
+	device_ser->amd.gcr3_glx = gcr3_info->glx;
+
+	return 0;
+}
+
+/**
+ * amd_iommu_unpreserve_device - Release per-device live-update state
+ * @dev: The device whose state is being released
+ * @device_ser: Serialized per-device record (unused)
+ *
+ * Reverse of amd_iommu_preserve_device(): unpins the GCR3 tree of a
+ * PASID-capable device. The DTE itself lives in the shared Device Table
+ * released by amd_iommu_unpreserve().
+ */
+void amd_iommu_unpreserve_device(struct device *dev,
+				 struct iommu_device_ser *device_ser)
+{
+	struct gcr3_tbl_info *gcr3_info;
+	struct iommu_dev_data *dev_data;
+
+	if (!dev_is_pci(dev))
+		return;
+
+	dev_data = dev_iommu_priv_get(dev);
+	if (!dev_data)
+		return;
+
+	gcr3_info = &dev_data->gcr3_info;
+	if (!gcr3_info->gcr3_tbl)
+		return;
+
+	unpreserve_gcr3_level(gcr3_info->gcr3_tbl, gcr3_info->glx);
+}
-- 
2.43.0




More information about the kexec mailing list