[RFC PATCH 4/7] iommu/amd: preserve IOMMU and device state for live update
Ankit Soni
Ankit.Soni at amd.com
Sun Oct 4 23:40:14 PDT 2026
Implement the .preserve and .preserve_device callbacks. Pin the PCI
segment's device table for KHO and record each device's domain ID,
page-table mode and GCR3 tree, so the next kernel can rebuild them.
The device table is shared by every IOMMU in a segment while the
callbacks run once per IOMMU, so the pin is reference counted.
Signed-off-by: Ankit Soni <Ankit.Soni at amd.com>
---
drivers/iommu/amd/Makefile | 1 +
drivers/iommu/amd/amd_iommu.h | 12 ++
drivers/iommu/amd/amd_iommu_types.h | 14 ++
drivers/iommu/amd/iommu.c | 7 +
drivers/iommu/amd/liveupdate.c | 281 ++++++++++++++++++++++++++++
5 files changed, 315 insertions(+)
create mode 100644 drivers/iommu/amd/liveupdate.c
diff --git a/drivers/iommu/amd/Makefile b/drivers/iommu/amd/Makefile
index 94b8ef2acb18..227bbe920c26 100644
--- a/drivers/iommu/amd/Makefile
+++ b/drivers/iommu/amd/Makefile
@@ -2,3 +2,4 @@
obj-y += iommu.o init.o quirks.o ppr.o pasid.o
obj-$(CONFIG_AMD_IOMMU_IOMMUFD) += iommufd.o nested.o
obj-$(CONFIG_AMD_IOMMU_DEBUGFS) += debugfs.o
+obj-$(CONFIG_IOMMU_LIVEUPDATE) += liveupdate.o
diff --git a/drivers/iommu/amd/amd_iommu.h b/drivers/iommu/amd/amd_iommu.h
index a2fe804b038b..5cf32e4898dc 100644
--- a/drivers/iommu/amd/amd_iommu.h
+++ b/drivers/iommu/amd/amd_iommu.h
@@ -225,4 +225,16 @@ amd_iommu_make_clear_dte(struct iommu_dev_data *dev_data, struct dev_table_entry
struct iommu_domain *
amd_iommu_alloc_domain_nested(struct iommufd_viommu *viommu, u32 flags,
const struct iommu_user_data *user_data);
+
+#ifdef CONFIG_IOMMU_LIVEUPDATE
+/* LIVE UPDATE (drivers/iommu/amd/liveupdate.c) */
+int amd_iommu_preserve(struct iommu_device *iommu_dev,
+ struct iommu_hw_ser *ser);
+void amd_iommu_unpreserve(struct iommu_device *iommu_dev,
+ struct iommu_hw_ser *ser);
+int amd_iommu_preserve_device(struct device *dev,
+ struct iommu_device_ser *device_ser);
+void amd_iommu_unpreserve_device(struct device *dev,
+ struct iommu_device_ser *device_ser);
+#endif /* CONFIG_IOMMU_LIVEUPDATE */
#endif /* AMD_IOMMU_H */
diff --git a/drivers/iommu/amd/amd_iommu_types.h b/drivers/iommu/amd/amd_iommu_types.h
index 3dbe20023456..fc9d98dfc6cd 100644
--- a/drivers/iommu/amd/amd_iommu_types.h
+++ b/drivers/iommu/amd/amd_iommu_types.h
@@ -596,6 +596,20 @@ struct amd_iommu_pci_seg {
*/
struct dev_table_entry *dev_table;
+#ifdef CONFIG_IOMMU_LIVEUPDATE
+ /*
+ * The device table is shared by every IOMMU in this PCI segment, but
+ * the live-update .preserve callback runs once per IOMMU, and each of
+ * those IOMMUs is preserved and unpreserved independently by the core.
+ * KHO preservation is not refcounted, so the references are counted
+ * here and the shared pages are pinned on the first and unpinned only
+ * on the last -- otherwise one IOMMU being unpreserved would drop the
+ * pin from underneath the devices still preserved behind every other
+ * IOMMU in the segment.
+ */
+ unsigned int dev_table_preserve_count;
+#endif
+
/*
* The rlookup iommu table is used to find the IOMMU which is
* responsible for a specific device. It is indexed by the PCI
diff --git a/drivers/iommu/amd/iommu.c b/drivers/iommu/amd/iommu.c
index a83ce4521f7f..72df97b99589 100644
--- a/drivers/iommu/amd/iommu.c
+++ b/drivers/iommu/amd/iommu.c
@@ -32,6 +32,7 @@
#include <linux/percpu.h>
#include <linux/cc_platform.h>
#include <linux/crash_dump.h>
+#include <linux/iommu-liveupdate.h>
#include <asm/irq_remapping.h>
#include <asm/io_apic.h>
#include <asm/apic.h>
@@ -3221,6 +3222,12 @@ const struct iommu_ops amd_iommu_ops = {
.page_response = amd_iommu_page_response,
.get_viommu_size = amd_iommufd_get_viommu_size,
.viommu_init = amd_iommufd_viommu_init,
+#ifdef CONFIG_IOMMU_LIVEUPDATE
+ .preserve_device = amd_iommu_preserve_device,
+ .unpreserve_device = amd_iommu_unpreserve_device,
+ .preserve = amd_iommu_preserve,
+ .unpreserve = amd_iommu_unpreserve,
+#endif
};
#ifdef CONFIG_IRQ_REMAP
diff --git a/drivers/iommu/amd/liveupdate.c b/drivers/iommu/amd/liveupdate.c
new file mode 100644
index 000000000000..096a23bb4e7b
--- /dev/null
+++ b/drivers/iommu/amd/liveupdate.c
@@ -0,0 +1,281 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * AMD IOMMU (AMD-Vi) live update support.
+ *
+ * Copyright (C) 2026 Advanced Micro Devices, Inc.
+ * Author: Ankit Soni <Ankit.Soni at amd.com>
+ */
+
+#include <linux/iommu-liveupdate.h>
+#include <linux/pci.h>
+
+#include "amd_iommu.h"
+#include "../iommu-pages.h"
+
+/* Each 4K GCR3 table level holds 512 u64 entries. */
+#define GCR3_ENTRIES_PER_LEVEL 512
+
+/**
+ * amd_iommu_preserve - Preserve one AMD IOMMU instance for live update
+ * @iommu_dev: Core handle for the IOMMU whose state is being preserved
+ * @ser: Serialized IOMMU-instance record to fill in
+ *
+ * Pins the PCI-segment Device Table. Command, event, PPR and GA buffers are
+ * not preserved: they are drained or quiesced at shutdown and the next kernel
+ * allocates fresh ones, matching Intel phase 1.
+ *
+ * Return: 0 on success, negative errno on failure (any pages pinned by this
+ * call are released before returning).
+ */
+int amd_iommu_preserve(struct iommu_device *iommu_dev, struct iommu_hw_ser *ser)
+{
+ struct amd_iommu *iommu = container_of(iommu_dev, struct amd_iommu, iommu);
+ struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
+ int ret;
+
+ if (!pci_seg->dev_table_preserve_count) {
+ ret = iommu_preserve_pages(pci_seg->dev_table);
+ if (ret)
+ return ret;
+ }
+ pci_seg->dev_table_preserve_count++;
+
+ ser->type = IOMMU_AMD;
+ ser->token = iommu->mmio_phys;
+ ser->amd.mmio_phys = iommu->mmio_phys;
+ ser->amd.dev_table_phys = __pa(pci_seg->dev_table);
+ ser->amd.dev_table_size = pci_seg->dev_table_size;
+ ser->amd.pci_seg_id = pci_seg->id;
+ ser->amd.efr = iommu->features;
+ ser->amd.efr2 = iommu->features2;
+
+ return 0;
+}
+
+/**
+ * amd_iommu_unpreserve - Release live-update state of one AMD IOMMU instance
+ * @iommu_dev: Core handle for the IOMMU whose state is being released
+ * @ser: Serialized IOMMU-instance record (unused; state is derived from @iommu)
+ *
+ * Reverse of amd_iommu_preserve(): once the last IOMMU of the segment has
+ * dropped its reference, unpins the shared Device Table.
+ */
+void amd_iommu_unpreserve(struct iommu_device *iommu_dev,
+ struct iommu_hw_ser *ser)
+{
+ struct amd_iommu *iommu = container_of(iommu_dev, struct amd_iommu, iommu);
+ struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
+
+ if (WARN_ON(!pci_seg->dev_table_preserve_count))
+ return;
+
+ if (!--pci_seg->dev_table_preserve_count)
+ iommu_unpreserve_pages(pci_seg->dev_table);
+}
+
+/*
+ * Map the domain's page-table mode onto its handoff wire value. Done with an
+ * explicit switch rather than a cast so that renumbering
+ * enum protection_domain_mode cannot silently change the ABI.
+ */
+static u32 pd_mode_to_ser(enum protection_domain_mode pd_mode)
+{
+ switch (pd_mode) {
+ case PD_MODE_V1:
+ return IOMMU_AMD_SER_PD_MODE_V1;
+ case PD_MODE_V2:
+ return IOMMU_AMD_SER_PD_MODE_V2;
+ default:
+ return IOMMU_AMD_SER_PD_MODE_NONE;
+ }
+}
+
+static void unpreserve_gcr3_level(u64 *tbl, int level)
+{
+ int i;
+
+ if (level > 0) {
+ for (i = 0; i < GCR3_ENTRIES_PER_LEVEL; i++) {
+ u64 *child;
+
+ if (!(tbl[i] & GCR3_VALID))
+ continue;
+
+ child = iommu_phys_to_virt(tbl[i] & PAGE_MASK);
+ unpreserve_gcr3_level(child, level - 1);
+ }
+ }
+
+ iommu_unpreserve_pages(tbl);
+}
+
+static int preserve_gcr3_level(u64 *tbl, int level)
+{
+ u64 *child;
+ int i, ret;
+
+ ret = iommu_preserve_pages(tbl);
+ if (ret)
+ return ret;
+
+ if (level == 0)
+ return 0;
+
+ for (i = 0; i < GCR3_ENTRIES_PER_LEVEL; i++) {
+ if (!(tbl[i] & GCR3_VALID))
+ continue;
+
+ child = iommu_phys_to_virt(tbl[i] & PAGE_MASK);
+ ret = preserve_gcr3_level(child, level - 1);
+ if (ret)
+ goto err_unwind;
+ }
+
+ return 0;
+
+err_unwind:
+ while (--i >= 0) {
+ if (!(tbl[i] & GCR3_VALID))
+ continue;
+
+ child = iommu_phys_to_virt(tbl[i] & PAGE_MASK);
+ unpreserve_gcr3_level(child, level - 1);
+ }
+ iommu_unpreserve_pages(tbl);
+ return ret;
+}
+
+/*
+ * A nested attach programs the DTE from the guest's own descriptor instead of
+ * from dev_data: the DOMID is the viommu's host domain ID for that guest domain,
+ * the GCR3 pointer is the guest's, and dev_data->domain is never assigned, so it
+ * still refers to whatever was attached before -- the nest parent, or nothing at
+ * all (see set_dte_nested()). None of that is recoverable from the state
+ * amd_iommu_preserve_device() serializes, and the core cannot filter it out
+ * either: a device on a nested domain is preserved against the domain of its
+ * paging parent, which is a legitimately preserved domain, so every check up to
+ * this point passes (see find_hwpt_paging()).
+ *
+ * Guest translation that this kernel set up itself always has a host-allocated
+ * GCR3 table behind it, which is what tells the two apart.
+ */
+static bool dev_is_nested_attached(struct iommu_dev_data *dev_data)
+{
+ struct amd_iommu *iommu = get_amd_iommu_from_dev_data(dev_data);
+ struct dev_table_entry *dev_table = get_dev_table(iommu);
+
+ if (!dev_table)
+ return false;
+
+ return (READ_ONCE(dev_table[dev_data->devid].data[0]) & DTE_FLAG_GV) &&
+ !dev_data->gcr3_info.gcr3_tbl;
+}
+
+/**
+ * amd_iommu_preserve_device - Preserve per-device live-update state
+ * @dev: The device being preserved
+ * @device_ser: Serialized per-device record to fill in
+ *
+ * The device's DTE rides across the kexec inside the preserved Device Table and
+ * keeps being used by the hardware, so everything the DTE still points at
+ * afterwards must be KHO-pinned. For a PASID-capable device that is the GCR3
+ * directory tree; the domain's page tables are pinned by the core when it
+ * preserves the domain. Anything not pinned must instead be dropped from the
+ * DTE, which is what amd_iommu_clear_unpreserved_dtes() does at shutdown.
+ *
+ * Records the domain ID the DTE tags this device with, and the domain's
+ * page-table mode so the next kernel can confirm its own default agrees before
+ * adopting the preserved tables. Both describe the DTE only as long as it was
+ * programmed from dev_data, so a device whose DTE came from a guest descriptor
+ * instead is refused outright; see dev_is_nested_attached().
+ *
+ * The walk runs during the quiesced live-update window, where the device is
+ * owned by its userspace driver and no PASID is being attached or detached, so
+ * the GCR3 tree is stable and no iommu-group lock is taken.
+ *
+ * Return: 0 on success, negative errno otherwise.
+ */
+int amd_iommu_preserve_device(struct device *dev,
+ struct iommu_device_ser *device_ser)
+{
+ struct gcr3_tbl_info *gcr3_info;
+ struct iommu_dev_data *dev_data;
+ int ret;
+
+ if (!dev_is_pci(dev)) {
+ dev_err(dev, "cannot preserve non-PCI device\n");
+ return -EOPNOTSUPP;
+ }
+
+ dev_data = dev_iommu_priv_get(dev);
+ if (!dev_data)
+ return -EINVAL;
+
+ if (dev_is_nested_attached(dev_data)) {
+ dev_err(dev, "cannot preserve device attached to a nested domain\n");
+ return -EOPNOTSUPP;
+ }
+
+ if (!dev_data->domain)
+ return -EINVAL;
+
+ /* Page-table mode is a domain property, independent of PASID use. */
+ device_ser->amd.pd_mode = pd_mode_to_ser(dev_data->domain->pd_mode);
+
+ gcr3_info = &dev_data->gcr3_info;
+ if (!gcr3_info->gcr3_tbl) {
+ /*
+ * Non-PASID device: the DTE's DOMID field holds the plain
+ * protection-domain ID (see amd_iommu_set_dte_v1()), and there
+ * is no GCR3 tree to pin.
+ */
+ device_ser->domain_iommu_ser.attachment_id = dev_data->domain->id;
+ device_ser->amd.gcr3_tbl_phys = 0;
+ device_ser->amd.gcr3_glx = 0;
+ return 0;
+ }
+
+ /* PASID device: pin the whole GCR3 directory tree. */
+ ret = preserve_gcr3_level(gcr3_info->gcr3_tbl, gcr3_info->glx);
+ if (ret)
+ return ret;
+
+ /*
+ * For a GCR3 device the DTE's DOMID field holds gcr3_info->domid, not
+ * domain->id (see set_dte_gcr3_table()).
+ */
+ device_ser->domain_iommu_ser.attachment_id = gcr3_info->domid;
+ device_ser->amd.gcr3_tbl_phys = __pa(gcr3_info->gcr3_tbl);
+ device_ser->amd.gcr3_glx = gcr3_info->glx;
+
+ return 0;
+}
+
+/**
+ * amd_iommu_unpreserve_device - Release per-device live-update state
+ * @dev: The device whose state is being released
+ * @device_ser: Serialized per-device record (unused)
+ *
+ * Reverse of amd_iommu_preserve_device(): unpins the GCR3 tree of a
+ * PASID-capable device. The DTE itself lives in the shared Device Table
+ * released by amd_iommu_unpreserve().
+ */
+void amd_iommu_unpreserve_device(struct device *dev,
+ struct iommu_device_ser *device_ser)
+{
+ struct gcr3_tbl_info *gcr3_info;
+ struct iommu_dev_data *dev_data;
+
+ if (!dev_is_pci(dev))
+ return;
+
+ dev_data = dev_iommu_priv_get(dev);
+ if (!dev_data)
+ return;
+
+ gcr3_info = &dev_data->gcr3_info;
+ if (!gcr3_info->gcr3_tbl)
+ return;
+
+ unpreserve_gcr3_level(gcr3_info->gcr3_tbl, gcr3_info->glx);
+}
--
2.43.0
More information about the kexec
mailing list