[PATCH v5 00/12] mm: make userland page table freeing RCU-safe

Andrew Morton akpm at linux-foundation.org
Fri Sep 25 15:28:08 PDT 2026


On Fri, 25 Sep 2026 21:09:35 +0100 "Lorenzo Stoakes (ARM)" <ljs at kernel.org> wrote:

> The majority of architectures in the kernel defer page table freeing until
> an RCU grace period has elapsed, this series converts all remaining
> architectures to do so too and eliminates CONFIG_MMU_GATHER_RCU_TABLE_FREE
> altogether.
> 
> This is important because it enables safe lockless page table walking under
> RCU alone.
> 
> Doing so allows for reduced lock contention, avoids lock ordering concerns
> and enables fast, efficient and correct page table walking as a result.

Thanks, I've updated mm-unstable to this version.

> v5:
> * Collected tags (thanks all! :)
> * Updated 1/12 to correctly set result to SCAN_ALLOC_HUGE_PAGE_FAIL should
>   the page table allocation fail, as per Lance.
> * Updated 1/12 to rename alloc_deposit_pte() to alloc_deposit_pte_table()
>   and dropped the powerpc hash MMU paragraph from the commit message, as
>   per David.
> * Updated 9/12 to use a BH-disabling spinlock rather than an IRQ-safe one,
>   as per Matthew.
> * Updated the subject line of 11/12 to clarify that the patch removes
>   now-unneeded code, as per David.
> * Updated 12/12 documentation as per David, Lance.

Here's how v5 altered mm.git:


 Documentation/mm/process_addrs.rst |    9 +++++++--
 arch/m68k/mm/motorola.c            |   16 +++++++---------
 mm/khugepaged.c                    |    4 ++--
 3 files changed, 16 insertions(+), 13 deletions(-)

--- a/arch/m68k/mm/motorola.c~b
+++ a/arch/m68k/mm/motorola.c
@@ -181,7 +181,7 @@ static void *add_pointer_table(struct mm
 	new = PD_PTABLE(pt_addr);
 
 	PD_MARKBITS(new) = ptable_mask(type) - 1;
-	scoped_guard(spinlock_irqsave, &ptable_lock)
+	scoped_guard(spinlock_bh, &ptable_lock)
 		list_add(new, &ptable_list[type]);
 
 	return (pmd_t *)pt_addr;
@@ -191,16 +191,15 @@ void *get_pointer_table(struct mm_struct
 {
 	unsigned int tmp, off;
 	unsigned long mask;
-	unsigned long flags;
 	ptable_desc *dp;
 	void *ret;
 
-	spin_lock_irqsave(&ptable_lock, flags);
+	spin_lock_bh(&ptable_lock);
 	dp = ptable_list[type].next;
 	mask = list_empty(&ptable_list[type]) ? 0 : PD_MARKBITS(dp);
 
 	if (mask == 0) {
-		spin_unlock_irqrestore(&ptable_lock, flags);
+		spin_unlock_bh(&ptable_lock);
 		return add_pointer_table(mm, type);
 	}
 
@@ -213,7 +212,7 @@ void *get_pointer_table(struct mm_struct
 	}
 
 	ret = ptdesc_address(PD_PTDESC(dp)) + off;
-	spin_unlock_irqrestore(&ptable_lock, flags);
+	spin_unlock_bh(&ptable_lock);
 	return ret;
 }
 
@@ -223,9 +222,8 @@ int free_pointer_table(void *table, int
 	unsigned long ptable = (unsigned long)table;
 	unsigned long pt_addr = ptable & PAGE_MASK;
 	unsigned int mask = 1U << ((ptable - pt_addr)/ptable_size(type));
-	unsigned long flags;
 
-	spin_lock_irqsave(&ptable_lock, flags);
+	spin_lock_bh(&ptable_lock);
 
 	dp = PD_PTABLE(pt_addr);
 	if (PD_MARKBITS (dp) & mask)
@@ -236,7 +234,7 @@ int free_pointer_table(void *table, int
 	if (PD_MARKBITS(dp) == ptable_mask(type)) {
 		/* all tables in ptdesc are free, free ptdesc */
 		list_del(dp);
-		spin_unlock_irqrestore(&ptable_lock, flags);
+		spin_unlock_bh(&ptable_lock);
 
 		mmu_page_dtor((void *)pt_addr);
 		pagetable_dtor_free(virt_to_ptdesc((void *)pt_addr));
@@ -249,7 +247,7 @@ int free_pointer_table(void *table, int
 		list_move(dp, &ptable_list[type]);
 	}
 
-	spin_unlock_irqrestore(&ptable_lock, flags);
+	spin_unlock_bh(&ptable_lock);
 	return 0;
 }
 
--- a/Documentation/mm/process_addrs.rst~b
+++ a/Documentation/mm/process_addrs.rst
@@ -541,8 +541,13 @@ We establish basic locking rules when in
   after an RCU grace period has elapsed. However, any entry found must be
   revalidated after the page table lock is taken (such as the
   :c:func:`!pmd_same` recheck performed by :c:func:`!pte_offset_map_lock`)
-  before it is acted upon. Changing an entry requires the page table
-  lock and one of the locks that excludes teardown (mmap or VMA lock).
+  before it is acted upon. Changing an entry requires the page table lock
+  and one of the locks that excludes teardown (any one of the mmap, VMA or
+  rmap locks).
+* When traversing page tables under RCU alone it is important to take care
+  when operating upon leaf entries - if the value is operated upon (for
+  instance getting the folio associated with a PTE) an appropriate lock must
+  be taken to prevent concurrent modification.
 * Reads from and writes to page table entries must be *appropriately*
   atomic. See the section on atomicity below for details.
 * Populating previously empty entries requires that the mmap or VMA locks are
--- a/mm/khugepaged.c~b
+++ a/mm/khugepaged.c
@@ -1278,7 +1278,7 @@ static enum scan_result alloc_charge_fol
 	return SCAN_SUCCEED;
 }
 
-static pgtable_t alloc_deposit_pte(struct mm_struct *mm)
+static pgtable_t alloc_deposit_pte_table(struct mm_struct *mm)
 {
 	/*
 	 * khugepaged is run from a kernel thread, so need to manually set the
@@ -1328,7 +1328,7 @@ static enum scan_result collapse_huge_pa
 	}
 
 	if (is_pmd_order(order)) {
-		pgtable = alloc_deposit_pte(mm);
+		pgtable = alloc_deposit_pte_table(mm);
 		if (!pgtable) {
 			result = SCAN_ALLOC_HUGE_PAGE_FAIL;
 			goto out_nolock;
_




More information about the linux-um mailing list