Re: [PATCH v5 00/12] mm: make userland page table freeing RCU-safe
From: Andrew Morton
Date: Fri Sep 25 2026 - 18:28:39 EST
On Fri, 25 Sep 2026 21:09:35 +0100 "Lorenzo Stoakes (ARM)" <ljs@xxxxxxxxxx> wrote:
> The majority of architectures in the kernel defer page table freeing until
> an RCU grace period has elapsed, this series converts all remaining
> architectures to do so too and eliminates CONFIG_MMU_GATHER_RCU_TABLE_FREE
> altogether.
>
> This is important because it enables safe lockless page table walking under
> RCU alone.
>
> Doing so allows for reduced lock contention, avoids lock ordering concerns
> and enables fast, efficient and correct page table walking as a result.
Thanks, I've updated mm-unstable to this version.
> v5:
> * Collected tags (thanks all! :)
> * Updated 1/12 to correctly set result to SCAN_ALLOC_HUGE_PAGE_FAIL should
> the page table allocation fail, as per Lance.
> * Updated 1/12 to rename alloc_deposit_pte() to alloc_deposit_pte_table()
> and dropped the powerpc hash MMU paragraph from the commit message, as
> per David.
> * Updated 9/12 to use a BH-disabling spinlock rather than an IRQ-safe one,
> as per Matthew.
> * Updated the subject line of 11/12 to clarify that the patch removes
> now-unneeded code, as per David.
> * Updated 12/12 documentation as per David, Lance.
Here's how v5 altered mm.git:
Documentation/mm/process_addrs.rst | 9 +++++++--
arch/m68k/mm/motorola.c | 16 +++++++---------
mm/khugepaged.c | 4 ++--
3 files changed, 16 insertions(+), 13 deletions(-)
--- a/arch/m68k/mm/motorola.c~b
+++ a/arch/m68k/mm/motorola.c
@@ -181,7 +181,7 @@ static void *add_pointer_table(struct mm
new = PD_PTABLE(pt_addr);
PD_MARKBITS(new) = ptable_mask(type) - 1;
- scoped_guard(spinlock_irqsave, &ptable_lock)
+ scoped_guard(spinlock_bh, &ptable_lock)
list_add(new, &ptable_list[type]);
return (pmd_t *)pt_addr;
@@ -191,16 +191,15 @@ void *get_pointer_table(struct mm_struct
{
unsigned int tmp, off;
unsigned long mask;
- unsigned long flags;
ptable_desc *dp;
void *ret;
- spin_lock_irqsave(&ptable_lock, flags);
+ spin_lock_bh(&ptable_lock);
dp = ptable_list[type].next;
mask = list_empty(&ptable_list[type]) ? 0 : PD_MARKBITS(dp);
if (mask == 0) {
- spin_unlock_irqrestore(&ptable_lock, flags);
+ spin_unlock_bh(&ptable_lock);
return add_pointer_table(mm, type);
}
@@ -213,7 +212,7 @@ void *get_pointer_table(struct mm_struct
}
ret = ptdesc_address(PD_PTDESC(dp)) + off;
- spin_unlock_irqrestore(&ptable_lock, flags);
+ spin_unlock_bh(&ptable_lock);
return ret;
}
@@ -223,9 +222,8 @@ int free_pointer_table(void *table, int
unsigned long ptable = (unsigned long)table;
unsigned long pt_addr = ptable & PAGE_MASK;
unsigned int mask = 1U << ((ptable - pt_addr)/ptable_size(type));
- unsigned long flags;
- spin_lock_irqsave(&ptable_lock, flags);
+ spin_lock_bh(&ptable_lock);
dp = PD_PTABLE(pt_addr);
if (PD_MARKBITS (dp) & mask)
@@ -236,7 +234,7 @@ int free_pointer_table(void *table, int
if (PD_MARKBITS(dp) == ptable_mask(type)) {
/* all tables in ptdesc are free, free ptdesc */
list_del(dp);
- spin_unlock_irqrestore(&ptable_lock, flags);
+ spin_unlock_bh(&ptable_lock);
mmu_page_dtor((void *)pt_addr);
pagetable_dtor_free(virt_to_ptdesc((void *)pt_addr));
@@ -249,7 +247,7 @@ int free_pointer_table(void *table, int
list_move(dp, &ptable_list[type]);
}
- spin_unlock_irqrestore(&ptable_lock, flags);
+ spin_unlock_bh(&ptable_lock);
return 0;
}
--- a/Documentation/mm/process_addrs.rst~b
+++ a/Documentation/mm/process_addrs.rst
@@ -541,8 +541,13 @@ We establish basic locking rules when in
after an RCU grace period has elapsed. However, any entry found must be
revalidated after the page table lock is taken (such as the
:c:func:`!pmd_same` recheck performed by :c:func:`!pte_offset_map_lock`)
- before it is acted upon. Changing an entry requires the page table
- lock and one of the locks that excludes teardown (mmap or VMA lock).
+ before it is acted upon. Changing an entry requires the page table lock
+ and one of the locks that excludes teardown (any one of the mmap, VMA or
+ rmap locks).
+* When traversing page tables under RCU alone it is important to take care
+ when operating upon leaf entries - if the value is operated upon (for
+ instance getting the folio associated with a PTE) an appropriate lock must
+ be taken to prevent concurrent modification.
* Reads from and writes to page table entries must be *appropriately*
atomic. See the section on atomicity below for details.
* Populating previously empty entries requires that the mmap or VMA locks are
--- a/mm/khugepaged.c~b
+++ a/mm/khugepaged.c
@@ -1278,7 +1278,7 @@ static enum scan_result alloc_charge_fol
return SCAN_SUCCEED;
}
-static pgtable_t alloc_deposit_pte(struct mm_struct *mm)
+static pgtable_t alloc_deposit_pte_table(struct mm_struct *mm)
{
/*
* khugepaged is run from a kernel thread, so need to manually set the
@@ -1328,7 +1328,7 @@ static enum scan_result collapse_huge_pa
}
if (is_pmd_order(order)) {
- pgtable = alloc_deposit_pte(mm);
+ pgtable = alloc_deposit_pte_table(mm);
if (!pgtable) {
result = SCAN_ALLOC_HUGE_PAGE_FAIL;
goto out_nolock;
_