[PATCH v8 13/30] mm: handle PMD swap entries in fork path

From: Usama Arif

Date: Fri Oct 02 2026 - 06:06:56 EST


copy_huge_pmd() only knows about migration and device-private PMDs, so a
PMD swap entry would fall through to the present-PMD path and fork() would
duplicate it without taking a reference on the slots it points at.

Copy it the way copy_nonpresent_pte() copies a PTE swap entry: duplicate
the swap references, clear the exclusive marker on the source, put the
destination mm on mmlist, and account the child's slots to MM_SWAPENTS.

The GFP_ATOMIC extend-table allocation inside the dup can fail. Report that
as -EIO and let copy_pmd_range() retry with GFP_KERNEL, as
copy_nonpresent_pte() and copy_pte_range() already do for a PTE swap entry.
copy_huge_pmd() hands the entry back so the caller knows which range to
allocate for.

Only -ENOMEM is reported that way. The other failures mean the entry itself
is bad, and swap_retry_table_alloc_nr() returns 0 for those, so collapsing
them into -EIO as the PTE path does would spin in the caller's retry rather
than failing the fork.

While here, move the mm counter update into each entry-type arm, as the PTE
version does, so the swap arm can account MM_SWAPENTS instead of
MM_ANONPAGES.

Signed-off-by: Usama Arif <usama.arif@xxxxxxxxx>
---
include/linux/huge_mm.h | 3 +-
mm/huge_memory.c | 62 ++++++++++++++++++++++++++++++-----------
mm/memory.c | 12 +++++++-
3 files changed, 58 insertions(+), 19 deletions(-)

diff --git a/include/linux/huge_mm.h b/include/linux/huge_mm.h
index 8205e83f27771..7aa63d982af07 100644
--- a/include/linux/huge_mm.h
+++ b/include/linux/huge_mm.h
@@ -10,7 +10,8 @@
vm_fault_t do_huge_pmd_anonymous_page(struct vm_fault *vmf);
int copy_huge_pmd(struct mm_struct *dst_mm, struct mm_struct *src_mm,
pmd_t *dst_pmd, pmd_t *src_pmd, unsigned long addr,
- struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma);
+ struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma,
+ softleaf_t *entryp);
bool huge_pmd_set_accessed(struct vm_fault *vmf);
int copy_huge_pud(struct mm_struct *dst_mm, struct mm_struct *src_mm,
pud_t *dst_pud, pud_t *src_pud, unsigned long addr,
diff --git a/mm/huge_memory.c b/mm/huge_memory.c
index 24d116ae1fc30..80d18ca972ecf 100644
--- a/mm/huge_memory.c
+++ b/mm/huge_memory.c
@@ -1894,7 +1894,7 @@ bool touch_pmd(struct vm_area_struct *vma, unsigned long addr,
return false;
}

-static void copy_huge_non_present_pmd(
+static int copy_huge_non_present_pmd(
struct mm_struct *dst_mm, struct mm_struct *src_mm,
pmd_t *dst_pmd, pmd_t *src_pmd, unsigned long addr,
struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma,
@@ -1902,18 +1902,41 @@ static void copy_huge_non_present_pmd(
{
softleaf_t entry = softleaf_from_pmd(pmd);
struct folio *src_folio;
+ int err;

VM_WARN_ON_ONCE(!pmd_is_valid_softleaf(pmd));

- if (softleaf_is_migration_write(entry) ||
- softleaf_is_migration_read_exclusive(entry)) {
- entry = make_readable_migration_entry(swp_offset(entry));
- pmd = softleaf_to_pmd(entry);
- if (pmd_swp_soft_dirty(*src_pmd))
- pmd = pmd_swp_mksoft_dirty(pmd);
- if (pmd_swp_uffd(*src_pmd))
- pmd = pmd_swp_mkuffd(pmd);
- set_pmd_at(src_mm, addr, src_pmd, pmd);
+ if (softleaf_is_swap(entry)) {
+ /*
+ * A PMD swap entry only exists under CONFIG_THP_SWAP, where
+ * SWAPFILE_CLUSTER == HPAGE_PMD_NR, and it is cluster aligned,
+ * so these HPAGE_PMD_NR slots are exactly one cluster - which
+ * is what swap_dup_entries_direct() requires.
+ */
+ err = swap_dup_entries_direct(entry, HPAGE_PMD_NR);
+ if (err)
+ /* Only -ENOMEM is worth a GFP_KERNEL retry. */
+ return err == -ENOMEM ? -EIO : -ENOMEM;
+
+ mm_prepare_for_swap_entries(dst_mm);
+ /* Mark the swap entry as shared. */
+ if (pmd_swp_exclusive(pmd)) {
+ pmd = pmd_swp_clear_exclusive(pmd);
+ set_pmd_at(src_mm, addr, src_pmd, pmd);
+ }
+ add_mm_counter(dst_mm, MM_SWAPENTS, HPAGE_PMD_NR);
+ } else if (softleaf_is_migration(entry)) {
+ if (softleaf_is_migration_write(entry) ||
+ softleaf_is_migration_read_exclusive(entry)) {
+ entry = make_readable_migration_entry(swp_offset(entry));
+ pmd = softleaf_to_pmd(entry);
+ if (pmd_swp_soft_dirty(*src_pmd))
+ pmd = pmd_swp_mksoft_dirty(pmd);
+ if (pmd_swp_uffd(*src_pmd))
+ pmd = pmd_swp_mkuffd(pmd);
+ set_pmd_at(src_mm, addr, src_pmd, pmd);
+ }
+ add_mm_counter(dst_mm, MM_ANONPAGES, HPAGE_PMD_NR);
} else if (softleaf_is_device_private(entry)) {
/*
* For device private entries, since there are no
@@ -1940,19 +1963,21 @@ static void copy_huge_non_present_pmd(
*/
folio_try_dup_anon_rmap_pmd(src_folio, &src_folio->page,
dst_vma, src_vma);
+ add_mm_counter(dst_mm, MM_ANONPAGES, HPAGE_PMD_NR);
}

- add_mm_counter(dst_mm, MM_ANONPAGES, HPAGE_PMD_NR);
mm_inc_nr_ptes(dst_mm);
pgtable_trans_huge_deposit(dst_mm, dst_pmd, pgtable);
if (!userfaultfd_protected(dst_vma))
pmd = pmd_swp_clear_uffd(pmd);
set_pmd_at(dst_mm, addr, dst_pmd, pmd);
+ return 0;
}

int copy_huge_pmd(struct mm_struct *dst_mm, struct mm_struct *src_mm,
pmd_t *dst_pmd, pmd_t *src_pmd, unsigned long addr,
- struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma)
+ struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma,
+ softleaf_t *entryp)
{
spinlock_t *dst_ptl, *src_ptl;
struct page *src_page;
@@ -1995,11 +2020,14 @@ int copy_huge_pmd(struct mm_struct *dst_mm, struct mm_struct *src_mm,
ret = -EAGAIN;
pmd = *src_pmd;

- if (unlikely(thp_migration_supported() &&
- pmd_is_valid_softleaf(pmd))) {
- copy_huge_non_present_pmd(dst_mm, src_mm, dst_pmd, src_pmd, addr,
- dst_vma, src_vma, pmd, pgtable);
- ret = 0;
+ if (unlikely(pmd_is_valid_softleaf(pmd))) {
+ ret = copy_huge_non_present_pmd(dst_mm, src_mm, dst_pmd, src_pmd,
+ addr, dst_vma, src_vma, pmd,
+ pgtable);
+ if (ret) {
+ *entryp = softleaf_from_pmd(pmd);
+ pte_free(dst_mm, pgtable);
+ }
goto out_unlock;
}

diff --git a/mm/memory.c b/mm/memory.c
index 477d7e359b447..c0ad446d0cea4 100644
--- a/mm/memory.c
+++ b/mm/memory.c
@@ -1437,11 +1437,21 @@ copy_pmd_range(struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma,
do {
next = pmd_addr_end(addr, end);
if (pmd_is_huge(*src_pmd)) {
+ softleaf_t entry = softleaf_mk_none();
int err;

VM_BUG_ON_VMA(next-addr != HPAGE_PMD_SIZE, src_vma);
+again:
err = copy_huge_pmd(dst_mm, src_mm, dst_pmd, src_pmd,
- addr, dst_vma, src_vma);
+ addr, dst_vma, src_vma, &entry);
+ if (err == -EIO) {
+ VM_WARN_ON_ONCE(!entry.val);
+ if (swap_retry_table_alloc_nr(entry,
+ HPAGE_PMD_NR,
+ GFP_KERNEL) < 0)
+ return -ENOMEM;
+ goto again;
+ }
if (err == -ENOMEM)
return -ENOMEM;
if (!err)
--
2.53.0-Meta