[PATCH 07/10] mm/madvise: separate PTE-batch folio processing

From: Gregory Price

Date: Tue Sep 22 2026 - 19:59:46 EST


The PTE loop combines page-table iteration with validation and
processing of each folio batch. This hides the batching rule for large
folios and the ownership transferred when a partial folio must be split.

Move one batch into a PTL-locked helper. A split candidate is returned
locked and referenced so the caller can release the PTE mapping before
calling split_folio().

No functional change intended.

Assisted-by: LLM
Signed-off-by: Gregory Price (Meta) <gourry@xxxxxxxxxx>
---
mm/madvise.c | 132 +++++++++++++++++++++++++--------------------------
1 file changed, 65 insertions(+), 67 deletions(-)

diff --git a/mm/madvise.c b/mm/madvise.c
index 83b27258c9673..350b854ccc197 100644
--- a/mm/madvise.c
+++ b/mm/madvise.c
@@ -394,6 +394,54 @@ static bool madvise_lru_folio_is_filtered(struct folio *folio,
(pageout_anon_only && !folio_test_anon(folio));
}

+/* Return a split candidate locked and referenced for use after the PTL drop. */
+static struct folio *
+madvise_lru_pte_batch_locked(pte_t *pte, unsigned long addr,
+ unsigned long end, struct mm_walk *walk,
+ struct list_head *folio_list, bool pageout_anon_only, int *nr)
+{
+ const struct madvise_walk_private *private = walk->private;
+ struct vm_area_struct *vma = walk->vma;
+ struct folio *folio;
+ pte_t ptent;
+
+ *nr = 1;
+ ptent = ptep_get(pte);
+ if (pte_none(ptent) || !pte_present(ptent))
+ return NULL;
+
+ folio = vm_normal_folio(vma, addr, ptent);
+ if (!folio || folio_is_zone_device(folio))
+ return NULL;
+
+ /* Split PTE-mapped large folios before advising only part of them. */
+ if (folio_test_large(folio)) {
+ *nr = madvise_folio_pte_batch(addr, end, folio, pte, &ptent);
+ if (*nr < folio_nr_pages(folio)) {
+ if (madvise_lru_folio_is_filtered(folio, pageout_anon_only))
+ return NULL;
+ if (!folio_trylock(folio))
+ return NULL;
+ folio_get(folio);
+ return folio;
+ }
+ }
+
+ if (!folio_test_lru(folio) ||
+ folio_mapcount(folio) != folio_nr_pages(folio))
+ return NULL;
+ if (pageout_anon_only && !folio_test_anon(folio))
+ return NULL;
+
+ if (!private->pageout && pte_young(ptent)) {
+ clear_young_dirty_ptes(vma, addr, pte, *nr, CYDP_CLEAR_YOUNG);
+ tlb_remove_tlb_entries(private->tlb, pte, *nr, addr);
+ }
+
+ madvise_lru_folio(folio, private->pageout, folio_list);
+ return NULL;
+}
+
#ifdef CONFIG_TRANSPARENT_HUGEPAGE
static void madvise_cold_pmd(struct mmu_gather *tlb, struct vm_area_struct *vma,
pmd_t *pmd, unsigned long addr, pmd_t orig_pmd)
@@ -491,7 +539,7 @@ static int madvise_lru_pmd_entry(pmd_t *pmd, unsigned long addr,
bool pageout = private->pageout;
struct mm_struct *mm = tlb->mm;
struct vm_area_struct *vma = walk->vma;
- pte_t *start_pte, *pte, ptent;
+ pte_t *start_pte, *pte;
spinlock_t *ptl;
struct folio *folio = NULL;
LIST_HEAD(folio_list);
@@ -515,9 +563,6 @@ static int madvise_lru_pmd_entry(pmd_t *pmd, unsigned long addr,
flush_tlb_batched_pending(mm);
lazy_mmu_mode_enable();
for (; addr < end; pte += nr, addr += nr * PAGE_SIZE) {
- nr = 1;
- ptent = ptep_get(pte);
-
if (++batch_count == SWAP_CLUSTER_MAX) {
batch_count = 0;
if (need_resched()) {
@@ -528,71 +573,24 @@ static int madvise_lru_pmd_entry(pmd_t *pmd, unsigned long addr,
}
}

- if (pte_none(ptent))
- continue;
-
- if (!pte_present(ptent))
- continue;
-
- folio = vm_normal_folio(vma, addr, ptent);
- if (!folio || folio_is_zone_device(folio))
- continue;
-
- /*
- * If we encounter a large folio, only split it if it is not
- * fully mapped within the range we are operating on. Otherwise
- * leave it as is so that it can be swapped out whole. If we
- * fail to split a folio, leave it in place and advance to the
- * next pte in the range.
- */
- if (folio_test_large(folio)) {
- nr = madvise_folio_pte_batch(addr, end, folio, pte, &ptent);
- if (nr < folio_nr_pages(folio)) {
- int err;
-
- if (madvise_lru_folio_is_filtered(folio, pageout_anon_only))
- continue;
- if (!folio_trylock(folio))
- continue;
- folio_get(folio);
- lazy_mmu_mode_disable();
- pte_unmap_unlock(start_pte, ptl);
- start_pte = NULL;
- err = split_folio(folio);
- folio_unlock(folio);
- folio_put(folio);
- start_pte = pte =
- pte_offset_map_lock(mm, pmd, addr, &ptl);
- if (!start_pte)
- break;
- flush_tlb_batched_pending(mm);
- lazy_mmu_mode_enable();
- if (!err)
- nr = 0;
- continue;
- }
- }
-
- /*
- * Do not interfere with other mappings of this folio and
- * non-LRU folio. If we have a large folio at this point, we
- * know it is fully mapped so if its mapcount is the same as its
- * number of pages, it must be exclusive.
- */
- if (!folio_test_lru(folio) ||
- folio_mapcount(folio) != folio_nr_pages(folio))
- continue;
-
- if (pageout_anon_only && !folio_test_anon(folio))
+ folio = madvise_lru_pte_batch_locked(pte, addr, end, walk,
+ &folio_list, pageout_anon_only, &nr);
+ if (!folio)
continue;

- if (!pageout && pte_young(ptent)) {
- clear_young_dirty_ptes(vma, addr, pte, nr,
- CYDP_CLEAR_YOUNG);
- tlb_remove_tlb_entries(tlb, pte, nr, addr);
- }
-
- madvise_lru_folio(folio, pageout, &folio_list);
+ lazy_mmu_mode_disable();
+ pte_unmap_unlock(start_pte, ptl);
+ start_pte = NULL;
+ if (!split_folio(folio))
+ nr = 0;
+ folio_unlock(folio);
+ folio_put(folio);
+ start_pte = pte_offset_map_lock(mm, pmd, addr, &ptl);
+ if (!start_pte)
+ break;
+ pte = start_pte;
+ flush_tlb_batched_pending(mm);
+ lazy_mmu_mode_enable();
}

out:
--
2.53.0-Meta