[PATCH v4 07/14] mm, swap: free backing pages in xswap_unmap_clusters

From: Baoquan He

Date: Fri Oct 02 2026 - 20:35:20 EST


vm_area_unmap_pages() does not free the backing pages that
xswap_map_clusters() allocated, so they leaked on every unmap.

Collect the backing pages from the PTEs before unmapping and free them
after the PTEs are cleared. The collection array is allocated with
kvmalloc_array(GFP_KERNEL), which falls back to vmalloc() and may
reclaim; if it still fails, unmap in bounded batches with a stack array
instead. The unmap therefore cannot fail and the callers do not need a
retry loop.

Signed-off-by: Baoquan He <hebaoquan@xxxxxxxxxx>
---
mm/swapfile.c | 95 ++++++++++++++++++++++++++++++++++++++++++++++-----
1 file changed, 86 insertions(+), 9 deletions(-)

diff --git a/mm/swapfile.c b/mm/swapfile.c
index b8572f301d31..7c062e772f5b 100644
--- a/mm/swapfile.c
+++ b/mm/swapfile.c
@@ -84,6 +84,14 @@

static bool xswap_should_grow(struct swap_info_struct *si);
static unsigned long xswap_grow(struct swap_info_struct *si);
+
+/* Batch size for the fallback unmap; keeps the stack array small. */
+#define XSWAP_UNMAP_BATCH_PAGES 16
+#define XSWAP_UNMAP_BATCH_CLUSTERS \
+ max_t(unsigned long, \
+ ((XSWAP_UNMAP_BATCH_PAGES * PAGE_SIZE) / \
+ sizeof(struct swap_cluster_info)), 16)
+
static int xswap_map_clusters(struct swap_info_struct *si,
unsigned long start_idx, unsigned long nr);
static void xswap_unmap_clusters(struct swap_info_struct *si,
@@ -3341,9 +3349,8 @@ static void free_swap_cluster_info(struct swap_info_struct *si)
}
spin_unlock(&ci->lock);
}
- /* Unmap all mapped clusters and free the VM_SPARSE area */
- if (si->nr_clusters_mapped > 0)
- xswap_unmap_clusters(si, 0, si->nr_clusters_mapped);
+ /* free_vm_area() drops the mapping without freeing the pages. */
+ xswap_unmap_clusters(si, 0, si->nr_clusters_mapped);
free_vm_area(si->cluster_vm);
si->cluster_vm = NULL;
si->cluster_info = NULL;
@@ -4026,21 +4033,65 @@ static unsigned long xswap_grow(struct swap_info_struct *si)
return nr_new;
}

+struct xswap_page_data {
+ struct page **pages;
+ unsigned long nr;
+ unsigned long max;
+};
+
+static int xswap_collect_page(pte_t *pte, unsigned long addr, void *data)
+{
+ struct xswap_page_data *xpd = data;
+ pte_t pteval = ptep_get(pte);
+
+ if (!pte_present(pteval))
+ return 0;
+ if (WARN_ON_ONCE(xpd->nr >= xpd->max))
+ return 0;
+ xpd->pages[xpd->nr++] = pte_page(pteval);
+ return 0;
+}
+
+/*
+ * Free the backing pages mapped in [vm_start, vm_end) and clear the PTEs.
+ * @pages must have room for @max pointers.
+ */
+static void xswap_unmap_range(struct swap_info_struct *si,
+ unsigned long vm_start, unsigned long vm_end,
+ struct page **pages, unsigned long max)
+{
+ struct xswap_page_data xpd = {
+ .pages = pages,
+ .max = max,
+ };
+ unsigned long i;
+
+ apply_to_existing_page_range(&init_mm, vm_start, vm_end - vm_start,
+ xswap_collect_page, &xpd);
+ vm_area_unmap_pages(si->cluster_vm, vm_start, vm_end);
+
+ for (i = 0; i < xpd.nr; i++)
+ __free_page(pages[i]);
+}
+
static void xswap_unmap_clusters(struct swap_info_struct *si,
unsigned long start_idx, unsigned long nr)
{
unsigned long start_addr = (unsigned long)si->cluster_info +
(size_t)start_idx * sizeof(struct swap_cluster_info);
- unsigned long end_addr = start_addr + (size_t)nr * sizeof(struct swap_cluster_info);
+ unsigned long end_addr = start_addr +
+ (size_t)nr * sizeof(struct swap_cluster_info);
unsigned long vm_start = PAGE_ALIGN(start_addr);
unsigned long vm_end = PAGE_ALIGN(end_addr);
+ struct page **pages;
+ unsigned int noreclaim_flags;
+ unsigned long npages, idx;

mutex_lock(&si->xswap_lock);

if (vm_start >= vm_end) {
WRITE_ONCE(si->nr_clusters_mapped, start_idx);
- mutex_unlock(&si->xswap_lock);
- return;
+ goto out_unlock;
}

/*
@@ -4052,11 +4103,37 @@ static void xswap_unmap_clusters(struct swap_info_struct *si,
flush_percpu_swap_cluster(si);
synchronize_rcu();

- vm_area_unmap_pages(si->cluster_vm, vm_start, vm_end);
- /* vm_area_unmap_pages() clears PTEs but does not free pages. */
- /* TODO: free backing pages via page table walk or tracking bitmap */
+ npages = (vm_end - vm_start) >> PAGE_SHIFT;
+
+ /* Reclaim here would re-enter xswap_map_clusters() and its lock. */
+ noreclaim_flags = memalloc_noreclaim_save();
+ pages = kvmalloc_array(npages, sizeof(*pages), GFP_KERNEL);
+ memalloc_noreclaim_restore(noreclaim_flags);
+ if (pages) {
+ xswap_unmap_range(si, vm_start, vm_end, pages, npages);
+ kvfree(pages);
+ WRITE_ONCE(si->nr_clusters_mapped, start_idx);
+ goto out_unlock;
+ }
+
+ /* No memory for the array: unmap in bounded batches instead. */
+ for (idx = start_idx; idx < start_idx + nr; idx += XSWAP_UNMAP_BATCH_CLUSTERS) {
+ unsigned long batch = min(start_idx + nr - idx,
+ XSWAP_UNMAP_BATCH_CLUSTERS);
+ unsigned long addr = (unsigned long)si->cluster_info +
+ (size_t)idx * sizeof(struct swap_cluster_info);
+ unsigned long bstart = PAGE_ALIGN(addr);
+ unsigned long bend = PAGE_ALIGN(addr + (size_t)batch *
+ sizeof(struct swap_cluster_info));
+ struct page *batch_pages[XSWAP_UNMAP_BATCH_PAGES + 1];
+
+ if (bstart < bend)
+ xswap_unmap_range(si, bstart, bend, batch_pages,
+ ARRAY_SIZE(batch_pages));
+ }

WRITE_ONCE(si->nr_clusters_mapped, start_idx);
+out_unlock:
mutex_unlock(&si->xswap_lock);
}

--
2.54.0