Thread (119 messages) flat view 119 messages, 9 authors, 6d ago
COOLING6d

[RFC PATCH 12/57] mm/collapse: allocate a destination per candidate

From: Kiryl Shutsemau <hidden>
Date: 2026-08-16 22:46:43
Also in: bpf, linux-kselftest, linux-mm, lkml
Subsystem: memory management, memory management - thp (transparent huge page), the rest, tracing · Maintainers: Andrew Morton, David Hildenbrand, Linus Torvalds, Steven Rostedt, Masami Hiramatsu

From: "Kiryl Shutsemau (Meta)" <kas@kernel.org>

Fill in the allocation, which happens on both sides of the freeze.

A destination is a folio of the candidate's order, charged to the memcg,
with the memcg's deferred-split list entry taken up front while sleeping
is still allowed: the PMD-order install would otherwise need one under
the pmd lock.

collapse_alloc() does all of that for one candidate with the gfp it is
handed, and counts nothing when it fails: what a miss means is up to the
caller.

collapse_provision() is the caller inside the window.  The sources are
frozen by then and a faulter on any of them is waiting, so it asks
without __GFP_DIRECT_RECLAIM: reclaim entered there would be paid for by
that faulter.

A candidate the allocator cannot spare one for is declined rather than
failed.  It keeps its freeze and records SCAN_ALLOC_LIGHT_MISS, which
asks for the reclaiming gfp so a later round can allocate for it before
freezing anything.  Where the policy forbids reclaim there is nothing
better to retry with, so the miss is the verdict: the real result is
recorded and the failure counters fire.

collapse_reserve() honours those requests, before the round takes any
lock.  This is where reclaim belongs: nothing is held or frozen, so a
long compaction costs only khugepaged's own progress, which is why the
mechanism being replaced allocated here too.  Having asked the allocator
to try hard, a miss there is a failure.

Nothing sets cand->reclaim yet, so collapse_reserve() has nothing to do.
The request comes from the selection side, which queues a region refused
at one order for another attempt.

The page table a PMD-order candidate deposits cannot be deferred the same
way.  pte_alloc_one() allocates with GFP_PGTABLE_USER and takes no gfp to
strip, so it may reclaim and sleep whatever the order asked for.
collapse_deposit() secures it ahead of the freeze and refuses the
candidate when it cannot.  A PMD-order window is a whole table, so there
is at most one such candidate and it is the first.

Assisted-by: Claude-Code:claude-opus-5
Signed-off-by: Kiryl Shutsemau (Meta) <kas@kernel.org>
---
 include/trace/events/huge_memory.h |   3 +-
 mm/collapse.c                      | 126 +++++++++++++++++++++++++++++
 mm/collapse.h                      |   2 +
 mm/khugepaged.c                    |   4 +-
 4 files changed, 132 insertions(+), 3 deletions(-)
diff --git a/include/trace/events/huge_memory.h b/include/trace/events/huge_memory.h
index 778f5a56956c..68693eba82ef 100644
--- a/include/trace/events/huge_memory.h
+++ b/include/trace/events/huge_memory.h
@@ -40,7 +40,8 @@
 	EM( SCAN_STORE_FAILED,		"store_failed")			\
 	EM( SCAN_COPY_MC,		"copy_poisoned_page")		\
 	EM( SCAN_PAGE_FILLED,		"page_filled")			\
-	EMe(SCAN_PAGE_DIRTY_OR_WRITEBACK, "page_dirty_or_writeback")
+	EM( SCAN_PAGE_DIRTY_OR_WRITEBACK, "page_dirty_or_writeback")	\
+	EMe(SCAN_ALLOC_LIGHT_MISS,	"alloc_light_miss")
 
 #undef EM
 #undef EMe
diff --git a/mm/collapse.c b/mm/collapse.c
index 25c0f72a9a68..58c8d83f3468 100644
--- a/mm/collapse.c
+++ b/mm/collapse.c
@@ -17,6 +17,7 @@
 #include <linux/slab.h>
 #include <linux/swap.h>
 #include <linux/userfaultfd_k.h>
+#include <linux/vmstat.h>
 
 #include <asm/tlb.h>
 #include "collapse.h"
@@ -114,6 +115,12 @@
 	min(COLLAPSE_BATCH_BYTES >> (PAGE_SHIFT + COLLAPSE_MIN_MTHP_ORDER), \
 	    COLLAPSE_TABLE_WINDOWS)
 
+/* How far a candidate got, and so what a failure has to undo for it */
+enum collapse_candidate_state {
+	CAND_SELECTED,		/* collected; nothing held on its behalf yet */
+	CAND_SKIPPED,		/* refused; nothing of it left to undo */
+};
+
 /*
  * A candidate is an (addr, order) window selected for collapse.  Selection
  * counts in PTE offsets -- the bitmap it reads and the alignment it honours are
@@ -123,7 +130,12 @@
 struct collapse_candidate {
 	unsigned long addr;
 	unsigned int order;
+	/* The light allocation missed last round: this one may reclaim for it */
+	bool reclaim;
+	enum collapse_candidate_state state;
 	enum scan_result result;
+	struct folio *new_folio;
+	pgtable_t deposit;		/* PMD order: fresh table to deposit */
 };
 
 static unsigned long candidate_start(const struct collapse_candidate *cand)
@@ -218,6 +230,43 @@ static void collapse_freeze(struct vm_area_struct *vma,
 {
 }
 
+/*
+ * Allocate one candidate's destination with @gfp: a folio of its order, charged,
+ * with the memcg's deferred-split list heads in place so the install cannot need
+ * to allocate under the pmd lock.  Those heads cost only the first collapse in a
+ * memcg.
+ *
+ * A failure counts nothing and changes nothing: what a miss means is the caller's
+ * policy.
+ */
+static enum scan_result collapse_alloc(struct mm_struct *mm,
+				       struct collapse_control *cc,
+				       struct collapse_candidate *cand,
+				       gfp_t gfp)
+{
+	struct folio *folio;
+
+	folio = __folio_alloc(gfp, cand->order, collapse_find_target_node(cc),
+			      &cc->alloc_nmask);
+	if (!folio)
+		return SCAN_ALLOC_HUGE_PAGE_FAIL;
+
+	if (unlikely(mem_cgroup_charge(folio, mm, gfp)) ||
+	    folio_memcg_alloc_deferred(folio)) {
+		folio_put(folio);
+		return SCAN_CGROUP_CHARGE_FAIL;
+	}
+
+	if (is_pmd_order(cand->order)) {
+		count_vm_event(THP_COLLAPSE_ALLOC);
+		count_memcg_folio_events(folio, THP_COLLAPSE_ALLOC, 1);
+	}
+	count_mthp_stat(cand->order, MTHP_STAT_COLLAPSE_ALLOC);
+	cand->new_folio = folio;
+
+	return SCAN_SUCCEED;
+}
+
 /*
  * Allocate ahead of the freeze for the candidates whose light allocation missed
  * last round.  This is where reclaim belongs: nothing is held or frozen, so a
@@ -227,6 +276,31 @@ static void collapse_freeze(struct vm_area_struct *vma,
  */
 static void collapse_reserve(struct mm_struct *mm, struct collapse_control *cc)
 {
+	unsigned int i;
+
+	for (i = 0; i < cc->nr_candidates; i++) {
+		struct collapse_candidate *cand = &cc->candidates[i];
+		enum scan_result result;
+
+		if (!cand->reclaim)
+			continue;
+		cand->reclaim = false;
+
+		result = collapse_alloc(mm, cc, cand, cc->policy.gfp);
+		if (result == SCAN_SUCCEED)
+			continue;
+
+		if (result == SCAN_ALLOC_HUGE_PAGE_FAIL) {
+			/* Asked the allocator to try hard and it still missed */
+			if (is_pmd_order(cand->order))
+				count_vm_event(THP_COLLAPSE_ALLOC_FAILED);
+			count_mthp_stat(cand->order,
+					MTHP_STAT_COLLAPSE_ALLOC_FAILED);
+		}
+
+		cand->state = CAND_SKIPPED;
+		cand->result = result;
+	}
 }
 
 /*
@@ -235,9 +309,28 @@ static void collapse_reserve(struct mm_struct *mm, struct collapse_control *cc)
  * to strip: order-0 or not, it may reclaim and sleep, which is what the window
  * exists to keep out.  The destination folio has a light gfp to fall back on and
  * so can be deferred; this has none.
+ *
+ * A round is one table and a PMD-order window is the whole of it, so such a
+ * candidate cannot share a round: if there is one it is the only one, and it is
+ * candidates[0].  This secures one page table, never a batch of them.
  */
 static void collapse_deposit(struct mm_struct *mm, struct collapse_control *cc)
 {
+	struct collapse_candidate *cand = &cc->candidates[0];
+
+	if (!is_pmd_order(cand->order))
+		return;
+
+	VM_WARN_ON_ONCE(cc->nr_candidates != 1);
+
+	if (cand->state != CAND_SELECTED)
+		return;
+
+	cand->deposit = pte_alloc_one(mm);
+	if (!cand->deposit) {
+		cand->state = CAND_SKIPPED;
+		cand->result = SCAN_ALLOC_HUGE_PAGE_FAIL;
+	}
 }
 
 /*
@@ -253,6 +346,35 @@ static void collapse_deposit(struct mm_struct *mm, struct collapse_control *cc)
 static void collapse_provision(struct mm_struct *mm,
 			       struct collapse_control *cc)
 {
+	const gfp_t gfp = cc->policy.gfp & ~__GFP_DIRECT_RECLAIM;
+	const bool may_retry = gfp != cc->policy.gfp;
+	unsigned int i;
+
+	for (i = 0; i < cc->nr_candidates; i++) {
+		struct collapse_candidate *cand = &cc->candidates[i];
+		enum scan_result result;
+
+		if (cand->state != CAND_SELECTED || cand->new_folio)
+			continue;
+
+		result = collapse_alloc(mm, cc, cand, gfp);
+		if (result == SCAN_SUCCEED)
+			continue;
+
+		if (may_retry) {
+			/* A charge miss too: charging may reclaim when allowed */
+			cand->result = SCAN_ALLOC_LIGHT_MISS;
+		} else {
+			/* The gfp a retry would use, so this is the answer */
+			if (result == SCAN_ALLOC_HUGE_PAGE_FAIL) {
+				if (is_pmd_order(cand->order))
+					count_vm_event(THP_COLLAPSE_ALLOC_FAILED);
+				count_mthp_stat(cand->order,
+						MTHP_STAT_COLLAPSE_ALLOC_FAILED);
+			}
+			cand->result = result;
+		}
+	}
 }
 
 /*
@@ -761,7 +883,11 @@ static void collapse_add_candidate(struct collapse_control *cc,
 	cc->nr_candidates++;
 	cand->addr = addr;
 	cand->order = order;
+	cand->reclaim = false;
+	cand->state = CAND_SELECTED;
 	cand->result = SCAN_FAIL;
+	cand->new_folio = NULL;
+	cand->deposit = NULL;
 }
 
 /*
diff --git a/mm/collapse.h b/mm/collapse.h
index c61db86dc6c2..feb2e0d57339 100644
--- a/mm/collapse.h
+++ b/mm/collapse.h
@@ -46,6 +46,7 @@ enum scan_result {
 	SCAN_COPY_MC,
 	SCAN_PAGE_FILLED,
 	SCAN_PAGE_DIRTY_OR_WRITEBACK,
+	SCAN_ALLOC_LIGHT_MISS,
 };
 
 /*
@@ -148,6 +149,7 @@ unsigned long collapse_possible_orders(struct vm_area_struct *vma,
 		vm_flags_t vm_flags, enum tva_type tva_flags);
 enum scan_result find_pmd_or_thp_or_none(struct mm_struct *mm,
 		unsigned long address, pmd_t **pmd);
+int collapse_find_target_node(struct collapse_control *cc);
 bool collapse_scan_abort(int nid, struct collapse_control *cc);
 unsigned int collapse_max_ptes_none(struct collapse_control *cc,
 		struct vm_area_struct *vma, unsigned int order);
diff --git a/mm/khugepaged.c b/mm/khugepaged.c
index 43f6107c953a..50b520961b9b 100644
--- a/mm/khugepaged.c
+++ b/mm/khugepaged.c
@@ -999,7 +999,7 @@ static void collapse_policy_forced(struct collapse_policy *p)
 }
 
 #ifdef CONFIG_NUMA
-static int collapse_find_target_node(struct collapse_control *cc)
+int collapse_find_target_node(struct collapse_control *cc)
 {
 	int nid, target_node = 0, max_value = 0;
 
@@ -1018,7 +1018,7 @@ static int collapse_find_target_node(struct collapse_control *cc)
 	return target_node;
 }
 #else
-static int collapse_find_target_node(struct collapse_control *cc)
+int collapse_find_target_node(struct collapse_control *cc)
 {
 	return 0;
 }
-- 
2.54.0
Keyboard shortcuts
hback out one level
jnext message in thread
kprevious message in thread
ldrill in
Escclose help / fold thread tree
?toggle this help