From: "Kiryl Shutsemau (Meta)" <[email protected]>

Tests scattered through the collapse path decide what a collapse is
allowed to do by asking whether khugepaged started it.  Between them
they settle:

 - which VMAs are eligible, and how hard to try for a folio;
 - how many empty, swapped-out or shared PTEs a window may contain, and
   whether a sub-PMD window is held to a stricter rule than a PMD;
 - whether a range has to look used, and whether a MADV_FREE'd page is
   left alone;
 - whether the PMD is mapped as part of the request, and whether dirty
   pages are worth writing back and retrying.

None of those is a fact about khugepaged.  Each is something the caller
decided before asking, and the collapse code should not have to look up
who called to find out.

Add struct collapse_policy for the caller to fill: khugepaged from its
own settings, MADV_COLLAPSE from the fact that a user asked explicitly.
Every test becomes a read of a field.

khugepaged fills the policy once per scan pass, MADV_COLLAPSE once per
call.  That is the one change in behaviour: the tunables are sampled
once per pass rather than on every call, so a table scanned early in a
pass and one scanned late are judged alike.

cc->is_khugepaged stays, with a single reader left: the daemon's
collapse counter, which is bookkeeping and not policy.

collapse_file() also drops a NULL check on the collapse_control.  It has
one call site, reached only from collapse_single_pmd(), which
dereferences cc unconditionally, so the check was already dead.

Assisted-by: Claude-Code:claude-opus-5
Signed-off-by: Kiryl Shutsemau (Meta) <[email protected]>
---
 mm/collapse.h   |  47 ++++++++++++++++++++++
 mm/khugepaged.c | 105 +++++++++++++++++++++++++++---------------------
 2 files changed, 107 insertions(+), 45 deletions(-)

diff --git a/mm/collapse.h b/mm/collapse.h
index 9c82e71533df..44f52ea5bbb8 100644
--- a/mm/collapse.h
+++ b/mm/collapse.h
@@ -2,6 +2,7 @@
 #ifndef __MM_COLLAPSE_H
 #define __MM_COLLAPSE_H
 
+#include <linux/mm.h>
 #include <linux/nodemask.h>
 #include <linux/pgtable.h>
 #include <linux/types.h>
@@ -41,7 +42,53 @@ enum scan_result {
        SCAN_PAGE_DIRTY_OR_WRITEBACK,
 };
 
+/*
+ * What a collapse is allowed to do, decided by whoever asked for it, so the
+ * code doing it need not ask who its caller is: khugepaged fills this in from
+ * its own settings, MADV_COLLAPSE from the fact that a user asked explicitly.
+ */
+struct collapse_policy {
+       /* Limits, stated per PMD; HPAGE_PMD_NR means "no limit" */
+       unsigned int max_ptes_none;
+       unsigned int max_ptes_swap;
+       unsigned int max_ptes_shared;
+
+       /*
+        * Hold a sub-PMD window to a stricter rule than a PMD: no swapped-out
+        * and no shared PTEs at all, and max_ptes_none as
+        * collapse_max_ptes_none() scales it.  khugepaged holds mTHP collapse
+        * to that; an explicit request does not.
+        */
+       bool strict_sub_pmd;
+
+       /*
+        * Collapse only where it looks worth doing: require some sign the range
+        * is in use, and leave clean lazyfree folios for reclaim rather than
+        * collapsing them into a folio that is not lazyfree.  A user who asked
+        * for a collapse gets one either way.
+        */
+       bool skip_lazyfree;
+       bool require_referenced;
+
+       /*
+        * Finish the job rather than leaving it half done for a fault to pick
+        * up: map the PMD over a file collapse before returning, and write
+        * dirty pages back and retry once instead of refusing them.  Both cost
+        * latency the caller has asked to pay.
+        */
+       bool install_pmd;
+       bool writeback_dirty;
+
+       /* How hard to try for a destination folio */
+       gfp_t gfp;
+
+       /* Which VMAs are eligible, as thp_vma_allowable_orders() spells it */
+       enum tva_type tva_type;
+};
+
 struct collapse_control {
+       struct collapse_policy policy;
+
        bool is_khugepaged;
 
        /* Num pages scanned per node */
diff --git a/mm/khugepaged.c b/mm/khugepaged.c
index a12aafae8d9c..eebc044a930e 100644
--- a/mm/khugepaged.c
+++ b/mm/khugepaged.c
@@ -310,15 +310,12 @@ struct attribute_group khugepaged_attr_group = {
 static unsigned int collapse_max_ptes_none(struct collapse_control *cc,
                struct vm_area_struct *vma, unsigned int order)
 {
-       const unsigned int max_ptes_none = khugepaged_max_ptes_none;
+       const unsigned int max_ptes_none = cc->policy.max_ptes_none;
 
        if (vma && userfaultfd_armed(vma))
                return 0;
-       /* for MADV_COLLAPSE, allow any empty/shared zeropage PTEs */
-       if (!cc->is_khugepaged)
-               return HPAGE_PMD_NR;
-       /* for PMD collapse, respect the user defined maximum */
-       if (is_pmd_order(order))
+       /* The limit as given, at the PMD order and wherever it is not capped */
+       if (is_pmd_order(order) || !cc->policy.strict_sub_pmd)
                return max_ptes_none;
        /*
         * for mTHP collapse with the sysctl value set to 
KHUGEPAGED_MAX_PTES_LIMIT,
@@ -350,19 +347,12 @@ static unsigned int collapse_max_ptes_shared(struct 
collapse_control *cc,
                unsigned int order)
 {
        /*
-        * For MADV_COLLAPSE, do not restrict the number of PTEs that map shared
-        * anonymous pages.
+        * A sub-PMD window held to the strict rule takes no shared page at all:
+        * an mTHP is not worth the CoW-breaking.
         */
-       if (!cc->is_khugepaged)
-               return HPAGE_PMD_NR;
-       /*
-        * for mTHP collapse do not allow collapsing anonymous memory pages that
-        * are shared between processes.
-        */
-       if (!is_pmd_order(order))
+       if (!is_pmd_order(order) && cc->policy.strict_sub_pmd)
                return 0;
-       /* for PMD collapse, respect the user defined maximum */
-       return khugepaged_max_ptes_shared;
+       return cc->policy.max_ptes_shared;
 }
 
 /**
@@ -378,16 +368,12 @@ static unsigned int collapse_max_ptes_swap(struct 
collapse_control *cc,
                unsigned int order)
 {
        /*
-        * For MADV_COLLAPSE, do not restrict the number PTEs entries or
-        * pagecache entries that are non-present.
+        * A sub-PMD window held to the strict rule takes nothing non-present:
+        * reading pages back to build an mTHP is not worth the latency.
         */
-       if (!cc->is_khugepaged)
-               return HPAGE_PMD_NR;
-       /* for mTHP collapse do not allow any non-present PTEs or pagecache 
entries */
-       if (!is_pmd_order(order))
+       if (!is_pmd_order(order) && cc->policy.strict_sub_pmd)
                return 0;
-       /* for PMD collapse, respect the user defined maximum */
-       return khugepaged_max_ptes_swap;
+       return cc->policy.max_ptes_swap;
 }
 
 int hugepage_madvise(struct vm_area_struct *vma,
@@ -686,7 +672,7 @@ static enum scan_result __collapse_huge_page_isolate(struct 
vm_area_struct *vma,
                 * If the vma has the VM_DROPPABLE flag, the collapse will
                 * preserve the lazyfree property without needing to skip.
                 */
-               if (cc->is_khugepaged && !(vma->vm_flags & VM_DROPPABLE) &&
+               if (cc->policy.skip_lazyfree && !(vma->vm_flags & VM_DROPPABLE) 
&&
                    folio_test_lazyfree(folio) && !pte_dirty(pteval)) {
                        result = SCAN_PAGE_LAZYFREE;
                        goto out;
@@ -775,12 +761,12 @@ static enum scan_result 
__collapse_huge_page_isolate(struct vm_area_struct *vma,
                if (folio_test_large(folio))
                        list_add_tail(&folio->lru, compound_pagelist);
 next:
-               if (cc->is_khugepaged &&
+               if (cc->policy.require_referenced &&
                    folio_pte_referenced(folio, vma, addr, pteval))
                        referenced++;
        }
 
-       if (unlikely(cc->is_khugepaged && !referenced)) {
+       if (unlikely(cc->policy.require_referenced && !referenced)) {
                result = SCAN_LACK_REFERENCED_PAGE;
        } else {
                result = SCAN_SUCCEED;
@@ -984,6 +970,36 @@ static inline gfp_t alloc_hugepage_khugepaged_gfpmask(void)
        return khugepaged_defrag() ? GFP_TRANSHUGE : GFP_TRANSHUGE_LIGHT;
 }
 
+/* khugepaged collapses on its own initiative, so it obeys its own settings. */
+static void collapse_policy_khugepaged(struct collapse_policy *p)
+{
+       p->max_ptes_none = READ_ONCE(khugepaged_max_ptes_none);
+       p->max_ptes_swap = READ_ONCE(khugepaged_max_ptes_swap);
+       p->max_ptes_shared = READ_ONCE(khugepaged_max_ptes_shared);
+       p->strict_sub_pmd = true;
+       p->skip_lazyfree = true;
+       p->require_referenced = true;
+       p->install_pmd = false;
+       p->writeback_dirty = false;
+       p->gfp = alloc_hugepage_khugepaged_gfpmask();
+       p->tva_type = TVA_KHUGEPAGED;
+}
+
+/* MADV_COLLAPSE was asked for explicitly, so it is not held to those. */
+static void collapse_policy_forced(struct collapse_policy *p)
+{
+       p->max_ptes_none = HPAGE_PMD_NR;
+       p->max_ptes_swap = HPAGE_PMD_NR;
+       p->max_ptes_shared = HPAGE_PMD_NR;
+       p->strict_sub_pmd = false;
+       p->skip_lazyfree = false;
+       p->require_referenced = false;
+       p->install_pmd = true;
+       p->writeback_dirty = true;
+       p->gfp = GFP_TRANSHUGE;
+       p->tva_type = TVA_FORCED_COLLAPSE;
+}
+
 #ifdef CONFIG_NUMA
 static int collapse_find_target_node(struct collapse_control *cc)
 {
@@ -1021,8 +1037,7 @@ static enum scan_result hugepage_vma_revalidate(struct 
mm_struct *mm, unsigned l
                struct collapse_control *cc, unsigned int order)
 {
        struct vm_area_struct *vma;
-       enum tva_type type = cc->is_khugepaged ? TVA_KHUGEPAGED :
-                                TVA_FORCED_COLLAPSE;
+       enum tva_type type = cc->policy.tva_type;
 
        if (unlikely(collapse_test_exit_or_disable(mm)))
                return SCAN_ANY_PROCESS;
@@ -1205,8 +1220,7 @@ static enum scan_result 
__collapse_huge_page_swapin(struct mm_struct *mm,
 static enum scan_result alloc_charge_folio(struct folio **foliop, struct 
mm_struct *mm,
                struct collapse_control *cc, unsigned int order)
 {
-       gfp_t gfp = (cc->is_khugepaged ? alloc_hugepage_khugepaged_gfpmask() :
-                    GFP_TRANSHUGE);
+       gfp_t gfp = cc->policy.gfp;
        int node = collapse_find_target_node(cc);
        struct folio *folio;
 
@@ -1559,7 +1573,7 @@ static enum scan_result collapse_scan_pmd(struct 
mm_struct *mm,
        const unsigned int max_ptes_shared = collapse_max_ptes_shared(cc, 
HPAGE_PMD_ORDER);
        const unsigned int max_ptes_swap = collapse_max_ptes_swap(cc, 
HPAGE_PMD_ORDER);
        unsigned int max_ptes_none = collapse_max_ptes_none(cc, vma, 
HPAGE_PMD_ORDER);
-       enum tva_type tva_flags = cc->is_khugepaged ? TVA_KHUGEPAGED : 
TVA_FORCED_COLLAPSE;
+       enum tva_type tva_flags = cc->policy.tva_type;
        pmd_t *pmd;
        pte_t *pte, *_pte, pteval;
        int i;
@@ -1658,7 +1672,7 @@ static enum scan_result collapse_scan_pmd(struct 
mm_struct *mm,
                 * If the vma has the VM_DROPPABLE flag, the collapse will
                 * preserve the lazyfree property without needing to skip.
                 */
-               if (cc->is_khugepaged && !(vma->vm_flags & VM_DROPPABLE) &&
+               if (cc->policy.skip_lazyfree && !(vma->vm_flags & VM_DROPPABLE) 
&&
                    folio_test_lazyfree(folio) && !pte_dirty(pteval)) {
                        result = SCAN_PAGE_LAZYFREE;
                        goto out_unmap;
@@ -1716,11 +1730,11 @@ static enum scan_result collapse_scan_pmd(struct 
mm_struct *mm,
                        goto out_unmap;
                }
 
-               if (cc->is_khugepaged &&
+               if (cc->policy.require_referenced &&
                    folio_pte_referenced(folio, vma, addr, pteval))
                        referenced++;
        }
-       if (cc->is_khugepaged &&
+       if (cc->policy.require_referenced &&
                   (!referenced ||
                    (unmapped && referenced < HPAGE_PMD_NR / 2))) {
                result = SCAN_LACK_REFERENCED_PAGE;
@@ -2572,11 +2586,11 @@ static enum scan_result collapse_file(struct mm_struct 
*mm, unsigned long addr,
        xas_unlock_irq(&xas);
 
        /*
-        * Remove pte page tables, so we can re-fault the page as huge.
-        * If MADV_COLLAPSE, adjust result to call 
try_collapse_pte_mapped_thp().
+        * Remove pte page tables, so we can re-fault the page as huge.  A 
caller
+        * that wants the PMD mapped now is told to go and do that.
         */
        retract_page_tables(mapping, start);
-       if (cc && !cc->is_khugepaged)
+       if (cc->policy.install_pmd)
                result = SCAN_PTE_MAPPED_HUGEPAGE;
        folio_unlock(new_folio);
 
@@ -2760,11 +2774,8 @@ static enum scan_result collapse_single_pmd(unsigned 
long addr,
 retry:
        result = collapse_scan_file(mm, addr, file, pgoff, cc);
 
-       /*
-        * For MADV_COLLAPSE, when encountering dirty pages, try to writeback,
-        * then retry the collapse one time.
-        */
-       if (!cc->is_khugepaged && result == SCAN_PAGE_DIRTY_OR_WRITEBACK &&
+       /* Dirty pages are worth a writeback and one more try, if asked for */
+       if (cc->policy.writeback_dirty && result == 
SCAN_PAGE_DIRTY_OR_WRITEBACK &&
            !triggered_wb && mapping_can_writeback(file->f_mapping)) {
                const loff_t lstart = (loff_t)pgoff << PAGE_SHIFT;
                const loff_t lend = lstart + HPAGE_PMD_SIZE - 1;
@@ -2781,7 +2792,7 @@ static enum scan_result collapse_single_pmd(unsigned long 
addr,
                        result = SCAN_ANY_PROCESS;
                else
                        result = try_collapse_pte_mapped_thp(mm, addr,
-                                                            
!cc->is_khugepaged);
+                                                       cc->policy.install_pmd);
                if (result == SCAN_PMD_MAPPED)
                        result = SCAN_SUCCEED;
                mmap_read_unlock(mm);
@@ -2931,6 +2942,9 @@ static void khugepaged_do_scan(struct collapse_control 
*cc)
 
        lru_add_drain_all();
 
+       /* One policy for the whole pass, so every table is judged the same */
+       collapse_policy_khugepaged(&cc->policy);
+
        cc->progress = 0;
        while (true) {
                cond_resched();
@@ -3159,6 +3173,7 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned 
long start,
        if (!cc)
                return -ENOMEM;
        cc->is_khugepaged = false;
+       collapse_policy_forced(&cc->policy);
        cc->progress = 0;
 
        mmgrab(mm);
-- 
2.54.0


Reply via email to