The device DAX vmemmap population still reserves one extra tail vmemmap
page after the head page.

Drop that extra reservation and let the shared tail page cover all tail
vmemmap pages after the head page, so DAX follows the same reservation
model as HugeTLB.

This reduces the reserved vmemmap pages for optimized DAX mappings to
one and removes the now-unneeded first-tail population from the generic
and powerpc paths to simplify the code as well.

Signed-off-by: Muchun Song <[email protected]>
---
 arch/powerpc/mm/book3s64/radix_pgtable.c | 46 ++----------------------
 include/linux/mm.h                       |  3 +-
 mm/mm_init.c                             |  2 +-
 mm/sparse-vmemmap.c                      | 13 ++-----
 4 files changed, 7 insertions(+), 57 deletions(-)

diff --git a/arch/powerpc/mm/book3s64/radix_pgtable.c 
b/arch/powerpc/mm/book3s64/radix_pgtable.c
index 831c231a4a18..e7e751c48dd2 100644
--- a/arch/powerpc/mm/book3s64/radix_pgtable.c
+++ b/arch/powerpc/mm/book3s64/radix_pgtable.c
@@ -1218,39 +1218,6 @@ int __meminit radix__vmemmap_populate(unsigned long 
start, unsigned long end, in
        return 0;
 }
 
-static pte_t * __meminit radix__vmemmap_populate_address(unsigned long addr, 
int node,
-                                                        struct vmem_altmap 
*altmap,
-                                                        struct page *reuse)
-{
-       pgd_t *pgd;
-       p4d_t *p4d;
-       pud_t *pud;
-       pmd_t *pmd;
-       pte_t *pte;
-
-       pgd = pgd_offset_k(addr);
-       p4d = p4d_offset(pgd, addr);
-       pud = vmemmap_pud_alloc(p4d, node, addr);
-       if (!pud)
-               return NULL;
-       pmd = vmemmap_pmd_alloc(pud, node, addr);
-       if (!pmd)
-               return NULL;
-       if (pmd_leaf(*pmd))
-               /*
-                * The second page is mapped as a hugepage due to a nearby 
request.
-                * Force our mapping to page size without deduplication
-                */
-               return NULL;
-       pte = vmemmap_pte_alloc(pmd, node, addr);
-       if (!pte)
-               return NULL;
-       radix__vmemmap_pte_populate(pmd, addr, node, NULL, NULL);
-       vmemmap_verify(pte, node, addr, addr + PAGE_SIZE);
-
-       return pte;
-}
-
 int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn,
                                              unsigned long start,
                                              unsigned long end, int node,
@@ -1297,7 +1264,7 @@ int __meminit vmemmap_populate_compound_pages(unsigned 
long start_pfn,
                if (!pte_none(*pte)) {
                        /*
                         * This could be because we already have a compound
-                        * page whose VMEMMAP_RESERVE_NR pages were mapped and
+                        * page whose retained vmemmap page was mapped and
                         * this request fall in those pages.
                         */
                        next = addr + PAGE_SIZE;
@@ -1318,16 +1285,7 @@ int __meminit vmemmap_populate_compound_pages(unsigned 
long start_pfn,
                                        return -ENOMEM;
                                vmemmap_verify(pte, node, addr, addr + 
PAGE_SIZE);
 
-                               /*
-                                * Populate the tail pages vmemmap page
-                                * It can fall in different pmd, hence
-                                * vmemmap_populate_address()
-                                */
-                               pte = radix__vmemmap_populate_address(addr + 
PAGE_SIZE, node, NULL, NULL);
-                               if (!pte)
-                                       return -ENOMEM;
-
-                               next = addr + 2 * PAGE_SIZE;
+                               next = addr + PAGE_SIZE;
                                continue;
                        }
 
diff --git a/include/linux/mm.h b/include/linux/mm.h
index edadd7549b72..edc7b9ce9e79 100644
--- a/include/linux/mm.h
+++ b/include/linux/mm.h
@@ -5180,7 +5180,6 @@ static inline void vmem_altmap_free(struct vmem_altmap 
*altmap,
 }
 #endif
 
-#define VMEMMAP_RESERVE_NR     2
 #ifdef CONFIG_ARCH_WANT_OPTIMIZE_DAX_VMEMMAP
 static inline bool __vmemmap_can_optimize(struct vmem_altmap *altmap,
                                          struct dev_pagemap *pgmap)
@@ -5200,7 +5199,7 @@ static inline bool __vmemmap_can_optimize(struct 
vmem_altmap *altmap,
         * For vmemmap optimization with DAX we need minimum 2 vmemmap
         * pages. See layout diagram in Documentation/mm/vmemmap_dedup.rst
         */
-       return !altmap && (nr_vmemmap_pages > VMEMMAP_RESERVE_NR);
+       return !altmap && (nr_vmemmap_pages > VMEMMAP_OPTIMIZATION_PAGES);
 }
 /*
  * If we don't have an architecture override, use the generic rule
diff --git a/mm/mm_init.c b/mm/mm_init.c
index c4cd61978ce8..d520fd8de0df 100644
--- a/mm/mm_init.c
+++ b/mm/mm_init.c
@@ -1049,7 +1049,7 @@ static inline unsigned long compound_nr_pages(unsigned 
long pfn,
        if (!section_vmemmap_optimizable(ms))
                return pgmap_vmemmap_nr(pgmap);
 
-       return VMEMMAP_RESERVE_NR * (PAGE_SIZE / sizeof(struct page));
+       return VMEMMAP_OPTIMIZATION_PAGES * (PAGE_SIZE / sizeof(struct page));
 }
 
 static void __ref memmap_init_compound(struct page *head,
diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c
index 0201877a7f80..e655d9d1348f 100644
--- a/mm/sparse-vmemmap.c
+++ b/mm/sparse-vmemmap.c
@@ -136,7 +136,6 @@ int __meminit section_nr_vmemmap_pages(unsigned long pfn, 
unsigned long nr_pages
 {
        const struct mem_section *ms = __pfn_to_section(pfn);
        const int order = section_order(ms);
-       const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : 
VMEMMAP_OPTIMIZATION_PAGES;
        const unsigned long pages_per_compound = 1UL << order;
 
        VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION));
@@ -147,13 +146,13 @@ int __meminit section_nr_vmemmap_pages(unsigned long pfn, 
unsigned long nr_pages
 
        if (order < PFN_SECTION_SHIFT) {
                VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, 
pages_per_compound));
-               return vmemmap_pages * nr_pages / pages_per_compound;
+               return VMEMMAP_OPTIMIZATION_PAGES * nr_pages / 
pages_per_compound;
        }
 
        VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION));
 
        if (IS_ALIGNED(pfn, pages_per_compound))
-               return vmemmap_pages;
+               return VMEMMAP_OPTIMIZATION_PAGES;
 
        return 0;
 }
@@ -521,17 +520,11 @@ static int __meminit 
vmemmap_populate_compound_pages(unsigned long start_pfn,
                if (!pte)
                        return -ENOMEM;
 
-               /* Populate the tail pages vmemmap page */
-               next = addr + PAGE_SIZE;
-               pte = vmemmap_populate_address(next, node, NULL, -1, flags);
-               if (!pte)
-                       return -ENOMEM;
-
                /*
                 * Reuse the shared page for the rest of tail pages
                 * See layout diagram in Documentation/mm/vmemmap_dedup.rst
                 */
-               next += PAGE_SIZE;
+               next = addr + PAGE_SIZE;
                rc = vmemmap_populate_range(next, last, node, NULL,
                                            page_to_pfn(page), flags);
                if (rc)
-- 
2.54.0


Reply via email to