From: Mika Penttilä <[email protected]>

Implement the needed hmm_vma_handle_migrate_prepare() function
which is mostly carried over from migrate_device.c's
migrate_vma_collect_pmd() function.

Also implement the migrate_vma_split_folio(), for splitting
pte mapped large folios. It is also mostly from migrate_device.c,
with care taken to reference folio before relasing page table
lock.

With HMM pagewalk based migration, the idea is that
hmm_vma_handle_*() are responsible for faulting,
and the pfn collecting part. hmm_vma_handle_migrate_prepare*()
do the migration decisions (with HMM_PFN_MIGRATE), possibly split
folios, and insert migration ptes/pmds.

HMM pagewalk based migration is enabled in later commit, for now
now hmm_select_migrate() returns 0.

Cc: David Hildenbrand <[email protected]>
Cc: Jason Gunthorpe <[email protected]>
Cc: Leon Romanovsky <[email protected]>
Cc: Alistair Popple <[email protected]>
Cc: Balbir Singh <[email protected]>
Cc: Zi Yan <[email protected]>
Cc: Matthew Brost <[email protected]>
Suggested-by: Alistair Popple <[email protected]>
Signed-off-by: Mika Penttilä <[email protected]>
---
 mm/hmm.c | 262 ++++++++++++++++++++++++++++++++++++++++++++++++++++++-
 1 file changed, 261 insertions(+), 1 deletion(-)

diff --git a/mm/hmm.c b/mm/hmm.c
index 3621d1df51d3..2d902b8bcd58 100644
--- a/mm/hmm.c
+++ b/mm/hmm.c
@@ -486,6 +486,63 @@ static int hmm_vma_handle_absent_pmd(struct mm_walk *walk, 
unsigned long start,
 #endif  /* CONFIG_ARCH_ENABLE_THP_MIGRATION */
 
 #ifdef CONFIG_DEVICE_MIGRATION
+/**
+ * migrate_vma_split_folio() - Helper function to split a THP folio
+ * @folio: the folio to split
+ * @fault_page: struct page associated with the fault if any
+ * @hmm_vma_walk: walk in progress
+ * @ptep: pte_t * for unmap and unlock ptl
+ *
+ * Returns 0 on success
+ */
+static int migrate_vma_split_folio(struct folio *folio,
+                                  struct page *fault_page,
+                                  struct hmm_vma_walk *hmm_vma_walk,
+                                  pte_t *ptep)
+{
+       int ret;
+       struct folio *fault_folio = fault_page ? page_folio(fault_page) : NULL;
+       struct folio *new_fault_folio = NULL;
+
+       if (folio != fault_folio)
+               folio_get(folio);
+
+       pte_unmap_unlock(ptep, hmm_vma_walk->ptl);
+       hmm_vma_walk->ptelocked = false;
+
+       if (folio != fault_folio)
+               folio_lock(folio);
+
+       ret = split_folio(folio);
+       if (ret) {
+               if (folio != fault_folio) {
+                       folio_unlock(folio);
+                       folio_put(folio);
+               }
+               return ret;
+       }
+
+       new_fault_folio = fault_page ? page_folio(fault_page) : NULL;
+
+       /*
+        * Ensure the lock is held on the correct
+        * folio after the split
+        */
+       if (!new_fault_folio) {
+               folio_unlock(folio);
+               folio_put(folio);
+       } else if (folio != new_fault_folio) {
+               if (new_fault_folio != fault_folio) {
+                       folio_get(new_fault_folio);
+                       folio_lock(new_fault_folio);
+               }
+               folio_unlock(folio);
+               folio_put(folio);
+       }
+
+       return 0;
+}
+
 static int hmm_vma_handle_migrate_prepare_pmd(const struct mm_walk *walk,
                                              pmd_t *pmdp,
                                              unsigned long start,
@@ -496,6 +553,11 @@ static int hmm_vma_handle_migrate_prepare_pmd(const struct 
mm_walk *walk,
        return 0;
 }
 
+/*
+ * Install migration entries if migration requested, either from fault
+ * or migrate paths.
+ *
+ */
 static int hmm_vma_handle_migrate_prepare(const struct mm_walk *walk,
                                          pmd_t *pmdp,
                                          pte_t *ptep,
@@ -503,8 +565,206 @@ static int hmm_vma_handle_migrate_prepare(const struct 
mm_walk *walk,
                                          unsigned long *hmm_pfn,
                                          bool *unmapped)
 {
-       // TODO: implement migration entry insertion
+       struct hmm_vma_walk *hmm_vma_walk = walk->private;
+       struct hmm_range *range = hmm_vma_walk->range;
+       struct migrate_vma *migrate = range->migrate;
+       struct mm_struct *mm = walk->vma->vm_mm;
+       struct folio *fault_folio = NULL;
+       enum migrate_vma_info minfo;
+       struct dev_pagemap *pgmap;
+       bool anon_exclusive;
+       struct folio *folio;
+       unsigned long pfn;
+       struct page *page;
+       softleaf_t entry;
+       pte_t pte, swp_pte;
+       bool writable = false;
+
+       // Do we want to migrate at all?
+       minfo = hmm_select_migrate(range);
+       if (!minfo)
+               return 0;
+
+       WARN_ON_ONCE(!migrate);
+       HMM_ASSERT_PTE_LOCKED(hmm_vma_walk, true);
+
+       fault_folio = migrate->fault_page ?
+               page_folio(migrate->fault_page) : NULL;
+
+       pte = ptep_get(ptep);
+
+       if (pte_none(pte)) {
+               if (vma_is_anonymous(walk->vma)) {
+                       *hmm_pfn &= HMM_PFN_INOUT_FLAGS;
+                       *hmm_pfn |= HMM_PFN_MIGRATE;
+                       goto out;
+               }
+       }
+
+       if (!(hmm_pfn[0] & HMM_PFN_VALID))
+               goto out;
+
+       if (!pte_present(pte)) {
+               /*
+                * Only care about unaddressable device page special
+                * page table entry. Other special swap entries are not
+                * migratable, and we ignore regular swapped page.
+                */
+               entry = softleaf_from_pte(pte);
+               if (!softleaf_is_device_private(entry))
+                       goto out;
+
+               if (!(minfo & MIGRATE_VMA_SELECT_DEVICE_PRIVATE))
+                       goto out;
+
+               page = softleaf_to_page(entry);
+               folio = page_folio(page);
+               if (folio->pgmap->owner != migrate->pgmap_owner)
+                       goto out;
+
+               if (folio_test_large(folio)) {
+                       int ret;
+
+                       ret = migrate_vma_split_folio(folio,
+                                                     migrate->fault_page,
+                                                     hmm_vma_walk,
+                                                     ptep);
+                       if (ret)
+                               goto out_error;
+                       return -EAGAIN;
+               }
+
+               pfn = page_to_pfn(page);
+               if (softleaf_is_device_private_write(entry))
+                       writable = true;
+       } else {
+               pfn = pte_pfn(pte);
+               if (is_zero_pfn(pfn) &&
+                   (minfo & MIGRATE_VMA_SELECT_SYSTEM)) {
+                       *hmm_pfn = HMM_PFN_MIGRATE;
+                       goto out;
+               }
+               page = vm_normal_page(walk->vma, addr, pte);
+               if (page && !is_zone_device_page(page) &&
+                   !(minfo & MIGRATE_VMA_SELECT_SYSTEM)) {
+                       goto out;
+               } else if (page && is_device_coherent_page(page)) {
+                       pgmap = page_pgmap(page);
+
+                       if (!(minfo &
+                             MIGRATE_VMA_SELECT_DEVICE_COHERENT) ||
+                           pgmap->owner != migrate->pgmap_owner)
+                               goto out;
+               }
+
+               folio = page ? page_folio(page) : NULL;
+               if (folio && folio_test_large(folio)) {
+                       int ret;
+
+                       ret = migrate_vma_split_folio(folio,
+                                                     migrate->fault_page,
+                                                     hmm_vma_walk,
+                                                     ptep);
+                       if (ret)
+                               goto out_error;
+                       return -EAGAIN;
+               }
+
+               writable = pte_write(pte);
+       }
+
+       if (!page || !page->mapping)
+               goto out;
+
+       /*
+        * By getting a reference on the folio we pin it and that blocks
+        * any kind of migration. Side effect is that it "freezes" the
+        * pte.
+        *
+        * We drop this reference after isolating the folio from the lru
+        * for non device folio (device folio are not on the lru and thus
+        * can't be dropped from it).
+        */
+       folio = page_folio(page);
+       folio_get(folio);
+
+       /*
+        * We rely on folio_trylock() to avoid deadlock between
+        * concurrent migrations where each is waiting on the others
+        * folio lock. If we can't immediately lock the folio we fail this
+        * migration as it is only best effort anyway.
+        *
+        * If we can lock the folio it's safe to set up a migration entry
+        * now. In the common case where the folio is mapped once in a
+        * single process setting up the migration entry now is an
+        * optimisation to avoid walking the rmap later with
+        * try_to_migrate().
+        */
+
+       if (fault_folio == folio || folio_trylock(folio)) {
+               anon_exclusive = folio_test_anon(folio) &&
+                       PageAnonExclusive(page);
+
+               if (pte_present(pte))
+                       flush_cache_page(walk->vma, addr, pfn);
+
+               if (anon_exclusive) {
+                       pte = ptep_clear_flush(walk->vma, addr, ptep);
+
+                       if (folio_try_share_anon_rmap_pte(folio, page)) {
+                               set_pte_at(mm, addr, ptep, pte);
+                               folio_unlock(folio);
+                               folio_put(folio);
+                               goto out;
+                       }
+               } else {
+                       pte = ptep_get_and_clear(mm, addr, ptep);
+               }
+
+               if (pte_present(pte) && pte_dirty(pte))
+                       folio_mark_dirty(folio);
+
+               /* Setup special migration page table entry */
+               if (writable)
+                       entry = make_writable_migration_entry(pfn);
+               else if (anon_exclusive)
+                       entry = make_readable_exclusive_migration_entry(pfn);
+               else
+                       entry = make_readable_migration_entry(pfn);
+
+               if (pte_present(pte)) {
+                       if (pte_young(pte))
+                               entry = make_migration_entry_young(entry);
+                       if (pte_dirty(pte))
+                               entry = make_migration_entry_dirty(entry);
+               }
+
+               swp_pte = swp_entry_to_pte(entry);
+               if (pte_present(pte)) {
+                       if (pte_soft_dirty(pte))
+                               swp_pte = pte_swp_mksoft_dirty(swp_pte);
+                       if (pte_uffd_wp(pte))
+                               swp_pte = pte_swp_mkuffd_wp(swp_pte);
+               } else {
+                       if (pte_swp_soft_dirty(pte))
+                               swp_pte = pte_swp_mksoft_dirty(swp_pte);
+                       if (pte_swp_uffd_wp(pte))
+                               swp_pte = pte_swp_mkuffd_wp(swp_pte);
+               }
+
+               set_pte_at(mm, addr, ptep, swp_pte);
+               folio_remove_rmap_pte(folio, page, walk->vma);
+               folio_put(folio);
+               *hmm_pfn |= HMM_PFN_MIGRATE;
+               if (pte_present(pte))
+                       *unmapped = true;
+       } else {
+               folio_put(folio);
+       }
+out:
        return 0;
+out_error:
+       return -EFAULT;
 }
 
 static int hmm_vma_walk_split(pmd_t *pmdp,
-- 
2.55.0

Reply via email to