On Mon, Jul 20, 2026 at 03:34:20PM +0800, Gregory Price wrote:
> Provide a mechanism to opt private nodes into the reclaim process.
> 
> Reclaim as a "singular service" is actually made up of:
> 
>    - kswapd reclaim
>    - direct reclaim
>    - kcompactd compaction
>    - direct compaction
>    - both mglru / lru paths
>    - madvise reclaim hints
>    - damon reclaim operations
> 
> CAP_RECLAIM gates whether the kernel may do inter-node placement
> (compaction) or swap for the folios on that private node.
> 
> node_allows_reclaim() encapsulates the policy as a whole.  Ordinary
> nodes are always reclaimable, private nodes only when opted in.
> 
> With the exception of madvise and DAMON, reclaim operations are
> highly integrated with one another, so they are opted in/out of
> together - otherwise reclaim becomes unpredictable.
> 
> To prevent bisect debugging failures, this stays in a single commit.
> 
> For example: reclaim without compaction will OOM on high-order
> allocation failure despite a large amount of free memory. Normally
> compaction would (potentially) resolve this issue.
> 
> If a private node opts into reclaim, we create normal watermarks for
> that node - otherwise pgdat_balanced() is always true and reclaim
> thinks there is no work to do.
> 
> Cross-node operations (demotion, promotion, khugepaged) are NOT
> included in CAP_RECLAIM because some workflows may desire different
> behaviors for this kind of operation:
> 
>    - prefer direct-to-swap, do not demote
>    - remain resident and OOM
>    - __GFP_THISNODE: fail and let the driver augment reclaim
> 
> Normal swap-out is allowed because there are clear userland controls
> (not registering swap, cgroup.swap, etc) to control that per-workload.
> 
> Signed-off-by: Gregory Price <[email protected]>
> ---
>  include/linux/node_private.h | 33 +++++++++++++++++++++++++
>  mm/compaction.c              |  9 ++++---
>  mm/damon/paddr.c             |  4 +--
>  mm/huge_memory.c             |  2 +-
>  mm/internal.h                | 13 ++++++++++
>  mm/madvise.c                 |  6 ++---
>  mm/memory_hotplug.c          |  3 +--
>  mm/page_alloc.c              |  2 +-
>  mm/vmscan.c                  | 48 ++++++++++++++++++++++++++++++------
>  9 files changed, 100 insertions(+), 20 deletions(-)
> 
> diff --git a/include/linux/node_private.h b/include/linux/node_private.h
> index 475496c84249f..f7cbae1309904 100644
> --- a/include/linux/node_private.h
> +++ b/include/linux/node_private.h
> @@ -7,6 +7,13 @@
>  
>  struct page;
>  
> +/*
> + * Per-node service opt-ins (node_private.caps).  A private node is isolated
> + * from all general mm services by default; the registering driver sets these
> + * to let specific services operate on its node.
> + */
> +#define NODE_PRIVATE_CAP_RECLAIM     (1UL << 0)      /* allow mm reclaim */
> +
>  /**
>   * struct node_private - Per-node container for N_MEMORY_PRIVATE nodes
>   *
> @@ -41,6 +48,27 @@ static inline bool node_is_private(int nid)
>       return node_state(nid, N_MEMORY_PRIVATE);
>  }
>  
> +/**
> + * node_allows_reclaim - may the mm reclaim from this node?
> + * @nid: the node to test
> + *
> + * Only a private node is ever excluded.  Every other node can safely
> + * be operated on by reclaim.
> + */
> +static inline bool node_allows_reclaim(int nid)
> +{
> +     struct node_private *np;
> +     bool ret;
> +
> +     if (!node_state(nid, N_MEMORY_PRIVATE))
> +             return true;
> +     rcu_read_lock();
> +     np = rcu_dereference(NODE_DATA(nid)->node_private);
> +     ret = np && (np->caps & NODE_PRIVATE_CAP_RECLAIM);
> +     rcu_read_unlock();
> +     return ret;
> +}
> +
>  #else /* !CONFIG_NUMA */
>  
>  static inline bool folio_is_private_node(struct folio *folio)
> @@ -58,6 +86,11 @@ static inline bool node_is_private(int nid)
>       return false;
>  }
>  
> +static inline bool node_allows_reclaim(int nid)
> +{
> +     return true;
> +}
> +
>  #endif /* CONFIG_NUMA */
>  
>  #if defined(CONFIG_NUMA) && defined(CONFIG_MEMORY_HOTPLUG)
> diff --git a/mm/compaction.c b/mm/compaction.c
> index 8c1351cce7bcc..b9c25c599732f 100644
> --- a/mm/compaction.c
> +++ b/mm/compaction.c
> @@ -25,6 +25,7 @@
>  #include <linux/psi.h>
>  #include <linux/cpuset.h>
>  #include "page_alloc.h"
> +#include <linux/node_private.h>
>  #include "internal.h"
>  
>  #ifdef CONFIG_COMPACTION
> @@ -2462,7 +2463,7 @@ bool compaction_zonelist_suitable(struct alloc_context 
> *ac, int order,
>                   !__cpuset_zone_allowed(zone, gfp_mask))
>                       continue;
>  
> -             if (node_is_private(zone_to_nid(zone)))
> +             if (!node_allows_reclaim(zone_to_nid(zone)))
>                       continue;
>  
>               /*
> @@ -2856,7 +2857,7 @@ enum compact_result try_to_compact_pages(gfp_t 
> gfp_mask, unsigned int order,
>                       !__cpuset_zone_allowed(zone, gfp_mask))
>                               continue;
>  
> -             if (node_is_private(zone_to_nid(zone)))
> +             if (!node_allows_reclaim(zone_to_nid(zone)))
>                       continue;
>  
>               if (prio > MIN_COMPACT_PRIORITY
> @@ -2928,7 +2929,7 @@ static int compact_node(pg_data_t *pgdat, bool 
> proactive)
>               .proactive_compaction = proactive,
>       };
>  
> -     if (node_is_private(pgdat->node_id))
> +     if (!node_allows_reclaim(pgdat->node_id))
>               return 0;
>  
>       for (zoneid = 0; zoneid < MAX_NR_ZONES; zoneid++) {
> @@ -3026,7 +3027,7 @@ static ssize_t compact_store(struct device *dev,
>  {
>       int nid = dev->id;
>  
> -     if (node_is_private(nid))
> +     if (!node_allows_reclaim(nid))
>               return -EINVAL;
>  
>       if (nid >= 0 && nid < nr_node_ids && node_online(nid)) {
> diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c
> index c741a94319750..b668cf55d1b67 100644
> --- a/mm/damon/paddr.c
> +++ b/mm/damon/paddr.c
> @@ -251,8 +251,8 @@ static unsigned long damon_pa_pageout(struct damon_region 
> *r,
>                       continue;
>               }
>  
> -             /* private node memory is not reclaimable by default */
> -             if (folio_is_private_node(folio))
> +             /* DAMOS pageout is reclaim; gate a private node on CAP_RECLAIM 
> */
> +             if (!node_allows_reclaim(folio_nid(folio)))
>                       goto put_folio;
>  
>               if (damos_pa_filter_out(s, folio))
> diff --git a/mm/huge_memory.c b/mm/huge_memory.c
> index 1df91b4e5c2bc..cebc89eb0541f 100644
> --- a/mm/huge_memory.c
> +++ b/mm/huge_memory.c
> @@ -2340,7 +2340,7 @@ bool madvise_free_huge_pmd(struct mmu_gather *tlb, 
> struct vm_area_struct *vma,
>  
>       folio = pmd_folio(orig_pmd);
>  
> -     if (folio_is_private_node(folio))
> +     if (!node_allows_reclaim(folio_nid(folio)))
>               goto out;
>  
>       /*
> diff --git a/mm/internal.h b/mm/internal.h
> index 85c460296cea1..8329034ae561f 100644
> --- a/mm/internal.h
> +++ b/mm/internal.h
> @@ -110,6 +110,19 @@ static inline bool page_is_private_managed(struct page 
> *page)
>       return folio_is_private_managed(page_folio(page));
>  }
>  
> +/*
> + * folio_allows_madvise() - may madvise reclaim hints act on this folio?
> + *
> + * madvise reclaim hints (COLD/PAGEOUT/FREE) are userland-driven reclaim, so
> + * they follow reclaim opt-in: false for ZONE_DEVICE and for N_MEMORY_PRIVATE
> + * nodes without CAP_RECLAIM, true for all other normal folios.
> + */
> +static inline bool folio_allows_madvise(struct folio *folio)
> +{
> +     return !folio_is_zone_device(folio) &&
> +            node_allows_reclaim(folio_nid(folio));
> +}
> +
>  /*
>   * folio_allows_longterm_pin() - may this folio be long-term GUP-pinned?
>   *
> diff --git a/mm/madvise.c b/mm/madvise.c
> index 29f35a23919a0..56ca974542707 100644
> --- a/mm/madvise.c
> +++ b/mm/madvise.c
> @@ -396,7 +396,7 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd,
>  
>               folio = pmd_folio(orig_pmd);
>  
> -             if (folio_is_private_node(folio))
> +             if (!node_allows_reclaim(folio_nid(folio)))
>                       goto huge_unlock;
>  
>               /* Do not interfere with other mappings of this folio */
> @@ -478,7 +478,7 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd,
>                       continue;
>  
>               folio = vm_normal_folio(vma, addr, ptent);
> -             if (!folio || folio_is_private_managed(folio))
> +             if (!folio || !folio_allows_madvise(folio))
>                       continue;
>  
>               /*
> @@ -707,7 +707,7 @@ static int madvise_free_pte_range(pmd_t *pmd, unsigned 
> long addr,
>               }
>  
>               folio = vm_normal_folio(vma, addr, ptent);
> -             if (!folio || folio_is_private_managed(folio))
> +             if (!folio || !folio_allows_madvise(folio))
>                       continue;
>  
>               /*
> diff --git a/mm/memory_hotplug.c b/mm/memory_hotplug.c
> index be230ac9efe5a..1f42ed303366c 100644
> --- a/mm/memory_hotplug.c
> +++ b/mm/memory_hotplug.c
> @@ -1237,8 +1237,7 @@ int online_pages(unsigned long pfn, unsigned long 
> nr_pages,
>       /* reinitialise watermarks and update pcp limits */
>       init_per_zone_wmark_min();
>  
> -     /* Private nodes opt-out of reclaim/compaction by default */
> -     if (!node_is_private(nid)) {
> +     if (node_allows_reclaim(nid)) {
>               kswapd_run(nid);
>               kcompactd_run(nid);
>       }
> diff --git a/mm/page_alloc.c b/mm/page_alloc.c
> index 2b08bea2379a9..2667a4564b7ac 100644
> --- a/mm/page_alloc.c
> +++ b/mm/page_alloc.c
> @@ -6667,7 +6667,7 @@ static void __setup_per_zone_wmarks(void)
>               u64 tmp;
>  
>               spin_lock_irqsave(&zone->lock, flags);
> -             if (node_is_private(zone_to_nid(zone))) {
> +             if (!node_allows_reclaim(zone_to_nid(zone))) {
>                       zone->_watermark[WMARK_MIN] = 0;
>                       zone->_watermark[WMARK_LOW] = 0;
>                       zone->_watermark[WMARK_HIGH] = 0;
> diff --git a/mm/vmscan.c b/mm/vmscan.c
> index 86b2334c23b98..f1722693ac2db 100644
> --- a/mm/vmscan.c
> +++ b/mm/vmscan.c
> @@ -5396,6 +5396,21 @@ static const struct attribute_group lru_gen_attr_group 
> = {
>   *                          debugfs interface
>   
> ******************************************************************************/
>  
> +/*
> + * Nodes the lru_gen debugfs interface lists: ordinary memory nodes plus any
> + * N_MEMORY_PRIVATE nodes opted into reclaim.  run_cmd() already accepts the
> + * latter, so keep the listing in sync with what it accepts.
> + */
> +static void lru_gen_seq_nodes(nodemask_t *nodes)
> +{
> +     int nid;
> +
> +     *nodes = node_states[N_MEMORY];
> +     for_each_node_state(nid, N_MEMORY_PRIVATE)
> +             if (node_allows_reclaim(nid))
> +                     node_set(nid, *nodes);
> +}
> +
>  static void *lru_gen_seq_start(struct seq_file *m, loff_t *pos)
>  {
>       struct mem_cgroup *memcg;
> @@ -5407,9 +5422,11 @@ static void *lru_gen_seq_start(struct seq_file *m, 
> loff_t *pos)
>  
>       memcg = mem_cgroup_iter(NULL, NULL, NULL);
>       do {
> +             nodemask_t nodes;
>               int nid;
>  
> -             for_each_node_state(nid, N_MEMORY) {
> +             lru_gen_seq_nodes(&nodes);
> +             for_each_node_mask(nid, nodes) {
>                       if (!nr_to_skip--)
>                               return get_lruvec(memcg, nid);
>               }
> @@ -5431,16 +5448,18 @@ static void *lru_gen_seq_next(struct seq_file *m, 
> void *v, loff_t *pos)
>  {
>       int nid = lruvec_pgdat(v)->node_id;
>       struct mem_cgroup *memcg = lruvec_memcg(v);
> +     nodemask_t nodes;
>  
>       ++*pos;
>  
> -     nid = next_memory_node(nid);
> +     lru_gen_seq_nodes(&nodes);
> +     nid = next_node(nid, nodes);
>       if (nid == MAX_NUMNODES) {
>               memcg = mem_cgroup_iter(NULL, memcg, NULL);
>               if (!memcg)
>                       return NULL;
>  
> -             nid = first_memory_node;
> +             nid = first_node(nodes);
>       }
>  
>       return get_lruvec(memcg, nid);
> @@ -5509,10 +5528,12 @@ static int lru_gen_seq_show(struct seq_file *m, void 
> *v)
>       struct lru_gen_folio *lrugen = &lruvec->lrugen;
>       int nid = lruvec_pgdat(lruvec)->node_id;
>       struct mem_cgroup *memcg = lruvec_memcg(lruvec);
> +     nodemask_t nodes;
>       DEFINE_MAX_SEQ(lruvec);
>       DEFINE_MIN_SEQ(lruvec);
>  
> -     if (nid == first_memory_node) {
> +     lru_gen_seq_nodes(&nodes);
> +     if (nid == first_node(nodes)) {
>               const char *path = memcg ? m->private : "";
>  
>  #ifdef CONFIG_MEMCG
> @@ -5612,7 +5633,9 @@ static int run_cmd(char cmd, u64 memcg_id, int nid, 
> unsigned long seq,
>       int err = -EINVAL;
>       struct mem_cgroup *memcg = NULL;
>  
> -     if (nid < 0 || nid >= MAX_NUMNODES || !node_state(nid, N_MEMORY))
> +     if (nid < 0 || nid >= MAX_NUMNODES ||
> +         !(node_state(nid, N_MEMORY) ||
> +           (node_is_private(nid) && node_allows_reclaim(nid))))
>               return -EINVAL;
>  
>       if (!mem_cgroup_disabled()) {
> @@ -6145,7 +6168,7 @@ static void shrink_node(pg_data_t *pgdat, struct 
> scan_control *sc)
>        * Private nodes do not support reclaim by default, filtering here
>        * captures all normal reclaim paths that may attempt eviction.
>        */
> -     if (node_is_private(pgdat->node_id))
> +     if (!node_allows_reclaim(pgdat->node_id))
>               return;
>  
>       if ((lru_gen_enabled() || lru_gen_switching()) && root_reclaim(sc)) {
> @@ -6758,6 +6781,16 @@ unsigned long mem_cgroup_shrink_node(struct mem_cgroup 
> *memcg,
>       return sc.nr_reclaimed;
>  }
>  
> +static struct zonelist *memcg_reclaim_zonelist(int nid, gfp_t gfp_mask)
> +{
> +     unsigned int aflags = ALLOC_DEFAULT;
> +
> +     if (unlikely(!nodes_empty(node_states[N_MEMORY_PRIVATE])))
> +             aflags = ALLOC_ZONELIST_PRIVATE;
> +
> +     return select_zonelist(nid, gfp_mask, aflags);
> +}


I applied this code change and notice in mem_cgroup_shrink_node(), it calls
shrink_lruvec() directly and never passes through shrink_node().

This patch makes private nodes reachable through memcg reclaim zonelist, but
what prevents the direct per-node memcg path from reclaiming a private node
that has not set NODE_PRIVATE_CAP_RECLAIM ?

--Richard


> +
>  unsigned long try_to_free_mem_cgroup_pages(struct mem_cgroup *memcg,
>                                          unsigned long nr_pages,
>                                          gfp_t gfp_mask,
> @@ -6784,7 +6817,8 @@ unsigned long try_to_free_mem_cgroup_pages(struct 
> mem_cgroup *memcg,
>        * equal pressure on all the nodes. This is based on the assumption that
>        * the reclaim does not bail out early.
>        */
> -     struct zonelist *zonelist = node_zonelist(numa_node_id(), sc.gfp_mask);
> +     struct zonelist *zonelist = memcg_reclaim_zonelist(numa_node_id(),
> +                                                        sc.gfp_mask);
>  
>       set_task_reclaim_state(current, &sc.reclaim_state);
>       trace_mm_vmscan_memcg_reclaim_begin(sc.gfp_mask, 0, memcg);
> -- 
> 2.53.0-Meta
> 
> 

Reply via email to