On Thu, Sep 17, 2026 at 9:27 AM Lorenzo Stoakes (ARM) <[email protected]> wrote:
>
> The existing kernel page mapping mmap actions allow for partial and full
> mapping of an array of struct page pointers.
>
> However some drivers require the mapping of discontiguous ranges. Permit
> this by providing discontig_kernel_page_ops which allows a driver to
> specify how the operation should begin and how batches of pages should be
> retrieved.
>
> It uses the minimum exposed interface to do so, providing address, page
> offset and both vm_private_data state and a local private state object.
>
> ops->init can establish state for the operation, and ops->get outputs the
> pages to map and their count. Should an error arise the core unmaps the
> VMA, invoking vm_ops->close, which is therefore where any state established
> by ops->init is released.
>
> Batches may not exceed the VMA, but may map less than its full range in
> case the driver wishes to allow the user to map an area larger than the
> available data.
>
> To use it, users invoke mmap_action_map_discontig_kernel_pages() with
> initial local private state and a set of operations.
>
> Users can then use one of the provided helper functions to perform an
> action:
>
> * discontig_kernel_map_abort() - Abort and leave the mapping as it has
> been accumulated so far.
> * discontig_kernel_map_page() - Map a single page, or a compound page given
> its head page.
> * discontig_kernel_map_page_range() - Maps a struct page ** array of a
> specified count.
>
> The userland VMA tests are updated accordingly.
>
> Signed-off-by: Lorenzo Stoakes (ARM) <[email protected]>
> ---
> include/linux/mm.h | 45 +++++++++++++++++
> include/linux/mm_types.h | 44 +++++++++++++++-
> mm/internal.h | 3 ++
> mm/memory.c | 108
> ++++++++++++++++++++++++++++++++++++++--
> mm/util.c | 7 +++
> tools/testing/vma/include/dup.h | 11 +++-
> 6 files changed, 209 insertions(+), 9 deletions(-)
>
> diff --git a/include/linux/mm.h b/include/linux/mm.h
> index a1f2d375cf7d..2a92193ac6a5 100644
> --- a/include/linux/mm.h
> +++ b/include/linux/mm.h
> @@ -4647,10 +4647,55 @@ static inline void
> mmap_action_map_kernel_pages_full(struct vm_area_desc *desc,
> vma_desc_pages(desc));
> }
>
> +static inline
> +void mmap_action_map_discontig_kernel_pages(struct vm_area_desc *desc,
> + void *init_private, const struct discontig_kernel_page_ops
> *ops)
> +{
> + struct mmap_action *action = &desc->action;
> +
> + action->type = MMAP_DISCONTIG_KERNEL_PAGES;
> + action->map_kernel_discontig.init_private = init_private;
> + action->map_kernel_discontig.ops = ops;
> +}
> +
> int mmap_action_prepare(struct vm_area_desc *desc);
> int mmap_action_complete(struct vm_area_struct *vma,
> struct mmap_action *action, bool is_compat);
>
> +static inline void
> +discontig_kernel_map_abort(struct discontig_kernel_page_state *state)
> +{
> + state->action = DISCONTIG_KERNEL_PAGE_ABORT;
> +}
> +
> +static inline void
> +discontig_kernel_map_page(struct discontig_kernel_page_state *state,
> + struct page *page)
> +{
> + struct folio *folio = page_folio(page);
> +
> + if (folio_test_large(folio)) {
> + VM_WARN_ON_ONCE(page != folio_page(folio, 0));
> + state->action = DISCONTIG_KERNEL_PAGE_MAP_COMPOUND_PAGE;
> + state->__folio = folio;
> + state->__nr_pages = min(state->nr_pages_remain,
> + folio_nr_pages(folio));
> + } else {
> + state->action = DISCONTIG_KERNEL_PAGE_MAP_PAGE;
> + state->__page = page;
> + state->__nr_pages = 1;
> + }
> +}
> +
> +static inline void
> +discontig_kernel_map_page_range(struct discontig_kernel_page_state *state,
> + struct page **page_arr, unsigned long
> nr_pages)
> +{
> + state->action = DISCONTIG_KERNEL_PAGE_MAP_PAGE_RANGE;
> + state->__page_arr = page_arr;
> + state->__nr_pages = nr_pages;
Shouldn't we cap state->__nr_pages at state->nr_pages_remain like you
do in discontig_kernel_map_page()?
> +}
> +
> /* Look up the first VMA which exactly match the interval vm_start ...
> vm_end */
> static inline struct vm_area_struct *find_exact_vma(struct mm_struct *mm,
> unsigned long vm_start, unsigned long vm_end)
> diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h
> index 9ca2ea3664bc..0cb4f9603956 100644
> --- a/include/linux/mm_types.h
> +++ b/include/linux/mm_types.h
> @@ -818,8 +818,44 @@ enum mmap_action_type {
> MMAP_NOTHING,
> MMAP_REMAP_PFN,
> MMAP_IO_REMAP_PFN,
> - MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */
> - MMAP_KERNEL_PAGES, /* Map kernel page range from array. */
> + MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */
> + MMAP_KERNEL_PAGES, /* Map kernel page range from array.
> */
> + MMAP_DISCONTIG_KERNEL_PAGES, /* Map kernel discontig page range. */
> +};
> +
> +enum discontig_kernel_page_action {
> + DISCONTIG_KERNEL_PAGE_ABORT,
> + DISCONTIG_KERNEL_PAGE_MAP_PAGE,
> + DISCONTIG_KERNEL_PAGE_MAP_COMPOUND_PAGE,
> + DISCONTIG_KERNEL_PAGE_MAP_PAGE_RANGE,
> +};
> +
> +struct discontig_kernel_page_state {
> + /* Map state. */
> + const unsigned long start; /* Start address of VMA. */
> + const unsigned long end; /* End address of VMA. */
> + unsigned long addr; /* The current address to be mapped.
> */
> + pgoff_t pgoff; /* The current pgoff to be mapped. */
> + unsigned long nr_pages_mapped; /* The number of pages mapped. */
> + unsigned long nr_pages_remain; /* The number of pages remaining. */
> +
> + /* User-defined state. */
> + void *vm_private_data; /* VMA private data. */
> + void *private; /* Mapping private data. */
> +
> + /* Users should not touch these, use discontig_kernel_map_*()
> helpers. */
> + enum discontig_kernel_page_action action;
> + union {
> + struct page *__page;
> + struct folio *__folio;
> + struct page **__page_arr;
> + };
> + unsigned long __nr_pages;
> +};
> +
> +struct discontig_kernel_page_ops {
> + int (*init)(void *vm_private_data, void **private);
> + int (*get)(struct discontig_kernel_page_state *state);
> };
>
> /*
> @@ -844,6 +880,10 @@ struct mmap_action {
> unsigned long nr_pages;
> pgoff_t pgoff;
> } map_kernel;
> + struct {
> + void *init_private;
> + const struct discontig_kernel_page_ops *ops;
> + } map_kernel_discontig;
> };
> enum mmap_action_type type;
>
> diff --git a/mm/internal.h b/mm/internal.h
> index c767007c2187..b81fce5fe510 100644
> --- a/mm/internal.h
> +++ b/mm/internal.h
> @@ -1516,6 +1516,9 @@ int simple_ioremap_prepare(struct vm_area_desc *desc);
> int map_kernel_pages_prepare(struct vm_area_desc *desc);
> int map_kernel_pages_complete(struct vm_area_struct *vma,
> struct mmap_action *action);
> +int map_discontig_kernel_pages_prepare(struct vm_area_desc *desc);
> +int map_discontig_kernel_pages_complete(struct vm_area_struct *vma,
> + struct mmap_action *action);
>
> static inline int io_remap_pfn_range_prepare(struct vm_area_desc *desc)
> {
> diff --git a/mm/memory.c b/mm/memory.c
> index 448342883e9d..45b21bb04a18 100644
> --- a/mm/memory.c
> +++ b/mm/memory.c
> @@ -2609,17 +2609,23 @@ int vm_insert_pages(struct vm_area_struct *vma,
> unsigned long addr,
> }
> EXPORT_SYMBOL(vm_insert_pages);
>
> +static void __map_kernel_pages_prepare(struct vm_area_desc *desc)
> +{
> + if (vma_desc_test(desc, VMA_MIXEDMAP_BIT))
> + return;
> +
> + VM_WARN_ON_ONCE(mmap_read_trylock(desc->mm));
> + VM_WARN_ON_ONCE(vma_desc_test(desc, VMA_PFNMAP_BIT));
> + vma_desc_set_flags(desc, VMA_MIXEDMAP_BIT);
> +}
> +
> int map_kernel_pages_prepare(struct vm_area_desc *desc)
> {
> const struct mmap_action *action = &desc->action;
> const unsigned long addr = action->map_kernel.start;
> unsigned long nr_pages, end;
>
> - if (!vma_desc_test(desc, VMA_MIXEDMAP_BIT)) {
> - VM_WARN_ON_ONCE(mmap_read_trylock(desc->mm));
> - VM_WARN_ON_ONCE(vma_desc_test(desc, VMA_PFNMAP_BIT));
> - vma_desc_set_flags(desc, VMA_MIXEDMAP_BIT);
> - }
> + __map_kernel_pages_prepare(desc);
>
> nr_pages = action->map_kernel.nr_pages;
> end = addr + PAGE_SIZE * nr_pages;
> @@ -2640,6 +2646,98 @@ int map_kernel_pages_complete(struct vm_area_struct
> *vma,
> &nr_pages, vma->vm_page_prot);
> }
>
> +int map_discontig_kernel_pages_prepare(struct vm_area_desc *desc)
> +{
> + const struct mmap_action *action = &desc->action;
> + const struct discontig_kernel_page_ops *ops =
> + action->map_kernel_discontig.ops;
> +
> + /* At minimum need to be able to get pages. */
> + if (WARN_ON_ONCE(!ops || !ops->get))
> + return -EINVAL;
> +
> + __map_kernel_pages_prepare(desc);
> + return 0;
> +}
> +
> +static int apply_discontig_action(struct vm_area_struct *vma,
> + struct discontig_kernel_page_state *state)
> +{
> + unsigned long nr_pages = state->__nr_pages;
> + unsigned long addr = state->addr;
> + unsigned long i;
> +
> + if (state->action == DISCONTIG_KERNEL_PAGE_MAP_PAGE)
> + return insert_page(vma, addr, state->__page,
> + vma->vm_page_prot, /*mkwrite=*/false);
> + if (state->action == DISCONTIG_KERNEL_PAGE_MAP_PAGE_RANGE)
> + return insert_pages(vma, addr, state->__page_arr,
> + &nr_pages, vma->vm_page_prot);
> +
> + /* Compound folio - have to iterate through each page. */
> + for (i = 0; i < nr_pages; i++, addr += PAGE_SIZE) {
> + struct page *page = folio_page(state->__folio, i);
> + int err;
> +
> + err = insert_page(vma, addr, page, vma->vm_page_prot,
> + /*mkwrite=*/false);
> + if (err)
> + return err;
> + }
> + return 0;
> +}
> +
> +int map_discontig_kernel_pages_complete(struct vm_area_struct *vma,
> + struct mmap_action *action)
> +{
> + const struct discontig_kernel_page_ops *ops =
> + action->map_kernel_discontig.ops;
> + struct discontig_kernel_page_state state = {
> + .start = vma->vm_start,
> + .end = vma->vm_end,
> + .addr = vma->vm_start,
> + .pgoff = vma->vm_pgoff,
> + .nr_pages_mapped = 0,
> + .nr_pages_remain = vma_pages(vma),
> + .vm_private_data = vma->vm_private_data,
> + .private = action->map_kernel_discontig.init_private,
> + };
> + int err = 0;
> +
> + if (ops->init)
> + err = ops->init(vma->vm_private_data, &state.private);
> + if (err)
> + return err;
> +
> + do {
> + unsigned long end, pgoff_end;
> + unsigned long nr_pages;
> +
> + /* Default to abort. */
> + state.action = DISCONTIG_KERNEL_PAGE_ABORT;
> + err = ops->get(&state);
> + if (err || state.action == DISCONTIG_KERNEL_PAGE_ABORT)
> + return err;
> + nr_pages = state.__nr_pages;
> +
> + if (!nr_pages || nr_pages > state.nr_pages_remain)
> + return -EINVAL;
> + end = state.addr + PAGE_SIZE * nr_pages;
> + pgoff_end = state.pgoff + nr_pages;
> +
> + err = apply_discontig_action(vma, &state);
> + if (err)
> + return err;
> +
> + state.addr = end;
> + state.pgoff = pgoff_end;
> + state.nr_pages_mapped += nr_pages;
> + state.nr_pages_remain -= nr_pages;
> + } while (state.addr < vma->vm_end);
> +
> + return 0;
> +}
> +
> /**
> * vm_insert_page - insert single page into user vma
> * @vma: user vma to map to
> diff --git a/mm/util.c b/mm/util.c
> index 438170490e7f..c5ee52aede1e 100644
> --- a/mm/util.c
> +++ b/mm/util.c
> @@ -1469,6 +1469,8 @@ int mmap_action_prepare(struct vm_area_desc *desc)
> return simple_ioremap_prepare(desc);
> case MMAP_KERNEL_PAGES:
> return map_kernel_pages_prepare(desc);
> + case MMAP_DISCONTIG_KERNEL_PAGES:
> + return map_discontig_kernel_pages_prepare(desc);
> }
>
> WARN_ON_ONCE(1);
> @@ -1501,6 +1503,9 @@ int mmap_action_complete(struct vm_area_struct *vma,
> case MMAP_KERNEL_PAGES:
> err = map_kernel_pages_complete(vma, action);
> break;
> + case MMAP_DISCONTIG_KERNEL_PAGES:
> + err = map_discontig_kernel_pages_complete(vma, action);
> + break;
> case MMAP_IO_REMAP_PFN:
> case MMAP_SIMPLE_IO_REMAP:
> /* Should have been delegated. */
> @@ -1522,6 +1527,7 @@ int mmap_action_prepare(struct vm_area_desc *desc)
> case MMAP_IO_REMAP_PFN:
> case MMAP_SIMPLE_IO_REMAP:
> case MMAP_KERNEL_PAGES:
> + case MMAP_DISCONTIG_KERNEL_PAGES:
> WARN_ON_ONCE(1); /* nommu cannot handle these. */
> break;
> }
> @@ -1543,6 +1549,7 @@ int mmap_action_complete(struct vm_area_struct *vma,
> case MMAP_IO_REMAP_PFN:
> case MMAP_SIMPLE_IO_REMAP:
> case MMAP_KERNEL_PAGES:
> + case MMAP_DISCONTIG_KERNEL_PAGES:
> WARN_ON_ONCE(1); /* nommu cannot handle this. */
>
> err = -EINVAL;
> diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h
> index 1098655a5f4a..1d5f6b3cbd21 100644
> --- a/tools/testing/vma/include/dup.h
> +++ b/tools/testing/vma/include/dup.h
> @@ -457,14 +457,17 @@ enum mmap_action_type {
> MMAP_NOTHING,
> MMAP_REMAP_PFN,
> MMAP_IO_REMAP_PFN,
> - MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */
> - MMAP_KERNEL_PAGES, /* Map kernel page range from array. */
> + MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */
> + MMAP_KERNEL_PAGES, /* Map kernel page range from array.
> */
> + MMAP_DISCONTIG_KERNEL_PAGES, /* Map kernel discontig page range. */
> };
>
> /*
> * Describes an action an mmap_prepare hook can instruct to be taken to
> complete
> * the mapping of a VMA. Specified in vm_area_desc.
> */
> +struct discontig_kernel_page_ops;
> +
> struct mmap_action {
> union {
> struct {
> @@ -483,6 +486,10 @@ struct mmap_action {
> unsigned long nr_pages;
> pgoff_t pgoff;
> } map_kernel;
> + struct {
> + void *init_private;
> + const struct discontig_kernel_page_ops *ops;
> + } map_kernel_discontig;
> };
> enum mmap_action_type type;
>
>
> --
> 2.55.0
>