On Thu, Sep 17, 2026 at 12:05:37AM +0800, [email protected] wrote:
> From: Manish Honap <[email protected]>
> 
> A Type-2 accelerator issues ATS-translated DMA to addresses inside its
> own HDM window, so that coherent host range must be present in the
> guest's IOAS (the iommufd IOAS backing the nested SMMU stage-2). iommufd
> maps a struct-page-less range only by fd, via IOMMU_IOAS_MAP_FILE over a
> dma-buf; a userspace-VA IOMMU_IOAS_MAP of the HDM mmap is rejected
> because the VMA is VM_IO | VM_PFNMAP. Without a dma-buf the range could
> only be mapped through an out-of-tree PFNMAP work-around.
> 
> vfio-pci already exports BAR memory as a P2P dma-buf, but the exporter
> is BAR-only: vfio_pci_core_feature_dma_buf() rejects any region index at
> or above the ROM index, and vfio_pci_core_get_dmabuf_phys() resolves the
> physical range from a PCI BAR. The HDM memory region is a dynamic
> device-specific region, not a BAR.
> 
> Let a device-specific region reach the device's get_dmabuf_phys(): a
> region index at or above VFIO_PCI_NUM_REGIONS skips the BAR-resource
> check and is validated by the driver instead, bounded to the regions
> that exist. Install a CXL-aware get_dmabuf_phys() in the vfio-cxl
> provider that returns cxl->hpa_range for the HDM memory region and
> delegates real BARs to the core, keeping the BAR path unchanged and the
> core free of CXL knowledge.
> 
> The HDM window is coherent host memory with no p2pdma provider of its
> own, so borrow BAR 0's, matching nvgrace-gpu's handling of its non-BAR
> device memory. The iommufd importer does not consume the provider; the
> scatterlist map path (real peer DMA) is left to a follow-up once
> upstream grows a negotiated interconnect for coherent CXL memory.
> 
> Assisted-by: LLM
> Signed-off-by: Manish Honap <[email protected]>
> ---
>  drivers/vfio/pci/cxl/vfio_cxl_core.c | 60 ++++++++++++++++++++++++++++
>  drivers/vfio/pci/vfio_pci_dmabuf.c   | 27 +++++++++++--
>  2 files changed, 83 insertions(+), 4 deletions(-)
> 
> diff --git a/drivers/vfio/pci/cxl/vfio_cxl_core.c 
> b/drivers/vfio/pci/cxl/vfio_cxl_core.c
> index 5fe8e35c63c4..55fa1f86850d 100644
> --- a/drivers/vfio/pci/cxl/vfio_cxl_core.c
> +++ b/drivers/vfio/pci/cxl/vfio_cxl_core.c
> @@ -11,6 +11,7 @@
>  #include <linux/mm.h>
>  #include <linux/module.h>
>  #include <linux/pci.h>
> +#include <linux/pci-p2pdma.h>
>  #include <linux/range.h>
>  #include <linux/slab.h>
>  #include <linux/uaccess.h>
> @@ -27,6 +28,7 @@
>   * @hdm_regs: mapped HDM decoder registers, read live by the decoder region
>   * @hdm_len: length of the HDM decoder register block
>   * @hdm_valid: true when host CPU access to the HDM range is safe; under 
> memory_lock
> + * @mem_region_index: vfio region index of the mmap-able HDM memory region
>   */
>  struct vfio_cxl_state {
>       struct cxl_dev_state cxlds;
> @@ -36,6 +38,7 @@ struct vfio_cxl_state {
>       void __iomem *hdm_regs;
>       u32 hdm_len;
>       bool hdm_valid;
> +     unsigned int mem_region_index;
>  };
>  
>  static unsigned long vfio_cxl_mem_pgoff(struct vm_area_struct *vma,
> @@ -289,6 +292,50 @@ static void vfio_cxl_release_hpa(void *data)
>       release_mem_region(cxl->hpa_range.start, range_len(&cxl->hpa_range));
>  }
>  
> +/*
> + * Resolve the physical range that backs a dma-buf export. The core exporter
> + * only knows BARs; teach it the HDM memory region so a guest IOAS can map 
> the
> + * coherent window by fd (IOMMU_IOAS_MAP_FILE) instead of the removed PFNMAP
> + * work-around. Real BARs stay on the byte-identical core path.
> + */
> +static int vfio_cxl_get_dmabuf_phys(struct vfio_pci_core_device *vdev,
> +                                 struct p2pdma_provider **provider,
> +                                 unsigned int region_index,
> +                                 struct phys_vec *phys_vec,
> +                                 struct vfio_region_dma_range *dma_ranges,
> +                                 size_t nr_ranges)
> +{
> +     struct vfio_cxl_state *cxl = vdev->cxl;
> +
> +     /* Real BARs go through the core P2P exporter unchanged. */
> +     if (region_index < VFIO_PCI_NUM_REGIONS)
> +             return vfio_pci_core_get_dmabuf_phys(vdev, provider,
> +                                                  region_index, phys_vec,
> +                                                  dma_ranges, nr_ranges);
> +
> +     /* Of the device regions, only the HDM memory window is exportable. */
> +     if (region_index != cxl->mem_region_index)
> +             return -EINVAL;
> +
> +     /*
> +      * The HDM window is coherent host memory, not BAR MMIO, so it has no
> +      * p2pdma provider of its own. Borrow BAR 0's: the P2P properties match
> +      * and the iommufd importer does not consume the provider. The sgt map
> +      * path (real peer DMA) is not supported for the HDM window.
> +      */
> +     *provider = pcim_p2pdma_provider(vdev->pdev, 0);

I think of a weird scenario which might make peer DMA work, not sure if that's 
possible.

userspace can give this dma-buf fd to an NIC driver or so to do RDMA ? use HDM 
memory as
network buffer ?

If that's the case and direct P2P addressing is selected, it applies BAR 0's 
bus offset to
HDM physical address.

Maybe explicitly reject peer-DMA mapping would be safer ?

Best regards,
Richard Cheng.


> +     if (!*provider)
> +             return -EINVAL;
> +
> +     return vfio_pci_core_fill_phys_vec(phys_vec, dma_ranges, nr_ranges,
> +                                       cxl->hpa_range.start,
> +                                       range_len(&cxl->hpa_range));
> +}
> +
> +static const struct vfio_pci_device_ops vfio_cxl_pci_dev_ops = {
> +     .get_dmabuf_phys = vfio_cxl_get_dmabuf_phys,
> +};
> +
>  static int vfio_cxl_init_device(struct vfio_pci_core_device *vdev)
>  {
>       struct pci_dev *pdev = vdev->pdev;
> @@ -483,6 +530,19 @@ static int vfio_cxl_open_device(struct 
> vfio_pci_core_device *vdev)
>       if (ret)
>               return ret;
>  
> +     /* Record where the HDM memory region landed for the dma-buf export. */
> +     cxl->mem_region_index = VFIO_PCI_NUM_REGIONS + vdev->num_regions - 1;
> +
> +     /*
> +      * Override the device ops so a dma-buf export of the HDM memory region
> +      * resolves to the coherent host range. This is done at open, not init:
> +      * vfio_pci_probe() resets pci_ops after vfio_alloc_device() returns, so
> +      * an override installed during init would be clobbered. Only a CXL 
> device
> +      * reaches this hook (cxl_ops is set on init success), so a fallback to
> +      * plain vfio-pci keeps the core ops.
> +      */
> +     vdev->pci_ops = &vfio_cxl_pci_dev_ops;
> +
>       ret = vfio_cxl_add_region(vdev, VFIO_REGION_SUBTYPE_CXL_COMP_REGS,
>                                 &vfio_cxl_comp_regops, cxl->hdm_len,
>                                 VFIO_REGION_INFO_FLAG_READ |
> diff --git a/drivers/vfio/pci/vfio_pci_dmabuf.c 
> b/drivers/vfio/pci/vfio_pci_dmabuf.c
> index c16f460c01d6..436c616d5b66 100644
> --- a/drivers/vfio/pci/vfio_pci_dmabuf.c
> +++ b/drivers/vfio/pci/vfio_pci_dmabuf.c
> @@ -178,6 +178,15 @@ int vfio_pci_core_get_dmabuf_phys(struct 
> vfio_pci_core_device *vdev,
>  {
>       struct pci_dev *pdev = vdev->pdev;
>  
> +     /*
> +      * This resolver only handles PCI BARs. A device-specific region index
> +      * (>= PCI_STD_NUM_BARS) would index pdev->resource[] out of bounds via
> +      * pcim_p2pdma_provider(), so reject it; a driver that exports such a
> +      * region installs its own get_dmabuf_phys.
> +      */
> +     if (region_index >= PCI_STD_NUM_BARS)
> +             return -EINVAL;
> +
>       *provider = pcim_p2pdma_provider(pdev, region_index);
>       if (!*provider)
>               return -EINVAL;
> @@ -227,6 +236,7 @@ int vfio_pci_core_feature_dma_buf(struct 
> vfio_pci_core_device *vdev, u32 flags,
>       DEFINE_DMA_BUF_EXPORT_INFO(exp_info);
>       struct vfio_pci_dma_buf *priv;
>       size_t length;
> +     u32 index;
>       int ret;
>  
>       if (!vdev->pci_ops || !vdev->pci_ops->get_dmabuf_phys)
> @@ -243,13 +253,22 @@ int vfio_pci_core_feature_dma_buf(struct 
> vfio_pci_core_device *vdev, u32 flags,
>       if (!get_dma_buf.nr_ranges || get_dma_buf.flags)
>               return -EINVAL;
>  
> +     index = get_dma_buf.region_index;
> +
>       /*
> -      * For PCI the region_index is the BAR number like everything
> -      * else.  Check that PCI resources have been claimed for it.
> +      * A fixed region index is the BAR number; only a BAR can be exported
> +      * and its PCI resource must be claimed. A device-specific region (index
> +      * >= VFIO_PCI_NUM_REGIONS) has no BAR resource and is validated by the
> +      * device's get_dmabuf_phys instead, but the index must name a region
> +      * that exists.
>        */
> -     if (get_dma_buf.region_index >= VFIO_PCI_ROM_REGION_INDEX ||
> -         IS_ERR(vfio_pci_core_get_iomap(vdev, get_dma_buf.region_index)))
> +     if (index < VFIO_PCI_NUM_REGIONS) {
> +             if (index >= VFIO_PCI_ROM_REGION_INDEX ||
> +                 IS_ERR(vfio_pci_core_get_iomap(vdev, index)))
> +                     return -ENODEV;
> +     } else if (index - VFIO_PCI_NUM_REGIONS >= vdev->num_regions) {
>               return -ENODEV;
> +     }
>  
>       dma_ranges = memdup_array_user(&arg->dma_ranges, get_dma_buf.nr_ranges,
>                                      sizeof(*dma_ranges));
> -- 
> 2.25.1
> 
> 

Reply via email to