From: Manish Honap <[email protected]>

A Type-2 accelerator issues ATS-translated DMA to addresses inside its
own HDM window, so that coherent host range must be present in the
guest's IOAS (the iommufd IOAS backing the nested SMMU stage-2). iommufd
maps a struct-page-less range only by fd, via IOMMU_IOAS_MAP_FILE over a
dma-buf; a userspace-VA IOMMU_IOAS_MAP of the HDM mmap is rejected
because the VMA is VM_IO | VM_PFNMAP. Without a dma-buf the range could
only be mapped through an out-of-tree PFNMAP work-around.

vfio-pci already exports BAR memory as a P2P dma-buf, but the exporter
is BAR-only: vfio_pci_core_feature_dma_buf() rejects any region index at
or above the ROM index, and vfio_pci_core_get_dmabuf_phys() resolves the
physical range from a PCI BAR. The HDM memory region is a dynamic
device-specific region, not a BAR.

Let a device-specific region reach the device's get_dmabuf_phys(): a
region index at or above VFIO_PCI_NUM_REGIONS skips the BAR-resource
check and is validated by the driver instead, bounded to the regions
that exist. Install a CXL-aware get_dmabuf_phys() in the vfio-cxl
provider that returns cxl->hpa_range for the HDM memory region and
delegates real BARs to the core, keeping the BAR path unchanged and the
core free of CXL knowledge.

The HDM window is coherent host memory with no p2pdma provider of its
own, so borrow BAR 0's, matching nvgrace-gpu's handling of its non-BAR
device memory. The iommufd importer does not consume the provider; the
scatterlist map path (real peer DMA) is left to a follow-up once
upstream grows a negotiated interconnect for coherent CXL memory.

Assisted-by: LLM
Signed-off-by: Manish Honap <[email protected]>
---
 drivers/vfio/pci/cxl/vfio_cxl_core.c | 60 ++++++++++++++++++++++++++++
 drivers/vfio/pci/vfio_pci_dmabuf.c   | 27 +++++++++++--
 2 files changed, 83 insertions(+), 4 deletions(-)

diff --git a/drivers/vfio/pci/cxl/vfio_cxl_core.c 
b/drivers/vfio/pci/cxl/vfio_cxl_core.c
index 5fe8e35c63c4..55fa1f86850d 100644
--- a/drivers/vfio/pci/cxl/vfio_cxl_core.c
+++ b/drivers/vfio/pci/cxl/vfio_cxl_core.c
@@ -11,6 +11,7 @@
 #include <linux/mm.h>
 #include <linux/module.h>
 #include <linux/pci.h>
+#include <linux/pci-p2pdma.h>
 #include <linux/range.h>
 #include <linux/slab.h>
 #include <linux/uaccess.h>
@@ -27,6 +28,7 @@
  * @hdm_regs: mapped HDM decoder registers, read live by the decoder region
  * @hdm_len: length of the HDM decoder register block
  * @hdm_valid: true when host CPU access to the HDM range is safe; under 
memory_lock
+ * @mem_region_index: vfio region index of the mmap-able HDM memory region
  */
 struct vfio_cxl_state {
        struct cxl_dev_state cxlds;
@@ -36,6 +38,7 @@ struct vfio_cxl_state {
        void __iomem *hdm_regs;
        u32 hdm_len;
        bool hdm_valid;
+       unsigned int mem_region_index;
 };
 
 static unsigned long vfio_cxl_mem_pgoff(struct vm_area_struct *vma,
@@ -289,6 +292,50 @@ static void vfio_cxl_release_hpa(void *data)
        release_mem_region(cxl->hpa_range.start, range_len(&cxl->hpa_range));
 }
 
+/*
+ * Resolve the physical range that backs a dma-buf export. The core exporter
+ * only knows BARs; teach it the HDM memory region so a guest IOAS can map the
+ * coherent window by fd (IOMMU_IOAS_MAP_FILE) instead of the removed PFNMAP
+ * work-around. Real BARs stay on the byte-identical core path.
+ */
+static int vfio_cxl_get_dmabuf_phys(struct vfio_pci_core_device *vdev,
+                                   struct p2pdma_provider **provider,
+                                   unsigned int region_index,
+                                   struct phys_vec *phys_vec,
+                                   struct vfio_region_dma_range *dma_ranges,
+                                   size_t nr_ranges)
+{
+       struct vfio_cxl_state *cxl = vdev->cxl;
+
+       /* Real BARs go through the core P2P exporter unchanged. */
+       if (region_index < VFIO_PCI_NUM_REGIONS)
+               return vfio_pci_core_get_dmabuf_phys(vdev, provider,
+                                                    region_index, phys_vec,
+                                                    dma_ranges, nr_ranges);
+
+       /* Of the device regions, only the HDM memory window is exportable. */
+       if (region_index != cxl->mem_region_index)
+               return -EINVAL;
+
+       /*
+        * The HDM window is coherent host memory, not BAR MMIO, so it has no
+        * p2pdma provider of its own. Borrow BAR 0's: the P2P properties match
+        * and the iommufd importer does not consume the provider. The sgt map
+        * path (real peer DMA) is not supported for the HDM window.
+        */
+       *provider = pcim_p2pdma_provider(vdev->pdev, 0);
+       if (!*provider)
+               return -EINVAL;
+
+       return vfio_pci_core_fill_phys_vec(phys_vec, dma_ranges, nr_ranges,
+                                         cxl->hpa_range.start,
+                                         range_len(&cxl->hpa_range));
+}
+
+static const struct vfio_pci_device_ops vfio_cxl_pci_dev_ops = {
+       .get_dmabuf_phys = vfio_cxl_get_dmabuf_phys,
+};
+
 static int vfio_cxl_init_device(struct vfio_pci_core_device *vdev)
 {
        struct pci_dev *pdev = vdev->pdev;
@@ -483,6 +530,19 @@ static int vfio_cxl_open_device(struct 
vfio_pci_core_device *vdev)
        if (ret)
                return ret;
 
+       /* Record where the HDM memory region landed for the dma-buf export. */
+       cxl->mem_region_index = VFIO_PCI_NUM_REGIONS + vdev->num_regions - 1;
+
+       /*
+        * Override the device ops so a dma-buf export of the HDM memory region
+        * resolves to the coherent host range. This is done at open, not init:
+        * vfio_pci_probe() resets pci_ops after vfio_alloc_device() returns, so
+        * an override installed during init would be clobbered. Only a CXL 
device
+        * reaches this hook (cxl_ops is set on init success), so a fallback to
+        * plain vfio-pci keeps the core ops.
+        */
+       vdev->pci_ops = &vfio_cxl_pci_dev_ops;
+
        ret = vfio_cxl_add_region(vdev, VFIO_REGION_SUBTYPE_CXL_COMP_REGS,
                                  &vfio_cxl_comp_regops, cxl->hdm_len,
                                  VFIO_REGION_INFO_FLAG_READ |
diff --git a/drivers/vfio/pci/vfio_pci_dmabuf.c 
b/drivers/vfio/pci/vfio_pci_dmabuf.c
index c16f460c01d6..436c616d5b66 100644
--- a/drivers/vfio/pci/vfio_pci_dmabuf.c
+++ b/drivers/vfio/pci/vfio_pci_dmabuf.c
@@ -178,6 +178,15 @@ int vfio_pci_core_get_dmabuf_phys(struct 
vfio_pci_core_device *vdev,
 {
        struct pci_dev *pdev = vdev->pdev;
 
+       /*
+        * This resolver only handles PCI BARs. A device-specific region index
+        * (>= PCI_STD_NUM_BARS) would index pdev->resource[] out of bounds via
+        * pcim_p2pdma_provider(), so reject it; a driver that exports such a
+        * region installs its own get_dmabuf_phys.
+        */
+       if (region_index >= PCI_STD_NUM_BARS)
+               return -EINVAL;
+
        *provider = pcim_p2pdma_provider(pdev, region_index);
        if (!*provider)
                return -EINVAL;
@@ -227,6 +236,7 @@ int vfio_pci_core_feature_dma_buf(struct 
vfio_pci_core_device *vdev, u32 flags,
        DEFINE_DMA_BUF_EXPORT_INFO(exp_info);
        struct vfio_pci_dma_buf *priv;
        size_t length;
+       u32 index;
        int ret;
 
        if (!vdev->pci_ops || !vdev->pci_ops->get_dmabuf_phys)
@@ -243,13 +253,22 @@ int vfio_pci_core_feature_dma_buf(struct 
vfio_pci_core_device *vdev, u32 flags,
        if (!get_dma_buf.nr_ranges || get_dma_buf.flags)
                return -EINVAL;
 
+       index = get_dma_buf.region_index;
+
        /*
-        * For PCI the region_index is the BAR number like everything
-        * else.  Check that PCI resources have been claimed for it.
+        * A fixed region index is the BAR number; only a BAR can be exported
+        * and its PCI resource must be claimed. A device-specific region (index
+        * >= VFIO_PCI_NUM_REGIONS) has no BAR resource and is validated by the
+        * device's get_dmabuf_phys instead, but the index must name a region
+        * that exists.
         */
-       if (get_dma_buf.region_index >= VFIO_PCI_ROM_REGION_INDEX ||
-           IS_ERR(vfio_pci_core_get_iomap(vdev, get_dma_buf.region_index)))
+       if (index < VFIO_PCI_NUM_REGIONS) {
+               if (index >= VFIO_PCI_ROM_REGION_INDEX ||
+                   IS_ERR(vfio_pci_core_get_iomap(vdev, index)))
+                       return -ENODEV;
+       } else if (index - VFIO_PCI_NUM_REGIONS >= vdev->num_regions) {
                return -ENODEV;
+       }
 
        dma_ranges = memdup_array_user(&arg->dma_ranges, get_dma_buf.nr_ranges,
                                       sizeof(*dma_ranges));
-- 
2.25.1


Reply via email to