This helper, vfio_pci_core_mmap_prep_dmabuf(), creates a single-range
DMABUF for the purpose of mapping a PCI BAR.  This is used in a future
commit by VFIO's ordinary mmap() path.

This function transfers ownership of the VFIO device fd to the
DMABUF, which fput()s when it's released.

Refactor the existing vfio_pci_core_feature_dma_buf() to split out
export code common to the two paths, VFIO_DEVICE_FEATURE_DMA_BUF and
this new VFIO_BAR mmap().

By exchanging the VMA file, we lose the original device path in
/proc/<pid>/maps, lsof, etc.  Generate a debug-oriented synthetic
'filename' for BAR mappings based on the cdev, plus BDF, plus resource
index.  (This does not apply to explicitly-exported DMABUFs which are
named by DMA_BUF_SET_NAME.)

Signed-off-by: Matt Evans <[email protected]>
---
 drivers/vfio/pci/vfio_pci_dmabuf.c | 211 +++++++++++++++++++++++------
 drivers/vfio/pci/vfio_pci_priv.h   |   5 +
 2 files changed, 171 insertions(+), 45 deletions(-)

diff --git a/drivers/vfio/pci/vfio_pci_dmabuf.c 
b/drivers/vfio/pci/vfio_pci_dmabuf.c
index 9f10b10fc436..faa9239e66f8 100644
--- a/drivers/vfio/pci/vfio_pci_dmabuf.c
+++ b/drivers/vfio/pci/vfio_pci_dmabuf.c
@@ -3,6 +3,7 @@
  */
 #include <linux/dma-buf-mapping.h>
 #include <linux/pci-p2pdma.h>
+#include <linux/dma-buf.h>
 #include <linux/dma-resv.h>
 
 #include "vfio_pci_priv.h"
@@ -82,6 +83,8 @@ static void vfio_pci_dma_buf_release(struct dma_buf *dmabuf)
                up_write(&priv->vdev->dmabuf_lock);
                vfio_device_put_registration(&priv->vdev->vdev);
        }
+       if (priv->vfile)
+               fput(priv->vfile);
        kfree(priv->phys_vec);
        kfree(priv);
 }
@@ -246,6 +249,167 @@ int vfio_pci_dma_buf_find_pfn(struct vfio_pci_core_device 
*vdev,
        return ret;
 }
 
+/*
+ * Create a DMABUF corresponding to priv, add it to vdev->dmabufs list
+ * for tracking (meaning cleanup or revocation will zap it), and take
+ * a vfio_device registration.
+ */
+static int vfio_pci_dmabuf_export(struct vfio_pci_core_device *vdev,
+                                 struct vfio_pci_dma_buf *priv, u32 flags)
+{
+       DEFINE_DMA_BUF_EXPORT_INFO(exp_info);
+
+       if (!vfio_device_try_get_registration(&vdev->vdev))
+               return -ENODEV;
+
+       exp_info.ops = &vfio_pci_dmabuf_ops;
+       exp_info.size = priv->size;
+       exp_info.flags = flags;
+       exp_info.priv = priv;
+
+       priv->dmabuf = dma_buf_export(&exp_info);
+       if (IS_ERR(priv->dmabuf)) {
+               vfio_device_put_registration(&vdev->vdev);
+               return PTR_ERR(priv->dmabuf);
+       }
+
+       kref_init(&priv->kref);
+       init_completion(&priv->comp);
+
+       /* dma_buf_put() now frees priv */
+       INIT_LIST_HEAD(&priv->dmabufs_elm);
+
+       /*
+        * dmabuf_lock synchronises access (R) or updates (W) to the
+        * vdev->dmabufs list and to bars_revoked (see below).  The
+        * revocation state of DMABUF elements in the list is written
+        * holding both dmabuf_lock(W) and resv, and tested with
+        * either.
+        *
+        * (memory_lock, if held ->) dmabuf_lock -> resv
+        *
+        * NOTE: memory_lock is strictly avoided here, to avoid a
+        * dependency on memory_lock when mmap_lock is held, when
+        * mmap() leads to export.  vfio-pci variant drivers are
+        * permitted to hold memory_lock across actions that might
+        * fault (such as user access); a deadlock could result when
+        * that fault path attempts to take mmap_lock (if held by an
+        * export waiting for memory_lock).
+        *
+        * vdev->bars_revoked tracks the BAR revocation status updated
+        * via vfio_pci_dma_buf_move(), so the initial DMABUF state
+        * follows the same criteria that later update the DMABUF
+        * state (BAR zap, etc.).
+        */
+       lockdep_assert_not_held(&vdev->memory_lock);
+
+       down_write(&vdev->dmabuf_lock);
+       dma_resv_lock(priv->dmabuf->resv, NULL);
+       priv->revoked = vdev->bars_revoked;
+       list_add_tail(&priv->dmabufs_elm, &vdev->dmabufs);
+       dma_resv_unlock(priv->dmabuf->resv);
+       up_write(&vdev->dmabuf_lock);
+
+       return 0;
+}
+
+int vfio_pci_core_mmap_prep_dmabuf(struct vfio_pci_core_device *vdev,
+                                  struct vm_area_struct *vma,
+                                  u64 phys_start, u64 req_len,
+                                  unsigned int res_index)
+{
+       struct vfio_pci_dma_buf *priv;
+       unsigned long vma_pgoff = vma->vm_pgoff & (VFIO_PCI_OFFSET_MASK >> 
PAGE_SHIFT);
+       char *bufname;
+       int ret;
+
+       priv = kzalloc_obj(*priv);
+       if (!priv)
+               return -ENOMEM;
+
+       priv->phys_vec = kzalloc_obj(*priv->phys_vec);
+       if (!priv->phys_vec) {
+               ret = -ENOMEM;
+               goto err_free_priv;
+       }
+
+       /*
+        * Maximum size of the friendly debug name is
+        * vfio1048575:ffff:ff:1f.7/5 = 26.  This fits within
+        * DMA_BUF_NAME_LEN, so dma_buf_set_name() below won't fail.
+        */
+       bufname = kasprintf(GFP_KERNEL, "%s:%s/%x",
+                           dev_name(&vdev->vdev.device), pci_name(vdev->pdev),
+                           res_index);
+
+       if (!bufname) {
+               ret = -ENOMEM;
+               goto err_free_phys;
+       }
+
+       /*
+        * The DMABUF begins from the mmap()'s BAR offset, i.e. the
+        * start of the VMA corresponds to byte 0 of the DMABUF and
+        * byte (vma_pgoff << PAGE_SHIFT) of the BAR.
+        *
+        * vfio_pci_dma_buf_find_pfn() reverses this offset using
+        * vma_pgoff_adjust, so that ultimately a fault's offset from
+        * the start of the _VMA_ has a consistent usage whether the
+        * VMA originates from an mmap() of the VFIO device here or a
+        * direct DMABUF mmap().  Note vma_pgoff_adjust also includes
+        * the encoded VFIO region index, which cancels out the index
+        * encoded in vm_pgoff.
+        */
+       priv->vdev = vdev;
+       priv->size = req_len;
+       priv->nr_ranges = 1;
+       priv->vma_pgoff_adjust = vma->vm_pgoff;
+
+       priv->provider = pcim_p2pdma_provider(vdev->pdev, res_index);
+       if (!priv->provider) {
+               ret = -EINVAL;
+               goto err_free_name;
+       }
+
+       priv->phys_vec[0].paddr = phys_start + ((u64)vma_pgoff << PAGE_SHIFT);
+       priv->phys_vec[0].len = priv->size;
+
+       ret = vfio_pci_dmabuf_export(vdev, priv, O_RDWR);
+       if (ret)
+               goto err_free_name;
+
+       if (dma_buf_set_name(priv->dmabuf, bufname)) {
+               /* Shouldn't happen, but don't leak if it does: */
+               dev_dbg_ratelimited(&vdev->pdev->dev,
+                                   "Failed to set map name '%s'\n",
+                                   bufname);
+               kfree(bufname);
+       }
+
+       /*
+        * Ownership of the DMABUF file transfers to the VMA so that
+        * other users can locate the DMABUF via a VA.  Ownership of
+        * the original VFIO device file being mmap()ed transfers to
+        * priv, and is put when the DMABUF is released.  This
+        * intentionally does not use get_file()/vma_set_file()
+        * because the references are already held, and ownership
+        * moves.
+        */
+       priv->vfile = vma->vm_file;
+       vma->vm_file = priv->dmabuf->file;
+       vma->vm_private_data = priv;
+
+       return 0;
+
+err_free_name:
+       kfree(bufname);
+err_free_phys:
+       kfree(priv->phys_vec);
+err_free_priv:
+       kfree(priv);
+       return ret;
+}
+
 /*
  * This is a temporary "private interconnect" between VFIO DMABUF and iommufd.
  * It allows the two co-operating drivers to exchange the physical address of
@@ -364,7 +528,6 @@ int vfio_pci_core_feature_dma_buf(struct 
vfio_pci_core_device *vdev, u32 flags,
 {
        struct vfio_device_feature_dma_buf get_dma_buf = {};
        struct vfio_region_dma_range *dma_ranges;
-       DEFINE_DMA_BUF_EXPORT_INFO(exp_info);
        struct vfio_pci_dma_buf *priv;
        size_t length;
        int ret;
@@ -424,49 +587,9 @@ int vfio_pci_core_feature_dma_buf(struct 
vfio_pci_core_device *vdev, u32 flags,
        kfree(dma_ranges);
        dma_ranges = NULL;
 
-       if (!vfio_device_try_get_registration(&vdev->vdev)) {
-               ret = -ENODEV;
+       ret = vfio_pci_dmabuf_export(vdev, priv, get_dma_buf.open_flags);
+       if (ret)
                goto err_free_phys;
-       }
-
-       exp_info.ops = &vfio_pci_dmabuf_ops;
-       exp_info.size = priv->size;
-       exp_info.flags = get_dma_buf.open_flags;
-       exp_info.priv = priv;
-
-       priv->dmabuf = dma_buf_export(&exp_info);
-       if (IS_ERR(priv->dmabuf)) {
-               ret = PTR_ERR(priv->dmabuf);
-               goto err_dev_put;
-       }
-
-       kref_init(&priv->kref);
-       init_completion(&priv->comp);
-
-       /* dma_buf_put() now frees priv */
-       INIT_LIST_HEAD(&priv->dmabufs_elm);
-
-       /*
-        * dmabuf_lock synchronises access (R) or updates (W) to the
-        * vdev->dmabufs list and to bars_revoked (see below).  The
-        * revocation state of DMABUF elements in the list is written
-        * holding both dmabuf_lock(W) and resv, and tested with
-        * either.
-        *
-        * dmabuf_lock -> resv
-        *
-        * vdev->bars_revoked tracks the BAR revocation status updated
-        * via vfio_pci_dma_buf_move(), so the initial DMABUF state
-        * follows the same criteria that later update the DMABUF
-        * state (BAR zap, etc.).
-        */
-       down_write(&vdev->dmabuf_lock);
-       dma_resv_lock(priv->dmabuf->resv, NULL);
-       priv->revoked = vdev->bars_revoked;
-       list_add_tail(&priv->dmabufs_elm, &vdev->dmabufs);
-       dma_resv_unlock(priv->dmabuf->resv);
-       up_write(&vdev->dmabuf_lock);
-
        /*
         * dma_buf_fd() consumes the reference, when the file closes the dmabuf
         * will be released.
@@ -477,8 +600,6 @@ int vfio_pci_core_feature_dma_buf(struct 
vfio_pci_core_device *vdev, u32 flags,
 
        return ret;
 
-err_dev_put:
-       vfio_device_put_registration(&vdev->vdev);
 err_free_phys:
        kfree(priv->phys_vec);
 err_free_priv:
diff --git a/drivers/vfio/pci/vfio_pci_priv.h b/drivers/vfio/pci/vfio_pci_priv.h
index 48d9f574a3df..3ec676e12e21 100644
--- a/drivers/vfio/pci/vfio_pci_priv.h
+++ b/drivers/vfio/pci/vfio_pci_priv.h
@@ -30,6 +30,7 @@ struct vfio_pci_dma_buf {
        size_t size;
        struct phys_vec *phys_vec;
        struct p2pdma_provider *provider;
+       struct file *vfile;
        u32 nr_ranges;
        struct kref kref;
        struct completion comp;
@@ -134,6 +135,10 @@ int vfio_pci_dma_buf_find_pfn(struct vfio_pci_core_device 
*vdev,
                              unsigned long address,
                              unsigned int order,
                              unsigned long *out_pfn);
+int vfio_pci_core_mmap_prep_dmabuf(struct vfio_pci_core_device *vdev,
+                                  struct vm_area_struct *vma,
+                                  u64 phys_start, u64 req_len,
+                                  unsigned int res_index);
 
 #ifdef CONFIG_VFIO_PCI_DMABUF
 int vfio_pci_core_feature_dma_buf(struct vfio_pci_core_device *vdev, u32 flags,
-- 
2.50.1 (Apple Git-155)

Reply via email to