This helper, vfio_pci_core_mmap_prep_dmabuf(), creates a single-range DMABUF for the purpose of mapping a PCI BAR. This is used in a future commit by VFIO's ordinary mmap() path.
This function transfers ownership of the VFIO device fd to the DMABUF, which fput()s when it's released.
Refactor the existing vfio_pci_core_feature_dma_buf() to split out export code common to the two paths, VFIO_DEVICE_FEATURE_DMA_BUF and this new VFIO_BAR mmap().
By exchanging the VMA file, we lose the original device path in /proc/<pid>/maps, lsof, etc. Generate a debug-oriented synthetic 'filename' for BAR mappings based on the cdev, plus BDF, plus resource index. (This does not apply to explicitly-exported DMABUFs which are named by DMA_BUF_SET_NAME.)
Signed-off-by: Matt Evans matt@ozlabs.org --- drivers/vfio/pci/vfio_pci_dmabuf.c | 211 +++++++++++++++++++++++------ drivers/vfio/pci/vfio_pci_priv.h | 5 + 2 files changed, 171 insertions(+), 45 deletions(-)
diff --git a/drivers/vfio/pci/vfio_pci_dmabuf.c b/drivers/vfio/pci/vfio_pci_dmabuf.c index 9f10b10fc436..faa9239e66f8 100644 --- a/drivers/vfio/pci/vfio_pci_dmabuf.c +++ b/drivers/vfio/pci/vfio_pci_dmabuf.c @@ -3,6 +3,7 @@ */ #include <linux/dma-buf-mapping.h> #include <linux/pci-p2pdma.h> +#include <linux/dma-buf.h> #include <linux/dma-resv.h>
#include "vfio_pci_priv.h" @@ -82,6 +83,8 @@ static void vfio_pci_dma_buf_release(struct dma_buf *dmabuf) up_write(&priv->vdev->dmabuf_lock); vfio_device_put_registration(&priv->vdev->vdev); } + if (priv->vfile) + fput(priv->vfile); kfree(priv->phys_vec); kfree(priv); } @@ -246,6 +249,167 @@ int vfio_pci_dma_buf_find_pfn(struct vfio_pci_core_device *vdev, return ret; }
+/* + * Create a DMABUF corresponding to priv, add it to vdev->dmabufs list + * for tracking (meaning cleanup or revocation will zap it), and take + * a vfio_device registration. + */ +static int vfio_pci_dmabuf_export(struct vfio_pci_core_device *vdev, + struct vfio_pci_dma_buf *priv, u32 flags) +{ + DEFINE_DMA_BUF_EXPORT_INFO(exp_info); + + if (!vfio_device_try_get_registration(&vdev->vdev)) + return -ENODEV; + + exp_info.ops = &vfio_pci_dmabuf_ops; + exp_info.size = priv->size; + exp_info.flags = flags; + exp_info.priv = priv; + + priv->dmabuf = dma_buf_export(&exp_info); + if (IS_ERR(priv->dmabuf)) { + vfio_device_put_registration(&vdev->vdev); + return PTR_ERR(priv->dmabuf); + } + + kref_init(&priv->kref); + init_completion(&priv->comp); + + /* dma_buf_put() now frees priv */ + INIT_LIST_HEAD(&priv->dmabufs_elm); + + /* + * dmabuf_lock synchronises access (R) or updates (W) to the + * vdev->dmabufs list and to bars_revoked (see below). The + * revocation state of DMABUF elements in the list is written + * holding both dmabuf_lock(W) and resv, and tested with + * either. + * + * (memory_lock, if held ->) dmabuf_lock -> resv + * + * NOTE: memory_lock is strictly avoided here, to avoid a + * dependency on memory_lock when mmap_lock is held, when + * mmap() leads to export. vfio-pci variant drivers are + * permitted to hold memory_lock across actions that might + * fault (such as user access); a deadlock could result when + * that fault path attempts to take mmap_lock (if held by an + * export waiting for memory_lock). + * + * vdev->bars_revoked tracks the BAR revocation status updated + * via vfio_pci_dma_buf_move(), so the initial DMABUF state + * follows the same criteria that later update the DMABUF + * state (BAR zap, etc.). + */ + lockdep_assert_not_held(&vdev->memory_lock); + + down_write(&vdev->dmabuf_lock); + dma_resv_lock(priv->dmabuf->resv, NULL); + priv->revoked = vdev->bars_revoked; + list_add_tail(&priv->dmabufs_elm, &vdev->dmabufs); + dma_resv_unlock(priv->dmabuf->resv); + up_write(&vdev->dmabuf_lock); + + return 0; +} + +int vfio_pci_core_mmap_prep_dmabuf(struct vfio_pci_core_device *vdev, + struct vm_area_struct *vma, + u64 phys_start, u64 req_len, + unsigned int res_index) +{ + struct vfio_pci_dma_buf *priv; + unsigned long vma_pgoff = vma->vm_pgoff & (VFIO_PCI_OFFSET_MASK >> PAGE_SHIFT); + char *bufname; + int ret; + + priv = kzalloc_obj(*priv); + if (!priv) + return -ENOMEM; + + priv->phys_vec = kzalloc_obj(*priv->phys_vec); + if (!priv->phys_vec) { + ret = -ENOMEM; + goto err_free_priv; + } + + /* + * Maximum size of the friendly debug name is + * vfio1048575:ffff:ff:1f.7/5 = 26. This fits within + * DMA_BUF_NAME_LEN, so dma_buf_set_name() below won't fail. + */ + bufname = kasprintf(GFP_KERNEL, "%s:%s/%x", + dev_name(&vdev->vdev.device), pci_name(vdev->pdev), + res_index); + + if (!bufname) { + ret = -ENOMEM; + goto err_free_phys; + } + + /* + * The DMABUF begins from the mmap()'s BAR offset, i.e. the + * start of the VMA corresponds to byte 0 of the DMABUF and + * byte (vma_pgoff << PAGE_SHIFT) of the BAR. + * + * vfio_pci_dma_buf_find_pfn() reverses this offset using + * vma_pgoff_adjust, so that ultimately a fault's offset from + * the start of the _VMA_ has a consistent usage whether the + * VMA originates from an mmap() of the VFIO device here or a + * direct DMABUF mmap(). Note vma_pgoff_adjust also includes + * the encoded VFIO region index, which cancels out the index + * encoded in vm_pgoff. + */ + priv->vdev = vdev; + priv->size = req_len; + priv->nr_ranges = 1; + priv->vma_pgoff_adjust = vma->vm_pgoff; + + priv->provider = pcim_p2pdma_provider(vdev->pdev, res_index); + if (!priv->provider) { + ret = -EINVAL; + goto err_free_name; + } + + priv->phys_vec[0].paddr = phys_start + ((u64)vma_pgoff << PAGE_SHIFT); + priv->phys_vec[0].len = priv->size; + + ret = vfio_pci_dmabuf_export(vdev, priv, O_RDWR); + if (ret) + goto err_free_name; + + if (dma_buf_set_name(priv->dmabuf, bufname)) { + /* Shouldn't happen, but don't leak if it does: */ + dev_dbg_ratelimited(&vdev->pdev->dev, + "Failed to set map name '%s'\n", + bufname); + kfree(bufname); + } + + /* + * Ownership of the DMABUF file transfers to the VMA so that + * other users can locate the DMABUF via a VA. Ownership of + * the original VFIO device file being mmap()ed transfers to + * priv, and is put when the DMABUF is released. This + * intentionally does not use get_file()/vma_set_file() + * because the references are already held, and ownership + * moves. + */ + priv->vfile = vma->vm_file; + vma->vm_file = priv->dmabuf->file; + vma->vm_private_data = priv; + + return 0; + +err_free_name: + kfree(bufname); +err_free_phys: + kfree(priv->phys_vec); +err_free_priv: + kfree(priv); + return ret; +} + /* * This is a temporary "private interconnect" between VFIO DMABUF and iommufd. * It allows the two co-operating drivers to exchange the physical address of @@ -364,7 +528,6 @@ int vfio_pci_core_feature_dma_buf(struct vfio_pci_core_device *vdev, u32 flags, { struct vfio_device_feature_dma_buf get_dma_buf = {}; struct vfio_region_dma_range *dma_ranges; - DEFINE_DMA_BUF_EXPORT_INFO(exp_info); struct vfio_pci_dma_buf *priv; size_t length; int ret; @@ -424,49 +587,9 @@ int vfio_pci_core_feature_dma_buf(struct vfio_pci_core_device *vdev, u32 flags, kfree(dma_ranges); dma_ranges = NULL;
- if (!vfio_device_try_get_registration(&vdev->vdev)) { - ret = -ENODEV; + ret = vfio_pci_dmabuf_export(vdev, priv, get_dma_buf.open_flags); + if (ret) goto err_free_phys; - } - - exp_info.ops = &vfio_pci_dmabuf_ops; - exp_info.size = priv->size; - exp_info.flags = get_dma_buf.open_flags; - exp_info.priv = priv; - - priv->dmabuf = dma_buf_export(&exp_info); - if (IS_ERR(priv->dmabuf)) { - ret = PTR_ERR(priv->dmabuf); - goto err_dev_put; - } - - kref_init(&priv->kref); - init_completion(&priv->comp); - - /* dma_buf_put() now frees priv */ - INIT_LIST_HEAD(&priv->dmabufs_elm); - - /* - * dmabuf_lock synchronises access (R) or updates (W) to the - * vdev->dmabufs list and to bars_revoked (see below). The - * revocation state of DMABUF elements in the list is written - * holding both dmabuf_lock(W) and resv, and tested with - * either. - * - * dmabuf_lock -> resv - * - * vdev->bars_revoked tracks the BAR revocation status updated - * via vfio_pci_dma_buf_move(), so the initial DMABUF state - * follows the same criteria that later update the DMABUF - * state (BAR zap, etc.). - */ - down_write(&vdev->dmabuf_lock); - dma_resv_lock(priv->dmabuf->resv, NULL); - priv->revoked = vdev->bars_revoked; - list_add_tail(&priv->dmabufs_elm, &vdev->dmabufs); - dma_resv_unlock(priv->dmabuf->resv); - up_write(&vdev->dmabuf_lock); - /* * dma_buf_fd() consumes the reference, when the file closes the dmabuf * will be released. @@ -477,8 +600,6 @@ int vfio_pci_core_feature_dma_buf(struct vfio_pci_core_device *vdev, u32 flags,
return ret;
-err_dev_put: - vfio_device_put_registration(&vdev->vdev); err_free_phys: kfree(priv->phys_vec); err_free_priv: diff --git a/drivers/vfio/pci/vfio_pci_priv.h b/drivers/vfio/pci/vfio_pci_priv.h index 48d9f574a3df..3ec676e12e21 100644 --- a/drivers/vfio/pci/vfio_pci_priv.h +++ b/drivers/vfio/pci/vfio_pci_priv.h @@ -30,6 +30,7 @@ struct vfio_pci_dma_buf { size_t size; struct phys_vec *phys_vec; struct p2pdma_provider *provider; + struct file *vfile; u32 nr_ranges; struct kref kref; struct completion comp; @@ -134,6 +135,10 @@ int vfio_pci_dma_buf_find_pfn(struct vfio_pci_core_device *vdev, unsigned long address, unsigned int order, unsigned long *out_pfn); +int vfio_pci_core_mmap_prep_dmabuf(struct vfio_pci_core_device *vdev, + struct vm_area_struct *vma, + u64 phys_start, u64 req_len, + unsigned int res_index);
#ifdef CONFIG_VFIO_PCI_DMABUF int vfio_pci_core_feature_dma_buf(struct vfio_pci_core_device *vdev, u32 flags,