> -----Original Message-----
> From: Richard Cheng <[email protected]>
> Sent: Thursday, September 17, 2026 1:25 PM
> To: Manish Honap <[email protected]>
> Cc: [email protected]; [email protected]; Ankit Agrawal <[email protected]>;
> [email protected]; [email protected]; [email protected];
> Srirangan Madhavan <[email protected]>; [email protected];
> [email protected]; [email protected]; [email protected];
> [email protected]; [email protected]; [email protected]; Yishai
> Hadas <[email protected]>; Shameer Kolothum Thodi
> <[email protected]>; [email protected]; [email protected];
> [email protected]; [email protected]; [email protected]; Neo Jia
> <[email protected]>; Krishnakant Jaju <[email protected]>; Vikram Sethi
> <[email protected]>; Zhi Wang <[email protected]>; linux-
> [email protected]; [email protected]; [email protected];
> [email protected]; [email protected]; linux-
> [email protected]; [email protected]
> Subject: Re: [PATCH v5 24/27] vfio/cxl: Export the HDM memory region as a
> dma-buf
> 
> On Thu, Sep 17, 2026 at 12:05:37AM +0800, [email protected] wrote:
> > From: Manish Honap <[email protected]>
> >
> > A Type-2 accelerator issues ATS-translated DMA to addresses inside its
> > own HDM window, so that coherent host range must be present in the
> > guest's IOAS (the iommufd IOAS backing the nested SMMU stage-2).
> > iommufd maps a struct-page-less range only by fd, via
> > IOMMU_IOAS_MAP_FILE over a dma-buf; a userspace-VA
> IOMMU_IOAS_MAP of
> > the HDM mmap is rejected because the VMA is VM_IO | VM_PFNMAP.
> Without
> > a dma-buf the range could only be mapped through an out-of-tree PFNMAP
> work-around.
> >
> > vfio-pci already exports BAR memory as a P2P dma-buf, but the exporter
> > is BAR-only: vfio_pci_core_feature_dma_buf() rejects any region index
> > at or above the ROM index, and vfio_pci_core_get_dmabuf_phys()
> > resolves the physical range from a PCI BAR. The HDM memory region is a
> > dynamic device-specific region, not a BAR.
> >
> > Let a device-specific region reach the device's get_dmabuf_phys(): a
> > region index at or above VFIO_PCI_NUM_REGIONS skips the BAR-resource
> > check and is validated by the driver instead, bounded to the regions
> > that exist. Install a CXL-aware get_dmabuf_phys() in the vfio-cxl
> > provider that returns cxl->hpa_range for the HDM memory region and
> > delegates real BARs to the core, keeping the BAR path unchanged and
> > the core free of CXL knowledge.
> >
> > The HDM window is coherent host memory with no p2pdma provider of its
> > own, so borrow BAR 0's, matching nvgrace-gpu's handling of its non-BAR
> > device memory. The iommufd importer does not consume the provider; the
> > scatterlist map path (real peer DMA) is left to a follow-up once
> > upstream grows a negotiated interconnect for coherent CXL memory.
> >
> > Assisted-by: LLM
> > Signed-off-by: Manish Honap <[email protected]>
> > ---
> >  drivers/vfio/pci/cxl/vfio_cxl_core.c | 60
> ++++++++++++++++++++++++++++
> >  drivers/vfio/pci/vfio_pci_dmabuf.c   | 27 +++++++++++--
> >  2 files changed, 83 insertions(+), 4 deletions(-)
> >
> > diff --git a/drivers/vfio/pci/cxl/vfio_cxl_core.c
> > b/drivers/vfio/pci/cxl/vfio_cxl_core.c
> > index 5fe8e35c63c4..55fa1f86850d 100644
> > --- a/drivers/vfio/pci/cxl/vfio_cxl_core.c
> > +++ b/drivers/vfio/pci/cxl/vfio_cxl_core.c
> > @@ -11,6 +11,7 @@
> >  #include <linux/mm.h>
> >  #include <linux/module.h>
> >  #include <linux/pci.h>
> > +#include <linux/pci-p2pdma.h>
> >  #include <linux/range.h>
> >  #include <linux/slab.h>
> >  #include <linux/uaccess.h>
> > @@ -27,6 +28,7 @@
> >   * @hdm_regs: mapped HDM decoder registers, read live by the decoder
> region
> >   * @hdm_len: length of the HDM decoder register block
> >   * @hdm_valid: true when host CPU access to the HDM range is safe;
> > under memory_lock
> > + * @mem_region_index: vfio region index of the mmap-able HDM memory
> > + region
> >   */
> >  struct vfio_cxl_state {
> >     struct cxl_dev_state cxlds;
> > @@ -36,6 +38,7 @@ struct vfio_cxl_state {
> >     void __iomem *hdm_regs;
> >     u32 hdm_len;
> >     bool hdm_valid;
> > +   unsigned int mem_region_index;
> >  };
> >
> >  static unsigned long vfio_cxl_mem_pgoff(struct vm_area_struct *vma,
> > @@ -289,6 +292,50 @@ static void vfio_cxl_release_hpa(void *data)
> >     release_mem_region(cxl->hpa_range.start,
> > range_len(&cxl->hpa_range));  }
> >
> > +/*
> > + * Resolve the physical range that backs a dma-buf export. The core
> > +exporter
> > + * only knows BARs; teach it the HDM memory region so a guest IOAS
> > +can map the
> > + * coherent window by fd (IOMMU_IOAS_MAP_FILE) instead of the
> removed
> > +PFNMAP
> > + * work-around. Real BARs stay on the byte-identical core path.
> > + */
> > +static int vfio_cxl_get_dmabuf_phys(struct vfio_pci_core_device *vdev,
> > +                               struct p2pdma_provider **provider,
> > +                               unsigned int region_index,
> > +                               struct phys_vec *phys_vec,
> > +                               struct vfio_region_dma_range
> *dma_ranges,
> > +                               size_t nr_ranges)
> > +{
> > +   struct vfio_cxl_state *cxl = vdev->cxl;
> > +
> > +   /* Real BARs go through the core P2P exporter unchanged. */
> > +   if (region_index < VFIO_PCI_NUM_REGIONS)
> > +           return vfio_pci_core_get_dmabuf_phys(vdev, provider,
> > +                                                region_index, phys_vec,
> > +                                                dma_ranges, nr_ranges);
> > +
> > +   /* Of the device regions, only the HDM memory window is exportable.
> */
> > +   if (region_index != cxl->mem_region_index)
> > +           return -EINVAL;
> > +
> > +   /*
> > +    * The HDM window is coherent host memory, not BAR MMIO, so it
> has no
> > +    * p2pdma provider of its own. Borrow BAR 0's: the P2P properties
> match
> > +    * and the iommufd importer does not consume the provider. The sgt
> map
> > +    * path (real peer DMA) is not supported for the HDM window.
> > +    */
> > +   *provider = pcim_p2pdma_provider(vdev->pdev, 0);
> 
> I think of a weird scenario which might make peer DMA work, not sure if that's
> possible.
> 
> userspace can give this dma-buf fd to an NIC driver or so to do RDMA ? use
> HDM memory as network buffer ?
> 
> If that's the case and direct P2P addressing is selected, it applies BAR 0's 
> bus
> offset to HDM physical address.
> 
> Maybe explicitly reject peer-DMA mapping would be safer ?

Yes, I think this will require handling.

For v6, I will tag the HDM export with no_peer2peer and vfio_pci_dma_buf_map()
return -EOPNOTSUPP for it, so a peer-DMA mapping is refused cleanly instead of 
being
handed a wrong address.

Thanks,
Manish

> 
> Best regards,
> Richard Cheng.
> 
> 
> > +   if (!*provider)
> > +           return -EINVAL;
> > +
> > +   return vfio_pci_core_fill_phys_vec(phys_vec, dma_ranges, nr_ranges,
> > +                                     cxl->hpa_range.start,
> > +                                     range_len(&cxl->hpa_range));
> > +}
> > +
> > +static const struct vfio_pci_device_ops vfio_cxl_pci_dev_ops = {
> > +   .get_dmabuf_phys = vfio_cxl_get_dmabuf_phys, };
> > +
> >  static int vfio_cxl_init_device(struct vfio_pci_core_device *vdev)  {
> >     struct pci_dev *pdev = vdev->pdev;
> > @@ -483,6 +530,19 @@ static int vfio_cxl_open_device(struct
> vfio_pci_core_device *vdev)
> >     if (ret)
> >             return ret;
> >
> > +   /* Record where the HDM memory region landed for the dma-buf
> export. */
> > +   cxl->mem_region_index = VFIO_PCI_NUM_REGIONS + vdev-
> >num_regions -
> > +1;
> > +
> > +   /*
> > +    * Override the device ops so a dma-buf export of the HDM memory
> region
> > +    * resolves to the coherent host range. This is done at open, not init:
> > +    * vfio_pci_probe() resets pci_ops after vfio_alloc_device() returns, so
> > +    * an override installed during init would be clobbered. Only a CXL
> device
> > +    * reaches this hook (cxl_ops is set on init success), so a fallback to
> > +    * plain vfio-pci keeps the core ops.
> > +    */
> > +   vdev->pci_ops = &vfio_cxl_pci_dev_ops;
> > +
> >     ret = vfio_cxl_add_region(vdev,
> VFIO_REGION_SUBTYPE_CXL_COMP_REGS,
> >                               &vfio_cxl_comp_regops, cxl->hdm_len,
> >                               VFIO_REGION_INFO_FLAG_READ |
> > diff --git a/drivers/vfio/pci/vfio_pci_dmabuf.c
> > b/drivers/vfio/pci/vfio_pci_dmabuf.c
> > index c16f460c01d6..436c616d5b66 100644
> > --- a/drivers/vfio/pci/vfio_pci_dmabuf.c
> > +++ b/drivers/vfio/pci/vfio_pci_dmabuf.c
> > @@ -178,6 +178,15 @@ int vfio_pci_core_get_dmabuf_phys(struct
> > vfio_pci_core_device *vdev,  {
> >     struct pci_dev *pdev = vdev->pdev;
> >
> > +   /*
> > +    * This resolver only handles PCI BARs. A device-specific region index
> > +    * (>= PCI_STD_NUM_BARS) would index pdev->resource[] out of
> bounds via
> > +    * pcim_p2pdma_provider(), so reject it; a driver that exports such a
> > +    * region installs its own get_dmabuf_phys.
> > +    */
> > +   if (region_index >= PCI_STD_NUM_BARS)
> > +           return -EINVAL;
> > +
> >     *provider = pcim_p2pdma_provider(pdev, region_index);
> >     if (!*provider)
> >             return -EINVAL;
> > @@ -227,6 +236,7 @@ int vfio_pci_core_feature_dma_buf(struct
> vfio_pci_core_device *vdev, u32 flags,
> >     DEFINE_DMA_BUF_EXPORT_INFO(exp_info);
> >     struct vfio_pci_dma_buf *priv;
> >     size_t length;
> > +   u32 index;
> >     int ret;
> >
> >     if (!vdev->pci_ops || !vdev->pci_ops->get_dmabuf_phys) @@ -
> 243,13
> > +253,22 @@ int vfio_pci_core_feature_dma_buf(struct
> vfio_pci_core_device *vdev, u32 flags,
> >     if (!get_dma_buf.nr_ranges || get_dma_buf.flags)
> >             return -EINVAL;
> >
> > +   index = get_dma_buf.region_index;
> > +
> >     /*
> > -    * For PCI the region_index is the BAR number like everything
> > -    * else.  Check that PCI resources have been claimed for it.
> > +    * A fixed region index is the BAR number; only a BAR can be exported
> > +    * and its PCI resource must be claimed. A device-specific region (index
> > +    * >= VFIO_PCI_NUM_REGIONS) has no BAR resource and is validated
> by the
> > +    * device's get_dmabuf_phys instead, but the index must name a
> region
> > +    * that exists.
> >      */
> > -   if (get_dma_buf.region_index >= VFIO_PCI_ROM_REGION_INDEX ||
> > -       IS_ERR(vfio_pci_core_get_iomap(vdev,
> get_dma_buf.region_index)))
> > +   if (index < VFIO_PCI_NUM_REGIONS) {
> > +           if (index >= VFIO_PCI_ROM_REGION_INDEX ||
> > +               IS_ERR(vfio_pci_core_get_iomap(vdev, index)))
> > +                   return -ENODEV;
> > +   } else if (index - VFIO_PCI_NUM_REGIONS >= vdev->num_regions) {
> >             return -ENODEV;
> > +   }
> >
> >     dma_ranges = memdup_array_user(&arg->dma_ranges,
> get_dma_buf.nr_ranges,
> >                                    sizeof(*dma_ranges));
> > --
> > 2.25.1
> >
> >

Reply via email to