Re: [PATCH v4 08/14] nvme-pci: implement dma_token backed requests

Caleb Sander Mateos <[email protected]>
Newsgroups org.kernel.vger.linux-media,dev.linux.lists.dm-devel,dev.linux.lists.nvdimm,org.freedesktop.lists.dri-devel,org.infradead.lists.linux-nvme,org.kernel.vger.ceph-devel,org.kernel.vger.io-uring,org.kernel.vger.linux-block,org.kernel.vger.linux-btrfs,org.kernel.vger.linux-fsdevel,org.kernel.vger.linux-kernel
Message-ID <CADUfDZpwF0jp2e=+c3w_NjWPbnQ858bkSD731M6+x27jzgcJiw@mail.gmail.com>
On Tue, Jul 28, 2026 at 2:35 PM Pavel Begunkov <[email protected]> wrote:
>
> Enable BIO_DMABUF_MAP backed requests. It creates a prp list for the
> dmabuf when it's mapped, which is then used to initialise requests.
>
> Suggested-by: Keith Busch <[email protected]>
> Signed-off-by: Pavel Begunkov <[email protected]>
> ---
>  drivers/nvme/host/core.c |  12 ++
>  drivers/nvme/host/nvme.h |   2 +
>  drivers/nvme/host/pci.c  | 259 +++++++++++++++++++++++++++++++++++++++
>  3 files changed, 273 insertions(+)
>
> diff --git a/drivers/nvme/host/core.c b/drivers/nvme/host/core.c
> index 453c1f0b2dd0..ce66a1843bec 100644
> --- a/drivers/nvme/host/core.c
> +++ b/drivers/nvme/host/core.c
> @@ -2676,6 +2676,17 @@ static int nvme_report_zones(struct gendisk *disk, sector_t sector,
>  #define nvme_report_zones      NULL
>  #endif /* CONFIG_BLK_DEV_ZONED */
>
> +static int nvme_init_dma_buf_io_ctx(struct block_device *bdev,
> +                                   struct dma_buf_io_ctx *ctx)
> +{
> +       struct nvme_ns *ns = bdev->bd_disk->private_data;
> +       struct nvme_ctrl *ctrl = ns->ctrl;
> +
> +       if (!ctrl->ops->init_dma_buf_io_ctx)
> +               return -EINVAL;
> +       return ctrl->ops->init_dma_buf_io_ctx(ctrl, ctx);
> +}
> +
>  const struct block_device_operations nvme_bdev_ops = {
>         .owner          = THIS_MODULE,
>         .ioctl          = nvme_ioctl,
> @@ -2686,6 +2697,7 @@ const struct block_device_operations nvme_bdev_ops = {
>         .get_unique_id  = nvme_get_unique_id,
>         .report_zones   = nvme_report_zones,
>         .pr_ops         = &nvme_pr_ops,
> +       .init_dma_buf_io_ctx = nvme_init_dma_buf_io_ctx,
>  };
>
>  static int nvme_wait_ready(struct nvme_ctrl *ctrl, u32 mask, u32 val,
> diff --git a/drivers/nvme/host/nvme.h b/drivers/nvme/host/nvme.h
> index 824651cc898d..034c31af8fec 100644
> --- a/drivers/nvme/host/nvme.h
> +++ b/drivers/nvme/host/nvme.h
> @@ -652,6 +652,8 @@ struct nvme_ctrl_ops {
>         int (*get_address)(struct nvme_ctrl *ctrl, char *buf, int size);
>         void (*print_device_info)(struct nvme_ctrl *ctrl);
>         bool (*supports_pci_p2pdma)(struct nvme_ctrl *ctrl);
> +       int (*init_dma_buf_io_ctx)(struct nvme_ctrl *ctrl,
> +                                  struct dma_buf_io_ctx *ctx);
>         unsigned long (*get_virt_boundary)(struct nvme_ctrl *ctrl, bool is_admin);
>  };
>
> diff --git a/drivers/nvme/host/pci.c b/drivers/nvme/host/pci.c
> index 69932d640b53..f1b67c191892 100644
> --- a/drivers/nvme/host/pci.c
> +++ b/drivers/nvme/host/pci.c
> @@ -27,6 +27,8 @@
>  #include <linux/io-64-nonatomic-lo-hi.h>
>  #include <linux/io-64-nonatomic-hi-lo.h>
>  #include <linux/sed-opal.h>
> +#include <linux/dma-buf-io.h>
> +#include <linux/dma-resv.h>
>
>  #include "trace.h"
>  #include "nvme.h"
> @@ -393,6 +395,13 @@ struct nvme_queue {
>         struct completion delete_done;
>  };
>
> +struct nvme_dmabuf_map {
> +       struct dma_buf_io_map base;
> +       struct sg_table *sgt;
> +       unsigned nr_entries;
> +       dma_addr_t dma_list[];
> +};
> +
>  /* bits for iod->flags */
>  enum nvme_iod_flags {
>         /* this command has been aborted by the timeout handler */
> @@ -859,6 +868,138 @@ static void nvme_free_descriptors(struct request *req)
>         }
>  }
>
> +static inline struct nvme_dmabuf_map *
> +to_nvme_dmabuf_map(struct dma_buf_io_map *map)
> +{
> +       return container_of(map, struct nvme_dmabuf_map, base);
> +}
> +
> +static void nvme_dmabuf_map_sync_for_cpu(struct nvme_dev *nvme_dev,
> +                                        struct request *req)
> +{
> +       struct device *dev = nvme_dev->dev;
> +       enum dma_data_direction dma_dir;
> +       struct bio *bio = req->bio;
> +       struct nvme_dmabuf_map *map = to_nvme_dmabuf_map(bio->bi_dmabuf_map);
> +       dma_addr_t *dma_list = map->dma_list;
> +       unsigned offset = bio->bi_iter.bi_offset;
> +       unsigned map_idx = offset / NVME_CTRL_PAGE_SIZE;
> +       int length = blk_rq_payload_bytes(req) +
> +                    (offset & (NVME_CTRL_PAGE_SIZE - 1));
> +
> +       dma_dir = rq_data_dir(req) == READ ? DMA_FROM_DEVICE : DMA_TO_DEVICE;
> +
> +       while (length > 0) {
> +               dma_sync_single_for_cpu(dev, dma_list[map_idx++],
> +                                       NVME_CTRL_PAGE_SIZE, dma_dir);
> +               length -= NVME_CTRL_PAGE_SIZE;
> +       }
> +}
> +
> +static void nvme_dmabuf_map_sync_for_device(struct nvme_dev *nvme_dev,
> +                                           struct request *req)
> +{
> +       struct device *dev = nvme_dev->dev;
> +       enum dma_data_direction dma_dir;
> +       struct bio *bio = req->bio;
> +       struct nvme_dmabuf_map *map = to_nvme_dmabuf_map(bio->bi_dmabuf_map);
> +       dma_addr_t *dma_list = map->dma_list;
> +       unsigned offset = bio->bi_iter.bi_offset;
> +       unsigned map_idx = offset / NVME_CTRL_PAGE_SIZE;
> +       int length = blk_rq_payload_bytes(req) +
> +                    (offset & (NVME_CTRL_PAGE_SIZE - 1));
> +
> +       dma_dir = rq_data_dir(req) == READ ? DMA_FROM_DEVICE : DMA_TO_DEVICE;
> +
> +       while (length > 0) {
> +               dma_sync_single_for_device(dev, dma_list[map_idx++],
> +                                          NVME_CTRL_PAGE_SIZE, dma_dir);
> +               length -= NVME_CTRL_PAGE_SIZE;
> +       }
> +}
> +
> +static void nvme_rq_clean_dmabuf_map(struct nvme_dev *dev,
> +                                     struct request *req)
> +{
> +       struct nvme_iod *iod = blk_mq_rq_to_pdu(req);
> +
> +       nvme_dmabuf_map_sync_for_cpu(dev, req);
> +
> +       if (!(iod->flags & IOD_SINGLE_SEGMENT))
> +               nvme_free_descriptors(req);
> +}
> +
> +static blk_status_t nvme_rq_setup_dmabuf_map(struct request *req,
> +                                            struct nvme_queue *nvmeq)
> +{
> +       struct nvme_iod *iod = blk_mq_rq_to_pdu(req);
> +       struct bio *bio = req->bio;
> +       struct nvme_dmabuf_map *map = to_nvme_dmabuf_map(bio->bi_dmabuf_map);
> +       unsigned bvec_done = bio->bi_iter.bi_offset;
> +       unsigned map_idx = bvec_done / NVME_CTRL_PAGE_SIZE;
> +       unsigned offset = bvec_done & (NVME_CTRL_PAGE_SIZE - 1);
> +       int length = blk_rq_payload_bytes(req) - (NVME_CTRL_PAGE_SIZE - offset);
> +       dma_addr_t *dma_list = map->dma_list;
> +       u64 prp1_dma = dma_list[map_idx++] + offset;
> +       u64 dma_addr, prp2_dma;
> +       dma_addr_t prp_dma;
> +       __le64 *prp_list;
> +       unsigned i;
> +
> +       nvme_dmabuf_map_sync_for_device(nvmeq->dev, req);
> +
> +       if (length <= 0) {
> +               prp2_dma = 0;
> +               goto done;
> +       }
> +
> +       if (length <= NVME_CTRL_PAGE_SIZE) {
> +               prp2_dma = dma_list[map_idx];
> +               goto done;
> +       }
> +
> +       if (DIV_ROUND_UP(length, NVME_CTRL_PAGE_SIZE) <=
> +           NVME_SMALL_POOL_SIZE / sizeof(__le64))
> +               iod->flags |= IOD_SMALL_DESCRIPTOR;
> +
> +       prp_list = dma_pool_alloc(nvme_dma_pool(nvmeq, iod), GFP_ATOMIC,
> +                       &prp_dma);

I wonder if it's possible to perform the PRP list page allocations and
initialization in nvme_dma_buf_io_map() instead of the I/O path. The
NVMe dma_pools are per-NUMA-node linked lists protected by spinlocks,
making them significant CPU hotspots. If the dmabuf registration set
up a contiguous DMA-coherent list of PRP entries for the pages of the
registered buffer, any command using the dmabuf and requiring at most
1 PRP list page (i.e. length <= 2 MB) could just set its PRP list
pointer to an offset into the dmabuf's PRP list.

Best,
Caleb


> +       if (!prp_list)
> +               return BLK_STS_RESOURCE;
> +
> +       iod->descriptors[iod->nr_descriptors++] = prp_list;
> +       prp2_dma = prp_dma;
> +       i = 0;
> +       for (;;) {
> +               if (i == NVME_CTRL_PAGE_SIZE >> 3) {
> +                       __le64 *old_prp_list = prp_list;
> +
> +                       prp_list = dma_pool_alloc(nvmeq->descriptor_pools.large,
> +                                       GFP_ATOMIC, &prp_dma);
> +                       if (!prp_list)
> +                               goto free_prps;
> +                       iod->descriptors[iod->nr_descriptors++] = prp_list;
> +                       prp_list[0] = old_prp_list[i - 1];
> +                       old_prp_list[i - 1] = cpu_to_le64(prp_dma);
> +                       i = 1;
> +               }
> +
> +               dma_addr = dma_list[map_idx++];
> +               prp_list[i++] = cpu_to_le64(dma_addr);
> +
> +               length -= NVME_CTRL_PAGE_SIZE;
> +               if (length <= 0)
> +                       break;
> +       }
> +done:
> +       iod->cmd.common.dptr.prp1 = cpu_to_le64(prp1_dma);
> +       iod->cmd.common.dptr.prp2 = cpu_to_le64(prp2_dma);
> +       return BLK_STS_OK;
> +free_prps:
> +       nvme_free_descriptors(req);
> +       return BLK_STS_RESOURCE;
> +}
> +
>  static void nvme_free_prps(struct request *req, unsigned int attrs)
>  {
>         struct nvme_iod *iod = blk_mq_rq_to_pdu(req);
> @@ -937,6 +1078,11 @@ static void nvme_unmap_data(struct request *req)
>         struct device *dma_dev = nvmeq->dev->dev;
>         unsigned int attrs = 0;
>
> +       if (blk_mq_rq_is_dmabuf(req)) {
> +               nvme_rq_clean_dmabuf_map(nvmeq->dev, req);
> +               return;
> +       }
> +
>         if (iod->flags & IOD_SINGLE_SEGMENT) {
>                 static_assert(offsetof(union nvme_data_ptr, prp1) ==
>                                 offsetof(union nvme_data_ptr, sgl.addr));
> @@ -1251,6 +1397,9 @@ static blk_status_t nvme_map_data(struct request *req)
>         struct blk_dma_iter iter;
>         blk_status_t ret;
>
> +       if (blk_mq_rq_is_dmabuf(req))
> +               return nvme_rq_setup_dmabuf_map(req, nvmeq);
> +
>         /*
>          * Try to skip the DMA iterator for single segment requests, as that
>          * significantly improves performances for small I/O sizes.
> @@ -2269,6 +2418,112 @@ static int nvme_create_queue(struct nvme_queue *nvmeq, int qid, bool polled)
>         return result;
>  }
>
> +#if defined(CONFIG_DMA_SHARED_BUFFER)
> +static void nvme_dmabuf_invalidate_mappings(struct dma_buf_attachment *attach)
> +{
> +       struct dma_buf_io_ctx *ctx = attach->importer_priv;
> +
> +       dma_buf_io_invalidate_mappings(ctx);
> +}
> +
> +const struct dma_buf_attach_ops nvme_dmabuf_importer_ops = {
> +       .invalidate_mappings    = nvme_dmabuf_invalidate_mappings,
> +       .allow_peer2peer        = true,
> +};
> +
> +static struct dma_buf_io_map *nvme_dma_buf_io_map(struct dma_buf_io_ctx *ctx)
> +{
> +       unsigned nr_entries = ctx->dmabuf->size / NVME_CTRL_PAGE_SIZE;
> +       struct dma_buf_attachment *attach = ctx->dev_priv;
> +       unsigned long tmp, i = 0;
> +       struct nvme_dmabuf_map *map;
> +       struct scatterlist *sg;
> +       struct sg_table *sgt;
> +       int ret;
> +
> +       dma_resv_assert_held(ctx->dmabuf->resv);
> +
> +       map = kmalloc_flex(*map, dma_list, nr_entries);
> +       if (!map)
> +               return ERR_PTR(-ENOMEM);
> +
> +       sgt = dma_buf_map_attachment(attach, ctx->dir);
> +       if (IS_ERR(sgt)) {
> +               ret = PTR_ERR(sgt);
> +               sgt = NULL;
> +               goto err;
> +       }
> +
> +       for_each_sgtable_dma_sg(sgt, sg, tmp) {
> +               dma_addr_t dma_addr = sg_dma_address(sg);
> +               unsigned long sg_len = sg_dma_len(sg);
> +
> +               if (sg_len % NVME_CTRL_PAGE_SIZE) {
> +                       ret = -EINVAL;
> +                       goto err;
> +               }
> +
> +               while (sg_len) {
> +                       map->dma_list[i++] = dma_addr;
> +                       dma_addr += NVME_CTRL_PAGE_SIZE;
> +                       sg_len -= NVME_CTRL_PAGE_SIZE;
> +               }
> +       }
> +
> +       ret = dma_buf_io_init_map(ctx, &map->base);
> +       if (ret)
> +               goto err;
> +       map->nr_entries = nr_entries;
> +       map->sgt = sgt;
> +       return &map->base;
> +err:
> +       if (sgt)
> +               dma_buf_unmap_attachment(attach, sgt, ctx->dir);
> +       kfree(map);
> +       return ERR_PTR(ret);
> +}
> +
> +static void nvme_dma_buf_io_unmap(struct dma_buf_io_ctx *ctx,
> +                                 struct dma_buf_io_map *map_base)
> +{
> +       struct dma_buf_attachment *attach = ctx->dev_priv;
> +       struct nvme_dmabuf_map *map = to_nvme_dmabuf_map(map_base);
> +
> +       dma_resv_assert_held(ctx->dmabuf->resv);
> +
> +       dma_buf_unmap_attachment(attach, map->sgt, ctx->dir);
> +}
> +
> +static void nvme_dma_buf_io_release(struct dma_buf_io_ctx *ctx)
> +{
> +       struct dma_buf_attachment *attach = ctx->dev_priv;
> +
> +       dma_buf_detach(ctx->dmabuf, attach);
> +}
> +
> +const struct dma_buf_io_ops nvme_dma_buf_io_ops = {
> +       .map            = nvme_dma_buf_io_map,
> +       .unmap          = nvme_dma_buf_io_unmap,
> +       .release        = nvme_dma_buf_io_release,
> +};
> +
> +static int nvme_pci_init_dma_buf_io_ctx(struct nvme_ctrl *ctrl,
> +                                       struct dma_buf_io_ctx *ctx)
> +{
> +       struct dma_buf_attachment *attach;
> +       struct nvme_dev *dev = to_nvme_dev(ctrl);
> +
> +       attach = dma_buf_dynamic_attach(ctx->dmabuf, dev->dev,
> +                                       &nvme_dmabuf_importer_ops, ctx);
> +       if (IS_ERR(attach))
> +               return PTR_ERR(attach);
> +
> +       ctx->dev_priv = attach;
> +       ctx->dev_ops = &nvme_dma_buf_io_ops;
> +       return 0;
> +}
> +#endif
> +
>  static const struct blk_mq_ops nvme_mq_admin_ops = {
>         .queue_rq       = nvme_queue_rq,
>         .complete       = nvme_pci_complete_rq,
> @@ -3563,6 +3818,9 @@ static const struct nvme_ctrl_ops nvme_pci_ctrl_ops = {
>         .print_device_info      = nvme_pci_print_device_info,
>         .supports_pci_p2pdma    = nvme_pci_supports_pci_p2pdma,
>         .get_virt_boundary      = nvme_pci_get_virt_boundary,
> +#if defined(CONFIG_DMA_SHARED_BUFFER)
> +       .init_dma_buf_io_ctx    = nvme_pci_init_dma_buf_io_ctx,
> +#endif
>  };
>
>  static int nvme_dev_map(struct nvme_dev *dev)
> @@ -4327,5 +4585,6 @@ MODULE_AUTHOR("Matthew Wilcox <[email protected]>");
>  MODULE_LICENSE("GPL");
>  MODULE_VERSION("1.0");
>  MODULE_DESCRIPTION("NVMe host PCIe transport driver");
> +MODULE_IMPORT_NS("DMA_BUF");
>  module_init(nvme_init);
>  module_exit(nvme_exit);
> --
> 2.54.0
>
>
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.