diff --git a/drivers/nvme/host/nvfs-rdma.h b/drivers/nvme/host/nvfs-rdma.h index f9051e2ab22b..460beb2e8dc8 100644 --- a/drivers/nvme/host/nvfs-rdma.h +++ b/drivers/nvme/host/nvfs-rdma.h @@ -15,6 +15,8 @@ #ifndef NVFS_RDMA_H #define NVFS_RDMA_H +#define NVFS_GPU_PAGE_SIZE (64UL * 1024) + static bool nvme_rdma_nvfs_unmap_data(struct ib_device *ibdev, struct request *rq) @@ -35,15 +37,89 @@ static bool nvme_rdma_nvfs_unmap_data(struct ib_device *ibdev, return false; } -static int nvme_rdma_nvfs_map_data(struct ib_device *ibdev, struct request *rq, bool *is_nvfs_io, int* count) +static bool nvme_rdma_nvfs_first_page_is_gpu(struct request *rq) +{ + struct req_iterator iter; + struct bio_vec bvec; + + rq_for_each_segment(bvec, rq, iter) + return nvfs_ops->nvfs_is_gpu_page(bvec.bv_page); + + return false; +} + +/* + * nvidia-fs issues each I/O from a single iovec, so every bio covers one + * contiguous range of a GPU buffer and maps to at most one entry per 64K + * GPU page it touches, plus one when it starts inside a GPU page. A request + * can merge bios from unrelated ranges, so bound each bio separately. + */ +static unsigned int nvme_rdma_nvfs_gpu_nr_sg(struct request *rq) +{ + unsigned int nr_sg = 0; + struct bio *bio; + + __rq_for_each_bio(bio, rq) + nr_sg += DIV_ROUND_UP(bio->bi_iter.bi_size, + NVFS_GPU_PAGE_SIZE) + 1; + + return nr_sg; +} + +static int nvme_rdma_nvfs_map_data(struct ib_device *ibdev, + struct request *rq, bool *is_nvfs_io, + bool *sg_allocated, int *count) { struct nvme_rdma_request *req = blk_mq_rq_to_pdu(rq); enum dma_data_direction dma_dir = rq_dma_dir(rq); + unsigned int nr_sg; int ret = 0; *is_nvfs_io = false; + *sg_allocated = false; *count = 0; - if (!blk_integrity_rq(rq) && nvfs_get_ops()) { + + /* + * rq_for_each_segment() does not iterate RQF_SPECIAL_PAYLOAD. Keep + * those requests, and operations NVFS does not support, on the normal + * block mapping path. In particular, NVMe discard uses a data-less bio + * and stores its DSM descriptor in rq->special_vec. + */ + if (blk_integrity_rq(rq) || + (rq->rq_flags & RQF_SPECIAL_PAYLOAD) || + (req_op(rq) != REQ_OP_READ && req_op(rq) != REQ_OP_WRITE)) + return 0; + + if (nvfs_get_ops()) { + nr_sg = blk_rq_nr_phys_segments(rq); + + /* + * Use the first bvec only to select the SGL allocation size. Every + * eligible request still goes through the NVFS mapper so it can + * validate the complete request and reject mixed CPU/GPU I/O. + * + * If page classification is unavailable, retain the conservative + * allocation required for a possible GPU request. + */ + if (!NVIDIA_FS_CHECK_FT_GPU_PAGE(nvfs_ops) || + !nvfs_ops->nvfs_is_gpu_page || + nvme_rdma_nvfs_first_page_is_gpu(rq)) { + /* + * The block layer may merge physically contiguous proxy pages + * into fewer segments. GPU physical pages need not have the + * same contiguity. + */ + nr_sg = max(nr_sg, nvme_rdma_nvfs_gpu_nr_sg(rq)); + } + + ret = sg_alloc_table_chained(&req->data_sgl.sg_table, nr_sg, + req->data_sgl.sg_table.sgl, + NVME_INLINE_SG_CNT); + if (ret) { + nvfs_put_ops(); + return -ENOMEM; + } + *sg_allocated = true; // associates bio pages to scatterlist *count = nvfs_ops->nvfs_blk_rq_map_sg(rq->q, rq , req->data_sgl.sg_table.sgl); diff --git a/drivers/nvme/host/rdma.c b/drivers/nvme/host/rdma.c index 53b4823d57c5..5bd95efff43e 100644 --- a/drivers/nvme/host/rdma.c +++ b/drivers/nvme/host/rdma.c @@ -1478,26 +1478,39 @@ static int nvme_rdma_dma_map_req(struct ib_device *ibdev, struct request *rq, int *count, int *pi_count) { struct nvme_rdma_request *req = blk_mq_rq_to_pdu(rq); +#ifdef CONFIG_NVFS + bool sg_allocated = false; +#endif int ret; req->data_sgl.sg_table.sgl = (struct scatterlist *)(req + 1); - ret = sg_alloc_table_chained(&req->data_sgl.sg_table, - blk_rq_nr_phys_segments(rq), req->data_sgl.sg_table.sgl, - NVME_INLINE_SG_CNT); - if (ret) - return -ENOMEM; #ifdef CONFIG_NVFS { bool is_nvfs_io = false; - ret = nvme_rdma_nvfs_map_data(ibdev, rq, &is_nvfs_io, count); - if (is_nvfs_io) { - if (ret) + + ret = nvme_rdma_nvfs_map_data(ibdev, rq, &is_nvfs_io, + &sg_allocated, count); + if (ret) { + if (sg_allocated) goto out_free_table; - return 0; + return ret; } + if (is_nvfs_io) + return 0; } + if (!sg_allocated) { #endif + ret = sg_alloc_table_chained(&req->data_sgl.sg_table, + blk_rq_nr_phys_segments(rq), + req->data_sgl.sg_table.sgl, + NVME_INLINE_SG_CNT); + if (ret) + return -ENOMEM; +#ifdef CONFIG_NVFS + } +#endif + req->data_sgl.nents = blk_rq_map_sg(rq, req->data_sgl.sg_table.sgl); *count = ib_dma_map_sg(ibdev, req->data_sgl.sg_table.sgl,