Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
80 changes: 78 additions & 2 deletions drivers/nvme/host/nvfs-rdma.h
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,8 @@
#ifndef NVFS_RDMA_H
#define NVFS_RDMA_H

#define NVFS_GPU_PAGE_SIZE (64UL * 1024)

static bool nvme_rdma_nvfs_unmap_data(struct ib_device *ibdev,
struct request *rq)

Expand All @@ -35,15 +37,89 @@ static bool nvme_rdma_nvfs_unmap_data(struct ib_device *ibdev,
return false;
}

static int nvme_rdma_nvfs_map_data(struct ib_device *ibdev, struct request *rq, bool *is_nvfs_io, int* count)
static bool nvme_rdma_nvfs_first_page_is_gpu(struct request *rq)
{
struct req_iterator iter;
struct bio_vec bvec;

rq_for_each_segment(bvec, rq, iter)
return nvfs_ops->nvfs_is_gpu_page(bvec.bv_page);

return false;
}

/*
* nvidia-fs issues each I/O from a single iovec, so every bio covers one
* contiguous range of a GPU buffer and maps to at most one entry per 64K
* GPU page it touches, plus one when it starts inside a GPU page. A request
* can merge bios from unrelated ranges, so bound each bio separately.
*/
static unsigned int nvme_rdma_nvfs_gpu_nr_sg(struct request *rq)
{
unsigned int nr_sg = 0;
struct bio *bio;

__rq_for_each_bio(bio, rq)
nr_sg += DIV_ROUND_UP(bio->bi_iter.bi_size,
NVFS_GPU_PAGE_SIZE) + 1;

return nr_sg;
}

static int nvme_rdma_nvfs_map_data(struct ib_device *ibdev,
struct request *rq, bool *is_nvfs_io,
bool *sg_allocated, int *count)
{
struct nvme_rdma_request *req = blk_mq_rq_to_pdu(rq);
enum dma_data_direction dma_dir = rq_dma_dir(rq);
unsigned int nr_sg;
int ret = 0;

*is_nvfs_io = false;
*sg_allocated = false;
*count = 0;
if (!blk_integrity_rq(rq) && nvfs_get_ops()) {

/*
* rq_for_each_segment() does not iterate RQF_SPECIAL_PAYLOAD. Keep
* those requests, and operations NVFS does not support, on the normal
* block mapping path. In particular, NVMe discard uses a data-less bio
* and stores its DSM descriptor in rq->special_vec.
*/
if (blk_integrity_rq(rq) ||
(rq->rq_flags & RQF_SPECIAL_PAYLOAD) ||
(req_op(rq) != REQ_OP_READ && req_op(rq) != REQ_OP_WRITE))
return 0;

if (nvfs_get_ops()) {
nr_sg = blk_rq_nr_phys_segments(rq);

/*
* Use the first bvec only to select the SGL allocation size. Every
* eligible request still goes through the NVFS mapper so it can
* validate the complete request and reject mixed CPU/GPU I/O.
*
* If page classification is unavailable, retain the conservative
* allocation required for a possible GPU request.
*/
if (!NVIDIA_FS_CHECK_FT_GPU_PAGE(nvfs_ops) ||
!nvfs_ops->nvfs_is_gpu_page ||
nvme_rdma_nvfs_first_page_is_gpu(rq)) {
/*
* The block layer may merge physically contiguous proxy pages
* into fewer segments. GPU physical pages need not have the
* same contiguity.
*/
nr_sg = max(nr_sg, nvme_rdma_nvfs_gpu_nr_sg(rq));
}

ret = sg_alloc_table_chained(&req->data_sgl.sg_table, nr_sg,
req->data_sgl.sg_table.sgl,
NVME_INLINE_SG_CNT);
if (ret) {
nvfs_put_ops();
return -ENOMEM;
}
*sg_allocated = true;

// associates bio pages to scatterlist
*count = nvfs_ops->nvfs_blk_rq_map_sg(rq->q, rq , req->data_sgl.sg_table.sgl);
Expand Down
31 changes: 22 additions & 9 deletions drivers/nvme/host/rdma.c
Original file line number Diff line number Diff line change
Expand Up @@ -1478,26 +1478,39 @@ static int nvme_rdma_dma_map_req(struct ib_device *ibdev, struct request *rq,
int *count, int *pi_count)
{
struct nvme_rdma_request *req = blk_mq_rq_to_pdu(rq);
#ifdef CONFIG_NVFS
bool sg_allocated = false;
#endif
int ret;

req->data_sgl.sg_table.sgl = (struct scatterlist *)(req + 1);
ret = sg_alloc_table_chained(&req->data_sgl.sg_table,
blk_rq_nr_phys_segments(rq), req->data_sgl.sg_table.sgl,
NVME_INLINE_SG_CNT);
if (ret)
return -ENOMEM;

#ifdef CONFIG_NVFS
{
bool is_nvfs_io = false;
ret = nvme_rdma_nvfs_map_data(ibdev, rq, &is_nvfs_io, count);
if (is_nvfs_io) {
if (ret)

ret = nvme_rdma_nvfs_map_data(ibdev, rq, &is_nvfs_io,
&sg_allocated, count);
if (ret) {
if (sg_allocated)
goto out_free_table;
return 0;
return ret;
}
if (is_nvfs_io)
return 0;
}
if (!sg_allocated) {
#endif
ret = sg_alloc_table_chained(&req->data_sgl.sg_table,
blk_rq_nr_phys_segments(rq),
req->data_sgl.sg_table.sgl,
NVME_INLINE_SG_CNT);
if (ret)
return -ENOMEM;
#ifdef CONFIG_NVFS
}
#endif

req->data_sgl.nents = blk_rq_map_sg(rq, req->data_sgl.sg_table.sgl);

*count = ib_dma_map_sg(ibdev, req->data_sgl.sg_table.sgl,
Expand Down
Loading