Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
110 changes: 102 additions & 8 deletions src/brpc/rdma/rdma_endpoint.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -80,7 +80,25 @@ extern const size_t RESERVED_WR_NUM = 3;
// The local recv block size, set during GlobalInitialize.
uint32_t g_rdma_recv_block_size = 0;

// static const uint32_t MAX_INLINE_DATA = 64;
// Inline size asked for at QP creation, and on Mellanox/NVIDIA NICs the
// largest message sent inline: 256 - 16 (ctrl) - 4 (inline header). A bigger
// inlined WQE no longer fits the 256-byte BlueFlame buffer of mlx5 NICs.
static const uint32_t BF_MAX_INLINE_DATA = 256 - 16 - 4;
static const uint32_t MELLANOX_VENDOR_ID = 0x02c9;
// The verbs API cannot report the inline data limit. Devices whose kernel
// driver has a fixed limit try it first: Intel irdma 216, 101 or 48 depending
// on the generation (101 on the E810, 48 on the X722;
// drivers/infiniband/hw/irdma/ig3rdma_hw.h, user.h, i40iw_hw.h), Alibaba
// erdma 96 (drivers/infiniband/hw/erdma/erdma_verbs.h). Other devices start
// at BF_MAX_INLINE_DATA, and any refused size steps down by INLINE_DATA_STEP
// until one is accepted. libfabric's verbs provider also probes the limit
// (vrb_find_max_inline()).
static const struct {
uint32_t vendor_id;
uint32_t max_inline;
} KNOWN_INLINE_LIMITS[] = {
{0x8086, 216}, {0x8086, 101}, {0x8086, 48}, {0x1ded, 96}};
static const uint32_t INLINE_DATA_STEP = 16;
static const uint8_t MAX_HOP_LIMIT = 16;
static const uint8_t TIMEOUT = 14;
static const uint8_t RETRY_CNT = 7;
Expand Down Expand Up @@ -780,11 +798,12 @@ class RdmaIOBuf : public butil::IOBuf {
friend class RdmaEndpoint;
private:
// Cut the current IOBuf to ibv_sge list and `to' for at most first max_sge
// blocks or first max_len bytes.
// blocks or first max_len bytes. `*in_pool' is cleared if any block is
// outside the block pool (user registered memory).
// Return: the bytes included in the sglist, or -1 if failed
ssize_t cut_into_sglist_and_iobuf(ibv_sge* sglist, size_t* sge_index,
butil::IOBuf* to, size_t max_sge,
size_t max_len) {
size_t max_len, bool* in_pool) {
size_t len = 0;
while (*sge_index < max_sge) {
if (len == max_len || _ref_num() == 0) {
Expand All @@ -795,6 +814,7 @@ friend class RdmaEndpoint;
const void* start = fetch1();
uint32_t lkey = GetRegionId(start);
if (lkey == 0) { // get lkey for user registered memory
*in_pool = false;
uint64_t meta = get_first_data_meta();
if (meta <= UINT_MAX) {
lkey = (uint32_t)meta;
Expand Down Expand Up @@ -829,6 +849,14 @@ friend class RdmaEndpoint;
}
};

// Whether to post a message of len bytes inline. Only a message from the block
// pool (host memory) that fits the inline size the QP was granted is copied
// into the WQE. User registered memory may be device memory, which the CPU
// cannot copy from, so it is never inlined. Not static: exposed for UT.
bool ShouldPostInline(bool in_pool, size_t len, uint32_t max_inline_data) {
return in_pool && len <= max_inline_data;
}

// Note this function is coupled with the implementation of IOBuf
ssize_t RdmaEndpoint::CutFromIOBufList(butil::IOBuf** from, size_t ndata) {
if (BAIDU_UNLIKELY(g_skip_rdma_init)) {
Expand Down Expand Up @@ -868,6 +896,7 @@ ssize_t RdmaEndpoint::CutFromIOBufList(butil::IOBuf** from, size_t ndata) {

RdmaIOBuf* data = (RdmaIOBuf*)from[current];
size_t sge_index = 0;
bool in_pool = true;
while (sge_index < (uint32_t)max_sge &&
this_len < _remote_recv_block_size) {
if (data->empty()) {
Expand All @@ -881,7 +910,8 @@ ssize_t RdmaEndpoint::CutFromIOBufList(butil::IOBuf** from, size_t ndata) {
}

ssize_t len = data->cut_into_sglist_and_iobuf(
sglist, &sge_index, to, max_sge, _remote_recv_block_size - this_len);
sglist, &sge_index, to, max_sge, _remote_recv_block_size - this_len,
&in_pool);
if (len < 0) {
return -1;
}
Expand All @@ -894,6 +924,9 @@ ssize_t RdmaEndpoint::CutFromIOBufList(butil::IOBuf** from, size_t ndata) {
}

wr.num_sge = sge_index;
if (ShouldPostInline(in_pool, this_len, _resource->max_inline_data)) {
wr.send_flags |= IBV_SEND_INLINE;
}

uint32_t imm = _new_rq_wrs.exchange(0, butil::memory_order_relaxed);
wr.imm_data = butil::HostToNet32(imm);
Expand Down Expand Up @@ -1128,7 +1161,65 @@ int RdmaEndpoint::PostRecv(uint32_t num, bool zerocopy) {
return 0;
}

static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, uint32_t rq_size) {
// Inline size to request: on the first try, size capped at the vendor's first
// known limit; after a refusal, the vendor's next smaller known limit, else
// one step smaller. Not static: exposed for UT.
uint32_t InlineDataToRequest(uint32_t vendor_id, uint32_t size, bool refused) {
for (const auto& known : KNOWN_INLINE_LIMITS) {
if (known.vendor_id == vendor_id &&
(!refused || known.max_inline < size)) {
return std::min(known.max_inline, size);
}
}
if (refused) {
size = size > INLINE_DATA_STEP ? size - INLINE_DATA_STEP : 0;
}
return size;
}

// The inline size the last QP was created with (UINT32_MAX: none yet). All
// QPs use the same device and attributes apart from their queue sizes, so
// later QPs start here instead of repeating the step-down. Not static:
// exposed for UT.
butil::atomic<uint32_t> g_inline_data_request(UINT32_MAX);

// Creates a QP with attr, asking for the inline size the last QP was created
// with (at first BF_MAX_INLINE_DATA, capped by the vendor's known limit), or
// less if the device refuses that size, and sets *max_inline_data to the
// largest message to post inline on it. Not static: exposed for UT.
ibv_qp* CreateQpWithInlineData(ibv_pd* pd, ibv_qp_init_attr* attr,
uint32_t vendor_id, uint32_t* max_inline_data) {
uint32_t inline_size =
g_inline_data_request.load(butil::memory_order_relaxed);
if (inline_size == UINT32_MAX) {
inline_size = InlineDataToRequest(vendor_id, BF_MAX_INLINE_DATA, false);
}
attr->cap.max_inline_data = inline_size;
ibv_qp* qp = IbvCreateQp(pd, attr);
// A refused inline size fails with EINVAL; any other failure is left to
// the caller, which reports its errno.
while (qp == nullptr && errno == EINVAL && inline_size > 0) {
uint32_t next = InlineDataToRequest(vendor_id, inline_size, true);
LOG_IF(INFO, FLAGS_rdma_trace_verbose)
<< "ibv_create_qp refused " << inline_size
<< " bytes of inline data, trying " << next;
inline_size = next;
attr->cap.max_inline_data = inline_size;
qp = IbvCreateQp(pd, attr);
}
if (qp != nullptr) {
g_inline_data_request.store(inline_size, butil::memory_order_relaxed);
// ibv_create_qp writes the granted inline data size back into attr
*max_inline_data = attr->cap.max_inline_data;
if (vendor_id == MELLANOX_VENDOR_ID) {
*max_inline_data = std::min(*max_inline_data, BF_MAX_INLINE_DATA);
}
Comment on lines +1210 to +1216

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This is on purpose. ibv_create_qp() either fails or grants at least what was asked: ibv_create_qp(3) says "the values will be greater than or equal to the values requested". So a size that was accepted once is accepted again on the next QP with the same attributes, and the step-down doesn't repeat. Caching the request keeps every QP asking for exactly what the first one asked for. Each QP still reads its own grant from attr.cap.max_inline_data, and the Mellanox cap is applied to that.

}
return qp;
}

static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size,
uint32_t rq_size, uint32_t* max_inline_data) {
ibv_qp_init_attr attr;
memset(&attr, 0, sizeof(attr));
attr.send_cq = send_cq;
Expand All @@ -1138,7 +1229,8 @@ static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, ui
attr.cap.max_send_sge = GetRdmaMaxSge();
attr.cap.max_recv_sge = 1;
attr.qp_type = IBV_QPT_RC;
return IbvCreateQp(GetRdmaPd(), &attr);
return CreateQpWithInlineData(GetRdmaPd(), &attr, GetRdmaVendorId(),
Comment thread
chenBright marked this conversation as resolved.
max_inline_data);
}

static RdmaResource* AllocateQpCq(uint16_t sq_size, uint16_t rq_size) {
Expand Down Expand Up @@ -1173,7 +1265,8 @@ static RdmaResource* AllocateQpCq(uint16_t sq_size, uint16_t rq_size) {
return nullptr;
}

resource->qp = AllocateQp(resource->send_cq, resource->recv_cq, sq_size, rq_size);
resource->qp = AllocateQp(resource->send_cq, resource->recv_cq, sq_size, rq_size,
&resource->max_inline_data);
if (nullptr == resource->qp) {
PLOG(WARNING) << "Fail to create QP";
return nullptr;
Expand All @@ -1187,7 +1280,8 @@ static RdmaResource* AllocateQpCq(uint16_t sq_size, uint16_t rq_size) {
}
resource->qp = AllocateQp(resource->polling_cq,
resource->polling_cq,
sq_size, rq_size);
sq_size, rq_size,
&resource->max_inline_data);
if (nullptr == resource->qp) {
PLOG(WARNING) << "Fail to create QP";
return nullptr;
Expand Down
2 changes: 2 additions & 0 deletions src/brpc/rdma/rdma_endpoint.h
Original file line number Diff line number Diff line change
Expand Up @@ -87,6 +87,8 @@ struct RdmaResource {
ibv_cq* send_cq{nullptr};
ibv_cq* recv_cq{nullptr};
ibv_comp_channel* comp_channel{nullptr};
// Largest message posted inline on `qp'.
uint32_t max_inline_data{0};
RdmaResource() = default;
~RdmaResource();
DISALLOW_COPY_AND_ASSIGN(RdmaResource);
Expand Down
6 changes: 6 additions & 0 deletions src/brpc/rdma/rdma_helper.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -84,6 +84,7 @@ static uint8_t g_gid_index = 0;
static ibv_gid g_gid;
static uint16_t g_lid;
static int g_max_sge = 0;
static uint32_t g_vendor_id = 0;
static uint8_t g_port_num = 1;

static int g_comp_vector_index = 0;
Expand Down Expand Up @@ -550,6 +551,7 @@ static void GlobalRdmaInitializeOrDieImpl() {
} else {
g_max_sge = attr.max_sge;
}
g_vendor_id = attr.vendor_id;

// Initialize RDMA memory pool (block_pool)
butil::SetDefaultBlockSize(GetRdmaBlockSize());
Expand Down Expand Up @@ -644,6 +646,10 @@ int GetRdmaMaxSge() {
return g_max_sge;
}

uint32_t GetRdmaVendorId() {
return g_vendor_id;
}

int GetRdmaCompVector() {
if (!g_context) {
return 0;
Expand Down
3 changes: 3 additions & 0 deletions src/brpc/rdma/rdma_helper.h
Original file line number Diff line number Diff line change
Expand Up @@ -77,6 +77,9 @@ uint8_t GetRdmaPortNum();
// Get max_sge supported by the device
int GetRdmaMaxSge();

// Get the vendor ID of the device
uint32_t GetRdmaVendorId();

// Get suggested comp_vector for a new CQ
int GetCompVector();

Expand Down
Loading
Loading