diff --git a/src/brpc/rdma/rdma_endpoint.cpp b/src/brpc/rdma/rdma_endpoint.cpp index a602122e17..0dc41c9dd1 100644 --- a/src/brpc/rdma/rdma_endpoint.cpp +++ b/src/brpc/rdma/rdma_endpoint.cpp @@ -80,7 +80,25 @@ extern const size_t RESERVED_WR_NUM = 3; // The local recv block size, set during GlobalInitialize. uint32_t g_rdma_recv_block_size = 0; -// static const uint32_t MAX_INLINE_DATA = 64; +// Inline size asked for at QP creation, and on Mellanox/NVIDIA NICs the +// largest message sent inline: 256 - 16 (ctrl) - 4 (inline header). A bigger +// inlined WQE no longer fits the 256-byte BlueFlame buffer of mlx5 NICs. +static const uint32_t BF_MAX_INLINE_DATA = 256 - 16 - 4; +static const uint32_t MELLANOX_VENDOR_ID = 0x02c9; +// The verbs API cannot report the inline data limit. Devices whose kernel +// driver has a fixed limit try it first: Intel irdma 216, 101 or 48 depending +// on the generation (101 on the E810, 48 on the X722; +// drivers/infiniband/hw/irdma/ig3rdma_hw.h, user.h, i40iw_hw.h), Alibaba +// erdma 96 (drivers/infiniband/hw/erdma/erdma_verbs.h). Other devices start +// at BF_MAX_INLINE_DATA, and any refused size steps down by INLINE_DATA_STEP +// until one is accepted. libfabric's verbs provider also probes the limit +// (vrb_find_max_inline()). +static const struct { + uint32_t vendor_id; + uint32_t max_inline; +} KNOWN_INLINE_LIMITS[] = { + {0x8086, 216}, {0x8086, 101}, {0x8086, 48}, {0x1ded, 96}}; +static const uint32_t INLINE_DATA_STEP = 16; static const uint8_t MAX_HOP_LIMIT = 16; static const uint8_t TIMEOUT = 14; static const uint8_t RETRY_CNT = 7; @@ -780,11 +798,12 @@ class RdmaIOBuf : public butil::IOBuf { friend class RdmaEndpoint; private: // Cut the current IOBuf to ibv_sge list and `to' for at most first max_sge - // blocks or first max_len bytes. + // blocks or first max_len bytes. `*in_pool' is cleared if any block is + // outside the block pool (user registered memory). // Return: the bytes included in the sglist, or -1 if failed ssize_t cut_into_sglist_and_iobuf(ibv_sge* sglist, size_t* sge_index, butil::IOBuf* to, size_t max_sge, - size_t max_len) { + size_t max_len, bool* in_pool) { size_t len = 0; while (*sge_index < max_sge) { if (len == max_len || _ref_num() == 0) { @@ -795,6 +814,7 @@ friend class RdmaEndpoint; const void* start = fetch1(); uint32_t lkey = GetRegionId(start); if (lkey == 0) { // get lkey for user registered memory + *in_pool = false; uint64_t meta = get_first_data_meta(); if (meta <= UINT_MAX) { lkey = (uint32_t)meta; @@ -829,6 +849,14 @@ friend class RdmaEndpoint; } }; +// Whether to post a message of len bytes inline. Only a message from the block +// pool (host memory) that fits the inline size the QP was granted is copied +// into the WQE. User registered memory may be device memory, which the CPU +// cannot copy from, so it is never inlined. Not static: exposed for UT. +bool ShouldPostInline(bool in_pool, size_t len, uint32_t max_inline_data) { + return in_pool && len <= max_inline_data; +} + // Note this function is coupled with the implementation of IOBuf ssize_t RdmaEndpoint::CutFromIOBufList(butil::IOBuf** from, size_t ndata) { if (BAIDU_UNLIKELY(g_skip_rdma_init)) { @@ -868,6 +896,7 @@ ssize_t RdmaEndpoint::CutFromIOBufList(butil::IOBuf** from, size_t ndata) { RdmaIOBuf* data = (RdmaIOBuf*)from[current]; size_t sge_index = 0; + bool in_pool = true; while (sge_index < (uint32_t)max_sge && this_len < _remote_recv_block_size) { if (data->empty()) { @@ -881,7 +910,8 @@ ssize_t RdmaEndpoint::CutFromIOBufList(butil::IOBuf** from, size_t ndata) { } ssize_t len = data->cut_into_sglist_and_iobuf( - sglist, &sge_index, to, max_sge, _remote_recv_block_size - this_len); + sglist, &sge_index, to, max_sge, _remote_recv_block_size - this_len, + &in_pool); if (len < 0) { return -1; } @@ -894,6 +924,9 @@ ssize_t RdmaEndpoint::CutFromIOBufList(butil::IOBuf** from, size_t ndata) { } wr.num_sge = sge_index; + if (ShouldPostInline(in_pool, this_len, _resource->max_inline_data)) { + wr.send_flags |= IBV_SEND_INLINE; + } uint32_t imm = _new_rq_wrs.exchange(0, butil::memory_order_relaxed); wr.imm_data = butil::HostToNet32(imm); @@ -1128,7 +1161,65 @@ int RdmaEndpoint::PostRecv(uint32_t num, bool zerocopy) { return 0; } -static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, uint32_t rq_size) { +// Inline size to request: on the first try, size capped at the vendor's first +// known limit; after a refusal, the vendor's next smaller known limit, else +// one step smaller. Not static: exposed for UT. +uint32_t InlineDataToRequest(uint32_t vendor_id, uint32_t size, bool refused) { + for (const auto& known : KNOWN_INLINE_LIMITS) { + if (known.vendor_id == vendor_id && + (!refused || known.max_inline < size)) { + return std::min(known.max_inline, size); + } + } + if (refused) { + size = size > INLINE_DATA_STEP ? size - INLINE_DATA_STEP : 0; + } + return size; +} + +// The inline size the last QP was created with (UINT32_MAX: none yet). All +// QPs use the same device and attributes apart from their queue sizes, so +// later QPs start here instead of repeating the step-down. Not static: +// exposed for UT. +butil::atomic g_inline_data_request(UINT32_MAX); + +// Creates a QP with attr, asking for the inline size the last QP was created +// with (at first BF_MAX_INLINE_DATA, capped by the vendor's known limit), or +// less if the device refuses that size, and sets *max_inline_data to the +// largest message to post inline on it. Not static: exposed for UT. +ibv_qp* CreateQpWithInlineData(ibv_pd* pd, ibv_qp_init_attr* attr, + uint32_t vendor_id, uint32_t* max_inline_data) { + uint32_t inline_size = + g_inline_data_request.load(butil::memory_order_relaxed); + if (inline_size == UINT32_MAX) { + inline_size = InlineDataToRequest(vendor_id, BF_MAX_INLINE_DATA, false); + } + attr->cap.max_inline_data = inline_size; + ibv_qp* qp = IbvCreateQp(pd, attr); + // A refused inline size fails with EINVAL; any other failure is left to + // the caller, which reports its errno. + while (qp == nullptr && errno == EINVAL && inline_size > 0) { + uint32_t next = InlineDataToRequest(vendor_id, inline_size, true); + LOG_IF(INFO, FLAGS_rdma_trace_verbose) + << "ibv_create_qp refused " << inline_size + << " bytes of inline data, trying " << next; + inline_size = next; + attr->cap.max_inline_data = inline_size; + qp = IbvCreateQp(pd, attr); + } + if (qp != nullptr) { + g_inline_data_request.store(inline_size, butil::memory_order_relaxed); + // ibv_create_qp writes the granted inline data size back into attr + *max_inline_data = attr->cap.max_inline_data; + if (vendor_id == MELLANOX_VENDOR_ID) { + *max_inline_data = std::min(*max_inline_data, BF_MAX_INLINE_DATA); + } + } + return qp; +} + +static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, + uint32_t rq_size, uint32_t* max_inline_data) { ibv_qp_init_attr attr; memset(&attr, 0, sizeof(attr)); attr.send_cq = send_cq; @@ -1138,7 +1229,8 @@ static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, ui attr.cap.max_send_sge = GetRdmaMaxSge(); attr.cap.max_recv_sge = 1; attr.qp_type = IBV_QPT_RC; - return IbvCreateQp(GetRdmaPd(), &attr); + return CreateQpWithInlineData(GetRdmaPd(), &attr, GetRdmaVendorId(), + max_inline_data); } static RdmaResource* AllocateQpCq(uint16_t sq_size, uint16_t rq_size) { @@ -1173,7 +1265,8 @@ static RdmaResource* AllocateQpCq(uint16_t sq_size, uint16_t rq_size) { return nullptr; } - resource->qp = AllocateQp(resource->send_cq, resource->recv_cq, sq_size, rq_size); + resource->qp = AllocateQp(resource->send_cq, resource->recv_cq, sq_size, rq_size, + &resource->max_inline_data); if (nullptr == resource->qp) { PLOG(WARNING) << "Fail to create QP"; return nullptr; @@ -1187,7 +1280,8 @@ static RdmaResource* AllocateQpCq(uint16_t sq_size, uint16_t rq_size) { } resource->qp = AllocateQp(resource->polling_cq, resource->polling_cq, - sq_size, rq_size); + sq_size, rq_size, + &resource->max_inline_data); if (nullptr == resource->qp) { PLOG(WARNING) << "Fail to create QP"; return nullptr; diff --git a/src/brpc/rdma/rdma_endpoint.h b/src/brpc/rdma/rdma_endpoint.h index 6d6ac391cd..eda5e558ef 100644 --- a/src/brpc/rdma/rdma_endpoint.h +++ b/src/brpc/rdma/rdma_endpoint.h @@ -87,6 +87,8 @@ struct RdmaResource { ibv_cq* send_cq{nullptr}; ibv_cq* recv_cq{nullptr}; ibv_comp_channel* comp_channel{nullptr}; + // Largest message posted inline on `qp'. + uint32_t max_inline_data{0}; RdmaResource() = default; ~RdmaResource(); DISALLOW_COPY_AND_ASSIGN(RdmaResource); diff --git a/src/brpc/rdma/rdma_helper.cpp b/src/brpc/rdma/rdma_helper.cpp index 38bd58cac4..7e671a85ba 100644 --- a/src/brpc/rdma/rdma_helper.cpp +++ b/src/brpc/rdma/rdma_helper.cpp @@ -84,6 +84,7 @@ static uint8_t g_gid_index = 0; static ibv_gid g_gid; static uint16_t g_lid; static int g_max_sge = 0; +static uint32_t g_vendor_id = 0; static uint8_t g_port_num = 1; static int g_comp_vector_index = 0; @@ -550,6 +551,7 @@ static void GlobalRdmaInitializeOrDieImpl() { } else { g_max_sge = attr.max_sge; } + g_vendor_id = attr.vendor_id; // Initialize RDMA memory pool (block_pool) butil::SetDefaultBlockSize(GetRdmaBlockSize()); @@ -644,6 +646,10 @@ int GetRdmaMaxSge() { return g_max_sge; } +uint32_t GetRdmaVendorId() { + return g_vendor_id; +} + int GetRdmaCompVector() { if (!g_context) { return 0; diff --git a/src/brpc/rdma/rdma_helper.h b/src/brpc/rdma/rdma_helper.h index 052763325b..2383d2d8b4 100644 --- a/src/brpc/rdma/rdma_helper.h +++ b/src/brpc/rdma/rdma_helper.h @@ -77,6 +77,9 @@ uint8_t GetRdmaPortNum(); // Get max_sge supported by the device int GetRdmaMaxSge(); +// Get the vendor ID of the device +uint32_t GetRdmaVendorId(); + // Get suggested comp_vector for a new CQ int GetCompVector(); diff --git a/test/brpc_rdma_unittest.cpp b/test/brpc_rdma_unittest.cpp index 8886714268..6ec54ed90b 100644 --- a/test/brpc_rdma_unittest.cpp +++ b/test/brpc_rdma_unittest.cpp @@ -78,6 +78,14 @@ extern int (*IbvDestroyQp)(ibv_qp*); extern butil::atomic g_rdma_available; extern bool g_skip_rdma_init; extern bool g_fail_resource_alloc_for_test; +extern uint32_t InlineDataToRequest(uint32_t vendor_id, uint32_t size, + bool refused); +extern ibv_qp* CreateQpWithInlineData(ibv_pd* pd, ibv_qp_init_attr* attr, + uint32_t vendor_id, + uint32_t* max_inline_data); +extern butil::atomic g_inline_data_request; +extern bool ShouldPostInline(bool in_pool, size_t len, + uint32_t max_inline_data); } // namespace rdma } // namespace brpc @@ -3133,6 +3141,181 @@ INSTANTIATE_TEST_SUITE_P( return std::string("v") + std::to_string(info.param); }); +static const uint32_t MELLANOX = 0x02c9; +static const uint32_t INTEL = 0x8086; +static const uint32_t ALIBABA = 0x1ded; +static const uint32_t OTHER = 0xffffff; // e.g. rxe + +TEST(RdmaInlineDataTest, inline_data_to_request) { + // First try: the project's 236 bytes, capped at the vendor's first + // known limit. + EXPECT_EQ(236u, rdma::InlineDataToRequest(MELLANOX, 236, false)); + EXPECT_EQ(216u, rdma::InlineDataToRequest(INTEL, 236, false)); + EXPECT_EQ(96u, rdma::InlineDataToRequest(ALIBABA, 236, false)); + EXPECT_EQ(236u, rdma::InlineDataToRequest(OTHER, 236, false)); + EXPECT_EQ(64u, rdma::InlineDataToRequest(INTEL, 64, false)); + // After a refusal: the vendor's next smaller known limit, else 16 less. + EXPECT_EQ(101u, rdma::InlineDataToRequest(INTEL, 216, true)); + EXPECT_EQ(48u, rdma::InlineDataToRequest(INTEL, 101, true)); + EXPECT_EQ(32u, rdma::InlineDataToRequest(INTEL, 48, true)); + EXPECT_EQ(80u, rdma::InlineDataToRequest(ALIBABA, 96, true)); + EXPECT_EQ(220u, rdma::InlineDataToRequest(MELLANOX, 236, true)); + EXPECT_EQ(220u, rdma::InlineDataToRequest(OTHER, 236, true)); + EXPECT_EQ(0u, rdma::InlineDataToRequest(OTHER, 12, true)); + EXPECT_EQ(0u, rdma::InlineDataToRequest(OTHER, 0, true)); +} + +// Stub for IbvCreateQp: records every inline size requested, refuses a request +// above g_stub_limit (every request if g_stub_limit < 0) with g_stub_errno, and +// otherwise writes back g_stub_grant as the granted size (0: the request). +static std::vector g_stub_requests; +static int64_t g_stub_limit = 0; +static int g_stub_errno = EINVAL; +static uint32_t g_stub_grant = 0; +static ibv_qp g_stub_qp; + +static ibv_qp* StubCreateQp(ibv_pd*, ibv_qp_init_attr* attr) { + g_stub_requests.push_back(attr->cap.max_inline_data); + if (g_stub_limit < 0 || attr->cap.max_inline_data > g_stub_limit) { + errno = g_stub_errno; + return nullptr; + } + if (g_stub_grant != 0) { + attr->cap.max_inline_data = g_stub_grant; + } + return &g_stub_qp; +} + +// Runs CreateQpWithInlineData() against the stub. Returns whether a QP was +// created; *limit gets the inline limit, *err the errno left behind. Unless +// keep_cached, forgets the inline size cached by earlier QP creations first. +static bool CreateWithStub(uint32_t vendor_id, int64_t device_limit, + int refuse_errno, uint32_t grant, uint32_t* limit, + int* err, bool keep_cached = false) { + if (!keep_cached) { + rdma::g_inline_data_request.store(UINT32_MAX); + } + ibv_qp* (*saved)(ibv_pd*, ibv_qp_init_attr*) = rdma::IbvCreateQp; + rdma::IbvCreateQp = StubCreateQp; + g_stub_requests.clear(); + g_stub_limit = device_limit; + g_stub_errno = refuse_errno; + g_stub_grant = grant; + ibv_qp_init_attr attr; + memset(&attr, 0, sizeof(attr)); + *limit = 0; + errno = 0; + ibv_qp* qp = rdma::CreateQpWithInlineData(NULL, &attr, vendor_id, limit); + *err = errno; + rdma::IbvCreateQp = saved; + return qp != nullptr; +} + +TEST(RdmaInlineDataTest, create_qp_with_inline_data) { + uint32_t limit = 0; + int err = 0; + typedef std::vector Sizes; + + // Mellanox: 236 accepted on the first try; a larger grant (316) is capped + // at 236 so that the inlined WQE fits the BlueFlame buffer. + ASSERT_TRUE(CreateWithStub(MELLANOX, 1024, EINVAL, 316, &limit, &err)); + EXPECT_EQ(Sizes({236}), g_stub_requests); + EXPECT_EQ(236u, limit); + + // Another device keeps a larger grant (rxe grants 16 bytes per send SGE). + ASSERT_TRUE(CreateWithStub(OTHER, 1024, EINVAL, 512, &limit, &err)); + EXPECT_EQ(Sizes({236}), g_stub_requests); + EXPECT_EQ(512u, limit); + + // Intel irdma, whose limit is 216, 101 or 48 depending on the generation. + ASSERT_TRUE(CreateWithStub(INTEL, 216, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({216}), g_stub_requests); + EXPECT_EQ(216u, limit); + ASSERT_TRUE(CreateWithStub(INTEL, 101, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({216, 101}), g_stub_requests); + EXPECT_EQ(101u, limit); + ASSERT_TRUE(CreateWithStub(INTEL, 48, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({216, 101, 48}), g_stub_requests); + EXPECT_EQ(48u, limit); + + // Alibaba erdma (96). + ASSERT_TRUE(CreateWithStub(ALIBABA, 96, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({96}), g_stub_requests); + EXPECT_EQ(96u, limit); + + // An unknown device that takes at most 101 bytes: 16 bytes less each time. + ASSERT_TRUE(CreateWithStub(OTHER, 101, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({236, 220, 204, 188, 172, 156, 140, 124, 108, 92}), + g_stub_requests); + EXPECT_EQ(92u, limit); + + // A failure that is not EINVAL stops at once and leaves its errno. + ASSERT_FALSE(CreateWithStub(OTHER, -1, ENOMEM, 0, &limit, &err)); + EXPECT_EQ(Sizes({236}), g_stub_requests); + EXPECT_EQ(ENOMEM, err); + + // A device that refuses every size with EINVAL ends with a request of 0. + ASSERT_FALSE(CreateWithStub(OTHER, -1, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({236, 220, 204, 188, 172, 156, 140, 124, 108, 92, 76, 60, + 44, 28, 12, 0}), + g_stub_requests); + EXPECT_EQ(EINVAL, err); +} + +// Later QPs start at the inline size the last QP was created with. +TEST(RdmaInlineDataTest, create_qp_reuses_inline_size) { + uint32_t limit = 0; + int err = 0; + typedef std::vector Sizes; + + // The first QP on an Intel device limited to 48 bytes steps down; the next + // asks for 48 straight away. + ASSERT_TRUE(CreateWithStub(INTEL, 48, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({216, 101, 48}), g_stub_requests); + ASSERT_TRUE(CreateWithStub(INTEL, 48, EINVAL, 0, &limit, &err, true)); + EXPECT_EQ(Sizes({48}), g_stub_requests); + EXPECT_EQ(48u, limit); + + // If the cached size is refused (a QP whose other attributes leave less + // room), the step-down continues from it, and the new size is kept. + ASSERT_TRUE(CreateWithStub(INTEL, 40, EINVAL, 0, &limit, &err, true)); + EXPECT_EQ(Sizes({48, 32}), g_stub_requests); + EXPECT_EQ(32u, limit); + ASSERT_TRUE(CreateWithStub(INTEL, 40, EINVAL, 0, &limit, &err, true)); + EXPECT_EQ(Sizes({32}), g_stub_requests); + + // A failure other than EINVAL leaves the cached size as it was. + ASSERT_FALSE(CreateWithStub(INTEL, -1, ENOMEM, 0, &limit, &err, true)); + EXPECT_EQ(Sizes({32}), g_stub_requests); + ASSERT_TRUE(CreateWithStub(INTEL, 40, EINVAL, 0, &limit, &err, true)); + EXPECT_EQ(Sizes({32}), g_stub_requests); + + // Mellanox: 236 is accepted and asked for again; the grant is still capped. + ASSERT_TRUE(CreateWithStub(MELLANOX, 1024, EINVAL, 316, &limit, &err)); + ASSERT_TRUE(CreateWithStub(MELLANOX, 1024, EINVAL, 316, &limit, &err, + true)); + EXPECT_EQ(Sizes({236}), g_stub_requests); + EXPECT_EQ(236u, limit); + rdma::g_inline_data_request.store(UINT32_MAX); +} + +// The send-path decision of CutFromIOBufList(), which itself needs an +// initialized RDMA device. +TEST(RdmaInlineDataTest, should_post_inline) { + const uint32_t granted = 236; + // A message from the block pool at the granted size is inlined, one byte + // more is not. + EXPECT_TRUE(rdma::ShouldPostInline(true, granted, granted)); + EXPECT_FALSE(rdma::ShouldPostInline(true, granted + 1, granted)); + EXPECT_TRUE(rdma::ShouldPostInline(true, 1, granted)); + // A message in user registered memory, which may be device memory, is + // never inlined, however small. + EXPECT_FALSE(rdma::ShouldPostInline(false, 1, granted)); + EXPECT_FALSE(rdma::ShouldPostInline(false, granted, granted)); + // A QP granted no inline data inlines nothing. + EXPECT_FALSE(rdma::ShouldPostInline(true, 1, 0)); +} + #endif // if BRPC_WITH_RDMA int main(int argc, char* argv[]) {