From 7da4d1886cccd50d0740d69495f0ea1354d4d348 Mon Sep 17 00:00:00 2001 From: Andrei-Alexandru Bleortu <46914650+alxrxs@users.noreply.github.com> Date: Fri, 2 Oct 2026 15:35:49 +0300 Subject: [PATCH 1/6] Post small RDMA messages inline CutFromIOBufList() posts every message by reference, so the NIC has to read a small RPC from host memory before sending it. Request 236 bytes of inline data at QP creation, retrying without it if the device refuses, and inline a message that fits the granted size and comes entirely from the RDMA block pool. 236 bytes is the largest Send whose WQE fits the 256-byte BlueFlame buffer of current mlx5 NICs. Generated-by: Claude Code (Claude Opus 5.5) Co-Authored-By: Claude Opus 5.5 --- src/brpc/rdma/rdma_endpoint.cpp | 44 +++++++++++++++++++++++++++------ src/brpc/rdma/rdma_endpoint.h | 2 ++ 2 files changed, 38 insertions(+), 8 deletions(-) diff --git a/src/brpc/rdma/rdma_endpoint.cpp b/src/brpc/rdma/rdma_endpoint.cpp index a602122e17..1840500778 100644 --- a/src/brpc/rdma/rdma_endpoint.cpp +++ b/src/brpc/rdma/rdma_endpoint.cpp @@ -80,7 +80,10 @@ extern const size_t RESERVED_WR_NUM = 3; // The local recv block size, set during GlobalInitialize. uint32_t g_rdma_recv_block_size = 0; -// static const uint32_t MAX_INLINE_DATA = 64; +// Largest message sent inline, and the inline size asked for at QP +// creation: 256 - 16 (ctrl) - 4 (inline header). A bigger inlined WQE no +// longer fits the 256-byte BlueFlame buffer of current mlx5 NICs. +static const uint32_t BF_MAX_INLINE_DATA = 256 - 16 - 4; static const uint8_t MAX_HOP_LIMIT = 16; static const uint8_t TIMEOUT = 14; static const uint8_t RETRY_CNT = 7; @@ -780,11 +783,12 @@ class RdmaIOBuf : public butil::IOBuf { friend class RdmaEndpoint; private: // Cut the current IOBuf to ibv_sge list and `to' for at most first max_sge - // blocks or first max_len bytes. + // blocks or first max_len bytes. `*in_pool' is cleared if any block is + // outside the block pool (user registered memory). // Return: the bytes included in the sglist, or -1 if failed ssize_t cut_into_sglist_and_iobuf(ibv_sge* sglist, size_t* sge_index, butil::IOBuf* to, size_t max_sge, - size_t max_len) { + size_t max_len, bool* in_pool) { size_t len = 0; while (*sge_index < max_sge) { if (len == max_len || _ref_num() == 0) { @@ -795,6 +799,7 @@ friend class RdmaEndpoint; const void* start = fetch1(); uint32_t lkey = GetRegionId(start); if (lkey == 0) { // get lkey for user registered memory + *in_pool = false; uint64_t meta = get_first_data_meta(); if (meta <= UINT_MAX) { lkey = (uint32_t)meta; @@ -868,6 +873,7 @@ ssize_t RdmaEndpoint::CutFromIOBufList(butil::IOBuf** from, size_t ndata) { RdmaIOBuf* data = (RdmaIOBuf*)from[current]; size_t sge_index = 0; + bool in_pool = true; while (sge_index < (uint32_t)max_sge && this_len < _remote_recv_block_size) { if (data->empty()) { @@ -881,7 +887,8 @@ ssize_t RdmaEndpoint::CutFromIOBufList(butil::IOBuf** from, size_t ndata) { } ssize_t len = data->cut_into_sglist_and_iobuf( - sglist, &sge_index, to, max_sge, _remote_recv_block_size - this_len); + sglist, &sge_index, to, max_sge, _remote_recv_block_size - this_len, + &in_pool); if (len < 0) { return -1; } @@ -894,6 +901,12 @@ ssize_t RdmaEndpoint::CutFromIOBufList(butil::IOBuf** from, size_t ndata) { } wr.num_sge = sge_index; + // A small message from the block pool (host memory) is copied into + // the WQE. User registered memory may be device memory, which the + // CPU cannot copy from, so it is never inlined. + if (in_pool && this_len <= _resource->max_inline_data) { + wr.send_flags |= IBV_SEND_INLINE; + } uint32_t imm = _new_rq_wrs.exchange(0, butil::memory_order_relaxed); wr.imm_data = butil::HostToNet32(imm); @@ -1128,7 +1141,8 @@ int RdmaEndpoint::PostRecv(uint32_t num, bool zerocopy) { return 0; } -static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, uint32_t rq_size) { +static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, + uint32_t rq_size, uint32_t* max_inline_data) { ibv_qp_init_attr attr; memset(&attr, 0, sizeof(attr)); attr.send_cq = send_cq; @@ -1137,8 +1151,20 @@ static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, ui attr.cap.max_recv_wr = rq_size; attr.cap.max_send_sge = GetRdmaMaxSge(); attr.cap.max_recv_sge = 1; + attr.cap.max_inline_data = BF_MAX_INLINE_DATA; attr.qp_type = IBV_QPT_RC; - return IbvCreateQp(GetRdmaPd(), &attr); + ibv_qp* qp = IbvCreateQp(GetRdmaPd(), &attr); + if (qp == nullptr) { + // The device may not support inline data, try again without it + attr.cap.max_inline_data = 0; + qp = IbvCreateQp(GetRdmaPd(), &attr); + } + if (qp != nullptr) { + // ibv_create_qp writes the granted inline data size back into attr + *max_inline_data = + std::min(attr.cap.max_inline_data, BF_MAX_INLINE_DATA); + } + return qp; } static RdmaResource* AllocateQpCq(uint16_t sq_size, uint16_t rq_size) { @@ -1173,7 +1199,8 @@ static RdmaResource* AllocateQpCq(uint16_t sq_size, uint16_t rq_size) { return nullptr; } - resource->qp = AllocateQp(resource->send_cq, resource->recv_cq, sq_size, rq_size); + resource->qp = AllocateQp(resource->send_cq, resource->recv_cq, sq_size, rq_size, + &resource->max_inline_data); if (nullptr == resource->qp) { PLOG(WARNING) << "Fail to create QP"; return nullptr; @@ -1187,7 +1214,8 @@ static RdmaResource* AllocateQpCq(uint16_t sq_size, uint16_t rq_size) { } resource->qp = AllocateQp(resource->polling_cq, resource->polling_cq, - sq_size, rq_size); + sq_size, rq_size, + &resource->max_inline_data); if (nullptr == resource->qp) { PLOG(WARNING) << "Fail to create QP"; return nullptr; diff --git a/src/brpc/rdma/rdma_endpoint.h b/src/brpc/rdma/rdma_endpoint.h index 6d6ac391cd..eda5e558ef 100644 --- a/src/brpc/rdma/rdma_endpoint.h +++ b/src/brpc/rdma/rdma_endpoint.h @@ -87,6 +87,8 @@ struct RdmaResource { ibv_cq* send_cq{nullptr}; ibv_cq* recv_cq{nullptr}; ibv_comp_channel* comp_channel{nullptr}; + // Largest message posted inline on `qp'. + uint32_t max_inline_data{0}; RdmaResource() = default; ~RdmaResource(); DISALLOW_COPY_AND_ASSIGN(RdmaResource); From 03d303c46d497cf4a028f3066875a442bc99d9c8 Mon Sep 17 00:00:00 2001 From: Andrei-Alexandru Bleortu <46914650+alxrxs@users.noreply.github.com> Date: Sun, 4 Oct 2026 00:30:07 +0300 Subject: [PATCH 2/6] Refine the inline size limit and the QP creation fallback The 236-byte limit keeps an inlined Send within the 256-byte BlueFlame buffer of mlx5 NICs. Apply it only on Mellanox/NVIDIA devices (vendor ID 0x02c9, kept from the device query GlobalRdmaInitializeOrDie already does); on other devices the inline size the QP was granted is the limit. If QP creation with 236 bytes of inline data fails, try 64 bytes before falling back to none. irdma (Intel E810) rejects requests above 101 bytes, so on those NICs nothing was inlined. Generated-by: Claude Code (Claude Opus 5.5) Co-Authored-By: Claude Opus 5.5 --- src/brpc/rdma/rdma_endpoint.cpp | 18 +++++++++++++----- src/brpc/rdma/rdma_helper.cpp | 6 ++++++ src/brpc/rdma/rdma_helper.h | 3 +++ 3 files changed, 22 insertions(+), 5 deletions(-) diff --git a/src/brpc/rdma/rdma_endpoint.cpp b/src/brpc/rdma/rdma_endpoint.cpp index 1840500778..5260ad2914 100644 --- a/src/brpc/rdma/rdma_endpoint.cpp +++ b/src/brpc/rdma/rdma_endpoint.cpp @@ -80,10 +80,11 @@ extern const size_t RESERVED_WR_NUM = 3; // The local recv block size, set during GlobalInitialize. uint32_t g_rdma_recv_block_size = 0; -// Largest message sent inline, and the inline size asked for at QP -// creation: 256 - 16 (ctrl) - 4 (inline header). A bigger inlined WQE no -// longer fits the 256-byte BlueFlame buffer of current mlx5 NICs. +// Inline size asked for at QP creation, and on Mellanox/NVIDIA NICs the +// largest message sent inline: 256 - 16 (ctrl) - 4 (inline header). A bigger +// inlined WQE no longer fits the 256-byte BlueFlame buffer of mlx5 NICs. static const uint32_t BF_MAX_INLINE_DATA = 256 - 16 - 4; +static const uint32_t MELLANOX_VENDOR_ID = 0x02c9; static const uint8_t MAX_HOP_LIMIT = 16; static const uint8_t TIMEOUT = 14; static const uint8_t RETRY_CNT = 7; @@ -1154,6 +1155,11 @@ static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, attr.cap.max_inline_data = BF_MAX_INLINE_DATA; attr.qp_type = IBV_QPT_RC; ibv_qp* qp = IbvCreateQp(GetRdmaPd(), &attr); + if (qp == nullptr) { + // Some devices take less inline data (irdma: 101 bytes), try 64 + attr.cap.max_inline_data = 64; + qp = IbvCreateQp(GetRdmaPd(), &attr); + } if (qp == nullptr) { // The device may not support inline data, try again without it attr.cap.max_inline_data = 0; @@ -1161,8 +1167,10 @@ static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, } if (qp != nullptr) { // ibv_create_qp writes the granted inline data size back into attr - *max_inline_data = - std::min(attr.cap.max_inline_data, BF_MAX_INLINE_DATA); + *max_inline_data = attr.cap.max_inline_data; + if (GetRdmaVendorId() == MELLANOX_VENDOR_ID) { + *max_inline_data = std::min(*max_inline_data, BF_MAX_INLINE_DATA); + } } return qp; } diff --git a/src/brpc/rdma/rdma_helper.cpp b/src/brpc/rdma/rdma_helper.cpp index 38bd58cac4..7e671a85ba 100644 --- a/src/brpc/rdma/rdma_helper.cpp +++ b/src/brpc/rdma/rdma_helper.cpp @@ -84,6 +84,7 @@ static uint8_t g_gid_index = 0; static ibv_gid g_gid; static uint16_t g_lid; static int g_max_sge = 0; +static uint32_t g_vendor_id = 0; static uint8_t g_port_num = 1; static int g_comp_vector_index = 0; @@ -550,6 +551,7 @@ static void GlobalRdmaInitializeOrDieImpl() { } else { g_max_sge = attr.max_sge; } + g_vendor_id = attr.vendor_id; // Initialize RDMA memory pool (block_pool) butil::SetDefaultBlockSize(GetRdmaBlockSize()); @@ -644,6 +646,10 @@ int GetRdmaMaxSge() { return g_max_sge; } +uint32_t GetRdmaVendorId() { + return g_vendor_id; +} + int GetRdmaCompVector() { if (!g_context) { return 0; diff --git a/src/brpc/rdma/rdma_helper.h b/src/brpc/rdma/rdma_helper.h index 052763325b..2383d2d8b4 100644 --- a/src/brpc/rdma/rdma_helper.h +++ b/src/brpc/rdma/rdma_helper.h @@ -77,6 +77,9 @@ uint8_t GetRdmaPortNum(); // Get max_sge supported by the device int GetRdmaMaxSge(); +// Get the vendor ID of the device +uint32_t GetRdmaVendorId(); + // Get suggested comp_vector for a new CQ int GetCompVector(); From e6ce6cd3c93e9c8ab649fd122c452ae0689fba05 Mon Sep 17 00:00:00 2001 From: Andrei-Alexandru Bleortu <46914650+alxrxs@users.noreply.github.com> Date: Sun, 4 Oct 2026 02:52:56 +0300 Subject: [PATCH 3/6] Try known inline limits, then step down After a refused 236-byte inline request, QP creation was retried with 64 bytes, then with none. Devices take less than 236 by different amounts: Intel's irdma refuses more than 216, 101 or 48 bytes depending on the generation (101 on an E810, 48 on an X722), Alibaba's erdma more than 96. With 64 as the only middle step, an E810 or erdma QP got 64 bytes, a newer irdma QP 64 instead of 216, and an X722 QP none. The verbs API cannot report the limit, so step down to it. A device whose kernel driver has a fixed limit, found by the vendor ID kept from the device query GlobalRdmaInitializeOrDie already does, starts there: 216, then 101, then 48 on Intel, 96 on Alibaba, each capped at 236. Other devices start at 236. Each refused size is followed by the next one 16 bytes smaller, down to 0, so a failure unrelated to inline data still ends as before, after a few more attempts at QP creation. Devices that accept 236 (mlx5, rxe) are unchanged. libfabric's verbs provider also probes the limit by trial QP creation (vrb_find_max_inline()). Generated-by: Claude Code (Claude Opus 5.5) Co-Authored-By: Claude Opus 5.5 --- src/brpc/rdma/rdma_endpoint.cpp | 49 ++++++++++++++++++++++++++------- 1 file changed, 39 insertions(+), 10 deletions(-) diff --git a/src/brpc/rdma/rdma_endpoint.cpp b/src/brpc/rdma/rdma_endpoint.cpp index 5260ad2914..18b3c72aa0 100644 --- a/src/brpc/rdma/rdma_endpoint.cpp +++ b/src/brpc/rdma/rdma_endpoint.cpp @@ -85,6 +85,20 @@ uint32_t g_rdma_recv_block_size = 0; // inlined WQE no longer fits the 256-byte BlueFlame buffer of mlx5 NICs. static const uint32_t BF_MAX_INLINE_DATA = 256 - 16 - 4; static const uint32_t MELLANOX_VENDOR_ID = 0x02c9; +// The verbs API cannot report the inline data limit. Devices whose kernel +// driver has a fixed limit try it first: Intel irdma 216, 101 or 48 depending +// on the generation (101 on the E810, 48 on the X722; +// drivers/infiniband/hw/irdma/ig3rdma_hw.h, user.h, i40iw_hw.h), Alibaba +// erdma 96 (drivers/infiniband/hw/erdma/erdma_verbs.h). Other devices start +// at BF_MAX_INLINE_DATA, and any refused size steps down by INLINE_DATA_STEP +// until one is accepted. libfabric's verbs provider also probes the limit +// (vrb_find_max_inline()). +static const struct { + uint32_t vendor_id; + uint32_t max_inline; +} KNOWN_INLINE_LIMITS[] = { + {0x8086, 216}, {0x8086, 101}, {0x8086, 48}, {0x1ded, 96}}; +static const uint32_t INLINE_DATA_STEP = 16; static const uint8_t MAX_HOP_LIMIT = 16; static const uint8_t TIMEOUT = 14; static const uint8_t RETRY_CNT = 7; @@ -1142,8 +1156,28 @@ int RdmaEndpoint::PostRecv(uint32_t num, bool zerocopy) { return 0; } +// Inline size to request: on the first try, size capped at the vendor's first +// known limit; after a refusal, the vendor's next smaller known limit, else +// one step smaller. +static uint32_t InlineDataToRequest(uint32_t vendor_id, uint32_t size, + bool refused) { + for (const auto& known : KNOWN_INLINE_LIMITS) { + if (known.vendor_id == vendor_id && + (!refused || known.max_inline < size)) { + return std::min(known.max_inline, size); + } + } + if (refused) { + size = size > INLINE_DATA_STEP ? size - INLINE_DATA_STEP : 0; + } + return size; +} + static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, uint32_t rq_size, uint32_t* max_inline_data) { + const uint32_t vendor_id = GetRdmaVendorId(); + uint32_t inline_size = + InlineDataToRequest(vendor_id, BF_MAX_INLINE_DATA, false); ibv_qp_init_attr attr; memset(&attr, 0, sizeof(attr)); attr.send_cq = send_cq; @@ -1152,23 +1186,18 @@ static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, attr.cap.max_recv_wr = rq_size; attr.cap.max_send_sge = GetRdmaMaxSge(); attr.cap.max_recv_sge = 1; - attr.cap.max_inline_data = BF_MAX_INLINE_DATA; + attr.cap.max_inline_data = inline_size; attr.qp_type = IBV_QPT_RC; ibv_qp* qp = IbvCreateQp(GetRdmaPd(), &attr); - if (qp == nullptr) { - // Some devices take less inline data (irdma: 101 bytes), try 64 - attr.cap.max_inline_data = 64; - qp = IbvCreateQp(GetRdmaPd(), &attr); - } - if (qp == nullptr) { - // The device may not support inline data, try again without it - attr.cap.max_inline_data = 0; + while (qp == nullptr && inline_size > 0) { + inline_size = InlineDataToRequest(vendor_id, inline_size, true); + attr.cap.max_inline_data = inline_size; qp = IbvCreateQp(GetRdmaPd(), &attr); } if (qp != nullptr) { // ibv_create_qp writes the granted inline data size back into attr *max_inline_data = attr.cap.max_inline_data; - if (GetRdmaVendorId() == MELLANOX_VENDOR_ID) { + if (vendor_id == MELLANOX_VENDOR_ID) { *max_inline_data = std::min(*max_inline_data, BF_MAX_INLINE_DATA); } } From 9b2a7c8a3ce60691e18bd5bec4786f98e0be7f21 Mon Sep 17 00:00:00 2001 From: Andrei-Alexandru Bleortu <46914650+alxrxs@users.noreply.github.com> Date: Sun, 4 Oct 2026 09:48:16 +0300 Subject: [PATCH 4/6] Step down the inline size only on EINVAL, and add unit tests AllocateQp() retried with a smaller inline size after any ibv_create_qp failure, so a failure unrelated to inline data, such as ENOMEM or the device's QP limit, led to up to 16 failed calls, and the caller's PLOG reported the errno of the last attempt. Step down only when the failed call left errno == EINVAL, which is how providers refuse an inline size; any other failure is returned at once with its errno. Each step is logged under FLAGS_rdma_trace_verbose. QP creation moves into CreateQpWithInlineData(), which takes the vendor ID as a parameter. It and InlineDataToRequest() are no longer static, so that brpc_rdma_unittest can test them with a stubbed IbvCreateQp: the first request per vendor, the Mellanox cap at 236 bytes, Intel 216, 101 and 48, Alibaba 96, an unknown device stepping down to 92, a non-EINVAL failure stopping at once, and a device refusing every size ending at 0. Generated-by: Claude Code (Claude Opus 5.5) Co-Authored-By: Claude Opus 5.5 --- src/brpc/rdma/rdma_endpoint.cpp | 55 ++++++++------ test/brpc_rdma_unittest.cpp | 122 ++++++++++++++++++++++++++++++++ 2 files changed, 156 insertions(+), 21 deletions(-) diff --git a/src/brpc/rdma/rdma_endpoint.cpp b/src/brpc/rdma/rdma_endpoint.cpp index 18b3c72aa0..d77ab409cd 100644 --- a/src/brpc/rdma/rdma_endpoint.cpp +++ b/src/brpc/rdma/rdma_endpoint.cpp @@ -1158,9 +1158,8 @@ int RdmaEndpoint::PostRecv(uint32_t num, bool zerocopy) { // Inline size to request: on the first try, size capped at the vendor's first // known limit; after a refusal, the vendor's next smaller known limit, else -// one step smaller. -static uint32_t InlineDataToRequest(uint32_t vendor_id, uint32_t size, - bool refused) { +// one step smaller. Not static: exposed for UT. +uint32_t InlineDataToRequest(uint32_t vendor_id, uint32_t size, bool refused) { for (const auto& known : KNOWN_INLINE_LIMITS) { if (known.vendor_id == vendor_id && (!refused || known.max_inline < size)) { @@ -1173,11 +1172,38 @@ static uint32_t InlineDataToRequest(uint32_t vendor_id, uint32_t size, return size; } -static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, - uint32_t rq_size, uint32_t* max_inline_data) { - const uint32_t vendor_id = GetRdmaVendorId(); +// Creates a QP with attr, asking for BF_MAX_INLINE_DATA bytes of inline data, +// or less if the device refuses that size, and sets *max_inline_data to the +// largest message to post inline on it. Not static: exposed for UT. +ibv_qp* CreateQpWithInlineData(ibv_pd* pd, ibv_qp_init_attr* attr, + uint32_t vendor_id, uint32_t* max_inline_data) { uint32_t inline_size = InlineDataToRequest(vendor_id, BF_MAX_INLINE_DATA, false); + attr->cap.max_inline_data = inline_size; + ibv_qp* qp = IbvCreateQp(pd, attr); + // A refused inline size fails with EINVAL; any other failure is left to + // the caller, which reports its errno. + while (qp == nullptr && errno == EINVAL && inline_size > 0) { + uint32_t next = InlineDataToRequest(vendor_id, inline_size, true); + LOG_IF(INFO, FLAGS_rdma_trace_verbose) + << "ibv_create_qp refused " << inline_size + << " bytes of inline data, trying " << next; + inline_size = next; + attr->cap.max_inline_data = inline_size; + qp = IbvCreateQp(pd, attr); + } + if (qp != nullptr) { + // ibv_create_qp writes the granted inline data size back into attr + *max_inline_data = attr->cap.max_inline_data; + if (vendor_id == MELLANOX_VENDOR_ID) { + *max_inline_data = std::min(*max_inline_data, BF_MAX_INLINE_DATA); + } + } + return qp; +} + +static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, + uint32_t rq_size, uint32_t* max_inline_data) { ibv_qp_init_attr attr; memset(&attr, 0, sizeof(attr)); attr.send_cq = send_cq; @@ -1186,22 +1212,9 @@ static ibv_qp* AllocateQp(ibv_cq* send_cq, ibv_cq* recv_cq, uint32_t sq_size, attr.cap.max_recv_wr = rq_size; attr.cap.max_send_sge = GetRdmaMaxSge(); attr.cap.max_recv_sge = 1; - attr.cap.max_inline_data = inline_size; attr.qp_type = IBV_QPT_RC; - ibv_qp* qp = IbvCreateQp(GetRdmaPd(), &attr); - while (qp == nullptr && inline_size > 0) { - inline_size = InlineDataToRequest(vendor_id, inline_size, true); - attr.cap.max_inline_data = inline_size; - qp = IbvCreateQp(GetRdmaPd(), &attr); - } - if (qp != nullptr) { - // ibv_create_qp writes the granted inline data size back into attr - *max_inline_data = attr.cap.max_inline_data; - if (vendor_id == MELLANOX_VENDOR_ID) { - *max_inline_data = std::min(*max_inline_data, BF_MAX_INLINE_DATA); - } - } - return qp; + return CreateQpWithInlineData(GetRdmaPd(), &attr, GetRdmaVendorId(), + max_inline_data); } static RdmaResource* AllocateQpCq(uint16_t sq_size, uint16_t rq_size) { diff --git a/test/brpc_rdma_unittest.cpp b/test/brpc_rdma_unittest.cpp index 8886714268..546cc49b53 100644 --- a/test/brpc_rdma_unittest.cpp +++ b/test/brpc_rdma_unittest.cpp @@ -78,6 +78,11 @@ extern int (*IbvDestroyQp)(ibv_qp*); extern butil::atomic g_rdma_available; extern bool g_skip_rdma_init; extern bool g_fail_resource_alloc_for_test; +extern uint32_t InlineDataToRequest(uint32_t vendor_id, uint32_t size, + bool refused); +extern ibv_qp* CreateQpWithInlineData(ibv_pd* pd, ibv_qp_init_attr* attr, + uint32_t vendor_id, + uint32_t* max_inline_data); } // namespace rdma } // namespace brpc @@ -3133,6 +3138,123 @@ INSTANTIATE_TEST_SUITE_P( return std::string("v") + std::to_string(info.param); }); +static const uint32_t MELLANOX = 0x02c9; +static const uint32_t INTEL = 0x8086; +static const uint32_t ALIBABA = 0x1ded; +static const uint32_t OTHER = 0xffffff; // e.g. rxe + +TEST(RdmaInlineDataTest, inline_data_to_request) { + // First try: the project's 236 bytes, capped at the vendor's first + // known limit. + EXPECT_EQ(236u, rdma::InlineDataToRequest(MELLANOX, 236, false)); + EXPECT_EQ(216u, rdma::InlineDataToRequest(INTEL, 236, false)); + EXPECT_EQ(96u, rdma::InlineDataToRequest(ALIBABA, 236, false)); + EXPECT_EQ(236u, rdma::InlineDataToRequest(OTHER, 236, false)); + EXPECT_EQ(64u, rdma::InlineDataToRequest(INTEL, 64, false)); + // After a refusal: the vendor's next smaller known limit, else 16 less. + EXPECT_EQ(101u, rdma::InlineDataToRequest(INTEL, 216, true)); + EXPECT_EQ(48u, rdma::InlineDataToRequest(INTEL, 101, true)); + EXPECT_EQ(32u, rdma::InlineDataToRequest(INTEL, 48, true)); + EXPECT_EQ(80u, rdma::InlineDataToRequest(ALIBABA, 96, true)); + EXPECT_EQ(220u, rdma::InlineDataToRequest(MELLANOX, 236, true)); + EXPECT_EQ(220u, rdma::InlineDataToRequest(OTHER, 236, true)); + EXPECT_EQ(0u, rdma::InlineDataToRequest(OTHER, 12, true)); + EXPECT_EQ(0u, rdma::InlineDataToRequest(OTHER, 0, true)); +} + +// Stub for IbvCreateQp: records every inline size requested, refuses a request +// above g_stub_limit (every request if g_stub_limit < 0) with g_stub_errno, and +// otherwise writes back g_stub_grant as the granted size (0: the request). +static std::vector g_stub_requests; +static int64_t g_stub_limit = 0; +static int g_stub_errno = EINVAL; +static uint32_t g_stub_grant = 0; +static ibv_qp g_stub_qp; + +static ibv_qp* StubCreateQp(ibv_pd*, ibv_qp_init_attr* attr) { + g_stub_requests.push_back(attr->cap.max_inline_data); + if (g_stub_limit < 0 || attr->cap.max_inline_data > g_stub_limit) { + errno = g_stub_errno; + return nullptr; + } + if (g_stub_grant != 0) { + attr->cap.max_inline_data = g_stub_grant; + } + return &g_stub_qp; +} + +// Runs CreateQpWithInlineData() against the stub. Returns whether a QP was +// created; *limit gets the inline limit, *err the errno left behind. +static bool CreateWithStub(uint32_t vendor_id, int64_t device_limit, + int refuse_errno, uint32_t grant, uint32_t* limit, + int* err) { + ibv_qp* (*saved)(ibv_pd*, ibv_qp_init_attr*) = rdma::IbvCreateQp; + rdma::IbvCreateQp = StubCreateQp; + g_stub_requests.clear(); + g_stub_limit = device_limit; + g_stub_errno = refuse_errno; + g_stub_grant = grant; + ibv_qp_init_attr attr; + memset(&attr, 0, sizeof(attr)); + *limit = 0; + errno = 0; + ibv_qp* qp = rdma::CreateQpWithInlineData(NULL, &attr, vendor_id, limit); + *err = errno; + rdma::IbvCreateQp = saved; + return qp != nullptr; +} + +TEST(RdmaInlineDataTest, create_qp_with_inline_data) { + uint32_t limit = 0; + int err = 0; + typedef std::vector Sizes; + + // Mellanox: 236 accepted on the first try; a larger grant (316) is capped + // at 236 so that the inlined WQE fits the BlueFlame buffer. + ASSERT_TRUE(CreateWithStub(MELLANOX, 1024, EINVAL, 316, &limit, &err)); + EXPECT_EQ(Sizes({236}), g_stub_requests); + EXPECT_EQ(236u, limit); + + // Another device keeps a larger grant (rxe grants 16 bytes per send SGE). + ASSERT_TRUE(CreateWithStub(OTHER, 1024, EINVAL, 512, &limit, &err)); + EXPECT_EQ(Sizes({236}), g_stub_requests); + EXPECT_EQ(512u, limit); + + // Intel irdma, whose limit is 216, 101 or 48 depending on the generation. + ASSERT_TRUE(CreateWithStub(INTEL, 216, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({216}), g_stub_requests); + EXPECT_EQ(216u, limit); + ASSERT_TRUE(CreateWithStub(INTEL, 101, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({216, 101}), g_stub_requests); + EXPECT_EQ(101u, limit); + ASSERT_TRUE(CreateWithStub(INTEL, 48, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({216, 101, 48}), g_stub_requests); + EXPECT_EQ(48u, limit); + + // Alibaba erdma (96). + ASSERT_TRUE(CreateWithStub(ALIBABA, 96, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({96}), g_stub_requests); + EXPECT_EQ(96u, limit); + + // An unknown device that takes at most 101 bytes: 16 bytes less each time. + ASSERT_TRUE(CreateWithStub(OTHER, 101, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({236, 220, 204, 188, 172, 156, 140, 124, 108, 92}), + g_stub_requests); + EXPECT_EQ(92u, limit); + + // A failure that is not EINVAL stops at once and leaves its errno. + ASSERT_FALSE(CreateWithStub(OTHER, -1, ENOMEM, 0, &limit, &err)); + EXPECT_EQ(Sizes({236}), g_stub_requests); + EXPECT_EQ(ENOMEM, err); + + // A device that refuses every size with EINVAL ends with a request of 0. + ASSERT_FALSE(CreateWithStub(OTHER, -1, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({236, 220, 204, 188, 172, 156, 140, 124, 108, 92, 76, 60, + 44, 28, 12, 0}), + g_stub_requests); + EXPECT_EQ(EINVAL, err); +} + #endif // if BRPC_WITH_RDMA int main(int argc, char* argv[]) { From 8e40af7ffc66c61b97f653db8d1197c77ace5aee Mon Sep 17 00:00:00 2001 From: Andrei-Alexandru Bleortu <46914650+alxrxs@users.noreply.github.com> Date: Sun, 4 Oct 2026 13:34:15 +0300 Subject: [PATCH 5/6] Factor the inline send decision out and test it CutFromIOBufList() inlines a message only when it fits the inline size the QP was granted and all of its blocks come from the block pool; user registered memory may be device memory, which the CPU cannot copy from. Move that decision into ShouldPostInline(), exposed for UT, and test it in brpc_rdma_unittest: a pool-backed message at the granted size is inlined, one byte more is not, and a message in user registered memory never is, however small. CutFromIOBufList() itself needs an initialized RDMA device: it returns early in the unit test binary, sizes its SGE list from the device and looks blocks up in the registered block pool. So the decision is tested on its own; the send path is otherwise unchanged. Generated-by: Claude Code (Claude Opus 5.5) Co-Authored-By: Claude Opus 5.5 --- src/brpc/rdma/rdma_endpoint.cpp | 13 +++++++++---- test/brpc_rdma_unittest.cpp | 19 +++++++++++++++++++ 2 files changed, 28 insertions(+), 4 deletions(-) diff --git a/src/brpc/rdma/rdma_endpoint.cpp b/src/brpc/rdma/rdma_endpoint.cpp index d77ab409cd..fed2b616c8 100644 --- a/src/brpc/rdma/rdma_endpoint.cpp +++ b/src/brpc/rdma/rdma_endpoint.cpp @@ -849,6 +849,14 @@ friend class RdmaEndpoint; } }; +// Whether to post a message of len bytes inline. Only a message from the block +// pool (host memory) that fits the inline size the QP was granted is copied +// into the WQE. User registered memory may be device memory, which the CPU +// cannot copy from, so it is never inlined. Not static: exposed for UT. +bool ShouldPostInline(bool in_pool, size_t len, uint32_t max_inline_data) { + return in_pool && len <= max_inline_data; +} + // Note this function is coupled with the implementation of IOBuf ssize_t RdmaEndpoint::CutFromIOBufList(butil::IOBuf** from, size_t ndata) { if (BAIDU_UNLIKELY(g_skip_rdma_init)) { @@ -916,10 +924,7 @@ ssize_t RdmaEndpoint::CutFromIOBufList(butil::IOBuf** from, size_t ndata) { } wr.num_sge = sge_index; - // A small message from the block pool (host memory) is copied into - // the WQE. User registered memory may be device memory, which the - // CPU cannot copy from, so it is never inlined. - if (in_pool && this_len <= _resource->max_inline_data) { + if (ShouldPostInline(in_pool, this_len, _resource->max_inline_data)) { wr.send_flags |= IBV_SEND_INLINE; } diff --git a/test/brpc_rdma_unittest.cpp b/test/brpc_rdma_unittest.cpp index 546cc49b53..f5218c4041 100644 --- a/test/brpc_rdma_unittest.cpp +++ b/test/brpc_rdma_unittest.cpp @@ -83,6 +83,8 @@ extern uint32_t InlineDataToRequest(uint32_t vendor_id, uint32_t size, extern ibv_qp* CreateQpWithInlineData(ibv_pd* pd, ibv_qp_init_attr* attr, uint32_t vendor_id, uint32_t* max_inline_data); +extern bool ShouldPostInline(bool in_pool, size_t len, + uint32_t max_inline_data); } // namespace rdma } // namespace brpc @@ -3255,6 +3257,23 @@ TEST(RdmaInlineDataTest, create_qp_with_inline_data) { EXPECT_EQ(EINVAL, err); } +// The send-path decision of CutFromIOBufList(), which itself needs an +// initialized RDMA device. +TEST(RdmaInlineDataTest, should_post_inline) { + const uint32_t granted = 236; + // A message from the block pool at the granted size is inlined, one byte + // more is not. + EXPECT_TRUE(rdma::ShouldPostInline(true, granted, granted)); + EXPECT_FALSE(rdma::ShouldPostInline(true, granted + 1, granted)); + EXPECT_TRUE(rdma::ShouldPostInline(true, 1, granted)); + // A message in user registered memory, which may be device memory, is + // never inlined, however small. + EXPECT_FALSE(rdma::ShouldPostInline(false, 1, granted)); + EXPECT_FALSE(rdma::ShouldPostInline(false, granted, granted)); + // A QP granted no inline data inlines nothing. + EXPECT_FALSE(rdma::ShouldPostInline(true, 1, 0)); +} + #endif // if BRPC_WITH_RDMA int main(int argc, char* argv[]) { From 88871736977dcf9611cea19774755007a1da917a Mon Sep 17 00:00:00 2001 From: Andrei-Alexandru Bleortu <46914650+alxrxs@users.noreply.github.com> Date: Tue, 6 Oct 2026 13:55:05 +0300 Subject: [PATCH 6/6] Reuse the inline size of the last QP for new QPs All QPs are created on the same device with the same attributes apart from their queue sizes, so a device that refuses the first inline size made every one of the rdma_prepared_qp_cnt QPs repeat the same step-down (216, 101, 48 on an Intel device limited to 48 bytes). Remember the size the last QP was created with and start there. If that size is refused, the EINVAL step-down continues from it as before. Each QP still takes its own granted size from attr.cap.max_inline_data. Generated-by: Claude Code (Claude Opus 5.5) Co-Authored-By: Claude Opus 5.5 --- src/brpc/rdma/rdma_endpoint.cpp | 17 +++++++++--- test/brpc_rdma_unittest.cpp | 46 +++++++++++++++++++++++++++++++-- 2 files changed, 58 insertions(+), 5 deletions(-) diff --git a/src/brpc/rdma/rdma_endpoint.cpp b/src/brpc/rdma/rdma_endpoint.cpp index fed2b616c8..0dc41c9dd1 100644 --- a/src/brpc/rdma/rdma_endpoint.cpp +++ b/src/brpc/rdma/rdma_endpoint.cpp @@ -1177,13 +1177,23 @@ uint32_t InlineDataToRequest(uint32_t vendor_id, uint32_t size, bool refused) { return size; } -// Creates a QP with attr, asking for BF_MAX_INLINE_DATA bytes of inline data, -// or less if the device refuses that size, and sets *max_inline_data to the +// The inline size the last QP was created with (UINT32_MAX: none yet). All +// QPs use the same device and attributes apart from their queue sizes, so +// later QPs start here instead of repeating the step-down. Not static: +// exposed for UT. +butil::atomic g_inline_data_request(UINT32_MAX); + +// Creates a QP with attr, asking for the inline size the last QP was created +// with (at first BF_MAX_INLINE_DATA, capped by the vendor's known limit), or +// less if the device refuses that size, and sets *max_inline_data to the // largest message to post inline on it. Not static: exposed for UT. ibv_qp* CreateQpWithInlineData(ibv_pd* pd, ibv_qp_init_attr* attr, uint32_t vendor_id, uint32_t* max_inline_data) { uint32_t inline_size = - InlineDataToRequest(vendor_id, BF_MAX_INLINE_DATA, false); + g_inline_data_request.load(butil::memory_order_relaxed); + if (inline_size == UINT32_MAX) { + inline_size = InlineDataToRequest(vendor_id, BF_MAX_INLINE_DATA, false); + } attr->cap.max_inline_data = inline_size; ibv_qp* qp = IbvCreateQp(pd, attr); // A refused inline size fails with EINVAL; any other failure is left to @@ -1198,6 +1208,7 @@ ibv_qp* CreateQpWithInlineData(ibv_pd* pd, ibv_qp_init_attr* attr, qp = IbvCreateQp(pd, attr); } if (qp != nullptr) { + g_inline_data_request.store(inline_size, butil::memory_order_relaxed); // ibv_create_qp writes the granted inline data size back into attr *max_inline_data = attr->cap.max_inline_data; if (vendor_id == MELLANOX_VENDOR_ID) { diff --git a/test/brpc_rdma_unittest.cpp b/test/brpc_rdma_unittest.cpp index f5218c4041..6ec54ed90b 100644 --- a/test/brpc_rdma_unittest.cpp +++ b/test/brpc_rdma_unittest.cpp @@ -83,6 +83,7 @@ extern uint32_t InlineDataToRequest(uint32_t vendor_id, uint32_t size, extern ibv_qp* CreateQpWithInlineData(ibv_pd* pd, ibv_qp_init_attr* attr, uint32_t vendor_id, uint32_t* max_inline_data); +extern butil::atomic g_inline_data_request; extern bool ShouldPostInline(bool in_pool, size_t len, uint32_t max_inline_data); } // namespace rdma @@ -3186,10 +3187,14 @@ static ibv_qp* StubCreateQp(ibv_pd*, ibv_qp_init_attr* attr) { } // Runs CreateQpWithInlineData() against the stub. Returns whether a QP was -// created; *limit gets the inline limit, *err the errno left behind. +// created; *limit gets the inline limit, *err the errno left behind. Unless +// keep_cached, forgets the inline size cached by earlier QP creations first. static bool CreateWithStub(uint32_t vendor_id, int64_t device_limit, int refuse_errno, uint32_t grant, uint32_t* limit, - int* err) { + int* err, bool keep_cached = false) { + if (!keep_cached) { + rdma::g_inline_data_request.store(UINT32_MAX); + } ibv_qp* (*saved)(ibv_pd*, ibv_qp_init_attr*) = rdma::IbvCreateQp; rdma::IbvCreateQp = StubCreateQp; g_stub_requests.clear(); @@ -3257,6 +3262,43 @@ TEST(RdmaInlineDataTest, create_qp_with_inline_data) { EXPECT_EQ(EINVAL, err); } +// Later QPs start at the inline size the last QP was created with. +TEST(RdmaInlineDataTest, create_qp_reuses_inline_size) { + uint32_t limit = 0; + int err = 0; + typedef std::vector Sizes; + + // The first QP on an Intel device limited to 48 bytes steps down; the next + // asks for 48 straight away. + ASSERT_TRUE(CreateWithStub(INTEL, 48, EINVAL, 0, &limit, &err)); + EXPECT_EQ(Sizes({216, 101, 48}), g_stub_requests); + ASSERT_TRUE(CreateWithStub(INTEL, 48, EINVAL, 0, &limit, &err, true)); + EXPECT_EQ(Sizes({48}), g_stub_requests); + EXPECT_EQ(48u, limit); + + // If the cached size is refused (a QP whose other attributes leave less + // room), the step-down continues from it, and the new size is kept. + ASSERT_TRUE(CreateWithStub(INTEL, 40, EINVAL, 0, &limit, &err, true)); + EXPECT_EQ(Sizes({48, 32}), g_stub_requests); + EXPECT_EQ(32u, limit); + ASSERT_TRUE(CreateWithStub(INTEL, 40, EINVAL, 0, &limit, &err, true)); + EXPECT_EQ(Sizes({32}), g_stub_requests); + + // A failure other than EINVAL leaves the cached size as it was. + ASSERT_FALSE(CreateWithStub(INTEL, -1, ENOMEM, 0, &limit, &err, true)); + EXPECT_EQ(Sizes({32}), g_stub_requests); + ASSERT_TRUE(CreateWithStub(INTEL, 40, EINVAL, 0, &limit, &err, true)); + EXPECT_EQ(Sizes({32}), g_stub_requests); + + // Mellanox: 236 is accepted and asked for again; the grant is still capped. + ASSERT_TRUE(CreateWithStub(MELLANOX, 1024, EINVAL, 316, &limit, &err)); + ASSERT_TRUE(CreateWithStub(MELLANOX, 1024, EINVAL, 316, &limit, &err, + true)); + EXPECT_EQ(Sizes({236}), g_stub_requests); + EXPECT_EQ(236u, limit); + rdma::g_inline_data_request.store(UINT32_MAX); +} + // The send-path decision of CutFromIOBufList(), which itself needs an // initialized RDMA device. TEST(RdmaInlineDataTest, should_post_inline) {