From 2ed6760447479fa255a78e85b18bfd71304b9860 Mon Sep 17 00:00:00 2001 From: Vitaliy Filippov Date: Tue, 17 Feb 2026 02:32:12 +0300 Subject: [PATCH] Remove ODP support --- docs/config/network.en.md | 45 +++++-------- docs/config/network.ru.md | 48 +++++--------- docs/config/src/network.yml | 85 ++++++++++-------------- src/client/messenger.cpp | 3 +- src/client/messenger.h | 3 - src/client/msgr_rdma.cpp | 128 ++++++------------------------------ src/client/msgr_rdma.h | 6 +- src/client/msgr_rdmacm.cpp | 1 - 8 files changed, 93 insertions(+), 226 deletions(-) diff --git a/docs/config/network.en.md b/docs/config/network.en.md index ef5a19b7..cdf47d4f 100644 --- a/docs/config/network.en.md +++ b/docs/config/network.en.md @@ -22,7 +22,6 @@ between clients, OSDs and etcd. - [rdma_max_msg](#rdma_max_msg) - [rdma_max_recv](#rdma_max_recv) - [rdma_max_send](#rdma_max_send) -- [rdma_odp](#rdma_odp) - [peer_connect_interval](#peer_connect_interval) - [peer_connect_timeout](#peer_connect_timeout) - [osd_idle_timeout](#osd_idle_timeout) @@ -102,11 +101,6 @@ found or if `osd_network` is not specified. Auto-selection is also unsupported with old libibverbs < v32, like in Debian 10 Buster or CentOS 7. -Vitastor supports all adapters, even ones without ODP support, like -Mellanox ConnectX-3 and non-Mellanox cards. Versions up to Vitastor -1.2.0 required ODP which is only present in Mellanox ConnectX >= 4. -See also [rdma_odp](#rdma_odp). - Run `ibv_devinfo -v` as root to list available RDMA devices and their features. @@ -116,6 +110,23 @@ the manual of your network vendor for details about setting up the switch for RoCEv2 correctly. Usually it means setting up Lossless Ethernet with PFC (Priority Flow Control) and ECN (Explicit Congestion Notification). +Vitastor supports all adapters, even ones without ODP (On-Demand Paging) +support, like Mellanox ConnectX-3 and non-Mellanox cards. ODP is only present +in Mellanox ConnectX >= 4 adapters and allows to skip memory registration +for RDMA and thus, in theory, avoid memory copying. + +Versions up to Vitastor 1.2.0 required ODP, then it was disabled by default, +but it was still supported up to 3.0.3. Now ODP support is removed because it +actually only hurts performance: an example 3-node cluster with 8 NVMe in each +node and 2*25 GBit/s ConnectX-6 RDMA network pushed 3950000 read iops without +ODP, but only 239000 iops with ODP. + +This happens because Mellanox ODP implementation seems to be based on +message retransmissions when the adapter doesn't know about the buffer yet - +it likely uses standard "RNR retransmissions" (RNR = receiver not ready) +which is generally slow in RDMA/RoCE networks. Here's a presentation about +it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf + ## rdma_port_num - Type: integer @@ -187,28 +198,6 @@ less than `rdma_max_recv` so the receiving side doesn't run out of buffers. Doesn't affect memory usage - additional memory isn't allocated for send operations. -## rdma_odp - -- Type: boolean -- Default: false - -Use RDMA with On-Demand Paging. ODP is currently only available on Mellanox -ConnectX-4 and newer adapters. ODP allows to not register memory explicitly -for RDMA adapter to be able to use it. This, in turn, allows to skip memory -copying during sending. One would think this should improve performance, but -**in reality** RDMA performance with ODP is **drastically** worse. Example -3-node cluster with 8 NVMe in each node and 2*25 GBit/s ConnectX-6 RDMA network -without ODP pushes 3950000 read iops, but only 239000 iops with ODP... - -This happens because Mellanox ODP implementation seems to be based on -message retransmissions when the adapter doesn't know about the buffer yet - -it likely uses standard "RNR retransmissions" (RNR = receiver not ready) -which is generally slow in RDMA/RoCE networks. Here's a presentation about -it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf - -ODP support is retained in the code just in case a good ODP implementation -appears one day. - ## peer_connect_interval - Type: seconds diff --git a/docs/config/network.ru.md b/docs/config/network.ru.md index 70de102e..5e1bcbe2 100644 --- a/docs/config/network.ru.md +++ b/docs/config/network.ru.md @@ -22,7 +22,6 @@ - [rdma_max_msg](#rdma_max_msg) - [rdma_max_recv](#rdma_max_recv) - [rdma_max_send](#rdma_max_send) -- [rdma_odp](#rdma_odp) - [peer_connect_interval](#peer_connect_interval) - [peer_connect_timeout](#peer_connect_timeout) - [osd_idle_timeout](#osd_idle_timeout) @@ -101,12 +100,6 @@ RoCEv1/RoCEv2, и даже позволяет полностью отключи не задана. Также автовыбор не поддерживается со старыми версиями библиотеки libibverbs < v32, например в Debian 10 Buster или CentOS 7. -Vitastor поддерживает все модели адаптеров, включая те, у которых -нет поддержки ODP, то есть вы можете использовать RDMA с ConnectX-3 и -картами производства не Mellanox. Версии Vitastor до 1.2.0 включительно -требовали ODP, который есть только на Mellanox ConnectX 4 и более новых. -См. также [rdma_odp](#rdma_odp). - Запустите `ibv_devinfo -v` от имени суперпользователя, чтобы посмотреть список доступных RDMA-устройств, их параметры и возможности. @@ -117,6 +110,24 @@ Vitastor поддерживает все модели адаптеров, вкл подразумевает настройку сети без потерь на основе PFC (Priority Flow Control) и ECN (Explicit Congestion Notification). +Vitastor поддерживает все модели адаптеров, включая те, у которых нет +поддержки ODP (On-Demand Paging), например, ConnectX-3 и карты производства +не Mellanox. Функция ODP доступна только на адаптерах Mellanox ConnectX-4 и +более новых и позволяет не регистрировать память для её использования RDMA-картой, +благодаря чему в теории можно избежать лишних копирований памяти. + +Версии Vitastor до 1.2.0 включительно требовали ODP, потом функция был отключена +по умолчанию, но поддерживалась вплоть до версии 3.0.3. Сейчас поддержка ODP +полностью удалена, так как на самом деле она только портит производительность: +например, на 3-узловом кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на +чтение с RDMA без ODP удаётся снять 3950000 iops, а с ODP - всего 239000 iops. + +Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и +основана на повторной передаче сообщений, когда карте не известен буфер - +вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready). +А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука. +Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf + ## rdma_port_num - Тип: целое число @@ -192,29 +203,6 @@ OSD в любом случае согласовывают реальное зн Не влияет на потребление памяти - дополнительная память на операции отправки не выделяется. -## rdma_odp - -- Тип: булево (да/нет) -- Значение по умолчанию: false - -Использовать RDMA с On-Demand Paging. ODP - функция, доступная пока что -исключительно на адаптерах Mellanox ConnectX-4 и более новых. ODP позволяет -не регистрировать память для её использования RDMA-картой. Благодаря этому -можно не копировать данные при отправке их в сеть и, казалось бы, это должно -улучшать производительность - но **по факту** получается так, что -производительность только ухудшается, причём сильно. Пример - на 3-узловом -кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на чтение с RDMA без ODP -удаётся снять 3950000 iops, а с ODP - всего 239000 iops... - -Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и -основана на повторной передаче сообщений, когда карте не известен буфер - -вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready). -А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука. -Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf - -Возможность использования ODP сохранена в коде на случай, если вдруг в один -прекрасный день появится хорошая реализация ODP. - ## peer_connect_interval - Тип: секунды diff --git a/docs/config/src/network.yml b/docs/config/src/network.yml index 1dadc706..a4f89892 100644 --- a/docs/config/src/network.yml +++ b/docs/config/src/network.yml @@ -84,11 +84,6 @@ unsupported with old libibverbs < v32, like in Debian 10 Buster or CentOS 7. - Vitastor supports all adapters, even ones without ODP support, like - Mellanox ConnectX-3 and non-Mellanox cards. Versions up to Vitastor - 1.2.0 required ODP which is only present in Mellanox ConnectX >= 4. - See also [rdma_odp](#rdma_odp). - Run `ibv_devinfo -v` as root to list available RDMA devices and their features. @@ -97,6 +92,23 @@ the manual of your network vendor for details about setting up the switch for RoCEv2 correctly. Usually it means setting up Lossless Ethernet with PFC (Priority Flow Control) and ECN (Explicit Congestion Notification). + + Vitastor supports all adapters, even ones without ODP (On-Demand Paging) + support, like Mellanox ConnectX-3 and non-Mellanox cards. ODP is only present + in Mellanox ConnectX >= 4 adapters and allows to skip memory registration + for RDMA and thus, in theory, avoid memory copying. + + Versions up to Vitastor 1.2.0 required ODP, then it was disabled by default, + but it was still supported up to 3.0.3. Now ODP support is removed because it + actually only hurts performance: an example 3-node cluster with 8 NVMe in each + node and 2*25 GBit/s ConnectX-6 RDMA network pushed 3950000 read iops without + ODP, but only 239000 iops with ODP. + + This happens because Mellanox ODP implementation seems to be based on + message retransmissions when the adapter doesn't know about the buffer yet - + it likely uses standard "RNR retransmissions" (RNR = receiver not ready) + which is generally slow in RDMA/RoCE networks. Here's a presentation about + it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf info_ru: | Название RDMA-устройства для связи с Vitastor OSD (например, "rocep5s0f0"). Если не указано, Vitastor попробует найти RoCE-устройство, соответствующее @@ -105,12 +117,6 @@ не задана. Также автовыбор не поддерживается со старыми версиями библиотеки libibverbs < v32, например в Debian 10 Buster или CentOS 7. - Vitastor поддерживает все модели адаптеров, включая те, у которых - нет поддержки ODP, то есть вы можете использовать RDMA с ConnectX-3 и - картами производства не Mellanox. Версии Vitastor до 1.2.0 включительно - требовали ODP, который есть только на Mellanox ConnectX 4 и более новых. - См. также [rdma_odp](#rdma_odp). - Запустите `ibv_devinfo -v` от имени суперпользователя, чтобы посмотреть список доступных RDMA-устройств, их параметры и возможности. @@ -120,6 +126,24 @@ коммутатора для RoCEv2 ищите в документации производителя. Обычно это подразумевает настройку сети без потерь на основе PFC (Priority Flow Control) и ECN (Explicit Congestion Notification). + + Vitastor поддерживает все модели адаптеров, включая те, у которых нет + поддержки ODP (On-Demand Paging), например, ConnectX-3 и карты производства + не Mellanox. Функция ODP доступна только на адаптерах Mellanox ConnectX-4 и + более новых и позволяет не регистрировать память для её использования RDMA-картой, + благодаря чему в теории можно избежать лишних копирований памяти. + + Версии Vitastor до 1.2.0 включительно требовали ODP, потом функция был отключена + по умолчанию, но поддерживалась вплоть до версии 3.0.3. Сейчас поддержка ODP + полностью удалена, так как на самом деле она только портит производительность: + например, на 3-узловом кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на + чтение с RDMA без ODP удаётся снять 3950000 iops, а с ODP - всего 239000 iops. + + Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и + основана на повторной передаче сообщений, когда карте не известен буфер - + вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready). + А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука. + Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf - name: rdma_port_num type: int info: | @@ -218,45 +242,6 @@ у принимающей стороны в процессе работы не заканчивались буферы на приём. Не влияет на потребление памяти - дополнительная память на операции отправки не выделяется. -- name: rdma_odp - type: bool - default: false - online: false - info: | - Use RDMA with On-Demand Paging. ODP is currently only available on Mellanox - ConnectX-4 and newer adapters. ODP allows to not register memory explicitly - for RDMA adapter to be able to use it. This, in turn, allows to skip memory - copying during sending. One would think this should improve performance, but - **in reality** RDMA performance with ODP is **drastically** worse. Example - 3-node cluster with 8 NVMe in each node and 2*25 GBit/s ConnectX-6 RDMA network - without ODP pushes 3950000 read iops, but only 239000 iops with ODP... - - This happens because Mellanox ODP implementation seems to be based on - message retransmissions when the adapter doesn't know about the buffer yet - - it likely uses standard "RNR retransmissions" (RNR = receiver not ready) - which is generally slow in RDMA/RoCE networks. Here's a presentation about - it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf - - ODP support is retained in the code just in case a good ODP implementation - appears one day. - info_ru: | - Использовать RDMA с On-Demand Paging. ODP - функция, доступная пока что - исключительно на адаптерах Mellanox ConnectX-4 и более новых. ODP позволяет - не регистрировать память для её использования RDMA-картой. Благодаря этому - можно не копировать данные при отправке их в сеть и, казалось бы, это должно - улучшать производительность - но **по факту** получается так, что - производительность только ухудшается, причём сильно. Пример - на 3-узловом - кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на чтение с RDMA без ODP - удаётся снять 3950000 iops, а с ODP - всего 239000 iops... - - Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и - основана на повторной передаче сообщений, когда карте не известен буфер - - вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready). - А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука. - Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf - - Возможность использования ODP сохранена в коде на случай, если вдруг в один - прекрасный день появится хорошая реализация ODP. - name: peer_connect_interval type: sec min: 1 diff --git a/src/client/messenger.cpp b/src/client/messenger.cpp index 0325d297..55b3a42c 100644 --- a/src/client/messenger.cpp +++ b/src/client/messenger.cpp @@ -145,7 +145,7 @@ void osd_messenger_t::init() rdma_contexts = msgr_rdma_context_t::create_all( osd_num && osd_cluster_network_masks.size() ? osd_cluster_network_masks : osd_network_masks, rdma_device != "" ? rdma_device.c_str() : NULL, - rdma_port_num, rdma_gid_index, rdma_mtu, rdma_odp, log_level + rdma_port_num, rdma_gid_index, rdma_mtu, log_level ); if (!rdma_contexts.size()) { @@ -322,7 +322,6 @@ void osd_messenger_t::parse_config(const json11::Json & config) this->rdma_max_msg = config["rdma_max_msg"].uint64_value(); if (!this->rdma_max_msg || this->rdma_max_msg > 128*1024*1024) this->rdma_max_msg = 129*1024; - this->rdma_odp = config["rdma_odp"].bool_value(); #endif if (!osd_num) this->iothread_count = (uint32_t)config["client_iothread_count"].uint64_value(); diff --git a/src/client/messenger.h b/src/client/messenger.h index 0318f04e..b264c044 100644 --- a/src/client/messenger.h +++ b/src/client/messenger.h @@ -200,7 +200,6 @@ protected: std::vector rdma_contexts; uint64_t rdma_max_sge = 0, rdma_max_send = 0, rdma_max_recv = 0; uint64_t rdma_max_msg = 0; - bool rdma_odp = false; rdma_event_channel *rdmacm_evch = NULL; std::map rdmacm_connections; std::map rdmacm_connecting; @@ -287,8 +286,6 @@ protected: #ifdef WITH_RDMA void try_send_rdma(osd_client_t *cl); - void try_send_rdma_odp(osd_client_t *cl); - void try_send_rdma_nodp(osd_client_t *cl); bool init_recv_rdma(osd_client_t *cl); void handle_rdma_events(msgr_rdma_context_t *rdma_context); msgr_rdma_context_t* choose_rdma_context(osd_client_t *cl); diff --git a/src/client/msgr_rdma.cpp b/src/client/msgr_rdma.cpp index 92ee86fd..68e09464 100644 --- a/src/client/msgr_rdma.cpp +++ b/src/client/msgr_rdma.cpp @@ -59,8 +59,6 @@ msgr_rdma_context_t::~msgr_rdma_context_t() ibv_destroy_cq(cq); if (channel) ibv_destroy_comp_channel(channel); - if (mr) - ibv_dereg_mr(mr); if (pd) ibv_dealloc_pd(pd); if (context && !is_cm) @@ -182,7 +180,7 @@ static int match_port_gid(const std::vector & osd_network_masks, ib #endif std::vector msgr_rdma_context_t::create_all(const std::vector & osd_network_masks, - const char *sel_dev_name, int sel_port_num, int sel_gid_index, uint32_t sel_mtu, bool odp, int log_level) + const char *sel_dev_name, int sel_port_num, int sel_gid_index, uint32_t sel_mtu, int log_level) { int res; std::vector ret; @@ -271,7 +269,7 @@ std::vector msgr_rdma_context_t::create_all(const std::vec { if (log_level > 0) log_rdma_dev_port_gid(dev, port_num, best_gid_idx, port_mtu, best_gidx); - auto ctx = msgr_rdma_context_t::create(dev, portinfo, port_num, best_gid_idx, port_mtu, odp, log_level); + auto ctx = msgr_rdma_context_t::create(dev, portinfo, port_num, best_gid_idx, port_mtu, log_level); if (ctx) { ctx->net_mask = osd_network_masks[net_num]; @@ -291,7 +289,7 @@ std::vector msgr_rdma_context_t::create_all(const std::vec log_rdma_dev_port_gid(dev, port_num, best_gid_idx, port_mtu, gidx); } #endif - auto ctx = msgr_rdma_context_t::create(dev, portinfo, port_num, best_gid_idx, port_mtu, odp, log_level); + auto ctx = msgr_rdma_context_t::create(dev, portinfo, port_num, best_gid_idx, port_mtu, log_level); if (ctx) ret.push_back(ctx); } @@ -306,7 +304,7 @@ cleanup: return ret; } -msgr_rdma_context_t *msgr_rdma_context_t::create(ibv_device *dev, ibv_port_attr & portinfo, int ib_port, int gid_index, uint32_t mtu, bool odp, int log_level) +msgr_rdma_context_t *msgr_rdma_context_t::create(ibv_device *dev, ibv_port_attr & portinfo, int ib_port, int gid_index, uint32_t mtu, int log_level) { msgr_rdma_context_t *ctx = new msgr_rdma_context_t(); ibv_context *context = ibv_open_device(dev); @@ -346,30 +344,6 @@ msgr_rdma_context_t *msgr_rdma_context_t::create(ibv_device *dev, ibv_port_attr goto cleanup; } - ctx->odp = odp; - if (ctx->odp) - { - if (!(ctx->attrx.odp_caps.general_caps & IBV_ODP_SUPPORT) || - !(ctx->attrx.odp_caps.general_caps & IBV_ODP_SUPPORT_IMPLICIT) || - !(ctx->attrx.odp_caps.per_transport_caps.rc_odp_caps & IBV_ODP_SUPPORT_SEND) || - !(ctx->attrx.odp_caps.per_transport_caps.rc_odp_caps & IBV_ODP_SUPPORT_RECV)) - { - ctx->odp = false; - if (log_level > 0) - fprintf(stderr, "The RDMA device isn't implicit ODP (On-Demand Paging) capable, disabling it\n"); - } - } - - if (ctx->odp) - { - ctx->mr = ibv_reg_mr(ctx->pd, NULL, SIZE_MAX, IBV_ACCESS_LOCAL_WRITE | IBV_ACCESS_ON_DEMAND); - if (!ctx->mr) - { - fprintf(stderr, "Couldn't register RDMA memory region\n"); - goto cleanup; - } - } - ctx->channel = ibv_create_comp_channel(ctx->context); if (!ctx->channel) { @@ -603,52 +577,7 @@ static int try_send_rdma_copy(osd_client_t *cl, uint8_t *dst, int dst_len) return total_dst_len-dst_len; } -void osd_messenger_t::try_send_rdma_odp(osd_client_t *cl) -{ - auto rc = cl->rdma_conn; - if (!cl->send_list.size() || rc->cur_send >= rc->max_send) - { - return; - } - uint64_t op_size = 0, op_sge = 0; - ibv_sge sge[rc->max_sge]; - while (rc->send_pos < cl->send_list.size()) - { - iovec & iov = cl->send_list[rc->send_pos]; - if (op_size >= rc->max_msg || op_sge >= rc->max_sge) - { - rc->send_sizes.push_back(op_size); - try_send_rdma_wr(cl, sge, op_sge); - op_sge = 0; - op_size = 0; - if (rc->cur_send >= rc->max_send) - { - break; - } - } - uint32_t len = (uint32_t)(op_size+iov.iov_len-rc->send_buf_pos < rc->max_msg - ? iov.iov_len-rc->send_buf_pos : rc->max_msg-op_size); - sge[op_sge++] = { - .addr = (uintptr_t)((uint8_t*)iov.iov_base+rc->send_buf_pos), - .length = len, - .lkey = rc->ctx->mr->lkey, - }; - op_size += len; - rc->send_buf_pos += len; - if (rc->send_buf_pos >= iov.iov_len) - { - rc->send_pos++; - rc->send_buf_pos = 0; - } - } - if (op_sge > 0) - { - rc->send_sizes.push_back(op_size); - try_send_rdma_wr(cl, sge, op_sge); - } -} - -void osd_messenger_t::try_send_rdma_nodp(osd_client_t *cl) +void osd_messenger_t::try_send_rdma(osd_client_t *cl) { auto rc = cl->rdma_conn; if (!rc->send_out_size) @@ -656,14 +585,11 @@ void osd_messenger_t::try_send_rdma_nodp(osd_client_t *cl) // Allocate send ring buffer, if not yet rc->send_out_size = rc->max_msg*rdma_max_send; rc->send_out.buf = (uint8_t*)malloc_or_die(rc->send_out_size); - if (!rc->ctx->odp) + rc->send_out.mr = ibv_reg_mr(rc->ctx->pd, rc->send_out.buf, rc->send_out_size, 0); + if (!rc->send_out.mr) { - rc->send_out.mr = ibv_reg_mr(rc->ctx->pd, rc->send_out.buf, rc->send_out_size, 0); - if (!rc->send_out.mr) - { - fprintf(stderr, "Failed to register RDMA memory region: %s\n", strerror(errno)); - exit(1); - } + fprintf(stderr, "Failed to register RDMA memory region: %s\n", strerror(errno)); + exit(1); } } // Copy data into the buffer and send it @@ -688,7 +614,7 @@ void osd_messenger_t::try_send_rdma_nodp(osd_client_t *cl) ibv_sge sge = { .addr = (uintptr_t)dst, .length = (uint32_t)copied, - .lkey = rc->ctx->odp ? rc->ctx->mr->lkey : rc->send_out.mr->lkey, + .lkey = rc->send_out.mr->lkey, }; try_send_rdma_wr(cl, &sge, 1); rc->send_sizes.push_back(copied); @@ -696,20 +622,12 @@ void osd_messenger_t::try_send_rdma_nodp(osd_client_t *cl) } } -void osd_messenger_t::try_send_rdma(osd_client_t *cl) -{ - if (cl->rdma_conn->ctx->odp) - try_send_rdma_odp(cl); - else - try_send_rdma_nodp(cl); -} - static void try_recv_rdma_wr(osd_client_t *cl, void *buf) { ibv_sge sge = { .addr = (uintptr_t)buf, .length = (uint32_t)cl->rdma_conn->max_msg, - .lkey = cl->rdma_conn->ctx->odp ? cl->rdma_conn->ctx->mr->lkey : cl->rdma_conn->recv_buf.mr->lkey, + .lkey = cl->rdma_conn->recv_buf.mr->lkey, }; ibv_recv_wr *bad_wr = NULL; ibv_recv_wr wr = { @@ -731,14 +649,11 @@ bool osd_messenger_t::init_recv_rdma(osd_client_t *cl) auto rc = cl->rdma_conn; assert(!rc->recv_buf.buf); rc->recv_buf.buf = (uint8_t*)malloc_or_die(rc->max_msg * rc->max_recv); - if (!rc->ctx->odp) + rc->recv_buf.mr = ibv_reg_mr(rc->ctx->pd, rc->recv_buf.buf, rc->max_msg * rc->max_recv, IBV_ACCESS_LOCAL_WRITE); + if (!rc->recv_buf.mr) { - rc->recv_buf.mr = ibv_reg_mr(rc->ctx->pd, rc->recv_buf.buf, rc->max_msg * rc->max_recv, IBV_ACCESS_LOCAL_WRITE); - if (!rc->recv_buf.mr) - { - fprintf(stderr, "Failed to register RDMA memory region: %s\n", strerror(errno)); - exit(1); - } + fprintf(stderr, "Failed to register RDMA memory region: %s\n", strerror(errno)); + exit(1); } for (uint32_t i = 0; i < rc->max_recv; i++) { @@ -814,14 +729,11 @@ void osd_messenger_t::handle_rdma_events(msgr_rdma_context_t *rdma_context) rc->cur_send--; uint64_t sent_size = rc->send_sizes.at(0); rc->send_sizes.erase(rc->send_sizes.begin(), rc->send_sizes.begin()+1); - if (!rdma_context->odp) - { - rc->send_done_pos += sent_size; - rc->send_out_full = false; - if (rc->send_done_pos == rc->send_out_size) - rc->send_done_pos = 0; - assert(rc->send_done_pos < rc->send_out_size); - } + rc->send_done_pos += sent_size; + rc->send_out_full = false; + if (rc->send_done_pos == rc->send_out_size) + rc->send_done_pos = 0; + assert(rc->send_done_pos < rc->send_out_size); int send_pos = 0, send_buf_pos = 0; while (sent_size > 0) { diff --git a/src/client/msgr_rdma.h b/src/client/msgr_rdma.h index 0e93c57e..f18b8d10 100644 --- a/src/client/msgr_rdma.h +++ b/src/client/msgr_rdma.h @@ -26,8 +26,6 @@ struct msgr_rdma_context_t ibv_context *context = NULL; ibv_device_attr_ex attrx; ibv_pd *pd = NULL; - bool odp = false; - ibv_mr *mr = NULL; ibv_comp_channel *channel = NULL; ibv_cq *cq = NULL; ibv_port_attr portinfo; @@ -43,9 +41,9 @@ struct msgr_rdma_context_t int cm_refs = 0; static std::vector create_all(const std::vector & osd_network_masks, - const char *sel_dev_name, int sel_port_num, int sel_gid_index, uint32_t sel_mtu, bool odp, int log_level); + const char *sel_dev_name, int sel_port_num, int sel_gid_index, uint32_t sel_mtu, int log_level); static msgr_rdma_context_t *create(ibv_device *dev, ibv_port_attr & portinfo, - int ib_port, int gid_index, uint32_t mtu, bool odp, int log_level); + int ib_port, int gid_index, uint32_t mtu, int log_level); static msgr_rdma_context_t* create_cm(ibv_context *ctx); bool reserve_cqe(int n); diff --git a/src/client/msgr_rdmacm.cpp b/src/client/msgr_rdmacm.cpp index 9a402386..b539603c 100644 --- a/src/client/msgr_rdmacm.cpp +++ b/src/client/msgr_rdmacm.cpp @@ -178,7 +178,6 @@ msgr_rdma_context_t* msgr_rdma_context_t::create_cm(ibv_context *ctx) delete rdma_context; return NULL; } - rdma_context->odp = false; rdma_context->channel = ibv_create_comp_channel(rdma_context->context); if (!rdma_context->channel) {