Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
eb3df895e0 |
@@ -234,60 +234,6 @@ jobs:
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_etcd_fail_https:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 10
|
||||
run: ETCD_SCHEME=https /root/vitastor/tests/test_etcd_fail.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_etcd_fail_https_antietcd:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 10
|
||||
run: ETCD_SCHEME=https ANTIETCD=1 /root/vitastor/tests/test_etcd_fail.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_snapshot_https:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 3
|
||||
run: ETCD_SCHEME=https /root/vitastor/tests/test_snapshot.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_interrupted_rebalance:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
@@ -1260,96 +1206,6 @@ jobs:
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_checksum:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 3
|
||||
run: /root/vitastor/tests/test_checksum.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_checksum_xxhash:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 3
|
||||
run: TEST_NAME=xxhash OSD_ARGS="--data_csum_type xxh3_32" /root/vitastor/tests/test_checksum.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_old_checksum:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 3
|
||||
run: OLD=1 /root/vitastor/tests/test_checksum.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_corrupt_all:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 3
|
||||
run: /root/vitastor/tests/test_corrupt_all.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_old_corrupt_all:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 3
|
||||
run: OLD=1 /root/vitastor/tests/test_corrupt_all.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_reweight_half:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
|
||||
@@ -38,10 +38,6 @@ for my $line (<>)
|
||||
{
|
||||
$test_name .= '_antietcd';
|
||||
}
|
||||
elsif ($1 eq 'ETCD_SCHEME' && $2 eq 'https')
|
||||
{
|
||||
$test_name .= '_https';
|
||||
}
|
||||
elsif ($1 eq 'OLD')
|
||||
{
|
||||
$test_name =~ s/^test_/test_old_/s;
|
||||
|
||||
+1
-1
@@ -2,7 +2,7 @@ cmake_minimum_required(VERSION 2.8.12)
|
||||
|
||||
project(vitastor)
|
||||
|
||||
set(VITASTOR_VERSION "3.0.3")
|
||||
set(VITASTOR_VERSION "3.0.2")
|
||||
|
||||
include(CTest)
|
||||
|
||||
|
||||
+5
-5
@@ -1,5 +1,5 @@
|
||||
# Compile stage
|
||||
FROM golang:trixie AS build
|
||||
FROM golang:bookworm AS build
|
||||
|
||||
ADD go.sum go.mod /app/
|
||||
RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go mod download -x
|
||||
@@ -9,7 +9,7 @@ RUN perl -i -e '$/ = undef; while(<>) { s/\n\s*(\{\s*\n)/$1\n/g; s/\}(\s*\n\s*)e
|
||||
CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o vitastor-csi
|
||||
|
||||
# Final stage
|
||||
FROM debian:trixie
|
||||
FROM debian:bookworm
|
||||
|
||||
LABEL maintainers="Vitaliy Filippov <vitalif@yourcmc.ru>"
|
||||
LABEL description="Vitastor CSI Driver"
|
||||
@@ -32,13 +32,13 @@ RUN apt-get update && \
|
||||
|
||||
COPY --from=build /app/vitastor-csi /bin/
|
||||
|
||||
RUN (echo deb http://vitastor.io/debian trixie main > /etc/apt/sources.list.d/vitastor.list) && \
|
||||
RUN (echo deb http://vitastor.io/debian bookworm main > /etc/apt/sources.list.d/vitastor.list) && \
|
||||
((echo 'Package: *'; echo 'Pin: origin "vitastor.io"'; echo 'Pin-Priority: 1000') > /etc/apt/preferences.d/vitastor.pref) && \
|
||||
wget -q -O /etc/apt/trusted.gpg.d/vitastor.gpg https://vitastor.io/debian/pubkey.gpg && \
|
||||
apt-get update && \
|
||||
apt-get install -y vitastor-client ibverbs-providers && \
|
||||
wget https://vitastor.io/archive/qemu/qemu-trixie-10.0.2%2Bds-2%2Bvitastor1/qemu-utils_10.0.2%2Bds-2%2Bvitastor1_amd64.deb && \
|
||||
wget https://vitastor.io/archive/qemu/qemu-trixie-10.0.2%2Bds-2%2Bvitastor1/qemu-block-extra_10.0.2%2Bds-2%2Bvitastor1_amd64.deb && \
|
||||
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||
dpkg -x qemu-utils*.deb tmp1 && \
|
||||
dpkg -x qemu-block-extra*.deb tmp1 && \
|
||||
cp -a tmp1/usr/bin/qemu-storage-daemon /usr/bin/ && \
|
||||
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
VITASTOR_VERSION ?= v3.0.3
|
||||
VITASTOR_VERSION ?= v3.0.2
|
||||
|
||||
all: build push
|
||||
|
||||
|
||||
@@ -49,7 +49,7 @@ spec:
|
||||
capabilities:
|
||||
add: ["SYS_ADMIN"]
|
||||
allowPrivilegeEscalation: true
|
||||
image: vitalif/vitastor-csi:v3.0.3
|
||||
image: vitalif/vitastor-csi:v3.0.2
|
||||
args:
|
||||
- "--node=$(NODE_ID)"
|
||||
- "--endpoint=$(CSI_ENDPOINT)"
|
||||
|
||||
@@ -121,7 +121,7 @@ spec:
|
||||
privileged: true
|
||||
capabilities:
|
||||
add: ["SYS_ADMIN"]
|
||||
image: vitalif/vitastor-csi:v3.0.3
|
||||
image: vitalif/vitastor-csi:v3.0.2
|
||||
args:
|
||||
- "--node=$(NODE_ID)"
|
||||
- "--endpoint=$(CSI_ENDPOINT)"
|
||||
|
||||
+1
-1
@@ -5,7 +5,7 @@ package vitastor
|
||||
|
||||
const (
|
||||
vitastorCSIDriverName = "csi.vitastor.io"
|
||||
vitastorCSIDriverVersion = "3.0.3"
|
||||
vitastorCSIDriverVersion = "3.0.2"
|
||||
)
|
||||
|
||||
// Config struct fills the parameters of request or user input
|
||||
|
||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
||||
vitastor (3.0.3-1) unstable; urgency=medium
|
||||
vitastor (3.0.2-1) unstable; urgency=medium
|
||||
|
||||
* Bugfixes
|
||||
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
# Build Docker image with Vitastor packages
|
||||
|
||||
FROM debian:trixie
|
||||
FROM debian:bookworm
|
||||
|
||||
ADD etc/apt /etc/apt/
|
||||
RUN apt-get update && apt-get -y install vitastor ibverbs-providers udev systemd qemu-system-x86 qemu-system-common qemu-block-extra qemu-utils jq nfs-common && apt-get clean
|
||||
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
VITASTOR_VERSION ?= v3.0.3
|
||||
VITASTOR_VERSION ?= v3.0.2
|
||||
|
||||
all: build push
|
||||
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
#
|
||||
|
||||
# Desired Vitastor version
|
||||
VITASTOR_VERSION=v3.0.3
|
||||
VITASTOR_VERSION=v3.0.2
|
||||
|
||||
# Additional arguments for all containers
|
||||
# For example, you may want to specify a custom logging driver here
|
||||
|
||||
+28
-17
@@ -22,6 +22,7 @@ between clients, OSDs and etcd.
|
||||
- [rdma_max_msg](#rdma_max_msg)
|
||||
- [rdma_max_recv](#rdma_max_recv)
|
||||
- [rdma_max_send](#rdma_max_send)
|
||||
- [rdma_odp](#rdma_odp)
|
||||
- [peer_connect_interval](#peer_connect_interval)
|
||||
- [peer_connect_timeout](#peer_connect_timeout)
|
||||
- [osd_idle_timeout](#osd_idle_timeout)
|
||||
@@ -101,6 +102,11 @@ found or if `osd_network` is not specified. Auto-selection is also
|
||||
unsupported with old libibverbs < v32, like in Debian 10 Buster or
|
||||
CentOS 7.
|
||||
|
||||
Vitastor supports all adapters, even ones without ODP support, like
|
||||
Mellanox ConnectX-3 and non-Mellanox cards. Versions up to Vitastor
|
||||
1.2.0 required ODP which is only present in Mellanox ConnectX >= 4.
|
||||
See also [rdma_odp](#rdma_odp).
|
||||
|
||||
Run `ibv_devinfo -v` as root to list available RDMA devices and their
|
||||
features.
|
||||
|
||||
@@ -110,23 +116,6 @@ the manual of your network vendor for details about setting up the switch
|
||||
for RoCEv2 correctly. Usually it means setting up Lossless Ethernet with
|
||||
PFC (Priority Flow Control) and ECN (Explicit Congestion Notification).
|
||||
|
||||
Vitastor supports all adapters, even ones without ODP (On-Demand Paging)
|
||||
support, like Mellanox ConnectX-3 and non-Mellanox cards. ODP is only present
|
||||
in Mellanox ConnectX >= 4 adapters and allows to skip memory registration
|
||||
for RDMA and thus, in theory, avoid memory copying.
|
||||
|
||||
Versions up to Vitastor 1.2.0 required ODP, then it was disabled by default,
|
||||
but it was still supported up to 3.0.3. Now ODP support is removed because it
|
||||
actually only hurts performance: an example 3-node cluster with 8 NVMe in each
|
||||
node and 2*25 GBit/s ConnectX-6 RDMA network pushed 3950000 read iops without
|
||||
ODP, but only 239000 iops with ODP.
|
||||
|
||||
This happens because Mellanox ODP implementation seems to be based on
|
||||
message retransmissions when the adapter doesn't know about the buffer yet -
|
||||
it likely uses standard "RNR retransmissions" (RNR = receiver not ready)
|
||||
which is generally slow in RDMA/RoCE networks. Here's a presentation about
|
||||
it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
||||
|
||||
## rdma_port_num
|
||||
|
||||
- Type: integer
|
||||
@@ -198,6 +187,28 @@ less than `rdma_max_recv` so the receiving side doesn't run out of buffers.
|
||||
Doesn't affect memory usage - additional memory isn't allocated for send
|
||||
operations.
|
||||
|
||||
## rdma_odp
|
||||
|
||||
- Type: boolean
|
||||
- Default: false
|
||||
|
||||
Use RDMA with On-Demand Paging. ODP is currently only available on Mellanox
|
||||
ConnectX-4 and newer adapters. ODP allows to not register memory explicitly
|
||||
for RDMA adapter to be able to use it. This, in turn, allows to skip memory
|
||||
copying during sending. One would think this should improve performance, but
|
||||
**in reality** RDMA performance with ODP is **drastically** worse. Example
|
||||
3-node cluster with 8 NVMe in each node and 2*25 GBit/s ConnectX-6 RDMA network
|
||||
without ODP pushes 3950000 read iops, but only 239000 iops with ODP...
|
||||
|
||||
This happens because Mellanox ODP implementation seems to be based on
|
||||
message retransmissions when the adapter doesn't know about the buffer yet -
|
||||
it likely uses standard "RNR retransmissions" (RNR = receiver not ready)
|
||||
which is generally slow in RDMA/RoCE networks. Here's a presentation about
|
||||
it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
||||
|
||||
ODP support is retained in the code just in case a good ODP implementation
|
||||
appears one day.
|
||||
|
||||
## peer_connect_interval
|
||||
|
||||
- Type: seconds
|
||||
|
||||
+30
-18
@@ -22,6 +22,7 @@
|
||||
- [rdma_max_msg](#rdma_max_msg)
|
||||
- [rdma_max_recv](#rdma_max_recv)
|
||||
- [rdma_max_send](#rdma_max_send)
|
||||
- [rdma_odp](#rdma_odp)
|
||||
- [peer_connect_interval](#peer_connect_interval)
|
||||
- [peer_connect_timeout](#peer_connect_timeout)
|
||||
- [osd_idle_timeout](#osd_idle_timeout)
|
||||
@@ -100,6 +101,12 @@ RoCEv1/RoCEv2, и даже позволяет полностью отключи
|
||||
не задана. Также автовыбор не поддерживается со старыми версиями библиотеки
|
||||
libibverbs < v32, например в Debian 10 Buster или CentOS 7.
|
||||
|
||||
Vitastor поддерживает все модели адаптеров, включая те, у которых
|
||||
нет поддержки ODP, то есть вы можете использовать RDMA с ConnectX-3 и
|
||||
картами производства не Mellanox. Версии Vitastor до 1.2.0 включительно
|
||||
требовали ODP, который есть только на Mellanox ConnectX 4 и более новых.
|
||||
См. также [rdma_odp](#rdma_odp).
|
||||
|
||||
Запустите `ibv_devinfo -v` от имени суперпользователя, чтобы посмотреть
|
||||
список доступных RDMA-устройств, их параметры и возможности.
|
||||
|
||||
@@ -110,24 +117,6 @@ libibverbs < v32, например в Debian 10 Buster или CentOS 7.
|
||||
подразумевает настройку сети без потерь на основе PFC (Priority Flow
|
||||
Control) и ECN (Explicit Congestion Notification).
|
||||
|
||||
Vitastor поддерживает все модели адаптеров, включая те, у которых нет
|
||||
поддержки ODP (On-Demand Paging), например, ConnectX-3 и карты производства
|
||||
не Mellanox. Функция ODP доступна только на адаптерах Mellanox ConnectX-4 и
|
||||
более новых и позволяет не регистрировать память для её использования RDMA-картой,
|
||||
благодаря чему в теории можно избежать лишних копирований памяти.
|
||||
|
||||
Версии Vitastor до 1.2.0 включительно требовали ODP, потом функция был отключена
|
||||
по умолчанию, но поддерживалась вплоть до версии 3.0.3. Сейчас поддержка ODP
|
||||
полностью удалена, так как на самом деле она только портит производительность:
|
||||
например, на 3-узловом кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на
|
||||
чтение с RDMA без ODP удаётся снять 3950000 iops, а с ODP - всего 239000 iops.
|
||||
|
||||
Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и
|
||||
основана на повторной передаче сообщений, когда карте не известен буфер -
|
||||
вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready).
|
||||
А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука.
|
||||
Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
||||
|
||||
## rdma_port_num
|
||||
|
||||
- Тип: целое число
|
||||
@@ -203,6 +192,29 @@ OSD в любом случае согласовывают реальное зн
|
||||
Не влияет на потребление памяти - дополнительная память на операции отправки
|
||||
не выделяется.
|
||||
|
||||
## rdma_odp
|
||||
|
||||
- Тип: булево (да/нет)
|
||||
- Значение по умолчанию: false
|
||||
|
||||
Использовать RDMA с On-Demand Paging. ODP - функция, доступная пока что
|
||||
исключительно на адаптерах Mellanox ConnectX-4 и более новых. ODP позволяет
|
||||
не регистрировать память для её использования RDMA-картой. Благодаря этому
|
||||
можно не копировать данные при отправке их в сеть и, казалось бы, это должно
|
||||
улучшать производительность - но **по факту** получается так, что
|
||||
производительность только ухудшается, причём сильно. Пример - на 3-узловом
|
||||
кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на чтение с RDMA без ODP
|
||||
удаётся снять 3950000 iops, а с ODP - всего 239000 iops...
|
||||
|
||||
Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и
|
||||
основана на повторной передаче сообщений, когда карте не известен буфер -
|
||||
вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready).
|
||||
А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука.
|
||||
Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
||||
|
||||
Возможность использования ODP сохранена в коде на случай, если вдруг в один
|
||||
прекрасный день появится хорошая реализация ODP.
|
||||
|
||||
## peer_connect_interval
|
||||
|
||||
- Тип: секунды
|
||||
|
||||
+50
-35
@@ -84,6 +84,11 @@
|
||||
unsupported with old libibverbs < v32, like in Debian 10 Buster or
|
||||
CentOS 7.
|
||||
|
||||
Vitastor supports all adapters, even ones without ODP support, like
|
||||
Mellanox ConnectX-3 and non-Mellanox cards. Versions up to Vitastor
|
||||
1.2.0 required ODP which is only present in Mellanox ConnectX >= 4.
|
||||
See also [rdma_odp](#rdma_odp).
|
||||
|
||||
Run `ibv_devinfo -v` as root to list available RDMA devices and their
|
||||
features.
|
||||
|
||||
@@ -92,23 +97,6 @@
|
||||
the manual of your network vendor for details about setting up the switch
|
||||
for RoCEv2 correctly. Usually it means setting up Lossless Ethernet with
|
||||
PFC (Priority Flow Control) and ECN (Explicit Congestion Notification).
|
||||
|
||||
Vitastor supports all adapters, even ones without ODP (On-Demand Paging)
|
||||
support, like Mellanox ConnectX-3 and non-Mellanox cards. ODP is only present
|
||||
in Mellanox ConnectX >= 4 adapters and allows to skip memory registration
|
||||
for RDMA and thus, in theory, avoid memory copying.
|
||||
|
||||
Versions up to Vitastor 1.2.0 required ODP, then it was disabled by default,
|
||||
but it was still supported up to 3.0.3. Now ODP support is removed because it
|
||||
actually only hurts performance: an example 3-node cluster with 8 NVMe in each
|
||||
node and 2*25 GBit/s ConnectX-6 RDMA network pushed 3950000 read iops without
|
||||
ODP, but only 239000 iops with ODP.
|
||||
|
||||
This happens because Mellanox ODP implementation seems to be based on
|
||||
message retransmissions when the adapter doesn't know about the buffer yet -
|
||||
it likely uses standard "RNR retransmissions" (RNR = receiver not ready)
|
||||
which is generally slow in RDMA/RoCE networks. Here's a presentation about
|
||||
it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
||||
info_ru: |
|
||||
Название RDMA-устройства для связи с Vitastor OSD (например, "rocep5s0f0").
|
||||
Если не указано, Vitastor попробует найти RoCE-устройство, соответствующее
|
||||
@@ -117,6 +105,12 @@
|
||||
не задана. Также автовыбор не поддерживается со старыми версиями библиотеки
|
||||
libibverbs < v32, например в Debian 10 Buster или CentOS 7.
|
||||
|
||||
Vitastor поддерживает все модели адаптеров, включая те, у которых
|
||||
нет поддержки ODP, то есть вы можете использовать RDMA с ConnectX-3 и
|
||||
картами производства не Mellanox. Версии Vitastor до 1.2.0 включительно
|
||||
требовали ODP, который есть только на Mellanox ConnectX 4 и более новых.
|
||||
См. также [rdma_odp](#rdma_odp).
|
||||
|
||||
Запустите `ibv_devinfo -v` от имени суперпользователя, чтобы посмотреть
|
||||
список доступных RDMA-устройств, их параметры и возможности.
|
||||
|
||||
@@ -126,24 +120,6 @@
|
||||
коммутатора для RoCEv2 ищите в документации производителя. Обычно это
|
||||
подразумевает настройку сети без потерь на основе PFC (Priority Flow
|
||||
Control) и ECN (Explicit Congestion Notification).
|
||||
|
||||
Vitastor поддерживает все модели адаптеров, включая те, у которых нет
|
||||
поддержки ODP (On-Demand Paging), например, ConnectX-3 и карты производства
|
||||
не Mellanox. Функция ODP доступна только на адаптерах Mellanox ConnectX-4 и
|
||||
более новых и позволяет не регистрировать память для её использования RDMA-картой,
|
||||
благодаря чему в теории можно избежать лишних копирований памяти.
|
||||
|
||||
Версии Vitastor до 1.2.0 включительно требовали ODP, потом функция был отключена
|
||||
по умолчанию, но поддерживалась вплоть до версии 3.0.3. Сейчас поддержка ODP
|
||||
полностью удалена, так как на самом деле она только портит производительность:
|
||||
например, на 3-узловом кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на
|
||||
чтение с RDMA без ODP удаётся снять 3950000 iops, а с ODP - всего 239000 iops.
|
||||
|
||||
Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и
|
||||
основана на повторной передаче сообщений, когда карте не известен буфер -
|
||||
вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready).
|
||||
А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука.
|
||||
Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
||||
- name: rdma_port_num
|
||||
type: int
|
||||
info: |
|
||||
@@ -242,6 +218,45 @@
|
||||
у принимающей стороны в процессе работы не заканчивались буферы на приём.
|
||||
Не влияет на потребление памяти - дополнительная память на операции отправки
|
||||
не выделяется.
|
||||
- name: rdma_odp
|
||||
type: bool
|
||||
default: false
|
||||
online: false
|
||||
info: |
|
||||
Use RDMA with On-Demand Paging. ODP is currently only available on Mellanox
|
||||
ConnectX-4 and newer adapters. ODP allows to not register memory explicitly
|
||||
for RDMA adapter to be able to use it. This, in turn, allows to skip memory
|
||||
copying during sending. One would think this should improve performance, but
|
||||
**in reality** RDMA performance with ODP is **drastically** worse. Example
|
||||
3-node cluster with 8 NVMe in each node and 2*25 GBit/s ConnectX-6 RDMA network
|
||||
without ODP pushes 3950000 read iops, but only 239000 iops with ODP...
|
||||
|
||||
This happens because Mellanox ODP implementation seems to be based on
|
||||
message retransmissions when the adapter doesn't know about the buffer yet -
|
||||
it likely uses standard "RNR retransmissions" (RNR = receiver not ready)
|
||||
which is generally slow in RDMA/RoCE networks. Here's a presentation about
|
||||
it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
||||
|
||||
ODP support is retained in the code just in case a good ODP implementation
|
||||
appears one day.
|
||||
info_ru: |
|
||||
Использовать RDMA с On-Demand Paging. ODP - функция, доступная пока что
|
||||
исключительно на адаптерах Mellanox ConnectX-4 и более новых. ODP позволяет
|
||||
не регистрировать память для её использования RDMA-картой. Благодаря этому
|
||||
можно не копировать данные при отправке их в сеть и, казалось бы, это должно
|
||||
улучшать производительность - но **по факту** получается так, что
|
||||
производительность только ухудшается, причём сильно. Пример - на 3-узловом
|
||||
кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на чтение с RDMA без ODP
|
||||
удаётся снять 3950000 iops, а с ODP - всего 239000 iops...
|
||||
|
||||
Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и
|
||||
основана на повторной передаче сообщений, когда карте не известен буфер -
|
||||
вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready).
|
||||
А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука.
|
||||
Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
||||
|
||||
Возможность использования ODP сохранена в коде на случай, если вдруг в один
|
||||
прекрасный день появится хорошая реализация ODP.
|
||||
- name: peer_connect_interval
|
||||
type: sec
|
||||
min: 1
|
||||
|
||||
@@ -26,9 +26,9 @@ at Vitastor Kubernetes operator: https://github.com/Antilles7227/vitastor-operat
|
||||
The instruction is very simple.
|
||||
|
||||
1. Download a Docker image of the desired version: \
|
||||
`docker pull vitalif/vitastor:v3.0.3`
|
||||
`docker pull vitalif/vitastor:v3.0.2`
|
||||
2. Install scripts to the host system: \
|
||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.3 install.sh`
|
||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.2 install.sh`
|
||||
3. Reload udev rules: \
|
||||
`udevadm control --reload-rules`
|
||||
|
||||
|
||||
@@ -25,9 +25,9 @@ Vitastor можно установить в Docker/Podman. При этом etcd,
|
||||
Инструкция по установке максимально простая.
|
||||
|
||||
1. Скачайте Docker-образ желаемой версии: \
|
||||
`docker pull vitalif/vitastor:v3.0.3`
|
||||
`docker pull vitalif/vitastor:v3.0.2`
|
||||
2. Установите скрипты в хост-систему командой: \
|
||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.3 install.sh`
|
||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.2 install.sh`
|
||||
3. Перезагрузите правила udev: \
|
||||
`udevadm control --reload-rules`
|
||||
|
||||
|
||||
+7
-11
@@ -18,7 +18,7 @@ class AntiEtcdAdapter
|
||||
cluster = cluster ? (''+(cluster||'')).split(/,+/) : [];
|
||||
cluster = Object.keys(cluster.reduce((a, url) =>
|
||||
{
|
||||
a[url.toLowerCase().replace(/^(https?:\/\/)?(.*?)(\/.*)?$/, (m, m1, m2) => (m1||'http://')+m2)] = true;
|
||||
a[url.toLowerCase().replace(/^(https?:\/\/)/, '').replace(/\/.*$/, '')] = true;
|
||||
return a;
|
||||
}, {}));
|
||||
const cfg_port = config.antietcd_port;
|
||||
@@ -26,8 +26,7 @@ class AntiEtcdAdapter
|
||||
is_local['0.0.0.0'] = true;
|
||||
is_local['::'] = true;
|
||||
is_local[''] = true;
|
||||
// split :, 3 -> <schema>:<//ip>:<port>
|
||||
const selected = cluster.map(s => s.split(':', 3)).filter(ip => is_local[ip[1].substr(2)] && (!cfg_port || ip[2] == cfg_port));
|
||||
const selected = cluster.map(s => s.split(':', 2)).filter(ip => is_local[ip[0]] && (!cfg_port || ip[1] == cfg_port));
|
||||
if (selected.length > 1)
|
||||
{
|
||||
console.error('More than 1 etcd_address matches local IPs, please specify port');
|
||||
@@ -36,15 +35,12 @@ class AntiEtcdAdapter
|
||||
else if (selected.length == 1)
|
||||
{
|
||||
const antietcd_config = {
|
||||
ip: selected[0][1].substr(2),
|
||||
port: selected[0][2],
|
||||
cert: config.antietcd_cert,
|
||||
key: config.antietcd_key,
|
||||
ca: config.etcd_ca,
|
||||
data: config.antietcd_data_file || ((config.antietcd_data_dir || '/var/lib/vitastor') + '/mon_'+selected[0][2]+'.json.gz'),
|
||||
ip: selected[0][0],
|
||||
port: selected[0][1],
|
||||
data: config.antietcd_data_file || ((config.antietcd_data_dir || '/var/lib/vitastor') + '/mon_'+selected[0][1]+'.json.gz'),
|
||||
persist_filter: vitastor_persist_filter({ vitastor_prefix: config.etcd_prefix || '/vitastor' }),
|
||||
node_id: selected[0][1].substr(2)+':'+selected[0][2], // node_id = ip:port
|
||||
cluster: (cluster.length == 1 ? null : cluster.reduce((a, c) => { a[c.replace(/^(https?:\/\/)/, '')] = c; return a; }, {})),
|
||||
node_id: selected[0][0]+':'+selected[0][1], // node_id = ip:port
|
||||
cluster: (cluster.length == 1 ? null : cluster.reduce((a, c) => { a[c] = "http://"+c; return a; }, {})),
|
||||
cluster_key: (config.etcd_prefix || '/vitastor'),
|
||||
stale_read: 1,
|
||||
log_level: 1,
|
||||
|
||||
+6
-15
@@ -1,9 +1,7 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
const fs = require('fs');
|
||||
const http = require('http');
|
||||
const https = require('https');
|
||||
const WebSocket = require('ws');
|
||||
const { b64, local_ips } = require('./utils.js');
|
||||
|
||||
@@ -17,18 +15,11 @@ class EtcdAdapter
|
||||
this.ws = null;
|
||||
this.ws_alive = false;
|
||||
this.ws_keepalive_timer = null;
|
||||
this.opts = {};
|
||||
}
|
||||
|
||||
parse_config(config)
|
||||
{
|
||||
this.parse_etcd_addresses(config.etcd_address||config.etcd_url);
|
||||
if (config.etcd_client_cert)
|
||||
this.opts.cert = fs.readFileSync(config.etcd_client_cert, { encoding: 'utf-8' });
|
||||
if (config.etcd_client_key)
|
||||
this.opts.key = fs.readFileSync(config.etcd_client_key, { encoding: 'utf-8' });
|
||||
if (config.etcd_ca)
|
||||
this.opts.ca = fs.readFileSync(config.etcd_ca, { encoding: 'utf-8' });
|
||||
}
|
||||
|
||||
parse_etcd_addresses(addrs)
|
||||
@@ -48,7 +39,7 @@ class EtcdAdapter
|
||||
for (let url of addrs)
|
||||
{
|
||||
let scheme = 'http';
|
||||
url = url.trim().replace(/^(https?):\/\//i, (m, m1) => { scheme = m1.toLowerCase(); return ''; });
|
||||
url = url.trim().replace(/^(https?):\/\//, (m, m1) => { scheme = m1; return ''; });
|
||||
const slash = url.indexOf('/');
|
||||
const colon = url.indexOf(':');
|
||||
const is_local = is_local_ip[colon >= 0 ? url.substr(0, colon) : (slash >= 0 ? url.substr(0, slash) : url)];
|
||||
@@ -139,7 +130,7 @@ class EtcdAdapter
|
||||
}
|
||||
ok(false);
|
||||
}, this.mon.config.etcd_mon_timeout);
|
||||
this.ws = new WebSocket(base+'/watch', this.opts);
|
||||
this.ws = new WebSocket(base+'/watch');
|
||||
this.ws_used_url = cur_addr;
|
||||
const fail = () =>
|
||||
{
|
||||
@@ -281,7 +272,7 @@ class EtcdAdapter
|
||||
{
|
||||
throw new Error(MON_STOPPED);
|
||||
}
|
||||
const res = await POST(base+path, body, timeout, this.opts);
|
||||
const res = await POST(base+path, body, timeout);
|
||||
if (this.mon.stopped)
|
||||
{
|
||||
throw new Error(MON_STOPPED);
|
||||
@@ -307,7 +298,7 @@ class EtcdAdapter
|
||||
}
|
||||
}
|
||||
|
||||
function POST(url, body, timeout, opts)
|
||||
function POST(url, body, timeout)
|
||||
{
|
||||
return new Promise(ok =>
|
||||
{
|
||||
@@ -319,10 +310,10 @@ function POST(url, body, timeout, opts)
|
||||
req = null;
|
||||
ok({ error: 'timeout' });
|
||||
}, timeout) : null;
|
||||
let req = (url.substr(0, 5) == 'https' ? https : http).request(url, { method: 'POST', headers: {
|
||||
let req = http.request(url, { method: 'POST', headers: {
|
||||
'Content-Type': 'application/json',
|
||||
'Content-Length': body_text.length,
|
||||
}, ...(opts||{}) }, (res) =>
|
||||
} }, (res) =>
|
||||
{
|
||||
if (!req)
|
||||
{
|
||||
|
||||
+1
-4
@@ -45,10 +45,7 @@ const etcd_tree = {
|
||||
config_path: "/etc/vitastor/vitastor.conf",
|
||||
etcd_prefix: "/vitastor",
|
||||
// etcd connection - configurable online
|
||||
etcd_address: "http://10.0.115.10:2379/v3",
|
||||
etcd_client_cert: "",
|
||||
etcd_client_key: "",
|
||||
etcd_ca: "",
|
||||
etcd_address: "10.0.115.10:2379/v3",
|
||||
// mon
|
||||
etcd_mon_ttl: 5, // min: 1
|
||||
etcd_mon_timeout: 1000, // ms. min: 0
|
||||
|
||||
@@ -10,19 +10,16 @@ const NO_OSD = 'Z';
|
||||
async function lp_solve(text)
|
||||
{
|
||||
const cp = child_process.spawn('lp_solve');
|
||||
let stdout = '', stderr = '', finish_cb, finished = 0;
|
||||
let stdout = '', stderr = '', finish_cb;
|
||||
cp.stdout.on('data', buf => stdout += buf.toString());
|
||||
cp.stderr.on('data', buf => stderr += buf.toString());
|
||||
cp.stdout.on('end', () => finish_cb());
|
||||
cp.stderr.on('end', () => finish_cb());
|
||||
cp.on('exit', () => finish_cb && finish_cb());
|
||||
cp.stdin.write(text);
|
||||
cp.stdin.end();
|
||||
await new Promise(ok => (finish_cb = () =>
|
||||
if (cp.exitCode == null)
|
||||
{
|
||||
finished++;
|
||||
if (finished == 2)
|
||||
ok();
|
||||
}));
|
||||
await new Promise(ok => finish_cb = ok);
|
||||
}
|
||||
if (!stdout.trim())
|
||||
{
|
||||
return null;
|
||||
|
||||
+2
-2
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "vitastor-mon",
|
||||
"version": "3.0.3",
|
||||
"version": "3.0.2",
|
||||
"description": "Vitastor SDS monitor service",
|
||||
"main": "mon-main.js",
|
||||
"scripts": {
|
||||
@@ -9,7 +9,7 @@
|
||||
"author": "Vitaliy Filippov",
|
||||
"license": "UNLICENSED",
|
||||
"dependencies": {
|
||||
"antietcd": "^1.2.4",
|
||||
"antietcd": "^1.2.2",
|
||||
"sprintf-js": "^1.1.2",
|
||||
"ws": "^7.2.5"
|
||||
},
|
||||
|
||||
+1
-1
@@ -195,7 +195,7 @@ async function generate_pool_pgs(state, global_config, pool_id, osd_tree, levels
|
||||
const folded_tree = make_hier_tree(global_config, folded.nodes.reduce((a, c) => { a[c.id] = c; return a; }, {}));
|
||||
const old_pg_count = prev_pgs.length;
|
||||
const optimize_cfg = {
|
||||
osd_weights: folded.nodes.reduce((a, c) => { if (/^\d+$/.exec(c.id) && c.size != null) { a[c.id] = c.size||0; } return a; }, {}),
|
||||
osd_weights: folded.nodes.reduce((a, c) => { if (Number(c.id)) { a[c.id] = c.size; } return a; }, {}),
|
||||
combinator: use_rules
|
||||
// new algorithm:
|
||||
? new RuleCombinator(folded_tree, rules, pool_cfg.max_osd_combinations)
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "vitastor",
|
||||
"version": "3.0.3",
|
||||
"version": "3.0.2",
|
||||
"description": "Low-level native bindings to Vitastor client library",
|
||||
"main": "index.js",
|
||||
"keywords": [
|
||||
|
||||
@@ -50,7 +50,7 @@ from cinder.volume import configuration
|
||||
from cinder.volume import driver
|
||||
from cinder.volume import volume_utils
|
||||
|
||||
VITASTOR_VERSION = '3.0.3'
|
||||
VITASTOR_VERSION = '3.0.2'
|
||||
|
||||
LOG = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@ rpmbuild -bp fio.spec
|
||||
cd $VITASTOR
|
||||
VER=$(grep ^Version: rpm/vitastor-$REL.spec | awk '{print $2}')
|
||||
rm -rf fio
|
||||
ln -s $(ls -d ~/rpmbuild/BUILD/fio*/ | grep -v SPECPARTS) fio
|
||||
ln -s ~/rpmbuild/BUILD/fio*/ fio
|
||||
sh copy-fio-includes.sh
|
||||
rm fio
|
||||
mv fio-copy fio
|
||||
|
||||
@@ -1,17 +0,0 @@
|
||||
# Build packages for AlmaLinux 10 inside a container
|
||||
# cd ..
|
||||
# docker pull --platform=linux/amd64/v2 quay.io/almalinuxorg/almalinux:10
|
||||
# docker build -t vitastor-buildenv:el10 -f rpm/vitastor-el10.Dockerfile .
|
||||
# docker run -i --rm -v ./:/root/vitastor vitastor-buildenv:el10 /root/vitastor/rpm/vitastor-build.sh
|
||||
|
||||
FROM quay.io/almalinuxorg/almalinux:10
|
||||
|
||||
WORKDIR /root
|
||||
|
||||
RUN sed -i 's/enabled=0/enabled=1/' /etc/yum.repos.d/*.repo
|
||||
RUN dnf -y install epel-release dnf-plugins-core
|
||||
RUN dnf -y install https://vitastor.io/rpms/centos/10/vitastor-release-1.0-1.el10.noarch.rpm
|
||||
RUN dnf -y install gcc-c++ gperftools-devel fio nodejs rpm-build jerasure-devel isa-l-devel gf-complete-devel rdma-core-devel cmake libnl3-devel
|
||||
RUN dnf download --source fio
|
||||
RUN rpm --nomd5 -i fio*.src.rpm
|
||||
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --spec fio.spec
|
||||
@@ -1,198 +0,0 @@
|
||||
Name: vitastor
|
||||
Version: 3.0.3
|
||||
Release: 1%{?dist}
|
||||
Summary: Vitastor, a fast software-defined clustered block storage
|
||||
|
||||
License: Vitastor Network Public License 1.1
|
||||
URL: https://vitastor.io/
|
||||
Source0: vitastor-3.0.3.el10.tar.gz
|
||||
|
||||
BuildRequires: gperftools-devel
|
||||
BuildRequires: gcc-c++
|
||||
BuildRequires: nodejs >= 10
|
||||
BuildRequires: jerasure-devel
|
||||
BuildRequires: isa-l-devel
|
||||
BuildRequires: gf-complete-devel
|
||||
BuildRequires: rdma-core-devel
|
||||
BuildRequires: cmake
|
||||
BuildRequires: libnl3-devel
|
||||
Requires: vitastor-osd = %{version}-%{release}
|
||||
Requires: vitastor-mon = %{version}-%{release}
|
||||
Requires: vitastor-client = %{version}-%{release}
|
||||
Requires: vitastor-client-devel = %{version}-%{release}
|
||||
Requires: vitastor-fio = %{version}-%{release}
|
||||
|
||||
%description
|
||||
Vitastor is a small, simple and fast clustered block storage (storage for VM drives),
|
||||
architecturally similar to Ceph which means strong consistency, primary-replication,
|
||||
symmetric clustering and automatic data distribution over any number of drives of any
|
||||
size with configurable redundancy (replication or erasure codes/XOR).
|
||||
|
||||
|
||||
%package -n vitastor-osd
|
||||
Summary: Vitastor - OSD
|
||||
Requires: vitastor-client = %{version}-%{release}
|
||||
Requires: util-linux
|
||||
Requires: parted
|
||||
|
||||
|
||||
%description -n vitastor-osd
|
||||
Vitastor object storage daemon, i.e. server program that stores data.
|
||||
|
||||
|
||||
%package -n vitastor-mon
|
||||
Summary: Vitastor - monitor
|
||||
Requires: nodejs >= 10
|
||||
Requires: lpsolve
|
||||
|
||||
|
||||
%description -n vitastor-mon
|
||||
Vitastor monitor, i.e. server program responsible for watching cluster state and
|
||||
scheduling cluster-level operations.
|
||||
|
||||
|
||||
%package -n vitastor-client
|
||||
Summary: Vitastor - client
|
||||
|
||||
|
||||
%description -n vitastor-client
|
||||
Vitastor client library and command-line interface.
|
||||
|
||||
|
||||
%package -n vitastor-client-devel
|
||||
Summary: Vitastor - development files
|
||||
Group: Development/Libraries
|
||||
Requires: vitastor-client = %{version}-%{release}
|
||||
|
||||
|
||||
%description -n vitastor-client-devel
|
||||
Vitastor library headers for development.
|
||||
|
||||
|
||||
%package -n vitastor-fio
|
||||
Summary: Vitastor - fio drivers
|
||||
Group: Development/Libraries
|
||||
Requires: vitastor-client = %{version}-%{release}
|
||||
Requires: fio = 3.36-5.el10
|
||||
|
||||
|
||||
%description -n vitastor-fio
|
||||
Vitastor fio drivers for benchmarking.
|
||||
|
||||
|
||||
%package -n vitastor-opennebula
|
||||
Summary: Vitastor for OpenNebula
|
||||
Group: Development/Libraries
|
||||
Requires: vitastor-client
|
||||
Requires: jq
|
||||
Requires: python3-lxml
|
||||
Requires: patch
|
||||
Requires: qemu-kvm-block-vitastor
|
||||
|
||||
|
||||
%description -n vitastor-opennebula
|
||||
Vitastor storage plugin for OpenNebula.
|
||||
|
||||
|
||||
%prep
|
||||
%setup -q
|
||||
|
||||
|
||||
%build
|
||||
%cmake
|
||||
%cmake_build
|
||||
|
||||
|
||||
%install
|
||||
rm -rf $RPM_BUILD_ROOT
|
||||
%cmake_install
|
||||
cd mon
|
||||
npm install --production
|
||||
cd ..
|
||||
mkdir -p %buildroot/usr/lib/vitastor
|
||||
cp -r mon %buildroot/usr/lib/vitastor
|
||||
mv %buildroot/usr/lib/vitastor/mon/scripts/make-etcd %buildroot/usr/lib/vitastor/mon/
|
||||
mkdir -p %buildroot/lib/systemd/system
|
||||
cp mon/scripts/vitastor.target mon/scripts/vitastor-mon.service mon/scripts/vitastor-osd@.service %buildroot/lib/systemd/system
|
||||
mkdir -p %buildroot/lib/udev/rules.d
|
||||
cp mon/scripts/90-vitastor.rules %buildroot/lib/udev/rules.d
|
||||
mkdir -p %buildroot/var/lib/one
|
||||
cp -r opennebula/remotes %buildroot/var/lib/one
|
||||
cp opennebula/install.sh %buildroot/var/lib/one/remotes/datastore/vitastor/
|
||||
mkdir -p %buildroot/etc/
|
||||
cp -r opennebula/sudoers.d %buildroot/etc/
|
||||
|
||||
|
||||
%files
|
||||
%doc GPL-2.0.txt VNPL-1.1.txt README.md README-ru.md
|
||||
|
||||
|
||||
%files -n vitastor-osd
|
||||
%_bindir/vitastor-osd
|
||||
%_bindir/vitastor-disk
|
||||
%_bindir/vitastor-dump-journal
|
||||
/lib/systemd/system/vitastor-osd@.service
|
||||
/lib/systemd/system/vitastor.target
|
||||
/lib/udev/rules.d/90-vitastor.rules
|
||||
|
||||
|
||||
%pre -n vitastor-osd
|
||||
groupadd -r -f vitastor 2>/dev/null ||:
|
||||
useradd -r -g vitastor -s /sbin/nologin -c "Vitastor daemons" -M -d /nonexistent vitastor 2>/dev/null ||:
|
||||
install -o vitastor -g vitastor -d /var/log/vitastor
|
||||
mkdir -p /etc/vitastor
|
||||
|
||||
|
||||
%files -n vitastor-mon
|
||||
/usr/lib/vitastor/mon
|
||||
/lib/systemd/system/vitastor-mon.service
|
||||
|
||||
|
||||
%pre -n vitastor-mon
|
||||
groupadd -r -f vitastor 2>/dev/null ||:
|
||||
useradd -r -g vitastor -s /sbin/nologin -c "Vitastor daemons" -M -d /nonexistent vitastor 2>/dev/null ||:
|
||||
mkdir -p /etc/vitastor
|
||||
mkdir -p /var/lib/vitastor
|
||||
chown vitastor:vitastor /var/lib/vitastor
|
||||
|
||||
|
||||
%files -n vitastor-client
|
||||
%_bindir/vitastor-nbd
|
||||
%_bindir/vitastor-ublk
|
||||
%_bindir/vitastor-nfs
|
||||
%_bindir/vitastor-cli
|
||||
%_bindir/vitastor-rm
|
||||
%_bindir/vitastor-kv
|
||||
%_bindir/vitastor-kv-stress
|
||||
%_bindir/vita
|
||||
%_libdir/libvitastor_client.so*
|
||||
%_libdir/libvitastor_kv.so*
|
||||
|
||||
|
||||
%files -n vitastor-client-devel
|
||||
%_includedir/vitastor_c.h
|
||||
%_includedir/vitastor_kv.h
|
||||
%_libdir/pkgconfig
|
||||
|
||||
|
||||
%files -n vitastor-fio
|
||||
%_libdir/libfio_vitastor.so
|
||||
%_libdir/libfio_vitastor_blk.so
|
||||
%_libdir/libfio_vitastor_sec.so
|
||||
|
||||
|
||||
%files -n vitastor-opennebula
|
||||
/var/lib/one
|
||||
/etc/sudoers.d/opennebula-vitastor
|
||||
|
||||
|
||||
%triggerin -n vitastor-opennebula -- opennebula
|
||||
[ $2 = 0 ] || exit 0
|
||||
/var/lib/one/remotes/datastore/vitastor/install.sh
|
||||
|
||||
|
||||
# Turn off the brp-python-bytecompile script
|
||||
%global __os_install_post %(echo '%{__os_install_post}' | sed -e 's!/usr/lib[^[:space:]]*/brp-python-bytecompile[[:space:]].*$!!g')
|
||||
|
||||
|
||||
%changelog
|
||||
@@ -1,11 +1,11 @@
|
||||
Name: vitastor
|
||||
Version: 3.0.3
|
||||
Version: 3.0.2
|
||||
Release: 1%{?dist}
|
||||
Summary: Vitastor, a fast software-defined clustered block storage
|
||||
|
||||
License: Vitastor Network Public License 1.1
|
||||
URL: https://vitastor.io/
|
||||
Source0: vitastor-3.0.3.el7.tar.gz
|
||||
Source0: vitastor-3.0.2.el7.tar.gz
|
||||
|
||||
BuildRequires: gperftools-devel
|
||||
BuildRequires: devtoolset-9-gcc-c++
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
Name: vitastor
|
||||
Version: 3.0.3
|
||||
Version: 3.0.2
|
||||
Release: 1%{?dist}
|
||||
Summary: Vitastor, a fast software-defined clustered block storage
|
||||
|
||||
License: Vitastor Network Public License 1.1
|
||||
URL: https://vitastor.io/
|
||||
Source0: vitastor-3.0.3.el8.tar.gz
|
||||
Source0: vitastor-3.0.2.el8.tar.gz
|
||||
|
||||
BuildRequires: gperftools-devel
|
||||
BuildRequires: gcc-toolset-9-gcc-c++
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
Name: vitastor
|
||||
Version: 3.0.3
|
||||
Version: 3.0.2
|
||||
Release: 1%{?dist}
|
||||
Summary: Vitastor, a fast software-defined clustered block storage
|
||||
|
||||
License: Vitastor Network Public License 1.1
|
||||
URL: https://vitastor.io/
|
||||
Source0: vitastor-3.0.3.el9.tar.gz
|
||||
Source0: vitastor-3.0.2.el9.tar.gz
|
||||
|
||||
BuildRequires: gperftools-devel
|
||||
BuildRequires: gcc-c++
|
||||
|
||||
+1
-6
@@ -21,7 +21,7 @@ if("${CMAKE_INSTALL_PREFIX}" MATCHES "^/usr/local/?$")
|
||||
endif()
|
||||
set(ENABLE_COVERAGE false CACHE BOOL "Enable code coverage")
|
||||
|
||||
add_definitions(-DVITASTOR_VERSION="3.0.3")
|
||||
add_definitions(-DVITASTOR_VERSION="3.0.2")
|
||||
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
|
||||
add_link_options(-fno-omit-frame-pointer)
|
||||
if (${WITH_ASAN})
|
||||
@@ -75,11 +75,6 @@ if (RDMACM_LIBRARIES)
|
||||
add_definitions(-DWITH_RDMACM)
|
||||
endif (RDMACM_LIBRARIES)
|
||||
|
||||
find_package(OpenSSL)
|
||||
if (OPENSSL_FOUND)
|
||||
add_definitions(-DWITH_OPENSSL)
|
||||
endif (OPENSSL_FOUND)
|
||||
|
||||
if (${WITH_SYSTEM_LIBURING})
|
||||
pkg_check_modules(LIBURING REQUIRED liburing>=2.10)
|
||||
include_directories(${LIBURING_INCLUDE_DIRS})
|
||||
|
||||
@@ -4,7 +4,7 @@ project(vitastor)
|
||||
|
||||
# libvitastor_blk.a
|
||||
add_library(vitastor_blk STATIC
|
||||
../util/allocator.cpp ../util/crc32c.c ../util/xxhash.c ../util/ringloop.cpp
|
||||
../util/allocator.cpp ../util/crc32c.c ../util/ringloop.cpp
|
||||
multilist.cpp blockstore_heap.cpp blockstore_disk.cpp
|
||||
blockstore.cpp blockstore_impl.cpp blockstore_init.cpp blockstore_open.cpp
|
||||
blockstore_flush.cpp blockstore_read.cpp blockstore_stable.cpp blockstore_sync.cpp blockstore_write.cpp
|
||||
|
||||
@@ -187,6 +187,7 @@ public:
|
||||
// MUST be called only when nobody makes any modifications to the DB for this pool
|
||||
virtual void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit) = 0;
|
||||
virtual bool reshard_continue(void *reshard_state, uint64_t chunk_limit) = 0;
|
||||
virtual void reshard_abort(void *reshard_state) = 0;
|
||||
|
||||
// Event loop
|
||||
virtual void loop() = 0;
|
||||
|
||||
@@ -83,17 +83,13 @@ void blockstore_disk_t::parse_config(std::map<std::string, std::string> & config
|
||||
{
|
||||
data_csum_type = BLOCKSTORE_CSUM_CRC32C;
|
||||
}
|
||||
else if (config["data_csum_type"] == "xxh3_32")
|
||||
{
|
||||
data_csum_type = BLOCKSTORE_CSUM_XXH3_32;
|
||||
}
|
||||
else if (config["data_csum_type"] == "" || config["data_csum_type"] == "none")
|
||||
{
|
||||
data_csum_type = BLOCKSTORE_CSUM_NONE;
|
||||
}
|
||||
else
|
||||
{
|
||||
throw std::runtime_error("data_csum_type="+config["data_csum_type"]+" is unsupported, only \"crc32c\", \"xxh3_32\" and \"none\" are supported");
|
||||
throw std::runtime_error("data_csum_type="+config["data_csum_type"]+" is unsupported, only \"crc32c\" and \"none\" are supported");
|
||||
}
|
||||
csum_block_size = parse_size(config["csum_block_size"]);
|
||||
discard_on_start = config.find("discard_on_start") != config.end() &&
|
||||
|
||||
@@ -16,7 +16,6 @@
|
||||
#define BLOCKSTORE_CSUM_NONE 0
|
||||
// Lower byte of checksum type is its length
|
||||
#define BLOCKSTORE_CSUM_CRC32C 0x104
|
||||
#define BLOCKSTORE_CSUM_XXH3_32 0x204
|
||||
|
||||
#define MOCK_DATA_FD 1000
|
||||
#define MOCK_META_FD 1001
|
||||
|
||||
+127
-216
@@ -12,7 +12,6 @@
|
||||
#include "blockstore_heap.h"
|
||||
#include "../util/allocator.h"
|
||||
#include "../util/crc32c.h"
|
||||
#include "../util/xxhash.h"
|
||||
#include "../util/malloc_or_die.h"
|
||||
|
||||
#define BS_HEAP_FREE_MVCC 1
|
||||
@@ -63,19 +62,19 @@ uint32_t blockstore_heap_t::get_simple_entry_size()
|
||||
uint32_t blockstore_heap_t::get_big_entry_size()
|
||||
{
|
||||
return sizeof(heap_big_write_t) + dsk->clean_entry_bitmap_size*2 +
|
||||
(!dsk->csum_block_size ? 0 : dsk->data_block_size/dsk->csum_block_size * (dsk->data_csum_type & 0xFF));
|
||||
(!dsk->data_csum_type ? 0 : dsk->data_block_size/dsk->csum_block_size * (dsk->data_csum_type & 0xFF));
|
||||
}
|
||||
|
||||
uint32_t blockstore_heap_t::get_big_intent_entry_size()
|
||||
{
|
||||
return sizeof(heap_big_intent_t) + dsk->clean_entry_bitmap_size*2 +
|
||||
(!dsk->csum_block_size ? 4 : dsk->data_block_size/dsk->csum_block_size * (dsk->data_csum_type & 0xFF));
|
||||
(!dsk->data_csum_type ? 4 : dsk->data_block_size/dsk->csum_block_size * (dsk->data_csum_type & 0xFF));
|
||||
}
|
||||
|
||||
uint32_t blockstore_heap_t::get_small_entry_size(uint32_t offset, uint32_t len)
|
||||
{
|
||||
return sizeof(heap_small_write_t) + dsk->clean_entry_bitmap_size +
|
||||
(!dsk->csum_block_size ? 4 : (dsk->data_csum_type & 0xFF) *
|
||||
(!dsk->data_csum_type ? 4 : (dsk->data_csum_type & 0xFF) *
|
||||
((offset+len+dsk->csum_block_size-1)/dsk->csum_block_size - offset/dsk->csum_block_size));
|
||||
}
|
||||
|
||||
@@ -90,7 +89,7 @@ uint32_t blockstore_heap_t::get_csum_size(heap_entry_t *wr)
|
||||
|
||||
uint32_t blockstore_heap_t::get_csum_size(uint32_t entry_type, uint32_t offset, uint32_t len)
|
||||
{
|
||||
if (!dsk->csum_block_size)
|
||||
if (!dsk->data_csum_type)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -213,24 +212,15 @@ void heap_entry_t::set_big_location(blockstore_heap_t *heap, uint64_t location)
|
||||
big().block_num = location / heap->dsk->data_block_size;
|
||||
}
|
||||
|
||||
uint32_t heap_entry_t::calc_checksum(blockstore_disk_t *dsk)
|
||||
uint32_t heap_entry_t::calc_crc32c()
|
||||
{
|
||||
auto old_checksum = checksum;
|
||||
checksum = 0;
|
||||
uint32_t res = 0;
|
||||
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
|
||||
res = (uint32_t)XXH3_64bits(this, size);
|
||||
else
|
||||
res = ::crc32c(0, (uint8_t*)this, size);
|
||||
checksum = old_checksum;
|
||||
auto old_crc32c = crc32c;
|
||||
crc32c = 0;
|
||||
uint32_t res = ::crc32c(0, (uint8_t*)this, size);
|
||||
crc32c = old_crc32c;
|
||||
return res;
|
||||
}
|
||||
|
||||
uint32_t heap_entry_t::calc_checksum(blockstore_heap_t *heap)
|
||||
{
|
||||
return calc_checksum(heap->dsk);
|
||||
}
|
||||
|
||||
uint64_t blockstore_heap_t::get_pg_id(inode_t inode, uint64_t stripe)
|
||||
{
|
||||
uint64_t pg_num = 0;
|
||||
@@ -341,19 +331,7 @@ corrupted_block:
|
||||
block_num, block_offset, wr->size, sizeof(heap_entry_t));
|
||||
goto corrupted_block;
|
||||
}
|
||||
if (wr->is_garbage())
|
||||
{
|
||||
// Garbage collection is only performed when writing new entries into the block
|
||||
// because it needs a fake LSN and modified blocks require consecutive modified LSNs
|
||||
// That's why garbage entries may persist on disk
|
||||
if (log_level > 5)
|
||||
{
|
||||
fprintf(stderr, "Notice: skipping garbage entry %jx:%jx v%ju l%ju in metadata block %u at %u\n",
|
||||
wr->inode, wr->stripe, wr->version, wr->lsn, block_num, block_offset);
|
||||
}
|
||||
block_offset += wr->size;
|
||||
continue;
|
||||
}
|
||||
wr->entry_type &= ~BS_HEAP_GARBAGE;
|
||||
if ((wr->entry_type & BS_HEAP_TYPE) < BS_HEAP_BIG_WRITE ||
|
||||
(wr->entry_type & BS_HEAP_TYPE) > BS_HEAP_ROLLBACK ||
|
||||
(wr->entry_type & ~(BS_HEAP_TYPE|BS_HEAP_STABLE)) ||
|
||||
@@ -382,7 +360,7 @@ corrupted_object:
|
||||
{
|
||||
// Small writes require accessing offset & len to calculate correct length,
|
||||
// so require at least sizeof(heap_small_write_t) for them
|
||||
fprintf(stderr, "Error: entry %jx:%jx v%ju has invalid size in metadata block %u at %u (%u < min %zu bytes)\n",
|
||||
fprintf(stderr, "Error: entry %jx:%jx v%ju has invalid size in metadata block %u at %u (%u < min %zu bytes). Metadata is corrupted, aborting\n",
|
||||
wr->inode, wr->stripe, wr->version, block_num, block_offset, wr->size, sizeof(heap_small_write_t));
|
||||
goto corrupted_object;
|
||||
}
|
||||
@@ -393,12 +371,12 @@ corrupted_object:
|
||||
goto corrupted_object;
|
||||
}
|
||||
// Verify crc
|
||||
uint32_t expected_checksum = wr->calc_checksum(this);
|
||||
if (wr->checksum != expected_checksum)
|
||||
uint32_t expected_crc32c = wr->calc_crc32c();
|
||||
if (wr->crc32c != expected_crc32c)
|
||||
{
|
||||
fprintf(stderr, "Error: entry %jx:%jx v%ju l%ju in metadata block %u at %u is corrupt (checksum mismatch: expected %08x, got %08x). ",
|
||||
wr->inode, wr->stripe, wr->version, wr->lsn,
|
||||
block_num, block_offset, expected_checksum, wr->checksum);
|
||||
fprintf(stderr, "Error: entry %jx:%jx v%ju in metadata block %u at %u is corrupt (crc32c mismatch: expected %08x, got %08x). Metadata is corrupted, aborting\n",
|
||||
wr->inode, wr->stripe, wr->version,
|
||||
block_num, block_offset, expected_crc32c, wr->crc32c);
|
||||
goto corrupted_object;
|
||||
}
|
||||
// Verify offset & len
|
||||
@@ -407,7 +385,7 @@ corrupted_object:
|
||||
wr->small().offset % dsk->bitmap_granularity ||
|
||||
wr->small().len % dsk->bitmap_granularity))
|
||||
{
|
||||
fprintf(stderr, "Error: %s entry %jx:%jx v%ju has invalid offset/length: %u/%u. Metadata is incompatible with current parameters. ",
|
||||
fprintf(stderr, "Error: %s entry %jx:%jx v%ju has invalid offset/length: %u/%u. Metadata is incompatible with current parameters, aborting\n",
|
||||
wr->type() == BS_HEAP_SMALL_WRITE ? "small_write" : "intent_write",
|
||||
wr->inode, wr->stripe, wr->version, wr->small().offset, wr->small().len);
|
||||
goto corrupted_object;
|
||||
@@ -417,7 +395,7 @@ corrupted_object:
|
||||
wr->big_intent().offset % dsk->bitmap_granularity ||
|
||||
wr->big_intent().len % dsk->bitmap_granularity))
|
||||
{
|
||||
fprintf(stderr, "Error: big_intent entry %jx:%jx v%ju has invalid offset/length: %u/%u. Metadata is incompatible with current parameters. ",
|
||||
fprintf(stderr, "Error: big_intent entry %jx:%jx v%ju has invalid offset/length: %u/%u. Metadata is incompatible with current parameters, aborting\n",
|
||||
wr->inode, wr->stripe, wr->version, wr->big_intent().offset, wr->big_intent().len);
|
||||
goto corrupted_object;
|
||||
}
|
||||
@@ -444,7 +422,7 @@ int blockstore_heap_t::load_blocks(uint64_t disk_offset, uint64_t size, uint8_t
|
||||
next_lsn = wr->lsn;
|
||||
}
|
||||
entries_loaded++;
|
||||
loaded_list_items.push_back(li);
|
||||
insert_list_item(li);
|
||||
modify_alloc(block_num, [&](heap_block_info_t & inf)
|
||||
{
|
||||
if (!inf.entries.size())
|
||||
@@ -480,22 +458,19 @@ bool blockstore_heap_t::validate_object(heap_entry_t *obj)
|
||||
next_wr = wr;
|
||||
if (wr->type() == BS_HEAP_ROLLBACK)
|
||||
{
|
||||
if (commit_wr && wr->version > commit_wr->version)
|
||||
{
|
||||
// rollback may not come before commit with a smaller version
|
||||
fprintf(stderr, "Error: rollback entry %jx:%jx v%ju l%ju comes before a commit entry v%ju l%ju\n",
|
||||
wr->inode, wr->stripe, wr->version, wr->lsn, commit_wr->version, commit_wr->lsn);
|
||||
return false;
|
||||
}
|
||||
rollback_wr = wr;
|
||||
continue;
|
||||
}
|
||||
if (wr->type() == BS_HEAP_COMMIT)
|
||||
{
|
||||
if (commit_wr && wr->version > commit_wr->version)
|
||||
{
|
||||
// commit may not come before commit with a smaller version
|
||||
fprintf(stderr, "Error: commit entry %jx:%jx v%ju l%ju comes before a commit entry v%ju l%ju\n",
|
||||
wr->inode, wr->stripe, wr->version, wr->lsn, commit_wr->version, commit_wr->lsn);
|
||||
return false;
|
||||
}
|
||||
if (!commit_wr)
|
||||
{
|
||||
commit_wr = wr;
|
||||
}
|
||||
commit_wr = wr;
|
||||
continue;
|
||||
}
|
||||
if (wr->entry_type & BS_HEAP_STABLE)
|
||||
@@ -548,23 +523,6 @@ bool blockstore_heap_t::validate_object(heap_entry_t *obj)
|
||||
return true;
|
||||
}
|
||||
|
||||
void blockstore_heap_t::finish_load()
|
||||
{
|
||||
if (loaded_list_items.size())
|
||||
{
|
||||
// Sort everything and load in correct order
|
||||
std::sort(loaded_list_items.begin(), loaded_list_items.end(), [this](const heap_list_item_t* a, const heap_list_item_t* b)
|
||||
{
|
||||
return a->entry.lsn < b->entry.lsn;
|
||||
});
|
||||
for (auto & li: loaded_list_items)
|
||||
{
|
||||
insert_list_item(li);
|
||||
}
|
||||
loaded_list_items.clear();
|
||||
}
|
||||
}
|
||||
|
||||
void blockstore_heap_t::fill_recheck_queue()
|
||||
{
|
||||
for (auto & pgp: block_index)
|
||||
@@ -751,7 +709,6 @@ bool blockstore_heap_t::recheck_small_writes(std::function<void(bool is_data, ui
|
||||
}
|
||||
if (!recheck_queue_filled)
|
||||
{
|
||||
finish_load();
|
||||
fill_recheck_queue();
|
||||
recheck_queue_filled = true;
|
||||
}
|
||||
@@ -843,7 +800,7 @@ std::vector<uint32_t> blockstore_heap_t::get_recheck_modified_blocks()
|
||||
return modified;
|
||||
}
|
||||
|
||||
int blockstore_heap_t::finish_recheck()
|
||||
int blockstore_heap_t::finish_load(bool allow_corrupted)
|
||||
{
|
||||
if (!marked_used_blocks)
|
||||
{
|
||||
@@ -880,17 +837,14 @@ bool blockstore_heap_t::calc_checksums(heap_entry_t *wr, uint8_t *data, bool set
|
||||
{
|
||||
return true;
|
||||
}
|
||||
uint32_t len = 0;
|
||||
if (wr->type() == BS_HEAP_SMALL_WRITE || wr->type() == BS_HEAP_INTENT_WRITE)
|
||||
len = wr->small().len;
|
||||
else if (wr->type() == BS_HEAP_BIG_INTENT)
|
||||
len = wr->big_intent().len;
|
||||
else
|
||||
assert(0);
|
||||
uint32_t real_csum = 0;
|
||||
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
|
||||
real_csum = (uint32_t)XXH3_64bits(data, len);
|
||||
else
|
||||
real_csum = crc32c(0, data, len);
|
||||
uint32_t real_csum = crc32c(0, data, len);
|
||||
if (set)
|
||||
{
|
||||
*wr_csum = real_csum;
|
||||
@@ -900,14 +854,13 @@ bool blockstore_heap_t::calc_checksums(heap_entry_t *wr, uint8_t *data, bool set
|
||||
}
|
||||
if (wr->type() == BS_HEAP_BIG_WRITE)
|
||||
{
|
||||
assert(offset != UINT32_MAX && len != UINT32_MAX);
|
||||
return calc_block_checksums((uint32_t*)(wr->get_checksums(this) + offset/dsk->csum_block_size * (dsk->data_csum_type & 0xFF)),
|
||||
data, wr->get_int_bitmap(this), offset, offset+len, set, NULL);
|
||||
}
|
||||
if (wr->type() == BS_HEAP_BIG_INTENT)
|
||||
{
|
||||
auto & bi = wr->big_intent();
|
||||
return calc_block_checksums((uint32_t*)(wr->get_checksums(this) + bi.offset/dsk->csum_block_size * (dsk->data_csum_type & 0xFF)),
|
||||
return calc_block_checksums((uint32_t*)(wr->get_checksums(this) + offset/dsk->csum_block_size * (dsk->data_csum_type & 0xFF)),
|
||||
data, wr->get_int_bitmap(this), bi.offset, bi.offset+bi.len, set, NULL);
|
||||
}
|
||||
assert(wr->type() == BS_HEAP_SMALL_WRITE || wr->type() == BS_HEAP_INTENT_WRITE);
|
||||
@@ -940,26 +893,11 @@ static uint32_t crc32c_iter(uint32_t prev_crc, const std::function<uint8_t*(uint
|
||||
return prev_crc;
|
||||
}
|
||||
|
||||
static void xxh3_iter(XXH3_state_t* xxh3_state, const std::function<uint8_t*(uint32_t start, uint32_t & len)> & next, uint32_t pos, uint32_t size)
|
||||
{
|
||||
uint32_t cur_len = 0;
|
||||
while (size > 0)
|
||||
{
|
||||
uint8_t *data = next(pos, cur_len);
|
||||
assert(data);
|
||||
cur_len = (cur_len < size ? cur_len : size);
|
||||
XXH3_64bits_update(xxh3_state, data, cur_len);
|
||||
pos += cur_len;
|
||||
size -= cur_len;
|
||||
}
|
||||
}
|
||||
|
||||
bool blockstore_heap_t::calc_block_checksums(uint32_t *block_csums, uint8_t *bitmap,
|
||||
uint32_t start, uint32_t end, std::function<uint8_t*(uint32_t start, uint32_t & len)> next,
|
||||
bool set, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb)
|
||||
{
|
||||
bool res = true;
|
||||
XXH3_state_t* xxh3_state = NULL;
|
||||
uint32_t pos = start;
|
||||
uint32_t block_end = (start/dsk->csum_block_size + 1)*dsk->csum_block_size;
|
||||
uint32_t block_crc = 0;
|
||||
@@ -976,89 +914,42 @@ bool blockstore_heap_t::calc_block_checksums(uint32_t *block_csums, uint8_t *bit
|
||||
pos += dsk->bitmap_granularity;
|
||||
// zero padding at the beginning or at the end of the block is not counted
|
||||
if (pos > prev && prev > 0 && pos < block_end)
|
||||
{
|
||||
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
|
||||
{
|
||||
if (!xxh3_state)
|
||||
{
|
||||
xxh3_state = XXH3_createState();
|
||||
XXH3_64bits_reset(xxh3_state);
|
||||
}
|
||||
uint32_t zeropad = pos-prev;
|
||||
while (zeropad > 0)
|
||||
{
|
||||
uint32_t zerolen = zeropad > 4096 ? 4096 : zeropad;
|
||||
XXH3_64bits_update(xxh3_state, zero_page, zerolen);
|
||||
zeropad -= zerolen;
|
||||
}
|
||||
}
|
||||
else
|
||||
block_crc = crc32c_pad(block_crc, NULL, 0, pos-prev, 0);
|
||||
}
|
||||
block_crc = crc32c_pad(block_crc, NULL, 0, pos-prev, 0);
|
||||
prev = pos;
|
||||
while (pos < end && pos < block_end && (bitmap[pos/dsk->bitmap_granularity/8] & (1 << ((pos/dsk->bitmap_granularity) % 8))))
|
||||
pos += dsk->bitmap_granularity;
|
||||
if (pos > prev)
|
||||
{
|
||||
isset = true;
|
||||
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
|
||||
{
|
||||
if (!xxh3_state)
|
||||
{
|
||||
xxh3_state = XXH3_createState();
|
||||
XXH3_64bits_reset(xxh3_state);
|
||||
}
|
||||
xxh3_iter(xxh3_state, next, prev, pos-prev);
|
||||
}
|
||||
else
|
||||
block_crc = crc32c_iter(block_crc, next, prev, pos-prev);
|
||||
block_crc = crc32c_iter(block_crc, next, prev, pos-prev);
|
||||
}
|
||||
prev = pos;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
|
||||
{
|
||||
if (!xxh3_state)
|
||||
{
|
||||
xxh3_state = XXH3_createState();
|
||||
XXH3_64bits_reset(xxh3_state);
|
||||
}
|
||||
xxh3_iter(xxh3_state, next, pos, (end > block_end ? block_end : end)-pos);
|
||||
}
|
||||
else
|
||||
block_crc = crc32c_iter(block_crc, next, pos, (end > block_end ? block_end : end)-pos);
|
||||
block_crc = crc32c_iter(block_crc, next, pos, (end > block_end ? block_end : end)-pos);
|
||||
pos = (end > block_end ? block_end : end);
|
||||
isset = true;
|
||||
}
|
||||
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32 && xxh3_state)
|
||||
{
|
||||
block_crc = (uint32_t)XXH3_64bits_digest(xxh3_state);
|
||||
XXH3_64bits_reset(xxh3_state);
|
||||
}
|
||||
if (set)
|
||||
{
|
||||
*block_csums = block_crc;
|
||||
}
|
||||
else if (isset && block_crc != *block_csums)
|
||||
{
|
||||
res = false;
|
||||
if (bad_block_cb)
|
||||
{
|
||||
bad_block_cb(blk_start, *block_csums, block_crc);
|
||||
res = false;
|
||||
}
|
||||
else
|
||||
break;
|
||||
return false;
|
||||
}
|
||||
block_end += dsk->csum_block_size;
|
||||
block_crc = 0;
|
||||
block_csums++;
|
||||
}
|
||||
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32 && xxh3_state)
|
||||
{
|
||||
block_crc = (uint32_t)XXH3_64bits_digest(xxh3_state);
|
||||
XXH3_freeState(xxh3_state);
|
||||
xxh3_state = NULL;
|
||||
}
|
||||
return res;
|
||||
}
|
||||
|
||||
@@ -1186,6 +1077,16 @@ bool blockstore_heap_t::reshard_check(pool_id_t pool, uint32_t pg_count, uint32_
|
||||
set_it->second.pg_stripe_size == pg_stripe_size);
|
||||
}
|
||||
|
||||
void blockstore_heap_t::reshard_abort(void* reshard_state)
|
||||
{
|
||||
heap_reshard_state_t *st = (heap_reshard_state_t*)reshard_state;
|
||||
for (auto sh_it = st->old_shards.begin(); sh_it != st->old_shards.end(); sh_it++)
|
||||
{
|
||||
block_index[sh_it->first] = std::move(sh_it->second);
|
||||
}
|
||||
delete st;
|
||||
}
|
||||
|
||||
heap_entry_t *blockstore_heap_t::lock_and_read_entry(object_id oid)
|
||||
{
|
||||
auto obj = read_entry(oid);
|
||||
@@ -1198,6 +1099,27 @@ heap_entry_t *blockstore_heap_t::lock_and_read_entry(object_id oid)
|
||||
return obj;
|
||||
}
|
||||
|
||||
heap_entry_t *blockstore_heap_t::read_locked_entry(object_id oid, uint64_t lsn)
|
||||
{
|
||||
auto obj = read_entry(oid);
|
||||
assert(obj);
|
||||
for (auto wr = obj; wr; wr = prev(wr))
|
||||
{
|
||||
if (wr->is_overwrite())
|
||||
{
|
||||
if (lsn == wr->lsn)
|
||||
{
|
||||
return obj;
|
||||
}
|
||||
else
|
||||
{
|
||||
obj = prev(wr);
|
||||
}
|
||||
}
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
bool blockstore_heap_t::unlock_entry(object_id oid)
|
||||
{
|
||||
auto mvcc_it = object_mvcc.find(oid);
|
||||
@@ -1234,35 +1156,6 @@ heap_entry_t *blockstore_heap_t::read_entry(object_id oid)
|
||||
return &li->entry;
|
||||
}
|
||||
|
||||
void blockstore_heap_t::gc_block(heap_block_info_t & inf)
|
||||
{
|
||||
if (inf.has_garbage)
|
||||
{
|
||||
size_t i = 0, j = 0;
|
||||
for (; i < inf.entries.size(); i++)
|
||||
{
|
||||
if (inf.entries[i]->entry.is_garbage())
|
||||
{
|
||||
// old entry invalidated by a newer one, mark it as freeable on block write
|
||||
// assign a 'virtual' LSN to track GC completion
|
||||
assert(!inf.mod_lsn_to || inf.mod_lsn_to == next_lsn);
|
||||
uint64_t gc_lsn = ++next_lsn;
|
||||
inf.mod_lsn = inf.mod_lsn ? inf.mod_lsn : gc_lsn;
|
||||
inf.mod_lsn_to = gc_lsn;
|
||||
push_inflight_lsn(gc_lsn, &inf.entries[i]->entry, HEAP_INFLIGHT_GC);
|
||||
}
|
||||
else
|
||||
{
|
||||
if (j != i)
|
||||
inf.entries[j] = inf.entries[i];
|
||||
j++;
|
||||
}
|
||||
}
|
||||
inf.entries.resize(j);
|
||||
inf.has_garbage = false;
|
||||
}
|
||||
}
|
||||
|
||||
int blockstore_heap_t::allocate_entry(uint32_t entry_size, uint32_t *block_num, bool allow_last_free)
|
||||
{
|
||||
if (last_allocated_block != UINT32_MAX)
|
||||
@@ -1329,7 +1222,31 @@ int blockstore_heap_t::allocate_entry(uint32_t entry_size, uint32_t *block_num,
|
||||
}
|
||||
// Write into the same block
|
||||
auto & inf = block_info.at(last_allocated_block);
|
||||
gc_block(inf);
|
||||
if (inf.has_garbage)
|
||||
{
|
||||
size_t i = 0, j = 0;
|
||||
for (; i < inf.entries.size(); i++)
|
||||
{
|
||||
if (inf.entries[i]->entry.is_garbage())
|
||||
{
|
||||
// old entry invalidated by a newer one, mark it as freeable on block write
|
||||
// assign a 'virtual' LSN to track GC completion
|
||||
assert(!inf.mod_lsn_to || inf.mod_lsn_to == next_lsn);
|
||||
uint64_t gc_lsn = ++next_lsn;
|
||||
inf.mod_lsn = inf.mod_lsn ? inf.mod_lsn : gc_lsn;
|
||||
inf.mod_lsn_to = gc_lsn;
|
||||
push_inflight_lsn(gc_lsn, &inf.entries[i]->entry, HEAP_INFLIGHT_GC);
|
||||
}
|
||||
else
|
||||
{
|
||||
if (j != i)
|
||||
inf.entries[j] = inf.entries[i];
|
||||
j++;
|
||||
}
|
||||
}
|
||||
inf.entries.resize(j);
|
||||
inf.has_garbage = false;
|
||||
}
|
||||
*block_num = last_allocated_block;
|
||||
modify_alloc(last_allocated_block, [&](heap_block_info_t & inf)
|
||||
{
|
||||
@@ -1410,7 +1327,7 @@ int blockstore_heap_t::add_entry(uint32_t wr_size, uint32_t *modified_block,
|
||||
insert_list_item(li);
|
||||
li->block_num = block_num;
|
||||
new_wr->size = wr_size;
|
||||
new_wr->checksum = new_wr->calc_checksum(this);
|
||||
new_wr->crc32c = new_wr->calc_crc32c();
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -1430,6 +1347,7 @@ int blockstore_heap_t::add_small_write(object_id oid, heap_entry_t **obj_ptr, ui
|
||||
// Small writes are written in parallel with buffered data so they require explicit_complete
|
||||
return add_entry(wr_size, modified_block, false, true, [&](heap_entry_t *wr)
|
||||
{
|
||||
printf("add_small_write t%u %lx:%lx l%lu v%lu %u +%u loc:%lx\n", type, oid.inode, oid.stripe, wr->lsn, version, offset, len, location);
|
||||
wr->entry_type = type;
|
||||
wr->inode = oid.inode;
|
||||
wr->stripe = oid.stripe;
|
||||
@@ -1459,6 +1377,7 @@ int blockstore_heap_t::add_big_write(object_id oid, heap_entry_t *old_head, bool
|
||||
// Big writes are written after writing data so they don't require explicit_complete
|
||||
return add_entry(wr_size, modified_block, false, false, [&](heap_entry_t *wr)
|
||||
{
|
||||
printf("add_big_write %lx:%lx l%lu v%lu loc:%lx\n", oid.inode, oid.stripe, wr->lsn, version, location);
|
||||
wr->entry_type = BS_HEAP_BIG_WRITE | (stable ? BS_HEAP_STABLE : 0);
|
||||
wr->inode = oid.inode;
|
||||
wr->stripe = oid.stripe;
|
||||
@@ -1470,7 +1389,7 @@ int blockstore_heap_t::add_big_write(object_id oid, heap_entry_t *old_head, bool
|
||||
memset(wr->get_ext_bitmap(this), 0, dsk->clean_entry_bitmap_size);
|
||||
memset(wr->get_int_bitmap(this), 0, dsk->clean_entry_bitmap_size);
|
||||
bitmap_set(wr->get_int_bitmap(this), offset, len, dsk->bitmap_granularity);
|
||||
if (dsk->csum_block_size)
|
||||
if (dsk->data_csum_type)
|
||||
{
|
||||
memset(wr->get_checksums(this), 0, get_csum_size(wr));
|
||||
calc_checksums(wr, (uint8_t*)data, true, offset, len);
|
||||
@@ -1485,6 +1404,7 @@ int blockstore_heap_t::add_redirect_intent(object_id oid, heap_entry_t **obj_ptr
|
||||
// Big-redirect intents, just like regular big writes, are written after writing data so they don't require explicit_complete
|
||||
return add_entry(wr_size, modified_block, false, false, [&](heap_entry_t *wr)
|
||||
{
|
||||
printf("add_redir_intent %lx:%lx l%lu v%lu %u +%u loc:%lx\n", oid.inode, oid.stripe, wr->lsn, version, offset, len, location);
|
||||
wr->entry_type = BS_HEAP_BIG_INTENT|BS_HEAP_STABLE;
|
||||
wr->inode = oid.inode;
|
||||
wr->stripe = oid.stripe;
|
||||
@@ -1499,9 +1419,9 @@ int blockstore_heap_t::add_redirect_intent(object_id oid, heap_entry_t **obj_ptr
|
||||
memset(wr->get_ext_bitmap(this), 0, dsk->clean_entry_bitmap_size);
|
||||
memset(wr->get_int_bitmap(this), 0, dsk->clean_entry_bitmap_size);
|
||||
bitmap_set(wr->get_int_bitmap(this), offset, len, dsk->bitmap_granularity);
|
||||
if (dsk->csum_block_size)
|
||||
if (dsk->data_csum_type)
|
||||
memset(wr->get_checksums(this), 0, get_csum_size(wr));
|
||||
calc_checksums(wr, (uint8_t*)data, true);
|
||||
calc_checksums(wr, (uint8_t*)data, true, offset, len);
|
||||
*obj_ptr = wr;
|
||||
});
|
||||
}
|
||||
@@ -1521,6 +1441,7 @@ int blockstore_heap_t::add_big_intent(object_id oid, heap_entry_t **obj_ptr, uin
|
||||
// Big intents are written before writing data so they require explicit_complete
|
||||
return add_entry(wr_size, modified_block, false, true, [&](heap_entry_t *wr)
|
||||
{
|
||||
printf("add_big_intent %lx:%lx l%lu v%lu %u +%u loc:%lx\n", oid.inode, oid.stripe, wr->lsn, version, offset, len, obj->big_location(this));
|
||||
wr->entry_type = BS_HEAP_BIG_INTENT | BS_HEAP_STABLE;
|
||||
wr->inode = oid.inode;
|
||||
wr->stripe = oid.stripe;
|
||||
@@ -1537,14 +1458,14 @@ int blockstore_heap_t::add_big_intent(object_id oid, heap_entry_t **obj_ptr, uin
|
||||
memcpy(wr->get_ext_bitmap(this), obj->get_ext_bitmap(this), dsk->clean_entry_bitmap_size);
|
||||
memcpy(wr->get_int_bitmap(this), obj->get_int_bitmap(this), dsk->clean_entry_bitmap_size);
|
||||
bitmap_set(wr->get_int_bitmap(this), offset, len, dsk->bitmap_granularity);
|
||||
if (dsk->csum_block_size)
|
||||
if (dsk->data_csum_type)
|
||||
{
|
||||
if (checksums)
|
||||
memcpy(wr->get_checksums(this), checksums, get_csum_size(wr));
|
||||
memcpy(wr->get_checksums(this), checksums, dsk->clean_entry_bitmap_size);
|
||||
else
|
||||
{
|
||||
memcpy(wr->get_checksums(this), obj->get_checksums(this), get_csum_size(wr));
|
||||
calc_checksums(wr, (uint8_t*)data, true);
|
||||
memcpy(wr->get_checksums(this), obj->get_checksums(this), dsk->clean_entry_bitmap_size);
|
||||
calc_checksums(wr, (uint8_t*)data, true, offset, len);
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -1576,6 +1497,7 @@ int blockstore_heap_t::add_compact(heap_entry_t *obj, uint64_t compact_version,
|
||||
// Compaction entry is added after copying data so it doesn't require explicit_complete
|
||||
return add_entry(wr_size, modified_block, true, false, [&](heap_entry_t *new_wr)
|
||||
{
|
||||
printf("add_compact %lx:%lx l%lu v%lu loc:%lx\n", obj->inode, obj->stripe, compact_lsn, compact_version, compact_location);
|
||||
new_wr->entry_type = BS_HEAP_BIG_WRITE|BS_HEAP_STABLE;
|
||||
new_wr->inode = obj->inode;
|
||||
new_wr->stripe = obj->stripe;
|
||||
@@ -1584,7 +1506,7 @@ int blockstore_heap_t::add_compact(heap_entry_t *obj, uint64_t compact_version,
|
||||
new_wr->set_big_location(this, compact_location);
|
||||
memcpy(new_wr->get_int_bitmap(this), new_int_bitmap, dsk->clean_entry_bitmap_size);
|
||||
memcpy(new_wr->get_ext_bitmap(this), new_ext_bitmap, dsk->clean_entry_bitmap_size);
|
||||
if (dsk->csum_block_size && new_csums)
|
||||
if (dsk->data_csum_type && new_csums)
|
||||
memcpy(new_wr->get_checksums(this), new_csums, dsk->data_block_size/dsk->csum_block_size*(dsk->data_csum_type & 0xFF));
|
||||
});
|
||||
}
|
||||
@@ -1665,7 +1587,7 @@ int blockstore_heap_t::add_commit(heap_entry_t *obj, uint64_t version, uint32_t
|
||||
}
|
||||
if (!uncommitted)
|
||||
{
|
||||
return 0;
|
||||
return EBUSY;
|
||||
}
|
||||
return add_simple(obj, version, modified_block, BS_HEAP_COMMIT);
|
||||
}
|
||||
@@ -1675,32 +1597,23 @@ int blockstore_heap_t::add_rollback(heap_entry_t *obj, uint64_t version, uint32_
|
||||
heap_entry_t *wr = obj;
|
||||
bool found_uncommitted = false;
|
||||
uint64_t commit_version = 0;
|
||||
uint64_t rollback_version = UINT64_MAX;
|
||||
while (wr)
|
||||
while (wr && !wr->is_overwrite())
|
||||
{
|
||||
if (wr->type() == BS_HEAP_ROLLBACK)
|
||||
{
|
||||
if (wr->version <= version)
|
||||
{
|
||||
// All previous writes are already rolled back, stop
|
||||
break;
|
||||
}
|
||||
rollback_version = wr->version;
|
||||
auto rollback_version = wr->version;
|
||||
wr = prev(wr);
|
||||
while (wr->version > rollback_version)
|
||||
{
|
||||
assert(!(wr->entry_type & BS_HEAP_STABLE));
|
||||
wr = prev(wr);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (wr->type() == BS_HEAP_COMMIT)
|
||||
{
|
||||
if (commit_version < wr->version)
|
||||
{
|
||||
commit_version = wr->version;
|
||||
}
|
||||
wr = prev(wr);
|
||||
continue;
|
||||
}
|
||||
if (wr->version > rollback_version)
|
||||
{
|
||||
// Already rolled back, skip
|
||||
wr = prev(wr);
|
||||
continue;
|
||||
}
|
||||
@@ -1711,10 +1624,14 @@ int blockstore_heap_t::add_rollback(heap_entry_t *obj, uint64_t version, uint32_
|
||||
{
|
||||
return EBUSY;
|
||||
}
|
||||
else
|
||||
else if (wr->version == version)
|
||||
{
|
||||
break;
|
||||
}
|
||||
else if (wr->version < version)
|
||||
{
|
||||
return ENOENT;
|
||||
}
|
||||
}
|
||||
else if (wr->version > version)
|
||||
{
|
||||
@@ -1895,9 +1812,9 @@ void blockstore_heap_t::iterate_with_stable(heap_entry_t *obj, uint64_t max_lsn,
|
||||
}
|
||||
else
|
||||
{
|
||||
// 1) 1 2 3 ROLLBACK(2) COMMIT(3) -> 3 is unstable
|
||||
// 1) 1 2 3 ROLLBACK(2) COMMIT(3) -> impossible
|
||||
// 2) 1 2 3 4 ROLLBACK(3) COMMIT(2) -> OK
|
||||
// 3) 1 2 3 ROLLBACK(2) 3 COMMIT(3) -> first 3 is unstable
|
||||
// 3) 1 2 3 ROLLBACK(2) 3 COMMIT(3) -> first 3 shouldn't be treated as stable
|
||||
// 4) 1 2 3 COMMIT(3) ROLLBACK(2) -> impossible
|
||||
// I.e. a rollback always has version >= previous commit
|
||||
// 5) 1 2 3 4 5 ROLLBACK(4) 5 ROLLBACK(3)
|
||||
@@ -2133,13 +2050,7 @@ void blockstore_heap_t::free_data(inode_t inode, uint64_t location)
|
||||
inode = (INODE_POOL(inode) << POOL_ID_BITS);
|
||||
assert(data_alloc->get(location / dsk->data_block_size));
|
||||
data_alloc->set(location / dsk->data_block_size, false);
|
||||
auto sp_it = inode_space_stats.find(inode);
|
||||
if (sp_it != inode_space_stats.end())
|
||||
{
|
||||
sp_it->second -= dsk->data_block_size;
|
||||
if (sp_it->second == 0)
|
||||
inode_space_stats.erase(sp_it);
|
||||
}
|
||||
inode_space_stats[inode] -= dsk->data_block_size;
|
||||
data_used_space -= dsk->data_block_size;
|
||||
}
|
||||
|
||||
@@ -2422,7 +2333,7 @@ void blockstore_heap_t::set_no_inode_stats(const std::vector<uint64_t> & pool_id
|
||||
{
|
||||
// Recalculate if changed
|
||||
if (ps.second.no_inode_stats == 2 || ps.second.no_inode_stats == 1)
|
||||
recalc_inode_space_stats(ps.first, ps.second.no_inode_stats == 2);
|
||||
recalc_inode_space_stats(ps.first, ps.second.no_inode_stats == 1);
|
||||
ps.second.no_inode_stats &= 1;
|
||||
}
|
||||
}
|
||||
@@ -2433,8 +2344,8 @@ void blockstore_heap_t::recalc_inode_space_stats(uint64_t pool_id, bool per_inod
|
||||
auto sp_begin = inode_space_stats.lower_bound((pool_id << (64-POOL_ID_BITS)));
|
||||
auto sp_end = inode_space_stats.lower_bound(((pool_id+1) << (64-POOL_ID_BITS)));
|
||||
inode_space_stats.erase(sp_begin, sp_end);
|
||||
uint32_t pg_count = ps.pg_count;
|
||||
for (uint32_t pg_num = pg_count ? 1 : 0; pg_num <= pg_count; pg_num++)
|
||||
uint32_t pg_count = ps.pg_count ? ps.pg_count : 1;
|
||||
for (uint32_t pg_num = 1; pg_num <= pg_count; pg_num++)
|
||||
{
|
||||
auto & pg_idx = block_index[(pool_id << (64-POOL_ID_BITS)) | pg_num];
|
||||
for (auto & ip: pg_idx)
|
||||
|
||||
@@ -43,7 +43,7 @@ struct __attribute__((__packed__)) heap_entry_t
|
||||
{
|
||||
uint16_t size;
|
||||
uint16_t entry_type;
|
||||
uint32_t checksum;
|
||||
uint32_t crc32c;
|
||||
uint64_t lsn;
|
||||
uint64_t inode;
|
||||
uint64_t stripe;
|
||||
@@ -69,8 +69,7 @@ struct __attribute__((__packed__)) heap_entry_t
|
||||
uint32_t *get_checksum(blockstore_heap_t *heap);
|
||||
uint64_t big_location(blockstore_heap_t *heap);
|
||||
void set_big_location(blockstore_heap_t *heap, uint64_t location);
|
||||
uint32_t calc_checksum(blockstore_heap_t *heap);
|
||||
uint32_t calc_checksum(blockstore_disk_t *dsk);
|
||||
uint32_t calc_crc32c();
|
||||
};
|
||||
|
||||
struct __attribute__((__packed__)) heap_small_write_t
|
||||
@@ -81,7 +80,7 @@ struct __attribute__((__packed__)) heap_small_write_t
|
||||
uint32_t offset;
|
||||
uint32_t len;
|
||||
|
||||
// Also includes 1 bitmap and 1 checksum after the bitmap if block checksums are disabled
|
||||
// Also includes 1 bitmap and 1 crc32c after the bitmap if checksums are disabled
|
||||
};
|
||||
|
||||
struct __attribute__((__packed__)) heap_big_write_t
|
||||
@@ -99,7 +98,7 @@ struct __attribute__((__packed__)) heap_big_intent_t
|
||||
uint32_t offset;
|
||||
uint32_t len;
|
||||
|
||||
// Also includes 2 bitmaps and 1 checksums if block checksums are disabled
|
||||
// Also includes 2 bitmaps and 1 crc32c if checksums are disabled
|
||||
};
|
||||
|
||||
struct __attribute__((__packed__)) heap_list_item_t
|
||||
@@ -201,7 +200,6 @@ class blockstore_heap_t
|
||||
|
||||
bool marked_used_blocks = false;
|
||||
bool recheck_queue_filled = false;
|
||||
std::vector<heap_list_item_t*> loaded_list_items;
|
||||
std::set<uint32_t> recheck_modified_blocks;
|
||||
std::deque<heap_entry_t*> recheck_queue;
|
||||
int recheck_in_progress = 0;
|
||||
@@ -217,7 +215,6 @@ class blockstore_heap_t
|
||||
void defragment_block(uint32_t block_num);
|
||||
void reshard_add(heap_reshard_state_t *st, heap_list_item_t *li);
|
||||
|
||||
void gc_block(heap_block_info_t & inf);
|
||||
int allocate_entry(uint32_t entry_size, uint32_t *block_num, bool allow_last_free);
|
||||
void insert_list_item(heap_list_item_t *li);
|
||||
int add_entry(uint32_t wr_size, uint32_t *modified_block, bool allow_last_free,
|
||||
@@ -240,29 +237,31 @@ public:
|
||||
std::function<void(uint32_t, uint32_t, uint8_t*)> handle_block);
|
||||
int load_blocks(uint64_t disk_offset, uint64_t size, uint8_t *buf,
|
||||
bool allow_corrupted, uint64_t &entries_loaded);
|
||||
// finish loading - should be called after load_blocks
|
||||
void finish_load();
|
||||
// finish loading
|
||||
int finish_load(bool allow_corrupted = false);
|
||||
// get blocks which are modified during loading and should be written to the disk
|
||||
// before finishing initialization if not R/O
|
||||
std::vector<uint32_t> get_recheck_modified_blocks();
|
||||
// recheck small write data after reading the database from disk
|
||||
bool recheck_small_writes(std::function<void(bool is_data, uint64_t offset, uint64_t len, uint8_t* buf, std::function<void()>)> read_buffer, int queue_depth);
|
||||
int finish_recheck();
|
||||
// reshard database according to the pool's PG count
|
||||
void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit);
|
||||
bool reshard_continue(void* reshard_state, uint64_t chunk_limit);
|
||||
bool reshard_check(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size);
|
||||
void reshard_abort(void* reshard_state);
|
||||
void set_no_inode_stats(const std::vector<uint64_t> & pool_ids);
|
||||
void recalc_inode_space_stats(uint64_t pool_id, bool per_inode);
|
||||
// read an object entry and lock it against removal
|
||||
// in the future, may become asynchronous
|
||||
heap_entry_t *lock_and_read_entry(object_id oid);
|
||||
// re-read a locked object entry with the given lsn (pointer may be invalidated)
|
||||
heap_entry_t *read_locked_entry(object_id oid, uint64_t lsn);
|
||||
// read an object entry without locking it
|
||||
heap_entry_t *read_entry(object_id oid);
|
||||
// unlock an entry
|
||||
bool unlock_entry(object_id oid);
|
||||
// set or verify checksums in a write request
|
||||
bool calc_checksums(heap_entry_t *wr, uint8_t *data, bool set, uint32_t offset = UINT32_MAX, uint32_t len = UINT32_MAX);
|
||||
bool calc_checksums(heap_entry_t *wr, uint8_t *data, bool set, uint32_t offset = 0, uint32_t len = 0);
|
||||
// set or verify raw block checksums
|
||||
bool calc_block_checksums(uint32_t *block_csums, uint8_t *data, uint8_t *bitmap, uint32_t start, uint32_t end,
|
||||
bool set, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb);
|
||||
|
||||
@@ -406,3 +406,8 @@ bool blockstore_impl_t::reshard_continue(void *reshard_state, uint64_t chunk_lim
|
||||
{
|
||||
return heap->reshard_continue(reshard_state, chunk_limit);
|
||||
}
|
||||
|
||||
void blockstore_impl_t::reshard_abort(void *reshard_state)
|
||||
{
|
||||
return heap->reshard_abort(reshard_state);
|
||||
}
|
||||
|
||||
@@ -191,6 +191,7 @@ public:
|
||||
|
||||
void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit);
|
||||
bool reshard_continue(void *reshard_state, uint64_t chunk_limit);
|
||||
void reshard_abort(void *reshard_state);
|
||||
|
||||
// Event loop
|
||||
void loop();
|
||||
|
||||
@@ -239,7 +239,6 @@ resume_4:
|
||||
return 1;
|
||||
}
|
||||
// metadata read finished
|
||||
bs->heap->finish_load();
|
||||
printf("Metadata entries loaded: %ju, used blocks: %ju / %ju\n", entries_loaded, bs->heap->get_data_used_space() / bs->dsk.data_block_size, bs->dsk.block_count);
|
||||
if (zero_on_init && !bs->dsk.disable_meta_fsync)
|
||||
{
|
||||
@@ -285,7 +284,7 @@ resume_6:
|
||||
}, bs->meta_write_recheck_parallelism);
|
||||
return 1;
|
||||
resume_7:
|
||||
if (bs->heap->finish_recheck() != 0)
|
||||
if (bs->heap->finish_load() != 0)
|
||||
{
|
||||
exit(1);
|
||||
}
|
||||
|
||||
@@ -57,9 +57,9 @@ int blockstore_impl_t::dequeue_stable(blockstore_op_t *op)
|
||||
}
|
||||
assert(res == 0);
|
||||
}
|
||||
resume_1:
|
||||
if (priv->modified_block != UINT32_MAX && priv->modified_block2 != priv->modified_block)
|
||||
{
|
||||
resume_1:
|
||||
BS_SUBMIT_CHECK_SQES(1);
|
||||
prepare_meta_block_write(priv->modified_block);
|
||||
resume_2:
|
||||
|
||||
@@ -13,10 +13,13 @@ bool blockstore_impl_t::enqueue_write(blockstore_op_t *op)
|
||||
return true;
|
||||
}
|
||||
|
||||
void blockstore_impl_t::prepare_meta_block_write(uint32_t modified_block)
|
||||
bool blockstore_impl_t::prepare_meta_block_write(uint32_t modified_block)
|
||||
{
|
||||
if (modified_blocks.find(modified_block) != modified_blocks.end())
|
||||
return;
|
||||
auto mod_it = modified_blocks.find(modified_block);
|
||||
if (mod_it != modified_blocks.end())
|
||||
{
|
||||
return !mod_it->second.sent;
|
||||
}
|
||||
io_uring_sqe *sqe = get_sqe();
|
||||
assert(sqe != NULL);
|
||||
ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
||||
@@ -41,6 +44,7 @@ void blockstore_impl_t::prepare_meta_block_write(uint32_t modified_block)
|
||||
unsynced_meta_write_count++;
|
||||
pending_modified_blocks.push_back(modified_block);
|
||||
modified_blocks[modified_block] = { .sent = false, .buf = buf };
|
||||
return true;
|
||||
}
|
||||
|
||||
bool blockstore_impl_t::meta_block_is_pending(uint32_t modified_block)
|
||||
@@ -121,6 +125,7 @@ int blockstore_impl_t::dequeue_write(blockstore_op_t *op)
|
||||
heap_entry_t *obj = heap->read_entry(op->oid);
|
||||
if (op->opcode == BS_OP_DELETE)
|
||||
{
|
||||
return continue_delete(op, 0);
|
||||
// Delete
|
||||
if (!obj || obj->type() == BS_HEAP_DELETE)
|
||||
{
|
||||
@@ -133,17 +138,38 @@ int blockstore_impl_t::dequeue_write(blockstore_op_t *op)
|
||||
BS_SUBMIT_CHECK_SQES(1);
|
||||
int res = heap->add_delete(obj, &PRIV(op)->modified_block);
|
||||
if (res == ENOSPC)
|
||||
{
|
||||
goto enospc;
|
||||
}
|
||||
assert(res == 0);
|
||||
prepare_meta_block_write(PRIV(op)->modified_block);
|
||||
PRIV(op)->pending_ops++;
|
||||
PRIV(op)->op_state = 5;
|
||||
write_iodepth++;
|
||||
resume_1:
|
||||
while (!prepare_meta_block_write(PRIV(op)->modified_block))
|
||||
{
|
||||
PRIV(op)->op_state = 1;
|
||||
return 1;
|
||||
}
|
||||
rseume_2:
|
||||
while (meta_block_is_pending(PRIV(op)->modified_block))
|
||||
{
|
||||
PRIV(op)->op_state = 2;
|
||||
return 1;
|
||||
}
|
||||
resume_3:
|
||||
resume_4:
|
||||
if (!throttle_write(op, 3))
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
write_iodepth--;
|
||||
ack_write(op);
|
||||
return 2;
|
||||
}
|
||||
// FIXME: Allow to do initial writes as buffered, not redirected
|
||||
// FIXME: Allow to do direct writes over holes
|
||||
else if (!obj || obj->type() == BS_HEAP_DELETE || op->offset == 0 && op->len == dsk.data_block_size)
|
||||
{
|
||||
return continue_big_write(op, 10);
|
||||
// Big (redirect) write
|
||||
PRIV(op)->write_type = dsk.disable_data_fsync || op->opcode != BS_OP_WRITE_STABLE ? BS_HEAP_BIG_WRITE : _REDIRECT_INTENT;
|
||||
BS_SUBMIT_CHECK_SQES(1);
|
||||
@@ -163,6 +189,7 @@ enospc:
|
||||
flusher->request_trim();
|
||||
return 0;
|
||||
}
|
||||
write_iodepth++;
|
||||
uint64_t loc = PRIV(op)->location;
|
||||
#ifdef BLOCKSTORE_DEBUG
|
||||
printf(
|
||||
@@ -176,18 +203,72 @@ enospc:
|
||||
data->iov = (struct iovec){ op->buf, op->len };
|
||||
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
||||
io_uring_prep_writev(sqe, dsk.data_fd, &data->iov, 1, dsk.data_offset + loc + op->offset);
|
||||
if (PRIV(op)->write_type == BS_HEAP_BIG_WRITE)
|
||||
inflight_big++;
|
||||
PRIV(op)->pending_ops++;
|
||||
write_iodepth++;
|
||||
resume_10:
|
||||
if (PRIV(op)->pending_ops > 0)
|
||||
{
|
||||
PRIV(op)->op_state = 10;
|
||||
return 1;
|
||||
}
|
||||
if (PRIV(op)->write_type == BS_HEAP_BIG_WRITE)
|
||||
{
|
||||
PRIV(op)->op_state = 1;
|
||||
inflight_big++;
|
||||
inflight_big--;
|
||||
resume_11:
|
||||
resume_12:
|
||||
resume_13:
|
||||
if (!fsync_big_write(op, 11))
|
||||
return 1;
|
||||
}
|
||||
heap_entry_t *obj = heap->read_entry(op->oid);
|
||||
int res = 0;
|
||||
if (PRIV(op)->write_type == _REDIRECT_INTENT)
|
||||
{
|
||||
res = heap->add_redirect_intent(op->oid, &obj, op->version, op->offset, op->len,
|
||||
PRIV(op)->location, op->bitmap, (uint8_t*)op->buf, &PRIV(op)->modified_block);
|
||||
}
|
||||
else
|
||||
PRIV(op)->op_state = 3;
|
||||
{
|
||||
res = heap->add_big_write(op->oid, obj, op->opcode == BS_OP_WRITE_STABLE,
|
||||
op->version, op->offset, op->len, PRIV(op)->location, op->bitmap, (uint8_t*)op->buf, &PRIV(op)->modified_block);
|
||||
}
|
||||
if (res == ENOSPC)
|
||||
{
|
||||
if (!heap->get_to_compact_count())
|
||||
{
|
||||
// no space
|
||||
heap->free_data(op->oid.inode, PRIV(op)->location);
|
||||
write_iodepth--;
|
||||
op->retval = -ENOSPC;
|
||||
FINISH_OP(op);
|
||||
return 2;
|
||||
}
|
||||
PRIV(op)->wait_for = WAIT_COMPACTION;
|
||||
PRIV(op)->wait_detail = heap->get_compacted_count();
|
||||
flusher->request_trim();
|
||||
return 0;
|
||||
}
|
||||
assert(res == 0);
|
||||
resume_14:
|
||||
while (!prepare_meta_block_write(PRIV(op)->modified_block))
|
||||
{
|
||||
PRIV(op)->op_state = 14;
|
||||
return 1;
|
||||
}
|
||||
resume_15:
|
||||
while (meta_block_is_pending(PRIV(op)->modified_block))
|
||||
{
|
||||
PRIV(op)->op_state = 15;
|
||||
return 1;
|
||||
}
|
||||
write_iodepth--;
|
||||
ack_write(op);
|
||||
return 2;
|
||||
}
|
||||
else if (intent_write_allowed(op, obj))
|
||||
{
|
||||
return continue_intent_write(op, 20);
|
||||
// Direct intent-write
|
||||
BS_SUBMIT_CHECK_SQES(1);
|
||||
int res = 0;
|
||||
@@ -223,13 +304,41 @@ enospc:
|
||||
assert(res == 0);
|
||||
PRIV(op)->lsn = obj->lsn;
|
||||
}
|
||||
prepare_meta_block_write(PRIV(op)->modified_block);
|
||||
PRIV(op)->pending_ops++;
|
||||
PRIV(op)->op_state = 9;
|
||||
write_iodepth++;
|
||||
resume_20:
|
||||
while (!prepare_meta_block_write(PRIV(op)->modified_block))
|
||||
{
|
||||
PRIV(op)->op_state = 20;
|
||||
return 1;
|
||||
}
|
||||
resume_21:
|
||||
while (meta_block_is_pending(PRIV(op)->modified_block))
|
||||
{
|
||||
PRIV(op)->op_state = 21;
|
||||
return 1;
|
||||
}
|
||||
// Direct intent-write
|
||||
// LSN is not marked as completed so big_write won't be freed
|
||||
BS_SUBMIT_GET_SQE(sqe, data);
|
||||
data->iov = (struct iovec){ op->buf, op->len };
|
||||
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
||||
io_uring_prep_writev(sqe, dsk.data_fd, &data->iov, 1, dsk.data_offset + PRIV(op)->location + op->offset);
|
||||
if (dsk.use_atomic_flag)
|
||||
sqe->rw_flags = RWF_ATOMIC;
|
||||
PRIV(op)->pending_ops++;
|
||||
resume_22:
|
||||
if (PRIV(op)->pending_ops > 0)
|
||||
{
|
||||
PRIV(op)->op_state = 22;
|
||||
return 1;
|
||||
}
|
||||
write_iodepth--;
|
||||
ack_write(op);
|
||||
return 2;
|
||||
}
|
||||
else
|
||||
{
|
||||
return continue_small_write(op, 30);
|
||||
// Small (buffered) overwrite
|
||||
// First check if there is free buffer space
|
||||
PRIV(op)->write_type = BS_HEAP_SMALL_WRITE;
|
||||
@@ -242,7 +351,7 @@ enospc:
|
||||
return 0;
|
||||
}
|
||||
// There is sufficient space. Check SQE(s)
|
||||
BS_SUBMIT_CHECK_SQES(1 + (op->len > 0 ? 1 : 0));
|
||||
BS_SUBMIT_CHECK_SQES(1 + (op->len > 0 ? 1 : 0)); ---> refactor too
|
||||
int res = heap->add_small_write(op->oid, &obj, (BS_HEAP_SMALL_WRITE | (op->opcode == BS_OP_WRITE_STABLE ? BS_HEAP_STABLE : 0)),
|
||||
op->version, op->offset, op->len, loc, op->bitmap, (uint8_t*)op->buf, &PRIV(op)->modified_block);
|
||||
if (res == ENOSPC)
|
||||
@@ -311,27 +420,46 @@ again:
|
||||
goto again;
|
||||
}
|
||||
resume_2:
|
||||
|
||||
|
||||
resume_4:
|
||||
resume_6:
|
||||
|
||||
resume_8:
|
||||
ack
|
||||
return 2;
|
||||
resume_10:
|
||||
return 1;
|
||||
}
|
||||
|
||||
bool blockstore_impl_t::fsync_big_write(blockstore_op_t *op, int base_state)
|
||||
{
|
||||
if (PRIV(op)->state == base_state)
|
||||
goto resume_0;
|
||||
else if (PRIV(op)->state == base_state+1)
|
||||
goto resume_1;
|
||||
else if (PRIV(op)->state == base_state+2)
|
||||
goto resume_2;
|
||||
// We must fsync all big writes to avoid complex write workflows
|
||||
// It's OK for all HDDs and for server SSDs, but slightly worse for desktop SSDs
|
||||
inflight_big--;
|
||||
if (!dsk.disable_data_fsync)
|
||||
{
|
||||
// fsync data in a batch
|
||||
resume_11:
|
||||
resume_0:
|
||||
if (inflight_big > 0)
|
||||
{
|
||||
PRIV(op)->op_state = 11;
|
||||
return 1;
|
||||
PRIV(op)->op_state = base_state;
|
||||
return false;
|
||||
}
|
||||
if (fsyncing_data)
|
||||
{
|
||||
resume_12:
|
||||
resume_1:
|
||||
if (fsyncing_data)
|
||||
{
|
||||
PRIV(op)->op_state = 12;
|
||||
return 1;
|
||||
PRIV(op)->op_state = base_state+1;
|
||||
return false;
|
||||
}
|
||||
goto resume_4;
|
||||
return true;
|
||||
}
|
||||
fsyncing_data = true;
|
||||
BS_SUBMIT_GET_SQE(sqe, data);
|
||||
@@ -343,47 +471,23 @@ resume_12:
|
||||
handle_write_event(data, op);
|
||||
};
|
||||
PRIV(op)->pending_ops++;
|
||||
PRIV(op)->op_state = 3;
|
||||
return 1;
|
||||
resume_2:
|
||||
if (PRIV(op)->pending_ops > 0)
|
||||
{
|
||||
PRIV(op)->op_state = base_state+2;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
resume_4:
|
||||
{
|
||||
auto obj = heap->read_entry(op->oid);
|
||||
int res = 0;
|
||||
if (PRIV(op)->write_type == _REDIRECT_INTENT)
|
||||
{
|
||||
res = heap->add_redirect_intent(op->oid, &obj, op->version, op->offset, op->len,
|
||||
PRIV(op)->location, op->bitmap, (uint8_t*)op->buf, &PRIV(op)->modified_block);
|
||||
}
|
||||
else
|
||||
{
|
||||
res = heap->add_big_write(op->oid, obj, op->opcode == BS_OP_WRITE_STABLE,
|
||||
op->version, op->offset, op->len, PRIV(op)->location, op->bitmap, (uint8_t*)op->buf, &PRIV(op)->modified_block);
|
||||
}
|
||||
if (res == ENOSPC)
|
||||
{
|
||||
if (!heap->get_to_compact_count())
|
||||
{
|
||||
// no space
|
||||
heap->free_data(op->oid.inode, PRIV(op)->location);
|
||||
write_iodepth--;
|
||||
op->retval = -ENOSPC;
|
||||
FINISH_OP(op);
|
||||
return 2;
|
||||
}
|
||||
PRIV(op)->wait_for = WAIT_COMPACTION;
|
||||
PRIV(op)->wait_detail = heap->get_compacted_count();
|
||||
flusher->request_trim();
|
||||
return 0;
|
||||
}
|
||||
assert(res == 0);
|
||||
prepare_meta_block_write(PRIV(op)->modified_block);
|
||||
PRIV(op)->pending_ops++;
|
||||
PRIV(op)->op_state = 5;
|
||||
return 1;
|
||||
}
|
||||
resume_6:
|
||||
return true;
|
||||
}
|
||||
|
||||
bool blockstore_impl_t::throttle_write(blockstore_op_t *op, int base_state)
|
||||
{
|
||||
// Apply throttling to not fill the journal too quickly for the SSD+HDD case
|
||||
if (PRIV(op)->op_state >= base_state+1)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
if (PRIV(op)->write_type == BS_HEAP_SMALL_WRITE && throttle_small_writes)
|
||||
{
|
||||
// Apply throttling
|
||||
@@ -404,17 +508,21 @@ resume_6:
|
||||
if (ref_us > exec_us + throttle_threshold_us)
|
||||
{
|
||||
// Pause reply
|
||||
PRIV(op)->op_state = 7;
|
||||
PRIV(op)->op_state = base_state;
|
||||
// Remember that the timer can in theory be called right here
|
||||
tfd->set_timer_us(ref_us-exec_us, false, [this, op](int timer_id)
|
||||
{
|
||||
PRIV(op)->op_state = 8;
|
||||
PRIV(op)->op_state++;
|
||||
ringloop->wakeup();
|
||||
});
|
||||
return 1;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
resume_8:
|
||||
return true;
|
||||
}
|
||||
|
||||
void blockstore_impl_t::ack_write(blockstore_op_t *op)
|
||||
{
|
||||
// Acknowledge write
|
||||
#ifdef BLOCKSTORE_DEBUG
|
||||
printf("Ack write %jx:%jx v%ju\n", op->oid.inode, op->oid.stripe, op->version);
|
||||
@@ -441,21 +549,7 @@ resume_8:
|
||||
unsynced_data_write_count++;
|
||||
intent_write_counter++;
|
||||
}
|
||||
write_iodepth--;
|
||||
FINISH_OP(op);
|
||||
return 2;
|
||||
resume_10:
|
||||
// Direct intent-write
|
||||
// LSN is not marked as completed so big_write won't be freed
|
||||
BS_SUBMIT_GET_SQE(sqe, data);
|
||||
data->iov = (struct iovec){ op->buf, op->len };
|
||||
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
||||
io_uring_prep_writev(sqe, dsk.data_fd, &data->iov, 1, dsk.data_offset + PRIV(op)->location + op->offset);
|
||||
if (dsk.use_atomic_flag)
|
||||
sqe->rw_flags = RWF_ATOMIC;
|
||||
PRIV(op)->pending_ops++;
|
||||
PRIV(op)->op_state = 7;
|
||||
return 1;
|
||||
}
|
||||
|
||||
void blockstore_impl_t::handle_write_event(ring_data_t *data, blockstore_op_t *op)
|
||||
|
||||
@@ -480,6 +480,17 @@ resume_1:
|
||||
return true;
|
||||
}
|
||||
|
||||
void blockstore_impl_t::reshard_abort(void *reshard_state)
|
||||
{
|
||||
bs_reshard_state_t *st = (bs_reshard_state_t*)reshard_state;
|
||||
for (auto sh_it = st->old_shards.begin(); sh_it != st->old_shards.end(); sh_it++)
|
||||
{
|
||||
auto & to = clean_db_shards[sh_it->first];
|
||||
to.swap(sh_it->second);
|
||||
}
|
||||
delete st;
|
||||
}
|
||||
|
||||
void blockstore_impl_t::process_list(blockstore_op_t *op)
|
||||
{
|
||||
uint32_t list_pg = op->pg_number+1;
|
||||
|
||||
@@ -290,6 +290,7 @@ public:
|
||||
// Reshard database for a pool
|
||||
void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit);
|
||||
bool reshard_continue(void *reshard_state, uint64_t chunk_limit);
|
||||
void reshard_abort(void *reshard_state);
|
||||
|
||||
// Event loop
|
||||
void loop();
|
||||
|
||||
@@ -13,10 +13,10 @@ if (RDMACM_LIBRARIES)
|
||||
endif (RDMACM_LIBRARIES)
|
||||
add_library(vitastor_common STATIC
|
||||
../util/epoll_manager.cpp etcd_state_client.cpp messenger.cpp ../util/addr_util.cpp
|
||||
msgr_encrypt.cpp msgr_stop.cpp msgr_op.cpp msgr_send.cpp msgr_receive.cpp ../util/ringloop.cpp ../../json11/json11.cpp
|
||||
msgr_stop.cpp msgr_op.cpp msgr_send.cpp msgr_receive.cpp ../util/ringloop.cpp ../../json11/json11.cpp
|
||||
http_client.cpp osd_ops.cpp pg_states.cpp ../util/timerfd_manager.cpp ../util/str_util.cpp ../util/json_util.cpp ${MSGR_RDMA} ${MSGR_RDMACM}
|
||||
)
|
||||
target_link_libraries(vitastor_common pthread ${OPENSSL_LIBRARIES})
|
||||
target_link_libraries(vitastor_common pthread)
|
||||
target_compile_options(vitastor_common PUBLIC -fPIC)
|
||||
|
||||
# libvitastor_client.so
|
||||
@@ -33,7 +33,6 @@ target_link_libraries(vitastor_client
|
||||
${LIBURING_LIBRARIES}
|
||||
${IBVERBS_LIBRARIES}
|
||||
${RDMACM_LIBRARIES}
|
||||
${OPENSSL_LIBRARIES}
|
||||
)
|
||||
set_target_properties(vitastor_client PROPERTIES VERSION ${VITASTOR_VERSION} SOVERSION 0)
|
||||
configure_file(vitastor.pc.in vitastor.pc @ONLY)
|
||||
@@ -99,10 +98,9 @@ endif (${WITH_QEMU})
|
||||
add_executable(test_cluster_client
|
||||
EXCLUDE_FROM_ALL
|
||||
../test/test_cluster_client.cpp
|
||||
pg_states.cpp osd_ops.cpp cluster_client.cpp cluster_client_list.cpp cluster_client_wb.cpp msgr_op.cpp ../test/mock/messenger.cpp msgr_stop.cpp msgr_encrypt.cpp
|
||||
pg_states.cpp osd_ops.cpp cluster_client.cpp cluster_client_list.cpp cluster_client_wb.cpp msgr_op.cpp ../test/mock/messenger.cpp msgr_stop.cpp
|
||||
etcd_state_client.cpp ../util/timerfd_manager.cpp ../util/addr_util.cpp ../util/str_util.cpp ../util/json_util.cpp ../../json11/json11.cpp
|
||||
)
|
||||
target_link_libraries(test_cluster_client ${OPENSSL_LIBRARIES})
|
||||
target_compile_definitions(test_cluster_client PUBLIC -D__MOCK__)
|
||||
target_include_directories(test_cluster_client BEFORE PUBLIC ${CMAKE_SOURCE_DIR}/src/test/mock)
|
||||
add_dependencies(build_tests test_cluster_client)
|
||||
|
||||
@@ -958,22 +958,9 @@ bool cluster_client_t::check_rw(cluster_op_t *op)
|
||||
{
|
||||
op->flags |= OP_IMMEDIATE_COMMIT;
|
||||
}
|
||||
auto ino_it = st_cli.inode_config.find(op->inode);
|
||||
if (ino_it != st_cli.inode_config.end() && ino_it->second.enc)
|
||||
{
|
||||
// FIXME: Rework client API by adding open/close and cache inode information in the "FD"
|
||||
op->enc = ino_it->second.enc;
|
||||
if (!op->enc->bitmap_granularity)
|
||||
{
|
||||
op->enc->bitmap_granularity = pool_it->second.bitmap_granularity;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
op->enc.reset();
|
||||
}
|
||||
if ((op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) && !(op->flags & OSD_OP_IGNORE_READONLY))
|
||||
{
|
||||
auto ino_it = st_cli.inode_config.find(op->inode);
|
||||
if (ino_it != st_cli.inode_config.end() && ino_it->second.readonly)
|
||||
{
|
||||
op->retval = -EROFS;
|
||||
@@ -985,6 +972,7 @@ bool cluster_client_t::check_rw(cluster_op_t *op)
|
||||
op->deoptimise_snapshot = false;
|
||||
if (enable_writeback && (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP))
|
||||
{
|
||||
auto ino_it = st_cli.inode_config.find(op->inode);
|
||||
if (ino_it != st_cli.inode_config.end())
|
||||
{
|
||||
int chain_size = 0;
|
||||
@@ -1447,7 +1435,6 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
|
||||
? (uint8_t*)op->part_bitmaps + pg_bitmap_size*i : NULL),
|
||||
.bitmap_len = (unsigned)(op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP
|
||||
? pg_bitmap_size : 0),
|
||||
.enc = op->enc,
|
||||
.callback = cb ? cb : [this, part](osd_op_t *op_part)
|
||||
{
|
||||
handle_op_part(part);
|
||||
|
||||
@@ -71,7 +71,6 @@ protected:
|
||||
cluster_op_t *prev = NULL, *next = NULL;
|
||||
int prev_wait = 0;
|
||||
uint64_t flush_id = 0;
|
||||
std::shared_ptr<inode_enc_t> enc;
|
||||
friend class cluster_client_t;
|
||||
friend class writeback_cache_t;
|
||||
};
|
||||
|
||||
@@ -22,19 +22,14 @@ etcd_state_client_t::~etcd_state_client_t()
|
||||
stop_ws_keepalive();
|
||||
if (etcd_watch_ws)
|
||||
{
|
||||
http_destroy(etcd_watch_ws);
|
||||
http_close(etcd_watch_ws);
|
||||
etcd_watch_ws = NULL;
|
||||
}
|
||||
if (keepalive_client)
|
||||
{
|
||||
http_destroy(keepalive_client);
|
||||
http_close(keepalive_client);
|
||||
keepalive_client = NULL;
|
||||
}
|
||||
if (http_ctx)
|
||||
{
|
||||
http_context_destroy(http_ctx);
|
||||
http_ctx = NULL;
|
||||
}
|
||||
#endif
|
||||
if (load_pgs_timer_id >= 0)
|
||||
{
|
||||
@@ -77,27 +72,10 @@ std::vector<std::string> etcd_state_client_t::get_addresses()
|
||||
return addrs;
|
||||
}
|
||||
|
||||
http_context_t *etcd_state_client_t::get_http_ctx()
|
||||
{
|
||||
if (!http_ctx)
|
||||
{
|
||||
std::string error;
|
||||
http_ctx = http_context_init(etcd_client_cert, etcd_client_key, etcd_ca, true, error);
|
||||
if (!http_ctx)
|
||||
{
|
||||
fprintf(stderr, "Failed to initialize HTTP context: %s\n", error.c_str());
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
return http_ctx;
|
||||
}
|
||||
|
||||
void etcd_state_client_t::etcd_call_oneshot(std::string etcd_address, std::string api, json11::Json payload,
|
||||
int timeout, std::function<void(std::string, json11::Json)> callback)
|
||||
{
|
||||
std::string etcd_api_path;
|
||||
bool ssl = etcd_address.substr(0, 8) == "https://";
|
||||
etcd_address = etcd_address.substr(ssl ? 8 : 7);
|
||||
int pos = etcd_address.find('/');
|
||||
if (pos >= 0)
|
||||
{
|
||||
@@ -111,16 +89,16 @@ void etcd_state_client_t::etcd_call_oneshot(std::string etcd_address, std::strin
|
||||
"Content-Length: "+std::to_string(req.size())+"\r\n"
|
||||
"Connection: close\r\n"
|
||||
"\r\n"+req;
|
||||
auto http_cli = http_init(tfd, get_http_ctx());
|
||||
auto cb = [http_cli, callback](http_message_t *response)
|
||||
auto http_cli = http_init(tfd);
|
||||
auto cb = [http_cli, callback](const http_response_t *response)
|
||||
{
|
||||
std::string err;
|
||||
json11::Json data;
|
||||
response->parse_json_response(err, data);
|
||||
callback(err, data);
|
||||
http_destroy(http_cli);
|
||||
http_close(http_cli);
|
||||
};
|
||||
http_request(http_cli, etcd_address, req, { .timeout = timeout, .ssl = ssl }, cb);
|
||||
http_request(http_cli, etcd_address, req, { .timeout = timeout }, cb);
|
||||
}
|
||||
|
||||
void etcd_state_client_t::etcd_call(std::string api, json11::Json payload, int timeout,
|
||||
@@ -134,8 +112,6 @@ void etcd_state_client_t::etcd_call(std::string api, json11::Json payload, int t
|
||||
pick_next_etcd();
|
||||
std::string etcd_address = selected_etcd_address;
|
||||
std::string etcd_api_path;
|
||||
bool ssl = etcd_address.substr(0, 8) == "https://";
|
||||
etcd_address = etcd_address.substr(ssl ? 8 : 7);
|
||||
int pos = etcd_address.find('/');
|
||||
if (pos >= 0)
|
||||
{
|
||||
@@ -152,7 +128,7 @@ void etcd_state_client_t::etcd_call(std::string api, json11::Json payload, int t
|
||||
"\r\n"+req;
|
||||
retries--;
|
||||
auto cb = [this, api, payload, timeout, retries, interval, callback,
|
||||
cur_addr = selected_etcd_address](http_message_t *response)
|
||||
cur_addr = selected_etcd_address](const http_response_t *response)
|
||||
{
|
||||
std::string err;
|
||||
json11::Json data;
|
||||
@@ -188,21 +164,22 @@ void etcd_state_client_t::etcd_call(std::string api, json11::Json payload, int t
|
||||
callback(err, data);
|
||||
};
|
||||
if (!keepalive_client)
|
||||
keepalive_client = http_init(tfd, get_http_ctx());
|
||||
http_request(keepalive_client, etcd_address, req, { .timeout = timeout, .keepalive = true, .ssl = ssl }, cb);
|
||||
{
|
||||
keepalive_client = http_init(tfd);
|
||||
}
|
||||
http_request(keepalive_client, etcd_address, req, { .timeout = timeout, .keepalive = true }, cb);
|
||||
}
|
||||
|
||||
void etcd_state_client_t::add_etcd_url(std::string addr)
|
||||
{
|
||||
if (addr.length() > 0)
|
||||
{
|
||||
bool ssl = false;
|
||||
if (strtolower(addr.substr(0, 7)) == "http://")
|
||||
addr = addr.substr(7);
|
||||
else if (strtolower(addr.substr(0, 8)) == "https://")
|
||||
{
|
||||
addr = addr.substr(8);
|
||||
ssl = true;
|
||||
fprintf(stderr, "HTTPS is unsupported for etcd. Either use plain HTTP or setup a local proxy for etcd interaction\n");
|
||||
exit(1);
|
||||
}
|
||||
if (!local_ips.size())
|
||||
local_ips = getifaddr_list(std::vector<addr_mask_t>(), true);
|
||||
@@ -217,7 +194,6 @@ void etcd_state_client_t::add_etcd_url(std::string addr)
|
||||
check_addr = addr;
|
||||
if (pos == std::string::npos)
|
||||
addr += "/v3";
|
||||
addr = (ssl ? "https://" : "http://") + addr;
|
||||
bool local = false;
|
||||
int i;
|
||||
for (i = 0; i < local_ips.size(); i++)
|
||||
@@ -263,9 +239,6 @@ void etcd_state_client_t::parse_config(const json11::Json & config)
|
||||
add_etcd_url(ea.string_value());
|
||||
}
|
||||
}
|
||||
this->etcd_client_cert = config["etcd_client_cert"].string_value();
|
||||
this->etcd_client_key = config["etcd_client_key"].string_value();
|
||||
this->etcd_ca = config["etcd_ca"].string_value();
|
||||
this->etcd_prefix = config["etcd_prefix"].string_value();
|
||||
if (this->etcd_prefix == "")
|
||||
{
|
||||
@@ -358,8 +331,6 @@ void etcd_state_client_t::start_etcd_watcher()
|
||||
pick_next_etcd();
|
||||
std::string etcd_address = selected_etcd_address;
|
||||
std::string etcd_api_path;
|
||||
bool ssl = etcd_address.substr(0, 8) == "https://";
|
||||
etcd_address = etcd_address.substr(ssl ? 8 : 7);
|
||||
int pos = etcd_address.find('/');
|
||||
if (pos >= 0)
|
||||
{
|
||||
@@ -368,17 +339,18 @@ void etcd_state_client_t::start_etcd_watcher()
|
||||
}
|
||||
etcd_watches_initialised = 0;
|
||||
ws_alive = 1;
|
||||
if (etcd_watch_ws)
|
||||
{
|
||||
http_close(etcd_watch_ws);
|
||||
etcd_watch_ws = NULL;
|
||||
}
|
||||
if (this->log_level > 1)
|
||||
{
|
||||
fprintf(stderr, "Trying to connect to etcd websocket at %s, watch from revision %ju/%ju/%ju\n", etcd_address.c_str(),
|
||||
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
|
||||
}
|
||||
if (!etcd_watch_ws)
|
||||
etcd_watch_ws = http_init(tfd, get_http_ctx());
|
||||
else
|
||||
http_close(etcd_watch_ws);
|
||||
open_websocket(etcd_watch_ws, etcd_address, etcd_api_path+"/watch", { .timeout = etcd_slow_timeout, .ssl = ssl },
|
||||
[this, cur_addr = selected_etcd_address](http_message_t *msg)
|
||||
etcd_watch_ws = open_websocket(tfd, etcd_address, etcd_api_path+"/watch", etcd_slow_timeout,
|
||||
[this, cur_addr = selected_etcd_address](const http_response_t *msg)
|
||||
{
|
||||
if (msg->body.length())
|
||||
{
|
||||
@@ -421,6 +393,7 @@ void etcd_state_client_t::start_etcd_watcher()
|
||||
fprintf(stderr, "Revisions before %ju were compacted by etcd, reloading state\n",
|
||||
data["result"]["compact_revision"].uint64_value());
|
||||
http_close(etcd_watch_ws);
|
||||
etcd_watch_ws = NULL;
|
||||
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = 0;
|
||||
on_reload_hook();
|
||||
}
|
||||
@@ -494,6 +467,11 @@ void etcd_state_client_t::start_etcd_watcher()
|
||||
fprintf(stderr, "Disconnected from etcd %s\n", cur_addr.c_str());
|
||||
if (cur_addr == selected_etcd_address)
|
||||
selected_etcd_address = "";
|
||||
if (etcd_watch_ws)
|
||||
{
|
||||
http_close(etcd_watch_ws);
|
||||
etcd_watch_ws = NULL;
|
||||
}
|
||||
if (etcd_watches_initialised == 0)
|
||||
{
|
||||
// Connection not established, retry in <etcd_quick_timeout>
|
||||
@@ -570,6 +548,11 @@ void etcd_state_client_t::start_ws_keepalive()
|
||||
{
|
||||
fprintf(stderr, "Websocket ping failed, disconnecting from etcd %s\n", selected_etcd_address.c_str());
|
||||
}
|
||||
if (etcd_watch_ws)
|
||||
{
|
||||
http_close(etcd_watch_ws);
|
||||
etcd_watch_ws = NULL;
|
||||
}
|
||||
start_etcd_watcher();
|
||||
}
|
||||
else
|
||||
@@ -1296,16 +1279,6 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
|
||||
else
|
||||
parent_inode_num |= parent_pool_id << (64-POOL_ID_BITS);
|
||||
}
|
||||
std::shared_ptr<inode_enc_t> enc;
|
||||
if (!value["enc_key"].string_value().empty())
|
||||
{
|
||||
std::vector<uint8_t> k = hexdecode(value["enc_key"].string_value());
|
||||
if (k.size() == 512/8)
|
||||
{
|
||||
enc = std::make_shared<inode_enc_t>();
|
||||
enc->key = std::move(k);
|
||||
}
|
||||
}
|
||||
insert_inode_config((inode_config_t){
|
||||
.num = inode_num,
|
||||
.name = value["name"].string_value(),
|
||||
@@ -1313,7 +1286,6 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
|
||||
.parent_id = parent_inode_num,
|
||||
.readonly = value["readonly"].bool_value(),
|
||||
.deleted = value["deleted"].bool_value(),
|
||||
.enc = enc,
|
||||
.meta = value["meta"],
|
||||
.mod_revision = kv.mod_revision,
|
||||
});
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
#pragma once
|
||||
|
||||
#include <set>
|
||||
#include <memory>
|
||||
|
||||
#include "json11/json11.hpp"
|
||||
#include "osd_id.h"
|
||||
@@ -76,14 +75,6 @@ struct pool_config_t
|
||||
void *reshard_state = NULL;
|
||||
};
|
||||
|
||||
struct inode_enc_t
|
||||
{
|
||||
int refs = 0;
|
||||
std::vector<uint8_t> key;
|
||||
// FIXME It may also contain snapshot chain and key information
|
||||
uint32_t bitmap_granularity = 0;
|
||||
};
|
||||
|
||||
struct inode_config_t
|
||||
{
|
||||
uint64_t num = 0;
|
||||
@@ -92,7 +83,6 @@ struct inode_config_t
|
||||
inode_t parent_id = 0;
|
||||
bool readonly = false;
|
||||
bool deleted = false;
|
||||
std::shared_ptr<inode_enc_t> enc;
|
||||
// Arbitrary metadata
|
||||
json11::Json meta;
|
||||
// Change revision of the metadata in etcd
|
||||
@@ -106,7 +96,6 @@ struct inode_watch_t
|
||||
};
|
||||
|
||||
struct http_co_t;
|
||||
struct http_context_t;
|
||||
|
||||
struct __attribute__((visibility("default"))) etcd_state_client_t
|
||||
{
|
||||
@@ -136,13 +125,9 @@ public:
|
||||
uint32_t global_immediate_commit = IMMEDIATE_NONE;
|
||||
|
||||
std::string etcd_prefix;
|
||||
std::string etcd_client_cert;
|
||||
std::string etcd_client_key;
|
||||
std::string etcd_ca;
|
||||
int log_level = 0;
|
||||
timerfd_manager_t *tfd = NULL;
|
||||
|
||||
http_context_t *http_ctx = NULL;
|
||||
http_co_t *etcd_watch_ws = NULL, *keepalive_client = NULL;
|
||||
int etcd_watches_initialised = 0;
|
||||
uint64_t etcd_watch_revision_config = 0;
|
||||
@@ -175,7 +160,6 @@ public:
|
||||
json11::Json::object serialize_inode_cfg(inode_config_t *cfg);
|
||||
etcd_kv_t parse_etcd_kv(const json11::Json & kv_json);
|
||||
std::vector<std::string> get_addresses();
|
||||
http_context_t *get_http_ctx();
|
||||
void etcd_call_oneshot(std::string etcd_address, std::string api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback);
|
||||
void etcd_call(std::string api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
|
||||
void etcd_txn(json11::Json txn, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
|
||||
|
||||
+68
-433
@@ -10,17 +10,9 @@
|
||||
#include <unistd.h>
|
||||
#include <fcntl.h>
|
||||
#include <string.h>
|
||||
#include <assert.h>
|
||||
|
||||
#include <stdexcept>
|
||||
|
||||
#ifdef WITH_OPENSSL
|
||||
#include <openssl/bio.h>
|
||||
#include <openssl/err.h>
|
||||
#include <openssl/pem.h>
|
||||
#include <openssl/ssl.h>
|
||||
#endif
|
||||
|
||||
#include "addr_util.h"
|
||||
#include "str_util.h"
|
||||
#include "json_util.h"
|
||||
@@ -32,51 +24,14 @@
|
||||
|
||||
static std::string ws_format_frame(int type, uint64_t size);
|
||||
static bool ws_parse_frame(std::string & buf, uint8_t & type, std::string & res);
|
||||
static void parse_http_headers(std::string & res, http_message_t *parsed, bool is_request);
|
||||
|
||||
struct http_context_t
|
||||
{
|
||||
std::string ssl_cert;
|
||||
std::string ssl_key;
|
||||
std::string ssl_ca;
|
||||
|
||||
#ifdef WITH_OPENSSL
|
||||
SSL_CTX *ssl_ctx = NULL;
|
||||
#endif
|
||||
|
||||
~http_context_t()
|
||||
{
|
||||
#ifdef WITH_OPENSSL
|
||||
if (ssl_ctx)
|
||||
{
|
||||
SSL_CTX_free(ssl_ctx);
|
||||
ssl_ctx = NULL;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
};
|
||||
|
||||
struct http_call_t
|
||||
{
|
||||
std::string host;
|
||||
std::string request;
|
||||
http_options_t options;
|
||||
std::function<void(http_message_t *)> cb;
|
||||
};
|
||||
static void parse_http_headers(std::string & res, http_response_t *parsed);
|
||||
|
||||
struct http_co_t
|
||||
{
|
||||
http_context_t *ctx = NULL;
|
||||
#ifdef WITH_OPENSSL
|
||||
SSL *ssl_cli = NULL;
|
||||
BIO *ssl_bio = NULL;
|
||||
#endif
|
||||
|
||||
timerfd_manager_t *tfd;
|
||||
std::function<void(http_message_t*)> response_callback;
|
||||
std::function<void(const http_response_t*)> response_callback;
|
||||
|
||||
int request_timeout = 0;
|
||||
bool ssl = false;
|
||||
std::string host;
|
||||
std::string request;
|
||||
std::string ws_outbox;
|
||||
@@ -84,7 +39,7 @@ struct http_co_t
|
||||
bool want_streaming;
|
||||
bool keepalive;
|
||||
|
||||
std::vector<http_call_t> keepalive_queue;
|
||||
std::vector<std::function<void()>> keepalive_queue;
|
||||
|
||||
int state = 0;
|
||||
std::string connected_host;
|
||||
@@ -92,10 +47,10 @@ struct http_co_t
|
||||
int timeout_id = -1;
|
||||
int epoll_events = 0;
|
||||
int sent = 0;
|
||||
std::vector<uint8_t> rbuf;
|
||||
std::vector<char> rbuf;
|
||||
iovec read_iov, send_iov;
|
||||
msghdr read_msg = { 0 }, send_msg = { 0 };
|
||||
http_message_t parsed;
|
||||
http_response_t parsed;
|
||||
uint64_t target_response_size = 0;
|
||||
|
||||
int onstack = 0;
|
||||
@@ -115,14 +70,9 @@ struct http_co_t
|
||||
void submit_read(bool check_timeout);
|
||||
void submit_send();
|
||||
bool handle_read();
|
||||
#ifdef WITH_OPENSSL
|
||||
bool do_ssl_handshake(bool init_send);
|
||||
void on_ssl_error(int res);
|
||||
#endif
|
||||
void post_message(uint8_t type, const std::string & msg);
|
||||
void reply(const std::string & msg);
|
||||
void send_request(const std::string & host, const std::string & request,
|
||||
const http_options_t & options, std::function<void(http_message_t *response)> response_callback);
|
||||
const http_options_t & options, std::function<void(const http_response_t *response)> response_callback);
|
||||
};
|
||||
|
||||
#define HTTP_CO_CLOSED 0
|
||||
@@ -133,64 +83,20 @@ struct http_co_t
|
||||
#define HTTP_CO_WEBSOCKET 5
|
||||
#define HTTP_CO_CHUNKED 6
|
||||
#define HTTP_CO_KEEPALIVE 7
|
||||
#define HTTP_CO_SERVER 8
|
||||
#define HTTP_CO_REQ_HDR_RECEIVED 9
|
||||
#define HTTP_CO_REQUEST_RECEIVED 10
|
||||
|
||||
#define DEFAULT_TIMEOUT 5000
|
||||
|
||||
http_context_t* http_context_init(const std::string & ssl_cert, const std::string & ssl_key,
|
||||
const std::string & ssl_ca, bool verify_peer, std::string & error)
|
||||
{
|
||||
http_context_t *ctx = new http_context_t;
|
||||
#ifdef WITH_OPENSSL
|
||||
SSL_CTX *ssl_ctx = SSL_CTX_new(TLS_method());
|
||||
ctx->ssl_cert = ssl_cert;
|
||||
ctx->ssl_key = ssl_key;
|
||||
ctx->ssl_ca = ssl_ca;
|
||||
ctx->ssl_ctx = ssl_ctx;
|
||||
if (!ssl_ctx)
|
||||
goto init_err;
|
||||
SSL_CTX_set_verify(ssl_ctx, verify_peer ? SSL_VERIFY_PEER : SSL_VERIFY_NONE, NULL);
|
||||
if (!SSL_CTX_set_min_proto_version(ssl_ctx, TLS1_2_VERSION))
|
||||
goto init_err;
|
||||
if ((ssl_ca != "")
|
||||
? !SSL_CTX_load_verify_locations(ssl_ctx, ssl_ca.c_str(), NULL)
|
||||
: !SSL_CTX_set_default_verify_paths(ssl_ctx))
|
||||
goto init_err;
|
||||
if (ssl_cert != "" && ssl_key != "" &&
|
||||
(!SSL_CTX_use_certificate_file(ssl_ctx, ssl_cert.c_str(), SSL_FILETYPE_PEM) ||
|
||||
!SSL_CTX_use_PrivateKey_file(ssl_ctx, ssl_key.c_str(), SSL_FILETYPE_PEM)))
|
||||
goto init_err;
|
||||
#endif
|
||||
return ctx;
|
||||
init_err:
|
||||
error = std::string("openssl initialization failed: ")+ERR_error_string(ERR_get_error(), NULL);
|
||||
delete ctx;
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void http_context_destroy(http_context_t *ctx)
|
||||
{
|
||||
delete ctx;
|
||||
}
|
||||
|
||||
http_co_t *http_init(timerfd_manager_t *tfd, http_context_t *ctx)
|
||||
http_co_t *http_init(timerfd_manager_t *tfd)
|
||||
{
|
||||
http_co_t *handler = new http_co_t();
|
||||
handler->tfd = tfd;
|
||||
handler->state = HTTP_CO_CLOSED;
|
||||
handler->ctx = ctx;
|
||||
return handler;
|
||||
}
|
||||
|
||||
void open_websocket(http_co_t *handler, const std::string & host, const std::string & path,
|
||||
const http_options_t & options, std::function<void(http_message_t *msg)> response_callback)
|
||||
http_co_t* open_websocket(timerfd_manager_t *tfd, const std::string & host, const std::string & path,
|
||||
int timeout, std::function<void(const http_response_t *msg)> response_callback)
|
||||
{
|
||||
if (handler->state == HTTP_CO_KEEPALIVE && (handler->connected_host != host || handler->ssl != options.ssl))
|
||||
handler->close_connection();
|
||||
if (handler->state != HTTP_CO_KEEPALIVE && handler->state != HTTP_CO_CLOSED)
|
||||
throw std::runtime_error("Attempt to open websocket on a keepalive stream");
|
||||
std::string request = "GET "+path+" HTTP/1.1\r\n"
|
||||
"Host: "+host+"\r\n"
|
||||
"Upgrade: websocket\r\n"
|
||||
@@ -198,54 +104,29 @@ void open_websocket(http_co_t *handler, const std::string & host, const std::str
|
||||
"Sec-WebSocket-Key: x3JJHMbDL1EzLkh9GBhXDw==\r\n"
|
||||
"Sec-WebSocket-Version: 13\r\n"
|
||||
"\r\n";
|
||||
http_co_t *handler = new http_co_t();
|
||||
handler->tfd = tfd;
|
||||
handler->state = HTTP_CO_CLOSED;
|
||||
handler->host = host;
|
||||
handler->request_timeout = options.timeout < 0 ? -1 : (options.timeout == 0 ? DEFAULT_TIMEOUT : options.timeout);
|
||||
handler->request_timeout = timeout < 0 ? -1 : (timeout == 0 ? DEFAULT_TIMEOUT : timeout);
|
||||
handler->want_streaming = false;
|
||||
handler->keepalive = false;
|
||||
handler->ssl = options.ssl;
|
||||
handler->request = request;
|
||||
handler->response_callback = response_callback;
|
||||
handler->ws_outbox = "";
|
||||
handler->response = "";
|
||||
handler->sent = 0;
|
||||
handler->parsed = {};
|
||||
handler->start_ws_connection();
|
||||
return handler;
|
||||
}
|
||||
|
||||
void http_request(http_co_t *handler, const std::string & host, const std::string & request,
|
||||
const http_options_t & options, std::function<void(http_message_t *response)> response_callback)
|
||||
const http_options_t & options, std::function<void(const http_response_t *response)> response_callback)
|
||||
{
|
||||
handler->send_request(host, request, options, response_callback);
|
||||
}
|
||||
|
||||
void http_serve(http_co_t *handler, int peer_fd, const http_options_t & options, std::function<void(http_message_t *msg)> request_callback)
|
||||
{
|
||||
if (handler->state != HTTP_CO_SERVER || handler->peer_fd != peer_fd)
|
||||
handler->close_connection();
|
||||
handler->host = "";
|
||||
handler->request_timeout = options.timeout < 0 ? -1 : (options.timeout == 0 ? DEFAULT_TIMEOUT : options.timeout);
|
||||
handler->want_streaming = false;
|
||||
handler->keepalive = false;
|
||||
handler->ssl = options.ssl;
|
||||
handler->request = "";
|
||||
handler->response_callback = request_callback;
|
||||
handler->ws_outbox = "";
|
||||
handler->response = "";
|
||||
handler->sent = 0;
|
||||
handler->parsed = {};
|
||||
handler->peer_fd = peer_fd;
|
||||
handler->state = HTTP_CO_SERVER;
|
||||
handler->tfd->set_fd_handler(peer_fd, false, [handler](int peer_fd, int epoll_events)
|
||||
{
|
||||
handler->epoll_events |= epoll_events;
|
||||
handler->handle_events();
|
||||
});
|
||||
}
|
||||
|
||||
void http_co_t::run_cb_and_clear()
|
||||
{
|
||||
parsed.eof = true;
|
||||
std::function<void(http_message_t*)> cb;
|
||||
std::function<void(const http_response_t*)> cb;
|
||||
cb.swap(response_callback);
|
||||
// Call callback after clearing it because otherwise we may hit reenterability problems
|
||||
if (cb != NULL)
|
||||
@@ -254,7 +135,7 @@ void http_co_t::run_cb_and_clear()
|
||||
}
|
||||
|
||||
void http_co_t::send_request(const std::string & host, const std::string & request,
|
||||
const http_options_t & options, std::function<void(http_message_t *response)> response_callback)
|
||||
const http_options_t & options, std::function<void(const http_response_t *response)> response_callback)
|
||||
{
|
||||
stackin();
|
||||
if (state == HTTP_CO_WEBSOCKET)
|
||||
@@ -264,18 +145,20 @@ void http_co_t::send_request(const std::string & host, const std::string & reque
|
||||
}
|
||||
else if (state != HTTP_CO_KEEPALIVE && state != HTTP_CO_CLOSED)
|
||||
{
|
||||
keepalive_queue.emplace_back((http_call_t){ host, request, options, std::move(response_callback) });
|
||||
keepalive_queue.push_back([this, host, request, options, response_callback]()
|
||||
{
|
||||
this->send_request(host, request, options, response_callback);
|
||||
});
|
||||
stackout();
|
||||
return;
|
||||
}
|
||||
if (state == HTTP_CO_KEEPALIVE && (connected_host != host || ssl != options.ssl))
|
||||
if (state == HTTP_CO_KEEPALIVE && connected_host != host)
|
||||
{
|
||||
close_connection();
|
||||
}
|
||||
this->request_timeout = options.timeout < 0 ? 0 : (options.timeout == 0 ? DEFAULT_TIMEOUT : options.timeout);
|
||||
this->want_streaming = options.want_streaming;
|
||||
this->keepalive = options.keepalive;
|
||||
this->ssl = options.ssl;
|
||||
this->host = host;
|
||||
this->request = request;
|
||||
this->response = "";
|
||||
@@ -307,7 +190,7 @@ void http_co_t::send_request(const std::string & host, const std::string & reque
|
||||
else
|
||||
{
|
||||
close_connection();
|
||||
parsed = { .error = "HTTP request timed out", .status_code = ETIMEDOUT };
|
||||
parsed = { .error = "HTTP request timed out" };
|
||||
run_cb_and_clear();
|
||||
}
|
||||
stackout();
|
||||
@@ -321,11 +204,6 @@ void http_post_message(http_co_t *handler, uint8_t type, const std::string & msg
|
||||
handler->post_message(type, msg);
|
||||
}
|
||||
|
||||
void http_reply(http_co_t *handler, const std::string & reply)
|
||||
{
|
||||
handler->reply(reply);
|
||||
}
|
||||
|
||||
void http_co_t::post_message(uint8_t type, const std::string & msg)
|
||||
{
|
||||
stackin();
|
||||
@@ -335,8 +213,7 @@ void http_co_t::post_message(uint8_t type, const std::string & msg)
|
||||
request += msg;
|
||||
submit_send();
|
||||
}
|
||||
else if (state == HTTP_CO_KEEPALIVE || state == HTTP_CO_CHUNKED ||
|
||||
state == HTTP_CO_SERVER || state == HTTP_CO_REQ_HDR_RECEIVED || state == HTTP_CO_REQUEST_RECEIVED)
|
||||
else if (state == HTTP_CO_KEEPALIVE || state == HTTP_CO_CHUNKED)
|
||||
{
|
||||
throw std::runtime_error("Attempt to send websocket message on a regular HTTP connection");
|
||||
}
|
||||
@@ -348,29 +225,12 @@ void http_co_t::post_message(uint8_t type, const std::string & msg)
|
||||
stackout();
|
||||
}
|
||||
|
||||
void http_co_t::reply(const std::string & reply)
|
||||
{
|
||||
stackin();
|
||||
if (state != HTTP_CO_REQUEST_RECEIVED)
|
||||
{
|
||||
throw std::runtime_error("Attempt to send HTTP response in invalid connection state");
|
||||
}
|
||||
request += reply;
|
||||
submit_send();
|
||||
stackout();
|
||||
}
|
||||
|
||||
void http_destroy(http_co_t *handler)
|
||||
void http_close(http_co_t *handler)
|
||||
{
|
||||
handler->end();
|
||||
}
|
||||
|
||||
void http_close(http_co_t *handler)
|
||||
{
|
||||
handler->close_connection();
|
||||
}
|
||||
|
||||
void http_message_t::parse_json_response(std::string & error, json11::Json & r) const
|
||||
void http_response_t::parse_json_response(std::string & error, json11::Json & r) const
|
||||
{
|
||||
if (this->error != "")
|
||||
{
|
||||
@@ -417,15 +277,6 @@ void http_co_t::close_connection()
|
||||
close(peer_fd);
|
||||
peer_fd = -1;
|
||||
}
|
||||
#ifdef WITH_OPENSSL
|
||||
if (ssl_cli)
|
||||
{
|
||||
// Frees client and bios at once
|
||||
SSL_free(ssl_cli);
|
||||
ssl_cli = NULL;
|
||||
}
|
||||
ssl_bio = NULL;
|
||||
#endif
|
||||
state = HTTP_CO_CLOSED;
|
||||
connected_host = "";
|
||||
response = "";
|
||||
@@ -444,7 +295,7 @@ void http_co_t::start_ws_connection()
|
||||
if (state != HTTP_CO_WEBSOCKET)
|
||||
{
|
||||
close_connection();
|
||||
parsed = { .error = "Websocket connection timed out", .status_code = ETIMEDOUT };
|
||||
parsed = { .error = "Websocket connection timed out" };
|
||||
run_cb_and_clear();
|
||||
}
|
||||
stackout();
|
||||
@@ -460,7 +311,7 @@ void http_co_t::start_connection()
|
||||
if (!string_to_addr(host.c_str(), 1, 80, &addr))
|
||||
{
|
||||
close_connection();
|
||||
parsed = { .error = "Invalid address: "+host, .status_code = EINVAL };
|
||||
parsed = { .error = "Invalid address: "+host };
|
||||
run_cb_and_clear();
|
||||
stackout();
|
||||
return;
|
||||
@@ -469,56 +320,19 @@ void http_co_t::start_connection()
|
||||
if (peer_fd < 0)
|
||||
{
|
||||
close_connection();
|
||||
parsed = { .error = std::string("socket: ")+strerror(errno), .status_code = errno };
|
||||
parsed = { .error = std::string("socket: ")+strerror(errno) };
|
||||
run_cb_and_clear();
|
||||
stackout();
|
||||
return;
|
||||
}
|
||||
fcntl(peer_fd, F_SETFL, fcntl(peer_fd, F_GETFL, 0) | O_NONBLOCK);
|
||||
epoll_events = 0;
|
||||
#ifdef WITH_OPENSSL
|
||||
// https://wiki.openssl.org/index.php/Hostname_validation
|
||||
if (ssl)
|
||||
{
|
||||
if (!ctx)
|
||||
goto init_err;
|
||||
ssl_bio = BIO_new(BIO_s_socket());
|
||||
if (!ssl_bio)
|
||||
goto init_err;
|
||||
if (!BIO_set_fd(ssl_bio, peer_fd, BIO_NOCLOSE))
|
||||
goto init_err;
|
||||
ssl_cli = SSL_new(ctx->ssl_ctx);
|
||||
if (!ssl_cli)
|
||||
goto init_err;
|
||||
SSL_set_bio(ssl_cli, ssl_bio, ssl_bio);
|
||||
if (!SSL_set_tlsext_host_name(ssl_cli, host.c_str()))
|
||||
{
|
||||
init_err:
|
||||
if (ssl_cli)
|
||||
{
|
||||
SSL_free(ssl_cli);
|
||||
ssl_cli = NULL;
|
||||
}
|
||||
else if (ssl_bio)
|
||||
{
|
||||
BIO_free(ssl_bio);
|
||||
ssl_bio = NULL;
|
||||
}
|
||||
parsed = { .error = std::string("openssl initialization failed: ")+ERR_error_string(ERR_get_error(), NULL) };
|
||||
response_callback(&parsed);
|
||||
response_callback = NULL;
|
||||
stackout();
|
||||
return;
|
||||
}
|
||||
SSL_set_connect_state(ssl_cli);
|
||||
}
|
||||
#endif
|
||||
// Finally call connect
|
||||
int r = ::connect(peer_fd, (sockaddr*)&addr, sizeof(addr));
|
||||
if (r < 0 && errno != EINPROGRESS)
|
||||
{
|
||||
close_connection();
|
||||
parsed = { .error = std::string("connect: ")+strerror(errno), .status_code = errno };
|
||||
parsed = { .error = std::string("connect: ")+strerror(errno) };
|
||||
run_cb_and_clear();
|
||||
stackout();
|
||||
return;
|
||||
@@ -553,8 +367,6 @@ void http_co_t::handle_events()
|
||||
{
|
||||
if (state == HTTP_CO_HEADERS_RECEIVED)
|
||||
std::swap(parsed.body, response);
|
||||
else if (state == HTTP_CO_SERVER || state == HTTP_CO_REQ_HDR_RECEIVED || state == HTTP_CO_REQUEST_RECEIVED)
|
||||
parsed = { .error = "client has disconnected normally" };
|
||||
close_connection();
|
||||
run_cb_and_clear();
|
||||
break;
|
||||
@@ -576,7 +388,7 @@ void http_co_t::handle_connect_result()
|
||||
if (result != 0)
|
||||
{
|
||||
close_connection();
|
||||
parsed = { .error = std::string("connect: ")+strerror(result), .status_code = result };
|
||||
parsed = { .error = std::string("connect: ")+strerror(result) };
|
||||
run_cb_and_clear();
|
||||
stackout();
|
||||
return;
|
||||
@@ -596,44 +408,18 @@ void http_co_t::handle_connect_result()
|
||||
void http_co_t::submit_send()
|
||||
{
|
||||
stackin();
|
||||
ssize_t res = 0;
|
||||
int res;
|
||||
again:
|
||||
if (sent < request.size())
|
||||
{
|
||||
send_iov = (iovec){ .iov_base = (void*)(request.data()+sent), .iov_len = request.size()-sent };
|
||||
#ifdef WITH_OPENSSL
|
||||
if (!ssl)
|
||||
#endif
|
||||
send_iov = (iovec){ .iov_base = (void*)(request.c_str()+sent), .iov_len = request.size()-sent };
|
||||
send_msg.msg_iov = &send_iov;
|
||||
send_msg.msg_iovlen = 1;
|
||||
res = sendmsg(peer_fd, &send_msg, MSG_NOSIGNAL);
|
||||
if (res < 0)
|
||||
{
|
||||
send_msg.msg_iov = &send_iov;
|
||||
send_msg.msg_iovlen = 1;
|
||||
res = sendmsg(peer_fd, &send_msg, MSG_NOSIGNAL);
|
||||
if (res < 0)
|
||||
res = -errno;
|
||||
res = -errno;
|
||||
}
|
||||
#ifdef WITH_OPENSSL
|
||||
else
|
||||
{
|
||||
if (!do_ssl_handshake(false))
|
||||
goto out;
|
||||
int ok = SSL_write_ex(ssl_cli, send_iov.iov_base, send_iov.iov_len, (size_t*)&res);
|
||||
if (!ok)
|
||||
{
|
||||
res = SSL_get_error(ssl_cli, ok);
|
||||
if (res == SSL_ERROR_WANT_WRITE || res == 0)
|
||||
res = 0;
|
||||
else if (res == SSL_ERROR_WANT_READ)
|
||||
goto out;
|
||||
else if (res == SSL_ERROR_SYSCALL)
|
||||
res = -errno;
|
||||
else
|
||||
{
|
||||
on_ssl_error(res);
|
||||
goto out;
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
if (res == -EAGAIN || res == -EINTR)
|
||||
{
|
||||
res = 0;
|
||||
@@ -641,33 +427,13 @@ again:
|
||||
else if (res < 0)
|
||||
{
|
||||
close_connection();
|
||||
parsed = { .error = std::string("sendmsg: ")+strerror(errno), .status_code = errno };
|
||||
parsed = { .error = std::string("sendmsg: ")+strerror(errno) };
|
||||
run_cb_and_clear();
|
||||
stackout();
|
||||
return;
|
||||
}
|
||||
sent += res;
|
||||
if (state == HTTP_CO_REQUEST_RECEIVED)
|
||||
{
|
||||
if (sent >= request.size())
|
||||
{
|
||||
if (!keepalive)
|
||||
{
|
||||
close_connection();
|
||||
parsed = { .error = "connection is not keep-alive" };
|
||||
run_cb_and_clear();
|
||||
stackout();
|
||||
return;
|
||||
}
|
||||
state = HTTP_CO_SERVER;
|
||||
request = "";
|
||||
sent = 0;
|
||||
}
|
||||
else
|
||||
goto again;
|
||||
handle_read();
|
||||
}
|
||||
else if (state == HTTP_CO_SENDING_REQUEST)
|
||||
if (state == HTTP_CO_SENDING_REQUEST)
|
||||
{
|
||||
if (sent >= request.size())
|
||||
state = HTTP_CO_REQUEST_SENT;
|
||||
@@ -681,53 +447,26 @@ again:
|
||||
goto again;
|
||||
}
|
||||
}
|
||||
out:
|
||||
stackout();
|
||||
}
|
||||
|
||||
void http_co_t::submit_read(bool check_timeout)
|
||||
{
|
||||
stackin();
|
||||
ssize_t res = 0;
|
||||
int res;
|
||||
again:
|
||||
if (rbuf.size() != READ_BUFFER_SIZE)
|
||||
{
|
||||
rbuf.resize(READ_BUFFER_SIZE);
|
||||
}
|
||||
read_iov = { .iov_base = rbuf.data(), .iov_len = READ_BUFFER_SIZE };
|
||||
#ifdef WITH_OPENSSL
|
||||
if (!ssl)
|
||||
#endif
|
||||
read_msg.msg_iov = &read_iov;
|
||||
read_msg.msg_iovlen = 1;
|
||||
res = recvmsg(peer_fd, &read_msg, 0);
|
||||
if (res < 0)
|
||||
{
|
||||
read_msg.msg_iov = &read_iov;
|
||||
read_msg.msg_iovlen = 1;
|
||||
res = recvmsg(peer_fd, &read_msg, 0);
|
||||
if (res < 0)
|
||||
res = -errno;
|
||||
res = -errno;
|
||||
}
|
||||
#ifdef WITH_OPENSSL
|
||||
else
|
||||
{
|
||||
if (!do_ssl_handshake(true))
|
||||
goto out;
|
||||
int ok = SSL_read_ex(ssl_cli, read_iov.iov_base, read_iov.iov_len, (size_t*)&res);
|
||||
if (!ok)
|
||||
{
|
||||
res = SSL_get_error(ssl_cli, ok);
|
||||
if (res == SSL_ERROR_WANT_READ)
|
||||
res = -EAGAIN;
|
||||
else if (res == SSL_ERROR_SYSCALL)
|
||||
res = -errno;
|
||||
else if (res == SSL_ERROR_ZERO_RETURN)
|
||||
res = 0;
|
||||
else
|
||||
{
|
||||
on_ssl_error(res);
|
||||
goto out;
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
if (res == -EAGAIN || res == -EINTR)
|
||||
{
|
||||
if (check_timeout)
|
||||
@@ -738,7 +477,7 @@ again:
|
||||
{
|
||||
// Timeout happened and there is no data to read
|
||||
close_connection();
|
||||
parsed = { .error = "HTTP request timed out", .status_code = ETIMEDOUT };
|
||||
parsed = { .error = "HTTP request timed out" };
|
||||
run_cb_and_clear();
|
||||
}
|
||||
}
|
||||
@@ -753,76 +492,23 @@ again:
|
||||
epoll_events = epoll_events & ~EPOLLIN;
|
||||
if (state == HTTP_CO_HEADERS_RECEIVED)
|
||||
std::swap(parsed.body, response);
|
||||
if (res < 0)
|
||||
parsed = { .error = std::string("recvmsg: ")+strerror(-res), .status_code = (int)-res };
|
||||
else if (state == HTTP_CO_SERVER || state == HTTP_CO_REQ_HDR_RECEIVED || state == HTTP_CO_REQUEST_RECEIVED)
|
||||
parsed = { .error = "client has disconnected normally" };
|
||||
close_connection();
|
||||
if (res < 0)
|
||||
parsed = { .error = std::string("recvmsg: ")+strerror(-res) };
|
||||
run_cb_and_clear();
|
||||
}
|
||||
else
|
||||
{
|
||||
response += std::string((char*)rbuf.data(), res);
|
||||
response += std::string(rbuf.data(), res);
|
||||
handle_read();
|
||||
}
|
||||
out:
|
||||
stackout();
|
||||
}
|
||||
|
||||
#ifdef WITH_OPENSSL
|
||||
void http_co_t::on_ssl_error(int res)
|
||||
{
|
||||
close_connection();
|
||||
if (res == SSL_ERROR_ZERO_RETURN)
|
||||
{
|
||||
// Client closed the connection
|
||||
parsed = { .error = "peer closed the SSL connection" };
|
||||
}
|
||||
else
|
||||
parsed = { .error = std::string("SSL error: ")+ERR_error_string(ERR_get_error(), NULL), .status_code = EIO };
|
||||
run_cb_and_clear();
|
||||
}
|
||||
|
||||
bool http_co_t::do_ssl_handshake(bool init_send)
|
||||
{
|
||||
if (SSL_is_init_finished(ssl_cli))
|
||||
return true;
|
||||
int r;
|
||||
while (1)
|
||||
{
|
||||
r = SSL_do_handshake(ssl_cli);
|
||||
if (r > 0)
|
||||
{
|
||||
// OK
|
||||
if (init_send)
|
||||
submit_send();
|
||||
return true;
|
||||
}
|
||||
r = SSL_get_error(ssl_cli, r);
|
||||
if (r == SSL_ERROR_WANT_READ)
|
||||
{
|
||||
break;
|
||||
}
|
||||
else
|
||||
{
|
||||
int errcode = ERR_get_error();
|
||||
parsed = { .error = ERR_error_string(errcode, NULL), .status_code = EIO };
|
||||
close_connection();
|
||||
run_cb_and_clear();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
#endif
|
||||
|
||||
bool http_co_t::handle_read()
|
||||
{
|
||||
stackin();
|
||||
if (state == HTTP_CO_REQUEST_RECEIVED)
|
||||
{
|
||||
}
|
||||
else if (state == HTTP_CO_REQUEST_SENT)
|
||||
if (state == HTTP_CO_REQUEST_SENT)
|
||||
{
|
||||
int pos = response.find("\r\n\r\n");
|
||||
if (pos >= 0)
|
||||
@@ -834,7 +520,7 @@ bool http_co_t::handle_read()
|
||||
timeout_id = -1;
|
||||
}
|
||||
state = HTTP_CO_HEADERS_RECEIVED;
|
||||
parse_http_headers(response, &parsed, false);
|
||||
parse_http_headers(response, &parsed);
|
||||
if (parsed.status_code == 101 &&
|
||||
parsed.headers.find("sec-websocket-accept") != parsed.headers.end() &&
|
||||
parsed.headers["upgrade"] == "websocket" &&
|
||||
@@ -858,7 +544,7 @@ bool http_co_t::handle_read()
|
||||
{
|
||||
// Sorry, unsupported response
|
||||
close_connection();
|
||||
parsed = { .error = "Response has neither Connection: close, nor Transfer-Encoding: chunked nor Content-Length headers", .status_code = EINVAL };
|
||||
parsed = { .error = "Response has neither Connection: close, nor Transfer-Encoding: chunked nor Content-Length headers" };
|
||||
run_cb_and_clear();
|
||||
stackout();
|
||||
return false;
|
||||
@@ -870,57 +556,14 @@ bool http_co_t::handle_read()
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (state == HTTP_CO_SERVER)
|
||||
{
|
||||
int pos = response.find("\r\n\r\n");
|
||||
if (pos >= 0)
|
||||
{
|
||||
if (timeout_id >= 0)
|
||||
{
|
||||
// Timeout is cleared when headers are received
|
||||
tfd->clear_timer(timeout_id);
|
||||
timeout_id = -1;
|
||||
}
|
||||
state = HTTP_CO_REQ_HDR_RECEIVED;
|
||||
parse_http_headers(response, &parsed, true);
|
||||
auto conn_it = parsed.headers.find("connection");
|
||||
keepalive = (conn_it != parsed.headers.end() && conn_it->second == "keep-alive");
|
||||
auto enc_it = parsed.headers.find("transfer-encoding");
|
||||
if (enc_it != parsed.headers.end())
|
||||
{
|
||||
// Sorry, unsupported request
|
||||
close_connection();
|
||||
parsed = { .error = "Chunked requests are not supported", .status_code = EINVAL };
|
||||
run_cb_and_clear();
|
||||
stackout();
|
||||
return false;
|
||||
}
|
||||
auto len_it = parsed.headers.find("content-length");
|
||||
target_response_size = stoull_full(len_it != parsed.headers.end() ? len_it->second : "");
|
||||
if (!target_response_size)
|
||||
{
|
||||
state = HTTP_CO_REQUEST_RECEIVED;
|
||||
response_callback(&parsed);
|
||||
}
|
||||
}
|
||||
}
|
||||
if ((state == HTTP_CO_HEADERS_RECEIVED || state == HTTP_CO_REQ_HDR_RECEIVED) &&
|
||||
target_response_size > 0 && response.size() >= target_response_size)
|
||||
if (state == HTTP_CO_HEADERS_RECEIVED && target_response_size > 0 && response.size() >= target_response_size)
|
||||
{
|
||||
std::swap(parsed.body, response);
|
||||
if (state == HTTP_CO_REQ_HDR_RECEIVED)
|
||||
{
|
||||
state = HTTP_CO_REQUEST_RECEIVED;
|
||||
response_callback(&parsed);
|
||||
}
|
||||
if (!keepalive)
|
||||
close_connection();
|
||||
else
|
||||
{
|
||||
if (!keepalive)
|
||||
close_connection();
|
||||
else
|
||||
state = HTTP_CO_KEEPALIVE;
|
||||
run_cb_and_clear();
|
||||
}
|
||||
state = HTTP_CO_KEEPALIVE;
|
||||
run_cb_and_clear();
|
||||
}
|
||||
else if (state == HTTP_CO_CHUNKED && response.size() > 0)
|
||||
{
|
||||
@@ -983,34 +626,26 @@ void http_co_t::next_request()
|
||||
{
|
||||
if (keepalive_queue.size() > 0)
|
||||
{
|
||||
auto next = std::move(keepalive_queue[0]);
|
||||
keepalive_queue.erase(keepalive_queue.begin());
|
||||
send_request(next.host, next.request, next.options, next.cb);
|
||||
auto next = keepalive_queue[0];
|
||||
keepalive_queue.erase(keepalive_queue.begin(), keepalive_queue.begin()+1);
|
||||
next();
|
||||
}
|
||||
}
|
||||
|
||||
static void parse_http_headers(std::string & res, http_message_t *parsed, bool is_request)
|
||||
static void parse_http_headers(std::string & res, http_response_t *parsed)
|
||||
{
|
||||
int pos = res.find("\r\n");
|
||||
pos = pos < 0 ? res.length() : pos+2;
|
||||
std::string status_line = res.substr(0, pos);
|
||||
int http_version;
|
||||
char *status_text = NULL;
|
||||
if (!is_request)
|
||||
sscanf(status_line.c_str(), "HTTP/1.%d %d %ms", &http_version, &parsed->status_code, &status_text);
|
||||
if (status_text)
|
||||
{
|
||||
sscanf(status_line.c_str(), "HTTP/1.%d %d %ms", &http_version, &parsed->status_code, &status_text);
|
||||
if (status_text)
|
||||
{
|
||||
parsed->status_line = status_text;
|
||||
// %ms = allocate a buffer
|
||||
free(status_text);
|
||||
status_text = NULL;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// Should be GET/POST / HTTP/1.1
|
||||
parsed->status_line = status_line;
|
||||
parsed->status_line = status_text;
|
||||
// %ms = allocate a buffer
|
||||
free(status_text);
|
||||
status_text = NULL;
|
||||
}
|
||||
int prev = pos;
|
||||
while ((pos = res.find("\r\n", prev)) >= prev)
|
||||
|
||||
@@ -17,19 +17,14 @@
|
||||
|
||||
class timerfd_manager_t;
|
||||
|
||||
#pragma GCC visibility push(default)
|
||||
|
||||
struct http_options_t
|
||||
{
|
||||
int timeout;
|
||||
bool want_streaming;
|
||||
bool keepalive;
|
||||
bool ssl;
|
||||
};
|
||||
|
||||
struct http_context_t;
|
||||
|
||||
struct http_message_t
|
||||
struct http_response_t
|
||||
{
|
||||
std::string error;
|
||||
|
||||
@@ -46,19 +41,10 @@ struct http_message_t
|
||||
// Opened websocket or keepalive HTTP connection
|
||||
struct http_co_t;
|
||||
|
||||
http_context_t* http_context_init(const std::string & ssl_cert, const std::string & ssl_key,
|
||||
const std::string & ssl_ca, bool verify_peer, std::string & error);
|
||||
void http_context_destroy(http_context_t *ctx);
|
||||
http_co_t* http_init(timerfd_manager_t *tfd, http_context_t *ctx = NULL);
|
||||
void open_websocket(http_co_t *handler, const std::string & host, const std::string & path,
|
||||
const http_options_t & options, std::function<void(http_message_t *msg)> on_message);
|
||||
http_co_t* http_init(timerfd_manager_t *tfd);
|
||||
http_co_t* open_websocket(timerfd_manager_t *tfd, const std::string & host, const std::string & path,
|
||||
int timeout, std::function<void(const http_response_t *msg)> on_message);
|
||||
void http_request(http_co_t *handler, const std::string & host, const std::string & request,
|
||||
const http_options_t & options, std::function<void(http_message_t *response)> response_callback);
|
||||
const http_options_t & options, std::function<void(const http_response_t *response)> response_callback);
|
||||
void http_post_message(http_co_t *handler, uint8_t type, const std::string & msg);
|
||||
void http_serve(http_co_t *handler, int peer_fd, const http_options_t & options,
|
||||
std::function<void(http_message_t *msg)> request_callback);
|
||||
void http_reply(http_co_t *handler, const std::string & reply);
|
||||
void http_close(http_co_t *co);
|
||||
void http_destroy(http_co_t *co);
|
||||
|
||||
#pragma GCC visibility pop
|
||||
|
||||
@@ -145,7 +145,7 @@ void osd_messenger_t::init()
|
||||
rdma_contexts = msgr_rdma_context_t::create_all(
|
||||
osd_num && osd_cluster_network_masks.size() ? osd_cluster_network_masks : osd_network_masks,
|
||||
rdma_device != "" ? rdma_device.c_str() : NULL,
|
||||
rdma_port_num, rdma_gid_index, rdma_mtu, log_level
|
||||
rdma_port_num, rdma_gid_index, rdma_mtu, rdma_odp, log_level
|
||||
);
|
||||
if (!rdma_contexts.size())
|
||||
{
|
||||
@@ -289,16 +289,6 @@ osd_messenger_t::~osd_messenger_t()
|
||||
rdmacm_evch = NULL;
|
||||
}
|
||||
#endif
|
||||
#ifdef WITH_OPENSSL
|
||||
for (auto encrypt_ctx: encrypt_ctx_pool)
|
||||
{
|
||||
destroy_aes_xts_encrypt(encrypt_ctx);
|
||||
}
|
||||
for (auto decrypt_ctx: decrypt_ctx_pool)
|
||||
{
|
||||
destroy_aes_xts_decrypt(decrypt_ctx);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void osd_messenger_t::parse_config(const json11::Json & config)
|
||||
@@ -332,10 +322,8 @@ void osd_messenger_t::parse_config(const json11::Json & config)
|
||||
this->rdma_max_msg = config["rdma_max_msg"].uint64_value();
|
||||
if (!this->rdma_max_msg || this->rdma_max_msg > 128*1024*1024)
|
||||
this->rdma_max_msg = 129*1024;
|
||||
this->rdma_odp = config["rdma_odp"].bool_value();
|
||||
#endif
|
||||
this->max_aes_xts_pool_size = config["max_aes_xts_pool_size"].uint64_value();
|
||||
if (!this->max_aes_xts_pool_size)
|
||||
this->max_aes_xts_pool_size = 256;
|
||||
if (!osd_num)
|
||||
this->iothread_count = (uint32_t)config["client_iothread_count"].uint64_value();
|
||||
else
|
||||
@@ -515,7 +503,7 @@ void osd_messenger_t::try_connect_peer_tcp(osd_num_t peer_osd, const char *peer_
|
||||
clients[peer_fd]->peer_state = PEER_CONNECTING;
|
||||
clients[peer_fd]->connect_timeout_id = -1;
|
||||
clients[peer_fd]->osd_num = peer_osd;
|
||||
clients[peer_fd]->in_buf = (uint8_t*)malloc_or_die(receive_buffer_size);
|
||||
clients[peer_fd]->in_buf = malloc_or_die(receive_buffer_size);
|
||||
tfd->set_fd_handler(peer_fd, true, [this](int peer_fd, int epoll_events)
|
||||
{
|
||||
// Either OUT (connected) or HUP
|
||||
@@ -794,7 +782,7 @@ void osd_messenger_t::accept_connections(int listen_fd)
|
||||
cl->peer_port = ntohs(((sockaddr_in*)&addr)->sin_port);
|
||||
cl->peer_fd = peer_fd;
|
||||
cl->peer_state = PEER_CONNECTED;
|
||||
cl->in_buf = (uint8_t*)malloc_or_die(receive_buffer_size);
|
||||
cl->in_buf = malloc_or_die(receive_buffer_size);
|
||||
// Add FD to epoll
|
||||
tfd->set_fd_handler(peer_fd, false, [this](int peer_fd, int epoll_events)
|
||||
{
|
||||
|
||||
+14
-44
@@ -34,6 +34,9 @@
|
||||
|
||||
#define DEFAULT_MIN_ZEROCOPY_SEND_SIZE 32*1024
|
||||
|
||||
#define MSGR_SENDP_HDR 1
|
||||
#define MSGR_SENDP_FREE 2
|
||||
|
||||
struct msgr_sendp_t
|
||||
{
|
||||
osd_op_t *op;
|
||||
@@ -45,13 +48,6 @@ struct msgr_rdma_connection_t;
|
||||
struct msgr_rdma_context_t;
|
||||
#endif
|
||||
|
||||
#ifdef WITH_OPENSSL
|
||||
struct op_aes_xts_encrypt_t;
|
||||
struct op_aes_xts_decrypt_t;
|
||||
void destroy_aes_xts_encrypt(op_aes_xts_encrypt_t *encrypt_ctx);
|
||||
void destroy_aes_xts_decrypt(op_aes_xts_decrypt_t *decrypt_ctx);
|
||||
#endif
|
||||
|
||||
struct osd_client_t
|
||||
{
|
||||
int refs = 0;
|
||||
@@ -67,23 +63,20 @@ struct osd_client_t
|
||||
osd_num_t in_osd_num = 0;
|
||||
bool is_incoming = false;
|
||||
|
||||
uint8_t *in_buf = NULL;
|
||||
void *in_buf = NULL;
|
||||
|
||||
#ifdef WITH_RDMA
|
||||
msgr_rdma_connection_t *rdma_conn = NULL;
|
||||
#endif
|
||||
|
||||
// Read state
|
||||
op_aes_xts_decrypt_t *decrypt_ctx = NULL;
|
||||
int read_ready = 0;
|
||||
osd_op_t *read_op = NULL;
|
||||
size_t read_op_size = 0;
|
||||
size_t read_op_pos = 0;
|
||||
size_t read_op_inline_decrypt_pos = 0;
|
||||
iovec read_iov = { 0 };
|
||||
msghdr read_msg = { 0 };
|
||||
std::vector<iovec> recv_list;
|
||||
size_t recv_list_size = 0;
|
||||
int read_remaining = 0;
|
||||
int read_state = 0;
|
||||
osd_op_buf_list_t recv_list;
|
||||
uint64_t read_op_id = 1;
|
||||
bool check_sequencing = false;
|
||||
bool enable_pg_locks = false;
|
||||
@@ -99,15 +92,10 @@ struct osd_client_t
|
||||
std::set<pool_pg_num_t> dirty_pgs;
|
||||
|
||||
// Write state
|
||||
op_aes_xts_encrypt_t *encrypt_ctx = NULL;
|
||||
std::deque<osd_op_t *> write_ops;
|
||||
osd_op_t *write_op = NULL;
|
||||
size_t write_op_pos = 0;
|
||||
msghdr write_msg = { 0 };
|
||||
int write_state = 0;
|
||||
std::vector<iovec> send_list;
|
||||
size_t send_list_size = 0;
|
||||
std::deque<osd_op_t*> send_free_ops;
|
||||
std::vector<iovec> send_list, next_send_list;
|
||||
std::vector<msgr_sendp_t> outbox, next_outbox;
|
||||
std::vector<osd_op_t*> zc_free_list;
|
||||
|
||||
~osd_client_t();
|
||||
@@ -200,7 +188,6 @@ protected:
|
||||
bool use_sync_send_recv = false;
|
||||
int min_zerocopy_send_size = DEFAULT_MIN_ZEROCOPY_SEND_SIZE;
|
||||
int iothread_count = 0;
|
||||
int max_aes_xts_pool_size = 256;
|
||||
|
||||
#ifdef WITH_RDMA
|
||||
bool use_rdma = true;
|
||||
@@ -213,6 +200,7 @@ protected:
|
||||
std::vector<msgr_rdma_context_t *> rdma_contexts;
|
||||
uint64_t rdma_max_sge = 0, rdma_max_send = 0, rdma_max_recv = 0;
|
||||
uint64_t rdma_max_msg = 0;
|
||||
bool rdma_odp = false;
|
||||
rdma_event_channel *rdmacm_evch = NULL;
|
||||
std::map<rdma_cm_id*, osd_client_t*> rdmacm_connections;
|
||||
std::map<rdma_cm_id*, rdmacm_connecting_t*> rdmacm_connecting;
|
||||
@@ -224,11 +212,6 @@ protected:
|
||||
// We don't use ringloop->set_immediate here because we may have no ringloop in client :)
|
||||
std::vector<osd_op_t*> set_immediate_ops;
|
||||
|
||||
#ifdef WITH_OPENSSL
|
||||
std::vector<op_aes_xts_encrypt_t*> encrypt_ctx_pool;
|
||||
std::vector<op_aes_xts_decrypt_t*> decrypt_ctx_pool;
|
||||
#endif
|
||||
|
||||
public:
|
||||
timerfd_manager_t *tfd = NULL;
|
||||
ring_loop_t *ringloop = NULL;
|
||||
@@ -292,23 +275,9 @@ protected:
|
||||
|
||||
bool try_send(osd_client_t *cl);
|
||||
void handle_send(int result, bool prev, bool more, osd_client_t *cl);
|
||||
bool op_encrypted_copy_data_to(osd_client_t* cl, uint8_t *buf, size_t len, size_t from, size_t & done);
|
||||
size_t op_copy_to(osd_client_t *cl, uint8_t *dst, size_t dst_len);
|
||||
void op_get_write_buffers(osd_client_t *cl, std::vector<iovec> & lst);
|
||||
|
||||
void handle_read(int result, osd_client_t *cl);
|
||||
bool handle_read_buffer(osd_client_t *cl, uint8_t *curbuf, size_t bufsize);
|
||||
bool handle_hdr(osd_client_t *cl);
|
||||
bool allocate_op_buffers(osd_client_t *cl);
|
||||
bool allocate_reply_buffers(osd_client_t *cl, osd_op_t *op);
|
||||
size_t op_copy_from(osd_client_t *cl, uint8_t *src, size_t src_len, size_t & done);
|
||||
bool op_decrypted_copy_data_from(osd_client_t* cl, uint8_t *buf, size_t len, size_t from, size_t & done);
|
||||
void op_decrypt_start(osd_client_t* cl);
|
||||
void op_decrypt_inline(osd_client_t* cl);
|
||||
void op_decrypt_free(osd_client_t* cl);
|
||||
size_t op_get_read_buffers(osd_client_t *cl, std::vector<iovec> & lst);
|
||||
void handle_finished_op(osd_client_t *cl);
|
||||
|
||||
bool handle_read(int result, osd_client_t *cl);
|
||||
bool handle_read_buffer(osd_client_t *cl, void *curbuf, int remain);
|
||||
bool handle_finished_read(osd_client_t *cl);
|
||||
void handle_op_hdr(osd_client_t *cl);
|
||||
bool handle_reply_hdr(osd_client_t *cl);
|
||||
@@ -318,7 +287,8 @@ protected:
|
||||
|
||||
#ifdef WITH_RDMA
|
||||
void try_send_rdma(osd_client_t *cl);
|
||||
int try_send_rdma_copy(osd_client_t *cl, uint8_t *dst, int dst_len);
|
||||
void try_send_rdma_odp(osd_client_t *cl);
|
||||
void try_send_rdma_nodp(osd_client_t *cl);
|
||||
bool init_recv_rdma(osd_client_t *cl);
|
||||
void handle_rdma_events(msgr_rdma_context_t *rdma_context);
|
||||
msgr_rdma_context_t* choose_rdma_context(osd_client_t *cl);
|
||||
|
||||
@@ -1,328 +0,0 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2026+
|
||||
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||
|
||||
#define _XOPEN_SOURCE
|
||||
#include <limits.h>
|
||||
#include <assert.h>
|
||||
|
||||
#include "etcd_state_client.h"
|
||||
#include "messenger.h"
|
||||
#include "msgr_encrypt.h"
|
||||
|
||||
// FIXME Fuck, no streaming...
|
||||
|
||||
op_aes_xts_encrypt_t::op_aes_xts_encrypt_t()
|
||||
{
|
||||
if (!(ctx = EVP_CIPHER_CTX_new()))
|
||||
{
|
||||
ERR_print_errors_fp(stderr);
|
||||
abort();
|
||||
}
|
||||
EVP_CIPHER_CTX_set_padding(ctx, 0);
|
||||
if (EVP_EncryptInit_ex(ctx, EVP_aes_256_xts(), NULL, NULL, NULL) != 1)
|
||||
{
|
||||
ERR_print_errors_fp(stderr);
|
||||
abort();
|
||||
}
|
||||
}
|
||||
|
||||
op_aes_xts_encrypt_t::~op_aes_xts_encrypt_t()
|
||||
{
|
||||
EVP_CIPHER_CTX_free(ctx);
|
||||
}
|
||||
|
||||
void op_aes_xts_encrypt_t::start(const uint8_t *key, uint64_t start_offset, size_t block_size)
|
||||
{
|
||||
this->start_offset = start_offset;
|
||||
this->key = key;
|
||||
this->block_size = block_size;
|
||||
this->offset = 0;
|
||||
}
|
||||
|
||||
void op_aes_xts_encrypt_t::update(uint8_t *in, size_t max_in, uint8_t *out, size_t max_out, size_t & done_in, size_t & done_out)
|
||||
{
|
||||
if (max_in > block_size - offset%block_size)
|
||||
max_in = block_size - offset%block_size;
|
||||
size_t insize = max_in;
|
||||
size_t outsize = ((offset+insize)/16 - offset/16) * 16;
|
||||
if (outsize > max_out)
|
||||
{
|
||||
// encrypt is used to send data through temporary buffer(s),
|
||||
// so we don't care to support fragmenting output into < 16 b parts
|
||||
insize = (max_out < 16 ? 0 : (max_out & ~15) - offset%16);
|
||||
outsize = ((offset+insize)/16 - offset/16) * 16;
|
||||
}
|
||||
assert(insize <= max_in);
|
||||
assert(outsize <= max_out);
|
||||
if (!(offset % block_size))
|
||||
{
|
||||
uint8_t iv[16] = { 0 };
|
||||
*((uint64_t*)iv) = start_offset + offset;
|
||||
if (EVP_EncryptInit_ex(ctx, NULL, NULL, key, iv) != 1)
|
||||
{
|
||||
ERR_print_errors_fp(stderr);
|
||||
abort();
|
||||
}
|
||||
}
|
||||
int actual_out = 0;
|
||||
if (EVP_EncryptUpdate(ctx, out, &actual_out, in, insize) != 1)
|
||||
{
|
||||
ERR_print_errors_fp(stderr);
|
||||
abort();
|
||||
}
|
||||
assert(actual_out == outsize);
|
||||
done_in += insize;
|
||||
done_out += outsize;
|
||||
offset += insize;
|
||||
}
|
||||
|
||||
void destroy_aes_xts_encrypt(op_aes_xts_encrypt_t *encrypt_ctx)
|
||||
{
|
||||
delete encrypt_ctx;
|
||||
}
|
||||
|
||||
op_aes_xts_decrypt_t::op_aes_xts_decrypt_t()
|
||||
{
|
||||
if (!(ctx = EVP_CIPHER_CTX_new()))
|
||||
{
|
||||
ERR_print_errors_fp(stderr);
|
||||
abort();
|
||||
}
|
||||
EVP_CIPHER_CTX_set_padding(ctx, 0);
|
||||
if (EVP_DecryptInit_ex(ctx, EVP_aes_256_xts(), NULL, NULL, NULL) != 1)
|
||||
{
|
||||
ERR_print_errors_fp(stderr);
|
||||
abort();
|
||||
}
|
||||
}
|
||||
|
||||
op_aes_xts_decrypt_t::~op_aes_xts_decrypt_t()
|
||||
{
|
||||
EVP_CIPHER_CTX_free(ctx);
|
||||
}
|
||||
|
||||
void op_aes_xts_decrypt_t::start(const uint8_t *key, uint64_t start_offset, size_t block_size)
|
||||
{
|
||||
this->start_offset = start_offset;
|
||||
this->key = key;
|
||||
this->block_size = block_size;
|
||||
this->in_offset = 0;
|
||||
this->tmp_pos = 16;
|
||||
}
|
||||
|
||||
void op_aes_xts_decrypt_t::update(uint8_t *in, size_t max_in, uint8_t *out, size_t max_out, size_t & done_in, size_t & done_out)
|
||||
{
|
||||
if (max_in > block_size - in_offset%block_size)
|
||||
max_in = block_size - in_offset%block_size;
|
||||
int actual_out = 0;
|
||||
// Write previously buffered block to support small output buffers
|
||||
if (tmp_pos < 16)
|
||||
{
|
||||
size_t tmp_size = 16-tmp_pos;
|
||||
if (tmp_size > max_out)
|
||||
tmp_size = max_out;
|
||||
memcpy(out, tmp_buf+tmp_pos, tmp_size);
|
||||
tmp_pos += tmp_size;
|
||||
done_out += tmp_size;
|
||||
out += tmp_size;
|
||||
max_out -= tmp_size;
|
||||
if (!max_out)
|
||||
return;
|
||||
assert(tmp_pos == 16);
|
||||
}
|
||||
if (!(in_offset % block_size))
|
||||
{
|
||||
uint8_t iv[16] = { 0 };
|
||||
*((uint64_t*)iv) = start_offset+in_offset;
|
||||
if (EVP_DecryptInit_ex(ctx, NULL, NULL, key, iv) != 1)
|
||||
{
|
||||
ERR_print_errors_fp(stderr);
|
||||
abort();
|
||||
}
|
||||
}
|
||||
size_t insize = max_in;
|
||||
size_t outsize = ((in_offset+insize)/16 - in_offset/16) * 16;
|
||||
if (outsize > max_out)
|
||||
{
|
||||
if (max_out < 16)
|
||||
{
|
||||
// We can only decrypt a partial block
|
||||
insize = 16 - in_offset%16;
|
||||
if (EVP_DecryptUpdate(ctx, tmp_buf, &actual_out, in, insize) != 1)
|
||||
{
|
||||
ERR_print_errors_fp(stderr);
|
||||
abort();
|
||||
}
|
||||
assert(actual_out == 16);
|
||||
in_offset += insize;
|
||||
in += insize;
|
||||
max_in -= insize;
|
||||
tmp_pos = 0;
|
||||
return;
|
||||
}
|
||||
// Otherwise, we can decrypt at least some data into <out> directly
|
||||
insize = (max_out & ~15) - (in_offset % 16);
|
||||
outsize = (max_out & ~15);
|
||||
assert(insize < max_in);
|
||||
}
|
||||
if (EVP_DecryptUpdate(ctx, out, &actual_out, in, insize) != 1)
|
||||
{
|
||||
ERR_print_errors_fp(stderr);
|
||||
abort();
|
||||
}
|
||||
assert(actual_out == outsize);
|
||||
in_offset += insize;
|
||||
done_in += insize;
|
||||
done_out += actual_out;
|
||||
}
|
||||
|
||||
void destroy_aes_xts_decrypt(op_aes_xts_decrypt_t *decrypt_ctx)
|
||||
{
|
||||
delete decrypt_ctx;
|
||||
}
|
||||
|
||||
bool osd_messenger_t::op_encrypted_copy_data_to(osd_client_t* cl, uint8_t *enc_buf, size_t enc_len, size_t from, size_t & done)
|
||||
{
|
||||
auto op = cl->write_op;
|
||||
auto & op_pos = cl->write_op_pos;
|
||||
assert(op->req.hdr.opcode == OSD_OP_WRITE);
|
||||
if (!from)
|
||||
{
|
||||
if (!cl->encrypt_ctx)
|
||||
{
|
||||
if (encrypt_ctx_pool.size())
|
||||
{
|
||||
cl->encrypt_ctx = encrypt_ctx_pool.back();
|
||||
encrypt_ctx_pool.pop_back();
|
||||
}
|
||||
else
|
||||
cl->encrypt_ctx = new op_aes_xts_encrypt_t();
|
||||
}
|
||||
assert(op->enc->key.size() == 512/8);
|
||||
cl->encrypt_ctx->start(op->enc->key.data(), op->req.rw.offset, op->enc->bitmap_granularity);
|
||||
}
|
||||
for (int i = 0; i < op->iov.count; i++)
|
||||
{
|
||||
uint8_t *plain = (uint8_t*)op->iov.buf[i].iov_base;
|
||||
size_t plain_len = op->iov.buf[i].iov_len;
|
||||
while (from < plain_len)
|
||||
{
|
||||
size_t done_in = 0;
|
||||
size_t done_out = 0;
|
||||
cl->encrypt_ctx->update(plain+from, plain_len-from, enc_buf+done, enc_len-done, done_in, done_out);
|
||||
done += done_out;
|
||||
op_pos += done_in;
|
||||
from += done_in;
|
||||
if (!done_in)
|
||||
return false;
|
||||
}
|
||||
from -= plain_len;
|
||||
}
|
||||
if (cl->encrypt_ctx)
|
||||
{
|
||||
if (encrypt_ctx_pool.size() > max_aes_xts_pool_size)
|
||||
delete cl->encrypt_ctx;
|
||||
else
|
||||
encrypt_ctx_pool.push_back(cl->encrypt_ctx);
|
||||
cl->encrypt_ctx = NULL;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool osd_messenger_t::op_decrypted_copy_data_from(osd_client_t* cl, uint8_t *enc_buf, size_t enc_len, size_t from, size_t & done)
|
||||
{
|
||||
op_decrypt_start(cl);
|
||||
auto op = cl->read_op;
|
||||
auto & op_pos = cl->read_op_pos;
|
||||
assert(op->req.hdr.opcode == OSD_OP_READ);
|
||||
uint64_t offset = from;
|
||||
for (int i = 0; i < op->iov.count; i++)
|
||||
{
|
||||
uint8_t *plain = (uint8_t*)op->iov.buf[i].iov_base;
|
||||
size_t plain_len = op->iov.buf[i].iov_len;
|
||||
while (from < plain_len)
|
||||
{
|
||||
size_t done_in = 0;
|
||||
size_t done_out = 0;
|
||||
cl->decrypt_ctx->update(enc_buf+done, enc_len-done, plain+from, plain_len-from, done_in, done_out);
|
||||
done += done_in;
|
||||
offset += done_in;
|
||||
op_pos += done_out;
|
||||
from += done_out;
|
||||
if (!done_in)
|
||||
return false;
|
||||
}
|
||||
from -= plain_len;
|
||||
}
|
||||
op_decrypt_free(cl);
|
||||
return true;
|
||||
}
|
||||
|
||||
void osd_messenger_t::op_decrypt_start(osd_client_t* cl)
|
||||
{
|
||||
if (!cl->decrypt_ctx)
|
||||
{
|
||||
if (decrypt_ctx_pool.size())
|
||||
{
|
||||
cl->decrypt_ctx = decrypt_ctx_pool.back();
|
||||
decrypt_ctx_pool.pop_back();
|
||||
}
|
||||
else
|
||||
cl->decrypt_ctx = new op_aes_xts_decrypt_t();
|
||||
assert(cl->read_op->enc->key.size() == 512/8);
|
||||
cl->decrypt_ctx->start(cl->read_op->enc->key.data(), cl->read_op->req.rw.offset, cl->read_op->enc->bitmap_granularity);
|
||||
}
|
||||
}
|
||||
|
||||
void osd_messenger_t::op_decrypt_inline(osd_client_t* cl)
|
||||
{
|
||||
op_decrypt_start(cl);
|
||||
osd_op_t *op = cl->read_op;
|
||||
size_t from_in = cl->read_op_inline_decrypt_pos - OSD_PACKET_SIZE - op->reply.rw.bitmap_len;
|
||||
int i = 0;
|
||||
while (i < op->iov.count && from_in >= op->iov.buf[i].iov_len)
|
||||
{
|
||||
from_in -= op->iov.buf[i].iov_len;
|
||||
i++;
|
||||
}
|
||||
size_t from_out = from_in;
|
||||
int j = i;
|
||||
while (i < op->iov.count && j < op->iov.count)
|
||||
{
|
||||
uint8_t *in = (uint8_t*)op->iov.buf[i].iov_base + from_in;
|
||||
size_t in_len = op->iov.buf[i].iov_len - from_in;
|
||||
uint8_t *out = (uint8_t*)op->iov.buf[j].iov_base + from_out;
|
||||
size_t out_len = op->iov.buf[j].iov_len - from_out;
|
||||
size_t done_in = 0;
|
||||
size_t done_out = 0;
|
||||
cl->decrypt_ctx->update(in, in_len, out, out_len, done_in, done_out);
|
||||
if (done_in >= in_len)
|
||||
{
|
||||
i++;
|
||||
from_in = 0;
|
||||
}
|
||||
else
|
||||
from_in += done_in;
|
||||
if (done_out >= out_len)
|
||||
{
|
||||
j++;
|
||||
from_out = 0;
|
||||
}
|
||||
else
|
||||
from_out += done_out;
|
||||
}
|
||||
assert(j >= op->iov.count);
|
||||
op_decrypt_free(cl);
|
||||
}
|
||||
|
||||
void osd_messenger_t::op_decrypt_free(osd_client_t* cl)
|
||||
{
|
||||
if (cl->decrypt_ctx)
|
||||
{
|
||||
if (decrypt_ctx_pool.size() > max_aes_xts_pool_size)
|
||||
delete cl->decrypt_ctx;
|
||||
else
|
||||
decrypt_ctx_pool.push_back(cl->decrypt_ctx);
|
||||
cl->decrypt_ctx = NULL;
|
||||
}
|
||||
}
|
||||
@@ -1,44 +0,0 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2026+
|
||||
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#include <openssl/conf.h>
|
||||
#include <openssl/evp.h>
|
||||
#include <openssl/err.h>
|
||||
|
||||
struct op_aes_xts_encrypt_t
|
||||
{
|
||||
EVP_CIPHER_CTX *ctx = NULL;
|
||||
uint64_t start_offset = 0;
|
||||
const uint8_t *key = NULL;
|
||||
size_t offset = 0;
|
||||
size_t block_size = 0;
|
||||
|
||||
op_aes_xts_encrypt_t();
|
||||
~op_aes_xts_encrypt_t();
|
||||
|
||||
void start(const uint8_t *key, uint64_t start_offset, size_t block_size);
|
||||
void update(uint8_t *in, size_t max_in, uint8_t *out, size_t max_out, size_t & done_in, size_t & done_out);
|
||||
};
|
||||
|
||||
void destroy_aes_xts_encrypt(op_aes_xts_encrypt_t *encrypt_ctx);
|
||||
|
||||
struct op_aes_xts_decrypt_t
|
||||
{
|
||||
EVP_CIPHER_CTX *ctx = NULL;
|
||||
uint64_t start_offset = 0;
|
||||
const uint8_t *key = NULL;
|
||||
uint8_t tmp_buf[16];
|
||||
size_t tmp_pos = 16;
|
||||
size_t in_offset = 0;
|
||||
size_t block_size = 0;
|
||||
|
||||
op_aes_xts_decrypt_t();
|
||||
~op_aes_xts_decrypt_t();
|
||||
|
||||
void start(const uint8_t *key, uint64_t start_offset, size_t block_size);
|
||||
void update(uint8_t *in, size_t max_in, uint8_t *out, size_t max_out, size_t & done_in, size_t & done_out);
|
||||
};
|
||||
|
||||
void destroy_aes_xts_decrypt(op_aes_xts_decrypt_t *decrypt_ctx);
|
||||
@@ -23,10 +23,6 @@ osd_op_t::~osd_op_t()
|
||||
// So we don't reuse it, but free it every time
|
||||
free(buf);
|
||||
}
|
||||
if (enc_buf)
|
||||
{
|
||||
free(enc_buf);
|
||||
}
|
||||
}
|
||||
|
||||
bool osd_op_t::is_recovery_related()
|
||||
|
||||
@@ -3,8 +3,6 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <memory>
|
||||
|
||||
#include <sys/uio.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
@@ -154,8 +152,6 @@ struct blockstore_op_t;
|
||||
|
||||
struct osd_primary_op_data_t;
|
||||
|
||||
struct inode_enc_t;
|
||||
|
||||
struct __attribute__((visibility("default"))) osd_op_t
|
||||
{
|
||||
timespec tv_begin = { 0 }, tv_end = { 0 };
|
||||
@@ -171,8 +167,6 @@ struct __attribute__((visibility("default"))) osd_op_t
|
||||
unsigned bmp_data = 0;
|
||||
void *bitmap_buf = NULL;
|
||||
void *rmw_buf = NULL;
|
||||
std::shared_ptr<inode_enc_t> enc;
|
||||
uint8_t *enc_buf = NULL;
|
||||
osd_primary_op_data_t* op_data = NULL;
|
||||
std::function<void(osd_op_t*)> callback;
|
||||
|
||||
|
||||
+162
-44
@@ -59,6 +59,8 @@ msgr_rdma_context_t::~msgr_rdma_context_t()
|
||||
ibv_destroy_cq(cq);
|
||||
if (channel)
|
||||
ibv_destroy_comp_channel(channel);
|
||||
if (mr)
|
||||
ibv_dereg_mr(mr);
|
||||
if (pd)
|
||||
ibv_dealloc_pd(pd);
|
||||
if (context && !is_cm)
|
||||
@@ -180,7 +182,7 @@ static int match_port_gid(const std::vector<addr_mask_t> & osd_network_masks, ib
|
||||
#endif
|
||||
|
||||
std::vector<msgr_rdma_context_t*> msgr_rdma_context_t::create_all(const std::vector<addr_mask_t> & osd_network_masks,
|
||||
const char *sel_dev_name, int sel_port_num, int sel_gid_index, uint32_t sel_mtu, int log_level)
|
||||
const char *sel_dev_name, int sel_port_num, int sel_gid_index, uint32_t sel_mtu, bool odp, int log_level)
|
||||
{
|
||||
int res;
|
||||
std::vector<msgr_rdma_context_t*> ret;
|
||||
@@ -269,7 +271,7 @@ std::vector<msgr_rdma_context_t*> msgr_rdma_context_t::create_all(const std::vec
|
||||
{
|
||||
if (log_level > 0)
|
||||
log_rdma_dev_port_gid(dev, port_num, best_gid_idx, port_mtu, best_gidx);
|
||||
auto ctx = msgr_rdma_context_t::create(dev, portinfo, port_num, best_gid_idx, port_mtu, log_level);
|
||||
auto ctx = msgr_rdma_context_t::create(dev, portinfo, port_num, best_gid_idx, port_mtu, odp, log_level);
|
||||
if (ctx)
|
||||
{
|
||||
ctx->net_mask = osd_network_masks[net_num];
|
||||
@@ -289,7 +291,7 @@ std::vector<msgr_rdma_context_t*> msgr_rdma_context_t::create_all(const std::vec
|
||||
log_rdma_dev_port_gid(dev, port_num, best_gid_idx, port_mtu, gidx);
|
||||
}
|
||||
#endif
|
||||
auto ctx = msgr_rdma_context_t::create(dev, portinfo, port_num, best_gid_idx, port_mtu, log_level);
|
||||
auto ctx = msgr_rdma_context_t::create(dev, portinfo, port_num, best_gid_idx, port_mtu, odp, log_level);
|
||||
if (ctx)
|
||||
ret.push_back(ctx);
|
||||
}
|
||||
@@ -304,7 +306,7 @@ cleanup:
|
||||
return ret;
|
||||
}
|
||||
|
||||
msgr_rdma_context_t *msgr_rdma_context_t::create(ibv_device *dev, ibv_port_attr & portinfo, int ib_port, int gid_index, uint32_t mtu, int log_level)
|
||||
msgr_rdma_context_t *msgr_rdma_context_t::create(ibv_device *dev, ibv_port_attr & portinfo, int ib_port, int gid_index, uint32_t mtu, bool odp, int log_level)
|
||||
{
|
||||
msgr_rdma_context_t *ctx = new msgr_rdma_context_t();
|
||||
ibv_context *context = ibv_open_device(dev);
|
||||
@@ -344,6 +346,30 @@ msgr_rdma_context_t *msgr_rdma_context_t::create(ibv_device *dev, ibv_port_attr
|
||||
goto cleanup;
|
||||
}
|
||||
|
||||
ctx->odp = odp;
|
||||
if (ctx->odp)
|
||||
{
|
||||
if (!(ctx->attrx.odp_caps.general_caps & IBV_ODP_SUPPORT) ||
|
||||
!(ctx->attrx.odp_caps.general_caps & IBV_ODP_SUPPORT_IMPLICIT) ||
|
||||
!(ctx->attrx.odp_caps.per_transport_caps.rc_odp_caps & IBV_ODP_SUPPORT_SEND) ||
|
||||
!(ctx->attrx.odp_caps.per_transport_caps.rc_odp_caps & IBV_ODP_SUPPORT_RECV))
|
||||
{
|
||||
ctx->odp = false;
|
||||
if (log_level > 0)
|
||||
fprintf(stderr, "The RDMA device isn't implicit ODP (On-Demand Paging) capable, disabling it\n");
|
||||
}
|
||||
}
|
||||
|
||||
if (ctx->odp)
|
||||
{
|
||||
ctx->mr = ibv_reg_mr(ctx->pd, NULL, SIZE_MAX, IBV_ACCESS_LOCAL_WRITE | IBV_ACCESS_ON_DEMAND);
|
||||
if (!ctx->mr)
|
||||
{
|
||||
fprintf(stderr, "Couldn't register RDMA memory region\n");
|
||||
goto cleanup;
|
||||
}
|
||||
}
|
||||
|
||||
ctx->channel = ibv_create_comp_channel(ctx->context);
|
||||
if (!ctx->channel)
|
||||
{
|
||||
@@ -362,6 +388,8 @@ msgr_rdma_context_t *msgr_rdma_context_t::create(ibv_device *dev, ibv_port_attr
|
||||
return ctx;
|
||||
|
||||
cleanup:
|
||||
if (context)
|
||||
ibv_close_device(context);
|
||||
delete ctx;
|
||||
return NULL;
|
||||
}
|
||||
@@ -555,34 +583,74 @@ static void try_send_rdma_wr(osd_client_t *cl, ibv_sge *sge, int op_sge)
|
||||
cl->rdma_conn->cur_send++;
|
||||
}
|
||||
|
||||
int osd_messenger_t::try_send_rdma_copy(osd_client_t *cl, uint8_t *dst, int dst_len)
|
||||
static int try_send_rdma_copy(osd_client_t *cl, uint8_t *dst, int dst_len)
|
||||
{
|
||||
auto rc = cl->rdma_conn;
|
||||
int total_dst_len = dst_len;
|
||||
while (dst_len > 0 && cl->write_ops.size())
|
||||
while (dst_len > 0 && rc->send_pos < cl->send_list.size())
|
||||
{
|
||||
if (!cl->write_op)
|
||||
iovec & iov = cl->send_list[rc->send_pos];
|
||||
uint32_t len = (uint32_t)(iov.iov_len-rc->send_buf_pos < dst_len
|
||||
? iov.iov_len-rc->send_buf_pos : dst_len);
|
||||
memcpy(dst, (uint8_t*)iov.iov_base+rc->send_buf_pos, len);
|
||||
dst += len;
|
||||
dst_len -= len;
|
||||
rc->send_buf_pos += len;
|
||||
if (rc->send_buf_pos >= iov.iov_len)
|
||||
{
|
||||
cl->write_op = cl->write_ops.front();
|
||||
cl->write_ops.pop_front();
|
||||
}
|
||||
osd_op_t *op = cl->write_op;
|
||||
size_t copied = op_copy_to(cl, dst, dst_len);
|
||||
if (!copied)
|
||||
{
|
||||
break;
|
||||
}
|
||||
dst += copied;
|
||||
dst_len -= copied;
|
||||
if (!cl->write_op && op->op_type == OSD_OP_IN)
|
||||
{
|
||||
// this is a reply, free the op after sending it
|
||||
cl->send_free_ops.push_back(op);
|
||||
rc->send_pos++;
|
||||
rc->send_buf_pos = 0;
|
||||
}
|
||||
}
|
||||
return total_dst_len-dst_len;
|
||||
}
|
||||
|
||||
void osd_messenger_t::try_send_rdma(osd_client_t *cl)
|
||||
void osd_messenger_t::try_send_rdma_odp(osd_client_t *cl)
|
||||
{
|
||||
auto rc = cl->rdma_conn;
|
||||
if (!cl->send_list.size() || rc->cur_send >= rc->max_send)
|
||||
{
|
||||
return;
|
||||
}
|
||||
uint64_t op_size = 0, op_sge = 0;
|
||||
ibv_sge sge[rc->max_sge];
|
||||
while (rc->send_pos < cl->send_list.size())
|
||||
{
|
||||
iovec & iov = cl->send_list[rc->send_pos];
|
||||
if (op_size >= rc->max_msg || op_sge >= rc->max_sge)
|
||||
{
|
||||
rc->send_sizes.push_back(op_size);
|
||||
try_send_rdma_wr(cl, sge, op_sge);
|
||||
op_sge = 0;
|
||||
op_size = 0;
|
||||
if (rc->cur_send >= rc->max_send)
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
uint32_t len = (uint32_t)(op_size+iov.iov_len-rc->send_buf_pos < rc->max_msg
|
||||
? iov.iov_len-rc->send_buf_pos : rc->max_msg-op_size);
|
||||
sge[op_sge++] = {
|
||||
.addr = (uintptr_t)((uint8_t*)iov.iov_base+rc->send_buf_pos),
|
||||
.length = len,
|
||||
.lkey = rc->ctx->mr->lkey,
|
||||
};
|
||||
op_size += len;
|
||||
rc->send_buf_pos += len;
|
||||
if (rc->send_buf_pos >= iov.iov_len)
|
||||
{
|
||||
rc->send_pos++;
|
||||
rc->send_buf_pos = 0;
|
||||
}
|
||||
}
|
||||
if (op_sge > 0)
|
||||
{
|
||||
rc->send_sizes.push_back(op_size);
|
||||
try_send_rdma_wr(cl, sge, op_sge);
|
||||
}
|
||||
}
|
||||
|
||||
void osd_messenger_t::try_send_rdma_nodp(osd_client_t *cl)
|
||||
{
|
||||
auto rc = cl->rdma_conn;
|
||||
if (!rc->send_out_size)
|
||||
@@ -590,11 +658,14 @@ void osd_messenger_t::try_send_rdma(osd_client_t *cl)
|
||||
// Allocate send ring buffer, if not yet
|
||||
rc->send_out_size = rc->max_msg*rdma_max_send;
|
||||
rc->send_out.buf = (uint8_t*)malloc_or_die(rc->send_out_size);
|
||||
rc->send_out.mr = ibv_reg_mr(rc->ctx->pd, rc->send_out.buf, rc->send_out_size, 0);
|
||||
if (!rc->send_out.mr)
|
||||
if (!rc->ctx->odp)
|
||||
{
|
||||
fprintf(stderr, "Failed to register RDMA memory region: %s\n", strerror(errno));
|
||||
exit(1);
|
||||
rc->send_out.mr = ibv_reg_mr(rc->ctx->pd, rc->send_out.buf, rc->send_out_size, 0);
|
||||
if (!rc->send_out.mr)
|
||||
{
|
||||
fprintf(stderr, "Failed to register RDMA memory region: %s\n", strerror(errno));
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
}
|
||||
// Copy data into the buffer and send it
|
||||
@@ -619,20 +690,28 @@ void osd_messenger_t::try_send_rdma(osd_client_t *cl)
|
||||
ibv_sge sge = {
|
||||
.addr = (uintptr_t)dst,
|
||||
.length = (uint32_t)copied,
|
||||
.lkey = rc->send_out.mr->lkey,
|
||||
.lkey = rc->ctx->odp ? rc->ctx->mr->lkey : rc->send_out.mr->lkey,
|
||||
};
|
||||
try_send_rdma_wr(cl, &sge, 1);
|
||||
cl->send_free_ops.push_back(NULL); // end marker
|
||||
rc->send_sizes.push_back(copied);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void osd_messenger_t::try_send_rdma(osd_client_t *cl)
|
||||
{
|
||||
if (cl->rdma_conn->ctx->odp)
|
||||
try_send_rdma_odp(cl);
|
||||
else
|
||||
try_send_rdma_nodp(cl);
|
||||
}
|
||||
|
||||
static void try_recv_rdma_wr(osd_client_t *cl, void *buf)
|
||||
{
|
||||
ibv_sge sge = {
|
||||
.addr = (uintptr_t)buf,
|
||||
.length = (uint32_t)cl->rdma_conn->max_msg,
|
||||
.lkey = cl->rdma_conn->recv_buf.mr->lkey,
|
||||
.lkey = cl->rdma_conn->ctx->odp ? cl->rdma_conn->ctx->mr->lkey : cl->rdma_conn->recv_buf.mr->lkey,
|
||||
};
|
||||
ibv_recv_wr *bad_wr = NULL;
|
||||
ibv_recv_wr wr = {
|
||||
@@ -654,11 +733,14 @@ bool osd_messenger_t::init_recv_rdma(osd_client_t *cl)
|
||||
auto rc = cl->rdma_conn;
|
||||
assert(!rc->recv_buf.buf);
|
||||
rc->recv_buf.buf = (uint8_t*)malloc_or_die(rc->max_msg * rc->max_recv);
|
||||
rc->recv_buf.mr = ibv_reg_mr(rc->ctx->pd, rc->recv_buf.buf, rc->max_msg * rc->max_recv, IBV_ACCESS_LOCAL_WRITE);
|
||||
if (!rc->recv_buf.mr)
|
||||
if (!rc->ctx->odp)
|
||||
{
|
||||
fprintf(stderr, "Failed to register RDMA memory region: %s\n", strerror(errno));
|
||||
exit(1);
|
||||
rc->recv_buf.mr = ibv_reg_mr(rc->ctx->pd, rc->recv_buf.buf, rc->max_msg * rc->max_recv, IBV_ACCESS_LOCAL_WRITE);
|
||||
if (!rc->recv_buf.mr)
|
||||
{
|
||||
fprintf(stderr, "Failed to register RDMA memory region: %s\n", strerror(errno));
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
for (uint32_t i = 0; i < rc->max_recv; i++)
|
||||
{
|
||||
@@ -732,18 +814,54 @@ void osd_messenger_t::handle_rdma_events(msgr_rdma_context_t *rdma_context)
|
||||
else
|
||||
{
|
||||
rc->cur_send--;
|
||||
uint64_t sent_size = wc[i].byte_len;
|
||||
rc->send_done_pos += sent_size;
|
||||
rc->send_out_full = false;
|
||||
if (rc->send_done_pos == rc->send_out_size)
|
||||
rc->send_done_pos = 0;
|
||||
assert(rc->send_done_pos < rc->send_out_size);
|
||||
while (cl->send_free_ops.front())
|
||||
uint64_t sent_size = rc->send_sizes.at(0);
|
||||
rc->send_sizes.erase(rc->send_sizes.begin(), rc->send_sizes.begin()+1);
|
||||
if (!rdma_context->odp)
|
||||
{
|
||||
delete cl->send_free_ops.front();
|
||||
cl->send_free_ops.pop_front();
|
||||
rc->send_done_pos += sent_size;
|
||||
rc->send_out_full = false;
|
||||
if (rc->send_done_pos == rc->send_out_size)
|
||||
rc->send_done_pos = 0;
|
||||
assert(rc->send_done_pos < rc->send_out_size);
|
||||
}
|
||||
int send_pos = 0, send_buf_pos = 0;
|
||||
while (sent_size > 0)
|
||||
{
|
||||
if (sent_size >= cl->send_list.at(send_pos).iov_len)
|
||||
{
|
||||
sent_size -= cl->send_list[send_pos].iov_len;
|
||||
send_pos++;
|
||||
}
|
||||
else
|
||||
{
|
||||
send_buf_pos = sent_size;
|
||||
sent_size = 0;
|
||||
}
|
||||
}
|
||||
assert(rc->send_pos >= send_pos);
|
||||
if (rc->send_pos == send_pos)
|
||||
{
|
||||
rc->send_buf_pos -= send_buf_pos;
|
||||
}
|
||||
rc->send_pos -= send_pos;
|
||||
for (int i = 0; i < send_pos; i++)
|
||||
{
|
||||
if (cl->outbox[i].flags & MSGR_SENDP_FREE)
|
||||
{
|
||||
// Reply fully sent
|
||||
delete cl->outbox[i].op;
|
||||
}
|
||||
}
|
||||
if (send_pos > 0)
|
||||
{
|
||||
cl->send_list.erase(cl->send_list.begin(), cl->send_list.begin()+send_pos);
|
||||
cl->outbox.erase(cl->outbox.begin(), cl->outbox.begin()+send_pos);
|
||||
}
|
||||
if (send_buf_pos > 0)
|
||||
{
|
||||
cl->send_list[0].iov_base = (uint8_t*)cl->send_list[0].iov_base + send_buf_pos;
|
||||
cl->send_list[0].iov_len -= send_buf_pos;
|
||||
}
|
||||
cl->send_free_ops.pop_front();
|
||||
try_send_rdma(cl);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,8 +10,6 @@
|
||||
#include <vector>
|
||||
#include "addr_util.h"
|
||||
|
||||
struct osd_op_t;
|
||||
|
||||
struct msgr_rdma_address_t
|
||||
{
|
||||
ibv_gid gid;
|
||||
@@ -28,6 +26,8 @@ struct msgr_rdma_context_t
|
||||
ibv_context *context = NULL;
|
||||
ibv_device_attr_ex attrx;
|
||||
ibv_pd *pd = NULL;
|
||||
bool odp = false;
|
||||
ibv_mr *mr = NULL;
|
||||
ibv_comp_channel *channel = NULL;
|
||||
ibv_cq *cq = NULL;
|
||||
ibv_port_attr portinfo;
|
||||
@@ -43,9 +43,9 @@ struct msgr_rdma_context_t
|
||||
int cm_refs = 0;
|
||||
|
||||
static std::vector<msgr_rdma_context_t*> create_all(const std::vector<addr_mask_t> & osd_network_masks,
|
||||
const char *sel_dev_name, int sel_port_num, int sel_gid_index, uint32_t sel_mtu, int log_level);
|
||||
const char *sel_dev_name, int sel_port_num, int sel_gid_index, uint32_t sel_mtu, bool odp, int log_level);
|
||||
static msgr_rdma_context_t *create(ibv_device *dev, ibv_port_attr & portinfo,
|
||||
int ib_port, int gid_index, uint32_t mtu, int log_level);
|
||||
int ib_port, int gid_index, uint32_t mtu, bool odp, int log_level);
|
||||
static msgr_rdma_context_t* create_cm(ibv_context *ctx);
|
||||
bool reserve_cqe(int n);
|
||||
|
||||
@@ -74,8 +74,9 @@ struct msgr_rdma_connection_t
|
||||
int cur_send = 0, cur_recv = 0;
|
||||
int send_pos = 0, send_buf_pos = 0;
|
||||
int next_recv_buf = 0;
|
||||
std::vector<uint8_t*> recv_buffers;
|
||||
std::vector<void*> recv_buffers;
|
||||
msgr_rdma_buf_t recv_buf;
|
||||
std::vector<uint64_t> send_sizes;
|
||||
msgr_rdma_buf_t send_out;
|
||||
int send_out_pos = 0, send_done_pos = 0, send_out_size = 0;
|
||||
bool send_out_full = false;
|
||||
|
||||
@@ -178,6 +178,7 @@ msgr_rdma_context_t* msgr_rdma_context_t::create_cm(ibv_context *ctx)
|
||||
delete rdma_context;
|
||||
return NULL;
|
||||
}
|
||||
rdma_context->odp = false;
|
||||
rdma_context->channel = ibv_create_comp_channel(rdma_context->context);
|
||||
if (!rdma_context->channel)
|
||||
{
|
||||
@@ -515,7 +516,7 @@ void osd_messenger_t::rdmacm_established(rdma_cm_event *ev)
|
||||
cl->peer_state = PEER_RDMA;
|
||||
cl->connect_timeout_id = -1;
|
||||
cl->osd_num = peer_osd;
|
||||
cl->in_buf = (uint8_t*)malloc_or_die(receive_buffer_size);
|
||||
cl->in_buf = malloc_or_die(receive_buffer_size);
|
||||
cl->rdma_conn = rc;
|
||||
clients[conn->peer_fd] = cl;
|
||||
if (conn->timeout_id >= 0)
|
||||
|
||||
+211
-372
@@ -1,8 +1,6 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||
|
||||
#define _XOPEN_SOURCE
|
||||
#include <limits.h>
|
||||
#include "messenger.h"
|
||||
|
||||
void osd_messenger_t::read_requests()
|
||||
@@ -16,11 +14,7 @@ void osd_messenger_t::read_requests()
|
||||
continue;
|
||||
}
|
||||
auto cl = cl_it->second;
|
||||
if (cl->read_op && cl->read_op_size-(cl->read_op_pos-OSD_PACKET_SIZE) >= receive_buffer_size)
|
||||
{
|
||||
op_get_read_buffers(cl, cl->recv_list);
|
||||
}
|
||||
if (!cl->recv_list.size())
|
||||
if (cl->read_remaining < receive_buffer_size)
|
||||
{
|
||||
cl->read_iov.iov_base = cl->in_buf;
|
||||
cl->read_iov.iov_len = receive_buffer_size;
|
||||
@@ -30,11 +24,10 @@ void osd_messenger_t::read_requests()
|
||||
else
|
||||
{
|
||||
cl->read_iov.iov_base = 0;
|
||||
cl->read_iov.iov_len = 0;
|
||||
cl->read_msg.msg_iov = cl->recv_list.data();
|
||||
cl->read_msg.msg_iovlen = cl->recv_list.size();
|
||||
cl->read_iov.iov_len = cl->read_remaining;
|
||||
cl->read_msg.msg_iov = cl->recv_list.get_iovec();
|
||||
cl->read_msg.msg_iovlen = cl->recv_list.get_size();
|
||||
}
|
||||
assert(!cl->read_op || cl->read_op_pos < OSD_PACKET_SIZE || cl->read_op_size >= (cl->read_op_pos-OSD_PACKET_SIZE));
|
||||
cl->refs++;
|
||||
if (ringloop && !use_sync_send_recv)
|
||||
{
|
||||
@@ -56,7 +49,7 @@ void osd_messenger_t::read_requests()
|
||||
}
|
||||
ring_data_t* data = ((ring_data_t*)sqe->user_data);
|
||||
data->callback = [this, cl](ring_data_t *data) { handle_read(data->res, cl); };
|
||||
io_uring_prep_recvmsg(sqe, peer_fd, &cl->read_msg, cl->recv_list.size() ? MSG_WAITALL : 0);
|
||||
io_uring_prep_recvmsg(sqe, peer_fd, &cl->read_msg, 0);
|
||||
if (iothread)
|
||||
{
|
||||
iothread->add_sqe(sqe_local);
|
||||
@@ -76,9 +69,11 @@ void osd_messenger_t::read_requests()
|
||||
read_ready_clients.clear();
|
||||
}
|
||||
|
||||
void osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
||||
bool osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
||||
{
|
||||
bool ret = false;
|
||||
int peer_fd = cl->peer_fd;
|
||||
cl->read_msg.msg_iovlen = 0;
|
||||
cl->refs--;
|
||||
if (cl->peer_state == PEER_STOPPED)
|
||||
{
|
||||
@@ -86,7 +81,7 @@ void osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
||||
{
|
||||
delete cl;
|
||||
}
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
if (result <= 0 && result != -EAGAIN && result != -EINTR)
|
||||
{
|
||||
@@ -96,51 +91,9 @@ void osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
||||
fprintf(stderr, "Client %d socket read error: %d (%s). Disconnecting client\n", cl->peer_fd, -result, strerror(-result));
|
||||
}
|
||||
stop_client(cl->peer_fd);
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
bool full_read = false;
|
||||
if (result > 0)
|
||||
{
|
||||
if (cl->read_iov.iov_base == cl->in_buf)
|
||||
{
|
||||
full_read = result >= cl->read_iov.iov_len;
|
||||
if (!handle_read_buffer(cl, cl->in_buf, result))
|
||||
{
|
||||
clear_immediate_ops(peer_fd);
|
||||
handle_immediate_ops();
|
||||
return;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// Reset OSD ping state
|
||||
cl->ping_time_remaining = 0;
|
||||
cl->idle_time_remaining = osd_idle_timeout;
|
||||
// Long data
|
||||
size_t i = 0;
|
||||
while (i < cl->recv_list.size() && result >= cl->recv_list[i].iov_len)
|
||||
{
|
||||
result -= cl->recv_list[i].iov_len;
|
||||
i++;
|
||||
}
|
||||
if (i < cl->recv_list.size())
|
||||
{
|
||||
cl->recv_list[i].iov_base += result;
|
||||
cl->recv_list[i].iov_len -= result;
|
||||
}
|
||||
else
|
||||
{
|
||||
full_read = true;
|
||||
}
|
||||
cl->recv_list.erase(cl->recv_list.begin(), cl->recv_list.begin()+i);
|
||||
if (!cl->recv_list.size())
|
||||
{
|
||||
handle_finished_op(cl);
|
||||
}
|
||||
}
|
||||
}
|
||||
cl->read_msg.msg_iovlen = 0;
|
||||
if (result == -EAGAIN || result == -EINTR || !full_read)
|
||||
if (result == -EAGAIN || result == -EINTR || result < cl->read_iov.iov_len)
|
||||
{
|
||||
cl->read_ready--;
|
||||
if (cl->read_ready > 0)
|
||||
@@ -150,7 +103,39 @@ void osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
||||
{
|
||||
read_ready_clients.push_back(cl->peer_fd);
|
||||
}
|
||||
if (result > 0)
|
||||
{
|
||||
if (cl->read_iov.iov_base == cl->in_buf)
|
||||
{
|
||||
if (!handle_read_buffer(cl, cl->in_buf, result))
|
||||
{
|
||||
clear_immediate_ops(peer_fd);
|
||||
handle_immediate_ops();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// Long data
|
||||
cl->read_remaining -= result;
|
||||
cl->recv_list.eat(result);
|
||||
if (cl->recv_list.done >= cl->recv_list.count)
|
||||
{
|
||||
if (!handle_finished_read(cl))
|
||||
{
|
||||
clear_immediate_ops(peer_fd);
|
||||
handle_immediate_ops();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (result >= cl->read_iov.iov_len)
|
||||
{
|
||||
ret = true;
|
||||
}
|
||||
}
|
||||
handle_immediate_ops();
|
||||
return ret;
|
||||
}
|
||||
|
||||
void osd_messenger_t::clear_immediate_ops(int peer_fd)
|
||||
@@ -190,95 +175,113 @@ void osd_messenger_t::handle_immediate_ops()
|
||||
set_immediate_ops.clear();
|
||||
}
|
||||
|
||||
bool osd_messenger_t::handle_read_buffer(osd_client_t *cl, uint8_t *curbuf, size_t bufsize)
|
||||
bool osd_messenger_t::handle_read_buffer(osd_client_t *cl, void *curbuf, int remain)
|
||||
{
|
||||
// Reset OSD ping state
|
||||
cl->ping_time_remaining = 0;
|
||||
cl->idle_time_remaining = osd_idle_timeout;
|
||||
// Compose operation(s) from the buffer
|
||||
size_t done = 0;
|
||||
while (done < bufsize)
|
||||
while (remain > 0)
|
||||
{
|
||||
if (!cl->read_op)
|
||||
{
|
||||
cl->read_op = new osd_op_t;
|
||||
cl->read_op->peer_fd = cl->peer_fd;
|
||||
cl->read_op->op_type = OSD_OP_IN;
|
||||
cl->read_op_pos = 0;
|
||||
cl->read_op_size = 0;
|
||||
cl->read_op_inline_decrypt_pos = (size_t)-1;
|
||||
cl->recv_list.push_back(cl->read_op->req.buf, OSD_PACKET_SIZE);
|
||||
cl->read_remaining = OSD_PACKET_SIZE;
|
||||
cl->read_state = CL_READ_HDR;
|
||||
}
|
||||
if (cl->read_op_pos < OSD_PACKET_SIZE)
|
||||
while (cl->recv_list.done < cl->recv_list.count && remain > 0)
|
||||
{
|
||||
int len = OSD_PACKET_SIZE - cl->read_op_pos;
|
||||
if (len > bufsize-done)
|
||||
len = bufsize-done;
|
||||
memcpy(cl->read_op->req.buf + cl->read_op_pos, curbuf+done, len);
|
||||
done += len;
|
||||
cl->read_op_pos += len;
|
||||
if (cl->read_op_pos < OSD_PACKET_SIZE)
|
||||
return true;
|
||||
if (!handle_hdr(cl))
|
||||
iovec* cur = cl->recv_list.get_iovec();
|
||||
if (cur->iov_len > remain)
|
||||
{
|
||||
memcpy(cur->iov_base, curbuf, remain);
|
||||
cl->read_remaining -= remain;
|
||||
cur->iov_len -= remain;
|
||||
cur->iov_base = (uint8_t*)cur->iov_base + remain;
|
||||
remain = 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
memcpy(cur->iov_base, curbuf, cur->iov_len);
|
||||
curbuf = (uint8_t*)curbuf + cur->iov_len;
|
||||
cl->read_remaining -= cur->iov_len;
|
||||
remain -= cur->iov_len;
|
||||
cur->iov_len = 0;
|
||||
cl->recv_list.done++;
|
||||
}
|
||||
}
|
||||
if (cl->recv_list.done >= cl->recv_list.count)
|
||||
{
|
||||
if (!handle_finished_read(cl))
|
||||
{
|
||||
stop_client(cl->peer_fd);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
op_copy_from(cl, curbuf, bufsize, done);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool osd_messenger_t::handle_hdr(osd_client_t *cl)
|
||||
bool osd_messenger_t::handle_finished_read(osd_client_t *cl)
|
||||
{
|
||||
if (cl->read_op->req.hdr.magic == SECONDARY_OSD_REPLY_MAGIC)
|
||||
// Reset OSD ping state
|
||||
cl->ping_time_remaining = 0;
|
||||
cl->idle_time_remaining = osd_idle_timeout;
|
||||
cl->recv_list.reset();
|
||||
if (cl->read_state == CL_READ_HDR)
|
||||
{
|
||||
auto req_it = cl->sent_ops.find(cl->read_op->req.hdr.id);
|
||||
if (req_it == cl->sent_ops.end())
|
||||
if (cl->read_op->req.hdr.magic == SECONDARY_OSD_REPLY_MAGIC)
|
||||
return handle_reply_hdr(cl);
|
||||
else if (cl->read_op->req.hdr.magic == SECONDARY_OSD_OP_MAGIC)
|
||||
{
|
||||
// Command out of sync. Drop connection
|
||||
fprintf(stderr, "Client %d command out of sync: id %ju\n", cl->peer_fd, cl->read_op->req.hdr.id);
|
||||
return false;
|
||||
}
|
||||
osd_op_t *op = req_it->second;
|
||||
memcpy(op->reply.buf, cl->read_op->req.buf, OSD_PACKET_SIZE);
|
||||
if (!allocate_reply_buffers(cl, op))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
cl->sent_ops.erase(req_it);
|
||||
delete cl->read_op;
|
||||
cl->read_op = op;
|
||||
}
|
||||
else if (cl->read_op->req.hdr.magic == SECONDARY_OSD_OP_MAGIC)
|
||||
{
|
||||
if (cl->check_sequencing)
|
||||
{
|
||||
if (cl->read_op->req.hdr.id != cl->read_op_id)
|
||||
if (cl->check_sequencing)
|
||||
{
|
||||
fprintf(stderr, "Warning: operation sequencing is broken on client %d: expected num %ju, got %ju, stopping client\n", cl->peer_fd, cl->read_op_id, cl->read_op->req.hdr.id);
|
||||
return false;
|
||||
if (cl->read_op->req.hdr.id != cl->read_op_id)
|
||||
{
|
||||
fprintf(stderr, "Warning: operation sequencing is broken on client %d: expected num %ju, got %ju, stopping client\n", cl->peer_fd, cl->read_op_id, cl->read_op->req.hdr.id);
|
||||
stop_client(cl->peer_fd);
|
||||
return false;
|
||||
}
|
||||
cl->read_op_id++;
|
||||
}
|
||||
cl->read_op_id++;
|
||||
handle_op_hdr(cl);
|
||||
}
|
||||
if (!allocate_op_buffers(cl))
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "Received garbage: magic=%jx id=%ju opcode=%jx from %d\n", cl->read_op->req.hdr.magic, cl->read_op->req.hdr.id, cl->read_op->req.hdr.opcode, cl->peer_fd);
|
||||
stop_client(cl->peer_fd);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if (cl->read_state == CL_READ_DATA)
|
||||
{
|
||||
// Operation is ready
|
||||
cl->received_ops.push_back(cl->read_op);
|
||||
set_immediate_ops.push_back(cl->read_op);
|
||||
cl->read_op = NULL;
|
||||
cl->read_state = 0;
|
||||
}
|
||||
else if (cl->read_state == CL_READ_REPLY_DATA)
|
||||
{
|
||||
// Reply is ready
|
||||
handle_reply_ready(cl->read_op);
|
||||
cl->read_op = NULL;
|
||||
cl->read_state = 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "Received garbage: magic=%jx id=%ju opcode=%jx from %d\n", cl->read_op->req.hdr.magic, cl->read_op->req.hdr.id, cl->read_op->req.hdr.opcode, cl->peer_fd);
|
||||
return false;
|
||||
assert(0);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool osd_messenger_t::allocate_op_buffers(osd_client_t *cl)
|
||||
void osd_messenger_t::handle_op_hdr(osd_client_t *cl)
|
||||
{
|
||||
osd_op_t *cur_op = cl->read_op;
|
||||
cl->read_op_size = 0;
|
||||
if (cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
|
||||
if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ)
|
||||
{
|
||||
cl->read_remaining = 0;
|
||||
}
|
||||
else if (cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE)
|
||||
{
|
||||
if (cur_op->req.sec_rw.attr_len > 0)
|
||||
@@ -287,12 +290,14 @@ bool osd_messenger_t::allocate_op_buffers(osd_client_t *cl)
|
||||
cur_op->bitmap = cur_op->rmw_buf = malloc_or_die(cur_op->req.sec_rw.attr_len);
|
||||
else
|
||||
cur_op->bitmap = &cur_op->bmp_data;
|
||||
cl->recv_list.push_back(cur_op->bitmap, cur_op->req.sec_rw.attr_len);
|
||||
}
|
||||
if (cur_op->req.sec_rw.len > 0)
|
||||
{
|
||||
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.sec_rw.len);
|
||||
cl->recv_list.push_back(cur_op->buf, cur_op->req.sec_rw.len);
|
||||
}
|
||||
cl->read_op_size = cur_op->req.sec_rw.len + cur_op->req.sec_rw.attr_len;
|
||||
cl->read_remaining = cur_op->req.sec_rw.len + cur_op->req.sec_rw.attr_len;
|
||||
}
|
||||
else if (cur_op->req.hdr.opcode == OSD_OP_SEC_STABILIZE ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SEC_ROLLBACK)
|
||||
@@ -300,24 +305,27 @@ bool osd_messenger_t::allocate_op_buffers(osd_client_t *cl)
|
||||
if (cur_op->req.sec_stab.len > 0)
|
||||
{
|
||||
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.sec_stab.len);
|
||||
cl->recv_list.push_back(cur_op->buf, cur_op->req.sec_stab.len);
|
||||
}
|
||||
cl->read_op_size = cur_op->req.sec_stab.len;
|
||||
cl->read_remaining = cur_op->req.sec_stab.len;
|
||||
}
|
||||
else if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ_BMP)
|
||||
{
|
||||
if (cur_op->req.sec_read_bmp.len > 0)
|
||||
{
|
||||
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.sec_read_bmp.len);
|
||||
cl->recv_list.push_back(cur_op->buf, cur_op->req.sec_read_bmp.len);
|
||||
}
|
||||
cl->read_op_size = cur_op->req.sec_read_bmp.len;
|
||||
cl->read_remaining = cur_op->req.sec_read_bmp.len;
|
||||
}
|
||||
else if (cur_op->req.hdr.opcode == OSD_OP_WRITE)
|
||||
{
|
||||
if (cur_op->req.rw.len > 0)
|
||||
{
|
||||
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.rw.len);
|
||||
cl->recv_list.push_back(cur_op->buf, cur_op->req.rw.len);
|
||||
}
|
||||
cl->read_op_size = cur_op->req.rw.len;
|
||||
cl->read_remaining = cur_op->req.rw.len;
|
||||
}
|
||||
else if (cur_op->req.hdr.opcode == OSD_OP_SHOW_CONFIG)
|
||||
{
|
||||
@@ -325,15 +333,44 @@ bool osd_messenger_t::allocate_op_buffers(osd_client_t *cl)
|
||||
{
|
||||
cur_op->buf = malloc_or_die(cur_op->req.show_conf.json_len+1);
|
||||
((uint8_t*)cur_op->buf)[cur_op->req.show_conf.json_len] = 0;
|
||||
cl->recv_list.push_back(cur_op->buf, cur_op->req.show_conf.json_len);
|
||||
}
|
||||
cl->read_op_size = cur_op->req.show_conf.json_len;
|
||||
cl->read_remaining = cur_op->req.show_conf.json_len;
|
||||
}
|
||||
/*else if (cur_op->req.hdr.opcode == OSD_OP_READ ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SCRUB ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_DESCRIBE)
|
||||
{
|
||||
cl->read_remaining = 0;
|
||||
}*/
|
||||
if (cl->read_remaining > 0)
|
||||
{
|
||||
// Read data
|
||||
cl->read_state = CL_READ_DATA;
|
||||
}
|
||||
else
|
||||
{
|
||||
// Operation is ready
|
||||
cl->received_ops.push_back(cur_op);
|
||||
set_immediate_ops.push_back(cur_op);
|
||||
cl->read_op = NULL;
|
||||
cl->read_state = 0;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool osd_messenger_t::allocate_reply_buffers(osd_client_t *cl, osd_op_t *op)
|
||||
bool osd_messenger_t::handle_reply_hdr(osd_client_t *cl)
|
||||
{
|
||||
cl->read_op_size = 0;
|
||||
auto req_it = cl->sent_ops.find(cl->read_op->req.hdr.id);
|
||||
if (req_it == cl->sent_ops.end())
|
||||
{
|
||||
// Command out of sync. Drop connection
|
||||
fprintf(stderr, "Client %d command out of sync: id %ju\n", cl->peer_fd, cl->read_op->req.hdr.id);
|
||||
stop_client(cl->peer_fd);
|
||||
return false;
|
||||
}
|
||||
osd_op_t *op = req_it->second;
|
||||
memcpy(op->reply.buf, cl->read_op->req.buf, OSD_PACKET_SIZE);
|
||||
cl->sent_ops.erase(req_it);
|
||||
if (op->reply.hdr.opcode == OSD_OP_SEC_READ || op->reply.hdr.opcode == OSD_OP_READ)
|
||||
{
|
||||
// Read data. In this case we assume that the buffer is preallocated by the caller (!)
|
||||
@@ -344,295 +381,97 @@ bool osd_messenger_t::allocate_reply_buffers(osd_client_t *cl, osd_op_t *op)
|
||||
// Check reply length to not overflow the buffer
|
||||
fprintf(stderr, "Client %d read reply of different length: expected %u+%u, got %jd+%u\n",
|
||||
cl->peer_fd, expected_size, op->bitmap_len, op->reply.hdr.retval, bmp_len);
|
||||
cl->sent_ops[op->req.hdr.id] = op;
|
||||
stop_client(cl->peer_fd);
|
||||
return false;
|
||||
}
|
||||
if (bmp_len > 0)
|
||||
{
|
||||
assert(op->bitmap);
|
||||
cl->read_op_size += bmp_len;
|
||||
cl->recv_list.push_back(op->bitmap, bmp_len);
|
||||
cl->read_remaining += bmp_len;
|
||||
}
|
||||
if (op->reply.hdr.retval > 0)
|
||||
{
|
||||
assert(op->iov.count > 0);
|
||||
cl->read_op_size += op->reply.hdr.retval;
|
||||
cl->recv_list.append(op->iov);
|
||||
cl->read_remaining += op->reply.hdr.retval;
|
||||
}
|
||||
if (cl->read_remaining == 0)
|
||||
{
|
||||
goto reuse;
|
||||
}
|
||||
delete cl->read_op;
|
||||
cl->read_op = op;
|
||||
cl->read_state = CL_READ_REPLY_DATA;
|
||||
}
|
||||
else if (op->reply.hdr.opcode == OSD_OP_SEC_LIST && op->reply.hdr.retval > 0)
|
||||
{
|
||||
assert(!op->iov.count);
|
||||
cl->read_op_size = sizeof(obj_ver_id) * op->reply.hdr.retval;
|
||||
op->buf = memalign_or_die(MEM_ALIGNMENT, cl->read_op_size);
|
||||
delete cl->read_op;
|
||||
cl->read_op = op;
|
||||
cl->read_state = CL_READ_REPLY_DATA;
|
||||
cl->read_remaining = sizeof(obj_ver_id) * op->reply.hdr.retval;
|
||||
op->buf = memalign_or_die(MEM_ALIGNMENT, cl->read_remaining);
|
||||
cl->recv_list.push_back(op->buf, cl->read_remaining);
|
||||
}
|
||||
else if (op->reply.hdr.opcode == OSD_OP_SEC_READ_BMP && op->reply.hdr.retval > 0)
|
||||
{
|
||||
assert(!op->iov.count);
|
||||
cl->read_op_size = op->reply.hdr.retval;
|
||||
delete cl->read_op;
|
||||
cl->read_op = op;
|
||||
cl->read_state = CL_READ_REPLY_DATA;
|
||||
cl->read_remaining = op->reply.hdr.retval;
|
||||
free(op->buf);
|
||||
op->buf = memalign_or_die(MEM_ALIGNMENT, cl->read_op_size);
|
||||
op->buf = memalign_or_die(MEM_ALIGNMENT, cl->read_remaining);
|
||||
cl->recv_list.push_back(op->buf, cl->read_remaining);
|
||||
}
|
||||
else if (op->reply.hdr.opcode == OSD_OP_SHOW_CONFIG && op->reply.hdr.retval > 0)
|
||||
{
|
||||
cl->read_op_size = op->reply.hdr.retval;
|
||||
delete cl->read_op;
|
||||
cl->read_op = op;
|
||||
cl->read_state = CL_READ_REPLY_DATA;
|
||||
cl->read_remaining = op->reply.hdr.retval;
|
||||
free(op->buf);
|
||||
op->buf = malloc_or_die(op->reply.hdr.retval);
|
||||
cl->recv_list.push_back(op->buf, op->reply.hdr.retval);
|
||||
}
|
||||
else if (op->reply.hdr.opcode == OSD_OP_DESCRIBE && op->reply.describe.result_bytes > 0)
|
||||
{
|
||||
cl->read_op_size = op->reply.describe.result_bytes;
|
||||
delete cl->read_op;
|
||||
cl->read_op = op;
|
||||
cl->read_state = CL_READ_REPLY_DATA;
|
||||
cl->read_remaining = op->reply.describe.result_bytes;
|
||||
free(op->buf);
|
||||
op->buf = malloc_or_die(op->reply.describe.result_bytes);
|
||||
cl->recv_list.push_back(op->buf, op->reply.describe.result_bytes);
|
||||
}
|
||||
else
|
||||
{
|
||||
reuse:
|
||||
// It's fine to reuse cl->read_op for the next reply
|
||||
handle_reply_ready(op);
|
||||
cl->recv_list.push_back(cl->read_op->req.buf, OSD_PACKET_SIZE);
|
||||
cl->read_remaining = OSD_PACKET_SIZE;
|
||||
cl->read_state = CL_READ_HDR;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
size_t osd_messenger_t::op_copy_from(osd_client_t *cl, uint8_t *src, size_t src_len, size_t & done)
|
||||
void osd_messenger_t::handle_reply_ready(osd_op_t *op)
|
||||
{
|
||||
osd_op_t *op = cl->read_op;
|
||||
size_t from = cl->read_op_pos-OSD_PACKET_SIZE;
|
||||
auto op_read_buf = [&](uint8_t *dst, size_t dst_len)
|
||||
// Measure subop latency
|
||||
timespec tv_end;
|
||||
clock_gettime(CLOCK_REALTIME, &tv_end);
|
||||
stats.subop_stat_count[op->req.hdr.opcode]++;
|
||||
if (!stats.subop_stat_count[op->req.hdr.opcode])
|
||||
{
|
||||
if (from < dst_len)
|
||||
{
|
||||
size_t n = dst_len-from;
|
||||
if (n > src_len-done)
|
||||
n = src_len-done;
|
||||
memcpy(dst+from, src+done, n);
|
||||
done += n;
|
||||
cl->read_op_pos += n;
|
||||
from += n;
|
||||
if (from < dst_len)
|
||||
return false;
|
||||
from = 0;
|
||||
}
|
||||
else
|
||||
from -= dst_len;
|
||||
return true;
|
||||
};
|
||||
if (op->op_type == OSD_OP_IN)
|
||||
{
|
||||
if (op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
|
||||
op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->bitmap, op->req.sec_rw.attr_len))
|
||||
return done;
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->req.sec_rw.len))
|
||||
return done;
|
||||
}
|
||||
else if (op->req.hdr.opcode == OSD_OP_SEC_STABILIZE ||
|
||||
op->req.hdr.opcode == OSD_OP_SEC_ROLLBACK)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->req.sec_stab.len))
|
||||
return done;
|
||||
}
|
||||
else if (op->req.hdr.opcode == OSD_OP_SEC_READ_BMP)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->req.sec_read_bmp.len))
|
||||
return done;
|
||||
}
|
||||
else if (op->req.hdr.opcode == OSD_OP_WRITE)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->req.rw.len))
|
||||
return done;
|
||||
}
|
||||
else if (op->req.hdr.opcode == OSD_OP_SHOW_CONFIG)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->req.show_conf.json_len))
|
||||
return done;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if (op->reply.hdr.opcode == OSD_OP_SEC_READ)
|
||||
{
|
||||
if (op->reply.sec_rw.attr_len > 0)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->bitmap, op->reply.sec_rw.attr_len))
|
||||
return done;
|
||||
}
|
||||
if (op->reply.hdr.retval > 0)
|
||||
{
|
||||
for (int i = 0; i < op->iov.count; i++)
|
||||
if (!op_read_buf((uint8_t*)op->iov.buf[i].iov_base, op->iov.buf[i].iov_len))
|
||||
return done;
|
||||
}
|
||||
}
|
||||
else if (op->reply.hdr.opcode == OSD_OP_READ)
|
||||
{
|
||||
if (op->reply.rw.bitmap_len > 0)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->bitmap, op->reply.rw.bitmap_len))
|
||||
return done;
|
||||
}
|
||||
if (op->reply.hdr.retval > 0)
|
||||
{
|
||||
if (op->enc)
|
||||
{
|
||||
if (!op_decrypted_copy_data_from(cl, src, src_len, from, done))
|
||||
return done;
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int i = 0; i < op->iov.count; i++)
|
||||
if (!op_read_buf((uint8_t*)op->iov.buf[i].iov_base, op->iov.buf[i].iov_len))
|
||||
return done;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (op->reply.hdr.opcode == OSD_OP_SEC_LIST && op->reply.hdr.retval > 0)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, sizeof(obj_ver_id) * op->reply.hdr.retval))
|
||||
return done;
|
||||
}
|
||||
else if ((op->reply.hdr.opcode == OSD_OP_SEC_READ_BMP ||
|
||||
op->reply.hdr.opcode == OSD_OP_SHOW_CONFIG) && op->reply.hdr.retval > 0)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->reply.hdr.retval))
|
||||
return done;
|
||||
}
|
||||
else if (op->reply.hdr.opcode == OSD_OP_DESCRIBE && op->reply.describe.result_bytes > 0)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->reply.describe.result_bytes))
|
||||
return done;
|
||||
}
|
||||
}
|
||||
handle_finished_op(cl);
|
||||
return done;
|
||||
}
|
||||
|
||||
size_t osd_messenger_t::op_get_read_buffers(osd_client_t *cl, std::vector<iovec> & lst)
|
||||
{
|
||||
osd_op_t *op = cl->read_op;
|
||||
size_t from = cl->read_op_pos-OSD_PACKET_SIZE;
|
||||
size_t done = 0;
|
||||
auto op_read_buf = [&](uint8_t *dst, size_t dst_len)
|
||||
{
|
||||
if (lst.size() >= IOV_MAX)
|
||||
return false;
|
||||
if (from < dst_len)
|
||||
{
|
||||
lst.push_back((iovec){ .iov_base = dst+from, .iov_len = dst_len-from });
|
||||
cl->read_op_pos += dst_len-from;
|
||||
done += dst_len-from;
|
||||
from = 0;
|
||||
}
|
||||
else
|
||||
from -= dst_len;
|
||||
return true;
|
||||
};
|
||||
if (op->op_type == OSD_OP_IN)
|
||||
{
|
||||
if (op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
|
||||
op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->bitmap, op->req.sec_rw.attr_len))
|
||||
return done;
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->req.sec_rw.len))
|
||||
return done;
|
||||
}
|
||||
else if (op->req.hdr.opcode == OSD_OP_SEC_STABILIZE ||
|
||||
op->req.hdr.opcode == OSD_OP_SEC_ROLLBACK)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->req.sec_stab.len))
|
||||
return done;
|
||||
}
|
||||
else if (op->req.hdr.opcode == OSD_OP_SEC_READ_BMP)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->req.sec_read_bmp.len))
|
||||
return done;
|
||||
}
|
||||
else if (op->req.hdr.opcode == OSD_OP_WRITE)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->req.rw.len))
|
||||
return done;
|
||||
}
|
||||
else if (op->req.hdr.opcode == OSD_OP_SHOW_CONFIG)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->req.show_conf.json_len))
|
||||
return done;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if (op->reply.hdr.opcode == OSD_OP_SEC_READ)
|
||||
{
|
||||
if (op->reply.sec_rw.attr_len > 0)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->bitmap, op->reply.sec_rw.attr_len))
|
||||
return done;
|
||||
}
|
||||
if (op->reply.hdr.retval > 0)
|
||||
{
|
||||
for (int i = 0; i < op->iov.count; i++)
|
||||
if (!op_read_buf((uint8_t*)op->iov.buf[i].iov_base, op->iov.buf[i].iov_len))
|
||||
return done;
|
||||
}
|
||||
}
|
||||
else if (op->reply.hdr.opcode == OSD_OP_READ)
|
||||
{
|
||||
if (op->reply.rw.bitmap_len > 0)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->bitmap, op->reply.rw.bitmap_len))
|
||||
return done;
|
||||
}
|
||||
if (op->reply.hdr.retval > 0)
|
||||
{
|
||||
if (op->enc)
|
||||
cl->read_op_inline_decrypt_pos = cl->read_op_pos;
|
||||
for (int i = 0; i < op->iov.count; i++)
|
||||
if (!op_read_buf((uint8_t*)op->iov.buf[i].iov_base, op->iov.buf[i].iov_len))
|
||||
return done;
|
||||
}
|
||||
}
|
||||
else if (op->reply.hdr.opcode == OSD_OP_SEC_LIST && op->reply.hdr.retval > 0)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, sizeof(obj_ver_id) * op->reply.hdr.retval))
|
||||
return done;
|
||||
}
|
||||
else if ((op->reply.hdr.opcode == OSD_OP_SEC_READ_BMP ||
|
||||
op->reply.hdr.opcode == OSD_OP_SHOW_CONFIG) && op->reply.hdr.retval > 0)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->reply.hdr.retval))
|
||||
return done;
|
||||
}
|
||||
else if (op->reply.hdr.opcode == OSD_OP_DESCRIBE && op->reply.describe.result_bytes > 0)
|
||||
{
|
||||
if (!op_read_buf((uint8_t*)op->buf, op->reply.describe.result_bytes))
|
||||
return done;
|
||||
}
|
||||
}
|
||||
return done;
|
||||
}
|
||||
|
||||
void osd_messenger_t::handle_finished_op(osd_client_t *cl)
|
||||
{
|
||||
osd_op_t *op = cl->read_op;
|
||||
if (op->op_type == OSD_OP_IN)
|
||||
{
|
||||
// Operation is ready
|
||||
cl->received_ops.push_back(op);
|
||||
}
|
||||
else
|
||||
{
|
||||
// Inline decryption
|
||||
if (cl->read_op_inline_decrypt_pos != (size_t)-1)
|
||||
{
|
||||
op_decrypt_inline(cl);
|
||||
cl->read_op_inline_decrypt_pos = (size_t)-1;
|
||||
}
|
||||
// Measure subop (outbound op) latency
|
||||
timespec tv_end;
|
||||
clock_gettime(CLOCK_REALTIME, &tv_end);
|
||||
stats.subop_stat_count[op->req.hdr.opcode]++;
|
||||
if (!stats.subop_stat_count[op->req.hdr.opcode])
|
||||
{
|
||||
stats.subop_stat_count[op->req.hdr.opcode]++;
|
||||
stats.subop_stat_sum[op->req.hdr.opcode] = 0;
|
||||
}
|
||||
stats.subop_stat_sum[op->req.hdr.opcode] += (
|
||||
(tv_end.tv_sec - op->tv_begin.tv_sec)*1000000 +
|
||||
(tv_end.tv_nsec - op->tv_begin.tv_nsec)/1000
|
||||
);
|
||||
stats.subop_stat_sum[op->req.hdr.opcode] = 0;
|
||||
}
|
||||
stats.subop_stat_sum[op->req.hdr.opcode] += (
|
||||
(tv_end.tv_sec - op->tv_begin.tv_sec)*1000000 +
|
||||
(tv_end.tv_nsec - op->tv_begin.tv_nsec)/1000
|
||||
);
|
||||
set_immediate_ops.push_back(op);
|
||||
cl->read_op = NULL;
|
||||
}
|
||||
|
||||
+123
-199
@@ -15,7 +15,6 @@ void osd_messenger_t::outbox_push(osd_op_t *cur_op)
|
||||
{
|
||||
clock_gettime(CLOCK_REALTIME, &cur_op->tv_begin);
|
||||
cur_op->req.hdr.id = ++cl->send_op_id;
|
||||
cl->sent_ops[cur_op->req.hdr.id] = cur_op;
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -36,9 +35,77 @@ void osd_messenger_t::outbox_push(osd_op_t *cur_op)
|
||||
delete cur_op;
|
||||
return;
|
||||
}
|
||||
measure_exec(cur_op);
|
||||
}
|
||||
cl->write_ops.push_back(cur_op);
|
||||
auto & to_send_list = cl->write_msg.msg_iovlen ? cl->next_send_list : cl->send_list;
|
||||
auto & to_outbox = cl->write_msg.msg_iovlen ? cl->next_outbox : cl->outbox;
|
||||
if (cur_op->op_type == OSD_OP_IN)
|
||||
{
|
||||
measure_exec(cur_op);
|
||||
to_send_list.push_back((iovec){ .iov_base = cur_op->reply.buf, .iov_len = OSD_PACKET_SIZE });
|
||||
}
|
||||
else
|
||||
{
|
||||
to_send_list.push_back((iovec){ .iov_base = cur_op->req.buf, .iov_len = OSD_PACKET_SIZE });
|
||||
cl->sent_ops[cur_op->req.hdr.id] = cur_op;
|
||||
}
|
||||
to_outbox.push_back((msgr_sendp_t){ .op = cur_op, .flags = MSGR_SENDP_HDR });
|
||||
// Bitmap
|
||||
if (cur_op->op_type == OSD_OP_IN &&
|
||||
cur_op->req.hdr.opcode == OSD_OP_SEC_READ &&
|
||||
cur_op->reply.sec_rw.attr_len > 0)
|
||||
{
|
||||
to_send_list.push_back((iovec){
|
||||
.iov_base = cur_op->bitmap,
|
||||
.iov_len = cur_op->reply.sec_rw.attr_len,
|
||||
});
|
||||
to_outbox.push_back((msgr_sendp_t){ .op = cur_op, .flags = 0 });
|
||||
}
|
||||
else if (cur_op->op_type == OSD_OP_OUT &&
|
||||
(cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE || cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE) &&
|
||||
cur_op->req.sec_rw.attr_len > 0)
|
||||
{
|
||||
to_send_list.push_back((iovec){
|
||||
.iov_base = cur_op->bitmap,
|
||||
.iov_len = cur_op->req.sec_rw.attr_len,
|
||||
});
|
||||
to_outbox.push_back((msgr_sendp_t){ .op = cur_op, .flags = 0 });
|
||||
}
|
||||
// Operation data
|
||||
if ((cur_op->op_type == OSD_OP_IN
|
||||
? (cur_op->req.hdr.opcode == OSD_OP_READ ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SEC_READ ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SEC_LIST ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SHOW_CONFIG ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_DESCRIBE)
|
||||
: (cur_op->req.hdr.opcode == OSD_OP_WRITE ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SEC_STABILIZE ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SEC_ROLLBACK ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SHOW_CONFIG)) && cur_op->iov.count > 0)
|
||||
{
|
||||
for (int i = 0; i < cur_op->iov.count; i++)
|
||||
{
|
||||
if (cur_op->iov.buf[i].iov_len > 0)
|
||||
{
|
||||
assert(cur_op->iov.buf[i].iov_base);
|
||||
to_send_list.push_back(cur_op->iov.buf[i]);
|
||||
to_outbox.push_back((msgr_sendp_t){ .op = cur_op, .flags = 0 });
|
||||
}
|
||||
}
|
||||
}
|
||||
if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ_BMP)
|
||||
{
|
||||
if (cur_op->op_type == OSD_OP_IN && cur_op->reply.hdr.retval > 0)
|
||||
to_send_list.push_back((iovec){ .iov_base = cur_op->buf, .iov_len = (size_t)cur_op->reply.hdr.retval });
|
||||
else if (cur_op->op_type == OSD_OP_OUT && cur_op->req.sec_read_bmp.len > 0)
|
||||
to_send_list.push_back((iovec){ .iov_base = cur_op->buf, .iov_len = (size_t)cur_op->req.sec_read_bmp.len });
|
||||
to_outbox.push_back((msgr_sendp_t){ .op = cur_op, .flags = 0 });
|
||||
}
|
||||
if (cur_op->op_type == OSD_OP_IN)
|
||||
{
|
||||
to_outbox[to_outbox.size()-1].flags |= MSGR_SENDP_FREE;
|
||||
}
|
||||
#ifdef WITH_RDMA
|
||||
if (cl->peer_state == PEER_RDMA)
|
||||
{
|
||||
@@ -49,7 +116,7 @@ void osd_messenger_t::outbox_push(osd_op_t *cur_op)
|
||||
if (!ringloop)
|
||||
{
|
||||
// FIXME: It's worse because it doesn't allow batching
|
||||
while (cl->write_ops.size())
|
||||
while (cl->outbox.size())
|
||||
{
|
||||
try_send(cl);
|
||||
}
|
||||
@@ -117,25 +184,11 @@ void osd_messenger_t::measure_exec(osd_op_t *cur_op)
|
||||
bool osd_messenger_t::try_send(osd_client_t *cl)
|
||||
{
|
||||
int peer_fd = cl->peer_fd;
|
||||
if (!cl->write_op && !cl->write_ops.size() || cl->write_msg.msg_iovlen > 0)
|
||||
if (!cl->send_list.size() || cl->write_msg.msg_iovlen > 0)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
assert(cl->peer_state != PEER_RDMA);
|
||||
while ((cl->write_op || cl->write_ops.size()) && cl->send_list.size() < IOV_MAX)
|
||||
{
|
||||
if (!cl->write_op)
|
||||
{
|
||||
cl->write_op = cl->write_ops.front();
|
||||
cl->write_ops.pop_front();
|
||||
}
|
||||
osd_op_t *op = cl->write_op;
|
||||
op_get_write_buffers(cl, cl->send_list);
|
||||
if (!cl->write_op && op->op_type == OSD_OP_IN)
|
||||
{
|
||||
cl->send_free_ops.push_back(op);
|
||||
}
|
||||
}
|
||||
if (ringloop && !use_sync_send_recv)
|
||||
{
|
||||
auto iothread = iothreads.size() ? iothreads[peer_fd % iothreads.size()] : NULL;
|
||||
@@ -148,24 +201,20 @@ bool osd_messenger_t::try_send(osd_client_t *cl)
|
||||
data_local = {};
|
||||
}
|
||||
if (!sqe)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
cl->send_list_size = 0;
|
||||
for (auto & iov: cl->send_list)
|
||||
{
|
||||
cl->send_list_size += iov.iov_len;
|
||||
}
|
||||
cl->write_msg.msg_iov = cl->send_list.data();
|
||||
cl->write_msg.msg_iovlen = cl->send_list.size() < IOV_MAX ? cl->send_list.size() : IOV_MAX;
|
||||
cl->refs++;
|
||||
ring_data_t* data = ((ring_data_t*)sqe->user_data);
|
||||
data->callback = [this, cl](ring_data_t *data) { handle_send(data->res, data->prev, data->more, cl); };
|
||||
bool use_zc = has_sendmsg_zc && min_zerocopy_send_size >= 0;
|
||||
if (use_zc && min_zerocopy_send_size > 0 &&
|
||||
cl->send_list_size/cl->write_msg.msg_iovlen < min_zerocopy_send_size)
|
||||
if (use_zc && min_zerocopy_send_size > 0)
|
||||
{
|
||||
use_zc = false;
|
||||
size_t avg_size = 0;
|
||||
for (size_t i = 0; i < cl->write_msg.msg_iovlen; i++)
|
||||
avg_size += cl->write_msg.msg_iov[i].iov_len;
|
||||
if (avg_size/cl->write_msg.msg_iovlen < min_zerocopy_send_size)
|
||||
use_zc = false;
|
||||
}
|
||||
if (use_zc)
|
||||
{
|
||||
@@ -216,7 +265,6 @@ void osd_messenger_t::handle_send(int result, bool prev, bool more, osd_client_t
|
||||
if (!prev)
|
||||
{
|
||||
cl->write_msg.msg_iovlen = 0;
|
||||
cl->send_list.clear();
|
||||
}
|
||||
if (!more)
|
||||
{
|
||||
@@ -249,26 +297,57 @@ void osd_messenger_t::handle_send(int result, bool prev, bool more, osd_client_t
|
||||
cl->zc_free_list.erase(cl->zc_free_list.begin(), cl->zc_free_list.begin()+i+1);
|
||||
return;
|
||||
}
|
||||
if (cl->send_list_size > result)
|
||||
int done = 0;
|
||||
while (result > 0 && done < cl->send_list.size())
|
||||
{
|
||||
fprintf(stderr, "Client %d socket write error: expected to send "
|
||||
"%zu bytes with MSG_WAITALL but sent %u. Disconnecting client\n", cl->peer_fd, cl->send_list_size, result);
|
||||
stop_client(cl->peer_fd);
|
||||
return;
|
||||
}
|
||||
for (auto op: cl->send_free_ops)
|
||||
{
|
||||
if (more)
|
||||
cl->zc_free_list.push_back(op);
|
||||
iovec & iov = cl->send_list[done];
|
||||
if (iov.iov_len <= result)
|
||||
{
|
||||
if (cl->outbox[done].flags & MSGR_SENDP_FREE)
|
||||
{
|
||||
// Reply fully sent
|
||||
if (more)
|
||||
cl->zc_free_list.push_back(cl->outbox[done].op);
|
||||
else
|
||||
delete cl->outbox[done].op;
|
||||
}
|
||||
result -= iov.iov_len;
|
||||
done++;
|
||||
}
|
||||
else
|
||||
delete op;
|
||||
{
|
||||
iov.iov_len -= result;
|
||||
iov.iov_base = (uint8_t*)iov.iov_base + result;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (more)
|
||||
{
|
||||
int expected = cl->send_list.size() < IOV_MAX ? cl->send_list.size() : IOV_MAX;
|
||||
if (done != expected)
|
||||
{
|
||||
fprintf(stderr, "Client %d socket write error: expected to send "
|
||||
"%d iovecs with MSG_WAITALL but sent %d. Disconnecting client\n", cl->peer_fd, expected, done);
|
||||
stop_client(cl->peer_fd);
|
||||
return;
|
||||
}
|
||||
cl->zc_free_list.push_back(NULL); // end marker
|
||||
cl->send_free_ops.clear();
|
||||
cl->write_state = cl->write_op || cl->write_ops.size() ? CL_WRITE_READY : 0;
|
||||
}
|
||||
if (done > 0)
|
||||
{
|
||||
cl->send_list.erase(cl->send_list.begin(), cl->send_list.begin()+done);
|
||||
cl->outbox.erase(cl->outbox.begin(), cl->outbox.begin()+done);
|
||||
}
|
||||
if (cl->next_send_list.size())
|
||||
{
|
||||
cl->send_list.insert(cl->send_list.end(), cl->next_send_list.begin(), cl->next_send_list.end());
|
||||
cl->outbox.insert(cl->outbox.end(), cl->next_outbox.begin(), cl->next_outbox.end());
|
||||
cl->next_send_list.clear();
|
||||
cl->next_outbox.clear();
|
||||
}
|
||||
cl->write_state = cl->outbox.size() > 0 ? CL_WRITE_READY : 0;
|
||||
#ifdef WITH_RDMA
|
||||
if (cl->rdma_conn && !cl->write_op && !cl->write_ops.size() && cl->peer_state == PEER_RDMA_CONNECTING)
|
||||
if (cl->rdma_conn && !cl->outbox.size() && cl->peer_state == PEER_RDMA_CONNECTING)
|
||||
{
|
||||
// FIXME: Do something better than just forgetting the FD
|
||||
// FIXME: Ignore pings during RDMA state transition
|
||||
@@ -295,158 +374,3 @@ void osd_messenger_t::handle_send(int result, bool prev, bool more, osd_client_t
|
||||
write_ready_clients.push_back(cl->peer_fd);
|
||||
}
|
||||
}
|
||||
|
||||
static inline bool op_write_headers(osd_op_t *op, std::function<bool(uint8_t*, size_t)> op_write_buf)
|
||||
{
|
||||
// Header
|
||||
if (!op_write_buf((op->op_type == OSD_OP_IN ? op->reply.buf : op->req.buf), OSD_PACKET_SIZE))
|
||||
return false;
|
||||
// Bitmap
|
||||
if (op->op_type == OSD_OP_IN &&
|
||||
op->req.hdr.opcode == OSD_OP_SEC_READ &&
|
||||
op->reply.sec_rw.attr_len > 0)
|
||||
{
|
||||
if (!op_write_buf((uint8_t*)op->bitmap, op->reply.sec_rw.attr_len))
|
||||
return false;
|
||||
}
|
||||
else if (op->op_type == OSD_OP_OUT &&
|
||||
(op->req.hdr.opcode == OSD_OP_SEC_WRITE || op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE) &&
|
||||
op->req.sec_rw.attr_len > 0)
|
||||
{
|
||||
if (!op_write_buf((uint8_t*)op->bitmap, op->req.sec_rw.attr_len))
|
||||
return false;
|
||||
}
|
||||
if (op->req.hdr.opcode == OSD_OP_SEC_READ_BMP)
|
||||
{
|
||||
if (op->op_type == OSD_OP_IN && op->reply.hdr.retval > 0)
|
||||
{
|
||||
if (!op_write_buf((uint8_t*)op->buf, (size_t)op->reply.hdr.retval))
|
||||
return false;
|
||||
}
|
||||
else if (op->op_type == OSD_OP_OUT && op->req.sec_read_bmp.len > 0)
|
||||
{
|
||||
if (!op_write_buf((uint8_t*)op->buf, (size_t)op->req.sec_read_bmp.len))
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static inline bool op_has_data(osd_op_t *op)
|
||||
{
|
||||
return (op->op_type == OSD_OP_IN
|
||||
? (op->req.hdr.opcode == OSD_OP_READ ||
|
||||
op->req.hdr.opcode == OSD_OP_SEC_READ ||
|
||||
op->req.hdr.opcode == OSD_OP_SEC_LIST ||
|
||||
op->req.hdr.opcode == OSD_OP_SHOW_CONFIG ||
|
||||
op->req.hdr.opcode == OSD_OP_DESCRIBE)
|
||||
: (op->req.hdr.opcode == OSD_OP_WRITE ||
|
||||
op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
|
||||
op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE ||
|
||||
op->req.hdr.opcode == OSD_OP_SEC_STABILIZE ||
|
||||
op->req.hdr.opcode == OSD_OP_SEC_ROLLBACK ||
|
||||
op->req.hdr.opcode == OSD_OP_SHOW_CONFIG)) && op->iov.count > 0;
|
||||
}
|
||||
|
||||
size_t osd_messenger_t::op_copy_to(osd_client_t *cl, uint8_t *dst, size_t dst_len)
|
||||
{
|
||||
size_t done = 0;
|
||||
size_t from = cl->write_op_pos;
|
||||
auto op_write_buf = [&](uint8_t *src, size_t src_len)
|
||||
{
|
||||
if (from < src_len)
|
||||
{
|
||||
size_t n = src_len-from;
|
||||
if (n > dst_len-done)
|
||||
n = dst_len-done;
|
||||
memcpy(dst+done, src+from, n);
|
||||
done += n;
|
||||
cl->write_op_pos += n;
|
||||
from += n;
|
||||
if (from < src_len)
|
||||
return false;
|
||||
from = 0;
|
||||
}
|
||||
else
|
||||
from -= src_len;
|
||||
return true;
|
||||
};
|
||||
if (!op_write_headers(cl->write_op, op_write_buf))
|
||||
{
|
||||
return done;
|
||||
}
|
||||
// Operation data
|
||||
if (op_has_data(cl->write_op))
|
||||
{
|
||||
if (cl->write_op->enc)
|
||||
{
|
||||
if (!op_encrypted_copy_data_to(cl, dst, dst_len, from, done))
|
||||
{
|
||||
return done;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int i = 0; i < cl->write_op->iov.count; i++)
|
||||
{
|
||||
if (!op_write_buf((uint8_t*)cl->write_op->iov.buf[i].iov_base, cl->write_op->iov.buf[i].iov_len))
|
||||
return done;
|
||||
}
|
||||
}
|
||||
}
|
||||
cl->write_op = NULL;
|
||||
cl->write_op_pos = 0;
|
||||
return done;
|
||||
}
|
||||
|
||||
void osd_messenger_t::op_get_write_buffers(osd_client_t *cl, std::vector<iovec> & lst)
|
||||
{
|
||||
size_t from = cl->write_op_pos;
|
||||
auto op_write_buf = [&](uint8_t *src, size_t src_len)
|
||||
{
|
||||
if (lst.size() >= IOV_MAX)
|
||||
return false;
|
||||
if (from < src_len)
|
||||
{
|
||||
lst.push_back((iovec){ .iov_base = src+from, .iov_len = src_len-from });
|
||||
cl->write_op_pos += src_len-from;
|
||||
from = 0;
|
||||
}
|
||||
else
|
||||
from -= src_len;
|
||||
return true;
|
||||
};
|
||||
if (!op_write_headers(cl->write_op, op_write_buf))
|
||||
{
|
||||
return;
|
||||
}
|
||||
// Operation data
|
||||
if (op_has_data(cl->write_op))
|
||||
{
|
||||
if (cl->write_op->enc)
|
||||
{
|
||||
if (lst.size() >= IOV_MAX)
|
||||
return;
|
||||
// No way except to allocate a temporary buffer and encrypt data to it
|
||||
assert(cl->write_op->req.hdr.opcode == OSD_OP_WRITE);
|
||||
size_t remsize = cl->write_op->req.rw.len - from + (from % 16);
|
||||
assert(remsize > 0);
|
||||
assert(!cl->write_op->enc_buf);
|
||||
cl->write_op->enc_buf = (uint8_t*)malloc_or_die(remsize);
|
||||
size_t done = 0;
|
||||
bool end = op_encrypted_copy_data_to(cl, cl->write_op->enc_buf, remsize, from, done);
|
||||
assert(end);
|
||||
lst.push_back((iovec){ .iov_base = cl->write_op->enc_buf, .iov_len = remsize });
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int i = 0; i < cl->write_op->iov.count; i++)
|
||||
{
|
||||
if (!op_write_buf((uint8_t*)cl->write_op->iov.buf[i].iov_base, cl->write_op->iov.buf[i].iov_len))
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
cl->write_op = NULL;
|
||||
cl->write_op_pos = 0;
|
||||
}
|
||||
|
||||
@@ -72,22 +72,6 @@ void osd_messenger_t::stop_client(int peer_fd, bool force, bool force_delete)
|
||||
fprintf(stderr, "[OSD %ju] Stopping client %d (regular client)\n", osd_num, peer_fd);
|
||||
}
|
||||
}
|
||||
if (cl->encrypt_ctx)
|
||||
{
|
||||
if (encrypt_ctx_pool.size() > max_aes_xts_pool_size)
|
||||
destroy_aes_xts_encrypt(cl->encrypt_ctx);
|
||||
else
|
||||
encrypt_ctx_pool.push_back(cl->encrypt_ctx);
|
||||
cl->encrypt_ctx = NULL;
|
||||
}
|
||||
if (cl->decrypt_ctx)
|
||||
{
|
||||
if (decrypt_ctx_pool.size() > max_aes_xts_pool_size)
|
||||
destroy_aes_xts_decrypt(cl->decrypt_ctx);
|
||||
else
|
||||
decrypt_ctx_pool.push_back(cl->decrypt_ctx);
|
||||
cl->decrypt_ctx = NULL;
|
||||
}
|
||||
// First set state to STOPPED so another stop_client() call doesn't try to free it again
|
||||
cl->refs++;
|
||||
int prev_state = cl->peer_state;
|
||||
@@ -184,13 +168,6 @@ osd_client_t::~osd_client_t()
|
||||
}
|
||||
// Cancel outbound ops
|
||||
cancel_ops();
|
||||
for (osd_op_t *op: send_free_ops)
|
||||
{
|
||||
if (op)
|
||||
{
|
||||
delete op;
|
||||
}
|
||||
}
|
||||
for (osd_op_t *op: zc_free_list)
|
||||
{
|
||||
if (op)
|
||||
|
||||
@@ -6,7 +6,7 @@ includedir=${prefix}/@CMAKE_INSTALL_INCLUDEDIR@
|
||||
|
||||
Name: Vitastor
|
||||
Description: Vitastor client library
|
||||
Version: 3.0.3
|
||||
Version: 3.0.2
|
||||
Libs: -L${libdir} -lvitastor_client
|
||||
Cflags: -I${includedir}
|
||||
|
||||
|
||||
@@ -2,15 +2,6 @@ cmake_minimum_required(VERSION 2.8.12)
|
||||
|
||||
project(vitastor)
|
||||
|
||||
set(OPENAPI_JSON_H "${CMAKE_CURRENT_BINARY_DIR}/openapi.json.h")
|
||||
add_custom_command(
|
||||
OUTPUT ${OPENAPI_JSON_H}
|
||||
COMMAND ${CMAKE_COMMAND} -E echo const char* openapi_description = R\\\"json\\\( > ${OPENAPI_JSON_H}
|
||||
COMMAND ${CMAKE_COMMAND} -E cat ${CMAKE_CURRENT_SOURCE_DIR}/openapi.json >> ${OPENAPI_JSON_H}
|
||||
COMMAND ${CMAKE_COMMAND} -E echo "\\)json\\\"\\;" >> ${OPENAPI_JSON_H}
|
||||
DEPENDS openapi.json
|
||||
)
|
||||
|
||||
# libvitastor_cli.a
|
||||
add_library(vitastor_cli STATIC
|
||||
cli_common.cpp
|
||||
@@ -36,11 +27,8 @@ add_library(vitastor_cli STATIC
|
||||
cli_pool_ls.cpp
|
||||
cli_pool_modify.cpp
|
||||
cli_pool_rm.cpp
|
||||
cli_serve.cpp
|
||||
${OPENAPI_JSON_H}
|
||||
)
|
||||
target_compile_options(vitastor_cli PUBLIC -fPIC)
|
||||
target_include_directories(vitastor_cli PRIVATE ${CMAKE_CURRENT_BINARY_DIR})
|
||||
|
||||
# vitastor-cli
|
||||
add_executable(vitastor-cli
|
||||
|
||||
+34
-70
@@ -37,7 +37,6 @@ static const char* help_text =
|
||||
" --sort FIELD Sort by specified field (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)\n"
|
||||
" -r|--reverse Sort in descending order\n"
|
||||
" -n|--count N Only list first N items\n"
|
||||
" --ids ID1,ID2 Only list images with specified full IDs\n"
|
||||
" --tree Show image snapshot/clone tree\n"
|
||||
"\n"
|
||||
"vitastor-cli create -s|--size <size> [-p|--pool <id|name>] [--parent <parent_name>[@<snapshot>]] <name>\n"
|
||||
@@ -218,7 +217,7 @@ static const char* help_text =
|
||||
"vitastor-cli rm-pool|pool-rm [--force] <id|name>\n"
|
||||
" Remove a pool. Refuses to remove pools with images without --force.\n"
|
||||
"\n"
|
||||
"vitastor-cli ls-pools|pool-ls|ls-pool|pools [-l] [--detail] [--sort FIELD] [-r] [-n N] [<glob> ...]\n"
|
||||
"vitastor-cli ls-pools|pool-ls|ls-pool|pools [-l] [--detail] [--sort FIELD] [-r] [-n N] [--stats] [<glob> ...]\n"
|
||||
" List pools (only matching <glob> patterns if passed).\n"
|
||||
" -l|--long Also report I/O statistics\n"
|
||||
" --detail Use list format (not table), show all details\n"
|
||||
@@ -226,14 +225,6 @@ static const char* help_text =
|
||||
" -r|--reverse Sort in descending order\n"
|
||||
" -n|--count N Only list first N items\n"
|
||||
"\n"
|
||||
"vitastor-cli serve\n"
|
||||
" Start HTTP server able to handle CLI commands over a REST API. Options:\n"
|
||||
" --bind_address ADDR Specify server IP address or addresses, separated by space. Default is 127.0.0.1.\n"
|
||||
" --port 8080 Specify server port.\n"
|
||||
" --ssl_cert FILE Path to server SSL certificate file (PEM format).\n"
|
||||
" --ssl_key FILE Path to server SSL private key file.\n"
|
||||
" --ssl_ca FILE Path to file with SSL CA certificates used to validate client connections.\n"
|
||||
"\n"
|
||||
"Use vitastor-cli --help <command> for command details or vitastor-cli --help --all for all details.\n"
|
||||
"\n"
|
||||
"GLOBAL OPTIONS:\n"
|
||||
@@ -328,24 +319,27 @@ static json11::Json::object parse_args(int narg, const char *args[])
|
||||
return cfg;
|
||||
}
|
||||
|
||||
std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg, cli_result_t & result)
|
||||
static int run(cli_tool_t *p, json11::Json::object cfg)
|
||||
{
|
||||
cli_result_t result = {};
|
||||
p->is_command_line = true;
|
||||
p->parse_config(cfg);
|
||||
json11::Json::array cmd = cfg["command"].array_items();
|
||||
cfg.erase("command");
|
||||
std::function<bool(cli_result_t &)> action_cb;
|
||||
if (!cmd.size())
|
||||
{
|
||||
result = { .err = EOPNOTSUPP, .text = "command is missing" };
|
||||
result = { .err = EINVAL, .text = "command is missing" };
|
||||
}
|
||||
else if (cmd[0] == "status")
|
||||
{
|
||||
// Show cluster status
|
||||
action_cb = start_status(cfg);
|
||||
action_cb = p->start_status(cfg);
|
||||
}
|
||||
else if (cmd[0] == "df")
|
||||
{
|
||||
// Show pool space stats
|
||||
action_cb = start_pool_ls(cfg);
|
||||
action_cb = p->start_pool_ls(cfg);
|
||||
}
|
||||
else if (cmd[0] == "ls")
|
||||
{
|
||||
@@ -355,7 +349,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
cmd.erase(cmd.begin(), cmd.begin()+1);
|
||||
cfg["names"] = cmd;
|
||||
}
|
||||
action_cb = start_ls(cfg);
|
||||
action_cb = p->start_ls(cfg);
|
||||
}
|
||||
else if (cmd[0] == "snap-create")
|
||||
{
|
||||
@@ -370,7 +364,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
{
|
||||
cfg["image"] = name.substr(0, pos);
|
||||
cfg["snapshot"] = name.substr(pos + 1);
|
||||
action_cb = start_create(cfg);
|
||||
action_cb = p->start_create(cfg);
|
||||
}
|
||||
}
|
||||
else if (cmd[0] == "create")
|
||||
@@ -380,7 +374,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
{
|
||||
cfg["image"] = cmd[1];
|
||||
}
|
||||
action_cb = start_create(cfg);
|
||||
action_cb = p->start_create(cfg);
|
||||
}
|
||||
else if (cmd[0] == "modify")
|
||||
{
|
||||
@@ -389,12 +383,12 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
{
|
||||
cfg["image"] = cmd[1];
|
||||
}
|
||||
action_cb = start_modify(cfg);
|
||||
action_cb = p->start_modify(cfg);
|
||||
}
|
||||
else if (cmd[0] == "rm-data")
|
||||
{
|
||||
// Delete inode data
|
||||
action_cb = start_rm_data(cfg);
|
||||
action_cb = p->start_rm_data(cfg);
|
||||
}
|
||||
else if (cmd[0] == "rm-osd")
|
||||
{
|
||||
@@ -404,7 +398,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
cmd.erase(cmd.begin(), cmd.begin()+1);
|
||||
cfg["osd_id"] = cmd;
|
||||
}
|
||||
action_cb = start_rm_osd(cfg);
|
||||
action_cb = p->start_rm_osd(cfg);
|
||||
}
|
||||
else if (cmd[0] == "merge-data")
|
||||
{
|
||||
@@ -415,7 +409,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
if (cmd.size() > 2)
|
||||
cfg["to"] = cmd[2];
|
||||
}
|
||||
action_cb = start_merge(cfg);
|
||||
action_cb = p->start_merge(cfg);
|
||||
}
|
||||
else if (cmd[0] == "flatten")
|
||||
{
|
||||
@@ -424,7 +418,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
{
|
||||
cfg["image"] = cmd[1];
|
||||
}
|
||||
action_cb = start_flatten(cfg);
|
||||
action_cb = p->start_flatten(cfg);
|
||||
}
|
||||
else if (cmd[0] == "dd")
|
||||
{
|
||||
@@ -438,31 +432,16 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
cfg[arg.substr(0, p)] = arg.substr(p+1);
|
||||
}
|
||||
}
|
||||
action_cb = start_dd(cfg);
|
||||
action_cb = p->start_dd(cfg);
|
||||
}
|
||||
else if (cmd[0] == "rm")
|
||||
{
|
||||
// Remove multiple snapshots and rebase their children
|
||||
if (cfg["names"].is_array())
|
||||
{
|
||||
cfg["globs"] = cfg["names"];
|
||||
cfg.erase("names");
|
||||
cfg["exact"] = true;
|
||||
cfg["matching"] = false;
|
||||
action_cb = start_rm_wildcard(cfg);
|
||||
}
|
||||
else if (cfg["matching"].is_array())
|
||||
{
|
||||
cfg["globs"] = cfg["matching"];
|
||||
cfg["exact"] = false;
|
||||
cfg["matching"] = true;
|
||||
action_cb = start_rm_wildcard(cfg);
|
||||
}
|
||||
else if (cfg["exact"].bool_value() || cfg["matching"].bool_value())
|
||||
if (cfg["exact"].bool_value() || cfg["matching"].bool_value())
|
||||
{
|
||||
cmd.erase(cmd.begin(), cmd.begin()+1);
|
||||
cfg["globs"] = cmd;
|
||||
action_cb = start_rm_wildcard(cfg);
|
||||
action_cb = p->start_rm_wildcard(cfg);
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -472,41 +451,41 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
if (cmd.size() > 2)
|
||||
cfg["to"] = cmd[2];
|
||||
}
|
||||
action_cb = start_rm(cfg);
|
||||
action_cb = p->start_rm(cfg);
|
||||
}
|
||||
}
|
||||
else if (cmd[0] == "describe")
|
||||
{
|
||||
// Describe unclean objects
|
||||
action_cb = start_describe(cfg);
|
||||
action_cb = p->start_describe(cfg);
|
||||
}
|
||||
else if (cmd[0] == "fix")
|
||||
{
|
||||
// Fix inconsistent objects (by deleting some copies)
|
||||
action_cb = start_fix(cfg);
|
||||
action_cb = p->start_fix(cfg);
|
||||
}
|
||||
else if (cmd[0] == "alloc-osd")
|
||||
{
|
||||
// Allocate a new OSD number
|
||||
action_cb = start_alloc_osd(cfg);
|
||||
action_cb = p->start_alloc_osd(cfg);
|
||||
}
|
||||
else if (cmd[0] == "osd-tree")
|
||||
{
|
||||
// Print OSD tree
|
||||
cfg["as_tree"] = true;
|
||||
action_cb = start_osd_tree(cfg);
|
||||
action_cb = p->start_osd_tree(cfg);
|
||||
}
|
||||
else if (cmd[0] == "osds" || cmd[0] == "ls-osds" || cmd[0] == "ls-osd" || cmd[0] == "osd-ls")
|
||||
{
|
||||
// Print OSD list
|
||||
action_cb = start_osd_tree(cfg);
|
||||
cfg["flat"] = true;
|
||||
action_cb = p->start_osd_tree(cfg);
|
||||
}
|
||||
else if (cmd[0] == "modify-osd")
|
||||
{
|
||||
// Modify OSD configuration
|
||||
if (cmd.size() > 1)
|
||||
cfg["osd_num"] = cmd[1];
|
||||
action_cb = start_modify_osd(cfg);
|
||||
action_cb = p->start_modify_osd(cfg);
|
||||
}
|
||||
else if (cmd[0] == "pg-list" || cmd[0] == "pg-ls" || cmd[0] == "list-pg" || cmd[0] == "ls-pg" || cmd[0] == "ls-pgs" || cmd[0] == "pgs")
|
||||
{
|
||||
@@ -516,7 +495,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
cmd.erase(cmd.begin(), cmd.begin()+1);
|
||||
cfg["pg_state"] = cmd;
|
||||
}
|
||||
action_cb = start_pg_list(cfg);
|
||||
action_cb = p->start_pg_list(cfg);
|
||||
}
|
||||
else if (cmd[0] == "create-pool" || cmd[0] == "pool-create")
|
||||
{
|
||||
@@ -525,16 +504,16 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
{
|
||||
cfg["name"] = cmd[1];
|
||||
}
|
||||
action_cb = start_pool_create(cfg);
|
||||
action_cb = p->start_pool_create(cfg);
|
||||
}
|
||||
else if (cmd[0] == "modify-pool" || cmd[0] == "pool-modify")
|
||||
{
|
||||
// Modify existing pool
|
||||
if (cmd.size() > 1)
|
||||
{
|
||||
cfg["pool"] = cmd[1];
|
||||
cfg["old_name"] = cmd[1];
|
||||
}
|
||||
action_cb = start_pool_modify(cfg);
|
||||
action_cb = p->start_pool_modify(cfg);
|
||||
}
|
||||
else if (cmd[0] == "rm-pool" || cmd[0] == "pool-rm")
|
||||
{
|
||||
@@ -543,7 +522,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
{
|
||||
cfg["pool"] = cmd[1];
|
||||
}
|
||||
action_cb = start_pool_rm(cfg);
|
||||
action_cb = p->start_pool_rm(cfg);
|
||||
}
|
||||
else if (cmd[0] == "ls-pool" || cmd[0] == "pool-ls" || cmd[0] == "ls-pools" || cmd[0] == "pools")
|
||||
{
|
||||
@@ -554,26 +533,12 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
cmd.erase(cmd.begin(), cmd.begin()+1);
|
||||
cfg["names"] = cmd;
|
||||
}
|
||||
action_cb = start_pool_ls(cfg);
|
||||
}
|
||||
else if (cmd[0] == "serve")
|
||||
{
|
||||
// Start HTTP server
|
||||
action_cb = start_serve(cfg);
|
||||
action_cb = p->start_pool_ls(cfg);
|
||||
}
|
||||
else
|
||||
{
|
||||
result = { .err = EOPNOTSUPP, .text = "unknown command: "+cmd[0].string_value() };
|
||||
result = { .err = EINVAL, .text = "unknown command: "+cmd[0].string_value() };
|
||||
}
|
||||
return action_cb;
|
||||
}
|
||||
|
||||
static int run(cli_tool_t *p, json11::Json::object cfg)
|
||||
{
|
||||
cli_result_t result = {};
|
||||
p->is_command_line = true;
|
||||
p->parse_config(cfg);
|
||||
auto action_cb = p->start(cfg, result);
|
||||
if (action_cb != NULL)
|
||||
{
|
||||
// Create client
|
||||
@@ -585,7 +550,6 @@ static int run(cli_tool_t *p, json11::Json::object cfg)
|
||||
{
|
||||
result = r;
|
||||
action_cb = NULL;
|
||||
p->ringloop->submit();
|
||||
});
|
||||
// Loop until it completes
|
||||
while (action_cb != NULL)
|
||||
|
||||
+1
-4
@@ -47,7 +47,6 @@ public:
|
||||
json11::Json etcd_result;
|
||||
|
||||
void parse_config(json11::Json::object & cfg);
|
||||
void parse_api_opts(json11::Json::object & cfg);
|
||||
json11::Json parse_tags(std::string tags);
|
||||
|
||||
void change_parent(inode_t cur, inode_t new_parent, cli_result_t *result);
|
||||
@@ -58,10 +57,8 @@ public:
|
||||
friend struct snap_flattener_t;
|
||||
friend struct snap_remover_t;
|
||||
|
||||
std::function<bool(cli_result_t &)> start(json11::Json::object cfg, cli_result_t & result);
|
||||
std::function<bool(cli_result_t &)> start_alloc_osd(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_create(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_dd(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_describe(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_fix(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_flatten(json11::Json);
|
||||
@@ -79,8 +76,8 @@ public:
|
||||
std::function<bool(cli_result_t &)> start_rm_data(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_rm_osd(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_rm_wildcard(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_serve(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_status(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_dd(json11::Json);
|
||||
|
||||
// Should be called like loop_and_wait(start_status(), <completion callback>)
|
||||
void loop_and_wait(std::function<bool(cli_result_t &)> loop_cb, std::function<void(const cli_result_t &)> complete_cb);
|
||||
|
||||
+7
-11
@@ -101,16 +101,6 @@ inode_config_t* cli_tool_t::get_inode_cfg(const std::string & name)
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void cli_tool_t::parse_api_opts(json11::Json::object & cfg)
|
||||
{
|
||||
iodepth = cfg["iodepth"].uint64_value();
|
||||
if (!iodepth)
|
||||
iodepth = 32;
|
||||
parallel_osds = cfg["parallel_osds"].uint64_value();
|
||||
if (!parallel_osds)
|
||||
parallel_osds = 4;
|
||||
}
|
||||
|
||||
void cli_tool_t::parse_config(json11::Json::object & cfg)
|
||||
{
|
||||
for (auto kv_it = cfg.begin(); kv_it != cfg.end();)
|
||||
@@ -131,10 +121,15 @@ void cli_tool_t::parse_config(json11::Json::object & cfg)
|
||||
else
|
||||
color = isatty(1);
|
||||
json_output = cfg["json"].bool_value();
|
||||
iodepth = cfg["iodepth"].uint64_value();
|
||||
if (!iodepth)
|
||||
iodepth = 32;
|
||||
parallel_osds = cfg["parallel_osds"].uint64_value();
|
||||
if (!parallel_osds)
|
||||
parallel_osds = 4;
|
||||
log_level = cfg["log_level"].int64_value();
|
||||
progress = cfg["progress"].uint64_value() ? true : false;
|
||||
list_first = cfg["wait_list"].uint64_value() ? true : false;
|
||||
parse_api_opts(cfg);
|
||||
}
|
||||
|
||||
struct cli_result_looper_t
|
||||
@@ -158,6 +153,7 @@ void cli_tool_t::loop_and_wait(std::function<bool(cli_result_t &)> loop_cb, std:
|
||||
ringloop->unregister_consumer(&looper->consumer);
|
||||
looper->loop_cb = NULL;
|
||||
looper->complete_cb(looper->result);
|
||||
ringloop->submit();
|
||||
delete looper;
|
||||
return;
|
||||
}
|
||||
|
||||
+12
-47
@@ -64,8 +64,7 @@ struct image_creator_t
|
||||
}
|
||||
if (new_pool_id)
|
||||
{
|
||||
auto pool_it = pools.find(new_pool_id);
|
||||
if (pool_it == pools.end())
|
||||
if (pools.find(new_pool_id) == pools.end())
|
||||
{
|
||||
result = (cli_result_t){ .err = ENOENT, .text = "Pool "+std::to_string(new_pool_id)+" does not exist" };
|
||||
state = 100;
|
||||
@@ -195,31 +194,15 @@ resume_3:
|
||||
// Save into inode_config for library users to be able to take it from there immediately
|
||||
new_cfg.mod_revision = parent->etcd_result["header"]["revision"].uint64_value();
|
||||
parent->cli->st_cli.insert_inode_config(new_cfg);
|
||||
auto img = json11::Json::object {
|
||||
{ "inode_id", INODE_WITH_POOL(new_pool_id, new_id) },
|
||||
{ "inode_num", new_id },
|
||||
{ "name", image_name },
|
||||
{ "pool_id", (uint64_t)new_pool_id },
|
||||
{ "size", size },
|
||||
};
|
||||
{
|
||||
auto new_pool_it = parent->cli->st_cli.pool_config.find(new_pool_id);
|
||||
if (new_pool_it != parent->cli->st_cli.pool_config.end())
|
||||
{
|
||||
img["pool_name"] = new_pool_it->second.name;
|
||||
}
|
||||
}
|
||||
if (new_parent_id)
|
||||
{
|
||||
img["parent_name"] = new_parent;
|
||||
img["parent_inode_id"] = new_parent_id;
|
||||
img["parent_inode_num"] = INODE_NO_POOL(new_parent_id);
|
||||
img["parent_pool_id"] = (uint64_t)INODE_POOL(new_parent_id);
|
||||
}
|
||||
result = (cli_result_t){
|
||||
.err = 0,
|
||||
.text = "Image "+image_name+" created",
|
||||
.data = img,
|
||||
.data = json11::Json::object {
|
||||
{ "name", image_name },
|
||||
{ "pool", new_pool_name },
|
||||
{ "parent", new_parent },
|
||||
{ "size", size },
|
||||
}
|
||||
};
|
||||
state = 100;
|
||||
}
|
||||
@@ -289,23 +272,13 @@ resume_4:
|
||||
// Save into inode_config for library users to be able to take it from there immediately
|
||||
new_cfg.mod_revision = parent->etcd_result["header"]["revision"].uint64_value();
|
||||
parent->cli->st_cli.insert_inode_config(new_cfg);
|
||||
{
|
||||
auto new_pool_it = parent->cli->st_cli.pool_config.find(new_pool_id);
|
||||
new_pool_name = new_pool_it != parent->cli->st_cli.pool_config.end() ? new_pool_it->second.name : "";
|
||||
}
|
||||
result = (cli_result_t){
|
||||
.err = 0,
|
||||
.text = "Snapshot "+image_name+"@"+new_snap+" created",
|
||||
.data = json11::Json::object {
|
||||
{ "inode_id", INODE_WITH_POOL(new_pool_id, new_id) },
|
||||
{ "inode_num", new_id },
|
||||
{ "name", image_name },
|
||||
{ "pool_id", (uint64_t)new_pool_id },
|
||||
{ "pool_name", new_pool_name },
|
||||
{ "parent_name", image_name+"@"+new_snap },
|
||||
{ "parent_inode_id", INODE_WITH_POOL(old_pool_id, old_id) },
|
||||
{ "parent_inode_num", old_id },
|
||||
{ "parent_pool_id", (uint64_t)old_pool_id },
|
||||
{ "name", image_name+"@"+new_snap },
|
||||
{ "pool", (uint64_t)new_pool_id },
|
||||
{ "parent", new_parent },
|
||||
{ "size", size },
|
||||
}
|
||||
};
|
||||
@@ -581,16 +554,8 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_create(json11::Json cfg)
|
||||
auto image_creator = new image_creator_t();
|
||||
image_creator->parent = this;
|
||||
image_creator->image_name = cfg["image"].string_value();
|
||||
if (!cfg["pool"].is_null())
|
||||
{
|
||||
image_creator->new_pool_id = cfg["pool"].uint64_value();
|
||||
image_creator->new_pool_name = cfg["pool"].string_value();
|
||||
}
|
||||
else
|
||||
{
|
||||
image_creator->new_pool_id = cfg["pool_id"].uint64_value();
|
||||
image_creator->new_pool_name = cfg["pool_name"].string_value();
|
||||
}
|
||||
image_creator->new_pool_id = cfg["pool"].uint64_value();
|
||||
image_creator->new_pool_name = cfg["pool"].string_value();
|
||||
image_creator->force = cfg["force"].bool_value();
|
||||
image_creator->force_size = cfg["force_size"].bool_value();
|
||||
if (cfg["image_meta"].is_object())
|
||||
|
||||
+1
-1
@@ -864,7 +864,7 @@ resume_2:
|
||||
// Copy data
|
||||
if (iinfo.in_seekable && iseek >= iinfo.in_size)
|
||||
{
|
||||
result = (cli_result_t){ .err = EINVAL, .text = "Input seek position is beyond end of input" };
|
||||
result = (cli_result_t){ .err = -EINVAL, .text = "Input seek position is beyond end of input" };
|
||||
goto close_end;
|
||||
}
|
||||
if (!iinfo.iwatch && !iinfo.in_seekable && iseek)
|
||||
|
||||
@@ -57,24 +57,12 @@ struct cli_describe_t
|
||||
|
||||
void parse_options(json11::Json cfg)
|
||||
{
|
||||
uint64_t pool_id;
|
||||
std::string pool_name;
|
||||
if (!cfg["pool"].is_null())
|
||||
{
|
||||
pool_id = cfg["pool"].uint64_value();
|
||||
pool_name = pool_id ? "" : cfg["pool"].string_value();
|
||||
}
|
||||
else
|
||||
{
|
||||
pool_id = cfg["pool_id"].uint64_value();
|
||||
pool_name = pool_id ? "" : cfg["pool_name"].string_value();
|
||||
}
|
||||
only_pool = pool_id;
|
||||
if (!only_pool && pool_name != "")
|
||||
only_pool = cfg["pool"].uint64_value();
|
||||
if (!only_pool && cfg["pool"].is_string())
|
||||
{
|
||||
for (auto & pp: parent->cli->st_cli.pool_config)
|
||||
{
|
||||
if (pp.second.name == pool_name)
|
||||
if (pp.second.name == cfg["pool"].string_value())
|
||||
{
|
||||
only_pool = pp.first;
|
||||
break;
|
||||
@@ -118,22 +106,6 @@ struct cli_describe_t
|
||||
if (cfg["object_state"].string_value().find("misplaced") != std::string::npos)
|
||||
object_state |= OBJ_MISPLACED;
|
||||
}
|
||||
else if (!object_state && cfg["object_state"].is_array())
|
||||
{
|
||||
for (auto & st: cfg["object_state"].array_items())
|
||||
{
|
||||
if (st == "inconsistent")
|
||||
object_state |= OBJ_INCONSISTENT;
|
||||
else if (st == "corrupted")
|
||||
object_state |= OBJ_CORRUPTED;
|
||||
else if (st == "incomplete")
|
||||
object_state |= OBJ_INCOMPLETE;
|
||||
else if (st == "degraded")
|
||||
object_state |= OBJ_DEGRADED;
|
||||
else if (st == "misplaced")
|
||||
object_state |= OBJ_MISPLACED;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void loop()
|
||||
|
||||
+13
-39
@@ -17,7 +17,6 @@ struct image_lister_t
|
||||
std::string list_pool_name;
|
||||
std::string sort_field;
|
||||
std::set<std::string> only_names;
|
||||
std::vector<uint64_t> only_ids;
|
||||
bool reverse = false;
|
||||
bool exact = false;
|
||||
bool tree = false;
|
||||
@@ -64,6 +63,7 @@ struct image_lister_t
|
||||
auto item = json11::Json::object {
|
||||
{ "name", ic.second.name },
|
||||
{ "size", ic.second.size },
|
||||
{ "used_size", 0 },
|
||||
{ "readonly", ic.second.readonly },
|
||||
{ "pool_id", (uint64_t)INODE_POOL(ic.second.num) },
|
||||
{ "pool_name", good_pool ? pool_it->second.name : "? (ID:"+std::to_string(INODE_POOL(ic.second.num))+")" },
|
||||
@@ -202,33 +202,20 @@ resume_1:
|
||||
json11::Json::array to_list()
|
||||
{
|
||||
json11::Json::array list;
|
||||
if (only_ids.size())
|
||||
for (auto & kv: stats)
|
||||
{
|
||||
for (auto & id: only_ids)
|
||||
if (!only_names.size())
|
||||
{
|
||||
if (stats.find(id) != stats.end())
|
||||
{
|
||||
list.push_back(stats[id]);
|
||||
}
|
||||
list.push_back(kv.second);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
for (auto & kv: stats)
|
||||
else
|
||||
{
|
||||
if (!only_names.size())
|
||||
for (auto & glob: only_names)
|
||||
{
|
||||
list.push_back(kv.second);
|
||||
}
|
||||
else
|
||||
{
|
||||
for (auto & glob: only_names)
|
||||
if (exact ? (kv.second["name"].string_value() == glob) : stupid_glob(kv.second["name"].string_value(), glob))
|
||||
{
|
||||
if (exact ? (kv.second["name"].string_value() == glob) : stupid_glob(kv.second["name"].string_value(), glob))
|
||||
{
|
||||
list.push_back(kv.second);
|
||||
break;
|
||||
}
|
||||
list.push_back(kv.second);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -583,30 +570,17 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_ls(json11::Json cfg)
|
||||
lister->parent = this;
|
||||
lister->exact = cfg["exact"].bool_value();
|
||||
lister->tree = cfg["tree"].bool_value();
|
||||
if (!cfg["pool"].is_null())
|
||||
{
|
||||
lister->list_pool_id = cfg["pool"].uint64_value();
|
||||
lister->list_pool_name = lister->list_pool_id ? "" : cfg["pool"].as_string();
|
||||
}
|
||||
else
|
||||
{
|
||||
lister->list_pool_id = cfg["pool_id"].uint64_value();
|
||||
lister->list_pool_name = lister->list_pool_id ? "" : cfg["pool_name"].string_value();
|
||||
}
|
||||
lister->list_pool_id = cfg["pool"].uint64_value();
|
||||
lister->list_pool_name = lister->list_pool_id ? "" : cfg["pool"].as_string();
|
||||
lister->show_stats = cfg["long"].bool_value();
|
||||
lister->show_delete = cfg["del"].bool_value();
|
||||
lister->sort_field = cfg["sort"].string_value() != "" ? cfg["sort"].string_value() : "name";
|
||||
lister->reverse = cfg["reverse"].bool_value();
|
||||
lister->max_count = cfg["count"].uint64_value();
|
||||
if (cfg["names"].is_string())
|
||||
lister->only_names.insert(cfg["names"].string_value());
|
||||
for (auto & item: cfg["names"].array_items())
|
||||
{
|
||||
lister->only_names.insert(item.string_value());
|
||||
if (cfg["ids"].is_string())
|
||||
for (auto & item: explode(",", cfg["ids"].string_value(), true))
|
||||
lister->only_ids.push_back(stoull_full(item));
|
||||
for (auto & item: cfg["ids"].array_items())
|
||||
lister->only_ids.push_back(item.uint64_value());
|
||||
}
|
||||
return [lister](cli_result_t & result)
|
||||
{
|
||||
lister->loop();
|
||||
|
||||
@@ -374,7 +374,7 @@ struct snap_merger_t
|
||||
result = (cli_result_t){ .text = "Done, layers from "+from_name+" to "+to_name+" merged into "+target_name, .data = json11::Json::object {
|
||||
{ "from", from_name },
|
||||
{ "to", to_name },
|
||||
{ "target", target_name },
|
||||
{ "into", target_name },
|
||||
}};
|
||||
state = 100;
|
||||
resume_100:
|
||||
|
||||
+10
-34
@@ -88,7 +88,10 @@ struct image_changer_t
|
||||
(!new_size && !force_size || cfg.size == new_size || cfg.size >= new_size && inc_size) &&
|
||||
(new_name == "" || new_name == image_name))
|
||||
{
|
||||
result = (cli_result_t){ .err = 0, .text = "No change", .data = fill_img(cfg) };
|
||||
result = (cli_result_t){ .err = 0, .text = "No change", .data = json11::Json::object {
|
||||
{ "error_code", 0 },
|
||||
{ "error_text", "No change" },
|
||||
}};
|
||||
state = 100;
|
||||
return;
|
||||
}
|
||||
@@ -232,42 +235,15 @@ resume_2:
|
||||
result = (cli_result_t){
|
||||
.err = 0,
|
||||
.text = "Image "+image_name+" modified",
|
||||
.data = fill_img(cfg)
|
||||
.data = json11::Json::object {
|
||||
{ "name", image_name },
|
||||
{ "inode", INODE_NO_POOL(inode_num) },
|
||||
{ "pool", (uint64_t)INODE_POOL(inode_num) },
|
||||
{ "size", new_size },
|
||||
}
|
||||
};
|
||||
state = 100;
|
||||
}
|
||||
|
||||
json11::Json fill_img(inode_config_t & cfg)
|
||||
{
|
||||
auto img = json11::Json::object {
|
||||
{ "inode_id", inode_num },
|
||||
{ "inode_num", INODE_NO_POOL(inode_num) },
|
||||
{ "name", cfg.name },
|
||||
{ "pool_id", (uint64_t)INODE_POOL(inode_num) },
|
||||
{ "size", cfg.size },
|
||||
{ "readonly", cfg.readonly },
|
||||
{ "deleted", cfg.deleted },
|
||||
};
|
||||
{
|
||||
auto pool_it = parent->cli->st_cli.pool_config.find(INODE_POOL(inode_num));
|
||||
if (pool_it != parent->cli->st_cli.pool_config.end())
|
||||
{
|
||||
img["pool_name"] = pool_it->second.name;
|
||||
}
|
||||
}
|
||||
if (cfg.parent_id)
|
||||
{
|
||||
auto parent_it = parent->cli->st_cli.inode_config.find(cfg.parent_id);
|
||||
if (parent_it != parent->cli->st_cli.inode_config.end())
|
||||
{
|
||||
img["parent_name"] = parent_it->second.name;
|
||||
}
|
||||
img["parent_inode_id"] = cfg.parent_id;
|
||||
img["parent_inode_num"] = INODE_NO_POOL(cfg.parent_id);
|
||||
img["parent_pool_id"] = (uint64_t)INODE_POOL(cfg.parent_id);
|
||||
}
|
||||
return img;
|
||||
}
|
||||
};
|
||||
|
||||
std::function<bool(cli_result_t &)> cli_tool_t::start_modify(json11::Json cfg)
|
||||
|
||||
+18
-26
@@ -41,7 +41,7 @@ struct osd_tree_printer_t
|
||||
{
|
||||
cli_tool_t *parent;
|
||||
json11::Json cfg;
|
||||
bool as_tree = false;
|
||||
bool flat = false;
|
||||
bool show_stats = false;
|
||||
|
||||
int state = 0;
|
||||
@@ -209,14 +209,11 @@ resume_1:
|
||||
for (int i = 1; i < node_seq.size(); i++)
|
||||
{
|
||||
auto & node = placement_tree->nodes.at(node_seq[i]);
|
||||
if (as_tree)
|
||||
{
|
||||
fmt_items.push_back(json11::Json::object{
|
||||
{ "type", node.level },
|
||||
{ "name", node.name },
|
||||
{ "parent", node.parent },
|
||||
});
|
||||
}
|
||||
fmt_items.push_back(json11::Json::object{
|
||||
{ "type", node.level },
|
||||
{ "name", node.name },
|
||||
{ "parent", node.parent },
|
||||
});
|
||||
for (uint64_t osd_num: node.child_osds)
|
||||
{
|
||||
auto & osd = placement_tree->osds.at(osd_num);
|
||||
@@ -224,22 +221,17 @@ resume_1:
|
||||
{ "type", "osd" },
|
||||
{ "name", osd.num },
|
||||
{ "parent", node.name },
|
||||
{ "up", osd.up },
|
||||
{ "up", osd.up ? "up" : "down" },
|
||||
{ "size", osd.size },
|
||||
{ "free", osd.free },
|
||||
{ "reweight", osd.reweight },
|
||||
{ "noout", osd.noout },
|
||||
{ "tags", osd.tags },
|
||||
{ "data_block_size", (uint64_t)osd.block_size },
|
||||
{ "bitmap_granularity", (uint64_t)osd.bitmap_granularity },
|
||||
{ "immediate_commit", osd.immediate_commit == IMMEDIATE_NONE ? "none" : (osd.immediate_commit == IMMEDIATE_ALL ? "all" : "small") },
|
||||
{ "block", (uint64_t)osd.block_size },
|
||||
{ "bitmap", (uint64_t)osd.bitmap_granularity },
|
||||
{ "commit", osd.immediate_commit == IMMEDIATE_NONE ? "none" : (osd.immediate_commit == IMMEDIATE_ALL ? "all" : "small") },
|
||||
{ "op_stats", osd_stats[osd_num]["op_stats"] },
|
||||
};
|
||||
if (show_stats)
|
||||
{
|
||||
json_osd["op_stats"] = osd_stats[osd_num]["op_stats"];
|
||||
json_osd["subop_stats"] = osd_stats[osd_num]["subop_stats"];
|
||||
json_osd["recovery_stats"] = osd_stats[osd_num]["recovery_stats"];
|
||||
}
|
||||
if (osd_stats[osd_num]["slow_ops_primary"].uint64_value() > 0)
|
||||
{
|
||||
json_osd["slow_ops_primary"] = osd_stats[osd_num]["slow_ops_primary"];
|
||||
@@ -257,7 +249,7 @@ resume_1:
|
||||
for (int i = 1; i < node_seq.size(); i++)
|
||||
{
|
||||
auto & node = placement_tree->nodes.at(node_seq[i]);
|
||||
if (as_tree)
|
||||
if (!flat)
|
||||
{
|
||||
fmt_items.push_back(json11::Json::object{
|
||||
{ "type", str_repeat(" ", indents[i]) + node.level },
|
||||
@@ -265,7 +257,7 @@ resume_1:
|
||||
});
|
||||
}
|
||||
std::string parent = node.name;
|
||||
if (!as_tree)
|
||||
if (flat)
|
||||
{
|
||||
auto cur = &placement_tree->nodes.at(node.name);
|
||||
while (cur->parent != "" && cur->parent != node.name)
|
||||
@@ -278,7 +270,7 @@ resume_1:
|
||||
{
|
||||
auto & osd = placement_tree->osds.at(osd_num);
|
||||
auto fmt = json11::Json::object{
|
||||
{ "type", (!as_tree ? "osd" : str_repeat(" ", indents[i]+1) + "osd") },
|
||||
{ "type", (flat ? "osd" : str_repeat(" ", indents[i]+1) + "osd") },
|
||||
{ "name", osd.num },
|
||||
{ "parent", parent },
|
||||
{ "up", osd.up ? "up" : "down" },
|
||||
@@ -308,7 +300,7 @@ resume_1:
|
||||
}
|
||||
}
|
||||
json11::Json::array cols;
|
||||
if (as_tree)
|
||||
if (!flat)
|
||||
{
|
||||
cols.push_back(json11::Json::object{
|
||||
{ "key", "type" },
|
||||
@@ -317,9 +309,9 @@ resume_1:
|
||||
}
|
||||
cols.push_back(json11::Json::object{
|
||||
{ "key", "name" },
|
||||
{ "title", !as_tree ? "OSD" : "NAME" },
|
||||
{ "title", flat ? "OSD" : "NAME" },
|
||||
});
|
||||
if (!as_tree)
|
||||
if (flat)
|
||||
{
|
||||
cols.push_back(json11::Json::object{
|
||||
{ "key", "parent" },
|
||||
@@ -422,7 +414,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_osd_tree(json11::Json cfg)
|
||||
auto osd_tree_printer = new osd_tree_printer_t();
|
||||
osd_tree_printer->parent = this;
|
||||
osd_tree_printer->cfg = cfg;
|
||||
osd_tree_printer->as_tree = cfg["as_tree"].bool_value();
|
||||
osd_tree_printer->flat = cfg["flat"].bool_value();
|
||||
osd_tree_printer->show_stats = cfg["long"].bool_value();
|
||||
return [osd_tree_printer](cli_result_t & result)
|
||||
{
|
||||
|
||||
@@ -282,16 +282,10 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_pg_list(json11::Json cfg)
|
||||
{
|
||||
auto pg_lister = new pg_lister_t();
|
||||
pg_lister->parent = this;
|
||||
if (!cfg["pool"].is_null())
|
||||
{
|
||||
if (cfg["pool"].uint64_value())
|
||||
pg_lister->pool_id = cfg["pool"].uint64_value();
|
||||
pg_lister->pool_name = pg_lister->pool_id ? "" : cfg["pool"].string_value();
|
||||
}
|
||||
else
|
||||
{
|
||||
pg_lister->pool_id = cfg["pool_id"].uint64_value();
|
||||
pg_lister->pool_name = pg_lister->pool_id ? "" : cfg["pool_name"].string_value();
|
||||
}
|
||||
pg_lister->pool_name = cfg["pool"].string_value();
|
||||
for (auto & st: cfg["pg_state"].array_items())
|
||||
pg_lister->pg_state.push_back(st.string_value());
|
||||
if (cfg["pg_state"].is_string())
|
||||
|
||||
@@ -213,7 +213,7 @@ resume_3:
|
||||
if (failure_domain != "osd")
|
||||
pool_err += "\n- different parent '"+failure_domain+"' nodes";
|
||||
result = (cli_result_t){
|
||||
.err = EBUSY,
|
||||
.err = EINVAL,
|
||||
.text = pool_err,
|
||||
};
|
||||
state = 100;
|
||||
|
||||
@@ -206,7 +206,7 @@ resume_1:
|
||||
{ "space_efficiency", pool_stats[pool_cfg.id]["space_efficiency"].number_value() },
|
||||
{ "pg_real_size", pool_stats[pool_cfg.id]["pg_real_size"].uint64_value() },
|
||||
{ "osd_count", (uint64_t)pg_per_osd.size() },
|
||||
{ "backfillfull", !!pool_cfg.backfillfull },
|
||||
{ "backfillfull", pool_cfg.backfillfull },
|
||||
};
|
||||
}
|
||||
// Include full pool config
|
||||
@@ -546,10 +546,6 @@ resume_3:
|
||||
{ "write_fmt", "Write" },
|
||||
{ "delete_fmt", "Delete" },
|
||||
};
|
||||
if (sort_field == "osd_tags" || sort_field == "primary_affinity_tags")
|
||||
{
|
||||
sort_field += "_fmt";
|
||||
}
|
||||
auto list = to_list();
|
||||
size_t title_len = 0;
|
||||
for (auto & item: list)
|
||||
@@ -670,12 +666,15 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_pool_ls(json11::Json cfg)
|
||||
lister->show_stats = cfg["long"].bool_value();
|
||||
lister->detailed = cfg["detail"].bool_value();
|
||||
lister->sort_field = cfg["sort"].string_value();
|
||||
if ((lister->sort_field == "osd_tags") ||
|
||||
(lister->sort_field == "primary_affinity_tags" ))
|
||||
lister->sort_field = lister->sort_field + "_fmt";
|
||||
lister->reverse = cfg["reverse"].bool_value();
|
||||
lister->max_count = cfg["count"].uint64_value();
|
||||
if (cfg["names"].is_string())
|
||||
lister->only_names.insert(cfg["names"].string_value());
|
||||
for (auto & item: cfg["names"].array_items())
|
||||
{
|
||||
lister->only_names.insert(item.string_value());
|
||||
}
|
||||
return [lister](cli_result_t & result)
|
||||
{
|
||||
lister->loop();
|
||||
|
||||
+10
-15
@@ -39,21 +39,16 @@ struct pool_changer_t
|
||||
goto resume_1;
|
||||
else if (state == 2)
|
||||
goto resume_2;
|
||||
if (!cfg["pool"].is_null())
|
||||
pool_id = stoull_full(cfg["old_name"].string_value());
|
||||
if (!pool_id)
|
||||
{
|
||||
pool_id = cfg["pool"].uint64_value();
|
||||
pool_name = pool_id ? "" : cfg["pool"].string_value();
|
||||
}
|
||||
else
|
||||
{
|
||||
pool_id = cfg["id"].uint64_value();
|
||||
pool_name = pool_id ? "" : cfg["old_name"].string_value();
|
||||
}
|
||||
if (!pool_id && pool_name == "")
|
||||
{
|
||||
result = (cli_result_t){ .err = ENOENT, .text = "Pool ID or name is required to modify it" };
|
||||
state = 100;
|
||||
return;
|
||||
pool_name = cfg["old_name"].string_value();
|
||||
if (pool_name == "")
|
||||
{
|
||||
result = (cli_result_t){ .err = ENOENT, .text = "Pool ID or name is required to modify it" };
|
||||
state = 100;
|
||||
return;
|
||||
}
|
||||
}
|
||||
resume_0:
|
||||
// Get pools from etcd
|
||||
@@ -184,7 +179,7 @@ resume_2:
|
||||
result = (cli_result_t){
|
||||
.err = 0,
|
||||
.text = "Pool "+pool_name+" updated",
|
||||
.data = new_pools[std::to_string(pool_id)],
|
||||
.data = new_pools,
|
||||
};
|
||||
state = 100;
|
||||
}
|
||||
|
||||
+3
-13
@@ -196,9 +196,7 @@ resume_3:
|
||||
result = (cli_result_t){
|
||||
.err = 0,
|
||||
.text = "Pool "+pool_name+" deleted",
|
||||
.data = json11::Json::object{
|
||||
{"ok", true},
|
||||
},
|
||||
.data = new_pools
|
||||
};
|
||||
state = 100;
|
||||
}
|
||||
@@ -209,16 +207,8 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_pool_rm(json11::Json cfg)
|
||||
auto pool_remover = new pool_remover_t();
|
||||
pool_remover->parent = this;
|
||||
|
||||
if (!cfg["pool"].is_null())
|
||||
{
|
||||
pool_remover->pool_id = cfg["pool"].uint64_value();
|
||||
pool_remover->pool_name = pool_remover->pool_id ? "" : cfg["pool"].as_string();
|
||||
}
|
||||
else
|
||||
{
|
||||
pool_remover->pool_id = cfg["pool_id"].uint64_value();
|
||||
pool_remover->pool_name = pool_remover->pool_id ? "" : cfg["pool_name"].as_string();
|
||||
}
|
||||
pool_remover->pool_id = cfg["pool"].uint64_value();
|
||||
pool_remover->pool_name = pool_remover->pool_id ? "" : cfg["pool"].as_string();
|
||||
|
||||
pool_remover->force = !cfg["force"].is_null();
|
||||
|
||||
|
||||
@@ -264,7 +264,7 @@ struct rm_inode_t
|
||||
{
|
||||
// Error
|
||||
result = (cli_result_t){
|
||||
.err = EBUSY,
|
||||
.err = EIO,
|
||||
.text = "Failed: some blocks were not removed",
|
||||
.data = data,
|
||||
};
|
||||
|
||||
@@ -181,7 +181,6 @@ struct rm_osd_t
|
||||
if (dry_run || is_dataloss && !force_dataloss || is_warning && !force_warning)
|
||||
{
|
||||
result.err = is_dataloss && !force_dataloss || is_warning && !force_warning ? EBUSY : 0;
|
||||
result.data = json11::Json::object{ { "pool_effects", pool_effects } };
|
||||
state = 100;
|
||||
return;
|
||||
}
|
||||
@@ -333,9 +332,6 @@ struct rm_osd_t
|
||||
ids = (osd_ids.size() > 1 ? "OSDs " : "OSD ")+ids+(osd_ids.size() > 1 ? " are" : " is")+" removed from etcd";
|
||||
state = 100;
|
||||
result.text = (result.text != "" ? ids+"\n"+result.text : ids);
|
||||
result.data = json11::Json::object {
|
||||
{"deleted_osds", osd_ids},
|
||||
};
|
||||
result.err = 0;
|
||||
}
|
||||
|
||||
|
||||
@@ -195,8 +195,6 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_rm_wildcard(json11::Json c
|
||||
auto wildcard_remover = new wildcard_remover_t();
|
||||
wildcard_remover->parent = this;
|
||||
wildcard_remover->cfg = cfg;
|
||||
if (cfg["globs"].is_string())
|
||||
wildcard_remover->globs.push_back(cfg["globs"].string_value());
|
||||
for (auto & glob: cfg["globs"].array_items())
|
||||
wildcard_remover->globs.push_back(glob.string_value());
|
||||
wildcard_remover->exact = cfg["exact"].bool_value();
|
||||
|
||||
@@ -1,439 +0,0 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#include <signal.h>
|
||||
#include <sys/socket.h>
|
||||
#include <netinet/tcp.h>
|
||||
#include <stdexcept>
|
||||
#include "cli.h"
|
||||
#include "cluster_client.h"
|
||||
#include "epoll_manager.h"
|
||||
#include "http_client.h"
|
||||
#include "str_util.h"
|
||||
#include "json_util.h"
|
||||
#include "addr_util.h"
|
||||
#include "openapi.json.h"
|
||||
|
||||
struct cli_serve_conn_t
|
||||
{
|
||||
int peer_fd = 0;
|
||||
std::string peer_addr;
|
||||
http_co_t *co = NULL;
|
||||
cli_tool_t *p = NULL;
|
||||
cli_result_t result;
|
||||
bool keepalive = false;
|
||||
bool closed = false;
|
||||
timespec request_time;
|
||||
std::string request_method;
|
||||
std::string request_path;
|
||||
std::string request_body;
|
||||
std::string response_type;
|
||||
std::function<bool(cli_result_t &)> action_cb;
|
||||
};
|
||||
|
||||
struct cli_serve_path_t
|
||||
{
|
||||
std::string cmd;
|
||||
bool allow_get;
|
||||
};
|
||||
|
||||
// Serve vitastor-cli commands over HTTP in JSON format
|
||||
struct cli_serve_t
|
||||
{
|
||||
std::map<std::string, cli_serve_path_t> cmd_paths = {
|
||||
{"data/delete", {"rm-data", false}},
|
||||
{"data/describe", {"describe", true}},
|
||||
{"data/fix", {"fix", false}},
|
||||
{"data/merge", {"merge-data", false}},
|
||||
{"image/create", {"create", false}},
|
||||
{"image/delete", {"rm", false}},
|
||||
{"image/flatten", {"flatten", false}},
|
||||
{"image/list", {"ls", true}},
|
||||
{"image/modify", {"modify", false}},
|
||||
{"osd/alloc", {"alloc-osd", false}},
|
||||
{"osd/delete", {"rm-osd", false}},
|
||||
{"osd/list", {"ls-osd", true}},
|
||||
{"osd/modify", {"modify-osd", false}},
|
||||
{"pg/list", {"ls-pgs", true}},
|
||||
{"pool/create", {"create-pool", false}},
|
||||
{"pool/delete", {"rm-pool", false}},
|
||||
{"pool/list", {"pools", true}},
|
||||
{"pool/modify", {"modify-pool", false}},
|
||||
{"status", {"status", true}},
|
||||
};
|
||||
|
||||
cli_tool_t *parent = NULL;
|
||||
json11::Json options;
|
||||
cli_result_t result;
|
||||
|
||||
bool log_body = false;
|
||||
bool stop = false;
|
||||
std::vector<std::string> bind_addresses;
|
||||
int port = 0;
|
||||
int listen_backlog = 0;
|
||||
bool ssl = false;
|
||||
std::vector<int> listen_fds;
|
||||
http_context_t *http_ctx = NULL;
|
||||
std::set<cli_serve_conn_t*> connections;
|
||||
|
||||
int state = 0;
|
||||
|
||||
bool is_done()
|
||||
{
|
||||
return state == 100;
|
||||
}
|
||||
|
||||
void loop()
|
||||
{
|
||||
if (state == 1)
|
||||
goto resume_1;
|
||||
else if (state == 2)
|
||||
goto resume_2;
|
||||
else if (state == 100)
|
||||
return;
|
||||
if (options["bind_address"].is_string())
|
||||
bind_addresses = explode(" ", options["bind_address"].string_value(), true);
|
||||
else
|
||||
bind_addresses.push_back("127.0.0.1");
|
||||
port = options["port"].uint64_value();
|
||||
if (!port)
|
||||
port = 8080;
|
||||
else if (port < 0 || port > 65535)
|
||||
{
|
||||
result = (cli_result_t){ .err = EINVAL, .text = "HTTP port can't be larger than 65536" };
|
||||
state = 100;
|
||||
return;
|
||||
}
|
||||
listen_backlog = options["listen_backlog"].uint64_value();
|
||||
if (!listen_backlog)
|
||||
listen_backlog = 128;
|
||||
ssl = json_is_true(options["ssl"]);
|
||||
if (ssl)
|
||||
{
|
||||
std::string ssl_cert = options["ssl_cert"].string_value();
|
||||
std::string ssl_key = options["ssl_key"].string_value();
|
||||
std::string ssl_ca = options["ssl_ca"].string_value();
|
||||
std::string error;
|
||||
http_ctx = http_context_init(ssl_cert, ssl_key, ssl_ca, ssl_ca != "", error);
|
||||
if (error != "")
|
||||
{
|
||||
result = (cli_result_t){ .err = EINVAL, .text = error };
|
||||
state = 100;
|
||||
return;
|
||||
}
|
||||
}
|
||||
for (auto & bind_address: bind_addresses)
|
||||
{
|
||||
int listen_fd = create_and_bind_socket(bind_address, port, listen_backlog, NULL);
|
||||
fcntl(listen_fd, F_SETFL, fcntl(listen_fd, F_GETFL, 0) | O_NONBLOCK);
|
||||
parent->epmgr->set_fd_handler(listen_fd, false, [this](int fd, int events)
|
||||
{
|
||||
accept_connections(fd);
|
||||
});
|
||||
listen_fds.push_back(listen_fd);
|
||||
}
|
||||
resume_1:
|
||||
if (!stop)
|
||||
{
|
||||
state = 1;
|
||||
return;
|
||||
}
|
||||
for (auto conn: connections)
|
||||
{
|
||||
http_close(conn->co);
|
||||
}
|
||||
resume_2:
|
||||
// Wait for all connections to finish
|
||||
if (connections.size() > 0)
|
||||
{
|
||||
state = 2;
|
||||
return;
|
||||
}
|
||||
if (http_ctx)
|
||||
{
|
||||
http_context_destroy(http_ctx);
|
||||
http_ctx = NULL;
|
||||
}
|
||||
state = 100;
|
||||
}
|
||||
|
||||
void accept_connections(int listen_fd)
|
||||
{
|
||||
sockaddr_storage addr;
|
||||
socklen_t peer_addr_size = sizeof(addr);
|
||||
int peer_fd;
|
||||
while ((peer_fd = accept(listen_fd, (sockaddr*)&addr, &peer_addr_size)) >= 0)
|
||||
{
|
||||
auto peer_addr_str = addr_to_string(addr);
|
||||
assert(peer_fd != 0);
|
||||
timespec ts;
|
||||
clock_gettime(CLOCK_REALTIME, &ts);
|
||||
printf("[%s.%03ju] New connection %d from %s\n", format_datetime(ts.tv_sec).c_str(), (uint64_t)ts.tv_nsec/1000000,
|
||||
peer_fd, peer_addr_str.c_str());
|
||||
fcntl(peer_fd, F_SETFL, fcntl(peer_fd, F_GETFL, 0) | O_NONBLOCK);
|
||||
int one = 1;
|
||||
setsockopt(peer_fd, SOL_TCP, TCP_NODELAY, &one, sizeof(one));
|
||||
cli_serve_conn_t *conn = new cli_serve_conn_t;
|
||||
conn->peer_fd = peer_fd;
|
||||
conn->peer_addr = peer_addr_str;
|
||||
conn->co = http_init(parent->epmgr->tfd, http_ctx);
|
||||
http_serve(conn->co, peer_fd, (http_options_t){ .ssl = ssl }, [this, conn](http_message_t *msg)
|
||||
{
|
||||
process_request(conn, msg);
|
||||
});
|
||||
connections.insert(conn);
|
||||
// Try to accept next connection
|
||||
peer_addr_size = sizeof(addr);
|
||||
}
|
||||
if (peer_fd == -1 && errno != EAGAIN)
|
||||
{
|
||||
throw std::runtime_error(std::string("accept: ") + strerror(errno));
|
||||
}
|
||||
}
|
||||
|
||||
int map_to_http(int err, std::string *text)
|
||||
{
|
||||
int code = 0;
|
||||
if (err == EINVAL)
|
||||
{
|
||||
code = 400;
|
||||
if (text)
|
||||
*text = "Bad Request";
|
||||
}
|
||||
else if (err == EOPNOTSUPP)
|
||||
{
|
||||
code = 404;
|
||||
if (text)
|
||||
*text = "Not Found";
|
||||
}
|
||||
else if (err == ENOSYS)
|
||||
{
|
||||
code = 405;
|
||||
if (text)
|
||||
*text = "Method Not Allowed";
|
||||
}
|
||||
else if (err == EAGAIN)
|
||||
{
|
||||
code = 409;
|
||||
if (text)
|
||||
*text = "Update Conflict";
|
||||
}
|
||||
else if (err == ENOTEMPTY || err == EEXIST || err == ENOENT || err == EBUSY)
|
||||
{
|
||||
code = 412;
|
||||
if (text)
|
||||
*text = "Precondition Failed";
|
||||
}
|
||||
else /*if (err == EIO || err == EBADF)*/
|
||||
{
|
||||
code = 500;
|
||||
if (text)
|
||||
*text = "Internal Server Error";
|
||||
}
|
||||
return code;
|
||||
}
|
||||
|
||||
std::string cli_http_response(cli_serve_conn_t *conn)
|
||||
{
|
||||
timespec now;
|
||||
clock_gettime(CLOCK_REALTIME, &now);
|
||||
int code = 200;
|
||||
std::string response;
|
||||
if (conn->result.err)
|
||||
{
|
||||
std::string status_line;
|
||||
code = map_to_http(conn->result.err, &status_line);
|
||||
response = "HTTP/1.1 "+std::to_string(code)+" "+status_line+"\r\n";
|
||||
}
|
||||
else
|
||||
{
|
||||
response = "HTTP/1.1 200 OK\r\n";
|
||||
}
|
||||
response += (conn->keepalive
|
||||
? "Connection: keep-alive\r\n"
|
||||
: "Connection: close\r\n");
|
||||
std::string body;
|
||||
if (!conn->result.data.is_null())
|
||||
{
|
||||
response += "Content-Type: application/json\r\n";
|
||||
body = conn->result.data.dump();
|
||||
}
|
||||
else
|
||||
{
|
||||
if (!conn->response_type.empty())
|
||||
response += "Content-Type: "+conn->response_type+"\r\n";
|
||||
else
|
||||
response += "Content-Type: text/plain; charset=utf-8\r\n";
|
||||
body = conn->result.text;
|
||||
}
|
||||
response += "Content-Length: "+std::to_string(body.size())+"\r\n\r\n";
|
||||
response += body;
|
||||
if (conn->request_method.find("\n") != std::string::npos)
|
||||
conn->request_method = str_replace(conn->request_method, "\n", "%0a");
|
||||
if (conn->request_method.find(" ") != std::string::npos)
|
||||
conn->request_method = str_replace(conn->request_method, " ", "%20");
|
||||
if (conn->request_path.find("\n") != std::string::npos)
|
||||
conn->request_path = str_replace(conn->request_path, "\n", "%0a");
|
||||
if (conn->request_path.find(" ") != std::string::npos)
|
||||
conn->request_path = str_replace(conn->request_path, " ", "%20");
|
||||
uint64_t response_time = (now.tv_sec-conn->request_time.tv_sec)*1000 + (now.tv_nsec-conn->request_time.tv_nsec)/1000000;
|
||||
printf("[%s.%03ju] %s %s %s %d %.03f sec\n",
|
||||
format_datetime(now.tv_sec).c_str(), (uint64_t)now.tv_nsec/1000000,
|
||||
conn->peer_addr.c_str(), conn->request_method.c_str(), conn->request_path.c_str(), code,
|
||||
response_time/1000.0);
|
||||
if (log_body)
|
||||
{
|
||||
if (conn->request_body.find("\n") != std::string::npos)
|
||||
conn->request_body = str_replace(conn->request_body, "\n", " ");
|
||||
if (conn->request_body.size())
|
||||
printf(" %s\n", conn->request_body.c_str());
|
||||
printf(" %s\n", body.c_str());
|
||||
}
|
||||
return response;
|
||||
}
|
||||
|
||||
void process_request(cli_serve_conn_t *conn, http_message_t *msg)
|
||||
{
|
||||
timespec ts;
|
||||
clock_gettime(CLOCK_REALTIME, &ts);
|
||||
if (!msg->error.empty())
|
||||
{
|
||||
// connection is closed
|
||||
fprintf(stderr, "[%s.%03ju] Connection %d closed: %s\n", format_datetime(ts.tv_sec).c_str(),
|
||||
(uint64_t)ts.tv_nsec/1000000, conn->peer_fd, msg->error.c_str());
|
||||
if (conn->p)
|
||||
conn->closed = true;
|
||||
else
|
||||
{
|
||||
connections.erase(conn);
|
||||
http_destroy(conn->co);
|
||||
delete conn;
|
||||
}
|
||||
return;
|
||||
}
|
||||
conn->keepalive = msg->headers.find("connection") != msg->headers.end() &&
|
||||
msg->headers.at("connection") == "keep-alive";
|
||||
conn->p = new cli_tool_t;
|
||||
conn->p->iodepth = parent->iodepth;
|
||||
conn->p->parallel_osds = parent->parallel_osds;
|
||||
conn->p->json_output = true;
|
||||
conn->p->ringloop = parent->ringloop;
|
||||
conn->p->epmgr = parent->epmgr;
|
||||
conn->p->cli = parent->cli;
|
||||
conn->p->is_command_line = false;
|
||||
// Parse request
|
||||
auto req_line = explode(" ", msg->status_line, true);
|
||||
if (req_line.size() < 2)
|
||||
{
|
||||
if (req_line[0] == "")
|
||||
req_line[0] = "-";
|
||||
req_line.push_back("-");
|
||||
}
|
||||
conn->request_time = ts;
|
||||
conn->request_method = std::move(req_line[0]);
|
||||
conn->request_path = std::move(req_line[1]);
|
||||
conn->request_body = std::move(msg->body);
|
||||
conn->response_type = "";
|
||||
auto ctype = msg->headers["content-type"];
|
||||
if (conn->request_method != "GET" && conn->request_method != "POST")
|
||||
{
|
||||
conn->result = { .err = ENOSYS, .text = "Unsupported request method "+conn->request_method };
|
||||
}
|
||||
else if (ctype != (conn->request_method == "GET" ? "" : "application/json"))
|
||||
{
|
||||
conn->result = { .err = EINVAL, .text = "Unsupported Content-Type: "+ctype+" for "+conn->request_method+" requests" };
|
||||
}
|
||||
else
|
||||
{
|
||||
auto uri = explode("?", conn->request_path, true);
|
||||
uri[0] = trim(uri[0], "/");
|
||||
auto cmd_it = cmd_paths.find(uri[0]);
|
||||
if (uri[0] == "")
|
||||
{
|
||||
std::string text = "Supported APIs:\n\n- GET /openapi\n";
|
||||
for (auto & pp: cmd_paths)
|
||||
{
|
||||
text += (pp.second.allow_get ? "- GET" : "- POST") + (" /" + pp.first) + "\n";
|
||||
}
|
||||
conn->result = { .text = text };
|
||||
}
|
||||
else if (uri[0] == "openapi")
|
||||
{
|
||||
conn->response_type = "application/json";
|
||||
conn->result = { .text = openapi_description };
|
||||
}
|
||||
else if (cmd_it == cmd_paths.end())
|
||||
{
|
||||
conn->result = { .err = EOPNOTSUPP, .text = "unknown command: "+uri[0] };
|
||||
}
|
||||
else if (conn->request_method == "GET" && !cmd_it->second.allow_get)
|
||||
{
|
||||
conn->result = { .err = ENOSYS, .text = "method /"+uri[0]+" only allows POST requests" };
|
||||
}
|
||||
else
|
||||
{
|
||||
std::string error;
|
||||
json11::Json::object cfg;
|
||||
if (conn->request_method == "POST")
|
||||
{
|
||||
cfg = json11::Json::parse(conn->request_body, error).object_items();
|
||||
}
|
||||
else
|
||||
{
|
||||
// Parse URI
|
||||
cfg = parse_uri_params(uri[1]);
|
||||
}
|
||||
if (error != "")
|
||||
{
|
||||
conn->result = { .err = EINVAL, .text = "Invalid JSON in body: "+error };
|
||||
}
|
||||
else
|
||||
{
|
||||
cfg["command"] = json11::Json::array{cmd_it->second.cmd};
|
||||
conn->p->parse_api_opts(cfg);
|
||||
conn->action_cb = conn->p->start(cfg, conn->result);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!conn->action_cb)
|
||||
{
|
||||
http_reply(conn->co, cli_http_response(conn));
|
||||
delete conn->p;
|
||||
conn->p = NULL;
|
||||
return;
|
||||
}
|
||||
conn->p->loop_and_wait(conn->action_cb, [this, conn](const cli_result_t & r)
|
||||
{
|
||||
conn->result = r;
|
||||
conn->action_cb = NULL;
|
||||
delete conn->p;
|
||||
conn->p = NULL;
|
||||
if (!conn->closed)
|
||||
http_reply(conn->co, cli_http_response(conn));
|
||||
else
|
||||
{
|
||||
connections.erase(conn);
|
||||
http_destroy(conn->co);
|
||||
delete conn;
|
||||
}
|
||||
});
|
||||
}
|
||||
};
|
||||
|
||||
std::function<bool(cli_result_t &)> cli_tool_t::start_serve(json11::Json cfg)
|
||||
{
|
||||
auto server = new cli_serve_t();
|
||||
server->parent = this;
|
||||
server->options = cfg;
|
||||
return [server](cli_result_t & result)
|
||||
{
|
||||
server->loop();
|
||||
if (server->is_done())
|
||||
{
|
||||
result = server->result;
|
||||
delete server;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
};
|
||||
}
|
||||
@@ -18,7 +18,6 @@ struct status_printer_t
|
||||
{
|
||||
cli_tool_t *parent;
|
||||
|
||||
cli_result_t result;
|
||||
int state = 0;
|
||||
json11::Json::array mon_members;
|
||||
json11::Json agg_stats;
|
||||
@@ -90,7 +89,7 @@ resume_2:
|
||||
return;
|
||||
if (parent->etcd_err.err)
|
||||
{
|
||||
result = parent->etcd_err;
|
||||
fprintf(stderr, "%s\n", parent->etcd_err.text.c_str());
|
||||
state = 100;
|
||||
return;
|
||||
}
|
||||
@@ -108,8 +107,7 @@ resume_2:
|
||||
if (etcd_states[i]["error"].is_null())
|
||||
{
|
||||
etcd_alive++;
|
||||
uint64_t db_size = etcd_states[i]["dbSize"].uint64_value();
|
||||
etcd_db_size = db_size > etcd_db_size ? db_size : etcd_db_size;
|
||||
etcd_db_size = etcd_states[i]["dbSize"].uint64_value();
|
||||
}
|
||||
}
|
||||
int mon_count = 0;
|
||||
@@ -267,7 +265,7 @@ resume_2:
|
||||
obj_n = agg_stats["object_counts"][str].uint64_value() * parent->cli->st_cli.global_block_size;
|
||||
json_status[str+"_data"] = obj_n;
|
||||
}
|
||||
result.data = json11::Json(json_status);
|
||||
printf("%s\n", json11::Json(json_status).dump().c_str());
|
||||
state = 100;
|
||||
return;
|
||||
}
|
||||
@@ -408,7 +406,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_status(json11::Json cfg)
|
||||
printer->loop();
|
||||
if (printer->is_done())
|
||||
{
|
||||
result = printer->result;
|
||||
result = { .err = 0 };
|
||||
delete printer;
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -1,937 +0,0 @@
|
||||
{
|
||||
"openapi": "3.0.0",
|
||||
"info": {
|
||||
"description": "Vitastor control API",
|
||||
"version": "3.0.2",
|
||||
"license": {
|
||||
"name": "VNPL 1.1"
|
||||
}
|
||||
},
|
||||
"paths": {
|
||||
"/status": { "get": {
|
||||
"summary": "Show server status",
|
||||
"operationId": "status",
|
||||
"produces": [ "application/json" ],
|
||||
"parameters": [],
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"pool_count": { "type": "integer", "format": "uint64", "description": "Total number of pools" },
|
||||
"active_pool_count": { "type": "integer", "format": "uint64", "description": "Number of active pools" },
|
||||
"backfillfull_pools": { "type": "array", "items": { "type": "integer", "format": "uint64" }, "description": "ID(s) of backfillfull pools" },
|
||||
"clean_data": { "type": "integer", "format": "uint64", "description": "Clean data - user size in bytes" },
|
||||
"misplaced_data": { "type": "integer", "format": "uint64", "description": "Misplaced data - user size in bytes" },
|
||||
"degraded_data": { "type": "integer", "format": "uint64", "description": "Degraded data - user size in bytes" },
|
||||
"incomplete_data": { "type": "integer", "format": "uint64", "description": "Incomplete data - user size in bytes" },
|
||||
"osd_count": { "type": "integer", "format": "uint64", "description": "Total number of OSDs" },
|
||||
"osd_up": { "type": "integer", "format": "uint64", "description": "Number of active OSDs" },
|
||||
"osds_full": { "type": "integer", "format": "uint64", "description": "Number of full OSDs" },
|
||||
"osds_nearfull": { "type": "integer", "format": "uint64", "description": "Number of nearfull OSDs" },
|
||||
"osds_primary_slow_ops": { "type": "array", "items": { "type": "integer", "format": "uint64" }, "description": "Numbers of OSDs with slow/hung client operations" },
|
||||
"osds_secondary_slow_ops": { "type": "array", "items": { "type": "integer", "format": "uint64" }, "description": "Numbers of OSDs with slow/hung storage operations" },
|
||||
"total_raw": { "type": "integer", "format": "uint64", "description": "Total size of all OSDs in bytes" },
|
||||
"free_raw": { "type": "integer", "format": "uint64", "description": "Total free space on OSDs in bytes" },
|
||||
"down_raw": { "type": "integer", "format": "uint64", "description": "Total size of stopped OSDs in bytes" },
|
||||
"free_down_raw": { "type": "integer", "format": "uint64", "description": "Total free space on stopped OSDs in bytes" },
|
||||
"etcd_count": { "type": "integer", "format": "uint64", "description": "Total number of etcd endpoints in the config" },
|
||||
"etcd_alive": { "type": "integer", "format": "uint64", "description": "Number of healthy etcd endpoints" },
|
||||
"etcd_db_size": { "type": "integer", "format": "uint64", "description": "Estimated maximum size of etcd databases (dbSize)" },
|
||||
"mon_count": { "type": "integer", "format": "uint64", "description": "Number of active monitors" },
|
||||
"mon_master": { "type": "string", "description": "Name of the host with active monitor" },
|
||||
"readonly": { "type": "boolean", "description": "readonly flag" },
|
||||
"no_rebalance": { "type": "boolean", "description": "no_rebalance flag" },
|
||||
"no_recovery": { "type": "boolean", "description": "no_recovery flag" },
|
||||
"no_scrub": { "type": "boolean", "description": "no_scrub flag" },
|
||||
"object_counts": { "type": "object", "properties": {
|
||||
"object": { "type": "integer", "format": "uint64", "description": "Total number of objects" },
|
||||
"clean": { "type": "integer", "format": "uint64", "description": "Number of clean objects" },
|
||||
"degraded": { "type": "integer", "format": "uint64", "description": "Number of degraded objects" },
|
||||
"incomplete": { "type": "integer", "format": "uint64", "description": "Number of incomplete objects" },
|
||||
"misplaced": { "type": "integer", "format": "uint64", "description": "Number of misplaced objects" }
|
||||
} },
|
||||
"pg_states": { "type": "object", "additionalProperties": { "type": "string" }, "description": "Number of PGs by state" },
|
||||
"op_stats": { "$ref": "#/components/schemas/OpStats" },
|
||||
"recovery_stats": { "$ref": "#/components/schemas/RecoveryStats" }
|
||||
}
|
||||
} } } }
|
||||
}
|
||||
} },
|
||||
"/image/list": { "get": {
|
||||
"summary": "List images",
|
||||
"operationId": "imageListGet",
|
||||
"produces": [ "application/json" ],
|
||||
"parameters": [ {
|
||||
"name": "params",
|
||||
"in": "query",
|
||||
"schema": { "$ref": "#/components/schemas/ImageListParams" },
|
||||
"style": "form",
|
||||
"explode": true
|
||||
} ],
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/ImageList" } } } }
|
||||
}
|
||||
}, "post": {
|
||||
"summary": "List images",
|
||||
"operationId": "imageListPost",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/ImageListParams" } } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/ImageList" } } } }
|
||||
}
|
||||
} },
|
||||
"/image/create": { "post": {
|
||||
"summary": "Create an image, snapshot, or a clone",
|
||||
"operationId": "imageCreate",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"required": [ "image" ],
|
||||
"properties": {
|
||||
"image": { "type": "string", "description": "Image name" },
|
||||
"snapshot": { "type": "string", "description": "Snapshot name" },
|
||||
"size": {
|
||||
"oneOf": [
|
||||
{ "type": "integer", "format": "uint64" },
|
||||
{ "type": "string", "pattern": "^\\d+[KMGT]$" }
|
||||
],
|
||||
"description": "Image size in bytes or with K/M/G/T suffix (not for snapshots)"
|
||||
},
|
||||
"pool_id": { "type": "integer", "format": "uint64", "description": "Pool ID for the new image/snapshot" },
|
||||
"pool_name": { "type": "string", "description": "Pool name for the new image/snapshot" },
|
||||
"parent": { "type": "string", "description": "Create a clone with this parent image name" }
|
||||
}
|
||||
} } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Image" } } } },
|
||||
"400": { "$ref": "#/components/responses/Invalid" },
|
||||
"409": { "$ref": "#/components/responses/UpdateConflict" },
|
||||
"412": { "$ref": "#/components/responses/PreconditionFailed" }
|
||||
}
|
||||
} },
|
||||
"/image/modify": { "post": {
|
||||
"summary": "Modify an image",
|
||||
"operationId": "imageModify",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"required": [ "image" ],
|
||||
"properties": {
|
||||
"image": { "type": "string", "description": "Old full name of the image" },
|
||||
"rename": { "type": "string", "description": "New full image name" },
|
||||
"resize": {
|
||||
"oneOf": [
|
||||
{ "type": "integer", "format": "uint64" },
|
||||
{ "type": "string", "pattern": "^\\d+[KMGT]$" }
|
||||
],
|
||||
"description": "New image size in bytes or with K/M/G/T suffix"
|
||||
},
|
||||
"readonly": { "type": "boolean", "description": "Make the image read-only" },
|
||||
"readwrite": { "type": "boolean", "description": "Make the image read-write" },
|
||||
"deleted": { "type": "boolean", "description": "Set or clear the 'deleted' flag" },
|
||||
"force": { "type": "boolean", "description": "Proceed with shrinking or setting readwrite flag even if the image has children" },
|
||||
"down_ok": { "type": "boolean", "description": "Proceed with shrinking even if some data will be left on unavailable OSDs" }
|
||||
}
|
||||
} } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Image" } } } },
|
||||
"400": { "$ref": "#/components/responses/Invalid" },
|
||||
"409": { "$ref": "#/components/responses/UpdateConflict" },
|
||||
"412": { "$ref": "#/components/responses/PreconditionFailed" }
|
||||
}
|
||||
} },
|
||||
"/image/delete": { "post": {
|
||||
"summary": "Delete image(s)",
|
||||
"operationId": "imageDelete",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": {
|
||||
"oneOf": [ {
|
||||
"type": "object",
|
||||
"required": [ "names" ],
|
||||
"properties": {
|
||||
"names": { "type": "array", "items": { "type": "string" }, "description": "Exact image name list to delete" },
|
||||
"writers_stopped": { "type": "boolean", "description": "Allow 'inverse' snapshot deletion optimisation" },
|
||||
"down_ok": { "type": "boolean", "description": "Continue deletion/merging even if some data will be left on unavailable OSDs" }
|
||||
}
|
||||
}, {
|
||||
"type": "object",
|
||||
"required": [ "matching" ],
|
||||
"properties": {
|
||||
"matching": { "type": "array", "items": { "type": "string" }, "description": "Glob image name patterns to delete" },
|
||||
"writers_stopped": { "type": "boolean", "description": "Allow 'inverse' snapshot deletion optimisation" },
|
||||
"down_ok": { "type": "boolean", "description": "Continue deletion/merging even if some data will be left on unavailable OSDs" }
|
||||
}
|
||||
}, {
|
||||
"type": "object",
|
||||
"required": [ "from" ],
|
||||
"properties": {
|
||||
"from": { "type": "string", "description": "Parent image in a parent-child sequence to delete" },
|
||||
"to": { "type": "string", "description": "Child image in a parent-child sequence to delete" },
|
||||
"writers_stopped": { "type": "boolean", "description": "Allow 'inverse' snapshot deletion optimisation" },
|
||||
"down_ok": { "type": "boolean", "description": "Continue deletion/merging even if some data will be left on unavailable OSDs" }
|
||||
}
|
||||
} ]
|
||||
} } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"deleted_ids": { "type": "array", "items": { "type": "integer", "format": "uint64" }, "description": "Full inode ID(s) of deleted images" },
|
||||
"deleted_images": { "type": "array", "items": { "type": "string" }, "description": "Deleted image name(s)" },
|
||||
"rebased_images": { "type": "array", "items": { "type": "string" }, "description": "Rebased child image name(s)" },
|
||||
"renamed_from": { "type": "string", "description": "Name of the renamed snapshot when 'inverse' deletion is enabled" },
|
||||
"renamed_to": { "type": "string", "description": "Name of the rename destination when 'inverse' deletion is enabled" }
|
||||
}
|
||||
} } } },
|
||||
"400": { "$ref": "#/components/responses/Invalid" },
|
||||
"409": { "$ref": "#/components/responses/UpdateConflict" },
|
||||
"412": { "$ref": "#/components/responses/PreconditionFailed" }
|
||||
}
|
||||
} },
|
||||
"/image/flatten": { "post": {
|
||||
"summary": "Flatten an image",
|
||||
"operationId": "imageFlatten",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"required": [ "names" ],
|
||||
"properties": {
|
||||
"image": { "type": "string", "description": "Image name to flatten" }
|
||||
}
|
||||
} } } },
|
||||
"responses": {
|
||||
"200": { "content": { "text/plain": { "schema": { "type": "string", "description": "Empty response" } } } },
|
||||
"400": { "$ref": "#/components/responses/Invalid" },
|
||||
"409": { "$ref": "#/components/responses/UpdateConflict" },
|
||||
"412": { "$ref": "#/components/responses/PreconditionFailed" }
|
||||
}
|
||||
} },
|
||||
"/osd/list": { "get": {
|
||||
"summary": "List OSDs",
|
||||
"operationId": "osdListGet",
|
||||
"produces": [ "application/json" ],
|
||||
"parameters": [ {
|
||||
"name": "params",
|
||||
"in": "query",
|
||||
"schema": { "$ref": "#/components/schemas/OsdListParams" },
|
||||
"style": "form",
|
||||
"explode": true
|
||||
} ],
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/OsdList" } } } }
|
||||
}
|
||||
}, "post": {
|
||||
"summary": "List OSDs",
|
||||
"operationId": "osdListPost",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/OsdListParams" } } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/OsdList" } } } }
|
||||
}
|
||||
} },
|
||||
"/osd/alloc": { "post": {
|
||||
"summary": "Allocate a new OSD number",
|
||||
"operationId": "osdAlloc",
|
||||
"produces": [ "application/json" ],
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": {
|
||||
"type": "integer",
|
||||
"format": "uint64",
|
||||
"description": "New OSD number"
|
||||
} } } },
|
||||
"400": { "$ref": "#/components/responses/Invalid" },
|
||||
"412": { "$ref": "#/components/responses/PreconditionFailed" }
|
||||
}
|
||||
} },
|
||||
"/osd/delete": { "post": {
|
||||
"summary": "Delete stopped OSD(s) from etcd",
|
||||
"operationId": "osdDelete",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"force": { "type": "boolean", "description": "Allow to remove non-empty OSDs" },
|
||||
"allow_data_loss": { "type": "boolean", "description": "Allow to remove non-empty OSDs even if it will lead to data loss" },
|
||||
"dry_run": { "type": "boolean", "description": "Only check if the removal is possible but do not delete" }
|
||||
}
|
||||
} } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"deleted_osds": {
|
||||
"type": "array",
|
||||
"items": { "type": "integer", "format": "uint64" },
|
||||
"description": "Deleted OSD numbers"
|
||||
}
|
||||
},
|
||||
"format": "uint64",
|
||||
"description": "New OSD number"
|
||||
} } } },
|
||||
"400": { "$ref": "#/components/responses/Invalid" },
|
||||
"409": { "$ref": "#/components/responses/UpdateConflict" },
|
||||
"412": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"pool_effects": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"pool_id": { "type": "integer", "format": "uint64", "description": "Affected pool ID" },
|
||||
"pool_name": { "type": "string", "description": "Affected pool name" },
|
||||
"effect": {
|
||||
"type": "string",
|
||||
"enum": [ "incomplete", "has_incomplete", "offline", "degraded" ],
|
||||
"description": "Possible pool PG state after removing OSD(s)"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} } } }
|
||||
}
|
||||
} },
|
||||
"/osd/modify": { "post": {
|
||||
"summary": "Modify OSD reweight, tags or noout flag",
|
||||
"operationId": "osdDelete",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"osd_num": { "type": "integer", "format": "uint64", "description": "OSD number to modify" },
|
||||
"reweight": { "type": "number", "minimum": 0, "maximum": 1, "description": "New OSD weight" },
|
||||
"tags": { "type": "array", "items": { "type": "string" }, "description": "New OSD tags" },
|
||||
"noout": { "type": "boolean", "description": "New value of the noout flag" }
|
||||
}
|
||||
} } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"reweight": { "type": "number", "minimum": 0, "maximum": 1, "description": "OSD weight" },
|
||||
"tags": { "type": "array", "items": { "type": "string" }, "description": "OSD tags" },
|
||||
"noout": { "type": "boolean", "description": "noout flag" }
|
||||
},
|
||||
"format": "uint64",
|
||||
"description": "New OSD number"
|
||||
} } } },
|
||||
"400": { "$ref": "#/components/responses/Invalid" },
|
||||
"409": { "$ref": "#/components/responses/UpdateConflict" },
|
||||
"412": { "$ref": "#/components/responses/PreconditionFailed" }
|
||||
}
|
||||
} },
|
||||
"/pool/list": { "get": {
|
||||
"summary": "List pools",
|
||||
"operationId": "poolListGet",
|
||||
"produces": [ "application/json" ],
|
||||
"parameters": [ {
|
||||
"name": "params",
|
||||
"in": "query",
|
||||
"schema": { "$ref": "#/components/schemas/PoolListParams" },
|
||||
"style": "form",
|
||||
"explode": true
|
||||
} ],
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/PoolList" } } } }
|
||||
}
|
||||
}, "post": {
|
||||
"summary": "List pools",
|
||||
"operationId": "poolListPost",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/PoolListParams" } } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/PoolList" } } } }
|
||||
}
|
||||
} },
|
||||
"/pool/create": { "post": {
|
||||
"summary": "Create a pool",
|
||||
"operationId": "poolCreate",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": {
|
||||
"allOf": [ {
|
||||
"$ref": "#/components/schemas/PoolConfig"
|
||||
}, { "type": "object", "properties": {
|
||||
"wait": { "type": "boolean", "description": "Wait for the new pool to come online" },
|
||||
"force": { "type": "boolean", "description": "Do not check that cluster has enough OSDs to create the pool" }
|
||||
} } ]
|
||||
} } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": {
|
||||
"$ref": "#/components/schemas/PoolConfig"
|
||||
} } } },
|
||||
"400": { "$ref": "#/components/responses/Invalid" },
|
||||
"409": { "$ref": "#/components/responses/UpdateConflict" },
|
||||
"412": { "$ref": "#/components/responses/PreconditionFailed" }
|
||||
}
|
||||
} },
|
||||
"/pool/modify": { "post": {
|
||||
"summary": "Modify a pool",
|
||||
"operationId": "poolModify",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": {
|
||||
"allOf": [ {
|
||||
"$ref": "#/components/schemas/PoolConfig"
|
||||
}, { "type": "object", "properties": {
|
||||
"id": { "type": "integer", "format": "uint64", "description": "Pool ID to modify" },
|
||||
"old_name": { "type": "string", "description": "Pool name to modify" },
|
||||
"force": { "type": "boolean", "description": "Allow modifications which may lead to data loss" }
|
||||
} } ]
|
||||
} } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": {
|
||||
"$ref": "#/components/schemas/PoolConfig"
|
||||
} } } },
|
||||
"400": { "$ref": "#/components/responses/Invalid" },
|
||||
"409": { "$ref": "#/components/responses/UpdateConflict" },
|
||||
"412": { "$ref": "#/components/responses/PreconditionFailed" }
|
||||
}
|
||||
} },
|
||||
"/pool/delete": { "post": {
|
||||
"summary": "Delete an pool",
|
||||
"operationId": "poolDelete",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"pool_id": { "type": "integer", "format": "uint64", "description": "Pool ID to delete" },
|
||||
"pool_name": { "type": "string", "description": "Pool name to delete" },
|
||||
"force": { "type": "boolean", "description": "Allow to delete non-empty pools" }
|
||||
}
|
||||
} } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": { "ok": { "type": "boolean" } }
|
||||
} } } },
|
||||
"400": { "$ref": "#/components/responses/Invalid" },
|
||||
"409": { "$ref": "#/components/responses/UpdateConflict" },
|
||||
"412": { "$ref": "#/components/responses/PreconditionFailed" }
|
||||
}
|
||||
} },
|
||||
"/pg/list": { "get": {
|
||||
"summary": "List PGs",
|
||||
"operationId": "pgListGet",
|
||||
"produces": [ "application/json" ],
|
||||
"parameters": [ {
|
||||
"name": "params",
|
||||
"in": "query",
|
||||
"schema": { "$ref": "#/components/schemas/PgListParams" },
|
||||
"style": "form",
|
||||
"explode": true
|
||||
} ],
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/PgList" } } } }
|
||||
}
|
||||
}, "post": {
|
||||
"summary": "List PGs",
|
||||
"operationId": "pgListPost",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/PgListParams" } } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/PgList" } } } }
|
||||
}
|
||||
} },
|
||||
"/data/delete": { "post": {
|
||||
"summary": "Remove inode data without changing metadata",
|
||||
"operationId": "dataDelete",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"pool": { "type": "integer", "format": "uint64", "description": "Pool ID" },
|
||||
"inode": { "type": "integer", "format": "uint64", "description": "Inode number" },
|
||||
"min_offset": { "type": "integer", "format": "uint64", "description": "Purge only data starting with specified offset" },
|
||||
"max_offset": { "type": "integer", "format": "uint64", "description": "Purge only data before specified offset" }
|
||||
}
|
||||
} } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"removed_objects": { "type": "integer", "format": "uint64", "description": "Removed object count" }
|
||||
}
|
||||
} } } },
|
||||
"400": { "$ref": "#/components/responses/Invalid" },
|
||||
"412": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"total_objects": { "type": "integer", "format": "uint64", "description": "Total object count" },
|
||||
"removed_objects": { "type": "integer", "format": "uint64", "description": "Removed object count" },
|
||||
"error_count": { "type": "integer", "format": "uint64", "description": "Error count" },
|
||||
"inactive_osds": {
|
||||
"type": "array",
|
||||
"items": { "type": "integer", "format": "uint64" },
|
||||
"description": "Inactive OSDs possibly with some non-removed data left"
|
||||
},
|
||||
"inactive_pgs": {
|
||||
"type": "array",
|
||||
"items": { "type": "integer", "format": "uint64" },
|
||||
"description": "Inactive PGs possibly with some non-removed data left"
|
||||
}
|
||||
}
|
||||
} } } }
|
||||
}
|
||||
} },
|
||||
"/data/describe": { "get": {
|
||||
"summary": "Describe unclean object locations in the cluster",
|
||||
"operationId": "dataDescribeGet",
|
||||
"produces": [ "application/json" ],
|
||||
"parameters": [ {
|
||||
"name": "params",
|
||||
"in": "query",
|
||||
"schema": { "$ref": "#/components/schemas/DataDescribeParams" },
|
||||
"style": "form",
|
||||
"explode": true
|
||||
} ],
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/DataDescribe" } } } }
|
||||
}
|
||||
}, "post": {
|
||||
"summary": "Describe unclean object locations in the cluster",
|
||||
"operationId": "dataDescribePost",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/DataDescribeParams" } } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": { "$ref": "#/components/schemas/DataDescribe" } } } }
|
||||
}
|
||||
} },
|
||||
"/data/fix": { "post": {
|
||||
"summary": "Fix inconsistent objects in the cluster by deleting some copies",
|
||||
"operationId": "dataDelete",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"objects": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"inode": { "type": "integer", "format": "uint64", "description": "Full inode ID with pool" },
|
||||
"stripe": { "type": "integer", "format": "uint64", "description": "Object offset within inode" }
|
||||
}
|
||||
}
|
||||
},
|
||||
"bad_osds": {
|
||||
"type": "array",
|
||||
"items": { "type": "integer", "format": "uint64" },
|
||||
"description": "Remove inconsistent copies/parts of objects from these OSDs"
|
||||
},
|
||||
"part": { "type": "integer", "minimum": 0, "maximum": 255, "description": "Only remove EC object part with this number" },
|
||||
"check": { "type": "boolean", "default": true, "description": "Do safety checks before removing parts" }
|
||||
}
|
||||
} } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"inode": { "type": "integer", "format": "uint64", "description": "Full inode ID with pool" },
|
||||
"stripe": { "type": "integer", "format": "uint64", "description": "Object offset within inode" },
|
||||
"part": { "type": "integer", "minimum": 0, "maximum": 255, "description": "Object part number" },
|
||||
"osd_num": { "type": "integer", "format": "uint64", "description": "OSD number with this part" }
|
||||
}
|
||||
} } } },
|
||||
"400": { "$ref": "#/components/responses/Invalid" }
|
||||
}
|
||||
} },
|
||||
"/data/merge": { "post": {
|
||||
"summary": "Merge layer data without changing metadata",
|
||||
"operationId": "dataMerge",
|
||||
"consumes": [ "application/json" ],
|
||||
"produces": [ "application/json" ],
|
||||
"requestBody": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"from": { "type": "string", "description": "First parent layer to merge" },
|
||||
"to": { "type": "string", "description": "Last child layer to merge" },
|
||||
"target": { "type": "string", "description": "Merge target (must be between from and to)" },
|
||||
"delete_source": { "type": "boolean", "description": "Delete source data after merging" },
|
||||
"fsync_interval": { "type": "integer", "format": "uint64", "description": "Fsync every this number of writes during merge" },
|
||||
"cas": { "type": "boolean", "description": "Use CAS writes to allow online merging" }
|
||||
}
|
||||
} } } },
|
||||
"responses": {
|
||||
"200": { "content": { "application/json": { "schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"from": { "type": "string", "description": "First merged parent layer" },
|
||||
"to": { "type": "string", "description": "Last merged child layer" },
|
||||
"target": { "type": "string", "description": "Merge target" }
|
||||
}
|
||||
} } } },
|
||||
"400": { "$ref": "#/components/responses/Invalid" },
|
||||
"412": { "$ref": "#/components/responses/PreconditionFailed" }
|
||||
}
|
||||
} }
|
||||
},
|
||||
"components": {
|
||||
"responses": {
|
||||
"Invalid": {
|
||||
"description": "Invalid Arguments (EINVAL)",
|
||||
"content": {
|
||||
"text/plain": {
|
||||
"schema": { "type": "string", "description": "Error text" }
|
||||
}
|
||||
}
|
||||
},
|
||||
"PreconditionFailed": {
|
||||
"description": "Precondition Failed (ENOTEMPTY, EEXIST or ENOENT)",
|
||||
"content": {
|
||||
"text/plain": {
|
||||
"schema": { "type": "string", "description": "Error text" }
|
||||
}
|
||||
}
|
||||
},
|
||||
"UpdateConflict": {
|
||||
"description": "Update Conflict",
|
||||
"content": {
|
||||
"text/plain": {
|
||||
"schema": { "type": "string", "description": "Error text" }
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"schemas": {
|
||||
"OpStat": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"bytes": { "type": "integer", "format": "uint64", "description": "Total bytes processed" },
|
||||
"count": { "type": "integer", "format": "uint64", "description": "Number of processed operations" },
|
||||
"usec": { "type": "integer", "format": "uint64", "description": "Total operation processing time in microseconds" },
|
||||
"bps": { "type": "integer", "format": "uint64", "description": "Bytes per second (average since previous report)" },
|
||||
"iops": { "type": "integer", "format": "uint64", "description": "Operations per second (average since previous report)" },
|
||||
"lat": { "type": "integer", "format": "uint64", "description": "Operation latency (average since previous report)" }
|
||||
}
|
||||
},
|
||||
"SubOpStat": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"count": { "type": "integer", "format": "uint64", "description": "Number of processed operations" },
|
||||
"usec": { "type": "integer", "format": "uint64", "description": "Total operation processing time in microseconds" },
|
||||
"iops": { "type": "integer", "format": "uint64", "description": "Operations per second (average since previous report)" },
|
||||
"lat": { "type": "integer", "format": "uint64", "description": "Operation latency (average since previous report)" }
|
||||
}
|
||||
},
|
||||
"RecoveryStat": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"bytes": { "type": "integer", "format": "uint64", "description": "Total bytes processed" },
|
||||
"count": { "type": "integer", "format": "uint64", "description": "Number of processed operations" },
|
||||
"bps": { "type": "integer", "format": "uint64", "description": "Bytes per second (average since previous report)" },
|
||||
"iops": { "type": "integer", "format": "uint64", "description": "Operations per second (average since previous report)" }
|
||||
}
|
||||
},
|
||||
"RecoveryStats": {
|
||||
"description": "Recovery statistics",
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"misplaced": { "$ref": "#/components/schemas/RecoveryStat" },
|
||||
"degraded": { "$ref": "#/components/schemas/RecoveryStat" }
|
||||
}
|
||||
},
|
||||
"OpStats": {
|
||||
"description": "Operation statistics (processing time without client RTT)",
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"read": { "$ref": "#/components/schemas/OpStat" },
|
||||
"write": { "$ref": "#/components/schemas/OpStat" },
|
||||
"write_stable": { "$ref": "#/components/schemas/OpStat" },
|
||||
"sync": { "$ref": "#/components/schemas/OpStat" },
|
||||
"stabilize": { "$ref": "#/components/schemas/OpStat" },
|
||||
"rollback": { "$ref": "#/components/schemas/OpStat" },
|
||||
"delete": { "$ref": "#/components/schemas/OpStat" },
|
||||
"list": { "$ref": "#/components/schemas/OpStat" },
|
||||
"show_config": { "$ref": "#/components/schemas/OpStat" },
|
||||
"primary_read": { "$ref": "#/components/schemas/OpStat" },
|
||||
"primary_write": { "$ref": "#/components/schemas/OpStat" },
|
||||
"primary_sync": { "$ref": "#/components/schemas/OpStat" },
|
||||
"primary_delete": { "$ref": "#/components/schemas/OpStat" },
|
||||
"ping": { "$ref": "#/components/schemas/OpStat" },
|
||||
"sec_read_bmp": { "$ref": "#/components/schemas/OpStat" },
|
||||
"scrub": { "$ref": "#/components/schemas/OpStat" },
|
||||
"describe": { "$ref": "#/components/schemas/OpStat" },
|
||||
"sec_lock": { "$ref": "#/components/schemas/OpStat" }
|
||||
}
|
||||
},
|
||||
"SubOpStats": {
|
||||
"description": "Suboperation statistics (processing+RTT to secondary as primary sees it)",
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"read": { "$ref": "#/components/schemas/SubOpStat" },
|
||||
"write": { "$ref": "#/components/schemas/SubOpStat" },
|
||||
"write_stable": { "$ref": "#/components/schemas/SubOpStat" },
|
||||
"sync": { "$ref": "#/components/schemas/SubOpStat" },
|
||||
"stabilize": { "$ref": "#/components/schemas/SubOpStat" },
|
||||
"rollback": { "$ref": "#/components/schemas/SubOpStat" },
|
||||
"delete": { "$ref": "#/components/schemas/SubOpStat" },
|
||||
"list": { "$ref": "#/components/schemas/SubOpStat" },
|
||||
"show_config": { "$ref": "#/components/schemas/SubOpStat" },
|
||||
"ping": { "$ref": "#/components/schemas/SubOpStat" },
|
||||
"sec_read_bmp": { "$ref": "#/components/schemas/SubOpStat" },
|
||||
"sec_lock": { "$ref": "#/components/schemas/SubOpStat" }
|
||||
}
|
||||
},
|
||||
"ImageListParams": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"globs": { "type": "array", "items": { "type": "string" }, "description": "Glob patterns or exact name matches" },
|
||||
"exact": { "type": "boolean", "description": "Treat globs as exact name matches, not as patterns" },
|
||||
"pool_id": { "type": "integer", "format": "uint64", "description": "Pool ID to list" },
|
||||
"pool_name": { "type": "string", "description": "Pool name to list" },
|
||||
"long": { "type": "boolean", "description": "Include allocated size and I/O statistics" },
|
||||
"sort": {
|
||||
"type": "string",
|
||||
"description": "Sort by specified field",
|
||||
"enum": [
|
||||
"name", "size", "used_size",
|
||||
"read_iops", "read_bps", "read_lat", "read_queue",
|
||||
"write_iops", "write_bps", "write_lat", "write_queue",
|
||||
"delete_iops", "delete_bps", "delete_lat", "delete_queue"
|
||||
]
|
||||
},
|
||||
"reverse": { "type": "boolean", "description": "Sort in descending order" },
|
||||
"count": { "type": "integer", "format": "uint64", "description": "Only list first N items" }
|
||||
}
|
||||
},
|
||||
"Image": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"inode_id": { "type": "integer", "format": "uint64", "description": "Full inode ID with pool" },
|
||||
"inode_num": { "type": "integer", "format": "uint64", "description": "Inode number without pool" },
|
||||
"name": { "type": "string", "description": "Image name" },
|
||||
"pool_id": { "type": "integer", "format": "uint64", "description": "Pool ID" },
|
||||
"pool_name": { "type": "string", "description": "Pool name" },
|
||||
"parent_name": { "type": "string", "description": "Name of the parent image (layer)" },
|
||||
"parent_inode_id": { "type": "integer", "format": "uint64", "description": "Full ID of the parent inode" },
|
||||
"parent_inode_num": { "type": "integer", "format": "uint64", "description": "Parent inode number without pool" },
|
||||
"parent_pool_id": { "type": "integer", "format": "uint64", "description": "Pool ID of the parent inode" },
|
||||
"size": { "type": "integer", "format": "uint64", "description": "Image size in bytes" },
|
||||
"used_size": { "type": "integer", "format": "uint64", "description": "Used space in bytes" },
|
||||
"readonly": { "type": "boolean", "description": "Readonly flag" },
|
||||
"deleted": { "type": "boolean", "description": "Deleted flag" }
|
||||
}
|
||||
},
|
||||
"ImageList": {
|
||||
"type": "array",
|
||||
"items": { "$ref": "#/components/schemas/Image" }
|
||||
},
|
||||
"OsdListParams": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"as_tree": { "type": "boolean", "description": "Include all placement tree nodes, not just OSDs" },
|
||||
"long": { "type": "boolean", "description": "Include per-OSD I/O statistics" }
|
||||
}
|
||||
},
|
||||
"OsdList": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"name": { "type": "string", "description": "OSD number" },
|
||||
"type": { "type": "string", "description": "Placement tree node type (osd for OSDs)" },
|
||||
"up": { "type": "boolean", "description": "Whether the OSD is active" },
|
||||
"parent": { "type": "string", "description": "Parent placement tree node name (host name, etc)" },
|
||||
"size": { "type": "integer", "format": "uint64", "description": "OSD size in bytes" },
|
||||
"free": { "type": "integer", "format": "uint64", "description": "Free space on the OSD in bytes" },
|
||||
"reweight": { "type": "number", "minimum": 0, "maximum": 1, "description": "OSD weight in data distribution" },
|
||||
"tags": { "type": "array", "items": { "type": "string" }, "description": "OSD tags" },
|
||||
"data_block_size": { "type": "integer", "format": "uint64" },
|
||||
"bitmap_granularity": { "type": "integer", "format": "uint64" },
|
||||
"immediate_commit": { "type": "string" },
|
||||
"noout": { "type": "boolean", "description": "noout flag" },
|
||||
"slow_ops_primary": { "type": "integer", "format": "uint64", "description": "Number of slow/hung client operations on this OSD" },
|
||||
"slow_ops_secondary": { "type": "integer", "format": "uint64", "description": "Number of slow/hung storage operations on this OSD" },
|
||||
"op_stats": { "$ref": "#/components/schemas/OpStats" },
|
||||
"subop_stats": { "$ref": "#/components/schemas/SubOpStats" },
|
||||
"recovery_stats": { "$ref": "#/components/schemas/RecoveryStats" }
|
||||
}
|
||||
}
|
||||
},
|
||||
"PoolListParams": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"globs": { "type": "array", "items": { "type": "string" }, "description": "Glob patterns to restrict listing" },
|
||||
"long": { "type": "boolean", "description": "Include per-pool I/O statistics" },
|
||||
"sort": { "type": "string", "description": "Sort by specified field" },
|
||||
"reverse": { "type": "boolean", "description": "Sort in descending order" },
|
||||
"count": { "type": "integer", "format": "uint64", "description": "Only list first N items" }
|
||||
}
|
||||
},
|
||||
"PoolConfig": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"name": { "type": "string", "description": "Pool name" },
|
||||
"used_for_app": { "type": "string", "description": "Application name pool is used for" },
|
||||
"scheme": { "type": "string", "enum": [ "replicated", "ec", "xor" ], "description": "Redundancy scheme" },
|
||||
"pg_count": { "type": "integer", "format": "uint64", "description": "Configured PG count" },
|
||||
"pg_size": { "type": "integer", "minimum": 1, "maximum": 255, "description": "Number of replicas or total data and parity EC chunks" },
|
||||
"pg_minsize": { "type": "integer", "minimum": 1, "maximum": 255, "description": "Minimum active OSD count in a PG to allow activity" },
|
||||
"parity_chunks": { "type": "integer", "minimum": 1, "maximum": 255, "description": "Parity chunk count for EC" },
|
||||
"local_reads": { "type": "string", "enum": [ "primary", "nearest", "random" ], "description": "Local read policy" },
|
||||
"root_node": { "type": "string", "description": "Root placement tree node for pool OSDs" },
|
||||
"pg_stripe_size": { "type": "integer", "format": "uint64", "description": "Stripe size to map to one PG in bytes" },
|
||||
"failure_domain": { "type": "string", "description": "Failure domain" },
|
||||
"osd_tags": { "type": "array", "items": { "type": "string" }, "description": "Put pool only on OSDs tagged with all specified tags" },
|
||||
"primary_affinity_tags": { "type": "array", "items": { "type": "string" }, "description": "Tags to prefer for primary OSDs" },
|
||||
"block_size": { "type": "integer", "format": "uint64", "description": "Put pool only on OSDs with this data block size" },
|
||||
"bitmap_granularity": { "type": "integer", "format": "uint64", "description": "Put pool only on OSDs with this logical sector size" },
|
||||
"immediate_commit": { "type": "string", "enum": [ "none", "small", "all" ], "description": "Put pool only on OSDs with this or larger immediate_commit" },
|
||||
"scrub_interval": { "type": "string", "pattern": "^\\d+[smhdMy]$", "description": "Automatic scrub interval" },
|
||||
"level_placement": { "type": "string", "description": "Additional failure domain rules" },
|
||||
"raw_placement": { "type": "string", "description": "Raw PG generation rules" },
|
||||
"max_osd_combinations": { "type": "integer", "format": "uint64", "description": "Maximum number of random combinations during PG generation" }
|
||||
}
|
||||
},
|
||||
"PoolList": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"allOf": [ {
|
||||
"$ref": "#/components/schemas/PoolConfig"
|
||||
}, { "type": "object", "properties": {
|
||||
"id": { "type": "integer", "format": "uint64", "description": "Pool ID" },
|
||||
"status": {
|
||||
"type": "string",
|
||||
"enum": [ "active", "has_misplaced", "has_degraded", "degraded", "has_incomplete", "incomplete", "inactive" ],
|
||||
"description": "Pool status (worst status of pool PGs)"
|
||||
},
|
||||
"real_pg_count": { "type": "integer", "format": "uint64", "description": "Currently active PG count" },
|
||||
"total_raw": { "type": "integer", "format": "uint64", "description": "Total raw space on all pool OSDs" },
|
||||
"used_raw": { "type": "integer", "format": "uint64", "description": "Raw OSD space used by the pool data" },
|
||||
"max_available": { "type": "integer", "format": "uint64", "description": "Maximum available user space in the pool" },
|
||||
"raw_to_usable": { "type": "number", "minimum": 1, "description": "Raw to user space ratio" },
|
||||
"space_efficiency": { "type": "number", "minimum": 0, "maximum": 1, "description": "Space efficiency (1 - wasted due to unbalanced PGs)" },
|
||||
"pg_real_size": { "type": "integer", "format": "uint64", "description": "Minimum of PG size and available failure domain count" },
|
||||
"osd_count": { "type": "integer", "format": "uint64", "description": "Number of OSDs used for the pool" },
|
||||
"backfillfull": { "type": "boolean", "description": "Backfillfull (is rebalance stopped due to low space)?" },
|
||||
"misplaced_count": { "type": "integer", "format": "uint64", "description": "Misplaced object count" },
|
||||
"degraded_count": { "type": "integer", "format": "uint64", "description": "Degraded object count" },
|
||||
"incomplete_count": { "type": "integer", "format": "uint64", "description": "Incomplete object count" },
|
||||
"object_count": { "type": "integer", "format": "uint64", "description": "Total object count" },
|
||||
"read_bps": { "type": "integer", "format": "uint64", "description": "Read bytes per second" },
|
||||
"read_iops": { "type": "integer", "format": "uint64", "description": "Read operations per second" },
|
||||
"read_lat": { "type": "integer", "format": "uint64", "description": "Average read latency" },
|
||||
"write_bps": { "type": "integer", "format": "uint64", "description": "Write bytes per second" },
|
||||
"write_iops": { "type": "integer", "format": "uint64", "description": "Write operations per second" },
|
||||
"write_lat": { "type": "integer", "format": "uint64", "description": "Average write latency" },
|
||||
"delete_bps": { "type": "integer", "format": "uint64", "description": "Delete bytes per second" },
|
||||
"delete_iops": { "type": "integer", "format": "uint64", "description": "Delete operations per second" },
|
||||
"delete_lat": { "type": "integer", "format": "uint64", "description": "Average delete latency" }
|
||||
} } ]
|
||||
}
|
||||
},
|
||||
"PgListParams": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"pool_id": { "type": "integer", "format": "uint64", "description": "Pool ID" },
|
||||
"pool_name": { "type": "string", "description": "Pool name" },
|
||||
"pg_state": {
|
||||
"type": "array",
|
||||
"items": { "type": "string" },
|
||||
"description": "PG state combinations (OR) separated by + (AND) and possibly with ! for negation"
|
||||
},
|
||||
"min": { "type": "integer", "format": "uint64", "description": "First PG number to show" },
|
||||
"max": { "type": "integer", "format": "uint64", "description": "Last PG number to show" },
|
||||
"osd": {
|
||||
"type": "array",
|
||||
"items": { "type": "integer", "format": "uint64" },
|
||||
"description": "Only return PGs with data on one of these OSDs"
|
||||
}
|
||||
}
|
||||
},
|
||||
"PgList": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"pool_id": { "type": "integer", "format": "uint64", "description": "Pool ID" },
|
||||
"pool_name": { "type": "string", "description": "Pool name" },
|
||||
"pg_num": { "type": "integer", "format": "uint64", "description": "PG number" },
|
||||
"target_set": {
|
||||
"type": "array",
|
||||
"items": { "type": "integer", "format": "uint64" },
|
||||
"description": "Target OSD set for this PG"
|
||||
},
|
||||
"target_primary": { "type": "integer", "format": "uint64", "description": "Target primary OSD of this PG (0 = no primary)" },
|
||||
"epoch": { "type": "integer", "format": "uint64", "description": "PG epoch number" },
|
||||
"next_scrub": { "type": "integer", "format": "uint64", "description": "UNIX timestamp of the next planned scrub" },
|
||||
"target_history": {
|
||||
"type": "array",
|
||||
"items": { "type": "array", "items": { "type": "integer", "format": "uint64" } },
|
||||
"description": "Other OSD sets of this PG which possibly still have some data left"
|
||||
},
|
||||
"pause": { "type": "boolean", "description": "Pause PG control flag from monitor" },
|
||||
"cur_primary": { "type": "integer", "format": "uint64", "description": "Current primary OSD of this PG (0 = no primary)" },
|
||||
"state": {
|
||||
"type": "array",
|
||||
"items": { "type": "string", "enum": [
|
||||
"starting", "peering", "incomplete", "active", "repeering", "stopping", "offline",
|
||||
"degraded", "has_inconsistent", "has_corrupted", "has_incomplete", "has_degraded",
|
||||
"has_misplaced", "has_unclean", "has_invalid", "left_on_dead", "scrubbing"
|
||||
] },
|
||||
"description": "PG state"
|
||||
},
|
||||
"all_peers": {
|
||||
"type": "array",
|
||||
"items": { "type": "array", "items": { "type": "integer", "format": "uint64" } },
|
||||
"description": "Other OSDs of this PG which possibly still have some data left"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"DataDescribeParams": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"pool_id": { "type": "integer", "format": "uint64", "description": "Only list objects from the given pool (by ID)" },
|
||||
"pool_name": { "type": "string", "description": "Only list objects from the given pool (by name)" },
|
||||
"pg": { "type": "integer", "format": "uint64", "description": "Only list objects in the given PG of the pool" },
|
||||
"inode": { "type": "integer", "format": "uint64", "description": "Single inode number to list" },
|
||||
"min_inode": { "type": "integer", "format": "uint64", "description": "Minimum inode number to list" },
|
||||
"max_inode": { "type": "integer", "format": "uint64", "description": "Maximum inode number to list" },
|
||||
"object_state": {
|
||||
"type": "array",
|
||||
"items": { "type": "string", "enum": [ "degraded", "misplaced", "incomplete", "corrupted", "inconsistent" ] },
|
||||
"description": "Only list objects in given state(s)"
|
||||
},
|
||||
"osd": {
|
||||
"type": "array",
|
||||
"items": { "type": "integer", "format": "uint64" },
|
||||
"description": "Only list objects from these primary OSD(s)"
|
||||
}
|
||||
}
|
||||
},
|
||||
"DataDescribe": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"inode": { "type": "integer", "format": "uint64", "description": "Full inode ID with pool" },
|
||||
"stripe": { "type": "integer", "format": "uint64", "description": "Object offset within inode" },
|
||||
"part": { "type": "integer", "minimum": 0, "maximum": 255, "description": "Object part number" },
|
||||
"osd_num": { "type": "integer", "format": "uint64", "description": "OSD number with this part" },
|
||||
"corrupted": { "type": "boolean", "description": "Part is corrupted flag" },
|
||||
"inconsistent": { "type": "boolean", "description": "Inconsistency flag (parts do not match)" },
|
||||
"outdated": { "type": "boolean", "description": "Part on this OSD is outdated" }
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -7,7 +7,7 @@ add_executable(vitastor-disk
|
||||
disk_tool.cpp disk_simple_offsets.cpp
|
||||
disk_tool_discard.cpp disk_tool_journal.cpp disk_tool_meta.cpp disk_tool_prepare.cpp disk_tool_resize.cpp
|
||||
disk_tool_resize_auto.cpp disk_tool_udev.cpp disk_tool_utils.cpp disk_tool_upgrade.cpp
|
||||
../util/crc32c.c ../util/xxhash.c ../util/str_util.cpp ../util/json_util.cpp ../../json11/json11.cpp ../util/rw_blocking.cpp ../util/allocator.cpp ../util/ringloop.cpp
|
||||
../util/crc32c.c ../util/str_util.cpp ../util/json_util.cpp ../../json11/json11.cpp ../util/rw_blocking.cpp ../util/allocator.cpp ../util/ringloop.cpp
|
||||
../blockstore/blockstore_disk.cpp ../blockstore/blockstore_heap.cpp ../blockstore/multilist.cpp
|
||||
)
|
||||
target_link_libraries(vitastor-disk
|
||||
|
||||
@@ -136,8 +136,8 @@ struct disk_tool_t
|
||||
void choose_journal_block(uint32_t je_size);
|
||||
int resize_rebuild_journal();
|
||||
int resize_write_new_journal();
|
||||
void remap_big_write(blockstore_heap_t *heap, heap_entry_t *wr);
|
||||
void remap_small_write(blockstore_heap_t *heap, heap_entry_t *wr);
|
||||
void remap_big_write(heap_entry_t *wr);
|
||||
void remap_small_write(heap_entry_t *wr);
|
||||
void fill_old_clean_entry(blockstore_heap_t *heap, heap_entry_t *big_wr);
|
||||
void fill_old_journal_entry(blockstore_heap_t *heap, heap_entry_t *wr);
|
||||
int resize_rebuild_meta();
|
||||
@@ -175,6 +175,7 @@ struct disk_tool_t
|
||||
void disk_tool_simple_offsets(json11::Json cfg, bool json_output);
|
||||
|
||||
uint64_t sscanf_json(const char *fmt, const json11::Json & str);
|
||||
void fromhexstr(const std::string & from, int bytes, uint8_t *to);
|
||||
int disable_cache(const std::string & dev);
|
||||
uint64_t get_atomic_write_size(const std::string & dev);
|
||||
uint64_t get_device_size(const std::string & dev, bool should_exist = false);
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
#include <unistd.h>
|
||||
|
||||
#include "disk_tool.h"
|
||||
#include "str_util.h"
|
||||
#include "malloc_or_die.h"
|
||||
|
||||
int disk_tool_t::dump_journal()
|
||||
|
||||
@@ -5,7 +5,6 @@
|
||||
#include "rw_blocking.h"
|
||||
#include "osd_id.h"
|
||||
#include "json_util.h"
|
||||
#include "str_util.h"
|
||||
#include "malloc_or_die.h"
|
||||
|
||||
int disk_tool_t::process_meta(std::function<void(blockstore_meta_header_v3_t *)> hdr_fn,
|
||||
@@ -110,7 +109,6 @@ close_error:
|
||||
r = heap->load_blocks(meta_pos-dsk.meta_block_size, read_len, data, true, entries_loaded);
|
||||
meta_pos += read_len;
|
||||
}
|
||||
heap->finish_load();
|
||||
heap->iterate_objects([&](heap_entry_t* obj, uint32_t meta_block_num)
|
||||
{
|
||||
obj_fn(heap, obj, meta_block_num);
|
||||
@@ -753,7 +751,7 @@ close_err0:
|
||||
{
|
||||
*wr->get_checksum(&heap) = sscanf_json("%jx", write_entry["data_crc32c"]);
|
||||
}
|
||||
wr->checksum = wr->calc_checksum(&heap);
|
||||
wr->crc32c = wr->calc_crc32c();
|
||||
assert((uint8_t*)wr + wr->size == new_meta_buf + meta_offset + used_space);
|
||||
}
|
||||
}
|
||||
@@ -796,7 +794,7 @@ close_err:
|
||||
fromhexstr(meta_entry["ext_bitmap"].string_value(), new_clean_entry_bitmap_size, wr->get_ext_bitmap(&heap));
|
||||
if (new_meta_hdr->data_csum_type != 0)
|
||||
fromhexstr(meta_entry["data_csum"].string_value(), new_data_csum_size, wr->get_checksums(&heap));
|
||||
wr->checksum = wr->calc_checksum(&heap);
|
||||
wr->crc32c = wr->calc_crc32c();
|
||||
assert((uint8_t*)wr + wr->size == new_meta_buf + meta_offset + used_space);
|
||||
auto j_it = journal_by_object.find(oid);
|
||||
if (j_it != journal_by_object.end())
|
||||
@@ -862,7 +860,7 @@ close_err:
|
||||
assert(0);
|
||||
}
|
||||
wr->size = wr->get_size(&heap);
|
||||
wr->checksum = wr->calc_checksum(&heap);
|
||||
wr->crc32c = wr->calc_crc32c();
|
||||
assert((uint8_t*)wr + wr->size == new_meta_buf + meta_offset + used_space);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -531,7 +531,7 @@ int disk_tool_t::resize_write_new_journal()
|
||||
return 0;
|
||||
}
|
||||
|
||||
void disk_tool_t::remap_big_write(blockstore_heap_t *heap, heap_entry_t *wr)
|
||||
void disk_tool_t::remap_big_write(heap_entry_t *wr)
|
||||
{
|
||||
uint64_t block_num = wr->big().block_num;
|
||||
auto remap_it = data_remap.find(block_num);
|
||||
@@ -544,10 +544,10 @@ void disk_tool_t::remap_big_write(blockstore_heap_t *heap, heap_entry_t *wr)
|
||||
}
|
||||
block_num += data_idx_diff;
|
||||
wr->big().block_num = block_num;
|
||||
wr->checksum = wr->calc_checksum(heap);
|
||||
wr->crc32c = wr->calc_crc32c();
|
||||
}
|
||||
|
||||
void disk_tool_t::remap_small_write(blockstore_heap_t *heap, heap_entry_t *wr)
|
||||
void disk_tool_t::remap_small_write(heap_entry_t *wr)
|
||||
{
|
||||
if (new_meta_format == BLOCKSTORE_META_FORMAT_HEAP && wr->small().len > 0)
|
||||
{
|
||||
@@ -559,7 +559,7 @@ void disk_tool_t::remap_small_write(blockstore_heap_t *heap, heap_entry_t *wr)
|
||||
memcpy(new_journal_ptr, buffer_area+wr->small().location, wr->small().len);
|
||||
wr->small().location = new_journal_ptr-new_journal_buf;
|
||||
new_journal_ptr += wr->small().len;
|
||||
wr->checksum = wr->calc_checksum(heap);
|
||||
wr->crc32c = wr->calc_crc32c();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -673,11 +673,11 @@ int disk_tool_t::resize_rebuild_meta()
|
||||
{
|
||||
if (wr->type() == BS_HEAP_BIG_WRITE || wr->type() == BS_HEAP_BIG_INTENT)
|
||||
{
|
||||
remap_big_write(heap, wr);
|
||||
remap_big_write(wr);
|
||||
}
|
||||
else if (wr->type() == BS_HEAP_SMALL_WRITE)
|
||||
{
|
||||
remap_small_write(heap, wr);
|
||||
remap_small_write(wr);
|
||||
}
|
||||
else if (wr->type() != BS_HEAP_DELETE && new_meta_format != BLOCKSTORE_META_FORMAT_HEAP)
|
||||
{
|
||||
@@ -786,7 +786,7 @@ int disk_tool_t::resize_rebuild_meta()
|
||||
memcpy(((uint8_t*)wr) + sizeof(heap_big_write_t) + new_clean_entry_bitmap_size, bitmap+new_clean_entry_bitmap_size, new_clean_entry_bitmap_size);
|
||||
memcpy(((uint8_t*)wr) + sizeof(heap_big_write_t) + 2*new_clean_entry_bitmap_size, bitmap+2*new_clean_entry_bitmap_size, new_data_csum_size);
|
||||
}
|
||||
wr->checksum = wr->calc_checksum(&dsk);
|
||||
wr->crc32c = wr->calc_crc32c();
|
||||
new_meta_pos += wr->size;
|
||||
}
|
||||
else
|
||||
|
||||
@@ -22,6 +22,28 @@ uint64_t sscanf_json(const char *fmt, const json11::Json & str)
|
||||
return value;
|
||||
}
|
||||
|
||||
static int fromhex(char c)
|
||||
{
|
||||
if (c >= '0' && c <= '9')
|
||||
return (c-'0');
|
||||
else if (c >= 'a' && c <= 'f')
|
||||
return (c-'a'+10);
|
||||
else if (c >= 'A' && c <= 'F')
|
||||
return (c-'A'+10);
|
||||
return -1;
|
||||
}
|
||||
|
||||
void fromhexstr(const std::string & from, int bytes, uint8_t *to)
|
||||
{
|
||||
for (int i = 0; i < from.size() && i < bytes; i++)
|
||||
{
|
||||
int x = fromhex(from[2*i]), y = fromhex(from[2*i+1]);
|
||||
if (x < 0 || y < 0)
|
||||
break;
|
||||
to[i] = x*16 + y;
|
||||
}
|
||||
}
|
||||
|
||||
// returns 1 = check error, 0 = write through, -1 = write back
|
||||
// (similar to 1 = warning, -1 = error, 0 = success in disable_cache)
|
||||
static int check_queue_cache(std::string dev, std::string parent_dev)
|
||||
|
||||
@@ -15,7 +15,6 @@ target_link_libraries(vitastor-osd
|
||||
${ISAL_LIBRARIES}
|
||||
${IBVERBS_LIBRARIES}
|
||||
${RDMACM_LIBRARIES}
|
||||
${OPENSSL_LIBRARIES}
|
||||
)
|
||||
|
||||
# osd_rmw_test
|
||||
|
||||
@@ -809,17 +809,21 @@ again:
|
||||
goto again;
|
||||
}
|
||||
auto & pool_cfg = pool_it->second;
|
||||
bool done = bs->reshard_continue(pool_cfg.reshard_state, pg_reshard_chunk_size);
|
||||
if (done &&
|
||||
(pool_cfg.real_pg_count != pool_cfg.applied_pg_count ||
|
||||
pool_cfg.pg_stripe_size != pool_cfg.applied_pg_stripe_size))
|
||||
bool done = false;
|
||||
if (pool_cfg.real_pg_count != pool_cfg.applied_pg_count ||
|
||||
pool_cfg.pg_stripe_size != pool_cfg.applied_pg_stripe_size)
|
||||
{
|
||||
// PG count changed again, reshard again
|
||||
bs->reshard_abort(pool_cfg.reshard_state);
|
||||
pool_cfg.applied_pg_count = pool_cfg.real_pg_count;
|
||||
pool_cfg.applied_pg_stripe_size = pool_cfg.pg_stripe_size;
|
||||
pool_cfg.reshard_state = bs->reshard_start(pool_id, pool_cfg.real_pg_count, pool_cfg.pg_stripe_size, pg_reshard_chunk_size);
|
||||
done = !pool_cfg.reshard_state;
|
||||
}
|
||||
else
|
||||
{
|
||||
done = bs->reshard_continue(pool_cfg.reshard_state, pg_reshard_chunk_size);
|
||||
}
|
||||
if (done)
|
||||
{
|
||||
// Pool is resharded
|
||||
|
||||
@@ -334,44 +334,36 @@ pg_osd_set_state_t* pg_t::add_object_to_state(const object_id oid, const uint64_
|
||||
if (it == state_dict.end())
|
||||
{
|
||||
std::vector<osd_num_t> read_target;
|
||||
bool found = false;
|
||||
uint32_t bad_mask = (LOC_OUTDATED | LOC_CORRUPTED);
|
||||
retry:
|
||||
if (scheme == POOL_SCHEME_REPLICATED)
|
||||
{
|
||||
for (auto & o: osd_set)
|
||||
{
|
||||
if (!(o.loc_bad & bad_mask))
|
||||
if (!(o.loc_bad & (LOC_OUTDATED | LOC_CORRUPTED)))
|
||||
{
|
||||
read_target.push_back(o.osd_num);
|
||||
found = true;
|
||||
}
|
||||
}
|
||||
if (read_target.size() < pg_size)
|
||||
while (read_target.size() < pg_size)
|
||||
{
|
||||
// FIXME: This is because we then use .data() and assume it's at least <pg_size> long
|
||||
read_target.resize(pg_size);
|
||||
read_target.push_back(0);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
read_target.resize(pg_size);
|
||||
for (int i = 0; i < pg_size; i++)
|
||||
{
|
||||
read_target[i] = 0;
|
||||
}
|
||||
for (auto & o: osd_set)
|
||||
{
|
||||
if (!(o.loc_bad & bad_mask))
|
||||
if (!(o.loc_bad & (LOC_OUTDATED | LOC_CORRUPTED)))
|
||||
{
|
||||
read_target[o.role] = o.osd_num;
|
||||
found = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!found && (bad_mask & LOC_CORRUPTED))
|
||||
{
|
||||
// Allow to try reading corrupted copies in rare cases when the object is corrupted on all OSDs
|
||||
bad_mask = LOC_OUTDATED;
|
||||
read_target.clear();
|
||||
goto retry;
|
||||
}
|
||||
state_dict[osd_set] = {
|
||||
.read_target = read_target,
|
||||
.osd_set = osd_set,
|
||||
|
||||
+7
-13
@@ -271,6 +271,7 @@ resume_0:
|
||||
}
|
||||
cur_op->buf = alloc_read_buffer(op_data->stripes, pg ? pg->pg_data_size : 1, 0);
|
||||
submit_primary_subops(SUBMIT_RMW_READ, op_data->target_ver, op_data->prev_set, cur_op);
|
||||
op_data->st = 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -283,14 +284,11 @@ resume_0:
|
||||
op_data->degraded = 1;
|
||||
cur_op->buf = alloc_read_buffer(op_data->stripes, pg->pg_size, 0);
|
||||
submit_primary_subops(SUBMIT_RMW_READ, op_data->target_ver, op_data->prev_set, cur_op);
|
||||
op_data->st = 1;
|
||||
}
|
||||
}
|
||||
resume_1:
|
||||
if (op_data->n_subops > 0)
|
||||
{
|
||||
op_data->st = 1;
|
||||
return;
|
||||
}
|
||||
return;
|
||||
resume_2:
|
||||
if (op_data->errors > 0)
|
||||
{
|
||||
@@ -298,9 +296,8 @@ resume_2:
|
||||
{
|
||||
// I/O or checksum error
|
||||
// FIXME: ref = true ideally... because new_state != state is not necessarily true if it's freed and recreated
|
||||
auto new_object_state = mark_object_corrupted(*pg, op_data->oid, op_data->object_state, op_data->stripes, false);
|
||||
if (new_object_state != op_data->object_state)
|
||||
goto resume_0;
|
||||
op_data->object_state = mark_object_corrupted(*pg, op_data->oid, op_data->object_state, op_data->stripes, false);
|
||||
goto resume_0;
|
||||
}
|
||||
finish_op(cur_op, op_data->errcode);
|
||||
return;
|
||||
@@ -739,11 +736,8 @@ resume_1:
|
||||
submit_primary_subops(SUBMIT_RMW_READ, UINT64_MAX, op_data->prev_set, cur_op);
|
||||
op_data->prev_set = NULL;
|
||||
resume_2:
|
||||
if (op_data->n_subops > 0)
|
||||
{
|
||||
op_data->st = 2;
|
||||
return;
|
||||
}
|
||||
op_data->st = 2;
|
||||
return;
|
||||
resume_3:
|
||||
if (op_data->errors > 0)
|
||||
{
|
||||
|
||||
@@ -134,17 +134,11 @@ void osd_t::submit_primary_subops(int submit_type, uint64_t op_version, const ui
|
||||
n_subops = 1;
|
||||
else
|
||||
zero_read = -1;
|
||||
osd_op_t *subops = new osd_op_t[n_subops];
|
||||
op_data->fact_ver = 0;
|
||||
op_data->done = op_data->errors = op_data->drops = op_data->errcode = 0;
|
||||
op_data->n_subops = n_subops;
|
||||
if (!n_subops)
|
||||
{
|
||||
op_data->errcode = -EIO;
|
||||
op_data->subops = NULL;
|
||||
op_data->errors = 1;
|
||||
return;
|
||||
}
|
||||
op_data->subops = new osd_op_t[n_subops];
|
||||
op_data->subops = subops;
|
||||
int sent = submit_primary_subop_batch(submit_type, op_data->oid.inode, op_version, op_data->stripes, osd_set, cur_op, 0, zero_read);
|
||||
assert(sent == n_subops);
|
||||
}
|
||||
|
||||
@@ -108,31 +108,57 @@ retry_1:
|
||||
}
|
||||
}
|
||||
// Read required blocks
|
||||
submit_primary_subops(SUBMIT_RMW_READ, UINT64_MAX, op_data->prev_set, cur_op);
|
||||
resume_2:
|
||||
if (op_data->n_subops > 0)
|
||||
{
|
||||
op_data->st = 2;
|
||||
return;
|
||||
if (op_data->object_state && (op_data->object_state->state & OBJ_INCOMPLETE))
|
||||
{
|
||||
// Allow to read version number (just version number!) from corrupted chunks
|
||||
// to allow full overwrite of a corrupted object
|
||||
bool found = false;
|
||||
for (int role = 0; role < pg.pg_size; role++)
|
||||
{
|
||||
if (op_data->prev_set[role] != 0 || op_data->stripes[role].read_end > op_data->stripes[role].read_start)
|
||||
{
|
||||
found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!found)
|
||||
{
|
||||
osd_num_t corrupted_target[pg.pg_size];
|
||||
for (int role = 0; role < pg.pg_size; role++)
|
||||
{
|
||||
corrupted_target[role] = 0;
|
||||
}
|
||||
for (auto & loc: op_data->object_state->osd_set)
|
||||
{
|
||||
if (!(loc.loc_bad & LOC_OUTDATED) && !corrupted_target[loc.role])
|
||||
{
|
||||
corrupted_target[loc.role] = loc.osd_num;
|
||||
}
|
||||
}
|
||||
submit_primary_subops(SUBMIT_RMW_READ, UINT64_MAX, corrupted_target, cur_op);
|
||||
goto resume_2;
|
||||
}
|
||||
}
|
||||
submit_primary_subops(SUBMIT_RMW_READ, UINT64_MAX, op_data->prev_set, cur_op);
|
||||
}
|
||||
resume_2:
|
||||
op_data->st = 2;
|
||||
return;
|
||||
resume_3:
|
||||
if (op_data->errors > 0)
|
||||
{
|
||||
if (op_data->errcode == -EIO || op_data->errcode == -EDOM)
|
||||
{
|
||||
// Mark object corrupted and retry
|
||||
pg_osd_set_state_t *new_object_state = mark_object_corrupted(pg, op_data->oid, op_data->object_state, op_data->stripes, true);
|
||||
if (new_object_state != op_data->object_state)
|
||||
op_data->object_state = mark_object_corrupted(pg, op_data->oid, op_data->object_state, op_data->stripes, true);
|
||||
op_data->prev_set = op_data->object_state ? op_data->object_state->read_target.data() : pg.cur_set.data();
|
||||
if (cur_op->rmw_buf)
|
||||
{
|
||||
op_data->object_state = new_object_state;
|
||||
op_data->prev_set = op_data->object_state ? op_data->object_state->read_target.data() : pg.cur_set.data();
|
||||
if (cur_op->rmw_buf)
|
||||
{
|
||||
free(cur_op->rmw_buf);
|
||||
cur_op->rmw_buf = NULL;
|
||||
}
|
||||
goto retry_1;
|
||||
free(cur_op->rmw_buf);
|
||||
cur_op->rmw_buf = NULL;
|
||||
}
|
||||
goto retry_1;
|
||||
}
|
||||
deref_object_state(pg, &op_data->object_state, true);
|
||||
pg_cancel_write_queue(pg, cur_op, op_data->oid, op_data->errcode);
|
||||
|
||||
@@ -41,7 +41,6 @@ add_executable(test_heap
|
||||
../blockstore/multilist.cpp
|
||||
../blockstore/blockstore_heap.cpp
|
||||
../util/crc32c.c
|
||||
../util/xxhash.c
|
||||
../util/allocator.cpp
|
||||
../blockstore/blockstore_disk.cpp
|
||||
../util/str_util.cpp
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user