Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
538620b400 | ||
|
|
4c34a47179 | ||
|
|
df16ab627a | ||
|
|
44c895dc30 | ||
|
|
fdea595913 | ||
|
|
ef4c91ecc8 | ||
|
|
5192c6cf50 | ||
|
|
cb639a130d | ||
|
|
892e3a8b6d | ||
|
|
0bd9c26620 | ||
|
|
ae2b1f7802 | ||
|
|
883b2e45da | ||
|
|
09df74cfac | ||
|
|
2ef3f012c9 | ||
|
|
7175d99c64 | ||
|
|
230a26772e | ||
|
|
637684c579 | ||
|
|
c541dd422f | ||
|
|
22c39561cf | ||
|
|
d4c67b4879 | ||
|
|
d1b7167861 | ||
|
|
33b561b73a | ||
|
|
a95a60c600 | ||
|
|
b3bc815354 | ||
|
|
42fc45d6da | ||
|
|
3fbe2e7f8a | ||
|
|
310c512b43 | ||
|
|
fd3e3b4ef0 | ||
|
|
93cf69f89f | ||
|
|
a8dd8bf06c | ||
|
|
7e23b57014 | ||
|
|
b2427836a1 | ||
|
|
2ed6760447 | ||
|
|
a1d215ea2f | ||
|
|
4fa442a5de | ||
|
|
3a057ed5af | ||
|
|
0ad042b28b | ||
|
|
4d75c6c8f9 | ||
|
|
9ec18f0aa1 | ||
|
|
facaa2cc6a | ||
|
|
d2ac8e3827 | ||
|
|
d6dacc67db | ||
|
|
cbe51595cb | ||
|
|
6bd0830ab8 | ||
|
|
796e82f34d | ||
|
|
db39283970 | ||
|
|
9f8c686321 | ||
|
|
2618e559d1 | ||
|
|
037d2bc162 | ||
|
|
f0b64adb32 | ||
|
|
f06d64d879 | ||
|
|
d164499a1c | ||
|
|
725e9aa8ae | ||
|
|
0715feffa1 | ||
|
|
e9f37e8dc3 | ||
|
|
4fbe4b5654 | ||
|
|
9fb645693b | ||
|
|
9e47828383 | ||
|
|
ac2ce48cb2 | ||
|
|
9cc2beed95 | ||
|
|
fb1c870f5c | ||
|
|
2d616d8058 | ||
|
|
3f7f6f442b | ||
|
|
7e7b95eeb4 | ||
|
|
dd588a0783 | ||
|
|
028a6cab68 | ||
|
|
d75334ddf0 | ||
|
|
bf0875128e | ||
|
|
9a6a7b7f75 | ||
|
|
c4c17ee6fb | ||
|
|
2b801a7ffa | ||
|
|
233d2b2a09 | ||
|
|
8380d4c6a6 | ||
|
|
73f9c7293f | ||
|
|
1c66c3e5ba | ||
|
|
eddfa93c18 | ||
|
|
ddd755a0e6 | ||
|
|
819f5b7ec9 | ||
|
|
34d0a6d9b1 | ||
|
|
fe83825ead | ||
|
|
8ec7faa675 | ||
|
|
21cf5c8815 | ||
|
|
44eeb1ed13 | ||
|
|
cc6c445cf0 | ||
|
|
bd6af0db09 | ||
|
|
8860101e99 | ||
|
|
c92661b364 | ||
|
|
9a02a592e3 | ||
|
|
8b7fa3d3bc | ||
|
|
db5eaa2eee | ||
|
|
d35727dbb7 | ||
|
|
b78f526696 | ||
|
|
a23df12260 | ||
|
|
166e16102e | ||
|
|
06c602110c | ||
|
|
1de68c30af | ||
|
|
2a0aca6e94 | ||
|
|
e808332e12 | ||
|
|
fc5a183959 | ||
|
|
55de37e58a | ||
|
|
aacfdf0dec | ||
|
|
a67c415e0f | ||
|
|
495f3fb4cd | ||
|
|
934752d617 | ||
|
|
1e6e233426 | ||
|
|
a5768a8ef6 | ||
|
|
216707f101 | ||
|
|
8ffdb93ed3 | ||
|
|
2fb7022c78 | ||
|
|
2633978fec | ||
|
|
615d4825c0 | ||
|
|
e55ca26ff6 | ||
|
|
abe093b9a3 | ||
|
|
07e6eb0b16 | ||
|
|
0eacce1e1d | ||
|
|
3df410acc7 | ||
|
|
e55927076c | ||
|
|
60fcb168fe | ||
|
|
b17691ba02 | ||
|
|
f2cbe793e2 | ||
|
|
0053546f8b | ||
|
|
959f792f82 | ||
|
|
eaff4509ca | ||
|
|
04eefce30b | ||
|
|
746844d301 | ||
|
|
c8f5b6cb19 | ||
|
|
0f4837e9bb | ||
|
|
25ce82a729 | ||
|
|
944499135f | ||
|
|
08f21edaf7 | ||
|
|
e9f0639e62 | ||
|
|
4e0b203552 | ||
|
|
b772a3cd04 | ||
|
|
856ad79a02 | ||
|
|
39a8772d7f | ||
|
|
993f40de37 | ||
|
|
5607222921 | ||
|
|
378fff6f67 | ||
|
|
daf2cc3fb1 | ||
|
|
17d61c5868 | ||
|
|
e709657de4 | ||
|
|
3eecf9048c | ||
|
|
1852caaeec | ||
|
|
7b40561141 | ||
|
|
d80c12ced2 | ||
|
|
b9713deecd | ||
|
|
852734270e | ||
|
|
8f92979a18 | ||
|
|
b720af74c2 | ||
|
|
96df2966cc | ||
|
|
d3b171e047 | ||
|
|
7287b7fc25 | ||
|
|
e1bb670491 | ||
|
|
1e5a01def8 | ||
|
|
371e630f52 | ||
|
|
0a7ae616f3 | ||
|
|
82f5fb7edd | ||
|
|
ec8527c89d | ||
|
|
5afef7ca6d | ||
|
|
e512e1eeb1 | ||
|
|
7ab60c00ab | ||
|
|
ac8e0ef231 | ||
|
|
94f3634602 | ||
|
|
2b8a9e3f90 | ||
|
|
ef792608b0 | ||
|
|
04531bcfbb | ||
|
|
1b40fa1cee | ||
|
|
3ea9230ed0 | ||
|
|
97dfbfad75 | ||
|
|
c148f97ee4 | ||
|
|
b1f61eb5c8 | ||
|
|
9db8748647 | ||
|
|
e747319c1e | ||
|
|
0cb0e31ceb | ||
|
|
0703efd8b9 | ||
|
|
4a0720a231 | ||
|
|
57d83ecf7c | ||
|
|
d4f7bbb412 | ||
|
|
e5e71fc21a | ||
|
|
0ec4f1608a | ||
|
|
232e416658 | ||
|
|
23d5fec580 | ||
|
|
16881a3d6b | ||
|
|
7cde5c75b9 | ||
|
|
4c32244409 | ||
|
|
30c5a79772 | ||
|
|
6e856c2719 | ||
|
|
af710d3c65 | ||
|
|
8189c3a4ef | ||
|
|
552ba5d885 | ||
|
|
14bbf18ede | ||
|
|
c3eaaa4b94 | ||
|
|
8b35c09e12 | ||
|
|
74b8ea1303 | ||
|
|
6a30e6653a | ||
|
|
2d7da127ad | ||
|
|
4143d56db7 | ||
|
|
5067554e46 | ||
|
|
bc48eb1ff8 | ||
|
|
b0809b33aa | ||
|
|
d61cf2303f | ||
|
|
712f22d6f8 | ||
|
|
4340082315 | ||
|
|
a1a449686a | ||
|
|
fb87870734 | ||
|
|
99bddb976a | ||
|
|
4f997791b8 | ||
|
|
c980168d10 | ||
|
|
dce4a6c37a | ||
|
|
38d5175f66 | ||
|
|
7569fb959b | ||
|
|
aa1e62a502 | ||
|
|
dcbdb0ae33 | ||
|
|
6e5f990801 | ||
|
|
845e76a0ec | ||
|
|
a98aee7906 | ||
|
|
064a94166c | ||
|
|
38112e9012 | ||
|
|
353460cc83 | ||
|
|
92631bb6b3 | ||
|
|
ce050e7eda | ||
|
|
556fc9a876 | ||
|
|
274f9ecda5 | ||
|
|
e0d2705294 | ||
|
|
57d2f30303 | ||
|
|
7f4c541a6f | ||
|
|
b0b495a991 | ||
|
|
da8bf2b73b | ||
|
|
615e9d1274 | ||
|
|
ec5e93307d | ||
|
|
b599334c4a | ||
|
|
bcde273ca1 | ||
|
|
87fe1bc00f | ||
|
|
d4c465f786 | ||
|
|
62995243f3 | ||
|
|
7ea4884ef6 | ||
|
|
8823ddf48e | ||
|
|
db14037ac8 | ||
|
|
64db505357 | ||
|
|
8992eb57df | ||
|
|
f3048d0858 | ||
|
|
714b0783bf | ||
|
|
e8e2aa5dba | ||
|
|
9063bcaa41 | ||
|
|
c392e914e2 | ||
|
|
249e04ac0d | ||
|
|
16dee7c136 | ||
|
|
adddf9b3b1 | ||
|
|
6a5044ae36 | ||
|
|
49407afa17 | ||
|
|
a00fc1bc24 | ||
|
|
3a96d41c93 | ||
|
|
2e4d5ae5bb | ||
|
|
db458fc999 | ||
|
|
574520be0f | ||
|
|
715e7df51f | ||
|
|
ea5ee0c46f | ||
|
|
47bec8af47 | ||
|
|
ec3bf4ae6c | ||
|
|
cb45b1865d | ||
|
|
f656545f4a | ||
|
|
0241a61412 | ||
|
|
0bc81f5320 | ||
|
|
84961f6d0a | ||
|
|
cb085f9c8f | ||
|
|
a9773b1908 | ||
|
|
11783a2d7a | ||
|
|
7915605609 | ||
|
|
3ba3eed0cf | ||
|
|
5dc0b42146 | ||
|
|
b44c3a7971 | ||
|
|
59b7b2e0c3 | ||
|
|
7f0c78113b | ||
|
|
20bbeb4095 | ||
|
|
3c687a2993 | ||
|
|
7530bdbec7 | ||
|
|
c78b4d184d | ||
|
|
c7a6b77c21 | ||
|
|
3cb7ec69bc | ||
|
|
954e7b658d | ||
|
|
2c6ea8a521 | ||
|
|
bfd5575425 | ||
|
|
85e61c9c31 | ||
|
|
f580cee936 | ||
|
|
6d0460500a | ||
|
|
01ae800d34 | ||
|
|
dbb885a6b1 | ||
|
|
8ff2c268f7 | ||
|
|
8430104c19 | ||
|
|
4c11e3ad3d | ||
|
|
d88b49872b | ||
|
|
ca27b91919 | ||
|
|
94f31b96b8 | ||
|
|
76c7c26d32 | ||
|
|
8aa2c49202 | ||
|
|
0f330b10f1 | ||
|
|
477b54a0d8 | ||
|
|
5823a7de66 | ||
|
|
eb0deaa3f5 | ||
|
|
59e6527303 | ||
|
|
67ba9f9b7c | ||
|
|
3ad83e8d13 | ||
|
|
5d3f3f47a7 | ||
|
|
8ee7058ec8 | ||
|
|
1badc6ad13 | ||
|
|
be1858848e | ||
|
|
d75b1cb2d2 | ||
|
|
a1c17d90a3 | ||
|
|
f0112050ce | ||
|
|
d6b8d921d6 | ||
|
|
65872f5d0e | ||
|
|
15eef27d44 | ||
|
|
aa1e51de5f | ||
|
|
c164adb43c | ||
|
|
622631c146 | ||
|
|
4ab93d8481 | ||
|
|
bd64770317 | ||
|
|
34cb48d553 | ||
|
|
724d2ffa04 | ||
|
|
b6bfe1435d | ||
|
|
74a23dcb63 | ||
|
|
93fd23b2bb | ||
|
|
eedc700b83 | ||
|
|
c8cc17dbe9 | ||
|
|
b55d406386 | ||
|
|
0c46dbd333 | ||
|
|
ed94aa52cf | ||
|
|
555ae613c2 | ||
|
|
cad6ea0360 | ||
|
|
d60709dce1 | ||
|
|
a03ffd0d73 | ||
|
|
89b76a87b6 | ||
|
|
ceba343ac0 | ||
|
|
3bc04d8250 | ||
|
|
d228fbfb68 | ||
|
|
e3c8fd28b4 | ||
|
|
d87e7d1a37 | ||
|
|
59f87c3e30 | ||
|
|
eba383f66f | ||
|
|
4e5e8822c0 | ||
|
|
60933c1d00 | ||
|
|
1ad6933953 | ||
|
|
8a250f4fca | ||
|
|
94ddf20667 | ||
|
|
5f18496c04 | ||
|
|
08a3dcd587 | ||
|
|
3c5b9d2744 | ||
|
|
cff08d2c72 | ||
|
|
1e1f395947 | ||
|
|
e6c2628960 | ||
|
|
887f7c1530 | ||
|
|
2c6bddd831 | ||
|
|
e1715c33bb | ||
|
|
2ef80bf0b8 | ||
|
|
85ba710718 | ||
|
|
c16b0e7f92 | ||
|
|
b3d388228a | ||
|
|
bcde9de7da | ||
|
|
52bc3261e9 | ||
|
|
2d42f29385 | ||
|
|
17240c6144 | ||
|
|
9e627a4414 | ||
|
|
90b1019636 | ||
|
|
df604afbd5 | ||
|
|
47c7aa62de | ||
|
|
9f2dc48d0f | ||
|
|
6d951b21fb | ||
|
|
552f28cb3e | ||
|
|
e87b6e26f7 |
@@ -1,28 +1,29 @@
|
|||||||
FROM node:16-bullseye
|
FROM node:16-bookworm
|
||||||
|
|
||||||
WORKDIR /root
|
WORKDIR /root
|
||||||
|
|
||||||
ADD ./docker/vitastor.gpg /etc/apt/trusted.gpg.d
|
ADD ./docker/etc/apt/trusted.gpg.d /etc/apt/trusted.gpg.d
|
||||||
|
|
||||||
RUN echo 'deb http://deb.debian.org/debian bullseye-backports main' >> /etc/apt/sources.list; \
|
RUN echo 'deb http://deb.debian.org/debian bookworm-backports main' >> /etc/apt/sources.list; \
|
||||||
echo 'deb http://vitastor.io/debian bullseye main' >> /etc/apt/sources.list; \
|
echo 'deb http://vitastor.io/debian bookworm main' >> /etc/apt/sources.list; \
|
||||||
echo >> /etc/apt/preferences; \
|
echo >> /etc/apt/preferences; \
|
||||||
echo 'Package: *' >> /etc/apt/preferences; \
|
echo 'Package: *' >> /etc/apt/preferences; \
|
||||||
echo 'Pin: release a=bullseye-backports' >> /etc/apt/preferences; \
|
echo 'Pin: release n=bookworm-backports' >> /etc/apt/preferences; \
|
||||||
echo 'Pin-Priority: 500' >> /etc/apt/preferences; \
|
echo 'Pin-Priority: 500' >> /etc/apt/preferences; \
|
||||||
echo >> /etc/apt/preferences; \
|
echo >> /etc/apt/preferences; \
|
||||||
echo 'Package: *' >> /etc/apt/preferences; \
|
echo 'Package: *' >> /etc/apt/preferences; \
|
||||||
echo 'Pin: origin "vitastor.io"' >> /etc/apt/preferences; \
|
echo 'Pin: origin "vitastor.io"' >> /etc/apt/preferences; \
|
||||||
echo 'Pin-Priority: 1000' >> /etc/apt/preferences; \
|
echo 'Pin-Priority: 1000' >> /etc/apt/preferences; \
|
||||||
|
perl -i -pe 's/Types: deb$/Types: deb deb-src/' /etc/apt/sources.list.d/debian.sources; \
|
||||||
grep '^deb ' /etc/apt/sources.list | perl -pe 's/^deb/deb-src/' >> /etc/apt/sources.list; \
|
grep '^deb ' /etc/apt/sources.list | perl -pe 's/^deb/deb-src/' >> /etc/apt/sources.list; \
|
||||||
echo 'APT::Install-Recommends false;' >> /etc/apt/apt.conf; \
|
echo 'APT::Install-Recommends false;' >> /etc/apt/apt.conf; \
|
||||||
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
|
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
|
||||||
|
|
||||||
RUN apt-get update
|
RUN apt-get update
|
||||||
RUN apt-get -y install etcd qemu-system-x86 qemu-block-extra qemu-utils fio libasan5 \
|
RUN apt-get -y install etcd qemu-system-x86 qemu-block-extra qemu-utils fio libasan8 \
|
||||||
libgoogle-perftools-dev devscripts libjerasure-dev cmake libibverbs-dev libisal-dev
|
libgoogle-perftools-dev devscripts libjerasure-dev cmake libibverbs-dev libisal-dev
|
||||||
RUN apt-get -y build-dep fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
RUN apt-get -y build-dep fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
||||||
RUN apt-get update && apt-get -y install jq lp-solve sudo nfs-common fdisk parted
|
RUN apt-get update && apt-get -y install jq lp-solve sudo nfs-common fdisk parted libc-ares-dev udev
|
||||||
RUN apt-get --download-only source fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
RUN apt-get --download-only source fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
||||||
|
|
||||||
RUN set -ex; \
|
RUN set -ex; \
|
||||||
|
|||||||
+868
-4
@@ -306,6 +306,78 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_interrupted_rebalance:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_interrupted_rebalance.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_interrupted_rebalance_imm:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 IMMEDIATE_COMMIT=1 /root/vitastor/tests/test_interrupted_rebalance.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_interrupted_rebalance_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 SCHEME=ec /root/vitastor/tests/test_interrupted_rebalance.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_interrupted_rebalance_ec_imm:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 SCHEME=ec IMMEDIATE_COMMIT=1 /root/vitastor/tests/test_interrupted_rebalance.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_create_halfhost:
|
test_create_halfhost:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -342,6 +414,24 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_level_placement:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_level_placement.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_snapshot:
|
test_snapshot:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -378,6 +468,42 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_snapshot:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_snapshot.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_snapshot_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=ec /root/vitastor/tests/test_snapshot.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_minsize_1:
|
test_minsize_1:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -414,6 +540,42 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_move_reappear:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_move_reappear.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_degraded:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_degraded.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_rm:
|
test_rm:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -486,6 +648,42 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_snapshot_chain:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_snapshot_chain.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_snapshot_chain_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=ec /root/vitastor/tests/test_snapshot_chain.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_snapshot_down:
|
test_snapshot_down:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -522,6 +720,78 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_snapshot_down:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_snapshot_down.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_snapshot_down_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=ec /root/vitastor/tests/test_snapshot_down.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_kv_stress:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_kv_stress.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_kv_stress_imm:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: IMMEDIATE_COMMIT=1 /root/vitastor/tests/test_kv_stress.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_splitbrain:
|
test_splitbrain:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -612,6 +882,78 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_rebalance_verify:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_rebalance_verify.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_rebalance_verify_imm:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 IMMEDIATE_COMMIT=1 /root/vitastor/tests/test_rebalance_verify.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_rebalance_verify_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 SCHEME=ec /root/vitastor/tests/test_rebalance_verify.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_rebalance_verify_ec_imm:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 SCHEME=ec IMMEDIATE_COMMIT=1 /root/vitastor/tests/test_rebalance_verify.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_dd:
|
test_dd:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -720,7 +1062,7 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
test_write_no_same:
|
test_old_write:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
@@ -728,7 +1070,61 @@ jobs:
|
|||||||
- name: Run test
|
- name: Run test
|
||||||
id: test
|
id: test
|
||||||
timeout-minutes: 3
|
timeout-minutes: 3
|
||||||
run: /root/vitastor/tests/test_write_no_same.sh
|
run: OLD=1 /root/vitastor/tests/test_write.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_write_xor:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=xor /root/vitastor/tests/test_write.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_write_old_iothreads:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: TEST_NAME=old_iothreads OLD=1 GLOBAL_CONFIG=',"client_iothread_count":4' /root/vitastor/tests/test_write.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_write_no_same:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_write_no_same.sh
|
||||||
- name: Print logs
|
- name: Print logs
|
||||||
if: always() && steps.test.outcome == 'failure'
|
if: always() && steps.test.outcome == 'failure'
|
||||||
run: |
|
run: |
|
||||||
@@ -810,6 +1206,132 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_checksum:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_checksum.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_checksum:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_checksum.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_corrupt_all:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_corrupt_all.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_corrupt_all:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_corrupt_all.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_reweight_half:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_reweight_half.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_snapshot_pool2:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_snapshot_pool2.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_snapshot_read_bitmap:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_snapshot_read_bitmap.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_heal_csum_32k_dmj:
|
test_heal_csum_32k_dmj:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -954,7 +1476,7 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
test_snapshot_pool2:
|
test_old_resize:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
@@ -962,7 +1484,25 @@ jobs:
|
|||||||
- name: Run test
|
- name: Run test
|
||||||
id: test
|
id: test
|
||||||
timeout-minutes: 3
|
timeout-minutes: 3
|
||||||
run: /root/vitastor/tests/test_snapshot_pool2.sh
|
run: OLD=1 /root/vitastor/tests/test_resize.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_resize_auto:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_resize_auto.sh
|
||||||
- name: Print logs
|
- name: Print logs
|
||||||
if: always() && steps.test.outcome == 'failure'
|
if: always() && steps.test.outcome == 'failure'
|
||||||
run: |
|
run: |
|
||||||
@@ -1062,6 +1602,78 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_enospc:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_enospc.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_enospc_xor:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=xor /root/vitastor/tests/test_enospc.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_enospc_imm:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 IMMEDIATE_COMMIT=1 /root/vitastor/tests/test_enospc.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_enospc_imm_xor:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 IMMEDIATE_COMMIT=1 SCHEME=xor /root/vitastor/tests/test_enospc.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_scrub:
|
test_scrub:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -1170,6 +1782,240 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_scrub:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_scrub.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_scrub_zero_osd_2:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 ZERO_OSD=2 /root/vitastor/tests/test_scrub.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_scrub_xor:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=xor /root/vitastor/tests/test_scrub.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_scrub_pg_size_3:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 PG_SIZE=3 /root/vitastor/tests/test_scrub.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_scrub_pg_size_6_pg_minsize_4_osd_count_6_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 PG_SIZE=6 PG_MINSIZE=4 OSD_COUNT=6 SCHEME=ec /root/vitastor/tests/test_scrub.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_scrub_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=ec /root/vitastor/tests/test_scrub.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_partwr_csum:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_partwr_csum.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_heal_old_csum_32k_dmj:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: TEST_NAME=old_csum_32k_dmj OLD=1 OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k --inmemory_metadata false --inmemory_journal false" OFFSET_ARGS=$OSD_ARGS /root/vitastor/tests/test_heal.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_heal_old_csum_32k_dj:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: TEST_NAME=old_csum_32k_dj OLD=1 OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k --inmemory_journal false" OFFSET_ARGS=$OSD_ARGS /root/vitastor/tests/test_heal.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_heal_old_csum_32k:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: TEST_NAME=old_csum_32k OLD=1 OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k" OFFSET_ARGS=$OSD_ARGS /root/vitastor/tests/test_heal.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_heal_old_csum_4k_dmj:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: TEST_NAME=old_csum_4k_dmj OLD=1 OSD_ARGS="--data_csum_type crc32c --inmemory_metadata false --inmemory_journal false" OFFSET_ARGS=$OSD_ARGS /root/vitastor/tests/test_heal.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_heal_old_csum_4k_dj:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: TEST_NAME=old_csum_4k_dj OLD=1 OSD_ARGS="--data_csum_type crc32c --inmemory_journal false" OFFSET_ARGS=$OSD_ARGS /root/vitastor/tests/test_heal.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_heal_old_csum_4k:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: TEST_NAME=old_csum_4k OLD=1 OSD_ARGS="--data_csum_type crc32c" OFFSET_ARGS=$OSD_ARGS /root/vitastor/tests/test_heal.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_nfs:
|
test_nfs:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -1188,3 +2034,21 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_nfs_unaligned_append:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_nfs_unaligned_append.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
|||||||
@@ -38,6 +38,10 @@ for my $line (<>)
|
|||||||
{
|
{
|
||||||
$test_name .= '_antietcd';
|
$test_name .= '_antietcd';
|
||||||
}
|
}
|
||||||
|
elsif ($1 eq 'OLD')
|
||||||
|
{
|
||||||
|
$test_name =~ s/^test_/test_old_/s;
|
||||||
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
$test_name .= '_'.lc($1).'_'.$2;
|
$test_name .= '_'.lc($1).'_'.$2;
|
||||||
|
|||||||
+14
-1
@@ -2,6 +2,19 @@ cmake_minimum_required(VERSION 2.8.12)
|
|||||||
|
|
||||||
project(vitastor)
|
project(vitastor)
|
||||||
|
|
||||||
set(VITASTOR_VERSION "2.3.0")
|
set(VITASTOR_VERSION "3.0.6")
|
||||||
|
|
||||||
|
include(CTest)
|
||||||
|
|
||||||
|
add_custom_target(build_tests)
|
||||||
|
add_custom_target(test
|
||||||
|
COMMAND
|
||||||
|
echo leak:tcmalloc > ${CMAKE_CURRENT_BINARY_DIR}/lsan-suppress.txt &&
|
||||||
|
env LSAN_OPTIONS=suppressions=${CMAKE_CURRENT_BINARY_DIR}/lsan-suppress.txt ${CMAKE_CTEST_COMMAND}
|
||||||
|
)
|
||||||
|
# make -j16 -C ../../build test_heap && ../../build/src/test/test_heap
|
||||||
|
# make -j16 -C ../../build test_heap && rm -f $(find ../../build -name '*.gcda') && ctest -V -T test -T coverage -R heap --test-dir ../../build && (cd ../../build; gcovr -f ../src --html --html-nested -o coverage/index.html; cd ../src/test)
|
||||||
|
# make -j16 -C ../../build test_blockstore && rm -f $(find ../../build -name '*.gcda') && ctest -V -T test -T coverage -R blockstore --test-dir ../../build && (cd ../../build; gcovr -f ../src --html --html-nested -o coverage/index.html; cd ../src/test)
|
||||||
|
# kcov --include-path=../../../src ../../kcov ./test_blockstore
|
||||||
|
add_dependencies(test build_tests)
|
||||||
add_subdirectory(src)
|
add_subdirectory(src)
|
||||||
|
|||||||
+6
-2
@@ -26,11 +26,15 @@ Vitastor поддерживает QEMU-драйвер, протоколы UBLK,
|
|||||||
|
|
||||||
## Презентации и записи докладов
|
## Презентации и записи докладов
|
||||||
|
|
||||||
|
- KuberConf'2025: [видео](https://vitastor.io/presentation/kuberconf.webm)
|
||||||
|
- Highload'2025: [видео](https://vitastor.io/presentation/hl2025/hl2025.webm),
|
||||||
|
[на youtube](https://www.youtube.com/watch?v=0R8MLjFtz7g), презентация
|
||||||
|
([на русском](https://vitastor.io/presentation/hl2025/), [на английском](https://vitastor.io/presentation/hl2025/en.html))
|
||||||
|
- Highload'2022: презентация ([на русском](https://vitastor.io/presentation/highload/highload.html)),
|
||||||
|
[видео](https://vitastor.io/presentation/highload/talk.webm)
|
||||||
- DevOpsConf'2021: презентация ([на русском](https://vitastor.io/presentation/devopsconf/devopsconf.html),
|
- DevOpsConf'2021: презентация ([на русском](https://vitastor.io/presentation/devopsconf/devopsconf.html),
|
||||||
[на английском](https://vitastor.io/presentation/devopsconf/devopsconf_en.html)),
|
[на английском](https://vitastor.io/presentation/devopsconf/devopsconf_en.html)),
|
||||||
[видео](https://vitastor.io/presentation/devopsconf/talk.webm)
|
[видео](https://vitastor.io/presentation/devopsconf/talk.webm)
|
||||||
- Highload'2022: презентация ([на русском](https://vitastor.io/presentation/highload/highload.html)),
|
|
||||||
[видео](https://vitastor.io/presentation/highload/talk.webm)
|
|
||||||
|
|
||||||
## Документация
|
## Документация
|
||||||
|
|
||||||
|
|||||||
@@ -26,11 +26,15 @@ Read more details in the documentation. You can start from here: [Quick Start](d
|
|||||||
|
|
||||||
## Talks and presentations
|
## Talks and presentations
|
||||||
|
|
||||||
|
- KuberConf'2025: [video](https://vitastor.io/presentation/kuberconf.webm)
|
||||||
|
- Highload'2025: [video](https://vitastor.io/presentation/hl2025/hl2025.webm),
|
||||||
|
[youtube](https://www.youtube.com/watch?v=0R8MLjFtz7g), presentation
|
||||||
|
([in Russian](https://vitastor.io/presentation/hl2025/), [in English](https://vitastor.io/presentation/hl2025/en.html))
|
||||||
|
- Highload'2022: presentation ([in Russian](https://vitastor.io/presentation/highload/highload.html)),
|
||||||
|
[video](https://vitastor.io/presentation/highload/talk.webm)
|
||||||
- DevOpsConf'2021: presentation ([in Russian](https://vitastor.io/presentation/devopsconf/devopsconf.html),
|
- DevOpsConf'2021: presentation ([in Russian](https://vitastor.io/presentation/devopsconf/devopsconf.html),
|
||||||
[in English](https://vitastor.io/presentation/devopsconf/devopsconf_en.html)),
|
[in English](https://vitastor.io/presentation/devopsconf/devopsconf_en.html)),
|
||||||
[video](https://vitastor.io/presentation/devopsconf/talk.webm)
|
[video](https://vitastor.io/presentation/devopsconf/talk.webm)
|
||||||
- Highload'2022: presentation ([in Russian](https://vitastor.io/presentation/highload/highload.html)),
|
|
||||||
[video](https://vitastor.io/presentation/highload/talk.webm)
|
|
||||||
|
|
||||||
## Documentation
|
## Documentation
|
||||||
|
|
||||||
|
|||||||
+8
-8
@@ -1,5 +1,5 @@
|
|||||||
# Compile stage
|
# Compile stage
|
||||||
FROM golang:bookworm AS build
|
FROM golang:trixie AS build
|
||||||
|
|
||||||
ADD go.sum go.mod /app/
|
ADD go.sum go.mod /app/
|
||||||
RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go mod download -x
|
RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go mod download -x
|
||||||
@@ -9,7 +9,7 @@ RUN perl -i -e '$/ = undef; while(<>) { s/\n\s*(\{\s*\n)/$1\n/g; s/\}(\s*\n\s*)e
|
|||||||
CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o vitastor-csi
|
CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o vitastor-csi
|
||||||
|
|
||||||
# Final stage
|
# Final stage
|
||||||
FROM debian:bookworm
|
FROM debian:trixie
|
||||||
|
|
||||||
LABEL maintainers="Vitaliy Filippov <vitalif@yourcmc.ru>"
|
LABEL maintainers="Vitaliy Filippov <vitalif@yourcmc.ru>"
|
||||||
LABEL description="Vitastor CSI Driver"
|
LABEL description="Vitastor CSI Driver"
|
||||||
@@ -25,20 +25,20 @@ RUN apt-get update && \
|
|||||||
# NFS mount dependencies
|
# NFS mount dependencies
|
||||||
nfs-common netbase \
|
nfs-common netbase \
|
||||||
# dependencies of qemu-storage-daemon
|
# dependencies of qemu-storage-daemon
|
||||||
libnuma1 liburing2 libglib2.0-0 libfuse3-3 libaio1 libzstd1 libnettle8 \
|
libaio1t64 libc6 libfuse3-4 libglib2.0-0t64 libgmp10 libgnutls30t64 \
|
||||||
libgmp10 libhogweed6 libp11-kit0 libidn2-0 libunistring2 libtasn1-6 libpcre2-8-0 libffi8 && \
|
libhogweed6t64 libnettle8t64 libnuma1 libselinux1 liburing2 libzstd1 zlib1g && \
|
||||||
apt-get clean && \
|
apt-get clean && \
|
||||||
(echo options nbd nbds_max=128 > /etc/modprobe.d/nbd.conf)
|
(echo options nbd nbds_max=128 > /etc/modprobe.d/nbd.conf)
|
||||||
|
|
||||||
COPY --from=build /app/vitastor-csi /bin/
|
COPY --from=build /app/vitastor-csi /bin/
|
||||||
|
|
||||||
RUN (echo deb http://vitastor.io/debian bookworm main > /etc/apt/sources.list.d/vitastor.list) && \
|
RUN (echo deb http://vitastor.io/debian trixie main > /etc/apt/sources.list.d/vitastor.list) && \
|
||||||
((echo 'Package: *'; echo 'Pin: origin "vitastor.io"'; echo 'Pin-Priority: 1000') > /etc/apt/preferences.d/vitastor.pref) && \
|
((echo 'Package: *'; echo 'Pin: origin "vitastor.io"'; echo 'Pin-Priority: 1000') > /etc/apt/preferences.d/vitastor.pref) && \
|
||||||
wget -q -O /etc/apt/trusted.gpg.d/vitastor.gpg https://vitastor.io/debian/pubkey.gpg && \
|
wget -q -O /etc/apt/trusted.gpg.d/vitastor.gpg https://vitastor.io/debian/pubkey.gpg && \
|
||||||
apt-get update && \
|
apt-get update && \
|
||||||
apt-get install -y vitastor-client && \
|
apt-get install -y vitastor-client ibverbs-providers && \
|
||||||
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
wget https://vitastor.io/archive/qemu/qemu-trixie-10.0.2%2Bds-2%2Bvitastor1/qemu-utils_10.0.2%2Bds-2%2Bvitastor1_amd64.deb && \
|
||||||
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
wget https://vitastor.io/archive/qemu/qemu-trixie-10.0.2%2Bds-2%2Bvitastor1/qemu-block-extra_10.0.2%2Bds-2%2Bvitastor1_amd64.deb && \
|
||||||
dpkg -x qemu-utils*.deb tmp1 && \
|
dpkg -x qemu-utils*.deb tmp1 && \
|
||||||
dpkg -x qemu-block-extra*.deb tmp1 && \
|
dpkg -x qemu-block-extra*.deb tmp1 && \
|
||||||
cp -a tmp1/usr/bin/qemu-storage-daemon /usr/bin/ && \
|
cp -a tmp1/usr/bin/qemu-storage-daemon /usr/bin/ && \
|
||||||
|
|||||||
@@ -0,0 +1,49 @@
|
|||||||
|
# Compile stage
|
||||||
|
FROM golang:trixie AS build
|
||||||
|
|
||||||
|
ADD go.sum go.mod /app/
|
||||||
|
RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go mod download -x
|
||||||
|
ADD . /app
|
||||||
|
RUN perl -i -e '$/ = undef; while(<>) { s/\n\s*(\{\s*\n)/$1\n/g; s/\}(\s*\n\s*)else\b/$1} else/g; print; }' `find /app -name '*.go'` && \
|
||||||
|
cd /app && \
|
||||||
|
CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o vitastor-csi
|
||||||
|
|
||||||
|
# Final stage
|
||||||
|
FROM debian:trixie
|
||||||
|
|
||||||
|
LABEL maintainers="Vitaliy Filippov <vitalif@yourcmc.ru>"
|
||||||
|
LABEL description="Vitastor CSI Driver"
|
||||||
|
|
||||||
|
ENV NODE_ID=""
|
||||||
|
ENV CSI_ENDPOINT=""
|
||||||
|
|
||||||
|
RUN apt-get update && \
|
||||||
|
apt-get install -y wget && \
|
||||||
|
(echo "APT::Install-Recommends false;" > /etc/apt/apt.conf) && \
|
||||||
|
apt-get update && \
|
||||||
|
apt-get install -y e2fsprogs xfsprogs kmod iproute2 \
|
||||||
|
# NFS mount dependencies
|
||||||
|
nfs-common netbase \
|
||||||
|
# dependencies of qemu-storage-daemon
|
||||||
|
libnuma1 liburing2 libglib2.0-0 libfuse3-3 libaio1 libzstd1 libnettle8 \
|
||||||
|
libgmp10 libhogweed6 libp11-kit0 libidn2-0 libunistring2 libtasn1-6 libpcre2-8-0 libffi8 && \
|
||||||
|
apt-get clean && \
|
||||||
|
(echo options nbd nbds_max=128 > /etc/modprobe.d/nbd.conf)
|
||||||
|
|
||||||
|
COPY --from=build /app/vitastor-csi /bin/
|
||||||
|
|
||||||
|
ADD deb /deb
|
||||||
|
|
||||||
|
RUN apt-get update && \
|
||||||
|
apt-get -y install /deb/vitastor-client_*.deb && \
|
||||||
|
wget https://vitastor.io/archive/qemu/qemu-trixie-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||||
|
wget https://vitastor.io/archive/qemu/qemu-trixie-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||||
|
dpkg -x qemu-utils*.deb tmp1 && \
|
||||||
|
dpkg -x qemu-block-extra*.deb tmp1 && \
|
||||||
|
cp -a tmp1/usr/bin/qemu-storage-daemon /usr/bin/ && \
|
||||||
|
mkdir -p /usr/lib/x86_64-linux-gnu/qemu && \
|
||||||
|
cp -a tmp1/usr/lib/x86_64-linux-gnu/qemu/block-vitastor.so /usr/lib/x86_64-linux-gnu/qemu/ && \
|
||||||
|
rm -rf tmp1 *.deb && \
|
||||||
|
apt-get clean
|
||||||
|
|
||||||
|
ENTRYPOINT ["/bin/vitastor-csi"]
|
||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
VITASTOR_VERSION ?= v2.3.0
|
VITASTOR_VERSION ?= v3.0.6
|
||||||
|
|
||||||
all: build push
|
all: build push
|
||||||
|
|
||||||
|
|||||||
@@ -49,7 +49,7 @@ spec:
|
|||||||
capabilities:
|
capabilities:
|
||||||
add: ["SYS_ADMIN"]
|
add: ["SYS_ADMIN"]
|
||||||
allowPrivilegeEscalation: true
|
allowPrivilegeEscalation: true
|
||||||
image: vitalif/vitastor-csi:v2.3.0
|
image: vitalif/vitastor-csi:v3.0.6
|
||||||
args:
|
args:
|
||||||
- "--node=$(NODE_ID)"
|
- "--node=$(NODE_ID)"
|
||||||
- "--endpoint=$(CSI_ENDPOINT)"
|
- "--endpoint=$(CSI_ENDPOINT)"
|
||||||
|
|||||||
@@ -121,7 +121,7 @@ spec:
|
|||||||
privileged: true
|
privileged: true
|
||||||
capabilities:
|
capabilities:
|
||||||
add: ["SYS_ADMIN"]
|
add: ["SYS_ADMIN"]
|
||||||
image: vitalif/vitastor-csi:v2.3.0
|
image: vitalif/vitastor-csi:v3.0.6
|
||||||
args:
|
args:
|
||||||
- "--node=$(NODE_ID)"
|
- "--node=$(NODE_ID)"
|
||||||
- "--endpoint=$(CSI_ENDPOINT)"
|
- "--endpoint=$(CSI_ENDPOINT)"
|
||||||
|
|||||||
+1
-1
@@ -5,7 +5,7 @@ package vitastor
|
|||||||
|
|
||||||
const (
|
const (
|
||||||
vitastorCSIDriverName = "csi.vitastor.io"
|
vitastorCSIDriverName = "csi.vitastor.io"
|
||||||
vitastorCSIDriverVersion = "2.3.0"
|
vitastorCSIDriverVersion = "3.0.6"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Config struct fills the parameters of request or user input
|
// Config struct fills the parameters of request or user input
|
||||||
|
|||||||
+115
-20
@@ -33,7 +33,7 @@ import (
|
|||||||
type NodeServer struct
|
type NodeServer struct
|
||||||
{
|
{
|
||||||
*Driver
|
*Driver
|
||||||
useVduse bool
|
method MountMethod
|
||||||
stateDir string
|
stateDir string
|
||||||
nfsStageDir string
|
nfsStageDir string
|
||||||
mounter mount.Interface
|
mounter mount.Interface
|
||||||
@@ -81,16 +81,23 @@ func NewNodeServer(driver *Driver) *NodeServer
|
|||||||
}
|
}
|
||||||
ns := &NodeServer{
|
ns := &NodeServer{
|
||||||
Driver: driver,
|
Driver: driver,
|
||||||
useVduse: checkVduseSupport(),
|
method: selectMountMethod(),
|
||||||
stateDir: stateDir,
|
stateDir: stateDir,
|
||||||
nfsStageDir: nfsStageDir,
|
nfsStageDir: nfsStageDir,
|
||||||
mounter: mount.New(""),
|
mounter: mount.New(""),
|
||||||
volumeLocks: make(map[string]bool),
|
volumeLocks: make(map[string]bool),
|
||||||
}
|
}
|
||||||
ns.cond = sync.NewCond(&ns.mu)
|
ns.cond = sync.NewCond(&ns.mu)
|
||||||
if (ns.useVduse)
|
if (ns.method == MOUNT_VDUSE)
|
||||||
{
|
{
|
||||||
ns.restoreVduseDaemons()
|
ns.restoreVduseDaemons()
|
||||||
|
}
|
||||||
|
else if (ns.method == MOUNT_UBLK)
|
||||||
|
{
|
||||||
|
ns.restoreUblkDaemons()
|
||||||
|
}
|
||||||
|
if (ns.method == MOUNT_VDUSE || ns.method == MOUNT_UBLK)
|
||||||
|
{
|
||||||
dur, err := time.ParseDuration(os.Getenv("RESTART_INTERVAL"))
|
dur, err := time.ParseDuration(os.Getenv("RESTART_INTERVAL"))
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
@@ -136,7 +143,14 @@ func (ns *NodeServer) restarter()
|
|||||||
for
|
for
|
||||||
{
|
{
|
||||||
<-ticker.C
|
<-ticker.C
|
||||||
ns.restoreVduseDaemons()
|
if (ns.method == MOUNT_VDUSE)
|
||||||
|
{
|
||||||
|
ns.restoreVduseDaemons()
|
||||||
|
}
|
||||||
|
else if (ns.method == MOUNT_UBLK)
|
||||||
|
{
|
||||||
|
ns.restoreUblkDaemons()
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -231,6 +245,78 @@ func (ns *NodeServer) checkVduseState(stateFile string, devs map[string]interfac
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (ns *NodeServer) restoreUblkDaemons()
|
||||||
|
{
|
||||||
|
pattern := ns.stateDir+"vitastor-ublk-*.json"
|
||||||
|
stateFiles, err := filepath.Glob(pattern)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to list %s: %v", pattern, err)
|
||||||
|
}
|
||||||
|
if (len(stateFiles) == 0)
|
||||||
|
{
|
||||||
|
return
|
||||||
|
}
|
||||||
|
for _, stateFile := range stateFiles
|
||||||
|
{
|
||||||
|
deviceNum := stateFile[len(ns.stateDir) + len("vitastor-ublk-") :]
|
||||||
|
deviceNum = deviceNum[0:len(deviceNum)-5]
|
||||||
|
ns.checkUblkState(deviceNum)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (ns *NodeServer) checkUblkState(deviceNum string)
|
||||||
|
{
|
||||||
|
// Check if the ublk daemon is still active
|
||||||
|
|
||||||
|
// Read state file
|
||||||
|
stateFile := ns.stateDir + "vitastor-ublk-" + deviceNum + ".json"
|
||||||
|
stateJSON, err := os.ReadFile(stateFile)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Warningf("error reading state file %v: %v", stateFile, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
var state DeviceState
|
||||||
|
err = json.Unmarshal(stateJSON, &state)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Warningf("state file %v contains invalid JSON (error %v): %v", stateFile, err, string(stateJSON))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Lock volume
|
||||||
|
ns.lockVolume(state.ConfigPath+":block:"+state.Image)
|
||||||
|
defer ns.unlockVolume(state.ConfigPath+":block:"+state.Image)
|
||||||
|
|
||||||
|
// Recheck state file after locking
|
||||||
|
_, err = os.ReadFile(stateFile)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Warningf("state file %v disappeared, skipping volume", stateFile)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Check if the vitastor-ublk process is still active
|
||||||
|
pidFile := ns.stateDir + "vitastor-ublk-" + deviceNum + ".pid"
|
||||||
|
exists := false
|
||||||
|
proc, err := findByPidFile(pidFile)
|
||||||
|
if (err == nil)
|
||||||
|
{
|
||||||
|
exists = proc.Signal(syscall.Signal(0)) == nil
|
||||||
|
}
|
||||||
|
if (!exists)
|
||||||
|
{
|
||||||
|
// Restart daemon
|
||||||
|
klog.Warningf("recovering UBLK device /dev/ublkb%v for volume %v", deviceNum, state.Image)
|
||||||
|
_, err = mapUblk(ns.stateDir, state.Image, state.ConfigPath, state.Readonly, "/dev/ublkb"+deviceNum)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Warningf("failed to recover ublk device for volume %v: %v", state.Image, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func (ns *NodeServer) restoreNfsDaemons()
|
func (ns *NodeServer) restoreNfsDaemons()
|
||||||
{
|
{
|
||||||
pattern := ns.stateDir+"vitastor-nfs-*.json"
|
pattern := ns.stateDir+"vitastor-nfs-*.json"
|
||||||
@@ -417,14 +503,18 @@ func (ns *NodeServer) NodeStageVolume(ctx context.Context, req *csi.NodeStageVol
|
|||||||
}
|
}
|
||||||
|
|
||||||
var devicePath, vdpaId string
|
var devicePath, vdpaId string
|
||||||
if (!ns.useVduse)
|
if (ns.method == MOUNT_UBLK)
|
||||||
{
|
{
|
||||||
devicePath, err = mapNbd(volName, ctxVars, false)
|
devicePath, err = mapUblk(ns.stateDir, volName, ctxVars["configPath"], false, "")
|
||||||
}
|
}
|
||||||
else
|
else if (ns.method == MOUNT_VDUSE)
|
||||||
{
|
{
|
||||||
devicePath, vdpaId, err = mapVduse(ns.stateDir, volName, ctxVars, false)
|
devicePath, vdpaId, err = mapVduse(ns.stateDir, volName, ctxVars, false)
|
||||||
}
|
}
|
||||||
|
else /* if (ns.method == MOUNT_NBD) */
|
||||||
|
{
|
||||||
|
devicePath, err = mapNbd(volName, ctxVars, false)
|
||||||
|
}
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
return nil, err
|
return nil, err
|
||||||
@@ -439,7 +529,8 @@ func (ns *NodeServer) NodeStageVolume(ctx context.Context, req *csi.NodeStageVol
|
|||||||
else
|
else
|
||||||
{
|
{
|
||||||
// Check existing format
|
// Check existing format
|
||||||
existingFormat, err := diskMounter.GetDiskFormat(devicePath)
|
var existingFormat string
|
||||||
|
existingFormat, err = diskMounter.GetDiskFormat(devicePath)
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
klog.Errorf("failed to get disk format for path %s, error: %v", err)
|
klog.Errorf("failed to get disk format for path %s, error: %v", err)
|
||||||
@@ -495,10 +586,6 @@ func (ns *NodeServer) NodeStageVolume(ctx context.Context, req *csi.NodeStageVol
|
|||||||
case "xfs":
|
case "xfs":
|
||||||
_, err = systemCombined("xfs_growfs", devicePath)
|
_, err = systemCombined("xfs_growfs", devicePath)
|
||||||
}
|
}
|
||||||
if (err != nil)
|
|
||||||
{
|
|
||||||
goto unmap
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
@@ -512,14 +599,18 @@ func (ns *NodeServer) NodeStageVolume(ctx context.Context, req *csi.NodeStageVol
|
|||||||
return &csi.NodeStageVolumeResponse{}, nil
|
return &csi.NodeStageVolumeResponse{}, nil
|
||||||
|
|
||||||
unmap:
|
unmap:
|
||||||
if (!ns.useVduse || len(devicePath) >= 8 && devicePath[0:8] == "/dev/nbd")
|
if (ns.method == MOUNT_UBLK)
|
||||||
{
|
{
|
||||||
unmapNbd(devicePath)
|
unmapUblk(ns.stateDir, devicePath)
|
||||||
}
|
}
|
||||||
else
|
else if (ns.method == MOUNT_VDUSE)
|
||||||
{
|
{
|
||||||
unmapVduseById(ns.stateDir, vdpaId)
|
unmapVduseById(ns.stateDir, vdpaId)
|
||||||
}
|
}
|
||||||
|
else /* if (ns.method == MOUNT_NBD) */
|
||||||
|
{
|
||||||
|
unmapNbd(devicePath)
|
||||||
|
}
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -545,7 +636,7 @@ func (ns *NodeServer) NodeUnstageVolume(ctx context.Context, req *csi.NodeUnstag
|
|||||||
defer ns.unlockVolume(ctxVars["configPath"]+":block:"+volName)
|
defer ns.unlockVolume(ctxVars["configPath"]+":block:"+volName)
|
||||||
|
|
||||||
targetPath := req.GetStagingTargetPath()
|
targetPath := req.GetStagingTargetPath()
|
||||||
devicePath, _, err := mount.GetDeviceNameFromMount(ns.mounter, targetPath)
|
devicePath, err := GetDeviceNameFromMount(targetPath)
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
if (os.IsNotExist(err))
|
if (os.IsNotExist(err))
|
||||||
@@ -582,14 +673,18 @@ func (ns *NodeServer) NodeUnstageVolume(ctx context.Context, req *csi.NodeUnstag
|
|||||||
// unmap device
|
// unmap device
|
||||||
if (len(refList) == 0)
|
if (len(refList) == 0)
|
||||||
{
|
{
|
||||||
if (!ns.useVduse)
|
if (ns.method == MOUNT_UBLK)
|
||||||
{
|
{
|
||||||
unmapNbd(devicePath)
|
unmapUblk(ns.stateDir, devicePath)
|
||||||
}
|
}
|
||||||
else
|
else if (ns.method == MOUNT_VDUSE)
|
||||||
{
|
{
|
||||||
unmapVduse(ns.stateDir, devicePath)
|
unmapVduse(ns.stateDir, devicePath)
|
||||||
}
|
}
|
||||||
|
else /* if (ns.method == MOUNT_NBD) */
|
||||||
|
{
|
||||||
|
unmapNbd(devicePath)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return &csi.NodeUnstageVolumeResponse{}, nil
|
return &csi.NodeUnstageVolumeResponse{}, nil
|
||||||
@@ -897,7 +992,7 @@ func (ns *NodeServer) NodeUnpublishVolume(ctx context.Context, req *csi.NodeUnpu
|
|||||||
}
|
}
|
||||||
|
|
||||||
targetPath := req.GetTargetPath()
|
targetPath := req.GetTargetPath()
|
||||||
devicePath, _, err := mount.GetDeviceNameFromMount(ns.mounter, targetPath)
|
devicePath, err := GetDeviceNameFromMount(targetPath)
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
if (os.IsNotExist(err))
|
if (os.IsNotExist(err))
|
||||||
|
|||||||
+205
-26
@@ -16,10 +16,20 @@ import (
|
|||||||
"syscall"
|
"syscall"
|
||||||
|
|
||||||
"k8s.io/klog"
|
"k8s.io/klog"
|
||||||
|
"k8s.io/utils/mount"
|
||||||
|
|
||||||
"google.golang.org/grpc/codes"
|
"google.golang.org/grpc/codes"
|
||||||
"google.golang.org/grpc/status"
|
"google.golang.org/grpc/status"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
type MountMethod int
|
||||||
|
|
||||||
|
const (
|
||||||
|
MOUNT_NBD MountMethod = 0
|
||||||
|
MOUNT_VDUSE MountMethod = 1
|
||||||
|
MOUNT_UBLK MountMethod = 2
|
||||||
|
)
|
||||||
|
|
||||||
func Contains(list []string, s string) bool
|
func Contains(list []string, s string) bool
|
||||||
{
|
{
|
||||||
for i := 0; i < len(list); i++
|
for i := 0; i < len(list); i++
|
||||||
@@ -32,29 +42,26 @@ func Contains(list []string, s string) bool
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
func checkVduseSupport() bool
|
func selectMountMethod() MountMethod
|
||||||
{
|
{
|
||||||
|
// Check UBLK support (ublk_drv kernel module)
|
||||||
|
if (checkModule("ublk_drv"))
|
||||||
|
{
|
||||||
|
klog.Infof("UBLK support enabled successfully")
|
||||||
|
return MOUNT_UBLK
|
||||||
|
}
|
||||||
|
klog.Errorf(
|
||||||
|
"Your host apparently has no UBLK support. UBLK support disabled."+
|
||||||
|
" For UBLK you need at least Linux 6.0 and the ublk_drv kernel module.",
|
||||||
|
)
|
||||||
// Check VDUSE support (vdpa, vduse, virtio-vdpa kernel modules)
|
// Check VDUSE support (vdpa, vduse, virtio-vdpa kernel modules)
|
||||||
vduse := true
|
vduse := true
|
||||||
for _, mod := range []string{"vdpa", "vduse", "virtio-vdpa"}
|
for _, mod := range []string{"vdpa", "vduse", "virtio-vdpa"}
|
||||||
{
|
{
|
||||||
_, err := os.Stat("/sys/module/"+mod)
|
if (!checkModule(mod))
|
||||||
if (err != nil)
|
|
||||||
{
|
{
|
||||||
if (!errors.Is(err, os.ErrNotExist))
|
vduse = false
|
||||||
{
|
break
|
||||||
klog.Errorf("failed to check /sys/module/%s: %v", mod, err)
|
|
||||||
}
|
|
||||||
c := exec.Command("/sbin/modprobe", mod)
|
|
||||||
c.Stdout = os.Stderr
|
|
||||||
c.Stderr = os.Stderr
|
|
||||||
err := c.Run()
|
|
||||||
if (err != nil)
|
|
||||||
{
|
|
||||||
klog.Errorf("/sbin/modprobe %s failed: %v", mod, err)
|
|
||||||
vduse = false
|
|
||||||
break
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Check that vdpa tool functions
|
// Check that vdpa tool functions
|
||||||
@@ -69,18 +76,38 @@ func checkVduseSupport() bool
|
|||||||
vduse = false
|
vduse = false
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!vduse)
|
if (vduse)
|
||||||
{
|
|
||||||
klog.Errorf(
|
|
||||||
"Your host apparently has no VDUSE support. VDUSE support disabled, NBD will be used to map devices."+
|
|
||||||
" For VDUSE you need at least Linux 5.15 and the following kernel modules: vdpa, virtio-vdpa, vduse.",
|
|
||||||
)
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
{
|
||||||
klog.Infof("VDUSE support enabled successfully")
|
klog.Infof("VDUSE support enabled successfully")
|
||||||
|
return MOUNT_VDUSE
|
||||||
}
|
}
|
||||||
return vduse
|
klog.Errorf(
|
||||||
|
"Your host apparently has no VDUSE support. VDUSE support disabled, NBD will be used to map devices."+
|
||||||
|
" For VDUSE you need at least Linux 5.15 and the following kernel modules: vdpa, virtio-vdpa, vduse.",
|
||||||
|
)
|
||||||
|
return MOUNT_NBD
|
||||||
|
}
|
||||||
|
|
||||||
|
func checkModule(mod string) bool
|
||||||
|
{
|
||||||
|
_, err := os.Stat("/sys/module/"+mod)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
if (!errors.Is(err, os.ErrNotExist))
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to check /sys/module/%s: %v", mod, err)
|
||||||
|
}
|
||||||
|
c := exec.Command("/sbin/modprobe", mod)
|
||||||
|
c.Stdout = os.Stderr
|
||||||
|
c.Stderr = os.Stderr
|
||||||
|
err := c.Run()
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("/sbin/modprobe %s failed: %v", mod, err)
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
func mapNbd(volName string, ctxVars map[string]string, readonly bool) (string, error)
|
func mapNbd(volName string, ctxVars map[string]string, readonly bool) (string, error)
|
||||||
@@ -217,6 +244,7 @@ func mapVduse(stateDir string, volName string, ctxVars map[string]string, readon
|
|||||||
stateJSON, _ := json.Marshal(&DeviceState{
|
stateJSON, _ := json.Marshal(&DeviceState{
|
||||||
ConfigPath: ctxVars["configPath"],
|
ConfigPath: ctxVars["configPath"],
|
||||||
VdpaId: vdpaId,
|
VdpaId: vdpaId,
|
||||||
|
|
||||||
Image: volName,
|
Image: volName,
|
||||||
Blockdev: blockdev,
|
Blockdev: blockdev,
|
||||||
Readonly: readonly,
|
Readonly: readonly,
|
||||||
@@ -309,6 +337,117 @@ func unmapVduseById(stateDir, vdpaId string)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func mapUblk(stateDir string, volName string, configPath string, readonly bool, recoverDev string) (string, error)
|
||||||
|
{
|
||||||
|
pidFile := ""
|
||||||
|
if (recoverDev != "")
|
||||||
|
{
|
||||||
|
if (len(recoverDev) < 10 || recoverDev[0:10] != "/dev/ublkb")
|
||||||
|
{
|
||||||
|
return "", fmt.Errorf("recover: %s does not start with /dev/ublkb", recoverDev)
|
||||||
|
}
|
||||||
|
pidFile = stateDir + "vitastor-ublk-" + recoverDev[10:] + ".pid"
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
pidFd, err := os.CreateTemp(stateDir, "vitastor-tmp-*.pid")
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
pidFile = pidFd.Name()
|
||||||
|
pidFd.Close()
|
||||||
|
}
|
||||||
|
// Map device via vitastor-ublk
|
||||||
|
args := []string{
|
||||||
|
"map", "--image", volName, "--pidfile", pidFile,
|
||||||
|
}
|
||||||
|
if (configPath != "")
|
||||||
|
{
|
||||||
|
args = append(args, "--config_path", configPath)
|
||||||
|
}
|
||||||
|
if (readonly)
|
||||||
|
{
|
||||||
|
args = append(args, "--readonly")
|
||||||
|
}
|
||||||
|
if (recoverDev != "")
|
||||||
|
{
|
||||||
|
args = append(args, "--recover", recoverDev)
|
||||||
|
}
|
||||||
|
stdout, stderr, err := system("/usr/bin/vitastor-ublk", args...)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
devicePath := strings.TrimSpace(string(stdout))
|
||||||
|
if (devicePath == "")
|
||||||
|
{
|
||||||
|
return "", fmt.Errorf("vitastor-ublk did not return the name of the device. output: %s", stderr)
|
||||||
|
}
|
||||||
|
if (len(devicePath) >= 10 && devicePath[0:10] == "/dev/ublkb")
|
||||||
|
{
|
||||||
|
// Generate state file
|
||||||
|
devNum := devicePath[10:]
|
||||||
|
pidNew := stateDir + "vitastor-ublk-" + devNum + ".pid"
|
||||||
|
if (pidFile != pidNew)
|
||||||
|
{
|
||||||
|
err := os.Rename(pidFile, pidNew)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("Failed to rename PID file %s to %s: %v", pidFile, pidNew, err)
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
pidFile = pidNew
|
||||||
|
}
|
||||||
|
}
|
||||||
|
stateFile := stateDir + "vitastor-ublk-" + devNum + ".json"
|
||||||
|
stateJSON, _ := json.Marshal(&DeviceState{
|
||||||
|
ConfigPath: configPath,
|
||||||
|
Image: volName,
|
||||||
|
Readonly: readonly,
|
||||||
|
PidFile: pidFile,
|
||||||
|
})
|
||||||
|
err = os.WriteFile(stateFile, stateJSON, 0600)
|
||||||
|
if (err == nil)
|
||||||
|
{
|
||||||
|
klog.Infof("Attached volume %s via UBLK as %s", volName, devicePath)
|
||||||
|
return devicePath, nil
|
||||||
|
}
|
||||||
|
os.Remove(stateFile)
|
||||||
|
}
|
||||||
|
killErr := killByPidFile(pidFile)
|
||||||
|
if (killErr != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("Failed to kill started vitastor-ublk: %v", killErr)
|
||||||
|
}
|
||||||
|
os.Remove(pidFile)
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
func unmapUblk(stateDir, devicePath string)
|
||||||
|
{
|
||||||
|
if (len(devicePath) < 10 || devicePath[0:10] != "/dev/ublkb")
|
||||||
|
{
|
||||||
|
klog.Errorf("%s does not start with /dev/ublkb", devicePath)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
unmapOut, unmapErr := exec.Command("/usr/bin/vitastor-ublk", "unmap", devicePath).CombinedOutput()
|
||||||
|
if (unmapErr != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to unmap UBLK device %s: %s, error: %v", devicePath, unmapOut, unmapErr)
|
||||||
|
}
|
||||||
|
for _, ext := range []string{"json", "pid"}
|
||||||
|
{
|
||||||
|
fn := stateDir + "vitastor-ublk-" + devicePath[10:] + "." + ext
|
||||||
|
err := os.Remove(fn)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to remove %s: %v", fn, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func system(program string, args ...string) ([]byte, []byte, error)
|
func system(program string, args ...string) ([]byte, []byte, error)
|
||||||
{
|
{
|
||||||
klog.Infof("Running "+program+" "+strings.Join(args, " "))
|
klog.Infof("Running "+program+" "+strings.Join(args, " "))
|
||||||
@@ -340,3 +479,43 @@ func systemCombined(program string, args ...string) ([]byte, error)
|
|||||||
}
|
}
|
||||||
return out.Bytes(), nil
|
return out.Bytes(), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func GetDeviceNameFromMount(mountPath string) (string, error)
|
||||||
|
{
|
||||||
|
// Use /proc/self/mountinfo to correctly parse bind mounts for block device files
|
||||||
|
mps, err := mount.ParseMountInfo("/proc/self/mountinfo")
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
slTarget, err := filepath.EvalSymlinks(mountPath)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
slTarget = mountPath
|
||||||
|
}
|
||||||
|
|
||||||
|
device := ""
|
||||||
|
for _, mp := range mps
|
||||||
|
{
|
||||||
|
if (mp.MountPoint == slTarget)
|
||||||
|
{
|
||||||
|
device = mp.Source
|
||||||
|
if (device[0] != '/' && mp.Root != "/")
|
||||||
|
{
|
||||||
|
// Handle {Source=udev Root=/vdb MountPoint=/var/lib/kubelet/tralaleylo/tralala}
|
||||||
|
for _, other := range mps
|
||||||
|
{
|
||||||
|
if (other.Root == "/" && other.Source == mp.Source)
|
||||||
|
{
|
||||||
|
device = other.MountPoint + mp.Root
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return device, nil
|
||||||
|
}
|
||||||
|
|||||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
|
|
||||||
docker build --build-arg DISTRO=debian --build-arg REL=bookworm -t vitastor-buildenv:bookworm -f vitastor-buildenv.Dockerfile .
|
docker build --build-arg DISTRO=debian --build-arg REL=bookworm -t vitastor-buildenv:bookworm -f vitastor-buildenv.Dockerfile .
|
||||||
docker run -i --rm -e REL=bookworm -v `dirname $0`/../:/root/vitastor vitastor-buildenv:bookworm /root/vitastor/debian/vitastor-build.sh
|
docker run -it --rm -e REL=bookworm -v `dirname $0`/../:/root/vitastor vitastor-buildenv:bookworm /root/vitastor/debian/vitastor-build.sh
|
||||||
|
|||||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
|
|
||||||
docker build --build-arg DISTRO=debian --build-arg REL=bullseye -t vitastor-buildenv:bullseye -f vitastor-buildenv.Dockerfile .
|
docker build --build-arg DISTRO=debian --build-arg REL=bullseye -t vitastor-buildenv:bullseye -f vitastor-buildenv.Dockerfile .
|
||||||
docker run -i --rm -e REL=bullseye -v `dirname $0`/../:/root/vitastor vitastor-buildenv:bullseye /root/vitastor/debian/vitastor-build.sh
|
docker run -it --rm -e REL=bullseye -v `dirname $0`/../:/root/vitastor vitastor-buildenv:bullseye /root/vitastor/debian/vitastor-build.sh
|
||||||
|
|||||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
|
|
||||||
docker build --build-arg DISTRO=debian --build-arg REL=buster -t vitastor-buildenv:buster -f vitastor-buildenv.Dockerfile .
|
docker build --build-arg DISTRO=debian --build-arg REL=buster -t vitastor-buildenv:buster -f vitastor-buildenv.Dockerfile .
|
||||||
docker run -i --rm -e REL=buster -v `dirname $0`/../:/root/vitastor vitastor-buildenv:buster /root/vitastor/debian/vitastor-build.sh
|
docker run -it --rm -e REL=buster -v `dirname $0`/../:/root/vitastor vitastor-buildenv:buster /root/vitastor/debian/vitastor-build.sh
|
||||||
|
|||||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
|
|
||||||
docker build --build-arg DISTRO=debian --build-arg REL=trixie -t vitastor-buildenv:trixie -f vitastor-buildenv.Dockerfile .
|
docker build --build-arg DISTRO=debian --build-arg REL=trixie -t vitastor-buildenv:trixie -f vitastor-buildenv.Dockerfile .
|
||||||
docker run -i --rm -e REL=trixie -v `dirname $0`/../:/root/vitastor vitastor-buildenv:trixie /root/vitastor/debian/vitastor-build.sh
|
docker run -it --rm -e REL=trixie -v `dirname $0`/../:/root/vitastor vitastor-buildenv:trixie /root/vitastor/debian/vitastor-build.sh
|
||||||
|
|||||||
+1
-1
@@ -2,4 +2,4 @@
|
|||||||
# Ubuntu 22.04 Jammy Jellyfish
|
# Ubuntu 22.04 Jammy Jellyfish
|
||||||
|
|
||||||
docker build --build-arg DISTRO=ubuntu --build-arg REL=jammy -t vitastor-buildenv:jammy -f vitastor-buildenv.Dockerfile .
|
docker build --build-arg DISTRO=ubuntu --build-arg REL=jammy -t vitastor-buildenv:jammy -f vitastor-buildenv.Dockerfile .
|
||||||
docker run -i --rm -e REL=jammy -v `dirname $0`/../:/root/vitastor vitastor-buildenv:jammy /root/vitastor/debian/vitastor-build.sh
|
docker run -it --rm -e REL=jammy -v `dirname $0`/../:/root/vitastor vitastor-buildenv:jammy /root/vitastor/debian/vitastor-build.sh
|
||||||
|
|||||||
+1
-1
@@ -2,4 +2,4 @@
|
|||||||
# 24.04 Noble Numbat
|
# 24.04 Noble Numbat
|
||||||
|
|
||||||
docker build --build-arg DISTRO=ubuntu --build-arg REL=noble -t vitastor-buildenv:noble -f vitastor-buildenv.Dockerfile .
|
docker build --build-arg DISTRO=ubuntu --build-arg REL=noble -t vitastor-buildenv:noble -f vitastor-buildenv.Dockerfile .
|
||||||
docker run -i --rm -e REL=noble -v `dirname $0`/../:/root/vitastor vitastor-buildenv:noble /root/vitastor/debian/vitastor-build.sh
|
docker run -it --rm -e REL=noble -v `dirname $0`/../:/root/vitastor vitastor-buildenv:noble /root/vitastor/debian/vitastor-build.sh
|
||||||
|
|||||||
+5
@@ -0,0 +1,5 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# 25.10 Questing quokka
|
||||||
|
|
||||||
|
docker build --build-arg DISTRO=ubuntu --build-arg REL=questing -t vitastor-buildenv:questing -f vitastor-buildenv.Dockerfile .
|
||||||
|
docker run -it --rm -e REL=questing -v `dirname $0`/../:/root/vitastor vitastor-buildenv:questing /root/vitastor/debian/vitastor-build.sh
|
||||||
+5
@@ -0,0 +1,5 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# 26.04 Resolute Raccoon
|
||||||
|
|
||||||
|
docker build --build-arg DISTRO=ubuntu --build-arg REL=resolute -t vitastor-buildenv:resolute -f vitastor-buildenv.Dockerfile .
|
||||||
|
docker run -it --rm -e REL=resolute -v `dirname $0`/../:/root/vitastor vitastor-buildenv:resolute /root/vitastor/debian/vitastor-build.sh
|
||||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
vitastor (2.3.0-1) unstable; urgency=medium
|
vitastor (3.0.6-1) unstable; urgency=medium
|
||||||
|
|
||||||
* Bugfixes
|
* Bugfixes
|
||||||
|
|
||||||
|
|||||||
Vendored
+1
-1
@@ -4,7 +4,7 @@ Priority: optional
|
|||||||
Maintainer: Vitaliy Filippov <vitalif@yourcmc.ru>
|
Maintainer: Vitaliy Filippov <vitalif@yourcmc.ru>
|
||||||
Build-Depends: debhelper, g++ (>= 8), libstdc++6 (>= 8),
|
Build-Depends: debhelper, g++ (>= 8), libstdc++6 (>= 8),
|
||||||
linux-libc-dev, libgoogle-perftools-dev, libjerasure-dev, libgf-complete-dev,
|
linux-libc-dev, libgoogle-perftools-dev, libjerasure-dev, libgf-complete-dev,
|
||||||
libibverbs-dev, libisal-dev, cmake, pkg-config, libnl-3-dev, libnl-genl-3-dev,
|
libibverbs-dev, librdmacm-dev, libisal-dev, cmake, pkg-config, libnl-3-dev, libnl-genl-3-dev,
|
||||||
node-bindings <!nocheck>, node-gyp, node-nan
|
node-bindings <!nocheck>, node-gyp, node-nan
|
||||||
Standards-Version: 4.5.0
|
Standards-Version: 4.5.0
|
||||||
Homepage: https://vitastor.io/
|
Homepage: https://vitastor.io/
|
||||||
|
|||||||
Vendored
+6
-1
@@ -44,7 +44,12 @@ curl -s https://git.yourcmc.ru/vitalif/antietcd/archive/master.tar.gz | tar -zx
|
|||||||
curl -s https://git.yourcmc.ru/vitalif/tinyraft/archive/master.tar.gz | tar -zx
|
curl -s https://git.yourcmc.ru/vitalif/tinyraft/archive/master.tar.gz | tar -zx
|
||||||
|
|
||||||
cd /root/vitastor/packages/vitastor-$REL
|
cd /root/vitastor/packages/vitastor-$REL
|
||||||
tar --sort=name --mtime='2020-01-01' --owner=0 --group=0 --exclude=debian -cJf vitastor_$VER.orig.tar.xz vitastor-$VER
|
if [[ "$REL" = "trixie" && -e ../vitastor-bookworm/vitastor_$VER.orig.tar.xz ]]; then
|
||||||
|
# Fucking shit, archives differ between bookworm (xz 5.4.1) and trixie (xz 5.8.1)
|
||||||
|
cp ../vitastor-bookworm/vitastor_$VER.orig.tar.xz .
|
||||||
|
else
|
||||||
|
tar --sort=name --mtime='2020-01-01' --owner=0 --group=0 --exclude=debian -cJf vitastor_$VER.orig.tar.xz vitastor-$VER
|
||||||
|
fi
|
||||||
cd vitastor-$VER
|
cd vitastor-$VER
|
||||||
DEBEMAIL="Vitaliy Filippov <vitalif@yourcmc.ru>" dch -D $REL -v "$FULLVER""$REL" "Rebuild for $REL"
|
DEBEMAIL="Vitaliy Filippov <vitalif@yourcmc.ru>" dch -D $REL -v "$FULLVER""$REL" "Rebuild for $REL"
|
||||||
DEB_BUILD_OPTIONS=nocheck dpkg-buildpackage --jobs=auto -sa
|
DEB_BUILD_OPTIONS=nocheck dpkg-buildpackage --jobs=auto -sa
|
||||||
|
|||||||
+2
-2
@@ -1,9 +1,9 @@
|
|||||||
# Build Docker image with Vitastor packages
|
# Build Docker image with Vitastor packages
|
||||||
|
|
||||||
FROM debian:bookworm
|
FROM debian:trixie
|
||||||
|
|
||||||
ADD etc/apt /etc/apt/
|
ADD etc/apt /etc/apt/
|
||||||
RUN apt-get update && apt-get -y install vitastor udev systemd qemu-system-x86 qemu-system-common qemu-block-extra qemu-utils jq nfs-common && apt-get clean
|
RUN apt-get update && apt-get -y install vitastor ibverbs-providers udev systemd qemu-system-x86 qemu-system-common qemu-block-extra qemu-utils jq nfs-common && apt-get clean
|
||||||
ADD sleep.sh /usr/bin/
|
ADD sleep.sh /usr/bin/
|
||||||
ADD install.sh /usr/bin/
|
ADD install.sh /usr/bin/
|
||||||
ADD scripts /opt/scripts/
|
ADD scripts /opt/scripts/
|
||||||
|
|||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
VITASTOR_VERSION ?= v2.3.0
|
VITASTOR_VERSION ?= v3.0.6
|
||||||
|
|
||||||
all: build push
|
all: build push
|
||||||
|
|
||||||
|
|||||||
@@ -1,3 +1,3 @@
|
|||||||
Package: *
|
Package: *
|
||||||
Pin: release n=bookworm-backports
|
Pin: release n=trixie-backports
|
||||||
Pin-Priority: 500
|
Pin-Priority: 500
|
||||||
|
|||||||
@@ -1,2 +1,2 @@
|
|||||||
deb http://vitastor.io/debian bookworm main
|
deb http://vitastor.io/debian trixie main
|
||||||
deb http://http.debian.net/debian/ bookworm-backports main
|
#deb http://http.debian.net/debian/ trixie-backports main
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ PartOf=vitastor.target
|
|||||||
[Service]
|
[Service]
|
||||||
Restart=always
|
Restart=always
|
||||||
EnvironmentFile=/etc/vitastor/docker.conf
|
EnvironmentFile=/etc/vitastor/docker.conf
|
||||||
ExecStart=bash -c 'docker run --rm -i -v /etc/vitastor:/etc/vitastor -v /dev:/dev -v /run:/run \
|
ExecStart=bash -c 'docker run --rm -i -v /etc/vitastor:/etc/vitastor -v /dev:/dev -v /run:/run -e SYSTEMD_IN_CHROOT=0 \
|
||||||
--security-opt seccomp=unconfined --privileged --pid=host --log-driver none --network host --name vitastor vitastor:$VITASTOR_VERSION \
|
--security-opt seccomp=unconfined --privileged --pid=host --log-driver none --network host --name vitastor vitastor:$VITASTOR_VERSION \
|
||||||
sleep.sh'
|
sleep.sh'
|
||||||
ExecStartPost=udevadm trigger
|
ExecStartPost=udevadm trigger
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
#
|
#
|
||||||
|
|
||||||
# Desired Vitastor version
|
# Desired Vitastor version
|
||||||
VITASTOR_VERSION=v2.3.0
|
VITASTOR_VERSION=v3.0.6
|
||||||
|
|
||||||
# Additional arguments for all containers
|
# Additional arguments for all containers
|
||||||
# For example, you may want to specify a custom logging driver here
|
# For example, you may want to specify a custom logging driver here
|
||||||
|
|||||||
+2
-3
@@ -2,8 +2,7 @@
|
|||||||
|
|
||||||
set -e
|
set -e
|
||||||
|
|
||||||
cp -urv /etc/default /host-etc/
|
cp -urv /etc/systemd/system/vitastor* /host-etc/systemd/system/
|
||||||
cp -urv /etc/systemd /host-etc/
|
cp -urv /etc/udev/rules.d /host-etc/udev/
|
||||||
cp -urv /etc/udev /host-etc/
|
|
||||||
cp -urnv /etc/vitastor /host-etc/
|
cp -urnv /etc/vitastor /host-etc/
|
||||||
cp -urnv /opt/scripts/* /host-bin/
|
cp -urnv /opt/scripts/* /host-bin/
|
||||||
|
|||||||
@@ -9,6 +9,7 @@
|
|||||||
These parameters apply to OSDs, are fixed at the moment of OSD drive
|
These parameters apply to OSDs, are fixed at the moment of OSD drive
|
||||||
initialization and can't be changed after it without losing data.
|
initialization and can't be changed after it without losing data.
|
||||||
|
|
||||||
|
- [meta_format](#meta_format)
|
||||||
- [data_device](#data_device)
|
- [data_device](#data_device)
|
||||||
- [meta_device](#meta_device)
|
- [meta_device](#meta_device)
|
||||||
- [journal_device](#journal_device)
|
- [journal_device](#journal_device)
|
||||||
@@ -27,6 +28,21 @@ initialization and can't be changed after it without losing data.
|
|||||||
- [data_csum_type](#data_csum_type)
|
- [data_csum_type](#data_csum_type)
|
||||||
- [csum_block_size](#csum_block_size)
|
- [csum_block_size](#csum_block_size)
|
||||||
|
|
||||||
|
## meta_format
|
||||||
|
|
||||||
|
- Type: integer
|
||||||
|
- Default: 3
|
||||||
|
|
||||||
|
OSD store implementation version and on-disk metadata format.
|
||||||
|
|
||||||
|
Three versions are currently supported: 3, 2 and 1.
|
||||||
|
- 3 the new log-structured store, it's overall faster, has lower Write
|
||||||
|
Amplification, which may be even close to 1 (i.e. almost no extra writes)
|
||||||
|
if your SSDs support atomic writes (see [atomic_write_size](osd.en.md#atomic_write_size)).
|
||||||
|
- 2 is the old stable store from Vitastor 0.9-2.x.
|
||||||
|
- 1 is the same old store but with a legacy metadata format from Vitastor
|
||||||
|
versions to up 0.8.x, without any support for checksums.
|
||||||
|
|
||||||
## data_device
|
## data_device
|
||||||
|
|
||||||
- Type: string
|
- Type: string
|
||||||
|
|||||||
@@ -10,6 +10,7 @@
|
|||||||
дисковые параметры, задаются в момент инициализации дисков OSD и не могут быть
|
дисковые параметры, задаются в момент инициализации дисков OSD и не могут быть
|
||||||
изменены после этого без потери данных.
|
изменены после этого без потери данных.
|
||||||
|
|
||||||
|
- [meta_format](#meta_format)
|
||||||
- [data_device](#data_device)
|
- [data_device](#data_device)
|
||||||
- [meta_device](#meta_device)
|
- [meta_device](#meta_device)
|
||||||
- [journal_device](#journal_device)
|
- [journal_device](#journal_device)
|
||||||
@@ -28,6 +29,23 @@
|
|||||||
- [data_csum_type](#data_csum_type)
|
- [data_csum_type](#data_csum_type)
|
||||||
- [csum_block_size](#csum_block_size)
|
- [csum_block_size](#csum_block_size)
|
||||||
|
|
||||||
|
## meta_format
|
||||||
|
|
||||||
|
- Тип: целое число
|
||||||
|
- Значение по умолчанию: 3
|
||||||
|
|
||||||
|
Версия реализации дискового хранилища OSD и дискового формата метаданных.
|
||||||
|
|
||||||
|
Поддерживаются три версии: 3, 2 и 1.
|
||||||
|
- 3 - новое лог-структурированное хранилище, в целом более быстрое, со
|
||||||
|
сниженным фактором амплификации записи, который может составлять около 1
|
||||||
|
(то есть, практически без лишней служебной записи), если ваши SSD
|
||||||
|
поддерживают атомарную запись (см. [atomic_write_size](osd.ru.md#atomic_write_size)).
|
||||||
|
- 2 - старое стабильное хранилище из версий Vitastor 0.9-2.x.
|
||||||
|
- 1 - то же самое стабильное хранилище, но с ещё более старым форматом
|
||||||
|
метаданных из версий Vitastor до 0.8.x, без какой-либо поддержки
|
||||||
|
контрольных сумм.
|
||||||
|
|
||||||
## data_device
|
## data_device
|
||||||
|
|
||||||
- Тип: строка
|
- Тип: строка
|
||||||
|
|||||||
+17
-28
@@ -22,7 +22,6 @@ between clients, OSDs and etcd.
|
|||||||
- [rdma_max_msg](#rdma_max_msg)
|
- [rdma_max_msg](#rdma_max_msg)
|
||||||
- [rdma_max_recv](#rdma_max_recv)
|
- [rdma_max_recv](#rdma_max_recv)
|
||||||
- [rdma_max_send](#rdma_max_send)
|
- [rdma_max_send](#rdma_max_send)
|
||||||
- [rdma_odp](#rdma_odp)
|
|
||||||
- [peer_connect_interval](#peer_connect_interval)
|
- [peer_connect_interval](#peer_connect_interval)
|
||||||
- [peer_connect_timeout](#peer_connect_timeout)
|
- [peer_connect_timeout](#peer_connect_timeout)
|
||||||
- [osd_idle_timeout](#osd_idle_timeout)
|
- [osd_idle_timeout](#osd_idle_timeout)
|
||||||
@@ -102,11 +101,6 @@ found or if `osd_network` is not specified. Auto-selection is also
|
|||||||
unsupported with old libibverbs < v32, like in Debian 10 Buster or
|
unsupported with old libibverbs < v32, like in Debian 10 Buster or
|
||||||
CentOS 7.
|
CentOS 7.
|
||||||
|
|
||||||
Vitastor supports all adapters, even ones without ODP support, like
|
|
||||||
Mellanox ConnectX-3 and non-Mellanox cards. Versions up to Vitastor
|
|
||||||
1.2.0 required ODP which is only present in Mellanox ConnectX >= 4.
|
|
||||||
See also [rdma_odp](#rdma_odp).
|
|
||||||
|
|
||||||
Run `ibv_devinfo -v` as root to list available RDMA devices and their
|
Run `ibv_devinfo -v` as root to list available RDMA devices and their
|
||||||
features.
|
features.
|
||||||
|
|
||||||
@@ -116,6 +110,23 @@ the manual of your network vendor for details about setting up the switch
|
|||||||
for RoCEv2 correctly. Usually it means setting up Lossless Ethernet with
|
for RoCEv2 correctly. Usually it means setting up Lossless Ethernet with
|
||||||
PFC (Priority Flow Control) and ECN (Explicit Congestion Notification).
|
PFC (Priority Flow Control) and ECN (Explicit Congestion Notification).
|
||||||
|
|
||||||
|
Vitastor supports all adapters, even ones without ODP (On-Demand Paging)
|
||||||
|
support, like Mellanox ConnectX-3 and non-Mellanox cards. ODP is only present
|
||||||
|
in Mellanox ConnectX >= 4 adapters and allows to skip memory registration
|
||||||
|
for RDMA and thus, in theory, avoid memory copying.
|
||||||
|
|
||||||
|
Versions up to Vitastor 1.2.0 required ODP, then it was disabled by default,
|
||||||
|
but it was still supported up to 3.0.3. Now ODP support is removed because it
|
||||||
|
actually only hurts performance: an example 3-node cluster with 8 NVMe in each
|
||||||
|
node and 2*25 GBit/s ConnectX-6 RDMA network pushed 3950000 read iops without
|
||||||
|
ODP, but only 239000 iops with ODP.
|
||||||
|
|
||||||
|
This happens because Mellanox ODP implementation seems to be based on
|
||||||
|
message retransmissions when the adapter doesn't know about the buffer yet -
|
||||||
|
it likely uses standard "RNR retransmissions" (RNR = receiver not ready)
|
||||||
|
which is generally slow in RDMA/RoCE networks. Here's a presentation about
|
||||||
|
it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
||||||
|
|
||||||
## rdma_port_num
|
## rdma_port_num
|
||||||
|
|
||||||
- Type: integer
|
- Type: integer
|
||||||
@@ -187,28 +198,6 @@ less than `rdma_max_recv` so the receiving side doesn't run out of buffers.
|
|||||||
Doesn't affect memory usage - additional memory isn't allocated for send
|
Doesn't affect memory usage - additional memory isn't allocated for send
|
||||||
operations.
|
operations.
|
||||||
|
|
||||||
## rdma_odp
|
|
||||||
|
|
||||||
- Type: boolean
|
|
||||||
- Default: false
|
|
||||||
|
|
||||||
Use RDMA with On-Demand Paging. ODP is currently only available on Mellanox
|
|
||||||
ConnectX-4 and newer adapters. ODP allows to not register memory explicitly
|
|
||||||
for RDMA adapter to be able to use it. This, in turn, allows to skip memory
|
|
||||||
copying during sending. One would think this should improve performance, but
|
|
||||||
**in reality** RDMA performance with ODP is **drastically** worse. Example
|
|
||||||
3-node cluster with 8 NVMe in each node and 2*25 GBit/s ConnectX-6 RDMA network
|
|
||||||
without ODP pushes 3950000 read iops, but only 239000 iops with ODP...
|
|
||||||
|
|
||||||
This happens because Mellanox ODP implementation seems to be based on
|
|
||||||
message retransmissions when the adapter doesn't know about the buffer yet -
|
|
||||||
it likely uses standard "RNR retransmissions" (RNR = receiver not ready)
|
|
||||||
which is generally slow in RDMA/RoCE networks. Here's a presentation about
|
|
||||||
it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
|
||||||
|
|
||||||
ODP support is retained in the code just in case a good ODP implementation
|
|
||||||
appears one day.
|
|
||||||
|
|
||||||
## peer_connect_interval
|
## peer_connect_interval
|
||||||
|
|
||||||
- Type: seconds
|
- Type: seconds
|
||||||
|
|||||||
+18
-30
@@ -22,7 +22,6 @@
|
|||||||
- [rdma_max_msg](#rdma_max_msg)
|
- [rdma_max_msg](#rdma_max_msg)
|
||||||
- [rdma_max_recv](#rdma_max_recv)
|
- [rdma_max_recv](#rdma_max_recv)
|
||||||
- [rdma_max_send](#rdma_max_send)
|
- [rdma_max_send](#rdma_max_send)
|
||||||
- [rdma_odp](#rdma_odp)
|
|
||||||
- [peer_connect_interval](#peer_connect_interval)
|
- [peer_connect_interval](#peer_connect_interval)
|
||||||
- [peer_connect_timeout](#peer_connect_timeout)
|
- [peer_connect_timeout](#peer_connect_timeout)
|
||||||
- [osd_idle_timeout](#osd_idle_timeout)
|
- [osd_idle_timeout](#osd_idle_timeout)
|
||||||
@@ -101,12 +100,6 @@ RoCEv1/RoCEv2, и даже позволяет полностью отключи
|
|||||||
не задана. Также автовыбор не поддерживается со старыми версиями библиотеки
|
не задана. Также автовыбор не поддерживается со старыми версиями библиотеки
|
||||||
libibverbs < v32, например в Debian 10 Buster или CentOS 7.
|
libibverbs < v32, например в Debian 10 Buster или CentOS 7.
|
||||||
|
|
||||||
Vitastor поддерживает все модели адаптеров, включая те, у которых
|
|
||||||
нет поддержки ODP, то есть вы можете использовать RDMA с ConnectX-3 и
|
|
||||||
картами производства не Mellanox. Версии Vitastor до 1.2.0 включительно
|
|
||||||
требовали ODP, который есть только на Mellanox ConnectX 4 и более новых.
|
|
||||||
См. также [rdma_odp](#rdma_odp).
|
|
||||||
|
|
||||||
Запустите `ibv_devinfo -v` от имени суперпользователя, чтобы посмотреть
|
Запустите `ibv_devinfo -v` от имени суперпользователя, чтобы посмотреть
|
||||||
список доступных RDMA-устройств, их параметры и возможности.
|
список доступных RDMA-устройств, их параметры и возможности.
|
||||||
|
|
||||||
@@ -117,6 +110,24 @@ Vitastor поддерживает все модели адаптеров, вкл
|
|||||||
подразумевает настройку сети без потерь на основе PFC (Priority Flow
|
подразумевает настройку сети без потерь на основе PFC (Priority Flow
|
||||||
Control) и ECN (Explicit Congestion Notification).
|
Control) и ECN (Explicit Congestion Notification).
|
||||||
|
|
||||||
|
Vitastor поддерживает все модели адаптеров, включая те, у которых нет
|
||||||
|
поддержки ODP (On-Demand Paging), например, ConnectX-3 и карты производства
|
||||||
|
не Mellanox. Функция ODP доступна только на адаптерах Mellanox ConnectX-4 и
|
||||||
|
более новых и позволяет не регистрировать память для её использования RDMA-картой,
|
||||||
|
благодаря чему в теории можно избежать лишних копирований памяти.
|
||||||
|
|
||||||
|
Версии Vitastor до 1.2.0 включительно требовали ODP, потом функция был отключена
|
||||||
|
по умолчанию, но поддерживалась вплоть до версии 3.0.3. Сейчас поддержка ODP
|
||||||
|
полностью удалена, так как на самом деле она только портит производительность:
|
||||||
|
например, на 3-узловом кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на
|
||||||
|
чтение с RDMA без ODP удаётся снять 3950000 iops, а с ODP - всего 239000 iops.
|
||||||
|
|
||||||
|
Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и
|
||||||
|
основана на повторной передаче сообщений, когда карте не известен буфер -
|
||||||
|
вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready).
|
||||||
|
А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука.
|
||||||
|
Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
||||||
|
|
||||||
## rdma_port_num
|
## rdma_port_num
|
||||||
|
|
||||||
- Тип: целое число
|
- Тип: целое число
|
||||||
@@ -192,29 +203,6 @@ OSD в любом случае согласовывают реальное зн
|
|||||||
Не влияет на потребление памяти - дополнительная память на операции отправки
|
Не влияет на потребление памяти - дополнительная память на операции отправки
|
||||||
не выделяется.
|
не выделяется.
|
||||||
|
|
||||||
## rdma_odp
|
|
||||||
|
|
||||||
- Тип: булево (да/нет)
|
|
||||||
- Значение по умолчанию: false
|
|
||||||
|
|
||||||
Использовать RDMA с On-Demand Paging. ODP - функция, доступная пока что
|
|
||||||
исключительно на адаптерах Mellanox ConnectX-4 и более новых. ODP позволяет
|
|
||||||
не регистрировать память для её использования RDMA-картой. Благодаря этому
|
|
||||||
можно не копировать данные при отправке их в сеть и, казалось бы, это должно
|
|
||||||
улучшать производительность - но **по факту** получается так, что
|
|
||||||
производительность только ухудшается, причём сильно. Пример - на 3-узловом
|
|
||||||
кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на чтение с RDMA без ODP
|
|
||||||
удаётся снять 3950000 iops, а с ODP - всего 239000 iops...
|
|
||||||
|
|
||||||
Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и
|
|
||||||
основана на повторной передаче сообщений, когда карте не известен буфер -
|
|
||||||
вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready).
|
|
||||||
А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука.
|
|
||||||
Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
|
||||||
|
|
||||||
Возможность использования ODP сохранена в коде на случай, если вдруг в один
|
|
||||||
прекрасный день появится хорошая реализация ODP.
|
|
||||||
|
|
||||||
## peer_connect_interval
|
## peer_connect_interval
|
||||||
|
|
||||||
- Тип: секунды
|
- Тип: секунды
|
||||||
|
|||||||
+95
-8
@@ -38,6 +38,7 @@ with an OSD restart or, for some of them, even without restarting by updating co
|
|||||||
- [journal_io](#journal_io)
|
- [journal_io](#journal_io)
|
||||||
- [journal_sector_buffer_count](#journal_sector_buffer_count)
|
- [journal_sector_buffer_count](#journal_sector_buffer_count)
|
||||||
- [journal_no_same_sector_overwrites](#journal_no_same_sector_overwrites)
|
- [journal_no_same_sector_overwrites](#journal_no_same_sector_overwrites)
|
||||||
|
- [skip_corrupted_meta_entries](#skip_corrupted_meta_entries)
|
||||||
- [throttle_small_writes](#throttle_small_writes)
|
- [throttle_small_writes](#throttle_small_writes)
|
||||||
- [throttle_target_iops](#throttle_target_iops)
|
- [throttle_target_iops](#throttle_target_iops)
|
||||||
- [throttle_target_mbs](#throttle_target_mbs)
|
- [throttle_target_mbs](#throttle_target_mbs)
|
||||||
@@ -65,6 +66,10 @@ with an OSD restart or, for some of them, even without restarting by updating co
|
|||||||
- [allow_net_split](#allow_net_split)
|
- [allow_net_split](#allow_net_split)
|
||||||
- [enable_pg_locks](#enable_pg_locks)
|
- [enable_pg_locks](#enable_pg_locks)
|
||||||
- [pg_lock_retry_interval_ms](#pg_lock_retry_interval_ms)
|
- [pg_lock_retry_interval_ms](#pg_lock_retry_interval_ms)
|
||||||
|
- [atomic_write_size](#atomic_write_size)
|
||||||
|
- [use_atomic_flag](#use_atomic_flag)
|
||||||
|
- [pg_reshard_chunk_size](#pg_reshard_chunk_size)
|
||||||
|
- [pg_reshard_chunk_pause_ms](#pg_reshard_chunk_pause_ms)
|
||||||
|
|
||||||
## bind_address
|
## bind_address
|
||||||
|
|
||||||
@@ -275,13 +280,19 @@ Maximum number of journal flushers (see above min_flusher_count).
|
|||||||
- Type: boolean
|
- Type: boolean
|
||||||
- Default: true
|
- Default: true
|
||||||
|
|
||||||
This parameter makes Vitastor always keep metadata area of the block device
|
Only for the old store ([meta_format](layout-osd.en.md#meta_format) 2).
|
||||||
in memory. It's required for good performance because it allows to avoid
|
|
||||||
additional read-modify-write cycles during metadata modifications. Metadata
|
This parameter makes Vitastor keep a copy of metadata area in memory as it is
|
||||||
area size is currently roughly 224 MB per 1 TB of data. You can turn it off
|
on disk, in addition to the metadata database. When the option is enabled, every
|
||||||
to reduce memory usage by this value, but it will hurt performance. This
|
metadata entry is effectively stored in RAM twice. It's required for good performance
|
||||||
restriction is likely to be removed in the future along with the upgrade
|
because it allows to avoid additional read-modify-write cycles during metadata
|
||||||
of the metadata storage scheme.
|
modifications. Metadata area size with the old store is roughly 224 MB per 1 TB
|
||||||
|
of data. You can turn the option off to reduce memory usage by this value, but
|
||||||
|
it will reduce performance.
|
||||||
|
|
||||||
|
For the new store ([meta_format](layout-osd.en.md#meta_format) 3), the option
|
||||||
|
may be changed in the future to support operation without loading full metadata
|
||||||
|
database in memory.
|
||||||
|
|
||||||
## inmemory_journal
|
## inmemory_journal
|
||||||
|
|
||||||
@@ -360,6 +371,8 @@ blocks. The only situation when you should increase it to a larger value
|
|||||||
is when you enable journal_no_same_sector_overwrites. In this case set
|
is when you enable journal_no_same_sector_overwrites. In this case set
|
||||||
it to, for example, 1024.
|
it to, for example, 1024.
|
||||||
|
|
||||||
|
Not applicable to the new store ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
|
||||||
## journal_no_same_sector_overwrites
|
## journal_no_same_sector_overwrites
|
||||||
|
|
||||||
- Type: boolean
|
- Type: boolean
|
||||||
@@ -373,6 +386,17 @@ journal after writing it instead of possibly overwriting it the second time.
|
|||||||
|
|
||||||
Most (99%) other SSDs don't need this option.
|
Most (99%) other SSDs don't need this option.
|
||||||
|
|
||||||
|
Not applicable to the new store ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
|
||||||
|
## skip_corrupted_meta_entries
|
||||||
|
|
||||||
|
- Type: boolean
|
||||||
|
- Default: false
|
||||||
|
|
||||||
|
Only for the new store ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
Allow OSD to start when some metadata entries or blocks are corrupted by
|
||||||
|
skipping them. Should be only used as an emergency measure.
|
||||||
|
|
||||||
## throttle_small_writes
|
## throttle_small_writes
|
||||||
|
|
||||||
- Type: boolean
|
- Type: boolean
|
||||||
@@ -491,7 +515,7 @@ Can be used to slow down scrubbing if it affects user load too much.
|
|||||||
## scrub_list_limit
|
## scrub_list_limit
|
||||||
|
|
||||||
- Type: integer
|
- Type: integer
|
||||||
- Default: 1000
|
- Default: 262144
|
||||||
- Can be changed online: yes
|
- Can be changed online: yes
|
||||||
|
|
||||||
Number of objects to list in one listing operation during scrub.
|
Number of objects to list in one listing operation during scrub.
|
||||||
@@ -666,3 +690,66 @@ Use this parameter to enable or disable this function for all pools.
|
|||||||
- Default: 100
|
- Default: 100
|
||||||
|
|
||||||
Retry interval for failed PG lock attempts.
|
Retry interval for failed PG lock attempts.
|
||||||
|
|
||||||
|
## atomic_write_size
|
||||||
|
|
||||||
|
- Type: integer
|
||||||
|
- Default: 4096
|
||||||
|
|
||||||
|
Maximum data device atomic write size allowed for OSD to use.
|
||||||
|
|
||||||
|
Atomic writes allow to reduce the Write Amplification factor with the new store
|
||||||
|
([meta_format](layout-osd.en.md#meta_format)=3) to almost 1 (i.e. almost no extra writes)
|
||||||
|
with replicated pools and reach the best possible write performance.
|
||||||
|
|
||||||
|
Default value is auto-detected during OSD initialization from
|
||||||
|
`/sys/block/xx/queue/atomic_write_max_bytes` or assumed to be 4096 bytes
|
||||||
|
because all known disks support 4 KB atomic writes. Auto-detection is only used for
|
||||||
|
NVMe disks because SAS disks require the explicit WRITE ATOMIC command which requires
|
||||||
|
RWF_ATOMIC (see below [#use_atomic_flag]) but that flag works incorrectly in current
|
||||||
|
Linux versions.
|
||||||
|
|
||||||
|
You can also check if your NVMe drives support atomic writes by running
|
||||||
|
the command `nvme id-ctrl /dev/nvme0n1 | grep awupf`. If the reported value,
|
||||||
|
plus 1, multiplied by the currently selected block size of the NVMe,
|
||||||
|
is more than 4 KB, then the new store can utilize it for better performance.
|
||||||
|
The only drives known to support it currently are [Micron and Kioxia](../intro/quickstart.en.md).
|
||||||
|
|
||||||
|
Atomic writes allow to skip double data writes in replicated pools, thus
|
||||||
|
reducing Write Amplification and improving write performance up to 2 times.
|
||||||
|
|
||||||
|
## use_atomic_flag
|
||||||
|
|
||||||
|
- Type: boolean
|
||||||
|
|
||||||
|
This option controls whether Vitastor OSDs use RWF_ATOMIC write flag with atomic writes.
|
||||||
|
This flag is supported since Linux 6.11 and adds some safety to atomic writes - the kernel
|
||||||
|
guarantees to not fragment write requests with it and also to check them against the actual
|
||||||
|
device atomic write capabilities.
|
||||||
|
|
||||||
|
However, the option is disabled by default because the flag is currently UNUSABLE - Linux
|
||||||
|
incorrectly requires writes with that flag to be of power-of-2 length and length-aligned.
|
||||||
|
I.e., for example, 12 KB writes and not-8-KB aligned 8 KB writes are forbidden by the kernel,
|
||||||
|
even though the NVMe specification allows them.
|
||||||
|
|
||||||
|
For NVMe disks with `scheduler=none` writes aren't fragmented anyway so it's not a big deal.
|
||||||
|
However, you can rebuild your kernel with [this patch](../../patches/linux-fix-atomic-write-checks.diff)
|
||||||
|
and turn this option on. It will make your atomic writes a bit safer.
|
||||||
|
|
||||||
|
## pg_reshard_chunk_size
|
||||||
|
|
||||||
|
- Type: integer
|
||||||
|
- Default: 100000
|
||||||
|
|
||||||
|
Pool PG count change is a CPU-intensive operation because OSDs store the full object database
|
||||||
|
in memory and have to move all entries between old and new PGs. Thus it's performed in chunks,
|
||||||
|
with pauses between chunks to prevent blocking OSD's event loop and other clients' operations.
|
||||||
|
This option sets the maximum number of object is a chunk. Moving 100k objects usually takes
|
||||||
|
50-100ms. Chunk size equal to 0 means unlimited.
|
||||||
|
|
||||||
|
## pg_reshard_chunk_pause_ms
|
||||||
|
|
||||||
|
- Type: milliseconds
|
||||||
|
- Default: 100
|
||||||
|
|
||||||
|
This option sets the interval between handling two PG count change chunks.
|
||||||
|
|||||||
+102
-8
@@ -39,6 +39,7 @@
|
|||||||
- [journal_io](#journal_io)
|
- [journal_io](#journal_io)
|
||||||
- [journal_sector_buffer_count](#journal_sector_buffer_count)
|
- [journal_sector_buffer_count](#journal_sector_buffer_count)
|
||||||
- [journal_no_same_sector_overwrites](#journal_no_same_sector_overwrites)
|
- [journal_no_same_sector_overwrites](#journal_no_same_sector_overwrites)
|
||||||
|
- [skip_corrupted_meta_entries](#skip_corrupted_meta_entries)
|
||||||
- [throttle_small_writes](#throttle_small_writes)
|
- [throttle_small_writes](#throttle_small_writes)
|
||||||
- [throttle_target_iops](#throttle_target_iops)
|
- [throttle_target_iops](#throttle_target_iops)
|
||||||
- [throttle_target_mbs](#throttle_target_mbs)
|
- [throttle_target_mbs](#throttle_target_mbs)
|
||||||
@@ -66,6 +67,10 @@
|
|||||||
- [allow_net_split](#allow_net_split)
|
- [allow_net_split](#allow_net_split)
|
||||||
- [enable_pg_locks](#enable_pg_locks)
|
- [enable_pg_locks](#enable_pg_locks)
|
||||||
- [pg_lock_retry_interval_ms](#pg_lock_retry_interval_ms)
|
- [pg_lock_retry_interval_ms](#pg_lock_retry_interval_ms)
|
||||||
|
- [atomic_write_size](#atomic_write_size)
|
||||||
|
- [use_atomic_flag](#use_atomic_flag)
|
||||||
|
- [pg_reshard_chunk_size](#pg_reshard_chunk_size)
|
||||||
|
- [pg_reshard_chunk_pause_ms](#pg_reshard_chunk_pause_ms)
|
||||||
|
|
||||||
## bind_address
|
## bind_address
|
||||||
|
|
||||||
@@ -283,13 +288,19 @@ Flusher - это микро-поток (корутина), которая коп
|
|||||||
- Тип: булево (да/нет)
|
- Тип: булево (да/нет)
|
||||||
- Значение по умолчанию: true
|
- Значение по умолчанию: true
|
||||||
|
|
||||||
Данный параметр заставляет Vitastor всегда держать область метаданных диска
|
Только для старого хранилища ([meta_format](layout-osd.en.md#meta_format) 2).
|
||||||
в памяти. Это нужно, чтобы избегать дополнительных операций чтения с диска
|
|
||||||
при записи. Размер области метаданных на данный момент составляет примерно
|
Данный параметр заставляет Vitastor всегда держать копию области метаданных
|
||||||
224 МБ на 1 ТБ данных. При включении потребление памяти снизится примерно
|
в памяти в том же виде, как она лежит на диске, в дополнение к БД метаданных.
|
||||||
на эту величину, но при этом также снизится и производительность. В будущем,
|
То есть, с включённой опцией каждая запись метаданных хранится в памяти дважды.
|
||||||
после обновления схемы хранения метаданных, это ограничение, скорее всего,
|
Это нужно, чтобы избегать дополнительных операций чтения с диска при записи.
|
||||||
будет ликвидировано.
|
Размер области метаданных в старом хранилище составляет примерно 224 МБ на
|
||||||
|
1 ТБ данных. Вы можете отключить опцию, чтобы снизить потребление памяти
|
||||||
|
примерно на эту величину, но при этом также снизится и производительность.
|
||||||
|
|
||||||
|
Для нового хранилища ([meta_format](layout-osd.en.md#meta_format) 3) опция,
|
||||||
|
возможно, будет переработана в будущем для поддержки работы без полной
|
||||||
|
загрузки метаданных в памяти.
|
||||||
|
|
||||||
## inmemory_journal
|
## inmemory_journal
|
||||||
|
|
||||||
@@ -372,6 +383,8 @@ fsync небезопасным даже с режимом "directsync".
|
|||||||
нужно менять - это если вы включаете journal_no_same_sector_overwrites. В
|
нужно менять - это если вы включаете journal_no_same_sector_overwrites. В
|
||||||
этом случае установите данный параметр, например, в 1024.
|
этом случае установите данный параметр, например, в 1024.
|
||||||
|
|
||||||
|
Неприменимо к новому хранилищу ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
|
||||||
## journal_no_same_sector_overwrites
|
## journal_no_same_sector_overwrites
|
||||||
|
|
||||||
- Тип: булево (да/нет)
|
- Тип: булево (да/нет)
|
||||||
@@ -387,6 +400,18 @@ fsync небезопасным даже с режимом "directsync".
|
|||||||
|
|
||||||
Почти все другие SSD (99% моделей) не требуют данной опции.
|
Почти все другие SSD (99% моделей) не требуют данной опции.
|
||||||
|
|
||||||
|
Неприменимо к новому хранилищу ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
|
||||||
|
## skip_corrupted_meta_entries
|
||||||
|
|
||||||
|
- Тип: булево (да/нет)
|
||||||
|
- Значение по умолчанию: false
|
||||||
|
|
||||||
|
Только для нового хранилища ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
Разрешить OSD запускаться, даже если часть блоков или записей метаданных
|
||||||
|
повреждена, пропуская их. Опция предназначена для использования только в
|
||||||
|
целях аварийного восстановления.
|
||||||
|
|
||||||
## throttle_small_writes
|
## throttle_small_writes
|
||||||
|
|
||||||
- Тип: булево (да/нет)
|
- Тип: булево (да/нет)
|
||||||
@@ -514,7 +539,7 @@ fsync небезопасным даже с режимом "directsync".
|
|||||||
## scrub_list_limit
|
## scrub_list_limit
|
||||||
|
|
||||||
- Тип: целое число
|
- Тип: целое число
|
||||||
- Значение по умолчанию: 1000
|
- Значение по умолчанию: 262144
|
||||||
- Можно менять на лету: да
|
- Можно менять на лету: да
|
||||||
|
|
||||||
Размер загружаемых за одну операцию списков объектов в процессе фоновой
|
Размер загружаемых за одну операцию списков объектов в процессе фоновой
|
||||||
@@ -699,3 +724,72 @@ pg_minsize OSD во время переключений, что может по
|
|||||||
- Значение по умолчанию: 100
|
- Значение по умолчанию: 100
|
||||||
|
|
||||||
Интервал повтора неудачных попыток блокировки PG.
|
Интервал повтора неудачных попыток блокировки PG.
|
||||||
|
|
||||||
|
## atomic_write_size
|
||||||
|
|
||||||
|
- Тип: целое число
|
||||||
|
- Значение по умолчанию: 4096
|
||||||
|
|
||||||
|
Максимальный размер атомарной записи на диск данных, который OSD разрешено использовать.
|
||||||
|
|
||||||
|
Поддержка атомарной записи позволяет снизить мультипликатор записи (Write Amplification)
|
||||||
|
на диск с новым хранилищем ([meta_format](layout-osd.ru.md#meta_format)=3)
|
||||||
|
практически до 1 (то есть, почти до нулевого объёма лишней записи) в реплицированных
|
||||||
|
пулах и достигнуть наилучшей возможной производительности записи.
|
||||||
|
|
||||||
|
Значение по умолчанию авто-определяется во время инициализации OSD из
|
||||||
|
`/sys/block/xx/queue/atomic_write_max_bytes` либо принимается равным 4096,
|
||||||
|
так как все известные диски поддерживают атомарную запись 4 КБ блоков.
|
||||||
|
Автоопределение применяется только для NVMe-дисков, так как SAS диски требуют
|
||||||
|
использования отдельной команды WRITE ATOMIC, а для неё нужен флаг RWF_ATOMIC
|
||||||
|
(см. ниже [#use_atomic_flag]), а он в текущих версиях Linux работает некорректно.
|
||||||
|
|
||||||
|
Вы также можете проверить, поддерживают ли ваши NVMe-диски атомарную запись,
|
||||||
|
с помощью команды `nvme id-ctrl /dev/nvme0n1 | grep awupf`. Если значение awupf
|
||||||
|
плюс 1, умноженное на текущий выбранный размер блока NVMe-диска, больше 4 КБ,
|
||||||
|
то новое хранилище может использовать атомарные записи для достижения лучшей
|
||||||
|
производительности. Единственные известные диски, которые поддерживают это сейчас -
|
||||||
|
[Micron и Kioxia](../intro/quickstart.ru.md).
|
||||||
|
|
||||||
|
Атомарная запись позволяет не использовать двойную запись данных (в журнал и на
|
||||||
|
устройство данных) в реплицированных пулах и таким образом снижает амплификацию
|
||||||
|
записи (объём служебной записи на диск) и улучшает производительность записи
|
||||||
|
вплоть до 2-х кратного прироста.
|
||||||
|
|
||||||
|
## use_atomic_flag
|
||||||
|
|
||||||
|
- Тип: булево (да/нет)
|
||||||
|
|
||||||
|
Данная опция контролирует использование Vitastor OSD флага RWF_ATOMIC при атомарной записи
|
||||||
|
блоков. Этот флаг поддерживается, начиная с версии ядра Linux 6.11 и добавляет немного корректности
|
||||||
|
атомарным записям - ядро гарантирует отсутствие фрагментации запросов записи с этим флагом и
|
||||||
|
проверяет их на соответствие реальным возможностям устройства.
|
||||||
|
|
||||||
|
Однако, данная опция по умолчанию отключена, так как флаг в текущих версиях Linux работает
|
||||||
|
абсолютно НЕКОРРЕКТНО - при нём Linux требует, чтобы запросы записи имели длину, равную
|
||||||
|
степени двойки и были выровнены на эту длину. То есть, например, 12 КБ запросы записи, а также
|
||||||
|
8 КБ запросы записи по не-кратному 8 КБ смещению запрещаются ядром, хотя спецификация NVMe их
|
||||||
|
разрешает.
|
||||||
|
|
||||||
|
Для NVMe-дисков с `scheduler=none` запросы записи и так не фрагментируются, так что это не так
|
||||||
|
уж и важно, однако вы можете пересобрать своё ядро с [этим патчем](../../patches/linux-fix-atomic-write-checks.diff)
|
||||||
|
и включить данную опцию. Это сделает вашу атомарную запись капельку безопаснее.
|
||||||
|
|
||||||
|
## pg_reshard_chunk_size
|
||||||
|
|
||||||
|
- Тип: целое число
|
||||||
|
- Значение по умолчанию: 100000
|
||||||
|
|
||||||
|
Изменение числа PG в пуле заметно загружает процессор, так как OSD хранят полную базу данных
|
||||||
|
объектов в памяти и им приходится перемещать все записи объектов между старыми и новыми PG.
|
||||||
|
Поэтому изменение применяется порциями, с паузами между порциями, чтобы не блокировать обработку
|
||||||
|
событий OSD и операции остальных клиентов. Данная опция задаёт максимальное число объектов
|
||||||
|
в порции. Перемещение 100 тысяч объектов (значение по умолчанию) обычно занимает порядка
|
||||||
|
50-100 миллисекунд. Значение опции 0 отключает лимит размера порции.
|
||||||
|
|
||||||
|
## pg_reshard_chunk_pause_ms
|
||||||
|
|
||||||
|
- Тип: миллисекунды
|
||||||
|
- Значение по умолчанию: 100
|
||||||
|
|
||||||
|
Данная опция задаёт интервал между обработкой двух порций изменения числа PG пулов.
|
||||||
|
|||||||
@@ -1,3 +1,28 @@
|
|||||||
|
- name: meta_format
|
||||||
|
type: int
|
||||||
|
default: 3
|
||||||
|
info: |
|
||||||
|
OSD store implementation version and on-disk metadata format.
|
||||||
|
|
||||||
|
Three versions are currently supported: 3, 2 and 1.
|
||||||
|
- 3 the new log-structured store, it's overall faster, has lower Write
|
||||||
|
Amplification, which may be even close to 1 (i.e. almost no extra writes)
|
||||||
|
if your SSDs support atomic writes (see [atomic_write_size](osd.en.md#atomic_write_size)).
|
||||||
|
- 2 is the old stable store from Vitastor 0.9-2.x.
|
||||||
|
- 1 is the same old store but with a legacy metadata format from Vitastor
|
||||||
|
versions to up 0.8.x, without any support for checksums.
|
||||||
|
info_ru: |
|
||||||
|
Версия реализации дискового хранилища OSD и дискового формата метаданных.
|
||||||
|
|
||||||
|
Поддерживаются три версии: 3, 2 и 1.
|
||||||
|
- 3 - новое лог-структурированное хранилище, в целом более быстрое, со
|
||||||
|
сниженным фактором амплификации записи, который может составлять около 1
|
||||||
|
(то есть, практически без лишней служебной записи), если ваши SSD
|
||||||
|
поддерживают атомарную запись (см. [atomic_write_size](osd.ru.md#atomic_write_size)).
|
||||||
|
- 2 - старое стабильное хранилище из версий Vitastor 0.9-2.x.
|
||||||
|
- 1 - то же самое стабильное хранилище, но с ещё более старым форматом
|
||||||
|
метаданных из версий Vitastor до 0.8.x, без какой-либо поддержки
|
||||||
|
контрольных сумм.
|
||||||
- name: data_device
|
- name: data_device
|
||||||
type: string
|
type: string
|
||||||
info: |
|
info: |
|
||||||
|
|||||||
+35
-50
@@ -84,11 +84,6 @@
|
|||||||
unsupported with old libibverbs < v32, like in Debian 10 Buster or
|
unsupported with old libibverbs < v32, like in Debian 10 Buster or
|
||||||
CentOS 7.
|
CentOS 7.
|
||||||
|
|
||||||
Vitastor supports all adapters, even ones without ODP support, like
|
|
||||||
Mellanox ConnectX-3 and non-Mellanox cards. Versions up to Vitastor
|
|
||||||
1.2.0 required ODP which is only present in Mellanox ConnectX >= 4.
|
|
||||||
See also [rdma_odp](#rdma_odp).
|
|
||||||
|
|
||||||
Run `ibv_devinfo -v` as root to list available RDMA devices and their
|
Run `ibv_devinfo -v` as root to list available RDMA devices and their
|
||||||
features.
|
features.
|
||||||
|
|
||||||
@@ -97,6 +92,23 @@
|
|||||||
the manual of your network vendor for details about setting up the switch
|
the manual of your network vendor for details about setting up the switch
|
||||||
for RoCEv2 correctly. Usually it means setting up Lossless Ethernet with
|
for RoCEv2 correctly. Usually it means setting up Lossless Ethernet with
|
||||||
PFC (Priority Flow Control) and ECN (Explicit Congestion Notification).
|
PFC (Priority Flow Control) and ECN (Explicit Congestion Notification).
|
||||||
|
|
||||||
|
Vitastor supports all adapters, even ones without ODP (On-Demand Paging)
|
||||||
|
support, like Mellanox ConnectX-3 and non-Mellanox cards. ODP is only present
|
||||||
|
in Mellanox ConnectX >= 4 adapters and allows to skip memory registration
|
||||||
|
for RDMA and thus, in theory, avoid memory copying.
|
||||||
|
|
||||||
|
Versions up to Vitastor 1.2.0 required ODP, then it was disabled by default,
|
||||||
|
but it was still supported up to 3.0.3. Now ODP support is removed because it
|
||||||
|
actually only hurts performance: an example 3-node cluster with 8 NVMe in each
|
||||||
|
node and 2*25 GBit/s ConnectX-6 RDMA network pushed 3950000 read iops without
|
||||||
|
ODP, but only 239000 iops with ODP.
|
||||||
|
|
||||||
|
This happens because Mellanox ODP implementation seems to be based on
|
||||||
|
message retransmissions when the adapter doesn't know about the buffer yet -
|
||||||
|
it likely uses standard "RNR retransmissions" (RNR = receiver not ready)
|
||||||
|
which is generally slow in RDMA/RoCE networks. Here's a presentation about
|
||||||
|
it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
||||||
info_ru: |
|
info_ru: |
|
||||||
Название RDMA-устройства для связи с Vitastor OSD (например, "rocep5s0f0").
|
Название RDMA-устройства для связи с Vitastor OSD (например, "rocep5s0f0").
|
||||||
Если не указано, Vitastor попробует найти RoCE-устройство, соответствующее
|
Если не указано, Vitastor попробует найти RoCE-устройство, соответствующее
|
||||||
@@ -105,12 +117,6 @@
|
|||||||
не задана. Также автовыбор не поддерживается со старыми версиями библиотеки
|
не задана. Также автовыбор не поддерживается со старыми версиями библиотеки
|
||||||
libibverbs < v32, например в Debian 10 Buster или CentOS 7.
|
libibverbs < v32, например в Debian 10 Buster или CentOS 7.
|
||||||
|
|
||||||
Vitastor поддерживает все модели адаптеров, включая те, у которых
|
|
||||||
нет поддержки ODP, то есть вы можете использовать RDMA с ConnectX-3 и
|
|
||||||
картами производства не Mellanox. Версии Vitastor до 1.2.0 включительно
|
|
||||||
требовали ODP, который есть только на Mellanox ConnectX 4 и более новых.
|
|
||||||
См. также [rdma_odp](#rdma_odp).
|
|
||||||
|
|
||||||
Запустите `ibv_devinfo -v` от имени суперпользователя, чтобы посмотреть
|
Запустите `ibv_devinfo -v` от имени суперпользователя, чтобы посмотреть
|
||||||
список доступных RDMA-устройств, их параметры и возможности.
|
список доступных RDMA-устройств, их параметры и возможности.
|
||||||
|
|
||||||
@@ -120,6 +126,24 @@
|
|||||||
коммутатора для RoCEv2 ищите в документации производителя. Обычно это
|
коммутатора для RoCEv2 ищите в документации производителя. Обычно это
|
||||||
подразумевает настройку сети без потерь на основе PFC (Priority Flow
|
подразумевает настройку сети без потерь на основе PFC (Priority Flow
|
||||||
Control) и ECN (Explicit Congestion Notification).
|
Control) и ECN (Explicit Congestion Notification).
|
||||||
|
|
||||||
|
Vitastor поддерживает все модели адаптеров, включая те, у которых нет
|
||||||
|
поддержки ODP (On-Demand Paging), например, ConnectX-3 и карты производства
|
||||||
|
не Mellanox. Функция ODP доступна только на адаптерах Mellanox ConnectX-4 и
|
||||||
|
более новых и позволяет не регистрировать память для её использования RDMA-картой,
|
||||||
|
благодаря чему в теории можно избежать лишних копирований памяти.
|
||||||
|
|
||||||
|
Версии Vitastor до 1.2.0 включительно требовали ODP, потом функция был отключена
|
||||||
|
по умолчанию, но поддерживалась вплоть до версии 3.0.3. Сейчас поддержка ODP
|
||||||
|
полностью удалена, так как на самом деле она только портит производительность:
|
||||||
|
например, на 3-узловом кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на
|
||||||
|
чтение с RDMA без ODP удаётся снять 3950000 iops, а с ODP - всего 239000 iops.
|
||||||
|
|
||||||
|
Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и
|
||||||
|
основана на повторной передаче сообщений, когда карте не известен буфер -
|
||||||
|
вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready).
|
||||||
|
А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука.
|
||||||
|
Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
||||||
- name: rdma_port_num
|
- name: rdma_port_num
|
||||||
type: int
|
type: int
|
||||||
info: |
|
info: |
|
||||||
@@ -218,45 +242,6 @@
|
|||||||
у принимающей стороны в процессе работы не заканчивались буферы на приём.
|
у принимающей стороны в процессе работы не заканчивались буферы на приём.
|
||||||
Не влияет на потребление памяти - дополнительная память на операции отправки
|
Не влияет на потребление памяти - дополнительная память на операции отправки
|
||||||
не выделяется.
|
не выделяется.
|
||||||
- name: rdma_odp
|
|
||||||
type: bool
|
|
||||||
default: false
|
|
||||||
online: false
|
|
||||||
info: |
|
|
||||||
Use RDMA with On-Demand Paging. ODP is currently only available on Mellanox
|
|
||||||
ConnectX-4 and newer adapters. ODP allows to not register memory explicitly
|
|
||||||
for RDMA adapter to be able to use it. This, in turn, allows to skip memory
|
|
||||||
copying during sending. One would think this should improve performance, but
|
|
||||||
**in reality** RDMA performance with ODP is **drastically** worse. Example
|
|
||||||
3-node cluster with 8 NVMe in each node and 2*25 GBit/s ConnectX-6 RDMA network
|
|
||||||
without ODP pushes 3950000 read iops, but only 239000 iops with ODP...
|
|
||||||
|
|
||||||
This happens because Mellanox ODP implementation seems to be based on
|
|
||||||
message retransmissions when the adapter doesn't know about the buffer yet -
|
|
||||||
it likely uses standard "RNR retransmissions" (RNR = receiver not ready)
|
|
||||||
which is generally slow in RDMA/RoCE networks. Here's a presentation about
|
|
||||||
it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
|
||||||
|
|
||||||
ODP support is retained in the code just in case a good ODP implementation
|
|
||||||
appears one day.
|
|
||||||
info_ru: |
|
|
||||||
Использовать RDMA с On-Demand Paging. ODP - функция, доступная пока что
|
|
||||||
исключительно на адаптерах Mellanox ConnectX-4 и более новых. ODP позволяет
|
|
||||||
не регистрировать память для её использования RDMA-картой. Благодаря этому
|
|
||||||
можно не копировать данные при отправке их в сеть и, казалось бы, это должно
|
|
||||||
улучшать производительность - но **по факту** получается так, что
|
|
||||||
производительность только ухудшается, причём сильно. Пример - на 3-узловом
|
|
||||||
кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на чтение с RDMA без ODP
|
|
||||||
удаётся снять 3950000 iops, а с ODP - всего 239000 iops...
|
|
||||||
|
|
||||||
Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и
|
|
||||||
основана на повторной передаче сообщений, когда карте не известен буфер -
|
|
||||||
вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready).
|
|
||||||
А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука.
|
|
||||||
Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
|
|
||||||
|
|
||||||
Возможность использования ODP сохранена в коде на случай, если вдруг в один
|
|
||||||
прекрасный день появится хорошая реализация ODP.
|
|
||||||
- name: peer_connect_interval
|
- name: peer_connect_interval
|
||||||
type: sec
|
type: sec
|
||||||
min: 1
|
min: 1
|
||||||
|
|||||||
+152
-15
@@ -253,21 +253,33 @@
|
|||||||
type: bool
|
type: bool
|
||||||
default: true
|
default: true
|
||||||
info: |
|
info: |
|
||||||
This parameter makes Vitastor always keep metadata area of the block device
|
Only for the old store ([meta_format](layout-osd.en.md#meta_format) 2).
|
||||||
in memory. It's required for good performance because it allows to avoid
|
|
||||||
additional read-modify-write cycles during metadata modifications. Metadata
|
This parameter makes Vitastor keep a copy of metadata area in memory as it is
|
||||||
area size is currently roughly 224 MB per 1 TB of data. You can turn it off
|
on disk, in addition to the metadata database. When the option is enabled, every
|
||||||
to reduce memory usage by this value, but it will hurt performance. This
|
metadata entry is effectively stored in RAM twice. It's required for good performance
|
||||||
restriction is likely to be removed in the future along with the upgrade
|
because it allows to avoid additional read-modify-write cycles during metadata
|
||||||
of the metadata storage scheme.
|
modifications. Metadata area size with the old store is roughly 224 MB per 1 TB
|
||||||
|
of data. You can turn the option off to reduce memory usage by this value, but
|
||||||
|
it will reduce performance.
|
||||||
|
|
||||||
|
For the new store ([meta_format](layout-osd.en.md#meta_format) 3), the option
|
||||||
|
may be changed in the future to support operation without loading full metadata
|
||||||
|
database in memory.
|
||||||
info_ru: |
|
info_ru: |
|
||||||
Данный параметр заставляет Vitastor всегда держать область метаданных диска
|
Только для старого хранилища ([meta_format](layout-osd.en.md#meta_format) 2).
|
||||||
в памяти. Это нужно, чтобы избегать дополнительных операций чтения с диска
|
|
||||||
при записи. Размер области метаданных на данный момент составляет примерно
|
Данный параметр заставляет Vitastor всегда держать копию области метаданных
|
||||||
224 МБ на 1 ТБ данных. При включении потребление памяти снизится примерно
|
в памяти в том же виде, как она лежит на диске, в дополнение к БД метаданных.
|
||||||
на эту величину, но при этом также снизится и производительность. В будущем,
|
То есть, с включённой опцией каждая запись метаданных хранится в памяти дважды.
|
||||||
после обновления схемы хранения метаданных, это ограничение, скорее всего,
|
Это нужно, чтобы избегать дополнительных операций чтения с диска при записи.
|
||||||
будет ликвидировано.
|
Размер области метаданных в старом хранилище составляет примерно 224 МБ на
|
||||||
|
1 ТБ данных. Вы можете отключить опцию, чтобы снизить потребление памяти
|
||||||
|
примерно на эту величину, но при этом также снизится и производительность.
|
||||||
|
|
||||||
|
Для нового хранилища ([meta_format](layout-osd.en.md#meta_format) 3) опция,
|
||||||
|
возможно, будет переработана в будущем для поддержки работы без полной
|
||||||
|
загрузки метаданных в памяти.
|
||||||
- name: inmemory_journal
|
- name: inmemory_journal
|
||||||
type: bool
|
type: bool
|
||||||
default: true
|
default: true
|
||||||
@@ -386,11 +398,15 @@
|
|||||||
blocks. The only situation when you should increase it to a larger value
|
blocks. The only situation when you should increase it to a larger value
|
||||||
is when you enable journal_no_same_sector_overwrites. In this case set
|
is when you enable journal_no_same_sector_overwrites. In this case set
|
||||||
it to, for example, 1024.
|
it to, for example, 1024.
|
||||||
|
|
||||||
|
Not applicable to the new store ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
info_ru: |
|
info_ru: |
|
||||||
Максимальное число буферов, разрешённых для использования под записываемые
|
Максимальное число буферов, разрешённых для использования под записываемые
|
||||||
в журнал блоки метаданных. Единственная ситуация, в которой этот параметр
|
в журнал блоки метаданных. Единственная ситуация, в которой этот параметр
|
||||||
нужно менять - это если вы включаете journal_no_same_sector_overwrites. В
|
нужно менять - это если вы включаете journal_no_same_sector_overwrites. В
|
||||||
этом случае установите данный параметр, например, в 1024.
|
этом случае установите данный параметр, например, в 1024.
|
||||||
|
|
||||||
|
Неприменимо к новому хранилищу ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
- name: journal_no_same_sector_overwrites
|
- name: journal_no_same_sector_overwrites
|
||||||
type: bool
|
type: bool
|
||||||
default: false
|
default: false
|
||||||
@@ -402,6 +418,8 @@
|
|||||||
journal after writing it instead of possibly overwriting it the second time.
|
journal after writing it instead of possibly overwriting it the second time.
|
||||||
|
|
||||||
Most (99%) other SSDs don't need this option.
|
Most (99%) other SSDs don't need this option.
|
||||||
|
|
||||||
|
Not applicable to the new store ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
info_ru: |
|
info_ru: |
|
||||||
Включайте данную опцию для SSD вроде Intel D3-S4510 и D3-S4610, которые
|
Включайте данную опцию для SSD вроде Intel D3-S4510 и D3-S4610, которые
|
||||||
ОЧЕНЬ не любят, когда ПО перезаписывает один и тот же сектор несколько раз
|
ОЧЕНЬ не любят, когда ПО перезаписывает один и тот же сектор несколько раз
|
||||||
@@ -412,6 +430,20 @@
|
|||||||
самого сектора.
|
самого сектора.
|
||||||
|
|
||||||
Почти все другие SSD (99% моделей) не требуют данной опции.
|
Почти все другие SSD (99% моделей) не требуют данной опции.
|
||||||
|
|
||||||
|
Неприменимо к новому хранилищу ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
- name: skip_corrupted_meta_entries
|
||||||
|
type: bool
|
||||||
|
default: false
|
||||||
|
info: |
|
||||||
|
Only for the new store ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
Allow OSD to start when some metadata entries or blocks are corrupted by
|
||||||
|
skipping them. Should be only used as an emergency measure.
|
||||||
|
info_ru: |
|
||||||
|
Только для нового хранилища ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
Разрешить OSD запускаться, даже если часть блоков или записей метаданных
|
||||||
|
повреждена, пропуская их. Опция предназначена для использования только в
|
||||||
|
целях аварийного восстановления.
|
||||||
- name: throttle_small_writes
|
- name: throttle_small_writes
|
||||||
type: bool
|
type: bool
|
||||||
default: false
|
default: false
|
||||||
@@ -566,7 +598,7 @@
|
|||||||
сильно влияет на пользовательскую нагрузку.
|
сильно влияет на пользовательскую нагрузку.
|
||||||
- name: scrub_list_limit
|
- name: scrub_list_limit
|
||||||
type: int
|
type: int
|
||||||
default: 1000
|
default: 262144
|
||||||
online: true
|
online: true
|
||||||
info: |
|
info: |
|
||||||
Number of objects to list in one listing operation during scrub.
|
Number of objects to list in one listing operation during scrub.
|
||||||
@@ -801,3 +833,108 @@
|
|||||||
default: 100
|
default: 100
|
||||||
info: Retry interval for failed PG lock attempts.
|
info: Retry interval for failed PG lock attempts.
|
||||||
info_ru: Интервал повтора неудачных попыток блокировки PG.
|
info_ru: Интервал повтора неудачных попыток блокировки PG.
|
||||||
|
- name: atomic_write_size
|
||||||
|
type: int
|
||||||
|
default: 4096
|
||||||
|
info: |
|
||||||
|
Maximum data device atomic write size allowed for OSD to use.
|
||||||
|
|
||||||
|
Atomic writes allow to reduce the Write Amplification factor with the new store
|
||||||
|
([meta_format](layout-osd.en.md#meta_format)=3) to almost 1 (i.e. almost no extra writes)
|
||||||
|
with replicated pools and reach the best possible write performance.
|
||||||
|
|
||||||
|
Default value is auto-detected during OSD initialization from
|
||||||
|
`/sys/block/xx/queue/atomic_write_max_bytes` or assumed to be 4096 bytes
|
||||||
|
because all known disks support 4 KB atomic writes. Auto-detection is only used for
|
||||||
|
NVMe disks because SAS disks require the explicit WRITE ATOMIC command which requires
|
||||||
|
RWF_ATOMIC (see below [#use_atomic_flag]) but that flag works incorrectly in current
|
||||||
|
Linux versions.
|
||||||
|
|
||||||
|
You can also check if your NVMe drives support atomic writes by running
|
||||||
|
the command `nvme id-ctrl /dev/nvme0n1 | grep awupf`. If the reported value,
|
||||||
|
plus 1, multiplied by the currently selected block size of the NVMe,
|
||||||
|
is more than 4 KB, then the new store can utilize it for better performance.
|
||||||
|
The only drives known to support it currently are [Micron and Kioxia](../intro/quickstart.en.md).
|
||||||
|
|
||||||
|
Atomic writes allow to skip double data writes in replicated pools, thus
|
||||||
|
reducing Write Amplification and improving write performance up to 2 times.
|
||||||
|
info_ru: |
|
||||||
|
Максимальный размер атомарной записи на диск данных, который OSD разрешено использовать.
|
||||||
|
|
||||||
|
Поддержка атомарной записи позволяет снизить мультипликатор записи (Write Amplification)
|
||||||
|
на диск с новым хранилищем ([meta_format](layout-osd.ru.md#meta_format)=3)
|
||||||
|
практически до 1 (то есть, почти до нулевого объёма лишней записи) в реплицированных
|
||||||
|
пулах и достигнуть наилучшей возможной производительности записи.
|
||||||
|
|
||||||
|
Значение по умолчанию авто-определяется во время инициализации OSD из
|
||||||
|
`/sys/block/xx/queue/atomic_write_max_bytes` либо принимается равным 4096,
|
||||||
|
так как все известные диски поддерживают атомарную запись 4 КБ блоков.
|
||||||
|
Автоопределение применяется только для NVMe-дисков, так как SAS диски требуют
|
||||||
|
использования отдельной команды WRITE ATOMIC, а для неё нужен флаг RWF_ATOMIC
|
||||||
|
(см. ниже [#use_atomic_flag]), а он в текущих версиях Linux работает некорректно.
|
||||||
|
|
||||||
|
Вы также можете проверить, поддерживают ли ваши NVMe-диски атомарную запись,
|
||||||
|
с помощью команды `nvme id-ctrl /dev/nvme0n1 | grep awupf`. Если значение awupf
|
||||||
|
плюс 1, умноженное на текущий выбранный размер блока NVMe-диска, больше 4 КБ,
|
||||||
|
то новое хранилище может использовать атомарные записи для достижения лучшей
|
||||||
|
производительности. Единственные известные диски, которые поддерживают это сейчас -
|
||||||
|
[Micron и Kioxia](../intro/quickstart.ru.md).
|
||||||
|
|
||||||
|
Атомарная запись позволяет не использовать двойную запись данных (в журнал и на
|
||||||
|
устройство данных) в реплицированных пулах и таким образом снижает амплификацию
|
||||||
|
записи (объём служебной записи на диск) и улучшает производительность записи
|
||||||
|
вплоть до 2-х кратного прироста.
|
||||||
|
- name: use_atomic_flag
|
||||||
|
type: bool
|
||||||
|
info: |
|
||||||
|
This option controls whether Vitastor OSDs use RWF_ATOMIC write flag with atomic writes.
|
||||||
|
This flag is supported since Linux 6.11 and adds some safety to atomic writes - the kernel
|
||||||
|
guarantees to not fragment write requests with it and also to check them against the actual
|
||||||
|
device atomic write capabilities.
|
||||||
|
|
||||||
|
However, the option is disabled by default because the flag is currently UNUSABLE - Linux
|
||||||
|
incorrectly requires writes with that flag to be of power-of-2 length and length-aligned.
|
||||||
|
I.e., for example, 12 KB writes and not-8-KB aligned 8 KB writes are forbidden by the kernel,
|
||||||
|
even though the NVMe specification allows them.
|
||||||
|
|
||||||
|
For NVMe disks with `scheduler=none` writes aren't fragmented anyway so it's not a big deal.
|
||||||
|
However, you can rebuild your kernel with [this patch](../../patches/linux-fix-atomic-write-checks.diff)
|
||||||
|
and turn this option on. It will make your atomic writes a bit safer.
|
||||||
|
info_ru: |
|
||||||
|
Данная опция контролирует использование Vitastor OSD флага RWF_ATOMIC при атомарной записи
|
||||||
|
блоков. Этот флаг поддерживается, начиная с версии ядра Linux 6.11 и добавляет немного корректности
|
||||||
|
атомарным записям - ядро гарантирует отсутствие фрагментации запросов записи с этим флагом и
|
||||||
|
проверяет их на соответствие реальным возможностям устройства.
|
||||||
|
|
||||||
|
Однако, данная опция по умолчанию отключена, так как флаг в текущих версиях Linux работает
|
||||||
|
абсолютно НЕКОРРЕКТНО - при нём Linux требует, чтобы запросы записи имели длину, равную
|
||||||
|
степени двойки и были выровнены на эту длину. То есть, например, 12 КБ запросы записи, а также
|
||||||
|
8 КБ запросы записи по не-кратному 8 КБ смещению запрещаются ядром, хотя спецификация NVMe их
|
||||||
|
разрешает.
|
||||||
|
|
||||||
|
Для NVMe-дисков с `scheduler=none` запросы записи и так не фрагментируются, так что это не так
|
||||||
|
уж и важно, однако вы можете пересобрать своё ядро с [этим патчем](../../patches/linux-fix-atomic-write-checks.diff)
|
||||||
|
и включить данную опцию. Это сделает вашу атомарную запись капельку безопаснее.
|
||||||
|
- name: pg_reshard_chunk_size
|
||||||
|
type: int
|
||||||
|
default: 100000
|
||||||
|
info: |
|
||||||
|
Pool PG count change is a CPU-intensive operation because OSDs store the full object database
|
||||||
|
in memory and have to move all entries between old and new PGs. Thus it's performed in chunks,
|
||||||
|
with pauses between chunks to prevent blocking OSD's event loop and other clients' operations.
|
||||||
|
This option sets the maximum number of object is a chunk. Moving 100k objects usually takes
|
||||||
|
50-100ms. Chunk size equal to 0 means unlimited.
|
||||||
|
info_ru: |
|
||||||
|
Изменение числа PG в пуле заметно загружает процессор, так как OSD хранят полную базу данных
|
||||||
|
объектов в памяти и им приходится перемещать все записи объектов между старыми и новыми PG.
|
||||||
|
Поэтому изменение применяется порциями, с паузами между порциями, чтобы не блокировать обработку
|
||||||
|
событий OSD и операции остальных клиентов. Данная опция задаёт максимальное число объектов
|
||||||
|
в порции. Перемещение 100 тысяч объектов (значение по умолчанию) обычно занимает порядка
|
||||||
|
50-100 миллисекунд. Значение опции 0 отключает лимит размера порции.
|
||||||
|
- name: pg_reshard_chunk_pause_ms
|
||||||
|
type: ms
|
||||||
|
default: 100
|
||||||
|
info: |
|
||||||
|
This option sets the interval between handling two PG count change chunks.
|
||||||
|
info_ru: |
|
||||||
|
Данная опция задаёт интервал между обработкой двух порций изменения числа PG пулов.
|
||||||
|
|||||||
@@ -26,13 +26,37 @@ at Vitastor Kubernetes operator: https://github.com/Antilles7227/vitastor-operat
|
|||||||
The instruction is very simple.
|
The instruction is very simple.
|
||||||
|
|
||||||
1. Download a Docker image of the desired version: \
|
1. Download a Docker image of the desired version: \
|
||||||
`docker pull vitalif/vitastor:v2.3.0`
|
`docker pull vitalif/vitastor:v3.0.6`
|
||||||
2. Install scripts to the host system: \
|
2. Install scripts to the host system: \
|
||||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v2.3.0 install.sh`
|
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.6 install.sh`
|
||||||
3. Reload udev rules: \
|
3. Reload udev rules: \
|
||||||
`udevadm control --reload-rules`
|
`udevadm control --reload-rules`
|
||||||
|
4. Enable the vitastor-host service: \
|
||||||
|
`systemctl enable --now vitastor-host`
|
||||||
|
|
||||||
And you can return to [Quick Start](../intro/quickstart.en.md).
|
After these steps, you can return to [Quick Start](../intro/quickstart.en.md).
|
||||||
|
|
||||||
|
## Podman
|
||||||
|
|
||||||
|
If you use Podman, run the following commands as root before installing Vitastor containers:
|
||||||
|
|
||||||
|
```
|
||||||
|
ln -s podman /usr/bin/docker
|
||||||
|
|
||||||
|
mkdir -p /etc/systemd/system/systemd-udevd.service.d
|
||||||
|
|
||||||
|
cat >/etc/systemd/system/systemd-udevd.service.d/override.conf <<EOF
|
||||||
|
[Service]
|
||||||
|
CapabilityBoundingSet=~
|
||||||
|
SystemCallFilter=@mount capset
|
||||||
|
EOF
|
||||||
|
|
||||||
|
systemctl daemon-reload
|
||||||
|
|
||||||
|
systemctl restart systemd-udevd
|
||||||
|
```
|
||||||
|
|
||||||
|
Without it, udev fails to do calls into a Podman container and Vitastor disk detection doesn't work.
|
||||||
|
|
||||||
## Upgrading Containers
|
## Upgrading Containers
|
||||||
|
|
||||||
|
|||||||
@@ -25,14 +25,39 @@ Vitastor можно установить в Docker/Podman. При этом etcd,
|
|||||||
Инструкция по установке максимально простая.
|
Инструкция по установке максимально простая.
|
||||||
|
|
||||||
1. Скачайте Docker-образ желаемой версии: \
|
1. Скачайте Docker-образ желаемой версии: \
|
||||||
`docker pull vitalif/vitastor:v2.3.0`
|
`docker pull vitalif/vitastor:v3.0.6`
|
||||||
2. Установите скрипты в хост-систему командой: \
|
2. Установите скрипты в хост-систему командой: \
|
||||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v2.3.0 install.sh`
|
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.6 install.sh`
|
||||||
3. Перезагрузите правила udev: \
|
3. Перезагрузите правила udev: \
|
||||||
`udevadm control --reload-rules`
|
`udevadm control --reload-rules`
|
||||||
|
4. Включите сервис vitastor-host: \
|
||||||
|
`systemctl enable --now vitastor-host`
|
||||||
|
|
||||||
После этого вы можете возвращаться к разделу [Быстрый старт](../intro/quickstart.ru.md).
|
После этого вы можете возвращаться к разделу [Быстрый старт](../intro/quickstart.ru.md).
|
||||||
|
|
||||||
|
## Podman
|
||||||
|
|
||||||
|
Если вы используете Podman, перед установкой контейнеров Vitastor выполните следующие
|
||||||
|
команды от имени суперпользователя:
|
||||||
|
|
||||||
|
```
|
||||||
|
ln -s podman /usr/bin/docker
|
||||||
|
|
||||||
|
mkdir -p /etc/systemd/system/systemd-udevd.service.d
|
||||||
|
|
||||||
|
cat >/etc/systemd/system/systemd-udevd.service.d/override.conf <<EOF
|
||||||
|
[Service]
|
||||||
|
CapabilityBoundingSet=~
|
||||||
|
SystemCallFilter=@mount capset
|
||||||
|
EOF
|
||||||
|
|
||||||
|
systemctl daemon-reload
|
||||||
|
|
||||||
|
systemctl restart systemd-udevd
|
||||||
|
```
|
||||||
|
|
||||||
|
Без этих настроек udev не может делать вызовы внутрь Podman-контейнеров и определение дисков Vitastor не работает.
|
||||||
|
|
||||||
## Обновление контейнеров
|
## Обновление контейнеров
|
||||||
|
|
||||||
Сначала обязательно проверьте раздел [Обновление Vitastor](../usage/admin.ru.md#обновление-vitastor),
|
Сначала обязательно проверьте раздел [Обновление Vitastor](../usage/admin.ru.md#обновление-vitastor),
|
||||||
|
|||||||
@@ -30,7 +30,7 @@ volume_backend_name = vitastor-testcluster
|
|||||||
image_volume_cache_enabled = True
|
image_volume_cache_enabled = True
|
||||||
volume_clear = none
|
volume_clear = none
|
||||||
vitastor_etcd_address = 192.168.7.2:2379
|
vitastor_etcd_address = 192.168.7.2:2379
|
||||||
vitastor_etcd_prefix =
|
vitastor_etcd_prefix = /vitastor
|
||||||
vitastor_config_path = /etc/vitastor/vitastor.conf
|
vitastor_config_path = /etc/vitastor/vitastor.conf
|
||||||
vitastor_pool_id = 1
|
vitastor_pool_id = 1
|
||||||
image_upload_use_cinder_backend = True
|
image_upload_use_cinder_backend = True
|
||||||
|
|||||||
@@ -29,7 +29,7 @@ volume_backend_name = vitastor-testcluster
|
|||||||
image_volume_cache_enabled = True
|
image_volume_cache_enabled = True
|
||||||
volume_clear = none
|
volume_clear = none
|
||||||
vitastor_etcd_address = 192.168.7.2:2379
|
vitastor_etcd_address = 192.168.7.2:2379
|
||||||
vitastor_etcd_prefix =
|
vitastor_etcd_prefix = /vitastor
|
||||||
vitastor_config_path = /etc/vitastor/vitastor.conf
|
vitastor_config_path = /etc/vitastor/vitastor.conf
|
||||||
vitastor_pool_id = 1
|
vitastor_pool_id = 1
|
||||||
image_upload_use_cinder_backend = True
|
image_upload_use_cinder_backend = True
|
||||||
|
|||||||
@@ -33,15 +33,17 @@
|
|||||||
- CentOS 7: `yum install https://vitastor.io/rpms/centos/7/vitastor-release.rpm`
|
- CentOS 7: `yum install https://vitastor.io/rpms/centos/7/vitastor-release.rpm`
|
||||||
- CentOS 8: `dnf install https://vitastor.io/rpms/centos/8/vitastor-release.rpm`
|
- CentOS 8: `dnf install https://vitastor.io/rpms/centos/8/vitastor-release.rpm`
|
||||||
- AlmaLinux 9 and other RHEL 9 clones (Rocky, Oracle...): `dnf install https://vitastor.io/rpms/centos/9/vitastor-release.rpm`
|
- AlmaLinux 9 and other RHEL 9 clones (Rocky, Oracle...): `dnf install https://vitastor.io/rpms/centos/9/vitastor-release.rpm`
|
||||||
|
- AlmaLinux 10 and other RHEL 10 clones: `dnf install https://vitastor.io/rpms/centos/10/vitastor-release.rpm`
|
||||||
- Enable EPEL: `yum/dnf install epel-release`
|
- Enable EPEL: `yum/dnf install epel-release`
|
||||||
- Enable additional CentOS repositories:
|
- Enable additional CentOS repositories:
|
||||||
- CentOS 7: `yum install centos-release-scl`
|
- CentOS 7: `yum install centos-release-scl`
|
||||||
- CentOS 8: `dnf install centos-release-advanced-virtualization`
|
- CentOS 8: `dnf install centos-release-advanced-virtualization`
|
||||||
- RHEL 9 clones: not required
|
- RHEL 9/10 clones: not required
|
||||||
- Enable elrepo-kernel:
|
- Enable elrepo-kernel:
|
||||||
- CentOS 7: `yum install https://www.elrepo.org/elrepo-release-7.el7.elrepo.noarch.rpm`
|
- CentOS 7: `yum install https://www.elrepo.org/elrepo-release-7.el7.elrepo.noarch.rpm`
|
||||||
- CentOS 8: `dnf install https://www.elrepo.org/elrepo-release-8.el8.elrepo.noarch.rpm`
|
- CentOS 8: `dnf install https://www.elrepo.org/elrepo-release-8.el8.elrepo.noarch.rpm`
|
||||||
- RHEL 9 clones: `dnf install https://www.elrepo.org/elrepo-release-9.el9.elrepo.noarch.rpm`
|
- RHEL 9 clones: `dnf install https://www.elrepo.org/elrepo-release-9.el9.elrepo.noarch.rpm`
|
||||||
|
- RHEL 10 clones: not required
|
||||||
- Install packages: `yum/dnf install vitastor lpsolve etcd kernel-ml qemu-kvm`
|
- Install packages: `yum/dnf install vitastor lpsolve etcd kernel-ml qemu-kvm`
|
||||||
|
|
||||||
## Installation requirements
|
## Installation requirements
|
||||||
|
|||||||
@@ -33,15 +33,17 @@
|
|||||||
- CentOS 7: `yum install https://vitastor.io/rpms/centos/7/vitastor-release.rpm`
|
- CentOS 7: `yum install https://vitastor.io/rpms/centos/7/vitastor-release.rpm`
|
||||||
- CentOS 8: `dnf install https://vitastor.io/rpms/centos/8/vitastor-release.rpm`
|
- CentOS 8: `dnf install https://vitastor.io/rpms/centos/8/vitastor-release.rpm`
|
||||||
- AlmaLinux 9 и другие клоны RHEL 9 (Rocky, Oracle...): `dnf install https://vitastor.io/rpms/centos/9/vitastor-release.rpm`
|
- AlmaLinux 9 и другие клоны RHEL 9 (Rocky, Oracle...): `dnf install https://vitastor.io/rpms/centos/9/vitastor-release.rpm`
|
||||||
|
- AlmaLinux 10 и другие клоны RHEL 10: `dnf install https://vitastor.io/rpms/centos/10/vitastor-release.rpm`
|
||||||
- Включите EPEL: `yum/dnf install epel-release`
|
- Включите EPEL: `yum/dnf install epel-release`
|
||||||
- Включите дополнительные репозитории CentOS:
|
- Включите дополнительные репозитории CentOS:
|
||||||
- CentOS 7: `yum install centos-release-scl`
|
- CentOS 7: `yum install centos-release-scl`
|
||||||
- CentOS 8: `dnf install centos-release-advanced-virtualization`
|
- CentOS 8: `dnf install centos-release-advanced-virtualization`
|
||||||
- Клоны RHEL 9: не нужно
|
- Клоны RHEL 9/10: не нужно
|
||||||
- Включите elrepo-kernel:
|
- Включите elrepo-kernel:
|
||||||
- CentOS 7: `yum install https://www.elrepo.org/elrepo-release-7.el7.elrepo.noarch.rpm`
|
- CentOS 7: `yum install https://www.elrepo.org/elrepo-release-7.el7.elrepo.noarch.rpm`
|
||||||
- CentOS 8: `dnf install https://www.elrepo.org/elrepo-release-8.el8.elrepo.noarch.rpm`
|
- CentOS 8: `dnf install https://www.elrepo.org/elrepo-release-8.el8.elrepo.noarch.rpm`
|
||||||
- Клоны RHEL 9: `dnf install https://www.elrepo.org/elrepo-release-9.el9.elrepo.noarch.rpm`
|
- Клоны RHEL 9: `dnf install https://www.elrepo.org/elrepo-release-9.el9.elrepo.noarch.rpm`
|
||||||
|
- Клоны RHEL 10: не нужно
|
||||||
- Установите пакеты: `yum/dnf install vitastor lpsolve etcd kernel-ml qemu-kvm`
|
- Установите пакеты: `yum/dnf install vitastor lpsolve etcd kernel-ml qemu-kvm`
|
||||||
|
|
||||||
## Установочные требования
|
## Установочные требования
|
||||||
|
|||||||
@@ -6,7 +6,7 @@
|
|||||||
|
|
||||||
# Proxmox VE
|
# Proxmox VE
|
||||||
|
|
||||||
To enable Vitastor support in Proxmox Virtual Environment (6.4-8.x are supported):
|
To enable Vitastor support in Proxmox Virtual Environment (6.4-9.x are supported):
|
||||||
|
|
||||||
- Add the corresponding Vitastor Debian repository into sources.list on Proxmox hosts:
|
- Add the corresponding Vitastor Debian repository into sources.list on Proxmox hosts:
|
||||||
trixie for 9.0+, bookworm for 8.1+, pve8.0 for 8.0, bullseye for 7.4, pve7.3 for 7.3, pve7.2 for 7.2, pve7.1 for 7.1, buster for 6.4
|
trixie for 9.0+, bookworm for 8.1+, pve8.0 for 8.0, bullseye for 7.4, pve7.3 for 7.3, pve7.2 for 7.2, pve7.1 for 7.1, buster for 6.4
|
||||||
|
|||||||
@@ -6,7 +6,7 @@
|
|||||||
|
|
||||||
# Proxmox VE
|
# Proxmox VE
|
||||||
|
|
||||||
Чтобы подключить Vitastor к Proxmox Virtual Environment (поддерживаются версии 6.4-8.x):
|
Чтобы подключить Vitastor к Proxmox Virtual Environment (поддерживаются версии 6.4-9.x):
|
||||||
|
|
||||||
- Добавьте соответствующий Debian-репозиторий Vitastor в sources.list на хостах Proxmox:
|
- Добавьте соответствующий Debian-репозиторий Vitastor в sources.list на хостах Proxmox:
|
||||||
trixie для 9.0+, bookworm для 8.1+, pve8.0 для 8.0, bullseye для 7.4, pve7.3 для 7.3, pve7.2 для 7.2, pve7.1 для 7.1, buster для 6.4
|
trixie для 9.0+, bookworm для 8.1+, pve8.0 для 8.0, bullseye для 7.4, pve7.3 для 7.3, pve7.2 для 7.2, pve7.1 для 7.1, buster для 6.4
|
||||||
|
|||||||
@@ -14,6 +14,8 @@
|
|||||||
|
|
||||||
- Basic part: highly-available block storage with symmetric clustering and no SPOF
|
- Basic part: highly-available block storage with symmetric clustering and no SPOF
|
||||||
- [Performance](../performance/bench2.en.md) ;-D
|
- [Performance](../performance/bench2.en.md) ;-D
|
||||||
|
- [NVMe atomic write support](../config/osd.en.md#atomic_write_size) for reducing the amount
|
||||||
|
of "extra" disk writes to almost zero (Write Amplification = 1)
|
||||||
- [Multiple redundancy schemes](../config/pool.en.md#scheme): Replication, XOR n+1, Reed-Solomon erasure codes
|
- [Multiple redundancy schemes](../config/pool.en.md#scheme): Replication, XOR n+1, Reed-Solomon erasure codes
|
||||||
based on jerasure and ISA-L libraries with any number of data and parity drives in a group
|
based on jerasure and ISA-L libraries with any number of data and parity drives in a group
|
||||||
- Configuration via simple JSON data structures in etcd (parameters, pools and images)
|
- Configuration via simple JSON data structures in etcd (parameters, pools and images)
|
||||||
|
|||||||
@@ -14,6 +14,8 @@
|
|||||||
|
|
||||||
- Базовая часть - надёжное кластерное блочное хранилище без единой точки отказа
|
- Базовая часть - надёжное кластерное блочное хранилище без единой точки отказа
|
||||||
- [Производительность](../performance/bench2.ru.md) ;-D
|
- [Производительность](../performance/bench2.ru.md) ;-D
|
||||||
|
- [Поддержка атомарной записи NVMe](../config/osd.ru.md#atomic_write_size) для снижения объёма
|
||||||
|
служебной записи практически до нуля (Write Amplification = 1)
|
||||||
- [Несколько схем отказоустойчивости](../config/pool.ru.md#scheme): репликация, XOR n+1 (1 диск чётности), коды коррекции ошибок
|
- [Несколько схем отказоустойчивости](../config/pool.ru.md#scheme): репликация, XOR n+1 (1 диск чётности), коды коррекции ошибок
|
||||||
Рида-Соломона на основе библиотек jerasure и ISA-L с любым числом дисков данных и чётности в группе
|
Рида-Соломона на основе библиотек jerasure и ISA-L с любым числом дисков данных и чётности в группе
|
||||||
- Конфигурация через простые человекочитаемые JSON-структуры в etcd
|
- Конфигурация через простые человекочитаемые JSON-структуры в etcd
|
||||||
|
|||||||
@@ -18,9 +18,10 @@
|
|||||||
|
|
||||||
## Preparation
|
## Preparation
|
||||||
|
|
||||||
- Get some SATA or NVMe SSDs with capacitors (server-grade drives). You can use desktop SSDs
|
- Get some SATA or NVMe SSDs with capacitors (server-grade drives). The best performance
|
||||||
with lazy fsync, but prepare for inferior single-thread latency. Read more about capacitors
|
is achieved with Micron or Kioxia NVMes with atomic write support (see below). You can use desktop
|
||||||
[here](../config/layout-cluster.en.md#immediate_commit).
|
SSDs with lazy fsync, but prepare for inferior single-thread latency. Read more about
|
||||||
|
capacitors [here](../config/layout-cluster.en.md#immediate_commit).
|
||||||
- If you want to use HDDs, get modern HDDs with Media Cache or SSD Cache: HGST Ultrastar,
|
- If you want to use HDDs, get modern HDDs with Media Cache or SSD Cache: HGST Ultrastar,
|
||||||
Toshiba MG, Seagate EXOS or something similar. If your drives don't have such cache then
|
Toshiba MG, Seagate EXOS or something similar. If your drives don't have such cache then
|
||||||
you also need small SSDs for journal and metadata (even 2 GB per 1 TB of HDD space is enough).
|
you also need small SSDs for journal and metadata (even 2 GB per 1 TB of HDD space is enough).
|
||||||
@@ -30,9 +31,11 @@
|
|||||||
|
|
||||||
## Recommended drives
|
## Recommended drives
|
||||||
|
|
||||||
- SATA SSD: Micron 5100/5200/5300/5400, Samsung PM863/PM883/PM893, Intel D3-S4510/4520/4610/4620, Kingston DC500M
|
- NVMe with atomic write support (ideal!): Micron 7450/7500/7550, Kioxia CD6/CD7/CD8/CD9
|
||||||
- NVMe: Micron 9100/9200/9300/9400, Micron 7300/7450, Samsung PM983/PM9A3, Samsung PM1723/1735/1743,
|
- Other NVMe: Micron 9100/9200/9300/9400/9550, Micron 7300, Samsung PM983/PM9A3, Samsung PM1723/1735/1743,
|
||||||
Intel DC-P3700/P4500/P4600, Intel D5-P4320/P5530, Intel D7-P5500/P5600, Intel Optane, Kingston DC1000B/DC1500M
|
Intel DC-P3700/P4500/P4600, Intel/Solidigm D5-P4320/P5530, Intel/Solidigm D7-P5500/P5600, Solidigm D7-PS1010/PS1030/P5810,
|
||||||
|
Intel Optane, Kingston DC1000B/DC1500M, Kioxia CD6/CD7/CD8/CD9
|
||||||
|
- SATA SSD: Micron 5100/5200/5300/5400, Samsung PM863/PM883/PM893, Intel/Solidigm D3-S4510/4520/4610/4620, Kingston DC500M
|
||||||
- HDD: HGST Ultrastar, Toshiba MG, Seagate EXOS
|
- HDD: HGST Ultrastar, Toshiba MG, Seagate EXOS
|
||||||
|
|
||||||
## Configure monitors
|
## Configure monitors
|
||||||
|
|||||||
@@ -18,8 +18,9 @@
|
|||||||
|
|
||||||
## Подготовка
|
## Подготовка
|
||||||
|
|
||||||
- Возьмите серверы с SSD (SATA или NVMe), желательно с конденсаторами (серверные SSD). Можно
|
- Возьмите серверы с SSD (SATA или NVMe), желательно с конденсаторами (серверные SSD). Наилучшая
|
||||||
использовать и десктопные SSD, включив режим отложенного fsync, но производительность будет хуже.
|
производительность достигается на дисках Micron и Kioxia с поддержкой атомарной записи (см. ниже).
|
||||||
|
Можно использовать и десктопные SSD, включив режим отложенного fsync, но производительность будет хуже.
|
||||||
О конденсаторах читайте [здесь](../config/layout-cluster.ru.md#immediate_commit).
|
О конденсаторах читайте [здесь](../config/layout-cluster.ru.md#immediate_commit).
|
||||||
- Если хотите использовать HDD, берите современные модели с Media или SSD кэшем - HGST Ultrastar,
|
- Если хотите использовать HDD, берите современные модели с Media или SSD кэшем - HGST Ultrastar,
|
||||||
Toshiba MG, Seagate EXOS или что-то похожее. Если такого кэша у ваших дисков нет,
|
Toshiba MG, Seagate EXOS или что-то похожее. Если такого кэша у ваших дисков нет,
|
||||||
@@ -30,9 +31,11 @@
|
|||||||
|
|
||||||
## Рекомендуемые диски
|
## Рекомендуемые диски
|
||||||
|
|
||||||
- SATA SSD: Micron 5100/5200/5300/5400, Samsung PM863/PM883/PM893, Intel D3-S4510/4520/4610/4620, Kingston DC500M
|
- NVMe с поддержкой атомарной записи (идеально!): Micron 7450/7500/7550, Kioxia CD6/CD7/CD8/CD9
|
||||||
- NVMe: Micron 9100/9200/9300/9400, Micron 7300/7450, Samsung PM983/PM9A3, Samsung PM1723/1735/1743,
|
- Другие NVMe: Micron 9100/9200/9300/9400/9550, Micron 7300, Samsung PM983/PM9A3, Samsung PM1723/1735/1743,
|
||||||
Intel DC-P3700/P4500/P4600, Intel D5-P4320/P5530, Intel D7-P5500/P5600, Intel Optane, Kingston DC1000B/DC1500M
|
Intel DC-P3700/P4500/P4600, Intel/Solidigm D5-P4320/P5530, Intel/Solidigm D7-P5500/P5600, Solidigm D7-PS1010/PS1030/P5810,
|
||||||
|
Intel Optane, Kingston DC1000B/DC1500M, Kioxia CD6/CD7/CD8/CD9
|
||||||
|
- SATA SSD: Micron 5100/5200/5300/5400, Samsung PM863/PM883/PM893, Intel/Solidigm D3-S4510/4520/4610/4620, Kingston DC500M
|
||||||
- HDD: HGST Ultrastar, Toshiba MG, Seagate EXOS
|
- HDD: HGST Ultrastar, Toshiba MG, Seagate EXOS
|
||||||
|
|
||||||
## Настройте мониторы
|
## Настройте мониторы
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ Replicated setups:
|
|||||||
- Linear read: `min(total network bandwidth, sum(disk read MB/s))`.
|
- Linear read: `min(total network bandwidth, sum(disk read MB/s))`.
|
||||||
- Linear write: `min(total network bandwidth, sum(disk write MB/s / number of replicas))`.
|
- Linear write: `min(total network bandwidth, sum(disk write MB/s / number of replicas))`.
|
||||||
- Saturated parallel read iops: `min(total network bandwidth, sum(disk read iops))`.
|
- Saturated parallel read iops: `min(total network bandwidth, sum(disk read iops))`.
|
||||||
- Saturated parallel write iops: `min(total network bandwidth / number of replicas, sum(disk write iops / number of replicas / (write amplification = 4)))`.
|
- Saturated parallel write iops: `min(total network bandwidth / number of replicas, sum(disk write iops / number of replicas / write amplification))`.
|
||||||
|
|
||||||
EC/XOR setups (EC N+K):
|
EC/XOR setups (EC N+K):
|
||||||
- Single-threaded (T1Q1) read latency: 1.5 network roundtrips + 1 disk read.
|
- Single-threaded (T1Q1) read latency: 1.5 network roundtrips + 1 disk read.
|
||||||
@@ -26,28 +26,36 @@ EC/XOR setups (EC N+K):
|
|||||||
- Linear read: `min(total network bandwidth, sum(disk read MB/s))`.
|
- Linear read: `min(total network bandwidth, sum(disk read MB/s))`.
|
||||||
- Linear write: `min(total network bandwidth, sum(disk write MB/s * N/(N+K)))`.
|
- Linear write: `min(total network bandwidth, sum(disk write MB/s * N/(N+K)))`.
|
||||||
- Saturated parallel read iops: `min(total network bandwidth, sum(disk read iops))`.
|
- Saturated parallel read iops: `min(total network bandwidth, sum(disk read iops))`.
|
||||||
- Saturated parallel write iops: roughly `total iops / (N+K) / WA`. More exactly,
|
- Saturated parallel write iops: roughly `total iops / (N+K) / WA`. More exactly:
|
||||||
`min(total network bandwidth * N/(N+K), sum(disk randrw iops / (N*4 + K*5 + 1)))` with
|
- With the new store: `min(total network bandwidth * N/(N+K), sum(disk randrw iops / (2 + N-1 + K*2)))`,
|
||||||
random read/write mix corresponding to `(N-1)/(N*4 + K*5 + 1)*100 % reads`.
|
with random read/write mix corresponding to `(N-1)/(2 + N-1 + K*2)*100 % reads`.
|
||||||
- For example, with EC 2+1 it is: `(7% randrw iops) / 14`.
|
- For example, with EC 2+1 it is: `(20% randrw iops) / 5`.
|
||||||
- With EC 6+3 it is: `(12.5% randrw iops) / 40`.
|
- With EC 6+3 it is: `(38% randrw iops) / 13`.
|
||||||
|
- With the old store: `min(total network bandwidth * N/(N+K), sum(disk randrw iops / (3 + N-1 + K*3)))`,
|
||||||
|
with random read/write mix corresponding to `(N-1)/(3 + N-1 + K*3)*100 % reads`.
|
||||||
|
- For example, with EC 2+1 it is: `(14% randrw iops) / 7`.
|
||||||
|
- With EC 6+3 it is: `(30% randrw iops) / 17`.
|
||||||
|
|
||||||
Write amplification for 4 KB blocks is usually 3-5 in Vitastor:
|
Write Amplification factor:
|
||||||
1. Journal block write
|
- For the new store and for 4 KB writes: WA is always 1 unless you set [atomic_write_size](../config/osd.en.md#atomic_write_size) to 0 manually.
|
||||||
2. Journal data write
|
- For the new store and for 8-124 KB writes: WA is 1 if you use NVMe drives with atomic write support, or roughly 2 if you use other drives.
|
||||||
3. Metadata block write
|
- For the old store, WA is roughly `(2 * write size + 4 KB) / (write size)`. So, for 4 KB writes it's 3, and for 8-124 KB writes it's closer to 2.
|
||||||
4. Another journal block write for EC/XOR setups
|
- For both the new and the old store and for writes of [block_size](../config/layout-cluster.en.md#block_size): WA is almost 1.
|
||||||
5. Data block write
|
|
||||||
|
|
||||||
If you manage to get an SSD which handles 512 byte blocks well (Optane?) you may
|
Write Amplification consists of:
|
||||||
lower 1, 3 and 4 to 512 bytes (1/8 of data size) and get WA as low as 2.375.
|
- For the new store:
|
||||||
|
- Buffer block write if non-atomic
|
||||||
|
- Data block write
|
||||||
|
- Metadata write(s) (amortized)
|
||||||
|
- For the old store:
|
||||||
|
- Journal block write (amortized)
|
||||||
|
- Journal data write
|
||||||
|
- Metadata block write
|
||||||
|
- Another journal block write for EC/XOR setups (amortized)
|
||||||
|
- Data block write
|
||||||
|
|
||||||
Implemented NVDIMM support can basically eliminate WA at all - all extra writes will
|
Other possibilities to reduce WA would be to use SSDs with internal 512-byte blocks
|
||||||
go to DRAM memory. But this requires a test cluster with NVDIMM - please contact me
|
or NVDIMM, but both options seem unavailable on the market at the moment.
|
||||||
if you want to provide me with such cluster for tests.
|
|
||||||
|
|
||||||
Lazy fsync also reduces WA for parallel workloads because journal blocks are only
|
|
||||||
written when they fill up or fsync is requested.
|
|
||||||
|
|
||||||
## In Practice
|
## In Practice
|
||||||
|
|
||||||
|
|||||||
@@ -27,29 +27,36 @@
|
|||||||
- Линейное чтение: сумма МБ/с чтения всех дисков, либо общая производительность сети, если в сеть упрётся раньше.
|
- Линейное чтение: сумма МБ/с чтения всех дисков, либо общая производительность сети, если в сеть упрётся раньше.
|
||||||
- Линейная запись: сумма МБ/с записи всех дисков * N/(N+K), либо производительность сети * N / (N+K), если в сеть упрётся раньше.
|
- Линейная запись: сумма МБ/с записи всех дисков * N/(N+K), либо производительность сети * N / (N+K), если в сеть упрётся раньше.
|
||||||
- Параллельное случайное мелкое чтение: сумма IOPS чтения всех дисков либо производительность сети, если в сеть упрётся раньше.
|
- Параллельное случайное мелкое чтение: сумма IOPS чтения всех дисков либо производительность сети, если в сеть упрётся раньше.
|
||||||
- Параллельная случайная мелкая запись: грубо `(сумма IOPS / (N+K) / WA)`. Если точнее, то:
|
- Параллельная случайная мелкая запись: грубо `(сумма IOPS / (N+K) / WA)`.
|
||||||
сумма смешанного IOPS всех дисков при `(N-1)/(N*4 + K*5 + 1)*100 %` чтения, делённая на `(N*4 + K*5 + 1)`.
|
Либо `производительность сети * N/(N+K)`, если в сеть упрётся раньше. Если точнее, то:
|
||||||
Либо, производительность сети * N/(N+K), если в сеть упрётся раньше.
|
- С новым хранилищем: сумма смешанного IOPS всех дисков при `(N-1)/(2 + N-1 + K*2)*100 %` чтения, делённая на `(2 + N-1 + K*2)`.
|
||||||
- Например, при EC 2+1 это: `(сумма IOPS при 7% чтения) / 14`.
|
- Например, при EC 2+1 это: `(сумма IOPS при 20% чтения) / 5`.
|
||||||
- При EC 6+3 это: `(сумма IOPS при 12.5% чтения) / 40`.
|
- При EC 6+3 это: `(сумма IOPS при 38% чтения) / 13`.
|
||||||
|
- Со старым хранилищем: сумма смешанного IOPS всех дисков при `(N-1)/(3 + N-1 + K*3)*100 %` чтения, делённая на `(3 + N-1 + K*3)`.
|
||||||
|
- Например, при EC 2+1 это: `(сумма IOPS при 14% чтения) / 7`.
|
||||||
|
- При EC 6+3 это: `(сумма IOPS при 30% чтения) / 17`.
|
||||||
|
|
||||||
WA (мультипликатор записи) для 4 КБ блоков в Vitastor обычно составляет 3-5:
|
WA (Write Amplification, мультипликатор записи):
|
||||||
1. Запись метаданных в журнал
|
- С новым хранилищем для 4 КБ записи: WA всегда примерно 1, если только вы не установите [atomic_write_size](../config/osd.ru.md#atomic_write_size) вручную в 0.
|
||||||
2. Запись блока данных в журнал
|
- С новым хранилищем и большими записями (8-124 КБ): WA примерно 1, если вы используете NVMe-диски с поддержкой атомарной записи,
|
||||||
3. Запись метаданных в БД
|
или примерно 2, если вы используете другие диски.
|
||||||
4. Ещё одна запись метаданных в журнал при использовании EC
|
- Со старым хранилищем, WA примерно `(2 * размер записи + 4 КБ) / (размер записи)`. То есть, для 4 КБ записи WA=3, а для 8-124 КБ WA ближе к 2.
|
||||||
5. Запись блока данных на диск данных
|
- И с новым, и со старым хранилищем и для записи размером [block_size](../config/layout-cluster.ru.md#block_size): WA примерно равен 1.
|
||||||
|
|
||||||
Если вы найдёте SSD, хорошо работающий с 512-байтными блоками данных (Optane?),
|
Мультипликатор записи состоит из:
|
||||||
то 1, 3 и 4 можно снизить до 512 байт (1/8 от размера данных) и получить WA всего 2.375.
|
- С новым хранилищем:
|
||||||
|
- Запись блока буфера, если диски без поддержки атомарной записи
|
||||||
|
- Запись блока данных
|
||||||
|
- Запись(-и) блоков метаданных (амортизированные)
|
||||||
|
- Со старым хранилищем:
|
||||||
|
- Запись блока журнала (амортизированная)
|
||||||
|
- Запись данных в журнал
|
||||||
|
- Запись блока метаданных
|
||||||
|
- Ещё одна запись блока журнала для EC/XOR пулов (амортизированная)
|
||||||
|
- Запись блока данных
|
||||||
|
|
||||||
Если реализовать поддержку NVDIMM, то WA можно, условно говоря, ликвидировать вообще - все
|
Другими потенциальными возможностями снижения WA могли бы быть SSD с внутренним 512-байтным блоком
|
||||||
дополнительные операции записи смогут обслуживаться DRAM памятью. Но для этого необходим
|
либо NVDIMM, но и то, и другое сейчас выглядит недоступным на рынке.
|
||||||
тестовый кластер с NVDIMM - пишите, если готовы предоставить такой для тестов.
|
|
||||||
|
|
||||||
Кроме того, WA снижается при использовании отложенного/ленивого сброса при параллельной
|
|
||||||
нагрузке, т.к. блоки журнала записываются на диск только когда они заполняются или явным
|
|
||||||
образом запрашивается fsync.
|
|
||||||
|
|
||||||
## На практике
|
## На практике
|
||||||
|
|
||||||
|
|||||||
@@ -231,6 +231,18 @@ Upgrading from <= 0.5.x to >= 0.6.x is not supported.
|
|||||||
|
|
||||||
Downgrade are also allowed freely, except the following specific instructions:
|
Downgrade are also allowed freely, except the following specific instructions:
|
||||||
|
|
||||||
|
### 3.x -> 2.x
|
||||||
|
|
||||||
|
Versions 3.0.0 and newer contain two store implementations - an old one and a new
|
||||||
|
one, unsupported in 2.x and previous versions. So you should check your OSD store
|
||||||
|
versions before downgrading to 2.x with the following command:
|
||||||
|
|
||||||
|
`vitastor-disk read-sb /dev/vitastor/osdXX-data | jq -r .meta_format`
|
||||||
|
|
||||||
|
If it prints 3 then OSD uses the new store and you can't downgrade it to 2.x.
|
||||||
|
|
||||||
|
If it prints 2 or nothing then OSD uses the old store and the downgrade is allowed.
|
||||||
|
|
||||||
### 1.8.0 to 1.7.1
|
### 1.8.0 to 1.7.1
|
||||||
|
|
||||||
Before downgrading from version >= 1.8.0 to version <= 1.7.1
|
Before downgrading from version >= 1.8.0 to version <= 1.7.1
|
||||||
|
|||||||
@@ -228,6 +228,18 @@ done
|
|||||||
|
|
||||||
Откат (понижение версии) тоже свободно разрешён, кроме указанных ниже случаев:
|
Откат (понижение версии) тоже свободно разрешён, кроме указанных ниже случаев:
|
||||||
|
|
||||||
|
### 3.x -> 2.x
|
||||||
|
|
||||||
|
Версии 3.0.0 и более новые содержат две реализации хранилища - старую и новую, не
|
||||||
|
поддерживаемую в 2.x и предыдущих версиях. Таким образом, перед откатом на 2.x вам
|
||||||
|
следует проверить, какая версия хранилища используется вашими OSD - командой:
|
||||||
|
|
||||||
|
`vitastor-disk read-sb /dev/vitastor/osdXX-data | jq -r .meta_format`
|
||||||
|
|
||||||
|
Если выводится 3, это новое хранилище и откатить такой OSD до 2.x нельзя.
|
||||||
|
|
||||||
|
Если выводится 2 или не выводится ничего, это старое хранилище и откат разрешён.
|
||||||
|
|
||||||
### 1.8.0 -> 1.7.1
|
### 1.8.0 -> 1.7.1
|
||||||
|
|
||||||
Перед понижением версии с >= 1.8.0 до <= 1.7.1 вы должны скопировать ключ
|
Перед понижением версии с >= 1.8.0 до <= 1.7.1 вы должны скопировать ключ
|
||||||
|
|||||||
@@ -100,12 +100,14 @@ List images (only matching `<glob>` pattern(s) if passed).
|
|||||||
Options:
|
Options:
|
||||||
|
|
||||||
```
|
```
|
||||||
|
--exact Do not match glob patterns as names, select only exact name matches.
|
||||||
-p|--pool POOL Filter images by pool ID or name
|
-p|--pool POOL Filter images by pool ID or name
|
||||||
-l|--long Also report allocated size and I/O statistics
|
-l|--long Also report allocated size and I/O statistics
|
||||||
--del Also include delete operation statistics
|
--del Also include delete operation statistics
|
||||||
--sort FIELD Sort by specified field (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)
|
--sort FIELD Sort by specified field (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)
|
||||||
-r|--reverse Sort in descending order
|
-r|--reverse Sort in descending order
|
||||||
-n|--count N Only list first N items
|
-n|--count N Only list first N items
|
||||||
|
--tree Show image snapshot/clone tree
|
||||||
```
|
```
|
||||||
|
|
||||||
Example output:
|
Example output:
|
||||||
|
|||||||
@@ -102,12 +102,14 @@ kaveri 2/1 32 0 B 10 G 0 B 100% 0%
|
|||||||
Опции:
|
Опции:
|
||||||
|
|
||||||
```
|
```
|
||||||
|
--exact Не применять ФС-шаблоны к именам, выводить только точные совпадения
|
||||||
-p|--pool POOL Фильтровать образы по пулу (ID или имени)
|
-p|--pool POOL Фильтровать образы по пулу (ID или имени)
|
||||||
-l|--long Также выводить статистику занятого места и ввода-вывода
|
-l|--long Также выводить статистику занятого места и ввода-вывода
|
||||||
--del Также выводить статистику операций удаления
|
--del Также выводить статистику операций удаления
|
||||||
--sort FIELD Сортировать по заданному полю (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)
|
--sort FIELD Сортировать по заданному полю (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)
|
||||||
-r|--reverse Сортировать в обратном порядке
|
-r|--reverse Сортировать в обратном порядке
|
||||||
-n|--count N Показывать только первые N записей
|
-n|--count N Показывать только первые N записей
|
||||||
|
--tree Вывести снапшоты и клоны в виде дерева
|
||||||
```
|
```
|
||||||
|
|
||||||
Пример вывода:
|
Пример вывода:
|
||||||
|
|||||||
@@ -51,6 +51,9 @@ Options (automatic mode):
|
|||||||
```
|
```
|
||||||
--osd_per_disk <N>
|
--osd_per_disk <N>
|
||||||
Create <N> OSDs on each disk (default 1)
|
Create <N> OSDs on each disk (default 1)
|
||||||
|
--meta_format 3
|
||||||
|
Metadata store version. 3 is the new log-structured store, 2 is the stable store
|
||||||
|
from Vitastor 0.9-2.x, 1 is the legacy store from Vitastor 0.6-0.8.
|
||||||
--hybrid
|
--hybrid
|
||||||
Prepare hybrid (HDD+SSD, NVMe+SATA or etc) OSDs using provided devices. By default,
|
Prepare hybrid (HDD+SSD, NVMe+SATA or etc) OSDs using provided devices. By default,
|
||||||
any passed SSDs will be used for journals and metadata, HDDs will be used for data,
|
any passed SSDs will be used for journals and metadata, HDDs will be used for data,
|
||||||
@@ -73,6 +76,8 @@ Options (automatic mode):
|
|||||||
--max_other 10%
|
--max_other 10%
|
||||||
Use disks for OSD data even if they already have non-Vitastor partitions,
|
Use disks for OSD data even if they already have non-Vitastor partitions,
|
||||||
but only if these take up no more than this percent of disk space.
|
but only if these take up no more than this percent of disk space.
|
||||||
|
--dry-run
|
||||||
|
Check and print new OSD count for each disk but do not actually create them.
|
||||||
```
|
```
|
||||||
|
|
||||||
Options (single-device mode):
|
Options (single-device mode):
|
||||||
@@ -90,6 +95,8 @@ Options (single-device mode):
|
|||||||
Options (both modes):
|
Options (both modes):
|
||||||
|
|
||||||
```
|
```
|
||||||
|
--tags tag1,tag2 Set new OSD tag(s)
|
||||||
|
--weight <number> Set new OSD weight (between 0 to 1)
|
||||||
--journal_size 1G/32M Set journal size (area or partition size)
|
--journal_size 1G/32M Set journal size (area or partition size)
|
||||||
--block_size 1M/128k Set blockstore object size
|
--block_size 1M/128k Set blockstore object size
|
||||||
--bitmap_granularity 4k Set bitmap granularity
|
--bitmap_granularity 4k Set bitmap granularity
|
||||||
|
|||||||
@@ -50,6 +50,9 @@ vitastor-disk - инструмент командной строки для уп
|
|||||||
```
|
```
|
||||||
--osd_per_disk <N>
|
--osd_per_disk <N>
|
||||||
Создавать по несколько (<N>) OSD на каждом диске (по умолчанию 1)
|
Создавать по несколько (<N>) OSD на каждом диске (по умолчанию 1)
|
||||||
|
--meta_format 3
|
||||||
|
Версия хранилища метаданных. 3 - новое лог-структурированное хранилище,
|
||||||
|
2 - стабильное хранилище из Vitastor 0.9-2.x, 1 - старое хранилище из Vitastor 0.6-0.8.
|
||||||
--hybrid
|
--hybrid
|
||||||
Инициализировать гибридные (HDD+SSD, NVMe+SATA и т.п.) OSD на указанных дисках.
|
Инициализировать гибридные (HDD+SSD, NVMe+SATA и т.п.) OSD на указанных дисках.
|
||||||
По умолчанию, SSD будут использованы для журналов и метаданных, а HDD - для данных,
|
По умолчанию, SSD будут использованы для журналов и метаданных, а HDD - для данных,
|
||||||
@@ -74,6 +77,8 @@ vitastor-disk - инструмент командной строки для уп
|
|||||||
--max_other 10%
|
--max_other 10%
|
||||||
Использовать диски под данные OSD, даже если на них уже есть не-Vitastor-овые
|
Использовать диски под данные OSD, даже если на них уже есть не-Vitastor-овые
|
||||||
разделы, но только в случае, если они занимают не более данного процента диска.
|
разделы, но только в случае, если они занимают не более данного процента диска.
|
||||||
|
--dry-run
|
||||||
|
Проверить и вывести число новых OSD для каждого диска, но не создавать их.
|
||||||
```
|
```
|
||||||
|
|
||||||
Опции для режима одного OSD:
|
Опции для режима одного OSD:
|
||||||
@@ -91,6 +96,8 @@ vitastor-disk - инструмент командной строки для уп
|
|||||||
Опции для обоих режимов:
|
Опции для обоих режимов:
|
||||||
|
|
||||||
```
|
```
|
||||||
|
--tags tag1,tag2 Задать теги для новых OSD
|
||||||
|
--weight <number> Задать вес для новых OSD (от 0 до 1)
|
||||||
--journal_size 1G/32M Задать размер журнала (области или раздела журнала)
|
--journal_size 1G/32M Задать размер журнала (области или раздела журнала)
|
||||||
--block_size 1M/128k Задать размер объекта хранилища
|
--block_size 1M/128k Задать размер объекта хранилища
|
||||||
--bitmap_granularity 4k Задать гранулярность битовых карт
|
--bitmap_granularity 4k Задать гранулярность битовых карт
|
||||||
|
|||||||
+1
-1
@@ -16,7 +16,7 @@ async function create_http_server(cfg, handler)
|
|||||||
};
|
};
|
||||||
if (cfg.mon_https_ca)
|
if (cfg.mon_https_ca)
|
||||||
{
|
{
|
||||||
tls.mon_https_ca = await fsp.readFile(cfg.mon_https_ca);
|
tls.ca = await fsp.readFile(cfg.mon_https_ca);
|
||||||
}
|
}
|
||||||
if (cfg.mon_https_client_auth)
|
if (cfg.mon_https_client_auth)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -10,16 +10,19 @@ const NO_OSD = 'Z';
|
|||||||
async function lp_solve(text)
|
async function lp_solve(text)
|
||||||
{
|
{
|
||||||
const cp = child_process.spawn('lp_solve');
|
const cp = child_process.spawn('lp_solve');
|
||||||
let stdout = '', stderr = '', finish_cb;
|
let stdout = '', stderr = '', finish_cb, finished = 0;
|
||||||
cp.stdout.on('data', buf => stdout += buf.toString());
|
cp.stdout.on('data', buf => stdout += buf.toString());
|
||||||
cp.stderr.on('data', buf => stderr += buf.toString());
|
cp.stderr.on('data', buf => stderr += buf.toString());
|
||||||
cp.on('exit', () => finish_cb && finish_cb());
|
cp.stdout.on('end', () => finish_cb());
|
||||||
|
cp.stderr.on('end', () => finish_cb());
|
||||||
cp.stdin.write(text);
|
cp.stdin.write(text);
|
||||||
cp.stdin.end();
|
cp.stdin.end();
|
||||||
if (cp.exitCode == null)
|
await new Promise(ok => (finish_cb = () =>
|
||||||
{
|
{
|
||||||
await new Promise(ok => finish_cb = ok);
|
finished++;
|
||||||
}
|
if (finished == 2)
|
||||||
|
ok();
|
||||||
|
}));
|
||||||
if (!stdout.trim())
|
if (!stdout.trim())
|
||||||
{
|
{
|
||||||
return null;
|
return null;
|
||||||
|
|||||||
+2
-2
@@ -15,7 +15,7 @@ function get_osd_tree(global_config, state)
|
|||||||
const stat = state.osd.stats[osd_num];
|
const stat = state.osd.stats[osd_num];
|
||||||
const osd_cfg = state.config.osd[osd_num];
|
const osd_cfg = state.config.osd[osd_num];
|
||||||
let reweight = osd_cfg == null ? 1 : Number(osd_cfg.reweight);
|
let reweight = osd_cfg == null ? 1 : Number(osd_cfg.reweight);
|
||||||
if (isNaN(reweight) || reweight < 0 || reweight > 0)
|
if (isNaN(reweight) || reweight < 0 || reweight > 1)
|
||||||
reweight = 1;
|
reweight = 1;
|
||||||
if (stat && stat.size && reweight && (state.osd.state[osd_num] || Number(stat.time) >= down_time ||
|
if (stat && stat.size && reweight && (state.osd.state[osd_num] || Number(stat.time) >= down_time ||
|
||||||
osd_cfg && osd_cfg.noout))
|
osd_cfg && osd_cfg.noout))
|
||||||
@@ -87,7 +87,7 @@ function make_hier_tree(global_config, tree)
|
|||||||
tree[''] = { children: [] };
|
tree[''] = { children: [] };
|
||||||
for (const node_id in tree)
|
for (const node_id in tree)
|
||||||
{
|
{
|
||||||
if (node_id === '' || !(tree[node_id].children||[]).length && (tree[node_id].size||0) <= 0)
|
if (node_id === '')
|
||||||
{
|
{
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|||||||
+2
-2
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "vitastor-mon",
|
"name": "vitastor-mon",
|
||||||
"version": "2.3.0",
|
"version": "3.0.6",
|
||||||
"description": "Vitastor SDS monitor service",
|
"description": "Vitastor SDS monitor service",
|
||||||
"main": "mon-main.js",
|
"main": "mon-main.js",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
@@ -9,7 +9,7 @@
|
|||||||
"author": "Vitaliy Filippov",
|
"author": "Vitaliy Filippov",
|
||||||
"license": "UNLICENSED",
|
"license": "UNLICENSED",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"antietcd": "^1.1.3",
|
"antietcd": "^1.2.4",
|
||||||
"sprintf-js": "^1.1.2",
|
"sprintf-js": "^1.1.2",
|
||||||
"ws": "^7.2.5"
|
"ws": "^7.2.5"
|
||||||
},
|
},
|
||||||
|
|||||||
+17
-4
@@ -9,7 +9,6 @@ const LPOptimizer = require('./lp_optimizer/lp_optimizer.js');
|
|||||||
const { scale_pg_count } = require('./pg_utils.js');
|
const { scale_pg_count } = require('./pg_utils.js');
|
||||||
const { make_hier_tree, filter_osds_by_root_node,
|
const { make_hier_tree, filter_osds_by_root_node,
|
||||||
filter_osds_by_tags, filter_osds_by_block_layout, get_affinity_osds } = require('./osd_tree.js');
|
filter_osds_by_tags, filter_osds_by_block_layout, get_affinity_osds } = require('./osd_tree.js');
|
||||||
const { select_murmur3 } = require('./lp_optimizer/murmur3.js');
|
|
||||||
|
|
||||||
function pick_primary(pool_id, pg_num, pool_config, osd_set, up_osds, aff_osds)
|
function pick_primary(pool_id, pg_num, pool_config, osd_set, up_osds, aff_osds)
|
||||||
{
|
{
|
||||||
@@ -39,7 +38,7 @@ function pick_primary(pool_id, pg_num, pool_config, osd_set, up_osds, aff_osds)
|
|||||||
{
|
{
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
return alive_set[select_murmur3(alive_set.length, osd_num => pool_id+'/'+pg_num+'/'+osd_num)];
|
return alive_set[pg_num % alive_set.length];
|
||||||
}
|
}
|
||||||
|
|
||||||
function recheck_primary(state, global_config, up_osds, osd_tree)
|
function recheck_primary(state, global_config, up_osds, osd_tree)
|
||||||
@@ -53,6 +52,7 @@ function recheck_primary(state, global_config, up_osds, osd_tree)
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
const aff_osds = get_affinity_osds(pool_cfg, up_osds, osd_tree);
|
const aff_osds = get_affinity_osds(pool_cfg, up_osds, osd_tree);
|
||||||
|
let paused = false;
|
||||||
for (let pg_num = 1; pg_num <= pool_cfg.pg_count; pg_num++)
|
for (let pg_num = 1; pg_num <= pool_cfg.pg_count; pg_num++)
|
||||||
{
|
{
|
||||||
if (!state.pg.config.items[pool_id])
|
if (!state.pg.config.items[pool_id])
|
||||||
@@ -75,6 +75,19 @@ function recheck_primary(state, global_config, up_osds, osd_tree)
|
|||||||
);
|
);
|
||||||
new_pg_config.items[pool_id][pg_num].primary = new_primary;
|
new_pg_config.items[pool_id][pg_num].primary = new_primary;
|
||||||
}
|
}
|
||||||
|
paused = paused || !!pg_cfg.pause;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (paused)
|
||||||
|
{
|
||||||
|
if (!new_pg_config)
|
||||||
|
{
|
||||||
|
new_pg_config = JSON.parse(JSON.stringify(state.pg.config));
|
||||||
|
}
|
||||||
|
console.log(`Resuming paused pool ${pool_id}`);
|
||||||
|
for (const pg in new_pg_config.items[pool_id])
|
||||||
|
{
|
||||||
|
delete new_pg_config.items[pool_id][pg].pause;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -179,10 +192,10 @@ async function generate_pool_pgs(state, global_config, pool_id, osd_tree, levels
|
|||||||
const rules = use_rules ? get_pg_rules(pool_id, pool_cfg, global_config.placement_levels) : null;
|
const rules = use_rules ? get_pg_rules(pool_id, pool_cfg, global_config.placement_levels) : null;
|
||||||
const folded = fold_failure_domains(Object.values(pool_tree), use_rules ? rules : [ [ [ pool_cfg.failure_domain ] ] ]);
|
const folded = fold_failure_domains(Object.values(pool_tree), use_rules ? rules : [ [ [ pool_cfg.failure_domain ] ] ]);
|
||||||
// FIXME: Remove/merge make_hier_tree() step somewhere, however it's needed to remove empty nodes
|
// FIXME: Remove/merge make_hier_tree() step somewhere, however it's needed to remove empty nodes
|
||||||
const folded_tree = make_hier_tree(global_config, folded.nodes);
|
const folded_tree = make_hier_tree(global_config, folded.nodes.reduce((a, c) => { a[c.id] = c; return a; }, {}));
|
||||||
const old_pg_count = prev_pgs.length;
|
const old_pg_count = prev_pgs.length;
|
||||||
const optimize_cfg = {
|
const optimize_cfg = {
|
||||||
osd_weights: folded.nodes.reduce((a, c) => { if (Number(c.id)) { a[c.id] = c.size; } return a; }, {}),
|
osd_weights: folded.nodes.reduce((a, c) => { if (/^\d+$/.exec(c.id) && c.size != null) { a[c.id] = c.size||0; } return a; }, {}),
|
||||||
combinator: use_rules
|
combinator: use_rules
|
||||||
// new algorithm:
|
// new algorithm:
|
||||||
? new RuleCombinator(folded_tree, rules, pool_cfg.max_osd_combinations)
|
? new RuleCombinator(folded_tree, rules, pool_cfg.max_osd_combinations)
|
||||||
|
|||||||
@@ -52,15 +52,16 @@ async function run()
|
|||||||
process.exit(1);
|
process.exit(1);
|
||||||
}
|
}
|
||||||
const etcds = (config.etcd_address instanceof Array ? config.etcd_address : (''+config.etcd_address).split(/,/))
|
const etcds = (config.etcd_address instanceof Array ? config.etcd_address : (''+config.etcd_address).split(/,/))
|
||||||
.map(s => (''+s).replace(/^https?:\/\/\[?|\]?(:\d+)?(\/.*)?$/g, '').toLowerCase());
|
.map(s => (''+s).replace(/^https?:\/\/|(:\d+)?(\/.*)?$/g, '').replace(/^\[(.*)\]$/, '$1').toLowerCase());
|
||||||
const num = select_local_etcd(etcds);
|
const num = select_local_etcd(etcds);
|
||||||
if (num < 0)
|
if (num < 0)
|
||||||
{
|
{
|
||||||
console.log('No matching IPs in etcd_address from '+config_path);
|
console.log('No matching IPs in etcd_address from '+config_path);
|
||||||
process.exit(0);
|
process.exit(0);
|
||||||
}
|
}
|
||||||
|
const etcd_url = 'http://' + (etcds[num].indexOf(':') >= 0 ? '['+etcds[num]+']' : etcds[num]);
|
||||||
const etcd_name = 'etcd'+etcds[num].replace(/[^0-9a-z_]/ig, '_');
|
const etcd_name = 'etcd'+etcds[num].replace(/[^0-9a-z_]/ig, '_');
|
||||||
const etcd_cluster = etcds.map(e => `etcd${e.replace(/[^0-9a-z_]/ig, '_')}=http://${e}:2380`).join(',');
|
const etcd_cluster = etcds.map(e => `etcd${e.replace(/[^0-9a-z_]/ig, '_')}=http://${e.indexOf(':') >= 0 ? '['+e+']' : e}:2380`).join(',');
|
||||||
if (in_docker)
|
if (in_docker)
|
||||||
{
|
{
|
||||||
let etcd_conf = fs.readFileSync("/etc/vitastor/etcd.conf", { encoding: 'utf-8' });
|
let etcd_conf = fs.readFileSync("/etc/vitastor/etcd.conf", { encoding: 'utf-8' });
|
||||||
@@ -83,8 +84,8 @@ Wants=network-online.target local-fs.target time-sync.target
|
|||||||
Restart=always
|
Restart=always
|
||||||
Environment=GOGC=50
|
Environment=GOGC=50
|
||||||
ExecStart=etcd --name ${etcd_name} --data-dir /var/lib/etcd/vitastor \\
|
ExecStart=etcd --name ${etcd_name} --data-dir /var/lib/etcd/vitastor \\
|
||||||
--snapshot-count 10000 --advertise-client-urls http://${etcds[num]}:2379 --listen-client-urls http://${etcds[num]}:2379 \\
|
--snapshot-count 10000 --advertise-client-urls ${etcd_url}:2379 --listen-client-urls ${etcd_url}:2379 \\
|
||||||
--initial-advertise-peer-urls http://${etcds[num]}:2380 --listen-peer-urls http://${etcds[num]}:2380 \\
|
--initial-advertise-peer-urls ${etcd_url}:2380 --listen-peer-urls ${etcd_url}:2380 \\
|
||||||
--initial-cluster-token vitastor-etcd-1 --initial-cluster ${etcd_cluster} \\
|
--initial-cluster-token vitastor-etcd-1 --initial-cluster ${etcd_cluster} \\
|
||||||
--initial-cluster-state new --max-txn-ops=100000 --max-request-bytes=104857600 \\
|
--initial-cluster-state new --max-txn-ops=100000 --max-request-bytes=104857600 \\
|
||||||
--auto-compaction-retention=10 --auto-compaction-mode=revision
|
--auto-compaction-retention=10 --auto-compaction-mode=revision
|
||||||
|
|||||||
@@ -276,6 +276,10 @@ function sum_inode_stats(state, prev_stats)
|
|||||||
}
|
}
|
||||||
for (const pool_id in osd_diff.inode_stats)
|
for (const pool_id in osd_diff.inode_stats)
|
||||||
{
|
{
|
||||||
|
if (!inode_stats[pool_id])
|
||||||
|
{
|
||||||
|
continue;
|
||||||
|
}
|
||||||
for (const inode_num in prev_stats.osd_diff[osd].inode_stats[pool_id])
|
for (const inode_num in prev_stats.osd_diff[osd].inode_stats[pool_id])
|
||||||
{
|
{
|
||||||
inode_stats[pool_id][inode_num] = inode_stats[pool_id][inode_num] || inode_stub();
|
inode_stats[pool_id][inode_num] = inode_stats[pool_id][inode_num] || inode_stub();
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "vitastor",
|
"name": "vitastor",
|
||||||
"version": "2.3.0",
|
"version": "3.0.6",
|
||||||
"description": "Low-level native bindings to Vitastor client library",
|
"description": "Low-level native bindings to Vitastor client library",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"keywords": [
|
"keywords": [
|
||||||
|
|||||||
@@ -499,4 +499,55 @@ sub rename_volume
|
|||||||
return "${storeid}:${base_name}${target_volname}";
|
return "${storeid}:${base_name}${target_volname}";
|
||||||
}
|
}
|
||||||
|
|
||||||
|
sub _monkey_patch_qemu_blockdev_options
|
||||||
|
{
|
||||||
|
my ($cfg, $volid, $machine_version, $options) = @_;
|
||||||
|
my ($storeid, $volname) = PVE::Storage::parse_volume_id($volid);
|
||||||
|
|
||||||
|
my $scfg = PVE::Storage::storage_config($cfg, $storeid);
|
||||||
|
|
||||||
|
my $plugin = PVE::Storage::Plugin->lookup($scfg->{type});
|
||||||
|
|
||||||
|
my ($vtype) = $plugin->parse_volname($volname);
|
||||||
|
die "cannot use volume of type '$vtype' as a QEMU blockdevice\n"
|
||||||
|
if $vtype ne 'images' && $vtype ne 'iso' && $vtype ne 'import';
|
||||||
|
|
||||||
|
return $plugin->qemu_blockdev_options($scfg, $storeid, $volname, $machine_version, $options);
|
||||||
|
}
|
||||||
|
|
||||||
|
sub qemu_blockdev_options
|
||||||
|
{
|
||||||
|
my ($class, $scfg, $storeid, $volname, $machine_version, $options) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
my ($vtype, $name, $vmid) = $class->parse_volname($volname);
|
||||||
|
$name .= '@'.$options->{'snapshot-name'} if $options->{'snapshot-name'};
|
||||||
|
if ($scfg->{vitastor_nbd})
|
||||||
|
{
|
||||||
|
my $mapped = run_cli($scfg, [ 'ls' ], binary => '/usr/bin/vitastor-nbd');
|
||||||
|
my ($kerneldev) = grep { $mapped->{$_}->{image} eq $prefix.$name } keys %$mapped;
|
||||||
|
die "Image not mapped via NBD" if !$kerneldev;
|
||||||
|
return { driver => 'host_device', filename => $kerneldev };
|
||||||
|
}
|
||||||
|
my $blockdev = {
|
||||||
|
driver => 'vitastor',
|
||||||
|
image => $prefix.$name,
|
||||||
|
};
|
||||||
|
if ($scfg->{vitastor_config_path})
|
||||||
|
{
|
||||||
|
$blockdev->{'config-path'} = $scfg->{vitastor_config_path};
|
||||||
|
}
|
||||||
|
if ($scfg->{vitastor_etcd_address})
|
||||||
|
{
|
||||||
|
# FIXME This is the only exception: etcd_address -> etcd_host for qemu
|
||||||
|
$blockdev->{'etcd-host'} = $scfg->{vitastor_etcd_address};
|
||||||
|
}
|
||||||
|
if ($scfg->{vitastor_etcd_prefix})
|
||||||
|
{
|
||||||
|
$blockdev->{'etcd-prefix'} = $scfg->{vitastor_etcd_prefix};
|
||||||
|
}
|
||||||
|
return $blockdev;
|
||||||
|
}
|
||||||
|
|
||||||
|
*PVE::Storage::qemu_blockdev_options = *_monkey_patch_qemu_blockdev_options;
|
||||||
|
|
||||||
1;
|
1;
|
||||||
|
|||||||
+30
-232
@@ -50,7 +50,7 @@ from cinder.volume import configuration
|
|||||||
from cinder.volume import driver
|
from cinder.volume import driver
|
||||||
from cinder.volume import volume_utils
|
from cinder.volume import volume_utils
|
||||||
|
|
||||||
VITASTOR_VERSION = '2.3.0'
|
VITASTOR_VERSION = '3.0.6'
|
||||||
|
|
||||||
LOG = logging.getLogger(__name__)
|
LOG = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -275,7 +275,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
LOG.exception('error getting vitastor pool stats: '+str(e))
|
LOG.exception('error getting vitastor pool stats: '+str(e))
|
||||||
|
|
||||||
self._stats = stats
|
self._stats = stats
|
||||||
|
|
||||||
def get_volume_stats(self, refresh=False):
|
def get_volume_stats(self, refresh=False):
|
||||||
"""Get volume stats.
|
"""Get volume stats.
|
||||||
If 'refresh' is True, run update the stats first.
|
If 'refresh' is True, run update the stats first.
|
||||||
@@ -291,6 +291,14 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
else:
|
else:
|
||||||
return (1 + resp['kvs'][0]['value'], resp['kvs'][0]['mod_revision'])
|
return (1 + resp['kvs'][0]['value'], resp['kvs'][0]['mod_revision'])
|
||||||
|
|
||||||
|
def _cli(self, descr, *args):
|
||||||
|
args = [ 'vitastor-cli', *args, *(self._vitastor_args()) ]
|
||||||
|
try:
|
||||||
|
self._execute(*args)
|
||||||
|
except processutils.ProcessExecutionError as exc:
|
||||||
|
LOG.error("Failed to "+descr+": "+exc)
|
||||||
|
raise exception.VolumeBackendAPIException(data = exc.stderr)
|
||||||
|
|
||||||
def create_volume(self, volume):
|
def create_volume(self, volume):
|
||||||
"""Creates a logical volume."""
|
"""Creates a logical volume."""
|
||||||
|
|
||||||
@@ -302,7 +310,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
|
|
||||||
LOG.debug("creating volume '%s'", vol_name)
|
LOG.debug("creating volume '%s'", vol_name)
|
||||||
|
|
||||||
self._create_image(vol_name, { 'size': size })
|
self._cli('create volume', 'create', vol_name, '--size', size)
|
||||||
|
|
||||||
if volume.encryption_key_id:
|
if volume.encryption_key_id:
|
||||||
self._create_encrypted_volume(volume, volume.obj_context)
|
self._create_encrypted_volume(volume, volume.obj_context)
|
||||||
@@ -346,7 +354,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
snap_name = utils.convert_str(snapshot.name)
|
snap_name = utils.convert_str(snapshot.name)
|
||||||
if snap_name.find('@') >= 0 or snap_name.find('/') >= 0:
|
if snap_name.find('@') >= 0 or snap_name.find('/') >= 0:
|
||||||
raise exception.VolumeBackendAPIException(data = '@ and / are forbidden in volume and snapshot names')
|
raise exception.VolumeBackendAPIException(data = '@ and / are forbidden in volume and snapshot names')
|
||||||
self._create_snapshot(vol_name, vol_name+'@'+snap_name)
|
self._cli('create snapshot', 'snap-create', vol_name+'@'+snap_name)
|
||||||
|
|
||||||
def snapshot_revert_use_temp_snapshot(self):
|
def snapshot_revert_use_temp_snapshot(self):
|
||||||
"""Disable the use of a temporary snapshot on revert."""
|
"""Disable the use of a temporary snapshot on revert."""
|
||||||
@@ -359,21 +367,8 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
snap_name = utils.convert_str(snapshot.name)
|
snap_name = utils.convert_str(snapshot.name)
|
||||||
|
|
||||||
# Delete the image and recreate it from the snapshot
|
# Delete the image and recreate it from the snapshot
|
||||||
args = [ 'vitastor-cli', 'rm', vol_name, *(self._vitastor_args()) ]
|
self._cli('delete image', 'rm', vol_name)
|
||||||
try:
|
self._cli('recreate image', 'create', '--parent', vol_name+'@'+snap_name, vol_name)
|
||||||
self._execute(*args)
|
|
||||||
except processutils.ProcessExecutionError as exc:
|
|
||||||
LOG.error("Failed to delete image "+vol_name+": "+exc)
|
|
||||||
raise exception.VolumeBackendAPIException(data = exc.stderr)
|
|
||||||
args = [
|
|
||||||
'vitastor-cli', 'create', '--parent', vol_name+'@'+snap_name,
|
|
||||||
vol_name, *(self._vitastor_args())
|
|
||||||
]
|
|
||||||
try:
|
|
||||||
self._execute(*args)
|
|
||||||
except processutils.ProcessExecutionError as exc:
|
|
||||||
LOG.error("Failed to recreate image "+vol_name+" from "+vol_name+"@"+snap_name+": "+exc)
|
|
||||||
raise exception.VolumeBackendAPIException(data = exc.stderr)
|
|
||||||
|
|
||||||
def delete_snapshot(self, snapshot):
|
def delete_snapshot(self, snapshot):
|
||||||
"""Deletes a snapshot."""
|
"""Deletes a snapshot."""
|
||||||
@@ -381,15 +376,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
vol_name = utils.convert_str(snapshot.volume_name)
|
vol_name = utils.convert_str(snapshot.volume_name)
|
||||||
snap_name = utils.convert_str(snapshot.name)
|
snap_name = utils.convert_str(snapshot.name)
|
||||||
|
|
||||||
args = [
|
self._cli('remove snapshot', 'rm', vol_name+'@'+snap_name)
|
||||||
'vitastor-cli', 'rm', vol_name+'@'+snap_name,
|
|
||||||
*(self._vitastor_args())
|
|
||||||
]
|
|
||||||
try:
|
|
||||||
self._execute(*args)
|
|
||||||
except processutils.ProcessExecutionError as exc:
|
|
||||||
LOG.error("Failed to remove snapshot "+vol_name+'@'+snap_name+": "+exc)
|
|
||||||
raise exception.VolumeBackendAPIException(data = exc.stderr)
|
|
||||||
|
|
||||||
def _child_count(self, parents):
|
def _child_count(self, parents):
|
||||||
children = 0
|
children = 0
|
||||||
@@ -427,13 +414,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
if src_vref.admin_metadata.get('readonly') == 'True':
|
if src_vref.admin_metadata.get('readonly') == 'True':
|
||||||
# source volume is a volume-image cache entry or other readonly volume
|
# source volume is a volume-image cache entry or other readonly volume
|
||||||
# clone without intermediate snapshot
|
# clone without intermediate snapshot
|
||||||
src = self._get_image(src_name)
|
self._cli('create clone', 'create', '--parent', src_name, '--size', size, dest_name)
|
||||||
LOG.debug("creating image '%s' from '%s'", dest_name, src_name)
|
|
||||||
new_cfg = self._create_image(dest_name, {
|
|
||||||
'size': size,
|
|
||||||
'parent_id': src['idx']['id'],
|
|
||||||
'parent_pool_id': src['idx']['pool_id'],
|
|
||||||
})
|
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
clone_snap = "%s@%s.clone_snap" % (src_name, dest_name)
|
clone_snap = "%s@%s.clone_snap" % (src_name, dest_name)
|
||||||
@@ -446,15 +427,12 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
clone_snap = dest_name
|
clone_snap = dest_name
|
||||||
make_img = False
|
make_img = False
|
||||||
|
|
||||||
LOG.debug("creating layer '%s' under '%s'", clone_snap, src_name)
|
LOG.debug("creating snapshot '%s'", clone_snap)
|
||||||
new_cfg = self._create_snapshot(src_name, clone_snap, True)
|
self._cli('create base snapshot', 'snap-create', '--allow-existing', '1', clone_snap)
|
||||||
|
|
||||||
if make_img:
|
if make_img:
|
||||||
# Then create a clone from it
|
# Then create a clone from it
|
||||||
new_cfg = self._create_image(dest_name, {
|
self._cli('create clone', 'create', '--parent', clone_snap, '--size', size, dest_name)
|
||||||
'size': size,
|
|
||||||
'parent_id': new_cfg['parent_id'],
|
|
||||||
'parent_pool_id': new_cfg['parent_pool_id'],
|
|
||||||
})
|
|
||||||
|
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
@@ -464,7 +442,8 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
vol_name = utils.convert_str(volume.name)
|
vol_name = utils.convert_str(volume.name)
|
||||||
snap_name = utils.convert_str(snapshot.name)
|
snap_name = utils.convert_str(snapshot.name)
|
||||||
|
|
||||||
snap = self._get_image('volume-'+snapshot.volume_id+'@'+snap_name)
|
src_snap = 'volume-'+snapshot.volume_id+'@'+snap_name
|
||||||
|
snap = self._get_image(src_snap)
|
||||||
if not snap:
|
if not snap:
|
||||||
raise exception.SnapshotNotFound(snapshot_id = snap_name)
|
raise exception.SnapshotNotFound(snapshot_id = snap_name)
|
||||||
snap_inode_id = int(resp['responses'][0]['kvs'][0]['value']['id'])
|
snap_inode_id = int(resp['responses'][0]['kvs'][0]['value']['id'])
|
||||||
@@ -473,12 +452,8 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
size = snap['cfg']['size']
|
size = snap['cfg']['size']
|
||||||
if int(volume.size):
|
if int(volume.size):
|
||||||
size = int(volume.size) * units.Gi
|
size = int(volume.size) * units.Gi
|
||||||
new_cfg = self._create_image(vol_name, {
|
|
||||||
'size': size,
|
|
||||||
'parent_id': snap['idx']['id'],
|
|
||||||
'parent_pool_id': snap['idx']['pool_id'],
|
|
||||||
})
|
|
||||||
|
|
||||||
|
self._cli('create clone', 'create', vol_name, '--size', size, '--parent', src_snap)
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
def _vitastor_args(self):
|
def _vitastor_args(self):
|
||||||
@@ -505,49 +480,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
"""Deletes a logical volume."""
|
"""Deletes a logical volume."""
|
||||||
|
|
||||||
vol_name = utils.convert_str(volume.name)
|
vol_name = utils.convert_str(volume.name)
|
||||||
|
self._cli('delete volume', 'rm', '--matching', vol_name, vol_name+'@*', '--progress', '0')
|
||||||
# Find the volume and all its snapshots
|
|
||||||
range_end = b'index/image/' + vol_name.encode('utf-8')
|
|
||||||
range_end = range_end[0 : len(range_end)-1] + six.int2byte(range_end[len(range_end)-1] + 1)
|
|
||||||
resp = self._etcd_txn({ 'success': [
|
|
||||||
{ 'request_range': { 'key': 'index/image/'+vol_name, 'range_end': range_end } },
|
|
||||||
] })
|
|
||||||
if len(resp['responses'][0]['kvs']) == 0:
|
|
||||||
# already deleted
|
|
||||||
LOG.info("volume %s no longer exists in backend", vol_name)
|
|
||||||
return
|
|
||||||
layers = resp['responses'][0]['kvs']
|
|
||||||
layer_ids = {}
|
|
||||||
for kv in layers:
|
|
||||||
inode_id = int(kv['value']['id'])
|
|
||||||
pool_id = int(kv['value']['pool_id'])
|
|
||||||
inode_pool_id = (pool_id << 48) | (inode_id & 0xffffffffffff)
|
|
||||||
layer_ids[inode_pool_id] = True
|
|
||||||
|
|
||||||
# Check if the volume has clones and raise 'busy' if so
|
|
||||||
children = self._child_count(layer_ids)
|
|
||||||
if children > 0:
|
|
||||||
raise exception.VolumeIsBusy(volume_name = vol_name)
|
|
||||||
|
|
||||||
# Clear data
|
|
||||||
for kv in layers:
|
|
||||||
args = [
|
|
||||||
'vitastor-cli', 'rm-data', '--pool', str(kv['value']['pool_id']),
|
|
||||||
'--inode', str(kv['value']['id']), '--progress', '0',
|
|
||||||
*(self._vitastor_args())
|
|
||||||
]
|
|
||||||
try:
|
|
||||||
self._execute(*args)
|
|
||||||
except processutils.ProcessExecutionError as exc:
|
|
||||||
LOG.error("Failed to remove layer "+kv['key']+": "+exc)
|
|
||||||
raise exception.VolumeBackendAPIException(data = exc.stderr)
|
|
||||||
|
|
||||||
# Delete all layers from etcd
|
|
||||||
requests = []
|
|
||||||
for kv in layers:
|
|
||||||
requests.append({ 'request_delete_range': { 'key': kv['key'] } })
|
|
||||||
requests.append({ 'request_delete_range': { 'key': 'config/inode/'+str(kv['value']['pool_id'])+'/'+str(kv['value']['id']) } })
|
|
||||||
self._etcd_txn({ 'success': requests })
|
|
||||||
|
|
||||||
def retype(self, context, volume, new_type, diff, host):
|
def retype(self, context, volume, new_type, diff, host):
|
||||||
"""Change extra type specifications for a volume."""
|
"""Change extra type specifications for a volume."""
|
||||||
@@ -567,98 +500,6 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
"""Removes an export for a logical volume."""
|
"""Removes an export for a logical volume."""
|
||||||
pass
|
pass
|
||||||
|
|
||||||
def _create_image(self, vol_name, cfg):
|
|
||||||
pool_s = str(self.cfg['pool_id'])
|
|
||||||
image_id = 0
|
|
||||||
while image_id == 0:
|
|
||||||
# check if the image already exists and find a free ID
|
|
||||||
resp = self._etcd_txn({ 'success': [
|
|
||||||
{ 'request_range': { 'key': 'index/image/'+vol_name } },
|
|
||||||
{ 'request_range': { 'key': 'index/maxid/'+pool_s } },
|
|
||||||
] })
|
|
||||||
if len(resp['responses'][0]['kvs']) > 0:
|
|
||||||
# already exists
|
|
||||||
raise exception.VolumeBackendAPIException(data = 'Volume '+vol_name+' already exists')
|
|
||||||
image_id, id_mod = self._next_id(resp['responses'][1])
|
|
||||||
# try to create the image
|
|
||||||
resp = self._etcd_txn({ 'compare': [
|
|
||||||
{ 'target': 'MOD', 'mod_revision': id_mod, 'key': 'index/maxid/'+pool_s },
|
|
||||||
{ 'target': 'VERSION', 'version': 0, 'key': 'index/image/'+vol_name },
|
|
||||||
{ 'target': 'VERSION', 'version': 0, 'key': 'config/inode/'+pool_s+'/'+str(image_id) },
|
|
||||||
], 'success': [
|
|
||||||
{ 'request_put': { 'key': 'index/maxid/'+pool_s, 'value': image_id } },
|
|
||||||
{ 'request_put': { 'key': 'index/image/'+vol_name, 'value': json.dumps({
|
|
||||||
'id': image_id, 'pool_id': self.cfg['pool_id']
|
|
||||||
}) } },
|
|
||||||
{ 'request_put': { 'key': 'config/inode/'+pool_s+'/'+str(image_id), 'value': json.dumps({
|
|
||||||
**cfg, 'name': vol_name,
|
|
||||||
}) } },
|
|
||||||
] })
|
|
||||||
if not resp.get('succeeded'):
|
|
||||||
# repeat
|
|
||||||
image_id = 0
|
|
||||||
|
|
||||||
def _create_snapshot(self, vol_name, snap_vol_name, allow_existing = False):
|
|
||||||
while True:
|
|
||||||
# check if the image already exists and snapshot doesn't
|
|
||||||
resp = self._etcd_txn({ 'success': [
|
|
||||||
{ 'request_range': { 'key': 'index/image/'+vol_name } },
|
|
||||||
{ 'request_range': { 'key': 'index/image/'+snap_vol_name } },
|
|
||||||
] })
|
|
||||||
if len(resp['responses'][0]['kvs']) == 0:
|
|
||||||
raise exception.VolumeBackendAPIException(data = 'Volume '+vol_name+' does not exist')
|
|
||||||
if len(resp['responses'][1]['kvs']) > 0:
|
|
||||||
if allow_existing:
|
|
||||||
snap_idx = resp['responses'][1]['kvs'][0]['value']
|
|
||||||
resp = self._etcd_txn({ 'success': [
|
|
||||||
{ 'request_range': { 'key': 'config/inode/'+str(snap_idx['pool_id'])+'/'+str(snap_idx['id']) } },
|
|
||||||
] })
|
|
||||||
if len(resp['responses'][0]['kvs']) == 0:
|
|
||||||
raise exception.VolumeBackendAPIException(data =
|
|
||||||
'Volume '+snap_vol_name+' is already indexed, but does not exist'
|
|
||||||
)
|
|
||||||
return resp['responses'][0]['kvs'][0]['value']
|
|
||||||
raise exception.VolumeBackendAPIException(
|
|
||||||
data = 'Volume '+snap_vol_name+' already exists'
|
|
||||||
)
|
|
||||||
vol_idx = resp['responses'][0]['kvs'][0]['value']
|
|
||||||
vol_idx_mod = resp['responses'][0]['kvs'][0]['mod_revision']
|
|
||||||
# get image inode config and find a new ID
|
|
||||||
resp = self._etcd_txn({ 'success': [
|
|
||||||
{ 'request_range': { 'key': 'config/inode/'+str(vol_idx['pool_id'])+'/'+str(vol_idx['id']) } },
|
|
||||||
{ 'request_range': { 'key': 'index/maxid/'+str(self.cfg['pool_id']) } },
|
|
||||||
] })
|
|
||||||
if len(resp['responses'][0]['kvs']) == 0:
|
|
||||||
raise exception.VolumeBackendAPIException(data = 'Volume '+vol_name+' does not exist')
|
|
||||||
vol_cfg = resp['responses'][0]['kvs'][0]['value']
|
|
||||||
vol_mod = resp['responses'][0]['kvs'][0]['mod_revision']
|
|
||||||
new_id, id_mod = self._next_id(resp['responses'][1])
|
|
||||||
# try to redirect image to the new inode
|
|
||||||
new_cfg = {
|
|
||||||
**vol_cfg, 'name': vol_name, 'parent_id': vol_idx['id'], 'parent_pool_id': vol_idx['pool_id']
|
|
||||||
}
|
|
||||||
resp = self._etcd_txn({ 'compare': [
|
|
||||||
{ 'target': 'MOD', 'mod_revision': vol_idx_mod, 'key': 'index/image/'+vol_name },
|
|
||||||
{ 'target': 'MOD', 'mod_revision': vol_mod, 'key': 'config/inode/'+str(vol_idx['pool_id'])+'/'+str(vol_idx['id']) },
|
|
||||||
{ 'target': 'MOD', 'mod_revision': id_mod, 'key': 'index/maxid/'+str(self.cfg['pool_id']) },
|
|
||||||
{ 'target': 'VERSION', 'version': 0, 'key': 'index/image/'+snap_vol_name },
|
|
||||||
{ 'target': 'VERSION', 'version': 0, 'key': 'config/inode/'+str(self.cfg['pool_id'])+'/'+str(new_id) },
|
|
||||||
], 'success': [
|
|
||||||
{ 'request_put': { 'key': 'index/maxid/'+str(self.cfg['pool_id']), 'value': new_id } },
|
|
||||||
{ 'request_put': { 'key': 'index/image/'+vol_name, 'value': json.dumps({
|
|
||||||
'id': new_id, 'pool_id': self.cfg['pool_id']
|
|
||||||
}) } },
|
|
||||||
{ 'request_put': { 'key': 'config/inode/'+str(self.cfg['pool_id'])+'/'+str(new_id), 'value': json.dumps(new_cfg) } },
|
|
||||||
{ 'request_put': { 'key': 'index/image/'+snap_vol_name, 'value': json.dumps({
|
|
||||||
'id': vol_idx['id'], 'pool_id': vol_idx['pool_id']
|
|
||||||
}) } },
|
|
||||||
{ 'request_put': { 'key': 'config/inode/'+str(vol_idx['pool_id'])+'/'+str(vol_idx['id']), 'value': json.dumps({
|
|
||||||
**vol_cfg, 'name': snap_vol_name, 'readonly': True
|
|
||||||
}) } }
|
|
||||||
] })
|
|
||||||
if resp.get('succeeded'):
|
|
||||||
return new_cfg
|
|
||||||
|
|
||||||
def initialize_connection(self, volume, connector):
|
def initialize_connection(self, volume, connector):
|
||||||
data = {
|
data = {
|
||||||
'driver_volume_type': 'vitastor',
|
'driver_volume_type': 'vitastor',
|
||||||
@@ -697,13 +538,9 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
size = int(volume.size) * units.Gi
|
size = int(volume.size) * units.Gi
|
||||||
dest_name = utils.convert_str(volume.name)
|
dest_name = utils.convert_str(volume.name)
|
||||||
# Find or create the base snapshot
|
# Find or create the base snapshot
|
||||||
snap_cfg = self._create_snapshot(base_vol.name, base_vol.name+'@.clone_snap', True)
|
self._cli('create base snapshot', 'create', '--allow-existing', '1', base_vol.name+'@.clone_snap')
|
||||||
# Then create a clone from it
|
# Then create a clone from it
|
||||||
new_cfg = self._create_image(dest_name, {
|
self._cli('create clone', 'create', dest_name, '--size', size, '--parent', base_vol.name+'@.clone_snap')
|
||||||
'size': size,
|
|
||||||
'parent_id': snap_cfg['parent_id'],
|
|
||||||
'parent_pool_id': snap_cfg['parent_pool_id'],
|
|
||||||
})
|
|
||||||
return ({}, True)
|
return ({}, True)
|
||||||
return ({}, False)
|
return ({}, False)
|
||||||
|
|
||||||
@@ -770,26 +607,8 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
def extend_volume(self, volume, new_size):
|
def extend_volume(self, volume, new_size):
|
||||||
"""Extend an existing volume."""
|
"""Extend an existing volume."""
|
||||||
vol_name = utils.convert_str(volume.name)
|
vol_name = utils.convert_str(volume.name)
|
||||||
while True:
|
size = int(new_size) * units.Gi
|
||||||
vol = self._get_image(vol_name)
|
self._cli('extend volume', 'modify', vol_name, '--resize', new_size)
|
||||||
if not vol:
|
|
||||||
raise exception.VolumeBackendAPIException(data = 'Volume '+vol_name+' does not exist')
|
|
||||||
# change size
|
|
||||||
size = int(new_size) * units.Gi
|
|
||||||
if size == vol['cfg']['size']:
|
|
||||||
break
|
|
||||||
resp = self._etcd_txn({ 'compare': [ {
|
|
||||||
'target': 'MOD',
|
|
||||||
'mod_revision': vol['cfg_mod'],
|
|
||||||
'key': 'config/inode/'+str(vol['idx']['pool_id'])+'/'+str(vol['idx']['id']),
|
|
||||||
} ], 'success': [
|
|
||||||
{ 'request_put': {
|
|
||||||
'key': 'config/inode/'+str(vol['idx']['pool_id'])+'/'+str(vol['idx']['id']),
|
|
||||||
'value': json.dumps({ **vol['cfg'], 'size': size }),
|
|
||||||
} },
|
|
||||||
] })
|
|
||||||
if resp.get('succeeded'):
|
|
||||||
break
|
|
||||||
LOG.debug(
|
LOG.debug(
|
||||||
"Extend volume from %(old_size)s GB to %(new_size)s GB.",
|
"Extend volume from %(old_size)s GB to %(new_size)s GB.",
|
||||||
{'old_size': volume.size, 'new_size': new_size}
|
{'old_size': volume.size, 'new_size': new_size}
|
||||||
@@ -862,28 +681,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
"""
|
"""
|
||||||
from_name = self._get_existing_name(existing_ref)
|
from_name = self._get_existing_name(existing_ref)
|
||||||
to_name = utils.convert_str(volume.name)
|
to_name = utils.convert_str(volume.name)
|
||||||
self._rename(from_name, to_name)
|
self._cli('rename', 'modify', from_name, '--rename', to_name)
|
||||||
|
|
||||||
def _rename(self, from_name, to_name):
|
|
||||||
while True:
|
|
||||||
vol = self._get_image(from_name)
|
|
||||||
if not vol:
|
|
||||||
raise exception.VolumeBackendAPIException(data = 'Volume '+from_name+' does not exist')
|
|
||||||
to = self._get_image(to_name)
|
|
||||||
if to:
|
|
||||||
raise exception.VolumeBackendAPIException(data = 'Volume '+to_name+' already exists')
|
|
||||||
resp = self._etcd_txn({ 'compare': [
|
|
||||||
{ 'target': 'MOD', 'mod_revision': vol['idx_mod'], 'key': 'index/image/'+vol['cfg']['name'] },
|
|
||||||
{ 'target': 'MOD', 'mod_revision': vol['cfg_mod'], 'key': 'config/inode/'+str(vol['idx']['pool_id'])+'/'+str(vol['idx']['id']) },
|
|
||||||
{ 'target': 'VERSION', 'version': 0, 'key': 'index/image/'+to_name },
|
|
||||||
], 'success': [
|
|
||||||
{ 'request_delete_range': { 'key': 'index/image/'+vol['cfg']['name'] } },
|
|
||||||
{ 'request_put': { 'key': 'index/image/'+to_name, 'value': json.dumps(vol['idx']) } },
|
|
||||||
{ 'request_put': { 'key': 'config/inode/'+str(vol['idx']['pool_id'])+'/'+str(vol['idx']['id']),
|
|
||||||
'value': json.dumps({ **vol['cfg'], 'name': to_name }) } },
|
|
||||||
] })
|
|
||||||
if resp.get('succeeded'):
|
|
||||||
break
|
|
||||||
|
|
||||||
def unmanage(self, volume):
|
def unmanage(self, volume):
|
||||||
pass
|
pass
|
||||||
@@ -956,7 +754,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
snap_name = self._get_existing_name(existing_ref)
|
snap_name = self._get_existing_name(existing_ref)
|
||||||
from_name = vol_name+'@'+snap_name
|
from_name = vol_name+'@'+snap_name
|
||||||
to_name = vol_name+'@'+utils.convert_str(snapshot.name)
|
to_name = vol_name+'@'+utils.convert_str(snapshot.name)
|
||||||
self._rename(from_name, to_name)
|
self._cli('rename', 'modify', from_name, '--rename', to_name)
|
||||||
|
|
||||||
def unmanage_snapshot(self, snapshot):
|
def unmanage_snapshot(self, snapshot):
|
||||||
"""Removes the specified snapshot from Cinder management."""
|
"""Removes the specified snapshot from Cinder management."""
|
||||||
|
|||||||
@@ -0,0 +1,39 @@
|
|||||||
|
From 98d3f68a40130c438854f61db6025f9e9b099cb6 Mon Sep 17 00:00:00 2001
|
||||||
|
From: Vitaliy Filippov <vitalifster@gmail.com>
|
||||||
|
Date: Sat, 20 Dec 2025 14:44:35 +0300
|
||||||
|
Subject: [PATCH] Do not require atomic writes to be power of 2 sized and
|
||||||
|
aligned on length boundary
|
||||||
|
|
||||||
|
It contradicts NVMe specification where alignment is only required when atomic
|
||||||
|
write boundary (NABSPF/NABO) is set and highly limits usage of NVMe atomic writes
|
||||||
|
|
||||||
|
Signed-off-by: Vitaliy Filippov <vitalifster@gmail.com>
|
||||||
|
---
|
||||||
|
fs/read_write.c | 8 --------
|
||||||
|
1 file changed, 8 deletions(-)
|
||||||
|
|
||||||
|
diff --git a/fs/read_write.c b/fs/read_write.c
|
||||||
|
index 833bae068770..5467d710108d 100644
|
||||||
|
--- a/fs/read_write.c
|
||||||
|
+++ b/fs/read_write.c
|
||||||
|
@@ -1802,17 +1802,9 @@ int generic_file_rw_checks(struct file *file_in, struct file *file_out)
|
||||||
|
|
||||||
|
int generic_atomic_write_valid(struct kiocb *iocb, struct iov_iter *iter)
|
||||||
|
{
|
||||||
|
- size_t len = iov_iter_count(iter);
|
||||||
|
-
|
||||||
|
if (!iter_is_ubuf(iter))
|
||||||
|
return -EINVAL;
|
||||||
|
|
||||||
|
- if (!is_power_of_2(len))
|
||||||
|
- return -EINVAL;
|
||||||
|
-
|
||||||
|
- if (!IS_ALIGNED(iocb->ki_pos, len))
|
||||||
|
- return -EINVAL;
|
||||||
|
-
|
||||||
|
if (!(iocb->ki_flags & IOCB_DIRECT))
|
||||||
|
return -EOPNOTSUPP;
|
||||||
|
|
||||||
|
--
|
||||||
|
2.51.0
|
||||||
|
|
||||||
@@ -0,0 +1,172 @@
|
|||||||
|
Index: pve-qemu-kvm-10.1.2/block/meson.build
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.1.2.orig/block/meson.build
|
||||||
|
+++ pve-qemu-kvm-10.1.2/block/meson.build
|
||||||
|
@@ -126,6 +126,7 @@ foreach m : [
|
||||||
|
[libnfs, 'nfs', files('nfs.c')],
|
||||||
|
[libssh, 'ssh', files('ssh.c')],
|
||||||
|
[rbd, 'rbd', files('rbd.c')],
|
||||||
|
+ [vitastor, 'vitastor', files('vitastor.c')],
|
||||||
|
]
|
||||||
|
if m[0].found()
|
||||||
|
module_ss = ss.source_set()
|
||||||
|
Index: pve-qemu-kvm-10.1.2/meson.build
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.1.2.orig/meson.build
|
||||||
|
+++ pve-qemu-kvm-10.1.2/meson.build
|
||||||
|
@@ -1653,6 +1653,26 @@ if not get_option('rbd').auto() or have_
|
||||||
|
endif
|
||||||
|
endif
|
||||||
|
|
||||||
|
+vitastor = not_found
|
||||||
|
+if not get_option('vitastor').auto() or have_block
|
||||||
|
+ libvitastor_client = cc.find_library('vitastor_client', has_headers: ['vitastor_c.h'],
|
||||||
|
+ required: get_option('vitastor'))
|
||||||
|
+ if libvitastor_client.found()
|
||||||
|
+ if cc.links('''
|
||||||
|
+ #include <vitastor_c.h>
|
||||||
|
+ int main(void) {
|
||||||
|
+ vitastor_c_create_qemu(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0);
|
||||||
|
+ return 0;
|
||||||
|
+ }''', dependencies: libvitastor_client)
|
||||||
|
+ vitastor = declare_dependency(dependencies: libvitastor_client)
|
||||||
|
+ elif get_option('vitastor').enabled()
|
||||||
|
+ error('could not link libvitastor_client')
|
||||||
|
+ else
|
||||||
|
+ warning('could not link libvitastor_client, disabling')
|
||||||
|
+ endif
|
||||||
|
+ endif
|
||||||
|
+endif
|
||||||
|
+
|
||||||
|
glusterfs = not_found
|
||||||
|
glusterfs_ftruncate_has_stat = false
|
||||||
|
glusterfs_iocb_has_stat = false
|
||||||
|
@@ -2552,6 +2572,7 @@ endif
|
||||||
|
config_host_data.set('CONFIG_OPENGL', opengl.found())
|
||||||
|
config_host_data.set('CONFIG_PLUGIN', get_option('plugins'))
|
||||||
|
config_host_data.set('CONFIG_RBD', rbd.found())
|
||||||
|
+config_host_data.set('CONFIG_VITASTOR', vitastor.found())
|
||||||
|
config_host_data.set('CONFIG_RDMA', rdma.found())
|
||||||
|
config_host_data.set('CONFIG_RELOCATABLE', get_option('relocatable'))
|
||||||
|
config_host_data.set('CONFIG_SAFESTACK', get_option('safe_stack'))
|
||||||
|
@@ -4984,6 +5005,7 @@ summary_info += {'fdt support': fd
|
||||||
|
summary_info += {'libcap-ng support': libcap_ng}
|
||||||
|
summary_info += {'bpf support': libbpf}
|
||||||
|
summary_info += {'rbd support': rbd}
|
||||||
|
+summary_info += {'vitastor support': vitastor}
|
||||||
|
summary_info += {'smartcard support': cacard}
|
||||||
|
summary_info += {'U2F support': u2f}
|
||||||
|
summary_info += {'libusb': libusb}
|
||||||
|
Index: pve-qemu-kvm-10.1.2/meson_options.txt
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.1.2.orig/meson_options.txt
|
||||||
|
+++ pve-qemu-kvm-10.1.2/meson_options.txt
|
||||||
|
@@ -202,6 +202,8 @@ option('pvg', type: 'feature', value: 'a
|
||||||
|
description: 'macOS paravirtualized graphics support')
|
||||||
|
option('rbd', type : 'feature', value : 'auto',
|
||||||
|
description: 'Ceph block device driver')
|
||||||
|
+option('vitastor', type : 'feature', value : 'auto',
|
||||||
|
+ description: 'Vitastor block device driver')
|
||||||
|
option('opengl', type : 'feature', value : 'auto',
|
||||||
|
description: 'OpenGL support')
|
||||||
|
option('rdma', type : 'feature', value : 'auto',
|
||||||
|
Index: pve-qemu-kvm-10.1.2/qapi/block-core.json
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.1.2.orig/qapi/block-core.json
|
||||||
|
+++ pve-qemu-kvm-10.1.2/qapi/block-core.json
|
||||||
|
@@ -3647,7 +3647,7 @@
|
||||||
|
'raw', 'rbd',
|
||||||
|
{ 'name': 'replication', 'if': 'CONFIG_REPLICATION' },
|
||||||
|
'pbs',
|
||||||
|
- 'ssh', 'throttle', 'vdi', 'vhdx',
|
||||||
|
+ 'ssh', 'throttle', 'vdi', 'vhdx', 'vitastor',
|
||||||
|
{ 'name': 'virtio-blk-vfio-pci', 'if': 'CONFIG_BLKIO' },
|
||||||
|
{ 'name': 'virtio-blk-vhost-user', 'if': 'CONFIG_BLKIO' },
|
||||||
|
{ 'name': 'virtio-blk-vhost-vdpa', 'if': 'CONFIG_BLKIO' },
|
||||||
|
@@ -4773,6 +4773,28 @@
|
||||||
|
'*server': ['InetSocketAddressBase'] } }
|
||||||
|
|
||||||
|
##
|
||||||
|
+# @BlockdevOptionsVitastor:
|
||||||
|
+#
|
||||||
|
+# Driver specific block device options for vitastor
|
||||||
|
+#
|
||||||
|
+# @image: Image name
|
||||||
|
+# @inode: Inode number
|
||||||
|
+# @pool: Pool ID
|
||||||
|
+# @size: Desired image size in bytes
|
||||||
|
+# @config-path: Path to Vitastor configuration
|
||||||
|
+# @etcd-host: etcd connection address(es)
|
||||||
|
+# @etcd-prefix: etcd key/value prefix
|
||||||
|
+##
|
||||||
|
+{ 'struct': 'BlockdevOptionsVitastor',
|
||||||
|
+ 'data': { '*inode': 'uint64',
|
||||||
|
+ '*pool': 'uint64',
|
||||||
|
+ '*size': 'uint64',
|
||||||
|
+ '*image': 'str',
|
||||||
|
+ '*config-path': 'str',
|
||||||
|
+ '*etcd-host': 'str',
|
||||||
|
+ '*etcd-prefix': 'str' } }
|
||||||
|
+
|
||||||
|
+##
|
||||||
|
# @ReplicationMode:
|
||||||
|
#
|
||||||
|
# An enumeration of replication modes.
|
||||||
|
@@ -5242,6 +5264,7 @@
|
||||||
|
'throttle': 'BlockdevOptionsThrottle',
|
||||||
|
'vdi': 'BlockdevOptionsGenericFormat',
|
||||||
|
'vhdx': 'BlockdevOptionsGenericFormat',
|
||||||
|
+ 'vitastor': 'BlockdevOptionsVitastor',
|
||||||
|
'virtio-blk-vfio-pci':
|
||||||
|
{ 'type': 'BlockdevOptionsVirtioBlkVfioPci',
|
||||||
|
'if': 'CONFIG_BLKIO' },
|
||||||
|
@@ -5722,6 +5745,20 @@
|
||||||
|
'*encrypt' : 'RbdEncryptionCreateOptions' } }
|
||||||
|
|
||||||
|
##
|
||||||
|
+# @BlockdevCreateOptionsVitastor:
|
||||||
|
+#
|
||||||
|
+# Driver specific image creation options for Vitastor.
|
||||||
|
+#
|
||||||
|
+# @location: Where to store the new image file. This location cannot
|
||||||
|
+# point to a snapshot.
|
||||||
|
+#
|
||||||
|
+# @size: Size of the virtual disk in bytes
|
||||||
|
+##
|
||||||
|
+{ 'struct': 'BlockdevCreateOptionsVitastor',
|
||||||
|
+ 'data': { 'location': 'BlockdevOptionsVitastor',
|
||||||
|
+ 'size': 'size' } }
|
||||||
|
+
|
||||||
|
+##
|
||||||
|
# @BlockdevVmdkSubformat:
|
||||||
|
#
|
||||||
|
# Subformat options for VMDK images
|
||||||
|
@@ -5943,6 +5980,7 @@
|
||||||
|
'ssh': 'BlockdevCreateOptionsSsh',
|
||||||
|
'vdi': 'BlockdevCreateOptionsVdi',
|
||||||
|
'vhdx': 'BlockdevCreateOptionsVhdx',
|
||||||
|
+ 'vitastor': 'BlockdevCreateOptionsVitastor',
|
||||||
|
'vmdk': 'BlockdevCreateOptionsVmdk',
|
||||||
|
'vpc': 'BlockdevCreateOptionsVpc'
|
||||||
|
} }
|
||||||
|
Index: pve-qemu-kvm-10.1.2/scripts/meson-buildoptions.sh
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.1.2.orig/scripts/meson-buildoptions.sh
|
||||||
|
+++ pve-qemu-kvm-10.1.2/scripts/meson-buildoptions.sh
|
||||||
|
@@ -175,6 +175,7 @@ meson_options_help() {
|
||||||
|
printf "%s\n" ' qga-vss build QGA VSS support (broken with MinGW)'
|
||||||
|
printf "%s\n" ' qpl Query Processing Library support'
|
||||||
|
printf "%s\n" ' rbd Ceph block device driver'
|
||||||
|
+ printf "%s\n" ' vitastor Vitastor block device driver'
|
||||||
|
printf "%s\n" ' rdma Enable RDMA-based migration'
|
||||||
|
printf "%s\n" ' replication replication support'
|
||||||
|
printf "%s\n" ' rust Rust support'
|
||||||
|
@@ -459,6 +460,8 @@ _meson_option_parse() {
|
||||||
|
--disable-qpl) printf "%s" -Dqpl=disabled ;;
|
||||||
|
--enable-rbd) printf "%s" -Drbd=enabled ;;
|
||||||
|
--disable-rbd) printf "%s" -Drbd=disabled ;;
|
||||||
|
+ --enable-vitastor) printf "%s" -Dvitastor=enabled ;;
|
||||||
|
+ --disable-vitastor) printf "%s" -Dvitastor=disabled ;;
|
||||||
|
--enable-rdma) printf "%s" -Drdma=enabled ;;
|
||||||
|
--disable-rdma) printf "%s" -Drdma=disabled ;;
|
||||||
|
--enable-relocatable) printf "%s" -Drelocatable=true ;;
|
||||||
@@ -0,0 +1,29 @@
|
|||||||
|
diff --git a/src/client/qemu_driver.c b/src/client/qemu_driver.c
|
||||||
|
index d8356dab..5f4cd50d 100644
|
||||||
|
--- a/src/client/qemu_driver.c
|
||||||
|
+++ b/src/client/qemu_driver.c
|
||||||
|
@@ -974,14 +974,21 @@ static void vitastor_co_read_bitmap_cb(void *opaque, long retval, uint8_t *bitma
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
-static int coroutine_fn vitastor_co_block_status(
|
||||||
|
- BlockDriverState *bs, bool want_zero, int64_t offset, int64_t bytes,
|
||||||
|
- int64_t *pnum, int64_t *map, BlockDriverState **file)
|
||||||
|
+static int coroutine_fn vitastor_co_block_status(BlockDriverState *bs,
|
||||||
|
+#if QEMU_VERSION_MAJOR > 10 || QEMU_VERSION_MAJOR == 10 && QEMU_VERSION_MINOR >= 1
|
||||||
|
+ unsigned int mode,
|
||||||
|
+#else
|
||||||
|
+ bool want_zero,
|
||||||
|
+#endif
|
||||||
|
+ int64_t offset, int64_t bytes, int64_t *pnum, int64_t *map, BlockDriverState **file)
|
||||||
|
{
|
||||||
|
// Allocated => return BDRV_BLOCK_DATA|BDRV_BLOCK_OFFSET_VALID
|
||||||
|
// Not allocated => return 0
|
||||||
|
// Error => return -errno
|
||||||
|
// Set pnum to length of the extent, `*map` = `offset`, `*file` = `bs`
|
||||||
|
+#if QEMU_VERSION_MAJOR > 10 || QEMU_VERSION_MAJOR == 10 && QEMU_VERSION_MINOR >= 1
|
||||||
|
+ int want_zero = (mode == BDRV_WANT_PRECISE);
|
||||||
|
+#endif
|
||||||
|
VitastorRPC task;
|
||||||
|
VitastorClient *client = bs->opaque;
|
||||||
|
uint64_t inode = client->watch ? vitastor_c_inode_get_num(client->watch) : client->inode;
|
||||||
@@ -21,7 +21,7 @@ rpmbuild -bp fio.spec
|
|||||||
cd $VITASTOR
|
cd $VITASTOR
|
||||||
VER=$(grep ^Version: rpm/vitastor-$REL.spec | awk '{print $2}')
|
VER=$(grep ^Version: rpm/vitastor-$REL.spec | awk '{print $2}')
|
||||||
rm -rf fio
|
rm -rf fio
|
||||||
ln -s ~/rpmbuild/BUILD/fio*/ fio
|
ln -s $(ls -d ~/rpmbuild/BUILD/fio*/ | grep -v SPECPARTS) fio
|
||||||
sh copy-fio-includes.sh
|
sh copy-fio-includes.sh
|
||||||
rm fio
|
rm fio
|
||||||
mv fio-copy fio
|
mv fio-copy fio
|
||||||
|
|||||||
@@ -0,0 +1,17 @@
|
|||||||
|
# Build packages for AlmaLinux 10 inside a container
|
||||||
|
# cd ..
|
||||||
|
# docker pull --platform=linux/amd64/v2 quay.io/almalinuxorg/almalinux:10
|
||||||
|
# docker build -t vitastor-buildenv:el10 -f rpm/vitastor-el10.Dockerfile .
|
||||||
|
# docker run -i --rm -v ./:/root/vitastor vitastor-buildenv:el10 /root/vitastor/rpm/vitastor-build.sh
|
||||||
|
|
||||||
|
FROM quay.io/almalinuxorg/almalinux:10
|
||||||
|
|
||||||
|
WORKDIR /root
|
||||||
|
|
||||||
|
RUN sed -i 's/enabled=0/enabled=1/' /etc/yum.repos.d/*.repo
|
||||||
|
RUN dnf -y install epel-release dnf-plugins-core
|
||||||
|
RUN dnf -y install https://vitastor.io/rpms/centos/10/vitastor-release-1.0-1.el10.noarch.rpm
|
||||||
|
RUN dnf -y install gcc-c++ gperftools-devel fio nodejs rpm-build jerasure-devel isa-l-devel gf-complete-devel rdma-core-devel cmake libnl3-devel
|
||||||
|
RUN dnf download --source fio
|
||||||
|
RUN rpm --nomd5 -i fio*.src.rpm
|
||||||
|
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --spec fio.spec
|
||||||
@@ -0,0 +1,198 @@
|
|||||||
|
Name: vitastor
|
||||||
|
Version: 3.0.6
|
||||||
|
Release: 1%{?dist}
|
||||||
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
|
License: Vitastor Network Public License 1.1
|
||||||
|
URL: https://vitastor.io/
|
||||||
|
Source0: vitastor-3.0.6.el10.tar.gz
|
||||||
|
|
||||||
|
BuildRequires: gperftools-devel
|
||||||
|
BuildRequires: gcc-c++
|
||||||
|
BuildRequires: nodejs >= 10
|
||||||
|
BuildRequires: jerasure-devel
|
||||||
|
BuildRequires: isa-l-devel
|
||||||
|
BuildRequires: gf-complete-devel
|
||||||
|
BuildRequires: rdma-core-devel
|
||||||
|
BuildRequires: cmake
|
||||||
|
BuildRequires: libnl3-devel
|
||||||
|
Requires: vitastor-osd = %{version}-%{release}
|
||||||
|
Requires: vitastor-mon = %{version}-%{release}
|
||||||
|
Requires: vitastor-client = %{version}-%{release}
|
||||||
|
Requires: vitastor-client-devel = %{version}-%{release}
|
||||||
|
Requires: vitastor-fio = %{version}-%{release}
|
||||||
|
|
||||||
|
%description
|
||||||
|
Vitastor is a small, simple and fast clustered block storage (storage for VM drives),
|
||||||
|
architecturally similar to Ceph which means strong consistency, primary-replication,
|
||||||
|
symmetric clustering and automatic data distribution over any number of drives of any
|
||||||
|
size with configurable redundancy (replication or erasure codes/XOR).
|
||||||
|
|
||||||
|
|
||||||
|
%package -n vitastor-osd
|
||||||
|
Summary: Vitastor - OSD
|
||||||
|
Requires: vitastor-client = %{version}-%{release}
|
||||||
|
Requires: util-linux
|
||||||
|
Requires: parted
|
||||||
|
|
||||||
|
|
||||||
|
%description -n vitastor-osd
|
||||||
|
Vitastor object storage daemon, i.e. server program that stores data.
|
||||||
|
|
||||||
|
|
||||||
|
%package -n vitastor-mon
|
||||||
|
Summary: Vitastor - monitor
|
||||||
|
Requires: nodejs >= 10
|
||||||
|
Requires: lpsolve
|
||||||
|
|
||||||
|
|
||||||
|
%description -n vitastor-mon
|
||||||
|
Vitastor monitor, i.e. server program responsible for watching cluster state and
|
||||||
|
scheduling cluster-level operations.
|
||||||
|
|
||||||
|
|
||||||
|
%package -n vitastor-client
|
||||||
|
Summary: Vitastor - client
|
||||||
|
|
||||||
|
|
||||||
|
%description -n vitastor-client
|
||||||
|
Vitastor client library and command-line interface.
|
||||||
|
|
||||||
|
|
||||||
|
%package -n vitastor-client-devel
|
||||||
|
Summary: Vitastor - development files
|
||||||
|
Group: Development/Libraries
|
||||||
|
Requires: vitastor-client = %{version}-%{release}
|
||||||
|
|
||||||
|
|
||||||
|
%description -n vitastor-client-devel
|
||||||
|
Vitastor library headers for development.
|
||||||
|
|
||||||
|
|
||||||
|
%package -n vitastor-fio
|
||||||
|
Summary: Vitastor - fio drivers
|
||||||
|
Group: Development/Libraries
|
||||||
|
Requires: vitastor-client = %{version}-%{release}
|
||||||
|
Requires: fio = 3.36-5.el10
|
||||||
|
|
||||||
|
|
||||||
|
%description -n vitastor-fio
|
||||||
|
Vitastor fio drivers for benchmarking.
|
||||||
|
|
||||||
|
|
||||||
|
%package -n vitastor-opennebula
|
||||||
|
Summary: Vitastor for OpenNebula
|
||||||
|
Group: Development/Libraries
|
||||||
|
Requires: vitastor-client
|
||||||
|
Requires: jq
|
||||||
|
Requires: python3-lxml
|
||||||
|
Requires: patch
|
||||||
|
Requires: qemu-kvm-block-vitastor
|
||||||
|
|
||||||
|
|
||||||
|
%description -n vitastor-opennebula
|
||||||
|
Vitastor storage plugin for OpenNebula.
|
||||||
|
|
||||||
|
|
||||||
|
%prep
|
||||||
|
%setup -q
|
||||||
|
|
||||||
|
|
||||||
|
%build
|
||||||
|
%cmake
|
||||||
|
%cmake_build
|
||||||
|
|
||||||
|
|
||||||
|
%install
|
||||||
|
rm -rf $RPM_BUILD_ROOT
|
||||||
|
%cmake_install
|
||||||
|
cd mon
|
||||||
|
npm install --production
|
||||||
|
cd ..
|
||||||
|
mkdir -p %buildroot/usr/lib/vitastor
|
||||||
|
cp -r mon %buildroot/usr/lib/vitastor
|
||||||
|
mv %buildroot/usr/lib/vitastor/mon/scripts/make-etcd %buildroot/usr/lib/vitastor/mon/
|
||||||
|
mkdir -p %buildroot/lib/systemd/system
|
||||||
|
cp mon/scripts/vitastor.target mon/scripts/vitastor-mon.service mon/scripts/vitastor-osd@.service %buildroot/lib/systemd/system
|
||||||
|
mkdir -p %buildroot/lib/udev/rules.d
|
||||||
|
cp mon/scripts/90-vitastor.rules %buildroot/lib/udev/rules.d
|
||||||
|
mkdir -p %buildroot/var/lib/one
|
||||||
|
cp -r opennebula/remotes %buildroot/var/lib/one
|
||||||
|
cp opennebula/install.sh %buildroot/var/lib/one/remotes/datastore/vitastor/
|
||||||
|
mkdir -p %buildroot/etc/
|
||||||
|
cp -r opennebula/sudoers.d %buildroot/etc/
|
||||||
|
|
||||||
|
|
||||||
|
%files
|
||||||
|
%doc GPL-2.0.txt VNPL-1.1.txt README.md README-ru.md
|
||||||
|
|
||||||
|
|
||||||
|
%files -n vitastor-osd
|
||||||
|
%_bindir/vitastor-osd
|
||||||
|
%_bindir/vitastor-disk
|
||||||
|
%_bindir/vitastor-dump-journal
|
||||||
|
/lib/systemd/system/vitastor-osd@.service
|
||||||
|
/lib/systemd/system/vitastor.target
|
||||||
|
/lib/udev/rules.d/90-vitastor.rules
|
||||||
|
|
||||||
|
|
||||||
|
%pre -n vitastor-osd
|
||||||
|
groupadd -r -f vitastor 2>/dev/null ||:
|
||||||
|
useradd -r -g vitastor -s /sbin/nologin -c "Vitastor daemons" -M -d /nonexistent vitastor 2>/dev/null ||:
|
||||||
|
install -o vitastor -g vitastor -d /var/log/vitastor
|
||||||
|
mkdir -p /etc/vitastor
|
||||||
|
|
||||||
|
|
||||||
|
%files -n vitastor-mon
|
||||||
|
/usr/lib/vitastor/mon
|
||||||
|
/lib/systemd/system/vitastor-mon.service
|
||||||
|
|
||||||
|
|
||||||
|
%pre -n vitastor-mon
|
||||||
|
groupadd -r -f vitastor 2>/dev/null ||:
|
||||||
|
useradd -r -g vitastor -s /sbin/nologin -c "Vitastor daemons" -M -d /nonexistent vitastor 2>/dev/null ||:
|
||||||
|
mkdir -p /etc/vitastor
|
||||||
|
mkdir -p /var/lib/vitastor
|
||||||
|
chown vitastor:vitastor /var/lib/vitastor
|
||||||
|
|
||||||
|
|
||||||
|
%files -n vitastor-client
|
||||||
|
%_bindir/vitastor-nbd
|
||||||
|
%_bindir/vitastor-ublk
|
||||||
|
%_bindir/vitastor-nfs
|
||||||
|
%_bindir/vitastor-cli
|
||||||
|
%_bindir/vitastor-rm
|
||||||
|
%_bindir/vitastor-kv
|
||||||
|
%_bindir/vitastor-kv-stress
|
||||||
|
%_bindir/vita
|
||||||
|
%_libdir/libvitastor_client.so*
|
||||||
|
%_libdir/libvitastor_kv.so*
|
||||||
|
|
||||||
|
|
||||||
|
%files -n vitastor-client-devel
|
||||||
|
%_includedir/vitastor_c.h
|
||||||
|
%_includedir/vitastor_kv.h
|
||||||
|
%_libdir/pkgconfig
|
||||||
|
|
||||||
|
|
||||||
|
%files -n vitastor-fio
|
||||||
|
%_libdir/libfio_vitastor.so
|
||||||
|
%_libdir/libfio_vitastor_blk.so
|
||||||
|
%_libdir/libfio_vitastor_sec.so
|
||||||
|
|
||||||
|
|
||||||
|
%files -n vitastor-opennebula
|
||||||
|
/var/lib/one
|
||||||
|
/etc/sudoers.d/opennebula-vitastor
|
||||||
|
|
||||||
|
|
||||||
|
%triggerin -n vitastor-opennebula -- opennebula
|
||||||
|
[ $2 = 0 ] || exit 0
|
||||||
|
/var/lib/one/remotes/datastore/vitastor/install.sh
|
||||||
|
|
||||||
|
|
||||||
|
# Turn off the brp-python-bytecompile script
|
||||||
|
%global __os_install_post %(echo '%{__os_install_post}' | sed -e 's!/usr/lib[^[:space:]]*/brp-python-bytecompile[[:space:]].*$!!g')
|
||||||
|
|
||||||
|
|
||||||
|
%changelog
|
||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 2.3.0
|
Version: 3.0.6
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-2.3.0.el7.tar.gz
|
Source0: vitastor-3.0.6.el7.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: devtoolset-9-gcc-c++
|
BuildRequires: devtoolset-9-gcc-c++
|
||||||
@@ -171,7 +171,6 @@ chown vitastor:vitastor /var/lib/vitastor
|
|||||||
%_bindir/vitastor-kv
|
%_bindir/vitastor-kv
|
||||||
%_bindir/vitastor-kv-stress
|
%_bindir/vitastor-kv-stress
|
||||||
%_bindir/vita
|
%_bindir/vita
|
||||||
%_libdir/libvitastor_blk.so*
|
|
||||||
%_libdir/libvitastor_client.so*
|
%_libdir/libvitastor_client.so*
|
||||||
%_libdir/libvitastor_kv.so*
|
%_libdir/libvitastor_kv.so*
|
||||||
|
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 2.3.0
|
Version: 3.0.6
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-2.3.0.el8.tar.gz
|
Source0: vitastor-3.0.6.el8.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: gcc-toolset-9-gcc-c++
|
BuildRequires: gcc-toolset-9-gcc-c++
|
||||||
@@ -168,7 +168,6 @@ chown vitastor:vitastor /var/lib/vitastor
|
|||||||
%_bindir/vitastor-kv
|
%_bindir/vitastor-kv
|
||||||
%_bindir/vitastor-kv-stress
|
%_bindir/vitastor-kv-stress
|
||||||
%_bindir/vita
|
%_bindir/vita
|
||||||
%_libdir/libvitastor_blk.so*
|
|
||||||
%_libdir/libvitastor_client.so*
|
%_libdir/libvitastor_client.so*
|
||||||
%_libdir/libvitastor_kv.so*
|
%_libdir/libvitastor_kv.so*
|
||||||
|
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 2.3.0
|
Version: 3.0.6
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-2.3.0.el9.tar.gz
|
Source0: vitastor-3.0.6.el9.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: gcc-c++
|
BuildRequires: gcc-c++
|
||||||
@@ -165,7 +165,6 @@ chown vitastor:vitastor /var/lib/vitastor
|
|||||||
%_bindir/vitastor-kv
|
%_bindir/vitastor-kv
|
||||||
%_bindir/vitastor-kv-stress
|
%_bindir/vitastor-kv-stress
|
||||||
%_bindir/vita
|
%_bindir/vita
|
||||||
%_libdir/libvitastor_blk.so*
|
|
||||||
%_libdir/libvitastor_client.so*
|
%_libdir/libvitastor_client.so*
|
||||||
%_libdir/libvitastor_kv.so*
|
%_libdir/libvitastor_kv.so*
|
||||||
|
|
||||||
|
|||||||
+8
-10
@@ -19,8 +19,9 @@ if("${CMAKE_INSTALL_PREFIX}" MATCHES "^/usr/local/?$")
|
|||||||
endif()
|
endif()
|
||||||
set(CMAKE_INSTALL_RPATH "${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}")
|
set(CMAKE_INSTALL_RPATH "${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}")
|
||||||
endif()
|
endif()
|
||||||
|
set(ENABLE_COVERAGE false CACHE BOOL "Enable code coverage")
|
||||||
|
|
||||||
add_definitions(-DVITASTOR_VERSION="2.3.0")
|
add_definitions(-DVITASTOR_VERSION="3.0.6")
|
||||||
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
|
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
|
||||||
add_link_options(-fno-omit-frame-pointer)
|
add_link_options(-fno-omit-frame-pointer)
|
||||||
if (${WITH_ASAN})
|
if (${WITH_ASAN})
|
||||||
@@ -31,6 +32,11 @@ set(CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fvisibility-inlines-hid
|
|||||||
set(CMAKE_CXX_FLAGS_MINSIZEREL "${CMAKE_CXX_FLAGS_MINSIZEREL} -fvisibility-inlines-hidden")
|
set(CMAKE_CXX_FLAGS_MINSIZEREL "${CMAKE_CXX_FLAGS_MINSIZEREL} -fvisibility-inlines-hidden")
|
||||||
set(CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fvisibility-inlines-hidden")
|
set(CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fvisibility-inlines-hidden")
|
||||||
|
|
||||||
|
if (${ENABLE_COVERAGE})
|
||||||
|
add_definitions(-coverage)
|
||||||
|
add_link_options(-coverage)
|
||||||
|
endif()
|
||||||
|
|
||||||
set(CMAKE_BUILD_TYPE RelWithDebInfo)
|
set(CMAKE_BUILD_TYPE RelWithDebInfo)
|
||||||
string(REGEX REPLACE "([\\/\\-]O)[^ \t\r\n]*" "\\13" CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE}")
|
string(REGEX REPLACE "([\\/\\-]O)[^ \t\r\n]*" "\\13" CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE}")
|
||||||
string(REGEX REPLACE "([\\/\\-]O)[^ \t\r\n]*" "\\13" CMAKE_CXX_FLAGS_MINSIZEREL "${CMAKE_CXX_FLAGS_MINSIZEREL}")
|
string(REGEX REPLACE "([\\/\\-]O)[^ \t\r\n]*" "\\13" CMAKE_CXX_FLAGS_MINSIZEREL "${CMAKE_CXX_FLAGS_MINSIZEREL}")
|
||||||
@@ -78,14 +84,6 @@ else()
|
|||||||
set(LIBURING_LIBRARIES uring)
|
set(LIBURING_LIBRARIES uring)
|
||||||
endif (${WITH_SYSTEM_LIBURING})
|
endif (${WITH_SYSTEM_LIBURING})
|
||||||
|
|
||||||
add_custom_target(build_tests)
|
|
||||||
add_custom_target(test
|
|
||||||
COMMAND
|
|
||||||
echo leak:tcmalloc > ${CMAKE_CURRENT_BINARY_DIR}/lsan-suppress.txt &&
|
|
||||||
env LSAN_OPTIONS=suppressions=${CMAKE_CURRENT_BINARY_DIR}/lsan-suppress.txt ${CMAKE_CTEST_COMMAND}
|
|
||||||
)
|
|
||||||
add_dependencies(test build_tests)
|
|
||||||
|
|
||||||
include_directories(
|
include_directories(
|
||||||
../
|
../
|
||||||
${CMAKE_SOURCE_DIR}/src/blockstore
|
${CMAKE_SOURCE_DIR}/src/blockstore
|
||||||
@@ -117,7 +115,7 @@ install_symlink(vitastor-disk ${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_BINDIR}/vi
|
|||||||
install_symlink(vitastor-cli ${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_BINDIR}/vitastor-rm)
|
install_symlink(vitastor-cli ${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_BINDIR}/vitastor-rm)
|
||||||
install_symlink(vitastor-cli ${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_BINDIR}/vita)
|
install_symlink(vitastor-cli ${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_BINDIR}/vita)
|
||||||
install(
|
install(
|
||||||
TARGETS vitastor_blk vitastor_client vitastor_kv
|
TARGETS vitastor_client vitastor_kv
|
||||||
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||||
PUBLIC_HEADER DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}
|
PUBLIC_HEADER DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -2,15 +2,18 @@ cmake_minimum_required(VERSION 2.8.12)
|
|||||||
|
|
||||||
project(vitastor)
|
project(vitastor)
|
||||||
|
|
||||||
# libvitastor_blk.so
|
# libvitastor_blk.a
|
||||||
add_library(vitastor_blk SHARED
|
add_library(vitastor_blk STATIC
|
||||||
../util/allocator.cpp blockstore.cpp blockstore_impl.cpp blockstore_disk.cpp blockstore_init.cpp blockstore_open.cpp blockstore_journal.cpp blockstore_read.cpp
|
../util/allocator.cpp ../util/crc32c.c ../util/ringloop.cpp
|
||||||
blockstore_write.cpp blockstore_sync.cpp blockstore_stable.cpp blockstore_rollback.cpp blockstore_flush.cpp ../util/crc32c.c ../util/ringloop.cpp
|
multilist.cpp blockstore_heap.cpp blockstore_disk.cpp
|
||||||
|
blockstore.cpp blockstore_impl.cpp blockstore_init.cpp blockstore_open.cpp
|
||||||
|
blockstore_flush.cpp blockstore_read.cpp blockstore_stable.cpp blockstore_sync.cpp blockstore_write.cpp
|
||||||
|
v1/flush.cpp v1/impl.cpp v1/init.cpp v1/journal.cpp v1/open.cpp v1/read.cpp v1/rollback.cpp v1/stable.cpp v1/sync.cpp v1/write.cpp
|
||||||
)
|
)
|
||||||
|
target_compile_options(vitastor_blk PUBLIC -fPIC)
|
||||||
target_link_libraries(vitastor_blk
|
target_link_libraries(vitastor_blk
|
||||||
${LIBURING_LIBRARIES}
|
${LIBURING_LIBRARIES}
|
||||||
${ISAL_LIBRARIES}
|
${ISAL_LIBRARIES}
|
||||||
tcmalloc_minimal
|
|
||||||
# for timerfd_manager
|
# for timerfd_manager
|
||||||
vitastor_common
|
vitastor_common
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -1,89 +1,16 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.1 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
|
#include "str_util.h"
|
||||||
|
|
||||||
#include "blockstore_impl.h"
|
#include "blockstore_impl.h"
|
||||||
|
#include "v1/impl.h"
|
||||||
|
|
||||||
blockstore_t::blockstore_t(blockstore_config_t & config, ring_loop_t *ringloop, timerfd_manager_t *tfd)
|
blockstore_i* blockstore_i::create(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd)
|
||||||
{
|
{
|
||||||
impl = new blockstore_impl_t(config, ringloop, tfd);
|
auto meta_format = stoull_full(config["meta_format"]);
|
||||||
}
|
if (meta_format == BLOCKSTORE_META_FORMAT_HEAP)
|
||||||
|
return new blockstore_impl_t(config, ringloop, tfd);
|
||||||
blockstore_t::~blockstore_t()
|
else
|
||||||
{
|
return new v1::blockstore_impl_t(config, ringloop, tfd);
|
||||||
delete impl;
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_t::parse_config(blockstore_config_t & config)
|
|
||||||
{
|
|
||||||
impl->parse_config(config, false);
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_t::loop()
|
|
||||||
{
|
|
||||||
impl->loop();
|
|
||||||
}
|
|
||||||
|
|
||||||
bool blockstore_t::is_started()
|
|
||||||
{
|
|
||||||
return impl->is_started();
|
|
||||||
}
|
|
||||||
|
|
||||||
bool blockstore_t::is_stalled()
|
|
||||||
{
|
|
||||||
return impl->is_stalled();
|
|
||||||
}
|
|
||||||
|
|
||||||
bool blockstore_t::is_safe_to_stop()
|
|
||||||
{
|
|
||||||
return impl->is_safe_to_stop();
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_t::enqueue_op(blockstore_op_t *op)
|
|
||||||
{
|
|
||||||
impl->enqueue_op(op);
|
|
||||||
}
|
|
||||||
|
|
||||||
int blockstore_t::read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version)
|
|
||||||
{
|
|
||||||
return impl->read_bitmap(oid, target_version, bitmap, result_version);
|
|
||||||
}
|
|
||||||
|
|
||||||
std::map<uint64_t, uint64_t> & blockstore_t::get_inode_space_stats()
|
|
||||||
{
|
|
||||||
return impl->inode_space_stats;
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_t::dump_diagnostics()
|
|
||||||
{
|
|
||||||
return impl->dump_diagnostics();
|
|
||||||
}
|
|
||||||
|
|
||||||
uint32_t blockstore_t::get_block_size()
|
|
||||||
{
|
|
||||||
return impl->get_block_size();
|
|
||||||
}
|
|
||||||
|
|
||||||
uint64_t blockstore_t::get_block_count()
|
|
||||||
{
|
|
||||||
return impl->get_block_count();
|
|
||||||
}
|
|
||||||
|
|
||||||
uint64_t blockstore_t::get_free_block_count()
|
|
||||||
{
|
|
||||||
return impl->get_free_block_count();
|
|
||||||
}
|
|
||||||
|
|
||||||
uint64_t blockstore_t::get_journal_size()
|
|
||||||
{
|
|
||||||
return impl->get_journal_size();
|
|
||||||
}
|
|
||||||
|
|
||||||
uint32_t blockstore_t::get_bitmap_granularity()
|
|
||||||
{
|
|
||||||
return impl->get_bitmap_granularity();
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_t::set_no_inode_stats(const std::vector<uint64_t> & pool_ids)
|
|
||||||
{
|
|
||||||
impl->set_no_inode_stats(pool_ids);
|
|
||||||
}
|
}
|
||||||
|
|||||||
+41
-33
@@ -17,22 +17,14 @@
|
|||||||
#include "ringloop.h"
|
#include "ringloop.h"
|
||||||
#include "timerfd_manager.h"
|
#include "timerfd_manager.h"
|
||||||
|
|
||||||
// Memory alignment for direct I/O (usually 512 bytes)
|
|
||||||
#ifndef DIRECT_IO_ALIGNMENT
|
|
||||||
#define DIRECT_IO_ALIGNMENT 512
|
|
||||||
#endif
|
|
||||||
|
|
||||||
// Memory allocation alignment (page size is usually optimal)
|
|
||||||
#ifndef MEM_ALIGNMENT
|
|
||||||
#define MEM_ALIGNMENT 4096
|
|
||||||
#endif
|
|
||||||
|
|
||||||
// Default block size is 128 KB, current allowed range is 4K - 128M
|
// Default block size is 128 KB, current allowed range is 4K - 128M
|
||||||
#define DEFAULT_DATA_BLOCK_ORDER 17
|
#define DEFAULT_DATA_BLOCK_ORDER 17
|
||||||
#define MIN_DATA_BLOCK_SIZE 4*1024
|
#define MIN_DATA_BLOCK_SIZE 4*1024
|
||||||
#define MAX_DATA_BLOCK_SIZE 128*1024*1024
|
#define MAX_DATA_BLOCK_SIZE 128*1024*1024
|
||||||
#define DEFAULT_BITMAP_GRANULARITY 4096
|
#define DEFAULT_BITMAP_GRANULARITY 4096
|
||||||
|
|
||||||
|
#define MIN_JOURNAL_SIZE 1024*1024
|
||||||
|
|
||||||
#define BS_OP_MIN 1
|
#define BS_OP_MIN 1
|
||||||
#define BS_OP_READ 1
|
#define BS_OP_READ 1
|
||||||
#define BS_OP_WRITE 2
|
#define BS_OP_WRITE 2
|
||||||
@@ -46,8 +38,18 @@
|
|||||||
|
|
||||||
#define BS_OP_PRIVATE_DATA_SIZE 256
|
#define BS_OP_PRIVATE_DATA_SIZE 256
|
||||||
|
|
||||||
|
#define IMMEDIATE_NONE 0
|
||||||
|
#define IMMEDIATE_SMALL 1
|
||||||
|
#define IMMEDIATE_ALL 2
|
||||||
|
|
||||||
/*
|
/*
|
||||||
|
|
||||||
|
All operations may be submitted in any order, because reads only see completed writes,
|
||||||
|
syncs only sync completed writes and writes don't depend on each other.
|
||||||
|
|
||||||
|
The only restriction is that the external code MUST NOT submit multiple writes for one
|
||||||
|
object in parallel. This is a natural restriction because `version` numbers are used though.
|
||||||
|
|
||||||
Blockstore opcode documentation:
|
Blockstore opcode documentation:
|
||||||
|
|
||||||
## BS_OP_READ / BS_OP_WRITE / BS_OP_WRITE_STABLE
|
## BS_OP_READ / BS_OP_WRITE / BS_OP_WRITE_STABLE
|
||||||
@@ -162,8 +164,8 @@ struct __attribute__ ((visibility("default"))) blockstore_op_t
|
|||||||
uint32_t list_stable_limit;
|
uint32_t list_stable_limit;
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
void *buf = NULL;
|
uint8_t *buf = NULL;
|
||||||
void *bitmap = NULL;
|
uint8_t *bitmap = NULL;
|
||||||
int retval = 0;
|
int retval = 0;
|
||||||
|
|
||||||
uint8_t private_data[BS_OP_PRIVATE_DATA_SIZE];
|
uint8_t private_data[BS_OP_PRIVATE_DATA_SIZE];
|
||||||
@@ -171,53 +173,59 @@ struct __attribute__ ((visibility("default"))) blockstore_op_t
|
|||||||
|
|
||||||
typedef std::map<std::string, std::string> blockstore_config_t;
|
typedef std::map<std::string, std::string> blockstore_config_t;
|
||||||
|
|
||||||
class blockstore_impl_t;
|
class __attribute__((visibility("default"))) blockstore_i
|
||||||
|
|
||||||
class __attribute__((visibility("default"))) blockstore_t
|
|
||||||
{
|
{
|
||||||
blockstore_impl_t *impl;
|
|
||||||
public:
|
public:
|
||||||
blockstore_t(blockstore_config_t & config, ring_loop_t *ringloop, timerfd_manager_t *tfd);
|
static blockstore_i* create(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd);
|
||||||
~blockstore_t();
|
|
||||||
|
virtual ~blockstore_i() = default;
|
||||||
|
|
||||||
// Update configuration
|
// Update configuration
|
||||||
void parse_config(blockstore_config_t & config);
|
virtual void parse_config(blockstore_config_t & config) = 0;
|
||||||
|
|
||||||
|
// Reshard database for a pool in chunks
|
||||||
|
// MUST be called only when nobody makes any modifications to the DB for this pool
|
||||||
|
virtual void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit) = 0;
|
||||||
|
virtual bool reshard_continue(void *reshard_state, uint64_t chunk_limit) = 0;
|
||||||
|
|
||||||
// Event loop
|
// Event loop
|
||||||
void loop();
|
virtual void loop() = 0;
|
||||||
|
|
||||||
// Returns true when blockstore is ready to process operations
|
// Returns true when blockstore is ready to process operations
|
||||||
// (Although you're free to enqueue them before that)
|
// (Although you're free to enqueue them before that)
|
||||||
bool is_started();
|
virtual bool is_started() = 0;
|
||||||
|
|
||||||
// Returns true when blockstore is stalled
|
// Returns true when blockstore is stalled
|
||||||
bool is_stalled();
|
virtual bool is_stalled() = 0;
|
||||||
|
|
||||||
// Returns true when it's safe to destroy the instance. If destroying the instance
|
// Returns true when it's safe to destroy the instance. If destroying the instance
|
||||||
// requires to purge some queues, starts that process. Should be called in the event
|
// requires to purge some queues, starts that process. Should be called in the event
|
||||||
// loop until it returns true.
|
// loop until it returns true.
|
||||||
bool is_safe_to_stop();
|
virtual bool is_safe_to_stop() = 0;
|
||||||
|
|
||||||
// Submission
|
// Submission
|
||||||
void enqueue_op(blockstore_op_t *op);
|
virtual void enqueue_op(blockstore_op_t *op) = 0;
|
||||||
|
|
||||||
// Simplified synchronous operation: get object bitmap & current version
|
// Simplified synchronous operation: get object bitmap & current version
|
||||||
int read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version = NULL);
|
virtual int read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version = NULL) = 0;
|
||||||
|
|
||||||
// Get per-inode space usage statistics
|
// Get per-inode space usage statistics
|
||||||
std::map<uint64_t, uint64_t> & get_inode_space_stats();
|
virtual const std::map<uint64_t, uint64_t> & get_inode_space_stats() = 0;
|
||||||
|
|
||||||
// Set per-pool no_inode_stats
|
// Set per-pool no_inode_stats
|
||||||
void set_no_inode_stats(const std::vector<uint64_t> & pool_ids);
|
virtual void set_no_inode_stats(const std::vector<uint64_t> & pool_ids) = 0;
|
||||||
|
|
||||||
// Print diagnostics to stdout
|
// Print diagnostics to stdout
|
||||||
void dump_diagnostics();
|
virtual void dump_diagnostics() = 0;
|
||||||
|
|
||||||
uint32_t get_block_size();
|
// Get diagnostic string for an operation
|
||||||
uint64_t get_block_count();
|
virtual std::string get_op_diag(blockstore_op_t *op) = 0;
|
||||||
uint64_t get_free_block_count();
|
|
||||||
|
|
||||||
uint64_t get_journal_size();
|
virtual uint32_t get_block_size() = 0;
|
||||||
|
virtual uint64_t get_block_count() = 0;
|
||||||
|
virtual uint64_t get_free_block_count() = 0;
|
||||||
|
|
||||||
uint32_t get_bitmap_granularity();
|
virtual uint64_t get_journal_size() = 0;
|
||||||
|
|
||||||
|
virtual uint32_t get_bitmap_granularity() = 0;
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -2,11 +2,15 @@
|
|||||||
// License: VNPL-1.1 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
#include <sys/file.h>
|
#include <sys/file.h>
|
||||||
|
#include <sys/ioctl.h>
|
||||||
|
#include <unistd.h>
|
||||||
|
|
||||||
#include <stdexcept>
|
#include <stdexcept>
|
||||||
|
|
||||||
#include "blockstore_impl.h"
|
#include "blockstore.h"
|
||||||
|
#include "ondisk_formats.h"
|
||||||
#include "blockstore_disk.h"
|
#include "blockstore_disk.h"
|
||||||
|
#include "blockstore_heap.h"
|
||||||
#include "str_util.h"
|
#include "str_util.h"
|
||||||
#include "allocator.h"
|
#include "allocator.h"
|
||||||
|
|
||||||
@@ -46,6 +50,10 @@ void blockstore_disk_t::parse_config(std::map<std::string, std::string> & config
|
|||||||
meta_block_size = parse_size(config["meta_block_size"]);
|
meta_block_size = parse_size(config["meta_block_size"]);
|
||||||
bitmap_granularity = parse_size(config["bitmap_granularity"]);
|
bitmap_granularity = parse_size(config["bitmap_granularity"]);
|
||||||
meta_format = stoull_full(config["meta_format"]);
|
meta_format = stoull_full(config["meta_format"]);
|
||||||
|
atomic_write_size = (config.find("atomic_write_size") != config.end()
|
||||||
|
? parse_size(config["atomic_write_size"]) : 4096);
|
||||||
|
use_atomic_flag = config.find("use_atomic_flag") != config.end() &&
|
||||||
|
(config["use_atomic_flag"] == "true" || config["use_atomic_flag"] == "1" || config["use_atomic_flag"] == "yes");
|
||||||
if (config.find("data_io") == config.end() &&
|
if (config.find("data_io") == config.end() &&
|
||||||
config.find("meta_io") == config.end() &&
|
config.find("meta_io") == config.end() &&
|
||||||
config.find("journal_io") == config.end())
|
config.find("journal_io") == config.end())
|
||||||
@@ -90,12 +98,28 @@ void blockstore_disk_t::parse_config(std::map<std::string, std::string> & config
|
|||||||
if (!min_discard_size)
|
if (!min_discard_size)
|
||||||
min_discard_size = 1024*1024;
|
min_discard_size = 1024*1024;
|
||||||
discard_granularity = parse_size(config["discard_granularity"]);
|
discard_granularity = parse_size(config["discard_granularity"]);
|
||||||
|
inmemory_meta = config["inmemory_metadata"] != "false" && config["inmemory_metadata"] != "0" &&
|
||||||
|
config["inmemory_metadata"] != "no";
|
||||||
|
inmemory_journal = config["inmemory_journal"] != "false" && config["inmemory_journal"] != "0" &&
|
||||||
|
config["inmemory_journal"] != "no";
|
||||||
|
disable_data_fsync = config["disable_data_fsync"] == "true" || config["disable_data_fsync"] == "1" || config["disable_data_fsync"] == "yes";
|
||||||
|
disable_meta_fsync = config["disable_meta_fsync"] == "true" || config["disable_meta_fsync"] == "1" || config["disable_meta_fsync"] == "yes";
|
||||||
|
disable_journal_fsync = config["disable_journal_fsync"] == "true" || config["disable_journal_fsync"] == "1" || config["disable_journal_fsync"] == "yes";
|
||||||
|
if (mock_mode)
|
||||||
|
{
|
||||||
|
data_device_size = parse_size(config["data_device_size"]);
|
||||||
|
data_device_sect = parse_size(config["data_device_sect"]);
|
||||||
|
meta_device_size = parse_size(config["meta_device_size"]);
|
||||||
|
meta_device_sect = parse_size(config["meta_device_sect"]);
|
||||||
|
journal_device_size = parse_size(config["journal_device_size"]);
|
||||||
|
journal_device_sect = parse_size(config["journal_device_sect"]);
|
||||||
|
}
|
||||||
// Validate
|
// Validate
|
||||||
if (!data_block_size)
|
if (!data_block_size)
|
||||||
{
|
{
|
||||||
data_block_size = (1 << DEFAULT_DATA_BLOCK_ORDER);
|
data_block_size = (1 << DEFAULT_DATA_BLOCK_ORDER);
|
||||||
}
|
}
|
||||||
if ((block_order = is_power_of_two(data_block_size)) >= 64 || data_block_size < MIN_DATA_BLOCK_SIZE || data_block_size >= MAX_DATA_BLOCK_SIZE)
|
if (is_power_of_two(data_block_size) >= 64 || data_block_size < MIN_DATA_BLOCK_SIZE || data_block_size >= MAX_DATA_BLOCK_SIZE)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("Bad block size");
|
throw std::runtime_error("Bad block size");
|
||||||
}
|
}
|
||||||
@@ -147,6 +171,10 @@ void blockstore_disk_t::parse_config(std::map<std::string, std::string> & config
|
|||||||
{
|
{
|
||||||
throw std::runtime_error("Data block size must be a multiple of sparse write tracking granularity");
|
throw std::runtime_error("Data block size must be a multiple of sparse write tracking granularity");
|
||||||
}
|
}
|
||||||
|
if (data_block_size / bitmap_granularity < 8)
|
||||||
|
{
|
||||||
|
throw std::runtime_error("Data block size must be at least bitmap_granularity*8");
|
||||||
|
}
|
||||||
if (!data_csum_type)
|
if (!data_csum_type)
|
||||||
{
|
{
|
||||||
csum_block_size = 0;
|
csum_block_size = 0;
|
||||||
@@ -179,17 +207,25 @@ void blockstore_disk_t::parse_config(std::map<std::string, std::string> & config
|
|||||||
{
|
{
|
||||||
throw std::runtime_error("journal_offset must be a multiple of journal_block_size = "+std::to_string(journal_block_size));
|
throw std::runtime_error("journal_offset must be a multiple of journal_block_size = "+std::to_string(journal_block_size));
|
||||||
}
|
}
|
||||||
|
if (meta_device == data_device)
|
||||||
|
{
|
||||||
|
disable_meta_fsync = disable_data_fsync;
|
||||||
|
}
|
||||||
|
if (journal_device == meta_device)
|
||||||
|
{
|
||||||
|
disable_journal_fsync = disable_meta_fsync;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void blockstore_disk_t::calc_lengths(bool skip_meta_check)
|
void blockstore_disk_t::calc_lengths(bool skip_meta_check)
|
||||||
{
|
{
|
||||||
// data
|
// data
|
||||||
data_len = data_device_size - data_offset;
|
data_len = data_device_size - data_offset;
|
||||||
if (data_fd == meta_fd && data_offset < meta_offset)
|
if (data_device == meta_device && data_offset < meta_offset)
|
||||||
{
|
{
|
||||||
data_len = meta_offset - data_offset;
|
data_len = meta_offset - data_offset;
|
||||||
}
|
}
|
||||||
if (data_fd == journal_fd && data_offset < journal_offset)
|
if (data_device == journal_device && data_offset < journal_offset)
|
||||||
{
|
{
|
||||||
data_len = data_len < journal_offset-data_offset
|
data_len = data_len < journal_offset-data_offset
|
||||||
? data_len : journal_offset-data_offset;
|
? data_len : journal_offset-data_offset;
|
||||||
@@ -204,63 +240,84 @@ void blockstore_disk_t::calc_lengths(bool skip_meta_check)
|
|||||||
data_len = cfg_data_size;
|
data_len = cfg_data_size;
|
||||||
}
|
}
|
||||||
// meta
|
// meta
|
||||||
uint64_t meta_area_size = (meta_fd == data_fd ? data_device_size : meta_device_size) - meta_offset;
|
meta_area_size = (meta_device == data_device ? data_device_size : meta_device_size) - meta_offset;
|
||||||
if (meta_fd == data_fd && meta_offset <= data_offset)
|
if (meta_device == data_device && meta_offset <= data_offset)
|
||||||
{
|
{
|
||||||
meta_area_size = data_offset - meta_offset;
|
meta_area_size = data_offset - meta_offset;
|
||||||
}
|
}
|
||||||
if (meta_fd == journal_fd && meta_offset <= journal_offset)
|
if (meta_device == journal_device && meta_offset <= journal_offset)
|
||||||
{
|
{
|
||||||
meta_area_size = meta_area_size < journal_offset-meta_offset
|
meta_area_size = meta_area_size < journal_offset-meta_offset
|
||||||
? meta_area_size : journal_offset-meta_offset;
|
? meta_area_size : journal_offset-meta_offset;
|
||||||
}
|
}
|
||||||
// journal
|
// journal
|
||||||
journal_len = (journal_fd == data_fd ? data_device_size : (journal_fd == meta_fd ? meta_device_size : journal_device_size)) - journal_offset;
|
journal_len = (journal_device == data_device ? data_device_size : (journal_device == meta_device ? meta_device_size : journal_device_size)) - journal_offset;
|
||||||
if (journal_fd == data_fd && journal_offset <= data_offset)
|
if (journal_device == data_device && journal_offset <= data_offset)
|
||||||
{
|
{
|
||||||
journal_len = data_offset - journal_offset;
|
journal_len = data_offset - journal_offset;
|
||||||
}
|
}
|
||||||
if (journal_fd == meta_fd && journal_offset <= meta_offset)
|
if (journal_device == meta_device && journal_offset <= meta_offset)
|
||||||
{
|
{
|
||||||
journal_len = journal_len < meta_offset-journal_offset
|
journal_len = journal_len < meta_offset-journal_offset
|
||||||
? journal_len : meta_offset-journal_offset;
|
? journal_len : meta_offset-journal_offset;
|
||||||
}
|
}
|
||||||
// required metadata size
|
// required metadata size
|
||||||
block_count = data_len / data_block_size;
|
block_count = data_len / data_block_size;
|
||||||
clean_entry_bitmap_size = data_block_size / bitmap_granularity / 8;
|
clean_entry_bitmap_size = (data_block_size / bitmap_granularity + 7) / 8;
|
||||||
clean_dyn_size = clean_entry_bitmap_size*2 + (csum_block_size
|
clean_dyn_size = clean_entry_bitmap_size*2 + (csum_block_size
|
||||||
? data_block_size/csum_block_size*(data_csum_type & 0xFF) : 0);
|
? data_block_size/csum_block_size*(data_csum_type & 0xFF) : 0);
|
||||||
clean_entry_size = sizeof(clean_disk_entry) + clean_dyn_size + 4 /*entry_csum*/;
|
recalc:
|
||||||
meta_len = (1 + (block_count - 1 + meta_block_size / clean_entry_size) / (meta_block_size / clean_entry_size)) * meta_block_size;
|
if (meta_format == BLOCKSTORE_META_FORMAT_HEAP)
|
||||||
bool new_doesnt_fit = (!meta_format && !skip_meta_check && meta_area_size < meta_len && !data_csum_type);
|
|
||||||
if (meta_format == BLOCKSTORE_META_FORMAT_V1 || new_doesnt_fit)
|
|
||||||
{
|
{
|
||||||
uint64_t clean_entry_v0_size = sizeof(clean_disk_entry) + 2*clean_entry_bitmap_size;
|
uint32_t entries_per_block = meta_block_size / (sizeof(heap_big_write_t) + clean_dyn_size);
|
||||||
uint64_t meta_v0_len = (1 + (block_count - 1 + meta_block_size / clean_entry_v0_size)
|
min_meta_len = (block_count+entries_per_block-1) / entries_per_block * meta_block_size;
|
||||||
/ (meta_block_size / clean_entry_v0_size)) * meta_block_size;
|
}
|
||||||
if (meta_format == BLOCKSTORE_META_FORMAT_V1 || meta_area_size >= meta_v0_len)
|
else if (meta_format == BLOCKSTORE_META_FORMAT_V1)
|
||||||
|
{
|
||||||
|
clean_entry_size = 24 /*sizeof(clean_disk_entry)*/ + 2*clean_entry_bitmap_size;
|
||||||
|
min_meta_len = (1 + (block_count - 1 + meta_block_size / clean_entry_size)
|
||||||
|
/ (meta_block_size / clean_entry_size)) * meta_block_size;
|
||||||
|
if (!skip_meta_check && meta_area_size < min_meta_len)
|
||||||
{
|
{
|
||||||
// Old metadata fits.
|
too_small:
|
||||||
if (new_doesnt_fit)
|
throw std::runtime_error("Metadata area is too small, need at least "+std::to_string(min_meta_len)+
|
||||||
|
" bytes, have only "+std::to_string(meta_area_size)+" bytes");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else if (meta_format == BLOCKSTORE_META_FORMAT_V2 || !meta_format)
|
||||||
|
{
|
||||||
|
meta_format = BLOCKSTORE_META_FORMAT_V2;
|
||||||
|
clean_entry_size = 24 /*sizeof(clean_disk_entry)*/ + clean_dyn_size + 4 /*entry_csum*/;
|
||||||
|
min_meta_len = (1 + (block_count - 1 + meta_block_size / clean_entry_size) / (meta_block_size / clean_entry_size)) * meta_block_size;
|
||||||
|
if (!skip_meta_check && meta_area_size < min_meta_len)
|
||||||
|
{
|
||||||
|
if (!data_csum_type)
|
||||||
{
|
{
|
||||||
printf("Warning: Using old metadata format without checksums because the new format"
|
printf("Warning: Using old metadata format without checksums because the new format"
|
||||||
" doesn't fit into provided area (%ju bytes required, %ju bytes available)\n", meta_len, meta_area_size);
|
" doesn't fit into provided area (%ju bytes required, %ju bytes available)\n", min_meta_len, meta_area_size);
|
||||||
|
meta_format = BLOCKSTORE_META_FORMAT_V1;
|
||||||
|
goto recalc;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
goto too_small;
|
||||||
}
|
}
|
||||||
clean_entry_size = clean_entry_v0_size;
|
|
||||||
meta_len = meta_v0_len;
|
|
||||||
meta_format = BLOCKSTORE_META_FORMAT_V1;
|
|
||||||
}
|
}
|
||||||
else
|
|
||||||
meta_format = BLOCKSTORE_META_FORMAT_V2;
|
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
meta_format = BLOCKSTORE_META_FORMAT_V2;
|
|
||||||
if (!skip_meta_check && meta_area_size < meta_len)
|
|
||||||
{
|
{
|
||||||
throw std::runtime_error("Metadata area is too small, need at least "+std::to_string(meta_len)+" bytes, have only "+std::to_string(meta_area_size)+" bytes");
|
throw std::runtime_error("meta_format = "+std::to_string(meta_format)+" is not supported");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void blockstore_disk_t::check_lengths()
|
||||||
|
{
|
||||||
|
if (meta_area_size < min_meta_len)
|
||||||
|
{
|
||||||
|
throw std::runtime_error("Metadata area is too small, need at least "+std::to_string(min_meta_len)+" bytes, have only "+std::to_string(meta_area_size)+" bytes");
|
||||||
}
|
}
|
||||||
// requested journal size
|
// requested journal size
|
||||||
if (!skip_meta_check && cfg_journal_size > journal_len)
|
if (cfg_journal_size > journal_len)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("Requested journal_size is too large");
|
throw std::runtime_error("Requested journal_size is too large");
|
||||||
}
|
}
|
||||||
@@ -321,12 +378,19 @@ static int bs_openmode(const std::string & mode)
|
|||||||
|
|
||||||
void blockstore_disk_t::open_data()
|
void blockstore_disk_t::open_data()
|
||||||
{
|
{
|
||||||
data_fd = open(data_device.c_str(), bs_openmode(data_io) | O_RDWR);
|
if (data_fd >= 0)
|
||||||
|
{
|
||||||
|
throw std::runtime_error("data device is already opened");
|
||||||
|
}
|
||||||
|
data_fd = mock_mode ? MOCK_DATA_FD : open(data_device.c_str(), bs_openmode(data_io) | O_RDWR);
|
||||||
if (data_fd == -1)
|
if (data_fd == -1)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("Failed to open data device "+data_device+": "+std::string(strerror(errno)));
|
throw std::runtime_error("Failed to open data device "+data_device+": "+std::string(strerror(errno)));
|
||||||
}
|
}
|
||||||
check_size(data_fd, &data_device_size, &data_device_sect, "data device");
|
if (!mock_mode)
|
||||||
|
{
|
||||||
|
check_size(data_fd, &data_device_size, &data_device_sect, "data device");
|
||||||
|
}
|
||||||
if (disk_alignment % data_device_sect)
|
if (disk_alignment % data_device_sect)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(
|
throw std::runtime_error(
|
||||||
@@ -338,7 +402,7 @@ void blockstore_disk_t::open_data()
|
|||||||
{
|
{
|
||||||
throw std::runtime_error("data_offset exceeds device size = "+std::to_string(data_device_size));
|
throw std::runtime_error("data_offset exceeds device size = "+std::to_string(data_device_size));
|
||||||
}
|
}
|
||||||
if (!disable_flock && flock(data_fd, LOCK_EX|LOCK_NB) != 0)
|
if (!mock_mode && !disable_flock && flock(data_fd, LOCK_EX|LOCK_NB) != 0)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(std::string("Failed to lock data device: ") + strerror(errno));
|
throw std::runtime_error(std::string("Failed to lock data device: ") + strerror(errno));
|
||||||
}
|
}
|
||||||
@@ -346,19 +410,26 @@ void blockstore_disk_t::open_data()
|
|||||||
|
|
||||||
void blockstore_disk_t::open_meta()
|
void blockstore_disk_t::open_meta()
|
||||||
{
|
{
|
||||||
|
if (meta_fd >= 0)
|
||||||
|
{
|
||||||
|
throw std::runtime_error("metadata device is already opened");
|
||||||
|
}
|
||||||
if (meta_device != data_device || meta_io != data_io)
|
if (meta_device != data_device || meta_io != data_io)
|
||||||
{
|
{
|
||||||
meta_fd = open(meta_device.c_str(), bs_openmode(meta_io) | O_RDWR);
|
meta_fd = mock_mode ? MOCK_META_FD : open(meta_device.c_str(), bs_openmode(meta_io) | O_RDWR);
|
||||||
if (meta_fd == -1)
|
if (meta_fd == -1)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("Failed to open metadata device "+meta_device+": "+std::string(strerror(errno)));
|
throw std::runtime_error("Failed to open metadata device "+meta_device+": "+std::string(strerror(errno)));
|
||||||
}
|
}
|
||||||
check_size(meta_fd, &meta_device_size, &meta_device_sect, "metadata device");
|
if (!mock_mode)
|
||||||
|
{
|
||||||
|
check_size(meta_fd, &meta_device_size, &meta_device_sect, "metadata device");
|
||||||
|
}
|
||||||
if (meta_offset >= meta_device_size)
|
if (meta_offset >= meta_device_size)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("meta_offset exceeds device size = "+std::to_string(meta_device_size));
|
throw std::runtime_error("meta_offset exceeds device size = "+std::to_string(meta_device_size));
|
||||||
}
|
}
|
||||||
if (!disable_flock && meta_device != data_device && flock(meta_fd, LOCK_EX|LOCK_NB) != 0)
|
if (!mock_mode && !disable_flock && meta_device != data_device && flock(meta_fd, LOCK_EX|LOCK_NB) != 0)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(std::string("Failed to lock metadata device: ") + strerror(errno));
|
throw std::runtime_error(std::string("Failed to lock metadata device: ") + strerror(errno));
|
||||||
}
|
}
|
||||||
@@ -384,15 +455,26 @@ void blockstore_disk_t::open_meta()
|
|||||||
|
|
||||||
void blockstore_disk_t::open_journal()
|
void blockstore_disk_t::open_journal()
|
||||||
{
|
{
|
||||||
|
if (journal_fd >= 0)
|
||||||
|
{
|
||||||
|
throw std::runtime_error("journal device is already opened");
|
||||||
|
}
|
||||||
if (journal_device != meta_device || journal_io != meta_io)
|
if (journal_device != meta_device || journal_io != meta_io)
|
||||||
{
|
{
|
||||||
journal_fd = open(journal_device.c_str(), bs_openmode(journal_io) | O_RDWR);
|
journal_fd = mock_mode ? MOCK_JOURNAL_FD : open(journal_device.c_str(), bs_openmode(journal_io) | O_RDWR);
|
||||||
if (journal_fd == -1)
|
if (journal_fd == -1)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("Failed to open journal device "+journal_device+": "+std::string(strerror(errno)));
|
throw std::runtime_error("Failed to open journal device "+journal_device+": "+std::string(strerror(errno)));
|
||||||
}
|
}
|
||||||
check_size(journal_fd, &journal_device_size, &journal_device_sect, "journal device");
|
if (!mock_mode)
|
||||||
if (!disable_flock && journal_device != meta_device && flock(journal_fd, LOCK_EX|LOCK_NB) != 0)
|
{
|
||||||
|
check_size(journal_fd, &journal_device_size, &journal_device_sect, "journal device");
|
||||||
|
}
|
||||||
|
if (journal_offset >= journal_device_size)
|
||||||
|
{
|
||||||
|
throw std::runtime_error("journal_offset exceeds device size = "+std::to_string(journal_device_size));
|
||||||
|
}
|
||||||
|
if (!mock_mode && !disable_flock && journal_device != meta_device && flock(journal_fd, LOCK_EX|LOCK_NB) != 0)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(std::string("Failed to lock journal device: ") + strerror(errno));
|
throw std::runtime_error(std::string("Failed to lock journal device: ") + strerror(errno));
|
||||||
}
|
}
|
||||||
@@ -418,25 +500,32 @@ void blockstore_disk_t::open_journal()
|
|||||||
|
|
||||||
void blockstore_disk_t::close_all()
|
void blockstore_disk_t::close_all()
|
||||||
{
|
{
|
||||||
if (data_fd >= 0)
|
if (!mock_mode)
|
||||||
close(data_fd);
|
{
|
||||||
if (meta_fd >= 0 && meta_fd != data_fd)
|
if (data_fd >= 0)
|
||||||
close(meta_fd);
|
close(data_fd);
|
||||||
if (journal_fd >= 0 && journal_fd != meta_fd)
|
if (meta_fd >= 0 && meta_fd != data_fd)
|
||||||
close(journal_fd);
|
close(meta_fd);
|
||||||
|
if (journal_fd >= 0 && journal_fd != meta_fd)
|
||||||
|
close(journal_fd);
|
||||||
|
}
|
||||||
data_fd = meta_fd = journal_fd = -1;
|
data_fd = meta_fd = journal_fd = -1;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Sadly DISCARD only works through ioctl(), but it seems to always block the device queue,
|
// Sadly DISCARD only works through ioctl(), but it seems to always block the device queue,
|
||||||
// so it's not a big deal that we can only run it synchronously.
|
// so it's not a big deal that we can only run it synchronously.
|
||||||
int blockstore_disk_t::trim_data(allocator_t *alloc)
|
int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_free)
|
||||||
{
|
{
|
||||||
|
if (mock_mode)
|
||||||
|
{
|
||||||
|
return -EINVAL;
|
||||||
|
}
|
||||||
int r = 0;
|
int r = 0;
|
||||||
uint64_t j = 0, i = 0;
|
uint64_t j = 0, i = 0;
|
||||||
uint64_t discarded = 0;
|
uint64_t discarded = 0;
|
||||||
for (; i <= block_count; i++)
|
for (; i <= block_count; i++)
|
||||||
{
|
{
|
||||||
if (i >= block_count || alloc->get(i))
|
if (i >= block_count || is_free(i))
|
||||||
{
|
{
|
||||||
if (i > j && (i-j)*data_block_size >= min_discard_size)
|
if (i > j && (i-j)*data_block_size >= min_discard_size)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -8,16 +8,25 @@
|
|||||||
#include <string>
|
#include <string>
|
||||||
#include <map>
|
#include <map>
|
||||||
|
|
||||||
|
// Memory alignment for direct I/O (usually 512 bytes)
|
||||||
|
#ifndef DIRECT_IO_ALIGNMENT
|
||||||
|
#define DIRECT_IO_ALIGNMENT 512
|
||||||
|
#endif
|
||||||
|
|
||||||
#define BLOCKSTORE_CSUM_NONE 0
|
#define BLOCKSTORE_CSUM_NONE 0
|
||||||
// Lower byte of checksum type is its length
|
// Lower byte of checksum type is its length
|
||||||
#define BLOCKSTORE_CSUM_CRC32C 0x104
|
#define BLOCKSTORE_CSUM_CRC32C 0x104
|
||||||
|
|
||||||
|
#define MOCK_DATA_FD 1000
|
||||||
|
#define MOCK_META_FD 1001
|
||||||
|
#define MOCK_JOURNAL_FD 1002
|
||||||
|
|
||||||
class allocator_t;
|
class allocator_t;
|
||||||
|
|
||||||
struct blockstore_disk_t
|
struct blockstore_disk_t
|
||||||
{
|
{
|
||||||
std::string data_device, meta_device, journal_device;
|
std::string data_device, meta_device, journal_device;
|
||||||
uint32_t data_block_size;
|
uint64_t data_block_size;
|
||||||
uint64_t cfg_journal_size, cfg_data_size;
|
uint64_t cfg_journal_size, cfg_data_size;
|
||||||
// Required write alignment and journal/metadata/data areas' location alignment
|
// Required write alignment and journal/metadata/data areas' location alignment
|
||||||
uint32_t disk_alignment = 4096;
|
uint32_t disk_alignment = 4096;
|
||||||
@@ -25,8 +34,12 @@ struct blockstore_disk_t
|
|||||||
uint64_t journal_block_size = 4096;
|
uint64_t journal_block_size = 4096;
|
||||||
// Metadata block size - minimum_io_size of the metadata device is the best choice
|
// Metadata block size - minimum_io_size of the metadata device is the best choice
|
||||||
uint64_t meta_block_size = 4096;
|
uint64_t meta_block_size = 4096;
|
||||||
|
// Atomic write size of the data block device
|
||||||
|
uint32_t atomic_write_size = 4096;
|
||||||
|
// Whether we should set RWF_ATOMIC on atomic writes
|
||||||
|
bool use_atomic_flag = false;
|
||||||
// Sparse write tracking granularity. 4 KB is a good choice. Must be a multiple of disk_alignment
|
// Sparse write tracking granularity. 4 KB is a good choice. Must be a multiple of disk_alignment
|
||||||
uint64_t bitmap_granularity = 4096;
|
uint32_t bitmap_granularity = 4096;
|
||||||
// Data checksum type, BLOCKSTORE_CSUM_NONE or BLOCKSTORE_CSUM_CRC32C
|
// Data checksum type, BLOCKSTORE_CSUM_NONE or BLOCKSTORE_CSUM_CRC32C
|
||||||
uint32_t data_csum_type = BLOCKSTORE_CSUM_NONE;
|
uint32_t data_csum_type = BLOCKSTORE_CSUM_NONE;
|
||||||
// Checksum block size, must be a multiple of bitmap_granularity
|
// Checksum block size, must be a multiple of bitmap_granularity
|
||||||
@@ -36,27 +49,37 @@ struct blockstore_disk_t
|
|||||||
// I/O modes for data, metadata and journal: direct or "" = O_DIRECT, cached = O_SYNC, directsync = O_DIRECT|O_SYNC
|
// I/O modes for data, metadata and journal: direct or "" = O_DIRECT, cached = O_SYNC, directsync = O_DIRECT|O_SYNC
|
||||||
// O_SYNC without O_DIRECT = use Linux page cache for reads and writes
|
// O_SYNC without O_DIRECT = use Linux page cache for reads and writes
|
||||||
std::string data_io, meta_io, journal_io;
|
std::string data_io, meta_io, journal_io;
|
||||||
|
// It is safe to disable fsync() if drive write cache is writethrough
|
||||||
|
bool disable_data_fsync = false, disable_meta_fsync = false, disable_journal_fsync = false;
|
||||||
|
// Keep journal (buffered data) in memory?
|
||||||
|
bool inmemory_meta = true;
|
||||||
|
// Keep metadata in memory?
|
||||||
|
bool inmemory_journal = true;
|
||||||
// Data discard granularity and minimum size (for the sake of performance)
|
// Data discard granularity and minimum size (for the sake of performance)
|
||||||
bool discard_on_start = false;
|
bool discard_on_start = false;
|
||||||
uint64_t min_discard_size = 1024*1024;
|
uint64_t min_discard_size = 1024*1024;
|
||||||
uint64_t discard_granularity = 0;
|
uint64_t discard_granularity = 0;
|
||||||
|
|
||||||
int meta_fd = -1, data_fd = -1, journal_fd = -1;
|
int meta_fd = -1, data_fd = -1, journal_fd = -1;
|
||||||
uint64_t meta_offset, meta_device_sect, meta_device_size, meta_len, meta_format = 0;
|
uint64_t meta_offset = 0, meta_device_sect = 0, meta_device_size = 0, meta_area_size = 0, min_meta_len = 0;
|
||||||
uint64_t data_offset, data_device_sect, data_device_size, data_len;
|
uint64_t data_offset = 0, data_device_sect = 0, data_device_size = 0, data_len = 0;
|
||||||
uint64_t journal_offset, journal_device_sect, journal_device_size, journal_len;
|
uint64_t journal_offset = 0, journal_device_sect = 0, journal_device_size = 0, journal_len = 0;
|
||||||
|
uint64_t meta_format = 0;
|
||||||
|
|
||||||
uint32_t block_order = 0;
|
|
||||||
uint64_t block_count = 0;
|
uint64_t block_count = 0;
|
||||||
uint32_t clean_entry_bitmap_size = 0, clean_entry_size = 0, clean_dyn_size = 0;
|
uint32_t clean_entry_bitmap_size = 0;
|
||||||
|
uint32_t clean_entry_size = 0, clean_dyn_size = 0; // for meta_v1/2
|
||||||
|
|
||||||
|
bool mock_mode = false;
|
||||||
|
|
||||||
void parse_config(std::map<std::string, std::string> & config);
|
void parse_config(std::map<std::string, std::string> & config);
|
||||||
void open_data();
|
void open_data();
|
||||||
void open_meta();
|
void open_meta();
|
||||||
void open_journal();
|
void open_journal();
|
||||||
void calc_lengths(bool skip_meta_check = false);
|
void calc_lengths(bool skip_meta_check = false);
|
||||||
|
void check_lengths();
|
||||||
void close_all();
|
void close_all();
|
||||||
int trim_data(allocator_t *alloc);
|
int trim_data(std::function<bool(uint64_t)> is_free);
|
||||||
|
|
||||||
inline uint64_t dirty_dyn_size(uint64_t offset, uint64_t len)
|
inline uint64_t dirty_dyn_size(uint64_t offset, uint64_t len)
|
||||||
{
|
{
|
||||||
|
|||||||
+614
-1282
File diff suppressed because it is too large
Load Diff
@@ -1,22 +1,12 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.1 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
#define COPY_BUF_JOURNAL 1
|
|
||||||
#define COPY_BUF_DATA 2
|
|
||||||
#define COPY_BUF_ZERO 4
|
|
||||||
#define COPY_BUF_CSUM_FILL 8
|
|
||||||
#define COPY_BUF_COALESCED 16
|
|
||||||
#define COPY_BUF_META_BLOCK 32
|
|
||||||
#define COPY_BUF_JOURNALED_BIG 64
|
|
||||||
|
|
||||||
struct copy_buffer_t
|
struct copy_buffer_t
|
||||||
{
|
{
|
||||||
int copy_flags;
|
uint32_t copy_flags;
|
||||||
uint64_t offset, len, disk_offset;
|
uint64_t offset, len, disk_loc, disk_offset, disk_len;
|
||||||
uint64_t journal_sector; // only for reads: sector+1 if used and !journal.inmemory, otherwise 0
|
uint8_t *buf;
|
||||||
void *buf;
|
heap_entry_t *wr;
|
||||||
uint8_t *csum_buf;
|
|
||||||
int *dyn_data;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
struct meta_sector_t
|
struct meta_sector_t
|
||||||
@@ -27,13 +17,6 @@ struct meta_sector_t
|
|||||||
int usage_count;
|
int usage_count;
|
||||||
};
|
};
|
||||||
|
|
||||||
struct flusher_sync_t
|
|
||||||
{
|
|
||||||
bool fsync_meta;
|
|
||||||
int ready_count;
|
|
||||||
int state;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct flusher_meta_write_t
|
struct flusher_meta_write_t
|
||||||
{
|
{
|
||||||
uint64_t sector, pos;
|
uint64_t sector, pos;
|
||||||
@@ -49,94 +32,73 @@ class journal_flusher_co
|
|||||||
{
|
{
|
||||||
blockstore_impl_t *bs;
|
blockstore_impl_t *bs;
|
||||||
journal_flusher_t *flusher;
|
journal_flusher_t *flusher;
|
||||||
int wait_state, wait_count, wait_journal_count;
|
int co_id;
|
||||||
|
int wait_state, wait_count;
|
||||||
struct io_uring_sqe *sqe;
|
struct io_uring_sqe *sqe;
|
||||||
struct ring_data_t *data;
|
struct ring_data_t *data;
|
||||||
|
uint8_t *new_csums = NULL;
|
||||||
|
uint8_t *new_bmp = NULL;
|
||||||
|
uint8_t *punch_bmp = NULL;
|
||||||
|
uint8_t *new_ext_bmp = NULL;
|
||||||
|
|
||||||
std::list<flusher_sync_t>::iterator cur_sync;
|
std::function<void(ring_data_t*)> simple_callback_r, simple_callback_w;
|
||||||
|
|
||||||
obj_ver_id cur;
|
object_id cur_oid;
|
||||||
std::map<obj_ver_id, dirty_entry>::iterator dirty_it, dirty_start, dirty_end;
|
heap_entry_t *cur_obj;
|
||||||
std::map<object_id, uint64_t>::iterator repeat_it;
|
uint64_t fsynced_lsn;
|
||||||
std::function<void(ring_data_t*)> simple_callback_r, simple_callback_rj, simple_callback_w;
|
heap_compact_t compact_info;
|
||||||
|
uint64_t clean_loc;
|
||||||
|
uint32_t modified_block;
|
||||||
|
bool bitmap_copied;
|
||||||
|
bool should_repeat;
|
||||||
|
|
||||||
bool try_trim = false;
|
std::vector<copy_buffer_t> read_vec;
|
||||||
bool skip_copy, has_delete, has_writes;
|
std::vector<heap_entry_t*> csum_copy;
|
||||||
std::vector<copy_buffer_t> v;
|
uint32_t overwrite_start, overwrite_end;
|
||||||
std::vector<copy_buffer_t>::iterator it;
|
int i, res;
|
||||||
int i;
|
bool read_to_fill_incomplete;
|
||||||
bool fill_incomplete, cleared_incomplete;
|
|
||||||
int read_to_fill_incomplete;
|
|
||||||
int copy_count;
|
int copy_count;
|
||||||
uint64_t clean_loc, clean_ver, old_clean_loc, old_clean_ver;
|
|
||||||
flusher_meta_write_t meta_old, meta_new;
|
|
||||||
bool clean_init_bitmap;
|
|
||||||
uint64_t clean_bitmap_offset, clean_bitmap_len;
|
|
||||||
uint8_t *clean_init_dyn_ptr;
|
|
||||||
uint8_t *new_clean_bitmap;
|
|
||||||
|
|
||||||
uint64_t new_trim_pos;
|
|
||||||
|
|
||||||
friend class journal_flusher_t;
|
friend class journal_flusher_t;
|
||||||
void scan_dirty();
|
|
||||||
bool read_dirty(int wait_base);
|
void iterate_checksum_holes(std::function<void(int & pos, uint32_t hole_start, uint32_t hole_end)> cb);
|
||||||
bool modify_meta_do_reads(int wait_base);
|
void fill_partial_checksum_blocks();
|
||||||
bool wait_meta_reads(int wait_base);
|
|
||||||
bool modify_meta_read(uint64_t meta_loc, flusher_meta_write_t &wr, int wait_base);
|
|
||||||
bool clear_incomplete_csum_block_bits(int wait_base);
|
|
||||||
void calc_block_checksums(uint32_t *new_data_csums, bool skip_overwrites);
|
|
||||||
void update_metadata_entry();
|
|
||||||
bool write_meta_block(flusher_meta_write_t & meta_block, int wait_base);
|
|
||||||
void update_clean_db();
|
|
||||||
void free_data_blocks();
|
|
||||||
bool fsync_batch(bool fsync_meta, int wait_base);
|
|
||||||
bool trim_journal(int wait_base);
|
|
||||||
void free_buffers();
|
void free_buffers();
|
||||||
|
int check_and_punch_checksums();
|
||||||
|
bool calc_block_checksums();
|
||||||
|
bool write_meta_block(int wait_base);
|
||||||
|
bool read_buffered(int wait_base);
|
||||||
|
bool fsync_meta(int wait_base);
|
||||||
|
bool fsync_buffer(int wait_base);
|
||||||
|
bool trim_lsn(int wait_base);
|
||||||
public:
|
public:
|
||||||
journal_flusher_co();
|
journal_flusher_co();
|
||||||
|
~journal_flusher_co();
|
||||||
bool loop();
|
bool loop();
|
||||||
};
|
};
|
||||||
|
|
||||||
// Journal flusher itself
|
// Journal flusher itself
|
||||||
class journal_flusher_t
|
class journal_flusher_t
|
||||||
{
|
{
|
||||||
int trim_wanted = 0;
|
int force_start = 0;
|
||||||
bool dequeuing;
|
int min_flusher_count = 0, max_flusher_count = 0, cur_flusher_count = 0, target_flusher_count = 0;
|
||||||
int min_flusher_count, max_flusher_count, cur_flusher_count, target_flusher_count;
|
|
||||||
int flusher_start_threshold;
|
|
||||||
journal_flusher_co *co;
|
journal_flusher_co *co;
|
||||||
blockstore_impl_t *bs;
|
blockstore_impl_t *bs;
|
||||||
friend class journal_flusher_co;
|
friend class journal_flusher_co;
|
||||||
|
|
||||||
int journal_trim_counter;
|
robin_hood::unordered_flat_set<object_id> flushing;
|
||||||
bool trimming;
|
int active_flushers = 0;
|
||||||
void* journal_superblock;
|
int wanting_meta_fsync = 0;
|
||||||
|
bool fsyncing_meta = false;
|
||||||
int active_flushers;
|
int syncing_buffer = 0;
|
||||||
int syncing_flushers;
|
|
||||||
std::list<flusher_sync_t> syncs;
|
|
||||||
std::map<object_id, uint64_t> sync_to_repeat;
|
|
||||||
|
|
||||||
std::map<uint64_t, meta_sector_t> meta_sectors;
|
|
||||||
std::deque<object_id> flush_queue;
|
|
||||||
std::unordered_map<object_id, uint64_t> flush_versions;
|
|
||||||
std::unordered_set<uint64_t> inflight_meta_sectors;
|
|
||||||
|
|
||||||
bool try_find_older(std::map<obj_ver_id, dirty_entry>::iterator & dirty_end, obj_ver_id & cur);
|
|
||||||
bool try_find_other(std::map<obj_ver_id, dirty_entry>::iterator & dirty_end, obj_ver_id & cur);
|
|
||||||
|
|
||||||
public:
|
public:
|
||||||
journal_flusher_t(blockstore_impl_t *bs);
|
journal_flusher_t(blockstore_impl_t *bs);
|
||||||
~journal_flusher_t();
|
~journal_flusher_t();
|
||||||
void loop();
|
void loop();
|
||||||
bool is_trim_wanted() { return trim_wanted; }
|
int get_syncing_buffer();
|
||||||
bool is_active();
|
bool is_active();
|
||||||
void mark_trim_possible();
|
|
||||||
void request_trim();
|
void request_trim();
|
||||||
void release_trim();
|
void release_trim();
|
||||||
void enqueue_flush(obj_ver_id oid);
|
|
||||||
void unshift_flush(obj_ver_id oid, bool force);
|
|
||||||
void remove_flush(object_id oid);
|
|
||||||
void dump_diagnostics();
|
void dump_diagnostics();
|
||||||
bool is_mutated(uint64_t clean_loc);
|
|
||||||
};
|
};
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,358 @@
|
|||||||
|
// Metadata storage version 3 ("lsm heap")
|
||||||
|
// Copyright (c) Vitaliy Filippov, 2025+
|
||||||
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include <map>
|
||||||
|
#include <unordered_map>
|
||||||
|
#include <set>
|
||||||
|
#include <deque>
|
||||||
|
#include <vector>
|
||||||
|
|
||||||
|
#include "../client/object_id.h"
|
||||||
|
#include "../util/robin_hood.h"
|
||||||
|
#include "blockstore_disk.h"
|
||||||
|
#include "multilist.h"
|
||||||
|
|
||||||
|
struct pool_shard_settings_t
|
||||||
|
{
|
||||||
|
uint32_t pg_count;
|
||||||
|
uint32_t pg_stripe_size;
|
||||||
|
uint32_t no_inode_stats;
|
||||||
|
};
|
||||||
|
|
||||||
|
#define BS_HEAP_TYPE 0x07
|
||||||
|
#define BS_HEAP_BIG_WRITE 1
|
||||||
|
#define BS_HEAP_SMALL_WRITE 2
|
||||||
|
#define BS_HEAP_INTENT_WRITE 3
|
||||||
|
#define BS_HEAP_BIG_INTENT 4
|
||||||
|
#define BS_HEAP_DELETE 5
|
||||||
|
#define BS_HEAP_COMMIT 6
|
||||||
|
#define BS_HEAP_ROLLBACK 7
|
||||||
|
#define BS_HEAP_STABLE 0x40
|
||||||
|
#define BS_HEAP_GARBAGE 0x80
|
||||||
|
|
||||||
|
class blockstore_heap_t;
|
||||||
|
|
||||||
|
struct heap_small_write_t;
|
||||||
|
struct heap_big_write_t;
|
||||||
|
struct heap_big_intent_t;
|
||||||
|
|
||||||
|
struct __attribute__((__packed__)) heap_entry_t
|
||||||
|
{
|
||||||
|
uint16_t size;
|
||||||
|
uint16_t entry_type;
|
||||||
|
uint32_t crc32c;
|
||||||
|
uint64_t lsn;
|
||||||
|
uint64_t inode;
|
||||||
|
uint64_t stripe;
|
||||||
|
uint64_t version;
|
||||||
|
|
||||||
|
// uint8_t[] external_bitmap
|
||||||
|
// uint8_t[] internal_bitmap
|
||||||
|
// uint32_t[] checksums
|
||||||
|
|
||||||
|
inline uint8_t type() const { return (entry_type & BS_HEAP_TYPE); }
|
||||||
|
inline heap_small_write_t& small() { return *(heap_small_write_t*)this; }
|
||||||
|
inline heap_big_write_t& big() { return *(heap_big_write_t*)this; }
|
||||||
|
inline heap_big_intent_t& big_intent() { return *(heap_big_intent_t*)this; }
|
||||||
|
bool is_garbage();
|
||||||
|
void set_garbage();
|
||||||
|
bool is_overwrite();
|
||||||
|
bool is_compactable();
|
||||||
|
bool is_before(heap_entry_t *other);
|
||||||
|
uint32_t get_size(blockstore_heap_t *heap);
|
||||||
|
uint8_t *get_ext_bitmap(blockstore_heap_t *heap);
|
||||||
|
uint8_t *get_int_bitmap(blockstore_heap_t *heap);
|
||||||
|
uint8_t *get_checksums(blockstore_heap_t *heap);
|
||||||
|
uint32_t *get_checksum(blockstore_heap_t *heap);
|
||||||
|
uint64_t big_location(blockstore_heap_t *heap);
|
||||||
|
void set_big_location(blockstore_heap_t *heap, uint64_t location);
|
||||||
|
uint32_t calc_crc32c();
|
||||||
|
};
|
||||||
|
|
||||||
|
struct __attribute__((__packed__)) heap_small_write_t
|
||||||
|
{
|
||||||
|
heap_entry_t hdr;
|
||||||
|
|
||||||
|
uint64_t location;
|
||||||
|
uint32_t offset;
|
||||||
|
uint32_t len;
|
||||||
|
|
||||||
|
// Also includes 1 bitmap and 1 crc32c after the bitmap if checksums are disabled
|
||||||
|
};
|
||||||
|
|
||||||
|
struct __attribute__((__packed__)) heap_big_write_t
|
||||||
|
{
|
||||||
|
heap_entry_t hdr;
|
||||||
|
|
||||||
|
uint32_t block_num;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct __attribute__((__packed__)) heap_big_intent_t
|
||||||
|
{
|
||||||
|
heap_entry_t hdr;
|
||||||
|
|
||||||
|
uint32_t block_num;
|
||||||
|
uint32_t offset;
|
||||||
|
uint32_t len;
|
||||||
|
|
||||||
|
// Also includes 2 bitmaps and 1 crc32c if checksums are disabled
|
||||||
|
};
|
||||||
|
|
||||||
|
struct __attribute__((__packed__)) heap_list_item_t
|
||||||
|
{
|
||||||
|
heap_list_item_t *prev;
|
||||||
|
heap_list_item_t *next;
|
||||||
|
uint32_t block_num;
|
||||||
|
heap_entry_t entry;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct heap_object_mvcc_t
|
||||||
|
{
|
||||||
|
uint32_t readers = 0;
|
||||||
|
heap_entry_t *garbage_entry = NULL;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct heap_block_info_t
|
||||||
|
{
|
||||||
|
uint32_t used_space = 0;
|
||||||
|
uint64_t mod_lsn = 0, mod_lsn_to = 0; // only 1 block write of LSN sequence is allowed at a moment
|
||||||
|
bool is_writing: 1;
|
||||||
|
bool has_garbage: 1;
|
||||||
|
std::vector<heap_list_item_t*> entries;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct heap_inflight_lsn_t
|
||||||
|
{
|
||||||
|
uint64_t flags;
|
||||||
|
heap_entry_t *wr;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct heap_compact_t
|
||||||
|
{
|
||||||
|
uint64_t compact_lsn, compact_version;
|
||||||
|
heap_entry_t *clean_wr;
|
||||||
|
bool do_delete;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct heap_reshard_state_t;
|
||||||
|
|
||||||
|
struct heap_li_hash
|
||||||
|
{
|
||||||
|
size_t operator()(const heap_list_item_t* li) const noexcept
|
||||||
|
{
|
||||||
|
return robin_hood::hash_int(li->entry.stripe);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
struct heap_li_equal
|
||||||
|
{
|
||||||
|
constexpr bool operator()(const heap_list_item_t* a, const heap_list_item_t* b) const noexcept
|
||||||
|
{
|
||||||
|
return a->entry.stripe == b->entry.stripe;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
using i64hash_t = robin_hood::hash<uint64_t>;
|
||||||
|
using heap_inode_map_t = robin_hood::unordered_flat_set<heap_list_item_t*, heap_li_hash, heap_li_equal, 88>;
|
||||||
|
using heap_block_index_t = robin_hood::unordered_flat_map<uint64_t,
|
||||||
|
robin_hood::unordered_flat_map<inode_t, void*, i64hash_t>, i64hash_t>;
|
||||||
|
using heap_mvcc_map_t = robin_hood::unordered_flat_map<object_id, heap_object_mvcc_t>;
|
||||||
|
|
||||||
|
class blockstore_heap_t
|
||||||
|
{
|
||||||
|
friend struct heap_entry_t;
|
||||||
|
|
||||||
|
blockstore_disk_t *dsk = NULL;
|
||||||
|
uint8_t* buffer_area = NULL;
|
||||||
|
int log_level = 0;
|
||||||
|
const uint32_t meta_block_count = 0;
|
||||||
|
const uint32_t max_entry_size = 0;
|
||||||
|
|
||||||
|
robin_hood::unordered_flat_map<pool_id_t, pool_shard_settings_t> pool_shard_settings;
|
||||||
|
// PG => inode => stripe => block number
|
||||||
|
heap_block_index_t block_index;
|
||||||
|
std::vector<heap_block_info_t> block_info;
|
||||||
|
allocator_t *data_alloc = NULL;
|
||||||
|
multilist_index_t *meta_alloc = NULL;
|
||||||
|
uint32_t meta_nearfull_blocks = 0;
|
||||||
|
uint64_t meta_used_space = 0;
|
||||||
|
multilist_alloc_t *buffer_alloc = NULL;
|
||||||
|
std::map<uint64_t, uint64_t> inode_space_stats;
|
||||||
|
uint64_t buffer_area_used_space = 0;
|
||||||
|
uint64_t data_used_space = 0;
|
||||||
|
|
||||||
|
uint64_t next_lsn = 0;
|
||||||
|
uint32_t last_allocated_block = UINT32_MAX;
|
||||||
|
heap_mvcc_map_t object_mvcc;
|
||||||
|
|
||||||
|
// LSN queue: inflight (writing) -> completed [-> fsynced]
|
||||||
|
std::deque<heap_inflight_lsn_t> inflight_lsn;
|
||||||
|
uint32_t to_compact_count = 0;
|
||||||
|
uint64_t compacted_count = 0;
|
||||||
|
uint32_t inflight_overwrite_count = 0;
|
||||||
|
uint64_t first_inflight_lsn = 0;
|
||||||
|
uint64_t completed_lsn = 0;
|
||||||
|
uint64_t fsynced_lsn = 0;
|
||||||
|
std::deque<object_id> compact_queue;
|
||||||
|
|
||||||
|
bool marked_used_blocks = false;
|
||||||
|
bool recheck_queue_filled = false;
|
||||||
|
std::vector<heap_list_item_t*> loaded_list_items;
|
||||||
|
std::set<uint32_t> recheck_modified_blocks;
|
||||||
|
std::deque<heap_entry_t*> recheck_queue;
|
||||||
|
int recheck_in_progress = 0;
|
||||||
|
bool in_recheck = false;
|
||||||
|
std::function<void(bool is_data, uint64_t offset, uint64_t len, uint8_t* buf, std::function<void()>)> recheck_cb;
|
||||||
|
int recheck_queue_depth = 0;
|
||||||
|
|
||||||
|
uint64_t get_pg_id(inode_t inode, uint64_t stripe);
|
||||||
|
bool validate_object(heap_entry_t *obj);
|
||||||
|
void fill_recheck_queue();
|
||||||
|
int mark_used_blocks();
|
||||||
|
void recheck_buffer(heap_entry_t *cwr, uint8_t *buf);
|
||||||
|
void defragment_block(uint32_t block_num);
|
||||||
|
void reshard_add(heap_reshard_state_t *st, heap_list_item_t *li);
|
||||||
|
|
||||||
|
void gc_block(heap_block_info_t & inf);
|
||||||
|
int allocate_entry(uint32_t entry_size, uint32_t *block_num, bool allow_last_free);
|
||||||
|
void insert_list_item(heap_list_item_t *li);
|
||||||
|
int add_entry(uint32_t wr_size, uint32_t *modified_block, bool allow_last_free,
|
||||||
|
bool explicit_complete, std::function<void(heap_entry_t *wr)> fill_entry);
|
||||||
|
int add_simple(heap_entry_t *obj, uint64_t version, uint32_t *modified_block, uint32_t entry_type);
|
||||||
|
uint32_t meta_alloc_pos(const heap_block_info_t & inf);
|
||||||
|
void modify_alloc(uint32_t block_num, std::function<void(heap_block_info_t &)> change_cb);
|
||||||
|
void mark_garbage_up_to(heap_entry_t *wr);
|
||||||
|
void mark_garbage(uint32_t block_num, heap_entry_t *prev_wr, uint32_t used_big);
|
||||||
|
void push_inflight_lsn(uint64_t lsn, heap_entry_t *wr, uint64_t flags);
|
||||||
|
void mark_completed_lsns(uint64_t mod_lsn);
|
||||||
|
void apply_inflight(heap_inflight_lsn_t & inflight);
|
||||||
|
public:
|
||||||
|
blockstore_heap_t(blockstore_disk_t *dsk, uint8_t *buffer_area, int log_level = 0);
|
||||||
|
~blockstore_heap_t();
|
||||||
|
void start_load(uint64_t completed_lsn);
|
||||||
|
// load data from the disk, returns EDOM on corruption
|
||||||
|
int read_blocks(uint64_t disk_offset, uint64_t size, uint8_t *buf, bool allow_corrupted,
|
||||||
|
std::function<void(uint32_t block_num, heap_entry_t* wr)> handle_write,
|
||||||
|
std::function<void(uint32_t, uint32_t, uint8_t*)> handle_block);
|
||||||
|
int load_blocks(uint64_t disk_offset, uint64_t size, uint8_t *buf,
|
||||||
|
bool allow_corrupted, uint64_t &entries_loaded);
|
||||||
|
// finish loading - should be called after load_blocks
|
||||||
|
void finish_load();
|
||||||
|
// get blocks which are modified during loading and should be written to the disk
|
||||||
|
// before finishing initialization if not R/O
|
||||||
|
std::vector<uint32_t> get_recheck_modified_blocks();
|
||||||
|
// recheck small write data after reading the database from disk
|
||||||
|
bool recheck_small_writes(std::function<void(bool is_data, uint64_t offset, uint64_t len, uint8_t* buf, std::function<void()>)> read_buffer, int queue_depth);
|
||||||
|
int finish_recheck();
|
||||||
|
// reshard database according to the pool's PG count
|
||||||
|
void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit);
|
||||||
|
bool reshard_continue(void* reshard_state, uint64_t chunk_limit);
|
||||||
|
bool reshard_check(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size);
|
||||||
|
void set_no_inode_stats(const std::vector<uint64_t> & pool_ids);
|
||||||
|
void recalc_inode_space_stats(uint64_t pool_id, bool per_inode);
|
||||||
|
// read an object entry and lock it against removal
|
||||||
|
// in the future, may become asynchronous
|
||||||
|
heap_entry_t *lock_and_read_entry(object_id oid);
|
||||||
|
// read an object entry without locking it
|
||||||
|
heap_entry_t *read_entry(object_id oid);
|
||||||
|
// unlock an entry
|
||||||
|
bool unlock_entry(object_id oid);
|
||||||
|
// set or verify checksums in a write request
|
||||||
|
bool calc_checksums(heap_entry_t *wr, uint8_t *data, bool set, uint32_t offset = UINT32_MAX, uint32_t len = UINT32_MAX);
|
||||||
|
// set or verify raw block checksums
|
||||||
|
bool calc_block_checksums(uint32_t *block_csums, uint8_t *data, uint8_t *bitmap, uint32_t start, uint32_t end,
|
||||||
|
bool set, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb);
|
||||||
|
bool calc_block_checksums(uint32_t *block_csums, uint8_t *bitmap,
|
||||||
|
uint32_t start, uint32_t end, std::function<uint8_t*(uint32_t start, uint32_t & len)> next,
|
||||||
|
bool set, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb);
|
||||||
|
// adds a small_write or intent_write entry to an object
|
||||||
|
// return 0 if OK, or maybe ENOSPC
|
||||||
|
int add_small_write(object_id oid, heap_entry_t **obj_ptr, uint16_t type, uint64_t version,
|
||||||
|
uint32_t offset, uint32_t len, uint64_t location, uint8_t *bitmap, uint8_t *data, uint32_t *modified_block);
|
||||||
|
// adds a big_write (overwrite) entry to an object
|
||||||
|
int add_big_write(object_id oid, heap_entry_t *old_head, bool stable, uint64_t version,
|
||||||
|
uint32_t offset, uint32_t len, uint64_t location, uint8_t *bitmap, uint8_t *data, uint32_t *modified_block);
|
||||||
|
// adds a "redirecting" big_intent entry to an object (same as big_write, used to avoid fsync on desktop SSDs)
|
||||||
|
int add_redirect_intent(object_id oid, heap_entry_t **obj_ptr, uint64_t version,
|
||||||
|
uint32_t offset, uint32_t len, uint64_t location, uint8_t *bitmap, uint8_t *data, uint32_t *modified_block);
|
||||||
|
// adds a big_intent (atomic partial modification) entry to an object
|
||||||
|
int add_big_intent(object_id oid, heap_entry_t **obj_ptr, uint64_t version,
|
||||||
|
uint32_t offset, uint32_t len, uint8_t *bitmap, uint8_t *data, uint8_t *checksums, uint32_t *modified_block);
|
||||||
|
// adds a compacted up to <version> entry to an object
|
||||||
|
int add_compact(heap_entry_t *obj, uint64_t compact_version, uint64_t compact_lsn, uint64_t compact_location,
|
||||||
|
bool do_delete, uint32_t *modified_block, uint8_t *new_int_bitmap, uint8_t *new_ext_bitmap, uint8_t *new_csums);
|
||||||
|
// "punch holes" in a big_entry
|
||||||
|
int punch_holes(heap_entry_t *wr, uint8_t *new_bitmap, uint8_t *new_csums, uint32_t *modified_block);
|
||||||
|
// stabilize an unstable object version
|
||||||
|
// return 0 if OK, ENOENT if not exists
|
||||||
|
int add_commit(heap_entry_t *obj, uint64_t version, uint32_t *modified_block);
|
||||||
|
// rollback an unstable object version
|
||||||
|
// return 0 if OK, ENOENT if not exists, EBUSY if already stable
|
||||||
|
int add_rollback(heap_entry_t *obj, uint64_t version, uint32_t *modified_block);
|
||||||
|
// forget an object
|
||||||
|
// return error code
|
||||||
|
int add_delete(heap_entry_t *obj, uint32_t *modified_block);
|
||||||
|
// get the next object to compact
|
||||||
|
// guaranteed to return objects in min lsn order
|
||||||
|
// returns 0 if OK, ENOENT if nothing to compact
|
||||||
|
int get_next_compact(object_id & oid);
|
||||||
|
void iterate_with_stable(heap_entry_t *obj, uint64_t max_lsn, std::function<bool(heap_entry_t*, bool stable)> cb);
|
||||||
|
// iterate compactable entries
|
||||||
|
heap_compact_t iterate_compaction(heap_entry_t *obj, uint64_t fsynced_lsn, bool under_pressure,
|
||||||
|
std::function<void(heap_entry_t*)> small_wr_cb);
|
||||||
|
// iterate all objects
|
||||||
|
void iterate_objects(std::function<void(heap_entry_t*, uint32_t block_num)> cb);
|
||||||
|
// retrieve object listing from a PG
|
||||||
|
int list_objects(uint32_t pg_num, object_id min_oid, object_id max_oid,
|
||||||
|
obj_ver_id **result_list, size_t *stable_count, size_t *unstable_count);
|
||||||
|
|
||||||
|
// inflight write tracking
|
||||||
|
void start_block_write(uint32_t block_num);
|
||||||
|
void complete_block_write(uint32_t block_num);
|
||||||
|
void complete_lsn_write(uint64_t lsn);
|
||||||
|
bool is_lsn_completed(uint64_t lsn);
|
||||||
|
uint64_t get_completed_lsn();
|
||||||
|
uint64_t get_fsynced_lsn();
|
||||||
|
void mark_lsn_fsynced(uint64_t lsn);
|
||||||
|
|
||||||
|
// data device block allocator functions
|
||||||
|
uint64_t find_free_data();
|
||||||
|
bool is_data_used(uint64_t location);
|
||||||
|
void use_data(inode_t inode, uint64_t location);
|
||||||
|
void free_data(inode_t inode, uint64_t location);
|
||||||
|
|
||||||
|
// buffer device allocator functions
|
||||||
|
uint64_t find_free_buffer_area(uint64_t size);
|
||||||
|
bool is_buffer_area_free(uint64_t location, uint64_t size);
|
||||||
|
void use_buffer_area(inode_t inode, uint64_t location, uint64_t size);
|
||||||
|
void free_buffer_area(inode_t inode, uint64_t location, uint64_t size);
|
||||||
|
uint64_t get_buffer_area_used_space();
|
||||||
|
|
||||||
|
// get metadata block data buffer and used space
|
||||||
|
void get_meta_block(uint32_t block_num, uint8_t *buffer);
|
||||||
|
void fill_block_empty_space(uint8_t *buffer, uint32_t pos);
|
||||||
|
uint32_t get_meta_block_used_space(uint32_t block_num);
|
||||||
|
|
||||||
|
// get space usage statistics
|
||||||
|
uint64_t get_data_used_space();
|
||||||
|
const std::map<uint64_t, uint64_t> & get_inode_space_stats();
|
||||||
|
uint64_t get_meta_total_space();
|
||||||
|
uint64_t get_meta_used_space();
|
||||||
|
uint32_t get_meta_nearfull_blocks();
|
||||||
|
uint32_t get_compact_queue_size();
|
||||||
|
uint32_t get_to_compact_count();
|
||||||
|
uint64_t get_compacted_count();
|
||||||
|
|
||||||
|
uint64_t entry_pos(uint32_t block_num, uint32_t offset);
|
||||||
|
heap_entry_t *entry_from_pos(uint64_t entry_pos, bool allow_unallocated = false);
|
||||||
|
heap_entry_t *prev(heap_entry_t *wr);
|
||||||
|
uint32_t get_simple_entry_size();
|
||||||
|
uint32_t get_big_entry_size();
|
||||||
|
uint32_t get_big_intent_entry_size();
|
||||||
|
uint32_t get_small_entry_size(uint32_t offset, uint32_t len);
|
||||||
|
uint32_t get_csum_size(heap_entry_t *wr);
|
||||||
|
uint32_t get_csum_size(uint32_t entry_type, uint32_t offset = 0, uint32_t len = 0);
|
||||||
|
};
|
||||||
+115
-494
@@ -1,13 +1,18 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.1 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
#include "blockstore_impl.h"
|
#include <stdexcept>
|
||||||
|
|
||||||
blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_t *ringloop, timerfd_manager_t *tfd)
|
#include "blockstore_impl.h"
|
||||||
|
#include "blockstore_internal.h"
|
||||||
|
#include "crc32c.h"
|
||||||
|
|
||||||
|
blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd, bool mock_mode)
|
||||||
{
|
{
|
||||||
assert(sizeof(blockstore_op_private_t) <= BS_OP_PRIVATE_DATA_SIZE);
|
assert(sizeof(blockstore_op_private_t) <= BS_OP_PRIVATE_DATA_SIZE);
|
||||||
this->tfd = tfd;
|
this->tfd = tfd;
|
||||||
this->ringloop = ringloop;
|
this->ringloop = ringloop;
|
||||||
|
dsk.mock_mode = mock_mode;
|
||||||
ring_consumer.loop = [this]() { loop(); };
|
ring_consumer.loop = [this]() { loop(); };
|
||||||
ringloop->register_consumer(&ring_consumer);
|
ringloop->register_consumer(&ring_consumer);
|
||||||
initialized = 0;
|
initialized = 0;
|
||||||
@@ -17,31 +22,37 @@ blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_t *
|
|||||||
dsk.open_data();
|
dsk.open_data();
|
||||||
dsk.open_meta();
|
dsk.open_meta();
|
||||||
dsk.open_journal();
|
dsk.open_journal();
|
||||||
calc_lengths();
|
dsk.calc_lengths();
|
||||||
alloc_dyn_data = dsk.clean_dyn_size > sizeof(void*) || dsk.csum_block_size > 0;
|
dsk.check_lengths();
|
||||||
zero_object = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.data_block_size);
|
|
||||||
data_alloc = new allocator_t(dsk.block_count);
|
|
||||||
}
|
}
|
||||||
catch (std::exception & e)
|
catch (std::exception & e)
|
||||||
{
|
{
|
||||||
dsk.close_all();
|
dsk.close_all();
|
||||||
throw;
|
throw;
|
||||||
}
|
}
|
||||||
|
meta_superblock = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.meta_block_size);
|
||||||
|
memset(meta_superblock, 0, dsk.meta_block_size);
|
||||||
flusher = new journal_flusher_t(this);
|
flusher = new journal_flusher_t(this);
|
||||||
|
if (dsk.inmemory_journal)
|
||||||
|
{
|
||||||
|
buffer_area = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.journal_len);
|
||||||
|
}
|
||||||
|
heap = new blockstore_heap_t(&dsk, buffer_area, log_level);
|
||||||
|
ringloop->wakeup();
|
||||||
}
|
}
|
||||||
|
|
||||||
blockstore_impl_t::~blockstore_impl_t()
|
blockstore_impl_t::~blockstore_impl_t()
|
||||||
{
|
{
|
||||||
delete data_alloc;
|
if (flusher)
|
||||||
delete flusher;
|
delete flusher;
|
||||||
if (zero_object)
|
if (heap)
|
||||||
free(zero_object);
|
delete heap;
|
||||||
|
if (buffer_area)
|
||||||
|
free(buffer_area);
|
||||||
|
if (meta_superblock)
|
||||||
|
free(meta_superblock);
|
||||||
ringloop->unregister_consumer(&ring_consumer);
|
ringloop->unregister_consumer(&ring_consumer);
|
||||||
dsk.close_all();
|
dsk.close_all();
|
||||||
if (metadata_buffer)
|
|
||||||
free(metadata_buffer);
|
|
||||||
if (clean_bitmaps)
|
|
||||||
free(clean_bitmaps);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
bool blockstore_impl_t::is_started()
|
bool blockstore_impl_t::is_started()
|
||||||
@@ -57,10 +68,9 @@ bool blockstore_impl_t::is_stalled()
|
|||||||
// main event loop - produce requests
|
// main event loop - produce requests
|
||||||
void blockstore_impl_t::loop()
|
void blockstore_impl_t::loop()
|
||||||
{
|
{
|
||||||
// FIXME: initialized == 10 is ugly
|
|
||||||
if (initialized != 10)
|
if (initialized != 10)
|
||||||
{
|
{
|
||||||
// read metadata, then journal
|
// read metadata
|
||||||
if (initialized == 0)
|
if (initialized == 0)
|
||||||
{
|
{
|
||||||
metadata_init_reader = new blockstore_init_meta(this);
|
metadata_init_reader = new blockstore_init_meta(this);
|
||||||
@@ -73,69 +83,41 @@ void blockstore_impl_t::loop()
|
|||||||
{
|
{
|
||||||
delete metadata_init_reader;
|
delete metadata_init_reader;
|
||||||
metadata_init_reader = NULL;
|
metadata_init_reader = NULL;
|
||||||
journal_init_reader = new blockstore_init_journal(this);
|
|
||||||
initialized = 2;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (initialized == 2)
|
|
||||||
{
|
|
||||||
int res = journal_init_reader->loop();
|
|
||||||
if (!res)
|
|
||||||
{
|
|
||||||
delete journal_init_reader;
|
|
||||||
journal_init_reader = NULL;
|
|
||||||
initialized = 3;
|
initialized = 3;
|
||||||
ringloop->wakeup();
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (initialized == 3)
|
if (initialized == 3)
|
||||||
{
|
{
|
||||||
if (!readonly && dsk.discard_on_start)
|
if (!readonly && dsk.discard_on_start)
|
||||||
dsk.trim_data(data_alloc);
|
|
||||||
if (journal.flush_journal)
|
|
||||||
initialized = 4;
|
|
||||||
else
|
|
||||||
initialized = 10;
|
|
||||||
}
|
|
||||||
if (initialized == 4)
|
|
||||||
{
|
|
||||||
if (readonly)
|
|
||||||
{
|
{
|
||||||
printf("Can't flush the journal in readonly mode\n");
|
dsk.trim_data([this](uint64_t block_num){ return heap->is_data_used(block_num * dsk.data_block_size); });
|
||||||
exit(1);
|
|
||||||
}
|
}
|
||||||
flusher->loop();
|
initialized = 10;
|
||||||
ringloop->submit();
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
// try to submit ops
|
// try to submit ops
|
||||||
unsigned initial_ring_space = ringloop->space_left();
|
unsigned initial_ring_space = ringloop->space_left();
|
||||||
// has_writes == 0 - no writes before the current queue item
|
int op_idx = 0, new_idx = 0;
|
||||||
// has_writes == 1 - some writes in progress
|
bool has_unfinished_writes = false;
|
||||||
// has_writes == 2 - tried to submit some writes, but failed
|
|
||||||
int has_writes = 0, op_idx = 0, new_idx = 0;
|
|
||||||
for (; op_idx < submit_queue.size(); op_idx++, new_idx++)
|
for (; op_idx < submit_queue.size(); op_idx++, new_idx++)
|
||||||
{
|
{
|
||||||
auto op = submit_queue[op_idx];
|
auto op = submit_queue[op_idx];
|
||||||
submit_queue[new_idx] = op;
|
submit_queue[new_idx] = op;
|
||||||
// FIXME: This needs some simplification
|
|
||||||
// Writes should not block reads if the ring is not full and reads don't depend on them
|
|
||||||
// In all other cases we should stop submission
|
|
||||||
if (PRIV(op)->wait_for)
|
if (PRIV(op)->wait_for)
|
||||||
{
|
{
|
||||||
check_wait(op);
|
check_wait(op);
|
||||||
if (PRIV(op)->wait_for == WAIT_SQE)
|
if (PRIV(op)->wait_for == WAIT_SQE)
|
||||||
{
|
{
|
||||||
|
// ring is full, stop submission
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
else if (PRIV(op)->wait_for)
|
else if (PRIV(op)->wait_for)
|
||||||
{
|
{
|
||||||
if (op->opcode == BS_OP_WRITE || op->opcode == BS_OP_WRITE_STABLE || op->opcode == BS_OP_DELETE)
|
has_unfinished_writes = has_unfinished_writes || op->opcode == BS_OP_WRITE ||
|
||||||
{
|
op->opcode == BS_OP_WRITE_STABLE || op->opcode == BS_OP_DELETE ||
|
||||||
has_writes = 2;
|
op->opcode == BS_OP_STABLE || op->opcode == BS_OP_ROLLBACK;
|
||||||
}
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -148,46 +130,33 @@ void blockstore_impl_t::loop()
|
|||||||
{
|
{
|
||||||
wr_st = dequeue_read(op);
|
wr_st = dequeue_read(op);
|
||||||
}
|
}
|
||||||
else if (op->opcode == BS_OP_WRITE || op->opcode == BS_OP_WRITE_STABLE)
|
else if (op->opcode == BS_OP_WRITE || op->opcode == BS_OP_WRITE_STABLE || op->opcode == BS_OP_DELETE)
|
||||||
{
|
{
|
||||||
if (has_writes == 2)
|
|
||||||
{
|
|
||||||
// Some writes already could not be submitted
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
wr_st = dequeue_write(op);
|
wr_st = dequeue_write(op);
|
||||||
has_writes = wr_st > 0 ? 1 : 2;
|
has_unfinished_writes = has_unfinished_writes || (wr_st != 2);
|
||||||
}
|
|
||||||
else if (op->opcode == BS_OP_DELETE)
|
|
||||||
{
|
|
||||||
if (has_writes == 2)
|
|
||||||
{
|
|
||||||
// Some writes already could not be submitted
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
wr_st = dequeue_del(op);
|
|
||||||
has_writes = wr_st > 0 ? 1 : 2;
|
|
||||||
}
|
}
|
||||||
else if (op->opcode == BS_OP_SYNC)
|
else if (op->opcode == BS_OP_SYNC)
|
||||||
{
|
{
|
||||||
// sync only completed writes?
|
// syncs only completed writes, so doesn't have to be blocked by anything
|
||||||
// wait for the data device fsync to complete, then submit journal writes for big writes
|
|
||||||
// then submit an fsync operation
|
|
||||||
wr_st = continue_sync(op);
|
wr_st = continue_sync(op);
|
||||||
}
|
}
|
||||||
else if (op->opcode == BS_OP_STABLE)
|
else if (op->opcode == BS_OP_STABLE || op->opcode == BS_OP_ROLLBACK)
|
||||||
{
|
{
|
||||||
wr_st = dequeue_stable(op);
|
wr_st = dequeue_stable(op);
|
||||||
}
|
has_unfinished_writes = has_unfinished_writes || (wr_st != 2);
|
||||||
else if (op->opcode == BS_OP_ROLLBACK)
|
|
||||||
{
|
|
||||||
wr_st = dequeue_rollback(op);
|
|
||||||
}
|
}
|
||||||
else if (op->opcode == BS_OP_LIST)
|
else if (op->opcode == BS_OP_LIST)
|
||||||
{
|
{
|
||||||
// LIST doesn't have to be blocked by previous modifications
|
// LIST has to be blocked by previous writes and commits/rollbacks
|
||||||
process_list(op);
|
if (!has_unfinished_writes)
|
||||||
wr_st = 2;
|
{
|
||||||
|
process_list(op);
|
||||||
|
wr_st = 2;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
wr_st = 0;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if (wr_st == 2)
|
if (wr_st == 2)
|
||||||
{
|
{
|
||||||
@@ -196,16 +165,13 @@ void blockstore_impl_t::loop()
|
|||||||
}
|
}
|
||||||
if (wr_st == 0)
|
if (wr_st == 0)
|
||||||
{
|
{
|
||||||
|
PRIV(op)->pending_ops = 0;
|
||||||
ringloop->restore(prev_sqe_pos);
|
ringloop->restore(prev_sqe_pos);
|
||||||
if (PRIV(op)->wait_for == WAIT_SQE)
|
if (PRIV(op)->wait_for == WAIT_SQE)
|
||||||
{
|
{
|
||||||
// ring is full, stop submission
|
// ring is full, stop submission
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
else if (PRIV(op)->wait_for == WAIT_JOURNAL)
|
|
||||||
{
|
|
||||||
PRIV(op)->wait_detail2 = (unstable_writes.size()+unstable_unsynced);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (op_idx != new_idx)
|
if (op_idx != new_idx)
|
||||||
@@ -220,17 +186,19 @@ void blockstore_impl_t::loop()
|
|||||||
{
|
{
|
||||||
flusher->loop();
|
flusher->loop();
|
||||||
}
|
}
|
||||||
|
for (auto & block_num: pending_modified_blocks)
|
||||||
|
{
|
||||||
|
auto & mb = modified_blocks[block_num];
|
||||||
|
heap->get_meta_block(block_num, mb.buf);
|
||||||
|
heap->start_block_write(block_num);
|
||||||
|
mb.sent = true;
|
||||||
|
}
|
||||||
|
pending_modified_blocks.clear();
|
||||||
int ret = ringloop->submit();
|
int ret = ringloop->submit();
|
||||||
if (ret < 0)
|
if (ret < 0)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(std::string("io_uring_submit: ") + strerror(-ret));
|
throw std::runtime_error(std::string("io_uring_submit: ") + strerror(-ret));
|
||||||
}
|
}
|
||||||
for (auto s: journal.submitting_sectors)
|
|
||||||
{
|
|
||||||
// Mark journal sector writes as submitted
|
|
||||||
journal.sector_info[s].submit_id = 0;
|
|
||||||
}
|
|
||||||
journal.submitting_sectors.clear();
|
|
||||||
if ((initial_ring_space - ringloop->space_left()) > 0)
|
if ((initial_ring_space - ringloop->space_left()) > 0)
|
||||||
{
|
{
|
||||||
live = true;
|
live = true;
|
||||||
@@ -248,7 +216,7 @@ bool blockstore_impl_t::is_safe_to_stop()
|
|||||||
{
|
{
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
if (unsynced_big_writes.size() > 0 || unsynced_small_writes.size() > 0)
|
if (has_unsynced())
|
||||||
{
|
{
|
||||||
if (!readonly && !stop_sync_submitted)
|
if (!readonly && !stop_sync_submitted)
|
||||||
{
|
{
|
||||||
@@ -272,7 +240,7 @@ void blockstore_impl_t::check_wait(blockstore_op_t *op)
|
|||||||
{
|
{
|
||||||
if (PRIV(op)->wait_for == WAIT_SQE)
|
if (PRIV(op)->wait_for == WAIT_SQE)
|
||||||
{
|
{
|
||||||
if (ringloop->sqes_left() < PRIV(op)->wait_detail)
|
if (ringloop->space_left() < PRIV(op)->wait_detail)
|
||||||
{
|
{
|
||||||
// stop submission if there's still no free space
|
// stop submission if there's still no free space
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
#ifdef BLOCKSTORE_DEBUG
|
||||||
@@ -282,40 +250,13 @@ void blockstore_impl_t::check_wait(blockstore_op_t *op)
|
|||||||
}
|
}
|
||||||
PRIV(op)->wait_for = 0;
|
PRIV(op)->wait_for = 0;
|
||||||
}
|
}
|
||||||
else if (PRIV(op)->wait_for == WAIT_JOURNAL)
|
else if (PRIV(op)->wait_for == WAIT_COMPACTION)
|
||||||
{
|
{
|
||||||
if (journal.used_start == PRIV(op)->wait_detail &&
|
if (heap->get_compacted_count() <= PRIV(op)->wait_detail)
|
||||||
(unstable_writes.size()+unstable_unsynced) == PRIV(op)->wait_detail2)
|
|
||||||
{
|
{
|
||||||
// do not submit
|
// do not submit
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
#ifdef BLOCKSTORE_DEBUG
|
||||||
printf("Still waiting to flush journal offset %08jx\n", PRIV(op)->wait_detail);
|
printf("Still waiting for more flushes\n");
|
||||||
#endif
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
flusher->release_trim();
|
|
||||||
PRIV(op)->wait_for = 0;
|
|
||||||
}
|
|
||||||
else if (PRIV(op)->wait_for == WAIT_JOURNAL_BUFFER)
|
|
||||||
{
|
|
||||||
int next = ((journal.cur_sector + 1) % journal.sector_count);
|
|
||||||
if (journal.sector_info[next].flush_count > 0 ||
|
|
||||||
journal.sector_info[next].dirty)
|
|
||||||
{
|
|
||||||
// do not submit
|
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
|
||||||
printf("Still waiting for a journal buffer\n");
|
|
||||||
#endif
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
PRIV(op)->wait_for = 0;
|
|
||||||
}
|
|
||||||
else if (PRIV(op)->wait_for == WAIT_FREE)
|
|
||||||
{
|
|
||||||
if (!data_alloc->get_free_count() && big_to_flush > 0)
|
|
||||||
{
|
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
|
||||||
printf("Still waiting for free space on the data device\n");
|
|
||||||
#endif
|
#endif
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
@@ -334,7 +275,8 @@ void blockstore_impl_t::enqueue_op(blockstore_op_t *op)
|
|||||||
((op->opcode == BS_OP_READ || op->opcode == BS_OP_WRITE || op->opcode == BS_OP_WRITE_STABLE) && (
|
((op->opcode == BS_OP_READ || op->opcode == BS_OP_WRITE || op->opcode == BS_OP_WRITE_STABLE) && (
|
||||||
op->offset >= dsk.data_block_size ||
|
op->offset >= dsk.data_block_size ||
|
||||||
op->len > dsk.data_block_size-op->offset ||
|
op->len > dsk.data_block_size-op->offset ||
|
||||||
(op->len % dsk.disk_alignment)
|
(op->offset % dsk.bitmap_granularity) ||
|
||||||
|
(op->len % dsk.bitmap_granularity)
|
||||||
)) ||
|
)) ||
|
||||||
readonly && op->opcode != BS_OP_READ && op->opcode != BS_OP_LIST)
|
readonly && op->opcode != BS_OP_READ && op->opcode != BS_OP_LIST)
|
||||||
{
|
{
|
||||||
@@ -361,75 +303,11 @@ void blockstore_impl_t::init_op(blockstore_op_t *op)
|
|||||||
{
|
{
|
||||||
// Call constructor without allocating memory. We'll call destructor before returning op back
|
// Call constructor without allocating memory. We'll call destructor before returning op back
|
||||||
new ((void*)op->private_data) blockstore_op_private_t;
|
new ((void*)op->private_data) blockstore_op_private_t;
|
||||||
PRIV(op)->min_flushed_journal_sector = PRIV(op)->max_flushed_journal_sector = 0;
|
|
||||||
PRIV(op)->wait_for = 0;
|
PRIV(op)->wait_for = 0;
|
||||||
PRIV(op)->op_state = 0;
|
PRIV(op)->op_state = 0;
|
||||||
PRIV(op)->pending_ops = 0;
|
PRIV(op)->pending_ops = 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
static bool replace_stable(object_id oid, uint64_t version, int search_start, int search_end, obj_ver_id* list)
|
|
||||||
{
|
|
||||||
while (search_start < search_end)
|
|
||||||
{
|
|
||||||
int pos = search_start+(search_end-search_start)/2;
|
|
||||||
if (oid < list[pos].oid)
|
|
||||||
{
|
|
||||||
search_end = pos;
|
|
||||||
}
|
|
||||||
else if (list[pos].oid < oid)
|
|
||||||
{
|
|
||||||
search_start = pos+1;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
list[pos].version = version;
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
blockstore_clean_db_t& blockstore_impl_t::clean_db_shard(object_id oid)
|
|
||||||
{
|
|
||||||
uint64_t pg_num = 0;
|
|
||||||
uint64_t pool_id = (oid.inode >> (64-POOL_ID_BITS));
|
|
||||||
auto sh_it = clean_db_settings.find(pool_id);
|
|
||||||
if (sh_it != clean_db_settings.end())
|
|
||||||
{
|
|
||||||
// like map_to_pg()
|
|
||||||
pg_num = (oid.stripe / sh_it->second.pg_stripe_size) % sh_it->second.pg_count + 1;
|
|
||||||
}
|
|
||||||
return clean_db_shards[(pool_id << (64-POOL_ID_BITS)) | pg_num];
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_impl_t::reshard_clean_db(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size)
|
|
||||||
{
|
|
||||||
uint64_t pool_id = (uint64_t)pool;
|
|
||||||
std::map<pool_pg_id_t, blockstore_clean_db_t> new_shards;
|
|
||||||
auto sh_it = clean_db_shards.lower_bound((pool_id << (64-POOL_ID_BITS)));
|
|
||||||
while (sh_it != clean_db_shards.end() &&
|
|
||||||
(sh_it->first >> (64-POOL_ID_BITS)) == pool_id)
|
|
||||||
{
|
|
||||||
for (auto & pair: sh_it->second)
|
|
||||||
{
|
|
||||||
// like map_to_pg()
|
|
||||||
uint64_t pg_num = (pair.first.stripe / pg_stripe_size) % pg_count + 1;
|
|
||||||
uint64_t shard_id = (pool_id << (64-POOL_ID_BITS)) | pg_num;
|
|
||||||
new_shards[shard_id][pair.first] = pair.second;
|
|
||||||
}
|
|
||||||
clean_db_shards.erase(sh_it++);
|
|
||||||
}
|
|
||||||
for (sh_it = new_shards.begin(); sh_it != new_shards.end(); sh_it++)
|
|
||||||
{
|
|
||||||
auto & to = clean_db_shards[sh_it->first];
|
|
||||||
to.swap(sh_it->second);
|
|
||||||
}
|
|
||||||
clean_db_settings[pool_id] = (pool_shard_settings_t){
|
|
||||||
.pg_count = pg_count,
|
|
||||||
.pg_stripe_size = pg_stripe_size,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_impl_t::process_list(blockstore_op_t *op)
|
void blockstore_impl_t::process_list(blockstore_op_t *op)
|
||||||
{
|
{
|
||||||
uint32_t list_pg = op->pg_number+1;
|
uint32_t list_pg = op->pg_number+1;
|
||||||
@@ -438,258 +316,58 @@ void blockstore_impl_t::process_list(blockstore_op_t *op)
|
|||||||
uint64_t min_inode = op->min_oid.inode;
|
uint64_t min_inode = op->min_oid.inode;
|
||||||
uint64_t max_inode = op->max_oid.inode;
|
uint64_t max_inode = op->max_oid.inode;
|
||||||
// Check PG
|
// Check PG
|
||||||
if (pg_count != 0 && (pg_stripe_size < MIN_DATA_BLOCK_SIZE || list_pg > pg_count))
|
if (!pg_count || (pg_stripe_size < MIN_DATA_BLOCK_SIZE || list_pg > pg_count) ||
|
||||||
|
!INODE_POOL(min_inode) || INODE_POOL(min_inode) != INODE_POOL(max_inode))
|
||||||
{
|
{
|
||||||
op->retval = -EINVAL;
|
op->retval = -EINVAL;
|
||||||
FINISH_OP(op);
|
FINISH_OP(op);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
// Check if the DB needs resharding
|
// Check if the DB is sharded correctly
|
||||||
// (we don't know about PGs from the beginning, we only create "shards" here)
|
if (!heap->reshard_check(INODE_POOL(min_inode), pg_count, pg_stripe_size))
|
||||||
uint64_t first_shard = 0, last_shard = UINT64_MAX;
|
|
||||||
if (min_inode != 0 &&
|
|
||||||
// Check if min_inode == max_inode == pool_id<<N, i.e. this is a pool listing
|
|
||||||
(min_inode >> (64-POOL_ID_BITS)) == (max_inode >> (64-POOL_ID_BITS)))
|
|
||||||
{
|
{
|
||||||
pool_id_t pool_id = (min_inode >> (64-POOL_ID_BITS));
|
op->retval = -EAGAIN;
|
||||||
if (pg_count > 1)
|
|
||||||
{
|
|
||||||
// Per-pg listing
|
|
||||||
auto sh_it = clean_db_settings.find(pool_id);
|
|
||||||
if (sh_it == clean_db_settings.end() ||
|
|
||||||
sh_it->second.pg_count != pg_count ||
|
|
||||||
sh_it->second.pg_stripe_size != pg_stripe_size)
|
|
||||||
{
|
|
||||||
reshard_clean_db(pool_id, pg_count, pg_stripe_size);
|
|
||||||
}
|
|
||||||
first_shard = last_shard = ((uint64_t)pool_id << (64-POOL_ID_BITS)) | list_pg;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// Per-pool listing
|
|
||||||
first_shard = ((uint64_t)pool_id << (64-POOL_ID_BITS));
|
|
||||||
last_shard = ((uint64_t)(pool_id+1) << (64-POOL_ID_BITS)) - 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Copy clean_db entries
|
|
||||||
int stable_count = 0, stable_alloc = 0;
|
|
||||||
if (min_inode != max_inode)
|
|
||||||
{
|
|
||||||
for (auto shard_it = clean_db_shards.lower_bound(first_shard);
|
|
||||||
shard_it != clean_db_shards.end() && shard_it->first <= last_shard;
|
|
||||||
shard_it++)
|
|
||||||
{
|
|
||||||
auto & clean_db = shard_it->second;
|
|
||||||
stable_alloc += clean_db.size();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (op->list_stable_limit > 0)
|
|
||||||
{
|
|
||||||
stable_alloc = op->list_stable_limit;
|
|
||||||
if (stable_alloc > 1024*1024)
|
|
||||||
stable_alloc = 1024*1024;
|
|
||||||
}
|
|
||||||
if (stable_alloc < 32768)
|
|
||||||
{
|
|
||||||
stable_alloc = 32768;
|
|
||||||
}
|
|
||||||
obj_ver_id *stable = (obj_ver_id*)malloc(sizeof(obj_ver_id) * stable_alloc);
|
|
||||||
if (!stable)
|
|
||||||
{
|
|
||||||
op->retval = -ENOMEM;
|
|
||||||
FINISH_OP(op);
|
FINISH_OP(op);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
auto max_oid = op->max_oid;
|
obj_ver_id *result = NULL;
|
||||||
bool limited = false;
|
size_t stable_count = 0, unstable_count = 0;
|
||||||
pool_pg_id_t last_shard_id = 0;
|
int res = heap->list_objects(list_pg, op->min_oid, op->max_oid, &result, &stable_count, &unstable_count);
|
||||||
for (auto shard_it = clean_db_shards.lower_bound(first_shard);
|
if (op->list_stable_limit)
|
||||||
shard_it != clean_db_shards.end() && shard_it->first <= last_shard;
|
|
||||||
shard_it++)
|
|
||||||
{
|
{
|
||||||
auto & clean_db = shard_it->second;
|
// Ordered result is expected - used by scrub
|
||||||
auto clean_it = clean_db.begin(), clean_end = clean_db.end();
|
// We use an unordered map
|
||||||
if (op->min_oid.inode != 0 || op->min_oid.stripe != 0)
|
std::sort(result, result + stable_count);
|
||||||
|
if (stable_count > op->list_stable_limit)
|
||||||
{
|
{
|
||||||
clean_it = clean_db.lower_bound(op->min_oid);
|
memmove(result + op->list_stable_limit, result + stable_count, unstable_count);
|
||||||
}
|
stable_count = op->list_stable_limit;
|
||||||
if ((max_oid.inode != 0 || max_oid.stripe != 0) && !(max_oid < op->min_oid))
|
|
||||||
{
|
|
||||||
clean_end = clean_db.upper_bound(max_oid);
|
|
||||||
}
|
|
||||||
for (; clean_it != clean_end; clean_it++)
|
|
||||||
{
|
|
||||||
if (stable_count >= stable_alloc)
|
|
||||||
{
|
|
||||||
stable_alloc *= 2;
|
|
||||||
obj_ver_id* nst = (obj_ver_id*)realloc(stable, sizeof(obj_ver_id) * stable_alloc);
|
|
||||||
if (!nst)
|
|
||||||
{
|
|
||||||
op->retval = -ENOMEM;
|
|
||||||
FINISH_OP(op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
stable = nst;
|
|
||||||
}
|
|
||||||
stable[stable_count++] = {
|
|
||||||
.oid = clean_it->first,
|
|
||||||
.version = clean_it->second.version,
|
|
||||||
};
|
|
||||||
if (op->list_stable_limit > 0 && stable_count >= op->list_stable_limit)
|
|
||||||
{
|
|
||||||
if (!limited)
|
|
||||||
{
|
|
||||||
limited = true;
|
|
||||||
max_oid = stable[stable_count-1].oid;
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (op->list_stable_limit > 0)
|
|
||||||
{
|
|
||||||
// To maintain the order, we have to include objects in the same range from other shards
|
|
||||||
if (last_shard_id != 0 && last_shard_id != shard_it->first)
|
|
||||||
std::sort(stable, stable+stable_count);
|
|
||||||
if (stable_count > op->list_stable_limit)
|
|
||||||
stable_count = op->list_stable_limit;
|
|
||||||
}
|
|
||||||
last_shard_id = shard_it->first;
|
|
||||||
}
|
|
||||||
if (op->list_stable_limit == 0 && first_shard != last_shard)
|
|
||||||
{
|
|
||||||
// If that's not a per-PG listing, sort clean entries (already sorted if list_stable_limit != 0)
|
|
||||||
std::sort(stable, stable+stable_count);
|
|
||||||
}
|
|
||||||
int clean_stable_count = stable_count;
|
|
||||||
// Copy dirty_db entries (sorted, too)
|
|
||||||
int unstable_count = 0, unstable_alloc = 0;
|
|
||||||
obj_ver_id *unstable = NULL;
|
|
||||||
{
|
|
||||||
auto dirty_it = dirty_db.begin(), dirty_end = dirty_db.end();
|
|
||||||
if (op->min_oid.inode != 0 || op->min_oid.stripe != 0)
|
|
||||||
{
|
|
||||||
dirty_it = dirty_db.lower_bound({
|
|
||||||
.oid = op->min_oid,
|
|
||||||
.version = 0,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
if ((max_oid.inode != 0 || max_oid.stripe != 0) && !(max_oid < op->min_oid))
|
|
||||||
{
|
|
||||||
dirty_end = dirty_db.upper_bound({
|
|
||||||
.oid = max_oid,
|
|
||||||
.version = UINT64_MAX,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
for (; dirty_it != dirty_end; dirty_it++)
|
|
||||||
{
|
|
||||||
if (!pg_count || ((dirty_it->first.oid.stripe / pg_stripe_size) % pg_count + 1) == list_pg) // like map_to_pg()
|
|
||||||
{
|
|
||||||
if (IS_DELETE(dirty_it->second.state))
|
|
||||||
{
|
|
||||||
// Deletions are always stable, so try to zero out two possible entries
|
|
||||||
if (!replace_stable(dirty_it->first.oid, 0, 0, clean_stable_count, stable))
|
|
||||||
{
|
|
||||||
replace_stable(dirty_it->first.oid, 0, clean_stable_count, stable_count, stable);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else if (IS_STABLE(dirty_it->second.state) || (dirty_it->second.state & BS_ST_INSTANT))
|
|
||||||
{
|
|
||||||
// First try to replace a clean stable version in the first part of the list
|
|
||||||
if (!replace_stable(dirty_it->first.oid, dirty_it->first.version, 0, clean_stable_count, stable))
|
|
||||||
{
|
|
||||||
// Then try to replace the last dirty stable version in the second part of the list
|
|
||||||
if (stable_count > 0 && stable[stable_count-1].oid == dirty_it->first.oid)
|
|
||||||
{
|
|
||||||
stable[stable_count-1].version = dirty_it->first.version;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
if (stable_count >= stable_alloc)
|
|
||||||
{
|
|
||||||
stable_alloc += 32768;
|
|
||||||
obj_ver_id *nst = (obj_ver_id*)realloc(stable, sizeof(obj_ver_id) * stable_alloc);
|
|
||||||
if (!nst)
|
|
||||||
{
|
|
||||||
if (unstable)
|
|
||||||
free(unstable);
|
|
||||||
op->retval = -ENOMEM;
|
|
||||||
FINISH_OP(op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
stable = nst;
|
|
||||||
}
|
|
||||||
stable[stable_count++] = dirty_it->first;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (op->list_stable_limit > 0 && stable_count >= op->list_stable_limit)
|
|
||||||
{
|
|
||||||
// Stop here
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
if (unstable_count >= unstable_alloc)
|
|
||||||
{
|
|
||||||
unstable_alloc += 32768;
|
|
||||||
obj_ver_id *nst = (obj_ver_id*)realloc(unstable, sizeof(obj_ver_id) * unstable_alloc);
|
|
||||||
if (!nst)
|
|
||||||
{
|
|
||||||
if (stable)
|
|
||||||
free(stable);
|
|
||||||
op->retval = -ENOMEM;
|
|
||||||
FINISH_OP(op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
unstable = nst;
|
|
||||||
}
|
|
||||||
unstable[unstable_count++] = dirty_it->first;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Remove zeroed out stable entries
|
|
||||||
int j = 0;
|
|
||||||
for (int i = 0; i < stable_count; i++)
|
|
||||||
{
|
|
||||||
if (stable[i].version != 0)
|
|
||||||
{
|
|
||||||
stable[j++] = stable[i];
|
|
||||||
}
|
|
||||||
}
|
|
||||||
stable_count = j;
|
|
||||||
if (stable_count+unstable_count > stable_alloc)
|
|
||||||
{
|
|
||||||
stable_alloc = stable_count+unstable_count;
|
|
||||||
obj_ver_id *nst = (obj_ver_id*)realloc(stable, sizeof(obj_ver_id) * stable_alloc);
|
|
||||||
if (!nst)
|
|
||||||
{
|
|
||||||
if (unstable)
|
|
||||||
free(unstable);
|
|
||||||
op->retval = -ENOMEM;
|
|
||||||
FINISH_OP(op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
stable = nst;
|
|
||||||
}
|
|
||||||
// Copy unstable entries
|
|
||||||
for (int i = 0; i < unstable_count; i++)
|
|
||||||
{
|
|
||||||
stable[j++] = unstable[i];
|
|
||||||
}
|
|
||||||
free(unstable);
|
|
||||||
op->version = stable_count;
|
op->version = stable_count;
|
||||||
op->retval = stable_count+unstable_count;
|
op->retval = res == 0 ? stable_count+unstable_count : -res;
|
||||||
op->buf = stable;
|
op->buf = (uint8_t*)result;
|
||||||
FINISH_OP(op);
|
FINISH_OP(op);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void blockstore_impl_t::set_no_inode_stats(const std::vector<uint64_t> & pool_ids)
|
||||||
|
{
|
||||||
|
heap->set_no_inode_stats(pool_ids);
|
||||||
|
}
|
||||||
|
|
||||||
void blockstore_impl_t::dump_diagnostics()
|
void blockstore_impl_t::dump_diagnostics()
|
||||||
{
|
{
|
||||||
journal.dump_diagnostics();
|
|
||||||
flusher->dump_diagnostics();
|
flusher->dump_diagnostics();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void blockstore_meta_header_v3_t::set_crc32c()
|
||||||
|
{
|
||||||
|
header_csum = 0;
|
||||||
|
uint32_t calc = crc32c(0, this, version == BLOCKSTORE_META_FORMAT_HEAP
|
||||||
|
? sizeof(blockstore_meta_header_v3_t) : sizeof(blockstore_meta_header_v2_t));
|
||||||
|
header_csum = calc;
|
||||||
|
}
|
||||||
|
|
||||||
void blockstore_impl_t::disk_error_abort(const char *op, int retval, int expected)
|
void blockstore_impl_t::disk_error_abort(const char *op, int retval, int expected)
|
||||||
{
|
{
|
||||||
if (retval == -EAGAIN)
|
if (retval == -EAGAIN)
|
||||||
@@ -703,85 +381,28 @@ void blockstore_impl_t::disk_error_abort(const char *op, int retval, int expecte
|
|||||||
exit(1);
|
exit(1);
|
||||||
}
|
}
|
||||||
|
|
||||||
void blockstore_impl_t::set_no_inode_stats(const std::vector<uint64_t> & pool_ids)
|
uint64_t blockstore_impl_t::get_free_block_count()
|
||||||
{
|
{
|
||||||
for (auto & np: no_inode_stats)
|
return dsk.block_count - heap->get_data_used_space()/dsk.data_block_size;
|
||||||
{
|
|
||||||
np.second = 2;
|
|
||||||
}
|
|
||||||
for (auto pool_id: pool_ids)
|
|
||||||
{
|
|
||||||
if (!no_inode_stats[pool_id])
|
|
||||||
recalc_inode_space_stats(pool_id, false);
|
|
||||||
no_inode_stats[pool_id] = 1;
|
|
||||||
}
|
|
||||||
for (auto np_it = no_inode_stats.begin(); np_it != no_inode_stats.end(); )
|
|
||||||
{
|
|
||||||
if (np_it->second == 2)
|
|
||||||
{
|
|
||||||
recalc_inode_space_stats(np_it->first, true);
|
|
||||||
no_inode_stats.erase(np_it++);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
np_it++;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
void blockstore_impl_t::recalc_inode_space_stats(uint64_t pool_id, bool per_inode)
|
std::string blockstore_impl_t::get_op_diag(blockstore_op_t *op)
|
||||||
{
|
{
|
||||||
auto sp_begin = inode_space_stats.lower_bound((pool_id << (64-POOL_ID_BITS)));
|
char buf[256];
|
||||||
auto sp_end = inode_space_stats.lower_bound(((pool_id+1) << (64-POOL_ID_BITS)));
|
auto priv = PRIV(op);
|
||||||
inode_space_stats.erase(sp_begin, sp_end);
|
if (priv->wait_for)
|
||||||
auto sh_it = clean_db_shards.lower_bound((pool_id << (64-POOL_ID_BITS)));
|
snprintf(buf, sizeof(buf), "state=%d wait=%d (detail=%ju)", priv->op_state, priv->wait_for, priv->wait_detail);
|
||||||
while (sh_it != clean_db_shards.end() &&
|
else
|
||||||
(sh_it->first >> (64-POOL_ID_BITS)) == pool_id)
|
snprintf(buf, sizeof(buf), "state=%d", priv->op_state);
|
||||||
{
|
return std::string(buf);
|
||||||
for (auto & pair: sh_it->second)
|
}
|
||||||
{
|
|
||||||
uint64_t space_id = per_inode ? pair.first.inode : (pool_id << (64-POOL_ID_BITS));
|
void* blockstore_impl_t::reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit)
|
||||||
inode_space_stats[space_id] += dsk.data_block_size;
|
{
|
||||||
}
|
return heap->reshard_start(pool, pg_count, pg_stripe_size, chunk_limit);
|
||||||
sh_it++;
|
}
|
||||||
}
|
|
||||||
object_id last_oid = {};
|
bool blockstore_impl_t::reshard_continue(void *reshard_state, uint64_t chunk_limit)
|
||||||
bool last_exists = false;
|
{
|
||||||
auto dirty_it = dirty_db.lower_bound((obj_ver_id){ .oid = { .inode = (pool_id << (64-POOL_ID_BITS)) } });
|
return heap->reshard_continue(reshard_state, chunk_limit);
|
||||||
while (dirty_it != dirty_db.end() && (dirty_it->first.oid.inode >> (64-POOL_ID_BITS)) == pool_id)
|
|
||||||
{
|
|
||||||
if (IS_STABLE(dirty_it->second.state) && (IS_BIG_WRITE(dirty_it->second.state) || IS_DELETE(dirty_it->second.state)))
|
|
||||||
{
|
|
||||||
bool exists = false;
|
|
||||||
if (last_oid == dirty_it->first.oid)
|
|
||||||
{
|
|
||||||
exists = last_exists;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
auto & clean_db = clean_db_shard(dirty_it->first.oid);
|
|
||||||
auto clean_it = clean_db.find(dirty_it->first.oid);
|
|
||||||
exists = clean_it != clean_db.end();
|
|
||||||
}
|
|
||||||
uint64_t space_id = per_inode ? dirty_it->first.oid.inode : (pool_id << (64-POOL_ID_BITS));
|
|
||||||
if (IS_BIG_WRITE(dirty_it->second.state))
|
|
||||||
{
|
|
||||||
if (!exists)
|
|
||||||
inode_space_stats[space_id] += dsk.data_block_size;
|
|
||||||
last_exists = true;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
if (exists)
|
|
||||||
{
|
|
||||||
auto & sp = inode_space_stats[space_id];
|
|
||||||
if (sp > dsk.data_block_size)
|
|
||||||
sp -= dsk.data_block_size;
|
|
||||||
else
|
|
||||||
inode_space_stats.erase(space_id);
|
|
||||||
}
|
|
||||||
last_exists = false;
|
|
||||||
}
|
|
||||||
last_oid = dirty_it->first.oid;
|
|
||||||
}
|
|
||||||
dirty_it++;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -5,6 +5,8 @@
|
|||||||
|
|
||||||
#include "blockstore.h"
|
#include "blockstore.h"
|
||||||
#include "blockstore_disk.h"
|
#include "blockstore_disk.h"
|
||||||
|
#include "blockstore_heap.h"
|
||||||
|
#include "ondisk_formats.h"
|
||||||
|
|
||||||
#include <sys/types.h>
|
#include <sys/types.h>
|
||||||
#include <sys/ioctl.h>
|
#include <sys/ioctl.h>
|
||||||
@@ -21,240 +23,67 @@
|
|||||||
#include <unordered_map>
|
#include <unordered_map>
|
||||||
#include <unordered_set>
|
#include <unordered_set>
|
||||||
|
|
||||||
#include "cpp-btree/btree_map.h"
|
|
||||||
|
|
||||||
#include "malloc_or_die.h"
|
#include "malloc_or_die.h"
|
||||||
#include "allocator.h"
|
|
||||||
|
class blockstore_impl_t;
|
||||||
|
|
||||||
//#define BLOCKSTORE_DEBUG
|
//#define BLOCKSTORE_DEBUG
|
||||||
|
|
||||||
// States are not stored on disk. Instead, they're deduced from the journal
|
|
||||||
|
|
||||||
#define BS_ST_SMALL_WRITE 0x01
|
|
||||||
#define BS_ST_BIG_WRITE 0x02
|
|
||||||
#define BS_ST_DELETE 0x03
|
|
||||||
|
|
||||||
#define BS_ST_WAIT_DEL 0x10
|
|
||||||
#define BS_ST_WAIT_BIG 0x20
|
|
||||||
#define BS_ST_IN_FLIGHT 0x30
|
|
||||||
#define BS_ST_SUBMITTED 0x40
|
|
||||||
#define BS_ST_WRITTEN 0x50
|
|
||||||
#define BS_ST_SYNCED 0x60
|
|
||||||
#define BS_ST_STABLE 0x70
|
|
||||||
|
|
||||||
#define BS_ST_INSTANT 0x100
|
|
||||||
|
|
||||||
#define IMMEDIATE_NONE 0
|
|
||||||
#define IMMEDIATE_SMALL 1
|
|
||||||
#define IMMEDIATE_ALL 2
|
|
||||||
|
|
||||||
#define BS_ST_TYPE_MASK 0x0F
|
|
||||||
#define BS_ST_WORKFLOW_MASK 0xF0
|
|
||||||
#define IS_IN_FLIGHT(st) (((st) & 0xF0) <= BS_ST_SUBMITTED)
|
|
||||||
#define IS_STABLE(st) (((st) & 0xF0) == BS_ST_STABLE)
|
|
||||||
#define IS_SYNCED(st) (((st) & 0xF0) >= BS_ST_SYNCED)
|
|
||||||
#define IS_JOURNAL(st) (((st) & 0x0F) == BS_ST_SMALL_WRITE)
|
|
||||||
#define IS_BIG_WRITE(st) (((st) & 0x0F) == BS_ST_BIG_WRITE)
|
|
||||||
#define IS_DELETE(st) (((st) & 0x0F) == BS_ST_DELETE)
|
|
||||||
#define IS_INSTANT(st) (((st) & BS_ST_TYPE_MASK) == BS_ST_DELETE || ((st) & BS_ST_INSTANT))
|
|
||||||
|
|
||||||
#define BS_SUBMIT_CHECK_SQES(n) \
|
|
||||||
if (ringloop->sqes_left() < (n))\
|
|
||||||
{\
|
|
||||||
/* Pause until there are more requests available */\
|
|
||||||
PRIV(op)->wait_detail = (n);\
|
|
||||||
PRIV(op)->wait_for = WAIT_SQE;\
|
|
||||||
return 0;\
|
|
||||||
}
|
|
||||||
|
|
||||||
#define BS_SUBMIT_GET_SQE(sqe, data) \
|
|
||||||
BS_SUBMIT_GET_ONLY_SQE(sqe); \
|
|
||||||
struct ring_data_t *data = ((ring_data_t*)sqe->user_data)
|
|
||||||
|
|
||||||
#define BS_SUBMIT_GET_ONLY_SQE(sqe) \
|
|
||||||
struct io_uring_sqe *sqe = get_sqe();\
|
|
||||||
if (!sqe)\
|
|
||||||
{\
|
|
||||||
/* Pause until there are more requests available */\
|
|
||||||
PRIV(op)->wait_detail = 1;\
|
|
||||||
PRIV(op)->wait_for = WAIT_SQE;\
|
|
||||||
return 0;\
|
|
||||||
}
|
|
||||||
|
|
||||||
#define BS_SUBMIT_GET_SQE_DECL(sqe) \
|
|
||||||
sqe = get_sqe();\
|
|
||||||
if (!sqe)\
|
|
||||||
{\
|
|
||||||
/* Pause until there are more requests available */\
|
|
||||||
PRIV(op)->wait_detail = 1;\
|
|
||||||
PRIV(op)->wait_for = WAIT_SQE;\
|
|
||||||
return 0;\
|
|
||||||
}
|
|
||||||
|
|
||||||
#include "blockstore_journal.h"
|
|
||||||
|
|
||||||
// "VITAstor"
|
|
||||||
#define BLOCKSTORE_META_MAGIC_V1 0x726F747341544956l
|
|
||||||
#define BLOCKSTORE_META_FORMAT_V1 1
|
|
||||||
#define BLOCKSTORE_META_FORMAT_V2 2
|
|
||||||
|
|
||||||
// metadata header (superblock)
|
|
||||||
struct __attribute__((__packed__)) blockstore_meta_header_v1_t
|
|
||||||
{
|
|
||||||
uint64_t zero;
|
|
||||||
uint64_t magic;
|
|
||||||
uint64_t version;
|
|
||||||
uint32_t meta_block_size;
|
|
||||||
uint32_t data_block_size;
|
|
||||||
uint32_t bitmap_granularity;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct __attribute__((__packed__)) blockstore_meta_header_v2_t
|
|
||||||
{
|
|
||||||
uint64_t zero;
|
|
||||||
uint64_t magic;
|
|
||||||
uint64_t version;
|
|
||||||
uint32_t meta_block_size;
|
|
||||||
uint32_t data_block_size;
|
|
||||||
uint32_t bitmap_granularity;
|
|
||||||
uint32_t data_csum_type;
|
|
||||||
uint32_t csum_block_size;
|
|
||||||
uint32_t header_csum;
|
|
||||||
};
|
|
||||||
|
|
||||||
// 32 bytes = 24 bytes + block bitmap (4 bytes by default) + external attributes (also bitmap, 4 bytes by default)
|
|
||||||
// per "clean" entry on disk with fixed metadata tables
|
|
||||||
struct __attribute__((__packed__)) clean_disk_entry
|
|
||||||
{
|
|
||||||
object_id oid;
|
|
||||||
uint64_t version;
|
|
||||||
uint8_t bitmap[];
|
|
||||||
// Two more fields come after bitmap in metadata version 2:
|
|
||||||
// uint32_t data_csum[];
|
|
||||||
// uint32_t entry_csum;
|
|
||||||
};
|
|
||||||
|
|
||||||
// 32 = 16 + 16 bytes per "clean" entry in memory (object_id => clean_entry)
|
|
||||||
struct __attribute__((__packed__)) clean_entry
|
|
||||||
{
|
|
||||||
uint64_t version;
|
|
||||||
uint64_t location;
|
|
||||||
};
|
|
||||||
|
|
||||||
// 64 = 24 + 40 bytes per dirty entry in memory (obj_ver_id => dirty_entry). Plus checksums
|
|
||||||
struct __attribute__((__packed__)) dirty_entry
|
|
||||||
{
|
|
||||||
uint32_t state;
|
|
||||||
uint32_t flags; // unneeded, but present for alignment
|
|
||||||
uint64_t location; // location in either journal or data -> in BYTES
|
|
||||||
uint32_t offset; // data offset within object (stripe)
|
|
||||||
uint32_t len; // data length
|
|
||||||
uint64_t journal_sector; // journal sector used for this entry
|
|
||||||
void* dyn_data; // dynamic data: external bitmap and data block checksums. may be a pointer to the in-memory journal
|
|
||||||
};
|
|
||||||
|
|
||||||
// - Sync must be submitted after previous writes/deletes (not before!)
|
|
||||||
// - Reads to the same object must be submitted after previous writes/deletes
|
|
||||||
// are written (not necessarily synced) in their location. This is because we
|
|
||||||
// rely on read-modify-write for erasure coding and we must return new data
|
|
||||||
// to calculate parity for subsequent writes
|
|
||||||
// - Writes may be submitted in any order, because they don't overlap. Each write
|
|
||||||
// goes into a new location - either on the journal device or on the data device
|
|
||||||
// - Stable (stabilize) must be submitted after sync of that object is completed
|
|
||||||
// It's even OK to return an error to the caller if that object is not synced yet
|
|
||||||
// - Journal trim may be processed only after all versions are moved to
|
|
||||||
// the main storage AND after all read operations for older versions complete
|
|
||||||
// - If an operation can not be submitted because the ring is full
|
|
||||||
// we should stop submission of other operations. Otherwise some "scatter" reads
|
|
||||||
// may end up blocked for a long time.
|
|
||||||
// Otherwise, the submit order is free, that is all operations may be submitted immediately
|
|
||||||
// In fact, adding a write operation must immediately result in dirty_db being populated
|
|
||||||
|
|
||||||
// Suspend operation until there are more free SQEs
|
|
||||||
#define WAIT_SQE 1
|
|
||||||
// Suspend operation until there are <wait_detail> bytes of free space in the journal on disk
|
|
||||||
#define WAIT_JOURNAL 3
|
|
||||||
// Suspend operation until the next journal sector buffer is free
|
|
||||||
#define WAIT_JOURNAL_BUFFER 4
|
|
||||||
// Suspend operation until there is some free space on the data device
|
|
||||||
#define WAIT_FREE 5
|
|
||||||
|
|
||||||
struct used_clean_obj_t
|
|
||||||
{
|
|
||||||
int refs;
|
|
||||||
bool was_freed; // was freed by a parallel flush?
|
|
||||||
bool was_changed; // was changed by a parallel flush?
|
|
||||||
};
|
|
||||||
|
|
||||||
// https://github.com/algorithm-ninja/cpp-btree
|
|
||||||
// https://github.com/greg7mdp/sparsepp/ was used previously, but it was TERRIBLY slow after resizing
|
|
||||||
// with sparsepp, random reads dropped to ~700 iops very fast with just as much as ~32k objects in the DB
|
|
||||||
typedef btree::btree_map<object_id, clean_entry> blockstore_clean_db_t;
|
|
||||||
typedef std::map<obj_ver_id, dirty_entry> blockstore_dirty_db_t;
|
|
||||||
|
|
||||||
#include "blockstore_init.h"
|
#include "blockstore_init.h"
|
||||||
|
|
||||||
#include "blockstore_flush.h"
|
#include "blockstore_flush.h"
|
||||||
|
|
||||||
#define PRIV(op) ((blockstore_op_private_t*)(op)->private_data)
|
|
||||||
#define FINISH_OP(op) PRIV(op)->~blockstore_op_private_t(); std::function<void (blockstore_op_t*)>(op->callback)(op)
|
|
||||||
|
|
||||||
struct blockstore_op_private_t
|
struct blockstore_op_private_t
|
||||||
{
|
{
|
||||||
// Wait status
|
// Wait status
|
||||||
int wait_for;
|
int wait_for;
|
||||||
uint64_t wait_detail, wait_detail2;
|
uint64_t wait_detail;
|
||||||
int pending_ops;
|
int pending_ops;
|
||||||
int op_state;
|
int op_state;
|
||||||
|
|
||||||
|
// Write, sync, stabilize
|
||||||
|
uint32_t modified_block, modified_block2;
|
||||||
|
|
||||||
// Read
|
// Read
|
||||||
uint64_t clean_block_used;
|
|
||||||
std::vector<copy_buffer_t> read_vec;
|
std::vector<copy_buffer_t> read_vec;
|
||||||
|
|
||||||
// Sync, write
|
// Read, write
|
||||||
uint64_t min_flushed_journal_sector, max_flushed_journal_sector;
|
uint64_t lsn;
|
||||||
|
|
||||||
|
// Write
|
||||||
|
uint64_t location;
|
||||||
|
uint32_t write_type;
|
||||||
|
|
||||||
|
// Stabilize, rollback
|
||||||
|
int stab_pos;
|
||||||
|
|
||||||
// Write
|
// Write
|
||||||
struct iovec iov_zerofill[3];
|
|
||||||
// Warning: must not have a default value here because it's written to before calling constructor in blockstore_write.cpp O_o
|
|
||||||
uint64_t real_version;
|
|
||||||
timespec tv_begin;
|
timespec tv_begin;
|
||||||
|
|
||||||
// Sync
|
|
||||||
std::vector<obj_ver_id> sync_big_writes, sync_small_writes;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef uint32_t pool_id_t;
|
struct bs_modified_block_t
|
||||||
typedef uint64_t pool_pg_id_t;
|
|
||||||
|
|
||||||
#define POOL_ID_BITS 16
|
|
||||||
|
|
||||||
struct pool_shard_settings_t
|
|
||||||
{
|
{
|
||||||
uint32_t pg_count;
|
bool sent;
|
||||||
uint32_t pg_stripe_size;
|
uint8_t *buf;
|
||||||
};
|
};
|
||||||
|
|
||||||
#define STAB_SPLIT_DONE 1
|
class blockstore_impl_t: public blockstore_i
|
||||||
#define STAB_SPLIT_WAIT 2
|
|
||||||
#define STAB_SPLIT_SYNC 3
|
|
||||||
#define STAB_SPLIT_TODO 4
|
|
||||||
|
|
||||||
class blockstore_impl_t
|
|
||||||
{
|
{
|
||||||
|
public:
|
||||||
blockstore_disk_t dsk;
|
blockstore_disk_t dsk;
|
||||||
|
|
||||||
/******* OPTIONS *******/
|
/******* OPTIONS *******/
|
||||||
bool readonly = false;
|
bool readonly = false;
|
||||||
// It is safe to disable fsync() if drive write cache is writethrough
|
|
||||||
bool disable_data_fsync = false, disable_meta_fsync = false, disable_journal_fsync = false;
|
|
||||||
// Enable if you want every operation to be executed with an "implicit fsync"
|
// Enable if you want every operation to be executed with an "implicit fsync"
|
||||||
// Suitable only for server SSDs with capacitors, requires disabled data and journal fsyncs
|
// Suitable only for server SSDs with capacitors, requires disabled data and journal fsyncs
|
||||||
int immediate_commit = IMMEDIATE_NONE;
|
int immediate_commit = IMMEDIATE_NONE;
|
||||||
bool inmemory_meta = false;
|
bool inmemory_meta = false;
|
||||||
|
bool skip_corrupted_meta_entries = false;
|
||||||
|
uint32_t meta_write_recheck_parallelism = 0;
|
||||||
// Maximum and minimum flusher count
|
// Maximum and minimum flusher count
|
||||||
unsigned max_flusher_count, min_flusher_count;
|
unsigned max_flusher_count = 0, min_flusher_count = 0;
|
||||||
unsigned journal_trim_interval;
|
unsigned journal_trim_interval = 0;
|
||||||
|
unsigned flusher_start_threshold = 0;
|
||||||
// Maximum queue depth
|
// Maximum queue depth
|
||||||
unsigned max_write_iodepth = 128;
|
unsigned max_write_iodepth = 128;
|
||||||
// Enable small (journaled) write throttling, useful for the SSD+HDD case
|
// Enable small (journaled) write throttling, useful for the SSD+HDD case
|
||||||
@@ -269,143 +98,101 @@ class blockstore_impl_t
|
|||||||
uint64_t autosync_writes = 128;
|
uint64_t autosync_writes = 128;
|
||||||
// Log level (0-10)
|
// Log level (0-10)
|
||||||
int log_level = 0;
|
int log_level = 0;
|
||||||
|
// Enable correct block checksum validation on objects updated with small writes when checksum block
|
||||||
|
// is larger than bitmap_granularity, at the expense of extra metadata fsyncs during compaction
|
||||||
|
bool perfect_csum_update = false;
|
||||||
/******* END OF OPTIONS *******/
|
/******* END OF OPTIONS *******/
|
||||||
|
|
||||||
struct ring_consumer_t ring_consumer;
|
struct ring_consumer_t ring_consumer;
|
||||||
|
|
||||||
std::map<pool_id_t, pool_shard_settings_t> clean_db_settings;
|
blockstore_heap_t *heap = NULL;
|
||||||
std::map<pool_pg_id_t, blockstore_clean_db_t> clean_db_shards;
|
uint8_t* meta_superblock = NULL;
|
||||||
std::map<uint64_t, int> no_inode_stats;
|
uint8_t *buffer_area = NULL;
|
||||||
uint8_t *clean_bitmaps = NULL;
|
|
||||||
blockstore_dirty_db_t dirty_db;
|
|
||||||
std::vector<blockstore_op_t*> submit_queue;
|
std::vector<blockstore_op_t*> submit_queue;
|
||||||
std::vector<obj_ver_id> unsynced_big_writes, unsynced_small_writes;
|
int unsynced_data_write_count = 0, unsynced_buffer_write_count = 0, unsynced_meta_write_count = 0;
|
||||||
int unsynced_big_write_count = 0, unstable_unsynced = 0;
|
|
||||||
int unsynced_queued_ops = 0;
|
int unsynced_queued_ops = 0;
|
||||||
allocator_t *data_alloc = NULL;
|
|
||||||
uint64_t used_blocks = 0;
|
|
||||||
uint8_t *zero_object = NULL;
|
|
||||||
|
|
||||||
void *metadata_buffer = NULL;
|
std::vector<uint32_t> pending_modified_blocks;
|
||||||
|
robin_hood::unordered_flat_map<uint32_t, bs_modified_block_t> modified_blocks;
|
||||||
|
|
||||||
struct journal_t journal;
|
|
||||||
journal_flusher_t *flusher;
|
journal_flusher_t *flusher;
|
||||||
int big_to_flush = 0;
|
|
||||||
int write_iodepth = 0;
|
int write_iodepth = 0;
|
||||||
bool alloc_dyn_data = false;
|
int inflight_big = 0;
|
||||||
|
int intent_write_counter = 0;
|
||||||
// clean data blocks referenced by read operations
|
bool fsyncing_data = false;
|
||||||
std::map<uint64_t, used_clean_obj_t> used_clean_objects;
|
|
||||||
|
|
||||||
bool live = false, queue_stall = false;
|
bool live = false, queue_stall = false;
|
||||||
ring_loop_t *ringloop;
|
ring_loop_i *ringloop = NULL;
|
||||||
timerfd_manager_t *tfd;
|
timerfd_manager_t *tfd = NULL;
|
||||||
|
|
||||||
bool stop_sync_submitted;
|
bool stop_sync_submitted = false;
|
||||||
|
|
||||||
inline struct io_uring_sqe* get_sqe()
|
inline struct io_uring_sqe* get_sqe()
|
||||||
{
|
{
|
||||||
return ringloop->get_sqe();
|
return ringloop->get_sqe();
|
||||||
}
|
}
|
||||||
|
|
||||||
friend class blockstore_init_meta;
|
|
||||||
friend class blockstore_init_journal;
|
|
||||||
friend struct blockstore_journal_check_t;
|
|
||||||
friend class journal_flusher_t;
|
|
||||||
friend class journal_flusher_co;
|
|
||||||
|
|
||||||
void calc_lengths();
|
|
||||||
void open_data();
|
void open_data();
|
||||||
void open_meta();
|
void open_meta();
|
||||||
void open_journal();
|
void open_journal();
|
||||||
uint8_t* get_clean_entry_bitmap(uint64_t block_loc, int offset);
|
|
||||||
|
|
||||||
blockstore_clean_db_t& clean_db_shard(object_id oid);
|
|
||||||
void reshard_clean_db(pool_id_t pool_id, uint32_t pg_count, uint32_t pg_stripe_size);
|
|
||||||
void recalc_inode_space_stats(uint64_t pool_id, bool per_inode);
|
|
||||||
|
|
||||||
// Journaling
|
|
||||||
void prepare_journal_sector_write(int sector, blockstore_op_t *op);
|
|
||||||
void handle_journal_write(ring_data_t *data, uint64_t flush_id);
|
|
||||||
void disk_error_abort(const char *op, int retval, int expected);
|
void disk_error_abort(const char *op, int retval, int expected);
|
||||||
|
|
||||||
// Asynchronous init
|
// Asynchronous init
|
||||||
int initialized;
|
int initialized;
|
||||||
int metadata_buf_size;
|
int metadata_buf_size;
|
||||||
blockstore_init_meta* metadata_init_reader;
|
blockstore_init_meta* metadata_init_reader;
|
||||||
blockstore_init_journal* journal_init_reader;
|
|
||||||
|
|
||||||
void check_wait(blockstore_op_t *op);
|
void check_wait(blockstore_op_t *op);
|
||||||
void init_op(blockstore_op_t *op);
|
void init_op(blockstore_op_t *op);
|
||||||
|
|
||||||
// Read
|
// Read
|
||||||
int dequeue_read(blockstore_op_t *read_op);
|
int dequeue_read(blockstore_op_t *op);
|
||||||
|
int fulfill_read(blockstore_op_t *op);
|
||||||
|
uint32_t prepare_read(std::vector<copy_buffer_t> & read_vec, heap_entry_t *obj, heap_entry_t *wr, uint32_t start, uint32_t end, uint32_t skip_csum);
|
||||||
|
uint32_t prepare_read_with_bitmaps(std::vector<copy_buffer_t> & read_vec, heap_entry_t *obj, heap_entry_t *wr, uint32_t start, uint32_t end, uint32_t skip_csum);
|
||||||
|
uint32_t prepare_read_zero(std::vector<copy_buffer_t> & read_vec, uint32_t start, uint32_t end);
|
||||||
|
uint32_t prepare_read_simple(std::vector<copy_buffer_t> & read_vec, heap_entry_t *obj, heap_entry_t *wr, uint32_t start, uint32_t end, uint32_t skip_csum);
|
||||||
|
void prepare_disk_read(std::vector<copy_buffer_t> & read_vec, int pos, heap_entry_t *obj, heap_entry_t *wr,
|
||||||
|
uint32_t blk_start, uint32_t blk_end, uint32_t start, uint32_t end, uint32_t copy_flags);
|
||||||
void find_holes(std::vector<copy_buffer_t> & read_vec, uint32_t item_start, uint32_t item_end,
|
void find_holes(std::vector<copy_buffer_t> & read_vec, uint32_t item_start, uint32_t item_end,
|
||||||
std::function<int(int, bool, uint32_t, uint32_t)> callback);
|
std::function<void(int&, uint32_t, uint32_t)> callback);
|
||||||
int fulfill_read(blockstore_op_t *read_op,
|
void free_read_buffers(std::vector<copy_buffer_t> & rv);
|
||||||
uint64_t &fulfilled, uint32_t item_start, uint32_t item_end,
|
|
||||||
uint32_t item_state, uint64_t item_version, uint64_t item_location,
|
|
||||||
uint64_t journal_sector, uint8_t *csum, int *dyn_data);
|
|
||||||
bool fulfill_clean_read(blockstore_op_t *read_op, uint64_t & fulfilled,
|
|
||||||
uint8_t *clean_entry_bitmap, int *dyn_data,
|
|
||||||
uint32_t item_start, uint32_t item_end, uint64_t clean_loc, uint64_t clean_ver);
|
|
||||||
int fill_partial_checksum_blocks(std::vector<copy_buffer_t> & rv, uint64_t & fulfilled,
|
|
||||||
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf, uint64_t read_offset, uint64_t read_end);
|
|
||||||
int pad_journal_read(std::vector<copy_buffer_t> & rv, copy_buffer_t & cp,
|
|
||||||
uint64_t dirty_offset, uint64_t dirty_end, uint64_t dirty_loc, uint8_t *csum_ptr, int *dyn_data,
|
|
||||||
uint64_t offset, uint64_t submit_len, uint64_t & blk_begin, uint64_t & blk_end, uint8_t* & blk_buf);
|
|
||||||
bool read_range_fulfilled(std::vector<copy_buffer_t> & rv, uint64_t & fulfilled, uint8_t *read_buf,
|
|
||||||
uint8_t *clean_entry_bitmap, uint32_t item_start, uint32_t item_end);
|
|
||||||
bool read_checksum_block(blockstore_op_t *op, int rv_pos, uint64_t &fulfilled, uint64_t clean_loc);
|
|
||||||
uint8_t* read_clean_meta_block(blockstore_op_t *read_op, uint64_t clean_loc, int rv_pos);
|
|
||||||
bool verify_padded_checksums(uint8_t *clean_entry_bitmap, uint8_t *csum_buf, uint32_t offset,
|
|
||||||
iovec *iov, int n_iov, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb);
|
|
||||||
bool verify_journal_checksums(uint8_t *csums, uint32_t offset,
|
|
||||||
iovec *iov, int n_iov, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb);
|
|
||||||
bool verify_clean_padded_checksums(blockstore_op_t *op, uint64_t clean_loc, uint8_t *dyn_data, bool from_journal,
|
|
||||||
iovec *iov, int n_iov, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb);
|
|
||||||
int fulfill_read_push(blockstore_op_t *op, void *buf, uint64_t offset, uint64_t len,
|
|
||||||
uint32_t item_state, uint64_t item_version);
|
|
||||||
void handle_read_event(ring_data_t *data, blockstore_op_t *op);
|
void handle_read_event(ring_data_t *data, blockstore_op_t *op);
|
||||||
|
bool verify_read_checksums(blockstore_op_t *op);
|
||||||
|
|
||||||
// Write
|
// Write
|
||||||
bool enqueue_write(blockstore_op_t *op);
|
bool enqueue_write(blockstore_op_t *op);
|
||||||
void cancel_all_writes(blockstore_op_t *op, blockstore_dirty_db_t::iterator dirty_it, int retval);
|
void prepare_meta_block_write(uint32_t modified_block);
|
||||||
|
bool meta_block_is_pending(uint32_t modified_block);
|
||||||
|
bool intent_write_allowed(blockstore_op_t *op, heap_entry_t *obj);
|
||||||
int dequeue_write(blockstore_op_t *op);
|
int dequeue_write(blockstore_op_t *op);
|
||||||
int dequeue_del(blockstore_op_t *op);
|
|
||||||
int continue_write(blockstore_op_t *op);
|
int continue_write(blockstore_op_t *op);
|
||||||
void release_journal_sectors(blockstore_op_t *op);
|
|
||||||
void handle_write_event(ring_data_t *data, blockstore_op_t *op);
|
void handle_write_event(ring_data_t *data, blockstore_op_t *op);
|
||||||
|
|
||||||
// Sync
|
// Sync
|
||||||
int continue_sync(blockstore_op_t *op);
|
int continue_sync(blockstore_op_t *op);
|
||||||
void ack_sync(blockstore_op_t *op);
|
bool submit_fsyncs(int & wait_count);
|
||||||
|
int do_sync(blockstore_op_t *op, int base_state);
|
||||||
|
bool has_unsynced();
|
||||||
|
|
||||||
// Stabilize
|
// Stabilize
|
||||||
int dequeue_stable(blockstore_op_t *op);
|
int dequeue_stable(blockstore_op_t *op);
|
||||||
int continue_stable(blockstore_op_t *op);
|
|
||||||
void mark_stable(obj_ver_id ov, bool forget_dirty = false);
|
|
||||||
void stabilize_object(object_id oid, uint64_t max_ver);
|
|
||||||
blockstore_op_t* selective_sync(blockstore_op_t *op);
|
|
||||||
int split_stab_op(blockstore_op_t *op, std::function<int(obj_ver_id v)> decider);
|
|
||||||
|
|
||||||
// Rollback
|
|
||||||
int dequeue_rollback(blockstore_op_t *op);
|
|
||||||
int continue_rollback(blockstore_op_t *op);
|
|
||||||
void mark_rolled_back(const obj_ver_id & ov);
|
|
||||||
void erase_dirty(blockstore_dirty_db_t::iterator dirty_start, blockstore_dirty_db_t::iterator dirty_end, uint64_t clean_loc);
|
|
||||||
void free_dirty_dyn_data(dirty_entry & e);
|
|
||||||
|
|
||||||
// List
|
// List
|
||||||
void process_list(blockstore_op_t *op);
|
void process_list(blockstore_op_t *op);
|
||||||
|
|
||||||
public:
|
/*public:*/
|
||||||
|
|
||||||
blockstore_impl_t(blockstore_config_t & config, ring_loop_t *ringloop, timerfd_manager_t *tfd);
|
blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd, bool mock_mode = false);
|
||||||
~blockstore_impl_t();
|
~blockstore_impl_t();
|
||||||
|
|
||||||
|
void parse_config(blockstore_config_t & config);
|
||||||
void parse_config(blockstore_config_t & config, bool init);
|
void parse_config(blockstore_config_t & config, bool init);
|
||||||
|
|
||||||
|
void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit);
|
||||||
|
bool reshard_continue(void *reshard_state, uint64_t chunk_limit);
|
||||||
|
|
||||||
// Event loop
|
// Event loop
|
||||||
void loop();
|
void loop();
|
||||||
|
|
||||||
@@ -427,21 +214,19 @@ public:
|
|||||||
// Simplified synchronous operation: get object bitmap & current version
|
// Simplified synchronous operation: get object bitmap & current version
|
||||||
int read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version = NULL);
|
int read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version = NULL);
|
||||||
|
|
||||||
// Unstable writes are added here (map of object_id -> version)
|
|
||||||
std::unordered_map<object_id, uint64_t> unstable_writes;
|
|
||||||
|
|
||||||
// Space usage statistics
|
|
||||||
std::map<uint64_t, uint64_t> inode_space_stats;
|
|
||||||
|
|
||||||
// Set per-pool no_inode_stats
|
// Set per-pool no_inode_stats
|
||||||
void set_no_inode_stats(const std::vector<uint64_t> & pool_ids);
|
void set_no_inode_stats(const std::vector<uint64_t> & pool_ids);
|
||||||
|
|
||||||
// Print diagnostics to stdout
|
// Print diagnostics to stdout
|
||||||
void dump_diagnostics();
|
void dump_diagnostics();
|
||||||
|
|
||||||
|
// Get diagnostic string for an operation
|
||||||
|
std::string get_op_diag(blockstore_op_t *op);
|
||||||
|
|
||||||
|
const std::map<uint64_t, uint64_t> & get_inode_space_stats() { return heap->get_inode_space_stats(); }
|
||||||
inline uint32_t get_block_size() { return dsk.data_block_size; }
|
inline uint32_t get_block_size() { return dsk.data_block_size; }
|
||||||
inline uint64_t get_block_count() { return dsk.block_count; }
|
inline uint64_t get_block_count() { return dsk.block_count; }
|
||||||
inline uint64_t get_free_block_count() { return dsk.block_count - used_blocks; }
|
uint64_t get_free_block_count();
|
||||||
inline uint32_t get_bitmap_granularity() { return dsk.disk_alignment; }
|
inline uint32_t get_bitmap_granularity() { return dsk.bitmap_granularity; }
|
||||||
inline uint64_t get_journal_size() { return dsk.journal_len; }
|
inline uint64_t get_journal_size() { return dsk.journal_len; }
|
||||||
};
|
};
|
||||||
|
|||||||
+144
-1026
File diff suppressed because it is too large
Load Diff
@@ -15,6 +15,7 @@ class blockstore_init_meta
|
|||||||
{
|
{
|
||||||
blockstore_impl_t *bs;
|
blockstore_impl_t *bs;
|
||||||
int wait_state = 0;
|
int wait_state = 0;
|
||||||
|
int wait_count = 0;
|
||||||
bool zero_on_init = false;
|
bool zero_on_init = false;
|
||||||
void *metadata_buffer = NULL;
|
void *metadata_buffer = NULL;
|
||||||
blockstore_init_meta_buf bufs[2] = {};
|
blockstore_init_meta_buf bufs[2] = {};
|
||||||
@@ -25,47 +26,11 @@ class blockstore_init_meta
|
|||||||
uint64_t next_offset = 0;
|
uint64_t next_offset = 0;
|
||||||
uint64_t last_read_offset = 0;
|
uint64_t last_read_offset = 0;
|
||||||
uint64_t entries_loaded = 0;
|
uint64_t entries_loaded = 0;
|
||||||
unsigned entries_per_block = 0;
|
std::vector<uint32_t> recheck_mod;
|
||||||
int i = 0, j = 0;
|
int i = 0, j = 0;
|
||||||
std::vector<uint64_t> entries_to_zero;
|
|
||||||
bool handle_meta_block(uint8_t *buf, uint64_t count, uint64_t done_cnt);
|
bool handle_meta_block(uint8_t *buf, uint64_t count, uint64_t done_cnt);
|
||||||
void handle_event(ring_data_t *data, int buf_num);
|
void handle_event(ring_data_t *data, int buf_num);
|
||||||
public:
|
public:
|
||||||
blockstore_init_meta(blockstore_impl_t *bs);
|
blockstore_init_meta(blockstore_impl_t *bs);
|
||||||
int loop();
|
int loop();
|
||||||
};
|
};
|
||||||
|
|
||||||
struct bs_init_journal_done
|
|
||||||
{
|
|
||||||
void *buf;
|
|
||||||
uint64_t pos, len;
|
|
||||||
};
|
|
||||||
|
|
||||||
class blockstore_init_journal
|
|
||||||
{
|
|
||||||
blockstore_impl_t *bs;
|
|
||||||
int wait_state = 0, wait_count = 0, handle_res = 0;
|
|
||||||
uint64_t entries_loaded = 0;
|
|
||||||
uint32_t crc32_last = 0;
|
|
||||||
bool started = false;
|
|
||||||
uint64_t next_free;
|
|
||||||
std::vector<bs_init_journal_done> done;
|
|
||||||
std::vector<obj_ver_id> double_allocs;
|
|
||||||
std::vector<iovec> small_write_data;
|
|
||||||
uint64_t journal_pos = 0;
|
|
||||||
uint64_t continue_pos = 0;
|
|
||||||
void *init_write_buf = NULL;
|
|
||||||
uint64_t init_write_sector = 0;
|
|
||||||
bool wrapped = false;
|
|
||||||
void *submitted_buf;
|
|
||||||
struct io_uring_sqe *sqe;
|
|
||||||
struct ring_data_t *data;
|
|
||||||
journal_entry_start *je_start;
|
|
||||||
std::function<void(ring_data_t*)> simple_callback;
|
|
||||||
int handle_journal_part(void *buf, uint64_t done_pos, uint64_t len);
|
|
||||||
void handle_event(ring_data_t *data);
|
|
||||||
void erase_dirty_object(blockstore_dirty_db_t::iterator dirty_it);
|
|
||||||
public:
|
|
||||||
blockstore_init_journal(blockstore_impl_t* bs);
|
|
||||||
int loop();
|
|
||||||
};
|
|
||||||
|
|||||||
@@ -0,0 +1,61 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#define BS_SUBMIT_CHECK_SQES(n) \
|
||||||
|
if (ringloop->space_left() < (n))\
|
||||||
|
{\
|
||||||
|
/* Pause until there are more requests available */\
|
||||||
|
PRIV(op)->wait_detail = (n);\
|
||||||
|
PRIV(op)->wait_for = WAIT_SQE;\
|
||||||
|
return 0;\
|
||||||
|
}
|
||||||
|
|
||||||
|
#define BS_SUBMIT_GET_SQE(sqe, data) \
|
||||||
|
BS_SUBMIT_GET_ONLY_SQE(sqe); \
|
||||||
|
struct ring_data_t *data = ((ring_data_t*)sqe->user_data)
|
||||||
|
|
||||||
|
#define BS_SUBMIT_GET_ONLY_SQE(sqe) \
|
||||||
|
struct io_uring_sqe *sqe = get_sqe();\
|
||||||
|
if (!sqe)\
|
||||||
|
{\
|
||||||
|
/* Pause until there are more requests available */\
|
||||||
|
PRIV(op)->wait_detail = 1;\
|
||||||
|
PRIV(op)->wait_for = WAIT_SQE;\
|
||||||
|
return 0;\
|
||||||
|
}
|
||||||
|
|
||||||
|
#define BS_SUBMIT_GET_SQE_DECL(sqe) \
|
||||||
|
sqe = get_sqe();\
|
||||||
|
if (!sqe)\
|
||||||
|
{\
|
||||||
|
/* Pause until there are more requests available */\
|
||||||
|
PRIV(op)->wait_detail = 1;\
|
||||||
|
PRIV(op)->wait_for = WAIT_SQE;\
|
||||||
|
return 0;\
|
||||||
|
}
|
||||||
|
|
||||||
|
#define PRIV(op) ((blockstore_op_private_t*)(op)->private_data)
|
||||||
|
#define FINISH_OP(op) PRIV(op)->~blockstore_op_private_t(); std::function<void (blockstore_op_t*)>(op->callback)(op)
|
||||||
|
|
||||||
|
// Suspend operation until there are more free SQEs
|
||||||
|
#define WAIT_SQE 1
|
||||||
|
// Suspend operation until there are <wait_detail> bytes of free space in the journal on disk
|
||||||
|
#define WAIT_COMPACTION 2
|
||||||
|
|
||||||
|
#define COPY_BUF_JOURNAL 0x01
|
||||||
|
#define COPY_BUF_DATA 0x02
|
||||||
|
#define COPY_BUF_ZERO 0x04
|
||||||
|
#define COPY_BUF_CSUM_FILL 0x08
|
||||||
|
#define COPY_BUF_COALESCED 0x10
|
||||||
|
#define COPY_BUF_PADDED 0x20
|
||||||
|
#define COPY_BUF_SKIP_CSUM 0x40
|
||||||
|
|
||||||
|
#ifndef RWF_ATOMIC
|
||||||
|
#define RWF_ATOMIC 0x40
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#ifndef RWF_DSYNC
|
||||||
|
#define RWF_DSYNC 0x02
|
||||||
|
#endif
|
||||||
@@ -2,8 +2,14 @@
|
|||||||
// License: VNPL-1.1 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
#include <sys/file.h>
|
#include <sys/file.h>
|
||||||
|
#include <stdexcept>
|
||||||
#include "blockstore_impl.h"
|
#include "blockstore_impl.h"
|
||||||
|
|
||||||
|
void blockstore_impl_t::parse_config(blockstore_config_t & config)
|
||||||
|
{
|
||||||
|
return parse_config(config, false);
|
||||||
|
}
|
||||||
|
|
||||||
void blockstore_impl_t::parse_config(blockstore_config_t & config, bool init)
|
void blockstore_impl_t::parse_config(blockstore_config_t & config, bool init)
|
||||||
{
|
{
|
||||||
// Online-configurable options:
|
// Online-configurable options:
|
||||||
@@ -14,12 +20,15 @@ void blockstore_impl_t::parse_config(blockstore_config_t & config, bool init)
|
|||||||
}
|
}
|
||||||
min_flusher_count = strtoull(config["min_flusher_count"].c_str(), NULL, 10);
|
min_flusher_count = strtoull(config["min_flusher_count"].c_str(), NULL, 10);
|
||||||
journal_trim_interval = strtoull(config["journal_trim_interval"].c_str(), NULL, 10);
|
journal_trim_interval = strtoull(config["journal_trim_interval"].c_str(), NULL, 10);
|
||||||
|
flusher_start_threshold = strtoull(config["flusher_start_threshold"].c_str(), NULL, 10);
|
||||||
max_write_iodepth = strtoull(config["max_write_iodepth"].c_str(), NULL, 10);
|
max_write_iodepth = strtoull(config["max_write_iodepth"].c_str(), NULL, 10);
|
||||||
throttle_small_writes = config["throttle_small_writes"] == "true" || config["throttle_small_writes"] == "1" || config["throttle_small_writes"] == "yes";
|
throttle_small_writes = config["throttle_small_writes"] == "true" || config["throttle_small_writes"] == "1" || config["throttle_small_writes"] == "yes";
|
||||||
throttle_target_iops = strtoull(config["throttle_target_iops"].c_str(), NULL, 10);
|
throttle_target_iops = strtoull(config["throttle_target_iops"].c_str(), NULL, 10);
|
||||||
throttle_target_mbs = strtoull(config["throttle_target_mbs"].c_str(), NULL, 10);
|
throttle_target_mbs = strtoull(config["throttle_target_mbs"].c_str(), NULL, 10);
|
||||||
throttle_target_parallelism = strtoull(config["throttle_target_parallelism"].c_str(), NULL, 10);
|
throttle_target_parallelism = strtoull(config["throttle_target_parallelism"].c_str(), NULL, 10);
|
||||||
throttle_threshold_us = strtoull(config["throttle_threshold_us"].c_str(), NULL, 10);
|
throttle_threshold_us = strtoull(config["throttle_threshold_us"].c_str(), NULL, 10);
|
||||||
|
perfect_csum_update = config["perfect_csum_update"] == "true" || config["perfect_csum_update"] == "1" || config["perfect_csum_update"] == "yes";
|
||||||
|
skip_corrupted_meta_entries = config["skip_corrupted_meta_entries"] == "true" || config["skip_corrupted_meta_entries"] == "1" || config["skip_corrupted_meta_entries"] == "yes";
|
||||||
if (config["autosync_writes"] != "")
|
if (config["autosync_writes"] != "")
|
||||||
{
|
{
|
||||||
autosync_writes = strtoull(config["autosync_writes"].c_str(), NULL, 10);
|
autosync_writes = strtoull(config["autosync_writes"].c_str(), NULL, 10);
|
||||||
@@ -28,13 +37,17 @@ void blockstore_impl_t::parse_config(blockstore_config_t & config, bool init)
|
|||||||
{
|
{
|
||||||
max_flusher_count = 256;
|
max_flusher_count = 256;
|
||||||
}
|
}
|
||||||
if (!min_flusher_count || journal.flush_journal)
|
if (!min_flusher_count)
|
||||||
{
|
{
|
||||||
min_flusher_count = 1;
|
min_flusher_count = 1;
|
||||||
}
|
}
|
||||||
if (!journal_trim_interval)
|
if (!journal_trim_interval)
|
||||||
{
|
{
|
||||||
journal_trim_interval = 512;
|
journal_trim_interval = 4096;
|
||||||
|
}
|
||||||
|
if (!flusher_start_threshold)
|
||||||
|
{
|
||||||
|
flusher_start_threshold = 32;
|
||||||
}
|
}
|
||||||
if (!max_write_iodepth)
|
if (!max_write_iodepth)
|
||||||
{
|
{
|
||||||
@@ -68,23 +81,6 @@ void blockstore_impl_t::parse_config(blockstore_config_t & config, bool init)
|
|||||||
{
|
{
|
||||||
readonly = true;
|
readonly = true;
|
||||||
}
|
}
|
||||||
if (config["disable_data_fsync"] == "true" || config["disable_data_fsync"] == "1" || config["disable_data_fsync"] == "yes")
|
|
||||||
{
|
|
||||||
disable_data_fsync = true;
|
|
||||||
}
|
|
||||||
if (config["disable_meta_fsync"] == "true" || config["disable_meta_fsync"] == "1" || config["disable_meta_fsync"] == "yes")
|
|
||||||
{
|
|
||||||
disable_meta_fsync = true;
|
|
||||||
}
|
|
||||||
if (config["disable_journal_fsync"] == "true" || config["disable_journal_fsync"] == "1" || config["disable_journal_fsync"] == "yes")
|
|
||||||
{
|
|
||||||
disable_journal_fsync = true;
|
|
||||||
}
|
|
||||||
if (config["flush_journal"] == "true" || config["flush_journal"] == "1" || config["flush_journal"] == "yes")
|
|
||||||
{
|
|
||||||
// Only flush journal and exit
|
|
||||||
journal.flush_journal = true;
|
|
||||||
}
|
|
||||||
if (config["immediate_commit"] == "all")
|
if (config["immediate_commit"] == "all")
|
||||||
{
|
{
|
||||||
immediate_commit = IMMEDIATE_ALL;
|
immediate_commit = IMMEDIATE_ALL;
|
||||||
@@ -94,85 +90,27 @@ void blockstore_impl_t::parse_config(blockstore_config_t & config, bool init)
|
|||||||
immediate_commit = IMMEDIATE_SMALL;
|
immediate_commit = IMMEDIATE_SMALL;
|
||||||
}
|
}
|
||||||
metadata_buf_size = strtoull(config["meta_buf_size"].c_str(), NULL, 10);
|
metadata_buf_size = strtoull(config["meta_buf_size"].c_str(), NULL, 10);
|
||||||
inmemory_meta = config["inmemory_metadata"] != "false" && config["inmemory_metadata"] != "0" &&
|
meta_write_recheck_parallelism = strtoull(config["meta_write_recheck_parallelism"].c_str(), NULL, 10);
|
||||||
config["inmemory_metadata"] != "no";
|
|
||||||
journal.sector_count = strtoull(config["journal_sector_buffer_count"].c_str(), NULL, 10);
|
|
||||||
journal.no_same_sector_overwrites = config["journal_no_same_sector_overwrites"] == "true" ||
|
|
||||||
config["journal_no_same_sector_overwrites"] == "1" || config["journal_no_same_sector_overwrites"] == "yes";
|
|
||||||
journal.inmemory = config["inmemory_journal"] != "false" && config["inmemory_journal"] != "0" &&
|
|
||||||
config["inmemory_journal"] != "no";
|
|
||||||
log_level = strtoull(config["log_level"].c_str(), NULL, 10);
|
log_level = strtoull(config["log_level"].c_str(), NULL, 10);
|
||||||
// Validate
|
// Validate
|
||||||
if (journal.sector_count < 2)
|
|
||||||
{
|
|
||||||
journal.sector_count = 32;
|
|
||||||
}
|
|
||||||
if (metadata_buf_size < 65536)
|
if (metadata_buf_size < 65536)
|
||||||
{
|
{
|
||||||
metadata_buf_size = 4*1024*1024;
|
metadata_buf_size = 4*1024*1024;
|
||||||
}
|
}
|
||||||
if (dsk.meta_device == dsk.data_device)
|
if (metadata_buf_size % dsk.meta_block_size)
|
||||||
{
|
{
|
||||||
disable_meta_fsync = disable_data_fsync;
|
throw std::runtime_error("metadata_buf_size should be a multiple of meta_block_size");
|
||||||
}
|
}
|
||||||
if (dsk.journal_device == dsk.meta_device)
|
if (!meta_write_recheck_parallelism)
|
||||||
{
|
{
|
||||||
disable_journal_fsync = disable_meta_fsync;
|
meta_write_recheck_parallelism = 16;
|
||||||
}
|
}
|
||||||
if (immediate_commit != IMMEDIATE_NONE && !disable_journal_fsync)
|
if (immediate_commit != IMMEDIATE_NONE && !dsk.disable_journal_fsync)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("immediate_commit requires disable_journal_fsync");
|
throw std::runtime_error("immediate_commit requires disable_journal_fsync");
|
||||||
}
|
}
|
||||||
if (immediate_commit == IMMEDIATE_ALL && !disable_data_fsync)
|
if (immediate_commit == IMMEDIATE_ALL && !dsk.disable_data_fsync)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("immediate_commit=all requires disable_journal_fsync and disable_data_fsync");
|
throw std::runtime_error("immediate_commit=all requires disable_journal_fsync and disable_data_fsync");
|
||||||
}
|
}
|
||||||
// init some fields
|
|
||||||
journal.block_size = dsk.journal_block_size;
|
|
||||||
journal.next_free = dsk.journal_block_size;
|
|
||||||
journal.used_start = dsk.journal_block_size;
|
|
||||||
// no free space because sector is initially unmapped
|
|
||||||
journal.in_sector_pos = dsk.journal_block_size;
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_impl_t::calc_lengths()
|
|
||||||
{
|
|
||||||
dsk.calc_lengths();
|
|
||||||
journal.len = dsk.journal_len;
|
|
||||||
journal.block_size = dsk.journal_block_size;
|
|
||||||
journal.offset = dsk.journal_offset;
|
|
||||||
if (inmemory_meta)
|
|
||||||
{
|
|
||||||
metadata_buffer = memalign(MEM_ALIGNMENT, dsk.meta_len);
|
|
||||||
if (!metadata_buffer)
|
|
||||||
throw std::runtime_error("Failed to allocate memory for the metadata ("+std::to_string(dsk.meta_len/1024/1024)+" MB)");
|
|
||||||
}
|
|
||||||
else if (dsk.clean_entry_bitmap_size || dsk.data_csum_type)
|
|
||||||
{
|
|
||||||
clean_bitmaps = (uint8_t*)malloc(dsk.block_count * 2 * dsk.clean_entry_bitmap_size);
|
|
||||||
if (!clean_bitmaps)
|
|
||||||
{
|
|
||||||
throw std::runtime_error(
|
|
||||||
"Failed to allocate memory for the metadata sparse write bitmap ("+
|
|
||||||
std::to_string(dsk.block_count * 2 * dsk.clean_entry_bitmap_size / 1024 / 1024)+" MB)"
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (journal.inmemory)
|
|
||||||
{
|
|
||||||
journal.buffer = memalign(MEM_ALIGNMENT, journal.len);
|
|
||||||
if (!journal.buffer)
|
|
||||||
throw std::runtime_error("Failed to allocate memory for journal ("+std::to_string(journal.len/1024/1024)+" MB)");
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
journal.sector_buf = (uint8_t*)memalign(MEM_ALIGNMENT, journal.sector_count * dsk.journal_block_size);
|
|
||||||
if (!journal.sector_buf)
|
|
||||||
throw std::bad_alloc();
|
|
||||||
}
|
|
||||||
journal.sector_info = (journal_sector_info_t*)calloc(journal.sector_count, sizeof(journal_sector_info_t));
|
|
||||||
if (!journal.sector_info)
|
|
||||||
{
|
|
||||||
throw std::bad_alloc();
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
+397
-937
File diff suppressed because it is too large
Load Diff
@@ -2,560 +2,91 @@
|
|||||||
// License: VNPL-1.1 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
#include "blockstore_impl.h"
|
#include "blockstore_impl.h"
|
||||||
|
#include "blockstore_internal.h"
|
||||||
|
|
||||||
// Stabilize small write:
|
// Handles both stabilize (commit) and rollback
|
||||||
// 1) Copy data from the journal to the data device
|
|
||||||
// 2) Increase version on the metadata device and sync it
|
|
||||||
// 3) Advance clean_db entry's version, clear previous journal entries
|
|
||||||
//
|
|
||||||
// This makes 1 4K small write+sync look like:
|
|
||||||
// 512b+4K (journal) + sync + 512b (journal) + sync + 4K (data) [+ sync?] + 512b (metadata) + sync.
|
|
||||||
// WA = 2.375. It's not the best, SSD FTL-like redirect-write could probably be lower
|
|
||||||
// even with defragmentation. But it's fixed and it's still better than in Ceph. :)
|
|
||||||
// except for HDD-only clusters, because each write results in 3 seeks.
|
|
||||||
|
|
||||||
// Stabilize big write:
|
|
||||||
// 1) Copy metadata from the journal to the metadata device
|
|
||||||
// 2) Move dirty_db entry to clean_db and clear previous journal entries
|
|
||||||
//
|
|
||||||
// This makes 1 128K big write+sync look like:
|
|
||||||
// 128K (data) + sync + 512b (journal) + sync + 512b (journal) + sync + 512b (metadata) + sync.
|
|
||||||
// WA = 1.012. Very good :)
|
|
||||||
|
|
||||||
// Stabilize delete:
|
|
||||||
// 1) Remove metadata entry and sync it
|
|
||||||
// 2) Remove dirty_db entry and clear previous journal entries
|
|
||||||
// We have 2 problems here:
|
|
||||||
// - In the cluster environment, we must store the "tombstones" of deleted objects until
|
|
||||||
// all replicas (not just quorum) agrees about their deletion. That is, "stabilize" is
|
|
||||||
// not possible for deletes in degraded placement groups
|
|
||||||
// - With simple "fixed" metadata tables we can't just clear the metadata entry of the latest
|
|
||||||
// object version. We must clear all previous entries, too.
|
|
||||||
// FIXME Fix both problems - probably, by switching from "fixed" metadata tables to "dynamic"
|
|
||||||
|
|
||||||
// AND We must do it in batches, for the sake of reduced fsync call count
|
|
||||||
// AND We must know what we stabilize. Basic workflow is like:
|
|
||||||
// 1) primary OSD receives sync request
|
|
||||||
// 2) it submits syncs to blockstore and peers
|
|
||||||
// 3) after everyone acks sync it acks sync to the client
|
|
||||||
// 4) after a while it takes his synced object list and sends stabilize requests
|
|
||||||
// to peers and to its own blockstore, thus freeing the old version
|
|
||||||
|
|
||||||
struct ver_vector_t
|
|
||||||
{
|
|
||||||
obj_ver_id *items = NULL;
|
|
||||||
uint64_t alloc = 0, size = 0;
|
|
||||||
};
|
|
||||||
|
|
||||||
static void init_versions(ver_vector_t & vec, obj_ver_id *start, obj_ver_id *end, uint64_t len)
|
|
||||||
{
|
|
||||||
if (!vec.items)
|
|
||||||
{
|
|
||||||
vec.alloc = len;
|
|
||||||
vec.items = (obj_ver_id*)malloc_or_die(sizeof(obj_ver_id) * vec.alloc);
|
|
||||||
for (auto sv = start; sv < end; sv++)
|
|
||||||
{
|
|
||||||
vec.items[vec.size++] = *sv;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
static void append_version(ver_vector_t & vec, obj_ver_id ov)
|
|
||||||
{
|
|
||||||
if (vec.size >= vec.alloc)
|
|
||||||
{
|
|
||||||
vec.alloc = !vec.alloc ? 4 : vec.alloc*2;
|
|
||||||
vec.items = (obj_ver_id*)realloc_or_die(vec.items, sizeof(obj_ver_id) * vec.alloc);
|
|
||||||
}
|
|
||||||
vec.items[vec.size++] = ov;
|
|
||||||
}
|
|
||||||
|
|
||||||
static bool check_unsynced(std::vector<obj_ver_id> & check, obj_ver_id ov, std::vector<obj_ver_id> & to, int *count)
|
|
||||||
{
|
|
||||||
bool found = false;
|
|
||||||
int j = 0, k = 0;
|
|
||||||
while (j < check.size())
|
|
||||||
{
|
|
||||||
if (check[j] == ov)
|
|
||||||
found = true;
|
|
||||||
if (check[j].oid == ov.oid && check[j].version <= ov.version)
|
|
||||||
{
|
|
||||||
to.push_back(check[j++]);
|
|
||||||
if (count)
|
|
||||||
(*count)--;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
check[k++] = check[j++];
|
|
||||||
}
|
|
||||||
check.resize(k);
|
|
||||||
return found;
|
|
||||||
}
|
|
||||||
|
|
||||||
blockstore_op_t* blockstore_impl_t::selective_sync(blockstore_op_t *op)
|
|
||||||
{
|
|
||||||
unsynced_big_write_count -= unsynced_big_writes.size();
|
|
||||||
unsynced_big_writes.swap(PRIV(op)->sync_big_writes);
|
|
||||||
unsynced_big_write_count += unsynced_big_writes.size();
|
|
||||||
unsynced_small_writes.swap(PRIV(op)->sync_small_writes);
|
|
||||||
// Create a sync operation, insert into the end of the queue
|
|
||||||
// And move ourselves into the end too!
|
|
||||||
// Rather hacky but that's what we need...
|
|
||||||
blockstore_op_t *sync_op = new blockstore_op_t;
|
|
||||||
sync_op->opcode = BS_OP_SYNC;
|
|
||||||
sync_op->buf = NULL;
|
|
||||||
sync_op->callback = [](blockstore_op_t *sync_op)
|
|
||||||
{
|
|
||||||
delete sync_op;
|
|
||||||
};
|
|
||||||
init_op(sync_op);
|
|
||||||
int sync_res = continue_sync(sync_op);
|
|
||||||
if (sync_res != 2)
|
|
||||||
{
|
|
||||||
// Put SYNC into the queue if it's not finished yet
|
|
||||||
submit_queue.push_back(sync_op);
|
|
||||||
}
|
|
||||||
// Restore unsynced_writes
|
|
||||||
unsynced_small_writes.swap(PRIV(op)->sync_small_writes);
|
|
||||||
unsynced_big_write_count -= unsynced_big_writes.size();
|
|
||||||
unsynced_big_writes.swap(PRIV(op)->sync_big_writes);
|
|
||||||
unsynced_big_write_count += unsynced_big_writes.size();
|
|
||||||
if (sync_res == 2)
|
|
||||||
{
|
|
||||||
// Sync is immediately completed
|
|
||||||
return NULL;
|
|
||||||
}
|
|
||||||
return sync_op;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Returns: 2 = stop processing and dequeue, 0 = stop processing and do not dequeue, 1 = proceed with op itself
|
|
||||||
int blockstore_impl_t::split_stab_op(blockstore_op_t *op, std::function<int(obj_ver_id v)> decider)
|
|
||||||
{
|
|
||||||
bool add_sync = false;
|
|
||||||
ver_vector_t good_vers, bad_vers;
|
|
||||||
obj_ver_id* v;
|
|
||||||
int i, todo = 0;
|
|
||||||
for (i = 0, v = (obj_ver_id*)op->buf; i < op->len; i++, v++)
|
|
||||||
{
|
|
||||||
int action = decider(*v);
|
|
||||||
if (action < 0)
|
|
||||||
{
|
|
||||||
// Rollback changes
|
|
||||||
for (auto & ov: PRIV(op)->sync_big_writes)
|
|
||||||
{
|
|
||||||
unsynced_big_writes.push_back(ov);
|
|
||||||
unsynced_big_write_count++;
|
|
||||||
}
|
|
||||||
for (auto & ov: PRIV(op)->sync_small_writes)
|
|
||||||
{
|
|
||||||
unsynced_small_writes.push_back(ov);
|
|
||||||
}
|
|
||||||
free(good_vers.items);
|
|
||||||
good_vers.items = NULL;
|
|
||||||
free(bad_vers.items);
|
|
||||||
bad_vers.items = NULL;
|
|
||||||
// Error
|
|
||||||
op->retval = action;
|
|
||||||
FINISH_OP(op);
|
|
||||||
return 2;
|
|
||||||
}
|
|
||||||
else if (action == STAB_SPLIT_DONE)
|
|
||||||
{
|
|
||||||
// Already done
|
|
||||||
init_versions(good_vers, (obj_ver_id*)op->buf, v, op->len);
|
|
||||||
}
|
|
||||||
else if (action == STAB_SPLIT_WAIT)
|
|
||||||
{
|
|
||||||
// Already in progress, we just have to wait until it finishes
|
|
||||||
init_versions(good_vers, (obj_ver_id*)op->buf, v, op->len);
|
|
||||||
append_version(bad_vers, *v);
|
|
||||||
}
|
|
||||||
else if (action == STAB_SPLIT_SYNC)
|
|
||||||
{
|
|
||||||
// Needs a SYNC, we have to send a SYNC if not already in progress
|
|
||||||
//
|
|
||||||
// If the object is not present in unsynced_(big|small)_writes then
|
|
||||||
// it's currently being synced. If it's present then we can initiate
|
|
||||||
// its sync ourselves.
|
|
||||||
init_versions(good_vers, (obj_ver_id*)op->buf, v, op->len);
|
|
||||||
append_version(bad_vers, *v);
|
|
||||||
if (!add_sync)
|
|
||||||
{
|
|
||||||
PRIV(op)->sync_big_writes.clear();
|
|
||||||
PRIV(op)->sync_small_writes.clear();
|
|
||||||
add_sync = true;
|
|
||||||
}
|
|
||||||
check_unsynced(unsynced_small_writes, *v, PRIV(op)->sync_small_writes, NULL);
|
|
||||||
check_unsynced(unsynced_big_writes, *v, PRIV(op)->sync_big_writes, &unsynced_big_write_count);
|
|
||||||
}
|
|
||||||
else /* if (action == STAB_SPLIT_TODO) */
|
|
||||||
{
|
|
||||||
if (good_vers.items)
|
|
||||||
{
|
|
||||||
// If we're selecting versions then append it
|
|
||||||
// Main idea is that 99% of the time all versions passed to BS_OP_STABLE are synced
|
|
||||||
// And we don't want to select/allocate anything in that optimistic case
|
|
||||||
append_version(good_vers, *v);
|
|
||||||
}
|
|
||||||
todo++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// In a pessimistic scenario, an operation may be split into 3:
|
|
||||||
// - Stabilize synced entries
|
|
||||||
// - Sync unsynced entries
|
|
||||||
// - Continue for unsynced entries after sync
|
|
||||||
add_sync = add_sync && (PRIV(op)->sync_big_writes.size() || PRIV(op)->sync_small_writes.size());
|
|
||||||
if (!todo && !bad_vers.size)
|
|
||||||
{
|
|
||||||
// Already stable
|
|
||||||
op->retval = 0;
|
|
||||||
FINISH_OP(op);
|
|
||||||
return 2;
|
|
||||||
}
|
|
||||||
op->retval = 0;
|
|
||||||
if (!todo && !add_sync)
|
|
||||||
{
|
|
||||||
// Only wait for inflight writes or current in-progress syncs
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
blockstore_op_t *sync_op = NULL, *split_stab_op = NULL;
|
|
||||||
if (add_sync)
|
|
||||||
{
|
|
||||||
// Initiate a selective sync for PRIV(op)->sync_(big|small)_writes
|
|
||||||
sync_op = selective_sync(op);
|
|
||||||
}
|
|
||||||
if (bad_vers.size)
|
|
||||||
{
|
|
||||||
// Split part of the request into a separate operation
|
|
||||||
split_stab_op = new blockstore_op_t;
|
|
||||||
split_stab_op->opcode = op->opcode;
|
|
||||||
split_stab_op->buf = bad_vers.items;
|
|
||||||
split_stab_op->len = bad_vers.size;
|
|
||||||
init_op(split_stab_op);
|
|
||||||
submit_queue.push_back(split_stab_op);
|
|
||||||
}
|
|
||||||
if (sync_op || split_stab_op || good_vers.items)
|
|
||||||
{
|
|
||||||
void *orig_buf = op->buf;
|
|
||||||
if (good_vers.items)
|
|
||||||
{
|
|
||||||
op->buf = good_vers.items;
|
|
||||||
op->len = good_vers.size;
|
|
||||||
}
|
|
||||||
// Make a wrapped callback
|
|
||||||
int *split_op_counter = (int*)malloc_or_die(sizeof(int));
|
|
||||||
*split_op_counter = (sync_op ? 1 : 0) + (split_stab_op ? 1 : 0) + (todo ? 1 : 0);
|
|
||||||
auto cb = [op, good_items = good_vers.items,
|
|
||||||
bad_items = bad_vers.items, split_op_counter,
|
|
||||||
orig_buf, real_cb = op->callback](blockstore_op_t *split_op)
|
|
||||||
{
|
|
||||||
if (split_op->retval != 0)
|
|
||||||
op->retval = split_op->retval;
|
|
||||||
(*split_op_counter)--;
|
|
||||||
assert((*split_op_counter) >= 0);
|
|
||||||
if (op != split_op)
|
|
||||||
delete split_op;
|
|
||||||
if (!*split_op_counter)
|
|
||||||
{
|
|
||||||
free(good_items);
|
|
||||||
free(bad_items);
|
|
||||||
free(split_op_counter);
|
|
||||||
op->buf = orig_buf;
|
|
||||||
real_cb(op);
|
|
||||||
}
|
|
||||||
};
|
|
||||||
if (sync_op)
|
|
||||||
{
|
|
||||||
sync_op->callback = cb;
|
|
||||||
}
|
|
||||||
if (split_stab_op)
|
|
||||||
{
|
|
||||||
split_stab_op->callback = cb;
|
|
||||||
}
|
|
||||||
op->callback = cb;
|
|
||||||
}
|
|
||||||
if (!todo)
|
|
||||||
{
|
|
||||||
// All work is postponed
|
|
||||||
op->callback = NULL;
|
|
||||||
return 2;
|
|
||||||
}
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
int blockstore_impl_t::dequeue_stable(blockstore_op_t *op)
|
int blockstore_impl_t::dequeue_stable(blockstore_op_t *op)
|
||||||
{
|
{
|
||||||
if (PRIV(op)->op_state)
|
obj_ver_id *v = (obj_ver_id*)op->buf;
|
||||||
{
|
auto priv = PRIV(op);
|
||||||
return continue_stable(op);
|
if (priv->op_state == 1) goto resume_1;
|
||||||
}
|
else if (priv->op_state == 2) goto resume_2;
|
||||||
int r = split_stab_op(op, [this](obj_ver_id ov)
|
else if (priv->op_state == 3) goto resume_3;
|
||||||
{
|
else if (priv->op_state == 4) goto resume_4;
|
||||||
auto dirty_it = dirty_db.find(ov);
|
else if (priv->op_state == 5) goto resume_5;
|
||||||
if (dirty_it == dirty_db.end())
|
assert(!priv->op_state);
|
||||||
{
|
|
||||||
auto & clean_db = clean_db_shard(ov.oid);
|
|
||||||
auto clean_it = clean_db.find(ov.oid);
|
|
||||||
if (clean_it == clean_db.end() || clean_it->second.version < ov.version)
|
|
||||||
{
|
|
||||||
// No such object version
|
|
||||||
printf("Error: %jx:%jx v%ju not found while stabilizing\n", ov.oid.inode, ov.oid.stripe, ov.version);
|
|
||||||
return -ENOENT;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// Already stable
|
|
||||||
return STAB_SPLIT_DONE;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else if (IS_STABLE(dirty_it->second.state))
|
|
||||||
{
|
|
||||||
// Already stable
|
|
||||||
return STAB_SPLIT_DONE;
|
|
||||||
}
|
|
||||||
while (true)
|
|
||||||
{
|
|
||||||
if (IS_IN_FLIGHT(dirty_it->second.state))
|
|
||||||
{
|
|
||||||
// Object write is still in progress. Wait until the write request completes
|
|
||||||
return STAB_SPLIT_WAIT;
|
|
||||||
}
|
|
||||||
else if (!IS_SYNCED(dirty_it->second.state))
|
|
||||||
{
|
|
||||||
// Object not synced yet - sync it
|
|
||||||
// In previous versions we returned EBUSY here and required
|
|
||||||
// the caller (OSD) to issue a global sync first. But a global sync
|
|
||||||
// waits for all writes in the queue including inflight writes. And
|
|
||||||
// inflight writes may themselves be blocked by unstable writes being
|
|
||||||
// still present in the journal and not flushed away from it.
|
|
||||||
// So we must sync specific objects here.
|
|
||||||
//
|
|
||||||
// Even more, we have to process "stabilize" request in parts. That is,
|
|
||||||
// we must stabilize all objects which are already synced. Otherwise
|
|
||||||
// they may block objects which are NOT synced yet.
|
|
||||||
return STAB_SPLIT_SYNC;
|
|
||||||
}
|
|
||||||
else if (IS_STABLE(dirty_it->second.state))
|
|
||||||
{
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
// Check previous versions too
|
|
||||||
if (dirty_it == dirty_db.begin())
|
|
||||||
{
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
dirty_it--;
|
|
||||||
if (dirty_it->first.oid != ov.oid)
|
|
||||||
{
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return STAB_SPLIT_TODO;
|
|
||||||
});
|
|
||||||
if (r != 1)
|
|
||||||
{
|
|
||||||
return r;
|
|
||||||
}
|
|
||||||
// Check journal space
|
|
||||||
blockstore_journal_check_t space_check(this);
|
|
||||||
if (!space_check.check_available(op, op->len, sizeof(journal_entry_stable), 0))
|
|
||||||
{
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
// There is sufficient space. Check SQEs
|
|
||||||
BS_SUBMIT_CHECK_SQES(space_check.sectors_to_write);
|
|
||||||
// Prepare and submit journal entries
|
|
||||||
int s = 0;
|
|
||||||
auto v = (obj_ver_id*)op->buf;
|
|
||||||
for (int i = 0; i < op->len; i++, v++)
|
|
||||||
{
|
|
||||||
if (!journal.entry_fits(sizeof(journal_entry_stable)) &&
|
|
||||||
journal.sector_info[journal.cur_sector].dirty)
|
|
||||||
{
|
|
||||||
prepare_journal_sector_write(journal.cur_sector, op);
|
|
||||||
s++;
|
|
||||||
}
|
|
||||||
journal_entry_stable *je = (journal_entry_stable*)
|
|
||||||
prefill_single_journal_entry(journal, JE_STABLE, sizeof(journal_entry_stable));
|
|
||||||
je->oid = v->oid;
|
|
||||||
je->version = v->version;
|
|
||||||
je->crc32 = je_crc32((journal_entry*)je);
|
|
||||||
journal.crc32_last = je->crc32;
|
|
||||||
}
|
|
||||||
prepare_journal_sector_write(journal.cur_sector, op);
|
|
||||||
s++;
|
|
||||||
assert(s == space_check.sectors_to_write);
|
|
||||||
PRIV(op)->op_state = 1;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
int blockstore_impl_t::continue_stable(blockstore_op_t *op)
|
|
||||||
{
|
|
||||||
if (PRIV(op)->op_state == 2)
|
|
||||||
goto resume_2;
|
|
||||||
else if (PRIV(op)->op_state == 4)
|
|
||||||
goto resume_4;
|
|
||||||
else
|
|
||||||
return 1;
|
|
||||||
resume_2:
|
|
||||||
if (!disable_journal_fsync)
|
|
||||||
{
|
|
||||||
BS_SUBMIT_GET_SQE(sqe, data);
|
|
||||||
io_uring_prep_fsync(sqe, dsk.journal_fd, IORING_FSYNC_DATASYNC);
|
|
||||||
data->iov = { 0 };
|
|
||||||
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
|
||||||
PRIV(op)->min_flushed_journal_sector = PRIV(op)->max_flushed_journal_sector = 0;
|
|
||||||
PRIV(op)->pending_ops = 1;
|
|
||||||
PRIV(op)->op_state = 3;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
resume_4:
|
|
||||||
// Mark dirty_db entries as stable, acknowledge op completion
|
|
||||||
obj_ver_id* v;
|
|
||||||
int i;
|
|
||||||
for (i = 0, v = (obj_ver_id*)op->buf; i < op->len; i++, v++)
|
|
||||||
{
|
|
||||||
// Mark all dirty_db entries up to op->version as stable
|
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
|
||||||
printf("Stabilize %jx:%jx v%ju\n", v->oid.inode, v->oid.stripe, v->version);
|
|
||||||
#endif
|
|
||||||
mark_stable(*v);
|
|
||||||
}
|
|
||||||
// Acknowledge op
|
|
||||||
op->retval = 0;
|
op->retval = 0;
|
||||||
|
priv->modified_block = priv->modified_block2 = UINT32_MAX;
|
||||||
|
for (priv->stab_pos = 0; priv->stab_pos < op->len; priv->stab_pos++)
|
||||||
|
{
|
||||||
|
{
|
||||||
|
auto obj = heap->read_entry(v[priv->stab_pos].oid);
|
||||||
|
if (!obj)
|
||||||
|
{
|
||||||
|
op->retval = -ENOENT;
|
||||||
|
FINISH_OP(op);
|
||||||
|
return 2;
|
||||||
|
}
|
||||||
|
int res = op->opcode == BS_OP_STABLE
|
||||||
|
? heap->add_commit(obj, v[priv->stab_pos].version, &priv->modified_block2)
|
||||||
|
: heap->add_rollback(obj, v[priv->stab_pos].version, &priv->modified_block2);
|
||||||
|
if (res == EBUSY)
|
||||||
|
{
|
||||||
|
op->retval = -EBUSY;
|
||||||
|
FINISH_OP(op);
|
||||||
|
return 2;
|
||||||
|
}
|
||||||
|
if (res == ENOSPC)
|
||||||
|
{
|
||||||
|
if (!heap->get_to_compact_count())
|
||||||
|
{
|
||||||
|
// no space
|
||||||
|
op->retval = -ENOSPC;
|
||||||
|
FINISH_OP(op);
|
||||||
|
return 2;
|
||||||
|
}
|
||||||
|
if (priv->modified_block2 != UINT32_MAX)
|
||||||
|
{
|
||||||
|
priv->stab_pos--;
|
||||||
|
goto resume_1;
|
||||||
|
}
|
||||||
|
priv->wait_for = WAIT_COMPACTION;
|
||||||
|
priv->wait_detail = heap->get_compacted_count();
|
||||||
|
flusher->request_trim();
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
assert(res == 0);
|
||||||
|
}
|
||||||
|
resume_1:
|
||||||
|
if (priv->modified_block != UINT32_MAX && priv->modified_block2 != priv->modified_block)
|
||||||
|
{
|
||||||
|
BS_SUBMIT_CHECK_SQES(1);
|
||||||
|
prepare_meta_block_write(priv->modified_block);
|
||||||
|
resume_2:
|
||||||
|
if (meta_block_is_pending(priv->modified_block))
|
||||||
|
{
|
||||||
|
priv->op_state = 2;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
priv->modified_block = priv->modified_block2;
|
||||||
|
if (priv->stab_pos == op->len-1 && priv->modified_block2 != UINT32_MAX)
|
||||||
|
{
|
||||||
|
priv->modified_block2 = UINT32_MAX;
|
||||||
|
goto resume_1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Fsync, just because our semantics imply that commit (stabilize) is immediately fsynced
|
||||||
|
priv->op_state = 3;
|
||||||
|
resume_3:
|
||||||
|
resume_4:
|
||||||
|
resume_5:
|
||||||
|
int res = do_sync(op, 3);
|
||||||
|
if (res != 2)
|
||||||
|
{
|
||||||
|
return res;
|
||||||
|
}
|
||||||
|
// Done. Don't touch op->retval - if anything resulted in ENOENT, return it as is
|
||||||
FINISH_OP(op);
|
FINISH_OP(op);
|
||||||
return 2;
|
return 2;
|
||||||
}
|
}
|
||||||
|
|
||||||
void blockstore_impl_t::mark_stable(obj_ver_id v, bool forget_dirty)
|
|
||||||
{
|
|
||||||
auto dirty_it = dirty_db.find(v);
|
|
||||||
if (dirty_it != dirty_db.end())
|
|
||||||
{
|
|
||||||
if (IS_INSTANT(dirty_it->second.state))
|
|
||||||
{
|
|
||||||
// 'Instant' (non-EC) operations may complete and try to become stable out of order. Prevent it.
|
|
||||||
auto back_it = dirty_it;
|
|
||||||
while (back_it != dirty_db.begin())
|
|
||||||
{
|
|
||||||
back_it--;
|
|
||||||
if (back_it->first.oid != v.oid)
|
|
||||||
{
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
if (!IS_STABLE(back_it->second.state))
|
|
||||||
{
|
|
||||||
// There are preceding unstable versions, can't flush <v>
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
while (true)
|
|
||||||
{
|
|
||||||
dirty_it++;
|
|
||||||
if (dirty_it == dirty_db.end() || dirty_it->first.oid != v.oid ||
|
|
||||||
!IS_SYNCED(dirty_it->second.state))
|
|
||||||
{
|
|
||||||
dirty_it--;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
v.version = dirty_it->first.version;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
while (1)
|
|
||||||
{
|
|
||||||
bool was_stable = IS_STABLE(dirty_it->second.state);
|
|
||||||
if ((dirty_it->second.state & BS_ST_WORKFLOW_MASK) == BS_ST_SYNCED)
|
|
||||||
{
|
|
||||||
dirty_it->second.state = (dirty_it->second.state & ~BS_ST_WORKFLOW_MASK) | BS_ST_STABLE;
|
|
||||||
// Allocations and deletions are counted when they're stabilized
|
|
||||||
if (IS_BIG_WRITE(dirty_it->second.state))
|
|
||||||
{
|
|
||||||
int exists = -1;
|
|
||||||
if (dirty_it != dirty_db.begin())
|
|
||||||
{
|
|
||||||
auto prev_it = dirty_it;
|
|
||||||
prev_it--;
|
|
||||||
if (prev_it->first.oid == v.oid)
|
|
||||||
{
|
|
||||||
exists = IS_DELETE(prev_it->second.state) ? 0 : 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (exists == -1)
|
|
||||||
{
|
|
||||||
auto & clean_db = clean_db_shard(v.oid);
|
|
||||||
auto clean_it = clean_db.find(v.oid);
|
|
||||||
exists = clean_it != clean_db.end() ? 1 : 0;
|
|
||||||
}
|
|
||||||
if (!exists)
|
|
||||||
{
|
|
||||||
uint64_t space_id = dirty_it->first.oid.inode;
|
|
||||||
if (no_inode_stats[dirty_it->first.oid.inode >> (64-POOL_ID_BITS)])
|
|
||||||
space_id = space_id & ~(((uint64_t)1 << (64-POOL_ID_BITS)) - 1);
|
|
||||||
inode_space_stats[space_id] += dsk.data_block_size;
|
|
||||||
used_blocks++;
|
|
||||||
}
|
|
||||||
big_to_flush++;
|
|
||||||
}
|
|
||||||
else if (IS_DELETE(dirty_it->second.state))
|
|
||||||
{
|
|
||||||
uint64_t space_id = dirty_it->first.oid.inode;
|
|
||||||
if (no_inode_stats[dirty_it->first.oid.inode >> (64-POOL_ID_BITS)])
|
|
||||||
space_id = space_id & ~(((uint64_t)1 << (64-POOL_ID_BITS)) - 1);
|
|
||||||
auto & sp = inode_space_stats[space_id];
|
|
||||||
if (sp > dsk.data_block_size)
|
|
||||||
sp -= dsk.data_block_size;
|
|
||||||
else
|
|
||||||
inode_space_stats.erase(space_id);
|
|
||||||
used_blocks--;
|
|
||||||
big_to_flush++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else if (IS_IN_FLIGHT(dirty_it->second.state))
|
|
||||||
{
|
|
||||||
// mark_stable should never be called for in-flight or submitted writes
|
|
||||||
printf(
|
|
||||||
"BUG: Attempt to mark_stable object %jx:%jx v%ju state of which is %x\n",
|
|
||||||
dirty_it->first.oid.inode, dirty_it->first.oid.stripe, dirty_it->first.version,
|
|
||||||
dirty_it->second.state
|
|
||||||
);
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
if (forget_dirty && (IS_BIG_WRITE(dirty_it->second.state) ||
|
|
||||||
IS_DELETE(dirty_it->second.state)))
|
|
||||||
{
|
|
||||||
// Big write overrides all previous dirty entries
|
|
||||||
auto erase_end = dirty_it;
|
|
||||||
while (dirty_it != dirty_db.begin())
|
|
||||||
{
|
|
||||||
dirty_it--;
|
|
||||||
if (dirty_it->first.oid != v.oid)
|
|
||||||
{
|
|
||||||
dirty_it++;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
auto & clean_db = clean_db_shard(v.oid);
|
|
||||||
auto clean_it = clean_db.find(v.oid);
|
|
||||||
uint64_t clean_loc = clean_it != clean_db.end()
|
|
||||||
? clean_it->second.location : UINT64_MAX;
|
|
||||||
erase_dirty(dirty_it, erase_end, clean_loc);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
if (was_stable || dirty_it == dirty_db.begin())
|
|
||||||
{
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
dirty_it--;
|
|
||||||
if (dirty_it->first.oid != v.oid)
|
|
||||||
{
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
flusher->enqueue_flush(v);
|
|
||||||
}
|
|
||||||
auto unstab_it = unstable_writes.find(v.oid);
|
|
||||||
if (unstab_it != unstable_writes.end() &&
|
|
||||||
unstab_it->second <= v.version)
|
|
||||||
{
|
|
||||||
unstable_writes.erase(unstab_it);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
+108
-218
@@ -2,232 +2,122 @@
|
|||||||
// License: VNPL-1.1 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
#include "blockstore_impl.h"
|
#include "blockstore_impl.h"
|
||||||
|
#include "blockstore_internal.h"
|
||||||
#define SYNC_HAS_SMALL 1
|
|
||||||
#define SYNC_HAS_BIG 2
|
|
||||||
#define SYNC_DATA_SYNC_SENT 3
|
|
||||||
#define SYNC_DATA_SYNC_DONE 4
|
|
||||||
#define SYNC_JOURNAL_WRITE_SENT 5
|
|
||||||
#define SYNC_JOURNAL_WRITE_DONE 6
|
|
||||||
#define SYNC_JOURNAL_SYNC_SENT 7
|
|
||||||
#define SYNC_DONE 8
|
|
||||||
|
|
||||||
int blockstore_impl_t::continue_sync(blockstore_op_t *op)
|
int blockstore_impl_t::continue_sync(blockstore_op_t *op)
|
||||||
{
|
{
|
||||||
if (immediate_commit == IMMEDIATE_ALL)
|
if (!PRIV(op)->op_state)
|
||||||
{
|
{
|
||||||
// We can return immediately because sync is only dequeued after all previous writes
|
|
||||||
op->retval = 0;
|
op->retval = 0;
|
||||||
|
}
|
||||||
|
int res = do_sync(op, 0);
|
||||||
|
if (res == 2)
|
||||||
|
{
|
||||||
FINISH_OP(op);
|
FINISH_OP(op);
|
||||||
return 2;
|
|
||||||
}
|
}
|
||||||
if (PRIV(op)->op_state == 0)
|
return res;
|
||||||
{
|
|
||||||
stop_sync_submitted = false;
|
|
||||||
unsynced_big_write_count -= unsynced_big_writes.size();
|
|
||||||
PRIV(op)->sync_big_writes.swap(unsynced_big_writes);
|
|
||||||
PRIV(op)->sync_small_writes.swap(unsynced_small_writes);
|
|
||||||
unsynced_big_writes.clear();
|
|
||||||
unsynced_small_writes.clear();
|
|
||||||
if (PRIV(op)->sync_big_writes.size() > 0)
|
|
||||||
PRIV(op)->op_state = SYNC_HAS_BIG;
|
|
||||||
else if (PRIV(op)->sync_small_writes.size() > 0)
|
|
||||||
PRIV(op)->op_state = SYNC_HAS_SMALL;
|
|
||||||
else
|
|
||||||
PRIV(op)->op_state = SYNC_DONE;
|
|
||||||
}
|
|
||||||
if (PRIV(op)->op_state == SYNC_HAS_SMALL)
|
|
||||||
{
|
|
||||||
// No big writes, just fsync the journal
|
|
||||||
if (journal.sector_info[journal.cur_sector].dirty)
|
|
||||||
{
|
|
||||||
// Write out the last journal sector if it happens to be dirty
|
|
||||||
BS_SUBMIT_CHECK_SQES(1);
|
|
||||||
prepare_journal_sector_write(journal.cur_sector, op);
|
|
||||||
PRIV(op)->op_state = SYNC_JOURNAL_WRITE_SENT;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
PRIV(op)->op_state = SYNC_JOURNAL_WRITE_DONE;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (PRIV(op)->op_state == SYNC_HAS_BIG)
|
|
||||||
{
|
|
||||||
// 1st step: fsync data
|
|
||||||
if (!disable_data_fsync)
|
|
||||||
{
|
|
||||||
BS_SUBMIT_GET_SQE(sqe, data);
|
|
||||||
io_uring_prep_fsync(sqe, dsk.data_fd, IORING_FSYNC_DATASYNC);
|
|
||||||
data->iov = { 0 };
|
|
||||||
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
|
||||||
PRIV(op)->min_flushed_journal_sector = PRIV(op)->max_flushed_journal_sector = 0;
|
|
||||||
PRIV(op)->pending_ops = 1;
|
|
||||||
PRIV(op)->op_state = SYNC_DATA_SYNC_SENT;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
PRIV(op)->op_state = SYNC_DATA_SYNC_DONE;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (PRIV(op)->op_state == SYNC_DATA_SYNC_DONE)
|
|
||||||
{
|
|
||||||
// 2nd step: Data device is synced, prepare & write journal entries
|
|
||||||
// Check space in the journal and journal memory buffers
|
|
||||||
blockstore_journal_check_t space_check(this);
|
|
||||||
if (dsk.csum_block_size)
|
|
||||||
{
|
|
||||||
// More complex check because all journal entries have different lengths
|
|
||||||
int left = PRIV(op)->sync_big_writes.size();
|
|
||||||
for (auto & sbw: PRIV(op)->sync_big_writes)
|
|
||||||
{
|
|
||||||
left--;
|
|
||||||
auto & dirty_entry = dirty_db.at(sbw);
|
|
||||||
uint64_t dyn_size = dsk.dirty_dyn_size(dirty_entry.offset, dirty_entry.len);
|
|
||||||
if (!space_check.check_available(op, 1, sizeof(journal_entry_big_write) + dyn_size, 0))
|
|
||||||
{
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else if (!space_check.check_available(op, PRIV(op)->sync_big_writes.size(),
|
|
||||||
sizeof(journal_entry_big_write) + dsk.clean_entry_bitmap_size, 0))
|
|
||||||
{
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
// Check SQEs. Don't bother about merging, submit each journal sector as a separate request
|
|
||||||
BS_SUBMIT_CHECK_SQES(space_check.sectors_to_write);
|
|
||||||
// Prepare and submit journal entries
|
|
||||||
auto it = PRIV(op)->sync_big_writes.begin();
|
|
||||||
int s = 0;
|
|
||||||
while (it != PRIV(op)->sync_big_writes.end())
|
|
||||||
{
|
|
||||||
auto & dirty_entry = dirty_db.at(*it);
|
|
||||||
uint64_t dyn_size = dsk.dirty_dyn_size(dirty_entry.offset, dirty_entry.len);
|
|
||||||
if (!journal.entry_fits(sizeof(journal_entry_big_write) + dyn_size) &&
|
|
||||||
journal.sector_info[journal.cur_sector].dirty)
|
|
||||||
{
|
|
||||||
prepare_journal_sector_write(journal.cur_sector, op);
|
|
||||||
s++;
|
|
||||||
}
|
|
||||||
journal_entry_big_write *je = (journal_entry_big_write*)prefill_single_journal_entry(
|
|
||||||
journal, (dirty_entry.state & BS_ST_INSTANT) ? JE_BIG_WRITE_INSTANT : JE_BIG_WRITE,
|
|
||||||
sizeof(journal_entry_big_write) + dyn_size
|
|
||||||
);
|
|
||||||
auto jsec = dirty_entry.journal_sector = journal.sector_info[journal.cur_sector].offset;
|
|
||||||
assert(journal.next_free >= journal.used_start
|
|
||||||
? (jsec >= journal.used_start && jsec < journal.next_free)
|
|
||||||
: (jsec >= journal.used_start || jsec < journal.next_free));
|
|
||||||
journal.used_sectors[journal.sector_info[journal.cur_sector].offset]++;
|
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
|
||||||
printf(
|
|
||||||
"journal offset %08jx is used by %jx:%jx v%ju (%ju refs)\n",
|
|
||||||
dirty_entry.journal_sector, it->oid.inode, it->oid.stripe, it->version,
|
|
||||||
journal.used_sectors[journal.sector_info[journal.cur_sector].offset]
|
|
||||||
);
|
|
||||||
#endif
|
|
||||||
je->oid = it->oid;
|
|
||||||
je->version = it->version;
|
|
||||||
je->offset = dirty_entry.offset;
|
|
||||||
je->len = dirty_entry.len;
|
|
||||||
je->location = dirty_entry.location;
|
|
||||||
memcpy((void*)(je+1), (alloc_dyn_data
|
|
||||||
? (uint8_t*)dirty_entry.dyn_data+sizeof(int) : (uint8_t*)&dirty_entry.dyn_data), dyn_size);
|
|
||||||
je->crc32 = je_crc32((journal_entry*)je);
|
|
||||||
journal.crc32_last = je->crc32;
|
|
||||||
it++;
|
|
||||||
}
|
|
||||||
prepare_journal_sector_write(journal.cur_sector, op);
|
|
||||||
s++;
|
|
||||||
assert(s == space_check.sectors_to_write);
|
|
||||||
PRIV(op)->op_state = SYNC_JOURNAL_WRITE_SENT;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
if (PRIV(op)->op_state == SYNC_JOURNAL_WRITE_DONE)
|
|
||||||
{
|
|
||||||
if (!disable_journal_fsync)
|
|
||||||
{
|
|
||||||
BS_SUBMIT_GET_SQE(sqe, data);
|
|
||||||
io_uring_prep_fsync(sqe, dsk.journal_fd, IORING_FSYNC_DATASYNC);
|
|
||||||
data->iov = { 0 };
|
|
||||||
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
|
||||||
PRIV(op)->min_flushed_journal_sector = PRIV(op)->max_flushed_journal_sector = 0;
|
|
||||||
PRIV(op)->pending_ops = 1;
|
|
||||||
PRIV(op)->op_state = SYNC_JOURNAL_SYNC_SENT;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
PRIV(op)->op_state = SYNC_DONE;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (PRIV(op)->op_state == SYNC_DONE)
|
|
||||||
{
|
|
||||||
ack_sync(op);
|
|
||||||
return 2;
|
|
||||||
}
|
|
||||||
return 1;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
void blockstore_impl_t::ack_sync(blockstore_op_t *op)
|
bool blockstore_impl_t::has_unsynced()
|
||||||
{
|
{
|
||||||
// Handle states
|
bool data = (!dsk.disable_data_fsync && unsynced_data_write_count);
|
||||||
for (auto it = PRIV(op)->sync_big_writes.begin(); it != PRIV(op)->sync_big_writes.end(); it++)
|
bool buffer = (!dsk.disable_journal_fsync && unsynced_buffer_write_count);
|
||||||
{
|
bool meta = (!dsk.disable_meta_fsync && unsynced_meta_write_count);
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
return data || buffer || meta;
|
||||||
printf("Ack sync big %jx:%jx v%ju\n", it->oid.inode, it->oid.stripe, it->version);
|
}
|
||||||
#endif
|
|
||||||
auto & unstab = unstable_writes[it->oid];
|
bool blockstore_impl_t::submit_fsyncs(int & wait_count)
|
||||||
unstab = unstab < it->version ? it->version : unstab;
|
{
|
||||||
auto dirty_it = dirty_db.find(*it);
|
int n = (unsynced_meta_write_count > 0 && !dsk.disable_meta_fsync) +
|
||||||
dirty_it->second.state = ((dirty_it->second.state & ~BS_ST_WORKFLOW_MASK) | BS_ST_SYNCED);
|
(unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync && dsk.journal_fd != dsk.meta_fd) +
|
||||||
if (dirty_it->second.state & BS_ST_INSTANT)
|
(unsynced_data_write_count > 0 && !dsk.disable_data_fsync && dsk.data_fd != dsk.meta_fd && dsk.data_fd != dsk.journal_fd);
|
||||||
{
|
if (ringloop->space_left() < n)
|
||||||
mark_stable(dirty_it->first);
|
{
|
||||||
}
|
return false;
|
||||||
else
|
}
|
||||||
{
|
if (!n)
|
||||||
unstable_unsynced--;
|
{
|
||||||
assert(unstable_unsynced >= 0);
|
return true;
|
||||||
}
|
}
|
||||||
dirty_it++;
|
auto cb = [this, & wait_count](ring_data_t *data)
|
||||||
while (dirty_it != dirty_db.end() && dirty_it->first.oid == it->oid)
|
{
|
||||||
{
|
if (data->res != 0)
|
||||||
if ((dirty_it->second.state & BS_ST_WORKFLOW_MASK) == BS_ST_WAIT_BIG)
|
disk_error_abort("sync meta", data->res, 0);
|
||||||
{
|
wait_count--;
|
||||||
dirty_it->second.state = (dirty_it->second.state & ~BS_ST_WORKFLOW_MASK) | BS_ST_IN_FLIGHT;
|
assert(wait_count >= 0);
|
||||||
}
|
if (!wait_count)
|
||||||
dirty_it++;
|
ringloop->wakeup();
|
||||||
}
|
};
|
||||||
}
|
if (unsynced_meta_write_count > 0 && !dsk.disable_meta_fsync)
|
||||||
for (auto it = PRIV(op)->sync_small_writes.begin(); it != PRIV(op)->sync_small_writes.end(); it++)
|
{
|
||||||
{
|
// fsync meta
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
io_uring_sqe *sqe = get_sqe();
|
||||||
printf("Ack sync small %jx:%jx v%ju\n", it->oid.inode, it->oid.stripe, it->version);
|
assert(sqe);
|
||||||
#endif
|
ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
||||||
auto & unstab = unstable_writes[it->oid];
|
io_uring_prep_fsync(sqe, dsk.meta_fd, IORING_FSYNC_DATASYNC);
|
||||||
unstab = unstab < it->version ? it->version : unstab;
|
data->iov = { 0 };
|
||||||
if (dirty_db[*it].state == (BS_ST_DELETE | BS_ST_WRITTEN))
|
data->callback = cb;
|
||||||
{
|
wait_count++;
|
||||||
dirty_db[*it].state = (BS_ST_DELETE | BS_ST_SYNCED);
|
}
|
||||||
// Deletions are treated as immediately stable
|
if (unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync && dsk.meta_fd != dsk.journal_fd)
|
||||||
mark_stable(*it);
|
{
|
||||||
}
|
// fsync buffer
|
||||||
else /* (BS_ST_INSTANT?) | BS_ST_SMALL_WRITE | BS_ST_WRITTEN */
|
io_uring_sqe *sqe = get_sqe();
|
||||||
{
|
assert(sqe);
|
||||||
dirty_db[*it].state = (dirty_db[*it].state & ~BS_ST_WORKFLOW_MASK) | BS_ST_SYNCED;
|
ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
||||||
if (dirty_db[*it].state & BS_ST_INSTANT)
|
io_uring_prep_fsync(sqe, dsk.journal_fd, IORING_FSYNC_DATASYNC);
|
||||||
{
|
data->iov = { 0 };
|
||||||
mark_stable(*it);
|
data->callback = cb;
|
||||||
}
|
wait_count++;
|
||||||
else
|
}
|
||||||
{
|
if (unsynced_data_write_count > 0 && !dsk.disable_data_fsync && dsk.data_fd != dsk.meta_fd && dsk.data_fd != dsk.journal_fd)
|
||||||
unstable_unsynced--;
|
{
|
||||||
assert(unstable_unsynced >= 0);
|
// fsync data
|
||||||
}
|
io_uring_sqe *sqe = get_sqe();
|
||||||
}
|
assert(sqe);
|
||||||
}
|
ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
||||||
op->retval = 0;
|
io_uring_prep_fsync(sqe, dsk.data_fd, IORING_FSYNC_DATASYNC);
|
||||||
FINISH_OP(op);
|
data->iov = { 0 };
|
||||||
|
data->callback = cb;
|
||||||
|
wait_count++;
|
||||||
|
}
|
||||||
|
unsynced_data_write_count = 0;
|
||||||
|
unsynced_buffer_write_count = 0;
|
||||||
|
unsynced_meta_write_count = 0;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
int blockstore_impl_t::do_sync(blockstore_op_t *op, int base_state)
|
||||||
|
{
|
||||||
|
int op_state = PRIV(op)->op_state - base_state;
|
||||||
|
if (op_state == 1) goto resume_1;
|
||||||
|
if (op_state == 2) goto resume_2;
|
||||||
|
assert(!op_state);
|
||||||
|
if (flusher->get_syncing_buffer())
|
||||||
|
{
|
||||||
|
// Wait for flusher-initiated sync
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
if (!has_unsynced())
|
||||||
|
{
|
||||||
|
// We can return immediately because sync only syncs previous writes
|
||||||
|
unsynced_data_write_count = unsynced_buffer_write_count = unsynced_meta_write_count = 0;
|
||||||
|
return 2;
|
||||||
|
}
|
||||||
|
PRIV(op)->modified_block = heap->get_completed_lsn();
|
||||||
|
if (!submit_fsyncs(PRIV(op)->pending_ops))
|
||||||
|
{
|
||||||
|
PRIV(op)->wait_detail = 1;
|
||||||
|
PRIV(op)->wait_for = WAIT_SQE;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
resume_1:
|
||||||
|
if (PRIV(op)->pending_ops > 0)
|
||||||
|
{
|
||||||
|
PRIV(op)->op_state = base_state+1;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
resume_2:
|
||||||
|
heap->mark_lsn_fsynced(PRIV(op)->modified_block);
|
||||||
|
return 2;
|
||||||
}
|
}
|
||||||
|
|||||||
+356
-701
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user