Compare commits
342
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2ccda97d0b | ||
|
|
477ae56287 | ||
|
|
141fbe1054 | ||
|
|
77ba706a73 | ||
|
|
d75334ddf0 | ||
|
|
bf0875128e | ||
|
|
9a6a7b7f75 | ||
|
|
c4c17ee6fb | ||
|
|
2b801a7ffa | ||
|
|
233d2b2a09 | ||
|
|
8380d4c6a6 | ||
|
|
73f9c7293f | ||
|
|
1c66c3e5ba | ||
|
|
eddfa93c18 | ||
|
|
ddd755a0e6 | ||
|
|
819f5b7ec9 | ||
|
|
34d0a6d9b1 | ||
|
|
fe83825ead | ||
|
|
8ec7faa675 | ||
|
|
21cf5c8815 | ||
|
|
44eeb1ed13 | ||
|
|
cc6c445cf0 | ||
|
|
bd6af0db09 | ||
|
|
8860101e99 | ||
|
|
c92661b364 | ||
|
|
9a02a592e3 | ||
|
|
8b7fa3d3bc | ||
|
|
db5eaa2eee | ||
|
|
d35727dbb7 | ||
|
|
b78f526696 | ||
|
|
a23df12260 | ||
|
|
166e16102e | ||
|
|
06c602110c | ||
|
|
1de68c30af | ||
|
|
2a0aca6e94 | ||
|
|
e808332e12 | ||
|
|
fc5a183959 | ||
|
|
55de37e58a | ||
|
|
aacfdf0dec | ||
|
|
a67c415e0f | ||
|
|
495f3fb4cd | ||
|
|
934752d617 | ||
|
|
1e6e233426 | ||
|
|
a5768a8ef6 | ||
|
|
216707f101 | ||
|
|
8ffdb93ed3 | ||
|
|
2fb7022c78 | ||
|
|
2633978fec | ||
|
|
615d4825c0 | ||
|
|
e55ca26ff6 | ||
|
|
abe093b9a3 | ||
|
|
07e6eb0b16 | ||
|
|
0eacce1e1d | ||
|
|
3df410acc7 | ||
|
|
e55927076c | ||
|
|
60fcb168fe | ||
|
|
b17691ba02 | ||
|
|
f2cbe793e2 | ||
|
|
0053546f8b | ||
|
|
959f792f82 | ||
|
|
eaff4509ca | ||
|
|
04eefce30b | ||
|
|
746844d301 | ||
|
|
c8f5b6cb19 | ||
|
|
0f4837e9bb | ||
|
|
25ce82a729 | ||
|
|
944499135f | ||
|
|
08f21edaf7 | ||
|
|
e9f0639e62 | ||
|
|
4e0b203552 | ||
|
|
b772a3cd04 | ||
|
|
856ad79a02 | ||
|
|
39a8772d7f | ||
|
|
993f40de37 | ||
|
|
5607222921 | ||
|
|
378fff6f67 | ||
|
|
daf2cc3fb1 | ||
|
|
17d61c5868 | ||
|
|
e709657de4 | ||
|
|
3eecf9048c | ||
|
|
1852caaeec | ||
|
|
7b40561141 | ||
|
|
d80c12ced2 | ||
|
|
b9713deecd | ||
|
|
852734270e | ||
|
|
8f92979a18 | ||
|
|
b720af74c2 | ||
|
|
96df2966cc | ||
|
|
d3b171e047 | ||
|
|
7287b7fc25 | ||
|
|
e1bb670491 | ||
|
|
1e5a01def8 | ||
|
|
371e630f52 | ||
|
|
0a7ae616f3 | ||
|
|
82f5fb7edd | ||
|
|
ec8527c89d | ||
|
|
5afef7ca6d | ||
|
|
e512e1eeb1 | ||
|
|
7ab60c00ab | ||
|
|
ac8e0ef231 | ||
|
|
94f3634602 | ||
|
|
2b8a9e3f90 | ||
|
|
ef792608b0 | ||
|
|
04531bcfbb | ||
|
|
1b40fa1cee | ||
|
|
3ea9230ed0 | ||
|
|
97dfbfad75 | ||
|
|
c148f97ee4 | ||
|
|
b1f61eb5c8 | ||
|
|
9db8748647 | ||
|
|
e747319c1e | ||
|
|
0cb0e31ceb | ||
|
|
0703efd8b9 | ||
|
|
4a0720a231 | ||
|
|
57d83ecf7c | ||
|
|
d4f7bbb412 | ||
|
|
e5e71fc21a | ||
|
|
0ec4f1608a | ||
|
|
232e416658 | ||
|
|
23d5fec580 | ||
|
|
16881a3d6b | ||
|
|
7cde5c75b9 | ||
|
|
4c32244409 | ||
|
|
30c5a79772 | ||
|
|
6e856c2719 | ||
|
|
af710d3c65 | ||
|
|
8189c3a4ef | ||
|
|
552ba5d885 | ||
|
|
14bbf18ede | ||
|
|
c3eaaa4b94 | ||
|
|
8b35c09e12 | ||
|
|
74b8ea1303 | ||
|
|
6a30e6653a | ||
|
|
2d7da127ad | ||
|
|
4143d56db7 | ||
|
|
5067554e46 | ||
|
|
bc48eb1ff8 | ||
|
|
b0809b33aa | ||
|
|
d61cf2303f | ||
|
|
712f22d6f8 | ||
|
|
4340082315 | ||
|
|
a1a449686a | ||
|
|
fb87870734 | ||
|
|
99bddb976a | ||
|
|
4f997791b8 | ||
|
|
c980168d10 | ||
|
|
dce4a6c37a | ||
|
|
38d5175f66 | ||
|
|
7569fb959b | ||
|
|
aa1e62a502 | ||
|
|
dcbdb0ae33 | ||
|
|
6e5f990801 | ||
|
|
845e76a0ec | ||
|
|
a98aee7906 | ||
|
|
064a94166c | ||
|
|
38112e9012 | ||
|
|
353460cc83 | ||
|
|
92631bb6b3 | ||
|
|
ce050e7eda | ||
|
|
556fc9a876 | ||
|
|
274f9ecda5 | ||
|
|
e0d2705294 | ||
|
|
57d2f30303 | ||
|
|
7f4c541a6f | ||
|
|
b0b495a991 | ||
|
|
da8bf2b73b | ||
|
|
615e9d1274 | ||
|
|
ec5e93307d | ||
|
|
b599334c4a | ||
|
|
bcde273ca1 | ||
|
|
87fe1bc00f | ||
|
|
d4c465f786 | ||
|
|
62995243f3 | ||
|
|
7ea4884ef6 | ||
|
|
8823ddf48e | ||
|
|
db14037ac8 | ||
|
|
64db505357 | ||
|
|
8992eb57df | ||
|
|
f3048d0858 | ||
|
|
714b0783bf | ||
|
|
e8e2aa5dba | ||
|
|
9063bcaa41 | ||
|
|
c392e914e2 | ||
|
|
249e04ac0d | ||
|
|
16dee7c136 | ||
|
|
adddf9b3b1 | ||
|
|
6a5044ae36 | ||
|
|
49407afa17 | ||
|
|
a00fc1bc24 | ||
|
|
3a96d41c93 | ||
|
|
2e4d5ae5bb | ||
|
|
db458fc999 | ||
|
|
574520be0f | ||
|
|
715e7df51f | ||
|
|
ea5ee0c46f | ||
|
|
47bec8af47 | ||
|
|
ec3bf4ae6c | ||
|
|
cb45b1865d | ||
|
|
f656545f4a | ||
|
|
0241a61412 | ||
|
|
0bc81f5320 | ||
|
|
84961f6d0a | ||
|
|
cb085f9c8f | ||
|
|
a9773b1908 | ||
|
|
11783a2d7a | ||
|
|
7915605609 | ||
|
|
3ba3eed0cf | ||
|
|
5dc0b42146 | ||
|
|
b44c3a7971 | ||
|
|
59b7b2e0c3 | ||
|
|
7f0c78113b | ||
|
|
20bbeb4095 | ||
|
|
3c687a2993 | ||
|
|
7530bdbec7 | ||
|
|
c78b4d184d | ||
|
|
c7a6b77c21 | ||
|
|
3cb7ec69bc | ||
|
|
954e7b658d | ||
|
|
2c6ea8a521 | ||
|
|
bfd5575425 | ||
|
|
85e61c9c31 | ||
|
|
f580cee936 | ||
|
|
6d0460500a | ||
|
|
01ae800d34 | ||
|
|
dbb885a6b1 | ||
|
|
8ff2c268f7 | ||
|
|
8430104c19 | ||
|
|
4c11e3ad3d | ||
|
|
d88b49872b | ||
|
|
ca27b91919 | ||
|
|
94f31b96b8 | ||
|
|
76c7c26d32 | ||
|
|
8aa2c49202 | ||
|
|
0f330b10f1 | ||
|
|
477b54a0d8 | ||
|
|
5823a7de66 | ||
|
|
eb0deaa3f5 | ||
|
|
59e6527303 | ||
|
|
67ba9f9b7c | ||
|
|
3ad83e8d13 | ||
|
|
5d3f3f47a7 | ||
|
|
8ee7058ec8 | ||
|
|
1badc6ad13 | ||
|
|
be1858848e | ||
|
|
d75b1cb2d2 | ||
|
|
a1c17d90a3 | ||
|
|
f0112050ce | ||
|
|
d6b8d921d6 | ||
|
|
65872f5d0e | ||
|
|
15eef27d44 | ||
|
|
aa1e51de5f | ||
|
|
c164adb43c | ||
|
|
622631c146 | ||
|
|
4ab93d8481 | ||
|
|
bd64770317 | ||
|
|
34cb48d553 | ||
|
|
724d2ffa04 | ||
|
|
b6bfe1435d | ||
|
|
74a23dcb63 | ||
|
|
93fd23b2bb | ||
|
|
eedc700b83 | ||
|
|
c8cc17dbe9 | ||
|
|
b55d406386 | ||
|
|
0c46dbd333 | ||
|
|
ed94aa52cf | ||
|
|
555ae613c2 | ||
|
|
cad6ea0360 | ||
|
|
d60709dce1 | ||
|
|
a03ffd0d73 | ||
|
|
89b76a87b6 | ||
|
|
ceba343ac0 | ||
|
|
3bc04d8250 | ||
|
|
d228fbfb68 | ||
|
|
e3c8fd28b4 | ||
|
|
d87e7d1a37 | ||
|
|
59f87c3e30 | ||
|
|
eba383f66f | ||
|
|
4e5e8822c0 | ||
|
|
60933c1d00 | ||
|
|
1ad6933953 | ||
|
|
8a250f4fca | ||
|
|
94ddf20667 | ||
|
|
5f18496c04 | ||
|
|
08a3dcd587 | ||
|
|
3c5b9d2744 | ||
|
|
cff08d2c72 | ||
|
|
1e1f395947 | ||
|
|
e6c2628960 | ||
|
|
887f7c1530 | ||
|
|
2c6bddd831 | ||
|
|
e1715c33bb | ||
|
|
2ef80bf0b8 | ||
|
|
85ba710718 | ||
|
|
c16b0e7f92 | ||
|
|
b3d388228a | ||
|
|
bcde9de7da | ||
|
|
52bc3261e9 | ||
|
|
2d42f29385 | ||
|
|
17240c6144 | ||
|
|
9e627a4414 | ||
|
|
90b1019636 | ||
|
|
df604afbd5 | ||
|
|
47c7aa62de | ||
|
|
9f2dc48d0f | ||
|
|
6d951b21fb | ||
|
|
552f28cb3e | ||
|
|
e87b6e26f7 | ||
|
|
0c89886374 | ||
|
|
e79bef8751 | ||
|
|
ad76f84e1c | ||
|
|
db827cb34c | ||
|
|
e5c6d85ea1 | ||
|
|
6cc44c1f54 | ||
|
|
c20450c1f1 | ||
|
|
db63e58b3d | ||
|
|
31b7021330 | ||
|
|
2ebe3a468c | ||
|
|
9892fccfb0 | ||
|
|
0be86a306d | ||
|
|
d77a775948 | ||
|
|
8cc82bab39 | ||
|
|
f9d5e33ddd | ||
|
|
f83418d93e | ||
|
|
fbf14fb0cb | ||
|
|
fb1c3e00f4 | ||
|
|
d8332171e9 | ||
|
|
c24cc9bf0b | ||
|
|
9f57c75acf | ||
|
|
53b12641d1 | ||
|
|
5c5c8825dc | ||
|
|
3a261ac3fc | ||
|
|
04514435de | ||
|
|
07303020fc | ||
|
|
feaf7a15cf | ||
|
|
29dda5066f | ||
|
|
1de53ef7e6 | ||
|
|
4793dbe9c3 | ||
|
|
918ea34af2 | ||
|
|
2db8184cd8 | ||
|
|
0e964b3c8c | ||
|
|
1b9296ff6c | ||
|
|
6bf136c199 |
@@ -20,7 +20,7 @@ RUN echo 'deb http://deb.debian.org/debian bullseye-backports main' >> /etc/apt/
|
|||||||
|
|
||||||
RUN apt-get update
|
RUN apt-get update
|
||||||
RUN apt-get -y install etcd qemu-system-x86 qemu-block-extra qemu-utils fio libasan5 \
|
RUN apt-get -y install etcd qemu-system-x86 qemu-block-extra qemu-utils fio libasan5 \
|
||||||
liburing1 liburing-dev libgoogle-perftools-dev devscripts libjerasure-dev cmake libibverbs-dev libisal-dev
|
libgoogle-perftools-dev devscripts libjerasure-dev cmake libibverbs-dev libisal-dev
|
||||||
RUN apt-get -y build-dep fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
RUN apt-get -y build-dep fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
||||||
RUN apt-get update && apt-get -y install jq lp-solve sudo nfs-common fdisk parted
|
RUN apt-get update && apt-get -y install jq lp-solve sudo nfs-common fdisk parted
|
||||||
RUN apt-get --download-only source fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
RUN apt-get --download-only source fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
||||||
|
|||||||
+796
-4
@@ -306,6 +306,78 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_interrupted_rebalance:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_interrupted_rebalance.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_interrupted_rebalance_imm:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 IMMEDIATE_COMMIT=1 /root/vitastor/tests/test_interrupted_rebalance.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_interrupted_rebalance_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 SCHEME=ec /root/vitastor/tests/test_interrupted_rebalance.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_interrupted_rebalance_ec_imm:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 SCHEME=ec IMMEDIATE_COMMIT=1 /root/vitastor/tests/test_interrupted_rebalance.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_create_halfhost:
|
test_create_halfhost:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -342,6 +414,24 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_level_placement:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_level_placement.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_snapshot:
|
test_snapshot:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -378,6 +468,42 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_snapshot:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_snapshot.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_snapshot_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=ec /root/vitastor/tests/test_snapshot.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_minsize_1:
|
test_minsize_1:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -414,6 +540,42 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_move_reappear:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_move_reappear.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_degraded:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_degraded.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_rm:
|
test_rm:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -486,6 +648,42 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_snapshot_chain:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_snapshot_chain.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_snapshot_chain_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=ec /root/vitastor/tests/test_snapshot_chain.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_snapshot_down:
|
test_snapshot_down:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -522,6 +720,78 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_snapshot_down:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_snapshot_down.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_snapshot_down_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=ec /root/vitastor/tests/test_snapshot_down.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_kv_stress:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_kv_stress.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_kv_stress_imm:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: IMMEDIATE_COMMIT=1 /root/vitastor/tests/test_kv_stress.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_splitbrain:
|
test_splitbrain:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -612,6 +882,78 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_rebalance_verify:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_rebalance_verify.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_rebalance_verify_imm:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 IMMEDIATE_COMMIT=1 /root/vitastor/tests/test_rebalance_verify.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_rebalance_verify_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 SCHEME=ec /root/vitastor/tests/test_rebalance_verify.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_rebalance_verify_ec_imm:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: OLD=1 SCHEME=ec IMMEDIATE_COMMIT=1 /root/vitastor/tests/test_rebalance_verify.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_dd:
|
test_dd:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -720,7 +1062,7 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
test_write_no_same:
|
test_old_write:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
@@ -728,7 +1070,61 @@ jobs:
|
|||||||
- name: Run test
|
- name: Run test
|
||||||
id: test
|
id: test
|
||||||
timeout-minutes: 3
|
timeout-minutes: 3
|
||||||
run: /root/vitastor/tests/test_write_no_same.sh
|
run: OLD=1 /root/vitastor/tests/test_write.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_write_xor:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=xor /root/vitastor/tests/test_write.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_write_old_iothreads:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: TEST_NAME=old_iothreads OLD=1 GLOBAL_CONFIG=',"client_iothread_count":4' /root/vitastor/tests/test_write.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_write_no_same:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_write_no_same.sh
|
||||||
- name: Print logs
|
- name: Print logs
|
||||||
if: always() && steps.test.outcome == 'failure'
|
if: always() && steps.test.outcome == 'failure'
|
||||||
run: |
|
run: |
|
||||||
@@ -810,6 +1206,60 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_reweight_half:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_reweight_half.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_snapshot_pool2:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_snapshot_pool2.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_snapshot_read_bitmap:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_snapshot_read_bitmap.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_heal_csum_32k_dmj:
|
test_heal_csum_32k_dmj:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -954,7 +1404,7 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
test_snapshot_pool2:
|
test_old_resize:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
@@ -962,7 +1412,25 @@ jobs:
|
|||||||
- name: Run test
|
- name: Run test
|
||||||
id: test
|
id: test
|
||||||
timeout-minutes: 3
|
timeout-minutes: 3
|
||||||
run: /root/vitastor/tests/test_snapshot_pool2.sh
|
run: OLD=1 /root/vitastor/tests/test_resize.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_resize_auto:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_resize_auto.sh
|
||||||
- name: Print logs
|
- name: Print logs
|
||||||
if: always() && steps.test.outcome == 'failure'
|
if: always() && steps.test.outcome == 'failure'
|
||||||
run: |
|
run: |
|
||||||
@@ -1062,6 +1530,78 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_enospc:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_enospc.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_enospc_xor:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=xor /root/vitastor/tests/test_enospc.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_enospc_imm:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 IMMEDIATE_COMMIT=1 /root/vitastor/tests/test_enospc.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_enospc_imm_xor:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 IMMEDIATE_COMMIT=1 SCHEME=xor /root/vitastor/tests/test_enospc.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_scrub:
|
test_scrub:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -1170,6 +1710,240 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_scrub:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_scrub.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_scrub_zero_osd_2:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 ZERO_OSD=2 /root/vitastor/tests/test_scrub.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_scrub_xor:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=xor /root/vitastor/tests/test_scrub.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_scrub_pg_size_3:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 PG_SIZE=3 /root/vitastor/tests/test_scrub.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_scrub_pg_size_6_pg_minsize_4_osd_count_6_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 PG_SIZE=6 PG_MINSIZE=4 OSD_COUNT=6 SCHEME=ec /root/vitastor/tests/test_scrub.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_scrub_ec:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 SCHEME=ec /root/vitastor/tests/test_scrub.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_partwr_csum:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_partwr_csum.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_heal_old_csum_32k_dmj:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: TEST_NAME=old_csum_32k_dmj OLD=1 OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k --inmemory_metadata false --inmemory_journal false" OFFSET_ARGS=$OSD_ARGS /root/vitastor/tests/test_heal.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_heal_old_csum_32k_dj:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: TEST_NAME=old_csum_32k_dj OLD=1 OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k --inmemory_journal false" OFFSET_ARGS=$OSD_ARGS /root/vitastor/tests/test_heal.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_heal_old_csum_32k:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: TEST_NAME=old_csum_32k OLD=1 OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k" OFFSET_ARGS=$OSD_ARGS /root/vitastor/tests/test_heal.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_heal_old_csum_4k_dmj:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: TEST_NAME=old_csum_4k_dmj OLD=1 OSD_ARGS="--data_csum_type crc32c --inmemory_metadata false --inmemory_journal false" OFFSET_ARGS=$OSD_ARGS /root/vitastor/tests/test_heal.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_heal_old_csum_4k_dj:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: TEST_NAME=old_csum_4k_dj OLD=1 OSD_ARGS="--data_csum_type crc32c --inmemory_journal false" OFFSET_ARGS=$OSD_ARGS /root/vitastor/tests/test_heal.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_heal_old_csum_4k:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: TEST_NAME=old_csum_4k OLD=1 OSD_ARGS="--data_csum_type crc32c" OFFSET_ARGS=$OSD_ARGS /root/vitastor/tests/test_heal.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_nfs:
|
test_nfs:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -1188,3 +1962,21 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_nfs_unaligned_append:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_nfs_unaligned_append.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
|||||||
@@ -38,6 +38,10 @@ for my $line (<>)
|
|||||||
{
|
{
|
||||||
$test_name .= '_antietcd';
|
$test_name .= '_antietcd';
|
||||||
}
|
}
|
||||||
|
elsif ($1 eq 'OLD')
|
||||||
|
{
|
||||||
|
$test_name =~ s/^test_/test_old_/s;
|
||||||
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
$test_name .= '_'.lc($1).'_'.$2;
|
$test_name .= '_'.lc($1).'_'.$2;
|
||||||
|
|||||||
+14
-1
@@ -2,6 +2,19 @@ cmake_minimum_required(VERSION 2.8.12)
|
|||||||
|
|
||||||
project(vitastor)
|
project(vitastor)
|
||||||
|
|
||||||
set(VITASTOR_VERSION "2.2.3")
|
set(VITASTOR_VERSION "3.0.2")
|
||||||
|
|
||||||
|
include(CTest)
|
||||||
|
|
||||||
|
add_custom_target(build_tests)
|
||||||
|
add_custom_target(test
|
||||||
|
COMMAND
|
||||||
|
echo leak:tcmalloc > ${CMAKE_CURRENT_BINARY_DIR}/lsan-suppress.txt &&
|
||||||
|
env LSAN_OPTIONS=suppressions=${CMAKE_CURRENT_BINARY_DIR}/lsan-suppress.txt ${CMAKE_CTEST_COMMAND}
|
||||||
|
)
|
||||||
|
# make -j16 -C ../../build test_heap && ../../build/src/test/test_heap
|
||||||
|
# make -j16 -C ../../build test_heap && rm -f $(find ../../build -name '*.gcda') && ctest -V -T test -T coverage -R heap --test-dir ../../build && (cd ../../build; gcovr -f ../src --html --html-nested -o coverage/index.html; cd ../src/test)
|
||||||
|
# make -j16 -C ../../build test_blockstore && rm -f $(find ../../build -name '*.gcda') && ctest -V -T test -T coverage -R blockstore --test-dir ../../build && (cd ../../build; gcovr -f ../src --html --html-nested -o coverage/index.html; cd ../src/test)
|
||||||
|
# kcov --include-path=../../../src ../../kcov ./test_blockstore
|
||||||
|
add_dependencies(test build_tests)
|
||||||
add_subdirectory(src)
|
add_subdirectory(src)
|
||||||
|
|||||||
+10
-5
@@ -19,18 +19,22 @@ Vitastor нацелен в первую очередь на SSD и SSD+HDD кл
|
|||||||
TCP и RDMA и на хорошем железе может достигать задержки 4 КБ чтения и записи на уровне ~0.1 мс,
|
TCP и RDMA и на хорошем железе может достигать задержки 4 КБ чтения и записи на уровне ~0.1 мс,
|
||||||
что примерно в 10 раз быстрее, чем Ceph и другие популярные программные СХД.
|
что примерно в 10 раз быстрее, чем Ceph и другие популярные программные СХД.
|
||||||
|
|
||||||
Vitastor поддерживает QEMU-драйвер, протоколы NBD и NFS, драйверы OpenStack, OpenNebula, Proxmox, Kubernetes.
|
Vitastor поддерживает QEMU-драйвер, протоколы UBLK, NBD и NFS, драйверы OpenStack, OpenNebula, Proxmox, Kubernetes.
|
||||||
Другие драйверы могут также быть легко реализованы.
|
Другие драйверы могут также быть легко реализованы.
|
||||||
|
|
||||||
Подробности смотрите в документации по ссылкам. Можете начать отсюда: [Быстрый старт](docs/intro/quickstart.ru.md).
|
Подробности смотрите в документации по ссылкам. Можете начать отсюда: [Быстрый старт](docs/intro/quickstart.ru.md).
|
||||||
|
|
||||||
## Презентации и записи докладов
|
## Презентации и записи докладов
|
||||||
|
|
||||||
|
- KuberConf'2025: [видео](https://vitastor.io/presentation/kuberconf.webm)
|
||||||
|
- Highload'2025: [видео](https://vitastor.io/presentation/hl2025/hl2025.webm),
|
||||||
|
[на youtube](https://www.youtube.com/watch?v=0R8MLjFtz7g), презентация
|
||||||
|
([на русском](https://vitastor.io/presentation/hl2025/), [на английском](https://vitastor.io/presentation/hl2025/en.html))
|
||||||
|
- Highload'2022: презентация ([на русском](https://vitastor.io/presentation/highload/highload.html)),
|
||||||
|
[видео](https://vitastor.io/presentation/highload/talk.webm)
|
||||||
- DevOpsConf'2021: презентация ([на русском](https://vitastor.io/presentation/devopsconf/devopsconf.html),
|
- DevOpsConf'2021: презентация ([на русском](https://vitastor.io/presentation/devopsconf/devopsconf.html),
|
||||||
[на английском](https://vitastor.io/presentation/devopsconf/devopsconf_en.html)),
|
[на английском](https://vitastor.io/presentation/devopsconf/devopsconf_en.html)),
|
||||||
[видео](https://vitastor.io/presentation/devopsconf/talk.webm)
|
[видео](https://vitastor.io/presentation/devopsconf/talk.webm)
|
||||||
- Highload'2022: презентация ([на русском](https://vitastor.io/presentation/highload/highload.html)),
|
|
||||||
[видео](https://vitastor.io/presentation/highload/talk.webm)
|
|
||||||
|
|
||||||
## Документация
|
## Документация
|
||||||
|
|
||||||
@@ -64,8 +68,9 @@ Vitastor поддерживает QEMU-драйвер, протоколы NBD и
|
|||||||
- [vitastor-cli](docs/usage/cli.ru.md) (консольный интерфейс)
|
- [vitastor-cli](docs/usage/cli.ru.md) (консольный интерфейс)
|
||||||
- [vitastor-disk](docs/usage/disk.ru.md) (управление дисками)
|
- [vitastor-disk](docs/usage/disk.ru.md) (управление дисками)
|
||||||
- [fio](docs/usage/fio.ru.md) для тестов производительности
|
- [fio](docs/usage/fio.ru.md) для тестов производительности
|
||||||
- [NBD](docs/usage/nbd.ru.md) для монтирования ядром
|
- [UBLK](docs/usage/ublk.ru.md) для монтирования ядром
|
||||||
- [QEMU и qemu-img](docs/usage/qemu.ru.md)
|
- [NBD](docs/usage/nbd.ru.md) - старый интерфейс для монтирования ядром
|
||||||
|
- [QEMU, qemu-img и VDUSE](docs/usage/qemu.ru.md)
|
||||||
- [NFS](docs/usage/nfs.ru.md) кластерная файловая система и псевдо-ФС прокси
|
- [NFS](docs/usage/nfs.ru.md) кластерная файловая система и псевдо-ФС прокси
|
||||||
- [Администрирование](docs/usage/admin.ru.md)
|
- [Администрирование](docs/usage/admin.ru.md)
|
||||||
- Производительность
|
- Производительность
|
||||||
|
|||||||
@@ -19,18 +19,22 @@ supports TCP and RDMA and may achieve 4 KB read and write latency as low as ~0.1
|
|||||||
with proper hardware which is ~10 times faster than other popular SDS's like Ceph
|
with proper hardware which is ~10 times faster than other popular SDS's like Ceph
|
||||||
or internal systems of public clouds.
|
or internal systems of public clouds.
|
||||||
|
|
||||||
Vitastor supports QEMU, NBD, NFS protocols, OpenStack, OpenNebula, Proxmox, Kubernetes drivers.
|
Vitastor supports QEMU, UBLK, NBD, NFS protocols, OpenStack, OpenNebula, Proxmox, Kubernetes drivers.
|
||||||
More drivers may be created easily.
|
More drivers may be created easily.
|
||||||
|
|
||||||
Read more details in the documentation. You can start from here: [Quick Start](docs/intro/quickstart.en.md).
|
Read more details in the documentation. You can start from here: [Quick Start](docs/intro/quickstart.en.md).
|
||||||
|
|
||||||
## Talks and presentations
|
## Talks and presentations
|
||||||
|
|
||||||
|
- KuberConf'2025: [video](https://vitastor.io/presentation/kuberconf.webm)
|
||||||
|
- Highload'2025: [video](https://vitastor.io/presentation/hl2025/hl2025.webm),
|
||||||
|
[youtube](https://www.youtube.com/watch?v=0R8MLjFtz7g), presentation
|
||||||
|
([in Russian](https://vitastor.io/presentation/hl2025/), [in English](https://vitastor.io/presentation/hl2025/en.html))
|
||||||
|
- Highload'2022: presentation ([in Russian](https://vitastor.io/presentation/highload/highload.html)),
|
||||||
|
[video](https://vitastor.io/presentation/highload/talk.webm)
|
||||||
- DevOpsConf'2021: presentation ([in Russian](https://vitastor.io/presentation/devopsconf/devopsconf.html),
|
- DevOpsConf'2021: presentation ([in Russian](https://vitastor.io/presentation/devopsconf/devopsconf.html),
|
||||||
[in English](https://vitastor.io/presentation/devopsconf/devopsconf_en.html)),
|
[in English](https://vitastor.io/presentation/devopsconf/devopsconf_en.html)),
|
||||||
[video](https://vitastor.io/presentation/devopsconf/talk.webm)
|
[video](https://vitastor.io/presentation/devopsconf/talk.webm)
|
||||||
- Highload'2022: presentation ([in Russian](https://vitastor.io/presentation/highload/highload.html)),
|
|
||||||
[video](https://vitastor.io/presentation/highload/talk.webm)
|
|
||||||
|
|
||||||
## Documentation
|
## Documentation
|
||||||
|
|
||||||
@@ -64,8 +68,9 @@ Read more details in the documentation. You can start from here: [Quick Start](d
|
|||||||
- [vitastor-cli](docs/usage/cli.en.md) (command-line interface)
|
- [vitastor-cli](docs/usage/cli.en.md) (command-line interface)
|
||||||
- [vitastor-disk](docs/usage/disk.en.md) (disk management tool)
|
- [vitastor-disk](docs/usage/disk.en.md) (disk management tool)
|
||||||
- [fio](docs/usage/fio.en.md) for benchmarks
|
- [fio](docs/usage/fio.en.md) for benchmarks
|
||||||
- [NBD](docs/usage/nbd.en.md) for kernel mounts
|
- [UBLK](docs/usage/ublk.en.md) for kernel mounts
|
||||||
- [QEMU and qemu-img](docs/usage/qemu.en.md)
|
- [NBD](docs/usage/nbd.en.md) - old interface for kernel mounts
|
||||||
|
- [QEMU, qemu-img and VDUSE](docs/usage/qemu.en.md)
|
||||||
- [NFS](docs/usage/nfs.en.md) clustered file system and pseudo-FS proxy
|
- [NFS](docs/usage/nfs.en.md) clustered file system and pseudo-FS proxy
|
||||||
- [Administration](docs/usage/admin.en.md)
|
- [Administration](docs/usage/admin.en.md)
|
||||||
- Performance
|
- Performance
|
||||||
|
|||||||
+1
-1
@@ -36,7 +36,7 @@ RUN (echo deb http://vitastor.io/debian bookworm main > /etc/apt/sources.list.d/
|
|||||||
((echo 'Package: *'; echo 'Pin: origin "vitastor.io"'; echo 'Pin-Priority: 1000') > /etc/apt/preferences.d/vitastor.pref) && \
|
((echo 'Package: *'; echo 'Pin: origin "vitastor.io"'; echo 'Pin-Priority: 1000') > /etc/apt/preferences.d/vitastor.pref) && \
|
||||||
wget -q -O /etc/apt/trusted.gpg.d/vitastor.gpg https://vitastor.io/debian/pubkey.gpg && \
|
wget -q -O /etc/apt/trusted.gpg.d/vitastor.gpg https://vitastor.io/debian/pubkey.gpg && \
|
||||||
apt-get update && \
|
apt-get update && \
|
||||||
apt-get install -y vitastor-client && \
|
apt-get install -y vitastor-client ibverbs-providers && \
|
||||||
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||||
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||||
dpkg -x qemu-utils*.deb tmp1 && \
|
dpkg -x qemu-utils*.deb tmp1 && \
|
||||||
|
|||||||
@@ -0,0 +1,49 @@
|
|||||||
|
# Compile stage
|
||||||
|
FROM golang:bookworm AS build
|
||||||
|
|
||||||
|
ADD go.sum go.mod /app/
|
||||||
|
RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go mod download -x
|
||||||
|
ADD . /app
|
||||||
|
RUN perl -i -e '$/ = undef; while(<>) { s/\n\s*(\{\s*\n)/$1\n/g; s/\}(\s*\n\s*)else\b/$1} else/g; print; }' `find /app -name '*.go'` && \
|
||||||
|
cd /app && \
|
||||||
|
CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o vitastor-csi
|
||||||
|
|
||||||
|
# Final stage
|
||||||
|
FROM debian:bookworm
|
||||||
|
|
||||||
|
LABEL maintainers="Vitaliy Filippov <vitalif@yourcmc.ru>"
|
||||||
|
LABEL description="Vitastor CSI Driver"
|
||||||
|
|
||||||
|
ENV NODE_ID=""
|
||||||
|
ENV CSI_ENDPOINT=""
|
||||||
|
|
||||||
|
RUN apt-get update && \
|
||||||
|
apt-get install -y wget && \
|
||||||
|
(echo "APT::Install-Recommends false;" > /etc/apt/apt.conf) && \
|
||||||
|
apt-get update && \
|
||||||
|
apt-get install -y e2fsprogs xfsprogs kmod iproute2 \
|
||||||
|
# NFS mount dependencies
|
||||||
|
nfs-common netbase \
|
||||||
|
# dependencies of qemu-storage-daemon
|
||||||
|
libnuma1 liburing2 libglib2.0-0 libfuse3-3 libaio1 libzstd1 libnettle8 \
|
||||||
|
libgmp10 libhogweed6 libp11-kit0 libidn2-0 libunistring2 libtasn1-6 libpcre2-8-0 libffi8 && \
|
||||||
|
apt-get clean && \
|
||||||
|
(echo options nbd nbds_max=128 > /etc/modprobe.d/nbd.conf)
|
||||||
|
|
||||||
|
COPY --from=build /app/vitastor-csi /bin/
|
||||||
|
|
||||||
|
ADD deb /deb
|
||||||
|
|
||||||
|
RUN apt-get update && \
|
||||||
|
apt-get -y install /deb/vitastor-client_*.deb && \
|
||||||
|
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||||
|
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||||
|
dpkg -x qemu-utils*.deb tmp1 && \
|
||||||
|
dpkg -x qemu-block-extra*.deb tmp1 && \
|
||||||
|
cp -a tmp1/usr/bin/qemu-storage-daemon /usr/bin/ && \
|
||||||
|
mkdir -p /usr/lib/x86_64-linux-gnu/qemu && \
|
||||||
|
cp -a tmp1/usr/lib/x86_64-linux-gnu/qemu/block-vitastor.so /usr/lib/x86_64-linux-gnu/qemu/ && \
|
||||||
|
rm -rf tmp1 *.deb && \
|
||||||
|
apt-get clean
|
||||||
|
|
||||||
|
ENTRYPOINT ["/bin/vitastor-csi"]
|
||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
VITASTOR_VERSION ?= v2.2.3
|
VITASTOR_VERSION ?= v3.0.2
|
||||||
|
|
||||||
all: build push
|
all: build push
|
||||||
|
|
||||||
|
|||||||
@@ -49,7 +49,7 @@ spec:
|
|||||||
capabilities:
|
capabilities:
|
||||||
add: ["SYS_ADMIN"]
|
add: ["SYS_ADMIN"]
|
||||||
allowPrivilegeEscalation: true
|
allowPrivilegeEscalation: true
|
||||||
image: vitalif/vitastor-csi:v2.2.3
|
image: vitalif/vitastor-csi:v3.0.2
|
||||||
args:
|
args:
|
||||||
- "--node=$(NODE_ID)"
|
- "--node=$(NODE_ID)"
|
||||||
- "--endpoint=$(CSI_ENDPOINT)"
|
- "--endpoint=$(CSI_ENDPOINT)"
|
||||||
|
|||||||
@@ -121,7 +121,7 @@ spec:
|
|||||||
privileged: true
|
privileged: true
|
||||||
capabilities:
|
capabilities:
|
||||||
add: ["SYS_ADMIN"]
|
add: ["SYS_ADMIN"]
|
||||||
image: vitalif/vitastor-csi:v2.2.3
|
image: vitalif/vitastor-csi:v3.0.2
|
||||||
args:
|
args:
|
||||||
- "--node=$(NODE_ID)"
|
- "--node=$(NODE_ID)"
|
||||||
- "--endpoint=$(CSI_ENDPOINT)"
|
- "--endpoint=$(CSI_ENDPOINT)"
|
||||||
|
|||||||
+1
-1
@@ -5,7 +5,7 @@ package vitastor
|
|||||||
|
|
||||||
const (
|
const (
|
||||||
vitastorCSIDriverName = "csi.vitastor.io"
|
vitastorCSIDriverName = "csi.vitastor.io"
|
||||||
vitastorCSIDriverVersion = "2.2.3"
|
vitastorCSIDriverVersion = "3.0.2"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Config struct fills the parameters of request or user input
|
// Config struct fills the parameters of request or user input
|
||||||
|
|||||||
+115
-20
@@ -33,7 +33,7 @@ import (
|
|||||||
type NodeServer struct
|
type NodeServer struct
|
||||||
{
|
{
|
||||||
*Driver
|
*Driver
|
||||||
useVduse bool
|
method MountMethod
|
||||||
stateDir string
|
stateDir string
|
||||||
nfsStageDir string
|
nfsStageDir string
|
||||||
mounter mount.Interface
|
mounter mount.Interface
|
||||||
@@ -81,16 +81,23 @@ func NewNodeServer(driver *Driver) *NodeServer
|
|||||||
}
|
}
|
||||||
ns := &NodeServer{
|
ns := &NodeServer{
|
||||||
Driver: driver,
|
Driver: driver,
|
||||||
useVduse: checkVduseSupport(),
|
method: selectMountMethod(),
|
||||||
stateDir: stateDir,
|
stateDir: stateDir,
|
||||||
nfsStageDir: nfsStageDir,
|
nfsStageDir: nfsStageDir,
|
||||||
mounter: mount.New(""),
|
mounter: mount.New(""),
|
||||||
volumeLocks: make(map[string]bool),
|
volumeLocks: make(map[string]bool),
|
||||||
}
|
}
|
||||||
ns.cond = sync.NewCond(&ns.mu)
|
ns.cond = sync.NewCond(&ns.mu)
|
||||||
if (ns.useVduse)
|
if (ns.method == MOUNT_VDUSE)
|
||||||
{
|
{
|
||||||
ns.restoreVduseDaemons()
|
ns.restoreVduseDaemons()
|
||||||
|
}
|
||||||
|
else if (ns.method == MOUNT_UBLK)
|
||||||
|
{
|
||||||
|
ns.restoreUblkDaemons()
|
||||||
|
}
|
||||||
|
if (ns.method == MOUNT_VDUSE || ns.method == MOUNT_UBLK)
|
||||||
|
{
|
||||||
dur, err := time.ParseDuration(os.Getenv("RESTART_INTERVAL"))
|
dur, err := time.ParseDuration(os.Getenv("RESTART_INTERVAL"))
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
@@ -136,7 +143,14 @@ func (ns *NodeServer) restarter()
|
|||||||
for
|
for
|
||||||
{
|
{
|
||||||
<-ticker.C
|
<-ticker.C
|
||||||
ns.restoreVduseDaemons()
|
if (ns.method == MOUNT_VDUSE)
|
||||||
|
{
|
||||||
|
ns.restoreVduseDaemons()
|
||||||
|
}
|
||||||
|
else if (ns.method == MOUNT_UBLK)
|
||||||
|
{
|
||||||
|
ns.restoreUblkDaemons()
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -231,6 +245,78 @@ func (ns *NodeServer) checkVduseState(stateFile string, devs map[string]interfac
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (ns *NodeServer) restoreUblkDaemons()
|
||||||
|
{
|
||||||
|
pattern := ns.stateDir+"vitastor-ublk-*.json"
|
||||||
|
stateFiles, err := filepath.Glob(pattern)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to list %s: %v", pattern, err)
|
||||||
|
}
|
||||||
|
if (len(stateFiles) == 0)
|
||||||
|
{
|
||||||
|
return
|
||||||
|
}
|
||||||
|
for _, stateFile := range stateFiles
|
||||||
|
{
|
||||||
|
deviceNum := stateFile[len(ns.stateDir) + len("vitastor-ublk-") :]
|
||||||
|
deviceNum = deviceNum[0:len(deviceNum)-5]
|
||||||
|
ns.checkUblkState(deviceNum)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (ns *NodeServer) checkUblkState(deviceNum string)
|
||||||
|
{
|
||||||
|
// Check if the ublk daemon is still active
|
||||||
|
|
||||||
|
// Read state file
|
||||||
|
stateFile := ns.stateDir + "vitastor-ublk-" + deviceNum + ".json"
|
||||||
|
stateJSON, err := os.ReadFile(stateFile)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Warningf("error reading state file %v: %v", stateFile, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
var state DeviceState
|
||||||
|
err = json.Unmarshal(stateJSON, &state)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Warningf("state file %v contains invalid JSON (error %v): %v", stateFile, err, string(stateJSON))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Lock volume
|
||||||
|
ns.lockVolume(state.ConfigPath+":block:"+state.Image)
|
||||||
|
defer ns.unlockVolume(state.ConfigPath+":block:"+state.Image)
|
||||||
|
|
||||||
|
// Recheck state file after locking
|
||||||
|
_, err = os.ReadFile(stateFile)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Warningf("state file %v disappeared, skipping volume", stateFile)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Check if the vitastor-ublk process is still active
|
||||||
|
pidFile := ns.stateDir + "vitastor-ublk-" + deviceNum + ".pid"
|
||||||
|
exists := false
|
||||||
|
proc, err := findByPidFile(pidFile)
|
||||||
|
if (err == nil)
|
||||||
|
{
|
||||||
|
exists = proc.Signal(syscall.Signal(0)) == nil
|
||||||
|
}
|
||||||
|
if (!exists)
|
||||||
|
{
|
||||||
|
// Restart daemon
|
||||||
|
klog.Warningf("recovering UBLK device /dev/ublkb%v for volume %v", deviceNum, state.Image)
|
||||||
|
_, err = mapUblk(ns.stateDir, state.Image, state.ConfigPath, state.Readonly, "/dev/ublkb"+deviceNum)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Warningf("failed to recover ublk device for volume %v: %v", state.Image, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func (ns *NodeServer) restoreNfsDaemons()
|
func (ns *NodeServer) restoreNfsDaemons()
|
||||||
{
|
{
|
||||||
pattern := ns.stateDir+"vitastor-nfs-*.json"
|
pattern := ns.stateDir+"vitastor-nfs-*.json"
|
||||||
@@ -417,14 +503,18 @@ func (ns *NodeServer) NodeStageVolume(ctx context.Context, req *csi.NodeStageVol
|
|||||||
}
|
}
|
||||||
|
|
||||||
var devicePath, vdpaId string
|
var devicePath, vdpaId string
|
||||||
if (!ns.useVduse)
|
if (ns.method == MOUNT_UBLK)
|
||||||
{
|
{
|
||||||
devicePath, err = mapNbd(volName, ctxVars, false)
|
devicePath, err = mapUblk(ns.stateDir, volName, ctxVars["configPath"], false, "")
|
||||||
}
|
}
|
||||||
else
|
else if (ns.method == MOUNT_VDUSE)
|
||||||
{
|
{
|
||||||
devicePath, vdpaId, err = mapVduse(ns.stateDir, volName, ctxVars, false)
|
devicePath, vdpaId, err = mapVduse(ns.stateDir, volName, ctxVars, false)
|
||||||
}
|
}
|
||||||
|
else /* if (ns.method == MOUNT_NBD) */
|
||||||
|
{
|
||||||
|
devicePath, err = mapNbd(volName, ctxVars, false)
|
||||||
|
}
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
return nil, err
|
return nil, err
|
||||||
@@ -439,7 +529,8 @@ func (ns *NodeServer) NodeStageVolume(ctx context.Context, req *csi.NodeStageVol
|
|||||||
else
|
else
|
||||||
{
|
{
|
||||||
// Check existing format
|
// Check existing format
|
||||||
existingFormat, err := diskMounter.GetDiskFormat(devicePath)
|
var existingFormat string
|
||||||
|
existingFormat, err = diskMounter.GetDiskFormat(devicePath)
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
klog.Errorf("failed to get disk format for path %s, error: %v", err)
|
klog.Errorf("failed to get disk format for path %s, error: %v", err)
|
||||||
@@ -495,10 +586,6 @@ func (ns *NodeServer) NodeStageVolume(ctx context.Context, req *csi.NodeStageVol
|
|||||||
case "xfs":
|
case "xfs":
|
||||||
_, err = systemCombined("xfs_growfs", devicePath)
|
_, err = systemCombined("xfs_growfs", devicePath)
|
||||||
}
|
}
|
||||||
if (err != nil)
|
|
||||||
{
|
|
||||||
goto unmap
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
@@ -512,14 +599,18 @@ func (ns *NodeServer) NodeStageVolume(ctx context.Context, req *csi.NodeStageVol
|
|||||||
return &csi.NodeStageVolumeResponse{}, nil
|
return &csi.NodeStageVolumeResponse{}, nil
|
||||||
|
|
||||||
unmap:
|
unmap:
|
||||||
if (!ns.useVduse || len(devicePath) >= 8 && devicePath[0:8] == "/dev/nbd")
|
if (ns.method == MOUNT_UBLK)
|
||||||
{
|
{
|
||||||
unmapNbd(devicePath)
|
unmapUblk(ns.stateDir, devicePath)
|
||||||
}
|
}
|
||||||
else
|
else if (ns.method == MOUNT_VDUSE)
|
||||||
{
|
{
|
||||||
unmapVduseById(ns.stateDir, vdpaId)
|
unmapVduseById(ns.stateDir, vdpaId)
|
||||||
}
|
}
|
||||||
|
else /* if (ns.method == MOUNT_NBD) */
|
||||||
|
{
|
||||||
|
unmapNbd(devicePath)
|
||||||
|
}
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -545,7 +636,7 @@ func (ns *NodeServer) NodeUnstageVolume(ctx context.Context, req *csi.NodeUnstag
|
|||||||
defer ns.unlockVolume(ctxVars["configPath"]+":block:"+volName)
|
defer ns.unlockVolume(ctxVars["configPath"]+":block:"+volName)
|
||||||
|
|
||||||
targetPath := req.GetStagingTargetPath()
|
targetPath := req.GetStagingTargetPath()
|
||||||
devicePath, _, err := mount.GetDeviceNameFromMount(ns.mounter, targetPath)
|
devicePath, err := GetDeviceNameFromMount(targetPath)
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
if (os.IsNotExist(err))
|
if (os.IsNotExist(err))
|
||||||
@@ -582,14 +673,18 @@ func (ns *NodeServer) NodeUnstageVolume(ctx context.Context, req *csi.NodeUnstag
|
|||||||
// unmap device
|
// unmap device
|
||||||
if (len(refList) == 0)
|
if (len(refList) == 0)
|
||||||
{
|
{
|
||||||
if (!ns.useVduse)
|
if (ns.method == MOUNT_UBLK)
|
||||||
{
|
{
|
||||||
unmapNbd(devicePath)
|
unmapUblk(ns.stateDir, devicePath)
|
||||||
}
|
}
|
||||||
else
|
else if (ns.method == MOUNT_VDUSE)
|
||||||
{
|
{
|
||||||
unmapVduse(ns.stateDir, devicePath)
|
unmapVduse(ns.stateDir, devicePath)
|
||||||
}
|
}
|
||||||
|
else /* if (ns.method == MOUNT_NBD) */
|
||||||
|
{
|
||||||
|
unmapNbd(devicePath)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return &csi.NodeUnstageVolumeResponse{}, nil
|
return &csi.NodeUnstageVolumeResponse{}, nil
|
||||||
@@ -897,7 +992,7 @@ func (ns *NodeServer) NodeUnpublishVolume(ctx context.Context, req *csi.NodeUnpu
|
|||||||
}
|
}
|
||||||
|
|
||||||
targetPath := req.GetTargetPath()
|
targetPath := req.GetTargetPath()
|
||||||
devicePath, _, err := mount.GetDeviceNameFromMount(ns.mounter, targetPath)
|
devicePath, err := GetDeviceNameFromMount(targetPath)
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
if (os.IsNotExist(err))
|
if (os.IsNotExist(err))
|
||||||
|
|||||||
+205
-26
@@ -16,10 +16,20 @@ import (
|
|||||||
"syscall"
|
"syscall"
|
||||||
|
|
||||||
"k8s.io/klog"
|
"k8s.io/klog"
|
||||||
|
"k8s.io/utils/mount"
|
||||||
|
|
||||||
"google.golang.org/grpc/codes"
|
"google.golang.org/grpc/codes"
|
||||||
"google.golang.org/grpc/status"
|
"google.golang.org/grpc/status"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
type MountMethod int
|
||||||
|
|
||||||
|
const (
|
||||||
|
MOUNT_NBD MountMethod = 0
|
||||||
|
MOUNT_VDUSE MountMethod = 1
|
||||||
|
MOUNT_UBLK MountMethod = 2
|
||||||
|
)
|
||||||
|
|
||||||
func Contains(list []string, s string) bool
|
func Contains(list []string, s string) bool
|
||||||
{
|
{
|
||||||
for i := 0; i < len(list); i++
|
for i := 0; i < len(list); i++
|
||||||
@@ -32,29 +42,26 @@ func Contains(list []string, s string) bool
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
func checkVduseSupport() bool
|
func selectMountMethod() MountMethod
|
||||||
{
|
{
|
||||||
|
// Check UBLK support (ublk_drv kernel module)
|
||||||
|
if (checkModule("ublk_drv"))
|
||||||
|
{
|
||||||
|
klog.Infof("UBLK support enabled successfully")
|
||||||
|
return MOUNT_UBLK
|
||||||
|
}
|
||||||
|
klog.Errorf(
|
||||||
|
"Your host apparently has no UBLK support. UBLK support disabled."+
|
||||||
|
" For UBLK you need at least Linux 6.0 and the ublk_drv kernel module.",
|
||||||
|
)
|
||||||
// Check VDUSE support (vdpa, vduse, virtio-vdpa kernel modules)
|
// Check VDUSE support (vdpa, vduse, virtio-vdpa kernel modules)
|
||||||
vduse := true
|
vduse := true
|
||||||
for _, mod := range []string{"vdpa", "vduse", "virtio-vdpa"}
|
for _, mod := range []string{"vdpa", "vduse", "virtio-vdpa"}
|
||||||
{
|
{
|
||||||
_, err := os.Stat("/sys/module/"+mod)
|
if (!checkModule(mod))
|
||||||
if (err != nil)
|
|
||||||
{
|
{
|
||||||
if (!errors.Is(err, os.ErrNotExist))
|
vduse = false
|
||||||
{
|
break
|
||||||
klog.Errorf("failed to check /sys/module/%s: %v", mod, err)
|
|
||||||
}
|
|
||||||
c := exec.Command("/sbin/modprobe", mod)
|
|
||||||
c.Stdout = os.Stderr
|
|
||||||
c.Stderr = os.Stderr
|
|
||||||
err := c.Run()
|
|
||||||
if (err != nil)
|
|
||||||
{
|
|
||||||
klog.Errorf("/sbin/modprobe %s failed: %v", mod, err)
|
|
||||||
vduse = false
|
|
||||||
break
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Check that vdpa tool functions
|
// Check that vdpa tool functions
|
||||||
@@ -69,18 +76,38 @@ func checkVduseSupport() bool
|
|||||||
vduse = false
|
vduse = false
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!vduse)
|
if (vduse)
|
||||||
{
|
|
||||||
klog.Errorf(
|
|
||||||
"Your host apparently has no VDUSE support. VDUSE support disabled, NBD will be used to map devices."+
|
|
||||||
" For VDUSE you need at least Linux 5.15 and the following kernel modules: vdpa, virtio-vdpa, vduse.",
|
|
||||||
)
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
{
|
||||||
klog.Infof("VDUSE support enabled successfully")
|
klog.Infof("VDUSE support enabled successfully")
|
||||||
|
return MOUNT_VDUSE
|
||||||
}
|
}
|
||||||
return vduse
|
klog.Errorf(
|
||||||
|
"Your host apparently has no VDUSE support. VDUSE support disabled, NBD will be used to map devices."+
|
||||||
|
" For VDUSE you need at least Linux 5.15 and the following kernel modules: vdpa, virtio-vdpa, vduse.",
|
||||||
|
)
|
||||||
|
return MOUNT_NBD
|
||||||
|
}
|
||||||
|
|
||||||
|
func checkModule(mod string) bool
|
||||||
|
{
|
||||||
|
_, err := os.Stat("/sys/module/"+mod)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
if (!errors.Is(err, os.ErrNotExist))
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to check /sys/module/%s: %v", mod, err)
|
||||||
|
}
|
||||||
|
c := exec.Command("/sbin/modprobe", mod)
|
||||||
|
c.Stdout = os.Stderr
|
||||||
|
c.Stderr = os.Stderr
|
||||||
|
err := c.Run()
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("/sbin/modprobe %s failed: %v", mod, err)
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
func mapNbd(volName string, ctxVars map[string]string, readonly bool) (string, error)
|
func mapNbd(volName string, ctxVars map[string]string, readonly bool) (string, error)
|
||||||
@@ -217,6 +244,7 @@ func mapVduse(stateDir string, volName string, ctxVars map[string]string, readon
|
|||||||
stateJSON, _ := json.Marshal(&DeviceState{
|
stateJSON, _ := json.Marshal(&DeviceState{
|
||||||
ConfigPath: ctxVars["configPath"],
|
ConfigPath: ctxVars["configPath"],
|
||||||
VdpaId: vdpaId,
|
VdpaId: vdpaId,
|
||||||
|
|
||||||
Image: volName,
|
Image: volName,
|
||||||
Blockdev: blockdev,
|
Blockdev: blockdev,
|
||||||
Readonly: readonly,
|
Readonly: readonly,
|
||||||
@@ -309,6 +337,117 @@ func unmapVduseById(stateDir, vdpaId string)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func mapUblk(stateDir string, volName string, configPath string, readonly bool, recoverDev string) (string, error)
|
||||||
|
{
|
||||||
|
pidFile := ""
|
||||||
|
if (recoverDev != "")
|
||||||
|
{
|
||||||
|
if (len(recoverDev) < 10 || recoverDev[0:10] != "/dev/ublkb")
|
||||||
|
{
|
||||||
|
return "", fmt.Errorf("recover: %s does not start with /dev/ublkb", recoverDev)
|
||||||
|
}
|
||||||
|
pidFile = stateDir + "vitastor-ublk-" + recoverDev[10:] + ".pid"
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
pidFd, err := os.CreateTemp(stateDir, "vitastor-tmp-*.pid")
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
pidFile = pidFd.Name()
|
||||||
|
pidFd.Close()
|
||||||
|
}
|
||||||
|
// Map device via vitastor-ublk
|
||||||
|
args := []string{
|
||||||
|
"map", "--image", volName, "--pidfile", pidFile,
|
||||||
|
}
|
||||||
|
if (configPath != "")
|
||||||
|
{
|
||||||
|
args = append(args, "--config_path", configPath)
|
||||||
|
}
|
||||||
|
if (readonly)
|
||||||
|
{
|
||||||
|
args = append(args, "--readonly")
|
||||||
|
}
|
||||||
|
if (recoverDev != "")
|
||||||
|
{
|
||||||
|
args = append(args, "--recover", recoverDev)
|
||||||
|
}
|
||||||
|
stdout, stderr, err := system("/usr/bin/vitastor-ublk", args...)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
devicePath := strings.TrimSpace(string(stdout))
|
||||||
|
if (devicePath == "")
|
||||||
|
{
|
||||||
|
return "", fmt.Errorf("vitastor-ublk did not return the name of the device. output: %s", stderr)
|
||||||
|
}
|
||||||
|
if (len(devicePath) >= 10 && devicePath[0:10] == "/dev/ublkb")
|
||||||
|
{
|
||||||
|
// Generate state file
|
||||||
|
devNum := devicePath[10:]
|
||||||
|
pidNew := stateDir + "vitastor-ublk-" + devNum + ".pid"
|
||||||
|
if (pidFile != pidNew)
|
||||||
|
{
|
||||||
|
err := os.Rename(pidFile, pidNew)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("Failed to rename PID file %s to %s: %v", pidFile, pidNew, err)
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
pidFile = pidNew
|
||||||
|
}
|
||||||
|
}
|
||||||
|
stateFile := stateDir + "vitastor-ublk-" + devNum + ".json"
|
||||||
|
stateJSON, _ := json.Marshal(&DeviceState{
|
||||||
|
ConfigPath: configPath,
|
||||||
|
Image: volName,
|
||||||
|
Readonly: readonly,
|
||||||
|
PidFile: pidFile,
|
||||||
|
})
|
||||||
|
err = os.WriteFile(stateFile, stateJSON, 0600)
|
||||||
|
if (err == nil)
|
||||||
|
{
|
||||||
|
klog.Infof("Attached volume %s via UBLK as %s", volName, devicePath)
|
||||||
|
return devicePath, nil
|
||||||
|
}
|
||||||
|
os.Remove(stateFile)
|
||||||
|
}
|
||||||
|
killErr := killByPidFile(pidFile)
|
||||||
|
if (killErr != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("Failed to kill started vitastor-ublk: %v", killErr)
|
||||||
|
}
|
||||||
|
os.Remove(pidFile)
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
func unmapUblk(stateDir, devicePath string)
|
||||||
|
{
|
||||||
|
if (len(devicePath) < 10 || devicePath[0:10] != "/dev/ublkb")
|
||||||
|
{
|
||||||
|
klog.Errorf("%s does not start with /dev/ublkb", devicePath)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
unmapOut, unmapErr := exec.Command("/usr/bin/vitastor-ublk", "unmap", devicePath).CombinedOutput()
|
||||||
|
if (unmapErr != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to unmap UBLK device %s: %s, error: %v", devicePath, unmapOut, unmapErr)
|
||||||
|
}
|
||||||
|
for _, ext := range []string{"json", "pid"}
|
||||||
|
{
|
||||||
|
fn := stateDir + "vitastor-ublk-" + devicePath[10:] + "." + ext
|
||||||
|
err := os.Remove(fn)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to remove %s: %v", fn, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func system(program string, args ...string) ([]byte, []byte, error)
|
func system(program string, args ...string) ([]byte, []byte, error)
|
||||||
{
|
{
|
||||||
klog.Infof("Running "+program+" "+strings.Join(args, " "))
|
klog.Infof("Running "+program+" "+strings.Join(args, " "))
|
||||||
@@ -340,3 +479,43 @@ func systemCombined(program string, args ...string) ([]byte, error)
|
|||||||
}
|
}
|
||||||
return out.Bytes(), nil
|
return out.Bytes(), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func GetDeviceNameFromMount(mountPath string) (string, error)
|
||||||
|
{
|
||||||
|
// Use /proc/self/mountinfo to correctly parse bind mounts for block device files
|
||||||
|
mps, err := mount.ParseMountInfo("/proc/self/mountinfo")
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
slTarget, err := filepath.EvalSymlinks(mountPath)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
slTarget = mountPath
|
||||||
|
}
|
||||||
|
|
||||||
|
device := ""
|
||||||
|
for _, mp := range mps
|
||||||
|
{
|
||||||
|
if (mp.MountPoint == slTarget)
|
||||||
|
{
|
||||||
|
device = mp.Source
|
||||||
|
if (device[0] != '/' && mp.Root != "/")
|
||||||
|
{
|
||||||
|
// Handle {Source=udev Root=/vdb MountPoint=/var/lib/kubelet/tralaleylo/tralala}
|
||||||
|
for _, other := range mps
|
||||||
|
{
|
||||||
|
if (other.Root == "/" && other.Source == mp.Source)
|
||||||
|
{
|
||||||
|
device = other.MountPoint + mp.Root
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return device, nil
|
||||||
|
}
|
||||||
|
|||||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
|
|
||||||
docker build --build-arg DISTRO=debian --build-arg REL=bookworm -t vitastor-buildenv:bookworm -f vitastor-buildenv.Dockerfile .
|
docker build --build-arg DISTRO=debian --build-arg REL=bookworm -t vitastor-buildenv:bookworm -f vitastor-buildenv.Dockerfile .
|
||||||
docker run -i --rm -e REL=bookworm -v `dirname $0`/../:/root/vitastor vitastor-buildenv:bookworm /root/vitastor/debian/vitastor-build.sh
|
docker run -it --rm -e REL=bookworm -v `dirname $0`/../:/root/vitastor vitastor-buildenv:bookworm /root/vitastor/debian/vitastor-build.sh
|
||||||
|
|||||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
|
|
||||||
docker build --build-arg DISTRO=debian --build-arg REL=bullseye -t vitastor-buildenv:bullseye -f vitastor-buildenv.Dockerfile .
|
docker build --build-arg DISTRO=debian --build-arg REL=bullseye -t vitastor-buildenv:bullseye -f vitastor-buildenv.Dockerfile .
|
||||||
docker run -i --rm -e REL=bullseye -v `dirname $0`/../:/root/vitastor vitastor-buildenv:bullseye /root/vitastor/debian/vitastor-build.sh
|
docker run -it --rm -e REL=bullseye -v `dirname $0`/../:/root/vitastor vitastor-buildenv:bullseye /root/vitastor/debian/vitastor-build.sh
|
||||||
|
|||||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
|
|
||||||
docker build --build-arg DISTRO=debian --build-arg REL=buster -t vitastor-buildenv:buster -f vitastor-buildenv.Dockerfile .
|
docker build --build-arg DISTRO=debian --build-arg REL=buster -t vitastor-buildenv:buster -f vitastor-buildenv.Dockerfile .
|
||||||
docker run -i --rm -e REL=buster -v `dirname $0`/../:/root/vitastor vitastor-buildenv:buster /root/vitastor/debian/vitastor-build.sh
|
docker run -it --rm -e REL=buster -v `dirname $0`/../:/root/vitastor vitastor-buildenv:buster /root/vitastor/debian/vitastor-build.sh
|
||||||
|
|||||||
+4
@@ -0,0 +1,4 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
docker build --build-arg DISTRO=debian --build-arg REL=trixie -t vitastor-buildenv:trixie -f vitastor-buildenv.Dockerfile .
|
||||||
|
docker run -it --rm -e REL=trixie -v `dirname $0`/../:/root/vitastor vitastor-buildenv:trixie /root/vitastor/debian/vitastor-build.sh
|
||||||
+1
-1
@@ -2,4 +2,4 @@
|
|||||||
# Ubuntu 22.04 Jammy Jellyfish
|
# Ubuntu 22.04 Jammy Jellyfish
|
||||||
|
|
||||||
docker build --build-arg DISTRO=ubuntu --build-arg REL=jammy -t vitastor-buildenv:jammy -f vitastor-buildenv.Dockerfile .
|
docker build --build-arg DISTRO=ubuntu --build-arg REL=jammy -t vitastor-buildenv:jammy -f vitastor-buildenv.Dockerfile .
|
||||||
docker run -i --rm -e REL=jammy -v `dirname $0`/../:/root/vitastor vitastor-buildenv:jammy /root/vitastor/debian/vitastor-build.sh
|
docker run -it --rm -e REL=jammy -v `dirname $0`/../:/root/vitastor vitastor-buildenv:jammy /root/vitastor/debian/vitastor-build.sh
|
||||||
|
|||||||
+1
-1
@@ -2,4 +2,4 @@
|
|||||||
# 24.04 Noble Numbat
|
# 24.04 Noble Numbat
|
||||||
|
|
||||||
docker build --build-arg DISTRO=ubuntu --build-arg REL=noble -t vitastor-buildenv:noble -f vitastor-buildenv.Dockerfile .
|
docker build --build-arg DISTRO=ubuntu --build-arg REL=noble -t vitastor-buildenv:noble -f vitastor-buildenv.Dockerfile .
|
||||||
docker run -i --rm -e REL=noble -v `dirname $0`/../:/root/vitastor vitastor-buildenv:noble /root/vitastor/debian/vitastor-build.sh
|
docker run -it --rm -e REL=noble -v `dirname $0`/../:/root/vitastor vitastor-buildenv:noble /root/vitastor/debian/vitastor-build.sh
|
||||||
|
|||||||
+5
@@ -0,0 +1,5 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# 25.10 Questing quokka
|
||||||
|
|
||||||
|
docker build --build-arg DISTRO=ubuntu --build-arg REL=questing -t vitastor-buildenv:questing -f vitastor-buildenv.Dockerfile .
|
||||||
|
docker run -it --rm -e REL=questing -v `dirname $0`/../:/root/vitastor vitastor-buildenv:questing /root/vitastor/debian/vitastor-build.sh
|
||||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
vitastor (2.2.3-1) unstable; urgency=medium
|
vitastor (3.0.2-1) unstable; urgency=medium
|
||||||
|
|
||||||
* Bugfixes
|
* Bugfixes
|
||||||
|
|
||||||
|
|||||||
Vendored
+2
-2
@@ -2,9 +2,9 @@ Source: vitastor
|
|||||||
Section: admin
|
Section: admin
|
||||||
Priority: optional
|
Priority: optional
|
||||||
Maintainer: Vitaliy Filippov <vitalif@yourcmc.ru>
|
Maintainer: Vitaliy Filippov <vitalif@yourcmc.ru>
|
||||||
Build-Depends: debhelper, liburing-dev (>= 0.6), g++ (>= 8), libstdc++6 (>= 8),
|
Build-Depends: debhelper, g++ (>= 8), libstdc++6 (>= 8),
|
||||||
linux-libc-dev, libgoogle-perftools-dev, libjerasure-dev, libgf-complete-dev,
|
linux-libc-dev, libgoogle-perftools-dev, libjerasure-dev, libgf-complete-dev,
|
||||||
libibverbs-dev, libisal-dev, cmake, pkg-config, libnl-3-dev, libnl-genl-3-dev,
|
libibverbs-dev, librdmacm-dev, libisal-dev, cmake, pkg-config, libnl-3-dev, libnl-genl-3-dev,
|
||||||
node-bindings <!nocheck>, node-gyp, node-nan
|
node-bindings <!nocheck>, node-gyp, node-nan
|
||||||
Standards-Version: 4.5.0
|
Standards-Version: 4.5.0
|
||||||
Homepage: https://vitastor.io/
|
Homepage: https://vitastor.io/
|
||||||
|
|||||||
Vendored
+1
-1
@@ -26,7 +26,7 @@ RUN if [ "$REL" = "buster" -o "$REL" = "bullseye" -o "$REL" = "bookworm" ]; then
|
|||||||
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
|
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
|
||||||
|
|
||||||
RUN apt-get update
|
RUN apt-get update
|
||||||
RUN DEBIAN_FRONTEND=noninteractive TZ=Europe/Moscow apt-get -y install fio liburing-dev libgoogle-perftools-dev devscripts
|
RUN DEBIAN_FRONTEND=noninteractive TZ=Europe/Moscow apt-get -y install fio libgoogle-perftools-dev devscripts
|
||||||
RUN DEBIAN_FRONTEND=noninteractive TZ=Europe/Moscow apt-get -y build-dep qemu
|
RUN DEBIAN_FRONTEND=noninteractive TZ=Europe/Moscow apt-get -y build-dep qemu
|
||||||
# To build a custom version
|
# To build a custom version
|
||||||
#RUN cp /root/packages/qemu-orig/* /root
|
#RUN cp /root/packages/qemu-orig/* /root
|
||||||
|
|||||||
Vendored
+7
-1
@@ -44,13 +44,19 @@ curl -s https://git.yourcmc.ru/vitalif/antietcd/archive/master.tar.gz | tar -zx
|
|||||||
curl -s https://git.yourcmc.ru/vitalif/tinyraft/archive/master.tar.gz | tar -zx
|
curl -s https://git.yourcmc.ru/vitalif/tinyraft/archive/master.tar.gz | tar -zx
|
||||||
|
|
||||||
cd /root/vitastor/packages/vitastor-$REL
|
cd /root/vitastor/packages/vitastor-$REL
|
||||||
tar --sort=name --mtime='2020-01-01' --owner=0 --group=0 --exclude=debian -cJf vitastor_$VER.orig.tar.xz vitastor-$VER
|
if [[ "$REL" = "trixie" && -e ../vitastor-bookworm/vitastor_$VER.orig.tar.xz ]]; then
|
||||||
|
# Fucking shit, archives differ between bookworm (xz 5.4.1) and trixie (xz 5.8.1)
|
||||||
|
cp ../vitastor-bookworm/vitastor_$VER.orig.tar.xz .
|
||||||
|
else
|
||||||
|
tar --sort=name --mtime='2020-01-01' --owner=0 --group=0 --exclude=debian -cJf vitastor_$VER.orig.tar.xz vitastor-$VER
|
||||||
|
fi
|
||||||
cd vitastor-$VER
|
cd vitastor-$VER
|
||||||
DEBEMAIL="Vitaliy Filippov <vitalif@yourcmc.ru>" dch -D $REL -v "$FULLVER""$REL" "Rebuild for $REL"
|
DEBEMAIL="Vitaliy Filippov <vitalif@yourcmc.ru>" dch -D $REL -v "$FULLVER""$REL" "Rebuild for $REL"
|
||||||
DEB_BUILD_OPTIONS=nocheck dpkg-buildpackage --jobs=auto -sa
|
DEB_BUILD_OPTIONS=nocheck dpkg-buildpackage --jobs=auto -sa
|
||||||
rm -rf /root/vitastor/packages/vitastor-$REL/vitastor-*/
|
rm -rf /root/vitastor/packages/vitastor-$REL/vitastor-*/
|
||||||
|
|
||||||
# Why does ubuntu rename debug packages to *.ddeb?
|
# Why does ubuntu rename debug packages to *.ddeb?
|
||||||
|
cd /root/vitastor/packages/vitastor-$REL
|
||||||
if ls *.ddeb >/dev/null; then
|
if ls *.ddeb >/dev/null; then
|
||||||
perl -i -pe 's/\.ddeb/.deb/' *.buildinfo *.changes
|
perl -i -pe 's/\.ddeb/.deb/' *.buildinfo *.changes
|
||||||
for i in *.ddeb; do
|
for i in *.ddeb; do
|
||||||
|
|||||||
Vendored
+1
-1
@@ -25,7 +25,7 @@ RUN set -e -x; \
|
|||||||
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
|
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
|
||||||
|
|
||||||
RUN apt-get update && \
|
RUN apt-get update && \
|
||||||
apt-get -y install fio liburing-dev libgoogle-perftools-dev devscripts libjerasure-dev cmake \
|
apt-get -y install fio libgoogle-perftools-dev devscripts libjerasure-dev cmake \
|
||||||
libibverbs-dev librdmacm-dev libisal-dev libnl-3-dev libnl-genl-3-dev curl nodejs npm node-nan node-bindings && \
|
libibverbs-dev librdmacm-dev libisal-dev libnl-3-dev libnl-genl-3-dev curl nodejs npm node-nan node-bindings && \
|
||||||
apt-get -y build-dep fio && \
|
apt-get -y build-dep fio && \
|
||||||
apt-get --download-only source fio
|
apt-get --download-only source fio
|
||||||
|
|||||||
Vendored
+1
@@ -2,6 +2,7 @@ usr/bin/vita
|
|||||||
usr/bin/vitastor-cli
|
usr/bin/vitastor-cli
|
||||||
usr/bin/vitastor-rm
|
usr/bin/vitastor-rm
|
||||||
usr/bin/vitastor-nbd
|
usr/bin/vitastor-nbd
|
||||||
|
usr/bin/vitastor-ublk
|
||||||
usr/bin/vitastor-nfs
|
usr/bin/vitastor-nfs
|
||||||
usr/bin/vitastor-kv
|
usr/bin/vitastor-kv
|
||||||
usr/bin/vitastor-kv-stress
|
usr/bin/vitastor-kv-stress
|
||||||
|
|||||||
+1
-1
@@ -3,7 +3,7 @@
|
|||||||
FROM debian:bookworm
|
FROM debian:bookworm
|
||||||
|
|
||||||
ADD etc/apt /etc/apt/
|
ADD etc/apt /etc/apt/
|
||||||
RUN apt-get update && apt-get -y install vitastor udev systemd qemu-system-x86 qemu-system-common qemu-block-extra qemu-utils jq nfs-common && apt-get clean
|
RUN apt-get update && apt-get -y install vitastor ibverbs-providers udev systemd qemu-system-x86 qemu-system-common qemu-block-extra qemu-utils jq nfs-common && apt-get clean
|
||||||
ADD sleep.sh /usr/bin/
|
ADD sleep.sh /usr/bin/
|
||||||
ADD install.sh /usr/bin/
|
ADD install.sh /usr/bin/
|
||||||
ADD scripts /opt/scripts/
|
ADD scripts /opt/scripts/
|
||||||
|
|||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
VITASTOR_VERSION ?= v2.2.3
|
VITASTOR_VERSION ?= v3.0.2
|
||||||
|
|
||||||
all: build push
|
all: build push
|
||||||
|
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
#
|
#
|
||||||
|
|
||||||
# Desired Vitastor version
|
# Desired Vitastor version
|
||||||
VITASTOR_VERSION=v2.2.3
|
VITASTOR_VERSION=v3.0.2
|
||||||
|
|
||||||
# Additional arguments for all containers
|
# Additional arguments for all containers
|
||||||
# For example, you may want to specify a custom logging driver here
|
# For example, you may want to specify a custom logging driver here
|
||||||
|
|||||||
@@ -25,6 +25,9 @@ affect their interaction with the cluster.
|
|||||||
- [nbd_max_part](#nbd_max_part)
|
- [nbd_max_part](#nbd_max_part)
|
||||||
- [osd_nearfull_ratio](#osd_nearfull_ratio)
|
- [osd_nearfull_ratio](#osd_nearfull_ratio)
|
||||||
- [hostname](#hostname)
|
- [hostname](#hostname)
|
||||||
|
- [ublk_queue_depth](#ublk_queue_depth)
|
||||||
|
- [ublk_max_io_size](#ublk_max_io_size)
|
||||||
|
- [qemu_file_mirror_path](#qemu_file_mirror_path)
|
||||||
|
|
||||||
## client_iothread_count
|
## client_iothread_count
|
||||||
|
|
||||||
@@ -225,3 +228,28 @@ without destroying and recreating OSDs.
|
|||||||
Clients use host name to find their distance to OSDs when [localized reads](pool.en.md#local_reads)
|
Clients use host name to find their distance to OSDs when [localized reads](pool.en.md#local_reads)
|
||||||
are enabled. By default, standard [gethostname](https://man7.org/linux/man-pages/man2/gethostname.2.html)
|
are enabled. By default, standard [gethostname](https://man7.org/linux/man-pages/man2/gethostname.2.html)
|
||||||
function is used to determine host name, but you can also override it with this parameter.
|
function is used to determine host name, but you can also override it with this parameter.
|
||||||
|
|
||||||
|
## ublk_queue_depth
|
||||||
|
|
||||||
|
- Type: integer
|
||||||
|
- Default: 256
|
||||||
|
|
||||||
|
Default queue depth for [Vitastor ublk servers](../usage/ublk.en.md).
|
||||||
|
|
||||||
|
## ublk_max_io_size
|
||||||
|
|
||||||
|
- Type: integer
|
||||||
|
|
||||||
|
Default maximum I/O size for Vitastor [ublk servers](../usage/ublk.en.md).
|
||||||
|
The largest of 1 MB and pool block size multiplied by EC data chunk count is used if not specified.
|
||||||
|
|
||||||
|
## qemu_file_mirror_path
|
||||||
|
|
||||||
|
- Type: string
|
||||||
|
|
||||||
|
When set to an FS directory path (for example, `/mnt/vitastor/`), `qemu-img info` and similar
|
||||||
|
QAPI commands return the name of the image inside this directory instead of normal
|
||||||
|
`vitastor://?image=abc` URI as `filename`.
|
||||||
|
|
||||||
|
This allows to then mount this path using [vitastor-nfs](../usage/nfs.en.md) and trick
|
||||||
|
third-party systems like Veeam which rely on `filename` in the image info but don't support Vitastor.
|
||||||
|
|||||||
@@ -25,6 +25,9 @@
|
|||||||
- [nbd_max_part](#nbd_max_part)
|
- [nbd_max_part](#nbd_max_part)
|
||||||
- [osd_nearfull_ratio](#osd_nearfull_ratio)
|
- [osd_nearfull_ratio](#osd_nearfull_ratio)
|
||||||
- [hostname](#hostname)
|
- [hostname](#hostname)
|
||||||
|
- [ublk_queue_depth](#ublk_queue_depth)
|
||||||
|
- [ublk_max_io_size](#ublk_max_io_size)
|
||||||
|
- [qemu_file_mirror_path](#qemu_file_mirror_path)
|
||||||
|
|
||||||
## client_iothread_count
|
## client_iothread_count
|
||||||
|
|
||||||
@@ -230,3 +233,30 @@ RDMA и хотите повысить пиковую производитель
|
|||||||
[локальные чтения](pool.ru.md#local_reads). По умолчанию для определения имени
|
[локальные чтения](pool.ru.md#local_reads). По умолчанию для определения имени
|
||||||
хоста используется стандартная функция [gethostname](https://man7.org/linux/man-pages/man2/gethostname.2.html),
|
хоста используется стандартная функция [gethostname](https://man7.org/linux/man-pages/man2/gethostname.2.html),
|
||||||
но вы также можете задать имя хоста вручную данным параметром.
|
но вы также можете задать имя хоста вручную данным параметром.
|
||||||
|
|
||||||
|
## ublk_queue_depth
|
||||||
|
|
||||||
|
- Тип: целое число
|
||||||
|
- Значение по умолчанию: 256
|
||||||
|
|
||||||
|
Глубина очереди по умолчанию для [ublk-серверов Vitastor](../usage/ublk.ru.md).
|
||||||
|
|
||||||
|
## ublk_max_io_size
|
||||||
|
|
||||||
|
- Тип: целое число
|
||||||
|
|
||||||
|
Максимальный размер запроса ввода-вывода для [ublk-серверов Vitastor](../usage/ublk.ru.md).
|
||||||
|
Если не задан, используется максимум из 1 МБ и размера блока пула, умноженного на число частей
|
||||||
|
данных EC-пула.
|
||||||
|
|
||||||
|
## qemu_file_mirror_path
|
||||||
|
|
||||||
|
- Тип: строка
|
||||||
|
|
||||||
|
Если установить эту опцию равной пути к каталогу в ФС, команда `qemu-img info` и подобные
|
||||||
|
команды QAPI будут возвращать в поле `filename` имя образа внутри заданного каталога вместо
|
||||||
|
обычного адреса типа `vitastor://?image=abc`.
|
||||||
|
|
||||||
|
Это позволяет смонтировать этот путь с помощью [vitastor-nfs](../usage/nfs.ru.md) и обмануть
|
||||||
|
сторонние системы типа Veeam, которые полагаются на поле `filename` в информации об образе QEMU,
|
||||||
|
но не поддерживают Vitastor.
|
||||||
|
|||||||
@@ -9,6 +9,7 @@
|
|||||||
These parameters apply to OSDs, are fixed at the moment of OSD drive
|
These parameters apply to OSDs, are fixed at the moment of OSD drive
|
||||||
initialization and can't be changed after it without losing data.
|
initialization and can't be changed after it without losing data.
|
||||||
|
|
||||||
|
- [meta_format](#meta_format)
|
||||||
- [data_device](#data_device)
|
- [data_device](#data_device)
|
||||||
- [meta_device](#meta_device)
|
- [meta_device](#meta_device)
|
||||||
- [journal_device](#journal_device)
|
- [journal_device](#journal_device)
|
||||||
@@ -27,6 +28,21 @@ initialization and can't be changed after it without losing data.
|
|||||||
- [data_csum_type](#data_csum_type)
|
- [data_csum_type](#data_csum_type)
|
||||||
- [csum_block_size](#csum_block_size)
|
- [csum_block_size](#csum_block_size)
|
||||||
|
|
||||||
|
## meta_format
|
||||||
|
|
||||||
|
- Type: integer
|
||||||
|
- Default: 3
|
||||||
|
|
||||||
|
OSD store implementation version and on-disk metadata format.
|
||||||
|
|
||||||
|
Three versions are currently supported: 3, 2 and 1.
|
||||||
|
- 3 the new log-structured store, it's overall faster, has lower Write
|
||||||
|
Amplification, which may be even close to 1 (i.e. almost no extra writes)
|
||||||
|
if your SSDs support atomic writes (see [atomic_write_size](osd.en.md#atomic_write_size)).
|
||||||
|
- 2 is the old stable store from Vitastor 0.9-2.x.
|
||||||
|
- 1 is the same old store but with a legacy metadata format from Vitastor
|
||||||
|
versions to up 0.8.x, without any support for checksums.
|
||||||
|
|
||||||
## data_device
|
## data_device
|
||||||
|
|
||||||
- Type: string
|
- Type: string
|
||||||
|
|||||||
@@ -10,6 +10,7 @@
|
|||||||
дисковые параметры, задаются в момент инициализации дисков OSD и не могут быть
|
дисковые параметры, задаются в момент инициализации дисков OSD и не могут быть
|
||||||
изменены после этого без потери данных.
|
изменены после этого без потери данных.
|
||||||
|
|
||||||
|
- [meta_format](#meta_format)
|
||||||
- [data_device](#data_device)
|
- [data_device](#data_device)
|
||||||
- [meta_device](#meta_device)
|
- [meta_device](#meta_device)
|
||||||
- [journal_device](#journal_device)
|
- [journal_device](#journal_device)
|
||||||
@@ -28,6 +29,23 @@
|
|||||||
- [data_csum_type](#data_csum_type)
|
- [data_csum_type](#data_csum_type)
|
||||||
- [csum_block_size](#csum_block_size)
|
- [csum_block_size](#csum_block_size)
|
||||||
|
|
||||||
|
## meta_format
|
||||||
|
|
||||||
|
- Тип: целое число
|
||||||
|
- Значение по умолчанию: 3
|
||||||
|
|
||||||
|
Версия реализации дискового хранилища OSD и дискового формата метаданных.
|
||||||
|
|
||||||
|
Поддерживаются три версии: 3, 2 и 1.
|
||||||
|
- 3 - новое лог-структурированное хранилище, в целом более быстрое, со
|
||||||
|
сниженным фактором амплификации записи, который может составлять около 1
|
||||||
|
(то есть, практически без лишней служебной записи), если ваши SSD
|
||||||
|
поддерживают атомарную запись (см. [atomic_write_size](osd.ru.md#atomic_write_size)).
|
||||||
|
- 2 - старое стабильное хранилище из версий Vitastor 0.9-2.x.
|
||||||
|
- 1 - то же самое стабильное хранилище, но с ещё более старым форматом
|
||||||
|
метаданных из версий Vitastor до 0.8.x, без какой-либо поддержки
|
||||||
|
контрольных сумм.
|
||||||
|
|
||||||
## data_device
|
## data_device
|
||||||
|
|
||||||
- Тип: строка
|
- Тип: строка
|
||||||
|
|||||||
+68
-1
@@ -65,6 +65,10 @@ with an OSD restart or, for some of them, even without restarting by updating co
|
|||||||
- [allow_net_split](#allow_net_split)
|
- [allow_net_split](#allow_net_split)
|
||||||
- [enable_pg_locks](#enable_pg_locks)
|
- [enable_pg_locks](#enable_pg_locks)
|
||||||
- [pg_lock_retry_interval_ms](#pg_lock_retry_interval_ms)
|
- [pg_lock_retry_interval_ms](#pg_lock_retry_interval_ms)
|
||||||
|
- [atomic_write_size](#atomic_write_size)
|
||||||
|
- [use_atomic_flag](#use_atomic_flag)
|
||||||
|
- [pg_reshard_chunk_size](#pg_reshard_chunk_size)
|
||||||
|
- [pg_reshard_chunk_pause_ms](#pg_reshard_chunk_pause_ms)
|
||||||
|
|
||||||
## bind_address
|
## bind_address
|
||||||
|
|
||||||
@@ -491,7 +495,7 @@ Can be used to slow down scrubbing if it affects user load too much.
|
|||||||
## scrub_list_limit
|
## scrub_list_limit
|
||||||
|
|
||||||
- Type: integer
|
- Type: integer
|
||||||
- Default: 1000
|
- Default: 262144
|
||||||
- Can be changed online: yes
|
- Can be changed online: yes
|
||||||
|
|
||||||
Number of objects to list in one listing operation during scrub.
|
Number of objects to list in one listing operation during scrub.
|
||||||
@@ -666,3 +670,66 @@ Use this parameter to enable or disable this function for all pools.
|
|||||||
- Default: 100
|
- Default: 100
|
||||||
|
|
||||||
Retry interval for failed PG lock attempts.
|
Retry interval for failed PG lock attempts.
|
||||||
|
|
||||||
|
## atomic_write_size
|
||||||
|
|
||||||
|
- Type: integer
|
||||||
|
- Default: 4096
|
||||||
|
|
||||||
|
Maximum data device atomic write size allowed for OSD to use.
|
||||||
|
|
||||||
|
Atomic writes allow to reduce the Write Amplification factor with the new store
|
||||||
|
([meta_format](layout-osd.en.md#meta_format)=3) to almost 1 (i.e. almost no extra writes)
|
||||||
|
with replicated pools and reach the best possible write performance.
|
||||||
|
|
||||||
|
Default value is auto-detected during OSD initialization from
|
||||||
|
`/sys/block/xx/queue/atomic_write_max_bytes` or assumed to be 4096 bytes
|
||||||
|
because all known disks support 4 KB atomic writes. Auto-detection is only used for
|
||||||
|
NVMe disks because SAS disks require the explicit WRITE ATOMIC command which requires
|
||||||
|
RWF_ATOMIC (see below [#use_atomic_flag]) but that flag works incorrectly in current
|
||||||
|
Linux versions.
|
||||||
|
|
||||||
|
You can also check if your NVMe drives support atomic writes by running
|
||||||
|
the command `nvme id-ctrl /dev/nvme0n1 | grep awupf`. If the reported value,
|
||||||
|
plus 1, multiplied by the currently selected block size of the NVMe,
|
||||||
|
is more than 4 KB, then the new store can utilize it for better performance.
|
||||||
|
The only drives known to support it currently are [Micron and Kioxia](../intro/quickstart.en.md).
|
||||||
|
|
||||||
|
Atomic writes allow to skip double data writes in replicated pools, thus
|
||||||
|
reducing Write Amplification and improving write performance up to 2 times.
|
||||||
|
|
||||||
|
## use_atomic_flag
|
||||||
|
|
||||||
|
- Type: boolean
|
||||||
|
|
||||||
|
This option controls whether Vitastor OSDs use RWF_ATOMIC write flag with atomic writes.
|
||||||
|
This flag is supported since Linux 6.11 and adds some safety to atomic writes - the kernel
|
||||||
|
guarantees to not fragment write requests with it and also to check them against the actual
|
||||||
|
device atomic write capabilities.
|
||||||
|
|
||||||
|
However, the option is disabled by default because the flag is currently UNUSABLE - Linux
|
||||||
|
incorrectly requires writes with that flag to be of power-of-2 length and length-aligned.
|
||||||
|
I.e., for example, 12 KB writes and not-8-KB aligned 8 KB writes are forbidden by the kernel,
|
||||||
|
even though the NVMe specification allows them.
|
||||||
|
|
||||||
|
For NVMe disks with `scheduler=none` writes aren't fragmented anyway so it's not a big deal.
|
||||||
|
However, you can rebuild your kernel with [this patch](../../patches/linux-fix-atomic-write-checks.diff)
|
||||||
|
and turn this option on. It will make your atomic writes a bit safer.
|
||||||
|
|
||||||
|
## pg_reshard_chunk_size
|
||||||
|
|
||||||
|
- Type: integer
|
||||||
|
- Default: 100000
|
||||||
|
|
||||||
|
Pool PG count change is a CPU-intensive operation because OSDs store the full object database
|
||||||
|
in memory and have to move all entries between old and new PGs. Thus it's performed in chunks,
|
||||||
|
with pauses between chunks to prevent blocking OSD's event loop and other clients' operations.
|
||||||
|
This option sets the maximum number of object is a chunk. Moving 100k objects usually takes
|
||||||
|
50-100ms. Chunk size equal to 0 means unlimited.
|
||||||
|
|
||||||
|
## pg_reshard_chunk_pause_ms
|
||||||
|
|
||||||
|
- Type: milliseconds
|
||||||
|
- Default: 100
|
||||||
|
|
||||||
|
This option sets the interval between handling two PG count change chunks.
|
||||||
|
|||||||
+74
-1
@@ -66,6 +66,10 @@
|
|||||||
- [allow_net_split](#allow_net_split)
|
- [allow_net_split](#allow_net_split)
|
||||||
- [enable_pg_locks](#enable_pg_locks)
|
- [enable_pg_locks](#enable_pg_locks)
|
||||||
- [pg_lock_retry_interval_ms](#pg_lock_retry_interval_ms)
|
- [pg_lock_retry_interval_ms](#pg_lock_retry_interval_ms)
|
||||||
|
- [atomic_write_size](#atomic_write_size)
|
||||||
|
- [use_atomic_flag](#use_atomic_flag)
|
||||||
|
- [pg_reshard_chunk_size](#pg_reshard_chunk_size)
|
||||||
|
- [pg_reshard_chunk_pause_ms](#pg_reshard_chunk_pause_ms)
|
||||||
|
|
||||||
## bind_address
|
## bind_address
|
||||||
|
|
||||||
@@ -514,7 +518,7 @@ fsync небезопасным даже с режимом "directsync".
|
|||||||
## scrub_list_limit
|
## scrub_list_limit
|
||||||
|
|
||||||
- Тип: целое число
|
- Тип: целое число
|
||||||
- Значение по умолчанию: 1000
|
- Значение по умолчанию: 262144
|
||||||
- Можно менять на лету: да
|
- Можно менять на лету: да
|
||||||
|
|
||||||
Размер загружаемых за одну операцию списков объектов в процессе фоновой
|
Размер загружаемых за одну операцию списков объектов в процессе фоновой
|
||||||
@@ -699,3 +703,72 @@ pg_minsize OSD во время переключений, что может по
|
|||||||
- Значение по умолчанию: 100
|
- Значение по умолчанию: 100
|
||||||
|
|
||||||
Интервал повтора неудачных попыток блокировки PG.
|
Интервал повтора неудачных попыток блокировки PG.
|
||||||
|
|
||||||
|
## atomic_write_size
|
||||||
|
|
||||||
|
- Тип: целое число
|
||||||
|
- Значение по умолчанию: 4096
|
||||||
|
|
||||||
|
Максимальный размер атомарной записи на диск данных, который OSD разрешено использовать.
|
||||||
|
|
||||||
|
Поддержка атомарной записи позволяет снизить мультипликатор записи (Write Amplification)
|
||||||
|
на диск с новым хранилищем ([meta_format](layout-osd.ru.md#meta_format)=3)
|
||||||
|
практически до 1 (то есть, почти до нулевого объёма лишней записи) в реплицированных
|
||||||
|
пулах и достигнуть наилучшей возможной производительности записи.
|
||||||
|
|
||||||
|
Значение по умолчанию авто-определяется во время инициализации OSD из
|
||||||
|
`/sys/block/xx/queue/atomic_write_max_bytes` либо принимается равным 4096,
|
||||||
|
так как все известные диски поддерживают атомарную запись 4 КБ блоков.
|
||||||
|
Автоопределение применяется только для NVMe-дисков, так как SAS диски требуют
|
||||||
|
использования отдельной команды WRITE ATOMIC, а для неё нужен флаг RWF_ATOMIC
|
||||||
|
(см. ниже [#use_atomic_flag]), а он в текущих версиях Linux работает некорректно.
|
||||||
|
|
||||||
|
Вы также можете проверить, поддерживают ли ваши NVMe-диски атомарную запись,
|
||||||
|
с помощью команды `nvme id-ctrl /dev/nvme0n1 | grep awupf`. Если значение awupf
|
||||||
|
плюс 1, умноженное на текущий выбранный размер блока NVMe-диска, больше 4 КБ,
|
||||||
|
то новое хранилище может использовать атомарные записи для достижения лучшей
|
||||||
|
производительности. Единственные известные диски, которые поддерживают это сейчас -
|
||||||
|
[Micron и Kioxia](../intro/quickstart.ru.md).
|
||||||
|
|
||||||
|
Атомарная запись позволяет не использовать двойную запись данных (в журнал и на
|
||||||
|
устройство данных) в реплицированных пулах и таким образом снижает амплификацию
|
||||||
|
записи (объём служебной записи на диск) и улучшает производительность записи
|
||||||
|
вплоть до 2-х кратного прироста.
|
||||||
|
|
||||||
|
## use_atomic_flag
|
||||||
|
|
||||||
|
- Тип: булево (да/нет)
|
||||||
|
|
||||||
|
Данная опция контролирует использование Vitastor OSD флага RWF_ATOMIC при атомарной записи
|
||||||
|
блоков. Этот флаг поддерживается, начиная с версии ядра Linux 6.11 и добавляет немного корректности
|
||||||
|
атомарным записям - ядро гарантирует отсутствие фрагментации запросов записи с этим флагом и
|
||||||
|
проверяет их на соответствие реальным возможностям устройства.
|
||||||
|
|
||||||
|
Однако, данная опция по умолчанию отключена, так как флаг в текущих версиях Linux работает
|
||||||
|
абсолютно НЕКОРРЕКТНО - при нём Linux требует, чтобы запросы записи имели длину, равную
|
||||||
|
степени двойки и были выровнены на эту длину. То есть, например, 12 КБ запросы записи, а также
|
||||||
|
8 КБ запросы записи по не-кратному 8 КБ смещению запрещаются ядром, хотя спецификация NVMe их
|
||||||
|
разрешает.
|
||||||
|
|
||||||
|
Для NVMe-дисков с `scheduler=none` запросы записи и так не фрагментируются, так что это не так
|
||||||
|
уж и важно, однако вы можете пересобрать своё ядро с [этим патчем](../../patches/linux-fix-atomic-write-checks.diff)
|
||||||
|
и включить данную опцию. Это сделает вашу атомарную запись капельку безопаснее.
|
||||||
|
|
||||||
|
## pg_reshard_chunk_size
|
||||||
|
|
||||||
|
- Тип: целое число
|
||||||
|
- Значение по умолчанию: 100000
|
||||||
|
|
||||||
|
Изменение числа PG в пуле заметно загружает процессор, так как OSD хранят полную базу данных
|
||||||
|
объектов в памяти и им приходится перемещать все записи объектов между старыми и новыми PG.
|
||||||
|
Поэтому изменение применяется порциями, с паузами между порциями, чтобы не блокировать обработку
|
||||||
|
событий OSD и операции остальных клиентов. Данная опция задаёт максимальное число объектов
|
||||||
|
в порции. Перемещение 100 тысяч объектов (значение по умолчанию) обычно занимает порядка
|
||||||
|
50-100 миллисекунд. Значение опции 0 отключает лимит размера порции.
|
||||||
|
|
||||||
|
## pg_reshard_chunk_pause_ms
|
||||||
|
|
||||||
|
- Тип: миллисекунды
|
||||||
|
- Значение по умолчанию: 100
|
||||||
|
|
||||||
|
Данная опция задаёт интервал между обработкой двух порций изменения числа PG пулов.
|
||||||
|
|||||||
@@ -283,3 +283,36 @@
|
|||||||
[локальные чтения](pool.ru.md#local_reads). По умолчанию для определения имени
|
[локальные чтения](pool.ru.md#local_reads). По умолчанию для определения имени
|
||||||
хоста используется стандартная функция [gethostname](https://man7.org/linux/man-pages/man2/gethostname.2.html),
|
хоста используется стандартная функция [gethostname](https://man7.org/linux/man-pages/man2/gethostname.2.html),
|
||||||
но вы также можете задать имя хоста вручную данным параметром.
|
но вы также можете задать имя хоста вручную данным параметром.
|
||||||
|
- name: ublk_queue_depth
|
||||||
|
type: int
|
||||||
|
default: 256
|
||||||
|
online: false
|
||||||
|
info: Default queue depth for [Vitastor ublk servers](../usage/ublk.en.md).
|
||||||
|
info_ru: Глубина очереди по умолчанию для [ublk-серверов Vitastor](../usage/ublk.ru.md).
|
||||||
|
- name: ublk_max_io_size
|
||||||
|
type: int
|
||||||
|
online: false
|
||||||
|
info: |
|
||||||
|
Default maximum I/O size for Vitastor [ublk servers](../usage/ublk.en.md).
|
||||||
|
The largest of 1 MB and pool block size multiplied by EC data chunk count is used if not specified.
|
||||||
|
info_ru: |
|
||||||
|
Максимальный размер запроса ввода-вывода для [ublk-серверов Vitastor](../usage/ublk.ru.md).
|
||||||
|
Если не задан, используется максимум из 1 МБ и размера блока пула, умноженного на число частей
|
||||||
|
данных EC-пула.
|
||||||
|
- name: qemu_file_mirror_path
|
||||||
|
type: string
|
||||||
|
info: |
|
||||||
|
When set to an FS directory path (for example, `/mnt/vitastor/`), `qemu-img info` and similar
|
||||||
|
QAPI commands return the name of the image inside this directory instead of normal
|
||||||
|
`vitastor://?image=abc` URI as `filename`.
|
||||||
|
|
||||||
|
This allows to then mount this path using [vitastor-nfs](../usage/nfs.en.md) and trick
|
||||||
|
third-party systems like Veeam which rely on `filename` in the image info but don't support Vitastor.
|
||||||
|
info_ru: |
|
||||||
|
Если установить эту опцию равной пути к каталогу в ФС, команда `qemu-img info` и подобные
|
||||||
|
команды QAPI будут возвращать в поле `filename` имя образа внутри заданного каталога вместо
|
||||||
|
обычного адреса типа `vitastor://?image=abc`.
|
||||||
|
|
||||||
|
Это позволяет смонтировать этот путь с помощью [vitastor-nfs](../usage/nfs.ru.md) и обмануть
|
||||||
|
сторонние системы типа Veeam, которые полагаются на поле `filename` в информации об образе QEMU,
|
||||||
|
но не поддерживают Vitastor.
|
||||||
|
|||||||
@@ -24,6 +24,8 @@
|
|||||||
|
|
||||||
{{../../installation/kubernetes.en.md}}
|
{{../../installation/kubernetes.en.md}}
|
||||||
|
|
||||||
|
{{../../installation/s3.en.md}}
|
||||||
|
|
||||||
{{../../installation/source.en.md}}
|
{{../../installation/source.en.md}}
|
||||||
|
|
||||||
{{../../config.en.md|indent=1}}
|
{{../../config.en.md|indent=1}}
|
||||||
@@ -54,6 +56,8 @@
|
|||||||
|
|
||||||
{{../../usage/fio.en.md}}
|
{{../../usage/fio.en.md}}
|
||||||
|
|
||||||
|
{{../../usage/ublk.en.md}}
|
||||||
|
|
||||||
{{../../usage/nbd.en.md}}
|
{{../../usage/nbd.en.md}}
|
||||||
|
|
||||||
{{../../usage/qemu.en.md}}
|
{{../../usage/qemu.en.md}}
|
||||||
|
|||||||
@@ -26,6 +26,8 @@
|
|||||||
|
|
||||||
{{../../installation/source.ru.md}}
|
{{../../installation/source.ru.md}}
|
||||||
|
|
||||||
|
{{../../installation/s3.ru.md}}
|
||||||
|
|
||||||
{{../../config.ru.md|indent=1}}
|
{{../../config.ru.md|indent=1}}
|
||||||
|
|
||||||
{{../../config/common.ru.md|indent=2}}
|
{{../../config/common.ru.md|indent=2}}
|
||||||
@@ -54,6 +56,8 @@
|
|||||||
|
|
||||||
{{../../usage/fio.ru.md}}
|
{{../../usage/fio.ru.md}}
|
||||||
|
|
||||||
|
{{../../usage/ublk.ru.md}}
|
||||||
|
|
||||||
{{../../usage/nbd.ru.md}}
|
{{../../usage/nbd.ru.md}}
|
||||||
|
|
||||||
{{../../usage/qemu.ru.md}}
|
{{../../usage/qemu.ru.md}}
|
||||||
|
|||||||
@@ -1,3 +1,28 @@
|
|||||||
|
- name: meta_format
|
||||||
|
type: int
|
||||||
|
default: 3
|
||||||
|
info: |
|
||||||
|
OSD store implementation version and on-disk metadata format.
|
||||||
|
|
||||||
|
Three versions are currently supported: 3, 2 and 1.
|
||||||
|
- 3 the new log-structured store, it's overall faster, has lower Write
|
||||||
|
Amplification, which may be even close to 1 (i.e. almost no extra writes)
|
||||||
|
if your SSDs support atomic writes (see [atomic_write_size](osd.en.md#atomic_write_size)).
|
||||||
|
- 2 is the old stable store from Vitastor 0.9-2.x.
|
||||||
|
- 1 is the same old store but with a legacy metadata format from Vitastor
|
||||||
|
versions to up 0.8.x, without any support for checksums.
|
||||||
|
info_ru: |
|
||||||
|
Версия реализации дискового хранилища OSD и дискового формата метаданных.
|
||||||
|
|
||||||
|
Поддерживаются три версии: 3, 2 и 1.
|
||||||
|
- 3 - новое лог-структурированное хранилище, в целом более быстрое, со
|
||||||
|
сниженным фактором амплификации записи, который может составлять около 1
|
||||||
|
(то есть, практически без лишней служебной записи), если ваши SSD
|
||||||
|
поддерживают атомарную запись (см. [atomic_write_size](osd.ru.md#atomic_write_size)).
|
||||||
|
- 2 - старое стабильное хранилище из версий Vitastor 0.9-2.x.
|
||||||
|
- 1 - то же самое стабильное хранилище, но с ещё более старым форматом
|
||||||
|
метаданных из версий Vitastor до 0.8.x, без какой-либо поддержки
|
||||||
|
контрольных сумм.
|
||||||
- name: data_device
|
- name: data_device
|
||||||
type: string
|
type: string
|
||||||
info: |
|
info: |
|
||||||
|
|||||||
+106
-1
@@ -566,7 +566,7 @@
|
|||||||
сильно влияет на пользовательскую нагрузку.
|
сильно влияет на пользовательскую нагрузку.
|
||||||
- name: scrub_list_limit
|
- name: scrub_list_limit
|
||||||
type: int
|
type: int
|
||||||
default: 1000
|
default: 262144
|
||||||
online: true
|
online: true
|
||||||
info: |
|
info: |
|
||||||
Number of objects to list in one listing operation during scrub.
|
Number of objects to list in one listing operation during scrub.
|
||||||
@@ -801,3 +801,108 @@
|
|||||||
default: 100
|
default: 100
|
||||||
info: Retry interval for failed PG lock attempts.
|
info: Retry interval for failed PG lock attempts.
|
||||||
info_ru: Интервал повтора неудачных попыток блокировки PG.
|
info_ru: Интервал повтора неудачных попыток блокировки PG.
|
||||||
|
- name: atomic_write_size
|
||||||
|
type: int
|
||||||
|
default: 4096
|
||||||
|
info: |
|
||||||
|
Maximum data device atomic write size allowed for OSD to use.
|
||||||
|
|
||||||
|
Atomic writes allow to reduce the Write Amplification factor with the new store
|
||||||
|
([meta_format](layout-osd.en.md#meta_format)=3) to almost 1 (i.e. almost no extra writes)
|
||||||
|
with replicated pools and reach the best possible write performance.
|
||||||
|
|
||||||
|
Default value is auto-detected during OSD initialization from
|
||||||
|
`/sys/block/xx/queue/atomic_write_max_bytes` or assumed to be 4096 bytes
|
||||||
|
because all known disks support 4 KB atomic writes. Auto-detection is only used for
|
||||||
|
NVMe disks because SAS disks require the explicit WRITE ATOMIC command which requires
|
||||||
|
RWF_ATOMIC (see below [#use_atomic_flag]) but that flag works incorrectly in current
|
||||||
|
Linux versions.
|
||||||
|
|
||||||
|
You can also check if your NVMe drives support atomic writes by running
|
||||||
|
the command `nvme id-ctrl /dev/nvme0n1 | grep awupf`. If the reported value,
|
||||||
|
plus 1, multiplied by the currently selected block size of the NVMe,
|
||||||
|
is more than 4 KB, then the new store can utilize it for better performance.
|
||||||
|
The only drives known to support it currently are [Micron and Kioxia](../intro/quickstart.en.md).
|
||||||
|
|
||||||
|
Atomic writes allow to skip double data writes in replicated pools, thus
|
||||||
|
reducing Write Amplification and improving write performance up to 2 times.
|
||||||
|
info_ru: |
|
||||||
|
Максимальный размер атомарной записи на диск данных, который OSD разрешено использовать.
|
||||||
|
|
||||||
|
Поддержка атомарной записи позволяет снизить мультипликатор записи (Write Amplification)
|
||||||
|
на диск с новым хранилищем ([meta_format](layout-osd.ru.md#meta_format)=3)
|
||||||
|
практически до 1 (то есть, почти до нулевого объёма лишней записи) в реплицированных
|
||||||
|
пулах и достигнуть наилучшей возможной производительности записи.
|
||||||
|
|
||||||
|
Значение по умолчанию авто-определяется во время инициализации OSD из
|
||||||
|
`/sys/block/xx/queue/atomic_write_max_bytes` либо принимается равным 4096,
|
||||||
|
так как все известные диски поддерживают атомарную запись 4 КБ блоков.
|
||||||
|
Автоопределение применяется только для NVMe-дисков, так как SAS диски требуют
|
||||||
|
использования отдельной команды WRITE ATOMIC, а для неё нужен флаг RWF_ATOMIC
|
||||||
|
(см. ниже [#use_atomic_flag]), а он в текущих версиях Linux работает некорректно.
|
||||||
|
|
||||||
|
Вы также можете проверить, поддерживают ли ваши NVMe-диски атомарную запись,
|
||||||
|
с помощью команды `nvme id-ctrl /dev/nvme0n1 | grep awupf`. Если значение awupf
|
||||||
|
плюс 1, умноженное на текущий выбранный размер блока NVMe-диска, больше 4 КБ,
|
||||||
|
то новое хранилище может использовать атомарные записи для достижения лучшей
|
||||||
|
производительности. Единственные известные диски, которые поддерживают это сейчас -
|
||||||
|
[Micron и Kioxia](../intro/quickstart.ru.md).
|
||||||
|
|
||||||
|
Атомарная запись позволяет не использовать двойную запись данных (в журнал и на
|
||||||
|
устройство данных) в реплицированных пулах и таким образом снижает амплификацию
|
||||||
|
записи (объём служебной записи на диск) и улучшает производительность записи
|
||||||
|
вплоть до 2-х кратного прироста.
|
||||||
|
- name: use_atomic_flag
|
||||||
|
type: bool
|
||||||
|
info: |
|
||||||
|
This option controls whether Vitastor OSDs use RWF_ATOMIC write flag with atomic writes.
|
||||||
|
This flag is supported since Linux 6.11 and adds some safety to atomic writes - the kernel
|
||||||
|
guarantees to not fragment write requests with it and also to check them against the actual
|
||||||
|
device atomic write capabilities.
|
||||||
|
|
||||||
|
However, the option is disabled by default because the flag is currently UNUSABLE - Linux
|
||||||
|
incorrectly requires writes with that flag to be of power-of-2 length and length-aligned.
|
||||||
|
I.e., for example, 12 KB writes and not-8-KB aligned 8 KB writes are forbidden by the kernel,
|
||||||
|
even though the NVMe specification allows them.
|
||||||
|
|
||||||
|
For NVMe disks with `scheduler=none` writes aren't fragmented anyway so it's not a big deal.
|
||||||
|
However, you can rebuild your kernel with [this patch](../../patches/linux-fix-atomic-write-checks.diff)
|
||||||
|
and turn this option on. It will make your atomic writes a bit safer.
|
||||||
|
info_ru: |
|
||||||
|
Данная опция контролирует использование Vitastor OSD флага RWF_ATOMIC при атомарной записи
|
||||||
|
блоков. Этот флаг поддерживается, начиная с версии ядра Linux 6.11 и добавляет немного корректности
|
||||||
|
атомарным записям - ядро гарантирует отсутствие фрагментации запросов записи с этим флагом и
|
||||||
|
проверяет их на соответствие реальным возможностям устройства.
|
||||||
|
|
||||||
|
Однако, данная опция по умолчанию отключена, так как флаг в текущих версиях Linux работает
|
||||||
|
абсолютно НЕКОРРЕКТНО - при нём Linux требует, чтобы запросы записи имели длину, равную
|
||||||
|
степени двойки и были выровнены на эту длину. То есть, например, 12 КБ запросы записи, а также
|
||||||
|
8 КБ запросы записи по не-кратному 8 КБ смещению запрещаются ядром, хотя спецификация NVMe их
|
||||||
|
разрешает.
|
||||||
|
|
||||||
|
Для NVMe-дисков с `scheduler=none` запросы записи и так не фрагментируются, так что это не так
|
||||||
|
уж и важно, однако вы можете пересобрать своё ядро с [этим патчем](../../patches/linux-fix-atomic-write-checks.diff)
|
||||||
|
и включить данную опцию. Это сделает вашу атомарную запись капельку безопаснее.
|
||||||
|
- name: pg_reshard_chunk_size
|
||||||
|
type: int
|
||||||
|
default: 100000
|
||||||
|
info: |
|
||||||
|
Pool PG count change is a CPU-intensive operation because OSDs store the full object database
|
||||||
|
in memory and have to move all entries between old and new PGs. Thus it's performed in chunks,
|
||||||
|
with pauses between chunks to prevent blocking OSD's event loop and other clients' operations.
|
||||||
|
This option sets the maximum number of object is a chunk. Moving 100k objects usually takes
|
||||||
|
50-100ms. Chunk size equal to 0 means unlimited.
|
||||||
|
info_ru: |
|
||||||
|
Изменение числа PG в пуле заметно загружает процессор, так как OSD хранят полную базу данных
|
||||||
|
объектов в памяти и им приходится перемещать все записи объектов между старыми и новыми PG.
|
||||||
|
Поэтому изменение применяется порциями, с паузами между порциями, чтобы не блокировать обработку
|
||||||
|
событий OSD и операции остальных клиентов. Данная опция задаёт максимальное число объектов
|
||||||
|
в порции. Перемещение 100 тысяч объектов (значение по умолчанию) обычно занимает порядка
|
||||||
|
50-100 миллисекунд. Значение опции 0 отключает лимит размера порции.
|
||||||
|
- name: pg_reshard_chunk_pause_ms
|
||||||
|
type: ms
|
||||||
|
default: 100
|
||||||
|
info: |
|
||||||
|
This option sets the interval between handling two PG count change chunks.
|
||||||
|
info_ru: |
|
||||||
|
Данная опция задаёт интервал между обработкой двух порций изменения числа PG пулов.
|
||||||
|
|||||||
@@ -26,9 +26,9 @@ at Vitastor Kubernetes operator: https://github.com/Antilles7227/vitastor-operat
|
|||||||
The instruction is very simple.
|
The instruction is very simple.
|
||||||
|
|
||||||
1. Download a Docker image of the desired version: \
|
1. Download a Docker image of the desired version: \
|
||||||
`docker pull vitalif/vitastor:v2.2.3`
|
`docker pull vitalif/vitastor:v3.0.2`
|
||||||
2. Install scripts to the host system: \
|
2. Install scripts to the host system: \
|
||||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v2.2.3 install.sh`
|
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.2 install.sh`
|
||||||
3. Reload udev rules: \
|
3. Reload udev rules: \
|
||||||
`udevadm control --reload-rules`
|
`udevadm control --reload-rules`
|
||||||
|
|
||||||
|
|||||||
@@ -25,9 +25,9 @@ Vitastor можно установить в Docker/Podman. При этом etcd,
|
|||||||
Инструкция по установке максимально простая.
|
Инструкция по установке максимально простая.
|
||||||
|
|
||||||
1. Скачайте Docker-образ желаемой версии: \
|
1. Скачайте Docker-образ желаемой версии: \
|
||||||
`docker pull vitalif/vitastor:v2.2.3`
|
`docker pull vitalif/vitastor:v3.0.2`
|
||||||
2. Установите скрипты в хост-систему командой: \
|
2. Установите скрипты в хост-систему командой: \
|
||||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v2.2.3 install.sh`
|
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.2 install.sh`
|
||||||
3. Перезагрузите правила udev: \
|
3. Перезагрузите правила udev: \
|
||||||
`udevadm control --reload-rules`
|
`udevadm control --reload-rules`
|
||||||
|
|
||||||
|
|||||||
@@ -30,7 +30,7 @@ volume_backend_name = vitastor-testcluster
|
|||||||
image_volume_cache_enabled = True
|
image_volume_cache_enabled = True
|
||||||
volume_clear = none
|
volume_clear = none
|
||||||
vitastor_etcd_address = 192.168.7.2:2379
|
vitastor_etcd_address = 192.168.7.2:2379
|
||||||
vitastor_etcd_prefix =
|
vitastor_etcd_prefix = /vitastor
|
||||||
vitastor_config_path = /etc/vitastor/vitastor.conf
|
vitastor_config_path = /etc/vitastor/vitastor.conf
|
||||||
vitastor_pool_id = 1
|
vitastor_pool_id = 1
|
||||||
image_upload_use_cinder_backend = True
|
image_upload_use_cinder_backend = True
|
||||||
|
|||||||
@@ -29,7 +29,7 @@ volume_backend_name = vitastor-testcluster
|
|||||||
image_volume_cache_enabled = True
|
image_volume_cache_enabled = True
|
||||||
volume_clear = none
|
volume_clear = none
|
||||||
vitastor_etcd_address = 192.168.7.2:2379
|
vitastor_etcd_address = 192.168.7.2:2379
|
||||||
vitastor_etcd_prefix =
|
vitastor_etcd_prefix = /vitastor
|
||||||
vitastor_config_path = /etc/vitastor/vitastor.conf
|
vitastor_config_path = /etc/vitastor/vitastor.conf
|
||||||
vitastor_pool_id = 1
|
vitastor_pool_id = 1
|
||||||
image_upload_use_cinder_backend = True
|
image_upload_use_cinder_backend = True
|
||||||
|
|||||||
@@ -11,13 +11,20 @@
|
|||||||
- Trust Vitastor package signing key:
|
- Trust Vitastor package signing key:
|
||||||
`wget https://vitastor.io/debian/pubkey.gpg -O /etc/apt/trusted.gpg.d/vitastor.gpg`
|
`wget https://vitastor.io/debian/pubkey.gpg -O /etc/apt/trusted.gpg.d/vitastor.gpg`
|
||||||
- Add Vitastor package repository to your /etc/apt/sources.list:
|
- Add Vitastor package repository to your /etc/apt/sources.list:
|
||||||
- Debian 12 (Bookworm/Sid): `deb https://vitastor.io/debian bookworm main`
|
- Debian 13 (Trixie/Sid): `deb https://vitastor.io/debian trixie main`
|
||||||
|
- Debian 12 (Bookworm): `deb https://vitastor.io/debian bookworm main`
|
||||||
- Debian 11 (Bullseye): `deb https://vitastor.io/debian bullseye main`
|
- Debian 11 (Bullseye): `deb https://vitastor.io/debian bullseye main`
|
||||||
- Debian 10 (Buster): `deb https://vitastor.io/debian buster main`
|
- Debian 10 (Buster): `deb https://vitastor.io/debian buster main`
|
||||||
- Ubuntu 22.04 (Jammy): `deb https://vitastor.io/debian jammy main`
|
- Ubuntu 22.04 (Jammy): `deb https://vitastor.io/debian jammy main`
|
||||||
- Ubuntu 24.04 (Noble): `deb https://vitastor.io/debian noble main`
|
- Ubuntu 24.04 (Noble): `deb https://vitastor.io/debian noble main`
|
||||||
- Add `-oldstable` to bookworm/bullseye/buster in this line to install the last
|
- Add `-oldstable` to bookworm/bullseye/buster in this line to install the last
|
||||||
stable version from 0.9.x branch instead of 1.x
|
stable version from 0.9.x branch instead of 1.x
|
||||||
|
- To always prefer vitastor-patched QEMU and Libvirt versions, add the following to `/etc/apt/preferences`:
|
||||||
|
```
|
||||||
|
Package: *
|
||||||
|
Pin: origin "vitastor.io"
|
||||||
|
Pin-Priority: 501
|
||||||
|
```
|
||||||
- Install packages: `apt update; apt install vitastor lp-solve etcd linux-image-amd64 qemu-system-x86`
|
- Install packages: `apt update; apt install vitastor lp-solve etcd linux-image-amd64 qemu-system-x86`
|
||||||
|
|
||||||
## CentOS
|
## CentOS
|
||||||
@@ -43,7 +50,6 @@
|
|||||||
recommended because io_uring is a relatively new technology and there is
|
recommended because io_uring is a relatively new technology and there is
|
||||||
at least one bug which reproduces with io_uring and HP SmartArray
|
at least one bug which reproduces with io_uring and HP SmartArray
|
||||||
controllers in 5.4
|
controllers in 5.4
|
||||||
- liburing 0.4 or newer
|
|
||||||
- lp_solve
|
- lp_solve
|
||||||
- etcd 3.4.15 or newer. Earlier versions won't work because of various bugs,
|
- etcd 3.4.15 or newer. Earlier versions won't work because of various bugs,
|
||||||
for example [#12402](https://github.com/etcd-io/etcd/pull/12402).
|
for example [#12402](https://github.com/etcd-io/etcd/pull/12402).
|
||||||
|
|||||||
@@ -11,13 +11,20 @@
|
|||||||
- Добавьте ключ репозитория Vitastor:
|
- Добавьте ключ репозитория Vitastor:
|
||||||
`wget https://vitastor.io/debian/pubkey.gpg -O /etc/apt/trusted.gpg.d/vitastor.gpg`
|
`wget https://vitastor.io/debian/pubkey.gpg -O /etc/apt/trusted.gpg.d/vitastor.gpg`
|
||||||
- Добавьте репозиторий Vitastor в /etc/apt/sources.list:
|
- Добавьте репозиторий Vitastor в /etc/apt/sources.list:
|
||||||
- Debian 12 (Bookworm/Sid): `deb https://vitastor.io/debian bookworm main`
|
- Debian 13 (Trixie/Sid): `deb https://vitastor.io/debian trixie main`
|
||||||
|
- Debian 12 (Bookworm): `deb https://vitastor.io/debian bookworm main`
|
||||||
- Debian 11 (Bullseye): `deb https://vitastor.io/debian bullseye main`
|
- Debian 11 (Bullseye): `deb https://vitastor.io/debian bullseye main`
|
||||||
- Debian 10 (Buster): `deb https://vitastor.io/debian buster main`
|
- Debian 10 (Buster): `deb https://vitastor.io/debian buster main`
|
||||||
- Ubuntu 22.04 (Jammy): `deb https://vitastor.io/debian jammy main`
|
- Ubuntu 22.04 (Jammy): `deb https://vitastor.io/debian jammy main`
|
||||||
- Ubuntu 24.04 (Noble): `deb https://vitastor.io/debian noble main`
|
- Ubuntu 24.04 (Noble): `deb https://vitastor.io/debian noble main`
|
||||||
- Добавьте `-oldstable` к слову bookworm/bullseye/buster в этой строке, чтобы
|
- Добавьте `-oldstable` к слову bookworm/bullseye/buster в этой строке, чтобы
|
||||||
установить последнюю стабильную версию из ветки 0.9.x вместо 1.x
|
установить последнюю стабильную версию из ветки 0.9.x вместо 1.x
|
||||||
|
- Чтобы всегда предпочитались версии пакетов QEMU и Libvirt с патчами Vitastor, добавьте в `/etc/apt/preferences`:
|
||||||
|
```
|
||||||
|
Package: *
|
||||||
|
Pin: origin "vitastor.io"
|
||||||
|
Pin-Priority: 501
|
||||||
|
```
|
||||||
- Установите пакеты: `apt update; apt install vitastor lp-solve etcd linux-image-amd64 qemu-system-x86`
|
- Установите пакеты: `apt update; apt install vitastor lp-solve etcd linux-image-amd64 qemu-system-x86`
|
||||||
|
|
||||||
## CentOS
|
## CentOS
|
||||||
@@ -42,7 +49,6 @@
|
|||||||
- Ядро Linux 5.4 или новее, для поддержки io_uring. Рекомендуется даже 5.8,
|
- Ядро Linux 5.4 или новее, для поддержки io_uring. Рекомендуется даже 5.8,
|
||||||
так как io_uring - относительно новый интерфейс и в версиях до 5.8 встречались
|
так как io_uring - относительно новый интерфейс и в версиях до 5.8 встречались
|
||||||
некоторые баги, например, зависание с io_uring и контроллером HP SmartArray
|
некоторые баги, например, зависание с io_uring и контроллером HP SmartArray
|
||||||
- liburing 0.4 или новее
|
|
||||||
- lp_solve
|
- lp_solve
|
||||||
- etcd 3.4.15 или новее. Более старые версии не будут работать из-за разных багов,
|
- etcd 3.4.15 или новее. Более старые версии не будут работать из-за разных багов,
|
||||||
например, [#12402](https://github.com/etcd-io/etcd/pull/12402).
|
например, [#12402](https://github.com/etcd-io/etcd/pull/12402).
|
||||||
|
|||||||
@@ -6,10 +6,10 @@
|
|||||||
|
|
||||||
# Proxmox VE
|
# Proxmox VE
|
||||||
|
|
||||||
To enable Vitastor support in Proxmox Virtual Environment (6.4-8.x are supported):
|
To enable Vitastor support in Proxmox Virtual Environment (6.4-9.x are supported):
|
||||||
|
|
||||||
- Add the corresponding Vitastor Debian repository into sources.list on Proxmox hosts:
|
- Add the corresponding Vitastor Debian repository into sources.list on Proxmox hosts:
|
||||||
bookworm for 8.1+, pve8.0 for 8.0, bullseye for 7.4, pve7.3 for 7.3, pve7.2 for 7.2, pve7.1 for 7.1, buster for 6.4
|
trixie for 9.0+, bookworm for 8.1+, pve8.0 for 8.0, bullseye for 7.4, pve7.3 for 7.3, pve7.2 for 7.2, pve7.1 for 7.1, buster for 6.4
|
||||||
- Install vitastor-client, pve-qemu-kvm, pve-storage-vitastor (* or see note) packages from Vitastor repository
|
- Install vitastor-client, pve-qemu-kvm, pve-storage-vitastor (* or see note) packages from Vitastor repository
|
||||||
- Define storage in `/etc/pve/storage.cfg` (see below)
|
- Define storage in `/etc/pve/storage.cfg` (see below)
|
||||||
- Block network access from VMs to Vitastor network (to OSDs and etcd),
|
- Block network access from VMs to Vitastor network (to OSDs and etcd),
|
||||||
|
|||||||
@@ -6,10 +6,10 @@
|
|||||||
|
|
||||||
# Proxmox VE
|
# Proxmox VE
|
||||||
|
|
||||||
Чтобы подключить Vitastor к Proxmox Virtual Environment (поддерживаются версии 6.4-8.x):
|
Чтобы подключить Vitastor к Proxmox Virtual Environment (поддерживаются версии 6.4-9.x):
|
||||||
|
|
||||||
- Добавьте соответствующий Debian-репозиторий Vitastor в sources.list на хостах Proxmox:
|
- Добавьте соответствующий Debian-репозиторий Vitastor в sources.list на хостах Proxmox:
|
||||||
bookworm для 8.1+, pve8.0 для 8.0, bullseye для 7.4, pve7.3 для 7.3, pve7.2 для 7.2, pve7.1 для 7.1, buster для 6.4
|
trixie для 9.0+, bookworm для 8.1+, pve8.0 для 8.0, bullseye для 7.4, pve7.3 для 7.3, pve7.2 для 7.2, pve7.1 для 7.1, buster для 6.4
|
||||||
- Установите пакеты vitastor-client, pve-qemu-kvm, pve-storage-vitastor (* или см. сноску) из репозитория Vitastor
|
- Установите пакеты vitastor-client, pve-qemu-kvm, pve-storage-vitastor (* или см. сноску) из репозитория Vitastor
|
||||||
- Определите тип хранилища в `/etc/pve/storage.cfg` (см. ниже)
|
- Определите тип хранилища в `/etc/pve/storage.cfg` (см. ниже)
|
||||||
- Обязательно заблокируйте доступ от виртуальных машин к сети Vitastor (OSD и etcd), т.к. Vitastor (пока) не поддерживает аутентификацию
|
- Обязательно заблокируйте доступ от виртуальных машин к сети Vitastor (OSD и etcd), т.к. Vitastor (пока) не поддерживает аутентификацию
|
||||||
|
|||||||
@@ -15,7 +15,7 @@
|
|||||||
- gcc and g++ 8 or newer, clang 10 or newer, or other compiler with C++11 plus
|
- gcc and g++ 8 or newer, clang 10 or newer, or other compiler with C++11 plus
|
||||||
designated initializers support from C++20
|
designated initializers support from C++20
|
||||||
- CMake
|
- CMake
|
||||||
- liburing, jerasure headers and libraries
|
- jerasure headers and libraries
|
||||||
- ISA-L, libibverbs and librdmacm headers and libraries (optional)
|
- ISA-L, libibverbs and librdmacm headers and libraries (optional)
|
||||||
- tcmalloc (google-perftools-dev)
|
- tcmalloc (google-perftools-dev)
|
||||||
|
|
||||||
|
|||||||
@@ -15,7 +15,7 @@
|
|||||||
- gcc и g++ >= 8, либо clang >= 10, либо другой компилятор с поддержкой C++11 плюс
|
- gcc и g++ >= 8, либо clang >= 10, либо другой компилятор с поддержкой C++11 плюс
|
||||||
назначенных инициализаторов (designated initializers) из C++20
|
назначенных инициализаторов (designated initializers) из C++20
|
||||||
- CMake
|
- CMake
|
||||||
- Заголовки и библиотеки liburing, jerasure
|
- Заголовки и библиотеки jerasure
|
||||||
- Опционально - заголовки и библиотеки ISA-L, libibverbs, librdmacm
|
- Опционально - заголовки и библиотеки ISA-L, libibverbs, librdmacm
|
||||||
- tcmalloc (google-perftools-dev)
|
- tcmalloc (google-perftools-dev)
|
||||||
|
|
||||||
|
|||||||
@@ -14,6 +14,8 @@
|
|||||||
|
|
||||||
- Basic part: highly-available block storage with symmetric clustering and no SPOF
|
- Basic part: highly-available block storage with symmetric clustering and no SPOF
|
||||||
- [Performance](../performance/bench2.en.md) ;-D
|
- [Performance](../performance/bench2.en.md) ;-D
|
||||||
|
- [NVMe atomic write support](../config/osd.en.md#atomic_write_size) for reducing the amount
|
||||||
|
of "extra" disk writes to almost zero (Write Amplification = 1)
|
||||||
- [Multiple redundancy schemes](../config/pool.en.md#scheme): Replication, XOR n+1, Reed-Solomon erasure codes
|
- [Multiple redundancy schemes](../config/pool.en.md#scheme): Replication, XOR n+1, Reed-Solomon erasure codes
|
||||||
based on jerasure and ISA-L libraries with any number of data and parity drives in a group
|
based on jerasure and ISA-L libraries with any number of data and parity drives in a group
|
||||||
- Configuration via simple JSON data structures in etcd (parameters, pools and images)
|
- Configuration via simple JSON data structures in etcd (parameters, pools and images)
|
||||||
@@ -52,7 +54,7 @@
|
|||||||
- Generic user-space client library
|
- Generic user-space client library
|
||||||
- [Native QEMU driver](../usage/qemu.en.md)
|
- [Native QEMU driver](../usage/qemu.en.md)
|
||||||
- [Loadable fio engine for benchmarks](../usage/fio.en.md)
|
- [Loadable fio engine for benchmarks](../usage/fio.en.md)
|
||||||
- [NBD proxy for kernel mounts](../usage/nbd.en.md)
|
- [UBLK](../usage/ublk.en.md) and [NBD](../usage/nbd.en.md) servers for kernel mounts
|
||||||
- [Simplified NFS proxy for file-based image access emulation (suitable for VMWare)](../usage/nfs.en.md#pseudo-fs)
|
- [Simplified NFS proxy for file-based image access emulation (suitable for VMWare)](../usage/nfs.en.md#pseudo-fs)
|
||||||
|
|
||||||
## Roadmap
|
## Roadmap
|
||||||
|
|||||||
@@ -14,6 +14,8 @@
|
|||||||
|
|
||||||
- Базовая часть - надёжное кластерное блочное хранилище без единой точки отказа
|
- Базовая часть - надёжное кластерное блочное хранилище без единой точки отказа
|
||||||
- [Производительность](../performance/bench2.ru.md) ;-D
|
- [Производительность](../performance/bench2.ru.md) ;-D
|
||||||
|
- [Поддержка атомарной записи NVMe](../config/osd.ru.md#atomic_write_size) для снижения объёма
|
||||||
|
служебной записи практически до нуля (Write Amplification = 1)
|
||||||
- [Несколько схем отказоустойчивости](../config/pool.ru.md#scheme): репликация, XOR n+1 (1 диск чётности), коды коррекции ошибок
|
- [Несколько схем отказоустойчивости](../config/pool.ru.md#scheme): репликация, XOR n+1 (1 диск чётности), коды коррекции ошибок
|
||||||
Рида-Соломона на основе библиотек jerasure и ISA-L с любым числом дисков данных и чётности в группе
|
Рида-Соломона на основе библиотек jerasure и ISA-L с любым числом дисков данных и чётности в группе
|
||||||
- Конфигурация через простые человекочитаемые JSON-структуры в etcd
|
- Конфигурация через простые человекочитаемые JSON-структуры в etcd
|
||||||
@@ -54,7 +56,7 @@
|
|||||||
- Общая пользовательская клиентская библиотека для работы с кластером
|
- Общая пользовательская клиентская библиотека для работы с кластером
|
||||||
- [Драйвер диска для QEMU](../usage/qemu.ru.md)
|
- [Драйвер диска для QEMU](../usage/qemu.ru.md)
|
||||||
- [Драйвер диска для утилиты тестирования производительности fio](../usage/fio.ru.md)
|
- [Драйвер диска для утилиты тестирования производительности fio](../usage/fio.ru.md)
|
||||||
- [NBD-прокси для монтирования образов ядром](../usage/nbd.ru.md) ("блочное устройство в режиме пользователя")
|
- [UBLK](../usage/ublk.ru.md) и [NBD](../usage/nbd.ru.md) серверы для монтирования образов ядром ("блочное устройство в режиме пользователя")
|
||||||
- [Упрощённая NFS-прокси для эмуляции файлового доступа к образам (подходит для VMWare)](../usage/nfs.ru.md#псевдо-фс)
|
- [Упрощённая NFS-прокси для эмуляции файлового доступа к образам (подходит для VMWare)](../usage/nfs.ru.md#псевдо-фс)
|
||||||
|
|
||||||
## Планы развития
|
## Планы развития
|
||||||
|
|||||||
@@ -18,9 +18,10 @@
|
|||||||
|
|
||||||
## Preparation
|
## Preparation
|
||||||
|
|
||||||
- Get some SATA or NVMe SSDs with capacitors (server-grade drives). You can use desktop SSDs
|
- Get some SATA or NVMe SSDs with capacitors (server-grade drives). The best performance
|
||||||
with lazy fsync, but prepare for inferior single-thread latency. Read more about capacitors
|
is achieved with Micron or Kioxia NVMes with atomic write support (see below). You can use desktop
|
||||||
[here](../config/layout-cluster.en.md#immediate_commit).
|
SSDs with lazy fsync, but prepare for inferior single-thread latency. Read more about
|
||||||
|
capacitors [here](../config/layout-cluster.en.md#immediate_commit).
|
||||||
- If you want to use HDDs, get modern HDDs with Media Cache or SSD Cache: HGST Ultrastar,
|
- If you want to use HDDs, get modern HDDs with Media Cache or SSD Cache: HGST Ultrastar,
|
||||||
Toshiba MG, Seagate EXOS or something similar. If your drives don't have such cache then
|
Toshiba MG, Seagate EXOS or something similar. If your drives don't have such cache then
|
||||||
you also need small SSDs for journal and metadata (even 2 GB per 1 TB of HDD space is enough).
|
you also need small SSDs for journal and metadata (even 2 GB per 1 TB of HDD space is enough).
|
||||||
@@ -30,9 +31,11 @@
|
|||||||
|
|
||||||
## Recommended drives
|
## Recommended drives
|
||||||
|
|
||||||
- SATA SSD: Micron 5100/5200/5300/5400, Samsung PM863/PM883/PM893, Intel D3-S4510/4520/4610/4620, Kingston DC500M
|
- NVMe with atomic write support (ideal!): Micron 7450/7500/7550, Kioxia CD6/CD7/CD8/CD9
|
||||||
- NVMe: Micron 9100/9200/9300/9400, Micron 7300/7450, Samsung PM983/PM9A3, Samsung PM1723/1735/1743,
|
- Other NVMe: Micron 9100/9200/9300/9400/9550, Micron 7300, Samsung PM983/PM9A3, Samsung PM1723/1735/1743,
|
||||||
Intel DC-P3700/P4500/P4600, Intel D5-P4320/P5530, Intel D7-P5500/P5600, Intel Optane, Kingston DC1000B/DC1500M
|
Intel DC-P3700/P4500/P4600, Intel/Solidigm D5-P4320/P5530, Intel/Solidigm D7-P5500/P5600, Solidigm D7-PS1010/PS1030/P5810,
|
||||||
|
Intel Optane, Kingston DC1000B/DC1500M, Kioxia CD6/CD7/CD8/CD9
|
||||||
|
- SATA SSD: Micron 5100/5200/5300/5400, Samsung PM863/PM883/PM893, Intel/Solidigm D3-S4510/4520/4610/4620, Kingston DC500M
|
||||||
- HDD: HGST Ultrastar, Toshiba MG, Seagate EXOS
|
- HDD: HGST Ultrastar, Toshiba MG, Seagate EXOS
|
||||||
|
|
||||||
## Configure monitors
|
## Configure monitors
|
||||||
|
|||||||
@@ -18,8 +18,9 @@
|
|||||||
|
|
||||||
## Подготовка
|
## Подготовка
|
||||||
|
|
||||||
- Возьмите серверы с SSD (SATA или NVMe), желательно с конденсаторами (серверные SSD). Можно
|
- Возьмите серверы с SSD (SATA или NVMe), желательно с конденсаторами (серверные SSD). Наилучшая
|
||||||
использовать и десктопные SSD, включив режим отложенного fsync, но производительность будет хуже.
|
производительность достигается на дисках Micron и Kioxia с поддержкой атомарной записи (см. ниже).
|
||||||
|
Можно использовать и десктопные SSD, включив режим отложенного fsync, но производительность будет хуже.
|
||||||
О конденсаторах читайте [здесь](../config/layout-cluster.ru.md#immediate_commit).
|
О конденсаторах читайте [здесь](../config/layout-cluster.ru.md#immediate_commit).
|
||||||
- Если хотите использовать HDD, берите современные модели с Media или SSD кэшем - HGST Ultrastar,
|
- Если хотите использовать HDD, берите современные модели с Media или SSD кэшем - HGST Ultrastar,
|
||||||
Toshiba MG, Seagate EXOS или что-то похожее. Если такого кэша у ваших дисков нет,
|
Toshiba MG, Seagate EXOS или что-то похожее. Если такого кэша у ваших дисков нет,
|
||||||
@@ -30,9 +31,11 @@
|
|||||||
|
|
||||||
## Рекомендуемые диски
|
## Рекомендуемые диски
|
||||||
|
|
||||||
- SATA SSD: Micron 5100/5200/5300/5400, Samsung PM863/PM883/PM893, Intel D3-S4510/4520/4610/4620, Kingston DC500M
|
- NVMe с поддержкой атомарной записи (идеально!): Micron 7450/7500/7550, Kioxia CD6/CD7/CD8/CD9
|
||||||
- NVMe: Micron 9100/9200/9300/9400, Micron 7300/7450, Samsung PM983/PM9A3, Samsung PM1723/1735/1743,
|
- Другие NVMe: Micron 9100/9200/9300/9400/9550, Micron 7300, Samsung PM983/PM9A3, Samsung PM1723/1735/1743,
|
||||||
Intel DC-P3700/P4500/P4600, Intel D5-P4320/P5530, Intel D7-P5500/P5600, Intel Optane, Kingston DC1000B/DC1500M
|
Intel DC-P3700/P4500/P4600, Intel/Solidigm D5-P4320/P5530, Intel/Solidigm D7-P5500/P5600, Solidigm D7-PS1010/PS1030/P5810,
|
||||||
|
Intel Optane, Kingston DC1000B/DC1500M, Kioxia CD6/CD7/CD8/CD9
|
||||||
|
- SATA SSD: Micron 5100/5200/5300/5400, Samsung PM863/PM883/PM893, Intel/Solidigm D3-S4510/4520/4610/4620, Kingston DC500M
|
||||||
- HDD: HGST Ultrastar, Toshiba MG, Seagate EXOS
|
- HDD: HGST Ultrastar, Toshiba MG, Seagate EXOS
|
||||||
|
|
||||||
## Настройте мониторы
|
## Настройте мониторы
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ Replicated setups:
|
|||||||
- Linear read: `min(total network bandwidth, sum(disk read MB/s))`.
|
- Linear read: `min(total network bandwidth, sum(disk read MB/s))`.
|
||||||
- Linear write: `min(total network bandwidth, sum(disk write MB/s / number of replicas))`.
|
- Linear write: `min(total network bandwidth, sum(disk write MB/s / number of replicas))`.
|
||||||
- Saturated parallel read iops: `min(total network bandwidth, sum(disk read iops))`.
|
- Saturated parallel read iops: `min(total network bandwidth, sum(disk read iops))`.
|
||||||
- Saturated parallel write iops: `min(total network bandwidth / number of replicas, sum(disk write iops / number of replicas / (write amplification = 4)))`.
|
- Saturated parallel write iops: `min(total network bandwidth / number of replicas, sum(disk write iops / number of replicas / write amplification))`.
|
||||||
|
|
||||||
EC/XOR setups (EC N+K):
|
EC/XOR setups (EC N+K):
|
||||||
- Single-threaded (T1Q1) read latency: 1.5 network roundtrips + 1 disk read.
|
- Single-threaded (T1Q1) read latency: 1.5 network roundtrips + 1 disk read.
|
||||||
@@ -26,28 +26,36 @@ EC/XOR setups (EC N+K):
|
|||||||
- Linear read: `min(total network bandwidth, sum(disk read MB/s))`.
|
- Linear read: `min(total network bandwidth, sum(disk read MB/s))`.
|
||||||
- Linear write: `min(total network bandwidth, sum(disk write MB/s * N/(N+K)))`.
|
- Linear write: `min(total network bandwidth, sum(disk write MB/s * N/(N+K)))`.
|
||||||
- Saturated parallel read iops: `min(total network bandwidth, sum(disk read iops))`.
|
- Saturated parallel read iops: `min(total network bandwidth, sum(disk read iops))`.
|
||||||
- Saturated parallel write iops: roughly `total iops / (N+K) / WA`. More exactly,
|
- Saturated parallel write iops: roughly `total iops / (N+K) / WA`. More exactly:
|
||||||
`min(total network bandwidth * N/(N+K), sum(disk randrw iops / (N*4 + K*5 + 1)))` with
|
- With the new store: `min(total network bandwidth * N/(N+K), sum(disk randrw iops / (2 + N-1 + K*2)))`,
|
||||||
random read/write mix corresponding to `(N-1)/(N*4 + K*5 + 1)*100 % reads`.
|
with random read/write mix corresponding to `(N-1)/(2 + N-1 + K*2)*100 % reads`.
|
||||||
- For example, with EC 2+1 it is: `(7% randrw iops) / 14`.
|
- For example, with EC 2+1 it is: `(20% randrw iops) / 5`.
|
||||||
- With EC 6+3 it is: `(12.5% randrw iops) / 40`.
|
- With EC 6+3 it is: `(38% randrw iops) / 13`.
|
||||||
|
- With the old store: `min(total network bandwidth * N/(N+K), sum(disk randrw iops / (3 + N-1 + K*3)))`,
|
||||||
|
with random read/write mix corresponding to `(N-1)/(3 + N-1 + K*3)*100 % reads`.
|
||||||
|
- For example, with EC 2+1 it is: `(14% randrw iops) / 7`.
|
||||||
|
- With EC 6+3 it is: `(30% randrw iops) / 17`.
|
||||||
|
|
||||||
Write amplification for 4 KB blocks is usually 3-5 in Vitastor:
|
Write Amplification factor:
|
||||||
1. Journal block write
|
- For the new store and for 4 KB writes: WA is always 1 unless you set [atomic_write_size](../config/osd.en.md#atomic_write_size) to 0 manually.
|
||||||
2. Journal data write
|
- For the new store and for 8-124 KB writes: WA is 1 if you use NVMe drives with atomic write support, or roughly 2 if you use other drives.
|
||||||
3. Metadata block write
|
- For the old store, WA is roughly `(2 * write size + 4 KB) / (write size)`. So, for 4 KB writes it's 3, and for 8-124 KB writes it's closer to 2.
|
||||||
4. Another journal block write for EC/XOR setups
|
- For both the new and the old store and for writes of [block_size](../config/layout-cluster.en.md#block_size): WA is almost 1.
|
||||||
5. Data block write
|
|
||||||
|
|
||||||
If you manage to get an SSD which handles 512 byte blocks well (Optane?) you may
|
Write Amplification consists of:
|
||||||
lower 1, 3 and 4 to 512 bytes (1/8 of data size) and get WA as low as 2.375.
|
- For the new store:
|
||||||
|
- Buffer block write if non-atomic
|
||||||
|
- Data block write
|
||||||
|
- Metadata write(s) (amortized)
|
||||||
|
- For the old store:
|
||||||
|
- Journal block write (amortized)
|
||||||
|
- Journal data write
|
||||||
|
- Metadata block write
|
||||||
|
- Another journal block write for EC/XOR setups (amortized)
|
||||||
|
- Data block write
|
||||||
|
|
||||||
Implemented NVDIMM support can basically eliminate WA at all - all extra writes will
|
Other possibilities to reduce WA would be to use SSDs with internal 512-byte blocks
|
||||||
go to DRAM memory. But this requires a test cluster with NVDIMM - please contact me
|
or NVDIMM, but both options seem unavailable on the market at the moment.
|
||||||
if you want to provide me with such cluster for tests.
|
|
||||||
|
|
||||||
Lazy fsync also reduces WA for parallel workloads because journal blocks are only
|
|
||||||
written when they fill up or fsync is requested.
|
|
||||||
|
|
||||||
## In Practice
|
## In Practice
|
||||||
|
|
||||||
|
|||||||
@@ -27,29 +27,36 @@
|
|||||||
- Линейное чтение: сумма МБ/с чтения всех дисков, либо общая производительность сети, если в сеть упрётся раньше.
|
- Линейное чтение: сумма МБ/с чтения всех дисков, либо общая производительность сети, если в сеть упрётся раньше.
|
||||||
- Линейная запись: сумма МБ/с записи всех дисков * N/(N+K), либо производительность сети * N / (N+K), если в сеть упрётся раньше.
|
- Линейная запись: сумма МБ/с записи всех дисков * N/(N+K), либо производительность сети * N / (N+K), если в сеть упрётся раньше.
|
||||||
- Параллельное случайное мелкое чтение: сумма IOPS чтения всех дисков либо производительность сети, если в сеть упрётся раньше.
|
- Параллельное случайное мелкое чтение: сумма IOPS чтения всех дисков либо производительность сети, если в сеть упрётся раньше.
|
||||||
- Параллельная случайная мелкая запись: грубо `(сумма IOPS / (N+K) / WA)`. Если точнее, то:
|
- Параллельная случайная мелкая запись: грубо `(сумма IOPS / (N+K) / WA)`.
|
||||||
сумма смешанного IOPS всех дисков при `(N-1)/(N*4 + K*5 + 1)*100 %` чтения, делённая на `(N*4 + K*5 + 1)`.
|
Либо `производительность сети * N/(N+K)`, если в сеть упрётся раньше. Если точнее, то:
|
||||||
Либо, производительность сети * N/(N+K), если в сеть упрётся раньше.
|
- С новым хранилищем: сумма смешанного IOPS всех дисков при `(N-1)/(2 + N-1 + K*2)*100 %` чтения, делённая на `(2 + N-1 + K*2)`.
|
||||||
- Например, при EC 2+1 это: `(сумма IOPS при 7% чтения) / 14`.
|
- Например, при EC 2+1 это: `(сумма IOPS при 20% чтения) / 5`.
|
||||||
- При EC 6+3 это: `(сумма IOPS при 12.5% чтения) / 40`.
|
- При EC 6+3 это: `(сумма IOPS при 38% чтения) / 13`.
|
||||||
|
- Со старым хранилищем: сумма смешанного IOPS всех дисков при `(N-1)/(3 + N-1 + K*3)*100 %` чтения, делённая на `(3 + N-1 + K*3)`.
|
||||||
|
- Например, при EC 2+1 это: `(сумма IOPS при 14% чтения) / 7`.
|
||||||
|
- При EC 6+3 это: `(сумма IOPS при 30% чтения) / 17`.
|
||||||
|
|
||||||
WA (мультипликатор записи) для 4 КБ блоков в Vitastor обычно составляет 3-5:
|
WA (Write Amplification, мультипликатор записи):
|
||||||
1. Запись метаданных в журнал
|
- С новым хранилищем для 4 КБ записи: WA всегда примерно 1, если только вы не установите [atomic_write_size](../config/osd.ru.md#atomic_write_size) вручную в 0.
|
||||||
2. Запись блока данных в журнал
|
- С новым хранилищем и большими записями (8-124 КБ): WA примерно 1, если вы используете NVMe-диски с поддержкой атомарной записи,
|
||||||
3. Запись метаданных в БД
|
или примерно 2, если вы используете другие диски.
|
||||||
4. Ещё одна запись метаданных в журнал при использовании EC
|
- Со старым хранилищем, WA примерно `(2 * размер записи + 4 КБ) / (размер записи)`. То есть, для 4 КБ записи WA=3, а для 8-124 КБ WA ближе к 2.
|
||||||
5. Запись блока данных на диск данных
|
- И с новым, и со старым хранилищем и для записи размером [block_size](../config/layout-cluster.ru.md#block_size): WA примерно равен 1.
|
||||||
|
|
||||||
Если вы найдёте SSD, хорошо работающий с 512-байтными блоками данных (Optane?),
|
Мультипликатор записи состоит из:
|
||||||
то 1, 3 и 4 можно снизить до 512 байт (1/8 от размера данных) и получить WA всего 2.375.
|
- С новым хранилищем:
|
||||||
|
- Запись блока буфера, если диски без поддержки атомарной записи
|
||||||
|
- Запись блока данных
|
||||||
|
- Запись(-и) блоков метаданных (амортизированные)
|
||||||
|
- Со старым хранилищем:
|
||||||
|
- Запись блока журнала (амортизированная)
|
||||||
|
- Запись данных в журнал
|
||||||
|
- Запись блока метаданных
|
||||||
|
- Ещё одна запись блока журнала для EC/XOR пулов (амортизированная)
|
||||||
|
- Запись блока данных
|
||||||
|
|
||||||
Если реализовать поддержку NVDIMM, то WA можно, условно говоря, ликвидировать вообще - все
|
Другими потенциальными возможностями снижения WA могли бы быть SSD с внутренним 512-байтным блоком
|
||||||
дополнительные операции записи смогут обслуживаться DRAM памятью. Но для этого необходим
|
либо NVDIMM, но и то, и другое сейчас выглядит недоступным на рынке.
|
||||||
тестовый кластер с NVDIMM - пишите, если готовы предоставить такой для тестов.
|
|
||||||
|
|
||||||
Кроме того, WA снижается при использовании отложенного/ленивого сброса при параллельной
|
|
||||||
нагрузке, т.к. блоки журнала записываются на диск только когда они заполняются или явным
|
|
||||||
образом запрашивается fsync.
|
|
||||||
|
|
||||||
## На практике
|
## На практике
|
||||||
|
|
||||||
|
|||||||
@@ -231,6 +231,18 @@ Upgrading from <= 0.5.x to >= 0.6.x is not supported.
|
|||||||
|
|
||||||
Downgrade are also allowed freely, except the following specific instructions:
|
Downgrade are also allowed freely, except the following specific instructions:
|
||||||
|
|
||||||
|
### 3.x -> 2.x
|
||||||
|
|
||||||
|
Versions 3.0.0 and newer contain two store implementations - an old one and a new
|
||||||
|
one, unsupported in 2.x and previous versions. So you should check your OSD store
|
||||||
|
versions before downgrading to 2.x with the following command:
|
||||||
|
|
||||||
|
`vitastor-disk read-sb /dev/vitastor/osdXX-data | jq -r .meta_format`
|
||||||
|
|
||||||
|
If it prints 3 then OSD uses the new store and you can't downgrade it to 2.x.
|
||||||
|
|
||||||
|
If it prints 2 or nothing then OSD uses the old store and the downgrade is allowed.
|
||||||
|
|
||||||
### 1.8.0 to 1.7.1
|
### 1.8.0 to 1.7.1
|
||||||
|
|
||||||
Before downgrading from version >= 1.8.0 to version <= 1.7.1
|
Before downgrading from version >= 1.8.0 to version <= 1.7.1
|
||||||
|
|||||||
@@ -228,6 +228,18 @@ done
|
|||||||
|
|
||||||
Откат (понижение версии) тоже свободно разрешён, кроме указанных ниже случаев:
|
Откат (понижение версии) тоже свободно разрешён, кроме указанных ниже случаев:
|
||||||
|
|
||||||
|
### 3.x -> 2.x
|
||||||
|
|
||||||
|
Версии 3.0.0 и более новые содержат две реализации хранилища - старую и новую, не
|
||||||
|
поддерживаемую в 2.x и предыдущих версиях. Таким образом, перед откатом на 2.x вам
|
||||||
|
следует проверить, какая версия хранилища используется вашими OSD - командой:
|
||||||
|
|
||||||
|
`vitastor-disk read-sb /dev/vitastor/osdXX-data | jq -r .meta_format`
|
||||||
|
|
||||||
|
Если выводится 3, это новое хранилище и откатить такой OSD до 2.x нельзя.
|
||||||
|
|
||||||
|
Если выводится 2 или не выводится ничего, это старое хранилище и откат разрешён.
|
||||||
|
|
||||||
### 1.8.0 -> 1.7.1
|
### 1.8.0 -> 1.7.1
|
||||||
|
|
||||||
Перед понижением версии с >= 1.8.0 до <= 1.7.1 вы должны скопировать ключ
|
Перед понижением версии с >= 1.8.0 до <= 1.7.1 вы должны скопировать ключ
|
||||||
|
|||||||
@@ -100,12 +100,14 @@ List images (only matching `<glob>` pattern(s) if passed).
|
|||||||
Options:
|
Options:
|
||||||
|
|
||||||
```
|
```
|
||||||
|
--exact Do not match glob patterns as names, select only exact name matches.
|
||||||
-p|--pool POOL Filter images by pool ID or name
|
-p|--pool POOL Filter images by pool ID or name
|
||||||
-l|--long Also report allocated size and I/O statistics
|
-l|--long Also report allocated size and I/O statistics
|
||||||
--del Also include delete operation statistics
|
--del Also include delete operation statistics
|
||||||
--sort FIELD Sort by specified field (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)
|
--sort FIELD Sort by specified field (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)
|
||||||
-r|--reverse Sort in descending order
|
-r|--reverse Sort in descending order
|
||||||
-n|--count N Only list first N items
|
-n|--count N Only list first N items
|
||||||
|
--tree Show image snapshot/clone tree
|
||||||
```
|
```
|
||||||
|
|
||||||
Example output:
|
Example output:
|
||||||
|
|||||||
@@ -102,12 +102,14 @@ kaveri 2/1 32 0 B 10 G 0 B 100% 0%
|
|||||||
Опции:
|
Опции:
|
||||||
|
|
||||||
```
|
```
|
||||||
|
--exact Не применять ФС-шаблоны к именам, выводить только точные совпадения
|
||||||
-p|--pool POOL Фильтровать образы по пулу (ID или имени)
|
-p|--pool POOL Фильтровать образы по пулу (ID или имени)
|
||||||
-l|--long Также выводить статистику занятого места и ввода-вывода
|
-l|--long Также выводить статистику занятого места и ввода-вывода
|
||||||
--del Также выводить статистику операций удаления
|
--del Также выводить статистику операций удаления
|
||||||
--sort FIELD Сортировать по заданному полю (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)
|
--sort FIELD Сортировать по заданному полю (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)
|
||||||
-r|--reverse Сортировать в обратном порядке
|
-r|--reverse Сортировать в обратном порядке
|
||||||
-n|--count N Показывать только первые N записей
|
-n|--count N Показывать только первые N записей
|
||||||
|
--tree Вывести снапшоты и клоны в виде дерева
|
||||||
```
|
```
|
||||||
|
|
||||||
Пример вывода:
|
Пример вывода:
|
||||||
|
|||||||
@@ -51,6 +51,9 @@ Options (automatic mode):
|
|||||||
```
|
```
|
||||||
--osd_per_disk <N>
|
--osd_per_disk <N>
|
||||||
Create <N> OSDs on each disk (default 1)
|
Create <N> OSDs on each disk (default 1)
|
||||||
|
--meta_format 3
|
||||||
|
Metadata store version. 3 is the new log-structured store, 2 is the stable store
|
||||||
|
from Vitastor 0.9-2.x, 1 is the legacy store from Vitastor 0.6-0.8.
|
||||||
--hybrid
|
--hybrid
|
||||||
Prepare hybrid (HDD+SSD, NVMe+SATA or etc) OSDs using provided devices. By default,
|
Prepare hybrid (HDD+SSD, NVMe+SATA or etc) OSDs using provided devices. By default,
|
||||||
any passed SSDs will be used for journals and metadata, HDDs will be used for data,
|
any passed SSDs will be used for journals and metadata, HDDs will be used for data,
|
||||||
@@ -73,6 +76,8 @@ Options (automatic mode):
|
|||||||
--max_other 10%
|
--max_other 10%
|
||||||
Use disks for OSD data even if they already have non-Vitastor partitions,
|
Use disks for OSD data even if they already have non-Vitastor partitions,
|
||||||
but only if these take up no more than this percent of disk space.
|
but only if these take up no more than this percent of disk space.
|
||||||
|
--dry-run
|
||||||
|
Check and print new OSD count for each disk but do not actually create them.
|
||||||
```
|
```
|
||||||
|
|
||||||
Options (single-device mode):
|
Options (single-device mode):
|
||||||
@@ -90,6 +95,8 @@ Options (single-device mode):
|
|||||||
Options (both modes):
|
Options (both modes):
|
||||||
|
|
||||||
```
|
```
|
||||||
|
--tags tag1,tag2 Set new OSD tag(s)
|
||||||
|
--weight <number> Set new OSD weight (between 0 to 1)
|
||||||
--journal_size 1G/32M Set journal size (area or partition size)
|
--journal_size 1G/32M Set journal size (area or partition size)
|
||||||
--block_size 1M/128k Set blockstore object size
|
--block_size 1M/128k Set blockstore object size
|
||||||
--bitmap_granularity 4k Set bitmap granularity
|
--bitmap_granularity 4k Set bitmap granularity
|
||||||
|
|||||||
@@ -50,6 +50,9 @@ vitastor-disk - инструмент командной строки для уп
|
|||||||
```
|
```
|
||||||
--osd_per_disk <N>
|
--osd_per_disk <N>
|
||||||
Создавать по несколько (<N>) OSD на каждом диске (по умолчанию 1)
|
Создавать по несколько (<N>) OSD на каждом диске (по умолчанию 1)
|
||||||
|
--meta_format 3
|
||||||
|
Версия хранилища метаданных. 3 - новое лог-структурированное хранилище,
|
||||||
|
2 - стабильное хранилище из Vitastor 0.9-2.x, 1 - старое хранилище из Vitastor 0.6-0.8.
|
||||||
--hybrid
|
--hybrid
|
||||||
Инициализировать гибридные (HDD+SSD, NVMe+SATA и т.п.) OSD на указанных дисках.
|
Инициализировать гибридные (HDD+SSD, NVMe+SATA и т.п.) OSD на указанных дисках.
|
||||||
По умолчанию, SSD будут использованы для журналов и метаданных, а HDD - для данных,
|
По умолчанию, SSD будут использованы для журналов и метаданных, а HDD - для данных,
|
||||||
@@ -74,6 +77,8 @@ vitastor-disk - инструмент командной строки для уп
|
|||||||
--max_other 10%
|
--max_other 10%
|
||||||
Использовать диски под данные OSD, даже если на них уже есть не-Vitastor-овые
|
Использовать диски под данные OSD, даже если на них уже есть не-Vitastor-овые
|
||||||
разделы, но только в случае, если они занимают не более данного процента диска.
|
разделы, но только в случае, если они занимают не более данного процента диска.
|
||||||
|
--dry-run
|
||||||
|
Проверить и вывести число новых OSD для каждого диска, но не создавать их.
|
||||||
```
|
```
|
||||||
|
|
||||||
Опции для режима одного OSD:
|
Опции для режима одного OSD:
|
||||||
@@ -91,6 +96,8 @@ vitastor-disk - инструмент командной строки для уп
|
|||||||
Опции для обоих режимов:
|
Опции для обоих режимов:
|
||||||
|
|
||||||
```
|
```
|
||||||
|
--tags tag1,tag2 Задать теги для новых OSD
|
||||||
|
--weight <number> Задать вес для новых OSD (от 0 до 1)
|
||||||
--journal_size 1G/32M Задать размер журнала (области или раздела журнала)
|
--journal_size 1G/32M Задать размер журнала (области или раздела журнала)
|
||||||
--block_size 1M/128k Задать размер объекта хранилища
|
--block_size 1M/128k Задать размер объекта хранилища
|
||||||
--bitmap_granularity 4k Задать гранулярность битовых карт
|
--bitmap_granularity 4k Задать гранулярность битовых карт
|
||||||
|
|||||||
@@ -89,6 +89,8 @@ POSIX features currently not implemented in VitastorFS:
|
|||||||
instead of actually allocated space
|
instead of actually allocated space
|
||||||
- Access times (`atime`) are not tracked (like `-o noatime`)
|
- Access times (`atime`) are not tracked (like `-o noatime`)
|
||||||
- Modification time (`mtime`) is updated lazily every second (like `-o lazytime`)
|
- Modification time (`mtime`) is updated lazily every second (like `-o lazytime`)
|
||||||
|
- Permission enforcement is disabled by default (and Linux NFS client doesn't
|
||||||
|
enforce them too). Use `--enforce 1` to enable it.
|
||||||
|
|
||||||
Other notable missing features which should be addressed in the future:
|
Other notable missing features which should be addressed in the future:
|
||||||
- Inode ID reuse. Currently inode IDs always grow, the limit is 2^48 inodes, so
|
- Inode ID reuse. Currently inode IDs always grow, the limit is 2^48 inodes, so
|
||||||
@@ -258,4 +260,5 @@ Options:
|
|||||||
| `--nfspath <PATH>` | set NFS export path to \<PATH> (default is /) |
|
| `--nfspath <PATH>` | set NFS export path to \<PATH> (default is /) |
|
||||||
| `--pidfile <FILE>` | write process ID to the specified file |
|
| `--pidfile <FILE>` | write process ID to the specified file |
|
||||||
| `--logfile <FILE>` | log to the specified file |
|
| `--logfile <FILE>` | log to the specified file |
|
||||||
|
| `--enforce 1` | enforce permissions at the server side (no by default) |
|
||||||
| `--foreground 1` | stay in foreground, do not daemonize |
|
| `--foreground 1` | stay in foreground, do not daemonize |
|
||||||
|
|||||||
@@ -91,6 +91,8 @@ JSON-формате :-). Для инспекции содержимого БД
|
|||||||
stat(2), так что `du` всегда показывает сумму размеров файлов, а не фактически занятое место
|
stat(2), так что `du` всегда показывает сумму размеров файлов, а не фактически занятое место
|
||||||
- Времена доступа (`atime`) не отслеживаются (как будто ФС смонтирована с `-o noatime`)
|
- Времена доступа (`atime`) не отслеживаются (как будто ФС смонтирована с `-o noatime`)
|
||||||
- Времена модификации (`mtime`) отслеживаются асинхронно (как будто ФС смонтирована с `-o lazytime`)
|
- Времена модификации (`mtime`) отслеживаются асинхронно (как будто ФС смонтирована с `-o lazytime`)
|
||||||
|
- Привилегии доступа по умолчанию не проверяются сервером (клиент NFS Linux их также не проверяет).
|
||||||
|
Чтобы включить проверки, используйте опцию `--enforce 1`.
|
||||||
|
|
||||||
Другие недостающие функции, которые нужно добавить в будущем:
|
Другие недостающие функции, которые нужно добавить в будущем:
|
||||||
- Переиспользование номеров инодов. В текущей реализации номера инодов всё время
|
- Переиспользование номеров инодов. В текущей реализации номера инодов всё время
|
||||||
@@ -270,4 +272,5 @@ VitastorFS из GPUDirect.
|
|||||||
| `--nfspath <PATH>` | установить путь NFS-экспорта в \<PATH> (по умолчанию /) |
|
| `--nfspath <PATH>` | установить путь NFS-экспорта в \<PATH> (по умолчанию /) |
|
||||||
| `--pidfile <FILE>` | записать ID процесса в заданный файл |
|
| `--pidfile <FILE>` | записать ID процесса в заданный файл |
|
||||||
| `--logfile <FILE>` | записывать логи в заданный файл |
|
| `--logfile <FILE>` | записывать логи в заданный файл |
|
||||||
|
| `--enforce 1` | проверять права доступа на стороне сервера (по умолчанию нет) |
|
||||||
| `--foreground 1` | не уходить в фон после запуска |
|
| `--foreground 1` | не уходить в фон после запуска |
|
||||||
|
|||||||
+17
-15
@@ -130,23 +130,16 @@ Linux kernel, starting with version 5.15, supports a new interface for attaching
|
|||||||
to the host - VDUSE (vDPA Device in Userspace). QEMU, starting with 7.2, has support for
|
to the host - VDUSE (vDPA Device in Userspace). QEMU, starting with 7.2, has support for
|
||||||
exporting QEMU block devices over this protocol using qemu-storage-daemon.
|
exporting QEMU block devices over this protocol using qemu-storage-daemon.
|
||||||
|
|
||||||
VDUSE is currently the best interface to attach Vitastor disks as kernel devices because:
|
VDUSE advantages:
|
||||||
- It avoids data copies and thus achieves much better performance than [NBD](nbd.en.md)
|
|
||||||
- It doesn't have NBD timeout problem - the device doesn't die if an operation executes for too long
|
- VDUSE copies memory 1 time instead of 2, and is thus faster than [NBD](nbd.en.md) for linear read/write.
|
||||||
|
- It doesn't have NBD timeout problem - the device doesn't die if an operation executes for too long.
|
||||||
- It doesn't have hung device problem - if the userspace process dies it can be restarted (!)
|
- It doesn't have hung device problem - if the userspace process dies it can be restarted (!)
|
||||||
and block device will continue operation
|
and block device will continue operation (UBLK can do it too).
|
||||||
- It doesn't seem to have the device number limit
|
- It doesn't seem to have the device number limit (UBLK also doesn't).
|
||||||
|
|
||||||
Example performance comparison:
|
At the same time, VDUSE may be slower or faster than [UBLK](ublk.en.md) for linear read/write,
|
||||||
|
and iops-wise it's sometimes even slower than NBD. See performance comparison examples at the page [UBLK](ublk.en.md).
|
||||||
| | direct fio | NBD | VDUSE |
|
|
||||||
|----------------------|-------------|-------------|-------------|
|
|
||||||
| linear write | 3.85 GB/s | 1.12 GB/s | 3.85 GB/s |
|
|
||||||
| 4k random write Q128 | 240000 iops | 120000 iops | 178000 iops |
|
|
||||||
| 4k random write Q1 | 9500 iops | 7620 iops | 7640 iops |
|
|
||||||
| linear read | 4.3 GB/s | 1.8 GB/s | 2.85 GB/s |
|
|
||||||
| 4k random read Q128 | 287000 iops | 140000 iops | 189000 iops |
|
|
||||||
| 4k random read Q1 | 9600 iops | 7640 iops | 7780 iops |
|
|
||||||
|
|
||||||
To try VDUSE you need at least Linux 5.15, built with VDUSE support
|
To try VDUSE you need at least Linux 5.15, built with VDUSE support
|
||||||
(CONFIG_VDPA=m, CONFIG_VDPA_USER=m, CONFIG_VIRTIO_VDPA=m).
|
(CONFIG_VDPA=m, CONFIG_VDPA_USER=m, CONFIG_VIRTIO_VDPA=m).
|
||||||
@@ -193,3 +186,12 @@ To remove the device:
|
|||||||
vdpa dev del test1
|
vdpa dev del test1
|
||||||
kill <qemu-storage-daemon_process_PID>
|
kill <qemu-storage-daemon_process_PID>
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## Veeam
|
||||||
|
|
||||||
|
Vitastor QEMU driver has a feature that allows to trick third-party systems like Veeam not able to parse qemu-img
|
||||||
|
vitastor URIs: [qemu_file_mirror_path](../config/client.en.md#qemu_file_mirror_path).
|
||||||
|
|
||||||
|
To make such systems work, you should set this option to an FS directory path (for example, `/mnt/vitastor/`) and
|
||||||
|
mount this directory using [`vitastor-nfs mount --block`](../usage/nfs.en.md). It will make them access
|
||||||
|
your images using files and, hopefully, succeed in doing their normal job :).
|
||||||
|
|||||||
+17
-16
@@ -132,24 +132,16 @@ qemu-system-x86_64 -enable-kvm -m 2048 -M accel=kvm,memory-backend=mem \
|
|||||||
к системе - VDUSE (vDPA Device in Userspace), а в QEMU, начиная с версии 7.2, есть поддержка
|
к системе - VDUSE (vDPA Device in Userspace), а в QEMU, начиная с версии 7.2, есть поддержка
|
||||||
экспорта блочных устройств QEMU по этому протоколу через qemu-storage-daemon.
|
экспорта блочных устройств QEMU по этому протоколу через qemu-storage-daemon.
|
||||||
|
|
||||||
VDUSE - на данный момент лучший интерфейс для подключения дисков Vitastor в виде блочных
|
Преимущества VDUSE:
|
||||||
устройств на уровне ядра, ибо:
|
|
||||||
- VDUSE не копирует данные и поэтому достигает значительно лучшей производительности, чем [NBD](nbd.ru.md)
|
|
||||||
- Также оно не имеет проблемы NBD-таймаута - устройство не умирает, если операция выполняется слишком долго
|
|
||||||
- Также оно не имеет проблемы подвисающих устройств - если процесс-обработчик умирает, его можно
|
|
||||||
перезапустить (!) и блочное устройство продолжит работать
|
|
||||||
- По-видимому, у него нет предела числа подключаемых в систему устройств
|
|
||||||
|
|
||||||
Пример сравнения производительности:
|
- VDUSE копирует данные 1 раз, а не 2, и поэтому он быстрее, чем [NBD](nbd.ru.md) при линейном доступе.
|
||||||
|
- VDUSE не имеет проблемы NBD-таймаута - устройство не умирает, если операция выполняется слишком долго.
|
||||||
|
- VDUSE не имеет проблемы подвисающих устройств - если процесс-обработчик умирает, его можно
|
||||||
|
перезапустить (!) и блочное устройство продолжит работать (в UBLK это тоже поддерживается).
|
||||||
|
- По-видимому, у него нет предела числа подключаемых в систему устройств (в UBLK лимита тоже нет).
|
||||||
|
|
||||||
| | Прямой fio | NBD | VDUSE |
|
Однако, при линейном доступе VDUSE может быть медленнее UBLK (а может быть и быстрее), а по iops
|
||||||
|--------------------------|-------------|-------------|-------------|
|
VDUSE иногда даже медленнее NBD. Пример сравнения производительности смотрите на странице [UBLK](ublk.ru.md).
|
||||||
| линейная запись | 3.85 GB/s | 1.12 GB/s | 3.85 GB/s |
|
|
||||||
| 4k случайная запись Q128 | 240000 iops | 120000 iops | 178000 iops |
|
|
||||||
| 4k случайная запись Q1 | 9500 iops | 7620 iops | 7640 iops |
|
|
||||||
| линейное чтение | 4.3 GB/s | 1.8 GB/s | 2.85 GB/s |
|
|
||||||
| 4k случайное чтение Q128 | 287000 iops | 140000 iops | 189000 iops |
|
|
||||||
| 4k случайное чтение Q1 | 9600 iops | 7640 iops | 7780 iops |
|
|
||||||
|
|
||||||
Чтобы попробовать VDUSE, вам нужно ядро Linux как минимум версии 5.15, собранное с поддержкой
|
Чтобы попробовать VDUSE, вам нужно ядро Linux как минимум версии 5.15, собранное с поддержкой
|
||||||
VDUSE (CONFIG_VDPA=m, CONFIG_VDPA_USER=m, CONFIG_VIRTIO_VDPA=m).
|
VDUSE (CONFIG_VDPA=m, CONFIG_VDPA_USER=m, CONFIG_VIRTIO_VDPA=m).
|
||||||
@@ -196,3 +188,12 @@ vdpa dev add name test1 mgmtdev vduse
|
|||||||
vdpa dev del test1
|
vdpa dev del test1
|
||||||
kill <PID_процесса_qemu-storage-daemon>
|
kill <PID_процесса_qemu-storage-daemon>
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## Veeam
|
||||||
|
|
||||||
|
Драйвер Vitastor QEMU имеет функцию, которая позволяет обманывать сторонние системы типа Veeam, которые
|
||||||
|
не могут сами по себе разобрать адреса дисков в vitastor: [qemu_file_mirror_path](../config/client.ru.md#qemu_file_mirror_path).
|
||||||
|
|
||||||
|
Чтобы заставить такие системы работать, вам нужно установить эту опцию равной пути к некоторому каталогу
|
||||||
|
в ФС (например, `/mnt/vitastor/`) и примонтировать этот каталог с помощью [`vitastor-nfs mount --block`](../usage/nfs.ru.md).
|
||||||
|
Они начнут обращаться к образам как к файлам и, вероятно, смогут заработать корректно :).
|
||||||
|
|||||||
@@ -0,0 +1,116 @@
|
|||||||
|
[Documentation](../../README.md#documentation) → Usage → UBLK
|
||||||
|
|
||||||
|
-----
|
||||||
|
|
||||||
|
[Читать на русском](ublk.ru.md)
|
||||||
|
|
||||||
|
# UBLK
|
||||||
|
|
||||||
|
[ublk](https://docs.kernel.org/block/ublk.html) is a new io_uring-based Linux interface
|
||||||
|
for user-space block device drivers, available since Linux 6.0.
|
||||||
|
|
||||||
|
It's not zero-copy, but it's still a fast implementation, outperforming both [NBD](nbd.en.md)
|
||||||
|
and [VDUSE](qemu.en.md#vduse) iops-wise and may or may not outperform VDUSE in linear I/O MB/s.
|
||||||
|
ublk also allows to recover devices even if the server (vitastor-ublk process) dies.
|
||||||
|
|
||||||
|
## Example performance comparison
|
||||||
|
|
||||||
|
TCP (100G), 3 hosts each with 6 NVMe OSDs, 3 replicas, single client
|
||||||
|
|
||||||
|
| | direct fio | NBD | VDUSE | UBLK |
|
||||||
|
|----------------------|-------------|-------------|------------|-------------|
|
||||||
|
| linear write | 3807 MB/s | 1832 MB/s | 3226 MB/s | 3027 MB/s |
|
||||||
|
| linear read | 3067 MB/s | 1885 MB/s | 1800 MB/s | 2076 MB/s |
|
||||||
|
| 4k random write Q128 | 128624 iops | 91060 iops | 94621 iops | 149450 iops |
|
||||||
|
| 4k random read Q128 | 117769 iops | 153408 iops | 93157 iops | 171987 iops |
|
||||||
|
| 4k random write Q1 | 8090 iops | 6442 iops | 6316 iops | 7272 iops |
|
||||||
|
| 4k random read Q1 | 9474 iops | 7200 iops | 6840 iops | 8038 iops |
|
||||||
|
|
||||||
|
RDMA (100G), 3 hosts each with 6 NVMe OSDs, 3 replicas, single client
|
||||||
|
|
||||||
|
| | direct fio | NBD | VDUSE | UBLK |
|
||||||
|
|----------------------|-------------|-------------|-------------|-------------|
|
||||||
|
| linear write | 6998 MB/s | 1878 MB/s | 4249 MB/s | 3140 MB/s |
|
||||||
|
| linear read | 8628 MB/s | 3389 MB/s | 5062 MB/s | 3674 MB/s |
|
||||||
|
| 4k random write Q128 | 222541 iops | 181589 iops | 138281 iops | 218222 iops |
|
||||||
|
| 4k random read Q128 | 412647 iops | 239987 iops | 151663 iops | 269583 iops |
|
||||||
|
| 4k random write Q1 | 11601 iops | 8592 iops | 9111 iops | 10000 iops |
|
||||||
|
| 4k random read Q1 | 10102 iops | 7788 iops | 8111 iops | 8965 iops |
|
||||||
|
|
||||||
|
## Commands
|
||||||
|
|
||||||
|
vitastor-ublk supports the following commands:
|
||||||
|
|
||||||
|
- [map](#map)
|
||||||
|
- [unmap](#unmap)
|
||||||
|
- [ls](#ls)
|
||||||
|
|
||||||
|
## map
|
||||||
|
|
||||||
|
To create a local block device for a Vitastor image run:
|
||||||
|
|
||||||
|
```
|
||||||
|
vitastor-ublk map [/dev/ublkbN] --image testimg
|
||||||
|
```
|
||||||
|
|
||||||
|
It will output a block device name like /dev/ublkb0 which you can then use as a normal disk.
|
||||||
|
|
||||||
|
You can also use `--pool <POOL> --inode <INODE> --size <SIZE>` instead of `--image <IMAGE>` if you want.
|
||||||
|
|
||||||
|
vitastor-ublk supports all usual Vitastor configuration options like `--config_path <path_to_config>` plus ublk-specific:
|
||||||
|
|
||||||
|
* `--recover` \
|
||||||
|
Recover a mapped device if the previous ublk server is dead.
|
||||||
|
* `--queue_depth 256` \
|
||||||
|
Maximum queue size for the device.
|
||||||
|
* `--max_io_size 1M` \
|
||||||
|
Maximum single I/O size for the device. Default: `max(1 MB, pool block size * EC part count)`.
|
||||||
|
* `--readonly` \
|
||||||
|
Make the device read-only.
|
||||||
|
* `--hdd` \
|
||||||
|
Mark the device as rotational.
|
||||||
|
* `--logfile /path/to/log/file.txt` \
|
||||||
|
Write log messages to the specified file instead of dropping them (in background mode)
|
||||||
|
or printing them to the standard output (in foreground mode).
|
||||||
|
* `--dev_num N` \
|
||||||
|
Use the specified device /dev/ublkbN instead of automatic selection (alternative syntax
|
||||||
|
to /dev/ublkbN positional parameter).
|
||||||
|
* `--foreground 1` \
|
||||||
|
Stay in foreground, do not daemonize.
|
||||||
|
|
||||||
|
Note that `ublk_queue_depth` and `ublk_max_io_size` may also be specified
|
||||||
|
in `/etc/vitastor/vitastor.conf` or in other configuration file specified with `--config_path`.
|
||||||
|
|
||||||
|
## unmap
|
||||||
|
|
||||||
|
To unmap the device run:
|
||||||
|
|
||||||
|
```
|
||||||
|
vitastor-ublk unmap /dev/ublkb0
|
||||||
|
```
|
||||||
|
|
||||||
|
## ls
|
||||||
|
|
||||||
|
```
|
||||||
|
vitastor-ublk ls [--json]
|
||||||
|
```
|
||||||
|
|
||||||
|
List mapped images.
|
||||||
|
|
||||||
|
Example output (normal format):
|
||||||
|
|
||||||
|
```
|
||||||
|
/dev/ublkb0
|
||||||
|
image: bench
|
||||||
|
pid: 584536
|
||||||
|
|
||||||
|
/dev/ublkb1
|
||||||
|
image: bench1
|
||||||
|
pid: 584546
|
||||||
|
```
|
||||||
|
|
||||||
|
Example output (JSON format):
|
||||||
|
|
||||||
|
```
|
||||||
|
{"/dev/ublkb0": {"image": "bench", "pid": 584536}, "/dev/ublkb1": {"image": "bench1", "pid": 584546}}
|
||||||
|
```
|
||||||
@@ -0,0 +1,121 @@
|
|||||||
|
[Документация](../../README-ru.md#документация) → Использование → UBLK
|
||||||
|
|
||||||
|
-----
|
||||||
|
|
||||||
|
[Read in English](ublk.en.md)
|
||||||
|
|
||||||
|
# UBLK
|
||||||
|
|
||||||
|
[ublk](https://docs.kernel.org/block/ublk.html) - это новый Linux-интерфейс на основе io_uring
|
||||||
|
для реализации блочных устройств в пространстве пользователя, доступный, начиная с Linux 6.0.
|
||||||
|
|
||||||
|
ublk тоже копирует память (т.е. не является zero-copy), но по IOPS всё равно обгоняет и
|
||||||
|
[NBD](nbd.ru.md), и [VDUSE](qemu.ru.md#vduse), и иногда может даже обгонять VDUSE по
|
||||||
|
скорости линейного доступа. Также ublk позволяет оживлять устройства, у которых умер
|
||||||
|
сервер (процесс-обработчик vitastor-ublk).
|
||||||
|
|
||||||
|
## Пример сравнения производительности
|
||||||
|
|
||||||
|
TCP (100G), 3 сервера с 6 NVMe OSD каждый, 3 реплики, один клиент
|
||||||
|
|
||||||
|
| | Прямой fio | NBD | VDUSE | UBLK |
|
||||||
|
|--------------------------|-------------|-------------|------------|-------------|
|
||||||
|
| линейная запись | 3807 MB/s | 1832 MB/s | 3226 MB/s | 3027 MB/s |
|
||||||
|
| линейное чтение | 3067 MB/s | 1885 MB/s | 1800 MB/s | 2076 MB/s |
|
||||||
|
| 4k случайная запись Q128 | 128624 iops | 91060 iops | 94621 iops | 149450 iops |
|
||||||
|
| 4k случайное чтение Q128 | 117769 iops | 153408 iops | 93157 iops | 171987 iops |
|
||||||
|
| 4k случайная запись Q1 | 8090 iops | 6442 iops | 6316 iops | 7272 iops |
|
||||||
|
| 4k случайное чтение Q1 | 9474 iops | 7200 iops | 6840 iops | 8038 iops |
|
||||||
|
|
||||||
|
RDMA (100G), 3 сервера с 6 NVMe OSD каждый, 3 реплики, один клиент
|
||||||
|
|
||||||
|
| | Прямой fio | NBD | VDUSE | UBLK |
|
||||||
|
|--------------------------|-------------|-------------|-------------|-------------|
|
||||||
|
| линейная запись | 6998 MB/s | 1878 MB/s | 4249 MB/s | 3140 MB/s |
|
||||||
|
| линейное чтение | 8628 MB/s | 3389 MB/s | 5062 MB/s | 3674 MB/s |
|
||||||
|
| 4k случайная запись Q128 | 222541 iops | 181589 iops | 138281 iops | 218222 iops |
|
||||||
|
| 4k случайное чтение Q128 | 412647 iops | 239987 iops | 151663 iops | 269583 iops |
|
||||||
|
| 4k случайная запись Q1 | 11601 iops | 8592 iops | 9111 iops | 10000 iops |
|
||||||
|
| 4k случайное чтение Q1 | 10102 iops | 7788 iops | 8111 iops | 8965 iops |
|
||||||
|
|
||||||
|
## Команды
|
||||||
|
|
||||||
|
vitastor-ublk поддерживает следующие команды:
|
||||||
|
|
||||||
|
- [map](#map)
|
||||||
|
- [unmap](#unmap)
|
||||||
|
- [ls](#ls)
|
||||||
|
|
||||||
|
## map
|
||||||
|
|
||||||
|
Чтобы создать локальное блочное устройство для образа, выполните команду:
|
||||||
|
|
||||||
|
```
|
||||||
|
vitastor-ublk map [/dev/ublkbN] --image testimg
|
||||||
|
```
|
||||||
|
|
||||||
|
Команда напечатает название блочного устройства вида /dev/ublkb0, которое потом можно
|
||||||
|
будет использовать как обычный диск.
|
||||||
|
|
||||||
|
Для обращения по номеру инода, аналогично другим командам, можно использовать опции
|
||||||
|
`--pool <POOL> --inode <INODE> --size <SIZE>` вместо `--image testimg`.
|
||||||
|
|
||||||
|
vitastor-ublk поддерживает все обычные опции Vitastor, например, `--config_path <path_to_config>`,
|
||||||
|
плюс специфичные для ublk:
|
||||||
|
|
||||||
|
* `--recover` \
|
||||||
|
Восстановить ранее подключённое устройство, у которого умер обработчик.
|
||||||
|
* `--queue_depth 256` \
|
||||||
|
Максимальная глубина очереди устройства.
|
||||||
|
* `--max_io_size 1M` \
|
||||||
|
Максимальный размер запроса ввода-вывода для устройства. По умолчанию: `max(1 MB, блок данных пула * число частей данных EC)`.
|
||||||
|
* `--readonly` \
|
||||||
|
Подключить устройство в режиме только для чтения.
|
||||||
|
* `--hdd` \
|
||||||
|
Пометить устройство как вращающийся жёсткий диск (флаг rotational).
|
||||||
|
* `--logfile /path/to/log/file.txt` \
|
||||||
|
Писать сообщения о процессе работы в заданный файл, вместо пропуска их
|
||||||
|
при фоновом режиме запуска или печати на стандартный вывод при запуске
|
||||||
|
в консоли с `--foreground 1`.
|
||||||
|
* `--dev_num N` \
|
||||||
|
Использовать заданное устройство `/dev/ublkbN` вместо автоматического подбора.
|
||||||
|
* `--foreground 1` \
|
||||||
|
Не уводить процесс в фоновый режим.
|
||||||
|
|
||||||
|
Обратите внимание, что опции `ublk_queue_depth` и `ublk_max_io_size` можно
|
||||||
|
также задавать в `/etc/vitastor/vitastor.conf` или в другом файле конфигурации,
|
||||||
|
заданном опцией `--config_path`.
|
||||||
|
|
||||||
|
## unmap
|
||||||
|
|
||||||
|
Для отключения устройства выполните:
|
||||||
|
|
||||||
|
```
|
||||||
|
vitastor-ublk unmap /dev/ublkb0
|
||||||
|
```
|
||||||
|
|
||||||
|
## ls
|
||||||
|
|
||||||
|
```
|
||||||
|
vitastor-ublk ls [--json]
|
||||||
|
```
|
||||||
|
|
||||||
|
Вывести подключённые устройства.
|
||||||
|
|
||||||
|
Пример вывода в обычном формате:
|
||||||
|
|
||||||
|
```
|
||||||
|
/dev/ublkb0
|
||||||
|
image: bench
|
||||||
|
pid: 584536
|
||||||
|
|
||||||
|
/dev/ublkb1
|
||||||
|
image: bench1
|
||||||
|
pid: 584546
|
||||||
|
```
|
||||||
|
|
||||||
|
Пример вывода в JSON-формате:
|
||||||
|
|
||||||
|
```
|
||||||
|
{"/dev/ublkb0": {"image": "bench", "pid": 584536}, "/dev/ublkb1": {"image": "bench1", "pid": 584546}}
|
||||||
|
```
|
||||||
+3
-3
@@ -15,7 +15,7 @@ function get_osd_tree(global_config, state)
|
|||||||
const stat = state.osd.stats[osd_num];
|
const stat = state.osd.stats[osd_num];
|
||||||
const osd_cfg = state.config.osd[osd_num];
|
const osd_cfg = state.config.osd[osd_num];
|
||||||
let reweight = osd_cfg == null ? 1 : Number(osd_cfg.reweight);
|
let reweight = osd_cfg == null ? 1 : Number(osd_cfg.reweight);
|
||||||
if (isNaN(reweight) || reweight < 0 || reweight > 0)
|
if (isNaN(reweight) || reweight < 0 || reweight > 1)
|
||||||
reweight = 1;
|
reweight = 1;
|
||||||
if (stat && stat.size && reweight && (state.osd.state[osd_num] || Number(stat.time) >= down_time ||
|
if (stat && stat.size && reweight && (state.osd.state[osd_num] || Number(stat.time) >= down_time ||
|
||||||
osd_cfg && osd_cfg.noout))
|
osd_cfg && osd_cfg.noout))
|
||||||
@@ -87,7 +87,7 @@ function make_hier_tree(global_config, tree)
|
|||||||
tree[''] = { children: [] };
|
tree[''] = { children: [] };
|
||||||
for (const node_id in tree)
|
for (const node_id in tree)
|
||||||
{
|
{
|
||||||
if (node_id === '' || !(tree[node_id].children||[]).length && (tree[node_id].size||0) <= 0)
|
if (node_id === '')
|
||||||
{
|
{
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -179,7 +179,7 @@ function filter_osds_by_block_layout(orig_tree, osd_stats, block_size, bitmap_gr
|
|||||||
if (orig_tree[osd].level === 'osd')
|
if (orig_tree[osd].level === 'osd')
|
||||||
{
|
{
|
||||||
const osd_stat = osd_stats[osd];
|
const osd_stat = osd_stats[osd];
|
||||||
if (osd_stat && (osd_stat.bs_block_size && osd_stat.bs_block_size != block_size ||
|
if (osd_stat && (osd_stat.data_block_size && osd_stat.data_block_size != block_size ||
|
||||||
osd_stat.bitmap_granularity && osd_stat.bitmap_granularity != bitmap_granularity ||
|
osd_stat.bitmap_granularity && osd_stat.bitmap_granularity != bitmap_granularity ||
|
||||||
osd_stat.immediate_commit == 'small' && immediate_commit == 'all' ||
|
osd_stat.immediate_commit == 'small' && immediate_commit == 'all' ||
|
||||||
osd_stat.immediate_commit == 'none' && immediate_commit != 'none'))
|
osd_stat.immediate_commit == 'none' && immediate_commit != 'none'))
|
||||||
|
|||||||
+2
-2
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "vitastor-mon",
|
"name": "vitastor-mon",
|
||||||
"version": "2.2.3",
|
"version": "3.0.2",
|
||||||
"description": "Vitastor SDS monitor service",
|
"description": "Vitastor SDS monitor service",
|
||||||
"main": "mon-main.js",
|
"main": "mon-main.js",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
@@ -9,7 +9,7 @@
|
|||||||
"author": "Vitaliy Filippov",
|
"author": "Vitaliy Filippov",
|
||||||
"license": "UNLICENSED",
|
"license": "UNLICENSED",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"antietcd": "^1.1.2",
|
"antietcd": "^1.2.2",
|
||||||
"sprintf-js": "^1.1.2",
|
"sprintf-js": "^1.1.2",
|
||||||
"ws": "^7.2.5"
|
"ws": "^7.2.5"
|
||||||
},
|
},
|
||||||
|
|||||||
+16
-3
@@ -9,7 +9,6 @@ const LPOptimizer = require('./lp_optimizer/lp_optimizer.js');
|
|||||||
const { scale_pg_count } = require('./pg_utils.js');
|
const { scale_pg_count } = require('./pg_utils.js');
|
||||||
const { make_hier_tree, filter_osds_by_root_node,
|
const { make_hier_tree, filter_osds_by_root_node,
|
||||||
filter_osds_by_tags, filter_osds_by_block_layout, get_affinity_osds } = require('./osd_tree.js');
|
filter_osds_by_tags, filter_osds_by_block_layout, get_affinity_osds } = require('./osd_tree.js');
|
||||||
const { select_murmur3 } = require('./lp_optimizer/murmur3.js');
|
|
||||||
|
|
||||||
function pick_primary(pool_id, pg_num, pool_config, osd_set, up_osds, aff_osds)
|
function pick_primary(pool_id, pg_num, pool_config, osd_set, up_osds, aff_osds)
|
||||||
{
|
{
|
||||||
@@ -39,7 +38,7 @@ function pick_primary(pool_id, pg_num, pool_config, osd_set, up_osds, aff_osds)
|
|||||||
{
|
{
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
return alive_set[select_murmur3(alive_set.length, osd_num => pool_id+'/'+pg_num+'/'+osd_num)];
|
return alive_set[pg_num % alive_set.length];
|
||||||
}
|
}
|
||||||
|
|
||||||
function recheck_primary(state, global_config, up_osds, osd_tree)
|
function recheck_primary(state, global_config, up_osds, osd_tree)
|
||||||
@@ -53,6 +52,7 @@ function recheck_primary(state, global_config, up_osds, osd_tree)
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
const aff_osds = get_affinity_osds(pool_cfg, up_osds, osd_tree);
|
const aff_osds = get_affinity_osds(pool_cfg, up_osds, osd_tree);
|
||||||
|
let paused = false;
|
||||||
for (let pg_num = 1; pg_num <= pool_cfg.pg_count; pg_num++)
|
for (let pg_num = 1; pg_num <= pool_cfg.pg_count; pg_num++)
|
||||||
{
|
{
|
||||||
if (!state.pg.config.items[pool_id])
|
if (!state.pg.config.items[pool_id])
|
||||||
@@ -75,6 +75,19 @@ function recheck_primary(state, global_config, up_osds, osd_tree)
|
|||||||
);
|
);
|
||||||
new_pg_config.items[pool_id][pg_num].primary = new_primary;
|
new_pg_config.items[pool_id][pg_num].primary = new_primary;
|
||||||
}
|
}
|
||||||
|
paused = paused || !!pg_cfg.pause;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (paused)
|
||||||
|
{
|
||||||
|
if (!new_pg_config)
|
||||||
|
{
|
||||||
|
new_pg_config = JSON.parse(JSON.stringify(state.pg.config));
|
||||||
|
}
|
||||||
|
console.log(`Resuming paused pool ${pool_id}`);
|
||||||
|
for (const pg in new_pg_config.items[pool_id])
|
||||||
|
{
|
||||||
|
delete new_pg_config.items[pool_id][pg].pause;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -179,7 +192,7 @@ async function generate_pool_pgs(state, global_config, pool_id, osd_tree, levels
|
|||||||
const rules = use_rules ? get_pg_rules(pool_id, pool_cfg, global_config.placement_levels) : null;
|
const rules = use_rules ? get_pg_rules(pool_id, pool_cfg, global_config.placement_levels) : null;
|
||||||
const folded = fold_failure_domains(Object.values(pool_tree), use_rules ? rules : [ [ [ pool_cfg.failure_domain ] ] ]);
|
const folded = fold_failure_domains(Object.values(pool_tree), use_rules ? rules : [ [ [ pool_cfg.failure_domain ] ] ]);
|
||||||
// FIXME: Remove/merge make_hier_tree() step somewhere, however it's needed to remove empty nodes
|
// FIXME: Remove/merge make_hier_tree() step somewhere, however it's needed to remove empty nodes
|
||||||
const folded_tree = make_hier_tree(global_config, folded.nodes);
|
const folded_tree = make_hier_tree(global_config, folded.nodes.reduce((a, c) => { a[c.id] = c; return a; }, {}));
|
||||||
const old_pg_count = prev_pgs.length;
|
const old_pg_count = prev_pgs.length;
|
||||||
const optimize_cfg = {
|
const optimize_cfg = {
|
||||||
osd_weights: folded.nodes.reduce((a, c) => { if (Number(c.id)) { a[c.id] = c.size; } return a; }, {}),
|
osd_weights: folded.nodes.reduce((a, c) => { if (Number(c.id)) { a[c.id] = c.size; } return a; }, {}),
|
||||||
|
|||||||
@@ -52,15 +52,16 @@ async function run()
|
|||||||
process.exit(1);
|
process.exit(1);
|
||||||
}
|
}
|
||||||
const etcds = (config.etcd_address instanceof Array ? config.etcd_address : (''+config.etcd_address).split(/,/))
|
const etcds = (config.etcd_address instanceof Array ? config.etcd_address : (''+config.etcd_address).split(/,/))
|
||||||
.map(s => (''+s).replace(/^https?:\/\/\[?|\]?(:\d+)?(\/.*)?$/g, '').toLowerCase());
|
.map(s => (''+s).replace(/^https?:\/\/|(:\d+)?(\/.*)?$/g, '').replace(/^\[(.*)\]$/, '$1').toLowerCase());
|
||||||
const num = select_local_etcd(etcds);
|
const num = select_local_etcd(etcds);
|
||||||
if (num < 0)
|
if (num < 0)
|
||||||
{
|
{
|
||||||
console.log('No matching IPs in etcd_address from '+config_path);
|
console.log('No matching IPs in etcd_address from '+config_path);
|
||||||
process.exit(0);
|
process.exit(0);
|
||||||
}
|
}
|
||||||
|
const etcd_url = 'http://' + (etcds[num].indexOf(':') >= 0 ? '['+etcds[num]+']' : etcds[num]);
|
||||||
const etcd_name = 'etcd'+etcds[num].replace(/[^0-9a-z_]/ig, '_');
|
const etcd_name = 'etcd'+etcds[num].replace(/[^0-9a-z_]/ig, '_');
|
||||||
const etcd_cluster = etcds.map(e => `etcd${e.replace(/[^0-9a-z_]/ig, '_')}=http://${e}:2380`).join(',');
|
const etcd_cluster = etcds.map(e => `etcd${e.replace(/[^0-9a-z_]/ig, '_')}=http://${e.indexOf(':') >= 0 ? '['+e+']' : e}:2380`).join(',');
|
||||||
if (in_docker)
|
if (in_docker)
|
||||||
{
|
{
|
||||||
let etcd_conf = fs.readFileSync("/etc/vitastor/etcd.conf", { encoding: 'utf-8' });
|
let etcd_conf = fs.readFileSync("/etc/vitastor/etcd.conf", { encoding: 'utf-8' });
|
||||||
@@ -83,8 +84,8 @@ Wants=network-online.target local-fs.target time-sync.target
|
|||||||
Restart=always
|
Restart=always
|
||||||
Environment=GOGC=50
|
Environment=GOGC=50
|
||||||
ExecStart=etcd --name ${etcd_name} --data-dir /var/lib/etcd/vitastor \\
|
ExecStart=etcd --name ${etcd_name} --data-dir /var/lib/etcd/vitastor \\
|
||||||
--snapshot-count 10000 --advertise-client-urls http://${etcds[num]}:2379 --listen-client-urls http://${etcds[num]}:2379 \\
|
--snapshot-count 10000 --advertise-client-urls ${etcd_url}:2379 --listen-client-urls ${etcd_url}:2379 \\
|
||||||
--initial-advertise-peer-urls http://${etcds[num]}:2380 --listen-peer-urls http://${etcds[num]}:2380 \\
|
--initial-advertise-peer-urls ${etcd_url}:2380 --listen-peer-urls ${etcd_url}:2380 \\
|
||||||
--initial-cluster-token vitastor-etcd-1 --initial-cluster ${etcd_cluster} \\
|
--initial-cluster-token vitastor-etcd-1 --initial-cluster ${etcd_cluster} \\
|
||||||
--initial-cluster-state new --max-txn-ops=100000 --max-request-bytes=104857600 \\
|
--initial-cluster-state new --max-txn-ops=100000 --max-request-bytes=104857600 \\
|
||||||
--auto-compaction-retention=10 --auto-compaction-mode=revision
|
--auto-compaction-retention=10 --auto-compaction-mode=revision
|
||||||
|
|||||||
@@ -276,6 +276,10 @@ function sum_inode_stats(state, prev_stats)
|
|||||||
}
|
}
|
||||||
for (const pool_id in osd_diff.inode_stats)
|
for (const pool_id in osd_diff.inode_stats)
|
||||||
{
|
{
|
||||||
|
if (!inode_stats[pool_id])
|
||||||
|
{
|
||||||
|
continue;
|
||||||
|
}
|
||||||
for (const inode_num in prev_stats.osd_diff[osd].inode_stats[pool_id])
|
for (const inode_num in prev_stats.osd_diff[osd].inode_stats[pool_id])
|
||||||
{
|
{
|
||||||
inode_stats[pool_id][inode_num] = inode_stats[pool_id][inode_num] || inode_stub();
|
inode_stats[pool_id][inode_num] = inode_stats[pool_id][inode_num] || inode_stub();
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "vitastor",
|
"name": "vitastor",
|
||||||
"version": "2.2.3",
|
"version": "3.0.2",
|
||||||
"description": "Low-level native bindings to Vitastor client library",
|
"description": "Low-level native bindings to Vitastor client library",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"keywords": [
|
"keywords": [
|
||||||
|
|||||||
@@ -261,7 +261,7 @@ sub free_image
|
|||||||
my ($vtype, $name, $vmid, undef, undef, undef) = $class->parse_volname($volname);
|
my ($vtype, $name, $vmid, undef, undef, undef) = $class->parse_volname($volname);
|
||||||
$class->deactivate_volume($storeid, $scfg, $volname);
|
$class->deactivate_volume($storeid, $scfg, $volname);
|
||||||
my $full_list = run_cli($scfg, [ 'ls', '-l' ]);
|
my $full_list = run_cli($scfg, [ 'ls', '-l' ]);
|
||||||
my $list = _process_list($scfg, $storeid, $full_list);
|
my $list = _process_list($scfg, $storeid, $full_list, 0);
|
||||||
# Remove image and all its snapshots
|
# Remove image and all its snapshots
|
||||||
my $rm_names = {
|
my $rm_names = {
|
||||||
map { ($prefix.$_->{name} => 1) }
|
map { ($prefix.$_->{name} => 1) }
|
||||||
@@ -269,6 +269,10 @@ sub free_image
|
|||||||
@$list
|
@$list
|
||||||
};
|
};
|
||||||
my $children = [ grep { $_->{parent_name} && $rm_names->{$_->{parent_name}} } @$full_list ];
|
my $children = [ grep { $_->{parent_name} && $rm_names->{$_->{parent_name}} } @$full_list ];
|
||||||
|
$children = [ grep {
|
||||||
|
substr($_->{name}, 0, length($prefix.$name)) ne $prefix.$name &&
|
||||||
|
substr($_->{name}, 0, length($prefix.$name)+1) ne $prefix.$name.'@'
|
||||||
|
} @$children ];
|
||||||
die "Image has children: ".join(', ', map {
|
die "Image has children: ".join(', ', map {
|
||||||
substr($_->{name}, 0, length $prefix) eq $prefix
|
substr($_->{name}, 0, length $prefix) eq $prefix
|
||||||
? substr($_->name, length $prefix)
|
? substr($_->name, length $prefix)
|
||||||
@@ -288,14 +292,15 @@ sub free_image
|
|||||||
|
|
||||||
sub _process_list
|
sub _process_list
|
||||||
{
|
{
|
||||||
my ($scfg, $storeid, $result) = @_;
|
my ($scfg, $storeid, $result, $skip_snapshot) = @_;
|
||||||
|
$skip_snapshot = 1 if !defined $skip_snapshot;
|
||||||
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
my $list = [];
|
my $list = [];
|
||||||
foreach my $el (@$result)
|
foreach my $el (@$result)
|
||||||
{
|
{
|
||||||
next if !$el->{name} || length($prefix) && substr($el->{name}, 0, length $prefix) ne $prefix;
|
next if !$el->{name} || length($prefix) && substr($el->{name}, 0, length $prefix) ne $prefix;
|
||||||
my $name = substr($el->{name}, length $prefix);
|
my $name = substr($el->{name}, length $prefix);
|
||||||
next if $name =~ /@/;
|
next if $skip_snapshot && $name =~ /@/;
|
||||||
my ($owner) = $name =~ /^(?:vm|base)-(\d+)-/s;
|
my ($owner) = $name =~ /^(?:vm|base)-(\d+)-/s;
|
||||||
next if !defined $owner;
|
next if !defined $owner;
|
||||||
my $parent = !defined $el->{parent_name}
|
my $parent = !defined $el->{parent_name}
|
||||||
@@ -494,4 +499,55 @@ sub rename_volume
|
|||||||
return "${storeid}:${base_name}${target_volname}";
|
return "${storeid}:${base_name}${target_volname}";
|
||||||
}
|
}
|
||||||
|
|
||||||
|
sub _monkey_patch_qemu_blockdev_options
|
||||||
|
{
|
||||||
|
my ($cfg, $volid, $machine_version, $options) = @_;
|
||||||
|
my ($storeid, $volname) = PVE::Storage::parse_volume_id($volid);
|
||||||
|
|
||||||
|
my $scfg = PVE::Storage::storage_config($cfg, $storeid);
|
||||||
|
|
||||||
|
my $plugin = PVE::Storage::Plugin->lookup($scfg->{type});
|
||||||
|
|
||||||
|
my ($vtype) = $plugin->parse_volname($volname);
|
||||||
|
die "cannot use volume of type '$vtype' as a QEMU blockdevice\n"
|
||||||
|
if $vtype ne 'images' && $vtype ne 'iso' && $vtype ne 'import';
|
||||||
|
|
||||||
|
return $plugin->qemu_blockdev_options($scfg, $storeid, $volname, $machine_version, $options);
|
||||||
|
}
|
||||||
|
|
||||||
|
sub qemu_blockdev_options
|
||||||
|
{
|
||||||
|
my ($class, $scfg, $storeid, $volname, $machine_version, $options) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
my ($vtype, $name, $vmid) = $class->parse_volname($volname);
|
||||||
|
$name .= '@'.$options->{'snapshot-name'} if $options->{'snapshot-name'};
|
||||||
|
if ($scfg->{vitastor_nbd})
|
||||||
|
{
|
||||||
|
my $mapped = run_cli($scfg, [ 'ls' ], binary => '/usr/bin/vitastor-nbd');
|
||||||
|
my ($kerneldev) = grep { $mapped->{$_}->{image} eq $prefix.$name } keys %$mapped;
|
||||||
|
die "Image not mapped via NBD" if !$kerneldev;
|
||||||
|
return { driver => 'host_device', filename => $kerneldev };
|
||||||
|
}
|
||||||
|
my $blockdev = {
|
||||||
|
driver => 'vitastor',
|
||||||
|
image => $prefix.$name,
|
||||||
|
};
|
||||||
|
if ($scfg->{vitastor_config_path})
|
||||||
|
{
|
||||||
|
$blockdev->{'config-path'} = $scfg->{vitastor_config_path};
|
||||||
|
}
|
||||||
|
if ($scfg->{vitastor_etcd_address})
|
||||||
|
{
|
||||||
|
# FIXME This is the only exception: etcd_address -> etcd_host for qemu
|
||||||
|
$blockdev->{'etcd-host'} = $scfg->{vitastor_etcd_address};
|
||||||
|
}
|
||||||
|
if ($scfg->{vitastor_etcd_prefix})
|
||||||
|
{
|
||||||
|
$blockdev->{'etcd-prefix'} = $scfg->{vitastor_etcd_prefix};
|
||||||
|
}
|
||||||
|
return $blockdev;
|
||||||
|
}
|
||||||
|
|
||||||
|
*PVE::Storage::qemu_blockdev_options = *_monkey_patch_qemu_blockdev_options;
|
||||||
|
|
||||||
1;
|
1;
|
||||||
|
|||||||
@@ -50,7 +50,7 @@ from cinder.volume import configuration
|
|||||||
from cinder.volume import driver
|
from cinder.volume import driver
|
||||||
from cinder.volume import volume_utils
|
from cinder.volume import volume_utils
|
||||||
|
|
||||||
VITASTOR_VERSION = '2.2.3'
|
VITASTOR_VERSION = '3.0.2'
|
||||||
|
|
||||||
LOG = logging.getLogger(__name__)
|
LOG = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,39 @@
|
|||||||
|
From 98d3f68a40130c438854f61db6025f9e9b099cb6 Mon Sep 17 00:00:00 2001
|
||||||
|
From: Vitaliy Filippov <vitalifster@gmail.com>
|
||||||
|
Date: Sat, 20 Dec 2025 14:44:35 +0300
|
||||||
|
Subject: [PATCH] Do not require atomic writes to be power of 2 sized and
|
||||||
|
aligned on length boundary
|
||||||
|
|
||||||
|
It contradicts NVMe specification where alignment is only required when atomic
|
||||||
|
write boundary (NABSPF/NABO) is set and highly limits usage of NVMe atomic writes
|
||||||
|
|
||||||
|
Signed-off-by: Vitaliy Filippov <vitalifster@gmail.com>
|
||||||
|
---
|
||||||
|
fs/read_write.c | 8 --------
|
||||||
|
1 file changed, 8 deletions(-)
|
||||||
|
|
||||||
|
diff --git a/fs/read_write.c b/fs/read_write.c
|
||||||
|
index 833bae068770..5467d710108d 100644
|
||||||
|
--- a/fs/read_write.c
|
||||||
|
+++ b/fs/read_write.c
|
||||||
|
@@ -1802,17 +1802,9 @@ int generic_file_rw_checks(struct file *file_in, struct file *file_out)
|
||||||
|
|
||||||
|
int generic_atomic_write_valid(struct kiocb *iocb, struct iov_iter *iter)
|
||||||
|
{
|
||||||
|
- size_t len = iov_iter_count(iter);
|
||||||
|
-
|
||||||
|
if (!iter_is_ubuf(iter))
|
||||||
|
return -EINVAL;
|
||||||
|
|
||||||
|
- if (!is_power_of_2(len))
|
||||||
|
- return -EINVAL;
|
||||||
|
-
|
||||||
|
- if (!IS_ALIGNED(iocb->ki_pos, len))
|
||||||
|
- return -EINVAL;
|
||||||
|
-
|
||||||
|
if (!(iocb->ki_flags & IOCB_DIRECT))
|
||||||
|
return -EOPNOTSUPP;
|
||||||
|
|
||||||
|
--
|
||||||
|
2.51.0
|
||||||
|
|
||||||
@@ -0,0 +1,172 @@
|
|||||||
|
Index: pve-qemu-kvm-10.0.2/block/meson.build
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.0.2.orig/block/meson.build
|
||||||
|
+++ pve-qemu-kvm-10.0.2/block/meson.build
|
||||||
|
@@ -126,6 +126,7 @@ foreach m : [
|
||||||
|
[libnfs, 'nfs', files('nfs.c')],
|
||||||
|
[libssh, 'ssh', files('ssh.c')],
|
||||||
|
[rbd, 'rbd', files('rbd.c')],
|
||||||
|
+ [vitastor, 'vitastor', files('vitastor.c')],
|
||||||
|
]
|
||||||
|
if m[0].found()
|
||||||
|
module_ss = ss.source_set()
|
||||||
|
Index: pve-qemu-kvm-10.0.2/meson.build
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.0.2.orig/meson.build
|
||||||
|
+++ pve-qemu-kvm-10.0.2/meson.build
|
||||||
|
@@ -1622,6 +1622,26 @@ if not get_option('rbd').auto() or have_
|
||||||
|
endif
|
||||||
|
endif
|
||||||
|
|
||||||
|
+vitastor = not_found
|
||||||
|
+if not get_option('vitastor').auto() or have_block
|
||||||
|
+ libvitastor_client = cc.find_library('vitastor_client', has_headers: ['vitastor_c.h'],
|
||||||
|
+ required: get_option('vitastor'))
|
||||||
|
+ if libvitastor_client.found()
|
||||||
|
+ if cc.links('''
|
||||||
|
+ #include <vitastor_c.h>
|
||||||
|
+ int main(void) {
|
||||||
|
+ vitastor_c_create_qemu(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0);
|
||||||
|
+ return 0;
|
||||||
|
+ }''', dependencies: libvitastor_client)
|
||||||
|
+ vitastor = declare_dependency(dependencies: libvitastor_client)
|
||||||
|
+ elif get_option('vitastor').enabled()
|
||||||
|
+ error('could not link libvitastor_client')
|
||||||
|
+ else
|
||||||
|
+ warning('could not link libvitastor_client, disabling')
|
||||||
|
+ endif
|
||||||
|
+ endif
|
||||||
|
+endif
|
||||||
|
+
|
||||||
|
glusterfs = not_found
|
||||||
|
glusterfs_ftruncate_has_stat = false
|
||||||
|
glusterfs_iocb_has_stat = false
|
||||||
|
@@ -2514,6 +2534,7 @@ endif
|
||||||
|
config_host_data.set('CONFIG_OPENGL', opengl.found())
|
||||||
|
config_host_data.set('CONFIG_PLUGIN', get_option('plugins'))
|
||||||
|
config_host_data.set('CONFIG_RBD', rbd.found())
|
||||||
|
+config_host_data.set('CONFIG_VITASTOR', vitastor.found())
|
||||||
|
config_host_data.set('CONFIG_RDMA', rdma.found())
|
||||||
|
config_host_data.set('CONFIG_RELOCATABLE', get_option('relocatable'))
|
||||||
|
config_host_data.set('CONFIG_SAFESTACK', get_option('safe_stack'))
|
||||||
|
@@ -4812,6 +4833,7 @@ summary_info += {'fdt support': fd
|
||||||
|
summary_info += {'libcap-ng support': libcap_ng}
|
||||||
|
summary_info += {'bpf support': libbpf}
|
||||||
|
summary_info += {'rbd support': rbd}
|
||||||
|
+summary_info += {'vitastor support': vitastor}
|
||||||
|
summary_info += {'smartcard support': cacard}
|
||||||
|
summary_info += {'U2F support': u2f}
|
||||||
|
summary_info += {'libusb': libusb}
|
||||||
|
Index: pve-qemu-kvm-10.0.2/meson_options.txt
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.0.2.orig/meson_options.txt
|
||||||
|
+++ pve-qemu-kvm-10.0.2/meson_options.txt
|
||||||
|
@@ -202,6 +202,8 @@ option('pvg', type: 'feature', value: 'a
|
||||||
|
description: 'macOS paravirtualized graphics support')
|
||||||
|
option('rbd', type : 'feature', value : 'auto',
|
||||||
|
description: 'Ceph block device driver')
|
||||||
|
+option('vitastor', type : 'feature', value : 'auto',
|
||||||
|
+ description: 'Vitastor block device driver')
|
||||||
|
option('opengl', type : 'feature', value : 'auto',
|
||||||
|
description: 'OpenGL support')
|
||||||
|
option('rdma', type : 'feature', value : 'auto',
|
||||||
|
Index: pve-qemu-kvm-10.0.2/qapi/block-core.json
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.0.2.orig/qapi/block-core.json
|
||||||
|
+++ pve-qemu-kvm-10.0.2/qapi/block-core.json
|
||||||
|
@@ -3599,7 +3599,7 @@
|
||||||
|
'raw', 'rbd',
|
||||||
|
{ 'name': 'replication', 'if': 'CONFIG_REPLICATION' },
|
||||||
|
'pbs',
|
||||||
|
- 'ssh', 'throttle', 'vdi', 'vhdx',
|
||||||
|
+ 'ssh', 'throttle', 'vdi', 'vhdx', 'vitastor',
|
||||||
|
{ 'name': 'virtio-blk-vfio-pci', 'if': 'CONFIG_BLKIO' },
|
||||||
|
{ 'name': 'virtio-blk-vhost-user', 'if': 'CONFIG_BLKIO' },
|
||||||
|
{ 'name': 'virtio-blk-vhost-vdpa', 'if': 'CONFIG_BLKIO' },
|
||||||
|
@@ -4725,6 +4725,28 @@
|
||||||
|
'*server': ['InetSocketAddressBase'] } }
|
||||||
|
|
||||||
|
##
|
||||||
|
+# @BlockdevOptionsVitastor:
|
||||||
|
+#
|
||||||
|
+# Driver specific block device options for vitastor
|
||||||
|
+#
|
||||||
|
+# @image: Image name
|
||||||
|
+# @inode: Inode number
|
||||||
|
+# @pool: Pool ID
|
||||||
|
+# @size: Desired image size in bytes
|
||||||
|
+# @config-path: Path to Vitastor configuration
|
||||||
|
+# @etcd-host: etcd connection address(es)
|
||||||
|
+# @etcd-prefix: etcd key/value prefix
|
||||||
|
+##
|
||||||
|
+{ 'struct': 'BlockdevOptionsVitastor',
|
||||||
|
+ 'data': { '*inode': 'uint64',
|
||||||
|
+ '*pool': 'uint64',
|
||||||
|
+ '*size': 'uint64',
|
||||||
|
+ '*image': 'str',
|
||||||
|
+ '*config-path': 'str',
|
||||||
|
+ '*etcd-host': 'str',
|
||||||
|
+ '*etcd-prefix': 'str' } }
|
||||||
|
+
|
||||||
|
+##
|
||||||
|
# @ReplicationMode:
|
||||||
|
#
|
||||||
|
# An enumeration of replication modes.
|
||||||
|
@@ -5194,6 +5216,7 @@
|
||||||
|
'throttle': 'BlockdevOptionsThrottle',
|
||||||
|
'vdi': 'BlockdevOptionsGenericFormat',
|
||||||
|
'vhdx': 'BlockdevOptionsGenericFormat',
|
||||||
|
+ 'vitastor': 'BlockdevOptionsVitastor',
|
||||||
|
'virtio-blk-vfio-pci':
|
||||||
|
{ 'type': 'BlockdevOptionsVirtioBlkVfioPci',
|
||||||
|
'if': 'CONFIG_BLKIO' },
|
||||||
|
@@ -5674,6 +5697,20 @@
|
||||||
|
'*encrypt' : 'RbdEncryptionCreateOptions' } }
|
||||||
|
|
||||||
|
##
|
||||||
|
+# @BlockdevCreateOptionsVitastor:
|
||||||
|
+#
|
||||||
|
+# Driver specific image creation options for Vitastor.
|
||||||
|
+#
|
||||||
|
+# @location: Where to store the new image file. This location cannot
|
||||||
|
+# point to a snapshot.
|
||||||
|
+#
|
||||||
|
+# @size: Size of the virtual disk in bytes
|
||||||
|
+##
|
||||||
|
+{ 'struct': 'BlockdevCreateOptionsVitastor',
|
||||||
|
+ 'data': { 'location': 'BlockdevOptionsVitastor',
|
||||||
|
+ 'size': 'size' } }
|
||||||
|
+
|
||||||
|
+##
|
||||||
|
# @BlockdevVmdkSubformat:
|
||||||
|
#
|
||||||
|
# Subformat options for VMDK images
|
||||||
|
@@ -5895,6 +5932,7 @@
|
||||||
|
'ssh': 'BlockdevCreateOptionsSsh',
|
||||||
|
'vdi': 'BlockdevCreateOptionsVdi',
|
||||||
|
'vhdx': 'BlockdevCreateOptionsVhdx',
|
||||||
|
+ 'vitastor': 'BlockdevCreateOptionsVitastor',
|
||||||
|
'vmdk': 'BlockdevCreateOptionsVmdk',
|
||||||
|
'vpc': 'BlockdevCreateOptionsVpc'
|
||||||
|
} }
|
||||||
|
Index: pve-qemu-kvm-10.0.2/scripts/meson-buildoptions.sh
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.0.2.orig/scripts/meson-buildoptions.sh
|
||||||
|
+++ pve-qemu-kvm-10.0.2/scripts/meson-buildoptions.sh
|
||||||
|
@@ -175,6 +175,7 @@ meson_options_help() {
|
||||||
|
printf "%s\n" ' qga-vss build QGA VSS support (broken with MinGW)'
|
||||||
|
printf "%s\n" ' qpl Query Processing Library support'
|
||||||
|
printf "%s\n" ' rbd Ceph block device driver'
|
||||||
|
+ printf "%s\n" ' vitastor Vitastor block device driver'
|
||||||
|
printf "%s\n" ' rdma Enable RDMA-based migration'
|
||||||
|
printf "%s\n" ' replication replication support'
|
||||||
|
printf "%s\n" ' rust Rust support'
|
||||||
|
@@ -458,6 +459,8 @@ _meson_option_parse() {
|
||||||
|
--disable-qpl) printf "%s" -Dqpl=disabled ;;
|
||||||
|
--enable-rbd) printf "%s" -Drbd=enabled ;;
|
||||||
|
--disable-rbd) printf "%s" -Drbd=disabled ;;
|
||||||
|
+ --enable-vitastor) printf "%s" -Dvitastor=enabled ;;
|
||||||
|
+ --disable-vitastor) printf "%s" -Dvitastor=disabled ;;
|
||||||
|
--enable-rdma) printf "%s" -Drdma=enabled ;;
|
||||||
|
--disable-rdma) printf "%s" -Drdma=disabled ;;
|
||||||
|
--enable-relocatable) printf "%s" -Drelocatable=true ;;
|
||||||
@@ -0,0 +1,172 @@
|
|||||||
|
Index: pve-qemu-kvm-10.1.2/block/meson.build
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.1.2.orig/block/meson.build
|
||||||
|
+++ pve-qemu-kvm-10.1.2/block/meson.build
|
||||||
|
@@ -126,6 +126,7 @@ foreach m : [
|
||||||
|
[libnfs, 'nfs', files('nfs.c')],
|
||||||
|
[libssh, 'ssh', files('ssh.c')],
|
||||||
|
[rbd, 'rbd', files('rbd.c')],
|
||||||
|
+ [vitastor, 'vitastor', files('vitastor.c')],
|
||||||
|
]
|
||||||
|
if m[0].found()
|
||||||
|
module_ss = ss.source_set()
|
||||||
|
Index: pve-qemu-kvm-10.1.2/meson.build
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.1.2.orig/meson.build
|
||||||
|
+++ pve-qemu-kvm-10.1.2/meson.build
|
||||||
|
@@ -1653,6 +1653,26 @@ if not get_option('rbd').auto() or have_
|
||||||
|
endif
|
||||||
|
endif
|
||||||
|
|
||||||
|
+vitastor = not_found
|
||||||
|
+if not get_option('vitastor').auto() or have_block
|
||||||
|
+ libvitastor_client = cc.find_library('vitastor_client', has_headers: ['vitastor_c.h'],
|
||||||
|
+ required: get_option('vitastor'))
|
||||||
|
+ if libvitastor_client.found()
|
||||||
|
+ if cc.links('''
|
||||||
|
+ #include <vitastor_c.h>
|
||||||
|
+ int main(void) {
|
||||||
|
+ vitastor_c_create_qemu(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0);
|
||||||
|
+ return 0;
|
||||||
|
+ }''', dependencies: libvitastor_client)
|
||||||
|
+ vitastor = declare_dependency(dependencies: libvitastor_client)
|
||||||
|
+ elif get_option('vitastor').enabled()
|
||||||
|
+ error('could not link libvitastor_client')
|
||||||
|
+ else
|
||||||
|
+ warning('could not link libvitastor_client, disabling')
|
||||||
|
+ endif
|
||||||
|
+ endif
|
||||||
|
+endif
|
||||||
|
+
|
||||||
|
glusterfs = not_found
|
||||||
|
glusterfs_ftruncate_has_stat = false
|
||||||
|
glusterfs_iocb_has_stat = false
|
||||||
|
@@ -2552,6 +2572,7 @@ endif
|
||||||
|
config_host_data.set('CONFIG_OPENGL', opengl.found())
|
||||||
|
config_host_data.set('CONFIG_PLUGIN', get_option('plugins'))
|
||||||
|
config_host_data.set('CONFIG_RBD', rbd.found())
|
||||||
|
+config_host_data.set('CONFIG_VITASTOR', vitastor.found())
|
||||||
|
config_host_data.set('CONFIG_RDMA', rdma.found())
|
||||||
|
config_host_data.set('CONFIG_RELOCATABLE', get_option('relocatable'))
|
||||||
|
config_host_data.set('CONFIG_SAFESTACK', get_option('safe_stack'))
|
||||||
|
@@ -4984,6 +5005,7 @@ summary_info += {'fdt support': fd
|
||||||
|
summary_info += {'libcap-ng support': libcap_ng}
|
||||||
|
summary_info += {'bpf support': libbpf}
|
||||||
|
summary_info += {'rbd support': rbd}
|
||||||
|
+summary_info += {'vitastor support': vitastor}
|
||||||
|
summary_info += {'smartcard support': cacard}
|
||||||
|
summary_info += {'U2F support': u2f}
|
||||||
|
summary_info += {'libusb': libusb}
|
||||||
|
Index: pve-qemu-kvm-10.1.2/meson_options.txt
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.1.2.orig/meson_options.txt
|
||||||
|
+++ pve-qemu-kvm-10.1.2/meson_options.txt
|
||||||
|
@@ -202,6 +202,8 @@ option('pvg', type: 'feature', value: 'a
|
||||||
|
description: 'macOS paravirtualized graphics support')
|
||||||
|
option('rbd', type : 'feature', value : 'auto',
|
||||||
|
description: 'Ceph block device driver')
|
||||||
|
+option('vitastor', type : 'feature', value : 'auto',
|
||||||
|
+ description: 'Vitastor block device driver')
|
||||||
|
option('opengl', type : 'feature', value : 'auto',
|
||||||
|
description: 'OpenGL support')
|
||||||
|
option('rdma', type : 'feature', value : 'auto',
|
||||||
|
Index: pve-qemu-kvm-10.1.2/qapi/block-core.json
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.1.2.orig/qapi/block-core.json
|
||||||
|
+++ pve-qemu-kvm-10.1.2/qapi/block-core.json
|
||||||
|
@@ -3647,7 +3647,7 @@
|
||||||
|
'raw', 'rbd',
|
||||||
|
{ 'name': 'replication', 'if': 'CONFIG_REPLICATION' },
|
||||||
|
'pbs',
|
||||||
|
- 'ssh', 'throttle', 'vdi', 'vhdx',
|
||||||
|
+ 'ssh', 'throttle', 'vdi', 'vhdx', 'vitastor',
|
||||||
|
{ 'name': 'virtio-blk-vfio-pci', 'if': 'CONFIG_BLKIO' },
|
||||||
|
{ 'name': 'virtio-blk-vhost-user', 'if': 'CONFIG_BLKIO' },
|
||||||
|
{ 'name': 'virtio-blk-vhost-vdpa', 'if': 'CONFIG_BLKIO' },
|
||||||
|
@@ -4773,6 +4773,28 @@
|
||||||
|
'*server': ['InetSocketAddressBase'] } }
|
||||||
|
|
||||||
|
##
|
||||||
|
+# @BlockdevOptionsVitastor:
|
||||||
|
+#
|
||||||
|
+# Driver specific block device options for vitastor
|
||||||
|
+#
|
||||||
|
+# @image: Image name
|
||||||
|
+# @inode: Inode number
|
||||||
|
+# @pool: Pool ID
|
||||||
|
+# @size: Desired image size in bytes
|
||||||
|
+# @config-path: Path to Vitastor configuration
|
||||||
|
+# @etcd-host: etcd connection address(es)
|
||||||
|
+# @etcd-prefix: etcd key/value prefix
|
||||||
|
+##
|
||||||
|
+{ 'struct': 'BlockdevOptionsVitastor',
|
||||||
|
+ 'data': { '*inode': 'uint64',
|
||||||
|
+ '*pool': 'uint64',
|
||||||
|
+ '*size': 'uint64',
|
||||||
|
+ '*image': 'str',
|
||||||
|
+ '*config-path': 'str',
|
||||||
|
+ '*etcd-host': 'str',
|
||||||
|
+ '*etcd-prefix': 'str' } }
|
||||||
|
+
|
||||||
|
+##
|
||||||
|
# @ReplicationMode:
|
||||||
|
#
|
||||||
|
# An enumeration of replication modes.
|
||||||
|
@@ -5242,6 +5264,7 @@
|
||||||
|
'throttle': 'BlockdevOptionsThrottle',
|
||||||
|
'vdi': 'BlockdevOptionsGenericFormat',
|
||||||
|
'vhdx': 'BlockdevOptionsGenericFormat',
|
||||||
|
+ 'vitastor': 'BlockdevOptionsVitastor',
|
||||||
|
'virtio-blk-vfio-pci':
|
||||||
|
{ 'type': 'BlockdevOptionsVirtioBlkVfioPci',
|
||||||
|
'if': 'CONFIG_BLKIO' },
|
||||||
|
@@ -5722,6 +5745,20 @@
|
||||||
|
'*encrypt' : 'RbdEncryptionCreateOptions' } }
|
||||||
|
|
||||||
|
##
|
||||||
|
+# @BlockdevCreateOptionsVitastor:
|
||||||
|
+#
|
||||||
|
+# Driver specific image creation options for Vitastor.
|
||||||
|
+#
|
||||||
|
+# @location: Where to store the new image file. This location cannot
|
||||||
|
+# point to a snapshot.
|
||||||
|
+#
|
||||||
|
+# @size: Size of the virtual disk in bytes
|
||||||
|
+##
|
||||||
|
+{ 'struct': 'BlockdevCreateOptionsVitastor',
|
||||||
|
+ 'data': { 'location': 'BlockdevOptionsVitastor',
|
||||||
|
+ 'size': 'size' } }
|
||||||
|
+
|
||||||
|
+##
|
||||||
|
# @BlockdevVmdkSubformat:
|
||||||
|
#
|
||||||
|
# Subformat options for VMDK images
|
||||||
|
@@ -5943,6 +5980,7 @@
|
||||||
|
'ssh': 'BlockdevCreateOptionsSsh',
|
||||||
|
'vdi': 'BlockdevCreateOptionsVdi',
|
||||||
|
'vhdx': 'BlockdevCreateOptionsVhdx',
|
||||||
|
+ 'vitastor': 'BlockdevCreateOptionsVitastor',
|
||||||
|
'vmdk': 'BlockdevCreateOptionsVmdk',
|
||||||
|
'vpc': 'BlockdevCreateOptionsVpc'
|
||||||
|
} }
|
||||||
|
Index: pve-qemu-kvm-10.1.2/scripts/meson-buildoptions.sh
|
||||||
|
===================================================================
|
||||||
|
--- pve-qemu-kvm-10.1.2.orig/scripts/meson-buildoptions.sh
|
||||||
|
+++ pve-qemu-kvm-10.1.2/scripts/meson-buildoptions.sh
|
||||||
|
@@ -175,6 +175,7 @@ meson_options_help() {
|
||||||
|
printf "%s\n" ' qga-vss build QGA VSS support (broken with MinGW)'
|
||||||
|
printf "%s\n" ' qpl Query Processing Library support'
|
||||||
|
printf "%s\n" ' rbd Ceph block device driver'
|
||||||
|
+ printf "%s\n" ' vitastor Vitastor block device driver'
|
||||||
|
printf "%s\n" ' rdma Enable RDMA-based migration'
|
||||||
|
printf "%s\n" ' replication replication support'
|
||||||
|
printf "%s\n" ' rust Rust support'
|
||||||
|
@@ -459,6 +460,8 @@ _meson_option_parse() {
|
||||||
|
--disable-qpl) printf "%s" -Dqpl=disabled ;;
|
||||||
|
--enable-rbd) printf "%s" -Drbd=enabled ;;
|
||||||
|
--disable-rbd) printf "%s" -Drbd=disabled ;;
|
||||||
|
+ --enable-vitastor) printf "%s" -Dvitastor=enabled ;;
|
||||||
|
+ --disable-vitastor) printf "%s" -Dvitastor=disabled ;;
|
||||||
|
--enable-rdma) printf "%s" -Drdma=enabled ;;
|
||||||
|
--disable-rdma) printf "%s" -Drdma=disabled ;;
|
||||||
|
--enable-relocatable) printf "%s" -Drelocatable=true ;;
|
||||||
@@ -0,0 +1,172 @@
|
|||||||
|
diff --git a/block/meson.build b/block/meson.build
|
||||||
|
index 34b1b2a306..24ca0f1e52 100644
|
||||||
|
--- a/block/meson.build
|
||||||
|
+++ b/block/meson.build
|
||||||
|
@@ -114,6 +114,7 @@ foreach m : [
|
||||||
|
[libnfs, 'nfs', files('nfs.c')],
|
||||||
|
[libssh, 'ssh', files('ssh.c')],
|
||||||
|
[rbd, 'rbd', files('rbd.c')],
|
||||||
|
+ [vitastor, 'vitastor', files('vitastor.c')],
|
||||||
|
]
|
||||||
|
if m[0].found()
|
||||||
|
module_ss = ss.source_set()
|
||||||
|
diff --git a/meson.build b/meson.build
|
||||||
|
index 41f68d3806..29eaed9ba4 100644
|
||||||
|
--- a/meson.build
|
||||||
|
+++ b/meson.build
|
||||||
|
@@ -1622,6 +1622,26 @@ if not get_option('rbd').auto() or have_block
|
||||||
|
endif
|
||||||
|
endif
|
||||||
|
|
||||||
|
+vitastor = not_found
|
||||||
|
+if not get_option('vitastor').auto() or have_block
|
||||||
|
+ libvitastor_client = cc.find_library('vitastor_client', has_headers: ['vitastor_c.h'],
|
||||||
|
+ required: get_option('vitastor'))
|
||||||
|
+ if libvitastor_client.found()
|
||||||
|
+ if cc.links('''
|
||||||
|
+ #include <vitastor_c.h>
|
||||||
|
+ int main(void) {
|
||||||
|
+ vitastor_c_create_qemu(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0);
|
||||||
|
+ return 0;
|
||||||
|
+ }''', dependencies: libvitastor_client)
|
||||||
|
+ vitastor = declare_dependency(dependencies: libvitastor_client)
|
||||||
|
+ elif get_option('vitastor').enabled()
|
||||||
|
+ error('could not link libvitastor_client')
|
||||||
|
+ else
|
||||||
|
+ warning('could not link libvitastor_client, disabling')
|
||||||
|
+ endif
|
||||||
|
+ endif
|
||||||
|
+endif
|
||||||
|
+
|
||||||
|
glusterfs = not_found
|
||||||
|
glusterfs_ftruncate_has_stat = false
|
||||||
|
glusterfs_iocb_has_stat = false
|
||||||
|
@@ -2506,6 +2526,7 @@ endif
|
||||||
|
config_host_data.set('CONFIG_OPENGL', opengl.found())
|
||||||
|
config_host_data.set('CONFIG_PLUGIN', get_option('plugins'))
|
||||||
|
config_host_data.set('CONFIG_RBD', rbd.found())
|
||||||
|
+config_host_data.set('CONFIG_VITASTOR', vitastor.found())
|
||||||
|
config_host_data.set('CONFIG_RDMA', rdma.found())
|
||||||
|
config_host_data.set('CONFIG_RELOCATABLE', get_option('relocatable'))
|
||||||
|
config_host_data.set('CONFIG_SAFESTACK', get_option('safe_stack'))
|
||||||
|
@@ -4813,6 +4834,7 @@ summary_info += {'fdt support': fdt_opt == 'internal' ? 'internal' : fdt}
|
||||||
|
summary_info += {'libcap-ng support': libcap_ng}
|
||||||
|
summary_info += {'bpf support': libbpf}
|
||||||
|
summary_info += {'rbd support': rbd}
|
||||||
|
+summary_info += {'vitastor support': vitastor}
|
||||||
|
summary_info += {'smartcard support': cacard}
|
||||||
|
summary_info += {'U2F support': u2f}
|
||||||
|
summary_info += {'libusb': libusb}
|
||||||
|
diff --git a/meson_options.txt b/meson_options.txt
|
||||||
|
index 59d973bca0..a3e7123980 100644
|
||||||
|
--- a/meson_options.txt
|
||||||
|
+++ b/meson_options.txt
|
||||||
|
@@ -202,6 +202,8 @@ option('pvg', type: 'feature', value: 'auto',
|
||||||
|
description: 'macOS paravirtualized graphics support')
|
||||||
|
option('rbd', type : 'feature', value : 'auto',
|
||||||
|
description: 'Ceph block device driver')
|
||||||
|
+option('vitastor', type : 'feature', value : 'auto',
|
||||||
|
+ description: 'Vitastor block device driver')
|
||||||
|
option('opengl', type : 'feature', value : 'auto',
|
||||||
|
description: 'OpenGL support')
|
||||||
|
option('rdma', type : 'feature', value : 'auto',
|
||||||
|
diff --git a/qapi/block-core.json b/qapi/block-core.json
|
||||||
|
index b1937780e1..a511193620 100644
|
||||||
|
--- a/qapi/block-core.json
|
||||||
|
+++ b/qapi/block-core.json
|
||||||
|
@@ -3216,7 +3216,7 @@
|
||||||
|
'parallels', 'preallocate', 'qcow', 'qcow2', 'qed', 'quorum',
|
||||||
|
'raw', 'rbd',
|
||||||
|
{ 'name': 'replication', 'if': 'CONFIG_REPLICATION' },
|
||||||
|
- 'ssh', 'throttle', 'vdi', 'vhdx',
|
||||||
|
+ 'ssh', 'throttle', 'vdi', 'vhdx', 'vitastor',
|
||||||
|
{ 'name': 'virtio-blk-vfio-pci', 'if': 'CONFIG_BLKIO' },
|
||||||
|
{ 'name': 'virtio-blk-vhost-user', 'if': 'CONFIG_BLKIO' },
|
||||||
|
{ 'name': 'virtio-blk-vhost-vdpa', 'if': 'CONFIG_BLKIO' },
|
||||||
|
@@ -4299,6 +4299,28 @@
|
||||||
|
'*key-secret': 'str',
|
||||||
|
'*server': ['InetSocketAddressBase'] } }
|
||||||
|
|
||||||
|
+##
|
||||||
|
+# @BlockdevOptionsVitastor:
|
||||||
|
+#
|
||||||
|
+# Driver specific block device options for vitastor
|
||||||
|
+#
|
||||||
|
+# @image: Image name
|
||||||
|
+# @inode: Inode number
|
||||||
|
+# @pool: Pool ID
|
||||||
|
+# @size: Desired image size in bytes
|
||||||
|
+# @config-path: Path to Vitastor configuration
|
||||||
|
+# @etcd-host: etcd connection address(es)
|
||||||
|
+# @etcd-prefix: etcd key/value prefix
|
||||||
|
+##
|
||||||
|
+{ 'struct': 'BlockdevOptionsVitastor',
|
||||||
|
+ 'data': { '*inode': 'uint64',
|
||||||
|
+ '*pool': 'uint64',
|
||||||
|
+ '*size': 'uint64',
|
||||||
|
+ '*image': 'str',
|
||||||
|
+ '*config-path': 'str',
|
||||||
|
+ '*etcd-host': 'str',
|
||||||
|
+ '*etcd-prefix': 'str' } }
|
||||||
|
+
|
||||||
|
##
|
||||||
|
# @ReplicationMode:
|
||||||
|
#
|
||||||
|
@@ -4767,6 +4789,7 @@
|
||||||
|
'throttle': 'BlockdevOptionsThrottle',
|
||||||
|
'vdi': 'BlockdevOptionsGenericFormat',
|
||||||
|
'vhdx': 'BlockdevOptionsGenericFormat',
|
||||||
|
+ 'vitastor': 'BlockdevOptionsVitastor',
|
||||||
|
'virtio-blk-vfio-pci':
|
||||||
|
{ 'type': 'BlockdevOptionsVirtioBlkVfioPci',
|
||||||
|
'if': 'CONFIG_BLKIO' },
|
||||||
|
@@ -5240,6 +5263,20 @@
|
||||||
|
'*cluster-size' : 'size',
|
||||||
|
'*encrypt' : 'RbdEncryptionCreateOptions' } }
|
||||||
|
|
||||||
|
+##
|
||||||
|
+# @BlockdevCreateOptionsVitastor:
|
||||||
|
+#
|
||||||
|
+# Driver specific image creation options for Vitastor.
|
||||||
|
+#
|
||||||
|
+# @location: Where to store the new image file. This location cannot
|
||||||
|
+# point to a snapshot.
|
||||||
|
+#
|
||||||
|
+# @size: Size of the virtual disk in bytes
|
||||||
|
+##
|
||||||
|
+{ 'struct': 'BlockdevCreateOptionsVitastor',
|
||||||
|
+ 'data': { 'location': 'BlockdevOptionsVitastor',
|
||||||
|
+ 'size': 'size' } }
|
||||||
|
+
|
||||||
|
##
|
||||||
|
# @BlockdevVmdkSubformat:
|
||||||
|
#
|
||||||
|
@@ -5462,6 +5499,7 @@
|
||||||
|
'ssh': 'BlockdevCreateOptionsSsh',
|
||||||
|
'vdi': 'BlockdevCreateOptionsVdi',
|
||||||
|
'vhdx': 'BlockdevCreateOptionsVhdx',
|
||||||
|
+ 'vitastor': 'BlockdevCreateOptionsVitastor',
|
||||||
|
'vmdk': 'BlockdevCreateOptionsVmdk',
|
||||||
|
'vpc': 'BlockdevCreateOptionsVpc'
|
||||||
|
} }
|
||||||
|
diff --git a/scripts/meson-buildoptions.sh b/scripts/meson-buildoptions.sh
|
||||||
|
index 3e8e00852b..45aff3b6a9 100644
|
||||||
|
--- a/scripts/meson-buildoptions.sh
|
||||||
|
+++ b/scripts/meson-buildoptions.sh
|
||||||
|
@@ -175,6 +175,7 @@ meson_options_help() {
|
||||||
|
printf "%s\n" ' qga-vss build QGA VSS support (broken with MinGW)'
|
||||||
|
printf "%s\n" ' qpl Query Processing Library support'
|
||||||
|
printf "%s\n" ' rbd Ceph block device driver'
|
||||||
|
+ printf "%s\n" ' vitastor Vitastor block device driver'
|
||||||
|
printf "%s\n" ' rdma Enable RDMA-based migration'
|
||||||
|
printf "%s\n" ' replication replication support'
|
||||||
|
printf "%s\n" ' rust Rust support'
|
||||||
|
@@ -458,6 +459,8 @@ _meson_option_parse() {
|
||||||
|
--disable-qpl) printf "%s" -Dqpl=disabled ;;
|
||||||
|
--enable-rbd) printf "%s" -Drbd=enabled ;;
|
||||||
|
--disable-rbd) printf "%s" -Drbd=disabled ;;
|
||||||
|
+ --enable-vitastor) printf "%s" -Dvitastor=enabled ;;
|
||||||
|
+ --disable-vitastor) printf "%s" -Dvitastor=disabled ;;
|
||||||
|
--enable-rdma) printf "%s" -Drdma=enabled ;;
|
||||||
|
--disable-rdma) printf "%s" -Drdma=disabled ;;
|
||||||
|
--enable-relocatable) printf "%s" -Drelocatable=true ;;
|
||||||
@@ -0,0 +1,29 @@
|
|||||||
|
diff --git a/src/client/qemu_driver.c b/src/client/qemu_driver.c
|
||||||
|
index d8356dab..5f4cd50d 100644
|
||||||
|
--- a/src/client/qemu_driver.c
|
||||||
|
+++ b/src/client/qemu_driver.c
|
||||||
|
@@ -974,14 +974,21 @@ static void vitastor_co_read_bitmap_cb(void *opaque, long retval, uint8_t *bitma
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
-static int coroutine_fn vitastor_co_block_status(
|
||||||
|
- BlockDriverState *bs, bool want_zero, int64_t offset, int64_t bytes,
|
||||||
|
- int64_t *pnum, int64_t *map, BlockDriverState **file)
|
||||||
|
+static int coroutine_fn vitastor_co_block_status(BlockDriverState *bs,
|
||||||
|
+#if QEMU_VERSION_MAJOR > 10 || QEMU_VERSION_MAJOR == 10 && QEMU_VERSION_MINOR >= 1
|
||||||
|
+ unsigned int mode,
|
||||||
|
+#else
|
||||||
|
+ bool want_zero,
|
||||||
|
+#endif
|
||||||
|
+ int64_t offset, int64_t bytes, int64_t *pnum, int64_t *map, BlockDriverState **file)
|
||||||
|
{
|
||||||
|
// Allocated => return BDRV_BLOCK_DATA|BDRV_BLOCK_OFFSET_VALID
|
||||||
|
// Not allocated => return 0
|
||||||
|
// Error => return -errno
|
||||||
|
// Set pnum to length of the extent, `*map` = `offset`, `*file` = `bs`
|
||||||
|
+#if QEMU_VERSION_MAJOR > 10 || QEMU_VERSION_MAJOR == 10 && QEMU_VERSION_MINOR >= 1
|
||||||
|
+ int want_zero = (mode == BDRV_WANT_PRECISE);
|
||||||
|
+#endif
|
||||||
|
VitastorRPC task;
|
||||||
|
VitastorClient *client = bs->opaque;
|
||||||
|
uint64_t inode = client->watch ? vitastor_c_inode_get_num(client->watch) : client->inode;
|
||||||
@@ -21,17 +21,3 @@ RUN rpm --nomd5 -i fio*.src.rpm
|
|||||||
RUN rm -f /etc/yum.repos.d/CentOS-Media.repo
|
RUN rm -f /etc/yum.repos.d/CentOS-Media.repo
|
||||||
RUN cd ~/rpmbuild/SPECS && yum-builddep -y fio.spec
|
RUN cd ~/rpmbuild/SPECS && yum-builddep -y fio.spec
|
||||||
RUN yum -y install cmake3
|
RUN yum -y install cmake3
|
||||||
|
|
||||||
ADD https://vitastor.io/rpms/liburing-el7/liburing-0.7-2.el7.src.rpm /root
|
|
||||||
|
|
||||||
RUN set -e; \
|
|
||||||
rpm -i liburing*.src.rpm; \
|
|
||||||
cd ~/rpmbuild/SPECS/; \
|
|
||||||
. /opt/rh/devtoolset-9/enable; \
|
|
||||||
rpmbuild -ba liburing.spec; \
|
|
||||||
mkdir -p /root/packages/liburing-el7; \
|
|
||||||
rm -rf /root/packages/liburing-el7/*; \
|
|
||||||
cp ~/rpmbuild/RPMS/*/liburing* /root/packages/liburing-el7/; \
|
|
||||||
cp ~/rpmbuild/SRPMS/liburing* /root/packages/liburing-el7/
|
|
||||||
|
|
||||||
RUN rpm -i `ls /root/packages/liburing-el7/liburing-*.x86_64.rpm | grep -v debug`
|
|
||||||
|
|||||||
@@ -1,13 +1,12 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 2.2.3
|
Version: 3.0.2
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-2.2.3.el7.tar.gz
|
Source0: vitastor-3.0.2.el7.tar.gz
|
||||||
|
|
||||||
BuildRequires: liburing-devel >= 0.6
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: devtoolset-9-gcc-c++
|
BuildRequires: devtoolset-9-gcc-c++
|
||||||
BuildRequires: rh-nodejs12
|
BuildRequires: rh-nodejs12
|
||||||
@@ -35,8 +34,6 @@ size with configurable redundancy (replication or erasure codes/XOR).
|
|||||||
Summary: Vitastor - OSD
|
Summary: Vitastor - OSD
|
||||||
Requires: libJerasure2
|
Requires: libJerasure2
|
||||||
Requires: libisa-l
|
Requires: libisa-l
|
||||||
Requires: liburing >= 0.6
|
|
||||||
Requires: liburing < 2
|
|
||||||
Requires: vitastor-client = %{version}-%{release}
|
Requires: vitastor-client = %{version}-%{release}
|
||||||
Requires: util-linux
|
Requires: util-linux
|
||||||
Requires: parted
|
Requires: parted
|
||||||
@@ -60,8 +57,6 @@ scheduling cluster-level operations.
|
|||||||
|
|
||||||
%package -n vitastor-client
|
%package -n vitastor-client
|
||||||
Summary: Vitastor - client
|
Summary: Vitastor - client
|
||||||
Requires: liburing >= 0.6
|
|
||||||
Requires: liburing < 2
|
|
||||||
|
|
||||||
|
|
||||||
%description -n vitastor-client
|
%description -n vitastor-client
|
||||||
@@ -169,13 +164,13 @@ chown vitastor:vitastor /var/lib/vitastor
|
|||||||
|
|
||||||
%files -n vitastor-client
|
%files -n vitastor-client
|
||||||
%_bindir/vitastor-nbd
|
%_bindir/vitastor-nbd
|
||||||
|
%_bindir/vitastor-ublk
|
||||||
%_bindir/vitastor-nfs
|
%_bindir/vitastor-nfs
|
||||||
%_bindir/vitastor-cli
|
%_bindir/vitastor-cli
|
||||||
%_bindir/vitastor-rm
|
%_bindir/vitastor-rm
|
||||||
%_bindir/vitastor-kv
|
%_bindir/vitastor-kv
|
||||||
%_bindir/vitastor-kv-stress
|
%_bindir/vitastor-kv-stress
|
||||||
%_bindir/vita
|
%_bindir/vita
|
||||||
%_libdir/libvitastor_blk.so*
|
|
||||||
%_libdir/libvitastor_client.so*
|
%_libdir/libvitastor_client.so*
|
||||||
%_libdir/libvitastor_kv.so*
|
%_libdir/libvitastor_kv.so*
|
||||||
|
|
||||||
|
|||||||
@@ -17,17 +17,3 @@ RUN dnf -y install gcc-toolset-9 gcc-toolset-9-gcc-c++ gperftools-devel \
|
|||||||
RUN dnf download --source fio
|
RUN dnf download --source fio
|
||||||
RUN rpm --nomd5 -i fio*.src.rpm
|
RUN rpm --nomd5 -i fio*.src.rpm
|
||||||
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --enablerepo=powertools --spec fio.spec
|
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --enablerepo=powertools --spec fio.spec
|
||||||
|
|
||||||
ADD https://vitastor.io/rpms/liburing-el7/liburing-0.7-2.el7.src.rpm /root
|
|
||||||
|
|
||||||
RUN set -e; \
|
|
||||||
rpm -i liburing*.src.rpm; \
|
|
||||||
cd ~/rpmbuild/SPECS/; \
|
|
||||||
. /opt/rh/gcc-toolset-9/enable; \
|
|
||||||
rpmbuild -ba liburing.spec; \
|
|
||||||
mkdir -p /root/packages/liburing-el8; \
|
|
||||||
rm -rf /root/packages/liburing-el8/*; \
|
|
||||||
cp ~/rpmbuild/RPMS/*/liburing* /root/packages/liburing-el8/; \
|
|
||||||
cp ~/rpmbuild/SRPMS/liburing* /root/packages/liburing-el8/
|
|
||||||
|
|
||||||
RUN rpm -i `ls /root/packages/liburing-el8/liburing-*.x86_64.rpm | grep -v debug`
|
|
||||||
|
|||||||
@@ -1,13 +1,12 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 2.2.3
|
Version: 3.0.2
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-2.2.3.el8.tar.gz
|
Source0: vitastor-3.0.2.el8.tar.gz
|
||||||
|
|
||||||
BuildRequires: liburing-devel >= 0.6
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: gcc-toolset-9-gcc-c++
|
BuildRequires: gcc-toolset-9-gcc-c++
|
||||||
BuildRequires: nodejs >= 10
|
BuildRequires: nodejs >= 10
|
||||||
@@ -34,8 +33,6 @@ size with configurable redundancy (replication or erasure codes/XOR).
|
|||||||
Summary: Vitastor - OSD
|
Summary: Vitastor - OSD
|
||||||
Requires: libJerasure2
|
Requires: libJerasure2
|
||||||
Requires: libisa-l
|
Requires: libisa-l
|
||||||
Requires: liburing >= 0.6
|
|
||||||
Requires: liburing < 2
|
|
||||||
Requires: vitastor-client = %{version}-%{release}
|
Requires: vitastor-client = %{version}-%{release}
|
||||||
Requires: util-linux
|
Requires: util-linux
|
||||||
Requires: parted
|
Requires: parted
|
||||||
@@ -58,8 +55,6 @@ scheduling cluster-level operations.
|
|||||||
|
|
||||||
%package -n vitastor-client
|
%package -n vitastor-client
|
||||||
Summary: Vitastor - client
|
Summary: Vitastor - client
|
||||||
Requires: liburing >= 0.6
|
|
||||||
Requires: liburing < 2
|
|
||||||
|
|
||||||
|
|
||||||
%description -n vitastor-client
|
%description -n vitastor-client
|
||||||
@@ -166,13 +161,13 @@ chown vitastor:vitastor /var/lib/vitastor
|
|||||||
|
|
||||||
%files -n vitastor-client
|
%files -n vitastor-client
|
||||||
%_bindir/vitastor-nbd
|
%_bindir/vitastor-nbd
|
||||||
|
%_bindir/vitastor-ublk
|
||||||
%_bindir/vitastor-nfs
|
%_bindir/vitastor-nfs
|
||||||
%_bindir/vitastor-cli
|
%_bindir/vitastor-cli
|
||||||
%_bindir/vitastor-rm
|
%_bindir/vitastor-rm
|
||||||
%_bindir/vitastor-kv
|
%_bindir/vitastor-kv
|
||||||
%_bindir/vitastor-kv-stress
|
%_bindir/vitastor-kv-stress
|
||||||
%_bindir/vita
|
%_bindir/vita
|
||||||
%_libdir/libvitastor_blk.so*
|
|
||||||
%_libdir/libvitastor_client.so*
|
%_libdir/libvitastor_client.so*
|
||||||
%_libdir/libvitastor_kv.so*
|
%_libdir/libvitastor_kv.so*
|
||||||
|
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ WORKDIR /root
|
|||||||
RUN sed -i 's/enabled=0/enabled=1/' /etc/yum.repos.d/*.repo
|
RUN sed -i 's/enabled=0/enabled=1/' /etc/yum.repos.d/*.repo
|
||||||
RUN dnf -y install epel-release dnf-plugins-core
|
RUN dnf -y install epel-release dnf-plugins-core
|
||||||
RUN dnf -y install https://vitastor.io/rpms/centos/9/vitastor-release-1.0-1.el9.noarch.rpm
|
RUN dnf -y install https://vitastor.io/rpms/centos/9/vitastor-release-1.0-1.el9.noarch.rpm
|
||||||
RUN dnf -y install gcc-c++ gperftools-devel fio nodejs rpm-build jerasure-devel libisa-l-devel gf-complete-devel rdma-core-devel libarchive liburing-devel cmake libnl3-devel
|
RUN dnf -y install gcc-c++ gperftools-devel fio nodejs rpm-build jerasure-devel libisa-l-devel gf-complete-devel rdma-core-devel libarchive cmake libnl3-devel
|
||||||
RUN dnf download --source fio
|
RUN dnf download --source fio
|
||||||
RUN rpm --nomd5 -i fio*.src.rpm
|
RUN rpm --nomd5 -i fio*.src.rpm
|
||||||
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --spec fio.spec
|
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --spec fio.spec
|
||||||
|
|||||||
@@ -1,13 +1,12 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 2.2.3
|
Version: 3.0.2
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-2.2.3.el9.tar.gz
|
Source0: vitastor-3.0.2.el9.tar.gz
|
||||||
|
|
||||||
BuildRequires: liburing-devel >= 0.6
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: gcc-c++
|
BuildRequires: gcc-c++
|
||||||
BuildRequires: nodejs >= 10
|
BuildRequires: nodejs >= 10
|
||||||
@@ -159,13 +158,13 @@ chown vitastor:vitastor /var/lib/vitastor
|
|||||||
|
|
||||||
%files -n vitastor-client
|
%files -n vitastor-client
|
||||||
%_bindir/vitastor-nbd
|
%_bindir/vitastor-nbd
|
||||||
|
%_bindir/vitastor-ublk
|
||||||
%_bindir/vitastor-nfs
|
%_bindir/vitastor-nfs
|
||||||
%_bindir/vitastor-cli
|
%_bindir/vitastor-cli
|
||||||
%_bindir/vitastor-rm
|
%_bindir/vitastor-rm
|
||||||
%_bindir/vitastor-kv
|
%_bindir/vitastor-kv
|
||||||
%_bindir/vitastor-kv-stress
|
%_bindir/vitastor-kv-stress
|
||||||
%_bindir/vita
|
%_bindir/vita
|
||||||
%_libdir/libvitastor_blk.so*
|
|
||||||
%_libdir/libvitastor_client.so*
|
%_libdir/libvitastor_client.so*
|
||||||
%_libdir/libvitastor_kv.so*
|
%_libdir/libvitastor_kv.so*
|
||||||
|
|
||||||
|
|||||||
+22
-13
@@ -12,20 +12,30 @@ set(WITH_QEMU false CACHE BOOL "Build QEMU driver inside Vitastor source tree")
|
|||||||
set(WITH_FIO true CACHE BOOL "Build FIO driver")
|
set(WITH_FIO true CACHE BOOL "Build FIO driver")
|
||||||
set(QEMU_PLUGINDIR qemu CACHE STRING "QEMU plugin directory suffix (qemu-kvm on RHEL)")
|
set(QEMU_PLUGINDIR qemu CACHE STRING "QEMU plugin directory suffix (qemu-kvm on RHEL)")
|
||||||
set(WITH_ASAN false CACHE BOOL "Build with AddressSanitizer")
|
set(WITH_ASAN false CACHE BOOL "Build with AddressSanitizer")
|
||||||
|
set(WITH_SYSTEM_LIBURING false CACHE BOOL "Use system liburing")
|
||||||
if("${CMAKE_INSTALL_PREFIX}" MATCHES "^/usr/local/?$")
|
if("${CMAKE_INSTALL_PREFIX}" MATCHES "^/usr/local/?$")
|
||||||
if(EXISTS "/etc/debian_version")
|
if(EXISTS "/etc/debian_version")
|
||||||
set(CMAKE_INSTALL_LIBDIR "lib/${CMAKE_LIBRARY_ARCHITECTURE}")
|
set(CMAKE_INSTALL_LIBDIR "lib/${CMAKE_LIBRARY_ARCHITECTURE}")
|
||||||
endif()
|
endif()
|
||||||
set(CMAKE_INSTALL_RPATH "${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}")
|
set(CMAKE_INSTALL_RPATH "${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}")
|
||||||
endif()
|
endif()
|
||||||
|
set(ENABLE_COVERAGE false CACHE BOOL "Enable code coverage")
|
||||||
|
|
||||||
add_definitions(-DVITASTOR_VERSION="2.2.3")
|
add_definitions(-DVITASTOR_VERSION="3.0.2")
|
||||||
add_definitions(-D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -I ${CMAKE_SOURCE_DIR}/src)
|
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
|
||||||
add_link_options(-fno-omit-frame-pointer)
|
add_link_options(-fno-omit-frame-pointer)
|
||||||
if (${WITH_ASAN})
|
if (${WITH_ASAN})
|
||||||
add_definitions(-fsanitize=address)
|
add_definitions(-fsanitize=address)
|
||||||
add_link_options(-fsanitize=address -fno-omit-frame-pointer)
|
add_link_options(-fsanitize=address -fno-omit-frame-pointer)
|
||||||
endif (${WITH_ASAN})
|
endif (${WITH_ASAN})
|
||||||
|
set(CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE} -fvisibility-inlines-hidden")
|
||||||
|
set(CMAKE_CXX_FLAGS_MINSIZEREL "${CMAKE_CXX_FLAGS_MINSIZEREL} -fvisibility-inlines-hidden")
|
||||||
|
set(CMAKE_CXX_FLAGS_RELWITHDEBINFO "${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fvisibility-inlines-hidden")
|
||||||
|
|
||||||
|
if (${ENABLE_COVERAGE})
|
||||||
|
add_definitions(-coverage)
|
||||||
|
add_link_options(-coverage)
|
||||||
|
endif()
|
||||||
|
|
||||||
set(CMAKE_BUILD_TYPE RelWithDebInfo)
|
set(CMAKE_BUILD_TYPE RelWithDebInfo)
|
||||||
string(REGEX REPLACE "([\\/\\-]O)[^ \t\r\n]*" "\\13" CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE}")
|
string(REGEX REPLACE "([\\/\\-]O)[^ \t\r\n]*" "\\13" CMAKE_CXX_FLAGS_RELEASE "${CMAKE_CXX_FLAGS_RELEASE}")
|
||||||
@@ -49,7 +59,6 @@ endmacro(install_symlink)
|
|||||||
check_include_file("linux/nbd-netlink.h" HAVE_NBD_NETLINK_H)
|
check_include_file("linux/nbd-netlink.h" HAVE_NBD_NETLINK_H)
|
||||||
|
|
||||||
find_package(PkgConfig)
|
find_package(PkgConfig)
|
||||||
pkg_check_modules(LIBURING REQUIRED liburing)
|
|
||||||
if (${WITH_QEMU})
|
if (${WITH_QEMU})
|
||||||
pkg_check_modules(GLIB REQUIRED glib-2.0)
|
pkg_check_modules(GLIB REQUIRED glib-2.0)
|
||||||
endif (${WITH_QEMU})
|
endif (${WITH_QEMU})
|
||||||
@@ -66,13 +75,14 @@ if (RDMACM_LIBRARIES)
|
|||||||
add_definitions(-DWITH_RDMACM)
|
add_definitions(-DWITH_RDMACM)
|
||||||
endif (RDMACM_LIBRARIES)
|
endif (RDMACM_LIBRARIES)
|
||||||
|
|
||||||
add_custom_target(build_tests)
|
if (${WITH_SYSTEM_LIBURING})
|
||||||
add_custom_target(test
|
pkg_check_modules(LIBURING REQUIRED liburing>=2.10)
|
||||||
COMMAND
|
include_directories(${LIBURING_INCLUDE_DIRS})
|
||||||
echo leak:tcmalloc > ${CMAKE_CURRENT_BINARY_DIR}/lsan-suppress.txt &&
|
else()
|
||||||
env LSAN_OPTIONS=suppressions=${CMAKE_CURRENT_BINARY_DIR}/lsan-suppress.txt ${CMAKE_CTEST_COMMAND}
|
include_directories(${CMAKE_SOURCE_DIR}/src/liburing/include)
|
||||||
)
|
add_subdirectory(liburing)
|
||||||
add_dependencies(test build_tests)
|
set(LIBURING_LIBRARIES uring)
|
||||||
|
endif (${WITH_SYSTEM_LIBURING})
|
||||||
|
|
||||||
include_directories(
|
include_directories(
|
||||||
../
|
../
|
||||||
@@ -86,7 +96,6 @@ include_directories(
|
|||||||
${CMAKE_SOURCE_DIR}/src/test
|
${CMAKE_SOURCE_DIR}/src/test
|
||||||
${CMAKE_SOURCE_DIR}/src/util
|
${CMAKE_SOURCE_DIR}/src/util
|
||||||
/usr/include/jerasure
|
/usr/include/jerasure
|
||||||
${LIBURING_INCLUDE_DIRS}
|
|
||||||
${IBVERBS_INCLUDE_DIRS}
|
${IBVERBS_INCLUDE_DIRS}
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -101,12 +110,12 @@ add_subdirectory(test)
|
|||||||
|
|
||||||
### Install
|
### Install
|
||||||
|
|
||||||
install(TARGETS vitastor-osd vitastor-disk vitastor-nbd vitastor-nfs vitastor-cli vitastor-kv vitastor-kv-stress RUNTIME DESTINATION ${CMAKE_INSTALL_BINDIR})
|
install(TARGETS vitastor-osd vitastor-disk vitastor-nbd vitastor-ublk vitastor-nfs vitastor-cli vitastor-kv vitastor-kv-stress RUNTIME DESTINATION ${CMAKE_INSTALL_BINDIR})
|
||||||
install_symlink(vitastor-disk ${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_BINDIR}/vitastor-dump-journal)
|
install_symlink(vitastor-disk ${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_BINDIR}/vitastor-dump-journal)
|
||||||
install_symlink(vitastor-cli ${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_BINDIR}/vitastor-rm)
|
install_symlink(vitastor-cli ${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_BINDIR}/vitastor-rm)
|
||||||
install_symlink(vitastor-cli ${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_BINDIR}/vita)
|
install_symlink(vitastor-cli ${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_BINDIR}/vita)
|
||||||
install(
|
install(
|
||||||
TARGETS vitastor_blk vitastor_client vitastor_kv
|
TARGETS vitastor_client vitastor_kv
|
||||||
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR}
|
||||||
PUBLIC_HEADER DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}
|
PUBLIC_HEADER DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -2,14 +2,18 @@ cmake_minimum_required(VERSION 2.8.12)
|
|||||||
|
|
||||||
project(vitastor)
|
project(vitastor)
|
||||||
|
|
||||||
# libvitastor_blk.so
|
# libvitastor_blk.a
|
||||||
add_library(vitastor_blk SHARED
|
add_library(vitastor_blk STATIC
|
||||||
../util/allocator.cpp blockstore.cpp blockstore_impl.cpp blockstore_disk.cpp blockstore_init.cpp blockstore_open.cpp blockstore_journal.cpp blockstore_read.cpp
|
../util/allocator.cpp ../util/crc32c.c ../util/ringloop.cpp
|
||||||
blockstore_write.cpp blockstore_sync.cpp blockstore_stable.cpp blockstore_rollback.cpp blockstore_flush.cpp ../util/crc32c.c ../util/ringloop.cpp
|
multilist.cpp blockstore_heap.cpp blockstore_disk.cpp
|
||||||
|
blockstore.cpp blockstore_impl.cpp blockstore_init.cpp blockstore_open.cpp
|
||||||
|
blockstore_flush.cpp blockstore_read.cpp blockstore_stable.cpp blockstore_sync.cpp blockstore_write.cpp
|
||||||
|
v1/flush.cpp v1/impl.cpp v1/init.cpp v1/journal.cpp v1/open.cpp v1/read.cpp v1/rollback.cpp v1/stable.cpp v1/sync.cpp v1/write.cpp
|
||||||
)
|
)
|
||||||
|
target_compile_options(vitastor_blk PUBLIC -fPIC)
|
||||||
target_link_libraries(vitastor_blk
|
target_link_libraries(vitastor_blk
|
||||||
${LIBURING_LIBRARIES}
|
${LIBURING_LIBRARIES}
|
||||||
tcmalloc_minimal
|
${ISAL_LIBRARIES}
|
||||||
# for timerfd_manager
|
# for timerfd_manager
|
||||||
vitastor_common
|
vitastor_common
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -1,89 +1,16 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.1 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
|
#include "str_util.h"
|
||||||
|
|
||||||
#include "blockstore_impl.h"
|
#include "blockstore_impl.h"
|
||||||
|
#include "v1/impl.h"
|
||||||
|
|
||||||
blockstore_t::blockstore_t(blockstore_config_t & config, ring_loop_t *ringloop, timerfd_manager_t *tfd)
|
blockstore_i* blockstore_i::create(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd)
|
||||||
{
|
{
|
||||||
impl = new blockstore_impl_t(config, ringloop, tfd);
|
auto meta_format = stoull_full(config["meta_format"]);
|
||||||
}
|
if (meta_format == BLOCKSTORE_META_FORMAT_HEAP)
|
||||||
|
return new blockstore_impl_t(config, ringloop, tfd);
|
||||||
blockstore_t::~blockstore_t()
|
else
|
||||||
{
|
return new v1::blockstore_impl_t(config, ringloop, tfd);
|
||||||
delete impl;
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_t::parse_config(blockstore_config_t & config)
|
|
||||||
{
|
|
||||||
impl->parse_config(config, false);
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_t::loop()
|
|
||||||
{
|
|
||||||
impl->loop();
|
|
||||||
}
|
|
||||||
|
|
||||||
bool blockstore_t::is_started()
|
|
||||||
{
|
|
||||||
return impl->is_started();
|
|
||||||
}
|
|
||||||
|
|
||||||
bool blockstore_t::is_stalled()
|
|
||||||
{
|
|
||||||
return impl->is_stalled();
|
|
||||||
}
|
|
||||||
|
|
||||||
bool blockstore_t::is_safe_to_stop()
|
|
||||||
{
|
|
||||||
return impl->is_safe_to_stop();
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_t::enqueue_op(blockstore_op_t *op)
|
|
||||||
{
|
|
||||||
impl->enqueue_op(op);
|
|
||||||
}
|
|
||||||
|
|
||||||
int blockstore_t::read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version)
|
|
||||||
{
|
|
||||||
return impl->read_bitmap(oid, target_version, bitmap, result_version);
|
|
||||||
}
|
|
||||||
|
|
||||||
std::map<uint64_t, uint64_t> & blockstore_t::get_inode_space_stats()
|
|
||||||
{
|
|
||||||
return impl->inode_space_stats;
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_t::dump_diagnostics()
|
|
||||||
{
|
|
||||||
return impl->dump_diagnostics();
|
|
||||||
}
|
|
||||||
|
|
||||||
uint32_t blockstore_t::get_block_size()
|
|
||||||
{
|
|
||||||
return impl->get_block_size();
|
|
||||||
}
|
|
||||||
|
|
||||||
uint64_t blockstore_t::get_block_count()
|
|
||||||
{
|
|
||||||
return impl->get_block_count();
|
|
||||||
}
|
|
||||||
|
|
||||||
uint64_t blockstore_t::get_free_block_count()
|
|
||||||
{
|
|
||||||
return impl->get_free_block_count();
|
|
||||||
}
|
|
||||||
|
|
||||||
uint64_t blockstore_t::get_journal_size()
|
|
||||||
{
|
|
||||||
return impl->get_journal_size();
|
|
||||||
}
|
|
||||||
|
|
||||||
uint32_t blockstore_t::get_bitmap_granularity()
|
|
||||||
{
|
|
||||||
return impl->get_bitmap_granularity();
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_t::set_no_inode_stats(const std::vector<uint64_t> & pool_ids)
|
|
||||||
{
|
|
||||||
impl->set_no_inode_stats(pool_ids);
|
|
||||||
}
|
}
|
||||||
|
|||||||
+43
-34
@@ -17,22 +17,14 @@
|
|||||||
#include "ringloop.h"
|
#include "ringloop.h"
|
||||||
#include "timerfd_manager.h"
|
#include "timerfd_manager.h"
|
||||||
|
|
||||||
// Memory alignment for direct I/O (usually 512 bytes)
|
|
||||||
#ifndef DIRECT_IO_ALIGNMENT
|
|
||||||
#define DIRECT_IO_ALIGNMENT 512
|
|
||||||
#endif
|
|
||||||
|
|
||||||
// Memory allocation alignment (page size is usually optimal)
|
|
||||||
#ifndef MEM_ALIGNMENT
|
|
||||||
#define MEM_ALIGNMENT 4096
|
|
||||||
#endif
|
|
||||||
|
|
||||||
// Default block size is 128 KB, current allowed range is 4K - 128M
|
// Default block size is 128 KB, current allowed range is 4K - 128M
|
||||||
#define DEFAULT_DATA_BLOCK_ORDER 17
|
#define DEFAULT_DATA_BLOCK_ORDER 17
|
||||||
#define MIN_DATA_BLOCK_SIZE 4*1024
|
#define MIN_DATA_BLOCK_SIZE 4*1024
|
||||||
#define MAX_DATA_BLOCK_SIZE 128*1024*1024
|
#define MAX_DATA_BLOCK_SIZE 128*1024*1024
|
||||||
#define DEFAULT_BITMAP_GRANULARITY 4096
|
#define DEFAULT_BITMAP_GRANULARITY 4096
|
||||||
|
|
||||||
|
#define MIN_JOURNAL_SIZE 1024*1024
|
||||||
|
|
||||||
#define BS_OP_MIN 1
|
#define BS_OP_MIN 1
|
||||||
#define BS_OP_READ 1
|
#define BS_OP_READ 1
|
||||||
#define BS_OP_WRITE 2
|
#define BS_OP_WRITE 2
|
||||||
@@ -46,8 +38,18 @@
|
|||||||
|
|
||||||
#define BS_OP_PRIVATE_DATA_SIZE 256
|
#define BS_OP_PRIVATE_DATA_SIZE 256
|
||||||
|
|
||||||
|
#define IMMEDIATE_NONE 0
|
||||||
|
#define IMMEDIATE_SMALL 1
|
||||||
|
#define IMMEDIATE_ALL 2
|
||||||
|
|
||||||
/*
|
/*
|
||||||
|
|
||||||
|
All operations may be submitted in any order, because reads only see completed writes,
|
||||||
|
syncs only sync completed writes and writes don't depend on each other.
|
||||||
|
|
||||||
|
The only restriction is that the external code MUST NOT submit multiple writes for one
|
||||||
|
object in parallel. This is a natural restriction because `version` numbers are used though.
|
||||||
|
|
||||||
Blockstore opcode documentation:
|
Blockstore opcode documentation:
|
||||||
|
|
||||||
## BS_OP_READ / BS_OP_WRITE / BS_OP_WRITE_STABLE
|
## BS_OP_READ / BS_OP_WRITE / BS_OP_WRITE_STABLE
|
||||||
@@ -135,7 +137,7 @@ Output:
|
|||||||
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
struct blockstore_op_t
|
struct __attribute__ ((visibility("default"))) blockstore_op_t
|
||||||
{
|
{
|
||||||
// operation
|
// operation
|
||||||
uint64_t opcode = 0;
|
uint64_t opcode = 0;
|
||||||
@@ -162,8 +164,8 @@ struct blockstore_op_t
|
|||||||
uint32_t list_stable_limit;
|
uint32_t list_stable_limit;
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
void *buf = NULL;
|
uint8_t *buf = NULL;
|
||||||
void *bitmap = NULL;
|
uint8_t *bitmap = NULL;
|
||||||
int retval = 0;
|
int retval = 0;
|
||||||
|
|
||||||
uint8_t private_data[BS_OP_PRIVATE_DATA_SIZE];
|
uint8_t private_data[BS_OP_PRIVATE_DATA_SIZE];
|
||||||
@@ -171,53 +173,60 @@ struct blockstore_op_t
|
|||||||
|
|
||||||
typedef std::map<std::string, std::string> blockstore_config_t;
|
typedef std::map<std::string, std::string> blockstore_config_t;
|
||||||
|
|
||||||
class blockstore_impl_t;
|
class __attribute__((visibility("default"))) blockstore_i
|
||||||
|
|
||||||
class blockstore_t
|
|
||||||
{
|
{
|
||||||
blockstore_impl_t *impl;
|
|
||||||
public:
|
public:
|
||||||
blockstore_t(blockstore_config_t & config, ring_loop_t *ringloop, timerfd_manager_t *tfd);
|
static blockstore_i* create(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd);
|
||||||
~blockstore_t();
|
|
||||||
|
virtual ~blockstore_i() = default;
|
||||||
|
|
||||||
// Update configuration
|
// Update configuration
|
||||||
void parse_config(blockstore_config_t & config);
|
virtual void parse_config(blockstore_config_t & config) = 0;
|
||||||
|
|
||||||
|
// Reshard database for a pool in chunks
|
||||||
|
// MUST be called only when nobody makes any modifications to the DB for this pool
|
||||||
|
virtual void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit) = 0;
|
||||||
|
virtual bool reshard_continue(void *reshard_state, uint64_t chunk_limit) = 0;
|
||||||
|
virtual void reshard_abort(void *reshard_state) = 0;
|
||||||
|
|
||||||
// Event loop
|
// Event loop
|
||||||
void loop();
|
virtual void loop() = 0;
|
||||||
|
|
||||||
// Returns true when blockstore is ready to process operations
|
// Returns true when blockstore is ready to process operations
|
||||||
// (Although you're free to enqueue them before that)
|
// (Although you're free to enqueue them before that)
|
||||||
bool is_started();
|
virtual bool is_started() = 0;
|
||||||
|
|
||||||
// Returns true when blockstore is stalled
|
// Returns true when blockstore is stalled
|
||||||
bool is_stalled();
|
virtual bool is_stalled() = 0;
|
||||||
|
|
||||||
// Returns true when it's safe to destroy the instance. If destroying the instance
|
// Returns true when it's safe to destroy the instance. If destroying the instance
|
||||||
// requires to purge some queues, starts that process. Should be called in the event
|
// requires to purge some queues, starts that process. Should be called in the event
|
||||||
// loop until it returns true.
|
// loop until it returns true.
|
||||||
bool is_safe_to_stop();
|
virtual bool is_safe_to_stop() = 0;
|
||||||
|
|
||||||
// Submission
|
// Submission
|
||||||
void enqueue_op(blockstore_op_t *op);
|
virtual void enqueue_op(blockstore_op_t *op) = 0;
|
||||||
|
|
||||||
// Simplified synchronous operation: get object bitmap & current version
|
// Simplified synchronous operation: get object bitmap & current version
|
||||||
int read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version = NULL);
|
virtual int read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version = NULL) = 0;
|
||||||
|
|
||||||
// Get per-inode space usage statistics
|
// Get per-inode space usage statistics
|
||||||
std::map<uint64_t, uint64_t> & get_inode_space_stats();
|
virtual const std::map<uint64_t, uint64_t> & get_inode_space_stats() = 0;
|
||||||
|
|
||||||
// Set per-pool no_inode_stats
|
// Set per-pool no_inode_stats
|
||||||
void set_no_inode_stats(const std::vector<uint64_t> & pool_ids);
|
virtual void set_no_inode_stats(const std::vector<uint64_t> & pool_ids) = 0;
|
||||||
|
|
||||||
// Print diagnostics to stdout
|
// Print diagnostics to stdout
|
||||||
void dump_diagnostics();
|
virtual void dump_diagnostics() = 0;
|
||||||
|
|
||||||
uint32_t get_block_size();
|
// Get diagnostic string for an operation
|
||||||
uint64_t get_block_count();
|
virtual std::string get_op_diag(blockstore_op_t *op) = 0;
|
||||||
uint64_t get_free_block_count();
|
|
||||||
|
|
||||||
uint64_t get_journal_size();
|
virtual uint32_t get_block_size() = 0;
|
||||||
|
virtual uint64_t get_block_count() = 0;
|
||||||
|
virtual uint64_t get_free_block_count() = 0;
|
||||||
|
|
||||||
uint32_t get_bitmap_granularity();
|
virtual uint64_t get_journal_size() = 0;
|
||||||
|
|
||||||
|
virtual uint32_t get_bitmap_granularity() = 0;
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -2,11 +2,15 @@
|
|||||||
// License: VNPL-1.1 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
#include <sys/file.h>
|
#include <sys/file.h>
|
||||||
|
#include <sys/ioctl.h>
|
||||||
|
#include <unistd.h>
|
||||||
|
|
||||||
#include <stdexcept>
|
#include <stdexcept>
|
||||||
|
|
||||||
#include "blockstore_impl.h"
|
#include "blockstore.h"
|
||||||
|
#include "ondisk_formats.h"
|
||||||
#include "blockstore_disk.h"
|
#include "blockstore_disk.h"
|
||||||
|
#include "blockstore_heap.h"
|
||||||
#include "str_util.h"
|
#include "str_util.h"
|
||||||
#include "allocator.h"
|
#include "allocator.h"
|
||||||
|
|
||||||
@@ -46,6 +50,10 @@ void blockstore_disk_t::parse_config(std::map<std::string, std::string> & config
|
|||||||
meta_block_size = parse_size(config["meta_block_size"]);
|
meta_block_size = parse_size(config["meta_block_size"]);
|
||||||
bitmap_granularity = parse_size(config["bitmap_granularity"]);
|
bitmap_granularity = parse_size(config["bitmap_granularity"]);
|
||||||
meta_format = stoull_full(config["meta_format"]);
|
meta_format = stoull_full(config["meta_format"]);
|
||||||
|
atomic_write_size = (config.find("atomic_write_size") != config.end()
|
||||||
|
? parse_size(config["atomic_write_size"]) : 4096);
|
||||||
|
use_atomic_flag = config.find("use_atomic_flag") != config.end() &&
|
||||||
|
(config["use_atomic_flag"] == "true" || config["use_atomic_flag"] == "1" || config["use_atomic_flag"] == "yes");
|
||||||
if (config.find("data_io") == config.end() &&
|
if (config.find("data_io") == config.end() &&
|
||||||
config.find("meta_io") == config.end() &&
|
config.find("meta_io") == config.end() &&
|
||||||
config.find("journal_io") == config.end())
|
config.find("journal_io") == config.end())
|
||||||
@@ -90,12 +98,28 @@ void blockstore_disk_t::parse_config(std::map<std::string, std::string> & config
|
|||||||
if (!min_discard_size)
|
if (!min_discard_size)
|
||||||
min_discard_size = 1024*1024;
|
min_discard_size = 1024*1024;
|
||||||
discard_granularity = parse_size(config["discard_granularity"]);
|
discard_granularity = parse_size(config["discard_granularity"]);
|
||||||
|
inmemory_meta = config["inmemory_metadata"] != "false" && config["inmemory_metadata"] != "0" &&
|
||||||
|
config["inmemory_metadata"] != "no";
|
||||||
|
inmemory_journal = config["inmemory_journal"] != "false" && config["inmemory_journal"] != "0" &&
|
||||||
|
config["inmemory_journal"] != "no";
|
||||||
|
disable_data_fsync = config["disable_data_fsync"] == "true" || config["disable_data_fsync"] == "1" || config["disable_data_fsync"] == "yes";
|
||||||
|
disable_meta_fsync = config["disable_meta_fsync"] == "true" || config["disable_meta_fsync"] == "1" || config["disable_meta_fsync"] == "yes";
|
||||||
|
disable_journal_fsync = config["disable_journal_fsync"] == "true" || config["disable_journal_fsync"] == "1" || config["disable_journal_fsync"] == "yes";
|
||||||
|
if (mock_mode)
|
||||||
|
{
|
||||||
|
data_device_size = parse_size(config["data_device_size"]);
|
||||||
|
data_device_sect = parse_size(config["data_device_sect"]);
|
||||||
|
meta_device_size = parse_size(config["meta_device_size"]);
|
||||||
|
meta_device_sect = parse_size(config["meta_device_sect"]);
|
||||||
|
journal_device_size = parse_size(config["journal_device_size"]);
|
||||||
|
journal_device_sect = parse_size(config["journal_device_sect"]);
|
||||||
|
}
|
||||||
// Validate
|
// Validate
|
||||||
if (!data_block_size)
|
if (!data_block_size)
|
||||||
{
|
{
|
||||||
data_block_size = (1 << DEFAULT_DATA_BLOCK_ORDER);
|
data_block_size = (1 << DEFAULT_DATA_BLOCK_ORDER);
|
||||||
}
|
}
|
||||||
if ((block_order = is_power_of_two(data_block_size)) >= 64 || data_block_size < MIN_DATA_BLOCK_SIZE || data_block_size >= MAX_DATA_BLOCK_SIZE)
|
if (is_power_of_two(data_block_size) >= 64 || data_block_size < MIN_DATA_BLOCK_SIZE || data_block_size >= MAX_DATA_BLOCK_SIZE)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("Bad block size");
|
throw std::runtime_error("Bad block size");
|
||||||
}
|
}
|
||||||
@@ -179,17 +203,25 @@ void blockstore_disk_t::parse_config(std::map<std::string, std::string> & config
|
|||||||
{
|
{
|
||||||
throw std::runtime_error("journal_offset must be a multiple of journal_block_size = "+std::to_string(journal_block_size));
|
throw std::runtime_error("journal_offset must be a multiple of journal_block_size = "+std::to_string(journal_block_size));
|
||||||
}
|
}
|
||||||
|
if (meta_device == data_device)
|
||||||
|
{
|
||||||
|
disable_meta_fsync = disable_data_fsync;
|
||||||
|
}
|
||||||
|
if (journal_device == meta_device)
|
||||||
|
{
|
||||||
|
disable_journal_fsync = disable_meta_fsync;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void blockstore_disk_t::calc_lengths(bool skip_meta_check)
|
void blockstore_disk_t::calc_lengths(bool skip_meta_check)
|
||||||
{
|
{
|
||||||
// data
|
// data
|
||||||
data_len = data_device_size - data_offset;
|
data_len = data_device_size - data_offset;
|
||||||
if (data_fd == meta_fd && data_offset < meta_offset)
|
if (data_device == meta_device && data_offset < meta_offset)
|
||||||
{
|
{
|
||||||
data_len = meta_offset - data_offset;
|
data_len = meta_offset - data_offset;
|
||||||
}
|
}
|
||||||
if (data_fd == journal_fd && data_offset < journal_offset)
|
if (data_device == journal_device && data_offset < journal_offset)
|
||||||
{
|
{
|
||||||
data_len = data_len < journal_offset-data_offset
|
data_len = data_len < journal_offset-data_offset
|
||||||
? data_len : journal_offset-data_offset;
|
? data_len : journal_offset-data_offset;
|
||||||
@@ -204,23 +236,23 @@ void blockstore_disk_t::calc_lengths(bool skip_meta_check)
|
|||||||
data_len = cfg_data_size;
|
data_len = cfg_data_size;
|
||||||
}
|
}
|
||||||
// meta
|
// meta
|
||||||
uint64_t meta_area_size = (meta_fd == data_fd ? data_device_size : meta_device_size) - meta_offset;
|
meta_area_size = (meta_device == data_device ? data_device_size : meta_device_size) - meta_offset;
|
||||||
if (meta_fd == data_fd && meta_offset <= data_offset)
|
if (meta_device == data_device && meta_offset <= data_offset)
|
||||||
{
|
{
|
||||||
meta_area_size = data_offset - meta_offset;
|
meta_area_size = data_offset - meta_offset;
|
||||||
}
|
}
|
||||||
if (meta_fd == journal_fd && meta_offset <= journal_offset)
|
if (meta_device == journal_device && meta_offset <= journal_offset)
|
||||||
{
|
{
|
||||||
meta_area_size = meta_area_size < journal_offset-meta_offset
|
meta_area_size = meta_area_size < journal_offset-meta_offset
|
||||||
? meta_area_size : journal_offset-meta_offset;
|
? meta_area_size : journal_offset-meta_offset;
|
||||||
}
|
}
|
||||||
// journal
|
// journal
|
||||||
journal_len = (journal_fd == data_fd ? data_device_size : (journal_fd == meta_fd ? meta_device_size : journal_device_size)) - journal_offset;
|
journal_len = (journal_device == data_device ? data_device_size : (journal_device == meta_device ? meta_device_size : journal_device_size)) - journal_offset;
|
||||||
if (journal_fd == data_fd && journal_offset <= data_offset)
|
if (journal_device == data_device && journal_offset <= data_offset)
|
||||||
{
|
{
|
||||||
journal_len = data_offset - journal_offset;
|
journal_len = data_offset - journal_offset;
|
||||||
}
|
}
|
||||||
if (journal_fd == meta_fd && journal_offset <= meta_offset)
|
if (journal_device == meta_device && journal_offset <= meta_offset)
|
||||||
{
|
{
|
||||||
journal_len = journal_len < meta_offset-journal_offset
|
journal_len = journal_len < meta_offset-journal_offset
|
||||||
? journal_len : meta_offset-journal_offset;
|
? journal_len : meta_offset-journal_offset;
|
||||||
@@ -230,37 +262,58 @@ void blockstore_disk_t::calc_lengths(bool skip_meta_check)
|
|||||||
clean_entry_bitmap_size = data_block_size / bitmap_granularity / 8;
|
clean_entry_bitmap_size = data_block_size / bitmap_granularity / 8;
|
||||||
clean_dyn_size = clean_entry_bitmap_size*2 + (csum_block_size
|
clean_dyn_size = clean_entry_bitmap_size*2 + (csum_block_size
|
||||||
? data_block_size/csum_block_size*(data_csum_type & 0xFF) : 0);
|
? data_block_size/csum_block_size*(data_csum_type & 0xFF) : 0);
|
||||||
clean_entry_size = sizeof(clean_disk_entry) + clean_dyn_size + 4 /*entry_csum*/;
|
recalc:
|
||||||
meta_len = (1 + (block_count - 1 + meta_block_size / clean_entry_size) / (meta_block_size / clean_entry_size)) * meta_block_size;
|
if (meta_format == BLOCKSTORE_META_FORMAT_HEAP)
|
||||||
bool new_doesnt_fit = (!meta_format && !skip_meta_check && meta_area_size < meta_len && !data_csum_type);
|
|
||||||
if (meta_format == BLOCKSTORE_META_FORMAT_V1 || new_doesnt_fit)
|
|
||||||
{
|
{
|
||||||
uint64_t clean_entry_v0_size = sizeof(clean_disk_entry) + 2*clean_entry_bitmap_size;
|
uint32_t entries_per_block = meta_block_size / (sizeof(heap_big_write_t) + clean_dyn_size);
|
||||||
uint64_t meta_v0_len = (1 + (block_count - 1 + meta_block_size / clean_entry_v0_size)
|
min_meta_len = (block_count+entries_per_block-1) / entries_per_block * meta_block_size;
|
||||||
/ (meta_block_size / clean_entry_v0_size)) * meta_block_size;
|
}
|
||||||
if (meta_format == BLOCKSTORE_META_FORMAT_V1 || meta_area_size >= meta_v0_len)
|
else if (meta_format == BLOCKSTORE_META_FORMAT_V1)
|
||||||
|
{
|
||||||
|
clean_entry_size = 24 /*sizeof(clean_disk_entry)*/ + 2*clean_entry_bitmap_size;
|
||||||
|
min_meta_len = (1 + (block_count - 1 + meta_block_size / clean_entry_size)
|
||||||
|
/ (meta_block_size / clean_entry_size)) * meta_block_size;
|
||||||
|
if (!skip_meta_check && meta_area_size < min_meta_len)
|
||||||
{
|
{
|
||||||
// Old metadata fits.
|
too_small:
|
||||||
if (new_doesnt_fit)
|
throw std::runtime_error("Metadata area is too small, need at least "+std::to_string(min_meta_len)+
|
||||||
|
" bytes, have only "+std::to_string(meta_area_size)+" bytes");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else if (meta_format == BLOCKSTORE_META_FORMAT_V2 || !meta_format)
|
||||||
|
{
|
||||||
|
meta_format = BLOCKSTORE_META_FORMAT_V2;
|
||||||
|
clean_entry_size = 24 /*sizeof(clean_disk_entry)*/ + clean_dyn_size + 4 /*entry_csum*/;
|
||||||
|
min_meta_len = (1 + (block_count - 1 + meta_block_size / clean_entry_size) / (meta_block_size / clean_entry_size)) * meta_block_size;
|
||||||
|
if (!skip_meta_check && meta_area_size < min_meta_len)
|
||||||
|
{
|
||||||
|
if (!data_csum_type)
|
||||||
{
|
{
|
||||||
printf("Warning: Using old metadata format without checksums because the new format"
|
printf("Warning: Using old metadata format without checksums because the new format"
|
||||||
" doesn't fit into provided area (%ju bytes required, %ju bytes available)\n", meta_len, meta_area_size);
|
" doesn't fit into provided area (%ju bytes required, %ju bytes available)\n", min_meta_len, meta_area_size);
|
||||||
|
meta_format = BLOCKSTORE_META_FORMAT_V1;
|
||||||
|
goto recalc;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
goto too_small;
|
||||||
}
|
}
|
||||||
clean_entry_size = clean_entry_v0_size;
|
|
||||||
meta_len = meta_v0_len;
|
|
||||||
meta_format = BLOCKSTORE_META_FORMAT_V1;
|
|
||||||
}
|
}
|
||||||
else
|
|
||||||
meta_format = BLOCKSTORE_META_FORMAT_V2;
|
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
meta_format = BLOCKSTORE_META_FORMAT_V2;
|
|
||||||
if (!skip_meta_check && meta_area_size < meta_len)
|
|
||||||
{
|
{
|
||||||
throw std::runtime_error("Metadata area is too small, need at least "+std::to_string(meta_len)+" bytes, have only "+std::to_string(meta_area_size)+" bytes");
|
throw std::runtime_error("meta_format = "+std::to_string(meta_format)+" is not supported");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void blockstore_disk_t::check_lengths()
|
||||||
|
{
|
||||||
|
if (meta_area_size < min_meta_len)
|
||||||
|
{
|
||||||
|
throw std::runtime_error("Metadata area is too small, need at least "+std::to_string(min_meta_len)+" bytes, have only "+std::to_string(meta_area_size)+" bytes");
|
||||||
}
|
}
|
||||||
// requested journal size
|
// requested journal size
|
||||||
if (!skip_meta_check && cfg_journal_size > journal_len)
|
if (cfg_journal_size > journal_len)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("Requested journal_size is too large");
|
throw std::runtime_error("Requested journal_size is too large");
|
||||||
}
|
}
|
||||||
@@ -321,12 +374,19 @@ static int bs_openmode(const std::string & mode)
|
|||||||
|
|
||||||
void blockstore_disk_t::open_data()
|
void blockstore_disk_t::open_data()
|
||||||
{
|
{
|
||||||
data_fd = open(data_device.c_str(), bs_openmode(data_io) | O_RDWR);
|
if (data_fd >= 0)
|
||||||
|
{
|
||||||
|
throw std::runtime_error("data device is already opened");
|
||||||
|
}
|
||||||
|
data_fd = mock_mode ? MOCK_DATA_FD : open(data_device.c_str(), bs_openmode(data_io) | O_RDWR);
|
||||||
if (data_fd == -1)
|
if (data_fd == -1)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("Failed to open data device "+data_device+": "+std::string(strerror(errno)));
|
throw std::runtime_error("Failed to open data device "+data_device+": "+std::string(strerror(errno)));
|
||||||
}
|
}
|
||||||
check_size(data_fd, &data_device_size, &data_device_sect, "data device");
|
if (!mock_mode)
|
||||||
|
{
|
||||||
|
check_size(data_fd, &data_device_size, &data_device_sect, "data device");
|
||||||
|
}
|
||||||
if (disk_alignment % data_device_sect)
|
if (disk_alignment % data_device_sect)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(
|
throw std::runtime_error(
|
||||||
@@ -338,7 +398,7 @@ void blockstore_disk_t::open_data()
|
|||||||
{
|
{
|
||||||
throw std::runtime_error("data_offset exceeds device size = "+std::to_string(data_device_size));
|
throw std::runtime_error("data_offset exceeds device size = "+std::to_string(data_device_size));
|
||||||
}
|
}
|
||||||
if (!disable_flock && flock(data_fd, LOCK_EX|LOCK_NB) != 0)
|
if (!mock_mode && !disable_flock && flock(data_fd, LOCK_EX|LOCK_NB) != 0)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(std::string("Failed to lock data device: ") + strerror(errno));
|
throw std::runtime_error(std::string("Failed to lock data device: ") + strerror(errno));
|
||||||
}
|
}
|
||||||
@@ -346,19 +406,26 @@ void blockstore_disk_t::open_data()
|
|||||||
|
|
||||||
void blockstore_disk_t::open_meta()
|
void blockstore_disk_t::open_meta()
|
||||||
{
|
{
|
||||||
|
if (meta_fd >= 0)
|
||||||
|
{
|
||||||
|
throw std::runtime_error("metadata device is already opened");
|
||||||
|
}
|
||||||
if (meta_device != data_device || meta_io != data_io)
|
if (meta_device != data_device || meta_io != data_io)
|
||||||
{
|
{
|
||||||
meta_fd = open(meta_device.c_str(), bs_openmode(meta_io) | O_RDWR);
|
meta_fd = mock_mode ? MOCK_META_FD : open(meta_device.c_str(), bs_openmode(meta_io) | O_RDWR);
|
||||||
if (meta_fd == -1)
|
if (meta_fd == -1)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("Failed to open metadata device "+meta_device+": "+std::string(strerror(errno)));
|
throw std::runtime_error("Failed to open metadata device "+meta_device+": "+std::string(strerror(errno)));
|
||||||
}
|
}
|
||||||
check_size(meta_fd, &meta_device_size, &meta_device_sect, "metadata device");
|
if (!mock_mode)
|
||||||
|
{
|
||||||
|
check_size(meta_fd, &meta_device_size, &meta_device_sect, "metadata device");
|
||||||
|
}
|
||||||
if (meta_offset >= meta_device_size)
|
if (meta_offset >= meta_device_size)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("meta_offset exceeds device size = "+std::to_string(meta_device_size));
|
throw std::runtime_error("meta_offset exceeds device size = "+std::to_string(meta_device_size));
|
||||||
}
|
}
|
||||||
if (!disable_flock && meta_device != data_device && flock(meta_fd, LOCK_EX|LOCK_NB) != 0)
|
if (!mock_mode && !disable_flock && meta_device != data_device && flock(meta_fd, LOCK_EX|LOCK_NB) != 0)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(std::string("Failed to lock metadata device: ") + strerror(errno));
|
throw std::runtime_error(std::string("Failed to lock metadata device: ") + strerror(errno));
|
||||||
}
|
}
|
||||||
@@ -384,15 +451,26 @@ void blockstore_disk_t::open_meta()
|
|||||||
|
|
||||||
void blockstore_disk_t::open_journal()
|
void blockstore_disk_t::open_journal()
|
||||||
{
|
{
|
||||||
|
if (journal_fd >= 0)
|
||||||
|
{
|
||||||
|
throw std::runtime_error("journal device is already opened");
|
||||||
|
}
|
||||||
if (journal_device != meta_device || journal_io != meta_io)
|
if (journal_device != meta_device || journal_io != meta_io)
|
||||||
{
|
{
|
||||||
journal_fd = open(journal_device.c_str(), bs_openmode(journal_io) | O_RDWR);
|
journal_fd = mock_mode ? MOCK_JOURNAL_FD : open(journal_device.c_str(), bs_openmode(journal_io) | O_RDWR);
|
||||||
if (journal_fd == -1)
|
if (journal_fd == -1)
|
||||||
{
|
{
|
||||||
throw std::runtime_error("Failed to open journal device "+journal_device+": "+std::string(strerror(errno)));
|
throw std::runtime_error("Failed to open journal device "+journal_device+": "+std::string(strerror(errno)));
|
||||||
}
|
}
|
||||||
check_size(journal_fd, &journal_device_size, &journal_device_sect, "journal device");
|
if (!mock_mode)
|
||||||
if (!disable_flock && journal_device != meta_device && flock(journal_fd, LOCK_EX|LOCK_NB) != 0)
|
{
|
||||||
|
check_size(journal_fd, &journal_device_size, &journal_device_sect, "journal device");
|
||||||
|
}
|
||||||
|
if (journal_offset >= journal_device_size)
|
||||||
|
{
|
||||||
|
throw std::runtime_error("journal_offset exceeds device size = "+std::to_string(journal_device_size));
|
||||||
|
}
|
||||||
|
if (!mock_mode && !disable_flock && journal_device != meta_device && flock(journal_fd, LOCK_EX|LOCK_NB) != 0)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(std::string("Failed to lock journal device: ") + strerror(errno));
|
throw std::runtime_error(std::string("Failed to lock journal device: ") + strerror(errno));
|
||||||
}
|
}
|
||||||
@@ -418,25 +496,32 @@ void blockstore_disk_t::open_journal()
|
|||||||
|
|
||||||
void blockstore_disk_t::close_all()
|
void blockstore_disk_t::close_all()
|
||||||
{
|
{
|
||||||
if (data_fd >= 0)
|
if (!mock_mode)
|
||||||
close(data_fd);
|
{
|
||||||
if (meta_fd >= 0 && meta_fd != data_fd)
|
if (data_fd >= 0)
|
||||||
close(meta_fd);
|
close(data_fd);
|
||||||
if (journal_fd >= 0 && journal_fd != meta_fd)
|
if (meta_fd >= 0 && meta_fd != data_fd)
|
||||||
close(journal_fd);
|
close(meta_fd);
|
||||||
|
if (journal_fd >= 0 && journal_fd != meta_fd)
|
||||||
|
close(journal_fd);
|
||||||
|
}
|
||||||
data_fd = meta_fd = journal_fd = -1;
|
data_fd = meta_fd = journal_fd = -1;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Sadly DISCARD only works through ioctl(), but it seems to always block the device queue,
|
// Sadly DISCARD only works through ioctl(), but it seems to always block the device queue,
|
||||||
// so it's not a big deal that we can only run it synchronously.
|
// so it's not a big deal that we can only run it synchronously.
|
||||||
int blockstore_disk_t::trim_data(allocator_t *alloc)
|
int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_free)
|
||||||
{
|
{
|
||||||
|
if (mock_mode)
|
||||||
|
{
|
||||||
|
return -EINVAL;
|
||||||
|
}
|
||||||
int r = 0;
|
int r = 0;
|
||||||
uint64_t j = 0, i = 0;
|
uint64_t j = 0, i = 0;
|
||||||
uint64_t discarded = 0;
|
uint64_t discarded = 0;
|
||||||
for (; i <= block_count; i++)
|
for (; i <= block_count; i++)
|
||||||
{
|
{
|
||||||
if (i >= block_count || alloc->get(i))
|
if (i >= block_count || is_free(i))
|
||||||
{
|
{
|
||||||
if (i > j && (i-j)*data_block_size >= min_discard_size)
|
if (i > j && (i-j)*data_block_size >= min_discard_size)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -8,10 +8,19 @@
|
|||||||
#include <string>
|
#include <string>
|
||||||
#include <map>
|
#include <map>
|
||||||
|
|
||||||
|
// Memory alignment for direct I/O (usually 512 bytes)
|
||||||
|
#ifndef DIRECT_IO_ALIGNMENT
|
||||||
|
#define DIRECT_IO_ALIGNMENT 512
|
||||||
|
#endif
|
||||||
|
|
||||||
#define BLOCKSTORE_CSUM_NONE 0
|
#define BLOCKSTORE_CSUM_NONE 0
|
||||||
// Lower byte of checksum type is its length
|
// Lower byte of checksum type is its length
|
||||||
#define BLOCKSTORE_CSUM_CRC32C 0x104
|
#define BLOCKSTORE_CSUM_CRC32C 0x104
|
||||||
|
|
||||||
|
#define MOCK_DATA_FD 1000
|
||||||
|
#define MOCK_META_FD 1001
|
||||||
|
#define MOCK_JOURNAL_FD 1002
|
||||||
|
|
||||||
class allocator_t;
|
class allocator_t;
|
||||||
|
|
||||||
struct blockstore_disk_t
|
struct blockstore_disk_t
|
||||||
@@ -22,11 +31,15 @@ struct blockstore_disk_t
|
|||||||
// Required write alignment and journal/metadata/data areas' location alignment
|
// Required write alignment and journal/metadata/data areas' location alignment
|
||||||
uint32_t disk_alignment = 4096;
|
uint32_t disk_alignment = 4096;
|
||||||
// Journal block size - minimum_io_size of the journal device is the best choice
|
// Journal block size - minimum_io_size of the journal device is the best choice
|
||||||
uint64_t journal_block_size = 4096;
|
uint32_t journal_block_size = 4096;
|
||||||
// Metadata block size - minimum_io_size of the metadata device is the best choice
|
// Metadata block size - minimum_io_size of the metadata device is the best choice
|
||||||
uint64_t meta_block_size = 4096;
|
uint32_t meta_block_size = 4096;
|
||||||
|
// Atomic write size of the data block device
|
||||||
|
uint32_t atomic_write_size = 4096;
|
||||||
|
// Whether we should set RWF_ATOMIC on atomic writes
|
||||||
|
bool use_atomic_flag = false;
|
||||||
// Sparse write tracking granularity. 4 KB is a good choice. Must be a multiple of disk_alignment
|
// Sparse write tracking granularity. 4 KB is a good choice. Must be a multiple of disk_alignment
|
||||||
uint64_t bitmap_granularity = 4096;
|
uint32_t bitmap_granularity = 4096;
|
||||||
// Data checksum type, BLOCKSTORE_CSUM_NONE or BLOCKSTORE_CSUM_CRC32C
|
// Data checksum type, BLOCKSTORE_CSUM_NONE or BLOCKSTORE_CSUM_CRC32C
|
||||||
uint32_t data_csum_type = BLOCKSTORE_CSUM_NONE;
|
uint32_t data_csum_type = BLOCKSTORE_CSUM_NONE;
|
||||||
// Checksum block size, must be a multiple of bitmap_granularity
|
// Checksum block size, must be a multiple of bitmap_granularity
|
||||||
@@ -36,27 +49,37 @@ struct blockstore_disk_t
|
|||||||
// I/O modes for data, metadata and journal: direct or "" = O_DIRECT, cached = O_SYNC, directsync = O_DIRECT|O_SYNC
|
// I/O modes for data, metadata and journal: direct or "" = O_DIRECT, cached = O_SYNC, directsync = O_DIRECT|O_SYNC
|
||||||
// O_SYNC without O_DIRECT = use Linux page cache for reads and writes
|
// O_SYNC without O_DIRECT = use Linux page cache for reads and writes
|
||||||
std::string data_io, meta_io, journal_io;
|
std::string data_io, meta_io, journal_io;
|
||||||
|
// It is safe to disable fsync() if drive write cache is writethrough
|
||||||
|
bool disable_data_fsync = false, disable_meta_fsync = false, disable_journal_fsync = false;
|
||||||
|
// Keep journal (buffered data) in memory?
|
||||||
|
bool inmemory_meta = true;
|
||||||
|
// Keep metadata in memory?
|
||||||
|
bool inmemory_journal = true;
|
||||||
// Data discard granularity and minimum size (for the sake of performance)
|
// Data discard granularity and minimum size (for the sake of performance)
|
||||||
bool discard_on_start = false;
|
bool discard_on_start = false;
|
||||||
uint64_t min_discard_size = 1024*1024;
|
uint64_t min_discard_size = 1024*1024;
|
||||||
uint64_t discard_granularity = 0;
|
uint64_t discard_granularity = 0;
|
||||||
|
|
||||||
int meta_fd = -1, data_fd = -1, journal_fd = -1;
|
int meta_fd = -1, data_fd = -1, journal_fd = -1;
|
||||||
uint64_t meta_offset, meta_device_sect, meta_device_size, meta_len, meta_format = 0;
|
uint64_t meta_offset = 0, meta_device_sect = 0, meta_device_size = 0, meta_area_size = 0, min_meta_len = 0;
|
||||||
uint64_t data_offset, data_device_sect, data_device_size, data_len;
|
uint64_t data_offset = 0, data_device_sect = 0, data_device_size = 0, data_len = 0;
|
||||||
uint64_t journal_offset, journal_device_sect, journal_device_size, journal_len;
|
uint64_t journal_offset = 0, journal_device_sect = 0, journal_device_size = 0, journal_len = 0;
|
||||||
|
uint64_t meta_format = 0;
|
||||||
|
|
||||||
uint32_t block_order = 0;
|
|
||||||
uint64_t block_count = 0;
|
uint64_t block_count = 0;
|
||||||
uint32_t clean_entry_bitmap_size = 0, clean_entry_size = 0, clean_dyn_size = 0;
|
uint32_t clean_entry_bitmap_size = 0;
|
||||||
|
uint32_t clean_entry_size = 0, clean_dyn_size = 0; // for meta_v1/2
|
||||||
|
|
||||||
|
bool mock_mode = false;
|
||||||
|
|
||||||
void parse_config(std::map<std::string, std::string> & config);
|
void parse_config(std::map<std::string, std::string> & config);
|
||||||
void open_data();
|
void open_data();
|
||||||
void open_meta();
|
void open_meta();
|
||||||
void open_journal();
|
void open_journal();
|
||||||
void calc_lengths(bool skip_meta_check = false);
|
void calc_lengths(bool skip_meta_check = false);
|
||||||
|
void check_lengths();
|
||||||
void close_all();
|
void close_all();
|
||||||
int trim_data(allocator_t *alloc);
|
int trim_data(std::function<bool(uint64_t)> is_free);
|
||||||
|
|
||||||
inline uint64_t dirty_dyn_size(uint64_t offset, uint64_t len)
|
inline uint64_t dirty_dyn_size(uint64_t offset, uint64_t len)
|
||||||
{
|
{
|
||||||
|
|||||||
+614
-1243
File diff suppressed because it is too large
Load Diff
@@ -1,22 +1,12 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.1 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
#define COPY_BUF_JOURNAL 1
|
|
||||||
#define COPY_BUF_DATA 2
|
|
||||||
#define COPY_BUF_ZERO 4
|
|
||||||
#define COPY_BUF_CSUM_FILL 8
|
|
||||||
#define COPY_BUF_COALESCED 16
|
|
||||||
#define COPY_BUF_META_BLOCK 32
|
|
||||||
#define COPY_BUF_JOURNALED_BIG 64
|
|
||||||
|
|
||||||
struct copy_buffer_t
|
struct copy_buffer_t
|
||||||
{
|
{
|
||||||
int copy_flags;
|
uint32_t copy_flags;
|
||||||
uint64_t offset, len, disk_offset;
|
uint64_t offset, len, disk_loc, disk_offset, disk_len;
|
||||||
uint64_t journal_sector; // only for reads: sector+1 if used and !journal.inmemory, otherwise 0
|
uint8_t *buf;
|
||||||
void *buf;
|
heap_entry_t *wr;
|
||||||
uint8_t *csum_buf;
|
|
||||||
int *dyn_data;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
struct meta_sector_t
|
struct meta_sector_t
|
||||||
@@ -27,13 +17,6 @@ struct meta_sector_t
|
|||||||
int usage_count;
|
int usage_count;
|
||||||
};
|
};
|
||||||
|
|
||||||
struct flusher_sync_t
|
|
||||||
{
|
|
||||||
bool fsync_meta;
|
|
||||||
int ready_count;
|
|
||||||
int state;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct flusher_meta_write_t
|
struct flusher_meta_write_t
|
||||||
{
|
{
|
||||||
uint64_t sector, pos;
|
uint64_t sector, pos;
|
||||||
@@ -49,93 +32,74 @@ class journal_flusher_co
|
|||||||
{
|
{
|
||||||
blockstore_impl_t *bs;
|
blockstore_impl_t *bs;
|
||||||
journal_flusher_t *flusher;
|
journal_flusher_t *flusher;
|
||||||
int wait_state, wait_count, wait_journal_count;
|
int co_id;
|
||||||
|
int wait_state, wait_count;
|
||||||
struct io_uring_sqe *sqe;
|
struct io_uring_sqe *sqe;
|
||||||
struct ring_data_t *data;
|
struct ring_data_t *data;
|
||||||
|
uint8_t *new_csums = NULL;
|
||||||
|
uint8_t *new_bmp = NULL;
|
||||||
|
uint8_t *punch_bmp = NULL;
|
||||||
|
uint8_t *new_ext_bmp = NULL;
|
||||||
|
|
||||||
std::list<flusher_sync_t>::iterator cur_sync;
|
std::function<void(ring_data_t*)> simple_callback_r, simple_callback_w;
|
||||||
|
|
||||||
obj_ver_id cur;
|
object_id cur_oid;
|
||||||
std::map<obj_ver_id, dirty_entry>::iterator dirty_it, dirty_start, dirty_end;
|
heap_entry_t *cur_obj;
|
||||||
std::map<object_id, uint64_t>::iterator repeat_it;
|
uint64_t fsynced_lsn;
|
||||||
std::function<void(ring_data_t*)> simple_callback_r, simple_callback_rj, simple_callback_w;
|
heap_compact_t compact_info;
|
||||||
|
uint64_t clean_loc;
|
||||||
|
uint32_t modified_block;
|
||||||
|
bool bitmap_copied;
|
||||||
|
bool should_repeat;
|
||||||
|
|
||||||
bool try_trim = false;
|
std::vector<copy_buffer_t> read_vec;
|
||||||
bool skip_copy, has_delete, has_writes;
|
std::vector<heap_entry_t*> csum_copy;
|
||||||
std::vector<copy_buffer_t> v;
|
uint32_t overwrite_start, overwrite_end;
|
||||||
std::vector<copy_buffer_t>::iterator it;
|
int i, res;
|
||||||
int i;
|
bool read_to_fill_incomplete;
|
||||||
bool fill_incomplete, cleared_incomplete;
|
|
||||||
int read_to_fill_incomplete;
|
|
||||||
int copy_count;
|
int copy_count;
|
||||||
uint64_t clean_loc, clean_ver, old_clean_loc, old_clean_ver;
|
bool do_repeat = false;
|
||||||
flusher_meta_write_t meta_old, meta_new;
|
|
||||||
bool clean_init_bitmap;
|
|
||||||
uint64_t clean_bitmap_offset, clean_bitmap_len;
|
|
||||||
uint8_t *clean_init_dyn_ptr;
|
|
||||||
uint8_t *new_clean_bitmap;
|
|
||||||
|
|
||||||
uint64_t new_trim_pos;
|
|
||||||
|
|
||||||
friend class journal_flusher_t;
|
friend class journal_flusher_t;
|
||||||
void scan_dirty();
|
|
||||||
bool read_dirty(int wait_base);
|
void iterate_checksum_holes(std::function<void(int & pos, uint32_t hole_start, uint32_t hole_end)> cb);
|
||||||
bool modify_meta_do_reads(int wait_base);
|
void fill_partial_checksum_blocks();
|
||||||
bool wait_meta_reads(int wait_base);
|
|
||||||
bool modify_meta_read(uint64_t meta_loc, flusher_meta_write_t &wr, int wait_base);
|
|
||||||
bool clear_incomplete_csum_block_bits(int wait_base);
|
|
||||||
void calc_block_checksums(uint32_t *new_data_csums, bool skip_overwrites);
|
|
||||||
void update_metadata_entry();
|
|
||||||
bool write_meta_block(flusher_meta_write_t & meta_block, int wait_base);
|
|
||||||
void update_clean_db();
|
|
||||||
void free_data_blocks();
|
|
||||||
bool fsync_batch(bool fsync_meta, int wait_base);
|
|
||||||
bool trim_journal(int wait_base);
|
|
||||||
void free_buffers();
|
void free_buffers();
|
||||||
|
int check_and_punch_checksums();
|
||||||
|
bool calc_block_checksums();
|
||||||
|
bool write_meta_block(int wait_base);
|
||||||
|
bool read_buffered(int wait_base);
|
||||||
|
bool fsync_meta(int wait_base);
|
||||||
|
bool fsync_buffer(int wait_base);
|
||||||
|
bool trim_lsn(int wait_base);
|
||||||
public:
|
public:
|
||||||
journal_flusher_co();
|
journal_flusher_co();
|
||||||
|
~journal_flusher_co();
|
||||||
bool loop();
|
bool loop();
|
||||||
};
|
};
|
||||||
|
|
||||||
// Journal flusher itself
|
// Journal flusher itself
|
||||||
class journal_flusher_t
|
class journal_flusher_t
|
||||||
{
|
{
|
||||||
int trim_wanted = 0;
|
int force_start = 0;
|
||||||
bool dequeuing;
|
int min_flusher_count = 0, max_flusher_count = 0, cur_flusher_count = 0, target_flusher_count = 0;
|
||||||
int min_flusher_count, max_flusher_count, cur_flusher_count, target_flusher_count;
|
|
||||||
int flusher_start_threshold;
|
|
||||||
journal_flusher_co *co;
|
journal_flusher_co *co;
|
||||||
blockstore_impl_t *bs;
|
blockstore_impl_t *bs;
|
||||||
friend class journal_flusher_co;
|
friend class journal_flusher_co;
|
||||||
|
|
||||||
int journal_trim_counter;
|
robin_hood::unordered_flat_set<object_id> flushing;
|
||||||
bool trimming;
|
int active_flushers = 0;
|
||||||
void* journal_superblock;
|
int wanting_meta_fsync = 0;
|
||||||
|
bool fsyncing_meta = false;
|
||||||
int active_flushers;
|
int syncing_buffer = 0;
|
||||||
int syncing_flushers;
|
|
||||||
std::list<flusher_sync_t> syncs;
|
|
||||||
std::map<object_id, uint64_t> sync_to_repeat;
|
|
||||||
|
|
||||||
std::map<uint64_t, meta_sector_t> meta_sectors;
|
|
||||||
std::deque<object_id> flush_queue;
|
|
||||||
std::map<object_id, uint64_t> flush_versions; // FIXME: consider unordered_map?
|
|
||||||
|
|
||||||
bool try_find_older(std::map<obj_ver_id, dirty_entry>::iterator & dirty_end, obj_ver_id & cur);
|
|
||||||
bool try_find_other(std::map<obj_ver_id, dirty_entry>::iterator & dirty_end, obj_ver_id & cur);
|
|
||||||
|
|
||||||
public:
|
public:
|
||||||
journal_flusher_t(blockstore_impl_t *bs);
|
journal_flusher_t(blockstore_impl_t *bs);
|
||||||
~journal_flusher_t();
|
~journal_flusher_t();
|
||||||
void loop();
|
void loop();
|
||||||
bool is_trim_wanted() { return trim_wanted; }
|
int get_syncing_buffer();
|
||||||
bool is_active();
|
bool is_active();
|
||||||
void mark_trim_possible();
|
|
||||||
void request_trim();
|
void request_trim();
|
||||||
void release_trim();
|
void release_trim();
|
||||||
void enqueue_flush(obj_ver_id oid);
|
|
||||||
void unshift_flush(obj_ver_id oid, bool force);
|
|
||||||
void remove_flush(object_id oid);
|
|
||||||
void dump_diagnostics();
|
void dump_diagnostics();
|
||||||
bool is_mutated(uint64_t clean_loc);
|
|
||||||
};
|
};
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,358 @@
|
|||||||
|
// Metadata storage version 3 ("lsm heap")
|
||||||
|
// Copyright (c) Vitaliy Filippov, 2025+
|
||||||
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include <map>
|
||||||
|
#include <unordered_map>
|
||||||
|
#include <set>
|
||||||
|
#include <deque>
|
||||||
|
#include <vector>
|
||||||
|
|
||||||
|
#include "../client/object_id.h"
|
||||||
|
#include "../util/robin_hood.h"
|
||||||
|
#include "blockstore_disk.h"
|
||||||
|
#include "multilist.h"
|
||||||
|
|
||||||
|
struct pool_shard_settings_t
|
||||||
|
{
|
||||||
|
uint32_t pg_count;
|
||||||
|
uint32_t pg_stripe_size;
|
||||||
|
uint32_t no_inode_stats;
|
||||||
|
};
|
||||||
|
|
||||||
|
#define BS_HEAP_TYPE 0x07
|
||||||
|
#define BS_HEAP_BIG_WRITE 1
|
||||||
|
#define BS_HEAP_SMALL_WRITE 2
|
||||||
|
#define BS_HEAP_INTENT_WRITE 3
|
||||||
|
#define BS_HEAP_BIG_INTENT 4
|
||||||
|
#define BS_HEAP_DELETE 5
|
||||||
|
#define BS_HEAP_COMMIT 6
|
||||||
|
#define BS_HEAP_ROLLBACK 7
|
||||||
|
#define BS_HEAP_STABLE 0x40
|
||||||
|
#define BS_HEAP_GARBAGE 0x80
|
||||||
|
|
||||||
|
class blockstore_heap_t;
|
||||||
|
|
||||||
|
struct heap_small_write_t;
|
||||||
|
struct heap_big_write_t;
|
||||||
|
struct heap_big_intent_t;
|
||||||
|
|
||||||
|
struct __attribute__((__packed__)) heap_entry_t
|
||||||
|
{
|
||||||
|
uint16_t size;
|
||||||
|
uint16_t entry_type;
|
||||||
|
uint32_t crc32c;
|
||||||
|
uint64_t lsn;
|
||||||
|
uint64_t inode;
|
||||||
|
uint64_t stripe;
|
||||||
|
uint64_t version;
|
||||||
|
|
||||||
|
// uint8_t[] external_bitmap
|
||||||
|
// uint8_t[] internal_bitmap
|
||||||
|
// uint32_t[] checksums
|
||||||
|
|
||||||
|
inline uint8_t type() const { return (entry_type & BS_HEAP_TYPE); }
|
||||||
|
inline heap_small_write_t& small() { return *(heap_small_write_t*)this; }
|
||||||
|
inline heap_big_write_t& big() { return *(heap_big_write_t*)this; }
|
||||||
|
inline heap_big_intent_t& big_intent() { return *(heap_big_intent_t*)this; }
|
||||||
|
bool is_garbage();
|
||||||
|
void set_garbage();
|
||||||
|
bool is_overwrite();
|
||||||
|
bool is_compactable();
|
||||||
|
bool is_before(heap_entry_t *other);
|
||||||
|
uint32_t get_size(blockstore_heap_t *heap);
|
||||||
|
uint8_t *get_ext_bitmap(blockstore_heap_t *heap);
|
||||||
|
uint8_t *get_int_bitmap(blockstore_heap_t *heap);
|
||||||
|
uint8_t *get_checksums(blockstore_heap_t *heap);
|
||||||
|
uint32_t *get_checksum(blockstore_heap_t *heap);
|
||||||
|
uint64_t big_location(blockstore_heap_t *heap);
|
||||||
|
void set_big_location(blockstore_heap_t *heap, uint64_t location);
|
||||||
|
uint32_t calc_crc32c();
|
||||||
|
};
|
||||||
|
|
||||||
|
struct __attribute__((__packed__)) heap_small_write_t
|
||||||
|
{
|
||||||
|
heap_entry_t hdr;
|
||||||
|
|
||||||
|
uint64_t location;
|
||||||
|
uint32_t offset;
|
||||||
|
uint32_t len;
|
||||||
|
|
||||||
|
// Also includes 1 bitmap and 1 crc32c after the bitmap if checksums are disabled
|
||||||
|
};
|
||||||
|
|
||||||
|
struct __attribute__((__packed__)) heap_big_write_t
|
||||||
|
{
|
||||||
|
heap_entry_t hdr;
|
||||||
|
|
||||||
|
uint32_t block_num;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct __attribute__((__packed__)) heap_big_intent_t
|
||||||
|
{
|
||||||
|
heap_entry_t hdr;
|
||||||
|
|
||||||
|
uint32_t block_num;
|
||||||
|
uint32_t offset;
|
||||||
|
uint32_t len;
|
||||||
|
|
||||||
|
// Also includes 2 bitmaps and 1 crc32c if checksums are disabled
|
||||||
|
};
|
||||||
|
|
||||||
|
struct __attribute__((__packed__)) heap_list_item_t
|
||||||
|
{
|
||||||
|
heap_list_item_t *prev;
|
||||||
|
heap_list_item_t *next;
|
||||||
|
uint32_t block_num;
|
||||||
|
heap_entry_t entry;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct heap_object_mvcc_t
|
||||||
|
{
|
||||||
|
uint32_t readers = 0;
|
||||||
|
heap_entry_t *garbage_entry = NULL;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct heap_block_info_t
|
||||||
|
{
|
||||||
|
uint32_t used_space = 0;
|
||||||
|
uint64_t mod_lsn = 0, mod_lsn_to = 0; // only 1 block write of LSN sequence is allowed at a moment
|
||||||
|
bool is_writing: 1;
|
||||||
|
bool has_garbage: 1;
|
||||||
|
std::vector<heap_list_item_t*> entries;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct heap_inflight_lsn_t
|
||||||
|
{
|
||||||
|
uint64_t flags;
|
||||||
|
heap_entry_t *wr;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct heap_compact_t
|
||||||
|
{
|
||||||
|
uint64_t compact_lsn, compact_version;
|
||||||
|
heap_entry_t *clean_wr;
|
||||||
|
bool do_delete;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct heap_reshard_state_t;
|
||||||
|
|
||||||
|
struct heap_li_hash
|
||||||
|
{
|
||||||
|
size_t operator()(const heap_list_item_t* li) const noexcept
|
||||||
|
{
|
||||||
|
return robin_hood::hash_int(li->entry.stripe);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
struct heap_li_equal
|
||||||
|
{
|
||||||
|
constexpr bool operator()(const heap_list_item_t* a, const heap_list_item_t* b) const noexcept
|
||||||
|
{
|
||||||
|
return a->entry.stripe == b->entry.stripe;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
using i64hash_t = robin_hood::hash<uint64_t>;
|
||||||
|
using heap_inode_map_t = robin_hood::unordered_flat_set<heap_list_item_t*, heap_li_hash, heap_li_equal, 88>;
|
||||||
|
using heap_block_index_t = robin_hood::unordered_flat_map<uint64_t,
|
||||||
|
robin_hood::unordered_flat_map<inode_t, void*, i64hash_t>, i64hash_t>;
|
||||||
|
using heap_mvcc_map_t = robin_hood::unordered_flat_map<object_id, heap_object_mvcc_t>;
|
||||||
|
|
||||||
|
class blockstore_heap_t
|
||||||
|
{
|
||||||
|
friend class heap_entry_t;
|
||||||
|
|
||||||
|
blockstore_disk_t *dsk = NULL;
|
||||||
|
uint8_t* buffer_area = NULL;
|
||||||
|
int log_level = 0;
|
||||||
|
const uint32_t meta_block_count = 0;
|
||||||
|
const uint32_t max_entry_size = 0;
|
||||||
|
|
||||||
|
robin_hood::unordered_flat_map<pool_id_t, pool_shard_settings_t> pool_shard_settings;
|
||||||
|
// PG => inode => stripe => block number
|
||||||
|
heap_block_index_t block_index;
|
||||||
|
std::vector<heap_block_info_t> block_info;
|
||||||
|
allocator_t *data_alloc = NULL;
|
||||||
|
multilist_index_t *meta_alloc = NULL;
|
||||||
|
uint32_t meta_nearfull_blocks = 0;
|
||||||
|
uint64_t meta_used_space = 0;
|
||||||
|
multilist_alloc_t *buffer_alloc = NULL;
|
||||||
|
std::map<uint64_t, uint64_t> inode_space_stats;
|
||||||
|
uint64_t buffer_area_used_space = 0;
|
||||||
|
uint64_t data_used_space = 0;
|
||||||
|
|
||||||
|
uint64_t next_lsn = 0;
|
||||||
|
uint32_t last_allocated_block = UINT32_MAX;
|
||||||
|
heap_mvcc_map_t object_mvcc;
|
||||||
|
|
||||||
|
// LSN queue: inflight (writing) -> completed [-> fsynced]
|
||||||
|
std::deque<heap_inflight_lsn_t> inflight_lsn;
|
||||||
|
uint32_t to_compact_count = 0;
|
||||||
|
uint64_t compacted_count = 0;
|
||||||
|
uint32_t inflight_overwrite_count = 0;
|
||||||
|
uint64_t first_inflight_lsn = 0;
|
||||||
|
uint64_t completed_lsn = 0;
|
||||||
|
uint64_t fsynced_lsn = 0;
|
||||||
|
std::deque<object_id> compact_queue;
|
||||||
|
|
||||||
|
bool marked_used_blocks = false;
|
||||||
|
bool recheck_queue_filled = false;
|
||||||
|
std::set<uint32_t> recheck_modified_blocks;
|
||||||
|
std::deque<heap_entry_t*> recheck_queue;
|
||||||
|
int recheck_in_progress = 0;
|
||||||
|
bool in_recheck = false;
|
||||||
|
std::function<void(bool is_data, uint64_t offset, uint64_t len, uint8_t* buf, std::function<void()>)> recheck_cb;
|
||||||
|
int recheck_queue_depth = 0;
|
||||||
|
|
||||||
|
uint64_t get_pg_id(inode_t inode, uint64_t stripe);
|
||||||
|
bool validate_object(heap_entry_t *obj);
|
||||||
|
void fill_recheck_queue();
|
||||||
|
int mark_used_blocks();
|
||||||
|
void recheck_buffer(heap_entry_t *cwr, uint8_t *buf);
|
||||||
|
void defragment_block(uint32_t block_num);
|
||||||
|
void reshard_add(heap_reshard_state_t *st, heap_list_item_t *li);
|
||||||
|
|
||||||
|
int allocate_entry(uint32_t entry_size, uint32_t *block_num, bool allow_last_free);
|
||||||
|
void insert_list_item(heap_list_item_t *li);
|
||||||
|
int add_entry(uint32_t wr_size, uint32_t *modified_block, bool allow_last_free,
|
||||||
|
bool explicit_complete, std::function<void(heap_entry_t *wr)> fill_entry);
|
||||||
|
int add_simple(heap_entry_t *obj, uint64_t version, uint32_t *modified_block, uint32_t entry_type);
|
||||||
|
uint32_t meta_alloc_pos(const heap_block_info_t & inf);
|
||||||
|
void modify_alloc(uint32_t block_num, std::function<void(heap_block_info_t &)> change_cb);
|
||||||
|
void mark_garbage_up_to(heap_entry_t *wr);
|
||||||
|
void mark_garbage(uint32_t block_num, heap_entry_t *prev_wr, uint32_t used_big);
|
||||||
|
void push_inflight_lsn(uint64_t lsn, heap_entry_t *wr, uint64_t flags);
|
||||||
|
void mark_completed_lsns(uint64_t mod_lsn);
|
||||||
|
void apply_inflight(heap_inflight_lsn_t & inflight);
|
||||||
|
public:
|
||||||
|
blockstore_heap_t(blockstore_disk_t *dsk, uint8_t *buffer_area, int log_level = 0);
|
||||||
|
~blockstore_heap_t();
|
||||||
|
void start_load(uint64_t completed_lsn);
|
||||||
|
// load data from the disk, returns EDOM on corruption
|
||||||
|
int read_blocks(uint64_t disk_offset, uint64_t size, uint8_t *buf, bool allow_corrupted,
|
||||||
|
std::function<void(uint32_t block_num, heap_entry_t* wr)> handle_write,
|
||||||
|
std::function<void(uint32_t, uint32_t, uint8_t*)> handle_block);
|
||||||
|
int load_blocks(uint64_t disk_offset, uint64_t size, uint8_t *buf,
|
||||||
|
bool allow_corrupted, uint64_t &entries_loaded);
|
||||||
|
// finish loading
|
||||||
|
int finish_load(bool allow_corrupted = false);
|
||||||
|
// get blocks which are modified during loading and should be written to the disk
|
||||||
|
// before finishing initialization if not R/O
|
||||||
|
std::vector<uint32_t> get_recheck_modified_blocks();
|
||||||
|
// recheck small write data after reading the database from disk
|
||||||
|
bool recheck_small_writes(std::function<void(bool is_data, uint64_t offset, uint64_t len, uint8_t* buf, std::function<void()>)> read_buffer, int queue_depth);
|
||||||
|
// reshard database according to the pool's PG count
|
||||||
|
void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit);
|
||||||
|
bool reshard_continue(void* reshard_state, uint64_t chunk_limit);
|
||||||
|
bool reshard_check(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size);
|
||||||
|
void reshard_abort(void* reshard_state);
|
||||||
|
void set_no_inode_stats(const std::vector<uint64_t> & pool_ids);
|
||||||
|
void recalc_inode_space_stats(uint64_t pool_id, bool per_inode);
|
||||||
|
// read an object entry and lock it against removal
|
||||||
|
// in the future, may become asynchronous
|
||||||
|
heap_entry_t *lock_and_read_entry(object_id oid);
|
||||||
|
// re-read a locked object entry with the given lsn (pointer may be invalidated)
|
||||||
|
heap_entry_t *read_locked_entry(object_id oid, uint64_t lsn);
|
||||||
|
// read an object entry without locking it
|
||||||
|
heap_entry_t *read_entry(object_id oid);
|
||||||
|
// unlock an entry
|
||||||
|
bool unlock_entry(object_id oid);
|
||||||
|
// set or verify checksums in a write request
|
||||||
|
bool calc_checksums(heap_entry_t *wr, uint8_t *data, bool set, uint32_t offset = 0, uint32_t len = 0);
|
||||||
|
// set or verify raw block checksums
|
||||||
|
bool calc_block_checksums(uint32_t *block_csums, uint8_t *data, uint8_t *bitmap, uint32_t start, uint32_t end,
|
||||||
|
bool set, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb);
|
||||||
|
bool calc_block_checksums(uint32_t *block_csums, uint8_t *bitmap,
|
||||||
|
uint32_t start, uint32_t end, std::function<uint8_t*(uint32_t start, uint32_t & len)> next,
|
||||||
|
bool set, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb);
|
||||||
|
// adds a small_write or intent_write entry to an object
|
||||||
|
// return 0 if OK, or maybe ENOSPC
|
||||||
|
int add_small_write(object_id oid, heap_entry_t **obj_ptr, uint16_t type, uint64_t version,
|
||||||
|
uint32_t offset, uint32_t len, uint64_t location, uint8_t *bitmap, uint8_t *data, uint32_t *modified_block);
|
||||||
|
// adds a big_write (overwrite) entry to an object
|
||||||
|
int add_big_write(object_id oid, heap_entry_t *old_head, bool stable, uint64_t version,
|
||||||
|
uint32_t offset, uint32_t len, uint64_t location, uint8_t *bitmap, uint8_t *data, uint32_t *modified_block);
|
||||||
|
// adds a "redirecting" big_intent entry to an object (same as big_write, used to avoid fsync on desktop SSDs)
|
||||||
|
int add_redirect_intent(object_id oid, heap_entry_t **obj_ptr, uint64_t version,
|
||||||
|
uint32_t offset, uint32_t len, uint64_t location, uint8_t *bitmap, uint8_t *data, uint32_t *modified_block);
|
||||||
|
// adds a big_intent (atomic partial modification) entry to an object
|
||||||
|
int add_big_intent(object_id oid, heap_entry_t **obj_ptr, uint64_t version,
|
||||||
|
uint32_t offset, uint32_t len, uint8_t *bitmap, uint8_t *data, uint8_t *checksums, uint32_t *modified_block);
|
||||||
|
// adds a compacted up to <version> entry to an object
|
||||||
|
int add_compact(heap_entry_t *obj, uint64_t compact_version, uint64_t compact_lsn, uint64_t compact_location,
|
||||||
|
bool do_delete, uint32_t *modified_block, uint8_t *new_int_bitmap, uint8_t *new_ext_bitmap, uint8_t *new_csums);
|
||||||
|
// "punch holes" in a big_entry
|
||||||
|
int punch_holes(heap_entry_t *wr, uint8_t *new_bitmap, uint8_t *new_csums, uint32_t *modified_block);
|
||||||
|
// stabilize an unstable object version
|
||||||
|
// return 0 if OK, ENOENT if not exists
|
||||||
|
int add_commit(heap_entry_t *obj, uint64_t version, uint32_t *modified_block);
|
||||||
|
// rollback an unstable object version
|
||||||
|
// return 0 if OK, ENOENT if not exists, EBUSY if already stable
|
||||||
|
int add_rollback(heap_entry_t *obj, uint64_t version, uint32_t *modified_block);
|
||||||
|
// forget an object
|
||||||
|
// return error code
|
||||||
|
int add_delete(heap_entry_t *obj, uint32_t *modified_block);
|
||||||
|
// get the next object to compact
|
||||||
|
// guaranteed to return objects in min lsn order
|
||||||
|
// returns 0 if OK, ENOENT if nothing to compact
|
||||||
|
int get_next_compact(object_id & oid);
|
||||||
|
void iterate_with_stable(heap_entry_t *obj, uint64_t max_lsn, std::function<bool(heap_entry_t*, bool stable)> cb);
|
||||||
|
// iterate compactable entries
|
||||||
|
heap_compact_t iterate_compaction(heap_entry_t *obj, uint64_t fsynced_lsn, bool under_pressure,
|
||||||
|
std::function<void(heap_entry_t*)> small_wr_cb);
|
||||||
|
// iterate all objects
|
||||||
|
void iterate_objects(std::function<void(heap_entry_t*, uint32_t block_num)> cb);
|
||||||
|
// retrieve object listing from a PG
|
||||||
|
int list_objects(uint32_t pg_num, object_id min_oid, object_id max_oid,
|
||||||
|
obj_ver_id **result_list, size_t *stable_count, size_t *unstable_count);
|
||||||
|
|
||||||
|
// inflight write tracking
|
||||||
|
void start_block_write(uint32_t block_num);
|
||||||
|
void complete_block_write(uint32_t block_num);
|
||||||
|
void complete_lsn_write(uint64_t lsn);
|
||||||
|
bool is_lsn_completed(uint64_t lsn);
|
||||||
|
uint64_t get_completed_lsn();
|
||||||
|
uint64_t get_fsynced_lsn();
|
||||||
|
void mark_lsn_fsynced(uint64_t lsn);
|
||||||
|
|
||||||
|
// data device block allocator functions
|
||||||
|
uint64_t find_free_data();
|
||||||
|
bool is_data_used(uint64_t location);
|
||||||
|
void use_data(inode_t inode, uint64_t location);
|
||||||
|
void free_data(inode_t inode, uint64_t location);
|
||||||
|
|
||||||
|
// buffer device allocator functions
|
||||||
|
uint64_t find_free_buffer_area(uint64_t size);
|
||||||
|
bool is_buffer_area_free(uint64_t location, uint64_t size);
|
||||||
|
void use_buffer_area(inode_t inode, uint64_t location, uint64_t size);
|
||||||
|
void free_buffer_area(inode_t inode, uint64_t location, uint64_t size);
|
||||||
|
uint64_t get_buffer_area_used_space();
|
||||||
|
|
||||||
|
// get metadata block data buffer and used space
|
||||||
|
void get_meta_block(uint32_t block_num, uint8_t *buffer);
|
||||||
|
void fill_block_empty_space(uint8_t *buffer, uint32_t pos);
|
||||||
|
uint32_t get_meta_block_used_space(uint32_t block_num);
|
||||||
|
|
||||||
|
// get space usage statistics
|
||||||
|
uint64_t get_data_used_space();
|
||||||
|
const std::map<uint64_t, uint64_t> & get_inode_space_stats();
|
||||||
|
uint64_t get_meta_total_space();
|
||||||
|
uint64_t get_meta_used_space();
|
||||||
|
uint32_t get_meta_nearfull_blocks();
|
||||||
|
uint32_t get_compact_queue_size();
|
||||||
|
uint32_t get_to_compact_count();
|
||||||
|
uint64_t get_compacted_count();
|
||||||
|
|
||||||
|
uint64_t entry_pos(uint32_t block_num, uint32_t offset);
|
||||||
|
heap_entry_t *entry_from_pos(uint64_t entry_pos, bool allow_unallocated = false);
|
||||||
|
heap_entry_t *prev(heap_entry_t *wr);
|
||||||
|
uint32_t get_simple_entry_size();
|
||||||
|
uint32_t get_big_entry_size();
|
||||||
|
uint32_t get_big_intent_entry_size();
|
||||||
|
uint32_t get_small_entry_size(uint32_t offset, uint32_t len);
|
||||||
|
uint32_t get_csum_size(heap_entry_t *wr);
|
||||||
|
uint32_t get_csum_size(uint32_t entry_type, uint32_t offset = 0, uint32_t len = 0);
|
||||||
|
};
|
||||||
+120
-494
@@ -1,13 +1,18 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.1 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
#include "blockstore_impl.h"
|
#include <stdexcept>
|
||||||
|
|
||||||
blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_t *ringloop, timerfd_manager_t *tfd)
|
#include "blockstore_impl.h"
|
||||||
|
#include "blockstore_internal.h"
|
||||||
|
#include "crc32c.h"
|
||||||
|
|
||||||
|
blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd, bool mock_mode)
|
||||||
{
|
{
|
||||||
assert(sizeof(blockstore_op_private_t) <= BS_OP_PRIVATE_DATA_SIZE);
|
assert(sizeof(blockstore_op_private_t) <= BS_OP_PRIVATE_DATA_SIZE);
|
||||||
this->tfd = tfd;
|
this->tfd = tfd;
|
||||||
this->ringloop = ringloop;
|
this->ringloop = ringloop;
|
||||||
|
dsk.mock_mode = mock_mode;
|
||||||
ring_consumer.loop = [this]() { loop(); };
|
ring_consumer.loop = [this]() { loop(); };
|
||||||
ringloop->register_consumer(&ring_consumer);
|
ringloop->register_consumer(&ring_consumer);
|
||||||
initialized = 0;
|
initialized = 0;
|
||||||
@@ -17,31 +22,37 @@ blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_t *
|
|||||||
dsk.open_data();
|
dsk.open_data();
|
||||||
dsk.open_meta();
|
dsk.open_meta();
|
||||||
dsk.open_journal();
|
dsk.open_journal();
|
||||||
calc_lengths();
|
dsk.calc_lengths();
|
||||||
alloc_dyn_data = dsk.clean_dyn_size > sizeof(void*) || dsk.csum_block_size > 0;
|
dsk.check_lengths();
|
||||||
zero_object = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.data_block_size);
|
|
||||||
data_alloc = new allocator_t(dsk.block_count);
|
|
||||||
}
|
}
|
||||||
catch (std::exception & e)
|
catch (std::exception & e)
|
||||||
{
|
{
|
||||||
dsk.close_all();
|
dsk.close_all();
|
||||||
throw;
|
throw;
|
||||||
}
|
}
|
||||||
|
meta_superblock = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.meta_block_size);
|
||||||
|
memset(meta_superblock, 0, dsk.meta_block_size);
|
||||||
flusher = new journal_flusher_t(this);
|
flusher = new journal_flusher_t(this);
|
||||||
|
if (dsk.inmemory_journal)
|
||||||
|
{
|
||||||
|
buffer_area = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.journal_len);
|
||||||
|
}
|
||||||
|
heap = new blockstore_heap_t(&dsk, buffer_area, log_level);
|
||||||
|
ringloop->wakeup();
|
||||||
}
|
}
|
||||||
|
|
||||||
blockstore_impl_t::~blockstore_impl_t()
|
blockstore_impl_t::~blockstore_impl_t()
|
||||||
{
|
{
|
||||||
delete data_alloc;
|
if (flusher)
|
||||||
delete flusher;
|
delete flusher;
|
||||||
if (zero_object)
|
if (heap)
|
||||||
free(zero_object);
|
delete heap;
|
||||||
|
if (buffer_area)
|
||||||
|
free(buffer_area);
|
||||||
|
if (meta_superblock)
|
||||||
|
free(meta_superblock);
|
||||||
ringloop->unregister_consumer(&ring_consumer);
|
ringloop->unregister_consumer(&ring_consumer);
|
||||||
dsk.close_all();
|
dsk.close_all();
|
||||||
if (metadata_buffer)
|
|
||||||
free(metadata_buffer);
|
|
||||||
if (clean_bitmaps)
|
|
||||||
free(clean_bitmaps);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
bool blockstore_impl_t::is_started()
|
bool blockstore_impl_t::is_started()
|
||||||
@@ -57,10 +68,9 @@ bool blockstore_impl_t::is_stalled()
|
|||||||
// main event loop - produce requests
|
// main event loop - produce requests
|
||||||
void blockstore_impl_t::loop()
|
void blockstore_impl_t::loop()
|
||||||
{
|
{
|
||||||
// FIXME: initialized == 10 is ugly
|
|
||||||
if (initialized != 10)
|
if (initialized != 10)
|
||||||
{
|
{
|
||||||
// read metadata, then journal
|
// read metadata
|
||||||
if (initialized == 0)
|
if (initialized == 0)
|
||||||
{
|
{
|
||||||
metadata_init_reader = new blockstore_init_meta(this);
|
metadata_init_reader = new blockstore_init_meta(this);
|
||||||
@@ -73,69 +83,41 @@ void blockstore_impl_t::loop()
|
|||||||
{
|
{
|
||||||
delete metadata_init_reader;
|
delete metadata_init_reader;
|
||||||
metadata_init_reader = NULL;
|
metadata_init_reader = NULL;
|
||||||
journal_init_reader = new blockstore_init_journal(this);
|
|
||||||
initialized = 2;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (initialized == 2)
|
|
||||||
{
|
|
||||||
int res = journal_init_reader->loop();
|
|
||||||
if (!res)
|
|
||||||
{
|
|
||||||
delete journal_init_reader;
|
|
||||||
journal_init_reader = NULL;
|
|
||||||
initialized = 3;
|
initialized = 3;
|
||||||
ringloop->wakeup();
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (initialized == 3)
|
if (initialized == 3)
|
||||||
{
|
{
|
||||||
if (!readonly && dsk.discard_on_start)
|
if (!readonly && dsk.discard_on_start)
|
||||||
dsk.trim_data(data_alloc);
|
|
||||||
if (journal.flush_journal)
|
|
||||||
initialized = 4;
|
|
||||||
else
|
|
||||||
initialized = 10;
|
|
||||||
}
|
|
||||||
if (initialized == 4)
|
|
||||||
{
|
|
||||||
if (readonly)
|
|
||||||
{
|
{
|
||||||
printf("Can't flush the journal in readonly mode\n");
|
dsk.trim_data([this](uint64_t block_num){ return heap->is_data_used(block_num * dsk.data_block_size); });
|
||||||
exit(1);
|
|
||||||
}
|
}
|
||||||
flusher->loop();
|
initialized = 10;
|
||||||
ringloop->submit();
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
// try to submit ops
|
// try to submit ops
|
||||||
unsigned initial_ring_space = ringloop->space_left();
|
unsigned initial_ring_space = ringloop->space_left();
|
||||||
// has_writes == 0 - no writes before the current queue item
|
int op_idx = 0, new_idx = 0;
|
||||||
// has_writes == 1 - some writes in progress
|
bool has_unfinished_writes = false;
|
||||||
// has_writes == 2 - tried to submit some writes, but failed
|
|
||||||
int has_writes = 0, op_idx = 0, new_idx = 0;
|
|
||||||
for (; op_idx < submit_queue.size(); op_idx++, new_idx++)
|
for (; op_idx < submit_queue.size(); op_idx++, new_idx++)
|
||||||
{
|
{
|
||||||
auto op = submit_queue[op_idx];
|
auto op = submit_queue[op_idx];
|
||||||
submit_queue[new_idx] = op;
|
submit_queue[new_idx] = op;
|
||||||
// FIXME: This needs some simplification
|
|
||||||
// Writes should not block reads if the ring is not full and reads don't depend on them
|
|
||||||
// In all other cases we should stop submission
|
|
||||||
if (PRIV(op)->wait_for)
|
if (PRIV(op)->wait_for)
|
||||||
{
|
{
|
||||||
check_wait(op);
|
check_wait(op);
|
||||||
if (PRIV(op)->wait_for == WAIT_SQE)
|
if (PRIV(op)->wait_for == WAIT_SQE)
|
||||||
{
|
{
|
||||||
|
// ring is full, stop submission
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
else if (PRIV(op)->wait_for)
|
else if (PRIV(op)->wait_for)
|
||||||
{
|
{
|
||||||
if (op->opcode == BS_OP_WRITE || op->opcode == BS_OP_WRITE_STABLE || op->opcode == BS_OP_DELETE)
|
has_unfinished_writes = has_unfinished_writes || op->opcode == BS_OP_WRITE ||
|
||||||
{
|
op->opcode == BS_OP_WRITE_STABLE || op->opcode == BS_OP_DELETE ||
|
||||||
has_writes = 2;
|
op->opcode == BS_OP_STABLE || op->opcode == BS_OP_ROLLBACK;
|
||||||
}
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -148,46 +130,33 @@ void blockstore_impl_t::loop()
|
|||||||
{
|
{
|
||||||
wr_st = dequeue_read(op);
|
wr_st = dequeue_read(op);
|
||||||
}
|
}
|
||||||
else if (op->opcode == BS_OP_WRITE || op->opcode == BS_OP_WRITE_STABLE)
|
else if (op->opcode == BS_OP_WRITE || op->opcode == BS_OP_WRITE_STABLE || op->opcode == BS_OP_DELETE)
|
||||||
{
|
{
|
||||||
if (has_writes == 2)
|
|
||||||
{
|
|
||||||
// Some writes already could not be submitted
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
wr_st = dequeue_write(op);
|
wr_st = dequeue_write(op);
|
||||||
has_writes = wr_st > 0 ? 1 : 2;
|
has_unfinished_writes = has_unfinished_writes || (wr_st != 2);
|
||||||
}
|
|
||||||
else if (op->opcode == BS_OP_DELETE)
|
|
||||||
{
|
|
||||||
if (has_writes == 2)
|
|
||||||
{
|
|
||||||
// Some writes already could not be submitted
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
wr_st = dequeue_del(op);
|
|
||||||
has_writes = wr_st > 0 ? 1 : 2;
|
|
||||||
}
|
}
|
||||||
else if (op->opcode == BS_OP_SYNC)
|
else if (op->opcode == BS_OP_SYNC)
|
||||||
{
|
{
|
||||||
// sync only completed writes?
|
// syncs only completed writes, so doesn't have to be blocked by anything
|
||||||
// wait for the data device fsync to complete, then submit journal writes for big writes
|
|
||||||
// then submit an fsync operation
|
|
||||||
wr_st = continue_sync(op);
|
wr_st = continue_sync(op);
|
||||||
}
|
}
|
||||||
else if (op->opcode == BS_OP_STABLE)
|
else if (op->opcode == BS_OP_STABLE || op->opcode == BS_OP_ROLLBACK)
|
||||||
{
|
{
|
||||||
wr_st = dequeue_stable(op);
|
wr_st = dequeue_stable(op);
|
||||||
}
|
has_unfinished_writes = has_unfinished_writes || (wr_st != 2);
|
||||||
else if (op->opcode == BS_OP_ROLLBACK)
|
|
||||||
{
|
|
||||||
wr_st = dequeue_rollback(op);
|
|
||||||
}
|
}
|
||||||
else if (op->opcode == BS_OP_LIST)
|
else if (op->opcode == BS_OP_LIST)
|
||||||
{
|
{
|
||||||
// LIST doesn't have to be blocked by previous modifications
|
// LIST has to be blocked by previous writes and commits/rollbacks
|
||||||
process_list(op);
|
if (!has_unfinished_writes)
|
||||||
wr_st = 2;
|
{
|
||||||
|
process_list(op);
|
||||||
|
wr_st = 2;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
wr_st = 0;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if (wr_st == 2)
|
if (wr_st == 2)
|
||||||
{
|
{
|
||||||
@@ -196,16 +165,13 @@ void blockstore_impl_t::loop()
|
|||||||
}
|
}
|
||||||
if (wr_st == 0)
|
if (wr_st == 0)
|
||||||
{
|
{
|
||||||
|
PRIV(op)->pending_ops = 0;
|
||||||
ringloop->restore(prev_sqe_pos);
|
ringloop->restore(prev_sqe_pos);
|
||||||
if (PRIV(op)->wait_for == WAIT_SQE)
|
if (PRIV(op)->wait_for == WAIT_SQE)
|
||||||
{
|
{
|
||||||
// ring is full, stop submission
|
// ring is full, stop submission
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
else if (PRIV(op)->wait_for == WAIT_JOURNAL)
|
|
||||||
{
|
|
||||||
PRIV(op)->wait_detail2 = (unstable_writes.size()+unstable_unsynced);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (op_idx != new_idx)
|
if (op_idx != new_idx)
|
||||||
@@ -220,17 +186,19 @@ void blockstore_impl_t::loop()
|
|||||||
{
|
{
|
||||||
flusher->loop();
|
flusher->loop();
|
||||||
}
|
}
|
||||||
|
for (auto & block_num: pending_modified_blocks)
|
||||||
|
{
|
||||||
|
auto & mb = modified_blocks[block_num];
|
||||||
|
heap->get_meta_block(block_num, mb.buf);
|
||||||
|
heap->start_block_write(block_num);
|
||||||
|
mb.sent = true;
|
||||||
|
}
|
||||||
int ret = ringloop->submit();
|
int ret = ringloop->submit();
|
||||||
if (ret < 0)
|
if (ret < 0)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(std::string("io_uring_submit: ") + strerror(-ret));
|
throw std::runtime_error(std::string("io_uring_submit: ") + strerror(-ret));
|
||||||
}
|
}
|
||||||
for (auto s: journal.submitting_sectors)
|
pending_modified_blocks.clear();
|
||||||
{
|
|
||||||
// Mark journal sector writes as submitted
|
|
||||||
journal.sector_info[s].submit_id = 0;
|
|
||||||
}
|
|
||||||
journal.submitting_sectors.clear();
|
|
||||||
if ((initial_ring_space - ringloop->space_left()) > 0)
|
if ((initial_ring_space - ringloop->space_left()) > 0)
|
||||||
{
|
{
|
||||||
live = true;
|
live = true;
|
||||||
@@ -248,7 +216,7 @@ bool blockstore_impl_t::is_safe_to_stop()
|
|||||||
{
|
{
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
if (unsynced_big_writes.size() > 0 || unsynced_small_writes.size() > 0)
|
if (has_unsynced())
|
||||||
{
|
{
|
||||||
if (!readonly && !stop_sync_submitted)
|
if (!readonly && !stop_sync_submitted)
|
||||||
{
|
{
|
||||||
@@ -272,7 +240,7 @@ void blockstore_impl_t::check_wait(blockstore_op_t *op)
|
|||||||
{
|
{
|
||||||
if (PRIV(op)->wait_for == WAIT_SQE)
|
if (PRIV(op)->wait_for == WAIT_SQE)
|
||||||
{
|
{
|
||||||
if (ringloop->sqes_left() < PRIV(op)->wait_detail)
|
if (ringloop->space_left() < PRIV(op)->wait_detail)
|
||||||
{
|
{
|
||||||
// stop submission if there's still no free space
|
// stop submission if there's still no free space
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
#ifdef BLOCKSTORE_DEBUG
|
||||||
@@ -282,40 +250,13 @@ void blockstore_impl_t::check_wait(blockstore_op_t *op)
|
|||||||
}
|
}
|
||||||
PRIV(op)->wait_for = 0;
|
PRIV(op)->wait_for = 0;
|
||||||
}
|
}
|
||||||
else if (PRIV(op)->wait_for == WAIT_JOURNAL)
|
else if (PRIV(op)->wait_for == WAIT_COMPACTION)
|
||||||
{
|
{
|
||||||
if (journal.used_start == PRIV(op)->wait_detail &&
|
if (heap->get_compacted_count() <= PRIV(op)->wait_detail)
|
||||||
(unstable_writes.size()+unstable_unsynced) == PRIV(op)->wait_detail2)
|
|
||||||
{
|
{
|
||||||
// do not submit
|
// do not submit
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
#ifdef BLOCKSTORE_DEBUG
|
||||||
printf("Still waiting to flush journal offset %08jx\n", PRIV(op)->wait_detail);
|
printf("Still waiting for more flushes\n");
|
||||||
#endif
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
flusher->release_trim();
|
|
||||||
PRIV(op)->wait_for = 0;
|
|
||||||
}
|
|
||||||
else if (PRIV(op)->wait_for == WAIT_JOURNAL_BUFFER)
|
|
||||||
{
|
|
||||||
int next = ((journal.cur_sector + 1) % journal.sector_count);
|
|
||||||
if (journal.sector_info[next].flush_count > 0 ||
|
|
||||||
journal.sector_info[next].dirty)
|
|
||||||
{
|
|
||||||
// do not submit
|
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
|
||||||
printf("Still waiting for a journal buffer\n");
|
|
||||||
#endif
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
PRIV(op)->wait_for = 0;
|
|
||||||
}
|
|
||||||
else if (PRIV(op)->wait_for == WAIT_FREE)
|
|
||||||
{
|
|
||||||
if (!data_alloc->get_free_count() && big_to_flush > 0)
|
|
||||||
{
|
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
|
||||||
printf("Still waiting for free space on the data device\n");
|
|
||||||
#endif
|
#endif
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
@@ -334,7 +275,8 @@ void blockstore_impl_t::enqueue_op(blockstore_op_t *op)
|
|||||||
((op->opcode == BS_OP_READ || op->opcode == BS_OP_WRITE || op->opcode == BS_OP_WRITE_STABLE) && (
|
((op->opcode == BS_OP_READ || op->opcode == BS_OP_WRITE || op->opcode == BS_OP_WRITE_STABLE) && (
|
||||||
op->offset >= dsk.data_block_size ||
|
op->offset >= dsk.data_block_size ||
|
||||||
op->len > dsk.data_block_size-op->offset ||
|
op->len > dsk.data_block_size-op->offset ||
|
||||||
(op->len % dsk.disk_alignment)
|
(op->offset % dsk.bitmap_granularity) ||
|
||||||
|
(op->len % dsk.bitmap_granularity)
|
||||||
)) ||
|
)) ||
|
||||||
readonly && op->opcode != BS_OP_READ && op->opcode != BS_OP_LIST)
|
readonly && op->opcode != BS_OP_READ && op->opcode != BS_OP_LIST)
|
||||||
{
|
{
|
||||||
@@ -361,75 +303,11 @@ void blockstore_impl_t::init_op(blockstore_op_t *op)
|
|||||||
{
|
{
|
||||||
// Call constructor without allocating memory. We'll call destructor before returning op back
|
// Call constructor without allocating memory. We'll call destructor before returning op back
|
||||||
new ((void*)op->private_data) blockstore_op_private_t;
|
new ((void*)op->private_data) blockstore_op_private_t;
|
||||||
PRIV(op)->min_flushed_journal_sector = PRIV(op)->max_flushed_journal_sector = 0;
|
|
||||||
PRIV(op)->wait_for = 0;
|
PRIV(op)->wait_for = 0;
|
||||||
PRIV(op)->op_state = 0;
|
PRIV(op)->op_state = 0;
|
||||||
PRIV(op)->pending_ops = 0;
|
PRIV(op)->pending_ops = 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
static bool replace_stable(object_id oid, uint64_t version, int search_start, int search_end, obj_ver_id* list)
|
|
||||||
{
|
|
||||||
while (search_start < search_end)
|
|
||||||
{
|
|
||||||
int pos = search_start+(search_end-search_start)/2;
|
|
||||||
if (oid < list[pos].oid)
|
|
||||||
{
|
|
||||||
search_end = pos;
|
|
||||||
}
|
|
||||||
else if (list[pos].oid < oid)
|
|
||||||
{
|
|
||||||
search_start = pos+1;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
list[pos].version = version;
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
blockstore_clean_db_t& blockstore_impl_t::clean_db_shard(object_id oid)
|
|
||||||
{
|
|
||||||
uint64_t pg_num = 0;
|
|
||||||
uint64_t pool_id = (oid.inode >> (64-POOL_ID_BITS));
|
|
||||||
auto sh_it = clean_db_settings.find(pool_id);
|
|
||||||
if (sh_it != clean_db_settings.end())
|
|
||||||
{
|
|
||||||
// like map_to_pg()
|
|
||||||
pg_num = (oid.stripe / sh_it->second.pg_stripe_size) % sh_it->second.pg_count + 1;
|
|
||||||
}
|
|
||||||
return clean_db_shards[(pool_id << (64-POOL_ID_BITS)) | pg_num];
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_impl_t::reshard_clean_db(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size)
|
|
||||||
{
|
|
||||||
uint64_t pool_id = (uint64_t)pool;
|
|
||||||
std::map<pool_pg_id_t, blockstore_clean_db_t> new_shards;
|
|
||||||
auto sh_it = clean_db_shards.lower_bound((pool_id << (64-POOL_ID_BITS)));
|
|
||||||
while (sh_it != clean_db_shards.end() &&
|
|
||||||
(sh_it->first >> (64-POOL_ID_BITS)) == pool_id)
|
|
||||||
{
|
|
||||||
for (auto & pair: sh_it->second)
|
|
||||||
{
|
|
||||||
// like map_to_pg()
|
|
||||||
uint64_t pg_num = (pair.first.stripe / pg_stripe_size) % pg_count + 1;
|
|
||||||
uint64_t shard_id = (pool_id << (64-POOL_ID_BITS)) | pg_num;
|
|
||||||
new_shards[shard_id][pair.first] = pair.second;
|
|
||||||
}
|
|
||||||
clean_db_shards.erase(sh_it++);
|
|
||||||
}
|
|
||||||
for (sh_it = new_shards.begin(); sh_it != new_shards.end(); sh_it++)
|
|
||||||
{
|
|
||||||
auto & to = clean_db_shards[sh_it->first];
|
|
||||||
to.swap(sh_it->second);
|
|
||||||
}
|
|
||||||
clean_db_settings[pool_id] = (pool_shard_settings_t){
|
|
||||||
.pg_count = pg_count,
|
|
||||||
.pg_stripe_size = pg_stripe_size,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_impl_t::process_list(blockstore_op_t *op)
|
void blockstore_impl_t::process_list(blockstore_op_t *op)
|
||||||
{
|
{
|
||||||
uint32_t list_pg = op->pg_number+1;
|
uint32_t list_pg = op->pg_number+1;
|
||||||
@@ -438,258 +316,58 @@ void blockstore_impl_t::process_list(blockstore_op_t *op)
|
|||||||
uint64_t min_inode = op->min_oid.inode;
|
uint64_t min_inode = op->min_oid.inode;
|
||||||
uint64_t max_inode = op->max_oid.inode;
|
uint64_t max_inode = op->max_oid.inode;
|
||||||
// Check PG
|
// Check PG
|
||||||
if (pg_count != 0 && (pg_stripe_size < MIN_DATA_BLOCK_SIZE || list_pg > pg_count))
|
if (!pg_count || (pg_stripe_size < MIN_DATA_BLOCK_SIZE || list_pg > pg_count) ||
|
||||||
|
!INODE_POOL(min_inode) || INODE_POOL(min_inode) != INODE_POOL(max_inode))
|
||||||
{
|
{
|
||||||
op->retval = -EINVAL;
|
op->retval = -EINVAL;
|
||||||
FINISH_OP(op);
|
FINISH_OP(op);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
// Check if the DB needs resharding
|
// Check if the DB is sharded correctly
|
||||||
// (we don't know about PGs from the beginning, we only create "shards" here)
|
if (!heap->reshard_check(INODE_POOL(min_inode), pg_count, pg_stripe_size))
|
||||||
uint64_t first_shard = 0, last_shard = UINT64_MAX;
|
|
||||||
if (min_inode != 0 &&
|
|
||||||
// Check if min_inode == max_inode == pool_id<<N, i.e. this is a pool listing
|
|
||||||
(min_inode >> (64-POOL_ID_BITS)) == (max_inode >> (64-POOL_ID_BITS)))
|
|
||||||
{
|
{
|
||||||
pool_id_t pool_id = (min_inode >> (64-POOL_ID_BITS));
|
op->retval = -EAGAIN;
|
||||||
if (pg_count > 1)
|
|
||||||
{
|
|
||||||
// Per-pg listing
|
|
||||||
auto sh_it = clean_db_settings.find(pool_id);
|
|
||||||
if (sh_it == clean_db_settings.end() ||
|
|
||||||
sh_it->second.pg_count != pg_count ||
|
|
||||||
sh_it->second.pg_stripe_size != pg_stripe_size)
|
|
||||||
{
|
|
||||||
reshard_clean_db(pool_id, pg_count, pg_stripe_size);
|
|
||||||
}
|
|
||||||
first_shard = last_shard = ((uint64_t)pool_id << (64-POOL_ID_BITS)) | list_pg;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// Per-pool listing
|
|
||||||
first_shard = ((uint64_t)pool_id << (64-POOL_ID_BITS));
|
|
||||||
last_shard = ((uint64_t)(pool_id+1) << (64-POOL_ID_BITS)) - 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Copy clean_db entries
|
|
||||||
int stable_count = 0, stable_alloc = 0;
|
|
||||||
if (min_inode != max_inode)
|
|
||||||
{
|
|
||||||
for (auto shard_it = clean_db_shards.lower_bound(first_shard);
|
|
||||||
shard_it != clean_db_shards.end() && shard_it->first <= last_shard;
|
|
||||||
shard_it++)
|
|
||||||
{
|
|
||||||
auto & clean_db = shard_it->second;
|
|
||||||
stable_alloc += clean_db.size();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (op->list_stable_limit > 0)
|
|
||||||
{
|
|
||||||
stable_alloc = op->list_stable_limit;
|
|
||||||
if (stable_alloc > 1024*1024)
|
|
||||||
stable_alloc = 1024*1024;
|
|
||||||
}
|
|
||||||
if (stable_alloc < 32768)
|
|
||||||
{
|
|
||||||
stable_alloc = 32768;
|
|
||||||
}
|
|
||||||
obj_ver_id *stable = (obj_ver_id*)malloc(sizeof(obj_ver_id) * stable_alloc);
|
|
||||||
if (!stable)
|
|
||||||
{
|
|
||||||
op->retval = -ENOMEM;
|
|
||||||
FINISH_OP(op);
|
FINISH_OP(op);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
auto max_oid = op->max_oid;
|
obj_ver_id *result = NULL;
|
||||||
bool limited = false;
|
size_t stable_count = 0, unstable_count = 0;
|
||||||
pool_pg_id_t last_shard_id = 0;
|
int res = heap->list_objects(list_pg, op->min_oid, op->max_oid, &result, &stable_count, &unstable_count);
|
||||||
for (auto shard_it = clean_db_shards.lower_bound(first_shard);
|
if (op->list_stable_limit)
|
||||||
shard_it != clean_db_shards.end() && shard_it->first <= last_shard;
|
|
||||||
shard_it++)
|
|
||||||
{
|
{
|
||||||
auto & clean_db = shard_it->second;
|
// Ordered result is expected - used by scrub
|
||||||
auto clean_it = clean_db.begin(), clean_end = clean_db.end();
|
// We use an unordered map
|
||||||
if (op->min_oid.inode != 0 || op->min_oid.stripe != 0)
|
std::sort(result, result + stable_count);
|
||||||
|
if (stable_count > op->list_stable_limit)
|
||||||
{
|
{
|
||||||
clean_it = clean_db.lower_bound(op->min_oid);
|
memmove(result + op->list_stable_limit, result + stable_count, unstable_count);
|
||||||
}
|
stable_count = op->list_stable_limit;
|
||||||
if ((max_oid.inode != 0 || max_oid.stripe != 0) && !(max_oid < op->min_oid))
|
|
||||||
{
|
|
||||||
clean_end = clean_db.upper_bound(max_oid);
|
|
||||||
}
|
|
||||||
for (; clean_it != clean_end; clean_it++)
|
|
||||||
{
|
|
||||||
if (stable_count >= stable_alloc)
|
|
||||||
{
|
|
||||||
stable_alloc *= 2;
|
|
||||||
obj_ver_id* nst = (obj_ver_id*)realloc(stable, sizeof(obj_ver_id) * stable_alloc);
|
|
||||||
if (!nst)
|
|
||||||
{
|
|
||||||
op->retval = -ENOMEM;
|
|
||||||
FINISH_OP(op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
stable = nst;
|
|
||||||
}
|
|
||||||
stable[stable_count++] = {
|
|
||||||
.oid = clean_it->first,
|
|
||||||
.version = clean_it->second.version,
|
|
||||||
};
|
|
||||||
if (op->list_stable_limit > 0 && stable_count >= op->list_stable_limit)
|
|
||||||
{
|
|
||||||
if (!limited)
|
|
||||||
{
|
|
||||||
limited = true;
|
|
||||||
max_oid = stable[stable_count-1].oid;
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (op->list_stable_limit > 0)
|
|
||||||
{
|
|
||||||
// To maintain the order, we have to include objects in the same range from other shards
|
|
||||||
if (last_shard_id != 0 && last_shard_id != shard_it->first)
|
|
||||||
std::sort(stable, stable+stable_count);
|
|
||||||
if (stable_count > op->list_stable_limit)
|
|
||||||
stable_count = op->list_stable_limit;
|
|
||||||
}
|
|
||||||
last_shard_id = shard_it->first;
|
|
||||||
}
|
|
||||||
if (op->list_stable_limit == 0 && first_shard != last_shard)
|
|
||||||
{
|
|
||||||
// If that's not a per-PG listing, sort clean entries (already sorted if list_stable_limit != 0)
|
|
||||||
std::sort(stable, stable+stable_count);
|
|
||||||
}
|
|
||||||
int clean_stable_count = stable_count;
|
|
||||||
// Copy dirty_db entries (sorted, too)
|
|
||||||
int unstable_count = 0, unstable_alloc = 0;
|
|
||||||
obj_ver_id *unstable = NULL;
|
|
||||||
{
|
|
||||||
auto dirty_it = dirty_db.begin(), dirty_end = dirty_db.end();
|
|
||||||
if (op->min_oid.inode != 0 || op->min_oid.stripe != 0)
|
|
||||||
{
|
|
||||||
dirty_it = dirty_db.lower_bound({
|
|
||||||
.oid = op->min_oid,
|
|
||||||
.version = 0,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
if ((max_oid.inode != 0 || max_oid.stripe != 0) && !(max_oid < op->min_oid))
|
|
||||||
{
|
|
||||||
dirty_end = dirty_db.upper_bound({
|
|
||||||
.oid = max_oid,
|
|
||||||
.version = UINT64_MAX,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
for (; dirty_it != dirty_end; dirty_it++)
|
|
||||||
{
|
|
||||||
if (!pg_count || ((dirty_it->first.oid.stripe / pg_stripe_size) % pg_count + 1) == list_pg) // like map_to_pg()
|
|
||||||
{
|
|
||||||
if (IS_DELETE(dirty_it->second.state))
|
|
||||||
{
|
|
||||||
// Deletions are always stable, so try to zero out two possible entries
|
|
||||||
if (!replace_stable(dirty_it->first.oid, 0, 0, clean_stable_count, stable))
|
|
||||||
{
|
|
||||||
replace_stable(dirty_it->first.oid, 0, clean_stable_count, stable_count, stable);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else if (IS_STABLE(dirty_it->second.state) || (dirty_it->second.state & BS_ST_INSTANT))
|
|
||||||
{
|
|
||||||
// First try to replace a clean stable version in the first part of the list
|
|
||||||
if (!replace_stable(dirty_it->first.oid, dirty_it->first.version, 0, clean_stable_count, stable))
|
|
||||||
{
|
|
||||||
// Then try to replace the last dirty stable version in the second part of the list
|
|
||||||
if (stable_count > 0 && stable[stable_count-1].oid == dirty_it->first.oid)
|
|
||||||
{
|
|
||||||
stable[stable_count-1].version = dirty_it->first.version;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
if (stable_count >= stable_alloc)
|
|
||||||
{
|
|
||||||
stable_alloc += 32768;
|
|
||||||
obj_ver_id *nst = (obj_ver_id*)realloc(stable, sizeof(obj_ver_id) * stable_alloc);
|
|
||||||
if (!nst)
|
|
||||||
{
|
|
||||||
if (unstable)
|
|
||||||
free(unstable);
|
|
||||||
op->retval = -ENOMEM;
|
|
||||||
FINISH_OP(op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
stable = nst;
|
|
||||||
}
|
|
||||||
stable[stable_count++] = dirty_it->first;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (op->list_stable_limit > 0 && stable_count >= op->list_stable_limit)
|
|
||||||
{
|
|
||||||
// Stop here
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
if (unstable_count >= unstable_alloc)
|
|
||||||
{
|
|
||||||
unstable_alloc += 32768;
|
|
||||||
obj_ver_id *nst = (obj_ver_id*)realloc(unstable, sizeof(obj_ver_id) * unstable_alloc);
|
|
||||||
if (!nst)
|
|
||||||
{
|
|
||||||
if (stable)
|
|
||||||
free(stable);
|
|
||||||
op->retval = -ENOMEM;
|
|
||||||
FINISH_OP(op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
unstable = nst;
|
|
||||||
}
|
|
||||||
unstable[unstable_count++] = dirty_it->first;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Remove zeroed out stable entries
|
|
||||||
int j = 0;
|
|
||||||
for (int i = 0; i < stable_count; i++)
|
|
||||||
{
|
|
||||||
if (stable[i].version != 0)
|
|
||||||
{
|
|
||||||
stable[j++] = stable[i];
|
|
||||||
}
|
|
||||||
}
|
|
||||||
stable_count = j;
|
|
||||||
if (stable_count+unstable_count > stable_alloc)
|
|
||||||
{
|
|
||||||
stable_alloc = stable_count+unstable_count;
|
|
||||||
obj_ver_id *nst = (obj_ver_id*)realloc(stable, sizeof(obj_ver_id) * stable_alloc);
|
|
||||||
if (!nst)
|
|
||||||
{
|
|
||||||
if (unstable)
|
|
||||||
free(unstable);
|
|
||||||
op->retval = -ENOMEM;
|
|
||||||
FINISH_OP(op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
stable = nst;
|
|
||||||
}
|
|
||||||
// Copy unstable entries
|
|
||||||
for (int i = 0; i < unstable_count; i++)
|
|
||||||
{
|
|
||||||
stable[j++] = unstable[i];
|
|
||||||
}
|
|
||||||
free(unstable);
|
|
||||||
op->version = stable_count;
|
op->version = stable_count;
|
||||||
op->retval = stable_count+unstable_count;
|
op->retval = res == 0 ? stable_count+unstable_count : -res;
|
||||||
op->buf = stable;
|
op->buf = (uint8_t*)result;
|
||||||
FINISH_OP(op);
|
FINISH_OP(op);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void blockstore_impl_t::set_no_inode_stats(const std::vector<uint64_t> & pool_ids)
|
||||||
|
{
|
||||||
|
heap->set_no_inode_stats(pool_ids);
|
||||||
|
}
|
||||||
|
|
||||||
void blockstore_impl_t::dump_diagnostics()
|
void blockstore_impl_t::dump_diagnostics()
|
||||||
{
|
{
|
||||||
journal.dump_diagnostics();
|
|
||||||
flusher->dump_diagnostics();
|
flusher->dump_diagnostics();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void blockstore_meta_header_v3_t::set_crc32c()
|
||||||
|
{
|
||||||
|
header_csum = 0;
|
||||||
|
uint32_t calc = crc32c(0, this, version == BLOCKSTORE_META_FORMAT_HEAP
|
||||||
|
? sizeof(blockstore_meta_header_v3_t) : sizeof(blockstore_meta_header_v2_t));
|
||||||
|
header_csum = calc;
|
||||||
|
}
|
||||||
|
|
||||||
void blockstore_impl_t::disk_error_abort(const char *op, int retval, int expected)
|
void blockstore_impl_t::disk_error_abort(const char *op, int retval, int expected)
|
||||||
{
|
{
|
||||||
if (retval == -EAGAIN)
|
if (retval == -EAGAIN)
|
||||||
@@ -703,85 +381,33 @@ void blockstore_impl_t::disk_error_abort(const char *op, int retval, int expecte
|
|||||||
exit(1);
|
exit(1);
|
||||||
}
|
}
|
||||||
|
|
||||||
void blockstore_impl_t::set_no_inode_stats(const std::vector<uint64_t> & pool_ids)
|
uint64_t blockstore_impl_t::get_free_block_count()
|
||||||
{
|
{
|
||||||
for (auto & np: no_inode_stats)
|
return dsk.block_count - heap->get_data_used_space()/dsk.data_block_size;
|
||||||
{
|
|
||||||
np.second = 2;
|
|
||||||
}
|
|
||||||
for (auto pool_id: pool_ids)
|
|
||||||
{
|
|
||||||
if (!no_inode_stats[pool_id])
|
|
||||||
recalc_inode_space_stats(pool_id, false);
|
|
||||||
no_inode_stats[pool_id] = 1;
|
|
||||||
}
|
|
||||||
for (auto np_it = no_inode_stats.begin(); np_it != no_inode_stats.end(); )
|
|
||||||
{
|
|
||||||
if (np_it->second == 2)
|
|
||||||
{
|
|
||||||
recalc_inode_space_stats(np_it->first, true);
|
|
||||||
no_inode_stats.erase(np_it++);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
np_it++;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
void blockstore_impl_t::recalc_inode_space_stats(uint64_t pool_id, bool per_inode)
|
std::string blockstore_impl_t::get_op_diag(blockstore_op_t *op)
|
||||||
{
|
{
|
||||||
auto sp_begin = inode_space_stats.lower_bound((pool_id << (64-POOL_ID_BITS)));
|
char buf[256];
|
||||||
auto sp_end = inode_space_stats.lower_bound(((pool_id+1) << (64-POOL_ID_BITS)));
|
auto priv = PRIV(op);
|
||||||
inode_space_stats.erase(sp_begin, sp_end);
|
if (priv->wait_for)
|
||||||
auto sh_it = clean_db_shards.lower_bound((pool_id << (64-POOL_ID_BITS)));
|
snprintf(buf, sizeof(buf), "state=%d wait=%d (detail=%ju)", priv->op_state, priv->wait_for, priv->wait_detail);
|
||||||
while (sh_it != clean_db_shards.end() &&
|
else
|
||||||
(sh_it->first >> (64-POOL_ID_BITS)) == pool_id)
|
snprintf(buf, sizeof(buf), "state=%d", priv->op_state);
|
||||||
{
|
return std::string(buf);
|
||||||
for (auto & pair: sh_it->second)
|
}
|
||||||
{
|
|
||||||
uint64_t space_id = per_inode ? pair.first.inode : (pool_id << (64-POOL_ID_BITS));
|
void* blockstore_impl_t::reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit)
|
||||||
inode_space_stats[space_id] += dsk.data_block_size;
|
{
|
||||||
}
|
return heap->reshard_start(pool, pg_count, pg_stripe_size, chunk_limit);
|
||||||
sh_it++;
|
}
|
||||||
}
|
|
||||||
object_id last_oid = {};
|
bool blockstore_impl_t::reshard_continue(void *reshard_state, uint64_t chunk_limit)
|
||||||
bool last_exists = false;
|
{
|
||||||
auto dirty_it = dirty_db.lower_bound((obj_ver_id){ .oid = { .inode = (pool_id << (64-POOL_ID_BITS)) } });
|
return heap->reshard_continue(reshard_state, chunk_limit);
|
||||||
while (dirty_it != dirty_db.end() && (dirty_it->first.oid.inode >> (64-POOL_ID_BITS)) == pool_id)
|
}
|
||||||
{
|
|
||||||
if (IS_STABLE(dirty_it->second.state) && (IS_BIG_WRITE(dirty_it->second.state) || IS_DELETE(dirty_it->second.state)))
|
void blockstore_impl_t::reshard_abort(void *reshard_state)
|
||||||
{
|
{
|
||||||
bool exists = false;
|
return heap->reshard_abort(reshard_state);
|
||||||
if (last_oid == dirty_it->first.oid)
|
|
||||||
{
|
|
||||||
exists = last_exists;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
auto & clean_db = clean_db_shard(dirty_it->first.oid);
|
|
||||||
auto clean_it = clean_db.find(dirty_it->first.oid);
|
|
||||||
exists = clean_it != clean_db.end();
|
|
||||||
}
|
|
||||||
uint64_t space_id = per_inode ? dirty_it->first.oid.inode : (pool_id << (64-POOL_ID_BITS));
|
|
||||||
if (IS_BIG_WRITE(dirty_it->second.state))
|
|
||||||
{
|
|
||||||
if (!exists)
|
|
||||||
inode_space_stats[space_id] += dsk.data_block_size;
|
|
||||||
last_exists = true;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
if (exists)
|
|
||||||
{
|
|
||||||
auto & sp = inode_space_stats[space_id];
|
|
||||||
if (sp > dsk.data_block_size)
|
|
||||||
sp -= dsk.data_block_size;
|
|
||||||
else
|
|
||||||
inode_space_stats.erase(space_id);
|
|
||||||
}
|
|
||||||
last_exists = false;
|
|
||||||
}
|
|
||||||
last_oid = dirty_it->first.oid;
|
|
||||||
}
|
|
||||||
dirty_it++;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -5,6 +5,8 @@
|
|||||||
|
|
||||||
#include "blockstore.h"
|
#include "blockstore.h"
|
||||||
#include "blockstore_disk.h"
|
#include "blockstore_disk.h"
|
||||||
|
#include "blockstore_heap.h"
|
||||||
|
#include "ondisk_formats.h"
|
||||||
|
|
||||||
#include <sys/types.h>
|
#include <sys/types.h>
|
||||||
#include <sys/ioctl.h>
|
#include <sys/ioctl.h>
|
||||||
@@ -19,241 +21,68 @@
|
|||||||
#include <deque>
|
#include <deque>
|
||||||
#include <new>
|
#include <new>
|
||||||
#include <unordered_map>
|
#include <unordered_map>
|
||||||
|
#include <unordered_set>
|
||||||
#include "cpp-btree/btree_map.h"
|
|
||||||
|
|
||||||
#include "malloc_or_die.h"
|
#include "malloc_or_die.h"
|
||||||
#include "allocator.h"
|
|
||||||
|
class blockstore_impl_t;
|
||||||
|
|
||||||
//#define BLOCKSTORE_DEBUG
|
//#define BLOCKSTORE_DEBUG
|
||||||
|
|
||||||
// States are not stored on disk. Instead, they're deduced from the journal
|
|
||||||
|
|
||||||
#define BS_ST_SMALL_WRITE 0x01
|
|
||||||
#define BS_ST_BIG_WRITE 0x02
|
|
||||||
#define BS_ST_DELETE 0x03
|
|
||||||
|
|
||||||
#define BS_ST_WAIT_DEL 0x10
|
|
||||||
#define BS_ST_WAIT_BIG 0x20
|
|
||||||
#define BS_ST_IN_FLIGHT 0x30
|
|
||||||
#define BS_ST_SUBMITTED 0x40
|
|
||||||
#define BS_ST_WRITTEN 0x50
|
|
||||||
#define BS_ST_SYNCED 0x60
|
|
||||||
#define BS_ST_STABLE 0x70
|
|
||||||
|
|
||||||
#define BS_ST_INSTANT 0x100
|
|
||||||
|
|
||||||
#define IMMEDIATE_NONE 0
|
|
||||||
#define IMMEDIATE_SMALL 1
|
|
||||||
#define IMMEDIATE_ALL 2
|
|
||||||
|
|
||||||
#define BS_ST_TYPE_MASK 0x0F
|
|
||||||
#define BS_ST_WORKFLOW_MASK 0xF0
|
|
||||||
#define IS_IN_FLIGHT(st) (((st) & 0xF0) <= BS_ST_SUBMITTED)
|
|
||||||
#define IS_STABLE(st) (((st) & 0xF0) == BS_ST_STABLE)
|
|
||||||
#define IS_SYNCED(st) (((st) & 0xF0) >= BS_ST_SYNCED)
|
|
||||||
#define IS_JOURNAL(st) (((st) & 0x0F) == BS_ST_SMALL_WRITE)
|
|
||||||
#define IS_BIG_WRITE(st) (((st) & 0x0F) == BS_ST_BIG_WRITE)
|
|
||||||
#define IS_DELETE(st) (((st) & 0x0F) == BS_ST_DELETE)
|
|
||||||
#define IS_INSTANT(st) (((st) & BS_ST_TYPE_MASK) == BS_ST_DELETE || ((st) & BS_ST_INSTANT))
|
|
||||||
|
|
||||||
#define BS_SUBMIT_CHECK_SQES(n) \
|
|
||||||
if (ringloop->sqes_left() < (n))\
|
|
||||||
{\
|
|
||||||
/* Pause until there are more requests available */\
|
|
||||||
PRIV(op)->wait_detail = (n);\
|
|
||||||
PRIV(op)->wait_for = WAIT_SQE;\
|
|
||||||
return 0;\
|
|
||||||
}
|
|
||||||
|
|
||||||
#define BS_SUBMIT_GET_SQE(sqe, data) \
|
|
||||||
BS_SUBMIT_GET_ONLY_SQE(sqe); \
|
|
||||||
struct ring_data_t *data = ((ring_data_t*)sqe->user_data)
|
|
||||||
|
|
||||||
#define BS_SUBMIT_GET_ONLY_SQE(sqe) \
|
|
||||||
struct io_uring_sqe *sqe = get_sqe();\
|
|
||||||
if (!sqe)\
|
|
||||||
{\
|
|
||||||
/* Pause until there are more requests available */\
|
|
||||||
PRIV(op)->wait_detail = 1;\
|
|
||||||
PRIV(op)->wait_for = WAIT_SQE;\
|
|
||||||
return 0;\
|
|
||||||
}
|
|
||||||
|
|
||||||
#define BS_SUBMIT_GET_SQE_DECL(sqe) \
|
|
||||||
sqe = get_sqe();\
|
|
||||||
if (!sqe)\
|
|
||||||
{\
|
|
||||||
/* Pause until there are more requests available */\
|
|
||||||
PRIV(op)->wait_detail = 1;\
|
|
||||||
PRIV(op)->wait_for = WAIT_SQE;\
|
|
||||||
return 0;\
|
|
||||||
}
|
|
||||||
|
|
||||||
#include "blockstore_journal.h"
|
|
||||||
|
|
||||||
// "VITAstor"
|
|
||||||
#define BLOCKSTORE_META_MAGIC_V1 0x726F747341544956l
|
|
||||||
#define BLOCKSTORE_META_FORMAT_V1 1
|
|
||||||
#define BLOCKSTORE_META_FORMAT_V2 2
|
|
||||||
|
|
||||||
// metadata header (superblock)
|
|
||||||
struct __attribute__((__packed__)) blockstore_meta_header_v1_t
|
|
||||||
{
|
|
||||||
uint64_t zero;
|
|
||||||
uint64_t magic;
|
|
||||||
uint64_t version;
|
|
||||||
uint32_t meta_block_size;
|
|
||||||
uint32_t data_block_size;
|
|
||||||
uint32_t bitmap_granularity;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct __attribute__((__packed__)) blockstore_meta_header_v2_t
|
|
||||||
{
|
|
||||||
uint64_t zero;
|
|
||||||
uint64_t magic;
|
|
||||||
uint64_t version;
|
|
||||||
uint32_t meta_block_size;
|
|
||||||
uint32_t data_block_size;
|
|
||||||
uint32_t bitmap_granularity;
|
|
||||||
uint32_t data_csum_type;
|
|
||||||
uint32_t csum_block_size;
|
|
||||||
uint32_t header_csum;
|
|
||||||
};
|
|
||||||
|
|
||||||
// 32 bytes = 24 bytes + block bitmap (4 bytes by default) + external attributes (also bitmap, 4 bytes by default)
|
|
||||||
// per "clean" entry on disk with fixed metadata tables
|
|
||||||
struct __attribute__((__packed__)) clean_disk_entry
|
|
||||||
{
|
|
||||||
object_id oid;
|
|
||||||
uint64_t version;
|
|
||||||
uint8_t bitmap[];
|
|
||||||
// Two more fields come after bitmap in metadata version 2:
|
|
||||||
// uint32_t data_csum[];
|
|
||||||
// uint32_t entry_csum;
|
|
||||||
};
|
|
||||||
|
|
||||||
// 32 = 16 + 16 bytes per "clean" entry in memory (object_id => clean_entry)
|
|
||||||
struct __attribute__((__packed__)) clean_entry
|
|
||||||
{
|
|
||||||
uint64_t version;
|
|
||||||
uint64_t location;
|
|
||||||
};
|
|
||||||
|
|
||||||
// 64 = 24 + 40 bytes per dirty entry in memory (obj_ver_id => dirty_entry). Plus checksums
|
|
||||||
struct __attribute__((__packed__)) dirty_entry
|
|
||||||
{
|
|
||||||
uint32_t state;
|
|
||||||
uint32_t flags; // unneeded, but present for alignment
|
|
||||||
uint64_t location; // location in either journal or data -> in BYTES
|
|
||||||
uint32_t offset; // data offset within object (stripe)
|
|
||||||
uint32_t len; // data length
|
|
||||||
uint64_t journal_sector; // journal sector used for this entry
|
|
||||||
void* dyn_data; // dynamic data: external bitmap and data block checksums. may be a pointer to the in-memory journal
|
|
||||||
};
|
|
||||||
|
|
||||||
// - Sync must be submitted after previous writes/deletes (not before!)
|
|
||||||
// - Reads to the same object must be submitted after previous writes/deletes
|
|
||||||
// are written (not necessarily synced) in their location. This is because we
|
|
||||||
// rely on read-modify-write for erasure coding and we must return new data
|
|
||||||
// to calculate parity for subsequent writes
|
|
||||||
// - Writes may be submitted in any order, because they don't overlap. Each write
|
|
||||||
// goes into a new location - either on the journal device or on the data device
|
|
||||||
// - Stable (stabilize) must be submitted after sync of that object is completed
|
|
||||||
// It's even OK to return an error to the caller if that object is not synced yet
|
|
||||||
// - Journal trim may be processed only after all versions are moved to
|
|
||||||
// the main storage AND after all read operations for older versions complete
|
|
||||||
// - If an operation can not be submitted because the ring is full
|
|
||||||
// we should stop submission of other operations. Otherwise some "scatter" reads
|
|
||||||
// may end up blocked for a long time.
|
|
||||||
// Otherwise, the submit order is free, that is all operations may be submitted immediately
|
|
||||||
// In fact, adding a write operation must immediately result in dirty_db being populated
|
|
||||||
|
|
||||||
// Suspend operation until there are more free SQEs
|
|
||||||
#define WAIT_SQE 1
|
|
||||||
// Suspend operation until there are <wait_detail> bytes of free space in the journal on disk
|
|
||||||
#define WAIT_JOURNAL 3
|
|
||||||
// Suspend operation until the next journal sector buffer is free
|
|
||||||
#define WAIT_JOURNAL_BUFFER 4
|
|
||||||
// Suspend operation until there is some free space on the data device
|
|
||||||
#define WAIT_FREE 5
|
|
||||||
|
|
||||||
struct used_clean_obj_t
|
|
||||||
{
|
|
||||||
int refs;
|
|
||||||
bool was_freed; // was freed by a parallel flush?
|
|
||||||
bool was_changed; // was changed by a parallel flush?
|
|
||||||
};
|
|
||||||
|
|
||||||
// https://github.com/algorithm-ninja/cpp-btree
|
|
||||||
// https://github.com/greg7mdp/sparsepp/ was used previously, but it was TERRIBLY slow after resizing
|
|
||||||
// with sparsepp, random reads dropped to ~700 iops very fast with just as much as ~32k objects in the DB
|
|
||||||
typedef btree::btree_map<object_id, clean_entry> blockstore_clean_db_t;
|
|
||||||
typedef std::map<obj_ver_id, dirty_entry> blockstore_dirty_db_t;
|
|
||||||
|
|
||||||
#include "blockstore_init.h"
|
#include "blockstore_init.h"
|
||||||
|
|
||||||
#include "blockstore_flush.h"
|
#include "blockstore_flush.h"
|
||||||
|
|
||||||
#define PRIV(op) ((blockstore_op_private_t*)(op)->private_data)
|
|
||||||
#define FINISH_OP(op) PRIV(op)->~blockstore_op_private_t(); std::function<void (blockstore_op_t*)>(op->callback)(op)
|
|
||||||
|
|
||||||
struct blockstore_op_private_t
|
struct blockstore_op_private_t
|
||||||
{
|
{
|
||||||
// Wait status
|
// Wait status
|
||||||
int wait_for;
|
int wait_for;
|
||||||
uint64_t wait_detail, wait_detail2;
|
uint64_t wait_detail;
|
||||||
int pending_ops;
|
int pending_ops;
|
||||||
int op_state;
|
int op_state;
|
||||||
|
|
||||||
|
// Write, sync, stabilize
|
||||||
|
uint32_t modified_block, modified_block2;
|
||||||
|
|
||||||
// Read
|
// Read
|
||||||
uint64_t clean_block_used;
|
|
||||||
std::vector<copy_buffer_t> read_vec;
|
std::vector<copy_buffer_t> read_vec;
|
||||||
|
|
||||||
// Sync, write
|
// Read, write
|
||||||
uint64_t min_flushed_journal_sector, max_flushed_journal_sector;
|
uint64_t lsn;
|
||||||
|
|
||||||
|
// Write
|
||||||
|
uint64_t location;
|
||||||
|
uint32_t write_type;
|
||||||
|
|
||||||
|
// Stabilize, rollback
|
||||||
|
int stab_pos;
|
||||||
|
|
||||||
// Write
|
// Write
|
||||||
struct iovec iov_zerofill[3];
|
|
||||||
// Warning: must not have a default value here because it's written to before calling constructor in blockstore_write.cpp O_o
|
|
||||||
uint64_t real_version;
|
|
||||||
timespec tv_begin;
|
timespec tv_begin;
|
||||||
|
|
||||||
// Sync
|
|
||||||
std::vector<obj_ver_id> sync_big_writes, sync_small_writes;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef uint32_t pool_id_t;
|
struct bs_modified_block_t
|
||||||
typedef uint64_t pool_pg_id_t;
|
|
||||||
|
|
||||||
#define POOL_ID_BITS 16
|
|
||||||
|
|
||||||
struct pool_shard_settings_t
|
|
||||||
{
|
{
|
||||||
uint32_t pg_count;
|
bool sent;
|
||||||
uint32_t pg_stripe_size;
|
uint8_t *buf;
|
||||||
};
|
};
|
||||||
|
|
||||||
#define STAB_SPLIT_DONE 1
|
class blockstore_impl_t: public blockstore_i
|
||||||
#define STAB_SPLIT_WAIT 2
|
|
||||||
#define STAB_SPLIT_SYNC 3
|
|
||||||
#define STAB_SPLIT_TODO 4
|
|
||||||
|
|
||||||
class blockstore_impl_t
|
|
||||||
{
|
{
|
||||||
|
public:
|
||||||
blockstore_disk_t dsk;
|
blockstore_disk_t dsk;
|
||||||
|
|
||||||
/******* OPTIONS *******/
|
/******* OPTIONS *******/
|
||||||
bool readonly = false;
|
bool readonly = false;
|
||||||
// It is safe to disable fsync() if drive write cache is writethrough
|
|
||||||
bool disable_data_fsync = false, disable_meta_fsync = false, disable_journal_fsync = false;
|
|
||||||
// Enable if you want every operation to be executed with an "implicit fsync"
|
// Enable if you want every operation to be executed with an "implicit fsync"
|
||||||
// Suitable only for server SSDs with capacitors, requires disabled data and journal fsyncs
|
// Suitable only for server SSDs with capacitors, requires disabled data and journal fsyncs
|
||||||
int immediate_commit = IMMEDIATE_NONE;
|
int immediate_commit = IMMEDIATE_NONE;
|
||||||
bool inmemory_meta = false;
|
bool inmemory_meta = false;
|
||||||
|
uint32_t meta_write_recheck_parallelism = 0;
|
||||||
// Maximum and minimum flusher count
|
// Maximum and minimum flusher count
|
||||||
unsigned max_flusher_count, min_flusher_count;
|
unsigned max_flusher_count = 0, min_flusher_count = 0;
|
||||||
unsigned journal_trim_interval;
|
unsigned journal_trim_interval = 0;
|
||||||
|
unsigned flusher_start_threshold = 0;
|
||||||
// Maximum queue depth
|
// Maximum queue depth
|
||||||
unsigned max_write_iodepth = 128;
|
unsigned max_write_iodepth = 128;
|
||||||
// Enable small (journaled) write throttling, useful for the SSD+HDD case
|
// Enable small (journaled) write throttling, useful for the SSD+HDD case
|
||||||
@@ -268,143 +97,102 @@ class blockstore_impl_t
|
|||||||
uint64_t autosync_writes = 128;
|
uint64_t autosync_writes = 128;
|
||||||
// Log level (0-10)
|
// Log level (0-10)
|
||||||
int log_level = 0;
|
int log_level = 0;
|
||||||
|
// Enable correct block checksum validation on objects updated with small writes when checksum block
|
||||||
|
// is larger than bitmap_granularity, at the expense of extra metadata fsyncs during compaction
|
||||||
|
bool perfect_csum_update = false;
|
||||||
/******* END OF OPTIONS *******/
|
/******* END OF OPTIONS *******/
|
||||||
|
|
||||||
struct ring_consumer_t ring_consumer;
|
struct ring_consumer_t ring_consumer;
|
||||||
|
|
||||||
std::map<pool_id_t, pool_shard_settings_t> clean_db_settings;
|
blockstore_heap_t *heap = NULL;
|
||||||
std::map<pool_pg_id_t, blockstore_clean_db_t> clean_db_shards;
|
uint8_t* meta_superblock = NULL;
|
||||||
std::map<uint64_t, int> no_inode_stats;
|
uint8_t *buffer_area = NULL;
|
||||||
uint8_t *clean_bitmaps = NULL;
|
|
||||||
blockstore_dirty_db_t dirty_db;
|
|
||||||
std::vector<blockstore_op_t*> submit_queue;
|
std::vector<blockstore_op_t*> submit_queue;
|
||||||
std::vector<obj_ver_id> unsynced_big_writes, unsynced_small_writes;
|
int unsynced_data_write_count = 0, unsynced_buffer_write_count = 0, unsynced_meta_write_count = 0;
|
||||||
int unsynced_big_write_count = 0, unstable_unsynced = 0;
|
|
||||||
int unsynced_queued_ops = 0;
|
int unsynced_queued_ops = 0;
|
||||||
allocator_t *data_alloc = NULL;
|
|
||||||
uint64_t used_blocks = 0;
|
|
||||||
uint8_t *zero_object = NULL;
|
|
||||||
|
|
||||||
void *metadata_buffer = NULL;
|
std::vector<uint32_t> pending_modified_blocks;
|
||||||
|
robin_hood::unordered_flat_map<uint32_t, bs_modified_block_t> modified_blocks;
|
||||||
|
|
||||||
struct journal_t journal;
|
|
||||||
journal_flusher_t *flusher;
|
journal_flusher_t *flusher;
|
||||||
int big_to_flush = 0;
|
|
||||||
int write_iodepth = 0;
|
int write_iodepth = 0;
|
||||||
bool alloc_dyn_data = false;
|
int inflight_big = 0;
|
||||||
|
int intent_write_counter = 0;
|
||||||
// clean data blocks referenced by read operations
|
bool fsyncing_data = false;
|
||||||
std::map<uint64_t, used_clean_obj_t> used_clean_objects;
|
|
||||||
|
|
||||||
bool live = false, queue_stall = false;
|
bool live = false, queue_stall = false;
|
||||||
ring_loop_t *ringloop;
|
ring_loop_i *ringloop = NULL;
|
||||||
timerfd_manager_t *tfd;
|
timerfd_manager_t *tfd = NULL;
|
||||||
|
|
||||||
bool stop_sync_submitted;
|
bool stop_sync_submitted = false;
|
||||||
|
|
||||||
inline struct io_uring_sqe* get_sqe()
|
inline struct io_uring_sqe* get_sqe()
|
||||||
{
|
{
|
||||||
return ringloop->get_sqe();
|
return ringloop->get_sqe();
|
||||||
}
|
}
|
||||||
|
|
||||||
friend class blockstore_init_meta;
|
|
||||||
friend class blockstore_init_journal;
|
|
||||||
friend struct blockstore_journal_check_t;
|
|
||||||
friend class journal_flusher_t;
|
|
||||||
friend class journal_flusher_co;
|
|
||||||
|
|
||||||
void calc_lengths();
|
|
||||||
void open_data();
|
void open_data();
|
||||||
void open_meta();
|
void open_meta();
|
||||||
void open_journal();
|
void open_journal();
|
||||||
uint8_t* get_clean_entry_bitmap(uint64_t block_loc, int offset);
|
|
||||||
|
|
||||||
blockstore_clean_db_t& clean_db_shard(object_id oid);
|
|
||||||
void reshard_clean_db(pool_id_t pool_id, uint32_t pg_count, uint32_t pg_stripe_size);
|
|
||||||
void recalc_inode_space_stats(uint64_t pool_id, bool per_inode);
|
|
||||||
|
|
||||||
// Journaling
|
|
||||||
void prepare_journal_sector_write(int sector, blockstore_op_t *op);
|
|
||||||
void handle_journal_write(ring_data_t *data, uint64_t flush_id);
|
|
||||||
void disk_error_abort(const char *op, int retval, int expected);
|
void disk_error_abort(const char *op, int retval, int expected);
|
||||||
|
|
||||||
// Asynchronous init
|
// Asynchronous init
|
||||||
int initialized;
|
int initialized;
|
||||||
int metadata_buf_size;
|
int metadata_buf_size;
|
||||||
blockstore_init_meta* metadata_init_reader;
|
blockstore_init_meta* metadata_init_reader;
|
||||||
blockstore_init_journal* journal_init_reader;
|
|
||||||
|
|
||||||
void check_wait(blockstore_op_t *op);
|
void check_wait(blockstore_op_t *op);
|
||||||
void init_op(blockstore_op_t *op);
|
void init_op(blockstore_op_t *op);
|
||||||
|
|
||||||
// Read
|
// Read
|
||||||
int dequeue_read(blockstore_op_t *read_op);
|
int dequeue_read(blockstore_op_t *op);
|
||||||
|
int fulfill_read(blockstore_op_t *op);
|
||||||
|
uint32_t prepare_read(std::vector<copy_buffer_t> & read_vec, heap_entry_t *obj, heap_entry_t *wr, uint32_t start, uint32_t end, uint32_t skip_csum);
|
||||||
|
uint32_t prepare_read_with_bitmaps(std::vector<copy_buffer_t> & read_vec, heap_entry_t *obj, heap_entry_t *wr, uint32_t start, uint32_t end, uint32_t skip_csum);
|
||||||
|
uint32_t prepare_read_zero(std::vector<copy_buffer_t> & read_vec, uint32_t start, uint32_t end);
|
||||||
|
uint32_t prepare_read_simple(std::vector<copy_buffer_t> & read_vec, heap_entry_t *obj, heap_entry_t *wr, uint32_t start, uint32_t end, uint32_t skip_csum);
|
||||||
|
void prepare_disk_read(std::vector<copy_buffer_t> & read_vec, int pos, heap_entry_t *obj, heap_entry_t *wr,
|
||||||
|
uint32_t blk_start, uint32_t blk_end, uint32_t start, uint32_t end, uint32_t copy_flags);
|
||||||
void find_holes(std::vector<copy_buffer_t> & read_vec, uint32_t item_start, uint32_t item_end,
|
void find_holes(std::vector<copy_buffer_t> & read_vec, uint32_t item_start, uint32_t item_end,
|
||||||
std::function<int(int, bool, uint32_t, uint32_t)> callback);
|
std::function<void(int&, uint32_t, uint32_t)> callback);
|
||||||
int fulfill_read(blockstore_op_t *read_op,
|
void free_read_buffers(std::vector<copy_buffer_t> & rv);
|
||||||
uint64_t &fulfilled, uint32_t item_start, uint32_t item_end,
|
|
||||||
uint32_t item_state, uint64_t item_version, uint64_t item_location,
|
|
||||||
uint64_t journal_sector, uint8_t *csum, int *dyn_data);
|
|
||||||
bool fulfill_clean_read(blockstore_op_t *read_op, uint64_t & fulfilled,
|
|
||||||
uint8_t *clean_entry_bitmap, int *dyn_data,
|
|
||||||
uint32_t item_start, uint32_t item_end, uint64_t clean_loc, uint64_t clean_ver);
|
|
||||||
int fill_partial_checksum_blocks(std::vector<copy_buffer_t> & rv, uint64_t & fulfilled,
|
|
||||||
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf, uint64_t read_offset, uint64_t read_end);
|
|
||||||
int pad_journal_read(std::vector<copy_buffer_t> & rv, copy_buffer_t & cp,
|
|
||||||
uint64_t dirty_offset, uint64_t dirty_end, uint64_t dirty_loc, uint8_t *csum_ptr, int *dyn_data,
|
|
||||||
uint64_t offset, uint64_t submit_len, uint64_t & blk_begin, uint64_t & blk_end, uint8_t* & blk_buf);
|
|
||||||
bool read_range_fulfilled(std::vector<copy_buffer_t> & rv, uint64_t & fulfilled, uint8_t *read_buf,
|
|
||||||
uint8_t *clean_entry_bitmap, uint32_t item_start, uint32_t item_end);
|
|
||||||
bool read_checksum_block(blockstore_op_t *op, int rv_pos, uint64_t &fulfilled, uint64_t clean_loc);
|
|
||||||
uint8_t* read_clean_meta_block(blockstore_op_t *read_op, uint64_t clean_loc, int rv_pos);
|
|
||||||
bool verify_padded_checksums(uint8_t *clean_entry_bitmap, uint8_t *csum_buf, uint32_t offset,
|
|
||||||
iovec *iov, int n_iov, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb);
|
|
||||||
bool verify_journal_checksums(uint8_t *csums, uint32_t offset,
|
|
||||||
iovec *iov, int n_iov, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb);
|
|
||||||
bool verify_clean_padded_checksums(blockstore_op_t *op, uint64_t clean_loc, uint8_t *dyn_data, bool from_journal,
|
|
||||||
iovec *iov, int n_iov, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb);
|
|
||||||
int fulfill_read_push(blockstore_op_t *op, void *buf, uint64_t offset, uint64_t len,
|
|
||||||
uint32_t item_state, uint64_t item_version);
|
|
||||||
void handle_read_event(ring_data_t *data, blockstore_op_t *op);
|
void handle_read_event(ring_data_t *data, blockstore_op_t *op);
|
||||||
|
bool verify_read_checksums(blockstore_op_t *op);
|
||||||
|
|
||||||
// Write
|
// Write
|
||||||
bool enqueue_write(blockstore_op_t *op);
|
bool enqueue_write(blockstore_op_t *op);
|
||||||
void cancel_all_writes(blockstore_op_t *op, blockstore_dirty_db_t::iterator dirty_it, int retval);
|
void prepare_meta_block_write(uint32_t modified_block);
|
||||||
|
bool meta_block_is_pending(uint32_t modified_block);
|
||||||
|
bool intent_write_allowed(blockstore_op_t *op, heap_entry_t *obj);
|
||||||
int dequeue_write(blockstore_op_t *op);
|
int dequeue_write(blockstore_op_t *op);
|
||||||
int dequeue_del(blockstore_op_t *op);
|
|
||||||
int continue_write(blockstore_op_t *op);
|
int continue_write(blockstore_op_t *op);
|
||||||
void release_journal_sectors(blockstore_op_t *op);
|
|
||||||
void handle_write_event(ring_data_t *data, blockstore_op_t *op);
|
void handle_write_event(ring_data_t *data, blockstore_op_t *op);
|
||||||
|
|
||||||
// Sync
|
// Sync
|
||||||
int continue_sync(blockstore_op_t *op);
|
int continue_sync(blockstore_op_t *op);
|
||||||
void ack_sync(blockstore_op_t *op);
|
bool submit_fsyncs(int & wait_count);
|
||||||
|
int do_sync(blockstore_op_t *op, int base_state);
|
||||||
|
bool has_unsynced();
|
||||||
|
|
||||||
// Stabilize
|
// Stabilize
|
||||||
int dequeue_stable(blockstore_op_t *op);
|
int dequeue_stable(blockstore_op_t *op);
|
||||||
int continue_stable(blockstore_op_t *op);
|
|
||||||
void mark_stable(obj_ver_id ov, bool forget_dirty = false);
|
|
||||||
void stabilize_object(object_id oid, uint64_t max_ver);
|
|
||||||
blockstore_op_t* selective_sync(blockstore_op_t *op);
|
|
||||||
int split_stab_op(blockstore_op_t *op, std::function<int(obj_ver_id v)> decider);
|
|
||||||
|
|
||||||
// Rollback
|
|
||||||
int dequeue_rollback(blockstore_op_t *op);
|
|
||||||
int continue_rollback(blockstore_op_t *op);
|
|
||||||
void mark_rolled_back(const obj_ver_id & ov);
|
|
||||||
void erase_dirty(blockstore_dirty_db_t::iterator dirty_start, blockstore_dirty_db_t::iterator dirty_end, uint64_t clean_loc);
|
|
||||||
void free_dirty_dyn_data(dirty_entry & e);
|
|
||||||
|
|
||||||
// List
|
// List
|
||||||
void process_list(blockstore_op_t *op);
|
void process_list(blockstore_op_t *op);
|
||||||
|
|
||||||
public:
|
/*public:*/
|
||||||
|
|
||||||
blockstore_impl_t(blockstore_config_t & config, ring_loop_t *ringloop, timerfd_manager_t *tfd);
|
blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd, bool mock_mode = false);
|
||||||
~blockstore_impl_t();
|
~blockstore_impl_t();
|
||||||
|
|
||||||
|
void parse_config(blockstore_config_t & config);
|
||||||
void parse_config(blockstore_config_t & config, bool init);
|
void parse_config(blockstore_config_t & config, bool init);
|
||||||
|
|
||||||
|
void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit);
|
||||||
|
bool reshard_continue(void *reshard_state, uint64_t chunk_limit);
|
||||||
|
void reshard_abort(void *reshard_state);
|
||||||
|
|
||||||
// Event loop
|
// Event loop
|
||||||
void loop();
|
void loop();
|
||||||
|
|
||||||
@@ -426,21 +214,19 @@ public:
|
|||||||
// Simplified synchronous operation: get object bitmap & current version
|
// Simplified synchronous operation: get object bitmap & current version
|
||||||
int read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version = NULL);
|
int read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version = NULL);
|
||||||
|
|
||||||
// Unstable writes are added here (map of object_id -> version)
|
|
||||||
std::unordered_map<object_id, uint64_t> unstable_writes;
|
|
||||||
|
|
||||||
// Space usage statistics
|
|
||||||
std::map<uint64_t, uint64_t> inode_space_stats;
|
|
||||||
|
|
||||||
// Set per-pool no_inode_stats
|
// Set per-pool no_inode_stats
|
||||||
void set_no_inode_stats(const std::vector<uint64_t> & pool_ids);
|
void set_no_inode_stats(const std::vector<uint64_t> & pool_ids);
|
||||||
|
|
||||||
// Print diagnostics to stdout
|
// Print diagnostics to stdout
|
||||||
void dump_diagnostics();
|
void dump_diagnostics();
|
||||||
|
|
||||||
|
// Get diagnostic string for an operation
|
||||||
|
std::string get_op_diag(blockstore_op_t *op);
|
||||||
|
|
||||||
|
const std::map<uint64_t, uint64_t> & get_inode_space_stats() { return heap->get_inode_space_stats(); }
|
||||||
inline uint32_t get_block_size() { return dsk.data_block_size; }
|
inline uint32_t get_block_size() { return dsk.data_block_size; }
|
||||||
inline uint64_t get_block_count() { return dsk.block_count; }
|
inline uint64_t get_block_count() { return dsk.block_count; }
|
||||||
inline uint64_t get_free_block_count() { return dsk.block_count - used_blocks; }
|
uint64_t get_free_block_count();
|
||||||
inline uint32_t get_bitmap_granularity() { return dsk.disk_alignment; }
|
inline uint32_t get_bitmap_granularity() { return dsk.bitmap_granularity; }
|
||||||
inline uint64_t get_journal_size() { return dsk.journal_len; }
|
inline uint64_t get_journal_size() { return dsk.journal_len; }
|
||||||
};
|
};
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user