Compare commits
509
Commits
v0.5.0
...
nfs-proxy-old
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6261809e87 | ||
|
|
d7e64e6ea1 | ||
|
|
98e3528a14 | ||
|
|
8e88f77101 | ||
|
|
caa2cc2e6c | ||
|
|
842ba8b831 | ||
|
|
1493823f9e | ||
|
|
c857272f44 | ||
|
|
340a4b4f27 | ||
|
|
5118980315 | ||
|
|
d71cc174e3 | ||
|
|
0eb929f1ba | ||
|
|
83146fa3e2 | ||
|
|
15dcaf7903 | ||
|
|
cd18ef7323 | ||
|
|
39531ef1a6 | ||
|
|
d334914948 | ||
|
|
c373425562 | ||
|
|
3615e57879 | ||
|
|
0edc6fe5a6 | ||
|
|
9c30df83e3 | ||
|
|
a420c77107 | ||
|
|
4100d829c7 | ||
|
|
79ebda933e | ||
|
|
65d08e067e | ||
|
|
d289753df4 | ||
|
|
85298ddae2 | ||
|
|
e23296a327 | ||
|
|
839ec9e6e0 | ||
|
|
7cbfdff41a | ||
|
|
951272f27f | ||
|
|
a3fb1d4c98 | ||
|
|
88402e6eb6 | ||
|
|
390239c51b | ||
|
|
b7b2adfa32 | ||
|
|
36c276358b | ||
|
|
117d6f0612 | ||
|
|
7d79c58095 | ||
|
|
46d2bc100f | ||
|
|
732e2804e9 | ||
|
|
abaec2008c | ||
|
|
8129d238a4 | ||
|
|
61ebed144a | ||
|
|
9d3ba113aa | ||
|
|
9788045dc9 | ||
|
|
d6b0d29af6 | ||
|
|
36f352f06f | ||
|
|
318cc463c2 | ||
|
|
145e5cfb86 | ||
|
|
73ae578981 | ||
|
|
20ee4ed758 | ||
|
|
63de79d1b2 | ||
|
|
f712967079 | ||
|
|
df0cd85352 | ||
|
|
ebaf4d7a72 | ||
|
|
d4bc10542c | ||
|
|
140309620a | ||
|
|
0a610ee943 | ||
|
|
f3ce166064 | ||
|
|
717d303370 | ||
|
|
d9857a5340 | ||
|
|
eb5d9153e8 | ||
|
|
ae6d1ed1d5 | ||
|
|
d123e58ea3 | ||
|
|
d9869d8116 | ||
|
|
4047ca606f | ||
|
|
218e294e9c | ||
|
|
c1929cabe0 | ||
|
|
cc6b24e03a | ||
|
|
0757ba630a | ||
|
|
2a0b881685 | ||
|
|
9a15b843ff | ||
|
|
8dc1ffb13b | ||
|
|
ba63af49b4 | ||
|
|
31b9c683ee | ||
|
|
3abcac058f | ||
|
|
e01c4db702 | ||
|
|
a5cf06acd0 | ||
|
|
9c3653b1e1 | ||
|
|
23e578b6a2 | ||
|
|
7920414bee | ||
|
|
098e369a3b | ||
|
|
a43ef525a2 | ||
|
|
8a6b07d8f7 | ||
|
|
2c930d55fb | ||
|
|
d798e0821e | ||
|
|
e591a3e9f7 | ||
|
|
77cc18420a | ||
|
|
7bdd92ca4f | ||
|
|
8f64fc61e7 | ||
|
|
4a9f001d9e | ||
|
|
8c908316d9 | ||
|
|
515a2e6e33 | ||
|
|
68b6763ebe | ||
|
|
9c6168bf17 | ||
|
|
08e467270a | ||
|
|
5473d5b4a2 | ||
|
|
c3304bce27 | ||
|
|
ec2852c598 | ||
|
|
b9f5c2a823 | ||
|
|
e9d2f79aa7 | ||
|
|
0785bdf8b3 | ||
|
|
b57e44748b | ||
|
|
1bbe62f29c | ||
|
|
3061c30132 | ||
|
|
20a4406acc | ||
|
|
f93491bc6c | ||
|
|
999bed8514 | ||
|
|
3f33095fd7 | ||
|
|
dd74c5ce1b | ||
|
|
c6d104ecd6 | ||
|
|
e544aef7d0 | ||
|
|
616c18c786 | ||
|
|
fa687d3878 | ||
|
|
2c7556e536 | ||
|
|
2020608a39 | ||
|
|
139b98d80f | ||
|
|
f54ff6ad5d | ||
|
|
b376ef2ed9 | ||
|
|
5a234588b9 | ||
|
|
b82c30328f | ||
|
|
0ee5e0a7fe | ||
|
|
0a1640d169 | ||
|
|
3482bb0860 | ||
|
|
526995f486 | ||
|
|
073b505928 | ||
|
|
a8b21a22d0 | ||
|
|
0b1ffba62b | ||
|
|
8dfbd7943c | ||
|
|
39e7f98e54 | ||
|
|
3a83a32cb7 | ||
|
|
20d5ed799a | ||
|
|
b262938bca | ||
|
|
7e54242251 | ||
|
|
c3c2e68cc1 | ||
|
|
aa1e21dd99 | ||
|
|
f4b57d487f | ||
|
|
711ecd2f8e | ||
|
|
9fca01dc62 | ||
|
|
0bd3a94efd | ||
|
|
9ffdeef93b | ||
|
|
589892d501 | ||
|
|
5fe3a40416 | ||
|
|
a453db9c8e | ||
|
|
e6498a52ca | ||
|
|
4bc41aed9d | ||
|
|
4da51f9c4c | ||
|
|
c6cee6f734 | ||
|
|
6fc08f5581 | ||
|
|
15957b7d13 | ||
|
|
09a3987e83 | ||
|
|
cd6820c439 | ||
|
|
dcd8f5e76c | ||
|
|
5859f913fc | ||
|
|
cac6a1d8d1 | ||
|
|
a0c32e7de9 | ||
|
|
8b37610dd0 | ||
|
|
ae82ca3b08 | ||
|
|
92362027a8 | ||
|
|
c4aeeda143 | ||
|
|
24f0f8278a | ||
|
|
95496d0845 | ||
|
|
94b1f09ef2 | ||
|
|
32b1312abb | ||
|
|
d5c8fde5de | ||
|
|
7a0b5212fe | ||
|
|
a8f5c71ae8 | ||
|
|
ce5b6253ab | ||
|
|
8398ad0117 | ||
|
|
fea451b4db | ||
|
|
6e12aca53b | ||
|
|
8b007d531f | ||
|
|
7b7f20fb89 | ||
|
|
300d507026 | ||
|
|
6886171289 | ||
|
|
43f8ea47a0 | ||
|
|
6e0e172e15 | ||
|
|
655a2c871d | ||
|
|
879fe9b2b4 | ||
|
|
660c3f7b0d | ||
|
|
f0ebfae3b8 | ||
|
|
eb7ad2c114 | ||
|
|
b4235b4edf | ||
|
|
cd21ff0b6a | ||
|
|
d3903f039c | ||
|
|
66fe1a469b | ||
|
|
24409bd4c4 | ||
|
|
c5029961ea | ||
|
|
1ca1143d4a | ||
|
|
920345f7b6 | ||
|
|
75b47a6298 | ||
|
|
6e446653ae | ||
|
|
e51edf2542 | ||
|
|
ce170af91f | ||
|
|
7eabc364bf | ||
|
|
a346f84c69 | ||
|
|
71a0c1a7b9 | ||
|
|
20e86c7d84 | ||
|
|
110b39900b | ||
|
|
697ee30a26 | ||
|
|
42479b4590 | ||
|
|
6e82044e84 | ||
|
|
2cb3e84882 | ||
|
|
32614c5bc8 | ||
|
|
aa436027c8 | ||
|
|
577a563b91 | ||
|
|
e4efa2c08a | ||
|
|
0f3f0a9d29 | ||
|
|
0544a16f95 | ||
|
|
30d8930958 | ||
|
|
baf003fbd3 | ||
|
|
ba39a38dc4 | ||
|
|
d528cd77f1 | ||
|
|
6e6f407df3 | ||
|
|
4d43774cbb | ||
|
|
a1488f7217 | ||
|
|
404e07d365 | ||
|
|
b3dcee0d43 | ||
|
|
609bd4eb59 | ||
|
|
8e445ddc9a | ||
|
|
ffb06536ff | ||
|
|
eeecab20c2 | ||
|
|
e889ac4209 | ||
|
|
cfe8de9b84 | ||
|
|
24b9b19066 | ||
|
|
ef645ee0c2 | ||
|
|
8a9bae5216 | ||
|
|
da99686a15 | ||
|
|
dcc03ee41f | ||
|
|
fb2f7a0d3c | ||
|
|
38d85da19a | ||
|
|
dc3caee284 | ||
|
|
89dcda1fed | ||
|
|
1526e2055e | ||
|
|
74cb3911db | ||
|
|
d5efbbb6b9 | ||
|
|
4319091bd3 | ||
|
|
6d307d5391 | ||
|
|
065dfef683 | ||
|
|
4d6b85fe67 | ||
|
|
2dd2f29f46 | ||
|
|
fc3a1e076a | ||
|
|
3a3e168c42 | ||
|
|
95c55da0ad | ||
|
|
5cf1157f16 | ||
|
|
acf637950c | ||
|
|
a02b02eb04 | ||
|
|
7d3d696110 | ||
|
|
712576ca75 | ||
|
|
28bd94d2c2 | ||
|
|
148ff04aa8 | ||
|
|
e86df4a2a2 | ||
|
|
e74af9745e | ||
|
|
0e0509e3da | ||
|
|
cb282d25e0 | ||
|
|
8b2a4c9539 | ||
|
|
b66a079892 | ||
|
|
e90bbe6385 | ||
|
|
4be761254c | ||
|
|
7a45c5f86c | ||
|
|
bff413584d | ||
|
|
bb31050ab5 | ||
|
|
b52dd6843a | ||
|
|
b66160a7ad | ||
|
|
30bb602681 | ||
|
|
eb0a3adafc | ||
|
|
24301b116c | ||
|
|
1d00c17d68 | ||
|
|
24f19c4b80 | ||
|
|
dfdf5c1f9c | ||
|
|
aad7792d3f | ||
|
|
6ca8afffe5 | ||
|
|
511a89948b | ||
|
|
3de553ecd7 | ||
|
|
9c45d43e74 | ||
|
|
891250d355 | ||
|
|
f9fe72d40a | ||
|
|
10ee4f7c1d | ||
|
|
fd8244699b | ||
|
|
eaac1fc5d1 | ||
|
|
57be1923d3 | ||
|
|
c467acc388 | ||
|
|
bf591ba3ee | ||
|
|
699a0fbbc7 | ||
|
|
6b2dd50f27 | ||
|
|
caf2f3c56f | ||
|
|
9174f188b1 | ||
|
|
d3978c6d0e | ||
|
|
4a7365660d | ||
|
|
818ae5d61d | ||
|
|
6810e93c3f | ||
|
|
f6f35f4127 | ||
|
|
72aa2fd819 | ||
|
|
5010b0dd75 | ||
|
|
483c5ab380 | ||
|
|
6a6fd6544d | ||
|
|
971aa4ae4f | ||
|
|
9e6cbc6ebc | ||
|
|
ce777319c3 | ||
|
|
f8ff39b0ab | ||
|
|
d749159585 | ||
|
|
9703773a63 | ||
|
|
5d8d486f7c | ||
|
|
2b546cdd55 | ||
|
|
bd7b177707 | ||
|
|
33f9d03d22 | ||
|
|
82e6aff17b | ||
|
|
57e2c503f7 | ||
|
|
715bc8d53d | ||
|
|
0af077701c | ||
|
|
cac976ce25 | ||
|
|
acf0646542 | ||
|
|
ede1c1d667 | ||
|
|
38bd51c97f | ||
|
|
8c9f32cd45 | ||
|
|
966fb763ca | ||
|
|
0b41ffc08d | ||
|
|
64eeb79051 | ||
|
|
2a02f3c4c7 | ||
|
|
f684d9101a | ||
|
|
c72fddd714 | ||
|
|
a1f2f19489 | ||
|
|
82c1a7ec67 | ||
|
|
2ab423d4ef | ||
|
|
4694811eab | ||
|
|
6b988de17d | ||
|
|
37efdc2a83 | ||
|
|
591cad09c9 | ||
|
|
b907ad50aa | ||
|
|
7308d6a6c0 | ||
|
|
5f5b6ef150 | ||
|
|
38a3df4a0e | ||
|
|
6950b8e3a0 | ||
|
|
0cea3576fb | ||
|
|
f01eea07d3 | ||
|
|
2c2f08aca2 | ||
|
|
d6524670e1 | ||
|
|
879ecfa74d | ||
|
|
aea2d19d35 | ||
|
|
04f86dc00b | ||
|
|
7aeb2cbac7 | ||
|
|
519f081006 | ||
|
|
e50f703e1d | ||
|
|
2612d3198a | ||
|
|
ab39ce2bbb | ||
|
|
d0c2e31312 | ||
|
|
9038d42327 | ||
|
|
691f066055 | ||
|
|
ffe1cd4c79 | ||
|
|
4ae1b84c67 | ||
|
|
c35963967f | ||
|
|
0aa2dd2890 | ||
|
|
6bf88883ac | ||
|
|
004f265393 | ||
|
|
860ac24762 | ||
|
|
6107a4d07b | ||
|
|
95c29b9dc3 | ||
|
|
d99407dcec | ||
|
|
6909807068 | ||
|
|
ec90fe6ec1 | ||
|
|
18c72f4835 | ||
|
|
59fbcef734 | ||
|
|
40b7c21fb1 | ||
|
|
efb3678606 | ||
|
|
462650134e | ||
|
|
8d87e32175 | ||
|
|
b0b2e7df3c | ||
|
|
97efb9e299 | ||
|
|
f6d705383a | ||
|
|
68567c0e1f | ||
|
|
04b00003e9 | ||
|
|
307c1731c1 | ||
|
|
75a6a556b5 | ||
|
|
a48e2bbf18 | ||
|
|
688821665a | ||
|
|
3e162d95a0 | ||
|
|
829381b335 | ||
|
|
54f2353f24 | ||
|
|
e47f6fba60 | ||
|
|
883bf84a16 | ||
|
|
52097c4856 | ||
|
|
e1355cbc74 | ||
|
|
8f8b90be7a | ||
|
|
ad9f619370 | ||
|
|
f4769ba7c7 | ||
|
|
843b7052d2 | ||
|
|
df99e232ee | ||
|
|
3a40fa4127 | ||
|
|
4095bcc558 | ||
|
|
564d64e271 | ||
|
|
cf54741c95 | ||
|
|
18a5fafa2a | ||
|
|
06f4978085 | ||
|
|
7ebf1588c5 | ||
|
|
b0ad1e1e6d | ||
|
|
0949f08407 | ||
|
|
04a1f18fa5 | ||
|
|
cf9a641d66 | ||
|
|
05db1308aa | ||
|
|
98b54ca948 | ||
|
|
23225c5e62 | ||
|
|
7e6e1a5a82 | ||
|
|
435045751d | ||
|
|
c5fb1d5987 | ||
|
|
9f59381bea | ||
|
|
9ac7e75178 | ||
|
|
88671cf745 | ||
|
|
fe1749c427 | ||
|
|
ceb9c28de7 | ||
|
|
299d7d7c95 | ||
|
|
d1526b415f | ||
|
|
f49fd53d55 | ||
|
|
dd76eda5e5 | ||
|
|
87dbd8fa57 | ||
|
|
b44f49aab2 | ||
|
|
036555638e | ||
|
|
af5155fcd9 | ||
|
|
0d2efbecc9 | ||
|
|
e62e8b6bae | ||
|
|
c4ba24c305 | ||
|
|
19e47a0279 | ||
|
|
bd178ac20f | ||
|
|
7006875a24 | ||
|
|
ad577c4aac | ||
|
|
836635c518 | ||
|
|
88a03f4e98 | ||
|
|
2a5036669d | ||
|
|
2e0c853180 | ||
|
|
e91ff2a9ec | ||
|
|
086667f568 | ||
|
|
73ce20e246 | ||
|
|
1be94da437 | ||
|
|
80e12358a2 | ||
|
|
36c935ace6 | ||
|
|
0d8b5e2ef9 | ||
|
|
98f1e2c277 | ||
|
|
21e7686037 | ||
|
|
ab21a1908b | ||
|
|
30d1ccd43e | ||
|
|
8bdd6d8d78 | ||
|
|
09b3e4e789 | ||
|
|
07912fd670 | ||
|
|
bc742ccf8c | ||
|
|
314b20437b | ||
|
|
29bac892ad | ||
|
|
cf7547faf3 | ||
|
|
ab90ed747f | ||
|
|
29d8ac8b1b | ||
|
|
97795ea1b1 | ||
|
|
24e7075f08 | ||
|
|
6155b23a7e | ||
|
|
7d49706c07 | ||
|
|
46e79f3306 | ||
|
|
41fd14e024 | ||
|
|
bb2d9a3afe | ||
|
|
e899ed2c25 | ||
|
|
e21b14b72c | ||
|
|
5af8eddaa9 | ||
|
|
4f5a94c07a | ||
|
|
e16b87ecc8 | ||
|
|
fcb4aa0a11 | ||
|
|
12adfa470c | ||
|
|
7f15e0c084 | ||
|
|
08d4bef419 | ||
|
|
2d73b19a6c | ||
|
|
69c87009e9 | ||
|
|
c974cb539c | ||
|
|
00e98f64f3 | ||
|
|
91a70dfb1b | ||
|
|
178388ac8c | ||
|
|
bf9a175efc | ||
|
|
08aed962de | ||
|
|
8c65e890b9 | ||
|
|
8cda70b889 | ||
|
|
61ab22403a | ||
|
|
16da663a66 | ||
|
|
4a2dcf7b6b | ||
|
|
8d48cc56b0 | ||
|
|
9f58f01425 | ||
|
|
b9e7d31aa1 | ||
|
|
2d9f09dcb6 | ||
|
|
7cc59260c5 | ||
|
|
ca0a11ec85 | ||
|
|
51c0b5afee | ||
|
|
e1e01d042e | ||
|
|
534a4a657e | ||
|
|
9b5d8b9ad4 | ||
|
|
e66ed47515 | ||
|
|
036c6d4c42 | ||
|
|
4cb79a3bf8 | ||
|
|
3bf53754c2 | ||
|
|
6023cac361 | ||
|
|
915d04c446 | ||
|
|
21e06ea40d | ||
|
|
9ef7f865b0 | ||
|
|
9dd20a31aa | ||
|
|
28be049909 | ||
|
|
78fbaacf1f | ||
|
|
1526c5a213 | ||
|
|
c7cc414c90 | ||
|
|
f4ea313707 | ||
|
|
b88b76f316 | ||
|
|
4a17a61d1f | ||
|
|
ccabbbfbcb | ||
|
|
26dac57083 | ||
|
|
44a53d8352 | ||
|
|
9d80bd2d98 | ||
|
|
322a38a144 | ||
|
|
1018764c91 |
@@ -1,5 +1,6 @@
|
|||||||
.git
|
.git
|
||||||
build
|
build
|
||||||
|
packages
|
||||||
mon/node_modules
|
mon/node_modules
|
||||||
*.o
|
*.o
|
||||||
*.so
|
*.so
|
||||||
@@ -15,3 +16,4 @@ fio
|
|||||||
qemu
|
qemu
|
||||||
rpm/*.Dockerfile
|
rpm/*.Dockerfile
|
||||||
debian/*.Dockerfile
|
debian/*.Dockerfile
|
||||||
|
Dockerfile
|
||||||
|
|||||||
+18
@@ -0,0 +1,18 @@
|
|||||||
|
*.o
|
||||||
|
*.so
|
||||||
|
package-lock.json
|
||||||
|
fio
|
||||||
|
qemu
|
||||||
|
osd
|
||||||
|
stub_osd
|
||||||
|
stub_uring_osd
|
||||||
|
stub_bench
|
||||||
|
osd_test
|
||||||
|
osd_peering_pg_test
|
||||||
|
dump_journal
|
||||||
|
nbd_proxy
|
||||||
|
rm_inode
|
||||||
|
test_allocator
|
||||||
|
test_blockstore
|
||||||
|
test_shit
|
||||||
|
osd_rmw_test
|
||||||
@@ -4,3 +4,6 @@
|
|||||||
[submodule "json11"]
|
[submodule "json11"]
|
||||||
path = json11
|
path = json11
|
||||||
url = ../json11.git
|
url = ../json11.git
|
||||||
|
[submodule "libnfs"]
|
||||||
|
path = libnfs
|
||||||
|
url = ../libnfs.git
|
||||||
|
|||||||
@@ -0,0 +1,7 @@
|
|||||||
|
cmake_minimum_required(VERSION 2.8)
|
||||||
|
|
||||||
|
project(vitastor)
|
||||||
|
|
||||||
|
set(VERSION "0.6.16")
|
||||||
|
|
||||||
|
add_subdirectory(src)
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
Copyright (c) Vitaliy Filippov (vitalif [at] yourcmc.ru), 2019+
|
||||||
|
|
||||||
|
All server-side code (OSD, Monitor and so on) is licensed under the terms of
|
||||||
|
Vitastor Network Public License 1.1 (VNPL 1.1), a copyleft license based on
|
||||||
|
GNU GPLv3.0 with the additional "Network Interaction" clause which requires
|
||||||
|
opensourcing all programs directly or indirectly interacting with Vitastor
|
||||||
|
through a computer network and expressly designed to be used in conjunction
|
||||||
|
with it ("Proxy Programs"). Proxy Programs may be made public not only under
|
||||||
|
the terms of the same license, but also under the terms of any GPL-Compatible
|
||||||
|
Free Software License, as listed by the Free Software Foundation.
|
||||||
|
This is a stricter copyleft license than the Affero GPL.
|
||||||
|
|
||||||
|
Please note that VNPL doesn't require you to open the code of proprietary
|
||||||
|
software running inside a VM if it's not specially designed to be used with
|
||||||
|
Vitastor.
|
||||||
|
|
||||||
|
Basically, you can't use the software in a proprietary environment to provide
|
||||||
|
its functionality to users without opensourcing all intermediary components
|
||||||
|
standing between the user and Vitastor or purchasing a commercial license
|
||||||
|
from the author 😀.
|
||||||
|
|
||||||
|
Client libraries (cluster_client and so on) are dual-licensed under the same
|
||||||
|
VNPL 1.1 and also GNU GPL 2.0 or later to allow for compatibility with GPLed
|
||||||
|
software like QEMU and fio.
|
||||||
|
|
||||||
|
You can find the full text of VNPL-1.1 in the file [VNPL-1.1.txt](VNPL-1.1.txt).
|
||||||
|
GPL 2.0 is also included in this repository as [GPL-2.0.txt](GPL-2.0.txt).
|
||||||
-46
@@ -1,46 +0,0 @@
|
|||||||
#!/usr/bin/perl
|
|
||||||
|
|
||||||
use strict;
|
|
||||||
|
|
||||||
my $deps = {};
|
|
||||||
for my $line (split /\n/, `grep '^#include "' *.cpp *.h`)
|
|
||||||
{
|
|
||||||
if ($line =~ /^([^:]+):\#include "([^"]+)"/s)
|
|
||||||
{
|
|
||||||
$deps->{$1}->{$2} = 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
my $added;
|
|
||||||
do
|
|
||||||
{
|
|
||||||
$added = 0;
|
|
||||||
for my $file (keys %$deps)
|
|
||||||
{
|
|
||||||
for my $dep (keys %{$deps->{$file}})
|
|
||||||
{
|
|
||||||
if ($deps->{$dep})
|
|
||||||
{
|
|
||||||
for my $subdep (keys %{$deps->{$dep}})
|
|
||||||
{
|
|
||||||
if (!$deps->{$file}->{$subdep})
|
|
||||||
{
|
|
||||||
$added = 1;
|
|
||||||
$deps->{$file}->{$subdep} = 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
} while ($added);
|
|
||||||
|
|
||||||
for my $file (sort keys %$deps)
|
|
||||||
{
|
|
||||||
if ($file =~ /\.cpp$/)
|
|
||||||
{
|
|
||||||
my $obj = $file;
|
|
||||||
$obj =~ s/\.cpp$/.o/s;
|
|
||||||
print "$obj: $file ".join(" ", sort keys %{$deps->{$file}})."\n";
|
|
||||||
print "\tg++ \$(CXXFLAGS) -c -o \$\@ \$\<\n";
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,195 +0,0 @@
|
|||||||
BINDIR ?= /usr/bin
|
|
||||||
LIBDIR ?= /usr/lib/x86_64-linux-gnu
|
|
||||||
QEMU_PLUGINDIR ?= /usr/lib/x86_64-linux-gnu/qemu
|
|
||||||
|
|
||||||
BLOCKSTORE_OBJS := allocator.o blockstore.o blockstore_impl.o blockstore_init.o blockstore_open.o blockstore_journal.o blockstore_read.o \
|
|
||||||
blockstore_write.o blockstore_sync.o blockstore_stable.o blockstore_rollback.o blockstore_flush.o crc32c.o ringloop.o
|
|
||||||
# -fsanitize=address
|
|
||||||
CXXFLAGS := -g -O3 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fPIC -fdiagnostics-color=always -I/usr/include/jerasure
|
|
||||||
all: libfio_blockstore.so osd libfio_sec_osd.so libfio_cluster.so stub_osd stub_uring_osd stub_bench osd_test dump_journal qemu_driver.so nbd_proxy rm_inode
|
|
||||||
clean:
|
|
||||||
rm -f *.o libblockstore.so libfio_blockstore.so osd libfio_sec_osd.so libfio_cluster.so stub_osd stub_uring_osd stub_bench osd_test dump_journal qemu_driver.so nbd_proxy rm_inode
|
|
||||||
|
|
||||||
install: all
|
|
||||||
mkdir -p $(DESTDIR)$(LIBDIR)/vitastor
|
|
||||||
install -m 0755 libfio_sec_osd.so $(DESTDIR)$(LIBDIR)/vitastor/
|
|
||||||
install -m 0755 libfio_cluster.so $(DESTDIR)$(LIBDIR)/vitastor/
|
|
||||||
install -m 0755 libfio_blockstore.so $(DESTDIR)$(LIBDIR)/vitastor/
|
|
||||||
install -m 0755 libblockstore.so $(DESTDIR)$(LIBDIR)/vitastor/
|
|
||||||
mkdir -p $(DESTDIR)$(BINDIR)
|
|
||||||
install -m 0755 osd $(DESTDIR)$(BINDIR)/vitastor-osd
|
|
||||||
install -m 0755 dump_journal $(DESTDIR)$(BINDIR)/vitastor-dump-journal
|
|
||||||
install -m 0755 nbd_proxy $(DESTDIR)$(BINDIR)/vitastor-nbd
|
|
||||||
install -m 0755 rm_inode $(DESTDIR)$(BINDIR)/vitastor-rm
|
|
||||||
mkdir -p $(DESTDIR)$(QEMU_PLUGINDIR)
|
|
||||||
install -m 0755 qemu_driver.so $(DESTDIR)$(QEMU_PLUGINDIR)/block-vitastor.so
|
|
||||||
|
|
||||||
dump_journal: dump_journal.cpp crc32c.o blockstore_journal.h
|
|
||||||
g++ $(CXXFLAGS) -o $@ $< crc32c.o
|
|
||||||
|
|
||||||
libblockstore.so: $(BLOCKSTORE_OBJS)
|
|
||||||
g++ $(CXXFLAGS) -o $@ -shared $(BLOCKSTORE_OBJS) -ltcmalloc_minimal -luring
|
|
||||||
libfio_blockstore.so: ./libblockstore.so fio_engine.o json11.o
|
|
||||||
g++ $(CXXFLAGS) -Wl,-rpath,'$(LIBDIR)/vitastor' -shared -o $@ fio_engine.o json11.o ./libblockstore.so -ltcmalloc_minimal -luring
|
|
||||||
|
|
||||||
OSD_OBJS := osd.o osd_secondary.o msgr_receive.o msgr_send.o osd_peering.o osd_flush.o osd_peering_pg.o \
|
|
||||||
osd_primary.o osd_primary_subops.o etcd_state_client.o messenger.o osd_cluster.o http_client.o osd_ops.o pg_states.o \
|
|
||||||
osd_rmw.o json11.o base64.o timerfd_manager.o epoll_manager.o
|
|
||||||
osd: ./libblockstore.so osd_main.cpp osd.h osd_ops.h $(OSD_OBJS)
|
|
||||||
g++ $(CXXFLAGS) -Wl,-rpath,'$(LIBDIR)/vitastor' -o $@ osd_main.cpp $(OSD_OBJS) ./libblockstore.so -ltcmalloc_minimal -luring -lJerasure
|
|
||||||
|
|
||||||
stub_osd: stub_osd.o rw_blocking.o
|
|
||||||
g++ $(CXXFLAGS) -o $@ stub_osd.o rw_blocking.o -ltcmalloc_minimal
|
|
||||||
|
|
||||||
osd_rmw_test: osd_rmw_test.o
|
|
||||||
g++ $(CXXFLAGS) -o $@ osd_rmw_test.o -lJerasure -fsanitize=address
|
|
||||||
|
|
||||||
STUB_URING_OSD_OBJS := stub_uring_osd.o epoll_manager.o messenger.o msgr_send.o msgr_receive.o ringloop.o timerfd_manager.o json11.o
|
|
||||||
stub_uring_osd: $(STUB_URING_OSD_OBJS)
|
|
||||||
g++ $(CXXFLAGS) -o $@ -ltcmalloc_minimal $(STUB_URING_OSD_OBJS) -luring
|
|
||||||
stub_bench: stub_bench.cpp osd_ops.h rw_blocking.o
|
|
||||||
g++ $(CXXFLAGS) -o $@ stub_bench.cpp rw_blocking.o -ltcmalloc_minimal
|
|
||||||
osd_test: osd_test.cpp osd_ops.h rw_blocking.o
|
|
||||||
g++ $(CXXFLAGS) -o $@ osd_test.cpp rw_blocking.o -ltcmalloc_minimal
|
|
||||||
osd_peering_pg_test: osd_peering_pg_test.cpp osd_peering_pg.o
|
|
||||||
g++ $(CXXFLAGS) -o $@ $< osd_peering_pg.o -ltcmalloc_minimal
|
|
||||||
|
|
||||||
libfio_sec_osd.so: fio_sec_osd.o rw_blocking.o
|
|
||||||
g++ $(CXXFLAGS) -ltcmalloc_minimal -shared -o $@ fio_sec_osd.o rw_blocking.o
|
|
||||||
|
|
||||||
FIO_CLUSTER_OBJS := cluster_client.o epoll_manager.o etcd_state_client.o \
|
|
||||||
messenger.o msgr_send.o msgr_receive.o ringloop.o json11.o http_client.o osd_ops.o pg_states.o timerfd_manager.o base64.o
|
|
||||||
libfio_cluster.so: fio_cluster.o $(FIO_CLUSTER_OBJS)
|
|
||||||
g++ $(CXXFLAGS) -ltcmalloc_minimal -shared -o $@ $< $(FIO_CLUSTER_OBJS) -luring
|
|
||||||
|
|
||||||
nbd_proxy: nbd_proxy.o $(FIO_CLUSTER_OBJS)
|
|
||||||
g++ $(CXXFLAGS) -ltcmalloc_minimal -o $@ $< $(FIO_CLUSTER_OBJS) -luring
|
|
||||||
|
|
||||||
rm_inode: rm_inode.o $(FIO_CLUSTER_OBJS)
|
|
||||||
g++ $(CXXFLAGS) -ltcmalloc_minimal -o $@ $< $(FIO_CLUSTER_OBJS) -luring
|
|
||||||
|
|
||||||
qemu_driver.o: qemu_driver.c qemu_proxy.h
|
|
||||||
gcc -I qemu/b/qemu `pkg-config glib-2.0 --cflags` \
|
|
||||||
-I qemu/include $(CXXFLAGS) -c -o $@ $<
|
|
||||||
|
|
||||||
qemu_driver.so: qemu_driver.o qemu_proxy.o $(FIO_CLUSTER_OBJS)
|
|
||||||
g++ $(CXXFLAGS) -ltcmalloc_minimal -shared -o $@ $(FIO_CLUSTER_OBJS) qemu_driver.o qemu_proxy.o -luring
|
|
||||||
|
|
||||||
test_blockstore: ./libblockstore.so test_blockstore.cpp timerfd_interval.o
|
|
||||||
g++ $(CXXFLAGS) -Wl,-rpath,'$(LIBDIR)/vitastor' -o test_blockstore test_blockstore.cpp timerfd_interval.o ./libblockstore.so -ltcmalloc_minimal -luring
|
|
||||||
test_shit: test_shit.cpp osd_peering_pg.o
|
|
||||||
g++ $(CXXFLAGS) -o test_shit test_shit.cpp -luring -lm
|
|
||||||
test_allocator: test_allocator.cpp allocator.o
|
|
||||||
g++ $(CXXFLAGS) -o test_allocator test_allocator.cpp allocator.o
|
|
||||||
|
|
||||||
crc32c.o: crc32c.c crc32c.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
json11.o: json11/json11.cpp
|
|
||||||
g++ $(CXXFLAGS) -c -o json11.o json11/json11.cpp
|
|
||||||
|
|
||||||
# Autogenerated
|
|
||||||
|
|
||||||
allocator.o: allocator.cpp allocator.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
base64.o: base64.cpp base64.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
blockstore.o: blockstore.cpp allocator.h blockstore.h blockstore_flush.h blockstore_impl.h blockstore_init.h blockstore_journal.h cpp-btree/btree_map.h crc32c.h malloc_or_die.h object_id.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
blockstore_flush.o: blockstore_flush.cpp allocator.h blockstore.h blockstore_flush.h blockstore_impl.h blockstore_init.h blockstore_journal.h cpp-btree/btree_map.h crc32c.h malloc_or_die.h object_id.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
blockstore_impl.o: blockstore_impl.cpp allocator.h blockstore.h blockstore_flush.h blockstore_impl.h blockstore_init.h blockstore_journal.h cpp-btree/btree_map.h crc32c.h malloc_or_die.h object_id.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
blockstore_init.o: blockstore_init.cpp allocator.h blockstore.h blockstore_flush.h blockstore_impl.h blockstore_init.h blockstore_journal.h cpp-btree/btree_map.h crc32c.h malloc_or_die.h object_id.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
blockstore_journal.o: blockstore_journal.cpp allocator.h blockstore.h blockstore_flush.h blockstore_impl.h blockstore_init.h blockstore_journal.h cpp-btree/btree_map.h crc32c.h malloc_or_die.h object_id.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
blockstore_open.o: blockstore_open.cpp allocator.h blockstore.h blockstore_flush.h blockstore_impl.h blockstore_init.h blockstore_journal.h cpp-btree/btree_map.h crc32c.h malloc_or_die.h object_id.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
blockstore_read.o: blockstore_read.cpp allocator.h blockstore.h blockstore_flush.h blockstore_impl.h blockstore_init.h blockstore_journal.h cpp-btree/btree_map.h crc32c.h malloc_or_die.h object_id.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
blockstore_rollback.o: blockstore_rollback.cpp allocator.h blockstore.h blockstore_flush.h blockstore_impl.h blockstore_init.h blockstore_journal.h cpp-btree/btree_map.h crc32c.h malloc_or_die.h object_id.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
blockstore_stable.o: blockstore_stable.cpp allocator.h blockstore.h blockstore_flush.h blockstore_impl.h blockstore_init.h blockstore_journal.h cpp-btree/btree_map.h crc32c.h malloc_or_die.h object_id.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
blockstore_sync.o: blockstore_sync.cpp allocator.h blockstore.h blockstore_flush.h blockstore_impl.h blockstore_init.h blockstore_journal.h cpp-btree/btree_map.h crc32c.h malloc_or_die.h object_id.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
blockstore_write.o: blockstore_write.cpp allocator.h blockstore.h blockstore_flush.h blockstore_impl.h blockstore_init.h blockstore_journal.h cpp-btree/btree_map.h crc32c.h malloc_or_die.h object_id.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
cluster_client.o: cluster_client.cpp cluster_client.h etcd_state_client.h http_client.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd_id.h osd_ops.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
dump_journal.o: dump_journal.cpp allocator.h blockstore.h blockstore_flush.h blockstore_impl.h blockstore_init.h blockstore_journal.h cpp-btree/btree_map.h crc32c.h malloc_or_die.h object_id.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
epoll_manager.o: epoll_manager.cpp epoll_manager.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
etcd_state_client.o: etcd_state_client.cpp base64.h etcd_state_client.h http_client.h json11/json11.hpp object_id.h osd_id.h osd_ops.h pg_states.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
fio_cluster.o: fio_cluster.cpp cluster_client.h epoll_manager.h etcd_state_client.h fio/arch/arch.h fio/fio.h fio/optgroup.h fio_headers.h http_client.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd_id.h osd_ops.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
fio_engine.o: fio_engine.cpp blockstore.h fio/arch/arch.h fio/fio.h fio/optgroup.h fio_headers.h json11/json11.hpp object_id.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
fio_sec_osd.o: fio_sec_osd.cpp fio/arch/arch.h fio/fio.h fio/optgroup.h fio_headers.h object_id.h osd_id.h osd_ops.h rw_blocking.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
http_client.o: http_client.cpp http_client.h json11/json11.hpp timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
messenger.o: messenger.cpp json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd_id.h osd_ops.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
msgr_receive.o: msgr_receive.cpp json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd_id.h osd_ops.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
msgr_send.o: msgr_send.cpp json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd_id.h osd_ops.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
nbd_proxy.o: nbd_proxy.cpp cluster_client.h epoll_manager.h etcd_state_client.h http_client.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd_id.h osd_ops.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd.o: osd.cpp blockstore.h cpp-btree/btree_map.h epoll_manager.h etcd_state_client.h http_client.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd.h osd_id.h osd_ops.h osd_peering_pg.h pg_states.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd_cluster.o: osd_cluster.cpp base64.h blockstore.h cpp-btree/btree_map.h epoll_manager.h etcd_state_client.h http_client.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd.h osd_id.h osd_ops.h osd_peering_pg.h pg_states.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd_flush.o: osd_flush.cpp blockstore.h cpp-btree/btree_map.h epoll_manager.h etcd_state_client.h http_client.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd.h osd_id.h osd_ops.h osd_peering_pg.h pg_states.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd_main.o: osd_main.cpp blockstore.h cpp-btree/btree_map.h epoll_manager.h etcd_state_client.h http_client.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd.h osd_id.h osd_ops.h osd_peering_pg.h pg_states.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd_ops.o: osd_ops.cpp object_id.h osd_id.h osd_ops.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd_peering.o: osd_peering.cpp base64.h blockstore.h cpp-btree/btree_map.h epoll_manager.h etcd_state_client.h http_client.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd.h osd_id.h osd_ops.h osd_peering_pg.h pg_states.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd_peering_pg.o: osd_peering_pg.cpp cpp-btree/btree_map.h object_id.h osd_id.h osd_ops.h osd_peering_pg.h pg_states.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd_peering_pg_test.o: osd_peering_pg_test.cpp cpp-btree/btree_map.h object_id.h osd_id.h osd_ops.h osd_peering_pg.h pg_states.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd_primary.o: osd_primary.cpp blockstore.h cpp-btree/btree_map.h epoll_manager.h etcd_state_client.h http_client.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd.h osd_id.h osd_ops.h osd_peering_pg.h osd_primary.h osd_rmw.h pg_states.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd_primary_subops.o: osd_primary_subops.cpp blockstore.h cpp-btree/btree_map.h epoll_manager.h etcd_state_client.h http_client.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd.h osd_id.h osd_ops.h osd_peering_pg.h osd_primary.h osd_rmw.h pg_states.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd_rmw.o: osd_rmw.cpp malloc_or_die.h object_id.h osd_id.h osd_rmw.h xor.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd_rmw_test.o: osd_rmw_test.cpp malloc_or_die.h object_id.h osd_id.h osd_rmw.cpp osd_rmw.h test_pattern.h xor.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd_secondary.o: osd_secondary.cpp blockstore.h cpp-btree/btree_map.h epoll_manager.h etcd_state_client.h http_client.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd.h osd_id.h osd_ops.h osd_peering_pg.h pg_states.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
osd_test.o: osd_test.cpp object_id.h osd_id.h osd_ops.h rw_blocking.h test_pattern.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
pg_states.o: pg_states.cpp pg_states.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
qemu_proxy.o: qemu_proxy.cpp cluster_client.h etcd_state_client.h http_client.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd_id.h osd_ops.h qemu_proxy.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
ringloop.o: ringloop.cpp ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
rm_inode.o: rm_inode.cpp cluster_client.h etcd_state_client.h http_client.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd_id.h osd_ops.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
rw_blocking.o: rw_blocking.cpp rw_blocking.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
stub_bench.o: stub_bench.cpp object_id.h osd_id.h osd_ops.h rw_blocking.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
stub_osd.o: stub_osd.cpp object_id.h osd_id.h osd_ops.h rw_blocking.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
stub_uring_osd.o: stub_uring_osd.cpp epoll_manager.h json11/json11.hpp malloc_or_die.h messenger.h object_id.h osd_id.h osd_ops.h ringloop.h timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
test_allocator.o: test_allocator.cpp allocator.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
test_blockstore.o: test_blockstore.cpp blockstore.h object_id.h ringloop.h timerfd_interval.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
test_shit.o: test_shit.cpp allocator.h blockstore.h blockstore_flush.h blockstore_impl.h blockstore_init.h blockstore_journal.h cpp-btree/btree_map.h crc32c.h malloc_or_die.h object_id.h osd_id.h osd_ops.h osd_peering_pg.h pg_states.h ringloop.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
timerfd_interval.o: timerfd_interval.cpp ringloop.h timerfd_interval.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
timerfd_manager.o: timerfd_manager.cpp timerfd_manager.h
|
|
||||||
g++ $(CXXFLAGS) -c -o $@ $<
|
|
||||||
+667
@@ -0,0 +1,667 @@
|
|||||||
|
## Vitastor
|
||||||
|
|
||||||
|
[Read English version](README.md)
|
||||||
|
|
||||||
|
## Идея
|
||||||
|
|
||||||
|
Я всего лишь хочу сделать качественную блочную SDS!
|
||||||
|
|
||||||
|
Vitastor - распределённая блочная SDS, прямой аналог Ceph RBD и внутренних СХД популярных
|
||||||
|
облачных провайдеров. Однако, в отличие от них, Vitastor быстрый и при этом простой.
|
||||||
|
Только пока маленький :-).
|
||||||
|
|
||||||
|
Архитектурная схожесть с Ceph означает заложенную на уровне алгоритмов записи строгую консистентность,
|
||||||
|
репликацию через первичный OSD, симметричную кластеризацию без единой точки отказа
|
||||||
|
и автоматическое распределение данных по любому числу дисков любого размера с настраиваемыми схемами
|
||||||
|
избыточности - репликацией или с произвольными кодами коррекции ошибок.
|
||||||
|
|
||||||
|
## Возможности
|
||||||
|
|
||||||
|
Vitastor на данный момент находится в статусе предварительного выпуска, расширенные
|
||||||
|
возможности пока отсутствуют, а в будущих версиях вероятны "ломающие" изменения.
|
||||||
|
|
||||||
|
Однако следующее уже реализовано:
|
||||||
|
|
||||||
|
- Базовая часть - надёжное кластерное блочное хранилище без единой точки отказа
|
||||||
|
- Производительность ;-D
|
||||||
|
- Несколько схем отказоустойчивости: репликация, XOR n+1 (1 диск чётности), коды коррекции ошибок
|
||||||
|
Рида-Соломона на основе библиотеки jerasure с любым числом дисков данных и чётности в группе
|
||||||
|
- Конфигурация через простые человекочитаемые JSON-структуры в etcd
|
||||||
|
- Автоматическое распределение данных по OSD, с поддержкой:
|
||||||
|
- Математической оптимизации для лучшей равномерности распределения и минимизации перемещений данных
|
||||||
|
- Нескольких пулов с разными схемами избыточности
|
||||||
|
- Дерева распределения, выбора OSD по тегам / классам устройств (только SSD, только HDD) и по поддереву
|
||||||
|
- Настраиваемых доменов отказа (диск/сервер/стойка и т.п.)
|
||||||
|
- Восстановление деградированных блоков
|
||||||
|
- Ребаланс, то есть перемещение данных между OSD (дисками)
|
||||||
|
- Поддержка "ленивого" fsync (fsync не на каждую операцию)
|
||||||
|
- Сбор статистики ввода/вывода в etcd
|
||||||
|
- Клиентская библиотека режима пользователя для ввода/вывода
|
||||||
|
- Драйвер диска для QEMU (собирается вне дерева исходников QEMU)
|
||||||
|
- Драйвер диска для утилиты тестирования производительности fio (также собирается вне дерева исходников fio)
|
||||||
|
- NBD-прокси для монтирования образов ядром ("блочное устройство в режиме пользователя")
|
||||||
|
- Утилита для удаления образов/инодов (vitastor-cli rm-data)
|
||||||
|
- Пакеты для Debian и CentOS
|
||||||
|
- Статистика операций ввода/вывода и занятого места в разрезе инодов
|
||||||
|
- Именование инодов через хранение их метаданных в etcd
|
||||||
|
- Снапшоты и copy-on-write клоны
|
||||||
|
- Сглаживание производительности случайной записи в SSD+HDD конфигурациях
|
||||||
|
- Поддержка RDMA/RoCEv2 через libibverbs
|
||||||
|
- CSI-плагин для Kubernetes
|
||||||
|
- Базовая поддержка OpenStack: драйвер Cinder, патчи для Nova и libvirt
|
||||||
|
- Слияние снапшотов (vitastor-cli {snap-rm,flatten,merge})
|
||||||
|
- Консольный интерфейс для управления образами (vitastor-cli {ls,create,modify})
|
||||||
|
- Плагин для Proxmox
|
||||||
|
|
||||||
|
## Планы развития
|
||||||
|
|
||||||
|
- Более корректные скрипты разметки дисков и автоматического запуска OSD
|
||||||
|
- Другие инструменты администрирования
|
||||||
|
- Плагины для OpenNebula и других облачных систем
|
||||||
|
- iSCSI-прокси
|
||||||
|
- Упрощённый NFS прокси
|
||||||
|
- Более быстрое переключение при отказах
|
||||||
|
- Фоновая проверка целостности без контрольных сумм (сверка реплик)
|
||||||
|
- Контрольные суммы
|
||||||
|
- Поддержка SSD-кэширования (tiered storage)
|
||||||
|
- Поддержка NVDIMM
|
||||||
|
- Web-интерфейс
|
||||||
|
- Возможно, сжатие
|
||||||
|
- Возможно, поддержка кэширования данных через системный page cache
|
||||||
|
|
||||||
|
## Архитектура
|
||||||
|
|
||||||
|
Так же, как и в Ceph, в Vitastor:
|
||||||
|
|
||||||
|
- Есть пулы (pools), PG, OSD, мониторы, домены отказа, дерево распределения (аналог crush-дерева).
|
||||||
|
- Образы делятся на блоки фиксированного размера (объекты), и эти объекты распределяются по OSD.
|
||||||
|
- У OSD есть журнал и метаданные и они тоже могут размещаться на отдельных быстрых дисках.
|
||||||
|
- Все операции записи тоже транзакционны. В Vitastor, правда, есть режим отложенного/ленивого fsync
|
||||||
|
(коммита), в котором fsync не вызывается на каждую операцию записи, что делает его более
|
||||||
|
пригодным для использования на "плохих" (десктопных) SSD. Однако все операции записи
|
||||||
|
в любом случае атомарны.
|
||||||
|
- Клиентская библиотека тоже старается ждать восстановления после любого отказа кластера, то есть,
|
||||||
|
вы тоже можете перезагрузить хоть весь кластер разом, и клиенты только на время зависнут,
|
||||||
|
но не отключатся.
|
||||||
|
|
||||||
|
Некоторые базовые термины для тех, кто не знаком с Ceph:
|
||||||
|
|
||||||
|
- OSD (Object Storage Daemon) - процесс, который хранит данные на одном диске и обрабатывает
|
||||||
|
запросы чтения/записи от клиентов.
|
||||||
|
- Пул (Pool) - контейнер для данных, имеющих одну и ту же схему избыточности и правила распределения по OSD.
|
||||||
|
- PG (Placement Group) - группа объектов, хранимых на одном и том же наборе реплик (OSD).
|
||||||
|
Несколько PG могут храниться на одном и том же наборе реплик, но объекты одной PG
|
||||||
|
в норме не хранятся на разных наборах OSD.
|
||||||
|
- Монитор - демон, хранящий состояние кластера.
|
||||||
|
- Домен отказа (Failure Domain) - группа OSD, которым вы разрешаете "упасть" всем вместе.
|
||||||
|
Иными словами, это группа OSD, в которые СХД не помещает разные копии одного и того же
|
||||||
|
блока данных. Например, если домен отказа - сервер, то на двух дисках одного сервера
|
||||||
|
никогда не окажется 2 и более копий одного и того же блока данных, а значит, даже
|
||||||
|
если в этом сервере откажут все диски, это будет равносильно потере только 1 копии
|
||||||
|
любого блока данных.
|
||||||
|
- Дерево распределения (Placement Tree / CRUSH Tree) - иерархическая группировка OSD
|
||||||
|
в узлы, которые далее можно использовать как домены отказа. То есть, диск (OSD) входит в
|
||||||
|
сервер, сервер входит в стойку, стойка входит в ряд, ряд в датацентр и т.п.
|
||||||
|
|
||||||
|
Чем Vitastor отличается от Ceph:
|
||||||
|
|
||||||
|
- Vitastor в первую очередь сфокусирован на SSD. Также Vitastor, вероятно, должен неплохо работать
|
||||||
|
с комбинацией SSD и HDD через bcache, а в будущем, возможно, будут добавлены и нативные способы
|
||||||
|
оптимизации под SSD+HDD. Однако хранилище на основе одних лишь жёстких дисков, вообще без SSD,
|
||||||
|
не в приоритете, поэтому оптимизации под этот кейс могут вообще не состояться.
|
||||||
|
- OSD Vitastor однопоточный и всегда таким останется, так как это самый оптимальный способ работы.
|
||||||
|
Если вам не хватает 1 ядра на 1 диск, просто делите диск на разделы и запускайте на нём несколько OSD.
|
||||||
|
Но, скорее всего, вам хватит и 1 ядра - Vitastor не так прожорлив к ресурсам CPU, как Ceph.
|
||||||
|
- Журнал и метаданные всегда размещаются в памяти, благодаря чему никогда не тратится лишнее время
|
||||||
|
на чтение метаданных с диска. Размер метаданных линейно зависит от размера диска и блока данных,
|
||||||
|
который задаётся в конфигурации кластера и по умолчанию составляет 128 КБ. С блоком 128 КБ метаданные
|
||||||
|
занимают примерно 512 МБ памяти на 1 ТБ дискового пространства (и это всё равно меньше, чем нужно Ceph-у).
|
||||||
|
Журнал вообще не должен быть большим, например, тесты производительности в данном документе проводились
|
||||||
|
с журналом размером всего 16 МБ. Большой журнал, вероятно, даже вреден, т.к. "грязные" записи (записи,
|
||||||
|
не сброшенные из журнала) тоже занимают память и могут немного замедлять работу.
|
||||||
|
- В Vitastor нет внутреннего copy-on-write. Я считаю, что реализация CoW-хранилища гораздо сложнее,
|
||||||
|
поэтому сложнее добиться устойчиво хороших результатов. Возможно, в один прекрасный день
|
||||||
|
я придумаю красивый алгоритм для CoW-хранилища, но пока нет - внутреннего CoW в Vitastor не будет.
|
||||||
|
Всё это не относится к "внешнему" CoW (снапшотам и клонам).
|
||||||
|
- Базовый слой Vitastor - простое блочное хранилище с блоками фиксированного размера, а не сложное
|
||||||
|
объектное хранилище с расширенными возможностями, как в Ceph (RADOS).
|
||||||
|
- В Vitastor есть режим "ленивых fsync", в котором OSD группирует запросы записи перед сбросом их
|
||||||
|
на диск, что позволяет получить лучшую производительность с дешёвыми настольными SSD без конденсаторов
|
||||||
|
("Advanced Power Loss Protection" / "Capacitor-Based Power Loss Protection").
|
||||||
|
Тем не менее, такой режим всё равно медленнее использования нормальных серверных SSD и мгновенного
|
||||||
|
fsync, так как приводит к дополнительным операциям передачи данных по сети, поэтому рекомендуется
|
||||||
|
всё-таки использовать хорошие серверные диски, тем более, стоят они почти так же, как десктопные.
|
||||||
|
- PG эфемерны. Это означает, что они не хранятся на дисках и существуют только в памяти работающих OSD.
|
||||||
|
- Процессы восстановления оперируют отдельными объектами, а не целыми PG.
|
||||||
|
- PGLOG-ов нет.
|
||||||
|
- "Мониторы" не хранят данные. Конфигурация и состояние кластера хранятся в etcd в простых человекочитаемых
|
||||||
|
JSON-структурах. Мониторы Vitastor только следят за состоянием кластера и управляют перемещением данных.
|
||||||
|
В этом смысле монитор Vitastor не является критичным компонентом системы и больше похож на Ceph-овский
|
||||||
|
менеджер (MGR). Монитор Vitastor написан на node.js.
|
||||||
|
- Распределение PG не основано на консистентных хешах. Вместо этого все маппинги PG хранятся прямо в etcd
|
||||||
|
(ибо нет никакой проблемы сохранить несколько сотен-тысяч записей в памяти, а не считать каждый раз хеши).
|
||||||
|
Перераспределение PG по OSD выполняется через математическую оптимизацию,
|
||||||
|
а конкретно, сведение задачи к ЛП (задаче линейного программирования) и решение оной с помощью утилиты
|
||||||
|
lp_solve. Такой подход позволяет обычно выравнивать распределение места почти идеально - равномерность
|
||||||
|
обычно составляет 96-99%, в отличие от Ceph, где на голом CRUSH-е без балансировщика обычно выходит 80-90%.
|
||||||
|
Также это позволяет минимизировать объём перемещения данных и случайность связей между OSD, а также менять
|
||||||
|
распределение вручную, не боясь сломать логику перебалансировки. В таком подходе есть и потенциальный
|
||||||
|
недостаток - есть предположение, что в очень большом кластере он может сломаться - однако вплоть до
|
||||||
|
нескольких сотен OSD подход точно работает нормально. Ну и, собственно, при необходимости легко
|
||||||
|
реализовать и консистентные хеши.
|
||||||
|
- Отдельный слой, подобный слою "CRUSH-правил", отсутствует. Вы настраиваете схемы отказоустойчивости,
|
||||||
|
домены отказа и правила выбора OSD напрямую в конфигурации пулов.
|
||||||
|
|
||||||
|
## Понимание сути производительности систем хранения
|
||||||
|
|
||||||
|
Вкратце: для быстрой хранилки задержки важнее, чем пиковые iops-ы.
|
||||||
|
|
||||||
|
Лучшая возможная задержка достигается при тестировании в 1 поток с глубиной очереди 1,
|
||||||
|
что приблизительно означает минимально нагруженное состояние кластера. В данном случае
|
||||||
|
IOPS = 1/задержка. Ни числом серверов, ни дисков, ни серверных процессов/потоков
|
||||||
|
задержка не масштабируется... Она зависит только от того, насколько быстро один
|
||||||
|
серверный процесс (и клиент) обрабатывают одну операцию.
|
||||||
|
|
||||||
|
Почему задержки важны? Потому, что некоторые приложения *не могут* использовать глубину
|
||||||
|
очереди больше 1, ибо их задача не параллелизуется. Важный пример - это все СУБД
|
||||||
|
с поддержкой консистентности (ACID), потому что все они обеспечивают её через
|
||||||
|
журналирование, а журналы пишутся последовательно и с fsync() после каждой операции.
|
||||||
|
|
||||||
|
fsync, кстати - это ещё одна очень важная вещь, про которую почти всегда забывают в тестах.
|
||||||
|
Смысл в том, что все современные диски имеют кэши/буферы записи и не гарантируют, что
|
||||||
|
данные реально физически записываются на носитель до того, как вы делаете fsync(),
|
||||||
|
который транслируется в команду сброса кэша операционной системой.
|
||||||
|
|
||||||
|
Дешёвые SSD для настольных ПК и ноутбуков очень быстрые без fsync - NVMe диски, например,
|
||||||
|
могут обработать порядка 80000 операций записи в секунду с глубиной очереди 1 без fsync.
|
||||||
|
Однако с fsync, когда они реально вынуждены писать каждый блок данных во флеш-память,
|
||||||
|
они выжимают лишь 1000-2000 операций записи в секунду (число практически постоянное
|
||||||
|
для всех моделей SSD).
|
||||||
|
|
||||||
|
Серверные SSD часто имеют суперконденсаторы, работающие как встроенный источник
|
||||||
|
бесперебойного питания и дающие дискам успеть сбросить их DRAM-кэш в постоянную
|
||||||
|
флеш-память при отключении питания. Благодаря этому диски с чистой совестью
|
||||||
|
*игнорируют fsync*, так как точно знают, что данные из кэша доедут до постоянной
|
||||||
|
памяти.
|
||||||
|
|
||||||
|
Все наиболее известные программные СХД, например, Ceph и внутренние СХД, используемые
|
||||||
|
такими облачными провайдерами, как Amazon, Google, Яндекс, медленные в смысле задержки.
|
||||||
|
В лучшем случае они дают задержки от 0.3мс на чтение и 0.6мс на запись 4 КБ блоками
|
||||||
|
даже при условии использования наилучшего возможного железа.
|
||||||
|
|
||||||
|
И это в эпоху SSD, когда вы можете пойти на рынок и купить там SSD, задержка которого
|
||||||
|
на чтение будет 0.1мс, а на запись - 0.04мс, за 100$ или даже дешевле.
|
||||||
|
|
||||||
|
Когда мне нужно быстро протестировать производительность дисковой подсистемы, я
|
||||||
|
использую следующие 6 команд, с небольшими вариациями:
|
||||||
|
|
||||||
|
- Линейная запись:
|
||||||
|
`fio -ioengine=libaio -direct=1 -invalidate=1 -name=test -bs=4M -iodepth=32 -rw=write -runtime=60 -filename=/dev/sdX`
|
||||||
|
- Линейное чтение:
|
||||||
|
`fio -ioengine=libaio -direct=1 -invalidate=1 -name=test -bs=4M -iodepth=32 -rw=read -runtime=60 -filename=/dev/sdX`
|
||||||
|
- Запись в 1 поток (T1Q1):
|
||||||
|
`fio -ioengine=libaio -direct=1 -invalidate=1 -name=test -bs=4k -iodepth=1 -fsync=1 -rw=randwrite -runtime=60 -filename=/dev/sdX`
|
||||||
|
- Чтение в 1 поток (T1Q1):
|
||||||
|
`fio -ioengine=libaio -direct=1 -invalidate=1 -name=test -bs=4k -iodepth=1 -rw=randread -runtime=60 -filename=/dev/sdX`
|
||||||
|
- Параллельная запись (numjobs используется, когда 1 ядро CPU не может насытить диск):
|
||||||
|
`fio -ioengine=libaio -direct=1 -invalidate=1 -name=test -bs=4k -iodepth=128 [-numjobs=4 -group_reporting] -rw=randwrite -runtime=60 -filename=/dev/sdX`
|
||||||
|
- Параллельное чтение (numjobs - аналогично):
|
||||||
|
`fio -ioengine=libaio -direct=1 -invalidate=1 -name=test -bs=4k -iodepth=128 [-numjobs=4 -group_reporting] -rw=randread -runtime=60 -filename=/dev/sdX`
|
||||||
|
|
||||||
|
## Теоретическая максимальная производительность Vitastor
|
||||||
|
|
||||||
|
При использовании репликации:
|
||||||
|
- Задержка чтения в 1 поток (T1Q1): 1 сетевой RTT + 1 чтение с диска.
|
||||||
|
- Запись+fsync в 1 поток:
|
||||||
|
- С мгновенным сбросом: 2 RTT + 1 запись.
|
||||||
|
- С отложенным ("ленивым") сбросом: 4 RTT + 1 запись + 1 fsync.
|
||||||
|
- Параллельное чтение: сумма IOPS всех дисков либо производительность сети, если в сеть упрётся раньше.
|
||||||
|
- Параллельная запись: сумма IOPS всех дисков / число реплик / WA либо производительность сети, если в сеть упрётся раньше.
|
||||||
|
|
||||||
|
При использовании кодов коррекции ошибок (EC):
|
||||||
|
- Задержка чтения в 1 поток (T1Q1): 1.5 RTT + 1 чтение.
|
||||||
|
- Запись+fsync в 1 поток:
|
||||||
|
- С мгновенным сбросом: 3.5 RTT + 1 чтение + 2 записи.
|
||||||
|
- С отложенным ("ленивым") сбросом: 5.5 RTT + 1 чтение + 2 записи + 2 fsync.
|
||||||
|
- Под 0.5 на самом деле подразумевается (k-1)/k, где k - число дисков данных,
|
||||||
|
что означает, что дополнительное обращение по сети не нужно, когда операция
|
||||||
|
чтения обслуживается локально.
|
||||||
|
- Параллельное чтение: сумма IOPS всех дисков либо производительность сети, если в сеть упрётся раньше.
|
||||||
|
- Параллельная запись: сумма IOPS всех дисков / общее число дисков данных и чётности / WA либо производительность сети, если в сеть упрётся раньше.
|
||||||
|
Примечание: IOPS дисков в данном случае надо брать в смешанном режиме чтения/записи в пропорции, аналогичной формулам выше.
|
||||||
|
|
||||||
|
WA (мультипликатор записи) для 4 КБ блоков в Vitastor обычно составляет 3-5:
|
||||||
|
1. Запись метаданных в журнал
|
||||||
|
2. Запись блока данных в журнал
|
||||||
|
3. Запись метаданных в БД
|
||||||
|
4. Ещё одна запись метаданных в журнал при использовании EC
|
||||||
|
5. Запись блока данных на диск данных
|
||||||
|
|
||||||
|
Если вы найдёте SSD, хорошо работающий с 512-байтными блоками данных (Optane?),
|
||||||
|
то 1, 3 и 4 можно снизить до 512 байт (1/8 от размера данных) и получить WA всего 2.375.
|
||||||
|
|
||||||
|
Кроме того, WA снижается при использовании отложенного/ленивого сброса при параллельной
|
||||||
|
нагрузке, т.к. блоки журнала записываются на диск только когда они заполняются или явным
|
||||||
|
образом запрашивается fsync.
|
||||||
|
|
||||||
|
## Пример сравнения с Ceph
|
||||||
|
|
||||||
|
Железо - 4 сервера, в каждом:
|
||||||
|
- 6x SATA SSD Intel D3-4510 3.84 TB
|
||||||
|
- 2x Xeon Gold 6242 (16 cores @ 2.8 GHz)
|
||||||
|
- 384 GB RAM
|
||||||
|
- 1x 25 GbE сетевая карта (Mellanox ConnectX-4 LX), подключённая к свитчу Juniper QFX5200
|
||||||
|
|
||||||
|
Экономия энергии CPU отключена. В тестах и Vitastor, и Ceph развёрнуто по 2 OSD на 1 SSD.
|
||||||
|
|
||||||
|
Все результаты ниже относятся к случайной нагрузке 4 КБ блоками (если явно не указано обратное).
|
||||||
|
|
||||||
|
Производительность голых дисков:
|
||||||
|
- T1Q1 запись ~27000 iops (задержка ~0.037ms)
|
||||||
|
- T1Q1 чтение ~9800 iops (задержка ~0.101ms)
|
||||||
|
- T1Q32 запись ~60000 iops
|
||||||
|
- T1Q32 чтение ~81700 iops
|
||||||
|
|
||||||
|
Ceph 15.2.4 (Bluestore):
|
||||||
|
- T1Q1 запись ~1000 iops (задержка ~1ms)
|
||||||
|
- T1Q1 чтение ~1750 iops (задержка ~0.57ms)
|
||||||
|
- T8Q64 запись ~100000 iops, потребление CPU процессами OSD около 40 ядер на каждом сервере
|
||||||
|
- T8Q64 чтение ~480000 iops, потребление CPU процессами OSD около 40 ядер на каждом сервере
|
||||||
|
|
||||||
|
Тесты в 8 потоков проводились на 8 400GB RBD образах со всех хостов (с каждого хоста запускалось 2 процесса fio).
|
||||||
|
Это нужно потому, что в Ceph несколько RBD-клиентов, пишущих в 1 образ, очень сильно замедляются.
|
||||||
|
|
||||||
|
Настройки RocksDB и Bluestore в Ceph не менялись, единственным изменением было отключение cephx_sign_messages.
|
||||||
|
|
||||||
|
На самом деле, результаты теста не такие уж и плохие для Ceph (могло быть хуже).
|
||||||
|
Собственно говоря, эти серверы как раз хорошо сбалансированы для Ceph - 6 SATA SSD как раз
|
||||||
|
утилизируют 25-гигабитную сеть, а без 2 мощных процессоров Ceph-у бы не хватило ядер,
|
||||||
|
чтобы выдать пристойный результат. Собственно, что и показывает жор 40 ядер в процессе
|
||||||
|
параллельного теста.
|
||||||
|
|
||||||
|
Vitastor:
|
||||||
|
- T1Q1 запись: 7087 iops (задержка 0.14ms)
|
||||||
|
- T1Q1 чтение: 6838 iops (задержка 0.145ms)
|
||||||
|
- T2Q64 запись: 162000 iops, потребление CPU - 3 ядра на каждом сервере
|
||||||
|
- T8Q64 чтение: 895000 iops, потребление CPU - 4 ядра на каждом сервере
|
||||||
|
- Линейная запись (4M T1Q32): 2800 МБ/с
|
||||||
|
- Линейное чтение (4M T1Q32): 1500 МБ/с
|
||||||
|
|
||||||
|
Тест на чтение в 8 потоков проводился на 1 большом образе (3.2 ТБ) со всех хостов (опять же, по 2 fio с каждого).
|
||||||
|
В Vitastor никакой разницы между 1 образом и 8-ю нет. Естественно, примерно 1/4 запросов чтения
|
||||||
|
в такой конфигурации, как и в тестах Ceph выше, обслуживалась с локальной машины. Если проводить
|
||||||
|
тест так, чтобы все операции всегда обращались к первичным OSD по сети - тест сильнее упирался
|
||||||
|
в сеть и результат составлял примерно 689000 iops.
|
||||||
|
|
||||||
|
Настройки Vitastor: `--disable_data_fsync true --immediate_commit all --flusher_count 8
|
||||||
|
--disk_alignment 4096 --journal_block_size 4096 --meta_block_size 4096
|
||||||
|
--journal_no_same_sector_overwrites true --journal_sector_buffer_count 1024
|
||||||
|
--journal_size 16777216`.
|
||||||
|
|
||||||
|
### EC/XOR 2+1
|
||||||
|
|
||||||
|
Vitastor:
|
||||||
|
- T1Q1 запись: 2808 iops (задержка ~0.355ms)
|
||||||
|
- T1Q1 чтение: 6190 iops (задержка ~0.16ms)
|
||||||
|
- T2Q64 запись: 85500 iops, потребление CPU - 3.4 ядра на каждом сервере
|
||||||
|
- T8Q64 чтение: 812000 iops, потребление CPU - 4.7 ядра на каждом сервере
|
||||||
|
- Линейная запись (4M T1Q32): 3200 МБ/с
|
||||||
|
- Линейное чтение (4M T1Q32): 1800 МБ/с
|
||||||
|
|
||||||
|
Ceph:
|
||||||
|
- T1Q1 запись: 730 iops (задержка ~1.37ms latency)
|
||||||
|
- T1Q1 чтение: 1500 iops с холодным кэшем метаданных (задержка ~0.66ms), 2300 iops через 2 минуты прогрева (задержка ~0.435ms)
|
||||||
|
- T4Q128 запись (4 RBD images): 45300 iops, потребление CPU - 30 ядер на каждом сервере
|
||||||
|
- T8Q64 чтение (4 RBD images): 278600 iops, потребление CPU - 40 ядер на каждом сервере
|
||||||
|
- Линейная запись (4M T1Q32): 1950 МБ/с в пустой образ, 2500 МБ/с в заполненный образ
|
||||||
|
- Линейное чтение (4M T1Q32): 2400 МБ/с
|
||||||
|
|
||||||
|
### NBD
|
||||||
|
|
||||||
|
NBD расшифровывается как "сетевое блочное устройство", но на самом деле оно также
|
||||||
|
работает просто как аналог FUSE для блочных устройств, то есть, представляет собой
|
||||||
|
"блочное устройство в пространстве пользователя".
|
||||||
|
|
||||||
|
NBD - на данный момент единственный способ монтировать Vitastor ядром Linux.
|
||||||
|
NBD немного снижает производительность, так как приводит к дополнительным копированиям
|
||||||
|
данных между ядром и пространством пользователя. Тем не менее, способ достаточно оптимален,
|
||||||
|
а производительность случайного доступа вообще затрагивается слабо.
|
||||||
|
|
||||||
|
Vitastor с однопоточной NBD прокси на том же стенде:
|
||||||
|
- T1Q1 запись: 6000 iops (задержка 0.166ms)
|
||||||
|
- T1Q1 чтение: 5518 iops (задержка 0.18ms)
|
||||||
|
- T1Q128 запись: 94400 iops
|
||||||
|
- T1Q128 чтение: 103000 iops
|
||||||
|
- Линейная запись (4M T1Q128): 1266 МБ/с (в сравнении с 2800 МБ/с через fio)
|
||||||
|
- Линейное чтение (4M T1Q128): 975 МБ/с (в сравнении с 1500 МБ/с через fio)
|
||||||
|
|
||||||
|
## Установка
|
||||||
|
|
||||||
|
### Debian
|
||||||
|
|
||||||
|
- Добавьте ключ репозитория Vitastor:
|
||||||
|
`wget -q -O - https://vitastor.io/debian/pubkey | sudo apt-key add -`
|
||||||
|
- Добавьте репозиторий Vitastor в /etc/apt/sources.list:
|
||||||
|
- Debian 11 (Bullseye/Sid): `deb https://vitastor.io/debian bullseye main`
|
||||||
|
- Debian 10 (Buster): `deb https://vitastor.io/debian buster main`
|
||||||
|
- Для Debian 10 (Buster) также включите репозиторий backports:
|
||||||
|
`deb http://deb.debian.org/debian buster-backports main`
|
||||||
|
- Установите пакеты: `apt update; apt install vitastor lp-solve etcd linux-image-amd64 qemu`
|
||||||
|
|
||||||
|
### CentOS
|
||||||
|
|
||||||
|
- Добавьте в систему репозиторий Vitastor:
|
||||||
|
- CentOS 7: `yum install https://vitastor.io/rpms/centos/7/vitastor-release-1.0-1.el7.noarch.rpm`
|
||||||
|
- CentOS 8: `dnf install https://vitastor.io/rpms/centos/8/vitastor-release-1.0-1.el8.noarch.rpm`
|
||||||
|
- Включите EPEL: `yum/dnf install epel-release`
|
||||||
|
- Включите дополнительные репозитории CentOS:
|
||||||
|
- CentOS 7: `yum install centos-release-scl`
|
||||||
|
- CentOS 8: `dnf install centos-release-advanced-virtualization`
|
||||||
|
- Включите elrepo-kernel:
|
||||||
|
- CentOS 7: `yum install https://www.elrepo.org/elrepo-release-7.el7.elrepo.noarch.rpm`
|
||||||
|
- CentOS 8: `dnf install https://www.elrepo.org/elrepo-release-8.el8.elrepo.noarch.rpm`
|
||||||
|
- Установите пакеты: `yum/dnf install vitastor lpsolve etcd kernel-ml qemu-kvm`
|
||||||
|
|
||||||
|
### Установка из исходников
|
||||||
|
|
||||||
|
- Установите ядро 5.4 или более новое, для поддержки io_uring. Желательно 5.8 или даже новее,
|
||||||
|
так как в 5.4 есть как минимум 1 известный баг, ведущий к зависанию с io_uring и контроллером HP SmartArray.
|
||||||
|
- Установите liburing 0.4 или более новый и его заголовки.
|
||||||
|
- Установите lp_solve.
|
||||||
|
- Установите etcd, версии не ниже 3.4.15. Более ранние версии работать не будут из-за различных багов,
|
||||||
|
например [#12402](https://github.com/etcd-io/etcd/pull/12402). Также вы можете взять версию 3.4.13 с
|
||||||
|
этим конкретным исправлением из ветки release-3.4 репозитория https://github.com/vitalif/etcd/.
|
||||||
|
- Установите node.js 10 или новее.
|
||||||
|
- Установите gcc и g++ 8.x или новее.
|
||||||
|
- Склонируйте данный репозиторий с подмодулями: `git clone https://yourcmc.ru/git/vitalif/vitastor/`.
|
||||||
|
- Желательно пересобрать QEMU с патчем, который делает необязательным запуск через LD_PRELOAD.
|
||||||
|
См `patches/qemu-*.*-vitastor.patch` - выберите версию, наиболее близкую вашей версии QEMU.
|
||||||
|
- Установите QEMU 3.0 или новее, возьмите исходные коды установленного пакета, начните его пересборку,
|
||||||
|
через некоторое время остановите её и скопируйте следующие заголовки:
|
||||||
|
- `<qemu>/include` → `<vitastor>/qemu/include`
|
||||||
|
- Debian:
|
||||||
|
* Берите qemu из основного репозитория
|
||||||
|
* `<qemu>/b/qemu/config-host.h` → `<vitastor>/qemu/b/qemu/config-host.h`
|
||||||
|
* `<qemu>/b/qemu/qapi` → `<vitastor>/qemu/b/qemu/qapi`
|
||||||
|
- CentOS 8:
|
||||||
|
* Берите qemu из репозитория Advanced-Virtualization. Чтобы включить его, запустите
|
||||||
|
`yum install centos-release-advanced-virtualization.noarch` и далее `yum install qemu`
|
||||||
|
* `<qemu>/config-host.h` → `<vitastor>/qemu/b/qemu/config-host.h`
|
||||||
|
* Для QEMU 3.0+: `<qemu>/qapi` → `<vitastor>/qemu/b/qemu/qapi`
|
||||||
|
* Для QEMU 2.0+: `<qemu>/qapi-types.h` → `<vitastor>/qemu/b/qemu/qapi-types.h`
|
||||||
|
- `config-host.h` и `qapi` нужны, т.к. в них содержатся автогенерируемые заголовки
|
||||||
|
- Установите fio 3.7 или новее, возьмите исходники пакета и сделайте на них симлинк с `<vitastor>/fio`.
|
||||||
|
- Соберите и установите Vitastor командой `mkdir build && cd build && cmake .. && make -j8 && make install`.
|
||||||
|
Обратите внимание на переменную cmake `QEMU_PLUGINDIR` - под RHEL её нужно установить равной `qemu-kvm`.
|
||||||
|
|
||||||
|
## Запуск
|
||||||
|
|
||||||
|
Внимание: процедура пока что достаточно нетривиальная, задавать конфигурацию и смещения
|
||||||
|
на диске нужно почти вручную. Это будет исправлено в ближайшем будущем.
|
||||||
|
|
||||||
|
- Желательны SATA SSD или NVMe диски с конденсаторами (серверные SSD). Можно использовать и
|
||||||
|
десктопные SSD, включив режим отложенного fsync, но производительность однопоточной записи
|
||||||
|
в этом случае пострадает.
|
||||||
|
- Быстрая сеть, минимум 10 гбит/с
|
||||||
|
- Для наилучшей производительности нужно отключить энергосбережение CPU: `cpupower idle-set -D 0 && cpupower frequency-set -g performance`.
|
||||||
|
- На хостах мониторов:
|
||||||
|
- Пропишите нужные вам значения в файле `/usr/lib/vitastor/mon/make-units.sh`
|
||||||
|
- Создайте юниты systemd для etcd и мониторов: `/usr/lib/vitastor/mon/make-units.sh`
|
||||||
|
- Запустите etcd и мониторы: `systemctl start etcd vitastor-mon`
|
||||||
|
- Пропишите etcd_address и osd_network в `/etc/vitastor/vitastor.conf`. Например:
|
||||||
|
```
|
||||||
|
{
|
||||||
|
"etcd_address": ["10.200.1.10:2379","10.200.1.11:2379","10.200.1.12:2379"],
|
||||||
|
"osd_network": "10.200.1.0/24"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
- Инициализуйте OSD:
|
||||||
|
- SSD: `/usr/lib/vitastor/make-osd.sh /dev/disk/by-partuuid/XXX [/dev/disk/by-partuuid/YYY ...]`
|
||||||
|
- Гибридные, HDD+SSD: `/usr/lib/vitastor/mon/make-osd-hybrid.js /dev/sda /dev/sdb ...` - передайте
|
||||||
|
все ваши SSD и HDD скрипту в командной строке подряд, скрипт автоматически выделит разделы под
|
||||||
|
журналы на SSD и данные на HDD. Скрипт пропускает HDD, на которых уже есть разделы
|
||||||
|
или вообще какие-то данные, поэтому если диски непустые, сначала очистите их с помощью
|
||||||
|
`wipefs -a`. SSD с таблицей разделов не пропускаются, но так как скрипт создаёт новые разделы
|
||||||
|
для журналов, на SSD должно быть доступно свободное нераспределённое место.
|
||||||
|
- Вы можете менять параметры OSD в юнитах systemd или в `vitastor.conf`. Смысл некоторых параметров:
|
||||||
|
- `disable_data_fsync 1` - отключает fsync, используется с SSD с конденсаторами.
|
||||||
|
- `immediate_commit all` - используется с SSD с конденсаторами.
|
||||||
|
Внимание: если установлено, также нужно установить его в то же значение в etcd в /vitastor/config/global
|
||||||
|
- `disable_device_lock 1` - отключает блокировку файла устройства, нужно, только если вы запускаете
|
||||||
|
несколько OSD на одном блочном устройстве.
|
||||||
|
- `flusher_count 256` - "flusher" - микропоток, удаляющий старые данные из журнала.
|
||||||
|
Не волнуйтесь об этой настройке, 256 теперь достаточно практически всегда.
|
||||||
|
- `disk_alignment`, `journal_block_size`, `meta_block_size` следует установить равными размеру
|
||||||
|
внутреннего блока SSD. Это почти всегда 4096.
|
||||||
|
- `journal_no_same_sector_overwrites true` запрещает перезапись одного и того же сектора журнала подряд
|
||||||
|
много раз в процессе записи. Большинство (99%) SSD не нуждаются в данной опции. Однако выяснилось, что
|
||||||
|
диски, используемые на одном из тестовых стендов - Intel D3-S4510 - очень сильно не любят такую
|
||||||
|
перезапись, и для них была добавлена эта опция. Когда данный режим включён, также нужно поднимать
|
||||||
|
значение `journal_sector_buffer_count`, так как иначе Vitastor не хватит буферов для записи в журнал.
|
||||||
|
- Создайте глобальную конфигурацию в etcd: `etcdctl --endpoints=... put /vitastor/config/global '{"immediate_commit":"all"}'`
|
||||||
|
(если все ваши диски - серверные с конденсаторами).
|
||||||
|
- Создайте пулы: `etcdctl --endpoints=... put /vitastor/config/pools '{"1":{"name":"testpool","scheme":"replicated","pg_size":2,"pg_minsize":1,"pg_count":256,"failure_domain":"host"}}'`.
|
||||||
|
Для jerasure EC-пулов конфигурация должна выглядеть так: `2:{"name":"ecpool","scheme":"jerasure","pg_size":4,"parity_chunks":2,"pg_minsize":2,"pg_count":256,"failure_domain":"host"}`.
|
||||||
|
- Запустите все OSD: `systemctl start vitastor.target`
|
||||||
|
- Ваш кластер должен быть готов - один из мониторов должен уже сконфигурировать PG, а OSD должны запустить их.
|
||||||
|
- Вы можете проверить состояние PG прямо в etcd: `etcdctl --endpoints=... get --prefix /vitastor/pg/state`. Все PG должны быть 'active'.
|
||||||
|
|
||||||
|
### Задать имя образу
|
||||||
|
|
||||||
|
```
|
||||||
|
etcdctl --endpoints=<etcd> put /vitastor/config/inode/<pool>/<inode> '{"name":"<name>","size":<size>[,"parent_id":<parent_inode_number>][,"readonly":true]}'
|
||||||
|
```
|
||||||
|
|
||||||
|
Например:
|
||||||
|
|
||||||
|
```
|
||||||
|
etcdctl --endpoints=http://10.115.0.10:2379/v3 put /vitastor/config/inode/1/1 '{"name":"testimg","size":2147483648}'
|
||||||
|
```
|
||||||
|
|
||||||
|
Если вы зададите parent_id, то образ станет CoW-клоном, т.е. все новые запросы записи пойдут в новый инод, а запросы
|
||||||
|
чтения будут проверять сначала его, а потом родительские слои по цепочке вверх. Чтобы случайно не перезаписать данные
|
||||||
|
в родительском слое, вы можете переключить его в режим "только чтение", добавив флаг `"readonly":true` в его запись
|
||||||
|
метаданных. В таком случае родительский образ становится просто снапшотом.
|
||||||
|
|
||||||
|
Таким образом, для создания снапшота вам нужно просто переименовать предыдущий inode (например, из testimg в testimg@0),
|
||||||
|
сделать его readonly и создать новый слой с исходным именем образа (testimg), ссылающийся на только что переименованный
|
||||||
|
в качестве родительского.
|
||||||
|
|
||||||
|
### Запуск тестов с fio
|
||||||
|
|
||||||
|
Пример команды для запуска тестов:
|
||||||
|
|
||||||
|
```
|
||||||
|
fio -thread -ioengine=libfio_vitastor.so -name=test -bs=4M -direct=1 -iodepth=16 -rw=write -etcd=10.115.0.10:2379/v3 -image=testimg
|
||||||
|
```
|
||||||
|
|
||||||
|
Если вы не хотите обращаться к образу по имени, вместо `-image=testimg` можно указать номер пула, номер инода и размер:
|
||||||
|
`-pool=1 -inode=1 -size=400G`.
|
||||||
|
|
||||||
|
### Загрузить образ диска ВМ в/из Vitastor
|
||||||
|
|
||||||
|
Используйте qemu-img и строку `vitastor:etcd_host=<HOST>:image=<IMAGE>` в качестве имени файла диска. Например:
|
||||||
|
|
||||||
|
```
|
||||||
|
qemu-img convert -f qcow2 debian10.qcow2 -p -O raw 'vitastor:etcd_host=10.115.0.10\:2379/v3:image=testimg'
|
||||||
|
```
|
||||||
|
|
||||||
|
Обратите внимание, что если вы используете немодифицированный QEMU, потребуется установить переменную окружения
|
||||||
|
`LD_PRELOAD=/usr/lib/x86_64-linux-gnu/qemu/block-vitastor.so`.
|
||||||
|
|
||||||
|
Если вы не хотите обращаться к образу по имени, вместо `:image=<IMAGE>` можно указать номер пула, номер инода и размер:
|
||||||
|
`:pool=<POOL>:inode=<INODE>:size=<SIZE>`.
|
||||||
|
|
||||||
|
### Запустить ВМ
|
||||||
|
|
||||||
|
Для запуска QEMU используйте опцию `-drive file=vitastor:etcd_host=<HOST>:image=<IMAGE>` (аналогично qemu-img)
|
||||||
|
и физический размер блока 4 KB.
|
||||||
|
|
||||||
|
Например:
|
||||||
|
|
||||||
|
```
|
||||||
|
qemu-system-x86_64 -enable-kvm -m 1024
|
||||||
|
-drive 'file=vitastor:etcd_host=10.115.0.10\:2379/v3:image=testimg',format=raw,if=none,id=drive-virtio-disk0,cache=none
|
||||||
|
-device virtio-blk-pci,scsi=off,bus=pci.0,addr=0x5,drive=drive-virtio-disk0,id=virtio-disk0,bootindex=1,write-cache=off,physical_block_size=4096,logical_block_size=512
|
||||||
|
-vnc 0.0.0.0:0
|
||||||
|
```
|
||||||
|
|
||||||
|
Обращение по номерам (`:pool=<POOL>:inode=<INODE>:size=<SIZE>` вместо `:image=<IMAGE>`) работает аналогично qemu-img.
|
||||||
|
|
||||||
|
### Удалить образ
|
||||||
|
|
||||||
|
Используйте утилиту vitastor-cli rm-data. Например:
|
||||||
|
|
||||||
|
```
|
||||||
|
vitastor-cli rm-data --etcd_address 10.115.0.10:2379/v3 --pool 1 --inode 1 --parallel_osds 16 --iodepth 32
|
||||||
|
```
|
||||||
|
|
||||||
|
### NBD
|
||||||
|
|
||||||
|
Чтобы создать локальное блочное устройство, используйте NBD. Например:
|
||||||
|
|
||||||
|
```
|
||||||
|
vitastor-nbd map --etcd_address 10.115.0.10:2379/v3 --image testimg
|
||||||
|
```
|
||||||
|
|
||||||
|
Команда напечатает название устройства вида /dev/nbd0, которое потом можно будет форматировать
|
||||||
|
и использовать как обычное блочное устройство.
|
||||||
|
|
||||||
|
Для обращения по номеру инода, аналогично другим командам, можно использовать опции
|
||||||
|
`--pool <POOL> --inode <INODE> --size <SIZE>` вместо `--image testimg`.
|
||||||
|
|
||||||
|
### Kubernetes
|
||||||
|
|
||||||
|
У Vitastor есть CSI-плагин для Kubernetes, поддерживающий RWO-тома.
|
||||||
|
|
||||||
|
Для установки возьмите манифесты из директории [csi/deploy/](csi/deploy/), поместите
|
||||||
|
вашу конфигурацию подключения к Vitastor в [csi/deploy/001-csi-config-map.yaml](001-csi-config-map.yaml),
|
||||||
|
настройте StorageClass в [csi/deploy/009-storage-class.yaml](009-storage-class.yaml)
|
||||||
|
и примените все `NNN-*.yaml` к вашей инсталляции Kubernetes.
|
||||||
|
|
||||||
|
```
|
||||||
|
for i in ./???-*.yaml; do kubectl apply -f $i; done
|
||||||
|
```
|
||||||
|
|
||||||
|
После этого вы сможете создавать PersistentVolume. Пример смотрите в файле [csi/deploy/example-pvc.yaml](csi/deploy/example-pvc.yaml).
|
||||||
|
|
||||||
|
### OpenStack
|
||||||
|
|
||||||
|
Чтобы подключить Vitastor к OpenStack:
|
||||||
|
|
||||||
|
- Установите пакеты vitastor-client, libvirt и QEMU из DEB или RPM репозитория Vitastor
|
||||||
|
- Примените патч `patches/nova-21.diff` или `patches/nova-23.diff` к вашей инсталляции Nova.
|
||||||
|
nova-21.diff подходит для Nova 21-22, nova-23.diff подходит для Nova 23-24.
|
||||||
|
- Скопируйте `patches/cinder-vitastor.py` в инсталляцию Cinder как `cinder/volume/drivers/vitastor.py`
|
||||||
|
- Создайте тип томов в cinder.conf (см. ниже)
|
||||||
|
- Обязательно заблокируйте доступ от виртуальных машин к сети Vitastor (OSD и etcd), т.к. Vitastor (пока) не поддерживает аутентификацию
|
||||||
|
- Перезапустите Cinder и Nova
|
||||||
|
|
||||||
|
Пример конфигурации Cinder:
|
||||||
|
|
||||||
|
```
|
||||||
|
[DEFAULT]
|
||||||
|
enabled_backends = lvmdriver-1, vitastor-testcluster
|
||||||
|
# ...
|
||||||
|
|
||||||
|
[vitastor-testcluster]
|
||||||
|
volume_driver = cinder.volume.drivers.vitastor.VitastorDriver
|
||||||
|
volume_backend_name = vitastor-testcluster
|
||||||
|
image_volume_cache_enabled = True
|
||||||
|
volume_clear = none
|
||||||
|
vitastor_etcd_address = 192.168.7.2:2379
|
||||||
|
vitastor_etcd_prefix =
|
||||||
|
vitastor_config_path = /etc/vitastor/vitastor.conf
|
||||||
|
vitastor_pool_id = 1
|
||||||
|
image_upload_use_cinder_backend = True
|
||||||
|
```
|
||||||
|
|
||||||
|
Чтобы помещать в Vitastor Glance-образы, нужно использовать
|
||||||
|
[https://docs.openstack.org/cinder/pike/admin/blockstorage-volume-backed-image.html](образы на основе томов Cinder),
|
||||||
|
однако, поддержка этой функции ещё не проверялась.
|
||||||
|
|
||||||
|
### Proxmox
|
||||||
|
|
||||||
|
Чтобы подключить Vitastor к Proxmox Virtual Environment (поддерживаются версии 6.4 и 7.1):
|
||||||
|
|
||||||
|
- Добавьте соответствующий Debian-репозиторий Vitastor в sources.list на хостах Proxmox
|
||||||
|
(buster для 6.4, bullseye для 7.1)
|
||||||
|
- Установите пакеты vitastor-client, pve-qemu-kvm, pve-storage-vitastor (* или см. сноску) из репозитория Vitastor
|
||||||
|
- Определите тип хранилища в `/etc/pve/storage.cfg` (см. ниже)
|
||||||
|
- Обязательно заблокируйте доступ от виртуальных машин к сети Vitastor (OSD и etcd), т.к. Vitastor (пока) не поддерживает аутентификацию
|
||||||
|
- Перезапустите демон Proxmox: `systemctl restart pvedaemon`
|
||||||
|
|
||||||
|
Пример `/etc/pve/storage.cfg` (единственная обязательная опция - vitastor_pool, все остальные
|
||||||
|
перечислены внизу для понимания значений по умолчанию):
|
||||||
|
|
||||||
|
```
|
||||||
|
vitastor: vitastor
|
||||||
|
# Пул, в который будут помещаться образы дисков
|
||||||
|
vitastor_pool testpool
|
||||||
|
# Путь к файлу конфигурации
|
||||||
|
vitastor_config_path /etc/vitastor/vitastor.conf
|
||||||
|
# Адрес(а) etcd, нужны, только если не указаны в vitastor.conf
|
||||||
|
vitastor_etcd_address 192.168.7.2:2379/v3
|
||||||
|
# Префикс ключей метаданных в etcd
|
||||||
|
vitastor_etcd_prefix /vitastor
|
||||||
|
# Префикс имён образов
|
||||||
|
vitastor_prefix pve/
|
||||||
|
# Монтировать образы через NBD прокси, через ядро (нужно только для контейнеров)
|
||||||
|
vitastor_nbd 0
|
||||||
|
```
|
||||||
|
|
||||||
|
\* Примечание: вместо установки пакета pve-storage-vitastor вы можете вручную скопировать файл
|
||||||
|
[patches/PVE_VitastorPlugin.pm](patches/PVE_VitastorPlugin.pm) на хосты Proxmox как
|
||||||
|
`/usr/share/perl5/PVE/Storage/Custom/VitastorPlugin.pm`.
|
||||||
|
|
||||||
|
## Известные проблемы
|
||||||
|
|
||||||
|
- Запросы удаления объектов могут в данный момент приводить к "неполным" объектам в EC-пулах,
|
||||||
|
если в процессе удаления произойдут отказы OSD или серверов, потому что правильная обработка
|
||||||
|
запросов удаления в кластере должна быть "трёхфазной", а это пока не реализовано. Если вы
|
||||||
|
столкнётесь с такой ситуацией, просто повторите запрос удаления.
|
||||||
|
|
||||||
|
## Принципы реализации
|
||||||
|
|
||||||
|
- Я люблю архитектурно простые решения. Vitastor проектируется именно так и я намерен
|
||||||
|
и далее следовать данному принципу.
|
||||||
|
- Если вы пришли сюда за идеальным кодом на C++, вы, вероятно, не по адресу. "Общепринятые"
|
||||||
|
практики написания C++ кода меня не очень волнуют, так как зачастую, опять-таки, ведут к
|
||||||
|
излишним усложнениям и код получается красивый... но медленный.
|
||||||
|
- По той же причине в коде иногда можно встретить велосипеды типа собственного упрощённого
|
||||||
|
HTTP-клиента для работы с etcd. Зато эти велосипеды маленькие и компактные и не требуют
|
||||||
|
использования десятка внешних библиотек.
|
||||||
|
- node.js для монитора - не случайный выбор. Он очень быстрый, имеет встроенную событийную
|
||||||
|
машину, приятный нейтральный C-подобный язык программирования и развитую инфраструктуру.
|
||||||
|
|
||||||
|
## Автор и лицензия
|
||||||
|
|
||||||
|
Автор: Виталий Филиппов (vitalif [at] yourcmc.ru), 2019+
|
||||||
|
|
||||||
|
Заходите в Telegram-чат Vitastor: https://t.me/vitastor
|
||||||
|
|
||||||
|
Лицензия: VNPL 1.1 на серверный код и двойная VNPL 1.1 + GPL 2.0+ на клиентский.
|
||||||
|
|
||||||
|
VNPL - "сетевой копилефт", собственная свободная копилефт-лицензия
|
||||||
|
Vitastor Network Public License 1.1, основанная на GNU GPL 3.0 с дополнительным
|
||||||
|
условием "Сетевого взаимодействия", требующим распространять все программы,
|
||||||
|
специально разработанные для использования вместе с Vitastor и взаимодействующие
|
||||||
|
с ним по сети, под лицензией VNPL или под любой другой свободной лицензией.
|
||||||
|
|
||||||
|
Идея VNPL - расширение действия копилефта не только на модули, явным образом
|
||||||
|
связываемые с кодом Vitastor, но также на модули, оформленные в виде микросервисов
|
||||||
|
и взаимодействующие с ним по сети.
|
||||||
|
|
||||||
|
Таким образом, если вы хотите построить на основе Vitastor сервис, содержаший
|
||||||
|
компоненты с закрытым кодом, взаимодействующие с Vitastor, вам нужна коммерческая
|
||||||
|
лицензия от автора 😀.
|
||||||
|
|
||||||
|
На Windows и любое другое ПО, не разработанное *специально* для использования
|
||||||
|
вместе с Vitastor, никакие ограничения не накладываются.
|
||||||
|
|
||||||
|
Клиентские библиотеки распространяются на условиях двойной лицензии VNPL 1.0
|
||||||
|
и также на условиях GNU GPL 2.0 или более поздней версии. Так сделано в целях
|
||||||
|
совместимости с таким ПО, как QEMU и fio.
|
||||||
|
|
||||||
|
Вы можете найти полный текст VNPL 1.1 в файле [VNPL-1.1.txt](VNPL-1.1.txt),
|
||||||
|
а GPL 2.0 в файле [GPL-2.0.txt](GPL-2.0.txt).
|
||||||
@@ -1,5 +1,7 @@
|
|||||||
## Vitastor
|
## Vitastor
|
||||||
|
|
||||||
|
[Читать на русском](README-ru.md)
|
||||||
|
|
||||||
## The Idea
|
## The Idea
|
||||||
|
|
||||||
Make Software-Defined Block Storage Great Again.
|
Make Software-Defined Block Storage Great Again.
|
||||||
@@ -16,7 +18,8 @@ breaking changes in the future. However, the following is implemented:
|
|||||||
|
|
||||||
- Basic part: highly-available block storage with symmetric clustering and no SPOF
|
- Basic part: highly-available block storage with symmetric clustering and no SPOF
|
||||||
- Performance ;-D
|
- Performance ;-D
|
||||||
- Two redundancy schemes: Replication and XOR n+1 (simplest case of EC)
|
- Multiple redundancy schemes: Replication, XOR n+1, Reed-Solomon erasure codes
|
||||||
|
based on jerasure library with any number of data and parity drives in a group
|
||||||
- Configuration via simple JSON data structures in etcd
|
- Configuration via simple JSON data structures in etcd
|
||||||
- Automatic data distribution over OSDs, with support for:
|
- Automatic data distribution over OSDs, with support for:
|
||||||
- Mathematical optimization for better uniformity and less data movement
|
- Mathematical optimization for better uniformity and less data movement
|
||||||
@@ -31,25 +34,32 @@ breaking changes in the future. However, the following is implemented:
|
|||||||
- QEMU driver (built out-of-tree)
|
- QEMU driver (built out-of-tree)
|
||||||
- Loadable fio engine for benchmarks (also built out-of-tree)
|
- Loadable fio engine for benchmarks (also built out-of-tree)
|
||||||
- NBD proxy for kernel mounts
|
- NBD proxy for kernel mounts
|
||||||
- Inode removal tool (vitastor-rm)
|
- Inode removal tool (vitastor-cli rm-data)
|
||||||
- Packaging for Debian and CentOS
|
- Packaging for Debian and CentOS
|
||||||
|
- Per-inode I/O and space usage statistics
|
||||||
|
- Inode metadata storage in etcd
|
||||||
|
- Snapshots and copy-on-write image clones
|
||||||
|
- Write throttling to smooth random write workloads in SSD+HDD configurations
|
||||||
|
- RDMA/RoCEv2 support via libibverbs
|
||||||
|
- CSI plugin for Kubernetes
|
||||||
|
- Basic OpenStack support: Cinder driver, Nova and libvirt patches
|
||||||
|
- Snapshot merge tool (vitastor-cli {snap-rm,flatten,merge})
|
||||||
|
- Image management CLI (vitastor-cli {ls,create,modify})
|
||||||
|
- Proxmox storage plugin
|
||||||
|
|
||||||
## Roadmap
|
## Roadmap
|
||||||
|
|
||||||
- OSD creation tool (OSDs currently have to be created by hand)
|
- Better OSD creation and auto-start tools
|
||||||
- Other administrative tools
|
- Other administrative tools
|
||||||
- Per-inode I/O and space usage statistics
|
- Plugins for OpenNebula and other cloud systems
|
||||||
- jerasure EC support with any number of data and parity drives in a group
|
|
||||||
- Parallel usage of multiple network interfaces
|
|
||||||
- Proxmox and OpenNebula plugins
|
|
||||||
- iSCSI proxy
|
- iSCSI proxy
|
||||||
- Inode metadata storage in etcd
|
- Simplified NFS proxy
|
||||||
- Snapshots and copy-on-write image clones
|
- Faster failover
|
||||||
- Operation timeouts and better failure detection
|
|
||||||
- Scrubbing without checksums (verification of replicas)
|
- Scrubbing without checksums (verification of replicas)
|
||||||
- Checksums
|
- Checksums
|
||||||
- SSD+HDD optimizations, possibly including tiered storage and soft journal flushes
|
- Tiered storage
|
||||||
- RDMA and NVDIMM support
|
- NVDIMM support
|
||||||
|
- Web GUI
|
||||||
- Compression (possibly)
|
- Compression (possibly)
|
||||||
- Read caching using system page cache (possibly)
|
- Read caching using system page cache (possibly)
|
||||||
|
|
||||||
@@ -291,7 +301,7 @@ Vitastor with single-thread NBD on the same hardware:
|
|||||||
- Debian 10 (Buster): `deb https://vitastor.io/debian buster main`
|
- Debian 10 (Buster): `deb https://vitastor.io/debian buster main`
|
||||||
- For Debian 10 (Buster) also enable backports repository:
|
- For Debian 10 (Buster) also enable backports repository:
|
||||||
`deb http://deb.debian.org/debian buster-backports main`
|
`deb http://deb.debian.org/debian buster-backports main`
|
||||||
- Install packages: `apt update; apt install vitastor lp-solve etcd linux-image-amd64`
|
- Install packages: `apt update; apt install vitastor lp-solve etcd linux-image-amd64 qemu`
|
||||||
|
|
||||||
### CentOS
|
### CentOS
|
||||||
|
|
||||||
@@ -313,10 +323,9 @@ Vitastor with single-thread NBD on the same hardware:
|
|||||||
there is at least one known io_uring hang with 5.4 and an HP SmartArray controller.
|
there is at least one known io_uring hang with 5.4 and an HP SmartArray controller.
|
||||||
- Install liburing 0.4 or newer and its headers.
|
- Install liburing 0.4 or newer and its headers.
|
||||||
- Install lp_solve.
|
- Install lp_solve.
|
||||||
- Install etcd. Attention: you need a fixed version from here: https://github.com/vitalif/etcd/,
|
- Install etcd, at least version 3.4.15. Earlier versions won't work because of various bugs,
|
||||||
branch release-3.4, because there is a bug in upstream etcd which makes Vitastor OSDs fail to
|
for example [#12402](https://github.com/etcd-io/etcd/pull/12402). You can also take 3.4.13
|
||||||
move PGs out of "starting" state if you have at least around ~500 PGs or so. The custom build
|
with this specific fix from here: https://github.com/vitalif/etcd/, branch release-3.4.
|
||||||
will be unnecessary when etcd merges the fix: https://github.com/etcd-io/etcd/pull/12402.
|
|
||||||
- Install node.js 10 or newer.
|
- Install node.js 10 or newer.
|
||||||
- Install gcc and g++ 8.x or newer.
|
- Install gcc and g++ 8.x or newer.
|
||||||
- Clone https://yourcmc.ru/git/vitalif/vitastor/ with submodules.
|
- Clone https://yourcmc.ru/git/vitalif/vitastor/ with submodules.
|
||||||
@@ -334,11 +343,10 @@ Vitastor with single-thread NBD on the same hardware:
|
|||||||
* For QEMU 2.0+: `<qemu>/qapi-types.h` → `<vitastor>/qemu/b/qemu/qapi-types.h`
|
* For QEMU 2.0+: `<qemu>/qapi-types.h` → `<vitastor>/qemu/b/qemu/qapi-types.h`
|
||||||
- `config-host.h` and `qapi` are required because they contain generated headers
|
- `config-host.h` and `qapi` are required because they contain generated headers
|
||||||
- You can also rebuild QEMU with a patch that makes LD_PRELOAD unnecessary to load vitastor driver.
|
- You can also rebuild QEMU with a patch that makes LD_PRELOAD unnecessary to load vitastor driver.
|
||||||
See `qemu-*.*-vitastor.patch`.
|
See `patches/qemu-*.*-vitastor.patch`.
|
||||||
- Install fio 3.7 or later, get its source and symlink it into `<vitastor>/fio`.
|
- Install fio 3.7 or later, get its source and symlink it into `<vitastor>/fio`.
|
||||||
- Build Vitastor with `make -j8`.
|
- Build & install Vitastor with `mkdir build && cd build && cmake .. && make -j8 && make install`.
|
||||||
- Run `make install` (optionally with `LIBDIR=/usr/lib64 QEMU_PLUGINDIR=/usr/lib64/qemu-kvm`
|
Pay attention to the `QEMU_PLUGINDIR` cmake option - it must be set to `qemu-kvm` on RHEL.
|
||||||
if you're using an RPM-based distro).
|
|
||||||
|
|
||||||
## Running
|
## Running
|
||||||
|
|
||||||
@@ -349,20 +357,31 @@ and calculate disk offsets almost by hand. This will be fixed in near future.
|
|||||||
with lazy fsync, but prepare for inferior single-thread latency.
|
with lazy fsync, but prepare for inferior single-thread latency.
|
||||||
- Get a fast network (at least 10 Gbit/s).
|
- Get a fast network (at least 10 Gbit/s).
|
||||||
- Disable CPU powersaving: `cpupower idle-set -D 0 && cpupower frequency-set -g performance`.
|
- Disable CPU powersaving: `cpupower idle-set -D 0 && cpupower frequency-set -g performance`.
|
||||||
- Start etcd with `--max-txn-ops=100000 --auto-compaction-retention=10 --auto-compaction-mode=revision` options.
|
- On the monitor hosts:
|
||||||
- Create global configuration in etcd: `etcdctl --endpoints=... put /vitastor/config/global '{"immediate_commit":"all"}'`
|
- Edit variables at the top of `/usr/lib/vitastor/mon/make-units.sh` to desired values.
|
||||||
(if all your drives have capacitors).
|
- Create systemd units for the monitor and etcd: `/usr/lib/vitastor/mon/make-units.sh`
|
||||||
- Create pool configuration in etcd: `etcdctl --endpoints=... put /vitastor/config/pools '{"1":{"name":"testpool","scheme":"replicated","pg_size":2,"pg_minsize":1,"pg_count":256,"failure_domain":"host"}}'`.
|
- Start etcd and monitors: `systemctl start etcd vitastor-mon`
|
||||||
- Calculate offsets for your drives with `node /usr/lib/vitastor/mon/simple-offsets.js --device /dev/sdX`.
|
- Put etcd_address and osd_network into `/etc/vitastor/vitastor.conf`. Example:
|
||||||
- Make systemd units for your OSDs. Look at `/usr/lib/vitastor/mon/make-units.sh` for example.
|
```
|
||||||
Notable configuration variables from the example:
|
{
|
||||||
|
"etcd_address": ["10.200.1.10:2379","10.200.1.11:2379","10.200.1.12:2379"],
|
||||||
|
"osd_network": "10.200.1.0/24"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
- Initialize OSDs:
|
||||||
|
- Simplest, SSD-only: `/usr/lib/vitastor/mon/make-osd.sh /dev/disk/by-partuuid/XXX [/dev/disk/by-partuuid/YYY ...]`
|
||||||
|
- Hybrid, HDD+SSD: `/usr/lib/vitastor/mon/make-osd-hybrid.js /dev/sda /dev/sdb ...` - pass all your
|
||||||
|
devices (HDD and SSD) to this script - it will partition disks and initialize journals on its own.
|
||||||
|
This script skips HDDs which are already partitioned so if you want to use non-empty disks for
|
||||||
|
Vitastor you should first wipe them with `wipefs -a`. SSDs with GPT partition table are not skipped,
|
||||||
|
but some free unpartitioned space must be available because the script creates new partitions for journals.
|
||||||
|
- You can change OSD configuration in units or in `vitastor.conf`. Notable configuration variables:
|
||||||
- `disable_data_fsync 1` - only safe with server-grade drives with capacitors.
|
- `disable_data_fsync 1` - only safe with server-grade drives with capacitors.
|
||||||
- `immediate_commit all` - use this if all your drives are server-grade.
|
- `immediate_commit all` - use this if all your drives are server-grade.
|
||||||
|
If all OSDs have it set to all then you should also put the same value in etcd into /vitastor/config/global
|
||||||
- `disable_device_lock 1` - only required if you run multiple OSDs on one block device.
|
- `disable_device_lock 1` - only required if you run multiple OSDs on one block device.
|
||||||
- `flusher_count 16` - flusher is a micro-thread that removes old data from the journal.
|
- `flusher_count 256` - flusher is a micro-thread that removes old data from the journal.
|
||||||
More flushers mean more aggressive journal flushing which allows for more throughput
|
You don't have to worry about this parameter anymore, 256 is enough.
|
||||||
but slightly hurts latency under less load. Flushing will probably be improved in the future
|
|
||||||
because currently high queue depths sometimes lead to performance degradation.
|
|
||||||
- `disk_alignment`, `journal_block_size`, `meta_block_size` should be set to the internal
|
- `disk_alignment`, `journal_block_size`, `meta_block_size` should be set to the internal
|
||||||
block size of your SSDs which is 4096 on most drives.
|
block size of your SSDs which is 4096 on most drives.
|
||||||
- `journal_no_same_sector_overwrites true` prevents multiple overwrites of the same journal sector.
|
- `journal_no_same_sector_overwrites true` prevents multiple overwrites of the same journal sector.
|
||||||
@@ -373,39 +392,186 @@ and calculate disk offsets almost by hand. This will be fixed in near future.
|
|||||||
setting is set, it is also required to raise `journal_sector_buffer_count` setting, which is the
|
setting is set, it is also required to raise `journal_sector_buffer_count` setting, which is the
|
||||||
number of dirty journal sectors that may be written to at the same time.
|
number of dirty journal sectors that may be written to at the same time.
|
||||||
- `systemctl start vitastor.target` everywhere.
|
- `systemctl start vitastor.target` everywhere.
|
||||||
- Start any number of monitors: `node /usr/lib/vitastor/mon/mon-main.js --etcd_url 'http://10.115.0.10:2379,http://10.115.0.11:2379,http://10.115.0.12:2379,http://10.115.0.13:2379' --etcd_prefix '/vitastor' --etcd_start_timeout 5`.
|
- Create global configuration in etcd: `etcdctl --endpoints=... put /vitastor/config/global '{"immediate_commit":"all"}'`
|
||||||
|
(if all your drives have capacitors).
|
||||||
|
- Create pool configuration in etcd: `etcdctl --endpoints=... put /vitastor/config/pools '{"1":{"name":"testpool","scheme":"replicated","pg_size":2,"pg_minsize":1,"pg_count":256,"failure_domain":"host"}}'`.
|
||||||
|
For jerasure pools the configuration should look like the following: `2:{"name":"ecpool","scheme":"jerasure","pg_size":4,"parity_chunks":2,"pg_minsize":2,"pg_count":256,"failure_domain":"host"}`.
|
||||||
- At this point, one of the monitors will configure PGs and OSDs will start them.
|
- At this point, one of the monitors will configure PGs and OSDs will start them.
|
||||||
- You can check PG states with `etcdctl --endpoints=... get --prefix /vitastor/pg/state`. All PGs should become 'active'.
|
- You can check PG states with `etcdctl --endpoints=... get --prefix /vitastor/pg/state`. All PGs should become 'active'.
|
||||||
- Run tests with (for example): `fio -thread -ioengine=/usr/lib/x86_64-linux-gnu/vitastor/libfio_cluster.so -name=test -bs=4M -direct=1 -iodepth=16 -rw=write -etcd=10.115.0.10:2379/v3 -pool=1 -inode=1 -size=400G`.
|
|
||||||
- Upload VM disk image with qemu-img (for example):
|
### Name an image
|
||||||
```
|
|
||||||
LD_PRELOAD=/usr/lib/x86_64-linux-gnu/qemu/block-vitastor.so qemu-img convert -f qcow2 debian10.qcow2 -p
|
```
|
||||||
-O raw 'vitastor:etcd_host=10.115.0.10\:2379/v3:pool=1:inode=1:size=2147483648'
|
etcdctl --endpoints=<etcd> put /vitastor/config/inode/<pool>/<inode> '{"name":"<name>","size":<size>[,"parent_id":<parent_inode_number>][,"readonly":true]}'
|
||||||
```
|
```
|
||||||
- Run QEMU with (for example):
|
|
||||||
```
|
For example:
|
||||||
LD_PRELOAD=/usr/lib/x86_64-linux-gnu/qemu/block-vitastor.so qemu-system-x86_64 -enable-kvm -m 1024
|
|
||||||
-drive 'file=vitastor:etcd_host=10.115.0.10\:2379/v3:pool=1:inode=1:size=2147483648',format=raw,if=none,id=drive-virtio-disk0,cache=none
|
```
|
||||||
-device virtio-blk-pci,scsi=off,bus=pci.0,addr=0x5,drive=drive-virtio-disk0,id=virtio-disk0,bootindex=1,write-cache=off,physical_block_size=4096,logical_block_size=512
|
etcdctl --endpoints=http://10.115.0.10:2379/v3 put /vitastor/config/inode/1/1 '{"name":"testimg","size":2147483648}'
|
||||||
-vnc 0.0.0.0:0
|
```
|
||||||
```
|
|
||||||
- Remove inode with (for example):
|
If you specify parent_id the image becomes a CoW clone. I.e. all writes go to the new inode and reads first check it
|
||||||
```
|
and then upper layers. You can then make parent readonly by updating its entry with `"readonly":true` for safety and
|
||||||
vitastor-rm --etcd_address 10.115.0.10:2379/v3 --pool 1 --inode 1 --parallel_osds 16 --iodepth 32
|
basically treat it as a snapshot.
|
||||||
```
|
|
||||||
|
So to create a snapshot you basically rename the previous upper layer (for example from testimg to testimg@0), make it readonly
|
||||||
|
and create a new top layer with the original name (testimg) and the previous one as a parent.
|
||||||
|
|
||||||
|
### Run fio benchmarks
|
||||||
|
|
||||||
|
fio command example:
|
||||||
|
|
||||||
|
```
|
||||||
|
fio -thread -ioengine=libfio_vitastor.so -name=test -bs=4M -direct=1 -iodepth=16 -rw=write -etcd=10.115.0.10:2379/v3 -image=testimg
|
||||||
|
```
|
||||||
|
|
||||||
|
If you don't want to access your image by name, you can specify pool number, inode number and size
|
||||||
|
(`-pool=1 -inode=1 -size=400G`) instead of the image name (`-image=testimg`).
|
||||||
|
|
||||||
|
### Upload VM image
|
||||||
|
|
||||||
|
Use qemu-img and `vitastor:etcd_host=<HOST>:image=<IMAGE>` disk filename. For example:
|
||||||
|
|
||||||
|
```
|
||||||
|
qemu-img convert -f qcow2 debian10.qcow2 -p -O raw 'vitastor:etcd_host=10.115.0.10\:2379/v3:image=testimg'
|
||||||
|
```
|
||||||
|
|
||||||
|
Note that the command requires to be run with `LD_PRELOAD=/usr/lib/x86_64-linux-gnu/qemu/block-vitastor.so qemu-img ...`
|
||||||
|
if you use unmodified QEMU.
|
||||||
|
|
||||||
|
You can also specify `:pool=<POOL>:inode=<INODE>:size=<SIZE>` instead of `:image=<IMAGE>`
|
||||||
|
if you don't want to use inode metadata.
|
||||||
|
|
||||||
|
### Start a VM
|
||||||
|
|
||||||
|
Run QEMU with `-drive file=vitastor:etcd_host=<HOST>:image=<IMAGE>` and use 4 KB physical block size.
|
||||||
|
|
||||||
|
For example:
|
||||||
|
|
||||||
|
```
|
||||||
|
qemu-system-x86_64 -enable-kvm -m 1024
|
||||||
|
-drive 'file=vitastor:etcd_host=10.115.0.10\:2379/v3:image=testimg',format=raw,if=none,id=drive-virtio-disk0,cache=none
|
||||||
|
-device virtio-blk-pci,scsi=off,bus=pci.0,addr=0x5,drive=drive-virtio-disk0,id=virtio-disk0,bootindex=1,write-cache=off,physical_block_size=4096,logical_block_size=512
|
||||||
|
-vnc 0.0.0.0:0
|
||||||
|
```
|
||||||
|
|
||||||
|
You can also specify `:pool=<POOL>:inode=<INODE>:size=<SIZE>` instead of `:image=<IMAGE>`,
|
||||||
|
just like in qemu-img.
|
||||||
|
|
||||||
|
### Remove inode
|
||||||
|
|
||||||
|
Use vitastor-rm / vitastor-cli rm-data. For example:
|
||||||
|
|
||||||
|
```
|
||||||
|
vitastor-cli rm-data --etcd_address 10.115.0.10:2379/v3 --pool 1 --inode 1 --parallel_osds 16 --iodepth 32
|
||||||
|
```
|
||||||
|
|
||||||
|
### NBD
|
||||||
|
|
||||||
|
To create a local block device for a Vitastor image, use NBD. For example:
|
||||||
|
|
||||||
|
```
|
||||||
|
vitastor-nbd map --etcd_address 10.115.0.10:2379/v3 --image testimg
|
||||||
|
```
|
||||||
|
|
||||||
|
It will output the device name, like /dev/nbd0 which you can then format and mount as a normal block device.
|
||||||
|
|
||||||
|
Again, you can use `--pool <POOL> --inode <INODE> --size <SIZE>` insteaf of `--image <IMAGE>` if you want.
|
||||||
|
|
||||||
|
### Kubernetes
|
||||||
|
|
||||||
|
Vitastor has a CSI plugin for Kubernetes which supports RWO volumes.
|
||||||
|
|
||||||
|
To deploy it, take manifests from [csi/deploy/](csi/deploy/) directory, put your
|
||||||
|
Vitastor configuration in [csi/deploy/001-csi-config-map.yaml](001-csi-config-map.yaml),
|
||||||
|
configure storage class in [csi/deploy/009-storage-class.yaml](009-storage-class.yaml)
|
||||||
|
and apply all `NNN-*.yaml` manifests to your Kubernetes installation:
|
||||||
|
|
||||||
|
```
|
||||||
|
for i in ./???-*.yaml; do kubectl apply -f $i; done
|
||||||
|
```
|
||||||
|
|
||||||
|
After that you'll be able to create PersistentVolumes. See example in [csi/deploy/example-pvc.yaml](csi/deploy/example-pvc.yaml).
|
||||||
|
|
||||||
|
### OpenStack
|
||||||
|
|
||||||
|
To enable Vitastor support in an OpenStack installation:
|
||||||
|
|
||||||
|
- Install vitastor-client, patched QEMU and libvirt packages from Vitastor DEB or RPM repository
|
||||||
|
- Use `patches/nova-21.diff` or `patches/nova-23.diff` to patch your Nova installation.
|
||||||
|
Patch 21 fits Nova 21-22, patch 23 fits Nova 23-24.
|
||||||
|
- Install `patches/cinder-vitastor.py` as `..../cinder/volume/drivers/vitastor.py`
|
||||||
|
- Define a volume type in cinder.conf (see below)
|
||||||
|
- Block network access from VMs to Vitastor network (to OSDs and etcd), because Vitastor doesn't support authentication (yet)
|
||||||
|
- Restart Cinder and Nova
|
||||||
|
|
||||||
|
Cinder volume type configuration example:
|
||||||
|
|
||||||
|
```
|
||||||
|
[DEFAULT]
|
||||||
|
enabled_backends = lvmdriver-1, vitastor-testcluster
|
||||||
|
# ...
|
||||||
|
|
||||||
|
[vitastor-testcluster]
|
||||||
|
volume_driver = cinder.volume.drivers.vitastor.VitastorDriver
|
||||||
|
volume_backend_name = vitastor-testcluster
|
||||||
|
image_volume_cache_enabled = True
|
||||||
|
volume_clear = none
|
||||||
|
vitastor_etcd_address = 192.168.7.2:2379
|
||||||
|
vitastor_etcd_prefix =
|
||||||
|
vitastor_config_path = /etc/vitastor/vitastor.conf
|
||||||
|
vitastor_pool_id = 1
|
||||||
|
image_upload_use_cinder_backend = True
|
||||||
|
```
|
||||||
|
|
||||||
|
To put Glance images in Vitastor, use [https://docs.openstack.org/cinder/pike/admin/blockstorage-volume-backed-image.html](volume-backed images),
|
||||||
|
although the support has not been verified yet.
|
||||||
|
|
||||||
|
### Proxmox
|
||||||
|
|
||||||
|
To enable Vitastor support in Proxmox Virtual Environment (6.4 and 7.1 are supported):
|
||||||
|
|
||||||
|
- Add the corresponding Vitastor Debian repository into sources.list on Proxmox hosts
|
||||||
|
(buster for 6.4, bullseye for 7.1)
|
||||||
|
- Install vitastor-client, pve-qemu-kvm, pve-storage-vitastor (* or see note) packages from Vitastor repository
|
||||||
|
- Define storage in `/etc/pve/storage.cfg` (see below)
|
||||||
|
- Block network access from VMs to Vitastor network (to OSDs and etcd), because Vitastor doesn't support authentication (yet)
|
||||||
|
- Restart pvedaemon: `systemctl restart pvedaemon`
|
||||||
|
|
||||||
|
`/etc/pve/storage.cfg` example (the only required option is vitastor_pool, all others
|
||||||
|
are listed below with their default values):
|
||||||
|
|
||||||
|
```
|
||||||
|
vitastor: vitastor
|
||||||
|
# pool to put new images into
|
||||||
|
vitastor_pool testpool
|
||||||
|
# path to the configuration file
|
||||||
|
vitastor_config_path /etc/vitastor/vitastor.conf
|
||||||
|
# etcd address(es), required only if missing in the configuration file
|
||||||
|
vitastor_etcd_address 192.168.7.2:2379/v3
|
||||||
|
# prefix for keys in etcd
|
||||||
|
vitastor_etcd_prefix /vitastor
|
||||||
|
# prefix for images
|
||||||
|
vitastor_prefix pve/
|
||||||
|
# use NBD mounter (only required for containers)
|
||||||
|
vitastor_nbd 0
|
||||||
|
```
|
||||||
|
|
||||||
|
\* Note: you can also manually copy [patches/PVE_VitastorPlugin.pm](patches/PVE_VitastorPlugin.pm) to Proxmox hosts
|
||||||
|
as `/usr/share/perl5/PVE/Storage/Custom/VitastorPlugin.pm` instead of installing pve-storage-vitastor.
|
||||||
|
|
||||||
## Known Problems
|
## Known Problems
|
||||||
|
|
||||||
- Object deletion requests may currently lead to 'incomplete' objects if your OSDs crash during
|
- Object deletion requests may currently lead to 'incomplete' objects in EC pools
|
||||||
deletion because proper handling of object cleanup in a cluster should be "three-phase"
|
if your OSDs crash during deletion because proper handling of object cleanup
|
||||||
and it's currently not implemented. Inode removal tool currently can't handle unclean
|
in a cluster should be "three-phase" and it's currently not implemented.
|
||||||
objects, so incomplete objects become undeletable. This will be fixed in near future
|
Just repeat the removal request again in this case.
|
||||||
by allowing the inode removal tool to delete unclean objects. With this problem fixed
|
|
||||||
you'll be able just to repeat the removal again.
|
|
||||||
|
|
||||||
## Implementation Principles
|
## Implementation Principles
|
||||||
|
|
||||||
- I like simple and stupid solutions, so expect Vitastor to stay simple.
|
- I like architecturally simple solutions. Vitastor is and will always be designed
|
||||||
|
exactly like that.
|
||||||
- I also like reinventing the wheel to some extent, like writing my own HTTP client
|
- I also like reinventing the wheel to some extent, like writing my own HTTP client
|
||||||
for etcd interaction instead of using prebuilt libraries, because in this case
|
for etcd interaction instead of using prebuilt libraries, because in this case
|
||||||
I'm confident about what my code does and what it doesn't do.
|
I'm confident about what my code does and what it doesn't do.
|
||||||
@@ -420,25 +586,30 @@ and calculate disk offsets almost by hand. This will be fixed in near future.
|
|||||||
|
|
||||||
Copyright (c) Vitaliy Filippov (vitalif [at] yourcmc.ru), 2019+
|
Copyright (c) Vitaliy Filippov (vitalif [at] yourcmc.ru), 2019+
|
||||||
|
|
||||||
You can also find me in the Russian Telegram Ceph chat: https://t.me/ceph_ru
|
Join Vitastor Telegram Chat: https://t.me/vitastor
|
||||||
|
|
||||||
All server-side code (OSD, Monitor and so on) is licensed under the terms of
|
All server-side code (OSD, Monitor and so on) is licensed under the terms of
|
||||||
Vitastor Network Public License 1.0 (VNPL 1.0), a copyleft license based on
|
Vitastor Network Public License 1.1 (VNPL 1.1), a copyleft license based on
|
||||||
GNU GPLv3.0 with the additional "Network Interaction" clause which requires
|
GNU GPLv3.0 with the additional "Network Interaction" clause which requires
|
||||||
opensourcing all programs directly or indirectly interacting with Vitastor
|
opensourcing all programs directly or indirectly interacting with Vitastor
|
||||||
through a computer network ("Proxy Programs"). Proxy Programs may be made public
|
through a computer network and expressly designed to be used in conjunction
|
||||||
not only under the terms of the same license, but also under the terms of any
|
with it ("Proxy Programs"). Proxy Programs may be made public not only under
|
||||||
GPL-Compatible Free Software License, as listed by the Free Software Foundation.
|
the terms of the same license, but also under the terms of any GPL-Compatible
|
||||||
|
Free Software License, as listed by the Free Software Foundation.
|
||||||
This is a stricter copyleft license than the Affero GPL.
|
This is a stricter copyleft license than the Affero GPL.
|
||||||
|
|
||||||
|
Please note that VNPL doesn't require you to open the code of proprietary
|
||||||
|
software running inside a VM if it's not specially designed to be used with
|
||||||
|
Vitastor.
|
||||||
|
|
||||||
Basically, you can't use the software in a proprietary environment to provide
|
Basically, you can't use the software in a proprietary environment to provide
|
||||||
its functionality to users without opensourcing all intermediary components
|
its functionality to users without opensourcing all intermediary components
|
||||||
standing between the user and Vitastor or purchasing a commercial license
|
standing between the user and Vitastor or purchasing a commercial license
|
||||||
from the author 😀.
|
from the author 😀.
|
||||||
|
|
||||||
Client libraries (cluster_client and so on) are dual-licensed under the same
|
Client libraries (cluster_client and so on) are dual-licensed under the same
|
||||||
VNPL 1.0 and also GNU GPL 2.0 or later to allow for compatibility with GPLed
|
VNPL 1.1 and also GNU GPL 2.0 or later to allow for compatibility with GPLed
|
||||||
software like QEMU and fio.
|
software like QEMU and fio.
|
||||||
|
|
||||||
You can find the full text of VNPL-1.0 in the file [VNPL-1.0.txt](VNPL-1.0.txt).
|
You can find the full text of VNPL-1.1 in the file [VNPL-1.1.txt](VNPL-1.1.txt).
|
||||||
GPL 2.0 is also included in this repository as [GPL-2.0.txt](GPL-2.0.txt).
|
GPL 2.0 is also included in this repository as [GPL-2.0.txt](GPL-2.0.txt).
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
VITASTOR NETWORK PUBLIC LICENSE
|
VITASTOR NETWORK PUBLIC LICENSE
|
||||||
Version 1, 17 September 2020
|
Version 1.1, 6 February 2021
|
||||||
|
|
||||||
Copyright (C) 2020 Vitaliy Filippov <vitalif@yourcmc.ru>
|
Copyright (C) 2021 Vitaliy Filippov <vitalif@yourcmc.ru>
|
||||||
Everyone is permitted to copy and distribute verbatim copies
|
Everyone is permitted to copy and distribute verbatim copies
|
||||||
of this license document, but changing it is not allowed.
|
of this license document, but changing it is not allowed.
|
||||||
|
|
||||||
@@ -540,12 +540,15 @@ License would be to refrain entirely from conveying the Program.
|
|||||||
|
|
||||||
13. Remote Network Interaction.
|
13. Remote Network Interaction.
|
||||||
|
|
||||||
Notwithstanding any other provision of this License, if you provide
|
A "Proxy Program" means a separate program which is specially designed to
|
||||||
any user an opportunity to interact with the covered work directly
|
be used in conjunction with the covered work and interacts with it directly
|
||||||
or indirectly through a computer network, an imitation of such network,
|
or indirectly through any kind of API (application programming interfaces),
|
||||||
or an additional program (hereinafter referred to as a "Proxy Program")
|
a computer network, an imitation of such network, or another Proxy Program
|
||||||
that, in turn, interacts with the covered work through a computer network,
|
itself.
|
||||||
an imitation of such network, or another Proxy Program itself,
|
|
||||||
|
Notwithstanding any other provision of this License, if you provide any user
|
||||||
|
with an opportunity to interact with the covered work through a computer
|
||||||
|
network, an imitation of such network, or any number of "Proxy Programs",
|
||||||
you must prominently offer that user an opportunity to receive the
|
you must prominently offer that user an opportunity to receive the
|
||||||
Corresponding Source of the covered work and all Proxy Programs from a
|
Corresponding Source of the covered work and all Proxy Programs from a
|
||||||
network server at no charge, through some standard or customary means of
|
network server at no charge, through some standard or customary means of
|
||||||
@@ -1,310 +0,0 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
|
||||||
// License: VNPL-1.0 (see README.md for details)
|
|
||||||
|
|
||||||
#include "blockstore_impl.h"
|
|
||||||
|
|
||||||
#define SYNC_HAS_SMALL 1
|
|
||||||
#define SYNC_HAS_BIG 2
|
|
||||||
#define SYNC_DATA_SYNC_SENT 3
|
|
||||||
#define SYNC_DATA_SYNC_DONE 4
|
|
||||||
#define SYNC_JOURNAL_WRITE_SENT 5
|
|
||||||
#define SYNC_JOURNAL_WRITE_DONE 6
|
|
||||||
#define SYNC_JOURNAL_SYNC_SENT 7
|
|
||||||
#define SYNC_DONE 8
|
|
||||||
|
|
||||||
int blockstore_impl_t::dequeue_sync(blockstore_op_t *op)
|
|
||||||
{
|
|
||||||
if (PRIV(op)->op_state == 0)
|
|
||||||
{
|
|
||||||
stop_sync_submitted = false;
|
|
||||||
PRIV(op)->sync_big_writes.swap(unsynced_big_writes);
|
|
||||||
PRIV(op)->sync_small_writes.swap(unsynced_small_writes);
|
|
||||||
PRIV(op)->sync_small_checked = 0;
|
|
||||||
PRIV(op)->sync_big_checked = 0;
|
|
||||||
unsynced_big_writes.clear();
|
|
||||||
unsynced_small_writes.clear();
|
|
||||||
if (PRIV(op)->sync_big_writes.size() > 0)
|
|
||||||
PRIV(op)->op_state = SYNC_HAS_BIG;
|
|
||||||
else if (PRIV(op)->sync_small_writes.size() > 0)
|
|
||||||
PRIV(op)->op_state = SYNC_HAS_SMALL;
|
|
||||||
else
|
|
||||||
PRIV(op)->op_state = SYNC_DONE;
|
|
||||||
// Always add sync to in_progress_syncs because we clear unsynced_big_writes and unsynced_small_writes
|
|
||||||
PRIV(op)->prev_sync_count = in_progress_syncs.size();
|
|
||||||
PRIV(op)->in_progress_ptr = in_progress_syncs.insert(in_progress_syncs.end(), op);
|
|
||||||
}
|
|
||||||
continue_sync(op);
|
|
||||||
// Always dequeue because we always add syncs to in_progress_syncs
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
int blockstore_impl_t::continue_sync(blockstore_op_t *op)
|
|
||||||
{
|
|
||||||
auto cb = [this, op](ring_data_t *data) { handle_sync_event(data, op); };
|
|
||||||
if (PRIV(op)->op_state == SYNC_HAS_SMALL)
|
|
||||||
{
|
|
||||||
// No big writes, just fsync the journal
|
|
||||||
for (; PRIV(op)->sync_small_checked < PRIV(op)->sync_small_writes.size(); PRIV(op)->sync_small_checked++)
|
|
||||||
{
|
|
||||||
if (IS_IN_FLIGHT(dirty_db[PRIV(op)->sync_small_writes[PRIV(op)->sync_small_checked]].state))
|
|
||||||
{
|
|
||||||
// Wait for small inflight writes to complete
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (journal.sector_info[journal.cur_sector].dirty)
|
|
||||||
{
|
|
||||||
// Write out the last journal sector if it happens to be dirty
|
|
||||||
BS_SUBMIT_GET_ONLY_SQE(sqe);
|
|
||||||
prepare_journal_sector_write(journal, journal.cur_sector, sqe, cb);
|
|
||||||
PRIV(op)->min_flushed_journal_sector = PRIV(op)->max_flushed_journal_sector = 1 + journal.cur_sector;
|
|
||||||
PRIV(op)->pending_ops = 1;
|
|
||||||
PRIV(op)->op_state = SYNC_JOURNAL_WRITE_SENT;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
PRIV(op)->op_state = SYNC_JOURNAL_WRITE_DONE;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (PRIV(op)->op_state == SYNC_HAS_BIG)
|
|
||||||
{
|
|
||||||
for (; PRIV(op)->sync_big_checked < PRIV(op)->sync_big_writes.size(); PRIV(op)->sync_big_checked++)
|
|
||||||
{
|
|
||||||
if (IS_IN_FLIGHT(dirty_db[PRIV(op)->sync_big_writes[PRIV(op)->sync_big_checked]].state))
|
|
||||||
{
|
|
||||||
// Wait for big inflight writes to complete
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// 1st step: fsync data
|
|
||||||
if (!disable_data_fsync)
|
|
||||||
{
|
|
||||||
BS_SUBMIT_GET_SQE(sqe, data);
|
|
||||||
my_uring_prep_fsync(sqe, data_fd, IORING_FSYNC_DATASYNC);
|
|
||||||
data->iov = { 0 };
|
|
||||||
data->callback = cb;
|
|
||||||
PRIV(op)->min_flushed_journal_sector = PRIV(op)->max_flushed_journal_sector = 0;
|
|
||||||
PRIV(op)->pending_ops = 1;
|
|
||||||
PRIV(op)->op_state = SYNC_DATA_SYNC_SENT;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
PRIV(op)->op_state = SYNC_DATA_SYNC_DONE;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (PRIV(op)->op_state == SYNC_DATA_SYNC_DONE)
|
|
||||||
{
|
|
||||||
for (; PRIV(op)->sync_small_checked < PRIV(op)->sync_small_writes.size(); PRIV(op)->sync_small_checked++)
|
|
||||||
{
|
|
||||||
if (IS_IN_FLIGHT(dirty_db[PRIV(op)->sync_small_writes[PRIV(op)->sync_small_checked]].state))
|
|
||||||
{
|
|
||||||
// Wait for small inflight writes to complete
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// 2nd step: Data device is synced, prepare & write journal entries
|
|
||||||
// Check space in the journal and journal memory buffers
|
|
||||||
blockstore_journal_check_t space_check(this);
|
|
||||||
if (!space_check.check_available(op, PRIV(op)->sync_big_writes.size(), sizeof(journal_entry_big_write), JOURNAL_STABILIZE_RESERVATION))
|
|
||||||
{
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
// Get SQEs. Don't bother about merging, submit each journal sector as a separate request
|
|
||||||
struct io_uring_sqe *sqe[space_check.sectors_required];
|
|
||||||
for (int i = 0; i < space_check.sectors_required; i++)
|
|
||||||
{
|
|
||||||
BS_SUBMIT_GET_SQE_DECL(sqe[i]);
|
|
||||||
}
|
|
||||||
// Prepare and submit journal entries
|
|
||||||
auto it = PRIV(op)->sync_big_writes.begin();
|
|
||||||
int s = 0, cur_sector = -1;
|
|
||||||
if ((journal_block_size - journal.in_sector_pos) < sizeof(journal_entry_big_write) &&
|
|
||||||
journal.sector_info[journal.cur_sector].dirty)
|
|
||||||
{
|
|
||||||
if (cur_sector == -1)
|
|
||||||
PRIV(op)->min_flushed_journal_sector = 1 + journal.cur_sector;
|
|
||||||
cur_sector = journal.cur_sector;
|
|
||||||
prepare_journal_sector_write(journal, cur_sector, sqe[s++], cb);
|
|
||||||
}
|
|
||||||
while (it != PRIV(op)->sync_big_writes.end())
|
|
||||||
{
|
|
||||||
journal_entry_big_write *je = (journal_entry_big_write*)prefill_single_journal_entry(
|
|
||||||
journal, (dirty_db[*it].state & BS_ST_INSTANT) ? JE_BIG_WRITE_INSTANT : JE_BIG_WRITE,
|
|
||||||
sizeof(journal_entry_big_write)
|
|
||||||
);
|
|
||||||
dirty_db[*it].journal_sector = journal.sector_info[journal.cur_sector].offset;
|
|
||||||
journal.sector_info[journal.cur_sector].dirty = false;
|
|
||||||
journal.used_sectors[journal.sector_info[journal.cur_sector].offset]++;
|
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
|
||||||
printf(
|
|
||||||
"journal offset %08lx is used by %lx:%lx v%lu (%lu refs)\n",
|
|
||||||
dirty_db[*it].journal_sector, it->oid.inode, it->oid.stripe, it->version,
|
|
||||||
journal.used_sectors[journal.sector_info[journal.cur_sector].offset]
|
|
||||||
);
|
|
||||||
#endif
|
|
||||||
je->oid = it->oid;
|
|
||||||
je->version = it->version;
|
|
||||||
je->offset = dirty_db[*it].offset;
|
|
||||||
je->len = dirty_db[*it].len;
|
|
||||||
je->location = dirty_db[*it].location;
|
|
||||||
je->crc32 = je_crc32((journal_entry*)je);
|
|
||||||
journal.crc32_last = je->crc32;
|
|
||||||
it++;
|
|
||||||
if (cur_sector != journal.cur_sector)
|
|
||||||
{
|
|
||||||
// Write previous sector. We should write the sector only after filling it,
|
|
||||||
// because otherwise we'll write a lot more sectors in the "no_same_sector_overwrite" mode
|
|
||||||
if (cur_sector != -1)
|
|
||||||
prepare_journal_sector_write(journal, cur_sector, sqe[s++], cb);
|
|
||||||
else
|
|
||||||
PRIV(op)->min_flushed_journal_sector = 1 + journal.cur_sector;
|
|
||||||
cur_sector = journal.cur_sector;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (cur_sector != -1)
|
|
||||||
prepare_journal_sector_write(journal, cur_sector, sqe[s++], cb);
|
|
||||||
PRIV(op)->max_flushed_journal_sector = 1 + journal.cur_sector;
|
|
||||||
PRIV(op)->pending_ops = s;
|
|
||||||
PRIV(op)->op_state = SYNC_JOURNAL_WRITE_SENT;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
if (PRIV(op)->op_state == SYNC_JOURNAL_WRITE_DONE)
|
|
||||||
{
|
|
||||||
if (!disable_journal_fsync)
|
|
||||||
{
|
|
||||||
BS_SUBMIT_GET_SQE(sqe, data);
|
|
||||||
my_uring_prep_fsync(sqe, journal.fd, IORING_FSYNC_DATASYNC);
|
|
||||||
data->iov = { 0 };
|
|
||||||
data->callback = cb;
|
|
||||||
PRIV(op)->pending_ops = 1;
|
|
||||||
PRIV(op)->op_state = SYNC_JOURNAL_SYNC_SENT;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
PRIV(op)->op_state = SYNC_DONE;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (PRIV(op)->op_state == SYNC_DONE)
|
|
||||||
{
|
|
||||||
return ack_sync(op);
|
|
||||||
}
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_impl_t::handle_sync_event(ring_data_t *data, blockstore_op_t *op)
|
|
||||||
{
|
|
||||||
live = true;
|
|
||||||
if (data->res != data->iov.iov_len)
|
|
||||||
{
|
|
||||||
throw std::runtime_error(
|
|
||||||
"write operation failed ("+std::to_string(data->res)+" != "+std::to_string(data->iov.iov_len)+
|
|
||||||
"). in-memory state is corrupted. AAAAAAAaaaaaaaaa!!!111"
|
|
||||||
);
|
|
||||||
}
|
|
||||||
PRIV(op)->pending_ops--;
|
|
||||||
if (PRIV(op)->pending_ops == 0)
|
|
||||||
{
|
|
||||||
// Release used journal sectors
|
|
||||||
release_journal_sectors(op);
|
|
||||||
// Handle states
|
|
||||||
if (PRIV(op)->op_state == SYNC_DATA_SYNC_SENT)
|
|
||||||
{
|
|
||||||
PRIV(op)->op_state = SYNC_DATA_SYNC_DONE;
|
|
||||||
}
|
|
||||||
else if (PRIV(op)->op_state == SYNC_JOURNAL_WRITE_SENT)
|
|
||||||
{
|
|
||||||
PRIV(op)->op_state = SYNC_JOURNAL_WRITE_DONE;
|
|
||||||
}
|
|
||||||
else if (PRIV(op)->op_state == SYNC_JOURNAL_SYNC_SENT)
|
|
||||||
{
|
|
||||||
PRIV(op)->op_state = SYNC_DONE;
|
|
||||||
ack_sync(op);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
throw std::runtime_error("BUG: unexpected sync op state");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
int blockstore_impl_t::ack_sync(blockstore_op_t *op)
|
|
||||||
{
|
|
||||||
if (PRIV(op)->op_state == SYNC_DONE && PRIV(op)->prev_sync_count == 0)
|
|
||||||
{
|
|
||||||
// Remove dependency of subsequent syncs
|
|
||||||
auto it = PRIV(op)->in_progress_ptr;
|
|
||||||
int done_syncs = 1;
|
|
||||||
++it;
|
|
||||||
// Acknowledge sync
|
|
||||||
ack_one_sync(op);
|
|
||||||
while (it != in_progress_syncs.end())
|
|
||||||
{
|
|
||||||
auto & next_sync = *it++;
|
|
||||||
PRIV(next_sync)->prev_sync_count -= done_syncs;
|
|
||||||
if (PRIV(next_sync)->prev_sync_count == 0 && PRIV(next_sync)->op_state == SYNC_DONE)
|
|
||||||
{
|
|
||||||
done_syncs++;
|
|
||||||
// Acknowledge next_sync
|
|
||||||
ack_one_sync(next_sync);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return 2;
|
|
||||||
}
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
void blockstore_impl_t::ack_one_sync(blockstore_op_t *op)
|
|
||||||
{
|
|
||||||
// Handle states
|
|
||||||
for (auto it = PRIV(op)->sync_big_writes.begin(); it != PRIV(op)->sync_big_writes.end(); it++)
|
|
||||||
{
|
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
|
||||||
printf("Ack sync big %lx:%lx v%lu\n", it->oid.inode, it->oid.stripe, it->version);
|
|
||||||
#endif
|
|
||||||
auto & unstab = unstable_writes[it->oid];
|
|
||||||
unstab = unstab < it->version ? it->version : unstab;
|
|
||||||
auto dirty_it = dirty_db.find(*it);
|
|
||||||
dirty_it->second.state = ((dirty_it->second.state & ~BS_ST_WORKFLOW_MASK) | BS_ST_SYNCED);
|
|
||||||
if (dirty_it->second.state & BS_ST_INSTANT)
|
|
||||||
{
|
|
||||||
mark_stable(dirty_it->first);
|
|
||||||
}
|
|
||||||
dirty_it++;
|
|
||||||
while (dirty_it != dirty_db.end() && dirty_it->first.oid == it->oid)
|
|
||||||
{
|
|
||||||
if ((dirty_it->second.state & BS_ST_WORKFLOW_MASK) == BS_ST_WAIT_BIG)
|
|
||||||
{
|
|
||||||
dirty_it->second.state = (dirty_it->second.state & ~BS_ST_WORKFLOW_MASK) | BS_ST_IN_FLIGHT;
|
|
||||||
}
|
|
||||||
dirty_it++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (auto it = PRIV(op)->sync_small_writes.begin(); it != PRIV(op)->sync_small_writes.end(); it++)
|
|
||||||
{
|
|
||||||
#ifdef BLOCKSTORE_DEBUG
|
|
||||||
printf("Ack sync small %lx:%lx v%lu\n", it->oid.inode, it->oid.stripe, it->version);
|
|
||||||
#endif
|
|
||||||
auto & unstab = unstable_writes[it->oid];
|
|
||||||
unstab = unstab < it->version ? it->version : unstab;
|
|
||||||
if (dirty_db[*it].state == (BS_ST_DELETE | BS_ST_WRITTEN))
|
|
||||||
{
|
|
||||||
dirty_db[*it].state = (BS_ST_DELETE | BS_ST_SYNCED);
|
|
||||||
// Deletions are treated as immediately stable
|
|
||||||
mark_stable(*it);
|
|
||||||
}
|
|
||||||
else /* (BS_ST_INSTANT?) | BS_ST_SMALL_WRITE | BS_ST_WRITTEN */
|
|
||||||
{
|
|
||||||
dirty_db[*it].state = (dirty_db[*it].state & ~BS_ST_WORKFLOW_MASK) | BS_ST_SYNCED;
|
|
||||||
if (dirty_db[*it].state & BS_ST_INSTANT)
|
|
||||||
{
|
|
||||||
mark_stable(*it);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
in_progress_syncs.erase(PRIV(op)->in_progress_ptr);
|
|
||||||
op->retval = 0;
|
|
||||||
FINISH_OP(op);
|
|
||||||
}
|
|
||||||
@@ -1,765 +0,0 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
|
||||||
// License: VNPL-1.0 or GNU GPL-2.0+ (see README.md for details)
|
|
||||||
|
|
||||||
#include <stdexcept>
|
|
||||||
#include "cluster_client.h"
|
|
||||||
|
|
||||||
cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json & config)
|
|
||||||
{
|
|
||||||
this->ringloop = ringloop;
|
|
||||||
this->tfd = tfd;
|
|
||||||
|
|
||||||
log_level = config["log_level"].int64_value();
|
|
||||||
|
|
||||||
msgr.osd_num = 0;
|
|
||||||
msgr.tfd = tfd;
|
|
||||||
msgr.ringloop = ringloop;
|
|
||||||
msgr.log_level = log_level;
|
|
||||||
msgr.repeer_pgs = [this](osd_num_t peer_osd)
|
|
||||||
{
|
|
||||||
if (msgr.osd_peer_fds.find(peer_osd) != msgr.osd_peer_fds.end())
|
|
||||||
{
|
|
||||||
// peer_osd just connected
|
|
||||||
continue_ops();
|
|
||||||
}
|
|
||||||
else if (unsynced_writes.size())
|
|
||||||
{
|
|
||||||
// peer_osd just dropped connection
|
|
||||||
for (auto op: syncing_writes)
|
|
||||||
{
|
|
||||||
for (auto & part: op->parts)
|
|
||||||
{
|
|
||||||
if (part.osd_num == peer_osd && part.done)
|
|
||||||
{
|
|
||||||
// repeat this operation
|
|
||||||
part.osd_num = 0;
|
|
||||||
part.done = false;
|
|
||||||
assert(!part.sent);
|
|
||||||
op->done_count--;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (auto op: unsynced_writes)
|
|
||||||
{
|
|
||||||
for (auto & part: op->parts)
|
|
||||||
{
|
|
||||||
if (part.osd_num == peer_osd && part.done)
|
|
||||||
{
|
|
||||||
// repeat this operation
|
|
||||||
part.osd_num = 0;
|
|
||||||
part.done = false;
|
|
||||||
assert(!part.sent);
|
|
||||||
op->done_count--;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (op->done_count < op->parts.size())
|
|
||||||
{
|
|
||||||
cur_ops.insert(op);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
continue_ops();
|
|
||||||
}
|
|
||||||
};
|
|
||||||
msgr.exec_op = [this](osd_op_t *op)
|
|
||||||
{
|
|
||||||
// Garbage in
|
|
||||||
printf("Incoming garbage from peer %d\n", op->peer_fd);
|
|
||||||
msgr.stop_client(op->peer_fd);
|
|
||||||
delete op;
|
|
||||||
};
|
|
||||||
msgr.use_sync_send_recv = config["use_sync_send_recv"].bool_value() ||
|
|
||||||
config["use_sync_send_recv"].uint64_value();
|
|
||||||
|
|
||||||
st_cli.tfd = tfd;
|
|
||||||
st_cli.on_load_config_hook = [this](json11::Json::object & cfg) { on_load_config_hook(cfg); };
|
|
||||||
st_cli.on_change_osd_state_hook = [this](uint64_t peer_osd) { on_change_osd_state_hook(peer_osd); };
|
|
||||||
st_cli.on_change_hook = [this](json11::Json::object & changes) { on_change_hook(changes); };
|
|
||||||
st_cli.on_load_pgs_hook = [this](bool success) { on_load_pgs_hook(success); };
|
|
||||||
|
|
||||||
st_cli.parse_config(config);
|
|
||||||
st_cli.load_global_config();
|
|
||||||
|
|
||||||
if (ringloop)
|
|
||||||
{
|
|
||||||
consumer.loop = [this]()
|
|
||||||
{
|
|
||||||
msgr.read_requests();
|
|
||||||
msgr.send_replies();
|
|
||||||
this->ringloop->submit();
|
|
||||||
};
|
|
||||||
ringloop->register_consumer(&consumer);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
cluster_client_t::~cluster_client_t()
|
|
||||||
{
|
|
||||||
if (ringloop)
|
|
||||||
{
|
|
||||||
ringloop->unregister_consumer(&consumer);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::stop()
|
|
||||||
{
|
|
||||||
while (msgr.clients.size() > 0)
|
|
||||||
{
|
|
||||||
msgr.stop_client(msgr.clients.begin()->first);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::continue_ops(bool up_retry)
|
|
||||||
{
|
|
||||||
for (auto op_it = cur_ops.begin(); op_it != cur_ops.end(); )
|
|
||||||
{
|
|
||||||
if ((*op_it)->up_wait)
|
|
||||||
{
|
|
||||||
if (up_retry)
|
|
||||||
{
|
|
||||||
(*op_it)->up_wait = false;
|
|
||||||
continue_rw(*op_it++);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
op_it++;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
continue_rw(*op_it++);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
static uint32_t is_power_of_two(uint64_t value)
|
|
||||||
{
|
|
||||||
uint32_t l = 0;
|
|
||||||
while (value > 1)
|
|
||||||
{
|
|
||||||
if (value & 1)
|
|
||||||
{
|
|
||||||
return 64;
|
|
||||||
}
|
|
||||||
value = value >> 1;
|
|
||||||
l++;
|
|
||||||
}
|
|
||||||
return l;
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::on_load_config_hook(json11::Json::object & config)
|
|
||||||
{
|
|
||||||
bs_block_size = config["block_size"].uint64_value();
|
|
||||||
bs_disk_alignment = config["disk_alignment"].uint64_value();
|
|
||||||
bs_bitmap_granularity = config["bitmap_granularity"].uint64_value();
|
|
||||||
if (!bs_block_size)
|
|
||||||
{
|
|
||||||
bs_block_size = DEFAULT_BLOCK_SIZE;
|
|
||||||
}
|
|
||||||
if (!bs_disk_alignment)
|
|
||||||
{
|
|
||||||
bs_disk_alignment = DEFAULT_DISK_ALIGNMENT;
|
|
||||||
}
|
|
||||||
if (!bs_bitmap_granularity)
|
|
||||||
{
|
|
||||||
bs_bitmap_granularity = DEFAULT_BITMAP_GRANULARITY;
|
|
||||||
}
|
|
||||||
uint32_t block_order;
|
|
||||||
if ((block_order = is_power_of_two(bs_block_size)) >= 64 || bs_block_size < MIN_BLOCK_SIZE || bs_block_size >= MAX_BLOCK_SIZE)
|
|
||||||
{
|
|
||||||
throw std::runtime_error("Bad block size");
|
|
||||||
}
|
|
||||||
if (config["immediate_commit"] == "all")
|
|
||||||
{
|
|
||||||
// Cluster-wide immediate_commit mode
|
|
||||||
immediate_commit = true;
|
|
||||||
}
|
|
||||||
else if (config.find("client_dirty_limit") != config.end())
|
|
||||||
{
|
|
||||||
client_dirty_limit = config["client_dirty_limit"].uint64_value();
|
|
||||||
}
|
|
||||||
if (!client_dirty_limit)
|
|
||||||
{
|
|
||||||
client_dirty_limit = DEFAULT_CLIENT_DIRTY_LIMIT;
|
|
||||||
}
|
|
||||||
up_wait_retry_interval = config["up_wait_retry_interval"].uint64_value();
|
|
||||||
if (!up_wait_retry_interval)
|
|
||||||
{
|
|
||||||
up_wait_retry_interval = 500;
|
|
||||||
}
|
|
||||||
else if (up_wait_retry_interval < 50)
|
|
||||||
{
|
|
||||||
up_wait_retry_interval = 50;
|
|
||||||
}
|
|
||||||
msgr.peer_connect_interval = config["peer_connect_interval"].uint64_value();
|
|
||||||
if (!msgr.peer_connect_interval)
|
|
||||||
{
|
|
||||||
msgr.peer_connect_interval = DEFAULT_PEER_CONNECT_INTERVAL;
|
|
||||||
}
|
|
||||||
msgr.peer_connect_timeout = config["peer_connect_timeout"].uint64_value();
|
|
||||||
if (!msgr.peer_connect_timeout)
|
|
||||||
{
|
|
||||||
msgr.peer_connect_timeout = DEFAULT_PEER_CONNECT_TIMEOUT;
|
|
||||||
}
|
|
||||||
st_cli.load_pgs();
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::on_load_pgs_hook(bool success)
|
|
||||||
{
|
|
||||||
for (auto pool_item: st_cli.pool_config)
|
|
||||||
{
|
|
||||||
pg_counts[pool_item.first] = pool_item.second.real_pg_count;
|
|
||||||
}
|
|
||||||
pgs_loaded = true;
|
|
||||||
for (auto fn: on_ready_hooks)
|
|
||||||
{
|
|
||||||
fn();
|
|
||||||
}
|
|
||||||
on_ready_hooks.clear();
|
|
||||||
for (auto op: offline_ops)
|
|
||||||
{
|
|
||||||
execute(op);
|
|
||||||
}
|
|
||||||
offline_ops.clear();
|
|
||||||
continue_ops();
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::on_change_hook(json11::Json::object & changes)
|
|
||||||
{
|
|
||||||
for (auto pool_item: st_cli.pool_config)
|
|
||||||
{
|
|
||||||
if (pg_counts[pool_item.first] != pool_item.second.real_pg_count)
|
|
||||||
{
|
|
||||||
// At this point, all pool operations should have been suspended
|
|
||||||
// And now they have to be resliced!
|
|
||||||
for (auto op: cur_ops)
|
|
||||||
{
|
|
||||||
if (INODE_POOL(op->inode) == pool_item.first)
|
|
||||||
{
|
|
||||||
op->needs_reslice = true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (auto op: unsynced_writes)
|
|
||||||
{
|
|
||||||
if (INODE_POOL(op->inode) == pool_item.first)
|
|
||||||
{
|
|
||||||
op->needs_reslice = true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (auto op: syncing_writes)
|
|
||||||
{
|
|
||||||
if (INODE_POOL(op->inode) == pool_item.first)
|
|
||||||
{
|
|
||||||
op->needs_reslice = true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
pg_counts[pool_item.first] = pool_item.second.real_pg_count;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
continue_ops();
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::on_change_osd_state_hook(uint64_t peer_osd)
|
|
||||||
{
|
|
||||||
if (msgr.wanted_peers.find(peer_osd) != msgr.wanted_peers.end())
|
|
||||||
{
|
|
||||||
msgr.connect_peer(peer_osd, st_cli.peer_states[peer_osd]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::on_ready(std::function<void(void)> fn)
|
|
||||||
{
|
|
||||||
if (pgs_loaded)
|
|
||||||
{
|
|
||||||
fn();
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
on_ready_hooks.push_back(fn);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* How writes are synced when immediate_commit is false
|
|
||||||
*
|
|
||||||
* 1) accept up to <client_dirty_limit> write operations for execution,
|
|
||||||
* queue all subsequent writes into <next_writes>
|
|
||||||
* 2) accept exactly one SYNC, queue all subsequent SYNCs into <next_writes>, too
|
|
||||||
* 3) "continue" all accepted writes
|
|
||||||
*
|
|
||||||
* "Continue" WRITE:
|
|
||||||
* 1) if the operation is not a copy yet - copy it (required for replay)
|
|
||||||
* 2) if the operation is not sliced yet - slice it
|
|
||||||
* 3) if the operation doesn't require reslice - try to connect & send all remaining parts
|
|
||||||
* 4) if any of them fail due to disconnected peers or PGs not up, repeat after reconnecting or small timeout
|
|
||||||
* 5) if any of them fail due to other errors, fail the operation and forget it from the current "unsynced batch"
|
|
||||||
* 6) if PG count changes before all parts are done, wait for all in-progress parts to finish,
|
|
||||||
* throw all results away, reslice and resubmit op
|
|
||||||
* 7) when all parts are done, try to "continue" the current SYNC
|
|
||||||
* 8) if the operation succeeds, but then some OSDs drop their connections, repeat
|
|
||||||
* parts from the current "unsynced batch" previously sent to those OSDs in any order
|
|
||||||
*
|
|
||||||
* "Continue" current SYNC:
|
|
||||||
* 1) take all unsynced operations from the current batch
|
|
||||||
* 2) check if all affected OSDs are still alive
|
|
||||||
* 3) if yes, send all SYNCs. otherwise, leave current SYNC as is.
|
|
||||||
* 4) if any of them fail due to disconnected peers, repeat SYNC after repeating all writes
|
|
||||||
* 5) if any of them fail due to other errors, fail the SYNC operation
|
|
||||||
*/
|
|
||||||
|
|
||||||
void cluster_client_t::execute(cluster_op_t *op)
|
|
||||||
{
|
|
||||||
if (!pgs_loaded)
|
|
||||||
{
|
|
||||||
// We're offline
|
|
||||||
offline_ops.push_back(op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
op->retval = 0;
|
|
||||||
if (op->opcode != OSD_OP_SYNC && op->opcode != OSD_OP_READ && op->opcode != OSD_OP_WRITE ||
|
|
||||||
(op->opcode == OSD_OP_READ || op->opcode == OSD_OP_WRITE) && (!op->inode || !op->len ||
|
|
||||||
op->offset % bs_disk_alignment || op->len % bs_disk_alignment))
|
|
||||||
{
|
|
||||||
op->retval = -EINVAL;
|
|
||||||
std::function<void(cluster_op_t*)>(op->callback)(op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (op->opcode == OSD_OP_SYNC)
|
|
||||||
{
|
|
||||||
execute_sync(op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (op->opcode == OSD_OP_WRITE && !immediate_commit)
|
|
||||||
{
|
|
||||||
if (next_writes.size() > 0)
|
|
||||||
{
|
|
||||||
assert(cur_sync);
|
|
||||||
next_writes.push_back(op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (queued_bytes >= client_dirty_limit)
|
|
||||||
{
|
|
||||||
// Push an extra SYNC operation to flush previous writes
|
|
||||||
next_writes.push_back(op);
|
|
||||||
cluster_op_t *sync_op = new cluster_op_t;
|
|
||||||
sync_op->is_internal = true;
|
|
||||||
sync_op->opcode = OSD_OP_SYNC;
|
|
||||||
sync_op->callback = [](cluster_op_t* sync_op) {};
|
|
||||||
execute_sync(sync_op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
queued_bytes += op->len;
|
|
||||||
}
|
|
||||||
cur_ops.insert(op);
|
|
||||||
continue_rw(op);
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::continue_rw(cluster_op_t *op)
|
|
||||||
{
|
|
||||||
pool_id_t pool_id = INODE_POOL(op->inode);
|
|
||||||
if (!pool_id)
|
|
||||||
{
|
|
||||||
op->retval = -EINVAL;
|
|
||||||
std::function<void(cluster_op_t*)>(op->callback)(op);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (st_cli.pool_config.find(pool_id) == st_cli.pool_config.end() ||
|
|
||||||
st_cli.pool_config[pool_id].real_pg_count == 0)
|
|
||||||
{
|
|
||||||
// Postpone operations to unknown pools
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (op->opcode == OSD_OP_WRITE && !immediate_commit && !op->is_internal)
|
|
||||||
{
|
|
||||||
// Save operation for replay when PG goes out of sync
|
|
||||||
// (primary OSD drops our connection in this case)
|
|
||||||
cluster_op_t *op_copy = new cluster_op_t();
|
|
||||||
op_copy->is_internal = true;
|
|
||||||
op_copy->orig_op = op;
|
|
||||||
op_copy->opcode = op->opcode;
|
|
||||||
op_copy->inode = op->inode;
|
|
||||||
op_copy->offset = op->offset;
|
|
||||||
op_copy->len = op->len;
|
|
||||||
op_copy->buf = malloc_or_die(op->len);
|
|
||||||
op_copy->iov.push_back(op_copy->buf, op->len);
|
|
||||||
op_copy->callback = [](cluster_op_t* op_copy)
|
|
||||||
{
|
|
||||||
if (op_copy->orig_op)
|
|
||||||
{
|
|
||||||
// Acknowledge write and forget the original pointer
|
|
||||||
op_copy->orig_op->retval = op_copy->retval;
|
|
||||||
std::function<void(cluster_op_t*)>(op_copy->orig_op->callback)(op_copy->orig_op);
|
|
||||||
op_copy->orig_op = NULL;
|
|
||||||
}
|
|
||||||
};
|
|
||||||
void *cur_buf = op_copy->buf;
|
|
||||||
for (int i = 0; i < op->iov.count; i++)
|
|
||||||
{
|
|
||||||
memcpy(cur_buf, op->iov.buf[i].iov_base, op->iov.buf[i].iov_len);
|
|
||||||
cur_buf += op->iov.buf[i].iov_len;
|
|
||||||
}
|
|
||||||
unsynced_writes.push_back(op_copy);
|
|
||||||
cur_ops.erase(op);
|
|
||||||
cur_ops.insert(op_copy);
|
|
||||||
op = op_copy;
|
|
||||||
}
|
|
||||||
if (!op->parts.size())
|
|
||||||
{
|
|
||||||
// Slice the operation into parts
|
|
||||||
slice_rw(op);
|
|
||||||
}
|
|
||||||
if (!op->needs_reslice)
|
|
||||||
{
|
|
||||||
// Send unsent parts, if they're not subject to change
|
|
||||||
for (auto & op_part: op->parts)
|
|
||||||
{
|
|
||||||
if (!op_part.sent && !op_part.done)
|
|
||||||
{
|
|
||||||
try_send(op, &op_part);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (!op->sent_count)
|
|
||||||
{
|
|
||||||
if (op->done_count >= op->parts.size())
|
|
||||||
{
|
|
||||||
// Finished successfully
|
|
||||||
// Even if the PG count has changed in meanwhile we treat it as success
|
|
||||||
// because if some operations were invalid for the new PG count we'd get errors
|
|
||||||
cur_ops.erase(op);
|
|
||||||
op->retval = op->len;
|
|
||||||
std::function<void(cluster_op_t*)>(op->callback)(op);
|
|
||||||
continue_sync();
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
else if (op->retval != 0 && op->retval != -EPIPE)
|
|
||||||
{
|
|
||||||
// Fatal error (not -EPIPE)
|
|
||||||
cur_ops.erase(op);
|
|
||||||
if (!immediate_commit && op->opcode == OSD_OP_WRITE)
|
|
||||||
{
|
|
||||||
for (int i = 0; i < unsynced_writes.size(); i++)
|
|
||||||
{
|
|
||||||
if (unsynced_writes[i] == op)
|
|
||||||
{
|
|
||||||
unsynced_writes.erase(unsynced_writes.begin()+i, unsynced_writes.begin()+i+1);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
bool del = op->is_internal;
|
|
||||||
std::function<void(cluster_op_t*)>(op->callback)(op);
|
|
||||||
if (del)
|
|
||||||
{
|
|
||||||
if (op->buf)
|
|
||||||
free(op->buf);
|
|
||||||
delete op;
|
|
||||||
}
|
|
||||||
continue_sync();
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// -EPIPE or no error - clear the error
|
|
||||||
op->retval = 0;
|
|
||||||
if (op->needs_reslice)
|
|
||||||
{
|
|
||||||
op->parts.clear();
|
|
||||||
op->done_count = 0;
|
|
||||||
op->needs_reslice = false;
|
|
||||||
continue_rw(op);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::slice_rw(cluster_op_t *op)
|
|
||||||
{
|
|
||||||
// Slice the request into individual object stripe requests
|
|
||||||
// Primary OSDs still operate individual stripes, but their size is multiplied by PG minsize in case of EC
|
|
||||||
auto & pool_cfg = st_cli.pool_config[INODE_POOL(op->inode)];
|
|
||||||
uint64_t pg_block_size = bs_block_size * (
|
|
||||||
pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_minsize
|
|
||||||
);
|
|
||||||
uint64_t first_stripe = (op->offset / pg_block_size) * pg_block_size;
|
|
||||||
uint64_t last_stripe = ((op->offset + op->len + pg_block_size - 1) / pg_block_size - 1) * pg_block_size;
|
|
||||||
op->retval = 0;
|
|
||||||
op->parts.resize((last_stripe - first_stripe) / pg_block_size + 1);
|
|
||||||
int iov_idx = 0;
|
|
||||||
size_t iov_pos = 0;
|
|
||||||
int i = 0;
|
|
||||||
for (uint64_t stripe = first_stripe; stripe <= last_stripe; stripe += pg_block_size)
|
|
||||||
{
|
|
||||||
pg_num_t pg_num = (op->inode + stripe/pool_cfg.pg_stripe_size) % pool_cfg.real_pg_count + 1;
|
|
||||||
uint64_t begin = (op->offset < stripe ? stripe : op->offset);
|
|
||||||
uint64_t end = (op->offset + op->len) > (stripe + pg_block_size)
|
|
||||||
? (stripe + pg_block_size) : (op->offset + op->len);
|
|
||||||
op->parts[i] = (cluster_op_part_t){
|
|
||||||
.parent = op,
|
|
||||||
.offset = begin,
|
|
||||||
.len = (uint32_t)(end - begin),
|
|
||||||
.pg_num = pg_num,
|
|
||||||
.sent = false,
|
|
||||||
.done = false,
|
|
||||||
};
|
|
||||||
int left = end-begin;
|
|
||||||
while (left > 0 && iov_idx < op->iov.count)
|
|
||||||
{
|
|
||||||
if (op->iov.buf[iov_idx].iov_len - iov_pos < left)
|
|
||||||
{
|
|
||||||
op->parts[i].iov.push_back(op->iov.buf[iov_idx].iov_base + iov_pos, op->iov.buf[iov_idx].iov_len - iov_pos);
|
|
||||||
left -= (op->iov.buf[iov_idx].iov_len - iov_pos);
|
|
||||||
iov_pos = 0;
|
|
||||||
iov_idx++;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
op->parts[i].iov.push_back(op->iov.buf[iov_idx].iov_base + iov_pos, left);
|
|
||||||
iov_pos += left;
|
|
||||||
left = 0;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
assert(left == 0);
|
|
||||||
i++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
bool cluster_client_t::try_send(cluster_op_t *op, cluster_op_part_t *part)
|
|
||||||
{
|
|
||||||
auto & pool_cfg = st_cli.pool_config[INODE_POOL(op->inode)];
|
|
||||||
auto pg_it = pool_cfg.pg_config.find(part->pg_num);
|
|
||||||
if (pg_it != pool_cfg.pg_config.end() &&
|
|
||||||
!pg_it->second.pause && pg_it->second.cur_primary)
|
|
||||||
{
|
|
||||||
osd_num_t primary_osd = pg_it->second.cur_primary;
|
|
||||||
auto peer_it = msgr.osd_peer_fds.find(primary_osd);
|
|
||||||
if (peer_it != msgr.osd_peer_fds.end())
|
|
||||||
{
|
|
||||||
int peer_fd = peer_it->second;
|
|
||||||
part->osd_num = primary_osd;
|
|
||||||
part->sent = true;
|
|
||||||
op->sent_count++;
|
|
||||||
part->op = (osd_op_t){
|
|
||||||
.op_type = OSD_OP_OUT,
|
|
||||||
.peer_fd = peer_fd,
|
|
||||||
.req = { .rw = {
|
|
||||||
.header = {
|
|
||||||
.magic = SECONDARY_OSD_OP_MAGIC,
|
|
||||||
.id = op_id++,
|
|
||||||
.opcode = op->opcode,
|
|
||||||
},
|
|
||||||
.inode = op->inode,
|
|
||||||
.offset = part->offset,
|
|
||||||
.len = part->len,
|
|
||||||
} },
|
|
||||||
.callback = [this, part](osd_op_t *op_part)
|
|
||||||
{
|
|
||||||
handle_op_part(part);
|
|
||||||
},
|
|
||||||
};
|
|
||||||
part->op.iov = part->iov;
|
|
||||||
msgr.outbox_push(&part->op);
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
else if (msgr.wanted_peers.find(primary_osd) == msgr.wanted_peers.end())
|
|
||||||
{
|
|
||||||
msgr.connect_peer(primary_osd, st_cli.peer_states[primary_osd]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::execute_sync(cluster_op_t *op)
|
|
||||||
{
|
|
||||||
if (immediate_commit)
|
|
||||||
{
|
|
||||||
// Syncs are not required in the immediate_commit mode
|
|
||||||
op->retval = 0;
|
|
||||||
std::function<void(cluster_op_t*)>(op->callback)(op);
|
|
||||||
}
|
|
||||||
else if (cur_sync != NULL)
|
|
||||||
{
|
|
||||||
next_writes.push_back(op);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
cur_sync = op;
|
|
||||||
continue_sync();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::continue_sync()
|
|
||||||
{
|
|
||||||
if (!cur_sync || cur_sync->parts.size() > 0)
|
|
||||||
{
|
|
||||||
// Already submitted
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
cur_sync->retval = 0;
|
|
||||||
std::set<osd_num_t> sync_osds;
|
|
||||||
for (auto prev_op: unsynced_writes)
|
|
||||||
{
|
|
||||||
if (prev_op->done_count < prev_op->parts.size())
|
|
||||||
{
|
|
||||||
// Writes not finished yet
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
for (auto & part: prev_op->parts)
|
|
||||||
{
|
|
||||||
if (part.osd_num)
|
|
||||||
{
|
|
||||||
sync_osds.insert(part.osd_num);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (!sync_osds.size())
|
|
||||||
{
|
|
||||||
// No dirty writes
|
|
||||||
finish_sync();
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
// Check that all OSD connections are still alive
|
|
||||||
for (auto sync_osd: sync_osds)
|
|
||||||
{
|
|
||||||
auto peer_it = msgr.osd_peer_fds.find(sync_osd);
|
|
||||||
if (peer_it == msgr.osd_peer_fds.end())
|
|
||||||
{
|
|
||||||
// SYNC is pointless to send to a non connected OSD
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
syncing_writes.swap(unsynced_writes);
|
|
||||||
// Post sync to affected OSDs
|
|
||||||
cur_sync->parts.resize(sync_osds.size());
|
|
||||||
int i = 0;
|
|
||||||
for (auto sync_osd: sync_osds)
|
|
||||||
{
|
|
||||||
cur_sync->parts[i] = {
|
|
||||||
.parent = cur_sync,
|
|
||||||
.osd_num = sync_osd,
|
|
||||||
.sent = false,
|
|
||||||
.done = false,
|
|
||||||
};
|
|
||||||
send_sync(cur_sync, &cur_sync->parts[i]);
|
|
||||||
i++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::finish_sync()
|
|
||||||
{
|
|
||||||
int retval = cur_sync->retval;
|
|
||||||
if (retval != 0)
|
|
||||||
{
|
|
||||||
for (auto op: syncing_writes)
|
|
||||||
{
|
|
||||||
if (op->done_count < op->parts.size())
|
|
||||||
{
|
|
||||||
cur_ops.insert(op);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
unsynced_writes.insert(unsynced_writes.begin(), syncing_writes.begin(), syncing_writes.end());
|
|
||||||
syncing_writes.clear();
|
|
||||||
}
|
|
||||||
if (retval == -EPIPE)
|
|
||||||
{
|
|
||||||
// Retry later
|
|
||||||
cur_sync->parts.clear();
|
|
||||||
cur_sync->retval = 0;
|
|
||||||
cur_sync->sent_count = 0;
|
|
||||||
cur_sync->done_count = 0;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
std::function<void(cluster_op_t*)>(cur_sync->callback)(cur_sync);
|
|
||||||
if (!retval)
|
|
||||||
{
|
|
||||||
for (auto op: syncing_writes)
|
|
||||||
{
|
|
||||||
assert(op->sent_count == 0);
|
|
||||||
if (op->is_internal)
|
|
||||||
{
|
|
||||||
if (op->buf)
|
|
||||||
free(op->buf);
|
|
||||||
delete op;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
syncing_writes.clear();
|
|
||||||
}
|
|
||||||
cur_sync = NULL;
|
|
||||||
queued_bytes = 0;
|
|
||||||
std::vector<cluster_op_t*> next_wr_copy;
|
|
||||||
next_wr_copy.swap(next_writes);
|
|
||||||
for (auto next_op: next_wr_copy)
|
|
||||||
{
|
|
||||||
execute(next_op);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::send_sync(cluster_op_t *op, cluster_op_part_t *part)
|
|
||||||
{
|
|
||||||
auto peer_it = msgr.osd_peer_fds.find(part->osd_num);
|
|
||||||
assert(peer_it != msgr.osd_peer_fds.end());
|
|
||||||
part->sent = true;
|
|
||||||
op->sent_count++;
|
|
||||||
part->op = (osd_op_t){
|
|
||||||
.op_type = OSD_OP_OUT,
|
|
||||||
.peer_fd = peer_it->second,
|
|
||||||
.req = {
|
|
||||||
.hdr = {
|
|
||||||
.magic = SECONDARY_OSD_OP_MAGIC,
|
|
||||||
.id = op_id++,
|
|
||||||
.opcode = OSD_OP_SYNC,
|
|
||||||
},
|
|
||||||
},
|
|
||||||
.callback = [this, part](osd_op_t *op_part)
|
|
||||||
{
|
|
||||||
handle_op_part(part);
|
|
||||||
},
|
|
||||||
};
|
|
||||||
msgr.outbox_push(&part->op);
|
|
||||||
}
|
|
||||||
|
|
||||||
void cluster_client_t::handle_op_part(cluster_op_part_t *part)
|
|
||||||
{
|
|
||||||
cluster_op_t *op = part->parent;
|
|
||||||
part->sent = false;
|
|
||||||
op->sent_count--;
|
|
||||||
int expected = part->op.req.hdr.opcode == OSD_OP_SYNC ? 0 : part->op.req.rw.len;
|
|
||||||
if (part->op.reply.hdr.retval != expected)
|
|
||||||
{
|
|
||||||
// Operation failed, retry
|
|
||||||
printf(
|
|
||||||
"Operation failed on OSD %lu: retval=%ld (expected %d), dropping connection\n",
|
|
||||||
part->osd_num, part->op.reply.hdr.retval, expected
|
|
||||||
);
|
|
||||||
msgr.stop_client(part->op.peer_fd);
|
|
||||||
if (part->op.reply.hdr.retval == -EPIPE)
|
|
||||||
{
|
|
||||||
op->up_wait = true;
|
|
||||||
if (!retry_timeout_id)
|
|
||||||
{
|
|
||||||
retry_timeout_id = tfd->set_timer(up_wait_retry_interval, false, [this](int)
|
|
||||||
{
|
|
||||||
retry_timeout_id = 0;
|
|
||||||
continue_ops(true);
|
|
||||||
});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (!op->retval || op->retval == -EPIPE)
|
|
||||||
{
|
|
||||||
// Don't overwrite other errors with -EPIPE
|
|
||||||
op->retval = part->op.reply.hdr.retval;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// OK
|
|
||||||
part->done = true;
|
|
||||||
op->done_count++;
|
|
||||||
}
|
|
||||||
if (op->sent_count == 0)
|
|
||||||
{
|
|
||||||
if (op->opcode == OSD_OP_SYNC)
|
|
||||||
{
|
|
||||||
assert(op == cur_sync);
|
|
||||||
finish_sync();
|
|
||||||
}
|
|
||||||
else if (!op->up_wait)
|
|
||||||
{
|
|
||||||
continue_rw(op);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,106 +0,0 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
|
||||||
// License: VNPL-1.0 or GNU GPL-2.0+ (see README.md for details)
|
|
||||||
|
|
||||||
#pragma once
|
|
||||||
|
|
||||||
#include "messenger.h"
|
|
||||||
#include "etcd_state_client.h"
|
|
||||||
|
|
||||||
#define MIN_BLOCK_SIZE 4*1024
|
|
||||||
#define MAX_BLOCK_SIZE 128*1024*1024
|
|
||||||
#define DEFAULT_DISK_ALIGNMENT 4096
|
|
||||||
#define DEFAULT_BITMAP_GRANULARITY 4096
|
|
||||||
#define DEFAULT_CLIENT_DIRTY_LIMIT 32*1024*1024
|
|
||||||
|
|
||||||
struct cluster_op_t;
|
|
||||||
|
|
||||||
struct cluster_op_part_t
|
|
||||||
{
|
|
||||||
cluster_op_t *parent;
|
|
||||||
uint64_t offset;
|
|
||||||
uint32_t len;
|
|
||||||
pg_num_t pg_num;
|
|
||||||
osd_num_t osd_num;
|
|
||||||
osd_op_buf_list_t iov;
|
|
||||||
bool sent;
|
|
||||||
bool done;
|
|
||||||
osd_op_t op;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct cluster_op_t
|
|
||||||
{
|
|
||||||
uint64_t opcode; // OSD_OP_READ, OSD_OP_WRITE, OSD_OP_SYNC
|
|
||||||
uint64_t inode;
|
|
||||||
uint64_t offset;
|
|
||||||
uint64_t len;
|
|
||||||
int retval;
|
|
||||||
osd_op_buf_list_t iov;
|
|
||||||
std::function<void(cluster_op_t*)> callback;
|
|
||||||
protected:
|
|
||||||
void *buf = NULL;
|
|
||||||
cluster_op_t *orig_op = NULL;
|
|
||||||
bool is_internal = false;
|
|
||||||
bool needs_reslice = false;
|
|
||||||
bool up_wait = false;
|
|
||||||
int sent_count = 0, done_count = 0;
|
|
||||||
std::vector<cluster_op_part_t> parts;
|
|
||||||
friend class cluster_client_t;
|
|
||||||
};
|
|
||||||
|
|
||||||
class cluster_client_t
|
|
||||||
{
|
|
||||||
timerfd_manager_t *tfd;
|
|
||||||
ring_loop_t *ringloop;
|
|
||||||
|
|
||||||
uint64_t bs_block_size = 0;
|
|
||||||
uint64_t bs_disk_alignment = 0;
|
|
||||||
uint64_t bs_bitmap_granularity = 0;
|
|
||||||
std::map<pool_id_t, uint64_t> pg_counts;
|
|
||||||
bool immediate_commit = false;
|
|
||||||
// FIXME: Implement inmemory_commit mode. Note that it requires to return overlapping reads from memory.
|
|
||||||
uint64_t client_dirty_limit = 0;
|
|
||||||
int log_level;
|
|
||||||
int up_wait_retry_interval = 500; // ms
|
|
||||||
|
|
||||||
uint64_t op_id = 1;
|
|
||||||
ring_consumer_t consumer;
|
|
||||||
// operations currently in progress
|
|
||||||
std::set<cluster_op_t*> cur_ops;
|
|
||||||
int retry_timeout_id = 0;
|
|
||||||
// unsynced operations are copied in memory to allow replay when cluster isn't in the immediate_commit mode
|
|
||||||
// unsynced_writes are replayed in any order (because only the SYNC operation guarantees ordering)
|
|
||||||
std::vector<cluster_op_t*> unsynced_writes;
|
|
||||||
std::vector<cluster_op_t*> syncing_writes;
|
|
||||||
cluster_op_t* cur_sync = NULL;
|
|
||||||
std::vector<cluster_op_t*> next_writes;
|
|
||||||
std::vector<cluster_op_t*> offline_ops;
|
|
||||||
uint64_t queued_bytes = 0;
|
|
||||||
|
|
||||||
bool pgs_loaded = false;
|
|
||||||
std::vector<std::function<void(void)>> on_ready_hooks;
|
|
||||||
|
|
||||||
public:
|
|
||||||
etcd_state_client_t st_cli;
|
|
||||||
osd_messenger_t msgr;
|
|
||||||
|
|
||||||
cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json & config);
|
|
||||||
~cluster_client_t();
|
|
||||||
void execute(cluster_op_t *op);
|
|
||||||
void on_ready(std::function<void(void)> fn);
|
|
||||||
void stop();
|
|
||||||
|
|
||||||
protected:
|
|
||||||
void continue_ops(bool up_retry = false);
|
|
||||||
void on_load_config_hook(json11::Json::object & config);
|
|
||||||
void on_load_pgs_hook(bool success);
|
|
||||||
void on_change_hook(json11::Json::object & changes);
|
|
||||||
void on_change_osd_state_hook(uint64_t peer_osd);
|
|
||||||
void continue_rw(cluster_op_t *op);
|
|
||||||
void slice_rw(cluster_op_t *op);
|
|
||||||
bool try_send(cluster_op_t *op, cluster_op_part_t *part);
|
|
||||||
void execute_sync(cluster_op_t *op);
|
|
||||||
void continue_sync();
|
|
||||||
void finish_sync();
|
|
||||||
void send_sync(cluster_op_t *op, cluster_op_part_t *part);
|
|
||||||
void handle_op_part(cluster_op_part_t *part);
|
|
||||||
};
|
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
|
|
||||||
gcc -E -o fio_headers.i fio_headers.h
|
gcc -I. -E -o fio_headers.i src/fio_headers.h
|
||||||
|
|
||||||
rm -rf fio-copy
|
rm -rf fio-copy
|
||||||
for i in `grep -Po 'fio/[^"]+' fio_headers.i | sort | uniq`; do
|
for i in `grep -Po 'fio/[^"]+' fio_headers.i | sort | uniq`; do
|
||||||
|
|||||||
@@ -5,7 +5,7 @@
|
|||||||
#cd b/qemu; make qapi
|
#cd b/qemu; make qapi
|
||||||
|
|
||||||
gcc -I qemu/b/qemu `pkg-config glib-2.0 --cflags` \
|
gcc -I qemu/b/qemu `pkg-config glib-2.0 --cflags` \
|
||||||
-I qemu/include -E -o qemu_driver.i qemu_driver.c
|
-I qemu/include -E -o qemu_driver.i src/qemu_driver.c
|
||||||
|
|
||||||
rm -rf qemu-copy
|
rm -rf qemu-copy
|
||||||
for i in `grep -Po 'qemu/[^"]+' qemu_driver.i | sort | uniq`; do
|
for i in `grep -Po 'qemu/[^"]+' qemu_driver.i | sort | uniq`; do
|
||||||
|
|||||||
+1
-1
Submodule cpp-btree updated: 5dc108754a...6e20146406
@@ -0,0 +1,2 @@
|
|||||||
|
vitastor-csi
|
||||||
|
Dockerfile
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
# Compile stage
|
||||||
|
FROM golang:buster AS build
|
||||||
|
|
||||||
|
ADD go.sum go.mod /app/
|
||||||
|
RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go mod download -x
|
||||||
|
ADD . /app
|
||||||
|
RUN perl -i -e '$/ = undef; while(<>) { s/\n\s*(\{\s*\n)/$1\n/g; s/\}(\s*\n\s*)else\b/$1} else/g; print; }' `find /app -name '*.go'`
|
||||||
|
RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o vitastor-csi
|
||||||
|
|
||||||
|
# Final stage
|
||||||
|
FROM debian:buster
|
||||||
|
|
||||||
|
LABEL maintainers="Vitaliy Filippov <vitalif@yourcmc.ru>"
|
||||||
|
LABEL description="Vitastor CSI Driver"
|
||||||
|
|
||||||
|
ENV NODE_ID=""
|
||||||
|
ENV CSI_ENDPOINT=""
|
||||||
|
|
||||||
|
RUN apt-get update && \
|
||||||
|
apt-get install -y wget && \
|
||||||
|
wget -q -O /etc/apt/trusted.gpg.d/vitastor.gpg https://vitastor.io/debian/pubkey.gpg && \
|
||||||
|
(echo deb http://vitastor.io/debian buster main > /etc/apt/sources.list.d/vitastor.list) && \
|
||||||
|
(echo deb http://deb.debian.org/debian buster-backports main > /etc/apt/sources.list.d/backports.list) && \
|
||||||
|
(echo "APT::Install-Recommends false;" > /etc/apt/apt.conf) && \
|
||||||
|
apt-get update && \
|
||||||
|
apt-get install -y e2fsprogs xfsprogs vitastor kmod && \
|
||||||
|
apt-get clean && \
|
||||||
|
(echo options nbd nbds_max=128 > /etc/modprobe.d/nbd.conf)
|
||||||
|
|
||||||
|
COPY --from=build /app/vitastor-csi /bin/
|
||||||
|
|
||||||
|
ENTRYPOINT ["/bin/vitastor-csi"]
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
VERSION ?= v0.6.16
|
||||||
|
|
||||||
|
all: build push
|
||||||
|
|
||||||
|
build:
|
||||||
|
@docker build --rm -t vitalif/vitastor-csi:$(VERSION) .
|
||||||
|
|
||||||
|
push:
|
||||||
|
@docker push vitalif/vitastor-csi:$(VERSION)
|
||||||
@@ -0,0 +1,5 @@
|
|||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
kind: Namespace
|
||||||
|
metadata:
|
||||||
|
name: vitastor-system
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
kind: ConfigMap
|
||||||
|
data:
|
||||||
|
vitastor.conf: |-
|
||||||
|
{"etcd_address":"http://192.168.7.2:2379","etcd_prefix":"/vitastor"}
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor-config
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
kind: ServiceAccount
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor-csi-nodeplugin
|
||||||
|
---
|
||||||
|
kind: ClusterRole
|
||||||
|
apiVersion: rbac.authorization.k8s.io/v1
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor-csi-nodeplugin
|
||||||
|
rules:
|
||||||
|
- apiGroups: [""]
|
||||||
|
resources: ["nodes"]
|
||||||
|
verbs: ["get"]
|
||||||
|
# allow to read Vault Token and connection options from the Tenants namespace
|
||||||
|
- apiGroups: [""]
|
||||||
|
resources: ["secrets"]
|
||||||
|
verbs: ["get"]
|
||||||
|
- apiGroups: [""]
|
||||||
|
resources: ["configmaps"]
|
||||||
|
verbs: ["get"]
|
||||||
|
---
|
||||||
|
kind: ClusterRoleBinding
|
||||||
|
apiVersion: rbac.authorization.k8s.io/v1
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor-csi-nodeplugin
|
||||||
|
subjects:
|
||||||
|
- kind: ServiceAccount
|
||||||
|
name: vitastor-csi-nodeplugin
|
||||||
|
namespace: vitastor-system
|
||||||
|
roleRef:
|
||||||
|
kind: ClusterRole
|
||||||
|
name: vitastor-csi-nodeplugin
|
||||||
|
apiGroup: rbac.authorization.k8s.io
|
||||||
@@ -0,0 +1,72 @@
|
|||||||
|
---
|
||||||
|
apiVersion: policy/v1beta1
|
||||||
|
kind: PodSecurityPolicy
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor-csi-nodeplugin-psp
|
||||||
|
spec:
|
||||||
|
allowPrivilegeEscalation: true
|
||||||
|
allowedCapabilities:
|
||||||
|
- 'SYS_ADMIN'
|
||||||
|
fsGroup:
|
||||||
|
rule: RunAsAny
|
||||||
|
privileged: true
|
||||||
|
hostNetwork: true
|
||||||
|
hostPID: true
|
||||||
|
runAsUser:
|
||||||
|
rule: RunAsAny
|
||||||
|
seLinux:
|
||||||
|
rule: RunAsAny
|
||||||
|
supplementalGroups:
|
||||||
|
rule: RunAsAny
|
||||||
|
volumes:
|
||||||
|
- 'configMap'
|
||||||
|
- 'emptyDir'
|
||||||
|
- 'projected'
|
||||||
|
- 'secret'
|
||||||
|
- 'downwardAPI'
|
||||||
|
- 'hostPath'
|
||||||
|
allowedHostPaths:
|
||||||
|
- pathPrefix: '/dev'
|
||||||
|
readOnly: false
|
||||||
|
- pathPrefix: '/run/mount'
|
||||||
|
readOnly: false
|
||||||
|
- pathPrefix: '/sys'
|
||||||
|
readOnly: false
|
||||||
|
- pathPrefix: '/lib/modules'
|
||||||
|
readOnly: true
|
||||||
|
- pathPrefix: '/var/lib/kubelet/pods'
|
||||||
|
readOnly: false
|
||||||
|
- pathPrefix: '/var/lib/kubelet/plugins/csi.vitastor.io'
|
||||||
|
readOnly: false
|
||||||
|
- pathPrefix: '/var/lib/kubelet/plugins_registry'
|
||||||
|
readOnly: false
|
||||||
|
- pathPrefix: '/var/lib/kubelet/plugins'
|
||||||
|
readOnly: false
|
||||||
|
|
||||||
|
---
|
||||||
|
kind: Role
|
||||||
|
apiVersion: rbac.authorization.k8s.io/v1
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor-csi-nodeplugin-psp
|
||||||
|
rules:
|
||||||
|
- apiGroups: ['policy']
|
||||||
|
resources: ['podsecuritypolicies']
|
||||||
|
verbs: ['use']
|
||||||
|
resourceNames: ['vitastor-csi-nodeplugin-psp']
|
||||||
|
|
||||||
|
---
|
||||||
|
kind: RoleBinding
|
||||||
|
apiVersion: rbac.authorization.k8s.io/v1
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor-csi-nodeplugin-psp
|
||||||
|
subjects:
|
||||||
|
- kind: ServiceAccount
|
||||||
|
name: vitastor-csi-nodeplugin
|
||||||
|
namespace: vitastor-system
|
||||||
|
roleRef:
|
||||||
|
kind: Role
|
||||||
|
name: vitastor-csi-nodeplugin-psp
|
||||||
|
apiGroup: rbac.authorization.k8s.io
|
||||||
@@ -0,0 +1,140 @@
|
|||||||
|
---
|
||||||
|
kind: DaemonSet
|
||||||
|
apiVersion: apps/v1
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: csi-vitastor
|
||||||
|
spec:
|
||||||
|
selector:
|
||||||
|
matchLabels:
|
||||||
|
app: csi-vitastor
|
||||||
|
template:
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
labels:
|
||||||
|
app: csi-vitastor
|
||||||
|
spec:
|
||||||
|
serviceAccountName: vitastor-csi-nodeplugin
|
||||||
|
hostNetwork: true
|
||||||
|
hostPID: true
|
||||||
|
priorityClassName: system-node-critical
|
||||||
|
# to use e.g. Rook orchestrated cluster, and mons' FQDN is
|
||||||
|
# resolved through k8s service, set dns policy to cluster first
|
||||||
|
dnsPolicy: ClusterFirstWithHostNet
|
||||||
|
containers:
|
||||||
|
- name: driver-registrar
|
||||||
|
# This is necessary only for systems with SELinux, where
|
||||||
|
# non-privileged sidecar containers cannot access unix domain socket
|
||||||
|
# created by privileged CSI driver container.
|
||||||
|
securityContext:
|
||||||
|
privileged: true
|
||||||
|
image: k8s.gcr.io/sig-storage/csi-node-driver-registrar:v2.2.0
|
||||||
|
args:
|
||||||
|
- "--v=5"
|
||||||
|
- "--csi-address=/csi/csi.sock"
|
||||||
|
- "--kubelet-registration-path=/var/lib/kubelet/plugins/csi.vitastor.io/csi.sock"
|
||||||
|
env:
|
||||||
|
- name: KUBE_NODE_NAME
|
||||||
|
valueFrom:
|
||||||
|
fieldRef:
|
||||||
|
fieldPath: spec.nodeName
|
||||||
|
volumeMounts:
|
||||||
|
- name: socket-dir
|
||||||
|
mountPath: /csi
|
||||||
|
- name: registration-dir
|
||||||
|
mountPath: /registration
|
||||||
|
- name: csi-vitastor
|
||||||
|
securityContext:
|
||||||
|
privileged: true
|
||||||
|
capabilities:
|
||||||
|
add: ["SYS_ADMIN"]
|
||||||
|
allowPrivilegeEscalation: true
|
||||||
|
image: vitalif/vitastor-csi:v0.6.16
|
||||||
|
args:
|
||||||
|
- "--node=$(NODE_ID)"
|
||||||
|
- "--endpoint=$(CSI_ENDPOINT)"
|
||||||
|
env:
|
||||||
|
- name: NODE_ID
|
||||||
|
valueFrom:
|
||||||
|
fieldRef:
|
||||||
|
fieldPath: spec.nodeName
|
||||||
|
- name: CSI_ENDPOINT
|
||||||
|
value: unix:///csi/csi.sock
|
||||||
|
imagePullPolicy: "IfNotPresent"
|
||||||
|
ports:
|
||||||
|
- containerPort: 9898
|
||||||
|
name: healthz
|
||||||
|
protocol: TCP
|
||||||
|
livenessProbe:
|
||||||
|
failureThreshold: 5
|
||||||
|
httpGet:
|
||||||
|
path: /healthz
|
||||||
|
port: healthz
|
||||||
|
initialDelaySeconds: 10
|
||||||
|
timeoutSeconds: 3
|
||||||
|
periodSeconds: 2
|
||||||
|
volumeMounts:
|
||||||
|
- name: socket-dir
|
||||||
|
mountPath: /csi
|
||||||
|
- mountPath: /dev
|
||||||
|
name: host-dev
|
||||||
|
- mountPath: /sys
|
||||||
|
name: host-sys
|
||||||
|
- mountPath: /run/mount
|
||||||
|
name: host-mount
|
||||||
|
- mountPath: /lib/modules
|
||||||
|
name: lib-modules
|
||||||
|
readOnly: true
|
||||||
|
- name: vitastor-config
|
||||||
|
mountPath: /etc/vitastor
|
||||||
|
- name: plugin-dir
|
||||||
|
mountPath: /var/lib/kubelet/plugins
|
||||||
|
mountPropagation: "Bidirectional"
|
||||||
|
- name: mountpoint-dir
|
||||||
|
mountPath: /var/lib/kubelet/pods
|
||||||
|
mountPropagation: "Bidirectional"
|
||||||
|
- name: liveness-probe
|
||||||
|
securityContext:
|
||||||
|
privileged: true
|
||||||
|
image: quay.io/k8scsi/livenessprobe:v1.1.0
|
||||||
|
args:
|
||||||
|
- "--csi-address=$(CSI_ENDPOINT)"
|
||||||
|
- "--health-port=9898"
|
||||||
|
env:
|
||||||
|
- name: CSI_ENDPOINT
|
||||||
|
value: unix://csi/csi.sock
|
||||||
|
volumeMounts:
|
||||||
|
- mountPath: /csi
|
||||||
|
name: socket-dir
|
||||||
|
volumes:
|
||||||
|
- name: socket-dir
|
||||||
|
hostPath:
|
||||||
|
path: /var/lib/kubelet/plugins/csi.vitastor.io
|
||||||
|
type: DirectoryOrCreate
|
||||||
|
- name: plugin-dir
|
||||||
|
hostPath:
|
||||||
|
path: /var/lib/kubelet/plugins
|
||||||
|
type: Directory
|
||||||
|
- name: mountpoint-dir
|
||||||
|
hostPath:
|
||||||
|
path: /var/lib/kubelet/pods
|
||||||
|
type: DirectoryOrCreate
|
||||||
|
- name: registration-dir
|
||||||
|
hostPath:
|
||||||
|
path: /var/lib/kubelet/plugins_registry/
|
||||||
|
type: Directory
|
||||||
|
- name: host-dev
|
||||||
|
hostPath:
|
||||||
|
path: /dev
|
||||||
|
- name: host-sys
|
||||||
|
hostPath:
|
||||||
|
path: /sys
|
||||||
|
- name: host-mount
|
||||||
|
hostPath:
|
||||||
|
path: /run/mount
|
||||||
|
- name: lib-modules
|
||||||
|
hostPath:
|
||||||
|
path: /lib/modules
|
||||||
|
- name: vitastor-config
|
||||||
|
configMap:
|
||||||
|
name: vitastor-config
|
||||||
@@ -0,0 +1,102 @@
|
|||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
kind: ServiceAccount
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor-csi-provisioner
|
||||||
|
|
||||||
|
---
|
||||||
|
kind: ClusterRole
|
||||||
|
apiVersion: rbac.authorization.k8s.io/v1
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor-external-provisioner-runner
|
||||||
|
rules:
|
||||||
|
- apiGroups: [""]
|
||||||
|
resources: ["nodes"]
|
||||||
|
verbs: ["get", "list", "watch"]
|
||||||
|
- apiGroups: [""]
|
||||||
|
resources: ["secrets"]
|
||||||
|
verbs: ["get", "list", "watch"]
|
||||||
|
- apiGroups: [""]
|
||||||
|
resources: ["events"]
|
||||||
|
verbs: ["list", "watch", "create", "update", "patch"]
|
||||||
|
- apiGroups: [""]
|
||||||
|
resources: ["persistentvolumes"]
|
||||||
|
verbs: ["get", "list", "watch", "create", "update", "delete", "patch"]
|
||||||
|
- apiGroups: [""]
|
||||||
|
resources: ["persistentvolumeclaims"]
|
||||||
|
verbs: ["get", "list", "watch", "update"]
|
||||||
|
- apiGroups: [""]
|
||||||
|
resources: ["persistentvolumeclaims/status"]
|
||||||
|
verbs: ["update", "patch"]
|
||||||
|
- apiGroups: ["storage.k8s.io"]
|
||||||
|
resources: ["storageclasses"]
|
||||||
|
verbs: ["get", "list", "watch"]
|
||||||
|
- apiGroups: ["snapshot.storage.k8s.io"]
|
||||||
|
resources: ["volumesnapshots"]
|
||||||
|
verbs: ["get", "list"]
|
||||||
|
- apiGroups: ["snapshot.storage.k8s.io"]
|
||||||
|
resources: ["volumesnapshotcontents"]
|
||||||
|
verbs: ["create", "get", "list", "watch", "update", "delete"]
|
||||||
|
- apiGroups: ["snapshot.storage.k8s.io"]
|
||||||
|
resources: ["volumesnapshotclasses"]
|
||||||
|
verbs: ["get", "list", "watch"]
|
||||||
|
- apiGroups: ["storage.k8s.io"]
|
||||||
|
resources: ["volumeattachments"]
|
||||||
|
verbs: ["get", "list", "watch", "update", "patch"]
|
||||||
|
- apiGroups: ["storage.k8s.io"]
|
||||||
|
resources: ["volumeattachments/status"]
|
||||||
|
verbs: ["patch"]
|
||||||
|
- apiGroups: ["storage.k8s.io"]
|
||||||
|
resources: ["csinodes"]
|
||||||
|
verbs: ["get", "list", "watch"]
|
||||||
|
- apiGroups: ["snapshot.storage.k8s.io"]
|
||||||
|
resources: ["volumesnapshotcontents/status"]
|
||||||
|
verbs: ["update"]
|
||||||
|
- apiGroups: [""]
|
||||||
|
resources: ["configmaps"]
|
||||||
|
verbs: ["get"]
|
||||||
|
---
|
||||||
|
kind: ClusterRoleBinding
|
||||||
|
apiVersion: rbac.authorization.k8s.io/v1
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor-csi-provisioner-role
|
||||||
|
subjects:
|
||||||
|
- kind: ServiceAccount
|
||||||
|
name: vitastor-csi-provisioner
|
||||||
|
namespace: vitastor-system
|
||||||
|
roleRef:
|
||||||
|
kind: ClusterRole
|
||||||
|
name: vitastor-external-provisioner-runner
|
||||||
|
apiGroup: rbac.authorization.k8s.io
|
||||||
|
|
||||||
|
---
|
||||||
|
kind: Role
|
||||||
|
apiVersion: rbac.authorization.k8s.io/v1
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor-external-provisioner-cfg
|
||||||
|
rules:
|
||||||
|
- apiGroups: [""]
|
||||||
|
resources: ["configmaps"]
|
||||||
|
verbs: ["get", "list", "watch", "create", "update", "delete"]
|
||||||
|
- apiGroups: ["coordination.k8s.io"]
|
||||||
|
resources: ["leases"]
|
||||||
|
verbs: ["get", "watch", "list", "delete", "update", "create"]
|
||||||
|
|
||||||
|
---
|
||||||
|
kind: RoleBinding
|
||||||
|
apiVersion: rbac.authorization.k8s.io/v1
|
||||||
|
metadata:
|
||||||
|
name: vitastor-csi-provisioner-role-cfg
|
||||||
|
namespace: vitastor-system
|
||||||
|
subjects:
|
||||||
|
- kind: ServiceAccount
|
||||||
|
name: vitastor-csi-provisioner
|
||||||
|
namespace: vitastor-system
|
||||||
|
roleRef:
|
||||||
|
kind: Role
|
||||||
|
name: vitastor-external-provisioner-cfg
|
||||||
|
apiGroup: rbac.authorization.k8s.io
|
||||||
@@ -0,0 +1,60 @@
|
|||||||
|
---
|
||||||
|
apiVersion: policy/v1beta1
|
||||||
|
kind: PodSecurityPolicy
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor-csi-provisioner-psp
|
||||||
|
spec:
|
||||||
|
allowPrivilegeEscalation: true
|
||||||
|
allowedCapabilities:
|
||||||
|
- 'SYS_ADMIN'
|
||||||
|
fsGroup:
|
||||||
|
rule: RunAsAny
|
||||||
|
privileged: true
|
||||||
|
runAsUser:
|
||||||
|
rule: RunAsAny
|
||||||
|
seLinux:
|
||||||
|
rule: RunAsAny
|
||||||
|
supplementalGroups:
|
||||||
|
rule: RunAsAny
|
||||||
|
volumes:
|
||||||
|
- 'configMap'
|
||||||
|
- 'emptyDir'
|
||||||
|
- 'projected'
|
||||||
|
- 'secret'
|
||||||
|
- 'downwardAPI'
|
||||||
|
- 'hostPath'
|
||||||
|
allowedHostPaths:
|
||||||
|
- pathPrefix: '/dev'
|
||||||
|
readOnly: false
|
||||||
|
- pathPrefix: '/sys'
|
||||||
|
readOnly: false
|
||||||
|
- pathPrefix: '/lib/modules'
|
||||||
|
readOnly: true
|
||||||
|
|
||||||
|
---
|
||||||
|
kind: Role
|
||||||
|
apiVersion: rbac.authorization.k8s.io/v1
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor-csi-provisioner-psp
|
||||||
|
rules:
|
||||||
|
- apiGroups: ['policy']
|
||||||
|
resources: ['podsecuritypolicies']
|
||||||
|
verbs: ['use']
|
||||||
|
resourceNames: ['vitastor-csi-provisioner-psp']
|
||||||
|
|
||||||
|
---
|
||||||
|
kind: RoleBinding
|
||||||
|
apiVersion: rbac.authorization.k8s.io/v1
|
||||||
|
metadata:
|
||||||
|
name: vitastor-csi-provisioner-psp
|
||||||
|
namespace: vitastor-system
|
||||||
|
subjects:
|
||||||
|
- kind: ServiceAccount
|
||||||
|
name: vitastor-csi-provisioner
|
||||||
|
namespace: vitastor-system
|
||||||
|
roleRef:
|
||||||
|
kind: Role
|
||||||
|
name: vitastor-csi-provisioner-psp
|
||||||
|
apiGroup: rbac.authorization.k8s.io
|
||||||
@@ -0,0 +1,159 @@
|
|||||||
|
---
|
||||||
|
kind: Service
|
||||||
|
apiVersion: v1
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: csi-vitastor-provisioner
|
||||||
|
labels:
|
||||||
|
app: csi-metrics
|
||||||
|
spec:
|
||||||
|
selector:
|
||||||
|
app: csi-vitastor-provisioner
|
||||||
|
ports:
|
||||||
|
- name: http-metrics
|
||||||
|
port: 8080
|
||||||
|
protocol: TCP
|
||||||
|
targetPort: 8680
|
||||||
|
|
||||||
|
---
|
||||||
|
kind: Deployment
|
||||||
|
apiVersion: apps/v1
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: csi-vitastor-provisioner
|
||||||
|
spec:
|
||||||
|
replicas: 3
|
||||||
|
selector:
|
||||||
|
matchLabels:
|
||||||
|
app: csi-vitastor-provisioner
|
||||||
|
template:
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
labels:
|
||||||
|
app: csi-vitastor-provisioner
|
||||||
|
spec:
|
||||||
|
affinity:
|
||||||
|
podAntiAffinity:
|
||||||
|
requiredDuringSchedulingIgnoredDuringExecution:
|
||||||
|
- labelSelector:
|
||||||
|
matchExpressions:
|
||||||
|
- key: app
|
||||||
|
operator: In
|
||||||
|
values:
|
||||||
|
- csi-vitastor-provisioner
|
||||||
|
topologyKey: "kubernetes.io/hostname"
|
||||||
|
serviceAccountName: vitastor-csi-provisioner
|
||||||
|
priorityClassName: system-cluster-critical
|
||||||
|
containers:
|
||||||
|
- name: csi-provisioner
|
||||||
|
image: k8s.gcr.io/sig-storage/csi-provisioner:v2.2.0
|
||||||
|
args:
|
||||||
|
- "--csi-address=$(ADDRESS)"
|
||||||
|
- "--v=5"
|
||||||
|
- "--timeout=150s"
|
||||||
|
- "--retry-interval-start=500ms"
|
||||||
|
- "--leader-election=true"
|
||||||
|
# set it to true to use topology based provisioning
|
||||||
|
- "--feature-gates=Topology=false"
|
||||||
|
# if fstype is not specified in storageclass, ext4 is default
|
||||||
|
- "--default-fstype=ext4"
|
||||||
|
- "--extra-create-metadata=true"
|
||||||
|
env:
|
||||||
|
- name: ADDRESS
|
||||||
|
value: unix:///csi/csi-provisioner.sock
|
||||||
|
imagePullPolicy: "IfNotPresent"
|
||||||
|
volumeMounts:
|
||||||
|
- name: socket-dir
|
||||||
|
mountPath: /csi
|
||||||
|
- name: csi-snapshotter
|
||||||
|
image: k8s.gcr.io/sig-storage/csi-snapshotter:v4.0.0
|
||||||
|
args:
|
||||||
|
- "--csi-address=$(ADDRESS)"
|
||||||
|
- "--v=5"
|
||||||
|
- "--timeout=150s"
|
||||||
|
- "--leader-election=true"
|
||||||
|
env:
|
||||||
|
- name: ADDRESS
|
||||||
|
value: unix:///csi/csi-provisioner.sock
|
||||||
|
imagePullPolicy: "IfNotPresent"
|
||||||
|
securityContext:
|
||||||
|
privileged: true
|
||||||
|
volumeMounts:
|
||||||
|
- name: socket-dir
|
||||||
|
mountPath: /csi
|
||||||
|
- name: csi-attacher
|
||||||
|
image: k8s.gcr.io/sig-storage/csi-attacher:v3.1.0
|
||||||
|
args:
|
||||||
|
- "--v=5"
|
||||||
|
- "--csi-address=$(ADDRESS)"
|
||||||
|
- "--leader-election=true"
|
||||||
|
- "--retry-interval-start=500ms"
|
||||||
|
env:
|
||||||
|
- name: ADDRESS
|
||||||
|
value: /csi/csi-provisioner.sock
|
||||||
|
imagePullPolicy: "IfNotPresent"
|
||||||
|
volumeMounts:
|
||||||
|
- name: socket-dir
|
||||||
|
mountPath: /csi
|
||||||
|
- name: csi-resizer
|
||||||
|
image: k8s.gcr.io/sig-storage/csi-resizer:v1.1.0
|
||||||
|
args:
|
||||||
|
- "--csi-address=$(ADDRESS)"
|
||||||
|
- "--v=5"
|
||||||
|
- "--timeout=150s"
|
||||||
|
- "--leader-election"
|
||||||
|
- "--retry-interval-start=500ms"
|
||||||
|
- "--handle-volume-inuse-error=false"
|
||||||
|
env:
|
||||||
|
- name: ADDRESS
|
||||||
|
value: unix:///csi/csi-provisioner.sock
|
||||||
|
imagePullPolicy: "IfNotPresent"
|
||||||
|
volumeMounts:
|
||||||
|
- name: socket-dir
|
||||||
|
mountPath: /csi
|
||||||
|
- name: csi-vitastor
|
||||||
|
securityContext:
|
||||||
|
privileged: true
|
||||||
|
capabilities:
|
||||||
|
add: ["SYS_ADMIN"]
|
||||||
|
image: vitalif/vitastor-csi:v0.6.16
|
||||||
|
args:
|
||||||
|
- "--node=$(NODE_ID)"
|
||||||
|
- "--endpoint=$(CSI_ENDPOINT)"
|
||||||
|
env:
|
||||||
|
- name: NODE_ID
|
||||||
|
valueFrom:
|
||||||
|
fieldRef:
|
||||||
|
fieldPath: spec.nodeName
|
||||||
|
- name: CSI_ENDPOINT
|
||||||
|
value: unix:///csi/csi-provisioner.sock
|
||||||
|
imagePullPolicy: "IfNotPresent"
|
||||||
|
volumeMounts:
|
||||||
|
- name: socket-dir
|
||||||
|
mountPath: /csi
|
||||||
|
- mountPath: /dev
|
||||||
|
name: host-dev
|
||||||
|
- mountPath: /sys
|
||||||
|
name: host-sys
|
||||||
|
- mountPath: /lib/modules
|
||||||
|
name: lib-modules
|
||||||
|
readOnly: true
|
||||||
|
- name: vitastor-config
|
||||||
|
mountPath: /etc/vitastor
|
||||||
|
volumes:
|
||||||
|
- name: host-dev
|
||||||
|
hostPath:
|
||||||
|
path: /dev
|
||||||
|
- name: host-sys
|
||||||
|
hostPath:
|
||||||
|
path: /sys
|
||||||
|
- name: lib-modules
|
||||||
|
hostPath:
|
||||||
|
path: /lib/modules
|
||||||
|
- name: socket-dir
|
||||||
|
emptyDir: {
|
||||||
|
medium: "Memory"
|
||||||
|
}
|
||||||
|
- name: vitastor-config
|
||||||
|
configMap:
|
||||||
|
name: vitastor-config
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
---
|
||||||
|
# if Kubernetes version is less than 1.18 change
|
||||||
|
# apiVersion to storage.k8s.io/v1betav1
|
||||||
|
apiVersion: storage.k8s.io/v1
|
||||||
|
kind: CSIDriver
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: csi.vitastor.io
|
||||||
|
spec:
|
||||||
|
attachRequired: true
|
||||||
|
podInfoOnMount: false
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
---
|
||||||
|
apiVersion: storage.k8s.io/v1
|
||||||
|
kind: StorageClass
|
||||||
|
metadata:
|
||||||
|
namespace: vitastor-system
|
||||||
|
name: vitastor
|
||||||
|
annotations:
|
||||||
|
storageclass.kubernetes.io/is-default-class: "true"
|
||||||
|
provisioner: csi.vitastor.io
|
||||||
|
volumeBindingMode: Immediate
|
||||||
|
parameters:
|
||||||
|
etcdVolumePrefix: ""
|
||||||
|
poolId: "1"
|
||||||
|
# you can choose other configuration file if you have it in the config map
|
||||||
|
#configPath: "/etc/vitastor/vitastor.conf"
|
||||||
|
# you can also specify etcdUrl here, maybe to connect to another Vitastor cluster
|
||||||
|
# multiple etcdUrls may be specified, delimited by comma
|
||||||
|
#etcdUrl: "http://192.168.7.2:2379"
|
||||||
|
#etcdPrefix: "/vitastor"
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
kind: PersistentVolumeClaim
|
||||||
|
metadata:
|
||||||
|
name: test-vitastor-pvc-block
|
||||||
|
spec:
|
||||||
|
storageClassName: vitastor
|
||||||
|
volumeMode: Block
|
||||||
|
accessModes:
|
||||||
|
- ReadWriteMany
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
storage: 10Gi
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
kind: PersistentVolumeClaim
|
||||||
|
metadata:
|
||||||
|
name: test-vitastor-pvc
|
||||||
|
spec:
|
||||||
|
storageClassName: vitastor
|
||||||
|
accessModes:
|
||||||
|
- ReadWriteOnce
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
storage: 10Gi
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
apiVersion: v1
|
||||||
|
kind: Pod
|
||||||
|
metadata:
|
||||||
|
name: vitastor-test-block-pvc
|
||||||
|
namespace: default
|
||||||
|
spec:
|
||||||
|
containers:
|
||||||
|
- name: vitastor-test-block-pvc
|
||||||
|
image: nginx
|
||||||
|
volumeDevices:
|
||||||
|
- name: data
|
||||||
|
devicePath: /dev/xvda
|
||||||
|
volumes:
|
||||||
|
- name: data
|
||||||
|
persistentVolumeClaim:
|
||||||
|
claimName: test-vitastor-pvc-block
|
||||||
|
readOnly: false
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
apiVersion: v1
|
||||||
|
kind: Pod
|
||||||
|
metadata:
|
||||||
|
name: vitastor-test-nginx
|
||||||
|
namespace: default
|
||||||
|
spec:
|
||||||
|
containers:
|
||||||
|
- name: vitastor-test-nginx
|
||||||
|
image: nginx
|
||||||
|
volumeMounts:
|
||||||
|
- mountPath: /usr/share/nginx/html/s3
|
||||||
|
name: data
|
||||||
|
volumes:
|
||||||
|
- name: data
|
||||||
|
persistentVolumeClaim:
|
||||||
|
claimName: test-vitastor-pvc
|
||||||
|
readOnly: false
|
||||||
+35
@@ -0,0 +1,35 @@
|
|||||||
|
module vitastor.io/csi
|
||||||
|
|
||||||
|
go 1.15
|
||||||
|
|
||||||
|
require (
|
||||||
|
github.com/container-storage-interface/spec v1.4.0
|
||||||
|
github.com/coreos/bbolt v0.0.0-00010101000000-000000000000 // indirect
|
||||||
|
github.com/coreos/etcd v3.3.25+incompatible // indirect
|
||||||
|
github.com/coreos/go-semver v0.3.0 // indirect
|
||||||
|
github.com/coreos/go-systemd v0.0.0-20191104093116-d3cd4ed1dbcf // indirect
|
||||||
|
github.com/coreos/pkg v0.0.0-20180928190104-399ea9e2e55f // indirect
|
||||||
|
github.com/dustin/go-humanize v1.0.0 // indirect
|
||||||
|
github.com/golang/glog v0.0.0-20160126235308-23def4e6c14b
|
||||||
|
github.com/gorilla/websocket v1.4.2 // indirect
|
||||||
|
github.com/grpc-ecosystem/go-grpc-middleware v1.3.0 // indirect
|
||||||
|
github.com/grpc-ecosystem/go-grpc-prometheus v1.2.0 // indirect
|
||||||
|
github.com/grpc-ecosystem/grpc-gateway v1.16.0 // indirect
|
||||||
|
github.com/jonboulle/clockwork v0.2.2 // indirect
|
||||||
|
github.com/kubernetes-csi/csi-lib-utils v0.9.1
|
||||||
|
github.com/soheilhy/cmux v0.1.5 // indirect
|
||||||
|
github.com/tmc/grpc-websocket-proxy v0.0.0-20201229170055-e5319fda7802 // indirect
|
||||||
|
github.com/xiang90/probing v0.0.0-20190116061207-43a291ad63a2 // indirect
|
||||||
|
go.etcd.io/bbolt v0.0.0-00010101000000-000000000000 // indirect
|
||||||
|
go.etcd.io/etcd v3.3.25+incompatible
|
||||||
|
golang.org/x/net v0.0.0-20201202161906-c7110b5ffcbb
|
||||||
|
google.golang.org/grpc v1.33.1
|
||||||
|
k8s.io/klog v1.0.0
|
||||||
|
k8s.io/utils v0.0.0-20210305010621-2afb4311ab10
|
||||||
|
)
|
||||||
|
|
||||||
|
replace github.com/coreos/bbolt => go.etcd.io/bbolt v1.3.5
|
||||||
|
|
||||||
|
replace go.etcd.io/bbolt => github.com/coreos/bbolt v1.3.5
|
||||||
|
|
||||||
|
replace google.golang.org/grpc => google.golang.org/grpc v1.25.1
|
||||||
+448
@@ -0,0 +1,448 @@
|
|||||||
|
cloud.google.com/go v0.34.0/go.mod h1:aQUYkXzVsufM+DwF1aE+0xfcU+56JwCaLick0ClmMTw=
|
||||||
|
cloud.google.com/go v0.38.0/go.mod h1:990N+gfupTy94rShfmMCWGDn0LpTmnzTp2qbd1dvSRU=
|
||||||
|
cloud.google.com/go v0.44.1/go.mod h1:iSa0KzasP4Uvy3f1mN/7PiObzGgflwredwwASm/v6AU=
|
||||||
|
cloud.google.com/go v0.44.2/go.mod h1:60680Gw3Yr4ikxnPRS/oxxkBccT6SA1yMk63TGekxKY=
|
||||||
|
cloud.google.com/go v0.45.1/go.mod h1:RpBamKRgapWJb87xiFSdk4g1CME7QZg3uwTez+TSTjc=
|
||||||
|
cloud.google.com/go v0.46.3/go.mod h1:a6bKKbmY7er1mI7TEI4lsAkts/mkhTSZK8w33B4RAg0=
|
||||||
|
cloud.google.com/go v0.51.0/go.mod h1:hWtGJ6gnXH+KgDv+V0zFGDvpi07n3z8ZNj3T1RW0Gcw=
|
||||||
|
cloud.google.com/go/bigquery v1.0.1/go.mod h1:i/xbL2UlR5RvWAURpBYZTtm/cXjCha9lbfbpx4poX+o=
|
||||||
|
cloud.google.com/go/datastore v1.0.0/go.mod h1:LXYbyblFSglQ5pkeyhO+Qmw7ukd3C+pD7TKLgZqpHYE=
|
||||||
|
cloud.google.com/go/pubsub v1.0.1/go.mod h1:R0Gpsv3s54REJCy4fxDixWD93lHJMoZTyQ2kNxGRt3I=
|
||||||
|
cloud.google.com/go/storage v1.0.0/go.mod h1:IhtSnM/ZTZV8YYJWCY8RULGVqBDmpoyjwiyrjsg+URw=
|
||||||
|
dmitri.shuralyov.com/gpu/mtl v0.0.0-20190408044501-666a987793e9/go.mod h1:H6x//7gZCb22OMCxBHrMx7a5I7Hp++hsVxbQ4BYO7hU=
|
||||||
|
github.com/Azure/go-ansiterm v0.0.0-20170929234023-d6e3b3328b78/go.mod h1:LmzpDX56iTiv29bbRTIsUNlaFfuhWRQBWjQdVyAevI8=
|
||||||
|
github.com/Azure/go-autorest/autorest v0.9.0/go.mod h1:xyHB1BMZT0cuDHU7I0+g046+BFDTQ8rEZB0s4Yfa6bI=
|
||||||
|
github.com/Azure/go-autorest/autorest v0.9.6/go.mod h1:/FALq9T/kS7b5J5qsQ+RSTUdAmGFqi0vUdVNNx8q630=
|
||||||
|
github.com/Azure/go-autorest/autorest/adal v0.5.0/go.mod h1:8Z9fGy2MpX0PvDjB1pEgQTmVqjGhiHBW7RJJEciWzS0=
|
||||||
|
github.com/Azure/go-autorest/autorest/adal v0.8.2/go.mod h1:ZjhuQClTqx435SRJ2iMlOxPYt3d2C/T/7TiQCVZSn3Q=
|
||||||
|
github.com/Azure/go-autorest/autorest/date v0.1.0/go.mod h1:plvfp3oPSKwf2DNjlBjWF/7vwR+cUD/ELuzDCXwHUVA=
|
||||||
|
github.com/Azure/go-autorest/autorest/date v0.2.0/go.mod h1:vcORJHLJEh643/Ioh9+vPmf1Ij9AEBM5FuBIXLmIy0g=
|
||||||
|
github.com/Azure/go-autorest/autorest/mocks v0.1.0/go.mod h1:OTyCOPRA2IgIlWxVYxBee2F5Gr4kF2zd2J5cFRaIDN0=
|
||||||
|
github.com/Azure/go-autorest/autorest/mocks v0.2.0/go.mod h1:OTyCOPRA2IgIlWxVYxBee2F5Gr4kF2zd2J5cFRaIDN0=
|
||||||
|
github.com/Azure/go-autorest/autorest/mocks v0.3.0/go.mod h1:a8FDP3DYzQ4RYfVAxAN3SVSiiO77gL2j2ronKKP0syM=
|
||||||
|
github.com/Azure/go-autorest/logger v0.1.0/go.mod h1:oExouG+K6PryycPJfVSxi/koC6LSNgds39diKLz7Vrc=
|
||||||
|
github.com/Azure/go-autorest/tracing v0.5.0/go.mod h1:r/s2XiOKccPW3HrqB+W0TQzfbtp2fGCgRFtBroKn4Dk=
|
||||||
|
github.com/BurntSushi/toml v0.3.1/go.mod h1:xHWCNGjB5oqiDr8zfno3MHue2Ht5sIBksp03qcyfWMU=
|
||||||
|
github.com/BurntSushi/xgb v0.0.0-20160522181843-27f122750802/go.mod h1:IVnqGOEym/WlBOVXweHU+Q+/VP0lqqI8lqeDx9IjBqo=
|
||||||
|
github.com/NYTimes/gziphandler v0.0.0-20170623195520-56545f4a5d46/go.mod h1:3wb06e3pkSAbeQ52E9H9iFoQsEEwGN64994WTCIhntQ=
|
||||||
|
github.com/PuerkitoBio/purell v1.0.0/go.mod h1:c11w/QuzBsJSee3cPx9rAFu61PvFxuPbtSwDGJws/X0=
|
||||||
|
github.com/PuerkitoBio/urlesc v0.0.0-20160726150825-5bd2802263f2/go.mod h1:uGdkoq3SwY9Y+13GIhn11/XLaGBb4BfwItxLd5jeuXE=
|
||||||
|
github.com/alecthomas/template v0.0.0-20160405071501-a0175ee3bccc/go.mod h1:LOuyumcjzFXgccqObfd/Ljyb9UuFJ6TxHnclSeseNhc=
|
||||||
|
github.com/alecthomas/template v0.0.0-20190718012654-fb15b899a751/go.mod h1:LOuyumcjzFXgccqObfd/Ljyb9UuFJ6TxHnclSeseNhc=
|
||||||
|
github.com/alecthomas/units v0.0.0-20151022065526-2efee857e7cf/go.mod h1:ybxpYRFXyAe+OPACYpWeL0wqObRcbAqCMya13uyzqw0=
|
||||||
|
github.com/alecthomas/units v0.0.0-20190717042225-c3de453c63f4/go.mod h1:ybxpYRFXyAe+OPACYpWeL0wqObRcbAqCMya13uyzqw0=
|
||||||
|
github.com/antihax/optional v1.0.0/go.mod h1:uupD/76wgC+ih3iEmQUL+0Ugr19nfwCT1kdvxnR2qWY=
|
||||||
|
github.com/beorn7/perks v0.0.0-20180321164747-3a771d992973/go.mod h1:Dwedo/Wpr24TaqPxmxbtue+5NUziq4I4S80YR8gNf3Q=
|
||||||
|
github.com/beorn7/perks v1.0.0/go.mod h1:KWe93zE9D1o94FZ5RNwFwVgaQK1VOXiVxmqh+CedLV8=
|
||||||
|
github.com/beorn7/perks v1.0.1 h1:VlbKKnNfV8bJzeqoa4cOKqO6bYr3WgKZxO8Z16+hsOM=
|
||||||
|
github.com/beorn7/perks v1.0.1/go.mod h1:G2ZrVWU2WbWT9wwq4/hrbKbnv/1ERSJQ0ibhJ6rlkpw=
|
||||||
|
github.com/blang/semver v3.5.0+incompatible/go.mod h1:kRBLl5iJ+tD4TcOOxsy/0fnwebNt5EWlYSAyrTnjyyk=
|
||||||
|
github.com/census-instrumentation/opencensus-proto v0.2.1/go.mod h1:f6KPmirojxKA12rnyqOA5BBL4O983OfeGPqjHWSTneU=
|
||||||
|
github.com/cespare/xxhash/v2 v2.1.1 h1:6MnRN8NT7+YBpUIWxHtefFZOKTAPgGjpQSxqLNn0+qY=
|
||||||
|
github.com/cespare/xxhash/v2 v2.1.1/go.mod h1:VGX0DQ3Q6kWi7AoAeZDth3/j3BFtOZR5XLFGgcrjCOs=
|
||||||
|
github.com/chzyer/logex v1.1.10/go.mod h1:+Ywpsq7O8HXn0nuIou7OrIPyXbp3wmkHB+jjWRnGsAI=
|
||||||
|
github.com/chzyer/readline v0.0.0-20180603132655-2972be24d48e/go.mod h1:nSuG5e5PlCu98SY8svDHJxuZscDgtXS6KTTbou5AhLI=
|
||||||
|
github.com/chzyer/test v0.0.0-20180213035817-a1ea475d72b1/go.mod h1:Q3SI9o4m/ZMnBNeIyt5eFwwo7qiLfzFZmjNmxjkiQlU=
|
||||||
|
github.com/container-storage-interface/spec v1.2.0/go.mod h1:6URME8mwIBbpVyZV93Ce5St17xBiQJQY67NDsuohiy4=
|
||||||
|
github.com/container-storage-interface/spec v1.4.0 h1:ozAshSKxpJnYUfmkpZCTYyF/4MYeYlhdXbAvPvfGmkg=
|
||||||
|
github.com/container-storage-interface/spec v1.4.0/go.mod h1:6URME8mwIBbpVyZV93Ce5St17xBiQJQY67NDsuohiy4=
|
||||||
|
github.com/coreos/bbolt v1.3.5 h1:XFv7xaq7701j8ZSEzR28VohFYSlyakMyqNMU5FQH6Ac=
|
||||||
|
github.com/coreos/bbolt v1.3.5/go.mod h1:G5EMThwa9y8QZGBClrRx5EY+Yw9kAhnjy3bSjsnlVTQ=
|
||||||
|
github.com/coreos/etcd v3.3.25+incompatible h1:0GQEw6h3YnuOVdtwygkIfJ+Omx0tZ8/QkVyXI4LkbeY=
|
||||||
|
github.com/coreos/etcd v3.3.25+incompatible/go.mod h1:uF7uidLiAD3TWHmW31ZFd/JWoc32PjwdhPthX9715RE=
|
||||||
|
github.com/coreos/go-semver v0.3.0 h1:wkHLiw0WNATZnSG7epLsujiMCgPAc9xhjJ4tgnAxmfM=
|
||||||
|
github.com/coreos/go-semver v0.3.0/go.mod h1:nnelYz7RCh+5ahJtPPxZlU+153eP4D4r3EedlOD2RNk=
|
||||||
|
github.com/coreos/go-systemd v0.0.0-20191104093116-d3cd4ed1dbcf h1:iW4rZ826su+pqaw19uhpSCzhj44qo35pNgKFGqzDKkU=
|
||||||
|
github.com/coreos/go-systemd v0.0.0-20191104093116-d3cd4ed1dbcf/go.mod h1:F5haX7vjVVG0kc13fIWeqUViNPyEJxv/OmvnBo0Yme4=
|
||||||
|
github.com/coreos/pkg v0.0.0-20180928190104-399ea9e2e55f h1:lBNOc5arjvs8E5mO2tbpBpLoyyu8B6e44T7hJy6potg=
|
||||||
|
github.com/coreos/pkg v0.0.0-20180928190104-399ea9e2e55f/go.mod h1:E3G3o1h8I7cfcXa63jLwjI0eiQQMgzzUDFVpN/nH/eA=
|
||||||
|
github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||||
|
github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c=
|
||||||
|
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||||
|
github.com/dgrijalva/jwt-go v3.2.0+incompatible h1:7qlOGliEKZXTDg6OTjfoBKDXWrumCAMpl/TFQ4/5kLM=
|
||||||
|
github.com/dgrijalva/jwt-go v3.2.0+incompatible/go.mod h1:E3ru+11k8xSBh+hMPgOLZmtrrCbhqsmaPHjLKYnJCaQ=
|
||||||
|
github.com/docker/spdystream v0.0.0-20160310174837-449fdfce4d96/go.mod h1:Qh8CwZgvJUkLughtfhJv5dyTYa91l1fOUCrgjqmcifM=
|
||||||
|
github.com/docopt/docopt-go v0.0.0-20180111231733-ee0de3bc6815/go.mod h1:WwZ+bS3ebgob9U8Nd0kOddGdZWjyMGR8Wziv+TBNwSE=
|
||||||
|
github.com/dustin/go-humanize v1.0.0 h1:VSnTsYCnlFHaM2/igO1h6X3HA71jcobQuxemgkq4zYo=
|
||||||
|
github.com/dustin/go-humanize v1.0.0/go.mod h1:HtrtbFcZ19U5GC7JDqmcUSB87Iq5E25KnS6fMYU6eOk=
|
||||||
|
github.com/elazarl/goproxy v0.0.0-20180725130230-947c36da3153/go.mod h1:/Zj4wYkgs4iZTTu3o/KG3Itv/qCCa8VVMlb3i9OVuzc=
|
||||||
|
github.com/emicklei/go-restful v0.0.0-20170410110728-ff4f55a20633/go.mod h1:otzb+WCGbkyDHkqmQmT5YD2WR4BBwUdeQoFo8l/7tVs=
|
||||||
|
github.com/envoyproxy/go-control-plane v0.9.0/go.mod h1:YTl/9mNaCwkRvm6d1a2C3ymFceY/DCBVvsKhRF0iEA4=
|
||||||
|
github.com/envoyproxy/protoc-gen-validate v0.1.0/go.mod h1:iSmxcyjqTsJpI2R4NaDN7+kN2VEUnK/pcBlmesArF7c=
|
||||||
|
github.com/evanphx/json-patch v4.9.0+incompatible/go.mod h1:50XU6AFN0ol/bzJsmQLiYLvXMP4fmwYFNcr97nuDLSk=
|
||||||
|
github.com/fsnotify/fsnotify v1.4.7/go.mod h1:jwhsz4b93w/PPRr/qN1Yymfu8t87LnFCMoQvtojpjFo=
|
||||||
|
github.com/fsnotify/fsnotify v1.4.9/go.mod h1:znqG4EE+3YCdAaPaxE2ZRY/06pZUdp0tY4IgpuI1SZQ=
|
||||||
|
github.com/ghodss/yaml v0.0.0-20150909031657-73d445a93680/go.mod h1:4dBDuWmgqj2HViK6kFavaiC9ZROes6MMH2rRYeMEF04=
|
||||||
|
github.com/ghodss/yaml v1.0.0/go.mod h1:4dBDuWmgqj2HViK6kFavaiC9ZROes6MMH2rRYeMEF04=
|
||||||
|
github.com/go-gl/glfw/v3.3/glfw v0.0.0-20191125211704-12ad95a8df72/go.mod h1:tQ2UAYgL5IevRw8kRxooKSPJfGvJ9fJQFa0TUsXzTg8=
|
||||||
|
github.com/go-kit/kit v0.8.0/go.mod h1:xBxKIO96dXMWWy0MnWVtmwkA9/13aqxPnvrjFYMA2as=
|
||||||
|
github.com/go-kit/kit v0.9.0/go.mod h1:xBxKIO96dXMWWy0MnWVtmwkA9/13aqxPnvrjFYMA2as=
|
||||||
|
github.com/go-logfmt/logfmt v0.3.0/go.mod h1:Qt1PoO58o5twSAckw1HlFXLmHsOX5/0LbT9GBnD5lWE=
|
||||||
|
github.com/go-logfmt/logfmt v0.4.0/go.mod h1:3RMwSq7FuexP4Kalkev3ejPJsZTpXXBr9+V4qmtdjCk=
|
||||||
|
github.com/go-logr/logr v0.1.0/go.mod h1:ixOQHD9gLJUVQQ2ZOR7zLEifBX6tGkNJF4QyIY7sIas=
|
||||||
|
github.com/go-logr/logr v0.2.0 h1:QvGt2nLcHH0WK9orKa+ppBPAxREcH364nPUedEpK0TY=
|
||||||
|
github.com/go-logr/logr v0.2.0/go.mod h1:z6/tIYblkpsD+a4lm/fGIIU9mZ+XfAiaFtq7xTgseGU=
|
||||||
|
github.com/go-openapi/jsonpointer v0.0.0-20160704185906-46af16f9f7b1/go.mod h1:+35s3my2LFTysnkMfxsJBAMHj/DoqoB9knIWoYG/Vk0=
|
||||||
|
github.com/go-openapi/jsonreference v0.0.0-20160704190145-13c6e3589ad9/go.mod h1:W3Z9FmVs9qj+KR4zFKmDPGiLdk1D9Rlm7cyMvf57TTg=
|
||||||
|
github.com/go-openapi/spec v0.0.0-20160808142527-6aced65f8501/go.mod h1:J8+jY1nAiCcj+friV/PDoE1/3eeccG9LYBs0tYvLOWc=
|
||||||
|
github.com/go-openapi/swag v0.0.0-20160704191624-1d0bd113de87/go.mod h1:DXUve3Dpr1UfpPtxFw+EFuQ41HhCWZfha5jSVRG7C7I=
|
||||||
|
github.com/go-stack/stack v1.8.0/go.mod h1:v0f6uXyyMGvRgIKkXu+yp6POWl0qKG85gN/melR3HDY=
|
||||||
|
github.com/gogo/protobuf v1.1.1/go.mod h1:r8qH/GZQm5c6nD/R0oafs1akxWv10x8SbQlK7atdtwQ=
|
||||||
|
github.com/gogo/protobuf v1.3.1 h1:DqDEcV5aeaTmdFBePNpYsp3FlcVH/2ISVVM9Qf8PSls=
|
||||||
|
github.com/gogo/protobuf v1.3.1/go.mod h1:SlYgWuQ5SjCEi6WLHjHCa1yvBfUnHcTbrrZtXPKa29o=
|
||||||
|
github.com/gogo/protobuf v1.3.2 h1:Ov1cvc58UF3b5XjBnZv7+opcTcQFZebYjWzi34vdm4Q=
|
||||||
|
github.com/gogo/protobuf v1.3.2/go.mod h1:P1XiOD3dCwIKUDQYPy72D8LYyHL2YPYrpS2s69NZV8Q=
|
||||||
|
github.com/golang/glog v0.0.0-20160126235308-23def4e6c14b h1:VKtxabqXZkF25pY9ekfRL6a582T4P37/31XEstQ5p58=
|
||||||
|
github.com/golang/glog v0.0.0-20160126235308-23def4e6c14b/go.mod h1:SBH7ygxi8pfUlaOkMMuAQtPIUF8ecWP5IEl/CR7VP2Q=
|
||||||
|
github.com/golang/groupcache v0.0.0-20190702054246-869f871628b6/go.mod h1:cIg4eruTrX1D+g88fzRXU5OdNfaM+9IcxsU14FzY7Hc=
|
||||||
|
github.com/golang/groupcache v0.0.0-20191227052852-215e87163ea7 h1:5ZkaAPbicIKTF2I64qf5Fh8Aa83Q/dnOafMYV0OMwjA=
|
||||||
|
github.com/golang/groupcache v0.0.0-20191227052852-215e87163ea7/go.mod h1:cIg4eruTrX1D+g88fzRXU5OdNfaM+9IcxsU14FzY7Hc=
|
||||||
|
github.com/golang/mock v1.1.1/go.mod h1:oTYuIxOrZwtPieC+H1uAHpcLFnEyAGVDL/k47Jfbm0A=
|
||||||
|
github.com/golang/mock v1.2.0/go.mod h1:oTYuIxOrZwtPieC+H1uAHpcLFnEyAGVDL/k47Jfbm0A=
|
||||||
|
github.com/golang/mock v1.3.1/go.mod h1:sBzyDLLjw3U8JLTeZvSv8jJB+tU5PVekmnlKIyFUx0Y=
|
||||||
|
github.com/golang/protobuf v1.2.0/go.mod h1:6lQm79b+lXiMfvg/cZm0SGofjICqVBUtrP5yJMmIC1U=
|
||||||
|
github.com/golang/protobuf v1.3.1/go.mod h1:6lQm79b+lXiMfvg/cZm0SGofjICqVBUtrP5yJMmIC1U=
|
||||||
|
github.com/golang/protobuf v1.3.2/go.mod h1:6lQm79b+lXiMfvg/cZm0SGofjICqVBUtrP5yJMmIC1U=
|
||||||
|
github.com/golang/protobuf v1.3.3/go.mod h1:vzj43D7+SQXF/4pzW/hwtAqwc6iTitCiVSaWz5lYuqw=
|
||||||
|
github.com/golang/protobuf v1.4.0-rc.1/go.mod h1:ceaxUfeHdC40wWswd/P6IGgMaK3YpKi5j83Wpe3EHw8=
|
||||||
|
github.com/golang/protobuf v1.4.0-rc.1.0.20200221234624-67d41d38c208/go.mod h1:xKAWHe0F5eneWXFV3EuXVDTCmh+JuBKY0li0aMyXATA=
|
||||||
|
github.com/golang/protobuf v1.4.0-rc.2/go.mod h1:LlEzMj4AhA7rCAGe4KMBDvJI+AwstrUpVNzEA03Pprs=
|
||||||
|
github.com/golang/protobuf v1.4.0-rc.4.0.20200313231945-b860323f09d0/go.mod h1:WU3c8KckQ9AFe+yFwt9sWVRKCVIyN9cPHBJSNnbL67w=
|
||||||
|
github.com/golang/protobuf v1.4.0/go.mod h1:jodUvKwWbYaEsadDk5Fwe5c77LiNKVO9IDvqG2KuDX0=
|
||||||
|
github.com/golang/protobuf v1.4.1/go.mod h1:U8fpvMrcmy5pZrNK1lt4xCsGvpyWQ/VVv6QDs8UjoX8=
|
||||||
|
github.com/golang/protobuf v1.4.2 h1:+Z5KGCizgyZCbGh1KZqA0fcLLkwbsjIzS4aV2v7wJX0=
|
||||||
|
github.com/golang/protobuf v1.4.2/go.mod h1:oDoupMAO8OvCJWAcko0GGGIgR6R6ocIYbsSw735rRwI=
|
||||||
|
github.com/google/btree v0.0.0-20180813153112-4030bb1f1f0c/go.mod h1:lNA+9X1NB3Zf8V7Ke586lFgjr2dZNuvo3lPJSGZ5JPQ=
|
||||||
|
github.com/google/btree v1.0.0 h1:0udJVsspx3VBr5FwtLhQQtuAsVc79tTq0ocGIPAU6qo=
|
||||||
|
github.com/google/btree v1.0.0/go.mod h1:lNA+9X1NB3Zf8V7Ke586lFgjr2dZNuvo3lPJSGZ5JPQ=
|
||||||
|
github.com/google/go-cmp v0.2.0/go.mod h1:oXzfMopK8JAjlY9xF4vHSVASa0yLyX7SntLO5aqRK0M=
|
||||||
|
github.com/google/go-cmp v0.3.0/go.mod h1:8QqcDgzrUqlUb/G2PQTWiueGozuR1884gddMywk6iLU=
|
||||||
|
github.com/google/go-cmp v0.3.1/go.mod h1:8QqcDgzrUqlUb/G2PQTWiueGozuR1884gddMywk6iLU=
|
||||||
|
github.com/google/go-cmp v0.4.0 h1:xsAVV57WRhGj6kEIi8ReJzQlHHqcBYCElAvkovg3B/4=
|
||||||
|
github.com/google/go-cmp v0.4.0/go.mod h1:v8dTdLbMG2kIc/vJvl+f65V22dbkXbowE6jgT/gNBxE=
|
||||||
|
github.com/google/gofuzz v1.0.0/go.mod h1:dBl0BpW6vV/+mYPU4Po3pmUjxk6FQPldtuIdl/M65Eg=
|
||||||
|
github.com/google/gofuzz v1.1.0/go.mod h1:dBl0BpW6vV/+mYPU4Po3pmUjxk6FQPldtuIdl/M65Eg=
|
||||||
|
github.com/google/martian v2.1.0+incompatible/go.mod h1:9I4somxYTbIHy5NJKHRl3wXiIaQGbYVAs8BPL6v8lEs=
|
||||||
|
github.com/google/pprof v0.0.0-20181206194817-3ea8567a2e57/go.mod h1:zfwlbNMJ+OItoe0UupaVj+oy1omPYYDuagoSzA8v9mc=
|
||||||
|
github.com/google/pprof v0.0.0-20190515194954-54271f7e092f/go.mod h1:zfwlbNMJ+OItoe0UupaVj+oy1omPYYDuagoSzA8v9mc=
|
||||||
|
github.com/google/pprof v0.0.0-20191218002539-d4f498aebedc/go.mod h1:ZgVRPoUq/hfqzAqh7sHMqb3I9Rq5C59dIz2SbBwJ4eM=
|
||||||
|
github.com/google/renameio v0.1.0/go.mod h1:KWCgfxg9yswjAJkECMjeO8J8rahYeXnNhOm40UhjYkI=
|
||||||
|
github.com/google/uuid v1.1.1 h1:Gkbcsh/GbpXz7lPftLA3P6TYMwjCLYm83jiFQZF/3gY=
|
||||||
|
github.com/google/uuid v1.1.1/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo=
|
||||||
|
github.com/googleapis/gax-go/v2 v2.0.4/go.mod h1:0Wqv26UfaUD9n4G6kQubkQ+KchISgw+vpHVxEJEs9eg=
|
||||||
|
github.com/googleapis/gax-go/v2 v2.0.5/go.mod h1:DWXyrwAJ9X0FpwwEdw+IPEYBICEFu5mhpdKc/us6bOk=
|
||||||
|
github.com/googleapis/gnostic v0.4.1/go.mod h1:LRhVm6pbyptWbWbuZ38d1eyptfvIytN3ir6b65WBswg=
|
||||||
|
github.com/gorilla/websocket v1.4.2 h1:+/TMaTYc4QFitKJxsQ7Yye35DkWvkdLcvGKqM+x0Ufc=
|
||||||
|
github.com/gorilla/websocket v1.4.2/go.mod h1:YR8l580nyteQvAITg2hZ9XVh4b55+EU/adAjf1fMHhE=
|
||||||
|
github.com/gregjones/httpcache v0.0.0-20180305231024-9cad4c3443a7/go.mod h1:FecbI9+v66THATjSRHfNgh1IVFe/9kFxbXtjV0ctIMA=
|
||||||
|
github.com/grpc-ecosystem/go-grpc-middleware v1.3.0 h1:+9834+KizmvFV7pXQGSXQTsaWhq2GjuNUt0aUU0YBYw=
|
||||||
|
github.com/grpc-ecosystem/go-grpc-middleware v1.3.0/go.mod h1:z0ButlSOZa5vEBq9m2m2hlwIgKw+rp3sdCBRoJY+30Y=
|
||||||
|
github.com/grpc-ecosystem/go-grpc-prometheus v1.2.0 h1:Ovs26xHkKqVztRpIrF/92BcuyuQ/YW4NSIpoGtfXNho=
|
||||||
|
github.com/grpc-ecosystem/go-grpc-prometheus v1.2.0/go.mod h1:8NvIoxWQoOIhqOTXgfV/d3M/q6VIi02HzZEHgUlZvzk=
|
||||||
|
github.com/grpc-ecosystem/grpc-gateway v1.16.0 h1:gmcG1KaJ57LophUzW0Hy8NmPhnMZb4M0+kPpLofRdBo=
|
||||||
|
github.com/grpc-ecosystem/grpc-gateway v1.16.0/go.mod h1:BDjrQk3hbvj6Nolgz8mAMFbcEtjT1g+wF4CSlocrBnw=
|
||||||
|
github.com/hashicorp/golang-lru v0.5.0/go.mod h1:/m3WP610KZHVQ1SGc6re/UDhFvYD7pJ4Ao+sR/qLZy8=
|
||||||
|
github.com/hashicorp/golang-lru v0.5.1/go.mod h1:/m3WP610KZHVQ1SGc6re/UDhFvYD7pJ4Ao+sR/qLZy8=
|
||||||
|
github.com/hpcloud/tail v1.0.0/go.mod h1:ab1qPbhIpdTxEkNHXyeSf5vhxWSCs/tWer42PpOxQnU=
|
||||||
|
github.com/ianlancetaylor/demangle v0.0.0-20181102032728-5e5cf60278f6/go.mod h1:aSSvb/t6k1mPoxDqO4vJh6VOCGPwU4O0C2/Eqndh1Sc=
|
||||||
|
github.com/imdario/mergo v0.3.5/go.mod h1:2EnlNZ0deacrJVfApfmtdGgDfMuh/nq6Ok1EcJh5FfA=
|
||||||
|
github.com/jonboulle/clockwork v0.2.2 h1:UOGuzwb1PwsrDAObMuhUnj0p5ULPj8V/xJ7Kx9qUBdQ=
|
||||||
|
github.com/jonboulle/clockwork v0.2.2/go.mod h1:Pkfl5aHPm1nk2H9h0bjmnJD/BcgbGXUBGnn1kMkgxc8=
|
||||||
|
github.com/json-iterator/go v1.1.6/go.mod h1:+SdeFBvtyEkXs7REEP0seUULqWtbJapLOCVDaaPEHmU=
|
||||||
|
github.com/json-iterator/go v1.1.10 h1:Kz6Cvnvv2wGdaG/V8yMvfkmNiXq9Ya2KUv4rouJJr68=
|
||||||
|
github.com/json-iterator/go v1.1.10/go.mod h1:KdQUCv79m/52Kvf8AW2vK1V8akMuk1QjK/uOdHXbAo4=
|
||||||
|
github.com/jstemmer/go-junit-report v0.0.0-20190106144839-af01ea7f8024/go.mod h1:6v2b51hI/fHJwM22ozAgKL4VKDeJcHhJFhtBdhmNjmU=
|
||||||
|
github.com/jstemmer/go-junit-report v0.9.1/go.mod h1:Brl9GWCQeLvo8nXZwPNNblvFj/XSXhF0NWZEnDohbsk=
|
||||||
|
github.com/julienschmidt/httprouter v1.2.0/go.mod h1:SYymIcj16QtmaHHD7aYtjjsJG7VTCxuUUipMqKk8s4w=
|
||||||
|
github.com/kisielk/errcheck v1.2.0/go.mod h1:/BMXB+zMLi60iA8Vv6Ksmxu/1UDYcXs4uQLJ+jE2L00=
|
||||||
|
github.com/kisielk/errcheck v1.5.0/go.mod h1:pFxgyoBC7bSaBwPgfKdkLd5X25qrDl4LWUI2bnpBCr8=
|
||||||
|
github.com/kisielk/gotool v1.0.0/go.mod h1:XhKaO+MFFWcvkIS/tQcRk01m1F5IRFswLeQ+oQHNcck=
|
||||||
|
github.com/konsorten/go-windows-terminal-sequences v1.0.1/go.mod h1:T0+1ngSBFLxvqU3pZ+m/2kptfBszLMUkC4ZK/EgS/cQ=
|
||||||
|
github.com/konsorten/go-windows-terminal-sequences v1.0.3 h1:CE8S1cTafDpPvMhIxNJKvHsGVBgn1xWYf1NbHQhywc8=
|
||||||
|
github.com/konsorten/go-windows-terminal-sequences v1.0.3/go.mod h1:T0+1ngSBFLxvqU3pZ+m/2kptfBszLMUkC4ZK/EgS/cQ=
|
||||||
|
github.com/kr/logfmt v0.0.0-20140226030751-b84e30acd515/go.mod h1:+0opPa2QZZtGFBFZlji/RkVcI2GknAs/DXo4wKdlNEc=
|
||||||
|
github.com/kr/pretty v0.1.0/go.mod h1:dAy3ld7l9f0ibDNOQOHHMYYIIbhfbHSm3C4ZsoJORNo=
|
||||||
|
github.com/kr/pretty v0.2.0 h1:s5hAObm+yFO5uHYt5dYjxi2rXrsnmRpJx4OYvIWUaQs=
|
||||||
|
github.com/kr/pretty v0.2.0/go.mod h1:ipq/a2n7PKx3OHsz4KJII5eveXtPO4qwEXGdVfWzfnI=
|
||||||
|
github.com/kr/pty v1.1.1/go.mod h1:pFQYn66WHrOpPYNljwOMqo10TkYh1fy3cYio2l3bCsQ=
|
||||||
|
github.com/kr/text v0.1.0 h1:45sCR5RtlFHMR4UwH9sdQ5TC8v0qDQCHnXt+kaKSTVE=
|
||||||
|
github.com/kr/text v0.1.0/go.mod h1:4Jbv+DJW3UT/LiOwJeYQe1efqtUx/iVham/4vfdArNI=
|
||||||
|
github.com/kubernetes-csi/csi-lib-utils v0.9.1 h1:sGq6ifVujfMSkfTsMZip44Ttv8SDXvsBlFk9GdYl/b8=
|
||||||
|
github.com/kubernetes-csi/csi-lib-utils v0.9.1/go.mod h1:8E2jVUX9j3QgspwHXa6LwyN7IHQDjW9jX3kwoWnSC+M=
|
||||||
|
github.com/mailru/easyjson v0.0.0-20160728113105-d5b7844b561a/go.mod h1:C1wdFJiN94OJF2b5HbByQZoLdCWB1Yqtg26g4irojpc=
|
||||||
|
github.com/matttproud/golang_protobuf_extensions v1.0.1/go.mod h1:D8He9yQNgCq6Z5Ld7szi9bcBfOoFv/3dc6xSMkL2PC0=
|
||||||
|
github.com/matttproud/golang_protobuf_extensions v1.0.2-0.20181231171920-c182affec369 h1:I0XW9+e1XWDxdcEniV4rQAIOPUGDq67JSCiRCgGCZLI=
|
||||||
|
github.com/matttproud/golang_protobuf_extensions v1.0.2-0.20181231171920-c182affec369/go.mod h1:BSXmuO+STAnVfrANrmjBb36TMTDstsz7MSK+HVaYKv4=
|
||||||
|
github.com/moby/term v0.0.0-20200312100748-672ec06f55cd/go.mod h1:DdlQx2hp0Ss5/fLikoLlEeIYiATotOjgB//nb973jeo=
|
||||||
|
github.com/modern-go/concurrent v0.0.0-20180228061459-e0a39a4cb421/go.mod h1:6dJC0mAP4ikYIbvyc7fijjWJddQyLn8Ig3JB5CqoB9Q=
|
||||||
|
github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd h1:TRLaZ9cD/w8PVh93nsPXa1VrQ6jlwL5oN8l14QlcNfg=
|
||||||
|
github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd/go.mod h1:6dJC0mAP4ikYIbvyc7fijjWJddQyLn8Ig3JB5CqoB9Q=
|
||||||
|
github.com/modern-go/reflect2 v0.0.0-20180701023420-4b7aa43c6742/go.mod h1:bx2lNnkwVCuqBIxFjflWJWanXIb3RllmbCylyMrvgv0=
|
||||||
|
github.com/modern-go/reflect2 v1.0.1 h1:9f412s+6RmYXLWZSEzVVgPGK7C2PphHj5RJrvfx9AWI=
|
||||||
|
github.com/modern-go/reflect2 v1.0.1/go.mod h1:bx2lNnkwVCuqBIxFjflWJWanXIb3RllmbCylyMrvgv0=
|
||||||
|
github.com/munnerz/goautoneg v0.0.0-20120707110453-a547fc61f48d/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ=
|
||||||
|
github.com/mwitkow/go-conntrack v0.0.0-20161129095857-cc309e4a2223/go.mod h1:qRWi+5nqEBWmkhHvq77mSJWrCKwh8bxhgT7d/eI7P4U=
|
||||||
|
github.com/mxk/go-flowrate v0.0.0-20140419014527-cca7078d478f/go.mod h1:ZdcZmHo+o7JKHSa8/e818NopupXU1YMK5fe1lsApnBw=
|
||||||
|
github.com/onsi/ginkgo v0.0.0-20170829012221-11459a886d9c/go.mod h1:lLunBs/Ym6LB5Z9jYTR76FiuTmxDTDusOGeTQH+WWjE=
|
||||||
|
github.com/onsi/ginkgo v1.6.0/go.mod h1:lLunBs/Ym6LB5Z9jYTR76FiuTmxDTDusOGeTQH+WWjE=
|
||||||
|
github.com/onsi/ginkgo v1.11.0/go.mod h1:lLunBs/Ym6LB5Z9jYTR76FiuTmxDTDusOGeTQH+WWjE=
|
||||||
|
github.com/onsi/gomega v0.0.0-20170829124025-dcabb60a477c/go.mod h1:C1qb7wdrVGGVU+Z6iS04AVkA3Q65CEZX59MT0QO5uiA=
|
||||||
|
github.com/onsi/gomega v1.7.0/go.mod h1:ex+gbHU/CVuBBDIJjb2X0qEXbFg53c61hWP/1CpauHY=
|
||||||
|
github.com/opentracing/opentracing-go v1.1.0/go.mod h1:UkNAQd3GIcIGf0SeVgPpRdFStlNbqXla1AfSYxPUl2o=
|
||||||
|
github.com/peterbourgon/diskv v2.0.1+incompatible/go.mod h1:uqqh8zWWbv1HBMNONnaR/tNboyR3/BZd58JJSHlUSCU=
|
||||||
|
github.com/pkg/errors v0.8.0/go.mod h1:bwawxfHBFNV+L2hUp1rHADufV3IMtnDRdf1r5NINEl0=
|
||||||
|
github.com/pkg/errors v0.8.1/go.mod h1:bwawxfHBFNV+L2hUp1rHADufV3IMtnDRdf1r5NINEl0=
|
||||||
|
github.com/pkg/errors v0.9.1 h1:FEBLx1zS214owpjy7qsBeixbURkuhQAwrK5UwLGTwt4=
|
||||||
|
github.com/pkg/errors v0.9.1/go.mod h1:bwawxfHBFNV+L2hUp1rHADufV3IMtnDRdf1r5NINEl0=
|
||||||
|
github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM=
|
||||||
|
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
|
||||||
|
github.com/prometheus/client_golang v0.9.1/go.mod h1:7SWBe2y4D6OKWSNQJUaRYU/AaXPKyh/dDVn+NZz0KFw=
|
||||||
|
github.com/prometheus/client_golang v1.0.0/go.mod h1:db9x61etRT2tGnBNRi70OPL5FsnadC4Ky3P0J6CfImo=
|
||||||
|
github.com/prometheus/client_golang v1.7.1 h1:NTGy1Ja9pByO+xAeH/qiWnLrKtr3hJPNjaVUwnjpdpA=
|
||||||
|
github.com/prometheus/client_golang v1.7.1/go.mod h1:PY5Wy2awLA44sXw4AOSfFBetzPP4j5+D6mVACh+pe2M=
|
||||||
|
github.com/prometheus/client_model v0.0.0-20180712105110-5c3871d89910/go.mod h1:MbSGuTsp3dbXC40dX6PRTWyKYBIrTGTE9sqQNg2J8bo=
|
||||||
|
github.com/prometheus/client_model v0.0.0-20190129233127-fd36f4220a90/go.mod h1:xMI15A0UPsDsEKsMN9yxemIoYk6Tm2C1GtYGdfGttqA=
|
||||||
|
github.com/prometheus/client_model v0.0.0-20190812154241-14fe0d1b01d4/go.mod h1:xMI15A0UPsDsEKsMN9yxemIoYk6Tm2C1GtYGdfGttqA=
|
||||||
|
github.com/prometheus/client_model v0.2.0 h1:uq5h0d+GuxiXLJLNABMgp2qUWDPiLvgCzz2dUR+/W/M=
|
||||||
|
github.com/prometheus/client_model v0.2.0/go.mod h1:xMI15A0UPsDsEKsMN9yxemIoYk6Tm2C1GtYGdfGttqA=
|
||||||
|
github.com/prometheus/common v0.4.1/go.mod h1:TNfzLD0ON7rHzMJeJkieUDPYmFC7Snx/y86RQel1bk4=
|
||||||
|
github.com/prometheus/common v0.10.0 h1:RyRA7RzGXQZiW+tGMr7sxa85G1z0yOpM1qq5c8lNawc=
|
||||||
|
github.com/prometheus/common v0.10.0/go.mod h1:Tlit/dnDKsSWFlCLTWaA1cyBgKHSMdTB80sz/V91rCo=
|
||||||
|
github.com/prometheus/procfs v0.0.0-20181005140218-185b4288413d/go.mod h1:c3At6R/oaqEKCNdg8wHV1ftS6bRYblBhIjjI8uT2IGk=
|
||||||
|
github.com/prometheus/procfs v0.0.2/go.mod h1:TjEm7ze935MbeOT/UhFTIMYKhuLP4wbCsTZCD3I8kEA=
|
||||||
|
github.com/prometheus/procfs v0.1.3 h1:F0+tqvhOksq22sc6iCHF5WGlWjdwj92p0udFh1VFBS8=
|
||||||
|
github.com/prometheus/procfs v0.1.3/go.mod h1:lV6e/gmhEcM9IjHGsFOCxxuZ+z1YqCvr4OA4YeYWdaU=
|
||||||
|
github.com/rogpeppe/fastuuid v1.2.0/go.mod h1:jVj6XXZzXRy/MSR5jhDC/2q6DgLz+nrA6LYCDYWNEvQ=
|
||||||
|
github.com/rogpeppe/go-internal v1.3.0/go.mod h1:M8bDsm7K2OlrFYOpmOWEs/qY81heoFRclV5y23lUDJ4=
|
||||||
|
github.com/sirupsen/logrus v1.2.0/go.mod h1:LxeOpSwHxABJmUn/MG1IvRgCAasNZTLOkJPxbbu5VWo=
|
||||||
|
github.com/sirupsen/logrus v1.4.2/go.mod h1:tLMulIdttU9McNUspp0xgXVQah82FyeX6MwdIuYE2rE=
|
||||||
|
github.com/sirupsen/logrus v1.6.0 h1:UBcNElsrwanuuMsnGSlYmtmgbb23qDR5dG+6X6Oo89I=
|
||||||
|
github.com/sirupsen/logrus v1.6.0/go.mod h1:7uNnSEd1DgxDLC74fIahvMZmmYsHGZGEOFrfsX/uA88=
|
||||||
|
github.com/soheilhy/cmux v0.1.5 h1:jjzc5WVemNEDTLwv9tlmemhC73tI08BNOIGwBOo10Js=
|
||||||
|
github.com/soheilhy/cmux v0.1.5/go.mod h1:T7TcVDs9LWfQgPlPsdngu6I6QIoyIFZDDC6sNE1GqG0=
|
||||||
|
github.com/spf13/afero v1.2.2/go.mod h1:9ZxEEn6pIJ8Rxe320qSDBk6AsU0r9pR7Q4OcevTdifk=
|
||||||
|
github.com/spf13/pflag v0.0.0-20170130214245-9ff6c6923cff/go.mod h1:DYY7MBk1bdzusC3SYhjObp+wFpr4gzcvqqNjLnInEg4=
|
||||||
|
github.com/spf13/pflag v1.0.3/go.mod h1:DYY7MBk1bdzusC3SYhjObp+wFpr4gzcvqqNjLnInEg4=
|
||||||
|
github.com/spf13/pflag v1.0.5/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg=
|
||||||
|
github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME=
|
||||||
|
github.com/stretchr/objx v0.1.1/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME=
|
||||||
|
github.com/stretchr/testify v1.2.2/go.mod h1:a8OnRcib4nhh0OaRAV+Yts87kKdq0PP7pXfy6kDkUVs=
|
||||||
|
github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI=
|
||||||
|
github.com/stretchr/testify v1.4.0/go.mod h1:j7eGeouHqKxXV5pUuKE4zz7dFj8WfuZ+81PSLYec5m4=
|
||||||
|
github.com/stretchr/testify v1.5.1 h1:nOGnQDM7FYENwehXlg/kFVnos3rEvtKTjRvOWSzb6H4=
|
||||||
|
github.com/stretchr/testify v1.5.1/go.mod h1:5W2xD1RspED5o8YsWQXVCued0rvSQ+mT+I5cxcmMvtA=
|
||||||
|
github.com/tmc/grpc-websocket-proxy v0.0.0-20201229170055-e5319fda7802 h1:uruHq4dN7GR16kFc5fp3d1RIYzJW5onx8Ybykw2YQFA=
|
||||||
|
github.com/tmc/grpc-websocket-proxy v0.0.0-20201229170055-e5319fda7802/go.mod h1:ncp9v5uamzpCO7NfCPTXjqaC+bZgJeR0sMTm6dMHP7U=
|
||||||
|
github.com/xiang90/probing v0.0.0-20190116061207-43a291ad63a2 h1:eY9dn8+vbi4tKz5Qo6v2eYzo7kUS51QINcR5jNpbZS8=
|
||||||
|
github.com/xiang90/probing v0.0.0-20190116061207-43a291ad63a2/go.mod h1:UETIi67q53MR2AWcXfiuqkDkRtnGDLqkBTpCHuJHxtU=
|
||||||
|
github.com/yuin/goldmark v1.1.27/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74=
|
||||||
|
github.com/yuin/goldmark v1.2.1/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74=
|
||||||
|
go.etcd.io/bbolt v1.3.5 h1:XAzx9gjCb0Rxj7EoqcClPD1d5ZBxZJk0jbuoPHenBt0=
|
||||||
|
go.etcd.io/bbolt v1.3.5/go.mod h1:G5EMThwa9y8QZGBClrRx5EY+Yw9kAhnjy3bSjsnlVTQ=
|
||||||
|
go.etcd.io/etcd v3.3.25+incompatible h1:V1RzkZJj9LqsJRy+TUBgpWSbZXITLB819lstuTFoZOY=
|
||||||
|
go.etcd.io/etcd v3.3.25+incompatible/go.mod h1:yaeTdrJi5lOmYerz05bd8+V7KubZs8YSFZfzsF9A6aI=
|
||||||
|
go.opencensus.io v0.21.0/go.mod h1:mSImk1erAIZhrmZN+AvHh14ztQfjbGwt4TtuofqLduU=
|
||||||
|
go.opencensus.io v0.22.0/go.mod h1:+kGneAE2xo2IficOXnaByMWTGM9T73dGwxeWcUqIpI8=
|
||||||
|
go.opencensus.io v0.22.2/go.mod h1:yxeiOL68Rb0Xd1ddK5vPZ/oVn4vY4Ynel7k9FzqtOIw=
|
||||||
|
go.uber.org/atomic v1.4.0 h1:cxzIVoETapQEqDhQu3QfnvXAV4AlzcvUCxkVUFw3+EU=
|
||||||
|
go.uber.org/atomic v1.4.0/go.mod h1:gD2HeocX3+yG+ygLZcrzQJaqmWj9AIm7n08wl/qW/PE=
|
||||||
|
go.uber.org/multierr v1.1.0 h1:HoEmRHQPVSqub6w2z2d2EOVs2fjyFRGyofhKuyDq0QI=
|
||||||
|
go.uber.org/multierr v1.1.0/go.mod h1:wR5kodmAFQ0UK8QlbwjlSNy0Z68gJhDJUG5sjR94q/0=
|
||||||
|
go.uber.org/zap v1.10.0 h1:ORx85nbTijNz8ljznvCMR1ZBIPKFn3jQrag10X2AsuM=
|
||||||
|
go.uber.org/zap v1.10.0/go.mod h1:vwi/ZaCAaUcBkycHslxD9B2zi4UTXhF60s6SWpuDF0Q=
|
||||||
|
golang.org/x/crypto v0.0.0-20180904163835-0709b304e793/go.mod h1:6SG95UA2DQfeDnfUPMdvaQW0Q7yPrPDi9nlGo2tz2b4=
|
||||||
|
golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACkg1iLfiJU5Ep61QUkGW8qpdssI0+w=
|
||||||
|
golang.org/x/crypto v0.0.0-20190510104115-cbcb75029529/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI=
|
||||||
|
golang.org/x/crypto v0.0.0-20190605123033-f99c8df09eb5/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI=
|
||||||
|
golang.org/x/crypto v0.0.0-20191011191535-87dc89f01550/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI=
|
||||||
|
golang.org/x/crypto v0.0.0-20191206172530-e9b2fee46413/go.mod h1:LzIPMQfyMNhhGPhUkYOs5KpL4U8rLKemX1yGLhDgUto=
|
||||||
|
golang.org/x/crypto v0.0.0-20200622213623-75b288015ac9 h1:psW17arqaxU48Z5kZ0CQnkZWQJsqcURM6tKiBApRjXI=
|
||||||
|
golang.org/x/crypto v0.0.0-20200622213623-75b288015ac9/go.mod h1:LzIPMQfyMNhhGPhUkYOs5KpL4U8rLKemX1yGLhDgUto=
|
||||||
|
golang.org/x/exp v0.0.0-20190121172915-509febef88a4/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||||
|
golang.org/x/exp v0.0.0-20190306152737-a1d7652674e8/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||||
|
golang.org/x/exp v0.0.0-20190510132918-efd6b22b2522/go.mod h1:ZjyILWgesfNpC6sMxTJOJm9Kp84zZh5NQWvqDGG3Qr8=
|
||||||
|
golang.org/x/exp v0.0.0-20190829153037-c13cbed26979/go.mod h1:86+5VVa7VpoJ4kLfm080zCjGlMRFzhUhsZKEZO7MGek=
|
||||||
|
golang.org/x/exp v0.0.0-20191227195350-da58074b4299/go.mod h1:2RIsYlXP63K8oxa1u096TMicItID8zy7Y6sNkU49FU4=
|
||||||
|
golang.org/x/image v0.0.0-20190227222117-0694c2d4d067/go.mod h1:kZ7UVZpmo3dzQBMxlp+ypCbDeSB+sBbTgSJuh5dn5js=
|
||||||
|
golang.org/x/image v0.0.0-20190802002840-cff245a6509b/go.mod h1:FeLwcggjj3mMvU+oOTbSwawSJRM1uh48EjtB4UJZlP0=
|
||||||
|
golang.org/x/lint v0.0.0-20190227174305-5b3e6a55c961/go.mod h1:wehouNa3lNwaWXcvxsM5YxQ5yQlVC4a0KAMCusXpPoU=
|
||||||
|
golang.org/x/lint v0.0.0-20190301231843-5614ed5bae6f/go.mod h1:UVdnD1Gm6xHRNCYTkRU2/jEulfH38KcIWyp/GAMgvoE=
|
||||||
|
golang.org/x/lint v0.0.0-20190313153728-d0100b6bd8b3/go.mod h1:6SW0HCj/g11FgYtHlgUYUwCkIfeOF89ocIRzGO/8vkc=
|
||||||
|
golang.org/x/lint v0.0.0-20190409202823-959b441ac422/go.mod h1:6SW0HCj/g11FgYtHlgUYUwCkIfeOF89ocIRzGO/8vkc=
|
||||||
|
golang.org/x/lint v0.0.0-20190909230951-414d861bb4ac/go.mod h1:6SW0HCj/g11FgYtHlgUYUwCkIfeOF89ocIRzGO/8vkc=
|
||||||
|
golang.org/x/lint v0.0.0-20191125180803-fdd1cda4f05f/go.mod h1:5qLYkcX4OjUUV8bRuDixDT3tpyyb+LUpUlRWLxfhWrs=
|
||||||
|
golang.org/x/mobile v0.0.0-20190312151609-d3739f865fa6/go.mod h1:z+o9i4GpDbdi3rU15maQ/Ox0txvL9dWGYEHz965HBQE=
|
||||||
|
golang.org/x/mobile v0.0.0-20190719004257-d2bd2a29d028/go.mod h1:E/iHnbuqvinMTCcRqshq8CkpyQDoeVncDDYHnLhea+o=
|
||||||
|
golang.org/x/mod v0.0.0-20190513183733-4bf6d317e70e/go.mod h1:mXi4GBBbnImb6dmsKGUJ2LatrhH/nqhxcFungHvyanc=
|
||||||
|
golang.org/x/mod v0.1.0/go.mod h1:0QHyrYULN0/3qlju5TqG8bIK38QM8yzMo5ekMj3DlcY=
|
||||||
|
golang.org/x/mod v0.1.1-0.20191105210325-c90efee705ee/go.mod h1:QqPTAvyqsEbceGzBzNggFXnrqF1CaUcvgkdR5Ot7KZg=
|
||||||
|
golang.org/x/mod v0.2.0/go.mod h1:s0Qsj1ACt9ePp/hMypM3fl4fZqREWJwdYDEqhRiZZUA=
|
||||||
|
golang.org/x/mod v0.3.0/go.mod h1:s0Qsj1ACt9ePp/hMypM3fl4fZqREWJwdYDEqhRiZZUA=
|
||||||
|
golang.org/x/net v0.0.0-20180724234803-3673e40ba225/go.mod h1:mL1N/T3taQHkDXs73rZJwtUhF3w3ftmwwsq0BUmARs4=
|
||||||
|
golang.org/x/net v0.0.0-20180906233101-161cd47e91fd/go.mod h1:mL1N/T3taQHkDXs73rZJwtUhF3w3ftmwwsq0BUmARs4=
|
||||||
|
golang.org/x/net v0.0.0-20181114220301-adae6a3d119a/go.mod h1:mL1N/T3taQHkDXs73rZJwtUhF3w3ftmwwsq0BUmARs4=
|
||||||
|
golang.org/x/net v0.0.0-20190108225652-1e06a53dbb7e/go.mod h1:mL1N/T3taQHkDXs73rZJwtUhF3w3ftmwwsq0BUmARs4=
|
||||||
|
golang.org/x/net v0.0.0-20190213061140-3a22650c66bd/go.mod h1:mL1N/T3taQHkDXs73rZJwtUhF3w3ftmwwsq0BUmARs4=
|
||||||
|
golang.org/x/net v0.0.0-20190311183353-d8887717615a/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg=
|
||||||
|
golang.org/x/net v0.0.0-20190404232315-eb5bcb51f2a3/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg=
|
||||||
|
golang.org/x/net v0.0.0-20190501004415-9ce7a6920f09/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg=
|
||||||
|
golang.org/x/net v0.0.0-20190503192946-f4e77d36d62c/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg=
|
||||||
|
golang.org/x/net v0.0.0-20190603091049-60506f45cf65/go.mod h1:HSz+uSET+XFnRR8LxR5pz3Of3rY3CfYBVs4xY44aLks=
|
||||||
|
golang.org/x/net v0.0.0-20190613194153-d28f0bde5980/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s=
|
||||||
|
golang.org/x/net v0.0.0-20190620200207-3b0461eec859/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s=
|
||||||
|
golang.org/x/net v0.0.0-20191209160850-c0dbc17a3553/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s=
|
||||||
|
golang.org/x/net v0.0.0-20200226121028-0de0cce0169b/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s=
|
||||||
|
golang.org/x/net v0.0.0-20200324143707-d3edc9973b7e/go.mod h1:qpuaurCH72eLCgpAm/N6yyVIVM9cpaDIP3A8BGJEC5A=
|
||||||
|
golang.org/x/net v0.0.0-20200707034311-ab3426394381 h1:VXak5I6aEWmAXeQjA+QSZzlgNrpq9mjcfDemuexIKsU=
|
||||||
|
golang.org/x/net v0.0.0-20200707034311-ab3426394381/go.mod h1:/O7V0waA8r7cgGh81Ro3o1hOxt32SMVPicZroKQ2sZA=
|
||||||
|
golang.org/x/net v0.0.0-20200822124328-c89045814202/go.mod h1:/O7V0waA8r7cgGh81Ro3o1hOxt32SMVPicZroKQ2sZA=
|
||||||
|
golang.org/x/net v0.0.0-20201021035429-f5854403a974/go.mod h1:sp8m0HH+o8qH0wwXwYZr8TS3Oi6o0r6Gce1SSxlDquU=
|
||||||
|
golang.org/x/net v0.0.0-20201202161906-c7110b5ffcbb h1:eBmm0M9fYhWpKZLjQUUKka/LtIxf46G4fxeEz5KJr9U=
|
||||||
|
golang.org/x/net v0.0.0-20201202161906-c7110b5ffcbb/go.mod h1:sp8m0HH+o8qH0wwXwYZr8TS3Oi6o0r6Gce1SSxlDquU=
|
||||||
|
golang.org/x/oauth2 v0.0.0-20180821212333-d2e6202438be/go.mod h1:N/0e6XlmueqKjAGxoOufVs8QHGRruUQn6yWY3a++T0U=
|
||||||
|
golang.org/x/oauth2 v0.0.0-20190226205417-e64efc72b421/go.mod h1:gOpvHmFTYa4IltrdGE7lF6nIHvwfUNPOp7c8zoXwtLw=
|
||||||
|
golang.org/x/oauth2 v0.0.0-20190604053449-0f29369cfe45/go.mod h1:gOpvHmFTYa4IltrdGE7lF6nIHvwfUNPOp7c8zoXwtLw=
|
||||||
|
golang.org/x/oauth2 v0.0.0-20191202225959-858c2ad4c8b6/go.mod h1:gOpvHmFTYa4IltrdGE7lF6nIHvwfUNPOp7c8zoXwtLw=
|
||||||
|
golang.org/x/oauth2 v0.0.0-20200107190931-bf48bf16ab8d/go.mod h1:gOpvHmFTYa4IltrdGE7lF6nIHvwfUNPOp7c8zoXwtLw=
|
||||||
|
golang.org/x/sync v0.0.0-20180314180146-1d60e4601c6f/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||||
|
golang.org/x/sync v0.0.0-20181108010431-42b317875d0f/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||||
|
golang.org/x/sync v0.0.0-20181221193216-37e7f081c4d4/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||||
|
golang.org/x/sync v0.0.0-20190227155943-e225da77a7e6/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||||
|
golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||||
|
golang.org/x/sync v0.0.0-20190911185100-cd5d95a43a6e/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||||
|
golang.org/x/sync v0.0.0-20201020160332-67f06af15bc9/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||||
|
golang.org/x/sys v0.0.0-20180905080454-ebe1bf3edb33/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
|
||||||
|
golang.org/x/sys v0.0.0-20180909124046-d0be0721c37e/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
|
||||||
|
golang.org/x/sys v0.0.0-20181116152217-5ac8a444bdc5/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
|
||||||
|
golang.org/x/sys v0.0.0-20190215142949-d0b11bdaac8a/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
|
||||||
|
golang.org/x/sys v0.0.0-20190312061237-fead79001313/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20190412213103-97732733099d/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20190422165155-953cdadca894/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20190502145724-3ef323f4f1fd/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20190507160741-ecd444e8653b/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20190606165138-5da285871e9c/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20190624142023-c5567b49c5d0/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20191005200804-aed5e4c7ecf9/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20191204072324-ce4227a45e2e/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20191228213918-04cbcbbfeed8/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20200106162015-b016eb3dc98e/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20200202164722-d101bd2416d5/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20200302150141-5c8b2ff67527/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20200323222414-85ca7c5b95cd/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20200615200032-f1bc736245b1/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20200622214017-ed371f2e16b4 h1:5/PjkGUjvEU5Gl6BxmvKRPpqo2uNMv4rcHBMwzk/st8=
|
||||||
|
golang.org/x/sys v0.0.0-20200622214017-ed371f2e16b4/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/sys v0.0.0-20200930185726-fdedc70b468f h1:+Nyd8tzPX9R7BWHguqsrbFdRx3WQ/1ib8I44HXV5yTA=
|
||||||
|
golang.org/x/sys v0.0.0-20200930185726-fdedc70b468f/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||||
|
golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
|
||||||
|
golang.org/x/text v0.3.1-0.20180807135948-17ff2d5776d2/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
|
||||||
|
golang.org/x/text v0.3.2/go.mod h1:bEr9sfX3Q8Zfm5fL9x+3itogRgK3+ptLWKqgva+5dAk=
|
||||||
|
golang.org/x/text v0.3.3 h1:cokOdA+Jmi5PJGXLlLllQSgYigAEfHXJAERHVMaCc2k=
|
||||||
|
golang.org/x/text v0.3.3/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ=
|
||||||
|
golang.org/x/time v0.0.0-20181108054448-85acf8d2951c/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||||
|
golang.org/x/time v0.0.0-20190308202827-9d24e82272b4/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||||
|
golang.org/x/time v0.0.0-20191024005414-555d28b269f0 h1:/5xXl8Y5W96D+TtHSlonuFqGHIWVuyCkGJLwGh9JJFs=
|
||||||
|
golang.org/x/time v0.0.0-20191024005414-555d28b269f0/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||||
|
golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
|
||||||
|
golang.org/x/tools v0.0.0-20181011042414-1f849cf54d09/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
|
||||||
|
golang.org/x/tools v0.0.0-20181030221726-6c7e314b6563/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
|
||||||
|
golang.org/x/tools v0.0.0-20190226205152-f727befe758c/go.mod h1:9Yl7xja0Znq3iFh3HoIrodX9oNMXvdceNzlUR8zjMvY=
|
||||||
|
golang.org/x/tools v0.0.0-20190311212946-11955173bddd/go.mod h1:LCzVGOaR6xXOjkQ3onu1FJEFr0SW1gC7cKk1uF8kGRs=
|
||||||
|
golang.org/x/tools v0.0.0-20190312151545-0bb0c0a6e846/go.mod h1:LCzVGOaR6xXOjkQ3onu1FJEFr0SW1gC7cKk1uF8kGRs=
|
||||||
|
golang.org/x/tools v0.0.0-20190312170243-e65039ee4138/go.mod h1:LCzVGOaR6xXOjkQ3onu1FJEFr0SW1gC7cKk1uF8kGRs=
|
||||||
|
golang.org/x/tools v0.0.0-20190425150028-36563e24a262/go.mod h1:RgjU9mgBXZiqYHBnxXauZ1Gv1EHHAz9KjViQ78xBX0Q=
|
||||||
|
golang.org/x/tools v0.0.0-20190506145303-2d16b83fe98c/go.mod h1:RgjU9mgBXZiqYHBnxXauZ1Gv1EHHAz9KjViQ78xBX0Q=
|
||||||
|
golang.org/x/tools v0.0.0-20190524140312-2c0ae7006135/go.mod h1:RgjU9mgBXZiqYHBnxXauZ1Gv1EHHAz9KjViQ78xBX0Q=
|
||||||
|
golang.org/x/tools v0.0.0-20190606124116-d0a3d012864b/go.mod h1:/rFqwRUd4F7ZHNgwSSTFct+R/Kf4OFW1sUzUTQQTgfc=
|
||||||
|
golang.org/x/tools v0.0.0-20190621195816-6e04913cbbac/go.mod h1:/rFqwRUd4F7ZHNgwSSTFct+R/Kf4OFW1sUzUTQQTgfc=
|
||||||
|
golang.org/x/tools v0.0.0-20190624222133-a101b041ded4/go.mod h1:/rFqwRUd4F7ZHNgwSSTFct+R/Kf4OFW1sUzUTQQTgfc=
|
||||||
|
golang.org/x/tools v0.0.0-20190628153133-6cdbf07be9d0/go.mod h1:/rFqwRUd4F7ZHNgwSSTFct+R/Kf4OFW1sUzUTQQTgfc=
|
||||||
|
golang.org/x/tools v0.0.0-20190816200558-6889da9d5479/go.mod h1:b+2E5dAYhXwXZwtnZ6UAqBI28+e2cm9otk0dWdXHAEo=
|
||||||
|
golang.org/x/tools v0.0.0-20190911174233-4f2ddba30aff/go.mod h1:b+2E5dAYhXwXZwtnZ6UAqBI28+e2cm9otk0dWdXHAEo=
|
||||||
|
golang.org/x/tools v0.0.0-20191012152004-8de300cfc20a/go.mod h1:b+2E5dAYhXwXZwtnZ6UAqBI28+e2cm9otk0dWdXHAEo=
|
||||||
|
golang.org/x/tools v0.0.0-20191119224855-298f0cb1881e/go.mod h1:b+2E5dAYhXwXZwtnZ6UAqBI28+e2cm9otk0dWdXHAEo=
|
||||||
|
golang.org/x/tools v0.0.0-20191125144606-a911d9008d1f/go.mod h1:b+2E5dAYhXwXZwtnZ6UAqBI28+e2cm9otk0dWdXHAEo=
|
||||||
|
golang.org/x/tools v0.0.0-20191227053925-7b8e75db28f4/go.mod h1:TB2adYChydJhpapKDTa4BR/hXlZSLoq2Wpct/0txZ28=
|
||||||
|
golang.org/x/tools v0.0.0-20200619180055-7c47624df98f/go.mod h1:EkVYQZoAsY45+roYkvgYkIh4xh/qjgUK9TdY2XT94GE=
|
||||||
|
golang.org/x/tools v0.0.0-20210106214847-113979e3529a/go.mod h1:emZCQorbCU4vsT4fOWvOPXz4eW1wZW4PmDk9uLelYpA=
|
||||||
|
golang.org/x/xerrors v0.0.0-20190717185122-a985d3407aa7/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
|
||||||
|
golang.org/x/xerrors v0.0.0-20191011141410-1b5146add898/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
|
||||||
|
golang.org/x/xerrors v0.0.0-20191204190536-9bdfabe68543 h1:E7g+9GITq07hpfrRu66IVDexMakfv52eLZ2CXBWiKr4=
|
||||||
|
golang.org/x/xerrors v0.0.0-20191204190536-9bdfabe68543/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
|
||||||
|
golang.org/x/xerrors v0.0.0-20200804184101-5ec99f83aff1 h1:go1bK/D/BFZV2I8cIQd1NKEZ+0owSTG1fDTci4IqFcE=
|
||||||
|
golang.org/x/xerrors v0.0.0-20200804184101-5ec99f83aff1/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
|
||||||
|
google.golang.org/api v0.4.0/go.mod h1:8k5glujaEP+g9n7WNsDg8QP6cUVNI86fCNMcbazEtwE=
|
||||||
|
google.golang.org/api v0.7.0/go.mod h1:WtwebWUNSVBH/HAw79HIFXZNqEvBhG+Ra+ax0hx3E3M=
|
||||||
|
google.golang.org/api v0.8.0/go.mod h1:o4eAsZoiT+ibD93RtjEohWalFOjRDx6CVaqeizhEnKg=
|
||||||
|
google.golang.org/api v0.9.0/go.mod h1:o4eAsZoiT+ibD93RtjEohWalFOjRDx6CVaqeizhEnKg=
|
||||||
|
google.golang.org/api v0.15.0/go.mod h1:iLdEw5Ide6rF15KTC1Kkl0iskquN2gFfn9o9XIsbkAI=
|
||||||
|
google.golang.org/appengine v1.4.0/go.mod h1:xpcJRLb0r/rnEns0DIKYYv+WjYCduHsrkT7/EB5XEv4=
|
||||||
|
google.golang.org/appengine v1.5.0/go.mod h1:xpcJRLb0r/rnEns0DIKYYv+WjYCduHsrkT7/EB5XEv4=
|
||||||
|
google.golang.org/appengine v1.6.1/go.mod h1:i06prIuMbXzDqacNJfV5OdTW448YApPu5ww/cMBSeb0=
|
||||||
|
google.golang.org/appengine v1.6.5/go.mod h1:8WjMMxjGQR8xUklV/ARdw2HLXBOI7O7uCIDZVag1xfc=
|
||||||
|
google.golang.org/genproto v0.0.0-20190307195333-5fe7a883aa19/go.mod h1:VzzqZJRnGkLBvHegQrXjBqPurQTc5/KpmUdxsrq26oE=
|
||||||
|
google.golang.org/genproto v0.0.0-20190418145605-e7d98fc518a7/go.mod h1:VzzqZJRnGkLBvHegQrXjBqPurQTc5/KpmUdxsrq26oE=
|
||||||
|
google.golang.org/genproto v0.0.0-20190425155659-357c62f0e4bb/go.mod h1:VzzqZJRnGkLBvHegQrXjBqPurQTc5/KpmUdxsrq26oE=
|
||||||
|
google.golang.org/genproto v0.0.0-20190502173448-54afdca5d873/go.mod h1:VzzqZJRnGkLBvHegQrXjBqPurQTc5/KpmUdxsrq26oE=
|
||||||
|
google.golang.org/genproto v0.0.0-20190801165951-fa694d86fc64/go.mod h1:DMBHOl98Agz4BDEuKkezgsaosCRResVns1a3J2ZsMNc=
|
||||||
|
google.golang.org/genproto v0.0.0-20190819201941-24fa4b261c55/go.mod h1:DMBHOl98Agz4BDEuKkezgsaosCRResVns1a3J2ZsMNc=
|
||||||
|
google.golang.org/genproto v0.0.0-20190911173649-1774047e7e51/go.mod h1:IbNlFCBrqXvoKpeg0TB2l7cyZUmoaFKYIwrEpbDKLA8=
|
||||||
|
google.golang.org/genproto v0.0.0-20191230161307-f3c370f40bfb/go.mod h1:n3cpQtvxv34hfy77yVDNjmbRyujviMdxYliBSkLhpCc=
|
||||||
|
google.golang.org/genproto v0.0.0-20200423170343-7949de9c1215/go.mod h1:55QSHmfGQM9UVYDPBsyGGes0y52j32PQ3BqQfXhyH3c=
|
||||||
|
google.golang.org/genproto v0.0.0-20200513103714-09dca8ec2884/go.mod h1:55QSHmfGQM9UVYDPBsyGGes0y52j32PQ3BqQfXhyH3c=
|
||||||
|
google.golang.org/genproto v0.0.0-20200526211855-cb27e3aa2013 h1:+kGHl1aib/qcwaRi1CbqBZ1rk19r85MNUf8HaBghugY=
|
||||||
|
google.golang.org/genproto v0.0.0-20200526211855-cb27e3aa2013/go.mod h1:NbSheEEYHJ7i3ixzK3sjbqSGDJWnxyFXZblF3eUsNvo=
|
||||||
|
google.golang.org/grpc v1.25.1 h1:wdKvqQk7IttEw92GoRyKG2IDrUIpgpj6H6m81yfeMW0=
|
||||||
|
google.golang.org/grpc v1.25.1/go.mod h1:c3i+UQWmh7LiEpx4sFZnkU36qjEYZ0imhYfXVyQciAY=
|
||||||
|
google.golang.org/protobuf v0.0.0-20200109180630-ec00e32a8dfd/go.mod h1:DFci5gLYBciE7Vtevhsrf46CRTquxDuWsQurQQe4oz8=
|
||||||
|
google.golang.org/protobuf v0.0.0-20200221191635-4d8936d0db64/go.mod h1:kwYJMbMJ01Woi6D6+Kah6886xMZcty6N08ah7+eCXa0=
|
||||||
|
google.golang.org/protobuf v0.0.0-20200228230310-ab0ca4ff8a60/go.mod h1:cfTl7dwQJ+fmap5saPgwCLgHXTUD7jkjRqWcaiX5VyM=
|
||||||
|
google.golang.org/protobuf v1.20.1-0.20200309200217-e05f789c0967/go.mod h1:A+miEFZTKqfCUM6K7xSMQL9OKL/b6hQv+e19PK+JZNE=
|
||||||
|
google.golang.org/protobuf v1.21.0/go.mod h1:47Nbq4nVaFHyn7ilMalzfO3qCViNmqZ2kzikPIcrTAo=
|
||||||
|
google.golang.org/protobuf v1.22.0/go.mod h1:EGpADcykh3NcUnDUJcl1+ZksZNG86OlYog2l/sGQquU=
|
||||||
|
google.golang.org/protobuf v1.23.0/go.mod h1:EGpADcykh3NcUnDUJcl1+ZksZNG86OlYog2l/sGQquU=
|
||||||
|
google.golang.org/protobuf v1.23.1-0.20200526195155-81db48ad09cc/go.mod h1:EGpADcykh3NcUnDUJcl1+ZksZNG86OlYog2l/sGQquU=
|
||||||
|
google.golang.org/protobuf v1.24.0 h1:UhZDfRO8JRQru4/+LlLE0BRKGF8L+PICnvYZmx/fEGA=
|
||||||
|
google.golang.org/protobuf v1.24.0/go.mod h1:r/3tXBNzIEhYS9I1OUVjXDlt8tc493IdKGjtUeSXeh4=
|
||||||
|
gopkg.in/alecthomas/kingpin.v2 v2.2.6/go.mod h1:FMv+mEhP44yOT+4EoQTLFTRgOQ1FBLkstjWtayDeSgw=
|
||||||
|
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||||
|
gopkg.in/check.v1 v1.0.0-20180628173108-788fd7840127/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||||
|
gopkg.in/check.v1 v1.0.0-20190902080502-41f04d3bba15 h1:YR8cESwS4TdDjEe65xsg0ogRM/Nc3DYOhEAlW+xobZo=
|
||||||
|
gopkg.in/check.v1 v1.0.0-20190902080502-41f04d3bba15/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||||
|
gopkg.in/errgo.v2 v2.1.0/go.mod h1:hNsd1EY+bozCKY1Ytp96fpM3vjJbqLJn88ws8XvfDNI=
|
||||||
|
gopkg.in/fsnotify.v1 v1.4.7/go.mod h1:Tz8NjZHkW78fSQdbUxIjBTcgA1z1m8ZHf0WmKUhAMys=
|
||||||
|
gopkg.in/inf.v0 v0.9.1/go.mod h1:cWUDdTG/fYaXco+Dcufb5Vnc6Gp2YChqWtbxRZE0mXw=
|
||||||
|
gopkg.in/tomb.v1 v1.0.0-20141024135613-dd632973f1e7/go.mod h1:dt/ZhP58zS4L8KSrWDmTeBkI65Dw0HsyUHuEVlX15mw=
|
||||||
|
gopkg.in/yaml.v2 v2.2.1/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
|
||||||
|
gopkg.in/yaml.v2 v2.2.2/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
|
||||||
|
gopkg.in/yaml.v2 v2.2.3/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
|
||||||
|
gopkg.in/yaml.v2 v2.2.4/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
|
||||||
|
gopkg.in/yaml.v2 v2.2.5/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
|
||||||
|
gopkg.in/yaml.v2 v2.2.8 h1:obN1ZagJSUGI0Ek/LBmuj4SNLPfIny3KsKFopxRdj10=
|
||||||
|
gopkg.in/yaml.v2 v2.2.8/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
|
||||||
|
gotest.tools v2.2.0+incompatible/go.mod h1:DsYFclhRJ6vuDpmuTbkuFWG+y2sxOXAzmJt81HFBacw=
|
||||||
|
gotest.tools/v3 v3.0.2/go.mod h1:3SzNCllyD9/Y+b5r9JIKQ474KzkZyqLqEfYqMsX94Bk=
|
||||||
|
honnef.co/go/tools v0.0.0-20190102054323-c2f93a96b099/go.mod h1:rf3lG4BRIbNafJWhAfAdb/ePZxsR/4RtNHQocxwk9r4=
|
||||||
|
honnef.co/go/tools v0.0.0-20190106161140-3f1c8253044a/go.mod h1:rf3lG4BRIbNafJWhAfAdb/ePZxsR/4RtNHQocxwk9r4=
|
||||||
|
honnef.co/go/tools v0.0.0-20190418001031-e561f6794a2a/go.mod h1:rf3lG4BRIbNafJWhAfAdb/ePZxsR/4RtNHQocxwk9r4=
|
||||||
|
honnef.co/go/tools v0.0.0-20190523083050-ea95bdfd59fc/go.mod h1:rf3lG4BRIbNafJWhAfAdb/ePZxsR/4RtNHQocxwk9r4=
|
||||||
|
honnef.co/go/tools v0.0.1-2019.2.3/go.mod h1:a3bituU0lyd329TUQxRnasdCoJDkEUEAqEt0JzvZhAg=
|
||||||
|
k8s.io/api v0.19.0/go.mod h1:I1K45XlvTrDjmj5LoM5LuP/KYrhWbjUKT/SoPG0qTjw=
|
||||||
|
k8s.io/apimachinery v0.19.0/go.mod h1:DnPGDnARWFvYa3pMHgSxtbZb7gpzzAZ1pTfaUNDVlmA=
|
||||||
|
k8s.io/client-go v0.19.0/go.mod h1:H9E/VT95blcFQnlyShFgnFT9ZnJOAceiUHM3MlRC+mU=
|
||||||
|
k8s.io/component-base v0.19.0/go.mod h1:dKsY8BxkA+9dZIAh2aWJLL/UdASFDNtGYTCItL4LM7Y=
|
||||||
|
k8s.io/gengo v0.0.0-20200413195148-3a45101e95ac/go.mod h1:ezvh/TsK7cY6rbqRK0oQQ8IAqLxYwwyPxAX1Pzy0ii0=
|
||||||
|
k8s.io/klog v1.0.0 h1:Pt+yjF5aB1xDSVbau4VsWe+dQNzA0qv1LlXdC2dF6Q8=
|
||||||
|
k8s.io/klog v1.0.0/go.mod h1:4Bi6QPql/J/LkTDqv7R/cd3hPo4k2DG6Ptcz060Ez5I=
|
||||||
|
k8s.io/klog/v2 v2.0.0/go.mod h1:PBfzABfn139FHAV07az/IF9Wp1bkk3vpT2XSJ76fSDE=
|
||||||
|
k8s.io/klog/v2 v2.2.0 h1:XRvcwJozkgZ1UQJmfMGpvRthQHOvihEhYtDfAaxMz/A=
|
||||||
|
k8s.io/klog/v2 v2.2.0/go.mod h1:Od+F08eJP+W3HUb4pSrPpgp9DGU4GzlpG/TmITuYh/Y=
|
||||||
|
k8s.io/kube-openapi v0.0.0-20200805222855-6aeccd4b50c6/go.mod h1:UuqjUnNftUyPE5H64/qeyjQoUZhGpeFDVdxjTeEVN2o=
|
||||||
|
k8s.io/utils v0.0.0-20200729134348-d5654de09c73/go.mod h1:jPW/WVKK9YHAvNhRxK0md/EJ228hCsBRufyofKtW8HA=
|
||||||
|
k8s.io/utils v0.0.0-20210305010621-2afb4311ab10 h1:u5rPykqiCpL+LBfjRkXvnK71gOgIdmq3eHUEkPrbeTI=
|
||||||
|
k8s.io/utils v0.0.0-20210305010621-2afb4311ab10/go.mod h1:jPW/WVKK9YHAvNhRxK0md/EJ228hCsBRufyofKtW8HA=
|
||||||
|
rsc.io/binaryregexp v0.2.0/go.mod h1:qTv7/COck+e2FymRvadv62gMdZztPaShugOCi3I+8D8=
|
||||||
|
sigs.k8s.io/structured-merge-diff/v4 v4.0.1/go.mod h1:bJZC9H9iH24zzfZ/41RGcq60oK1F7G282QMXDPYydCw=
|
||||||
|
sigs.k8s.io/yaml v1.1.0/go.mod h1:UJmg0vDUVViEyp3mgSv9WPwZCDxu4rQW1olrI1uml+o=
|
||||||
|
sigs.k8s.io/yaml v1.2.0 h1:kr/MCeFWJWTwyaHoR9c8EjH9OumOmoF9YGiZd7lFm/Q=
|
||||||
|
sigs.k8s.io/yaml v1.2.0/go.mod h1:yfXDCHCao9+ENCvLSE62v9VSji2MKu5jeNfTrofGhJc=
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
package vitastor
|
||||||
|
|
||||||
|
const (
|
||||||
|
vitastorCSIDriverName = "csi.vitastor.io"
|
||||||
|
vitastorCSIDriverVersion = "0.6.16"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Config struct fills the parameters of request or user input
|
||||||
|
type Config struct
|
||||||
|
{
|
||||||
|
Endpoint string
|
||||||
|
NodeID string
|
||||||
|
}
|
||||||
|
|
||||||
|
// NewConfig returns config struct to initialize new driver
|
||||||
|
func NewConfig() *Config
|
||||||
|
{
|
||||||
|
return &Config{}
|
||||||
|
}
|
||||||
@@ -0,0 +1,530 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
package vitastor
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"strings"
|
||||||
|
"bytes"
|
||||||
|
"strconv"
|
||||||
|
"time"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"io/ioutil"
|
||||||
|
|
||||||
|
"github.com/kubernetes-csi/csi-lib-utils/protosanitizer"
|
||||||
|
"k8s.io/klog"
|
||||||
|
|
||||||
|
"google.golang.org/grpc/codes"
|
||||||
|
"google.golang.org/grpc/status"
|
||||||
|
|
||||||
|
"go.etcd.io/etcd/clientv3"
|
||||||
|
|
||||||
|
"github.com/container-storage-interface/spec/lib/go/csi"
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
KB int64 = 1024
|
||||||
|
MB int64 = 1024 * KB
|
||||||
|
GB int64 = 1024 * MB
|
||||||
|
TB int64 = 1024 * GB
|
||||||
|
ETCD_TIMEOUT time.Duration = 15*time.Second
|
||||||
|
)
|
||||||
|
|
||||||
|
type InodeIndex struct
|
||||||
|
{
|
||||||
|
Id uint64 `json:"id"`
|
||||||
|
PoolId uint64 `json:"pool_id"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type InodeConfig struct
|
||||||
|
{
|
||||||
|
Name string `json:"name"`
|
||||||
|
Size uint64 `json:"size,omitempty"`
|
||||||
|
ParentPool uint64 `json:"parent_pool,omitempty"`
|
||||||
|
ParentId uint64 `json:"parent_id,omitempty"`
|
||||||
|
Readonly bool `json:"readonly,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type ControllerServer struct
|
||||||
|
{
|
||||||
|
*Driver
|
||||||
|
}
|
||||||
|
|
||||||
|
// NewControllerServer create new instance controller
|
||||||
|
func NewControllerServer(driver *Driver) *ControllerServer
|
||||||
|
{
|
||||||
|
return &ControllerServer{
|
||||||
|
Driver: driver,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func GetConnectionParams(params map[string]string) (map[string]string, []string, string)
|
||||||
|
{
|
||||||
|
ctxVars := make(map[string]string)
|
||||||
|
configPath := params["configPath"]
|
||||||
|
if (configPath == "")
|
||||||
|
{
|
||||||
|
configPath = "/etc/vitastor/vitastor.conf"
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
ctxVars["configPath"] = configPath
|
||||||
|
}
|
||||||
|
config := make(map[string]interface{})
|
||||||
|
if configFD, err := os.Open(configPath); err == nil
|
||||||
|
{
|
||||||
|
defer configFD.Close()
|
||||||
|
data, _ := ioutil.ReadAll(configFD)
|
||||||
|
json.Unmarshal(data, &config)
|
||||||
|
}
|
||||||
|
// Try to load prefix & etcd URL from the config
|
||||||
|
var etcdUrl []string
|
||||||
|
if (params["etcdUrl"] != "")
|
||||||
|
{
|
||||||
|
ctxVars["etcdUrl"] = params["etcdUrl"]
|
||||||
|
etcdUrl = strings.Split(params["etcdUrl"], ",")
|
||||||
|
}
|
||||||
|
if (len(etcdUrl) == 0)
|
||||||
|
{
|
||||||
|
switch config["etcd_address"].(type)
|
||||||
|
{
|
||||||
|
case string:
|
||||||
|
etcdUrl = strings.Split(config["etcd_address"].(string), ",")
|
||||||
|
case []string:
|
||||||
|
etcdUrl = config["etcd_address"].([]string)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
etcdPrefix := params["etcdPrefix"]
|
||||||
|
if (etcdPrefix == "")
|
||||||
|
{
|
||||||
|
etcdPrefix, _ = config["etcd_prefix"].(string)
|
||||||
|
if (etcdPrefix == "")
|
||||||
|
{
|
||||||
|
etcdPrefix = "/vitastor"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
ctxVars["etcdPrefix"] = etcdPrefix
|
||||||
|
}
|
||||||
|
return ctxVars, etcdUrl, etcdPrefix
|
||||||
|
}
|
||||||
|
|
||||||
|
// Create the volume
|
||||||
|
func (cs *ControllerServer) CreateVolume(ctx context.Context, req *csi.CreateVolumeRequest) (*csi.CreateVolumeResponse, error)
|
||||||
|
{
|
||||||
|
klog.Infof("received controller create volume request %+v", protosanitizer.StripSecrets(req))
|
||||||
|
if (req == nil)
|
||||||
|
{
|
||||||
|
return nil, status.Errorf(codes.InvalidArgument, "request cannot be empty")
|
||||||
|
}
|
||||||
|
if (req.GetName() == "")
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.InvalidArgument, "name is a required field")
|
||||||
|
}
|
||||||
|
volumeCapabilities := req.GetVolumeCapabilities()
|
||||||
|
if (volumeCapabilities == nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.InvalidArgument, "volume capabilities is a required field")
|
||||||
|
}
|
||||||
|
|
||||||
|
etcdVolumePrefix := req.Parameters["etcdVolumePrefix"]
|
||||||
|
poolId, _ := strconv.ParseUint(req.Parameters["poolId"], 10, 64)
|
||||||
|
if (poolId == 0)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.InvalidArgument, "poolId is missing in storage class configuration")
|
||||||
|
}
|
||||||
|
|
||||||
|
volName := etcdVolumePrefix + req.GetName()
|
||||||
|
volSize := 1 * GB
|
||||||
|
if capRange := req.GetCapacityRange(); capRange != nil
|
||||||
|
{
|
||||||
|
volSize = ((capRange.GetRequiredBytes() + MB - 1) / MB) * MB
|
||||||
|
}
|
||||||
|
|
||||||
|
// FIXME: The following should PROBABLY be implemented externally in a management tool
|
||||||
|
|
||||||
|
ctxVars, etcdUrl, etcdPrefix := GetConnectionParams(req.Parameters)
|
||||||
|
if (len(etcdUrl) == 0)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.InvalidArgument, "no etcdUrl in storage class configuration and no etcd_address in vitastor.conf")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Connect to etcd
|
||||||
|
cli, err := clientv3.New(clientv3.Config{
|
||||||
|
DialTimeout: ETCD_TIMEOUT,
|
||||||
|
Endpoints: etcdUrl,
|
||||||
|
})
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "failed to connect to etcd at "+strings.Join(etcdUrl, ",")+": "+err.Error())
|
||||||
|
}
|
||||||
|
defer cli.Close()
|
||||||
|
|
||||||
|
var imageId uint64 = 0
|
||||||
|
for
|
||||||
|
{
|
||||||
|
// Check if the image exists
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), ETCD_TIMEOUT)
|
||||||
|
resp, err := cli.Get(ctx, etcdPrefix+"/index/image/"+volName)
|
||||||
|
cancel()
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "failed to read key from etcd: "+err.Error())
|
||||||
|
}
|
||||||
|
if (len(resp.Kvs) > 0)
|
||||||
|
{
|
||||||
|
kv := resp.Kvs[0]
|
||||||
|
var v InodeIndex
|
||||||
|
err := json.Unmarshal(kv.Value, &v)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "invalid /index/image/"+volName+" key in etcd: "+err.Error())
|
||||||
|
}
|
||||||
|
poolId = v.PoolId
|
||||||
|
imageId = v.Id
|
||||||
|
inodeCfgKey := fmt.Sprintf("/config/inode/%d/%d", poolId, imageId)
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), ETCD_TIMEOUT)
|
||||||
|
resp, err := cli.Get(ctx, etcdPrefix+inodeCfgKey)
|
||||||
|
cancel()
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "failed to read key from etcd: "+err.Error())
|
||||||
|
}
|
||||||
|
if (len(resp.Kvs) == 0)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "missing "+inodeCfgKey+" key in etcd")
|
||||||
|
}
|
||||||
|
var inodeCfg InodeConfig
|
||||||
|
err = json.Unmarshal(resp.Kvs[0].Value, &inodeCfg)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "invalid "+inodeCfgKey+" key in etcd: "+err.Error())
|
||||||
|
}
|
||||||
|
if (inodeCfg.Size < uint64(volSize))
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "image "+volName+" is already created, but size is less than expected")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// Find a free ID
|
||||||
|
// Create image metadata in a transaction verifying that the image doesn't exist yet AND ID is still free
|
||||||
|
maxIdKey := fmt.Sprintf("%s/index/maxid/%d", etcdPrefix, poolId)
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), ETCD_TIMEOUT)
|
||||||
|
resp, err := cli.Get(ctx, maxIdKey)
|
||||||
|
cancel()
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "failed to read key from etcd: "+err.Error())
|
||||||
|
}
|
||||||
|
var modRev int64
|
||||||
|
var nextId uint64
|
||||||
|
if (len(resp.Kvs) > 0)
|
||||||
|
{
|
||||||
|
var err error
|
||||||
|
nextId, err = strconv.ParseUint(string(resp.Kvs[0].Value), 10, 64)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, maxIdKey+" contains invalid ID")
|
||||||
|
}
|
||||||
|
modRev = resp.Kvs[0].ModRevision
|
||||||
|
nextId++
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
nextId = 1
|
||||||
|
}
|
||||||
|
inodeIdxJson, _ := json.Marshal(InodeIndex{
|
||||||
|
Id: nextId,
|
||||||
|
PoolId: poolId,
|
||||||
|
})
|
||||||
|
inodeCfgJson, _ := json.Marshal(InodeConfig{
|
||||||
|
Name: volName,
|
||||||
|
Size: uint64(volSize),
|
||||||
|
})
|
||||||
|
ctx, cancel = context.WithTimeout(context.Background(), ETCD_TIMEOUT)
|
||||||
|
txnResp, err := cli.Txn(ctx).If(
|
||||||
|
clientv3.Compare(clientv3.ModRevision(fmt.Sprintf("%s/index/maxid/%d", etcdPrefix, poolId)), "=", modRev),
|
||||||
|
clientv3.Compare(clientv3.CreateRevision(fmt.Sprintf("%s/index/image/%s", etcdPrefix, volName)), "=", 0),
|
||||||
|
clientv3.Compare(clientv3.CreateRevision(fmt.Sprintf("%s/config/inode/%d/%d", etcdPrefix, poolId, nextId)), "=", 0),
|
||||||
|
).Then(
|
||||||
|
clientv3.OpPut(fmt.Sprintf("%s/index/maxid/%d", etcdPrefix, poolId), fmt.Sprintf("%d", nextId)),
|
||||||
|
clientv3.OpPut(fmt.Sprintf("%s/index/image/%s", etcdPrefix, volName), string(inodeIdxJson)),
|
||||||
|
clientv3.OpPut(fmt.Sprintf("%s/config/inode/%d/%d", etcdPrefix, poolId, nextId), string(inodeCfgJson)),
|
||||||
|
).Commit()
|
||||||
|
cancel()
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "failed to commit transaction in etcd: "+err.Error())
|
||||||
|
}
|
||||||
|
if (txnResp.Succeeded)
|
||||||
|
{
|
||||||
|
imageId = nextId
|
||||||
|
break
|
||||||
|
}
|
||||||
|
// Start over if the transaction fails
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
ctxVars["name"] = volName
|
||||||
|
volumeIdJson, _ := json.Marshal(ctxVars)
|
||||||
|
return &csi.CreateVolumeResponse{
|
||||||
|
Volume: &csi.Volume{
|
||||||
|
// Ugly, but VolumeContext isn't passed to DeleteVolume :-(
|
||||||
|
VolumeId: string(volumeIdJson),
|
||||||
|
CapacityBytes: volSize,
|
||||||
|
},
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// DeleteVolume deletes the given volume
|
||||||
|
func (cs *ControllerServer) DeleteVolume(ctx context.Context, req *csi.DeleteVolumeRequest) (*csi.DeleteVolumeResponse, error)
|
||||||
|
{
|
||||||
|
klog.Infof("received controller delete volume request %+v", protosanitizer.StripSecrets(req))
|
||||||
|
if (req == nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.InvalidArgument, "request cannot be empty")
|
||||||
|
}
|
||||||
|
|
||||||
|
ctxVars := make(map[string]string)
|
||||||
|
err := json.Unmarshal([]byte(req.VolumeId), &ctxVars)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "volume ID not in JSON format")
|
||||||
|
}
|
||||||
|
volName := ctxVars["name"]
|
||||||
|
|
||||||
|
_, etcdUrl, etcdPrefix := GetConnectionParams(ctxVars)
|
||||||
|
if (len(etcdUrl) == 0)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.InvalidArgument, "no etcdUrl in storage class configuration and no etcd_address in vitastor.conf")
|
||||||
|
}
|
||||||
|
|
||||||
|
cli, err := clientv3.New(clientv3.Config{
|
||||||
|
DialTimeout: ETCD_TIMEOUT,
|
||||||
|
Endpoints: etcdUrl,
|
||||||
|
})
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "failed to connect to etcd at "+strings.Join(etcdUrl, ",")+": "+err.Error())
|
||||||
|
}
|
||||||
|
defer cli.Close()
|
||||||
|
|
||||||
|
// Find inode by name
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), ETCD_TIMEOUT)
|
||||||
|
resp, err := cli.Get(ctx, etcdPrefix+"/index/image/"+volName)
|
||||||
|
cancel()
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "failed to read key from etcd: "+err.Error())
|
||||||
|
}
|
||||||
|
if (len(resp.Kvs) == 0)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.NotFound, "volume "+volName+" does not exist")
|
||||||
|
}
|
||||||
|
var idx InodeIndex
|
||||||
|
err = json.Unmarshal(resp.Kvs[0].Value, &idx)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "invalid /index/image/"+volName+" key in etcd: "+err.Error())
|
||||||
|
}
|
||||||
|
|
||||||
|
// Get inode config
|
||||||
|
inodeCfgKey := fmt.Sprintf("%s/config/inode/%d/%d", etcdPrefix, idx.PoolId, idx.Id)
|
||||||
|
ctx, cancel = context.WithTimeout(context.Background(), ETCD_TIMEOUT)
|
||||||
|
resp, err = cli.Get(ctx, inodeCfgKey)
|
||||||
|
cancel()
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "failed to read key from etcd: "+err.Error())
|
||||||
|
}
|
||||||
|
if (len(resp.Kvs) == 0)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.NotFound, "volume "+volName+" does not exist")
|
||||||
|
}
|
||||||
|
var inodeCfg InodeConfig
|
||||||
|
err = json.Unmarshal(resp.Kvs[0].Value, &inodeCfg)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "invalid "+inodeCfgKey+" key in etcd: "+err.Error())
|
||||||
|
}
|
||||||
|
|
||||||
|
// Delete inode data by invoking vitastor-cli
|
||||||
|
args := []string{
|
||||||
|
"rm-data", "--etcd_address", strings.Join(etcdUrl, ","),
|
||||||
|
"--pool", fmt.Sprintf("%d", idx.PoolId),
|
||||||
|
"--inode", fmt.Sprintf("%d", idx.Id),
|
||||||
|
}
|
||||||
|
if (ctxVars["configPath"] != "")
|
||||||
|
{
|
||||||
|
args = append(args, "--config_path", ctxVars["configPath"])
|
||||||
|
}
|
||||||
|
c := exec.Command("/usr/bin/vitastor-cli", args...)
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
c.Stdout = nil
|
||||||
|
c.Stderr = &stderr
|
||||||
|
err = c.Run()
|
||||||
|
stderrStr := string(stderr.Bytes())
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("vitastor-cli rm-data failed: %s, status %s\n", stderrStr, err)
|
||||||
|
return nil, status.Error(codes.Internal, stderrStr+" (status "+err.Error()+")")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Delete inode config in etcd
|
||||||
|
ctx, cancel = context.WithTimeout(context.Background(), ETCD_TIMEOUT)
|
||||||
|
txnResp, err := cli.Txn(ctx).Then(
|
||||||
|
clientv3.OpDelete(fmt.Sprintf("%s/index/image/%s", etcdPrefix, volName)),
|
||||||
|
clientv3.OpDelete(fmt.Sprintf("%s/config/inode/%d/%d", etcdPrefix, idx.PoolId, idx.Id)),
|
||||||
|
).Commit()
|
||||||
|
cancel()
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "failed to delete keys in etcd: "+err.Error())
|
||||||
|
}
|
||||||
|
if (!txnResp.Succeeded)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "failed to delete keys in etcd: transaction failed")
|
||||||
|
}
|
||||||
|
|
||||||
|
return &csi.DeleteVolumeResponse{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// ControllerPublishVolume return Unimplemented error
|
||||||
|
func (cs *ControllerServer) ControllerPublishVolume(ctx context.Context, req *csi.ControllerPublishVolumeRequest) (*csi.ControllerPublishVolumeResponse, error)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Unimplemented, "")
|
||||||
|
}
|
||||||
|
|
||||||
|
// ControllerUnpublishVolume return Unimplemented error
|
||||||
|
func (cs *ControllerServer) ControllerUnpublishVolume(ctx context.Context, req *csi.ControllerUnpublishVolumeRequest) (*csi.ControllerUnpublishVolumeResponse, error)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Unimplemented, "")
|
||||||
|
}
|
||||||
|
|
||||||
|
// ValidateVolumeCapabilities checks whether the volume capabilities requested are supported.
|
||||||
|
func (cs *ControllerServer) ValidateVolumeCapabilities(ctx context.Context, req *csi.ValidateVolumeCapabilitiesRequest) (*csi.ValidateVolumeCapabilitiesResponse, error)
|
||||||
|
{
|
||||||
|
klog.Infof("received controller validate volume capability request %+v", protosanitizer.StripSecrets(req))
|
||||||
|
if (req == nil)
|
||||||
|
{
|
||||||
|
return nil, status.Errorf(codes.InvalidArgument, "request is nil")
|
||||||
|
}
|
||||||
|
volumeID := req.GetVolumeId()
|
||||||
|
if (volumeID == "")
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.InvalidArgument, "volumeId is nil")
|
||||||
|
}
|
||||||
|
volumeCapabilities := req.GetVolumeCapabilities()
|
||||||
|
if (volumeCapabilities == nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.InvalidArgument, "volumeCapabilities is nil")
|
||||||
|
}
|
||||||
|
|
||||||
|
var volumeCapabilityAccessModes []*csi.VolumeCapability_AccessMode
|
||||||
|
for _, mode := range []csi.VolumeCapability_AccessMode_Mode{
|
||||||
|
csi.VolumeCapability_AccessMode_SINGLE_NODE_WRITER,
|
||||||
|
csi.VolumeCapability_AccessMode_MULTI_NODE_MULTI_WRITER,
|
||||||
|
} {
|
||||||
|
volumeCapabilityAccessModes = append(volumeCapabilityAccessModes, &csi.VolumeCapability_AccessMode{Mode: mode})
|
||||||
|
}
|
||||||
|
|
||||||
|
capabilitySupport := false
|
||||||
|
for _, capability := range volumeCapabilities
|
||||||
|
{
|
||||||
|
for _, volumeCapabilityAccessMode := range volumeCapabilityAccessModes
|
||||||
|
{
|
||||||
|
if (volumeCapabilityAccessMode.Mode == capability.AccessMode.Mode)
|
||||||
|
{
|
||||||
|
capabilitySupport = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!capabilitySupport)
|
||||||
|
{
|
||||||
|
return nil, status.Errorf(codes.NotFound, "%v not supported", req.GetVolumeCapabilities())
|
||||||
|
}
|
||||||
|
|
||||||
|
return &csi.ValidateVolumeCapabilitiesResponse{
|
||||||
|
Confirmed: &csi.ValidateVolumeCapabilitiesResponse_Confirmed{
|
||||||
|
VolumeCapabilities: req.VolumeCapabilities,
|
||||||
|
},
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// ListVolumes returns a list of volumes
|
||||||
|
func (cs *ControllerServer) ListVolumes(ctx context.Context, req *csi.ListVolumesRequest) (*csi.ListVolumesResponse, error)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Unimplemented, "")
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetCapacity returns the capacity of the storage pool
|
||||||
|
func (cs *ControllerServer) GetCapacity(ctx context.Context, req *csi.GetCapacityRequest) (*csi.GetCapacityResponse, error)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Unimplemented, "")
|
||||||
|
}
|
||||||
|
|
||||||
|
// ControllerGetCapabilities returns the capabilities of the controller service.
|
||||||
|
func (cs *ControllerServer) ControllerGetCapabilities(ctx context.Context, req *csi.ControllerGetCapabilitiesRequest) (*csi.ControllerGetCapabilitiesResponse, error)
|
||||||
|
{
|
||||||
|
functionControllerServerCapabilities := func(cap csi.ControllerServiceCapability_RPC_Type) *csi.ControllerServiceCapability
|
||||||
|
{
|
||||||
|
return &csi.ControllerServiceCapability{
|
||||||
|
Type: &csi.ControllerServiceCapability_Rpc{
|
||||||
|
Rpc: &csi.ControllerServiceCapability_RPC{
|
||||||
|
Type: cap,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
var controllerServerCapabilities []*csi.ControllerServiceCapability
|
||||||
|
for _, capability := range []csi.ControllerServiceCapability_RPC_Type{
|
||||||
|
csi.ControllerServiceCapability_RPC_CREATE_DELETE_VOLUME,
|
||||||
|
csi.ControllerServiceCapability_RPC_LIST_VOLUMES,
|
||||||
|
csi.ControllerServiceCapability_RPC_EXPAND_VOLUME,
|
||||||
|
csi.ControllerServiceCapability_RPC_CREATE_DELETE_SNAPSHOT,
|
||||||
|
} {
|
||||||
|
controllerServerCapabilities = append(controllerServerCapabilities, functionControllerServerCapabilities(capability))
|
||||||
|
}
|
||||||
|
|
||||||
|
return &csi.ControllerGetCapabilitiesResponse{
|
||||||
|
Capabilities: controllerServerCapabilities,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// CreateSnapshot create snapshot of an existing PV
|
||||||
|
func (cs *ControllerServer) CreateSnapshot(ctx context.Context, req *csi.CreateSnapshotRequest) (*csi.CreateSnapshotResponse, error)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Unimplemented, "")
|
||||||
|
}
|
||||||
|
|
||||||
|
// DeleteSnapshot delete provided snapshot of a PV
|
||||||
|
func (cs *ControllerServer) DeleteSnapshot(ctx context.Context, req *csi.DeleteSnapshotRequest) (*csi.DeleteSnapshotResponse, error)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Unimplemented, "")
|
||||||
|
}
|
||||||
|
|
||||||
|
// ListSnapshots list the snapshots of a PV
|
||||||
|
func (cs *ControllerServer) ListSnapshots(ctx context.Context, req *csi.ListSnapshotsRequest) (*csi.ListSnapshotsResponse, error)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Unimplemented, "")
|
||||||
|
}
|
||||||
|
|
||||||
|
// ControllerExpandVolume resizes a volume
|
||||||
|
func (cs *ControllerServer) ControllerExpandVolume(ctx context.Context, req *csi.ControllerExpandVolumeRequest) (*csi.ControllerExpandVolumeResponse, error)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Unimplemented, "")
|
||||||
|
}
|
||||||
|
|
||||||
|
// ControllerGetVolume get volume info
|
||||||
|
func (cs *ControllerServer) ControllerGetVolume(ctx context.Context, req *csi.ControllerGetVolumeRequest) (*csi.ControllerGetVolumeResponse, error)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Unimplemented, "")
|
||||||
|
}
|
||||||
+137
@@ -0,0 +1,137 @@
|
|||||||
|
/*
|
||||||
|
Copyright 2017 The Kubernetes Authors.
|
||||||
|
|
||||||
|
Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
you may not use this file except in compliance with the License.
|
||||||
|
You may obtain a copy of the License at
|
||||||
|
|
||||||
|
http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
|
||||||
|
Unless required by applicable law or agreed to in writing, software
|
||||||
|
distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
See the License for the specific language governing permissions and
|
||||||
|
limitations under the License.
|
||||||
|
*/
|
||||||
|
|
||||||
|
package vitastor
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"net"
|
||||||
|
"os"
|
||||||
|
"strings"
|
||||||
|
"sync"
|
||||||
|
|
||||||
|
"github.com/golang/glog"
|
||||||
|
"golang.org/x/net/context"
|
||||||
|
"google.golang.org/grpc"
|
||||||
|
|
||||||
|
"github.com/container-storage-interface/spec/lib/go/csi"
|
||||||
|
"github.com/kubernetes-csi/csi-lib-utils/protosanitizer"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Defines Non blocking GRPC server interfaces
|
||||||
|
type NonBlockingGRPCServer interface {
|
||||||
|
// Start services at the endpoint
|
||||||
|
Start(endpoint string, ids csi.IdentityServer, cs csi.ControllerServer, ns csi.NodeServer)
|
||||||
|
// Waits for the service to stop
|
||||||
|
Wait()
|
||||||
|
// Stops the service gracefully
|
||||||
|
Stop()
|
||||||
|
// Stops the service forcefully
|
||||||
|
ForceStop()
|
||||||
|
}
|
||||||
|
|
||||||
|
func NewNonBlockingGRPCServer() NonBlockingGRPCServer {
|
||||||
|
return &nonBlockingGRPCServer{}
|
||||||
|
}
|
||||||
|
|
||||||
|
// NonBlocking server
|
||||||
|
type nonBlockingGRPCServer struct {
|
||||||
|
wg sync.WaitGroup
|
||||||
|
server *grpc.Server
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *nonBlockingGRPCServer) Start(endpoint string, ids csi.IdentityServer, cs csi.ControllerServer, ns csi.NodeServer) {
|
||||||
|
|
||||||
|
s.wg.Add(1)
|
||||||
|
|
||||||
|
go s.serve(endpoint, ids, cs, ns)
|
||||||
|
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *nonBlockingGRPCServer) Wait() {
|
||||||
|
s.wg.Wait()
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *nonBlockingGRPCServer) Stop() {
|
||||||
|
s.server.GracefulStop()
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *nonBlockingGRPCServer) ForceStop() {
|
||||||
|
s.server.Stop()
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *nonBlockingGRPCServer) serve(endpoint string, ids csi.IdentityServer, cs csi.ControllerServer, ns csi.NodeServer) {
|
||||||
|
|
||||||
|
proto, addr, err := ParseEndpoint(endpoint)
|
||||||
|
if err != nil {
|
||||||
|
glog.Fatal(err.Error())
|
||||||
|
}
|
||||||
|
|
||||||
|
if proto == "unix" {
|
||||||
|
addr = "/" + addr
|
||||||
|
if err := os.Remove(addr); err != nil && !os.IsNotExist(err) {
|
||||||
|
glog.Fatalf("Failed to remove %s, error: %s", addr, err.Error())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
listener, err := net.Listen(proto, addr)
|
||||||
|
if err != nil {
|
||||||
|
glog.Fatalf("Failed to listen: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
opts := []grpc.ServerOption{
|
||||||
|
grpc.UnaryInterceptor(logGRPC),
|
||||||
|
}
|
||||||
|
server := grpc.NewServer(opts...)
|
||||||
|
s.server = server
|
||||||
|
|
||||||
|
if ids != nil {
|
||||||
|
csi.RegisterIdentityServer(server, ids)
|
||||||
|
}
|
||||||
|
if cs != nil {
|
||||||
|
csi.RegisterControllerServer(server, cs)
|
||||||
|
}
|
||||||
|
if ns != nil {
|
||||||
|
csi.RegisterNodeServer(server, ns)
|
||||||
|
}
|
||||||
|
|
||||||
|
glog.Infof("Listening for connections on address: %#v", listener.Addr())
|
||||||
|
|
||||||
|
server.Serve(listener)
|
||||||
|
}
|
||||||
|
|
||||||
|
func ParseEndpoint(ep string) (string, string, error) {
|
||||||
|
if strings.HasPrefix(strings.ToLower(ep), "unix://") || strings.HasPrefix(strings.ToLower(ep), "tcp://") {
|
||||||
|
s := strings.SplitN(ep, "://", 2)
|
||||||
|
if s[1] != "" {
|
||||||
|
return s[0], s[1], nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "", "", fmt.Errorf("Invalid endpoint: %v", ep)
|
||||||
|
}
|
||||||
|
|
||||||
|
func logGRPC(ctx context.Context, req interface{}, info *grpc.UnaryServerInfo, handler grpc.UnaryHandler) (interface{}, error) {
|
||||||
|
glog.V(3).Infof("GRPC call: %s", info.FullMethod)
|
||||||
|
glog.V(5).Infof("GRPC request: %s", protosanitizer.StripSecrets(req))
|
||||||
|
resp, err := handler(ctx, req)
|
||||||
|
if err != nil {
|
||||||
|
glog.Errorf("GRPC error: %v", err)
|
||||||
|
} else {
|
||||||
|
glog.V(5).Infof("GRPC response: %s", protosanitizer.StripSecrets(resp))
|
||||||
|
}
|
||||||
|
return resp, err
|
||||||
|
}
|
||||||
@@ -0,0 +1,60 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
package vitastor
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
|
||||||
|
"github.com/kubernetes-csi/csi-lib-utils/protosanitizer"
|
||||||
|
"k8s.io/klog"
|
||||||
|
|
||||||
|
"github.com/container-storage-interface/spec/lib/go/csi"
|
||||||
|
)
|
||||||
|
|
||||||
|
// IdentityServer struct of Vitastor CSI driver with supported methods of CSI identity server spec.
|
||||||
|
type IdentityServer struct
|
||||||
|
{
|
||||||
|
*Driver
|
||||||
|
}
|
||||||
|
|
||||||
|
// NewIdentityServer create new instance identity
|
||||||
|
func NewIdentityServer(driver *Driver) *IdentityServer
|
||||||
|
{
|
||||||
|
return &IdentityServer{
|
||||||
|
Driver: driver,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetPluginInfo returns metadata of the plugin
|
||||||
|
func (is *IdentityServer) GetPluginInfo(ctx context.Context, req *csi.GetPluginInfoRequest) (*csi.GetPluginInfoResponse, error)
|
||||||
|
{
|
||||||
|
klog.Infof("received identity plugin info request %+v", protosanitizer.StripSecrets(req))
|
||||||
|
return &csi.GetPluginInfoResponse{
|
||||||
|
Name: vitastorCSIDriverName,
|
||||||
|
VendorVersion: vitastorCSIDriverVersion,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetPluginCapabilities returns available capabilities of the plugin
|
||||||
|
func (is *IdentityServer) GetPluginCapabilities(ctx context.Context, req *csi.GetPluginCapabilitiesRequest) (*csi.GetPluginCapabilitiesResponse, error)
|
||||||
|
{
|
||||||
|
klog.Infof("received identity plugin capabilities request %+v", protosanitizer.StripSecrets(req))
|
||||||
|
return &csi.GetPluginCapabilitiesResponse{
|
||||||
|
Capabilities: []*csi.PluginCapability{
|
||||||
|
{
|
||||||
|
Type: &csi.PluginCapability_Service_{
|
||||||
|
Service: &csi.PluginCapability_Service{
|
||||||
|
Type: csi.PluginCapability_Service_CONTROLLER_SERVICE,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Probe returns the health and readiness of the plugin
|
||||||
|
func (is *IdentityServer) Probe(ctx context.Context, req *csi.ProbeRequest) (*csi.ProbeResponse, error)
|
||||||
|
{
|
||||||
|
return &csi.ProbeResponse{}, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,293 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
package vitastor
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"encoding/json"
|
||||||
|
"strings"
|
||||||
|
"bytes"
|
||||||
|
|
||||||
|
"google.golang.org/grpc/codes"
|
||||||
|
"google.golang.org/grpc/status"
|
||||||
|
"k8s.io/utils/mount"
|
||||||
|
utilexec "k8s.io/utils/exec"
|
||||||
|
|
||||||
|
"github.com/container-storage-interface/spec/lib/go/csi"
|
||||||
|
"github.com/kubernetes-csi/csi-lib-utils/protosanitizer"
|
||||||
|
"k8s.io/klog"
|
||||||
|
)
|
||||||
|
|
||||||
|
// NodeServer struct of Vitastor CSI driver with supported methods of CSI node server spec.
|
||||||
|
type NodeServer struct
|
||||||
|
{
|
||||||
|
*Driver
|
||||||
|
mounter mount.Interface
|
||||||
|
}
|
||||||
|
|
||||||
|
// NewNodeServer create new instance node
|
||||||
|
func NewNodeServer(driver *Driver) *NodeServer
|
||||||
|
{
|
||||||
|
return &NodeServer{
|
||||||
|
Driver: driver,
|
||||||
|
mounter: mount.New(""),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// NodeStageVolume mounts the volume to a staging path on the node.
|
||||||
|
func (ns *NodeServer) NodeStageVolume(ctx context.Context, req *csi.NodeStageVolumeRequest) (*csi.NodeStageVolumeResponse, error)
|
||||||
|
{
|
||||||
|
return &csi.NodeStageVolumeResponse{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// NodeUnstageVolume unstages the volume from the staging path
|
||||||
|
func (ns *NodeServer) NodeUnstageVolume(ctx context.Context, req *csi.NodeUnstageVolumeRequest) (*csi.NodeUnstageVolumeResponse, error)
|
||||||
|
{
|
||||||
|
return &csi.NodeUnstageVolumeResponse{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func Contains(list []string, s string) bool
|
||||||
|
{
|
||||||
|
for i := 0; i < len(list); i++
|
||||||
|
{
|
||||||
|
if (list[i] == s)
|
||||||
|
{
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// NodePublishVolume mounts the volume mounted to the staging path to the target path
|
||||||
|
func (ns *NodeServer) NodePublishVolume(ctx context.Context, req *csi.NodePublishVolumeRequest) (*csi.NodePublishVolumeResponse, error)
|
||||||
|
{
|
||||||
|
klog.Infof("received node publish volume request %+v", protosanitizer.StripSecrets(req))
|
||||||
|
|
||||||
|
targetPath := req.GetTargetPath()
|
||||||
|
isBlock := req.GetVolumeCapability().GetBlock() != nil
|
||||||
|
|
||||||
|
// Check that it's not already mounted
|
||||||
|
_, error := mount.IsNotMountPoint(ns.mounter, targetPath)
|
||||||
|
if (error != nil)
|
||||||
|
{
|
||||||
|
if (os.IsNotExist(error))
|
||||||
|
{
|
||||||
|
if (isBlock)
|
||||||
|
{
|
||||||
|
pathFile, err := os.OpenFile(targetPath, os.O_CREATE|os.O_RDWR, 0o600)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to create block device mount target %s with error: %v", targetPath, err)
|
||||||
|
return nil, status.Error(codes.Internal, err.Error())
|
||||||
|
}
|
||||||
|
err = pathFile.Close()
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to close %s with error: %v", targetPath, err)
|
||||||
|
return nil, status.Error(codes.Internal, err.Error())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
err := os.MkdirAll(targetPath, 0777)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to create fs mount target %s with error: %v", targetPath, err)
|
||||||
|
return nil, status.Error(codes.Internal, err.Error())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, error.Error())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
ctxVars := make(map[string]string)
|
||||||
|
err := json.Unmarshal([]byte(req.VolumeId), &ctxVars)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, "volume ID not in JSON format")
|
||||||
|
}
|
||||||
|
volName := ctxVars["name"]
|
||||||
|
|
||||||
|
_, etcdUrl, etcdPrefix := GetConnectionParams(ctxVars)
|
||||||
|
if (len(etcdUrl) == 0)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.InvalidArgument, "no etcdUrl in storage class configuration and no etcd_address in vitastor.conf")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Map NBD device
|
||||||
|
// FIXME: Check if already mapped
|
||||||
|
args := []string{
|
||||||
|
"map", "--etcd_address", strings.Join(etcdUrl, ","),
|
||||||
|
"--etcd_prefix", etcdPrefix,
|
||||||
|
"--image", volName,
|
||||||
|
};
|
||||||
|
if (ctxVars["configPath"] != "")
|
||||||
|
{
|
||||||
|
args = append(args, "--config_path", ctxVars["configPath"])
|
||||||
|
}
|
||||||
|
if (req.GetReadonly())
|
||||||
|
{
|
||||||
|
args = append(args, "--readonly", "1")
|
||||||
|
}
|
||||||
|
c := exec.Command("/usr/bin/vitastor-nbd", args...)
|
||||||
|
var stdout, stderr bytes.Buffer
|
||||||
|
c.Stdout, c.Stderr = &stdout, &stderr
|
||||||
|
err = c.Run()
|
||||||
|
stdoutStr, stderrStr := string(stdout.Bytes()), string(stderr.Bytes())
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("vitastor-nbd map failed: %s, status %s\n", stdoutStr+stderrStr, err)
|
||||||
|
return nil, status.Error(codes.Internal, stdoutStr+stderrStr+" (status "+err.Error()+")")
|
||||||
|
}
|
||||||
|
devicePath := strings.TrimSpace(stdoutStr)
|
||||||
|
|
||||||
|
// Check existing format
|
||||||
|
diskMounter := &mount.SafeFormatAndMount{Interface: ns.mounter, Exec: utilexec.New()}
|
||||||
|
existingFormat, err := diskMounter.GetDiskFormat(devicePath)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to get disk format for path %s, error: %v", err)
|
||||||
|
// unmap NBD device
|
||||||
|
unmapOut, unmapErr := exec.Command("/usr/bin/vitastor-nbd", "unmap", devicePath).CombinedOutput()
|
||||||
|
if (unmapErr != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to unmap NBD device %s: %s, error: %v", devicePath, unmapOut, unmapErr)
|
||||||
|
}
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
// Format the device (ext4 or xfs)
|
||||||
|
fsType := req.GetVolumeCapability().GetMount().GetFsType()
|
||||||
|
opt := req.GetVolumeCapability().GetMount().GetMountFlags()
|
||||||
|
opt = append(opt, "_netdev")
|
||||||
|
if ((req.VolumeCapability.AccessMode.Mode == csi.VolumeCapability_AccessMode_MULTI_NODE_READER_ONLY ||
|
||||||
|
req.VolumeCapability.AccessMode.Mode == csi.VolumeCapability_AccessMode_SINGLE_NODE_READER_ONLY) &&
|
||||||
|
!Contains(opt, "ro"))
|
||||||
|
{
|
||||||
|
opt = append(opt, "ro")
|
||||||
|
}
|
||||||
|
if (fsType == "xfs")
|
||||||
|
{
|
||||||
|
opt = append(opt, "nouuid")
|
||||||
|
}
|
||||||
|
readOnly := Contains(opt, "ro")
|
||||||
|
if (existingFormat == "" && !readOnly)
|
||||||
|
{
|
||||||
|
args := []string{}
|
||||||
|
switch fsType
|
||||||
|
{
|
||||||
|
case "ext4":
|
||||||
|
args = []string{"-m0", "-Enodiscard,lazy_itable_init=1,lazy_journal_init=1", devicePath}
|
||||||
|
case "xfs":
|
||||||
|
args = []string{"-K", devicePath}
|
||||||
|
}
|
||||||
|
if (len(args) > 0)
|
||||||
|
{
|
||||||
|
cmdOut, cmdErr := diskMounter.Exec.Command("mkfs."+fsType, args...).CombinedOutput()
|
||||||
|
if (cmdErr != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to run mkfs error: %v, output: %v", cmdErr, string(cmdOut))
|
||||||
|
// unmap NBD device
|
||||||
|
unmapOut, unmapErr := exec.Command("/usr/bin/vitastor-nbd", "unmap", devicePath).CombinedOutput()
|
||||||
|
if (unmapErr != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to unmap NBD device %s: %s, error: %v", devicePath, unmapOut, unmapErr)
|
||||||
|
}
|
||||||
|
return nil, status.Error(codes.Internal, cmdErr.Error())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (isBlock)
|
||||||
|
{
|
||||||
|
opt = append(opt, "bind")
|
||||||
|
err = diskMounter.Mount(devicePath, targetPath, fsType, opt)
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
err = diskMounter.FormatAndMount(devicePath, targetPath, fsType, opt)
|
||||||
|
}
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf(
|
||||||
|
"failed to mount device path (%s) to path (%s) for volume (%s) error: %s",
|
||||||
|
devicePath, targetPath, volName, err,
|
||||||
|
)
|
||||||
|
// unmap NBD device
|
||||||
|
unmapOut, unmapErr := exec.Command("/usr/bin/vitastor-nbd", "unmap", devicePath).CombinedOutput()
|
||||||
|
if (unmapErr != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to unmap NBD device %s: %s, error: %v", devicePath, unmapOut, unmapErr)
|
||||||
|
}
|
||||||
|
return nil, status.Error(codes.Internal, err.Error())
|
||||||
|
}
|
||||||
|
return &csi.NodePublishVolumeResponse{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// NodeUnpublishVolume unmounts the volume from the target path
|
||||||
|
func (ns *NodeServer) NodeUnpublishVolume(ctx context.Context, req *csi.NodeUnpublishVolumeRequest) (*csi.NodeUnpublishVolumeResponse, error)
|
||||||
|
{
|
||||||
|
klog.Infof("received node unpublish volume request %+v", protosanitizer.StripSecrets(req))
|
||||||
|
targetPath := req.GetTargetPath()
|
||||||
|
devicePath, refCount, err := mount.GetDeviceNameFromMount(ns.mounter, targetPath)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
if (os.IsNotExist(err))
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.NotFound, "Target path not found")
|
||||||
|
}
|
||||||
|
return nil, status.Error(codes.Internal, err.Error())
|
||||||
|
}
|
||||||
|
if (devicePath == "")
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.NotFound, "Volume not mounted")
|
||||||
|
}
|
||||||
|
// unmount
|
||||||
|
err = mount.CleanupMountPoint(targetPath, ns.mounter, false)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Internal, err.Error())
|
||||||
|
}
|
||||||
|
// unmap NBD device
|
||||||
|
if (refCount == 1)
|
||||||
|
{
|
||||||
|
unmapOut, unmapErr := exec.Command("/usr/bin/vitastor-nbd", "unmap", devicePath).CombinedOutput()
|
||||||
|
if (unmapErr != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to unmap NBD device %s: %s, error: %v", devicePath, unmapOut, unmapErr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return &csi.NodeUnpublishVolumeResponse{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// NodeGetVolumeStats returns volume capacity statistics available for the volume
|
||||||
|
func (ns *NodeServer) NodeGetVolumeStats(ctx context.Context, req *csi.NodeGetVolumeStatsRequest) (*csi.NodeGetVolumeStatsResponse, error)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Unimplemented, "")
|
||||||
|
}
|
||||||
|
|
||||||
|
// NodeExpandVolume expanding the file system on the node
|
||||||
|
func (ns *NodeServer) NodeExpandVolume(ctx context.Context, req *csi.NodeExpandVolumeRequest) (*csi.NodeExpandVolumeResponse, error)
|
||||||
|
{
|
||||||
|
return nil, status.Error(codes.Unimplemented, "")
|
||||||
|
}
|
||||||
|
|
||||||
|
// NodeGetCapabilities returns the supported capabilities of the node server
|
||||||
|
func (ns *NodeServer) NodeGetCapabilities(ctx context.Context, req *csi.NodeGetCapabilitiesRequest) (*csi.NodeGetCapabilitiesResponse, error)
|
||||||
|
{
|
||||||
|
return &csi.NodeGetCapabilitiesResponse{}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// NodeGetInfo returns NodeGetInfoResponse for CO.
|
||||||
|
func (ns *NodeServer) NodeGetInfo(ctx context.Context, req *csi.NodeGetInfoRequest) (*csi.NodeGetInfoResponse, error)
|
||||||
|
{
|
||||||
|
klog.Infof("received node get info request %+v", protosanitizer.StripSecrets(req))
|
||||||
|
return &csi.NodeGetInfoResponse{
|
||||||
|
NodeId: ns.NodeID,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,36 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
package vitastor
|
||||||
|
|
||||||
|
import (
|
||||||
|
"k8s.io/klog"
|
||||||
|
)
|
||||||
|
|
||||||
|
type Driver struct
|
||||||
|
{
|
||||||
|
*Config
|
||||||
|
}
|
||||||
|
|
||||||
|
// NewDriver create new instance driver
|
||||||
|
func NewDriver(config *Config) (*Driver, error)
|
||||||
|
{
|
||||||
|
if (config == nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("Vitastor CSI driver initialization failed")
|
||||||
|
return nil, nil
|
||||||
|
}
|
||||||
|
driver := &Driver{
|
||||||
|
Config: config,
|
||||||
|
}
|
||||||
|
klog.Infof("Vitastor CSI driver initialized")
|
||||||
|
return driver, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Start server
|
||||||
|
func (driver *Driver) Run()
|
||||||
|
{
|
||||||
|
server := NewNonBlockingGRPCServer()
|
||||||
|
server.Start(driver.Endpoint, NewIdentityServer(driver), NewControllerServer(driver), NewNodeServer(driver))
|
||||||
|
server.Wait()
|
||||||
|
}
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"flag"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"k8s.io/klog"
|
||||||
|
"vitastor.io/csi/src"
|
||||||
|
)
|
||||||
|
|
||||||
|
func main()
|
||||||
|
{
|
||||||
|
var config = vitastor.NewConfig()
|
||||||
|
flag.StringVar(&config.Endpoint, "endpoint", "", "CSI endpoint")
|
||||||
|
flag.StringVar(&config.NodeID, "node", "", "Node ID")
|
||||||
|
flag.Parse()
|
||||||
|
if (config.Endpoint == "")
|
||||||
|
{
|
||||||
|
config.Endpoint = os.Getenv("CSI_ENDPOINT")
|
||||||
|
}
|
||||||
|
if (config.NodeID == "")
|
||||||
|
{
|
||||||
|
config.NodeID = os.Getenv("NODE_ID")
|
||||||
|
}
|
||||||
|
if (config.Endpoint == "" && config.NodeID == "")
|
||||||
|
{
|
||||||
|
fmt.Fprintf(os.Stderr, "Please set -endpoint and -node / CSI_ENDPOINT & NODE_ID env vars\n")
|
||||||
|
os.Exit(1)
|
||||||
|
}
|
||||||
|
drv, err := vitastor.NewDriver(config)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Fatalln(err)
|
||||||
|
}
|
||||||
|
drv.Run()
|
||||||
|
}
|
||||||
+7
@@ -0,0 +1,7 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
cat < vitastor.Dockerfile > ../Dockerfile
|
||||||
|
cd ..
|
||||||
|
mkdir -p packages
|
||||||
|
sudo podman build --build-arg REL=bullseye -v `pwd`/packages:/root/packages -f Dockerfile .
|
||||||
|
rm Dockerfile
|
||||||
+7
@@ -0,0 +1,7 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
cat < vitastor.Dockerfile > ../Dockerfile
|
||||||
|
cd ..
|
||||||
|
mkdir -p packages
|
||||||
|
sudo podman build --build-arg REL=buster -v `pwd`/packages:/root/packages -f Dockerfile .
|
||||||
|
rm Dockerfile
|
||||||
Vendored
+22
@@ -1,3 +1,25 @@
|
|||||||
|
vitastor (0.6.16-1) unstable; urgency=medium
|
||||||
|
|
||||||
|
* RDMA support
|
||||||
|
* Bugfixes
|
||||||
|
|
||||||
|
-- Vitaliy Filippov <vitalif@yourcmc.ru> Sat, 01 May 2021 18:46:10 +0300
|
||||||
|
|
||||||
|
vitastor (0.6.0-1) unstable; urgency=medium
|
||||||
|
|
||||||
|
* Snapshots and Copy-on-Write clones
|
||||||
|
* Image metadata in etcd (name, size)
|
||||||
|
* Image I/O and space statistics in etcd
|
||||||
|
* Write throttling for smoothing random write workloads in SSD+HDD configurations
|
||||||
|
|
||||||
|
-- Vitaliy Filippov <vitalif@yourcmc.ru> Sun, 11 Apr 2021 00:49:18 +0300
|
||||||
|
|
||||||
|
vitastor (0.5.1-1) unstable; urgency=medium
|
||||||
|
|
||||||
|
* Add jerasure support
|
||||||
|
|
||||||
|
-- Vitaliy Filippov <vitalif@yourcmc.ru> Sat, 05 Dec 2020 17:02:26 +0300
|
||||||
|
|
||||||
vitastor (0.5-1) unstable; urgency=medium
|
vitastor (0.5-1) unstable; urgency=medium
|
||||||
|
|
||||||
* First packaging for Debian
|
* First packaging for Debian
|
||||||
|
|||||||
Vendored
+41
-3
@@ -2,16 +2,54 @@ Source: vitastor
|
|||||||
Section: admin
|
Section: admin
|
||||||
Priority: optional
|
Priority: optional
|
||||||
Maintainer: Vitaliy Filippov <vitalif@yourcmc.ru>
|
Maintainer: Vitaliy Filippov <vitalif@yourcmc.ru>
|
||||||
Build-Depends: debhelper, liburing-dev (>= 0.6), g++ (>= 8), libstdc++6 (>= 8), linux-libc-dev, libgoogle-perftools-dev
|
Build-Depends: debhelper, liburing-dev (>= 0.6), g++ (>= 8), libstdc++6 (>= 8), linux-libc-dev, libgoogle-perftools-dev, libjerasure-dev, libgf-complete-dev, libibverbs-dev
|
||||||
Standards-Version: 4.5.0
|
Standards-Version: 4.5.0
|
||||||
Homepage: https://vitastor.io/
|
Homepage: https://vitastor.io/
|
||||||
Rules-Requires-Root: no
|
Rules-Requires-Root: no
|
||||||
|
|
||||||
Package: vitastor
|
Package: vitastor
|
||||||
Architecture: any
|
Architecture: amd64
|
||||||
Depends: ${shlibs:Depends}, ${misc:Depends}, fio (= ${dep:fio}), qemu (= ${dep:qemu}), nodejs (>= 12), node-sprintf-js, node-ws (>= 7)
|
Depends: vitastor-osd, vitastor-mon, vitastor-client, vitastor-client-dev, vitastor-fio
|
||||||
Description: Vitastor, a fast software-defined clustered block storage
|
Description: Vitastor, a fast software-defined clustered block storage
|
||||||
Vitastor is a small, simple and fast clustered block storage (storage for VM drives),
|
Vitastor is a small, simple and fast clustered block storage (storage for VM drives),
|
||||||
architecturally similar to Ceph which means strong consistency, primary-replication,
|
architecturally similar to Ceph which means strong consistency, primary-replication,
|
||||||
symmetric clustering and automatic data distribution over any number of drives of any
|
symmetric clustering and automatic data distribution over any number of drives of any
|
||||||
size with configurable redundancy (replication or erasure codes/XOR).
|
size with configurable redundancy (replication or erasure codes/XOR).
|
||||||
|
|
||||||
|
Package: vitastor-osd
|
||||||
|
Architecture: amd64
|
||||||
|
Depends: ${shlibs:Depends}, ${misc:Depends}, vitastor-client (= ${binary:Version})
|
||||||
|
Description: Vitastor, a fast software-defined clustered block storage - object storage daemon
|
||||||
|
Vitastor object storage daemon, i.e. server program that stores data.
|
||||||
|
|
||||||
|
Package: vitastor-mon
|
||||||
|
Architecture: amd64
|
||||||
|
Depends: ${misc:Depends}, nodejs (>= 10), node-sprintf-js, node-ws (>= 7), lp-solve
|
||||||
|
Description: Vitastor, a fast software-defined clustered block storage - monitor
|
||||||
|
Vitastor monitor, i.e. server program responsible for watching cluster state and
|
||||||
|
scheduling cluster-level operations.
|
||||||
|
|
||||||
|
Package: vitastor-client
|
||||||
|
Architecture: amd64
|
||||||
|
Depends: ${shlibs:Depends}, ${misc:Depends}
|
||||||
|
Description: Vitastor, a fast software-defined clustered block storage - client
|
||||||
|
Vitastor client library and command-line interface.
|
||||||
|
|
||||||
|
Package: vitastor-client-dev
|
||||||
|
Section: devel
|
||||||
|
Architecture: amd64
|
||||||
|
Depends: ${misc:Depends}, vitastor-client (= ${binary:Version})
|
||||||
|
Description: Vitastor, a fast software-defined clustered block storage - development files
|
||||||
|
Vitastor library headers for development.
|
||||||
|
|
||||||
|
Package: vitastor-fio
|
||||||
|
Architecture: amd64
|
||||||
|
Depends: ${shlibs:Depends}, ${misc:Depends}, vitastor-client (= ${binary:Version}), fio (= ${dep:fio})
|
||||||
|
Description: Vitastor, a fast software-defined clustered block storage - fio drivers
|
||||||
|
Vitastor fio drivers for benchmarking.
|
||||||
|
|
||||||
|
Package: pve-storage-vitastor
|
||||||
|
Architecture: amd64
|
||||||
|
Depends: ${shlibs:Depends}, ${misc:Depends}, vitastor-client (= ${binary:Version})
|
||||||
|
Description: Vitastor Proxmox Virtual Environment storage plugin
|
||||||
|
Vitastor storage plugin for Proxmox Virtual Environment.
|
||||||
|
|||||||
Vendored
+7
-6
@@ -5,16 +5,17 @@ Source: https://vitastor.io
|
|||||||
|
|
||||||
Files: *
|
Files: *
|
||||||
Copyright: 2019+ Vitaliy Filippov <vitalif@yourcmc.ru>
|
Copyright: 2019+ Vitaliy Filippov <vitalif@yourcmc.ru>
|
||||||
License: Multiple licenses VNPL-1.0 and/or GPL-2.0+
|
License: Multiple licenses VNPL-1.1 and/or GPL-2.0+
|
||||||
All server-side code (OSD, Monitor and so on) is licensed under the terms of
|
All server-side code (OSD, Monitor and so on) is licensed under the terms of
|
||||||
Vitastor Network Public License 1.0 (VNPL 1.0), a copyleft license based on
|
Vitastor Network Public License 1.1 (VNPL 1.1), a copyleft license based on
|
||||||
GNU GPLv3.0 with the additional "Network Interaction" clause which requires
|
GNU GPLv3.0 with the additional "Network Interaction" clause which requires
|
||||||
opensourcing all programs directly or indirectly interacting with Vitastor
|
opensourcing all programs directly or indirectly interacting with Vitastor
|
||||||
through a computer network ("Proxy Programs"). Proxy Programs may be made public
|
through a computer network and expressly designed to be used in conjunction
|
||||||
not only under the terms of the same license, but also under the terms of any
|
with it ("Proxy Programs"). Proxy Programs may be made public not only under
|
||||||
GPL-Compatible Free Software License, as listed by the Free Software Foundation.
|
the terms of the same license, but also under the terms of any GPL-Compatible
|
||||||
|
Free Software License, as listed by the Free Software Foundation.
|
||||||
This is a stricter copyleft license than the Affero GPL.
|
This is a stricter copyleft license than the Affero GPL.
|
||||||
.
|
.
|
||||||
Client libraries (cluster_client and so on) are dual-licensed under the same
|
Client libraries (cluster_client and so on) are dual-licensed under the same
|
||||||
VNPL 1.0 and also GNU GPL 2.0 or later to allow for compatibility with GPLed
|
VNPL 1.1 and also GNU GPL 2.0 or later to allow for compatibility with GPLed
|
||||||
software like QEMU and fio.
|
software like QEMU and fio.
|
||||||
|
|||||||
Vendored
+1
@@ -0,0 +1 @@
|
|||||||
|
dep:fio=3.16-1
|
||||||
Vendored
+3
-2
@@ -1,3 +1,4 @@
|
|||||||
VNPL-1.0.txt usr/share/doc/vitastor
|
VNPL-1.1.txt usr/share/doc/vitastor
|
||||||
GPL-2.0.txt usr/share/doc/vitastor
|
GPL-2.0.txt usr/share/doc/vitastor
|
||||||
mon usr/lib/vitastor
|
README.md usr/share/doc/vitastor
|
||||||
|
README-ru.md usr/share/doc/vitastor
|
||||||
|
|||||||
Vendored
+40
@@ -0,0 +1,40 @@
|
|||||||
|
# Build patched libvirt for Debian Buster or Bullseye/Sid inside a container
|
||||||
|
# cd ..; podman build --build-arg REL=bullseye -v `pwd`/packages:/root/packages -f debian/libvirt.Dockerfile .
|
||||||
|
|
||||||
|
ARG REL=
|
||||||
|
FROM debian:$REL
|
||||||
|
ARG REL=
|
||||||
|
|
||||||
|
WORKDIR /root
|
||||||
|
|
||||||
|
RUN if [ "$REL" = "buster" -o "$REL" = "bullseye" ]; then \
|
||||||
|
echo "deb http://deb.debian.org/debian $REL-backports main" >> /etc/apt/sources.list; \
|
||||||
|
echo >> /etc/apt/preferences; \
|
||||||
|
echo 'Package: *' >> /etc/apt/preferences; \
|
||||||
|
echo "Pin: release a=$REL-backports" >> /etc/apt/preferences; \
|
||||||
|
echo 'Pin-Priority: 500' >> /etc/apt/preferences; \
|
||||||
|
fi; \
|
||||||
|
grep '^deb ' /etc/apt/sources.list | perl -pe 's/^deb/deb-src/' >> /etc/apt/sources.list; \
|
||||||
|
echo 'APT::Install-Recommends false;' >> /etc/apt/apt.conf; \
|
||||||
|
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
|
||||||
|
|
||||||
|
RUN apt-get update; apt-get -y install devscripts
|
||||||
|
RUN apt-get -y build-dep libvirt0
|
||||||
|
RUN apt-get -y install libglusterfs-dev
|
||||||
|
RUN apt-get --download-only source libvirt
|
||||||
|
|
||||||
|
ADD patches/libvirt-5.0-vitastor.diff patches/libvirt-7.0-vitastor.diff patches/libvirt-7.5-vitastor.diff patches/libvirt-7.6-vitastor.diff /root
|
||||||
|
RUN set -e; \
|
||||||
|
mkdir -p /root/packages/libvirt-$REL; \
|
||||||
|
rm -rf /root/packages/libvirt-$REL/*; \
|
||||||
|
cd /root/packages/libvirt-$REL; \
|
||||||
|
dpkg-source -x /root/libvirt*.dsc; \
|
||||||
|
D=$(ls -d libvirt-*/); \
|
||||||
|
V=$(ls -d libvirt-*/ | perl -pe 's/libvirt-(\d+\.\d+).*/$1/'); \
|
||||||
|
cp /root/libvirt-$V-vitastor.diff $D/debian/patches; \
|
||||||
|
echo libvirt-$V-vitastor.diff >> $D/debian/patches/series; \
|
||||||
|
cd $D; \
|
||||||
|
V=$(head -n1 debian/changelog | perl -pe 's/^.*\((.*?)(~bpo[\d\+]*)?(\+deb[u\d]+)?\).*$/$1/')+vitastor2; \
|
||||||
|
DEBEMAIL="Vitaliy Filippov <vitalif@yourcmc.ru>" dch -D $REL -v $V 'Add Vitastor support'; \
|
||||||
|
DEB_BUILD_OPTIONS=nocheck dpkg-buildpackage --jobs=auto -sa; \
|
||||||
|
rm -rf /root/packages/libvirt-$REL/$D
|
||||||
Vendored
+61
@@ -0,0 +1,61 @@
|
|||||||
|
# Build patched QEMU for Debian Buster or Bullseye/Sid inside a container
|
||||||
|
# cd ..; podman build --build-arg REL=bullseye -v `pwd`/packages:/root/packages -f debian/patched-qemu.Dockerfile .
|
||||||
|
|
||||||
|
ARG REL=
|
||||||
|
FROM debian:$REL
|
||||||
|
ARG REL=
|
||||||
|
|
||||||
|
WORKDIR /root
|
||||||
|
|
||||||
|
RUN if [ "$REL" = "buster" -o "$REL" = "bullseye" ]; then \
|
||||||
|
echo "deb http://deb.debian.org/debian $REL-backports main" >> /etc/apt/sources.list; \
|
||||||
|
echo >> /etc/apt/preferences; \
|
||||||
|
echo 'Package: *' >> /etc/apt/preferences; \
|
||||||
|
echo "Pin: release a=$REL-backports" >> /etc/apt/preferences; \
|
||||||
|
echo 'Pin-Priority: 500' >> /etc/apt/preferences; \
|
||||||
|
fi; \
|
||||||
|
grep '^deb ' /etc/apt/sources.list | perl -pe 's/^deb/deb-src/' >> /etc/apt/sources.list; \
|
||||||
|
echo 'APT::Install-Recommends false;' >> /etc/apt/apt.conf; \
|
||||||
|
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
|
||||||
|
|
||||||
|
RUN apt-get update
|
||||||
|
RUN apt-get -y install qemu fio liburing1 liburing-dev libgoogle-perftools-dev devscripts
|
||||||
|
RUN apt-get -y build-dep qemu
|
||||||
|
# To build a custom version
|
||||||
|
#RUN cp /root/packages/qemu-orig/* /root
|
||||||
|
RUN apt-get --download-only source qemu
|
||||||
|
|
||||||
|
ADD patches/qemu-5.0-vitastor.patch patches/qemu-5.1-vitastor.patch patches/qemu-6.1-vitastor.patch src/qemu_driver.c /root/vitastor/patches/
|
||||||
|
RUN set -e; \
|
||||||
|
apt-get install -y wget; \
|
||||||
|
wget -q -O /etc/apt/trusted.gpg.d/vitastor.gpg https://vitastor.io/debian/pubkey.gpg; \
|
||||||
|
(echo deb http://vitastor.io/debian $REL main > /etc/apt/sources.list.d/vitastor.list); \
|
||||||
|
(echo "APT::Install-Recommends false;" > /etc/apt/apt.conf) && \
|
||||||
|
apt-get update; \
|
||||||
|
apt-get install -y vitastor-client vitastor-client-dev quilt; \
|
||||||
|
mkdir -p /root/packages/qemu-$REL; \
|
||||||
|
rm -rf /root/packages/qemu-$REL/*; \
|
||||||
|
cd /root/packages/qemu-$REL; \
|
||||||
|
dpkg-source -x /root/qemu*.dsc; \
|
||||||
|
if ls -d /root/packages/qemu-$REL/qemu-5.0*; then \
|
||||||
|
D=$(ls -d /root/packages/qemu-$REL/qemu-5.0*); \
|
||||||
|
cp /root/vitastor/patches/qemu-5.0-vitastor.patch $D/debian/patches; \
|
||||||
|
echo qemu-5.0-vitastor.patch >> $D/debian/patches/series; \
|
||||||
|
elif ls /root/packages/qemu-$REL/qemu-6.1*; then \
|
||||||
|
D=$(ls -d /root/packages/qemu-$REL/qemu-6.1*); \
|
||||||
|
cp /root/vitastor/patches/qemu-6.1-vitastor.patch $D/debian/patches; \
|
||||||
|
echo qemu-6.1-vitastor.patch >> $D/debian/patches/series; \
|
||||||
|
else \
|
||||||
|
cp /root/vitastor/patches/qemu-5.1-vitastor.patch /root/packages/qemu-$REL/qemu-*/debian/patches; \
|
||||||
|
P=`ls -d /root/packages/qemu-$REL/qemu-*/debian/patches`; \
|
||||||
|
echo qemu-5.1-vitastor.patch >> $P/series; \
|
||||||
|
fi; \
|
||||||
|
cd /root/packages/qemu-$REL/qemu-*/; \
|
||||||
|
quilt push -a; \
|
||||||
|
quilt add block/vitastor.c; \
|
||||||
|
cp /root/vitastor/patches/qemu_driver.c block/vitastor.c; \
|
||||||
|
quilt refresh; \
|
||||||
|
V=$(head -n1 debian/changelog | perl -pe 's/^.*\((.*?)(~bpo[\d\+]*)?\).*$/$1/')+vitastor1; \
|
||||||
|
DEBEMAIL="Vitaliy Filippov <vitalif@yourcmc.ru>" dch -D $REL -v $V 'Plug Vitastor block driver'; \
|
||||||
|
DEB_BUILD_OPTIONS=nocheck dpkg-buildpackage --jobs=auto -sa; \
|
||||||
|
rm -rf /root/packages/qemu-$REL/qemu-*/
|
||||||
Vendored
+1
@@ -0,0 +1 @@
|
|||||||
|
patches/PVE_VitastorPlugin.pm usr/share/perl5/PVE/Storage/Custom/VitastorPlugin.pm
|
||||||
Vendored
+19
@@ -0,0 +1,19 @@
|
|||||||
|
/* Removed in Linux 5.14 */
|
||||||
|
|
||||||
|
/* SPDX-License-Identifier: GPL-2.0 WITH Linux-syscall-note */
|
||||||
|
#ifndef __LINUX_RAW_H
|
||||||
|
#define __LINUX_RAW_H
|
||||||
|
|
||||||
|
#include <linux/types.h>
|
||||||
|
|
||||||
|
#define RAW_SETBIND _IO( 0xac, 0 )
|
||||||
|
#define RAW_GETBIND _IO( 0xac, 1 )
|
||||||
|
|
||||||
|
struct raw_config_request
|
||||||
|
{
|
||||||
|
int raw_minor;
|
||||||
|
__u64 block_major;
|
||||||
|
__u64 block_minor;
|
||||||
|
};
|
||||||
|
|
||||||
|
#endif /* __LINUX_RAW_H */
|
||||||
Vendored
+2
-1
@@ -5,5 +5,6 @@ export DH_VERBOSE = 1
|
|||||||
dh $@
|
dh $@
|
||||||
|
|
||||||
override_dh_installdeb:
|
override_dh_installdeb:
|
||||||
cat debian/substvars >> debian/vitastor.substvars
|
cat debian/fio_version >> debian/vitastor-fio.substvars
|
||||||
|
[ -f debian/qemu_version ] && (cat debian/qemu_version >> debian/vitastor-qemu.substvars) || true
|
||||||
dh_installdeb
|
dh_installdeb
|
||||||
|
|||||||
Vendored
-2
@@ -1,2 +0,0 @@
|
|||||||
dep:fio=3.16-1
|
|
||||||
dep:qemu=1:5.1+dfsg-4+vitastor1
|
|
||||||
Vendored
-86
@@ -1,86 +0,0 @@
|
|||||||
# Build packages for Debian Bullseye/Sid inside a container
|
|
||||||
# cd ..; podman build -t vitastor-bullseye -v `pwd`/build:/root/build -f debian/vitastor-bullseye.Dockerfile .
|
|
||||||
|
|
||||||
ARG REL=bullseye
|
|
||||||
|
|
||||||
FROM debian:$REL
|
|
||||||
|
|
||||||
# again, it doesn't work otherwise
|
|
||||||
ARG REL=bullseye
|
|
||||||
|
|
||||||
WORKDIR /root
|
|
||||||
|
|
||||||
RUN grep '^deb ' /etc/apt/sources.list | perl -pe 's/^deb/deb-src/' >> /etc/apt/sources.list; \
|
|
||||||
echo 'APT::Install-Recommends false;' > /etc/apt/apt.conf
|
|
||||||
|
|
||||||
RUN apt-get update
|
|
||||||
RUN apt-get -y install qemu fio liburing1 liburing-dev libgoogle-perftools-dev devscripts
|
|
||||||
RUN apt-get -y build-dep qemu
|
|
||||||
RUN apt-get -y build-dep fio
|
|
||||||
RUN apt-get --download-only source qemu
|
|
||||||
RUN apt-get --download-only source fio
|
|
||||||
|
|
||||||
ADD qemu-5.0-vitastor.patch qemu-5.1-vitastor.patch /root/vitastor/
|
|
||||||
RUN set -e; \
|
|
||||||
mkdir -p /root/build/qemu-$REL; \
|
|
||||||
rm -rf /root/build/qemu-$REL/*; \
|
|
||||||
cd /root/build/qemu-$REL; \
|
|
||||||
dpkg-source -x /root/qemu*.dsc; \
|
|
||||||
if [ -d /root/build/qemu-$REL/qemu-5.0 ]; then \
|
|
||||||
cp /root/vitastor/qemu-5.0-vitastor.patch /root/build/qemu-$REL/qemu-5.0/debian/patches; \
|
|
||||||
echo qemu-5.0-vitastor.patch >> /root/build/qemu-$REL/qemu-5.0/debian/patches/series; \
|
|
||||||
else \
|
|
||||||
cp /root/vitastor/qemu-5.1-vitastor.patch /root/build/qemu-$REL/qemu-*/debian/patches; \
|
|
||||||
P=`ls -d /root/build/qemu-$REL/qemu-*/debian/patches`; \
|
|
||||||
echo qemu-5.1-vitastor.patch >> $P/series; \
|
|
||||||
fi; \
|
|
||||||
cd /root/build/qemu-$REL/qemu-*/; \
|
|
||||||
V=$(head -n1 debian/changelog | perl -pe 's/^.*\((.*?)(~bpo[\d\+]*)?\).*$/$1/')+vitastor1; \
|
|
||||||
echo ">>> VERSION: $V"; \
|
|
||||||
DEBFULLNAME="Vitaliy Filippov <vitalif@yourcmc.ru>" dch -D $REL -v $V 'Plug Vitastor block driver'; \
|
|
||||||
DEB_BUILD_OPTIONS=nocheck dpkg-buildpackage --jobs=auto -sa; \
|
|
||||||
rm -rf /root/build/qemu-$REL/qemu-*/
|
|
||||||
|
|
||||||
RUN cd /root/build/qemu-$REL && apt-get -y install ./qemu-system-data*.deb ./qemu-system-common_*.deb ./qemu-system-x86_*.deb ./qemu_*.deb
|
|
||||||
|
|
||||||
ADD . /root/vitastor
|
|
||||||
RUN set -e -x; \
|
|
||||||
mkdir -p /root/fio-build/; \
|
|
||||||
cd /root/fio-build/; \
|
|
||||||
rm -rf /root/fio-build/*; \
|
|
||||||
dpkg-source -x /root/fio*.dsc; \
|
|
||||||
cd /root/build/qemu-$REL/; \
|
|
||||||
rm -rf qemu*/; \
|
|
||||||
dpkg-source -x qemu*.dsc; \
|
|
||||||
cd /root/build/qemu-$REL/qemu*/; \
|
|
||||||
debian/rules b/configure-stamp; \
|
|
||||||
cd b/qemu; \
|
|
||||||
make -j8 qapi; \
|
|
||||||
mkdir -p /root/build/vitastor-$REL; \
|
|
||||||
rm -rf /root/build/vitastor-$REL/*; \
|
|
||||||
cd /root/build/vitastor-$REL; \
|
|
||||||
cp -r /root/vitastor vitastor-0.5; \
|
|
||||||
ln -s /root/build/qemu-$REL/qemu-*/ vitastor-0.5/qemu; \
|
|
||||||
ln -s /root/fio-build/fio-*/ vitastor-0.5/fio; \
|
|
||||||
cd vitastor-0.5; \
|
|
||||||
FIO=$(head -n1 fio/debian/changelog | perl -pe 's/^.*\((.*?)\).*$/$1/'); \
|
|
||||||
QEMU=$(head -n1 qemu/debian/changelog | perl -pe 's/^.*\((.*?)\).*$/$1/'); \
|
|
||||||
sh copy-qemu-includes.sh; \
|
|
||||||
sh copy-fio-includes.sh; \
|
|
||||||
rm qemu fio; \
|
|
||||||
mkdir -p a b debian/patches; \
|
|
||||||
mv qemu-copy b/qemu; \
|
|
||||||
mv fio-copy b/fio; \
|
|
||||||
diff -NaurpbB a b > debian/patches/qemu-fio-headers.patch || true; \
|
|
||||||
echo qemu-fio-headers.patch >> debian/patches/series; \
|
|
||||||
rm -rf a b; \
|
|
||||||
rm -rf /root/build/qemu-$REL/qemu*/; \
|
|
||||||
echo "dep:fio=$FIO" > debian/substvars; \
|
|
||||||
echo "dep:qemu=$QEMU" >> debian/substvars; \
|
|
||||||
cd /root/build/vitastor-$REL; \
|
|
||||||
tar --sort=name --mtime='2020-01-01' --owner=0 --group=0 --exclude=debian -cJf vitastor_0.5.orig.tar.xz vitastor-0.5; \
|
|
||||||
cd vitastor-0.5; \
|
|
||||||
V=$(head -n1 debian/changelog | perl -pe 's/^.*\((.*?)\).*$/$1/'); \
|
|
||||||
DEBFULLNAME="Vitaliy Filippov <vitalif@yourcmc.ru>" dch -D $REL -v "$V""$REL" "Rebuild for $REL"; \
|
|
||||||
DEB_BUILD_OPTIONS=nocheck dpkg-buildpackage --jobs=auto -sa; \
|
|
||||||
rm -rf /root/build/vitastor-$REL/vitastor-*/
|
|
||||||
Vendored
-80
@@ -1,80 +0,0 @@
|
|||||||
# Build packages for Debian 10 inside a container
|
|
||||||
# cd ..; podman build -t vitastor-buster -v `pwd`/build:/root/build -f debian/vitastor-buster.Dockerfile .
|
|
||||||
|
|
||||||
FROM debian:buster
|
|
||||||
|
|
||||||
WORKDIR /root
|
|
||||||
|
|
||||||
RUN echo 'deb http://deb.debian.org/debian buster-backports main' >> /etc/apt/sources.list; \
|
|
||||||
grep '^deb ' /etc/apt/sources.list | perl -pe 's/^deb/deb-src/' >> /etc/apt/sources.list; \
|
|
||||||
echo 'APT::Install-Recommends false;' > /etc/apt/apt.conf
|
|
||||||
|
|
||||||
RUN apt-get update
|
|
||||||
RUN apt-get -t buster-backports -y install qemu fio liburing1 liburing-dev libgoogle-perftools-dev devscripts
|
|
||||||
RUN apt-get -t buster-backports -y build-dep qemu
|
|
||||||
RUN apt-get -y build-dep fio
|
|
||||||
RUN apt-get -t buster-backports --download-only source qemu-kvm
|
|
||||||
RUN apt-get --download-only source fio
|
|
||||||
|
|
||||||
ADD qemu-5.0-vitastor.patch qemu-5.1-vitastor.patch /root/vitastor/
|
|
||||||
RUN set -e; \
|
|
||||||
mkdir -p /root/build/qemu-buster; \
|
|
||||||
rm -rf /root/build/qemu-buster/*; \
|
|
||||||
cd /root/build/qemu-buster; \
|
|
||||||
dpkg-source -x /root/qemu*.dsc; \
|
|
||||||
if [ -d /root/build/qemu-buster/qemu-5.0 ]; then \
|
|
||||||
cp /root/vitastor/qemu-5.0-vitastor.patch /root/build/qemu-buster/qemu-5.0/debian/patches; \
|
|
||||||
echo qemu-5.0-vitastor.patch >> /root/build/qemu-buster/qemu-5.0/debian/patches/series; \
|
|
||||||
else \
|
|
||||||
cp /root/vitastor/qemu-5.1-vitastor.patch /root/build/qemu-buster/qemu-*/debian/patches; \
|
|
||||||
echo qemu-5.1-vitastor.patch >> /root/build/qemu-buster/qemu-*/debian/patches/series; \
|
|
||||||
fi; \
|
|
||||||
cd /root/build/qemu-buster/qemu-*/; \
|
|
||||||
V=$(head -n1 debian/changelog | perl -pe 's/^.*\((.*?)(~bpo[\d\+]*)\).*$/$1/')+vitastor1; \
|
|
||||||
DEBFULLNAME="Vitaliy Filippov <vitalif@yourcmc.ru>" dch -D buster -v $V 'Plug Vitastor block driver'; \
|
|
||||||
DEB_BUILD_OPTIONS=nocheck dpkg-buildpackage --jobs=auto -sa; \
|
|
||||||
rm -rf /root/build/qemu-buster/qemu-*/
|
|
||||||
|
|
||||||
RUN cd /root/build/qemu-buster && apt-get -y -t buster-backports install ./qemu-system-data*.deb ./qemu-system-common_*.deb ./qemu-system-x86_*.deb ./qemu_*.deb
|
|
||||||
|
|
||||||
ADD . /root/vitastor
|
|
||||||
RUN set -e -x; \
|
|
||||||
mkdir -p /root/fio-build/; \
|
|
||||||
cd /root/fio-build/; \
|
|
||||||
rm -rf /root/fio-build/*; \
|
|
||||||
dpkg-source -x /root/fio*.dsc; \
|
|
||||||
cd /root/build/qemu-buster/; \
|
|
||||||
rm -rf qemu*/; \
|
|
||||||
dpkg-source -x qemu*.dsc; \
|
|
||||||
cd /root/build/qemu-buster/qemu*/; \
|
|
||||||
debian/rules b/configure-stamp; \
|
|
||||||
cd b/qemu; \
|
|
||||||
make -j8 qapi; \
|
|
||||||
mkdir -p /root/build/vitastor-buster; \
|
|
||||||
rm -rf /root/build/vitastor-buster/*; \
|
|
||||||
cd /root/build/vitastor-buster; \
|
|
||||||
cp -r /root/vitastor vitastor-0.5; \
|
|
||||||
ln -s /root/build/qemu-buster/qemu-*/ vitastor-0.5/qemu; \
|
|
||||||
ln -s /root/fio-build/fio-*/ vitastor-0.5/fio; \
|
|
||||||
cd vitastor-0.5; \
|
|
||||||
FIO=$(head -n1 fio/debian/changelog | perl -pe 's/^.*\((.*?)\).*$/$1/'); \
|
|
||||||
QEMU=$(head -n1 qemu/debian/changelog | perl -pe 's/^.*\((.*?)\).*$/$1/'); \
|
|
||||||
sh copy-qemu-includes.sh; \
|
|
||||||
sh copy-fio-includes.sh; \
|
|
||||||
rm qemu fio; \
|
|
||||||
mkdir -p a b debian/patches; \
|
|
||||||
mv qemu-copy b/qemu; \
|
|
||||||
mv fio-copy b/fio; \
|
|
||||||
diff -NaurpbB a b > debian/patches/qemu-fio-headers.patch || true; \
|
|
||||||
echo qemu-fio-headers.patch >> debian/patches/series; \
|
|
||||||
rm -rf a b; \
|
|
||||||
rm -rf /root/build/qemu-buster/qemu*/; \
|
|
||||||
echo "dep:fio=$FIO" > debian/substvars; \
|
|
||||||
echo "dep:qemu=$QEMU" >> debian/substvars; \
|
|
||||||
cd /root/build/vitastor-buster; \
|
|
||||||
tar --sort=name --mtime='2020-01-01' --owner=0 --group=0 --exclude=debian -cJf vitastor_0.5.orig.tar.xz vitastor-0.5; \
|
|
||||||
cd vitastor-0.5; \
|
|
||||||
V=$(head -n1 debian/changelog | perl -pe 's/^.*\((.*?)\).*$/$1/'); \
|
|
||||||
DEBFULLNAME="Vitaliy Filippov <vitalif@yourcmc.ru>" dch -D buster -v "$V""buster" "Rebuild for buster"; \
|
|
||||||
DEB_BUILD_OPTIONS=nocheck dpkg-buildpackage --jobs=auto -sa; \
|
|
||||||
rm -rf /root/build/vitastor-buster/vitastor-*/
|
|
||||||
Vendored
+2
@@ -0,0 +1,2 @@
|
|||||||
|
usr/include
|
||||||
|
usr/lib/*/pkgconfig
|
||||||
Vendored
+6
@@ -0,0 +1,6 @@
|
|||||||
|
usr/bin/vita
|
||||||
|
usr/bin/vitastor-cli
|
||||||
|
usr/bin/vitastor-rm
|
||||||
|
usr/bin/vitastor-nbd
|
||||||
|
usr/lib/*/libvitastor*.so*
|
||||||
|
mon/make-osd.sh /usr/lib/vitastor
|
||||||
Vendored
+1
@@ -0,0 +1 @@
|
|||||||
|
usr/lib/*/libfio*.so*
|
||||||
Vendored
+1
@@ -0,0 +1 @@
|
|||||||
|
mon usr/lib/vitastor
|
||||||
Vendored
+2
@@ -0,0 +1,2 @@
|
|||||||
|
usr/bin/vitastor-osd
|
||||||
|
usr/bin/vitastor-dump-journal
|
||||||
Vendored
+55
@@ -0,0 +1,55 @@
|
|||||||
|
# Build Vitastor packages for Debian Buster or Bullseye/Sid inside a container
|
||||||
|
# cd ..; podman build --build-arg REL=bullseye -v `pwd`/packages:/root/packages -f debian/vitastor.Dockerfile .
|
||||||
|
|
||||||
|
ARG REL=
|
||||||
|
FROM debian:$REL
|
||||||
|
ARG REL=
|
||||||
|
|
||||||
|
WORKDIR /root
|
||||||
|
|
||||||
|
RUN if [ "$REL" = "buster" -o "$REL" = "bullseye" ]; then \
|
||||||
|
echo "deb http://deb.debian.org/debian $REL-backports main" >> /etc/apt/sources.list; \
|
||||||
|
echo >> /etc/apt/preferences; \
|
||||||
|
echo 'Package: *' >> /etc/apt/preferences; \
|
||||||
|
echo "Pin: release a=$REL-backports" >> /etc/apt/preferences; \
|
||||||
|
echo 'Pin-Priority: 500' >> /etc/apt/preferences; \
|
||||||
|
fi; \
|
||||||
|
grep '^deb ' /etc/apt/sources.list | perl -pe 's/^deb/deb-src/' >> /etc/apt/sources.list; \
|
||||||
|
echo 'APT::Install-Recommends false;' >> /etc/apt/apt.conf; \
|
||||||
|
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
|
||||||
|
|
||||||
|
RUN apt-get update
|
||||||
|
RUN apt-get -y install fio liburing1 liburing-dev libgoogle-perftools-dev devscripts
|
||||||
|
RUN apt-get -y build-dep fio
|
||||||
|
RUN apt-get --download-only source fio
|
||||||
|
RUN apt-get update && apt-get -y install libjerasure-dev cmake libibverbs-dev
|
||||||
|
|
||||||
|
ADD . /root/vitastor
|
||||||
|
RUN set -e -x; \
|
||||||
|
mkdir -p /root/fio-build/; \
|
||||||
|
cd /root/fio-build/; \
|
||||||
|
rm -rf /root/fio-build/*; \
|
||||||
|
dpkg-source -x /root/fio*.dsc; \
|
||||||
|
mkdir -p /root/packages/vitastor-$REL; \
|
||||||
|
rm -rf /root/packages/vitastor-$REL/*; \
|
||||||
|
cd /root/packages/vitastor-$REL; \
|
||||||
|
cp -r /root/vitastor vitastor-0.6.16; \
|
||||||
|
cd vitastor-0.6.16; \
|
||||||
|
ln -s /root/fio-build/fio-*/ ./fio; \
|
||||||
|
FIO=$(head -n1 fio/debian/changelog | perl -pe 's/^.*\((.*?)\).*$/$1/'); \
|
||||||
|
ls /usr/include/linux/raw.h || cp ./debian/raw.h /usr/include/linux/raw.h; \
|
||||||
|
sh copy-fio-includes.sh; \
|
||||||
|
rm fio; \
|
||||||
|
mkdir -p a b debian/patches; \
|
||||||
|
mv fio-copy b/fio; \
|
||||||
|
diff -NaurpbB a b > debian/patches/fio-headers.patch || true; \
|
||||||
|
echo fio-headers.patch >> debian/patches/series; \
|
||||||
|
rm -rf a b; \
|
||||||
|
echo "dep:fio=$FIO" > debian/fio_version; \
|
||||||
|
cd /root/packages/vitastor-$REL; \
|
||||||
|
tar --sort=name --mtime='2020-01-01' --owner=0 --group=0 --exclude=debian -cJf vitastor_0.6.16.orig.tar.xz vitastor-0.6.16; \
|
||||||
|
cd vitastor-0.6.16; \
|
||||||
|
V=$(head -n1 debian/changelog | perl -pe 's/^.*\((.*?)\).*$/$1/'); \
|
||||||
|
DEBFULLNAME="Vitaliy Filippov <vitalif@yourcmc.ru>" dch -D $REL -v "$V""$REL" "Rebuild for $REL"; \
|
||||||
|
DEB_BUILD_OPTIONS=nocheck dpkg-buildpackage --jobs=auto -sa; \
|
||||||
|
rm -rf /root/packages/vitastor-$REL/vitastor-*/
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
# Build Docker image with Vitastor packages
|
||||||
|
|
||||||
|
FROM debian:bullseye
|
||||||
|
|
||||||
|
ADD vitastor.list /etc/apt/sources.list.d
|
||||||
|
ADD vitastor.gpg /etc/apt/trusted.gpg.d
|
||||||
|
ADD vitastor.pref /etc/apt/preferences.d
|
||||||
|
ADD apt.conf /etc/apt/
|
||||||
|
RUN apt-get update && apt-get -y install vitastor qemu-system-x86 qemu-system-common && apt-get clean
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
APT::Install-Recommends false;
|
||||||
Binary file not shown.
@@ -0,0 +1 @@
|
|||||||
|
deb http://vitastor.io/debian bullseye main
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
Package: *
|
||||||
|
Pin: origin "vitastor.io"
|
||||||
|
Pin-Priority: 1000
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
- name: config_path
|
||||||
|
type: string
|
||||||
|
default: "/etc/vitastor/vitastor.conf"
|
||||||
|
info: |
|
||||||
|
Path to the JSON configuration file. Configuration file is optional,
|
||||||
|
a non-existing configuration file does not prevent Vitastor from
|
||||||
|
running if required parameters are specified.
|
||||||
|
info_ru: |
|
||||||
|
Путь к файлу конфигурации в формате JSON. Файл конфигурации необязателен,
|
||||||
|
без него Vitastor тоже будет работать, если переданы необходимые параметры.
|
||||||
|
- name: etcd_address
|
||||||
|
type: string or array of strings
|
||||||
|
type_ru: строка или массив строк
|
||||||
|
info: |
|
||||||
|
etcd connection endpoint(s). Multiple endpoints may be delimited by "," or
|
||||||
|
specified in a JSON array `["10.0.115.10:2379/v3","10.0.115.11:2379/v3"]`.
|
||||||
|
Note that https is not supported for etcd connections yet.
|
||||||
|
info_ru: |
|
||||||
|
Адрес(а) подключения к etcd. Несколько адресов могут разделяться запятой
|
||||||
|
или указываться в виде JSON-массива `["10.0.115.10:2379/v3","10.0.115.11:2379/v3"]`.
|
||||||
|
- name: etcd_prefix
|
||||||
|
type: string
|
||||||
|
default: "/vitastor"
|
||||||
|
info: |
|
||||||
|
Prefix for all keys in etcd used by Vitastor. You can change prefix and, for
|
||||||
|
example, use a single etcd cluster for multiple Vitastor clusters.
|
||||||
|
info_ru: |
|
||||||
|
Префикс для ключей etcd, которые использует Vitastor. Вы можете задать другой
|
||||||
|
префикс, например, чтобы запустить несколько кластеров Vitastor с одним
|
||||||
|
кластером etcd.
|
||||||
|
- name: log_level
|
||||||
|
type: int
|
||||||
|
default: 0
|
||||||
|
info: Log level. Raise if you want more verbose output.
|
||||||
|
info_ru: Уровень логгирования. Повысьте, если хотите более подробный вывод.
|
||||||
@@ -0,0 +1,200 @@
|
|||||||
|
- name: block_size
|
||||||
|
type: int
|
||||||
|
default: 131072
|
||||||
|
info: |
|
||||||
|
Size of objects (data blocks) into which all physical and virtual drives are
|
||||||
|
subdivided in Vitastor. One of current main settings in Vitastor, affects
|
||||||
|
memory usage, write amplification and I/O load distribution effectiveness.
|
||||||
|
|
||||||
|
Recommended default block size is 128 KB for SSD and 4 MB for HDD. In fact,
|
||||||
|
it's possible to use 4 MB for SSD too - it will lower memory usage, but
|
||||||
|
may increase average WA and reduce linear performance.
|
||||||
|
|
||||||
|
OSDs with different block sizes (for example, SSD and SSD+HDD OSDs) can
|
||||||
|
currently coexist in one etcd instance only within separate Vitastor
|
||||||
|
clusters with different etcd_prefix'es.
|
||||||
|
|
||||||
|
Also block size can't be changed after OSD initialization without losing
|
||||||
|
data.
|
||||||
|
|
||||||
|
You must always specify block_size in etcd in /vitastor/config/global if
|
||||||
|
you change it so all clients can know about it.
|
||||||
|
|
||||||
|
OSD memory usage is roughly (SIZE / BLOCK * 68 bytes) which is roughly
|
||||||
|
544 MB per 1 TB of used disk space with the default 128 KB block size.
|
||||||
|
info_ru: |
|
||||||
|
Размер объектов (блоков данных), на которые делятся физические и виртуальные
|
||||||
|
диски в Vitastor. Одна из ключевых на данный момент настроек, влияет на
|
||||||
|
потребление памяти, объём избыточной записи (write amplification) и
|
||||||
|
эффективность распределения нагрузки по OSD.
|
||||||
|
|
||||||
|
Рекомендуемые по умолчанию размеры блока - 128 килобайт для SSD и 4
|
||||||
|
мегабайта для HDD. В принципе, для SSD можно тоже использовать 4 мегабайта,
|
||||||
|
это понизит использование памяти, но ухудшит распределение нагрузки и в
|
||||||
|
среднем увеличит WA.
|
||||||
|
|
||||||
|
OSD с разными размерами блока (например, SSD и SSD+HDD OSD) на данный
|
||||||
|
момент могут сосуществовать в рамках одного etcd только в виде двух независимых
|
||||||
|
кластеров Vitastor с разными etcd_prefix.
|
||||||
|
|
||||||
|
Также размер блока нельзя менять после инициализации OSD без потери данных.
|
||||||
|
|
||||||
|
Если вы меняете размер блока, обязательно прописывайте его в etcd в
|
||||||
|
/vitastor/config/global, дабы все клиенты его знали.
|
||||||
|
|
||||||
|
Потребление памяти OSD составляет примерно (РАЗМЕР / БЛОК * 68 байт),
|
||||||
|
т.е. примерно 544 МБ памяти на 1 ТБ занятого места на диске при
|
||||||
|
стандартном 128 КБ блоке.
|
||||||
|
- name: bitmap_granularity
|
||||||
|
type: int
|
||||||
|
default: 4096
|
||||||
|
info: |
|
||||||
|
Required virtual disk write alignment ("sector size"). Must be a multiple
|
||||||
|
of disk_alignment. It's called bitmap granularity because Vitastor tracks
|
||||||
|
an allocation bitmap for each object containing 2 bits per each
|
||||||
|
(bitmap_granularity) bytes.
|
||||||
|
|
||||||
|
This parameter can't be changed after OSD initialization without losing
|
||||||
|
data. Also it's fixed for the whole Vitastor cluster i.e. two different
|
||||||
|
values can't be used in a single Vitastor cluster.
|
||||||
|
|
||||||
|
Clients MUST be aware of this parameter value, so put it into etcd key
|
||||||
|
/vitastor/config/global if you change it for any reason.
|
||||||
|
info_ru: |
|
||||||
|
Требуемое выравнивание записи на виртуальные диски (размер их "сектора").
|
||||||
|
Должен быть кратен disk_alignment. Называется гранулярностью битовой карты
|
||||||
|
потому, что Vitastor хранит битовую карту для каждого объекта, содержащую
|
||||||
|
по 2 бита на каждые (bitmap_granularity) байт.
|
||||||
|
|
||||||
|
Данный параметр нельзя менять после инициализации OSD без потери данных.
|
||||||
|
Также он фиксирован для всего кластера Vitastor, т.е. разные значения
|
||||||
|
не могут сосуществовать в одном кластере.
|
||||||
|
|
||||||
|
Клиенты ДОЛЖНЫ знать правильное значение этого параметра, так что если вы
|
||||||
|
его меняете, обязательно прописывайте изменённое значение в etcd в ключ
|
||||||
|
/vitastor/config/global.
|
||||||
|
- name: immediate_commit
|
||||||
|
type: string
|
||||||
|
default: false
|
||||||
|
info: |
|
||||||
|
Another parameter which is really important for performance.
|
||||||
|
|
||||||
|
Desktop SSDs are very fast (100000+ iops) for simple random writes
|
||||||
|
without cache flush. However, they are really slow (only around 1000 iops)
|
||||||
|
if you try to fsync() each write, that is, when you want to guarantee that
|
||||||
|
each change gets immediately persisted to the physical media.
|
||||||
|
|
||||||
|
Server-grade SSDs with "Advanced/Enhanced Power Loss Protection" or with
|
||||||
|
"Supercapacitor-based Power Loss Protection", on the other hand, are equally
|
||||||
|
fast with and without fsync because their cache is protected from sudden
|
||||||
|
power loss by a built-in supercapacitor-based "UPS".
|
||||||
|
|
||||||
|
Some software-defined storage systems always fsync each write and thus are
|
||||||
|
really slow when used with desktop SSDs. Vitastor, however, can also
|
||||||
|
efficiently utilize desktop SSDs by postponing fsync until the client calls
|
||||||
|
it explicitly.
|
||||||
|
|
||||||
|
This is what this parameter regulates. When it's set to "all" the whole
|
||||||
|
Vitastor cluster commits each change to disks immediately and clients just
|
||||||
|
ignore fsyncs because they know for sure that they're unneeded. This reduces
|
||||||
|
the amount of network roundtrips performed by clients and improves
|
||||||
|
performance. So it's always better to use server grade SSDs with
|
||||||
|
supercapacitors even with Vitastor, especially given that they cost only
|
||||||
|
a bit more than desktop models.
|
||||||
|
|
||||||
|
There is also a common SATA SSD (and HDD too!) firmware bug (or feature)
|
||||||
|
that makes server SSDs which have supercapacitors slow with fsync. To check
|
||||||
|
if your SSDs are affected, compare benchmark results from `fio -name=test
|
||||||
|
-ioengine=libaio -direct=1 -bs=4k -rw=randwrite -iodepth=1` with and without
|
||||||
|
`-fsync=1`. Results should be the same. If fsync=1 result is worse you can
|
||||||
|
try to work around this bug by "disabling" drive write-back cache by running
|
||||||
|
`hdparm -W 0 /dev/sdXX` or `echo write through > /sys/block/sdXX/device/scsi_disk/*/cache_type`
|
||||||
|
(IMPORTANT: don't mistake it with `/sys/block/sdXX/queue/write_cache` - it's
|
||||||
|
unsafe to change by hand). The same may apply to newer HDDs with internal
|
||||||
|
SSD cache or "media-cache" - for example, a lot of Seagate EXOS drives have
|
||||||
|
it (they have internal SSD cache even though it's not stated in datasheets).
|
||||||
|
|
||||||
|
This parameter must be set both in etcd in /vitastor/config/global and in
|
||||||
|
OSD command line or configuration. Setting it to "all" or "small" requires
|
||||||
|
enabling disable_journal_fsync and disable_meta_fsync, setting it to "all"
|
||||||
|
also requires enabling disable_data_fsync.
|
||||||
|
|
||||||
|
TLDR: For optimal performance, set immediate_commit to "all" if you only use
|
||||||
|
SSDs with supercapacitor-based power loss protection (nonvolatile
|
||||||
|
write-through cache) for both data and journals in the whole Vitastor
|
||||||
|
cluster. Set it to "small" if you only use such SSDs for journals. Leave
|
||||||
|
empty if your drives have write-back cache.
|
||||||
|
info_ru: |
|
||||||
|
Ещё один важный для производительности параметр.
|
||||||
|
|
||||||
|
Модели SSD для настольных компьютеров очень быстрые (100000+ операций в
|
||||||
|
секунду) при простой случайной записи без сбросов кэша. Однако они очень
|
||||||
|
медленные (всего порядка 1000 iops), если вы пытаетесь сбрасывать кэш после
|
||||||
|
каждой записи, то есть, если вы пытаетесь гарантировать, что каждое
|
||||||
|
изменение физически записывается в энергонезависимую память.
|
||||||
|
|
||||||
|
С другой стороны, серверные SSD с конденсаторами - функцией, называемой
|
||||||
|
"Advanced/Enhanced Power Loss Protection" или просто "Supercapacitor-based
|
||||||
|
Power Loss Protection" - одинаково быстрые и со сбросом кэша, и без
|
||||||
|
него, потому что их кэш защищён от потери питания встроенным "источником
|
||||||
|
бесперебойного питания" на основе суперконденсаторов и на самом деле они
|
||||||
|
его никогда не сбрасывают.
|
||||||
|
|
||||||
|
Некоторые программные СХД всегда сбрасывают кэши дисков при каждой записи
|
||||||
|
и поэтому работают очень медленно с настольными SSD. Vitastor, однако, может
|
||||||
|
откладывать fsync до явного его вызова со стороны клиента и таким образом
|
||||||
|
эффективно утилизировать настольные SSD.
|
||||||
|
|
||||||
|
Данный параметр влияет как раз на это. Когда он установлен в значение "all",
|
||||||
|
весь кластер Vitastor мгновенно фиксирует каждое изменение на физические
|
||||||
|
носители и клиенты могут просто игнорировать запросы fsync, т.к. они точно
|
||||||
|
знают, что fsync-и не нужны. Это уменьшает число необходимых обращений к OSD
|
||||||
|
по сети и улучшает производительность. Поэтому даже с Vitastor лучше всегда
|
||||||
|
использовать только серверные модели SSD с суперконденсаторами, особенно
|
||||||
|
учитывая то, что стоят они ненамного дороже настольных.
|
||||||
|
|
||||||
|
Также в прошивках SATA SSD (и даже HDD!) очень часто встречается либо баг,
|
||||||
|
либо просто особенность логики, из-за которой серверные SSD, имеющие
|
||||||
|
конденсаторы и защиту от потери питания, всё равно медленно работают с
|
||||||
|
fsync. Чтобы понять, подвержены ли этой проблеме ваши SSD, сравните
|
||||||
|
результаты тестов `fio -name=test -ioengine=libaio -direct=1 -bs=4k
|
||||||
|
-rw=randwrite -iodepth=1` без и с опцией `-fsync=1`. Результаты должны
|
||||||
|
быть одинаковые. Если результат с `fsync=1` хуже, вы можете попробовать
|
||||||
|
обойти проблему, "отключив" кэш записи диска командой `hdparm -W 0 /dev/sdXX`
|
||||||
|
либо `echo write through > /sys/block/sdXX/device/scsi_disk/*/cache_type`
|
||||||
|
(ВАЖНО: не перепутайте с `/sys/block/sdXX/queue/write_cache` - этот параметр
|
||||||
|
менять руками небезопасно). Такая же проблема может встречаться и в новых
|
||||||
|
HDD-дисках с внутренним SSD или "медиа" кэшем - например, она встречается во
|
||||||
|
многих дисках Seagate EXOS (у них есть внутренний SSD-кэш, хотя это и не
|
||||||
|
указано в спецификациях).
|
||||||
|
|
||||||
|
Данный параметр нужно указывать и в etcd в /vitastor/config/global, и в
|
||||||
|
командной строке или конфигурации OSD. Значения "all" и "small" требуют
|
||||||
|
включения disable_journal_fsync и disable_meta_fsync, значение "all" также
|
||||||
|
требует включения disable_data_fsync.
|
||||||
|
|
||||||
|
Итого, вкратце: для оптимальной производительности установите
|
||||||
|
immediate_commit в значение "all", если вы используете в кластере только SSD
|
||||||
|
с суперконденсаторами и для данных, и для журналов. Если вы используете
|
||||||
|
такие SSD для всех журналов, но не для данных - можете установить параметр
|
||||||
|
в "small". Если и какие-то из дисков журналов имеют волатильный кэш записи -
|
||||||
|
оставьте параметр пустым.
|
||||||
|
- name: client_dirty_limit
|
||||||
|
type: int
|
||||||
|
default: 33554432
|
||||||
|
info: |
|
||||||
|
Without immediate_commit=all this parameter sets the limit of "dirty"
|
||||||
|
(not committed by fsync) data allowed by the client before forcing an
|
||||||
|
additional fsync and committing the data. Also note that the client always
|
||||||
|
holds a copy of uncommitted data in memory so this setting also affects
|
||||||
|
RAM usage of clients.
|
||||||
|
|
||||||
|
This parameter doesn't affect OSDs themselves.
|
||||||
|
info_ru: |
|
||||||
|
При работе без immediate_commit=all - это лимит объёма "грязных" (не
|
||||||
|
зафиксированных fsync-ом) данных, при достижении которого клиент будет
|
||||||
|
принудительно вызывать fsync и фиксировать данные. Также стоит иметь в виду,
|
||||||
|
что в этом случае до момента fsync клиент хранит копию незафиксированных
|
||||||
|
данных в памяти, то есть, настройка влияет на потребление памяти клиентами.
|
||||||
|
|
||||||
|
Параметр не влияет на сами OSD.
|
||||||
@@ -0,0 +1,205 @@
|
|||||||
|
- name: data_device
|
||||||
|
type: string
|
||||||
|
info: |
|
||||||
|
Path to the block device to use for data. It's highly recommendded to use
|
||||||
|
stable paths for all device names: `/dev/disk/by-partuuid/xxx...` instead
|
||||||
|
of just `/dev/sda` or `/dev/nvme0n1` to not mess up after server restart.
|
||||||
|
Files can also be used instead of block devices, but this is implemented
|
||||||
|
only for testing purposes and not for production.
|
||||||
|
info_ru: |
|
||||||
|
Путь к диску (блочному устройству) для хранения данных. Крайне рекомендуется
|
||||||
|
использовать стабильные пути: `/dev/disk/by-partuuid/xxx...` вместо простых
|
||||||
|
`/dev/sda` или `/dev/nvme0n1`, чтобы пути не могли спутаться после
|
||||||
|
перезагрузки сервера. Также вместо блочных устройств можно указывать файлы,
|
||||||
|
но это реализовано только для тестирования, а не для боевой среды.
|
||||||
|
- name: meta_device
|
||||||
|
type: string
|
||||||
|
info: |
|
||||||
|
Path to the block device to use for the metadata. Metadata must be on a fast
|
||||||
|
SSD or performance will suffer. If this option is skipped, `data_device` is
|
||||||
|
used for the metadata.
|
||||||
|
info_ru: |
|
||||||
|
Путь к диску метаданных. Метаданные должны располагаться на быстром
|
||||||
|
SSD-диске, иначе производительность пострадает. Если эта опция не указана,
|
||||||
|
для метаданных используется `data_device`.
|
||||||
|
- name: journal_device
|
||||||
|
type: string
|
||||||
|
info: |
|
||||||
|
Path to the block device to use for the journal. Journal must be on a fast
|
||||||
|
SSD or performance will suffer. If this option is skipped, `meta_device` is
|
||||||
|
used for the journal, and if it's also empty, journal is put on
|
||||||
|
`data_device`. It's almost always fine to put metadata and journal on the
|
||||||
|
same device, in this case you only need to set `meta_device`.
|
||||||
|
info_ru: |
|
||||||
|
Путь к диску журнала. Журнал должен располагаться на быстром SSD-диске,
|
||||||
|
иначе производительность пострадает. Если эта опция не указана,
|
||||||
|
для журнала используется `meta_device`, если же пуста и она, журнал
|
||||||
|
располагается на `data_device`. Нормально располагать журнал и метаданные
|
||||||
|
на одном устройстве, в этом случае достаточно указать только `meta_device`.
|
||||||
|
- name: journal_offset
|
||||||
|
type: int
|
||||||
|
default: 0
|
||||||
|
info: Offset on the device in bytes where the journal is stored.
|
||||||
|
info_ru: Смещение на устройстве в байтах, по которому располагается журнал.
|
||||||
|
- name: journal_size
|
||||||
|
type: int
|
||||||
|
info: |
|
||||||
|
Journal size in bytes. Doesn't have to be large, 16-32 MB is usually fine.
|
||||||
|
By default, the whole journal device will be used for the journal. You must
|
||||||
|
set it to some value manually (or use make-osd.sh) if you colocate the
|
||||||
|
journal with data or metadata.
|
||||||
|
info_ru: |
|
||||||
|
Размер журнала в байтах. Большим быть не обязан, 16-32 МБ обычно достаточно.
|
||||||
|
По умолчанию для журнала используется всё устройство журнала. Если же вы
|
||||||
|
размещаете журнал на устройстве данных или метаданных, то вы должны
|
||||||
|
установить эту опцию в какое-то значение сами (или использовать скрипт
|
||||||
|
make-osd.sh).
|
||||||
|
- name: meta_offset
|
||||||
|
type: int
|
||||||
|
default: 0
|
||||||
|
info: |
|
||||||
|
Offset on the device in bytes where the metadata area is stored.
|
||||||
|
Again, set it to something if you colocate metadata with journal or data.
|
||||||
|
info_ru: |
|
||||||
|
Смещение на устройстве в байтах, по которому располагаются метаданные.
|
||||||
|
Эту опцию нужно задать, если метаданные у вас хранятся на том же
|
||||||
|
устройстве, что данные или журнал.
|
||||||
|
- name: data_offset
|
||||||
|
type: int
|
||||||
|
default: 0
|
||||||
|
info: |
|
||||||
|
Offset on the device in bytes where the data area is stored.
|
||||||
|
Again, set it to something if you colocate data with journal or metadata.
|
||||||
|
info_ru: |
|
||||||
|
Смещение на устройстве в байтах, по которому располагаются данные.
|
||||||
|
Эту опцию нужно задать, если данные у вас хранятся на том же
|
||||||
|
устройстве, что метаданные или журнал.
|
||||||
|
- name: data_size
|
||||||
|
type: int
|
||||||
|
info: |
|
||||||
|
Data area size in bytes. By default, the whole data device up to the end
|
||||||
|
will be used for the data area, but you can restrict it if you want to use
|
||||||
|
a smaller part. Note that there is no option to set metadata area size -
|
||||||
|
it's derived from the data area size.
|
||||||
|
info_ru: |
|
||||||
|
Размер области данных в байтах. По умолчанию под данные будет использована
|
||||||
|
вся доступная область устройства данных до конца устройства, но вы можете
|
||||||
|
использовать эту опцию, чтобы ограничить её меньшим размером. Заметьте, что
|
||||||
|
опции размера области метаданных нет - она вычисляется из размера области
|
||||||
|
данных автоматически.
|
||||||
|
- name: meta_block_size
|
||||||
|
type: int
|
||||||
|
default: 4096
|
||||||
|
info: |
|
||||||
|
Physical block size of the metadata device. 4096 for most current
|
||||||
|
HDDs and SSDs.
|
||||||
|
info_ru: |
|
||||||
|
Размер физического блока устройства метаданных. 4096 для большинства
|
||||||
|
современных SSD и HDD.
|
||||||
|
- name: journal_block_size
|
||||||
|
type: int
|
||||||
|
default: 4096
|
||||||
|
info: |
|
||||||
|
Physical block size of the journal device. Must be a multiple of
|
||||||
|
`disk_alignment`. 4096 for most current HDDs and SSDs.
|
||||||
|
info_ru: |
|
||||||
|
Размер физического блока устройства журнала. Должен быть кратен
|
||||||
|
`disk_alignment`. 4096 для большинства современных SSD и HDD.
|
||||||
|
- name: disable_data_fsync
|
||||||
|
type: bool
|
||||||
|
default: false
|
||||||
|
info: |
|
||||||
|
Do not issue fsyncs to the data device, i.e. do not flush its cache.
|
||||||
|
Safe ONLY if your data device has write-through cache. If you disable
|
||||||
|
the cache yourself using `hdparm` or `scsi_disk/cache_type` then make sure
|
||||||
|
that the cache disable command is run every time before starting Vitastor
|
||||||
|
OSD, for example, in the systemd unit. See also `immediate_commit` option
|
||||||
|
for the instructions to disable cache and how to benefit from it.
|
||||||
|
info_ru: |
|
||||||
|
Не отправлять fsync-и устройству данных, т.е. не сбрасывать его кэш.
|
||||||
|
Безопасно, ТОЛЬКО если ваше устройство данных имеет кэш со сквозной
|
||||||
|
записью (write-through). Если вы отключаете кэш через `hdparm` или
|
||||||
|
`scsi_disk/cache_type`, то удостоверьтесь, что команда отключения кэша
|
||||||
|
выполняется перед каждым запуском Vitastor OSD, например, в systemd unit-е.
|
||||||
|
Смотрите также опцию `immediate_commit` для инструкций по отключению кэша
|
||||||
|
и о том, как из этого извлечь выгоду.
|
||||||
|
- name: disable_meta_fsync
|
||||||
|
type: bool
|
||||||
|
default: false
|
||||||
|
info: |
|
||||||
|
Same as disable_data_fsync, but for the metadata device. If the metadata
|
||||||
|
device is not set or if the data device is used for the metadata the option
|
||||||
|
is ignored and disable_data_fsync value is used instead of it.
|
||||||
|
info_ru: |
|
||||||
|
То же, что disable_data_fsync, но для устройства метаданных. Если устройство
|
||||||
|
метаданных не задано или если оно равно устройству данных, значение опции
|
||||||
|
игнорируется и вместо него используется значение опции disable_data_fsync.
|
||||||
|
- name: disable_journal_fsync
|
||||||
|
type: bool
|
||||||
|
default: false
|
||||||
|
info: |
|
||||||
|
Same as disable_data_fsync, but for the journal device. If the journal
|
||||||
|
device is not set or if the metadata device is used for the journal the
|
||||||
|
option is ignored and disable_meta_fsync value is used instead of it. If
|
||||||
|
the same device is used for data, metadata and journal the option is also
|
||||||
|
ignored and disable_data_fsync value is used instead of it.
|
||||||
|
info_ru: |
|
||||||
|
То же, что disable_data_fsync, но для устройства журнала. Если устройство
|
||||||
|
журнала не задано или если оно равно устройству метаданных, значение опции
|
||||||
|
игнорируется и вместо него используется значение опции disable_meta_fsync.
|
||||||
|
Если одно и то же устройство используется и под данные, и под журнал, и под
|
||||||
|
метаданные - значение опции также игнорируется и вместо него используется
|
||||||
|
значение опции disable_data_fsync.
|
||||||
|
- name: disable_device_lock
|
||||||
|
type: bool
|
||||||
|
default: false
|
||||||
|
info: |
|
||||||
|
Do not lock data, metadata and journal block devices exclusively with
|
||||||
|
flock(). Though it's not recommended, but you can use it you want to run
|
||||||
|
multiple OSD with a single device and different offsets, without using
|
||||||
|
partitions.
|
||||||
|
info_ru: |
|
||||||
|
Не блокировать устройства данных, метаданных и журнала от открытия их
|
||||||
|
другими OSD с помощью flock(). Так делать не рекомендуется, но теоретически
|
||||||
|
вы можете это использовать, чтобы запускать несколько OSD на одном
|
||||||
|
устройстве с разными смещениями и без использования разделов.
|
||||||
|
- name: disk_alignment
|
||||||
|
type: int
|
||||||
|
default: 4096
|
||||||
|
info: |
|
||||||
|
Required physical disk write alignment. Most current SSD and HDD drives
|
||||||
|
use 4 KB physical sectors even if they report 512 byte logical sector
|
||||||
|
size, so 4 KB is a good default setting.
|
||||||
|
|
||||||
|
Note, however, that physical sector size also affects WA, because with block
|
||||||
|
devices it's impossible to write anything smaller than a block. So, when
|
||||||
|
Vitastor has to write a single metadata entry that's only about 32 bytes in
|
||||||
|
size, it actually has to write the whole 4 KB sector.
|
||||||
|
|
||||||
|
Because of this it can actually be beneficial to use SSDs which work well
|
||||||
|
with 512 byte sectors and use 512 byte disk_alignment, journal_block_size
|
||||||
|
and meta_block_size. But the only SSD that may fit into this category is
|
||||||
|
Intel Optane (probably, not tested yet).
|
||||||
|
|
||||||
|
Clients don't need to be aware of disk_alignment, so it's not required to
|
||||||
|
put a modified value into etcd key /vitastor/config/global.
|
||||||
|
info_ru: |
|
||||||
|
Требуемое выравнивание записи на физические диски. Почти все современные
|
||||||
|
SSD и HDD диски используют 4 КБ физические секторы, даже если показывают
|
||||||
|
логический размер сектора 512 байт, поэтому 4 КБ - хорошее значение по
|
||||||
|
умолчанию.
|
||||||
|
|
||||||
|
Однако стоит понимать, что физический размер сектора тоже влияет на
|
||||||
|
избыточную запись (WA), потому что ничего меньше блока (сектора) на блочное
|
||||||
|
устройство записать невозможно. Таким образом, когда Vitastor-у нужно
|
||||||
|
записать на диск всего лишь одну 32-байтную запись метаданных, фактически
|
||||||
|
приходится перезаписывать 4 КБ сектор целиком.
|
||||||
|
|
||||||
|
Поэтому, на самом деле, может быть выгодно найти SSD, хорошо работающие с
|
||||||
|
меньшими, 512-байтными, блоками и использовать 512-байтные disk_alignment,
|
||||||
|
journal_block_size и meta_block_size. Однако единственные SSD, которые
|
||||||
|
теоретически могут попасть в эту категорию - это Intel Optane (но и это
|
||||||
|
пока не проверялось автором).
|
||||||
|
|
||||||
|
Клиентам не обязательно знать про disk_alignment, так что помещать значение
|
||||||
|
этого параметра в etcd в /vitastor/config/global не нужно.
|
||||||
@@ -0,0 +1,65 @@
|
|||||||
|
- name: etcd_mon_ttl
|
||||||
|
type: sec
|
||||||
|
min: 10
|
||||||
|
default: 30
|
||||||
|
info: Monitor etcd lease refresh interval in seconds
|
||||||
|
info_ru: Интервал обновления etcd резервации (lease) монитором
|
||||||
|
- name: etcd_mon_timeout
|
||||||
|
type: ms
|
||||||
|
default: 1000
|
||||||
|
info: etcd request timeout used by monitor
|
||||||
|
info_ru: Таймаут выполнения запросов к etcd от монитора
|
||||||
|
- name: etcd_mon_retries
|
||||||
|
type: int
|
||||||
|
default: 5
|
||||||
|
info: Maximum number of attempts for one monitor etcd request
|
||||||
|
info_ru: Максимальное число попыток выполнения запросов к etcd монитором
|
||||||
|
- name: mon_change_timeout
|
||||||
|
type: ms
|
||||||
|
min: 100
|
||||||
|
default: 1000
|
||||||
|
info: Optimistic retry interval for monitor etcd modification requests
|
||||||
|
info_ru: Время повтора при коллизиях при запросах модификации в etcd, производимых монитором
|
||||||
|
- name: mon_stats_timeout
|
||||||
|
type: ms
|
||||||
|
min: 100
|
||||||
|
default: 1000
|
||||||
|
info: |
|
||||||
|
Interval for monitor to wait before updating aggregated statistics in
|
||||||
|
etcd after receiving OSD statistics updates
|
||||||
|
info_ru: |
|
||||||
|
Интервал, который монитор ожидает при изменении статистики по отдельным
|
||||||
|
OSD перед обновлением агрегированной статистики в etcd
|
||||||
|
- name: osd_out_time
|
||||||
|
type: sec
|
||||||
|
default: 600
|
||||||
|
info: |
|
||||||
|
Time after which a failed OSD is removed from the data distribution.
|
||||||
|
I.e. time which the monitor waits before attempting to restore data
|
||||||
|
redundancy using other OSDs.
|
||||||
|
info_ru: |
|
||||||
|
Время, через которое отключенный OSD исключается из распределения данных.
|
||||||
|
То есть, время, которое монитор ожидает перед попыткой переместить данные
|
||||||
|
на другие OSD и таким образом восстановить избыточность хранения.
|
||||||
|
- name: placement_levels
|
||||||
|
type: json
|
||||||
|
default: '`{"host":100,"osd":101}`'
|
||||||
|
info: |
|
||||||
|
Levels for the placement tree. You can define arbitrary tree levels by
|
||||||
|
defining them in this parameter. The configuration parameter value should
|
||||||
|
contain a JSON object with level names as keys and integer priorities as
|
||||||
|
values. Smaller priority means higher level in tree. For example,
|
||||||
|
"datacenter" should have smaller priority than "osd". "host" and "osd"
|
||||||
|
levels are always predefined and can't be removed. If one of them is not
|
||||||
|
present in the configuration, then it is defined with the default priority
|
||||||
|
(100 for "host", 101 for "osd").
|
||||||
|
info_ru: |
|
||||||
|
Определения уровней для дерева размещения OSD. Вы можете определять
|
||||||
|
произвольные уровни, помещая их в данный параметр конфигурации. Значение
|
||||||
|
параметра должно содержать JSON-объект, ключи которого будут являться
|
||||||
|
названиями уровней, а значения - целочисленными приоритетами. Меньшие
|
||||||
|
приоритеты соответствуют верхним уровням дерева. Например, уровень
|
||||||
|
"датацентр" должен иметь меньший приоритет, чем "OSD". Уровни с названиями
|
||||||
|
"host" и "osd" являются предопределёнными и не могут быть удалены. Если
|
||||||
|
один из них отсутствует в конфигурации, он доопределяется с приоритетом по
|
||||||
|
умолчанию (100 для уровня "host", 101 для "osd").
|
||||||
@@ -0,0 +1,225 @@
|
|||||||
|
- name: tcp_header_buffer_size
|
||||||
|
type: int
|
||||||
|
default: 65536
|
||||||
|
info: |
|
||||||
|
Size of the buffer used to read data using an additional copy. Vitastor
|
||||||
|
packet headers are 128 bytes, payload is always at least 4 KB, so it is
|
||||||
|
usually beneficial to try to read multiple packets at once even though
|
||||||
|
it requires to copy the data an additional time. The rest of each packet
|
||||||
|
is received without an additional copy. You can try to play with this
|
||||||
|
parameter and see how it affects random iops and linear bandwidth if you
|
||||||
|
want.
|
||||||
|
info_ru: |
|
||||||
|
Размер буфера для чтения данных с дополнительным копированием. Пакеты
|
||||||
|
Vitastor содержат 128-байтные заголовки, за которыми следуют данные размером
|
||||||
|
от 4 КБ и для мелких операций ввода-вывода обычно выгодно за 1 вызов читать
|
||||||
|
сразу несколько пакетов, даже не смотря на то, что это требует лишний раз
|
||||||
|
скопировать данные. Часть каждого пакета за пределами значения данного
|
||||||
|
параметра читается без дополнительного копирования. Вы можете попробовать
|
||||||
|
поменять этот параметр и посмотреть, как он влияет на производительность
|
||||||
|
случайного и линейного доступа.
|
||||||
|
- name: use_sync_send_recv
|
||||||
|
type: bool
|
||||||
|
default: false
|
||||||
|
info: |
|
||||||
|
If true, synchronous send/recv syscalls are used instead of io_uring for
|
||||||
|
socket communication. Useless for OSDs because they require io_uring anyway,
|
||||||
|
but may be required for clients with old kernel versions.
|
||||||
|
info_ru: |
|
||||||
|
Если установлено в истину, то вместо io_uring для передачи данных по сети
|
||||||
|
будут использоваться обычные синхронные системные вызовы send/recv. Для OSD
|
||||||
|
это бессмысленно, так как OSD в любом случае нуждается в io_uring, но, в
|
||||||
|
принципе, это может применяться для клиентов со старыми версиями ядра.
|
||||||
|
- name: use_rdma
|
||||||
|
type: bool
|
||||||
|
default: true
|
||||||
|
info: |
|
||||||
|
Try to use RDMA for communication if it's available. Disable if you don't
|
||||||
|
want Vitastor to use RDMA. RDMA increases the performance, but TCP-only
|
||||||
|
clients can still talk to an RDMA-enabled cluster, so you don't need to
|
||||||
|
make sure that all clients support RDMA when enabling it.
|
||||||
|
info_ru: |
|
||||||
|
Пытаться использовать RDMA для связи при наличии доступных устройств.
|
||||||
|
Отключите, если вы не хотите, чтобы Vitastor использовал RDMA.
|
||||||
|
RDMA улучшает производительность, но
|
||||||
|
Клиенты и клиентов and TCP-only clients in the cluster at the
|
||||||
|
same time - TCP-only clients are still able to use an RDMA-enabled cluster.
|
||||||
|
- name: rdma_device
|
||||||
|
type: string
|
||||||
|
info: |
|
||||||
|
RDMA device name to use for Vitastor OSD communications (for example,
|
||||||
|
"rocep5s0f0"). Please note that Vitastor RDMA requires Implicit On-Demand
|
||||||
|
Paging (Implicit ODP) and Scatter/Gather (SG) support from the RDMA device
|
||||||
|
to work. For example, Mellanox ConnectX-3 and older adapters don't have
|
||||||
|
Implicit ODP, so they're unsupported by Vitastor. Run `ibv_devinfo -v` as
|
||||||
|
root to list available RDMA devices and their features.
|
||||||
|
info_ru: |
|
||||||
|
Название RDMA-устройства для связи с Vitastor OSD (например, "rocep5s0f0").
|
||||||
|
Имейте в виду, что поддержка RDMA в Vitastor требует функций устройства
|
||||||
|
Implicit On-Demand Paging (Implicit ODP) и Scatter/Gather (SG). Например,
|
||||||
|
адаптеры Mellanox ConnectX-3 и более старые не поддерживают Implicit ODP и
|
||||||
|
потому не поддерживаются в Vitastor. Запустите `ibv_devinfo -v` от имени
|
||||||
|
суперпользователя, чтобы посмотреть список доступных RDMA-устройств, их
|
||||||
|
параметры и возможности.
|
||||||
|
- name: rdma_port_num
|
||||||
|
type: int
|
||||||
|
default: 1
|
||||||
|
info: |
|
||||||
|
RDMA device port number to use. Only for devices that have more than 1 port.
|
||||||
|
See `phys_port_cnt` in `ibv_devinfo -v` output to determine how many ports
|
||||||
|
your device has.
|
||||||
|
info_ru: |
|
||||||
|
Номер порта RDMA-устройства, который следует использовать. Имеет смысл
|
||||||
|
только для устройств, у которых более 1 порта. Чтобы узнать, сколько портов
|
||||||
|
у вашего адаптера, посмотрите `phys_port_cnt` в выводе команды
|
||||||
|
`ibv_devinfo -v`.
|
||||||
|
- name: rdma_gid_index
|
||||||
|
type: int
|
||||||
|
default: 0
|
||||||
|
info: |
|
||||||
|
Global address identifier index of the RDMA device to use. Different GID
|
||||||
|
indexes may correspond to different protocols like RoCEv1, RoCEv2 and iWARP.
|
||||||
|
Search for "GID" in `ibv_devinfo -v` output to determine which GID index
|
||||||
|
you need.
|
||||||
|
|
||||||
|
**IMPORTANT:** If you want to use RoCEv2 (as recommended) then the correct
|
||||||
|
rdma_gid_index is usually 1 (IPv6) or 3 (IPv4).
|
||||||
|
info_ru: |
|
||||||
|
Номер глобального идентификатора адреса RDMA-устройства, который следует
|
||||||
|
использовать. Разным gid_index могут соответствовать разные протоколы связи:
|
||||||
|
RoCEv1, RoCEv2, iWARP. Чтобы понять, какой нужен вам - смотрите строчки со
|
||||||
|
словом "GID" в выводе команды `ibv_devinfo -v`.
|
||||||
|
|
||||||
|
**ВАЖНО:** Если вы хотите использовать RoCEv2 (как мы и рекомендуем), то
|
||||||
|
правильный rdma_gid_index, как правило, 1 (IPv6) или 3 (IPv4).
|
||||||
|
- name: rdma_mtu
|
||||||
|
type: int
|
||||||
|
default: 4096
|
||||||
|
info: |
|
||||||
|
RDMA Path MTU to use. Must be 1024, 2048 or 4096. There is usually no
|
||||||
|
sense to change it from the default 4096.
|
||||||
|
info_ru: |
|
||||||
|
Максимальная единица передачи (Path MTU) для RDMA. Должно быть равно 1024,
|
||||||
|
2048 или 4096. Обычно нет смысла менять значение по умолчанию, равное 4096.
|
||||||
|
- name: rdma_max_sge
|
||||||
|
type: int
|
||||||
|
default: 128
|
||||||
|
info: |
|
||||||
|
Maximum number of scatter/gather entries to use for RDMA. OSDs negotiate
|
||||||
|
the actual value when establishing connection anyway, so it's usually not
|
||||||
|
required to change this parameter.
|
||||||
|
info_ru: |
|
||||||
|
Максимальное число записей разделения/сборки (scatter/gather) для RDMA.
|
||||||
|
OSD в любом случае согласовывают реальное значение при установке соединения,
|
||||||
|
так что менять этот параметр обычно не нужно.
|
||||||
|
- name: rdma_max_msg
|
||||||
|
type: int
|
||||||
|
default: 1048576
|
||||||
|
info: Maximum size of a single RDMA send or receive operation in bytes.
|
||||||
|
info_ru: Максимальный размер одной RDMA-операции отправки или приёма.
|
||||||
|
- name: rdma_max_recv
|
||||||
|
type: int
|
||||||
|
default: 8
|
||||||
|
info: |
|
||||||
|
Maximum number of parallel RDMA receive operations. Note that this number
|
||||||
|
of receive buffers `rdma_max_msg` in size are allocated for each client,
|
||||||
|
so this setting actually affects memory usage. This is because RDMA receive
|
||||||
|
operations are (sadly) still not zero-copy in Vitastor. It may be fixed in
|
||||||
|
later versions.
|
||||||
|
info_ru: |
|
||||||
|
Максимальное число параллельных RDMA-операций получения данных. Следует
|
||||||
|
иметь в виду, что данное число буферов размером `rdma_max_msg` выделяется
|
||||||
|
для каждого подключённого клиентского соединения, так что данная настройка
|
||||||
|
влияет на потребление памяти. Это так потому, что RDMA-приём данных в
|
||||||
|
Vitastor, увы, всё равно не является zero-copy, т.е. всё равно 1 раз
|
||||||
|
копирует данные в памяти. Данная особенность, возможно, будет исправлена в
|
||||||
|
более новых версиях Vitastor.
|
||||||
|
- name: peer_connect_interval
|
||||||
|
type: sec
|
||||||
|
min: 1
|
||||||
|
default: 5
|
||||||
|
info: Interval before attempting to reconnect to an unavailable OSD.
|
||||||
|
info_ru: Время ожидания перед повторной попыткой соединиться с недоступным OSD.
|
||||||
|
- name: peer_connect_timeout
|
||||||
|
type: sec
|
||||||
|
min: 1
|
||||||
|
default: 5
|
||||||
|
info: Timeout for OSD connection attempts.
|
||||||
|
info_ru: Максимальное время ожидания попытки соединения с OSD.
|
||||||
|
- name: osd_idle_timeout
|
||||||
|
type: sec
|
||||||
|
min: 1
|
||||||
|
default: 5
|
||||||
|
info: |
|
||||||
|
OSD connection inactivity time after which clients and other OSDs send
|
||||||
|
keepalive requests to check state of the connection.
|
||||||
|
info_ru: |
|
||||||
|
Время неактивности соединения с OSD, после которого клиенты или другие OSD
|
||||||
|
посылают запрос проверки состояния соединения.
|
||||||
|
- name: osd_ping_timeout
|
||||||
|
type: sec
|
||||||
|
min: 1
|
||||||
|
default: 5
|
||||||
|
info: |
|
||||||
|
Maximum time to wait for OSD keepalive responses. If an OSD doesn't respond
|
||||||
|
within this time, the connection to it is dropped and a reconnection attempt
|
||||||
|
is scheduled.
|
||||||
|
info_ru: |
|
||||||
|
Максимальное время ожидания ответа на запрос проверки состояния соединения.
|
||||||
|
Если OSD не отвечает за это время, соединение отключается и производится
|
||||||
|
повторная попытка соединения.
|
||||||
|
- name: up_wait_retry_interval
|
||||||
|
type: ms
|
||||||
|
min: 50
|
||||||
|
default: 500
|
||||||
|
info: |
|
||||||
|
OSDs respond to clients with a special error code when they receive I/O
|
||||||
|
requests for a PG that's not synchronized and started. This parameter sets
|
||||||
|
the time for the clients to wait before re-attempting such I/O requests.
|
||||||
|
info_ru: |
|
||||||
|
Когда OSD получают от клиентов запросы ввода-вывода, относящиеся к не
|
||||||
|
поднятым на данный момент на них PG, либо к PG в процессе синхронизации,
|
||||||
|
они отвечают клиентам специальным кодом ошибки, означающим, что клиент
|
||||||
|
должен некоторое время подождать перед повторением запроса. Именно это время
|
||||||
|
ожидания задаёт данный параметр.
|
||||||
|
- name: max_etcd_attempts
|
||||||
|
type: int
|
||||||
|
default: 5
|
||||||
|
info: |
|
||||||
|
Maximum number of attempts for etcd requests which can't be retried
|
||||||
|
indefinitely.
|
||||||
|
info_ru: |
|
||||||
|
Максимальное число попыток выполнения запросов к etcd для тех запросов,
|
||||||
|
которые нельзя повторять бесконечно.
|
||||||
|
- name: etcd_quick_timeout
|
||||||
|
type: ms
|
||||||
|
default: 1000
|
||||||
|
info: |
|
||||||
|
Timeout for etcd requests which should complete quickly, like lease refresh.
|
||||||
|
info_ru: |
|
||||||
|
Максимальное время выполнения запросов к etcd, которые должны завершаться
|
||||||
|
быстро, таких, как обновление резервации (lease).
|
||||||
|
- name: etcd_slow_timeout
|
||||||
|
type: ms
|
||||||
|
default: 5000
|
||||||
|
info: Timeout for etcd requests which are allowed to wait for some time.
|
||||||
|
info_ru: |
|
||||||
|
Максимальное время выполнения запросов к etcd, для которых не обязательно
|
||||||
|
гарантировать быстрое выполнение.
|
||||||
|
- name: etcd_keepalive_timeout
|
||||||
|
type: sec
|
||||||
|
default: max(30, etcd_report_interval*2)
|
||||||
|
info: |
|
||||||
|
Timeout for etcd connection HTTP Keep-Alive. Should be higher than
|
||||||
|
etcd_report_interval to guarantee that keepalive actually works.
|
||||||
|
info_ru: |
|
||||||
|
Таймаут для HTTP Keep-Alive в соединениях к etcd. Должен быть больше, чем
|
||||||
|
etcd_report_interval, чтобы keepalive гарантированно работал.
|
||||||
|
- name: etcd_ws_keepalive_timeout
|
||||||
|
type: sec
|
||||||
|
default: 30
|
||||||
|
info: |
|
||||||
|
etcd websocket ping interval required to keep the connection alive and
|
||||||
|
detect disconnections quickly.
|
||||||
|
info_ru: |
|
||||||
|
Интервал проверки живости вебсокет-подключений к etcd.
|
||||||
@@ -0,0 +1,341 @@
|
|||||||
|
- name: etcd_report_interval
|
||||||
|
type: sec
|
||||||
|
default: 5
|
||||||
|
info: |
|
||||||
|
Interval at which OSDs report their state to etcd. Affects OSD lease time
|
||||||
|
and thus the failover speed. Lease time is equal to this parameter value
|
||||||
|
plus max_etcd_attempts * etcd_quick_timeout because it should be guaranteed
|
||||||
|
that every OSD always refreshes its lease in time.
|
||||||
|
info_ru: |
|
||||||
|
Интервал, с которым OSD обновляет своё состояние в etcd. Значение параметра
|
||||||
|
влияет на время резервации (lease) OSD и поэтому на скорость переключения
|
||||||
|
при падении OSD. Время lease равняется значению этого параметра плюс
|
||||||
|
max_etcd_attempts * etcd_quick_timeout.
|
||||||
|
- name: run_primary
|
||||||
|
type: bool
|
||||||
|
default: true
|
||||||
|
info: |
|
||||||
|
Start primary OSD logic on this OSD. As of now, can be turned off only for
|
||||||
|
debugging purposes. It's possible to implement additional feature for the
|
||||||
|
monitor which may allow to separate primary and secondary OSDs, but it's
|
||||||
|
unclear why anyone could need it, so it's not implemented.
|
||||||
|
info_ru: |
|
||||||
|
Запускать логику первичного OSD на данном OSD. На данный момент отключать
|
||||||
|
эту опцию может иметь смысл только в целях отладки. В теории, можно
|
||||||
|
реализовать дополнительный режим для монитора, который позволит отделять
|
||||||
|
первичные OSD от вторичных, но пока не понятно, зачем это может кому-то
|
||||||
|
понадобиться, поэтому это не реализовано.
|
||||||
|
- name: osd_network
|
||||||
|
type: string or array of strings
|
||||||
|
type_ru: строка или массив строк
|
||||||
|
info: |
|
||||||
|
Network mask of the network (IPv4 or IPv6) to use for OSDs. Note that
|
||||||
|
although it's possible to specify multiple networks here, this does not
|
||||||
|
mean that OSDs will create multiple listening sockets - they'll only
|
||||||
|
pick the first matching address of an UP + RUNNING interface. Separate
|
||||||
|
networks for cluster and client connections are also not implemented, but
|
||||||
|
they are mostly useless anyway, so it's not a big deal.
|
||||||
|
info_ru: |
|
||||||
|
Маска подсети (IPv4 или IPv6) для использования для соединений с OSD.
|
||||||
|
Имейте в виду, что хотя сейчас и можно передать в этот параметр несколько
|
||||||
|
подсетей, это не означает, что OSD будут создавать несколько слушающих
|
||||||
|
сокетов - они лишь будут выбирать адрес первого поднятого (состояние UP +
|
||||||
|
RUNNING), подходящий под заданную маску. Также не реализовано разделение
|
||||||
|
кластерной и публичной сетей OSD. Правда, от него обычно всё равно довольно
|
||||||
|
мало толку, так что особенной проблемы в этом нет.
|
||||||
|
- name: bind_address
|
||||||
|
type: string
|
||||||
|
default: "0.0.0.0"
|
||||||
|
info: |
|
||||||
|
Instead of the network mask, you can also set OSD listen address explicitly
|
||||||
|
using this parameter. May be useful if you want to start OSDs on interfaces
|
||||||
|
that are not UP + RUNNING.
|
||||||
|
info_ru: |
|
||||||
|
Этим параметром можно явным образом задать адрес, на котором будет ожидать
|
||||||
|
соединений OSD (вместо использования маски подсети). Может быть полезно,
|
||||||
|
например, чтобы запускать OSD на неподнятых интерфейсах (не UP + RUNNING).
|
||||||
|
- name: bind_port
|
||||||
|
type: int
|
||||||
|
info: |
|
||||||
|
By default, OSDs pick random ports to use for incoming connections
|
||||||
|
automatically. With this option you can set a specific port for a specific
|
||||||
|
OSD by hand.
|
||||||
|
info_ru: |
|
||||||
|
По умолчанию OSD сами выбирают случайные порты для входящих подключений.
|
||||||
|
С помощью данной опции вы можете задать порт для отдельного OSD вручную.
|
||||||
|
- name: autosync_interval
|
||||||
|
type: sec
|
||||||
|
default: 5
|
||||||
|
info: |
|
||||||
|
Time interval at which automatic fsyncs/flushes are issued by each OSD when
|
||||||
|
the immediate_commit mode if disabled. fsyncs are required because without
|
||||||
|
them OSDs quickly fill their journals, become unable to clear them and
|
||||||
|
stall. Also this option limits the amount of recent uncommitted changes
|
||||||
|
which OSDs may lose in case of a power outage in case when clients don't
|
||||||
|
issue fsyncs at all.
|
||||||
|
info_ru: |
|
||||||
|
Временной интервал отправки автоматических fsync-ов (операций очистки кэша)
|
||||||
|
каждым OSD для случая, когда режим immediate_commit отключён. fsync-и нужны
|
||||||
|
OSD, чтобы успевать очищать журнал - без них OSD быстро заполняют журналы и
|
||||||
|
перестают обрабатывать операции записи. Также эта опция ограничивает объём
|
||||||
|
недавних незафиксированных изменений, которые OSD могут терять при
|
||||||
|
отключении питания, если клиенты вообще не отправляют fsync.
|
||||||
|
- name: autosync_writes
|
||||||
|
type: int
|
||||||
|
default: 128
|
||||||
|
info: |
|
||||||
|
Same as autosync_interval, but sets the maximum number of uncommitted write
|
||||||
|
operations before issuing an fsync operation internally.
|
||||||
|
info_ru: |
|
||||||
|
Аналогично autosync_interval, но задаёт не временной интервал, а
|
||||||
|
максимальное количество незафиксированных операций записи перед
|
||||||
|
принудительной отправкой fsync-а.
|
||||||
|
- name: recovery_queue_depth
|
||||||
|
type: int
|
||||||
|
default: 4
|
||||||
|
info: |
|
||||||
|
Maximum recovery operations per one primary OSD at any given moment of time.
|
||||||
|
Currently it's the only parameter available to tune the speed or recovery
|
||||||
|
and rebalancing, but it's planned to implement more.
|
||||||
|
info_ru: |
|
||||||
|
Максимальное число операций восстановления на одном первичном OSD в любой
|
||||||
|
момент времени. На данный момент единственный параметр, который можно менять
|
||||||
|
для ускорения или замедления восстановления и перебалансировки данных, но
|
||||||
|
в планах реализация других параметров.
|
||||||
|
- name: recovery_sync_batch
|
||||||
|
type: int
|
||||||
|
default: 16
|
||||||
|
info: Maximum number of recovery operations before issuing an additional fsync.
|
||||||
|
info_ru: Максимальное число операций восстановления перед дополнительным fsync.
|
||||||
|
- name: readonly
|
||||||
|
type: bool
|
||||||
|
default: false
|
||||||
|
info: |
|
||||||
|
Read-only mode. If this is enabled, an OSD will never issue any writes to
|
||||||
|
the underlying device. This may be useful for recovery purposes.
|
||||||
|
info_ru: |
|
||||||
|
Режим "только чтение". Если включить этот режим, OSD не будет писать ничего
|
||||||
|
на диск. Может быть полезно в целях восстановления.
|
||||||
|
- name: no_recovery
|
||||||
|
type: bool
|
||||||
|
default: false
|
||||||
|
info: |
|
||||||
|
Disable automatic background recovery of objects. Note that it doesn't
|
||||||
|
affect implicit recovery of objects happening during writes - a write is
|
||||||
|
always made to a full set of at least pg_minsize OSDs.
|
||||||
|
info_ru: |
|
||||||
|
Отключить автоматическое фоновое восстановление объектов. Обратите внимание,
|
||||||
|
что эта опция не отключает восстановление объектов, происходящее при
|
||||||
|
записи - запись всегда производится в полный набор из как минимум pg_minsize
|
||||||
|
OSD.
|
||||||
|
- name: no_rebalance
|
||||||
|
type: bool
|
||||||
|
default: false
|
||||||
|
info: |
|
||||||
|
Disable background movement of data between different OSDs. Disabling it
|
||||||
|
means that PGs in the `has_misplaced` state will be left in it indefinitely.
|
||||||
|
info_ru: |
|
||||||
|
Отключить фоновое перемещение объектов между разными OSD. Отключение
|
||||||
|
означает, что PG, находящиеся в состоянии `has_misplaced`, будут оставлены
|
||||||
|
в нём на неопределённый срок.
|
||||||
|
- name: print_stats_interval
|
||||||
|
type: sec
|
||||||
|
default: 3
|
||||||
|
info: |
|
||||||
|
Time interval at which OSDs print simple human-readable operation
|
||||||
|
statistics on stdout.
|
||||||
|
info_ru: |
|
||||||
|
Временной интервал, с которым OSD печатают простую человекочитаемую
|
||||||
|
статистику выполнения операций в стандартный вывод.
|
||||||
|
- name: slow_log_interval
|
||||||
|
type: sec
|
||||||
|
default: 10
|
||||||
|
info: |
|
||||||
|
Time interval at which OSDs dump slow or stuck operations on stdout, if
|
||||||
|
they're any. Also it's the time after which an operation is considered
|
||||||
|
"slow".
|
||||||
|
info_ru: |
|
||||||
|
Временной интервал, с которым OSD выводят в стандартный вывод список
|
||||||
|
медленных или зависших операций, если таковые имеются. Также время, при
|
||||||
|
превышении которого операция считается "медленной".
|
||||||
|
- name: max_write_iodepth
|
||||||
|
type: int
|
||||||
|
default: 128
|
||||||
|
info: |
|
||||||
|
Parallel client write operation limit per one OSD. Operations that exceed
|
||||||
|
this limit are pushed to a temporary queue instead of being executed
|
||||||
|
immediately.
|
||||||
|
info_ru: |
|
||||||
|
Максимальное число одновременных клиентских операций записи на один OSD.
|
||||||
|
Операции, превышающие этот лимит, не исполняются сразу, а сохраняются во
|
||||||
|
временной очереди.
|
||||||
|
- name: min_flusher_count
|
||||||
|
type: int
|
||||||
|
default: 1
|
||||||
|
info: |
|
||||||
|
Flusher is a micro-thread that moves data from the journal to the data
|
||||||
|
area of the device. Their number is auto-tuned between minimum and maximum.
|
||||||
|
Minimum number is set by this parameter.
|
||||||
|
info_ru: |
|
||||||
|
Flusher - это микро-поток (корутина), которая копирует данные из журнала в
|
||||||
|
основную область устройства данных. Их число настраивается динамически между
|
||||||
|
минимальным и максимальным значением. Этот параметр задаёт минимальное число.
|
||||||
|
- name: max_flusher_count
|
||||||
|
type: int
|
||||||
|
default: 256
|
||||||
|
info: |
|
||||||
|
Maximum number of journal flushers (see above min_flusher_count).
|
||||||
|
info_ru: |
|
||||||
|
Максимальное число микро-потоков очистки журнала (см. выше min_flusher_count).
|
||||||
|
- name: inmemory_metadata
|
||||||
|
type: bool
|
||||||
|
default: true
|
||||||
|
info: |
|
||||||
|
This parameter makes Vitastor always keep metadata area of the block device
|
||||||
|
in memory. It's required for good performance because it allows to avoid
|
||||||
|
additional read-modify-write cycles during metadata modifications. Metadata
|
||||||
|
area size is currently roughly 224 MB per 1 TB of data. You can turn it off
|
||||||
|
to reduce memory usage by this value, but it will hurt performance. This
|
||||||
|
restriction is likely to be removed in the future along with the upgrade
|
||||||
|
of the metadata storage scheme.
|
||||||
|
info_ru: |
|
||||||
|
Данный параметр заставляет Vitastor всегда держать область метаданных диска
|
||||||
|
в памяти. Это нужно, чтобы избегать дополнительных операций чтения с диска
|
||||||
|
при записи. Размер области метаданных на данный момент составляет примерно
|
||||||
|
224 МБ на 1 ТБ данных. При включении потребление памяти снизится примерно
|
||||||
|
на эту величину, но при этом также снизится и производительность. В будущем,
|
||||||
|
после обновления схемы хранения метаданных, это ограничение, скорее всего,
|
||||||
|
будет ликвидировано.
|
||||||
|
- name: inmemory_journal
|
||||||
|
type: bool
|
||||||
|
default: true
|
||||||
|
info: |
|
||||||
|
This parameter make Vitastor always keep journal area of the block
|
||||||
|
device in memory. Turning it off will, again, reduce memory usage, but
|
||||||
|
hurt performance because flusher coroutines will have to read data from
|
||||||
|
the disk back before copying it into the main area. The memory usage benefit
|
||||||
|
is typically very small because it's sufficient to have 16-32 MB journal
|
||||||
|
for SSD OSDs. However, in theory it's possible that you'll want to turn it
|
||||||
|
off for hybrid (HDD+SSD) OSDs with large journals on quick devices.
|
||||||
|
info_ru: |
|
||||||
|
Данный параметр заставляет Vitastor всегда держать в памяти журналы OSD.
|
||||||
|
Отключение параметра, опять же, снижает потребление памяти, но ухудшает
|
||||||
|
производительность, так как для копирования данных из журнала в основную
|
||||||
|
область устройства OSD будут вынуждены читать их обратно с диска. Выигрыш
|
||||||
|
по памяти при этом обычно крайне низкий, так как для SSD OSD обычно
|
||||||
|
достаточно 16- или 32-мегабайтного журнала. Однако в теории отключение
|
||||||
|
параметра может оказаться полезным для гибридных OSD (HDD+SSD) с большими
|
||||||
|
журналами, расположенными на быстром по сравнению с HDD устройстве.
|
||||||
|
- name: journal_sector_buffer_count
|
||||||
|
type: int
|
||||||
|
default: 32
|
||||||
|
info: |
|
||||||
|
Maximum number of buffers that can be used for writing journal metadata
|
||||||
|
blocks. The only situation when you should increase it to a larger value
|
||||||
|
is when you enable journal_no_same_sector_overwrites. In this case set
|
||||||
|
it to, for example, 1024.
|
||||||
|
info_ru: |
|
||||||
|
Максимальное число буферов, разрешённых для использования под записываемые
|
||||||
|
в журнал блоки метаданных. Единственная ситуация, в которой этот параметр
|
||||||
|
нужно менять - это если вы включаете journal_no_same_sector_overwrites. В
|
||||||
|
этом случае установите данный параметр, например, в 1024.
|
||||||
|
- name: journal_no_same_sector_overwrites
|
||||||
|
type: bool
|
||||||
|
default: false
|
||||||
|
info: |
|
||||||
|
Enable this option for SSDs like Intel D3-S4510 and D3-S4610 which REALLY
|
||||||
|
don't like when a program overwrites the same sector multiple times in a
|
||||||
|
row and slow down significantly (from 25000+ iops to ~3000 iops). When
|
||||||
|
this option is set, Vitastor will always move to the next sector of the
|
||||||
|
journal after writing it instead of possibly overwriting it the second time.
|
||||||
|
info_ru: |
|
||||||
|
Включайте данную опцию для SSD вроде Intel D3-S4510 и D3-S4610, которые
|
||||||
|
ОЧЕНЬ не любят, когда ПО перезаписывает один и тот же сектор несколько раз
|
||||||
|
подряд. Такие SSD при многократной перезаписи одного и того же сектора
|
||||||
|
сильно замедляются - условно, с 25000 и более iops до 3000 iops. Когда
|
||||||
|
данная опция установлена, Vitastor всегда переходит к следующему сектору
|
||||||
|
журнала после записи вместо потенциально повторной перезаписи того же
|
||||||
|
самого сектора.
|
||||||
|
- name: throttle_small_writes
|
||||||
|
type: bool
|
||||||
|
default: false
|
||||||
|
info: |
|
||||||
|
Enable soft throttling of small journaled writes. Useful for hybrid OSDs
|
||||||
|
with fast journal/metadata devices and slow data devices. The idea is that
|
||||||
|
small writes complete very quickly because they're first written to the
|
||||||
|
journal device, but moving them to the main device is slow. So if an OSD
|
||||||
|
allows clients to issue a lot of small writes it will perform very good
|
||||||
|
for several seconds and then the journal will fill up and the performance
|
||||||
|
will drop to almost zero. Throttling is meant to prevent this problem by
|
||||||
|
artifically slowing quick writes down based on the amount of free space in
|
||||||
|
the journal. When throttling is used, the performance of small writes will
|
||||||
|
decrease smoothly instead of abrupt drop at the moment when the journal
|
||||||
|
fills up.
|
||||||
|
info_ru: |
|
||||||
|
Разрешить мягкое ограничение скорости журналируемой записи. Полезно для
|
||||||
|
гибридных OSD с быстрыми устройствами метаданных и медленными устройствами
|
||||||
|
данных. Идея заключается в том, что мелкие записи в этой ситуации могут
|
||||||
|
завершаться очень быстро, так как они изначально записываются на быстрое
|
||||||
|
журнальное устройство (SSD). Но перемещать их потом на основное медленное
|
||||||
|
устройство долго. Поэтому если OSD быстро примет от клиентов очень много
|
||||||
|
мелких операций записи, он быстро заполнит свой журнал, после чего
|
||||||
|
производительность записи резко упадёт практически до нуля. Ограничение
|
||||||
|
скорости записи призвано решить эту проблему с помощью искусственного
|
||||||
|
замедления операций записи на основании объёма свободного места в журнале.
|
||||||
|
Когда эта опция включена, производительность мелких операций записи будет
|
||||||
|
снижаться плавно, а не резко в момент окончательного заполнения журнала.
|
||||||
|
- name: throttle_target_iops
|
||||||
|
type: int
|
||||||
|
default: 100
|
||||||
|
info: |
|
||||||
|
Target maximum number of throttled operations per second under the condition
|
||||||
|
of full journal. Set it to approximate random write iops of your data devices
|
||||||
|
(HDDs).
|
||||||
|
info_ru: |
|
||||||
|
Расчётное максимальное число ограничиваемых операций в секунду при условии
|
||||||
|
отсутствия свободного места в журнале. Устанавливайте приблизительно равным
|
||||||
|
максимальной производительности случайной записи ваших устройств данных
|
||||||
|
(HDD) в операциях в секунду.
|
||||||
|
- name: throttle_target_mbs
|
||||||
|
type: int
|
||||||
|
default: 100
|
||||||
|
info: |
|
||||||
|
Target maximum bandwidth in MB/s of throttled operations per second under
|
||||||
|
the condition of full journal. Set it to approximate linear write
|
||||||
|
performance of your data devices (HDDs).
|
||||||
|
info_ru: |
|
||||||
|
Расчётный максимальный размер в МБ/с ограничиваемых операций в секунду при
|
||||||
|
условии отсутствия свободного места в журнале. Устанавливайте приблизительно
|
||||||
|
равным максимальной производительности линейной записи ваших устройств
|
||||||
|
данных (HDD).
|
||||||
|
- name: throttle_target_parallelism
|
||||||
|
type: int
|
||||||
|
default: 1
|
||||||
|
info: |
|
||||||
|
Target maximum parallelism of throttled operations under the condition of
|
||||||
|
full journal. Set it to approximate internal parallelism of your data
|
||||||
|
devices (1 for HDDs, 4-8 for SSDs).
|
||||||
|
info_ru: |
|
||||||
|
Расчётный максимальный параллелизм ограничиваемых операций в секунду при
|
||||||
|
условии отсутствия свободного места в журнале. Устанавливайте приблизительно
|
||||||
|
равным внутреннему параллелизму ваших устройств данных (1 для HDD, 4-8
|
||||||
|
для SSD).
|
||||||
|
- name: throttle_threshold_us
|
||||||
|
type: us
|
||||||
|
default: 50
|
||||||
|
info: |
|
||||||
|
Minimal computed delay to be applied to throttled operations. Usually
|
||||||
|
doesn't need to be changed.
|
||||||
|
info_ru: |
|
||||||
|
Минимальная применимая к ограничиваемым операциям задержка. Обычно не
|
||||||
|
требует изменений.
|
||||||
|
- name: osd_memlock
|
||||||
|
type: bool
|
||||||
|
default: false
|
||||||
|
info: >
|
||||||
|
Lock all OSD memory to prevent it from being unloaded into swap with
|
||||||
|
mlockall(). Requires sufficient ulimit -l (max locked memory).
|
||||||
|
info_ru: >
|
||||||
|
Блокировать всю память OSD с помощью mlockall, чтобы запретить её выгрузку
|
||||||
|
в пространство подкачки. Требует достаточного значения ulimit -l (лимита
|
||||||
|
заблокированной памяти).
|
||||||
@@ -1,599 +0,0 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
|
||||||
// License: VNPL-1.0 or GNU GPL-2.0+ (see README.md for details)
|
|
||||||
|
|
||||||
#include "osd_ops.h"
|
|
||||||
#include "pg_states.h"
|
|
||||||
#include "etcd_state_client.h"
|
|
||||||
#include "http_client.h"
|
|
||||||
#include "base64.h"
|
|
||||||
|
|
||||||
json_kv_t etcd_state_client_t::parse_etcd_kv(const json11::Json & kv_json)
|
|
||||||
{
|
|
||||||
json_kv_t kv;
|
|
||||||
kv.key = base64_decode(kv_json["key"].string_value());
|
|
||||||
std::string json_err, json_text = base64_decode(kv_json["value"].string_value());
|
|
||||||
kv.value = json_text == "" ? json11::Json() : json11::Json::parse(json_text, json_err);
|
|
||||||
if (json_err != "")
|
|
||||||
{
|
|
||||||
printf("Bad JSON in etcd key %s: %s (value: %s)\n", kv.key.c_str(), json_err.c_str(), json_text.c_str());
|
|
||||||
kv.key = "";
|
|
||||||
}
|
|
||||||
return kv;
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::etcd_txn(json11::Json txn, int timeout, std::function<void(std::string, json11::Json)> callback)
|
|
||||||
{
|
|
||||||
etcd_call("/kv/txn", txn, timeout, callback);
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::etcd_call(std::string api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback)
|
|
||||||
{
|
|
||||||
std::string etcd_address = etcd_addresses[rand() % etcd_addresses.size()];
|
|
||||||
std::string etcd_api_path;
|
|
||||||
int pos = etcd_address.find('/');
|
|
||||||
if (pos >= 0)
|
|
||||||
{
|
|
||||||
etcd_api_path = etcd_address.substr(pos);
|
|
||||||
etcd_address = etcd_address.substr(0, pos);
|
|
||||||
}
|
|
||||||
std::string req = payload.dump();
|
|
||||||
req = "POST "+etcd_api_path+api+" HTTP/1.1\r\n"
|
|
||||||
"Host: "+etcd_address+"\r\n"
|
|
||||||
"Content-Type: application/json\r\n"
|
|
||||||
"Content-Length: "+std::to_string(req.size())+"\r\n"
|
|
||||||
"Connection: close\r\n"
|
|
||||||
"\r\n"+req;
|
|
||||||
http_request_json(tfd, etcd_address, req, timeout, callback);
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::parse_config(json11::Json & config)
|
|
||||||
{
|
|
||||||
this->etcd_addresses.clear();
|
|
||||||
if (config["etcd_address"].is_string())
|
|
||||||
{
|
|
||||||
std::string ea = config["etcd_address"].string_value();
|
|
||||||
while (1)
|
|
||||||
{
|
|
||||||
int pos = ea.find(',');
|
|
||||||
std::string addr = pos >= 0 ? ea.substr(0, pos) : ea;
|
|
||||||
if (addr.length() > 0)
|
|
||||||
{
|
|
||||||
if (addr.find('/') < 0)
|
|
||||||
addr += "/v3";
|
|
||||||
this->etcd_addresses.push_back(addr);
|
|
||||||
}
|
|
||||||
if (pos >= 0)
|
|
||||||
ea = ea.substr(pos+1);
|
|
||||||
else
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else if (config["etcd_address"].array_items().size())
|
|
||||||
{
|
|
||||||
for (auto & ea: config["etcd_address"].array_items())
|
|
||||||
{
|
|
||||||
std::string addr = ea.string_value();
|
|
||||||
if (addr != "")
|
|
||||||
{
|
|
||||||
if (addr.find('/') < 0)
|
|
||||||
addr += "/v3";
|
|
||||||
this->etcd_addresses.push_back(addr);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
this->etcd_prefix = config["etcd_prefix"].string_value();
|
|
||||||
if (this->etcd_prefix == "")
|
|
||||||
{
|
|
||||||
this->etcd_prefix = "/vitastor";
|
|
||||||
}
|
|
||||||
else if (this->etcd_prefix[0] != '/')
|
|
||||||
{
|
|
||||||
this->etcd_prefix = "/"+this->etcd_prefix;
|
|
||||||
}
|
|
||||||
this->log_level = config["log_level"].int64_value();
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::start_etcd_watcher()
|
|
||||||
{
|
|
||||||
std::string etcd_address = etcd_addresses[rand() % etcd_addresses.size()];
|
|
||||||
std::string etcd_api_path;
|
|
||||||
int pos = etcd_address.find('/');
|
|
||||||
if (pos >= 0)
|
|
||||||
{
|
|
||||||
etcd_api_path = etcd_address.substr(pos);
|
|
||||||
etcd_address = etcd_address.substr(0, pos);
|
|
||||||
}
|
|
||||||
etcd_watches_initialised = 0;
|
|
||||||
etcd_watch_ws = open_websocket(tfd, etcd_address, etcd_api_path+"/watch", ETCD_SLOW_TIMEOUT, [this](const http_response_t *msg)
|
|
||||||
{
|
|
||||||
if (msg->body.length())
|
|
||||||
{
|
|
||||||
std::string json_err;
|
|
||||||
json11::Json data = json11::Json::parse(msg->body, json_err);
|
|
||||||
if (json_err != "")
|
|
||||||
{
|
|
||||||
printf("Bad JSON in etcd event: %s, ignoring event\n", json_err.c_str());
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
if (data["result"]["created"].bool_value())
|
|
||||||
{
|
|
||||||
etcd_watches_initialised++;
|
|
||||||
}
|
|
||||||
if (etcd_watches_initialised == 4)
|
|
||||||
{
|
|
||||||
etcd_watch_revision = data["result"]["header"]["revision"].uint64_value();
|
|
||||||
}
|
|
||||||
// First gather all changes into a hash to remove multiple overwrites
|
|
||||||
json11::Json::object changes;
|
|
||||||
for (auto & ev: data["result"]["events"].array_items())
|
|
||||||
{
|
|
||||||
auto kv = parse_etcd_kv(ev["kv"]);
|
|
||||||
if (kv.key != "")
|
|
||||||
{
|
|
||||||
changes[kv.key] = kv.value;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (auto & kv: changes)
|
|
||||||
{
|
|
||||||
if (this->log_level > 3)
|
|
||||||
{
|
|
||||||
printf("Incoming event: %s -> %s\n", kv.first.c_str(), kv.second.dump().c_str());
|
|
||||||
}
|
|
||||||
parse_state(kv.first, kv.second);
|
|
||||||
}
|
|
||||||
// React to changes
|
|
||||||
if (on_change_hook != NULL)
|
|
||||||
{
|
|
||||||
on_change_hook(changes);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (msg->eof)
|
|
||||||
{
|
|
||||||
etcd_watch_ws = NULL;
|
|
||||||
if (etcd_watches_initialised == 0)
|
|
||||||
{
|
|
||||||
// Connection not established, retry in <ETCD_SLOW_TIMEOUT>
|
|
||||||
tfd->set_timer(ETCD_SLOW_TIMEOUT, false, [this](int)
|
|
||||||
{
|
|
||||||
start_etcd_watcher();
|
|
||||||
});
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// Connection was live, retry immediately
|
|
||||||
start_etcd_watcher();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
});
|
|
||||||
etcd_watch_ws->post_message(WS_TEXT, json11::Json(json11::Json::object {
|
|
||||||
{ "create_request", json11::Json::object {
|
|
||||||
{ "key", base64_encode(etcd_prefix+"/config/") },
|
|
||||||
{ "range_end", base64_encode(etcd_prefix+"/config0") },
|
|
||||||
{ "start_revision", etcd_watch_revision+1 },
|
|
||||||
{ "watch_id", ETCD_CONFIG_WATCH_ID },
|
|
||||||
{ "progress_notify", true },
|
|
||||||
} }
|
|
||||||
}).dump());
|
|
||||||
etcd_watch_ws->post_message(WS_TEXT, json11::Json(json11::Json::object {
|
|
||||||
{ "create_request", json11::Json::object {
|
|
||||||
{ "key", base64_encode(etcd_prefix+"/osd/state/") },
|
|
||||||
{ "range_end", base64_encode(etcd_prefix+"/osd/state0") },
|
|
||||||
{ "start_revision", etcd_watch_revision+1 },
|
|
||||||
{ "watch_id", ETCD_OSD_STATE_WATCH_ID },
|
|
||||||
{ "progress_notify", true },
|
|
||||||
} }
|
|
||||||
}).dump());
|
|
||||||
etcd_watch_ws->post_message(WS_TEXT, json11::Json(json11::Json::object {
|
|
||||||
{ "create_request", json11::Json::object {
|
|
||||||
{ "key", base64_encode(etcd_prefix+"/pg/state/") },
|
|
||||||
{ "range_end", base64_encode(etcd_prefix+"/pg/state0") },
|
|
||||||
{ "start_revision", etcd_watch_revision+1 },
|
|
||||||
{ "watch_id", ETCD_PG_STATE_WATCH_ID },
|
|
||||||
{ "progress_notify", true },
|
|
||||||
} }
|
|
||||||
}).dump());
|
|
||||||
etcd_watch_ws->post_message(WS_TEXT, json11::Json(json11::Json::object {
|
|
||||||
{ "create_request", json11::Json::object {
|
|
||||||
{ "key", base64_encode(etcd_prefix+"/pg/history/") },
|
|
||||||
{ "range_end", base64_encode(etcd_prefix+"/pg/history0") },
|
|
||||||
{ "start_revision", etcd_watch_revision+1 },
|
|
||||||
{ "watch_id", ETCD_PG_HISTORY_WATCH_ID },
|
|
||||||
{ "progress_notify", true },
|
|
||||||
} }
|
|
||||||
}).dump());
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::load_global_config()
|
|
||||||
{
|
|
||||||
etcd_call("/kv/range", json11::Json::object {
|
|
||||||
{ "key", base64_encode(etcd_prefix+"/config/global") }
|
|
||||||
}, ETCD_SLOW_TIMEOUT, [this](std::string err, json11::Json data)
|
|
||||||
{
|
|
||||||
if (err != "")
|
|
||||||
{
|
|
||||||
printf("Error reading OSD configuration from etcd: %s\n", err.c_str());
|
|
||||||
tfd->set_timer(ETCD_SLOW_TIMEOUT, false, [this](int timer_id)
|
|
||||||
{
|
|
||||||
load_global_config();
|
|
||||||
});
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
json11::Json::object global_config;
|
|
||||||
if (data["kvs"].array_items().size() > 0)
|
|
||||||
{
|
|
||||||
auto kv = parse_etcd_kv(data["kvs"][0]);
|
|
||||||
if (kv.value.is_object())
|
|
||||||
{
|
|
||||||
global_config = kv.value.object_items();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
bs_block_size = global_config["block_size"].uint64_value();
|
|
||||||
if (!bs_block_size)
|
|
||||||
{
|
|
||||||
bs_block_size = DEFAULT_BLOCK_SIZE;
|
|
||||||
}
|
|
||||||
on_load_config_hook(global_config);
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::load_pgs()
|
|
||||||
{
|
|
||||||
json11::Json::array txn = {
|
|
||||||
json11::Json::object {
|
|
||||||
{ "request_range", json11::Json::object {
|
|
||||||
{ "key", base64_encode(etcd_prefix+"/config/pools") },
|
|
||||||
} }
|
|
||||||
},
|
|
||||||
json11::Json::object {
|
|
||||||
{ "request_range", json11::Json::object {
|
|
||||||
{ "key", base64_encode(etcd_prefix+"/config/pgs") },
|
|
||||||
} }
|
|
||||||
},
|
|
||||||
json11::Json::object {
|
|
||||||
{ "request_range", json11::Json::object {
|
|
||||||
{ "key", base64_encode(etcd_prefix+"/pg/history/") },
|
|
||||||
{ "range_end", base64_encode(etcd_prefix+"/pg/history0") },
|
|
||||||
} }
|
|
||||||
},
|
|
||||||
json11::Json::object {
|
|
||||||
{ "request_range", json11::Json::object {
|
|
||||||
{ "key", base64_encode(etcd_prefix+"/pg/state/") },
|
|
||||||
{ "range_end", base64_encode(etcd_prefix+"/pg/state0") },
|
|
||||||
} }
|
|
||||||
},
|
|
||||||
json11::Json::object {
|
|
||||||
{ "request_range", json11::Json::object {
|
|
||||||
{ "key", base64_encode(etcd_prefix+"/osd/state/") },
|
|
||||||
{ "range_end", base64_encode(etcd_prefix+"/osd/state0") },
|
|
||||||
} }
|
|
||||||
},
|
|
||||||
};
|
|
||||||
json11::Json::object req = { { "success", txn } };
|
|
||||||
json11::Json checks = load_pgs_checks_hook != NULL ? load_pgs_checks_hook() : json11::Json();
|
|
||||||
if (checks.array_items().size() > 0)
|
|
||||||
{
|
|
||||||
req["compare"] = checks;
|
|
||||||
}
|
|
||||||
etcd_txn(req, ETCD_SLOW_TIMEOUT, [this](std::string err, json11::Json data)
|
|
||||||
{
|
|
||||||
if (err != "")
|
|
||||||
{
|
|
||||||
printf("Error loading PGs from etcd: %s\n", err.c_str());
|
|
||||||
tfd->set_timer(ETCD_SLOW_TIMEOUT, false, [this](int timer_id)
|
|
||||||
{
|
|
||||||
load_pgs();
|
|
||||||
});
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (!data["succeeded"].bool_value())
|
|
||||||
{
|
|
||||||
on_load_pgs_hook(false);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (!etcd_watch_revision)
|
|
||||||
{
|
|
||||||
etcd_watch_revision = data["header"]["revision"].uint64_value();
|
|
||||||
}
|
|
||||||
for (auto & res: data["responses"].array_items())
|
|
||||||
{
|
|
||||||
for (auto & kv_json: res["response_range"]["kvs"].array_items())
|
|
||||||
{
|
|
||||||
auto kv = parse_etcd_kv(kv_json);
|
|
||||||
parse_state(kv.key, kv.value);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
on_load_pgs_hook(true);
|
|
||||||
start_etcd_watcher();
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::parse_state(const std::string & key, const json11::Json & value)
|
|
||||||
{
|
|
||||||
if (key == etcd_prefix+"/config/pools")
|
|
||||||
{
|
|
||||||
for (auto & pool_item: this->pool_config)
|
|
||||||
{
|
|
||||||
pool_item.second.exists = false;
|
|
||||||
}
|
|
||||||
for (auto & pool_item: value.object_items())
|
|
||||||
{
|
|
||||||
pool_config_t pc;
|
|
||||||
// ID
|
|
||||||
pool_id_t pool_id = stoull_full(pool_item.first);
|
|
||||||
if (!pool_id || pool_id >= POOL_ID_MAX)
|
|
||||||
{
|
|
||||||
printf("Pool ID %s is invalid (must be a number less than 0x%x), skipping pool\n", pool_item.first.c_str(), POOL_ID_MAX);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
pc.id = pool_id;
|
|
||||||
// Pool Name
|
|
||||||
pc.name = pool_item.second["name"].string_value();
|
|
||||||
if (pc.name == "")
|
|
||||||
{
|
|
||||||
printf("Pool %u has empty name, skipping pool\n", pool_id);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// Failure Domain
|
|
||||||
pc.failure_domain = pool_item.second["failure_domain"].string_value();
|
|
||||||
// Coding Scheme
|
|
||||||
if (pool_item.second["scheme"] == "replicated")
|
|
||||||
pc.scheme = POOL_SCHEME_REPLICATED;
|
|
||||||
else if (pool_item.second["scheme"] == "xor")
|
|
||||||
pc.scheme = POOL_SCHEME_XOR;
|
|
||||||
else if (pool_item.second["scheme"] == "jerasure")
|
|
||||||
pc.scheme = POOL_SCHEME_JERASURE;
|
|
||||||
else
|
|
||||||
{
|
|
||||||
printf("Pool %u has invalid coding scheme (one of \"xor\", \"replicated\" or \"jerasure\" required), skipping pool\n", pool_id);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// PG Size
|
|
||||||
pc.pg_size = pool_item.second["pg_size"].uint64_value();
|
|
||||||
if (pc.pg_size < 1 ||
|
|
||||||
pool_item.second["pg_size"].uint64_value() < 3 &&
|
|
||||||
(pc.scheme == POOL_SCHEME_XOR || pc.scheme == POOL_SCHEME_JERASURE) ||
|
|
||||||
pool_item.second["pg_size"].uint64_value() > 256)
|
|
||||||
{
|
|
||||||
printf("Pool %u has invalid pg_size, skipping pool\n", pool_id);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// Parity Chunks
|
|
||||||
pc.parity_chunks = pool_item.second["parity_chunks"].uint64_value();
|
|
||||||
if (pc.scheme == POOL_SCHEME_XOR)
|
|
||||||
{
|
|
||||||
if (pc.parity_chunks > 1)
|
|
||||||
{
|
|
||||||
printf("Pool %u has invalid parity_chunks (must be 1), skipping pool\n", pool_id);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
pc.parity_chunks = 1;
|
|
||||||
}
|
|
||||||
if (pc.scheme == POOL_SCHEME_JERASURE &&
|
|
||||||
(pc.parity_chunks < 1 || pc.parity_chunks > pc.pg_size-2))
|
|
||||||
{
|
|
||||||
printf("Pool %u has invalid parity_chunks (must be between 1 and pg_size-2), skipping pool\n", pool_id);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// PG MinSize
|
|
||||||
pc.pg_minsize = pool_item.second["pg_minsize"].uint64_value();
|
|
||||||
if (pc.pg_minsize < 1 || pc.pg_minsize > pc.pg_size ||
|
|
||||||
(pc.scheme == POOL_SCHEME_XOR || pc.scheme == POOL_SCHEME_JERASURE) &&
|
|
||||||
pc.pg_minsize < (pc.pg_size-pc.parity_chunks))
|
|
||||||
{
|
|
||||||
printf("Pool %u has invalid pg_minsize, skipping pool\n", pool_id);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// PG Count
|
|
||||||
pc.pg_count = pool_item.second["pg_count"].uint64_value();
|
|
||||||
if (pc.pg_count < 1)
|
|
||||||
{
|
|
||||||
printf("Pool %u has invalid pg_count, skipping pool\n", pool_id);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// Max OSD Combinations
|
|
||||||
pc.max_osd_combinations = pool_item.second["max_osd_combinations"].uint64_value();
|
|
||||||
if (!pc.max_osd_combinations)
|
|
||||||
pc.max_osd_combinations = 10000;
|
|
||||||
if (pc.max_osd_combinations > 0 && pc.max_osd_combinations < 100)
|
|
||||||
{
|
|
||||||
printf("Pool %u has invalid max_osd_combinations (must be at least 100), skipping pool\n", pool_id);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// PG Stripe Size
|
|
||||||
pc.pg_stripe_size = pool_item.second["pg_stripe_size"].uint64_value();
|
|
||||||
uint64_t min_stripe_size = bs_block_size * (pc.scheme == POOL_SCHEME_REPLICATED ? 1 : (pc.pg_size-pc.parity_chunks));
|
|
||||||
if (pc.pg_stripe_size < min_stripe_size)
|
|
||||||
pc.pg_stripe_size = min_stripe_size;
|
|
||||||
// Save
|
|
||||||
std::swap(pc.pg_config, this->pool_config[pool_id].pg_config);
|
|
||||||
std::swap(this->pool_config[pool_id], pc);
|
|
||||||
auto & parsed_cfg = this->pool_config[pool_id];
|
|
||||||
parsed_cfg.exists = true;
|
|
||||||
for (auto & pg_item: parsed_cfg.pg_config)
|
|
||||||
{
|
|
||||||
if (pg_item.second.target_set.size() != parsed_cfg.pg_size)
|
|
||||||
{
|
|
||||||
printf("Pool %u PG %u configuration is invalid: osd_set size %lu != pool pg_size %lu\n",
|
|
||||||
pool_id, pg_item.first, pg_item.second.target_set.size(), parsed_cfg.pg_size);
|
|
||||||
pg_item.second.pause = true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else if (key == etcd_prefix+"/config/pgs")
|
|
||||||
{
|
|
||||||
for (auto & pool_item: this->pool_config)
|
|
||||||
{
|
|
||||||
for (auto & pg_item: pool_item.second.pg_config)
|
|
||||||
{
|
|
||||||
pg_item.second.exists = false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (auto & pool_item: value["items"].object_items())
|
|
||||||
{
|
|
||||||
pool_id_t pool_id = stoull_full(pool_item.first);
|
|
||||||
if (!pool_id || pool_id >= POOL_ID_MAX)
|
|
||||||
{
|
|
||||||
printf("Pool ID %s is invalid in PG configuration (must be a number less than 0x%x), skipping pool\n", pool_item.first.c_str(), POOL_ID_MAX);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
for (auto & pg_item: pool_item.second.object_items())
|
|
||||||
{
|
|
||||||
pg_num_t pg_num = stoull_full(pg_item.first);
|
|
||||||
if (!pg_num)
|
|
||||||
{
|
|
||||||
printf("Bad key in pool %u PG configuration: %s (must be a number), skipped\n", pool_id, pg_item.first.c_str());
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
auto & parsed_cfg = this->pool_config[pool_id].pg_config[pg_num];
|
|
||||||
parsed_cfg.exists = true;
|
|
||||||
parsed_cfg.pause = pg_item.second["pause"].bool_value();
|
|
||||||
parsed_cfg.primary = pg_item.second["primary"].uint64_value();
|
|
||||||
parsed_cfg.target_set.clear();
|
|
||||||
for (auto & pg_osd: pg_item.second["osd_set"].array_items())
|
|
||||||
{
|
|
||||||
parsed_cfg.target_set.push_back(pg_osd.uint64_value());
|
|
||||||
}
|
|
||||||
if (parsed_cfg.target_set.size() != pool_config[pool_id].pg_size)
|
|
||||||
{
|
|
||||||
printf("Pool %u PG %u configuration is invalid: osd_set size %lu != pool pg_size %lu\n",
|
|
||||||
pool_id, pg_num, parsed_cfg.target_set.size(), pool_config[pool_id].pg_size);
|
|
||||||
parsed_cfg.pause = true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (auto & pool_item: this->pool_config)
|
|
||||||
{
|
|
||||||
int n = 0;
|
|
||||||
for (auto pg_it = pool_item.second.pg_config.begin(); pg_it != pool_item.second.pg_config.end(); pg_it++)
|
|
||||||
{
|
|
||||||
if (pg_it->second.exists && pg_it->first != ++n)
|
|
||||||
{
|
|
||||||
printf(
|
|
||||||
"Invalid pool %u PG configuration: PG numbers don't cover whole 1..%lu range\n",
|
|
||||||
pool_item.second.id, pool_item.second.pg_config.size()
|
|
||||||
);
|
|
||||||
for (pg_it = pool_item.second.pg_config.begin(); pg_it != pool_item.second.pg_config.end(); pg_it++)
|
|
||||||
{
|
|
||||||
pg_it->second.exists = false;
|
|
||||||
}
|
|
||||||
n = 0;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
pool_item.second.real_pg_count = n;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else if (key.substr(0, etcd_prefix.length()+12) == etcd_prefix+"/pg/history/")
|
|
||||||
{
|
|
||||||
// <etcd_prefix>/pg/history/%d/%d
|
|
||||||
pool_id_t pool_id = 0;
|
|
||||||
pg_num_t pg_num = 0;
|
|
||||||
char null_byte = 0;
|
|
||||||
sscanf(key.c_str() + etcd_prefix.length()+12, "%u/%u%c", &pool_id, &pg_num, &null_byte);
|
|
||||||
if (!pool_id || pool_id >= POOL_ID_MAX || !pg_num || null_byte != 0)
|
|
||||||
{
|
|
||||||
printf("Bad etcd key %s, ignoring\n", key.c_str());
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
auto & pg_cfg = this->pool_config[pool_id].pg_config[pg_num];
|
|
||||||
pg_cfg.target_history.clear();
|
|
||||||
pg_cfg.all_peers.clear();
|
|
||||||
// Refuse to start PG if any set of the <osd_sets> has no live OSDs
|
|
||||||
for (auto hist_item: value["osd_sets"].array_items())
|
|
||||||
{
|
|
||||||
std::vector<osd_num_t> history_set;
|
|
||||||
for (auto pg_osd: hist_item.array_items())
|
|
||||||
{
|
|
||||||
history_set.push_back(pg_osd.uint64_value());
|
|
||||||
}
|
|
||||||
pg_cfg.target_history.push_back(history_set);
|
|
||||||
}
|
|
||||||
// Include these additional OSDs when peering the PG
|
|
||||||
for (auto pg_osd: value["all_peers"].array_items())
|
|
||||||
{
|
|
||||||
pg_cfg.all_peers.push_back(pg_osd.uint64_value());
|
|
||||||
}
|
|
||||||
// Read epoch
|
|
||||||
pg_cfg.epoch = value["epoch"].uint64_value();
|
|
||||||
if (on_change_pg_history_hook != NULL)
|
|
||||||
{
|
|
||||||
on_change_pg_history_hook(pool_id, pg_num);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else if (key.substr(0, etcd_prefix.length()+10) == etcd_prefix+"/pg/state/")
|
|
||||||
{
|
|
||||||
// <etcd_prefix>/pg/state/%d/%d
|
|
||||||
pool_id_t pool_id = 0;
|
|
||||||
pg_num_t pg_num = 0;
|
|
||||||
char null_byte = 0;
|
|
||||||
sscanf(key.c_str() + etcd_prefix.length()+10, "%u/%u%c", &pool_id, &pg_num, &null_byte);
|
|
||||||
if (!pool_id || pool_id >= POOL_ID_MAX || !pg_num || null_byte != 0)
|
|
||||||
{
|
|
||||||
printf("Bad etcd key %s, ignoring\n", key.c_str());
|
|
||||||
}
|
|
||||||
else if (value.is_null())
|
|
||||||
{
|
|
||||||
this->pool_config[pool_id].pg_config[pg_num].cur_primary = 0;
|
|
||||||
this->pool_config[pool_id].pg_config[pg_num].cur_state = 0;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
osd_num_t cur_primary = value["primary"].uint64_value();
|
|
||||||
int state = 0;
|
|
||||||
for (auto & e: value["state"].array_items())
|
|
||||||
{
|
|
||||||
int i;
|
|
||||||
for (i = 0; i < pg_state_bit_count; i++)
|
|
||||||
{
|
|
||||||
if (e.string_value() == pg_state_names[i])
|
|
||||||
{
|
|
||||||
state = state | pg_state_bits[i];
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (i >= pg_state_bit_count)
|
|
||||||
{
|
|
||||||
printf("Unexpected pool %u PG %u state keyword in etcd: %s\n", pool_id, pg_num, e.dump().c_str());
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (!cur_primary || !value["state"].is_array() || !state ||
|
|
||||||
(state & PG_OFFLINE) && state != PG_OFFLINE ||
|
|
||||||
(state & PG_PEERING) && state != PG_PEERING ||
|
|
||||||
(state & PG_INCOMPLETE) && state != PG_INCOMPLETE)
|
|
||||||
{
|
|
||||||
printf("Unexpected pool %u PG %u state in etcd: primary=%lu, state=%s\n", pool_id, pg_num, cur_primary, value["state"].dump().c_str());
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
this->pool_config[pool_id].pg_config[pg_num].cur_primary = cur_primary;
|
|
||||||
this->pool_config[pool_id].pg_config[pg_num].cur_state = state;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else if (key.substr(0, etcd_prefix.length()+11) == etcd_prefix+"/osd/state/")
|
|
||||||
{
|
|
||||||
// <etcd_prefix>/osd/state/%d
|
|
||||||
osd_num_t peer_osd = std::stoull(key.substr(etcd_prefix.length()+11));
|
|
||||||
if (peer_osd > 0)
|
|
||||||
{
|
|
||||||
if (value.is_object() && value["state"] == "up" &&
|
|
||||||
value["addresses"].is_array() &&
|
|
||||||
value["port"].int64_value() > 0 && value["port"].int64_value() < 65536)
|
|
||||||
{
|
|
||||||
this->peer_states[peer_osd] = value;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
this->peer_states.erase(peer_osd);
|
|
||||||
}
|
|
||||||
if (on_change_osd_state_hook != NULL)
|
|
||||||
{
|
|
||||||
on_change_osd_state_hook(peer_osd);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,84 +0,0 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
|
||||||
// License: VNPL-1.0 or GNU GPL-2.0+ (see README.md for details)
|
|
||||||
|
|
||||||
#pragma once
|
|
||||||
|
|
||||||
#include "osd_id.h"
|
|
||||||
#include "http_client.h"
|
|
||||||
#include "timerfd_manager.h"
|
|
||||||
|
|
||||||
#define ETCD_CONFIG_WATCH_ID 1
|
|
||||||
#define ETCD_PG_STATE_WATCH_ID 2
|
|
||||||
#define ETCD_PG_HISTORY_WATCH_ID 3
|
|
||||||
#define ETCD_OSD_STATE_WATCH_ID 4
|
|
||||||
|
|
||||||
#define MAX_ETCD_ATTEMPTS 5
|
|
||||||
#define ETCD_SLOW_TIMEOUT 5000
|
|
||||||
#define ETCD_QUICK_TIMEOUT 1000
|
|
||||||
|
|
||||||
#define DEFAULT_BLOCK_SIZE 128*1024
|
|
||||||
|
|
||||||
struct json_kv_t
|
|
||||||
{
|
|
||||||
std::string key;
|
|
||||||
json11::Json value;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct pg_config_t
|
|
||||||
{
|
|
||||||
bool exists;
|
|
||||||
osd_num_t primary;
|
|
||||||
std::vector<osd_num_t> target_set;
|
|
||||||
std::vector<std::vector<osd_num_t>> target_history;
|
|
||||||
std::vector<osd_num_t> all_peers;
|
|
||||||
bool pause;
|
|
||||||
osd_num_t cur_primary;
|
|
||||||
int cur_state;
|
|
||||||
uint64_t epoch;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct pool_config_t
|
|
||||||
{
|
|
||||||
bool exists;
|
|
||||||
pool_id_t id;
|
|
||||||
std::string name;
|
|
||||||
uint64_t scheme;
|
|
||||||
uint64_t pg_size, pg_minsize, parity_chunks;
|
|
||||||
uint64_t pg_count;
|
|
||||||
uint64_t real_pg_count;
|
|
||||||
std::string failure_domain;
|
|
||||||
uint64_t max_osd_combinations;
|
|
||||||
uint64_t pg_stripe_size;
|
|
||||||
std::map<pg_num_t, pg_config_t> pg_config;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct etcd_state_client_t
|
|
||||||
{
|
|
||||||
std::vector<std::string> etcd_addresses;
|
|
||||||
std::string etcd_prefix;
|
|
||||||
int log_level = 0;
|
|
||||||
timerfd_manager_t *tfd = NULL;
|
|
||||||
|
|
||||||
int etcd_watches_initialised = 0;
|
|
||||||
uint64_t etcd_watch_revision = 0;
|
|
||||||
websocket_t *etcd_watch_ws = NULL;
|
|
||||||
uint64_t bs_block_size = 0;
|
|
||||||
std::map<pool_id_t, pool_config_t> pool_config;
|
|
||||||
std::map<osd_num_t, json11::Json> peer_states;
|
|
||||||
|
|
||||||
std::function<void(json11::Json::object &)> on_change_hook;
|
|
||||||
std::function<void(json11::Json::object &)> on_load_config_hook;
|
|
||||||
std::function<json11::Json()> load_pgs_checks_hook;
|
|
||||||
std::function<void(bool)> on_load_pgs_hook;
|
|
||||||
std::function<void(pool_id_t, pg_num_t)> on_change_pg_history_hook;
|
|
||||||
std::function<void(osd_num_t)> on_change_osd_state_hook;
|
|
||||||
|
|
||||||
json_kv_t parse_etcd_kv(const json11::Json & kv_json);
|
|
||||||
void etcd_call(std::string api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback);
|
|
||||||
void etcd_txn(json11::Json txn, int timeout, std::function<void(std::string, json11::Json)> callback);
|
|
||||||
void start_etcd_watcher();
|
|
||||||
void load_global_config();
|
|
||||||
void load_pgs();
|
|
||||||
void parse_state(const std::string & key, const json11::Json & value);
|
|
||||||
void parse_config(json11::Json & config);
|
|
||||||
};
|
|
||||||
+1
-1
Submodule json11 updated: 97f06cb20c...52a3af664f
@@ -1,51 +0,0 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
|
||||||
// License: VNPL-1.0 (see README.md for details)
|
|
||||||
|
|
||||||
#include <iostream>
|
|
||||||
#include <functional>
|
|
||||||
#include <array>
|
|
||||||
#include <cstdlib> // for malloc() and free()
|
|
||||||
using namespace std;
|
|
||||||
|
|
||||||
// replace operator new and delete to log allocations
|
|
||||||
void* operator new(std::size_t n)
|
|
||||||
{
|
|
||||||
cout << "Allocating " << n << " bytes" << endl;
|
|
||||||
return malloc(n);
|
|
||||||
}
|
|
||||||
|
|
||||||
void operator delete(void* p) throw()
|
|
||||||
{
|
|
||||||
free(p);
|
|
||||||
}
|
|
||||||
|
|
||||||
class test
|
|
||||||
{
|
|
||||||
public:
|
|
||||||
std::string s;
|
|
||||||
void a(std::function<void()> & f, const char *str)
|
|
||||||
{
|
|
||||||
auto l = [this, str]() { cout << str << " ? " << s << " from this\n"; };
|
|
||||||
cout << "Assigning lambda3 of size " << sizeof(l) << endl;
|
|
||||||
f = l;
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
int main()
|
|
||||||
{
|
|
||||||
std::array<char, 16> arr1;
|
|
||||||
auto lambda1 = [arr1](){};
|
|
||||||
cout << "Assigning lambda1 of size " << sizeof(lambda1) << endl;
|
|
||||||
std::function<void()> f1 = lambda1;
|
|
||||||
|
|
||||||
std::array<char, 17> arr2;
|
|
||||||
auto lambda2 = [arr2](){};
|
|
||||||
cout << "Assigning lambda2 of size " << sizeof(lambda2) << endl;
|
|
||||||
std::function<void()> f2 = lambda2;
|
|
||||||
|
|
||||||
test t;
|
|
||||||
std::function<void()> f3;
|
|
||||||
t.s = "str";
|
|
||||||
t.a(f3, "huyambda");
|
|
||||||
f3();
|
|
||||||
}
|
|
||||||
Submodule
+1
Submodule libnfs added at 5a991e1fcb
-424
@@ -1,424 +0,0 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
|
||||||
// License: VNPL-1.0 or GNU GPL-2.0+ (see README.md for details)
|
|
||||||
|
|
||||||
#include <unistd.h>
|
|
||||||
#include <fcntl.h>
|
|
||||||
#include <sys/socket.h>
|
|
||||||
#include <sys/epoll.h>
|
|
||||||
#include <netinet/tcp.h>
|
|
||||||
#include <stdexcept>
|
|
||||||
|
|
||||||
#include "messenger.h"
|
|
||||||
|
|
||||||
osd_op_t::~osd_op_t()
|
|
||||||
{
|
|
||||||
assert(!bs_op);
|
|
||||||
assert(!op_data);
|
|
||||||
if (rmw_buf)
|
|
||||||
{
|
|
||||||
free(rmw_buf);
|
|
||||||
}
|
|
||||||
if (buf)
|
|
||||||
{
|
|
||||||
// Note: reusing osd_op_t WILL currently lead to memory leaks
|
|
||||||
// So we don't reuse it, but free it every time
|
|
||||||
free(buf);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
osd_messenger_t::~osd_messenger_t()
|
|
||||||
{
|
|
||||||
while (clients.size() > 0)
|
|
||||||
{
|
|
||||||
stop_client(clients.begin()->first);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::connect_peer(uint64_t peer_osd, json11::Json peer_state)
|
|
||||||
{
|
|
||||||
if (wanted_peers.find(peer_osd) == wanted_peers.end())
|
|
||||||
{
|
|
||||||
wanted_peers[peer_osd] = (osd_wanted_peer_t){
|
|
||||||
.address_list = peer_state["addresses"],
|
|
||||||
.port = (int)peer_state["port"].int64_value(),
|
|
||||||
};
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
wanted_peers[peer_osd].address_list = peer_state["addresses"];
|
|
||||||
wanted_peers[peer_osd].port = (int)peer_state["port"].int64_value();
|
|
||||||
}
|
|
||||||
wanted_peers[peer_osd].address_changed = true;
|
|
||||||
if (!wanted_peers[peer_osd].connecting &&
|
|
||||||
(time(NULL) - wanted_peers[peer_osd].last_connect_attempt) >= peer_connect_interval)
|
|
||||||
{
|
|
||||||
try_connect_peer(peer_osd);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::try_connect_peer(uint64_t peer_osd)
|
|
||||||
{
|
|
||||||
auto wp_it = wanted_peers.find(peer_osd);
|
|
||||||
if (wp_it == wanted_peers.end())
|
|
||||||
{
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (osd_peer_fds.find(peer_osd) != osd_peer_fds.end())
|
|
||||||
{
|
|
||||||
wanted_peers.erase(peer_osd);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
auto & wp = wp_it->second;
|
|
||||||
if (wp.address_index >= wp.address_list.array_items().size())
|
|
||||||
{
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
wp.cur_addr = wp.address_list[wp.address_index].string_value();
|
|
||||||
wp.cur_port = wp.port;
|
|
||||||
wp.connecting = true;
|
|
||||||
try_connect_peer_addr(peer_osd, wp.cur_addr.c_str(), wp.cur_port);
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::try_connect_peer_addr(osd_num_t peer_osd, const char *peer_host, int peer_port)
|
|
||||||
{
|
|
||||||
assert(peer_osd != this->osd_num);
|
|
||||||
struct sockaddr_in addr;
|
|
||||||
int r;
|
|
||||||
if ((r = inet_pton(AF_INET, peer_host, &addr.sin_addr)) != 1)
|
|
||||||
{
|
|
||||||
on_connect_peer(peer_osd, -EINVAL);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
addr.sin_family = AF_INET;
|
|
||||||
addr.sin_port = htons(peer_port ? peer_port : 11203);
|
|
||||||
int peer_fd = socket(AF_INET, SOCK_STREAM, 0);
|
|
||||||
if (peer_fd < 0)
|
|
||||||
{
|
|
||||||
on_connect_peer(peer_osd, -errno);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
fcntl(peer_fd, F_SETFL, fcntl(peer_fd, F_GETFL, 0) | O_NONBLOCK);
|
|
||||||
r = connect(peer_fd, (sockaddr*)&addr, sizeof(addr));
|
|
||||||
if (r < 0 && errno != EINPROGRESS)
|
|
||||||
{
|
|
||||||
close(peer_fd);
|
|
||||||
on_connect_peer(peer_osd, -errno);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
int timeout_id = -1;
|
|
||||||
if (peer_connect_timeout > 0)
|
|
||||||
{
|
|
||||||
timeout_id = tfd->set_timer(1000*peer_connect_timeout, false, [this, peer_fd](int timer_id)
|
|
||||||
{
|
|
||||||
osd_num_t peer_osd = clients.at(peer_fd)->osd_num;
|
|
||||||
stop_client(peer_fd);
|
|
||||||
on_connect_peer(peer_osd, -EIO);
|
|
||||||
return;
|
|
||||||
});
|
|
||||||
}
|
|
||||||
clients[peer_fd] = new osd_client_t((osd_client_t){
|
|
||||||
.peer_addr = addr,
|
|
||||||
.peer_port = peer_port,
|
|
||||||
.peer_fd = peer_fd,
|
|
||||||
.peer_state = PEER_CONNECTING,
|
|
||||||
.connect_timeout_id = timeout_id,
|
|
||||||
.osd_num = peer_osd,
|
|
||||||
.in_buf = malloc_or_die(receive_buffer_size),
|
|
||||||
});
|
|
||||||
tfd->set_fd_handler(peer_fd, true, [this](int peer_fd, int epoll_events)
|
|
||||||
{
|
|
||||||
// Either OUT (connected) or HUP
|
|
||||||
handle_connect_epoll(peer_fd);
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::handle_connect_epoll(int peer_fd)
|
|
||||||
{
|
|
||||||
auto cl = clients[peer_fd];
|
|
||||||
if (cl->connect_timeout_id >= 0)
|
|
||||||
{
|
|
||||||
tfd->clear_timer(cl->connect_timeout_id);
|
|
||||||
cl->connect_timeout_id = -1;
|
|
||||||
}
|
|
||||||
osd_num_t peer_osd = cl->osd_num;
|
|
||||||
int result = 0;
|
|
||||||
socklen_t result_len = sizeof(result);
|
|
||||||
if (getsockopt(peer_fd, SOL_SOCKET, SO_ERROR, &result, &result_len) < 0)
|
|
||||||
{
|
|
||||||
result = errno;
|
|
||||||
}
|
|
||||||
if (result != 0)
|
|
||||||
{
|
|
||||||
stop_client(peer_fd);
|
|
||||||
on_connect_peer(peer_osd, -result);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
int one = 1;
|
|
||||||
setsockopt(peer_fd, SOL_TCP, TCP_NODELAY, &one, sizeof(one));
|
|
||||||
cl->peer_state = PEER_CONNECTED;
|
|
||||||
tfd->set_fd_handler(peer_fd, false, [this](int peer_fd, int epoll_events)
|
|
||||||
{
|
|
||||||
handle_peer_epoll(peer_fd, epoll_events);
|
|
||||||
});
|
|
||||||
// Check OSD number
|
|
||||||
check_peer_config(cl);
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::handle_peer_epoll(int peer_fd, int epoll_events)
|
|
||||||
{
|
|
||||||
// Mark client as ready (i.e. some data is available)
|
|
||||||
if (epoll_events & EPOLLRDHUP)
|
|
||||||
{
|
|
||||||
// Stop client
|
|
||||||
printf("[OSD %lu] client %d disconnected\n", this->osd_num, peer_fd);
|
|
||||||
stop_client(peer_fd);
|
|
||||||
}
|
|
||||||
else if (epoll_events & EPOLLIN)
|
|
||||||
{
|
|
||||||
// Mark client as ready (i.e. some data is available)
|
|
||||||
auto cl = clients[peer_fd];
|
|
||||||
cl->read_ready++;
|
|
||||||
if (cl->read_ready == 1)
|
|
||||||
{
|
|
||||||
read_ready_clients.push_back(cl->peer_fd);
|
|
||||||
if (ringloop)
|
|
||||||
ringloop->wakeup();
|
|
||||||
else
|
|
||||||
read_requests();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::on_connect_peer(osd_num_t peer_osd, int peer_fd)
|
|
||||||
{
|
|
||||||
auto & wp = wanted_peers.at(peer_osd);
|
|
||||||
wp.connecting = false;
|
|
||||||
if (peer_fd < 0)
|
|
||||||
{
|
|
||||||
printf("Failed to connect to peer OSD %lu address %s port %d: %s\n", peer_osd, wp.cur_addr.c_str(), wp.cur_port, strerror(-peer_fd));
|
|
||||||
if (wp.address_changed)
|
|
||||||
{
|
|
||||||
wp.address_changed = false;
|
|
||||||
wp.address_index = 0;
|
|
||||||
try_connect_peer(peer_osd);
|
|
||||||
}
|
|
||||||
else if (wp.address_index < wp.address_list.array_items().size()-1)
|
|
||||||
{
|
|
||||||
// Try other addresses
|
|
||||||
wp.address_index++;
|
|
||||||
try_connect_peer(peer_osd);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// Retry again in <peer_connect_interval> seconds
|
|
||||||
wp.last_connect_attempt = time(NULL);
|
|
||||||
wp.address_index = 0;
|
|
||||||
tfd->set_timer(1000*peer_connect_interval, false, [this, peer_osd](int)
|
|
||||||
{
|
|
||||||
try_connect_peer(peer_osd);
|
|
||||||
});
|
|
||||||
}
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (log_level > 0)
|
|
||||||
{
|
|
||||||
printf("[OSD %lu] Connected with peer OSD %lu (client %d)\n", osd_num, peer_osd, peer_fd);
|
|
||||||
}
|
|
||||||
wanted_peers.erase(peer_osd);
|
|
||||||
repeer_pgs(peer_osd);
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::check_peer_config(osd_client_t *cl)
|
|
||||||
{
|
|
||||||
osd_op_t *op = new osd_op_t();
|
|
||||||
op->op_type = OSD_OP_OUT;
|
|
||||||
op->peer_fd = cl->peer_fd;
|
|
||||||
op->req = (osd_any_op_t){
|
|
||||||
.show_conf = {
|
|
||||||
.header = {
|
|
||||||
.magic = SECONDARY_OSD_OP_MAGIC,
|
|
||||||
.id = this->next_subop_id++,
|
|
||||||
.opcode = OSD_OP_SHOW_CONFIG,
|
|
||||||
},
|
|
||||||
},
|
|
||||||
};
|
|
||||||
op->callback = [this, cl](osd_op_t *op)
|
|
||||||
{
|
|
||||||
std::string json_err;
|
|
||||||
json11::Json config;
|
|
||||||
bool err = false;
|
|
||||||
if (op->reply.hdr.retval < 0)
|
|
||||||
{
|
|
||||||
err = true;
|
|
||||||
printf("Failed to get config from OSD %lu (retval=%ld), disconnecting peer\n", cl->osd_num, op->reply.hdr.retval);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
config = json11::Json::parse(std::string((char*)op->buf), json_err);
|
|
||||||
if (json_err != "")
|
|
||||||
{
|
|
||||||
err = true;
|
|
||||||
printf("Failed to get config from OSD %lu: bad JSON: %s, disconnecting peer\n", cl->osd_num, json_err.c_str());
|
|
||||||
}
|
|
||||||
else if (config["osd_num"].uint64_value() != cl->osd_num)
|
|
||||||
{
|
|
||||||
err = true;
|
|
||||||
printf("Connected to OSD %lu instead of OSD %lu, peer state is outdated, disconnecting peer\n", config["osd_num"].uint64_value(), cl->osd_num);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (err)
|
|
||||||
{
|
|
||||||
osd_num_t osd_num = cl->osd_num;
|
|
||||||
stop_client(op->peer_fd);
|
|
||||||
on_connect_peer(osd_num, -1);
|
|
||||||
delete op;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
osd_peer_fds[cl->osd_num] = cl->peer_fd;
|
|
||||||
on_connect_peer(cl->osd_num, cl->peer_fd);
|
|
||||||
delete op;
|
|
||||||
};
|
|
||||||
outbox_push(op);
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::cancel_osd_ops(osd_client_t *cl)
|
|
||||||
{
|
|
||||||
for (auto p: cl->sent_ops)
|
|
||||||
{
|
|
||||||
cancel_op(p.second);
|
|
||||||
}
|
|
||||||
cl->sent_ops.clear();
|
|
||||||
cl->outbox.clear();
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::cancel_op(osd_op_t *op)
|
|
||||||
{
|
|
||||||
if (op->op_type == OSD_OP_OUT)
|
|
||||||
{
|
|
||||||
op->reply.hdr.magic = SECONDARY_OSD_REPLY_MAGIC;
|
|
||||||
op->reply.hdr.id = op->req.hdr.id;
|
|
||||||
op->reply.hdr.opcode = op->req.hdr.opcode;
|
|
||||||
op->reply.hdr.retval = -EPIPE;
|
|
||||||
// Copy lambda to be unaffected by `delete op`
|
|
||||||
std::function<void(osd_op_t*)>(op->callback)(op);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// This function is only called in stop_client(), so it's fine to destroy the operation
|
|
||||||
delete op;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::stop_client(int peer_fd)
|
|
||||||
{
|
|
||||||
assert(peer_fd != 0);
|
|
||||||
auto it = clients.find(peer_fd);
|
|
||||||
if (it == clients.end())
|
|
||||||
{
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
uint64_t repeer_osd = 0;
|
|
||||||
osd_client_t *cl = it->second;
|
|
||||||
if (cl->peer_state == PEER_CONNECTED)
|
|
||||||
{
|
|
||||||
if (cl->osd_num)
|
|
||||||
{
|
|
||||||
// Reload configuration from etcd when the connection is dropped
|
|
||||||
if (log_level > 0)
|
|
||||||
printf("[OSD %lu] Stopping client %d (OSD peer %lu)\n", osd_num, peer_fd, cl->osd_num);
|
|
||||||
repeer_osd = cl->osd_num;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
if (log_level > 0)
|
|
||||||
printf("[OSD %lu] Stopping client %d (regular client)\n", osd_num, peer_fd);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
cl->peer_state = PEER_STOPPED;
|
|
||||||
clients.erase(it);
|
|
||||||
tfd->set_fd_handler(peer_fd, false, NULL);
|
|
||||||
if (cl->connect_timeout_id >= 0)
|
|
||||||
{
|
|
||||||
tfd->clear_timer(cl->connect_timeout_id);
|
|
||||||
cl->connect_timeout_id = -1;
|
|
||||||
}
|
|
||||||
if (cl->osd_num)
|
|
||||||
{
|
|
||||||
osd_peer_fds.erase(cl->osd_num);
|
|
||||||
}
|
|
||||||
if (cl->read_op)
|
|
||||||
{
|
|
||||||
delete cl->read_op;
|
|
||||||
cl->read_op = NULL;
|
|
||||||
}
|
|
||||||
for (auto rit = read_ready_clients.begin(); rit != read_ready_clients.end(); rit++)
|
|
||||||
{
|
|
||||||
if (*rit == peer_fd)
|
|
||||||
{
|
|
||||||
read_ready_clients.erase(rit);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (auto wit = write_ready_clients.begin(); wit != write_ready_clients.end(); wit++)
|
|
||||||
{
|
|
||||||
if (*wit == peer_fd)
|
|
||||||
{
|
|
||||||
write_ready_clients.erase(wit);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
free(cl->in_buf);
|
|
||||||
cl->in_buf = NULL;
|
|
||||||
close(peer_fd);
|
|
||||||
if (repeer_osd)
|
|
||||||
{
|
|
||||||
// First repeer PGs as canceling OSD ops may push new operations
|
|
||||||
// and we need correct PG states when we do that
|
|
||||||
repeer_pgs(repeer_osd);
|
|
||||||
}
|
|
||||||
if (cl->osd_num)
|
|
||||||
{
|
|
||||||
// Cancel outbound operations
|
|
||||||
cancel_osd_ops(cl);
|
|
||||||
}
|
|
||||||
if (cl->refs <= 0)
|
|
||||||
{
|
|
||||||
delete cl;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::accept_connections(int listen_fd)
|
|
||||||
{
|
|
||||||
// Accept new connections
|
|
||||||
sockaddr_in addr;
|
|
||||||
socklen_t peer_addr_size = sizeof(addr);
|
|
||||||
int peer_fd;
|
|
||||||
while ((peer_fd = accept(listen_fd, (sockaddr*)&addr, &peer_addr_size)) >= 0)
|
|
||||||
{
|
|
||||||
assert(peer_fd != 0);
|
|
||||||
char peer_str[256];
|
|
||||||
printf("[OSD %lu] new client %d: connection from %s port %d\n", this->osd_num, peer_fd,
|
|
||||||
inet_ntop(AF_INET, &addr.sin_addr, peer_str, 256), ntohs(addr.sin_port));
|
|
||||||
fcntl(peer_fd, F_SETFL, fcntl(peer_fd, F_GETFL, 0) | O_NONBLOCK);
|
|
||||||
int one = 1;
|
|
||||||
setsockopt(peer_fd, SOL_TCP, TCP_NODELAY, &one, sizeof(one));
|
|
||||||
clients[peer_fd] = new osd_client_t((osd_client_t){
|
|
||||||
.peer_addr = addr,
|
|
||||||
.peer_port = ntohs(addr.sin_port),
|
|
||||||
.peer_fd = peer_fd,
|
|
||||||
.peer_state = PEER_CONNECTED,
|
|
||||||
.in_buf = malloc_or_die(receive_buffer_size),
|
|
||||||
});
|
|
||||||
// Add FD to epoll
|
|
||||||
tfd->set_fd_handler(peer_fd, false, [this](int peer_fd, int epoll_events)
|
|
||||||
{
|
|
||||||
handle_peer_epoll(peer_fd, epoll_events);
|
|
||||||
});
|
|
||||||
// Try to accept next connection
|
|
||||||
peer_addr_size = sizeof(addr);
|
|
||||||
}
|
|
||||||
if (peer_fd == -1 && errno != EAGAIN)
|
|
||||||
{
|
|
||||||
throw std::runtime_error(std::string("accept: ") + strerror(errno));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
-306
@@ -1,306 +0,0 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
|
||||||
// License: VNPL-1.0 or GNU GPL-2.0+ (see README.md for details)
|
|
||||||
|
|
||||||
#pragma once
|
|
||||||
|
|
||||||
#include <sys/types.h>
|
|
||||||
#include <stdint.h>
|
|
||||||
#include <arpa/inet.h>
|
|
||||||
|
|
||||||
#include <set>
|
|
||||||
#include <map>
|
|
||||||
#include <deque>
|
|
||||||
#include <vector>
|
|
||||||
|
|
||||||
#include "malloc_or_die.h"
|
|
||||||
#include "json11/json11.hpp"
|
|
||||||
#include "osd_ops.h"
|
|
||||||
#include "timerfd_manager.h"
|
|
||||||
#include "ringloop.h"
|
|
||||||
|
|
||||||
#define OSD_OP_IN 0
|
|
||||||
#define OSD_OP_OUT 1
|
|
||||||
|
|
||||||
#define CL_READ_HDR 1
|
|
||||||
#define CL_READ_DATA 2
|
|
||||||
#define CL_READ_REPLY_DATA 3
|
|
||||||
#define CL_WRITE_READY 1
|
|
||||||
#define CL_WRITE_REPLY 2
|
|
||||||
#define OSD_OP_INLINE_BUF_COUNT 16
|
|
||||||
|
|
||||||
#define PEER_CONNECTING 1
|
|
||||||
#define PEER_CONNECTED 2
|
|
||||||
#define PEER_STOPPED 3
|
|
||||||
|
|
||||||
#define DEFAULT_PEER_CONNECT_INTERVAL 5
|
|
||||||
#define DEFAULT_PEER_CONNECT_TIMEOUT 5
|
|
||||||
|
|
||||||
// Kind of a vector with small-list-optimisation
|
|
||||||
struct osd_op_buf_list_t
|
|
||||||
{
|
|
||||||
int count = 0, alloc = OSD_OP_INLINE_BUF_COUNT, done = 0;
|
|
||||||
iovec *buf = NULL;
|
|
||||||
iovec inline_buf[OSD_OP_INLINE_BUF_COUNT];
|
|
||||||
|
|
||||||
inline osd_op_buf_list_t()
|
|
||||||
{
|
|
||||||
buf = inline_buf;
|
|
||||||
}
|
|
||||||
|
|
||||||
inline osd_op_buf_list_t(const osd_op_buf_list_t & other)
|
|
||||||
{
|
|
||||||
buf = inline_buf;
|
|
||||||
append(other);
|
|
||||||
}
|
|
||||||
|
|
||||||
inline osd_op_buf_list_t & operator = (const osd_op_buf_list_t & other)
|
|
||||||
{
|
|
||||||
reset();
|
|
||||||
append(other);
|
|
||||||
return *this;
|
|
||||||
}
|
|
||||||
|
|
||||||
inline ~osd_op_buf_list_t()
|
|
||||||
{
|
|
||||||
if (buf && buf != inline_buf)
|
|
||||||
{
|
|
||||||
free(buf);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
inline void reset()
|
|
||||||
{
|
|
||||||
count = 0;
|
|
||||||
done = 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
inline iovec* get_iovec()
|
|
||||||
{
|
|
||||||
return buf + done;
|
|
||||||
}
|
|
||||||
|
|
||||||
inline int get_size()
|
|
||||||
{
|
|
||||||
return count - done;
|
|
||||||
}
|
|
||||||
|
|
||||||
inline void append(const osd_op_buf_list_t & other)
|
|
||||||
{
|
|
||||||
if (count+other.count > alloc)
|
|
||||||
{
|
|
||||||
if (buf == inline_buf)
|
|
||||||
{
|
|
||||||
int old = alloc;
|
|
||||||
alloc = (((count+other.count+15)/16)*16);
|
|
||||||
buf = (iovec*)malloc(sizeof(iovec) * alloc);
|
|
||||||
if (!buf)
|
|
||||||
{
|
|
||||||
printf("Failed to allocate %lu bytes\n", sizeof(iovec) * alloc);
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
memcpy(buf, inline_buf, sizeof(iovec) * old);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
alloc = (((count+other.count+15)/16)*16);
|
|
||||||
buf = (iovec*)realloc(buf, sizeof(iovec) * alloc);
|
|
||||||
if (!buf)
|
|
||||||
{
|
|
||||||
printf("Failed to allocate %lu bytes\n", sizeof(iovec) * alloc);
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (int i = 0; i < other.count; i++)
|
|
||||||
{
|
|
||||||
buf[count++] = other.buf[i];
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
inline void push_back(void *nbuf, size_t len)
|
|
||||||
{
|
|
||||||
if (count >= alloc)
|
|
||||||
{
|
|
||||||
if (buf == inline_buf)
|
|
||||||
{
|
|
||||||
int old = alloc;
|
|
||||||
alloc = ((alloc/16)*16 + 1);
|
|
||||||
buf = (iovec*)malloc(sizeof(iovec) * alloc);
|
|
||||||
if (!buf)
|
|
||||||
{
|
|
||||||
printf("Failed to allocate %lu bytes\n", sizeof(iovec) * alloc);
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
memcpy(buf, inline_buf, sizeof(iovec)*old);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
alloc = alloc < 16 ? 16 : (alloc+16);
|
|
||||||
buf = (iovec*)realloc(buf, sizeof(iovec) * alloc);
|
|
||||||
if (!buf)
|
|
||||||
{
|
|
||||||
printf("Failed to allocate %lu bytes\n", sizeof(iovec) * alloc);
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
buf[count++] = { .iov_base = nbuf, .iov_len = len };
|
|
||||||
}
|
|
||||||
|
|
||||||
inline void eat(int result)
|
|
||||||
{
|
|
||||||
while (result > 0 && done < count)
|
|
||||||
{
|
|
||||||
iovec & iov = buf[done];
|
|
||||||
if (iov.iov_len <= result)
|
|
||||||
{
|
|
||||||
result -= iov.iov_len;
|
|
||||||
done++;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
iov.iov_len -= result;
|
|
||||||
iov.iov_base += result;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
struct blockstore_op_t;
|
|
||||||
|
|
||||||
struct osd_primary_op_data_t;
|
|
||||||
|
|
||||||
struct osd_op_t
|
|
||||||
{
|
|
||||||
timespec tv_begin;
|
|
||||||
uint64_t op_type = OSD_OP_IN;
|
|
||||||
int peer_fd;
|
|
||||||
osd_any_op_t req;
|
|
||||||
osd_any_reply_t reply;
|
|
||||||
blockstore_op_t *bs_op = NULL;
|
|
||||||
void *buf = NULL;
|
|
||||||
void *rmw_buf = NULL;
|
|
||||||
osd_primary_op_data_t* op_data = NULL;
|
|
||||||
std::function<void(osd_op_t*)> callback;
|
|
||||||
|
|
||||||
osd_op_buf_list_t iov;
|
|
||||||
|
|
||||||
~osd_op_t();
|
|
||||||
};
|
|
||||||
|
|
||||||
struct osd_client_t
|
|
||||||
{
|
|
||||||
int refs = 0;
|
|
||||||
|
|
||||||
sockaddr_in peer_addr;
|
|
||||||
int peer_port;
|
|
||||||
int peer_fd;
|
|
||||||
int peer_state;
|
|
||||||
int connect_timeout_id = -1;
|
|
||||||
osd_num_t osd_num = 0;
|
|
||||||
|
|
||||||
void *in_buf = NULL;
|
|
||||||
|
|
||||||
// Read state
|
|
||||||
int read_ready = 0;
|
|
||||||
osd_op_t *read_op = NULL;
|
|
||||||
iovec read_iov = { 0 };
|
|
||||||
msghdr read_msg = { 0 };
|
|
||||||
int read_remaining = 0;
|
|
||||||
int read_state = 0;
|
|
||||||
osd_op_buf_list_t recv_list;
|
|
||||||
|
|
||||||
// Incoming operations
|
|
||||||
std::vector<osd_op_t*> received_ops;
|
|
||||||
|
|
||||||
// Outbound operations
|
|
||||||
std::map<uint64_t, osd_op_t*> sent_ops;
|
|
||||||
|
|
||||||
// PGs dirtied by this client's primary-writes
|
|
||||||
std::set<pool_pg_num_t> dirty_pgs;
|
|
||||||
|
|
||||||
// Write state
|
|
||||||
msghdr write_msg = { 0 };
|
|
||||||
int write_state = 0;
|
|
||||||
std::vector<iovec> send_list, next_send_list;
|
|
||||||
std::vector<osd_op_t*> outbox, next_outbox;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct osd_wanted_peer_t
|
|
||||||
{
|
|
||||||
json11::Json address_list;
|
|
||||||
int port;
|
|
||||||
time_t last_connect_attempt;
|
|
||||||
bool connecting, address_changed;
|
|
||||||
int address_index;
|
|
||||||
std::string cur_addr;
|
|
||||||
int cur_port;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct osd_op_stats_t
|
|
||||||
{
|
|
||||||
uint64_t op_stat_sum[OSD_OP_MAX+1] = { 0 };
|
|
||||||
uint64_t op_stat_count[OSD_OP_MAX+1] = { 0 };
|
|
||||||
uint64_t op_stat_bytes[OSD_OP_MAX+1] = { 0 };
|
|
||||||
uint64_t subop_stat_sum[OSD_OP_MAX+1] = { 0 };
|
|
||||||
uint64_t subop_stat_count[OSD_OP_MAX+1] = { 0 };
|
|
||||||
};
|
|
||||||
|
|
||||||
struct osd_messenger_t
|
|
||||||
{
|
|
||||||
timerfd_manager_t *tfd;
|
|
||||||
ring_loop_t *ringloop;
|
|
||||||
|
|
||||||
// osd_num_t is only for logging and asserts
|
|
||||||
osd_num_t osd_num;
|
|
||||||
// FIXME: make receive_buffer_size configurable
|
|
||||||
int receive_buffer_size = 64*1024;
|
|
||||||
int peer_connect_interval = DEFAULT_PEER_CONNECT_INTERVAL;
|
|
||||||
int peer_connect_timeout = DEFAULT_PEER_CONNECT_TIMEOUT;
|
|
||||||
int log_level = 0;
|
|
||||||
bool use_sync_send_recv = false;
|
|
||||||
|
|
||||||
std::map<osd_num_t, osd_wanted_peer_t> wanted_peers;
|
|
||||||
std::map<uint64_t, int> osd_peer_fds;
|
|
||||||
uint64_t next_subop_id = 1;
|
|
||||||
|
|
||||||
std::map<int, osd_client_t*> clients;
|
|
||||||
std::vector<int> read_ready_clients;
|
|
||||||
std::vector<int> write_ready_clients;
|
|
||||||
std::vector<std::function<void()>> set_immediate;
|
|
||||||
|
|
||||||
// op statistics
|
|
||||||
osd_op_stats_t stats;
|
|
||||||
|
|
||||||
public:
|
|
||||||
void connect_peer(uint64_t osd_num, json11::Json peer_state);
|
|
||||||
void stop_client(int peer_fd);
|
|
||||||
void outbox_push(osd_op_t *cur_op);
|
|
||||||
std::function<void(osd_op_t*)> exec_op;
|
|
||||||
std::function<void(osd_num_t)> repeer_pgs;
|
|
||||||
void handle_peer_epoll(int peer_fd, int epoll_events);
|
|
||||||
void read_requests();
|
|
||||||
void send_replies();
|
|
||||||
void accept_connections(int listen_fd);
|
|
||||||
~osd_messenger_t();
|
|
||||||
|
|
||||||
protected:
|
|
||||||
void try_connect_peer(uint64_t osd_num);
|
|
||||||
void try_connect_peer_addr(osd_num_t peer_osd, const char *peer_host, int peer_port);
|
|
||||||
void handle_connect_epoll(int peer_fd);
|
|
||||||
void on_connect_peer(osd_num_t peer_osd, int peer_fd);
|
|
||||||
void check_peer_config(osd_client_t *cl);
|
|
||||||
void cancel_osd_ops(osd_client_t *cl);
|
|
||||||
void cancel_op(osd_op_t *op);
|
|
||||||
|
|
||||||
bool try_send(osd_client_t *cl);
|
|
||||||
void measure_exec(osd_op_t *cur_op);
|
|
||||||
void handle_send(int result, osd_client_t *cl);
|
|
||||||
|
|
||||||
bool handle_read(int result, osd_client_t *cl);
|
|
||||||
bool handle_finished_read(osd_client_t *cl);
|
|
||||||
void handle_op_hdr(osd_client_t *cl);
|
|
||||||
bool handle_reply_hdr(osd_client_t *cl);
|
|
||||||
void handle_reply_ready(osd_op_t *op);
|
|
||||||
};
|
|
||||||
+53
-61
@@ -1,22 +1,59 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.0 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
module.exports = {
|
module.exports = {
|
||||||
scale_pg_count,
|
scale_pg_count,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
function add_pg_history(new_pg_history, new_pg, prev_pgs, prev_pg_history, old_pg)
|
||||||
|
{
|
||||||
|
if (!new_pg_history[new_pg])
|
||||||
|
{
|
||||||
|
new_pg_history[new_pg] = {
|
||||||
|
osd_sets: {},
|
||||||
|
all_peers: {},
|
||||||
|
epoch: 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
const nh = new_pg_history[new_pg], oh = prev_pg_history[old_pg];
|
||||||
|
nh.osd_sets[prev_pgs[old_pg].join(' ')] = prev_pgs[old_pg];
|
||||||
|
if (oh && oh.osd_sets && oh.osd_sets.length)
|
||||||
|
{
|
||||||
|
for (const pg of oh.osd_sets)
|
||||||
|
{
|
||||||
|
nh.osd_sets[pg.join(' ')] = pg;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (oh && oh.all_peers && oh.all_peers.length)
|
||||||
|
{
|
||||||
|
for (const osd_num of oh.all_peers)
|
||||||
|
{
|
||||||
|
nh.all_peers[osd_num] = Number(osd_num);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (oh && oh.epoch)
|
||||||
|
{
|
||||||
|
nh.epoch = nh.epoch < oh.epoch ? oh.epoch : nh.epoch;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function finish_pg_history(merged_history)
|
||||||
|
{
|
||||||
|
merged_history.osd_sets = Object.values(merged_history.osd_sets);
|
||||||
|
merged_history.all_peers = Object.values(merged_history.all_peers);
|
||||||
|
}
|
||||||
|
|
||||||
function scale_pg_count(prev_pgs, prev_pg_history, new_pg_history, new_pg_count)
|
function scale_pg_count(prev_pgs, prev_pg_history, new_pg_history, new_pg_count)
|
||||||
{
|
{
|
||||||
const old_pg_count = prev_pgs.length;
|
const old_pg_count = prev_pgs.length;
|
||||||
// Add all possibly intersecting PGs into the history of new PGs
|
// Add all possibly intersecting PGs to the history of new PGs
|
||||||
if (!(new_pg_count % old_pg_count))
|
if (!(new_pg_count % old_pg_count))
|
||||||
{
|
{
|
||||||
// New PG count is a multiple of the old PG count
|
// New PG count is a multiple of old PG count
|
||||||
const mul = (new_pg_count / old_pg_count);
|
|
||||||
for (let i = 0; i < new_pg_count; i++)
|
for (let i = 0; i < new_pg_count; i++)
|
||||||
{
|
{
|
||||||
const old_i = Math.floor(new_pg_count / mul);
|
add_pg_history(new_pg_history, i, prev_pgs, prev_pg_history, i % old_pg_count);
|
||||||
new_pg_history[i] = JSON.parse(JSON.stringify(prev_pg_history[1+old_i]));
|
finish_pg_history(new_pg_history[i]);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
else if (!(old_pg_count % new_pg_count))
|
else if (!(old_pg_count % new_pg_count))
|
||||||
@@ -25,68 +62,26 @@ function scale_pg_count(prev_pgs, prev_pg_history, new_pg_history, new_pg_count)
|
|||||||
const mul = (old_pg_count / new_pg_count);
|
const mul = (old_pg_count / new_pg_count);
|
||||||
for (let i = 0; i < new_pg_count; i++)
|
for (let i = 0; i < new_pg_count; i++)
|
||||||
{
|
{
|
||||||
new_pg_history[i] = {
|
|
||||||
osd_sets: [],
|
|
||||||
all_peers: [],
|
|
||||||
epoch: 0,
|
|
||||||
};
|
|
||||||
for (let j = 0; j < mul; j++)
|
for (let j = 0; j < mul; j++)
|
||||||
{
|
{
|
||||||
new_pg_history[i].osd_sets.push(prev_pgs[i*mul]);
|
add_pg_history(new_pg_history, i, prev_pgs, prev_pg_history, i+j*new_pg_count);
|
||||||
const hist = prev_pg_history[1+i*mul+j];
|
|
||||||
if (hist && hist.osd_sets && hist.osd_sets.length)
|
|
||||||
{
|
|
||||||
Array.prototype.push.apply(new_pg_history[i].osd_sets, hist.osd_sets);
|
|
||||||
}
|
|
||||||
if (hist && hist.all_peers && hist.all_peers.length)
|
|
||||||
{
|
|
||||||
Array.prototype.push.apply(new_pg_history[i].all_peers, hist.all_peers);
|
|
||||||
}
|
|
||||||
if (hist && hist.epoch)
|
|
||||||
{
|
|
||||||
new_pg_history[i].epoch = new_pg_history[i].epoch < hist.epoch ? hist.epoch : new_pg_history[i].epoch;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
finish_pg_history(new_pg_history[i]);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
// Any PG may intersect with any PG after non-multiple PG count change
|
// Any PG may intersect with any PG after non-multiple PG count change
|
||||||
// So, merge ALL PGs history
|
// So, merge ALL PGs history
|
||||||
let all_sets = {};
|
let merged_history = {};
|
||||||
let all_peers = {};
|
for (let i = 0; i < old_pg_count; i++)
|
||||||
let max_epoch = 0;
|
|
||||||
for (const pg of prev_pgs)
|
|
||||||
{
|
{
|
||||||
all_sets[pg.join(' ')] = pg;
|
add_pg_history(merged_history, 1, prev_pgs, prev_pg_history, i);
|
||||||
}
|
}
|
||||||
for (const pg in prev_pg_history)
|
finish_pg_history(merged_history[1]);
|
||||||
{
|
|
||||||
const hist = prev_pg_history[pg];
|
|
||||||
if (hist && hist.osd_sets)
|
|
||||||
{
|
|
||||||
for (const pg of hist.osd_sets)
|
|
||||||
{
|
|
||||||
all_sets[pg.join(' ')] = pg;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (hist && hist.all_peers)
|
|
||||||
{
|
|
||||||
for (const osd_num of hist.all_peers)
|
|
||||||
{
|
|
||||||
all_peers[osd_num] = Number(osd_num);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (hist && hist.epoch)
|
|
||||||
{
|
|
||||||
max_epoch = max_epoch < hist.epoch ? hist.epoch : max_epoch;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
all_sets = Object.values(all_sets);
|
|
||||||
all_peers = Object.values(all_peers);
|
|
||||||
for (let i = 0; i < new_pg_count; i++)
|
for (let i = 0; i < new_pg_count; i++)
|
||||||
{
|
{
|
||||||
new_pg_history[i] = { osd_sets: all_sets, all_peers, epoch: max_epoch };
|
new_pg_history[i] = { ...merged_history[1] };
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Mark history keys for removed PGs as removed
|
// Mark history keys for removed PGs as removed
|
||||||
@@ -94,19 +89,16 @@ function scale_pg_count(prev_pgs, prev_pg_history, new_pg_history, new_pg_count)
|
|||||||
{
|
{
|
||||||
new_pg_history[i] = null;
|
new_pg_history[i] = null;
|
||||||
}
|
}
|
||||||
|
// Just for the lp_solve optimizer - pick a "previous" PG for each "new" one
|
||||||
if (old_pg_count < new_pg_count)
|
if (old_pg_count < new_pg_count)
|
||||||
{
|
{
|
||||||
for (let i = new_pg_count-1; i >= 0; i--)
|
for (let i = old_pg_count; i < new_pg_count; i++)
|
||||||
{
|
{
|
||||||
prev_pgs[i] = prev_pgs[Math.floor(i/new_pg_count*old_pg_count)];
|
prev_pgs[i] = prev_pgs[i % old_pg_count];
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
else if (old_pg_count > new_pg_count)
|
else if (old_pg_count > new_pg_count)
|
||||||
{
|
{
|
||||||
for (let i = 0; i < new_pg_count; i++)
|
|
||||||
{
|
|
||||||
prev_pgs[i] = prev_pgs[Math.round(i/new_pg_count*old_pg_count)];
|
|
||||||
}
|
|
||||||
prev_pgs.splice(new_pg_count, old_pg_count-new_pg_count);
|
prev_pgs.splice(new_pg_count, old_pg_count-new_pg_count);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+23
-93
@@ -1,31 +1,16 @@
|
|||||||
// Functions to calculate Annualized Failure Rate of your cluster
|
// Functions to calculate Annualized Failure Rate of your cluster
|
||||||
// if you know AFR of your drives, number of drives, expected rebalance time
|
// if you know AFR of your drives, number of drives, expected rebalance time
|
||||||
// and replication factor
|
// and replication factor
|
||||||
// License: VNPL-1.0 (see README.md for details)
|
// License: VNPL-1.1 (see https://yourcmc.ru/git/vitalif/vitastor/src/branch/master/README.md for details) or AGPL-3.0
|
||||||
|
// Author: Vitaliy Filippov, 2020+
|
||||||
const { sprintf } = require('sprintf-js');
|
|
||||||
|
|
||||||
module.exports = {
|
module.exports = {
|
||||||
cluster_afr_fullmesh,
|
cluster_afr_fullmesh,
|
||||||
failure_rate_fullmesh,
|
failure_rate_fullmesh,
|
||||||
cluster_afr,
|
cluster_afr,
|
||||||
print_cluster_afr,
|
|
||||||
c_n_k,
|
c_n_k,
|
||||||
};
|
};
|
||||||
|
|
||||||
print_cluster_afr({ n_hosts: 4, n_drives: 6, afr_drive: 0.03, afr_host: 0.05, capacity: 4000, speed: 0.1, replicas: 2 });
|
|
||||||
print_cluster_afr({ n_hosts: 4, n_drives: 3, afr_drive: 0.03, capacity: 4000, speed: 0.1, replicas: 2 });
|
|
||||||
print_cluster_afr({ n_hosts: 4, n_drives: 3, afr_drive: 0.03, afr_host: 0.05, capacity: 4000, speed: 0.1, replicas: 2 });
|
|
||||||
print_cluster_afr({ n_hosts: 4, n_drives: 3, afr_drive: 0.03, capacity: 4000, speed: 0.1, ec: [ 2, 1 ] });
|
|
||||||
print_cluster_afr({ n_hosts: 4, n_drives: 3, afr_drive: 0.03, afr_host: 0.05, capacity: 4000, speed: 0.1, ec: [ 2, 1 ] });
|
|
||||||
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, capacity: 8000, speed: 0.02, replicas: 2 });
|
|
||||||
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, afr_host: 0.05, capacity: 8000, speed: 0.02, replicas: 2 });
|
|
||||||
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, capacity: 8000, speed: 0.02, replicas: 3 });
|
|
||||||
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, afr_host: 0.05, capacity: 8000, speed: 0.02, replicas: 3 });
|
|
||||||
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, capacity: 8000, speed: 0.02, replicas: 3, pgs: 100 });
|
|
||||||
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, afr_host: 0.05, capacity: 8000, speed: 0.02, replicas: 3, pgs: 100 });
|
|
||||||
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, afr_host: 0.05, capacity: 8000, speed: 0.02, replicas: 3, pgs: 100, degraded_replacement: 1 });
|
|
||||||
|
|
||||||
/******** "FULL MESH": ASSUME EACH OSD COMMUNICATES WITH ALL OTHER OSDS ********/
|
/******** "FULL MESH": ASSUME EACH OSD COMMUNICATES WITH ALL OTHER OSDS ********/
|
||||||
|
|
||||||
// Estimate AFR of the cluster
|
// Estimate AFR of the cluster
|
||||||
@@ -56,93 +41,38 @@ function failure_rate_fullmesh(n, a, f)
|
|||||||
/******** PGS: EACH OSD ONLY COMMUNICATES WITH <pgs> OTHER OSDs ********/
|
/******** PGS: EACH OSD ONLY COMMUNICATES WITH <pgs> OTHER OSDs ********/
|
||||||
|
|
||||||
// <n> hosts of <m> drives of <capacity> GB, each able to backfill at <speed> GB/s,
|
// <n> hosts of <m> drives of <capacity> GB, each able to backfill at <speed> GB/s,
|
||||||
// <k> replicas, <pgs> unique peer PGs per OSD
|
// <k> replicas, <pgs> unique peer PGs per OSD (~50 for 100 PG-per-OSD in a big cluster)
|
||||||
//
|
//
|
||||||
// For each of n*m drives: P(drive fails in a year) * P(any of its peers fail in <l*365> next days).
|
// For each of n*m drives: P(drive fails in a year) * P(any of its peers fail in <l*365> next days).
|
||||||
// More peers per OSD increase rebalance speed (more drives work together to resilver) if you
|
// More peers per OSD increase rebalance speed (more drives work together to resilver) if you
|
||||||
// let them finish rebalance BEFORE replacing the failed drive.
|
// let them finish rebalance BEFORE replacing the failed drive (degraded_replacement=false).
|
||||||
// At the same time, more peers per OSD increase probability of any of them to fail!
|
// At the same time, more peers per OSD increase probability of any of them to fail!
|
||||||
|
// osd_rm=true means that failed OSDs' data is rebalanced over all other hosts,
|
||||||
|
// not over the same host as it's in Ceph by default (dead OSDs are marked 'out').
|
||||||
//
|
//
|
||||||
// Probability of all except one drives in a replica group to fail is (AFR^(k-1)).
|
// Probability of all except one drives in a replica group to fail is (AFR^(k-1)).
|
||||||
// So with <x> PGs it becomes ~ (x * (AFR*L/365)^(k-1)). Interesting but reasonable consequence
|
// So with <x> PGs it becomes ~ (x * (AFR*L/365)^(k-1)). Interesting but reasonable consequence
|
||||||
// is that, with k=2, total failure rate doesn't depend on number of peers per OSD,
|
// is that, with k=2, total failure rate doesn't depend on number of peers per OSD,
|
||||||
// because it gets increased linearly by increased number of peers to fail
|
// because it gets increased linearly by increased number of peers to fail
|
||||||
// and decreased linearly by reduced rebalance time.
|
// and decreased linearly by reduced rebalance time.
|
||||||
function cluster_afr_pgs({ n_hosts, n_drives, afr_drive, capacity, speed, replicas, pgs = 1, degraded_replacement })
|
function cluster_afr({ n_hosts, n_drives, afr_drive, afr_host, capacity, speed, ec, ec_data, ec_parity, replicas, pgs = 1, osd_rm, degraded_replacement, down_out_interval = 600 })
|
||||||
{
|
{
|
||||||
pgs = Math.min(pgs, (n_hosts-1)*n_drives/(replicas-1));
|
const pg_size = (ec ? ec_data+ec_parity : replicas);
|
||||||
const l = capacity/(degraded_replacement ? 1 : pgs)/speed/86400/365;
|
pgs = Math.min(pgs, (n_hosts-1)*n_drives/(pg_size-1));
|
||||||
return 1 - (1 - afr_drive * (1-(1-(afr_drive*l)**(replicas-1))**pgs)) ** (n_hosts*n_drives);
|
const host_pgs = Math.min(pgs*n_drives, (n_hosts-1)*n_drives/(pg_size-1));
|
||||||
}
|
const resilver_disk = n_drives == 1 || osd_rm ? pgs : (n_drives-1);
|
||||||
|
const disk_heal_time = (down_out_interval + capacity/(degraded_replacement ? 1 : resilver_disk)/speed)/86400/365;
|
||||||
function cluster_afr_pgs_ec({ n_hosts, n_drives, afr_drive, capacity, speed, ec: [ ec_data, ec_parity ], pgs = 1, degraded_replacement })
|
const host_heal_time = (down_out_interval + n_drives*capacity/pgs/speed)/86400/365;
|
||||||
{
|
const disk_heal_fail = ((afr_drive+afr_host/n_drives)*disk_heal_time);
|
||||||
const ec_total = ec_data+ec_parity;
|
const host_heal_fail = ((afr_drive+afr_host/n_drives)*host_heal_time);
|
||||||
pgs = Math.min(pgs, (n_hosts-1)*n_drives/(ec_total-1));
|
const disk_pg_fail = ec
|
||||||
const l = capacity/(degraded_replacement ? 1 : pgs)/speed/86400/365;
|
? failure_rate_fullmesh(ec_data+ec_parity-1, disk_heal_fail, ec_parity)
|
||||||
return 1 - (1 - afr_drive * (1-(1-failure_rate_fullmesh(ec_total-1, afr_drive*l, ec_parity))**pgs)) ** (n_hosts*n_drives);
|
: disk_heal_fail**(replicas-1);
|
||||||
}
|
const host_pg_fail = ec
|
||||||
|
? failure_rate_fullmesh(ec_data+ec_parity-1, host_heal_fail, ec_parity)
|
||||||
// Same as above, but also take server failures into account
|
: host_heal_fail**(replicas-1);
|
||||||
function cluster_afr_pgs_hosts({ n_hosts, n_drives, afr_drive, afr_host, capacity, speed, replicas, pgs = 1, degraded_replacement })
|
return 1 - ((1 - afr_drive * (1-(1-disk_pg_fail)**pgs)) ** (n_hosts*n_drives))
|
||||||
{
|
* ((1 - afr_host * (1-(1-host_pg_fail)**host_pgs)) ** n_hosts);
|
||||||
let otherhosts = Math.min(pgs, (n_hosts-1)/(replicas-1));
|
|
||||||
pgs = Math.min(pgs, (n_hosts-1)*n_drives/(replicas-1));
|
|
||||||
let pgh = Math.min(pgs*n_drives, (n_hosts-1)*n_drives/(replicas-1));
|
|
||||||
const ld = capacity/(degraded_replacement ? 1 : pgs)/speed/86400/365;
|
|
||||||
const lh = n_drives*capacity/pgs/speed/86400/365;
|
|
||||||
const p1 = ((afr_drive+afr_host*pgs/otherhosts)*lh);
|
|
||||||
const p2 = ((afr_drive+afr_host*pgs/otherhosts)*ld);
|
|
||||||
return 1 - ((1 - afr_host * (1-(1-p1**(replicas-1))**pgh)) ** n_hosts) *
|
|
||||||
((1 - afr_drive * (1-(1-p2**(replicas-1))**pgs)) ** (n_hosts*n_drives));
|
|
||||||
}
|
|
||||||
|
|
||||||
function cluster_afr_pgs_ec_hosts({ n_hosts, n_drives, afr_drive, afr_host, capacity, speed, ec: [ ec_data, ec_parity ], pgs = 1, degraded_replacement })
|
|
||||||
{
|
|
||||||
const ec_total = ec_data+ec_parity;
|
|
||||||
const otherhosts = Math.min(pgs, (n_hosts-1)/(ec_total-1));
|
|
||||||
pgs = Math.min(pgs, (n_hosts-1)*n_drives/(ec_total-1));
|
|
||||||
const pgh = Math.min(pgs*n_drives, (n_hosts-1)*n_drives/(ec_total-1));
|
|
||||||
const ld = capacity/(degraded_replacement ? 1 : pgs)/speed/86400/365;
|
|
||||||
const lh = n_drives*capacity/pgs/speed/86400/365;
|
|
||||||
const p1 = ((afr_drive+afr_host*pgs/otherhosts)*lh);
|
|
||||||
const p2 = ((afr_drive+afr_host*pgs/otherhosts)*ld);
|
|
||||||
return 1 - ((1 - afr_host * (1-(1-failure_rate_fullmesh(ec_total-1, p1, ec_parity))**pgh)) ** n_hosts) *
|
|
||||||
((1 - afr_drive * (1-(1-failure_rate_fullmesh(ec_total-1, p2, ec_parity))**pgs)) ** (n_hosts*n_drives));
|
|
||||||
}
|
|
||||||
|
|
||||||
// Wrapper for 4 above functions
|
|
||||||
function cluster_afr(config)
|
|
||||||
{
|
|
||||||
if (config.ec && config.afr_host)
|
|
||||||
{
|
|
||||||
return cluster_afr_pgs_ec_hosts(config);
|
|
||||||
}
|
|
||||||
else if (config.ec)
|
|
||||||
{
|
|
||||||
return cluster_afr_pgs_ec(config);
|
|
||||||
}
|
|
||||||
else if (config.afr_host)
|
|
||||||
{
|
|
||||||
return cluster_afr_pgs_hosts(config);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
return cluster_afr_pgs(config);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
function print_cluster_afr(config)
|
|
||||||
{
|
|
||||||
console.log(
|
|
||||||
`${config.n_hosts} nodes with ${config.n_drives} ${sprintf("%.1f", config.capacity/1000)}TB drives`+
|
|
||||||
`, capable to backfill at ${sprintf("%.1f", config.speed*1000)} MB/s, drive AFR ${sprintf("%.1f", config.afr_drive*100)}%`+
|
|
||||||
(config.afr_host ? `, host AFR ${sprintf("%.1f", config.afr_host*100)}%` : '')+
|
|
||||||
(config.ec ? `, EC ${config.ec[0]}+${config.ec[1]}` : `, ${config.replicas} replicas`)+
|
|
||||||
`, ${config.pgs||1} PG per OSD`+
|
|
||||||
(config.degraded_replacement ? `\n...and you don't let the rebalance finish before replacing drives` : '')
|
|
||||||
);
|
|
||||||
console.log('-> '+sprintf("%.7f%%", 100*cluster_afr(config))+'\n');
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/******** UTILITY ********/
|
/******** UTILITY ********/
|
||||||
|
|||||||
@@ -0,0 +1,28 @@
|
|||||||
|
const { sprintf } = require('sprintf-js');
|
||||||
|
const { cluster_afr } = require('./afr.js');
|
||||||
|
|
||||||
|
print_cluster_afr({ n_hosts: 4, n_drives: 6, afr_drive: 0.03, afr_host: 0.05, capacity: 4000, speed: 0.1, replicas: 2 });
|
||||||
|
print_cluster_afr({ n_hosts: 4, n_drives: 3, afr_drive: 0.03, afr_host: 0, capacity: 4000, speed: 0.1, replicas: 2 });
|
||||||
|
print_cluster_afr({ n_hosts: 4, n_drives: 3, afr_drive: 0.03, afr_host: 0.05, capacity: 4000, speed: 0.1, replicas: 2 });
|
||||||
|
print_cluster_afr({ n_hosts: 4, n_drives: 3, afr_drive: 0.03, afr_host: 0, capacity: 4000, speed: 0.1, ec: true, ec_data: 2, ec_parity: 1 });
|
||||||
|
print_cluster_afr({ n_hosts: 4, n_drives: 3, afr_drive: 0.03, afr_host: 0.05, capacity: 4000, speed: 0.1, ec: true, ec_data: 2, ec_parity: 1 });
|
||||||
|
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, afr_host: 0, capacity: 8000, speed: 0.02, replicas: 2 });
|
||||||
|
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, afr_host: 0.05, capacity: 8000, speed: 0.02, replicas: 2 });
|
||||||
|
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, afr_host: 0, capacity: 8000, speed: 0.02, replicas: 3 });
|
||||||
|
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, afr_host: 0.05, capacity: 8000, speed: 0.02, replicas: 3 });
|
||||||
|
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, afr_host: 0, capacity: 8000, speed: 0.02, replicas: 3, pgs: 100 });
|
||||||
|
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, afr_host: 0.05, capacity: 8000, speed: 0.02, replicas: 3, pgs: 100 });
|
||||||
|
print_cluster_afr({ n_hosts: 10, n_drives: 10, afr_drive: 0.1, afr_host: 0.05, capacity: 8000, speed: 0.02, replicas: 3, pgs: 100, degraded_replacement: 1 });
|
||||||
|
|
||||||
|
function print_cluster_afr(config)
|
||||||
|
{
|
||||||
|
console.log(
|
||||||
|
`${config.n_hosts} nodes with ${config.n_drives} ${sprintf("%.1f", config.capacity/1000)}TB drives`+
|
||||||
|
`, capable to backfill at ${sprintf("%.1f", config.speed*1000)} MB/s, drive AFR ${sprintf("%.1f", config.afr_drive*100)}%`+
|
||||||
|
(config.afr_host ? `, host AFR ${sprintf("%.1f", config.afr_host*100)}%` : '')+
|
||||||
|
(config.ec ? `, EC ${config.ec_data}+${config.ec_parity}` : `, ${config.replicas} replicas`)+
|
||||||
|
`, ${config.pgs||1} PG per OSD`+
|
||||||
|
(config.degraded_replacement ? `\n...and you don't let the rebalance finish before replacing drives` : '')
|
||||||
|
);
|
||||||
|
console.log('-> '+sprintf("%.7f%%", 100*cluster_afr(config))+'\n');
|
||||||
|
}
|
||||||
+107
-33
@@ -1,5 +1,5 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.0 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
// Data distribution optimizer using linear programming (lp_solve)
|
// Data distribution optimizer using linear programming (lp_solve)
|
||||||
|
|
||||||
@@ -50,7 +50,7 @@ async function lp_solve(text)
|
|||||||
return { score, vars };
|
return { score, vars };
|
||||||
}
|
}
|
||||||
|
|
||||||
async function optimize_initial({ osd_tree, pg_count, pg_size = 3, pg_minsize = 2, max_combinations = 10000, parity_space = 1 })
|
async function optimize_initial({ osd_tree, pg_count, pg_size = 3, pg_minsize = 2, max_combinations = 10000, parity_space = 1, ordered = false })
|
||||||
{
|
{
|
||||||
if (!pg_count || !osd_tree)
|
if (!pg_count || !osd_tree)
|
||||||
{
|
{
|
||||||
@@ -58,7 +58,7 @@ async function optimize_initial({ osd_tree, pg_count, pg_size = 3, pg_minsize =
|
|||||||
}
|
}
|
||||||
const all_weights = Object.assign({}, ...Object.values(osd_tree));
|
const all_weights = Object.assign({}, ...Object.values(osd_tree));
|
||||||
const total_weight = Object.values(all_weights).reduce((a, c) => Number(a) + Number(c), 0);
|
const total_weight = Object.values(all_weights).reduce((a, c) => Number(a) + Number(c), 0);
|
||||||
const all_pgs = Object.values(random_combinations(osd_tree, pg_size, max_combinations));
|
const all_pgs = Object.values(random_combinations(osd_tree, pg_size, max_combinations, parity_space > 1));
|
||||||
const pg_per_osd = {};
|
const pg_per_osd = {};
|
||||||
for (const pg of all_pgs)
|
for (const pg of all_pgs)
|
||||||
{
|
{
|
||||||
@@ -92,7 +92,7 @@ async function optimize_initial({ osd_tree, pg_count, pg_size = 3, pg_minsize =
|
|||||||
console.log(lp);
|
console.log(lp);
|
||||||
throw new Error('Problem is infeasible or unbounded - is it a bug?');
|
throw new Error('Problem is infeasible or unbounded - is it a bug?');
|
||||||
}
|
}
|
||||||
const int_pgs = make_int_pgs(lp_result.vars, pg_count);
|
const int_pgs = make_int_pgs(lp_result.vars, pg_count, ordered);
|
||||||
const eff = pg_list_space_efficiency(int_pgs, all_weights, pg_minsize, parity_space);
|
const eff = pg_list_space_efficiency(int_pgs, all_weights, pg_minsize, parity_space);
|
||||||
const res = {
|
const res = {
|
||||||
score: lp_result.score,
|
score: lp_result.score,
|
||||||
@@ -104,7 +104,18 @@ async function optimize_initial({ osd_tree, pg_count, pg_size = 3, pg_minsize =
|
|||||||
return res;
|
return res;
|
||||||
}
|
}
|
||||||
|
|
||||||
function make_int_pgs(weights, pg_count)
|
function shuffle(array)
|
||||||
|
{
|
||||||
|
for (let i = array.length - 1, j, x; i > 0; i--)
|
||||||
|
{
|
||||||
|
j = Math.floor(Math.random() * (i + 1));
|
||||||
|
x = array[i];
|
||||||
|
array[i] = array[j];
|
||||||
|
array[j] = x;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function make_int_pgs(weights, pg_count, round_robin)
|
||||||
{
|
{
|
||||||
const total_weight = Object.values(weights).reduce((a, c) => Number(a) + Number(c), 0);
|
const total_weight = Object.values(weights).reduce((a, c) => Number(a) + Number(c), 0);
|
||||||
let int_pgs = [];
|
let int_pgs = [];
|
||||||
@@ -112,31 +123,37 @@ function make_int_pgs(weights, pg_count)
|
|||||||
let weight_left = total_weight;
|
let weight_left = total_weight;
|
||||||
for (const pg_name in weights)
|
for (const pg_name in weights)
|
||||||
{
|
{
|
||||||
|
let cur_pg = pg_name.substr(3).split('_');
|
||||||
let n = Math.round(weights[pg_name] / weight_left * pg_left);
|
let n = Math.round(weights[pg_name] / weight_left * pg_left);
|
||||||
for (let i = 0; i < n; i++)
|
for (let i = 0; i < n; i++)
|
||||||
{
|
{
|
||||||
int_pgs.push(pg_name.substr(3).split('_'));
|
int_pgs.push([ ...cur_pg ]);
|
||||||
|
if (round_robin)
|
||||||
|
{
|
||||||
|
cur_pg.push(cur_pg.shift());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
weight_left -= weights[pg_name];
|
weight_left -= weights[pg_name];
|
||||||
pg_left -= n;
|
pg_left -= n;
|
||||||
}
|
}
|
||||||
|
shuffle(int_pgs);
|
||||||
return int_pgs;
|
return int_pgs;
|
||||||
}
|
}
|
||||||
|
|
||||||
function calc_intersect_weights(pg_size, pg_count, prev_weights, all_pgs)
|
function calc_intersect_weights(old_pg_size, pg_size, pg_count, prev_weights, all_pgs, ordered)
|
||||||
{
|
{
|
||||||
const move_weights = {};
|
const move_weights = {};
|
||||||
if ((1 << pg_size) < pg_count)
|
if ((1 << old_pg_size) < pg_count)
|
||||||
{
|
{
|
||||||
const intersect = {};
|
const intersect = {};
|
||||||
for (const pg_name in prev_weights)
|
for (const pg_name in prev_weights)
|
||||||
{
|
{
|
||||||
const pg = pg_name.substr(3).split(/_/);
|
const pg = pg_name.substr(3).split(/_/);
|
||||||
for (let omit = 1; omit < (1 << pg_size); omit++)
|
for (let omit = 1; omit < (1 << old_pg_size); omit++)
|
||||||
{
|
{
|
||||||
let pg_omit = [ ...pg ];
|
let pg_omit = [ ...pg ];
|
||||||
let intersect_count = pg_size;
|
let intersect_count = old_pg_size;
|
||||||
for (let i = 0; i < pg_size; i++)
|
for (let i = 0; i < old_pg_size; i++)
|
||||||
{
|
{
|
||||||
if (omit & (1 << i))
|
if (omit & (1 << i))
|
||||||
{
|
{
|
||||||
@@ -144,6 +161,8 @@ function calc_intersect_weights(pg_size, pg_count, prev_weights, all_pgs)
|
|||||||
intersect_count--;
|
intersect_count--;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if (!ordered)
|
||||||
|
pg_omit = pg_omit.filter(n => n).sort();
|
||||||
pg_omit = pg_omit.join(':');
|
pg_omit = pg_omit.join(':');
|
||||||
intersect[pg_omit] = Math.max(intersect[pg_omit] || 0, intersect_count);
|
intersect[pg_omit] = Math.max(intersect[pg_omit] || 0, intersect_count);
|
||||||
}
|
}
|
||||||
@@ -157,10 +176,10 @@ function calc_intersect_weights(pg_size, pg_count, prev_weights, all_pgs)
|
|||||||
for (let i = 0; i < pg_size; i++)
|
for (let i = 0; i < pg_size; i++)
|
||||||
{
|
{
|
||||||
if (omit & (1 << i))
|
if (omit & (1 << i))
|
||||||
{
|
|
||||||
pg_omit[i] = '';
|
pg_omit[i] = '';
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
if (!ordered)
|
||||||
|
pg_omit = pg_omit.filter(n => n).sort();
|
||||||
pg_omit = pg_omit.join(':');
|
pg_omit = pg_omit.join(':');
|
||||||
max_int = Math.max(max_int, intersect[pg_omit] || 0);
|
max_int = Math.max(max_int, intersect[pg_omit] || 0);
|
||||||
}
|
}
|
||||||
@@ -169,15 +188,18 @@ function calc_intersect_weights(pg_size, pg_count, prev_weights, all_pgs)
|
|||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
const prev_pg_hashed = Object.keys(prev_weights).map(pg_name => pg_name.substr(3).split(/_/).reduce((a, c) => { a[c] = 1; return a; }, {}));
|
const prev_pg_hashed = Object.keys(prev_weights).map(pg_name => pg_name
|
||||||
|
.substr(3).split(/_/).reduce((a, c, i) => { a[c] = i+1; return a; }, {}));
|
||||||
for (const pg of all_pgs)
|
for (const pg of all_pgs)
|
||||||
{
|
{
|
||||||
if (!prev_weights['pg_'+pg.join('_')])
|
if (!prev_weights['pg_'+pg.join('_')])
|
||||||
{
|
{
|
||||||
let max_int = 0;
|
let max_int = 0;
|
||||||
for (const prev_hash in prev_pg_hashed)
|
for (const prev_hash of prev_pg_hashed)
|
||||||
{
|
{
|
||||||
const intersect_count = pg.reduce((a, osd) => a + (prev_hash[osd] ? 1 : 0), 0);
|
const intersect_count = ordered
|
||||||
|
? pg.reduce((a, osd, i) => a + (prev_hash[osd] == 1+i ? 1 : 0), 0)
|
||||||
|
: pg.reduce((a, osd, i) => a + (prev_hash[osd] ? 1 : 0), 0);
|
||||||
if (max_int < intersect_count)
|
if (max_int < intersect_count)
|
||||||
{
|
{
|
||||||
max_int = intersect_count;
|
max_int = intersect_count;
|
||||||
@@ -226,12 +248,13 @@ function add_valid_previous(osd_tree, prev_weights, all_pgs)
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Try to minimize data movement
|
// Try to minimize data movement
|
||||||
async function optimize_change({ prev_pgs: prev_int_pgs, osd_tree, pg_size = 3, pg_minsize = 2, max_combinations = 10000, parity_space = 1 })
|
async function optimize_change({ prev_pgs: prev_int_pgs, osd_tree, pg_size = 3, pg_minsize = 2, max_combinations = 10000, parity_space = 1, ordered = false })
|
||||||
{
|
{
|
||||||
if (!osd_tree)
|
if (!osd_tree)
|
||||||
{
|
{
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
// FIXME: use parity_chunks with parity_space instead of pg_minsize
|
||||||
const pg_effsize = Math.min(pg_minsize, Object.keys(osd_tree).length)
|
const pg_effsize = Math.min(pg_minsize, Object.keys(osd_tree).length)
|
||||||
+ Math.max(0, Math.min(pg_size, Object.keys(osd_tree).length) - pg_minsize) * parity_space;
|
+ Math.max(0, Math.min(pg_size, Object.keys(osd_tree).length) - pg_minsize) * parity_space;
|
||||||
const pg_count = prev_int_pgs.length;
|
const pg_count = prev_int_pgs.length;
|
||||||
@@ -248,9 +271,13 @@ async function optimize_change({ prev_pgs: prev_int_pgs, osd_tree, pg_size = 3,
|
|||||||
prev_pg_per_osd[osd].push([ pg_name, (i >= pg_minsize ? parity_space : 1) ]);
|
prev_pg_per_osd[osd].push([ pg_name, (i >= pg_minsize ? parity_space : 1) ]);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
const old_pg_size = prev_int_pgs[0].length;
|
||||||
// Get all combinations
|
// Get all combinations
|
||||||
let all_pgs = random_combinations(osd_tree, pg_size, max_combinations);
|
let all_pgs = random_combinations(osd_tree, pg_size, max_combinations, parity_space > 1);
|
||||||
add_valid_previous(osd_tree, prev_weights, all_pgs);
|
if (old_pg_size == pg_size)
|
||||||
|
{
|
||||||
|
add_valid_previous(osd_tree, prev_weights, all_pgs);
|
||||||
|
}
|
||||||
all_pgs = Object.values(all_pgs);
|
all_pgs = Object.values(all_pgs);
|
||||||
const pg_per_osd = {};
|
const pg_per_osd = {};
|
||||||
for (const pg of all_pgs)
|
for (const pg of all_pgs)
|
||||||
@@ -264,7 +291,7 @@ async function optimize_change({ prev_pgs: prev_int_pgs, osd_tree, pg_size = 3,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Penalize PGs based on their similarity to old PGs
|
// Penalize PGs based on their similarity to old PGs
|
||||||
const move_weights = calc_intersect_weights(pg_size, pg_count, prev_weights, all_pgs);
|
const move_weights = calc_intersect_weights(old_pg_size, pg_size, pg_count, prev_weights, all_pgs, ordered);
|
||||||
// Calculate total weight - old PG weights
|
// Calculate total weight - old PG weights
|
||||||
const all_pg_names = all_pgs.map(pg => 'pg_'+pg.join('_'));
|
const all_pg_names = all_pgs.map(pg => 'pg_'+pg.join('_'));
|
||||||
const all_pgs_hash = all_pg_names.reduce((a, c) => { a[c] = true; return a; }, {});
|
const all_pgs_hash = all_pg_names.reduce((a, c) => { a[c] = true; return a; }, {});
|
||||||
@@ -355,11 +382,35 @@ async function optimize_change({ prev_pgs: prev_int_pgs, osd_tree, pg_size = 3,
|
|||||||
{
|
{
|
||||||
differs++;
|
differs++;
|
||||||
}
|
}
|
||||||
for (let j = 0; j < pg_size; j++)
|
}
|
||||||
|
if (ordered)
|
||||||
|
{
|
||||||
|
for (let i = 0; i < pg_count; i++)
|
||||||
{
|
{
|
||||||
if (new_pgs[i][j] != prev_int_pgs[i][j])
|
for (let j = 0; j < pg_size; j++)
|
||||||
{
|
{
|
||||||
osd_differs++;
|
if (new_pgs[i][j] != prev_int_pgs[i][j])
|
||||||
|
{
|
||||||
|
osd_differs++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
for (let i = 0; i < pg_count; i++)
|
||||||
|
{
|
||||||
|
const old_map = prev_int_pgs[i].reduce((a, c) => { a[c] = (a[c]|0) + 1; return a; }, {});
|
||||||
|
for (let j = 0; j < pg_size; j++)
|
||||||
|
{
|
||||||
|
if ((0|old_map[new_pgs[i][j]]) > 0)
|
||||||
|
{
|
||||||
|
old_map[new_pgs[i][j]]--;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
osd_differs++;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -488,7 +539,8 @@ function extract_osds(osd_tree, levels, osd_level, osds = {})
|
|||||||
return osds;
|
return osds;
|
||||||
}
|
}
|
||||||
|
|
||||||
function random_combinations(osd_tree, pg_size, count)
|
// ordered = don't treat (x,y) and (y,x) as equal
|
||||||
|
function random_combinations(osd_tree, pg_size, count, ordered)
|
||||||
{
|
{
|
||||||
let seed = 0x5f020e43;
|
let seed = 0x5f020e43;
|
||||||
let rng = () =>
|
let rng = () =>
|
||||||
@@ -516,25 +568,47 @@ function random_combinations(osd_tree, pg_size, count)
|
|||||||
pg.push(osds[cur_hosts[next_host]][next_osd]);
|
pg.push(osds[cur_hosts[next_host]][next_osd]);
|
||||||
cur_hosts.splice(next_host, 1);
|
cur_hosts.splice(next_host, 1);
|
||||||
}
|
}
|
||||||
while (pg.length < pg_size)
|
const cyclic_pgs = [ pg ];
|
||||||
|
if (ordered)
|
||||||
{
|
{
|
||||||
pg.push(NO_OSD);
|
for (let i = 1; i < pg.size; i++)
|
||||||
|
{
|
||||||
|
cyclic_pgs.push([ ...pg.slice(i), ...pg.slice(0, i) ]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (const pg of cyclic_pgs)
|
||||||
|
{
|
||||||
|
while (pg.length < pg_size)
|
||||||
|
{
|
||||||
|
pg.push(NO_OSD);
|
||||||
|
}
|
||||||
|
r['pg_'+pg.join('_')] = pg;
|
||||||
}
|
}
|
||||||
r['pg_'+pg.join('_')] = pg;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Generate purely random combinations
|
// Generate purely random combinations
|
||||||
restart: while (count > 0)
|
while (count > 0)
|
||||||
{
|
{
|
||||||
let host_idx = [];
|
let host_idx = [];
|
||||||
for (let i = 0; i < pg_size && i < hosts.length; i++)
|
const cur_hosts = [ ...hosts.map((h, i) => i) ];
|
||||||
|
const max_hosts = pg_size < hosts.length ? pg_size : hosts.length;
|
||||||
|
if (ordered)
|
||||||
{
|
{
|
||||||
let start = i > 0 ? host_idx[i-1]+1 : 0;
|
for (let i = 0; i < max_hosts; i++)
|
||||||
if (start >= hosts.length)
|
|
||||||
{
|
{
|
||||||
continue restart;
|
const r = rng() % cur_hosts.length;
|
||||||
|
host_idx[i] = cur_hosts[r];
|
||||||
|
cur_hosts.splice(r, 1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
for (let i = 0; i < max_hosts; i++)
|
||||||
|
{
|
||||||
|
const r = rng() % (cur_hosts.length - (max_hosts - i - 1));
|
||||||
|
host_idx[i] = cur_hosts[r];
|
||||||
|
cur_hosts.splice(0, r+1);
|
||||||
}
|
}
|
||||||
host_idx[i] = start + rng() % (hosts.length-start);
|
|
||||||
}
|
}
|
||||||
let pg = host_idx.map(h => osds[hosts[h]][rng() % osds[hosts[h]].length]);
|
let pg = host_idx.map(h => osds[hosts[h]][rng() % osds[hosts[h]].length]);
|
||||||
while (pg.length < pg_size)
|
while (pg.length < pg_size)
|
||||||
|
|||||||
Executable
+414
@@ -0,0 +1,414 @@
|
|||||||
|
#!/usr/bin/nodejs
|
||||||
|
// systemd unit generator for hybrid (HDD+SSD) vitastor OSDs
|
||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1
|
||||||
|
|
||||||
|
// USAGE: nodejs make-osd-hybrid.js [--disable_ssd_cache 0] [--disable_hdd_cache 0] /dev/sda /dev/sdb /dev/sdc /dev/sdd ...
|
||||||
|
// I.e. - just pass all HDDs and SSDs mixed, the script will decide where
|
||||||
|
// to put journals on its own
|
||||||
|
|
||||||
|
const fs = require('fs');
|
||||||
|
const fsp = fs.promises;
|
||||||
|
const child_process = require('child_process');
|
||||||
|
|
||||||
|
const options = {
|
||||||
|
debug: 1,
|
||||||
|
journal_size: 1024*1024*1024,
|
||||||
|
min_meta_size: 1024*1024*1024,
|
||||||
|
object_size: 1024*1024,
|
||||||
|
bitmap_granularity: 4096,
|
||||||
|
device_block_size: 4096,
|
||||||
|
disable_ssd_cache: 1,
|
||||||
|
disable_hdd_cache: 1,
|
||||||
|
};
|
||||||
|
|
||||||
|
run().catch(console.fatal);
|
||||||
|
|
||||||
|
async function run()
|
||||||
|
{
|
||||||
|
const device_list = parse_options();
|
||||||
|
await system_or_die("mkdir -p /var/log/vitastor; chown vitastor /var/log/vitastor");
|
||||||
|
// Collect devices
|
||||||
|
const all_devices = await collect_devices(device_list);
|
||||||
|
const ssds = all_devices.filter(d => d.ssd);
|
||||||
|
const hdds = all_devices.filter(d => !d.ssd);
|
||||||
|
// Collect existing OSD units
|
||||||
|
const osd_units = await collect_osd_units();
|
||||||
|
// Count assigned HDD journals and unallocated space for each SSD
|
||||||
|
await check_journal_count(ssds, osd_units);
|
||||||
|
// Create new OSDs
|
||||||
|
await create_new_hybrid_osds(hdds, ssds, osd_units);
|
||||||
|
process.exit(0);
|
||||||
|
}
|
||||||
|
|
||||||
|
function parse_options()
|
||||||
|
{
|
||||||
|
const devices = [];
|
||||||
|
const opt = {};
|
||||||
|
for (let i = 2; i < process.argv.length; i++)
|
||||||
|
{
|
||||||
|
const arg = process.argv[i];
|
||||||
|
if (arg == '--help' || arg == '-h')
|
||||||
|
{
|
||||||
|
opt.help = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
else if (arg.substr(0, 2) == '--')
|
||||||
|
opt[arg.substr(2)] = process.argv[++i];
|
||||||
|
else
|
||||||
|
devices.push(arg);
|
||||||
|
}
|
||||||
|
if (opt.help || !devices.length)
|
||||||
|
{
|
||||||
|
console.log(
|
||||||
|
'Prepare hybrid (HDD+SSD) Vitastor OSDs\n'+
|
||||||
|
'(c) Vitaliy Filippov, 2019+, license: VNPL-1.1\n\n'+
|
||||||
|
'USAGE: nodejs make-osd-hybrid.js [OPTIONS] /dev/sda /dev/sdb /dev/sdc ...\n'+
|
||||||
|
'Just pass all your SSDs and HDDs in any order, the script will distribute OSDs for you.\n\n'+
|
||||||
|
'OPTIONS (with defaults):\n'+
|
||||||
|
Object.keys(options).map(k => ` --${k} ${options[k]}`).join('\n')
|
||||||
|
);
|
||||||
|
process.exit(0);
|
||||||
|
}
|
||||||
|
for (const k in opt)
|
||||||
|
options[k] = opt[k];
|
||||||
|
return devices;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Collect devices
|
||||||
|
async function collect_devices(devices_to_check)
|
||||||
|
{
|
||||||
|
const devices = [];
|
||||||
|
for (const dev of devices_to_check)
|
||||||
|
{
|
||||||
|
if (dev.substr(0, 5) != '/dev/')
|
||||||
|
{
|
||||||
|
console.log(`${dev} does not start with /dev/, skipping`);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (!await file_exists('/sys/block/'+dev.substr(5)))
|
||||||
|
{
|
||||||
|
console.log(`${dev} is a partition, skipping`);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
// Check if the device is an SSD
|
||||||
|
const rot = '/sys/block/'+dev.substr(5)+'/queue/rotational';
|
||||||
|
if (!await file_exists(rot))
|
||||||
|
{
|
||||||
|
console.log(`${dev} does not have ${rot} to check whether it's an SSD, skipping`);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
const ssd = !parseInt(await fsp.readFile(rot, { encoding: 'utf-8' }));
|
||||||
|
// Check if the device has partition table
|
||||||
|
let [ has_partition_table, parts ] = await system(`sfdisk --dump ${dev} --json`);
|
||||||
|
if (has_partition_table != 0)
|
||||||
|
{
|
||||||
|
// Check if the device has any data
|
||||||
|
const [ has_data, out ] = await system(`blkid ${dev}`);
|
||||||
|
if (has_data == 0)
|
||||||
|
{
|
||||||
|
console.log(`${dev} contains data, skipping:\n ${out.trim().replace(/\n/g, '\n ')}`);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
parts = parts ? JSON.parse(parts).partitiontable : null;
|
||||||
|
if (parts && parts.label != 'gpt')
|
||||||
|
{
|
||||||
|
console.log(`${dev} contains "${parts.label}" partition table, only GPT is supported, skipping`);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
devices.push({
|
||||||
|
path: dev,
|
||||||
|
ssd,
|
||||||
|
parts,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
return devices;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Collect existing OSD units
|
||||||
|
async function collect_osd_units()
|
||||||
|
{
|
||||||
|
const units = [];
|
||||||
|
for (const unit of (await system("ls /etc/systemd/system/vitastor-osd*.service"))[1].trim().split('\n'))
|
||||||
|
{
|
||||||
|
if (!unit)
|
||||||
|
{
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let cmd = /^ExecStart\s*=\s*(([^\n]*\\\n)*[^\n]*)/.exec(await fsp.readFile(unit, { encoding: 'utf-8' }));
|
||||||
|
if (!cmd)
|
||||||
|
{
|
||||||
|
console.log('ExecStart= not found in '+unit+', skipping')
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let kv = {}, key;
|
||||||
|
cmd = cmd[1].replace(/^bash\s+-c\s+'/, '')
|
||||||
|
.replace(/>>\s*\S+2>\s*&1\s*'$/, '')
|
||||||
|
.replace(/\s*\\\n\s*/g, ' ')
|
||||||
|
.replace(/([^\s']+)|'([^']+)'/g, (m, m1, m2) =>
|
||||||
|
{
|
||||||
|
m1 = m1||m2;
|
||||||
|
if (key == null)
|
||||||
|
{
|
||||||
|
if (m1.substr(0, 2) != '--')
|
||||||
|
{
|
||||||
|
console.log('Strange command line in '+unit+', stopping');
|
||||||
|
process.exit(1);
|
||||||
|
}
|
||||||
|
key = m1.substr(2);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
kv[key] = m1;
|
||||||
|
key = null;
|
||||||
|
}
|
||||||
|
});
|
||||||
|
units.push(kv);
|
||||||
|
}
|
||||||
|
return units;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Count assigned HDD journals and unallocated space for each SSD
|
||||||
|
async function check_journal_count(ssds, osd_units)
|
||||||
|
{
|
||||||
|
const units_by_journal = osd_units.reduce((a, c) =>
|
||||||
|
{
|
||||||
|
if (c.journal_device)
|
||||||
|
a[c.journal_device] = c;
|
||||||
|
return a;
|
||||||
|
}, {});
|
||||||
|
for (const dev of ssds)
|
||||||
|
{
|
||||||
|
dev.journals = 0;
|
||||||
|
if (dev.parts)
|
||||||
|
{
|
||||||
|
for (const part of dev.parts.partitions)
|
||||||
|
{
|
||||||
|
if (part.uuid && units_by_journal['/dev/disk/by-partuuid/'+part.uuid.toLowerCase()])
|
||||||
|
{
|
||||||
|
dev.journals++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
dev.free = free_from_parttable(dev.parts);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
dev.free = parseInt(await system_or_die("blockdev --getsize64 "+dev.path));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async function create_new_hybrid_osds(hdds, ssds, osd_units)
|
||||||
|
{
|
||||||
|
const units_by_disk = osd_units.reduce((a, c) => { a[c.data_device] = c; return a; }, {});
|
||||||
|
for (const dev of hdds)
|
||||||
|
{
|
||||||
|
if (!dev.parts)
|
||||||
|
{
|
||||||
|
// HDD is not partitioned yet, create a single partition
|
||||||
|
// + is the "default value" for sfdisk
|
||||||
|
await system_or_die('sfdisk '+dev.path, 'label: gpt\n\n+ +\n');
|
||||||
|
dev.parts = JSON.parse(await system_or_die('sfdisk --dump '+dev.path+' --json')).partitiontable;
|
||||||
|
}
|
||||||
|
if (dev.parts.partitions.length != 1)
|
||||||
|
{
|
||||||
|
console.log(dev.path+' has more than 1 partition, skipping');
|
||||||
|
}
|
||||||
|
else if ((dev.parts.partitions[0].start + dev.parts.partitions[0].size) != (1 + dev.parts.lastlba))
|
||||||
|
{
|
||||||
|
console.log(dev.path+'1 is not a whole-disk partition, skipping');
|
||||||
|
}
|
||||||
|
else if (!dev.parts.partitions[0].uuid)
|
||||||
|
{
|
||||||
|
console.log(dev.parts.partitions[0].node+' does not have UUID. Please repartition '+dev.path+' with GPT');
|
||||||
|
}
|
||||||
|
else if (!units_by_disk['/dev/disk/by-partuuid/'+dev.parts.partitions[0].uuid.toLowerCase()])
|
||||||
|
{
|
||||||
|
await create_hybrid_osd(dev, ssds);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async function create_hybrid_osd(dev, ssds)
|
||||||
|
{
|
||||||
|
// Create a new OSD
|
||||||
|
// Calculate metadata size
|
||||||
|
const data_device = '/dev/disk/by-partuuid/'+dev.parts.partitions[0].uuid.toLowerCase();
|
||||||
|
const data_size = dev.parts.partitions[0].size * dev.parts.sectorsize;
|
||||||
|
const meta_entry_size = 24 + 2*options.object_size/options.bitmap_granularity/8;
|
||||||
|
const entries_per_block = Math.floor(options.device_block_size / meta_entry_size);
|
||||||
|
const object_count = Math.floor(data_size / options.object_size);
|
||||||
|
let meta_size = Math.ceil(1 + object_count / entries_per_block) * options.device_block_size;
|
||||||
|
// Leave some extra space for future metadata formats and round metadata area size to multiples of 1 MB
|
||||||
|
meta_size = 2*meta_size;
|
||||||
|
meta_size = Math.ceil(meta_size/1024/1024) * 1024*1024;
|
||||||
|
if (meta_size < options.min_meta_size)
|
||||||
|
meta_size = options.min_meta_size;
|
||||||
|
let journal_size = Math.ceil(options.journal_size/1024/1024) * 1024*1024;
|
||||||
|
// Pick an SSD for journal, balancing the number of journals across SSDs
|
||||||
|
let selected_ssd;
|
||||||
|
for (const ssd of ssds)
|
||||||
|
if (ssd.free >= (meta_size+journal_size) && (!selected_ssd || selected_ssd.journals > ssd.journals))
|
||||||
|
selected_ssd = ssd;
|
||||||
|
if (!selected_ssd)
|
||||||
|
{
|
||||||
|
console.error('Could not find free space for SSD journal and metadata for '+dev.path);
|
||||||
|
process.exit(1);
|
||||||
|
}
|
||||||
|
// Allocate an OSD number
|
||||||
|
const osd_num = (await system_or_die("vitastor-cli alloc-osd")).trim();
|
||||||
|
if (!osd_num)
|
||||||
|
{
|
||||||
|
console.error('Failed to run vitastor-cli alloc-osd');
|
||||||
|
process.exit(1);
|
||||||
|
}
|
||||||
|
console.log('Creating OSD '+osd_num+' on '+dev.path+' (HDD) with journal and metadata on '+selected_ssd.path+' (SSD)');
|
||||||
|
// Add two partitions: journal and metadata
|
||||||
|
const new_parts = await add_partitions(selected_ssd, [ journal_size, meta_size ]);
|
||||||
|
selected_ssd.journals++;
|
||||||
|
const journal_device = '/dev/disk/by-partuuid/'+new_parts[0].uuid.toLowerCase();
|
||||||
|
const meta_device = '/dev/disk/by-partuuid/'+new_parts[1].uuid.toLowerCase();
|
||||||
|
// Wait until the device symlinks appear
|
||||||
|
while (!await file_exists(journal_device))
|
||||||
|
{
|
||||||
|
await new Promise(ok => setTimeout(ok, 100));
|
||||||
|
}
|
||||||
|
while (!await file_exists(meta_device))
|
||||||
|
{
|
||||||
|
await new Promise(ok => setTimeout(ok, 100));
|
||||||
|
}
|
||||||
|
// Zero out metadata and journal
|
||||||
|
await system_or_die("dd if=/dev/zero of="+journal_device+" bs=1M count="+(journal_size/1024/1024)+" oflag=direct");
|
||||||
|
await system_or_die("dd if=/dev/zero of="+meta_device+" bs=1M count="+(meta_size/1024/1024)+" oflag=direct");
|
||||||
|
// Create unit file for the OSD
|
||||||
|
const has_scsi_cache_type = options.disable_ssd_cache &&
|
||||||
|
(await system("ls /sys/block/"+selected_ssd.path.substr(5)+"/device/scsi_disk/*/cache_type"))[0] == 0;
|
||||||
|
const write_through = options.disable_ssd_cache && (
|
||||||
|
has_scsi_cache_type || selected_ssd.path.substr(5, 4) == 'nvme'
|
||||||
|
&& (await system_or_die("/sys/block/"+selected_ssd.path.substr(5)+"/queue/write_cache")).trim() == "write through");
|
||||||
|
await fsp.writeFile('/etc/systemd/system/vitastor-osd'+osd_num+'.service',
|
||||||
|
`[Unit]
|
||||||
|
Description=Vitastor object storage daemon osd.${osd_num}
|
||||||
|
After=network-online.target local-fs.target time-sync.target
|
||||||
|
Wants=network-online.target local-fs.target time-sync.target
|
||||||
|
PartOf=vitastor.target
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
LimitNOFILE=1048576
|
||||||
|
LimitNPROC=1048576
|
||||||
|
LimitMEMLOCK=infinity
|
||||||
|
ExecStart=bash -c '/usr/bin/vitastor-osd \\
|
||||||
|
--osd_num ${osd_num} ${write_through
|
||||||
|
? "--disable_meta_fsync 1 --disable_journal_fsync 1 --immediate_commit "+(options.disable_hdd_cache ? "all" : "small")
|
||||||
|
: ""} \\
|
||||||
|
--throttle_small_writes 1 \\
|
||||||
|
--disk_alignment ${options.device_block_size} \\
|
||||||
|
--journal_block_size ${options.device_block_size} \\
|
||||||
|
--meta_block_size ${options.device_block_size} \\
|
||||||
|
--journal_no_same_sector_overwrites true \\
|
||||||
|
--journal_sector_buffer_count 1024 \\
|
||||||
|
--block_size ${options.object_size} \\
|
||||||
|
--data_device ${data_device} \\
|
||||||
|
--journal_device ${journal_device} \\
|
||||||
|
--meta_device ${meta_device} >>/var/log/vitastor/osd${osd_num}.log 2>&1'
|
||||||
|
WorkingDirectory=/
|
||||||
|
ExecStartPre=+chown vitastor:vitastor ${data_device}
|
||||||
|
ExecStartPre=+chown vitastor:vitastor ${journal_device}
|
||||||
|
ExecStartPre=+chown vitastor:vitastor ${meta_device}${
|
||||||
|
has_scsi_cache_type
|
||||||
|
? "\nExecStartPre=+bash -c 'D=$$$(readlink "+journal_device+"); echo write through > $$$(dirname /sys/block/*/$$\${D##*/})/device/scsi_disk/*/cache_type'"
|
||||||
|
: ""}${
|
||||||
|
options.disable_hdd_cache
|
||||||
|
? "\nExecStartPre=+bash -c 'D=$$$(readlink "+data_device+"); echo write through > $$$(dirname /sys/block/*/$$\${D##*/})/device/scsi_disk/*/cache_type'"
|
||||||
|
: ""}
|
||||||
|
User=vitastor
|
||||||
|
PrivateTmp=false
|
||||||
|
TasksMax=infinity
|
||||||
|
Restart=always
|
||||||
|
StartLimitInterval=0
|
||||||
|
RestartSec=10
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=vitastor.target
|
||||||
|
`);
|
||||||
|
await system_or_die("systemctl enable vitastor-osd"+osd_num);
|
||||||
|
}
|
||||||
|
|
||||||
|
async function add_partitions(dev, sizes)
|
||||||
|
{
|
||||||
|
let script = 'label: gpt\n\n';
|
||||||
|
if (dev.parts)
|
||||||
|
{
|
||||||
|
// Old partitions
|
||||||
|
for (const part of dev.parts.partitions)
|
||||||
|
{
|
||||||
|
script += part.node+': '+Object.keys(part).map(k => k == 'node' ? '' : k+'='+part[k]).filter(k => k).join(', ')+'\n';
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// New partitions
|
||||||
|
for (const size of sizes)
|
||||||
|
{
|
||||||
|
script += '+ '+Math.ceil(size/1024)+'KiB\n';
|
||||||
|
}
|
||||||
|
await system_or_die('sfdisk '+dev.path, script);
|
||||||
|
// Get new partition table and find the new partition
|
||||||
|
const newpt = JSON.parse(await system_or_die('sfdisk --dump '+dev.path+' --json')).partitiontable;
|
||||||
|
const old_nodes = dev.parts ? dev.parts.partitions.reduce((a, c) => { a[c.uuid] = true; return a; }, {}) : {};
|
||||||
|
const new_nodes = newpt.partitions.filter(part => !old_nodes[part.uuid]);
|
||||||
|
if (new_nodes.length != sizes.length)
|
||||||
|
{
|
||||||
|
console.error('Failed to partition '+dev.path+': new partitions not found in table');
|
||||||
|
process.exit(1);
|
||||||
|
}
|
||||||
|
dev.parts = newpt;
|
||||||
|
dev.free = free_from_parttable(newpt);
|
||||||
|
return new_nodes;
|
||||||
|
}
|
||||||
|
|
||||||
|
function free_from_parttable(pt)
|
||||||
|
{
|
||||||
|
let free = pt.lastlba + 1 - pt.firstlba;
|
||||||
|
for (const part of pt.partitions)
|
||||||
|
{
|
||||||
|
free -= part.size;
|
||||||
|
}
|
||||||
|
free *= pt.sectorsize;
|
||||||
|
return free;
|
||||||
|
}
|
||||||
|
|
||||||
|
async function system_or_die(cmd, input = '')
|
||||||
|
{
|
||||||
|
let [ exitcode, stdout, stderr ] = await system(cmd, input);
|
||||||
|
if (exitcode != 0)
|
||||||
|
{
|
||||||
|
console.error(cmd+' failed: '+stderr);
|
||||||
|
process.exit(1);
|
||||||
|
}
|
||||||
|
return stdout;
|
||||||
|
}
|
||||||
|
|
||||||
|
async function system(cmd, input = '')
|
||||||
|
{
|
||||||
|
if (options.debug)
|
||||||
|
{
|
||||||
|
process.stderr.write('+ '+cmd+(input ? " <<EOF\n"+input.replace(/\s*$/, '\n')+"EOF" : '')+'\n');
|
||||||
|
}
|
||||||
|
const cp = child_process.spawn(cmd, { shell: true });
|
||||||
|
let stdout = '', stderr = '', finish_cb;
|
||||||
|
cp.stdout.on('data', buf => stdout += buf.toString());
|
||||||
|
cp.stderr.on('data', buf => stderr += buf.toString());
|
||||||
|
cp.on('exit', () => finish_cb && finish_cb());
|
||||||
|
cp.stdin.write(input);
|
||||||
|
cp.stdin.end();
|
||||||
|
if (cp.exitCode == null)
|
||||||
|
{
|
||||||
|
await new Promise(ok => finish_cb = ok);
|
||||||
|
}
|
||||||
|
return [ cp.exitCode, stdout, stderr ];
|
||||||
|
}
|
||||||
|
|
||||||
|
async function file_exists(filename)
|
||||||
|
{
|
||||||
|
return new Promise((ok, no) => fs.access(filename, fs.constants.R_OK, err => ok(!err)));
|
||||||
|
}
|
||||||
Executable
+66
@@ -0,0 +1,66 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# Very simple systemd unit generator for vitastor-osd services
|
||||||
|
# Not the final solution yet, mostly for tests
|
||||||
|
# Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
# License: MIT
|
||||||
|
|
||||||
|
# USAGE:
|
||||||
|
# 1) Put etcd_address and osd_network into /etc/vitastor/vitastor.conf. Example:
|
||||||
|
# {
|
||||||
|
# "etcd_address":["http://10.200.1.10:2379/v3","http://10.200.1.11:2379/v3","http://10.200.1.12:2379/v3"],
|
||||||
|
# "osd_network":"10.200.1.0/24"
|
||||||
|
# }
|
||||||
|
# 2) Run ./make-osd.sh /dev/disk/by-partuuid/xxx [ /dev/disk/by-partuuid/yyy]...
|
||||||
|
|
||||||
|
set -e -x
|
||||||
|
|
||||||
|
# Create OSDs on all passed devices
|
||||||
|
for DEV in $*; do
|
||||||
|
|
||||||
|
OSD_NUM=$(vitastor-cli alloc-osd)
|
||||||
|
|
||||||
|
echo Creating OSD $OSD_NUM on $DEV
|
||||||
|
|
||||||
|
OPT=$(vitastor-cli simple-offsets --format options $DEV | tr '\n' ' ')
|
||||||
|
META=$(vitastor-cli simple-offsets --format json $DEV | jq .data_offset)
|
||||||
|
dd if=/dev/zero of=$DEV bs=1048576 count=$(((META+1048575)/1048576)) oflag=direct
|
||||||
|
|
||||||
|
mkdir -p /var/log/vitastor
|
||||||
|
id vitastor &>/dev/null || useradd vitastor
|
||||||
|
chown vitastor /var/log/vitastor
|
||||||
|
|
||||||
|
cat >/etc/systemd/system/vitastor-osd$OSD_NUM.service <<EOF
|
||||||
|
[Unit]
|
||||||
|
Description=Vitastor object storage daemon osd.$OSD_NUM
|
||||||
|
After=network-online.target local-fs.target time-sync.target
|
||||||
|
Wants=network-online.target local-fs.target time-sync.target
|
||||||
|
PartOf=vitastor.target
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
LimitNOFILE=1048576
|
||||||
|
LimitNPROC=1048576
|
||||||
|
LimitMEMLOCK=infinity
|
||||||
|
ExecStart=bash -c '/usr/bin/vitastor-osd \\
|
||||||
|
--osd_num $OSD_NUM \\
|
||||||
|
--disable_data_fsync 1 \\
|
||||||
|
--immediate_commit all \\
|
||||||
|
--disk_alignment 4096 --journal_block_size 4096 --meta_block_size 4096 \\
|
||||||
|
--journal_no_same_sector_overwrites true \\
|
||||||
|
--journal_sector_buffer_count 1024 \\
|
||||||
|
$OPT >>/var/log/vitastor/osd$OSD_NUM.log 2>&1'
|
||||||
|
WorkingDirectory=/
|
||||||
|
ExecStartPre=+chown vitastor:vitastor $DEV
|
||||||
|
User=vitastor
|
||||||
|
PrivateTmp=false
|
||||||
|
TasksMax=infinity
|
||||||
|
Restart=always
|
||||||
|
StartLimitInterval=0
|
||||||
|
RestartSec=10
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=vitastor.target
|
||||||
|
EOF
|
||||||
|
|
||||||
|
systemctl enable vitastor-osd$OSD_NUM
|
||||||
|
|
||||||
|
done
|
||||||
Regular → Executable
+27
-114
@@ -1,19 +1,25 @@
|
|||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
# Example startup script generator
|
# Very simple systemd unit generator for etcd & vitastor-mon services
|
||||||
# Of course this isn't a production solution yet, this is just for tests
|
# Not the final solution yet, mostly for tests
|
||||||
# Copyright (c) Vitaliy Filippov, 2019+
|
# Copyright (c) Vitaliy Filippov, 2019+
|
||||||
# License: MIT
|
# License: MIT
|
||||||
|
|
||||||
IP=`ip -json a s | jq -r '.[].addr_info[] | select(.broadcast == "10.115.0.255") | .local'`
|
# USAGE: ./make-units.sh
|
||||||
|
|
||||||
|
IP_SUBSTR="10.200.1."
|
||||||
|
ETCD_HOSTS="etcd0=http://10.200.1.10:2380,etcd1=http://10.200.1.11:2380,etcd2=http://10.200.1.12:2380"
|
||||||
|
|
||||||
|
# determine IP
|
||||||
|
IP=`ip -json a s | jq -r '.[].addr_info[] | select(.local | startswith("'$IP_SUBSTR'")) | .local'`
|
||||||
[ "$IP" != "" ] || exit 1
|
[ "$IP" != "" ] || exit 1
|
||||||
|
ETCD_NUM=${ETCD_HOSTS/$IP*/}
|
||||||
|
[ "$ETCD_NUM" != "$ETCD_HOSTS" ] || exit 1
|
||||||
|
ETCD_NUM=$(echo $ETCD_NUM | tr -d -c , | wc -c)
|
||||||
|
|
||||||
BASE=${IP/*./}
|
# etcd
|
||||||
BASE=$((BASE-10))
|
|
||||||
|
|
||||||
useradd etcd
|
useradd etcd
|
||||||
|
|
||||||
mkdir -p /var/lib/etcd$BASE.etcd
|
mkdir -p /var/lib/etcd$ETCD_NUM.etcd
|
||||||
cat >/etc/systemd/system/etcd.service <<EOF
|
cat >/etc/systemd/system/etcd.service <<EOF
|
||||||
[Unit]
|
[Unit]
|
||||||
Description=etcd for vitastor
|
Description=etcd for vitastor
|
||||||
@@ -22,19 +28,19 @@ Wants=network-online.target local-fs.target time-sync.target
|
|||||||
|
|
||||||
[Service]
|
[Service]
|
||||||
Restart=always
|
Restart=always
|
||||||
ExecStart=/usr/local/bin/etcd -name etcd$BASE --data-dir /var/lib/etcd$BASE.etcd \\
|
ExecStart=/usr/local/bin/etcd -name etcd$ETCD_NUM --data-dir /var/lib/etcd$ETCD_NUM.etcd \\
|
||||||
--advertise-client-urls http://$IP:2379 --listen-client-urls http://$IP:2379 \\
|
--advertise-client-urls http://$IP:2379 --listen-client-urls http://$IP:2379 \\
|
||||||
--initial-advertise-peer-urls http://$IP:2380 --listen-peer-urls http://$IP:2380 \\
|
--initial-advertise-peer-urls http://$IP:2380 --listen-peer-urls http://$IP:2380 \\
|
||||||
--initial-cluster-token vitastor-etcd-1 --initial-cluster etcd0=http://10.115.0.10:2380,etcd1=http://10.115.0.11:2380,etcd2=http://10.115.0.12:2380,etcd3=http://10.115.0.13:2380 \\
|
--initial-cluster-token vitastor-etcd-1 --initial-cluster $ETCD_HOSTS \\
|
||||||
--initial-cluster-state new --max-txn-ops=100000 --auto-compaction-retention=10 --auto-compaction-mode=revision
|
--initial-cluster-state new --max-txn-ops=100000 --max-request-bytes=104857600 \\
|
||||||
WorkingDirectory=/var/lib/etcd$BASE.etcd
|
--auto-compaction-retention=10 --auto-compaction-mode=revision
|
||||||
ExecStartPre=+chown -R etcd /var/lib/etcd$BASE.etcd
|
WorkingDirectory=/var/lib/etcd$ETCD_NUM.etcd
|
||||||
|
ExecStartPre=+chown -R etcd /var/lib/etcd$ETCD_NUM.etcd
|
||||||
User=etcd
|
User=etcd
|
||||||
PrivateTmp=false
|
PrivateTmp=false
|
||||||
TasksMax=infinity
|
TasksMax=infinity
|
||||||
Restart=always
|
Restart=always
|
||||||
StartLimitInterval=0
|
StartLimitInterval=0
|
||||||
StartLimitIntervalSec=0
|
|
||||||
RestartSec=10
|
RestartSec=10
|
||||||
|
|
||||||
[Install]
|
[Install]
|
||||||
@@ -48,9 +54,7 @@ systemctl start etcd
|
|||||||
useradd vitastor
|
useradd vitastor
|
||||||
chmod 755 /root
|
chmod 755 /root
|
||||||
|
|
||||||
BASE=${IP/*./}
|
# Vitastor target
|
||||||
BASE=$(((BASE-10)*12))
|
|
||||||
|
|
||||||
cat >/etc/systemd/system/vitastor.target <<EOF
|
cat >/etc/systemd/system/vitastor.target <<EOF
|
||||||
[Unit]
|
[Unit]
|
||||||
Description=vitastor target
|
Description=vitastor target
|
||||||
@@ -58,116 +62,25 @@ Description=vitastor target
|
|||||||
WantedBy=multi-user.target
|
WantedBy=multi-user.target
|
||||||
EOF
|
EOF
|
||||||
|
|
||||||
i=1
|
# Monitor unit
|
||||||
for DEV in `ls /dev/disk/by-id/ | grep ata-INTEL_SSDSC2KB`; do
|
ETCD_MON=$(echo $ETCD_HOSTS | perl -pe 's/:2380/:2379/g; s/etcd\d*=//g;')
|
||||||
dd if=/dev/zero of=/dev/disk/by-id/$DEV bs=1048576 count=$(((427814912+1048575)/1048576+2))
|
cat >/etc/systemd/system/vitastor-mon.service <<EOF
|
||||||
dd if=/dev/zero of=/dev/disk/by-id/$DEV bs=1048576 count=$(((427814912+1048575)/1048576+2)) seek=$((1920377991168/1048576))
|
|
||||||
cat >/etc/systemd/system/vitastor-osd$((BASE+i)).service <<EOF
|
|
||||||
[Unit]
|
[Unit]
|
||||||
Description=Vitastor object storage daemon osd.$((BASE+i))
|
Description=Vitastor monitor
|
||||||
After=network-online.target local-fs.target time-sync.target
|
After=network-online.target local-fs.target time-sync.target
|
||||||
Wants=network-online.target local-fs.target time-sync.target
|
Wants=network-online.target local-fs.target time-sync.target
|
||||||
PartOf=vitastor.target
|
|
||||||
|
|
||||||
[Service]
|
[Service]
|
||||||
LimitNOFILE=1048576
|
Restart=always
|
||||||
LimitNPROC=1048576
|
ExecStart=node /usr/lib/vitastor/mon/mon-main.js --etcd_url '$ETCD_MON' --etcd_prefix '/vitastor' --etcd_start_timeout 5
|
||||||
LimitMEMLOCK=infinity
|
WorkingDirectory=/
|
||||||
ExecStart=/root/vitastor/osd \\
|
|
||||||
--etcd_address $IP:2379/v3 \\
|
|
||||||
--bind_address $IP \\
|
|
||||||
--osd_num $((BASE+i)) \\
|
|
||||||
--disable_data_fsync 1 \\
|
|
||||||
--disable_device_lock 1 \\
|
|
||||||
--immediate_commit all \\
|
|
||||||
--flusher_count 8 \\
|
|
||||||
--disk_alignment 4096 --journal_block_size 4096 --meta_block_size 4096 \\
|
|
||||||
--journal_no_same_sector_overwrites true \\
|
|
||||||
--journal_sector_buffer_count 1024 \\
|
|
||||||
--journal_offset 0 \\
|
|
||||||
--meta_offset 16777216 \\
|
|
||||||
--data_offset 427814912 \\
|
|
||||||
--data_size $((1920377991168-427814912)) \\
|
|
||||||
--data_device /dev/disk/by-id/$DEV
|
|
||||||
WorkingDirectory=/root/vitastor
|
|
||||||
ExecStartPre=+chown vitastor:vitastor /dev/disk/by-id/$DEV
|
|
||||||
User=vitastor
|
User=vitastor
|
||||||
PrivateTmp=false
|
PrivateTmp=false
|
||||||
TasksMax=infinity
|
TasksMax=infinity
|
||||||
Restart=always
|
Restart=always
|
||||||
StartLimitInterval=0
|
StartLimitInterval=0
|
||||||
StartLimitIntervalSec=0
|
|
||||||
RestartSec=10
|
RestartSec=10
|
||||||
|
|
||||||
[Install]
|
[Install]
|
||||||
WantedBy=vitastor.target
|
WantedBy=vitastor.target
|
||||||
EOF
|
EOF
|
||||||
systemctl enable vitastor-osd$((BASE+i))
|
|
||||||
i=$((i+1))
|
|
||||||
cat >/etc/systemd/system/vitastor-osd$((BASE+i)).service <<EOF
|
|
||||||
[Unit]
|
|
||||||
Description=Vitastor object storage daemon osd.$((BASE+i))
|
|
||||||
After=network-online.target local-fs.target time-sync.target
|
|
||||||
Wants=network-online.target local-fs.target time-sync.target
|
|
||||||
PartOf=vitastor.target
|
|
||||||
|
|
||||||
[Service]
|
|
||||||
LimitNOFILE=1048576
|
|
||||||
LimitNPROC=1048576
|
|
||||||
LimitMEMLOCK=infinity
|
|
||||||
ExecStart=/root/vitastor/osd \\
|
|
||||||
--etcd_address $IP:2379/v3 \\
|
|
||||||
--bind_address $IP \\
|
|
||||||
--osd_num $((BASE+i)) \\
|
|
||||||
--disable_data_fsync 1 \\
|
|
||||||
--immediate_commit all \\
|
|
||||||
--flusher_count 8 \\
|
|
||||||
--disk_alignment 4096 --journal_block_size 4096 --meta_block_size 4096 \\
|
|
||||||
--journal_no_same_sector_overwrites true \\
|
|
||||||
--journal_sector_buffer_count 1024 \\
|
|
||||||
--journal_offset 1920377991168 \\
|
|
||||||
--meta_offset $((1920377991168+16777216)) \\
|
|
||||||
--data_offset $((1920377991168+427814912)) \\
|
|
||||||
--data_size $((1920377991168-427814912)) \\
|
|
||||||
--data_device /dev/disk/by-id/$DEV
|
|
||||||
WorkingDirectory=/root/vitastor
|
|
||||||
ExecStartPre=+chown vitastor:vitastor /dev/disk/by-id/$DEV
|
|
||||||
User=vitastor
|
|
||||||
PrivateTmp=false
|
|
||||||
TasksMax=infinity
|
|
||||||
Restart=always
|
|
||||||
StartLimitInterval=0
|
|
||||||
StartLimitIntervalSec=0
|
|
||||||
RestartSec=10
|
|
||||||
|
|
||||||
[Install]
|
|
||||||
WantedBy=vitastor.target
|
|
||||||
EOF
|
|
||||||
systemctl enable vitastor-osd$((BASE+i))
|
|
||||||
i=$((i+1))
|
|
||||||
done
|
|
||||||
|
|
||||||
exit
|
|
||||||
|
|
||||||
node mon-main.js --etcd_url 'http://10.115.0.10:2379,http://10.115.0.11:2379,http://10.115.0.12:2379,http://10.115.0.13:2379' --etcd_prefix '/vitastor' --etcd_start_timeout 5
|
|
||||||
|
|
||||||
podman run -d --network host --restart always -v /var/lib/etcd0.etcd:/etcd0.etcd --name etcd quay.io/coreos/etcd:v3.4.13 etcd -name etcd0 \
|
|
||||||
-advertise-client-urls http://10.115.0.10:2379 -listen-client-urls http://10.115.0.10:2379 \
|
|
||||||
-initial-advertise-peer-urls http://10.115.0.10:2380 -listen-peer-urls http://10.115.0.10:2380 \
|
|
||||||
-initial-cluster-token vitastor-etcd-1 -initial-cluster etcd0=http://10.115.0.10:2380,etcd1=http://10.115.0.11:2380,etcd2=http://10.115.0.12:2380,etcd3=http://10.115.0.13:2380 \
|
|
||||||
-initial-cluster-state new --max-txn-ops=100000 --auto-compaction-retention=10 --auto-compaction-mode=revision
|
|
||||||
|
|
||||||
etcdctl --endpoints http://10.115.0.10:2379 put /vitastor/config/global '{"immediate_commit":"all"}'
|
|
||||||
|
|
||||||
etcdctl --endpoints http://10.115.0.10:2379 put /vitastor/config/pools '{"1":{"name":"testpool","scheme":"replicated","pg_size":2,"pg_minsize":1,"pg_count":48,"failure_domain":"host"}}'
|
|
||||||
|
|
||||||
#let pgs = {};
|
|
||||||
#for (let n = 0; n < 48; n++) { let i = n/2 | 0; pgs[1+n] = { osd_set: [ (1+i%12+(i/12 | 0)*24), (1+12+i%12+(i/12 | 0)*24) ], primary: (1+(n%2)*12+i%12+(i/12 | 0)*24) }; };
|
|
||||||
#console.log(JSON.stringify({ items: { 1: pgs } }));
|
|
||||||
#etcdctl --endpoints http://10.115.0.10:2379 put /vitastor/config/pgs ...
|
|
||||||
|
|
||||||
# --disk_alignment 4096 --journal_block_size 4096 --meta_block_size 4096 \\
|
|
||||||
# --data_offset 427814912 \\
|
|
||||||
|
|
||||||
# --disk_alignment 4096 --journal_block_size 512 --meta_block_size 512 \\
|
|
||||||
# --data_offset 433434624 \\
|
|
||||||
|
|||||||
@@ -0,0 +1,23 @@
|
|||||||
|
const fsp = require('fs').promises;
|
||||||
|
|
||||||
|
async function merge(file1, file2, out)
|
||||||
|
{
|
||||||
|
if (!out)
|
||||||
|
{
|
||||||
|
console.error('USAGE: nodejs merge.js layer1 layer2 output');
|
||||||
|
process.exit();
|
||||||
|
}
|
||||||
|
const layer1 = await fsp.readFile(file1);
|
||||||
|
const layer2 = await fsp.readFile(file2);
|
||||||
|
const zero = Buffer.alloc(4096);
|
||||||
|
for (let i = 0; i < layer2.length; i += 4096)
|
||||||
|
{
|
||||||
|
if (zero.compare(layer2, i, i+4096) != 0)
|
||||||
|
{
|
||||||
|
layer2.copy(layer1, i, i, i+4096);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
await fsp.writeFile(out, layer1);
|
||||||
|
}
|
||||||
|
|
||||||
|
merge(process.argv[2], process.argv[3], process.argv[4]);
|
||||||
+10
-9
@@ -1,7 +1,7 @@
|
|||||||
#!/usr/bin/node
|
#!/usr/bin/node
|
||||||
|
|
||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.0 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
const Mon = require('./mon.js');
|
const Mon = require('./mon.js');
|
||||||
|
|
||||||
@@ -9,17 +9,18 @@ const options = {};
|
|||||||
|
|
||||||
for (let i = 2; i < process.argv.length; i++)
|
for (let i = 2; i < process.argv.length; i++)
|
||||||
{
|
{
|
||||||
if (process.argv[i].substr(0, 2) == '--')
|
if (process.argv[i] === '-h' || process.argv[i] === '--help')
|
||||||
|
{
|
||||||
|
console.error('USAGE: '+process.argv[0]+' '+process.argv[1]+' [--verbose 1]'+
|
||||||
|
' [--etcd_address "http://127.0.0.1:2379,..."] [--config_file /etc/vitastor/vitastor.conf]'+
|
||||||
|
' [--etcd_prefix "/vitastor"] [--etcd_start_timeout 5]');
|
||||||
|
process.exit();
|
||||||
|
}
|
||||||
|
else if (process.argv[i].substr(0, 2) == '--')
|
||||||
{
|
{
|
||||||
options[process.argv[i].substr(2)] = process.argv[i+1];
|
options[process.argv[i].substr(2)] = process.argv[i+1];
|
||||||
i++;
|
i++;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if (!options.etcd_url)
|
new Mon(options).start().catch(e => { console.error(e); process.exit(1); });
|
||||||
{
|
|
||||||
console.error('USAGE: '+process.argv[0]+' '+process.argv[1]+' --etcd_url "http://127.0.0.1:2379,..." --etcd_prefix "/vitastor" --etcd_start_timeout 5 [--verbose 1]');
|
|
||||||
process.exit();
|
|
||||||
}
|
|
||||||
|
|
||||||
new Mon(options).start().catch(e => { console.error(e); process.exit(); });
|
|
||||||
|
|||||||
+685
-179
File diff suppressed because it is too large
Load Diff
+54
-13
@@ -4,6 +4,7 @@
|
|||||||
// Simple tool to calculate journal and metadata offsets for a single device
|
// Simple tool to calculate journal and metadata offsets for a single device
|
||||||
// Will be replaced by smarter tools in the future
|
// Will be replaced by smarter tools in the future
|
||||||
|
|
||||||
|
const fs = require('fs').promises;
|
||||||
const child_process = require('child_process');
|
const child_process = require('child_process');
|
||||||
|
|
||||||
async function run()
|
async function run()
|
||||||
@@ -15,6 +16,7 @@ async function run()
|
|||||||
device_block_size: 4096,
|
device_block_size: 4096,
|
||||||
journal_offset: 0,
|
journal_offset: 0,
|
||||||
device_size: 0,
|
device_size: 0,
|
||||||
|
format: 'text',
|
||||||
};
|
};
|
||||||
for (let i = 2; i < process.argv.length; i++)
|
for (let i = 2; i < process.argv.length; i++)
|
||||||
{
|
{
|
||||||
@@ -24,7 +26,22 @@ async function run()
|
|||||||
i++;
|
i++;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
const device_size = Number(options.device_size || await system("blockdev --getsize64 "+options.device));
|
if (!options.device)
|
||||||
|
{
|
||||||
|
process.stderr.write('USAGE: nodejs '+process.argv[1]+' --device /dev/sdXXX\n');
|
||||||
|
process.exit(1);
|
||||||
|
}
|
||||||
|
options.device_size = Number(options.device_size);
|
||||||
|
let device_size = options.device_size;
|
||||||
|
if (!device_size)
|
||||||
|
{
|
||||||
|
const st = await fs.stat(options.device);
|
||||||
|
options.device_block_size = st.blksize;
|
||||||
|
if (st.isBlockDevice())
|
||||||
|
device_size = Number(await system("/sbin/blockdev --getsize64 "+options.device))
|
||||||
|
else
|
||||||
|
device_size = st.size;
|
||||||
|
}
|
||||||
if (!device_size)
|
if (!device_size)
|
||||||
{
|
{
|
||||||
process.stderr.write('Failed to get device size\n');
|
process.stderr.write('Failed to get device size\n');
|
||||||
@@ -32,25 +49,49 @@ async function run()
|
|||||||
}
|
}
|
||||||
options.journal_offset = Math.ceil(options.journal_offset/options.device_block_size)*options.device_block_size;
|
options.journal_offset = Math.ceil(options.journal_offset/options.device_block_size)*options.device_block_size;
|
||||||
const meta_offset = options.journal_offset + Math.ceil(options.journal_size/options.device_block_size)*options.device_block_size;
|
const meta_offset = options.journal_offset + Math.ceil(options.journal_size/options.device_block_size)*options.device_block_size;
|
||||||
const entries_per_block = Math.floor(options.device_block_size / (24 + options.object_size/options.bitmap_granularity/8));
|
const meta_entry_size = 24 + 2*options.object_size/options.bitmap_granularity/8;
|
||||||
|
const entries_per_block = Math.floor(options.device_block_size / meta_entry_size);
|
||||||
const object_count = Math.floor((device_size-meta_offset)/options.object_size);
|
const object_count = Math.floor((device_size-meta_offset)/options.object_size);
|
||||||
const meta_size = Math.ceil(object_count / entries_per_block) * options.device_block_size;
|
const meta_size = Math.ceil(1 + object_count / entries_per_block) * options.device_block_size;
|
||||||
const data_offset = meta_offset + meta_size;
|
const data_offset = meta_offset + meta_size;
|
||||||
const meta_size_fmt = (meta_size > 1024*1024*1024 ? Math.round(meta_size/1024/1024/1024*100)/100+" GB"
|
const meta_size_fmt = (meta_size > 1024*1024*1024 ? Math.round(meta_size/1024/1024/1024*100)/100+" GB"
|
||||||
: Math.round(meta_size/1024/1024*100)/100+" MB");
|
: Math.round(meta_size/1024/1024*100)/100+" MB");
|
||||||
process.stdout.write(
|
if (options.format == 'text' || options.format == 'options')
|
||||||
`Metadata size: ${meta_size_fmt}\n`+
|
{
|
||||||
`Options for the OSD:\n`+
|
if (options.format == 'text')
|
||||||
` --journal_offset ${options.journal_offset}\n`+
|
{
|
||||||
` --meta_offset ${meta_offset}\n`+
|
process.stderr.write(
|
||||||
` --data_offset ${data_offset}\n`+
|
`Metadata size: ${meta_size_fmt}\n`+
|
||||||
(options.device_size ? ` --data_size ${device_size-data_offset}\n` : '')
|
`Options for the OSD:\n`
|
||||||
);
|
);
|
||||||
|
}
|
||||||
|
process.stdout.write(
|
||||||
|
(options.device_block_size != 4096 ?
|
||||||
|
` --meta_block_size ${options.device}\n`+
|
||||||
|
` --journal_block-size ${options.device}\n` : '')+
|
||||||
|
` --data_device ${options.device}\n`+
|
||||||
|
` --journal_offset ${options.journal_offset}\n`+
|
||||||
|
` --meta_offset ${meta_offset}\n`+
|
||||||
|
` --data_offset ${data_offset}\n`+
|
||||||
|
(options.device_size ? ` --data_size ${device_size-data_offset}\n` : '')
|
||||||
|
);
|
||||||
|
}
|
||||||
|
else if (options.format == 'env')
|
||||||
|
{
|
||||||
|
process.stdout.write(
|
||||||
|
`journal_offset=${options.journal_offset}\n`+
|
||||||
|
`meta_offset=${meta_offset}\n`+
|
||||||
|
`data_offset=${data_offset}\n`+
|
||||||
|
`data_size=${device_size-data_offset}\n`
|
||||||
|
);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
process.stdout.write('Unknown format: '+options.format);
|
||||||
}
|
}
|
||||||
|
|
||||||
function system(cmd)
|
function system(cmd)
|
||||||
{
|
{
|
||||||
return new Promise((ok, no) => child_process.exec(cmd, { maxBuffer: 64*1024*1024 }, (err, stdout, stderr) => (err ? no(err) : ok(stdout))));
|
return new Promise((ok, no) => child_process.exec(cmd, { maxBuffer: 64*1024*1024 }, (err, stdout, stderr) => (err ? no(err.message) : ok(stdout))));
|
||||||
}
|
}
|
||||||
|
|
||||||
run().catch(console.error);
|
run().catch(err => { console.error(err); process.exit(1); });
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.0 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
// Interesting real-world example coming from Ceph with EC and compression enabled.
|
// Interesting real-world example coming from Ceph with EC and compression enabled.
|
||||||
// EC parity chunks can't be compressed as efficiently as data chunks,
|
// EC parity chunks can't be compressed as efficiently as data chunks,
|
||||||
|
|||||||
@@ -0,0 +1,49 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
|
const LPOptimizer = require('./lp-optimizer.js');
|
||||||
|
|
||||||
|
async function run()
|
||||||
|
{
|
||||||
|
const osd_tree = {
|
||||||
|
100: { 1: 1 },
|
||||||
|
200: { 2: 1 },
|
||||||
|
300: { 3: 1 },
|
||||||
|
};
|
||||||
|
|
||||||
|
let res;
|
||||||
|
|
||||||
|
console.log('16 PGs, size=3');
|
||||||
|
res = await LPOptimizer.optimize_initial({ osd_tree, pg_size: 3, pg_count: 16, ordered: false });
|
||||||
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
assert(res.space == 3, 'Initial distribution');
|
||||||
|
console.log('\nChange size to 2');
|
||||||
|
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree, pg_size: 2, ordered: false });
|
||||||
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
assert(res.space >= 3*14/16 && res.osd_differs == 0, 'Redistribution');
|
||||||
|
console.log('\nRemove OSD 3');
|
||||||
|
const no3_tree = { ...osd_tree };
|
||||||
|
delete no3_tree['300'];
|
||||||
|
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree: no3_tree, pg_size: 2, ordered: false });
|
||||||
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
assert(res.space == 2, 'Redistribution after OSD removal');
|
||||||
|
|
||||||
|
console.log('\n16 PGs, size=3, ordered');
|
||||||
|
res = await LPOptimizer.optimize_initial({ osd_tree, pg_size: 3, pg_count: 16, ordered: true });
|
||||||
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
assert(res.space == 3, 'Initial distribution');
|
||||||
|
console.log('\nChange size to 2, ordered');
|
||||||
|
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree, pg_size: 2, ordered: true });
|
||||||
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
assert(res.space >= 3*14/16 && res.osd_differs < 8, 'Redistribution');
|
||||||
|
}
|
||||||
|
|
||||||
|
function assert(cond, txt)
|
||||||
|
{
|
||||||
|
if (!cond)
|
||||||
|
{
|
||||||
|
throw new Error((txt||'test')+' failed');
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
run().catch(console.error);
|
||||||
@@ -1,5 +1,5 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.0 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
const LPOptimizer = require('./lp-optimizer.js');
|
const LPOptimizer = require('./lp-optimizer.js');
|
||||||
|
|
||||||
@@ -45,30 +45,45 @@ async function run()
|
|||||||
console.log('Empty tree:');
|
console.log('Empty tree:');
|
||||||
let res = await LPOptimizer.optimize_initial({ osd_tree: cur_tree, pg_size: 3, pg_count: 256 });
|
let res = await LPOptimizer.optimize_initial({ osd_tree: cur_tree, pg_size: 3, pg_count: 256 });
|
||||||
LPOptimizer.print_change_stats(res, false);
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
assert(res.space == 0);
|
||||||
console.log('\nAdding 1st failure domain:');
|
console.log('\nAdding 1st failure domain:');
|
||||||
cur_tree['dom1'] = osd_tree['dom1'];
|
cur_tree['dom1'] = osd_tree['dom1'];
|
||||||
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree: cur_tree, pg_size: 3 });
|
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree: cur_tree, pg_size: 3 });
|
||||||
LPOptimizer.print_change_stats(res, false);
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
assert(res.space == 12 && res.total_space == 12);
|
||||||
console.log('\nAdding 2nd failure domain:');
|
console.log('\nAdding 2nd failure domain:');
|
||||||
cur_tree['dom2'] = osd_tree['dom2'];
|
cur_tree['dom2'] = osd_tree['dom2'];
|
||||||
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree: cur_tree, pg_size: 3 });
|
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree: cur_tree, pg_size: 3 });
|
||||||
LPOptimizer.print_change_stats(res, false);
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
assert(res.space == 24 && res.total_space == 24);
|
||||||
console.log('\nAdding 3rd failure domain:');
|
console.log('\nAdding 3rd failure domain:');
|
||||||
cur_tree['dom3'] = osd_tree['dom3'];
|
cur_tree['dom3'] = osd_tree['dom3'];
|
||||||
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree: cur_tree, pg_size: 3 });
|
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree: cur_tree, pg_size: 3 });
|
||||||
LPOptimizer.print_change_stats(res, false);
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
assert(res.space == 36 && res.total_space == 36);
|
||||||
console.log('\nRemoving 3rd failure domain:');
|
console.log('\nRemoving 3rd failure domain:');
|
||||||
delete cur_tree['dom3'];
|
delete cur_tree['dom3'];
|
||||||
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree: cur_tree, pg_size: 3 });
|
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree: cur_tree, pg_size: 3 });
|
||||||
LPOptimizer.print_change_stats(res, false);
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
assert(res.space == 24 && res.total_space == 24);
|
||||||
console.log('\nRemoving 2nd failure domain:');
|
console.log('\nRemoving 2nd failure domain:');
|
||||||
delete cur_tree['dom2'];
|
delete cur_tree['dom2'];
|
||||||
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree: cur_tree, pg_size: 3 });
|
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree: cur_tree, pg_size: 3 });
|
||||||
LPOptimizer.print_change_stats(res, false);
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
assert(res.space == 12 && res.total_space == 12);
|
||||||
console.log('\nRemoving 1st failure domain:');
|
console.log('\nRemoving 1st failure domain:');
|
||||||
delete cur_tree['dom1'];
|
delete cur_tree['dom1'];
|
||||||
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree: cur_tree, pg_size: 3 });
|
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree: cur_tree, pg_size: 3 });
|
||||||
LPOptimizer.print_change_stats(res, false);
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
assert(res.space == 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
function assert(cond, txt)
|
||||||
|
{
|
||||||
|
if (!cond)
|
||||||
|
{
|
||||||
|
throw new Error((txt||'test')+' failed');
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
run().catch(console.error);
|
run().catch(console.error);
|
||||||
|
|||||||
@@ -0,0 +1,33 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
|
const LPOptimizer = require('./lp-optimizer.js');
|
||||||
|
|
||||||
|
const osd_tree = {
|
||||||
|
100: {
|
||||||
|
1: 0.1,
|
||||||
|
2: 0.1,
|
||||||
|
3: 0.1,
|
||||||
|
},
|
||||||
|
200: {
|
||||||
|
4: 0.1,
|
||||||
|
5: 0.1,
|
||||||
|
6: 0.1,
|
||||||
|
},
|
||||||
|
};
|
||||||
|
|
||||||
|
async function run()
|
||||||
|
{
|
||||||
|
let res;
|
||||||
|
console.log('256 PGs, 3+3 OSDs, size=2');
|
||||||
|
res = await LPOptimizer.optimize_initial({ osd_tree, pg_size: 2, pg_count: 256 });
|
||||||
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
|
||||||
|
// Should NOT fail with the "unfeasible or unbounded" exception
|
||||||
|
console.log('\nRemoving osd.2');
|
||||||
|
delete osd_tree[100][2];
|
||||||
|
res = await LPOptimizer.optimize_change({ prev_pgs: res.int_pgs, osd_tree, pg_size: 2 });
|
||||||
|
LPOptimizer.print_change_stats(res, false);
|
||||||
|
}
|
||||||
|
|
||||||
|
run().catch(console.error);
|
||||||
@@ -1,5 +1,5 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.0 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
const LPOptimizer = require('./lp-optimizer.js');
|
const LPOptimizer = require('./lp-optimizer.js');
|
||||||
|
|
||||||
|
|||||||
-822
@@ -1,822 +0,0 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
|
||||||
// License: VNPL-1.0 (see README.md for details)
|
|
||||||
|
|
||||||
#include "osd_primary.h"
|
|
||||||
|
|
||||||
// read: read directly or read paired stripe(s), reconstruct, return
|
|
||||||
// write: read paired stripe(s), reconstruct, modify, calculate parity, write
|
|
||||||
//
|
|
||||||
// nuance: take care to read the same version from paired stripes!
|
|
||||||
// to do so, we remember "last readable" version until a write request completes
|
|
||||||
// and we postpone other write requests to the same stripe until completion of previous ones
|
|
||||||
//
|
|
||||||
// sync: sync peers, get unstable versions, stabilize them
|
|
||||||
|
|
||||||
bool osd_t::prepare_primary_rw(osd_op_t *cur_op)
|
|
||||||
{
|
|
||||||
// PG number is calculated from the offset
|
|
||||||
// Our EC scheme stores data in fixed chunks equal to (K*block size)
|
|
||||||
// K = (pg_size-parity_chunks) in case of EC/XOR, or 1 for replicated pools
|
|
||||||
pool_id_t pool_id = INODE_POOL(cur_op->req.rw.inode);
|
|
||||||
// FIXME: We have to access pool config here, so make sure that it doesn't change while its PGs are active...
|
|
||||||
auto pool_cfg_it = st_cli.pool_config.find(pool_id);
|
|
||||||
if (pool_cfg_it == st_cli.pool_config.end())
|
|
||||||
{
|
|
||||||
// Pool config is not loaded yet
|
|
||||||
finish_op(cur_op, -EPIPE);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
auto & pool_cfg = pool_cfg_it->second;
|
|
||||||
uint64_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
|
||||||
uint64_t pg_block_size = bs_block_size * pg_data_size;
|
|
||||||
object_id oid = {
|
|
||||||
.inode = cur_op->req.rw.inode,
|
|
||||||
// oid.stripe = starting offset of the parity stripe
|
|
||||||
.stripe = (cur_op->req.rw.offset/pg_block_size)*pg_block_size,
|
|
||||||
};
|
|
||||||
pg_num_t pg_num = (cur_op->req.rw.inode + oid.stripe/pool_cfg.pg_stripe_size) % pg_counts[pool_id] + 1;
|
|
||||||
auto pg_it = pgs.find({ .pool_id = pool_id, .pg_num = pg_num });
|
|
||||||
if (pg_it == pgs.end() || !(pg_it->second.state & PG_ACTIVE))
|
|
||||||
{
|
|
||||||
// This OSD is not primary for this PG or the PG is inactive
|
|
||||||
// FIXME: Allow reads from PGs degraded under pg_minsize, but don't allow writes
|
|
||||||
finish_op(cur_op, -EPIPE);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
if ((cur_op->req.rw.offset + cur_op->req.rw.len) > (oid.stripe + pg_block_size) ||
|
|
||||||
(cur_op->req.rw.offset % bs_disk_alignment) != 0 ||
|
|
||||||
(cur_op->req.rw.len % bs_disk_alignment) != 0)
|
|
||||||
{
|
|
||||||
finish_op(cur_op, -EINVAL);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
osd_primary_op_data_t *op_data = (osd_primary_op_data_t*)calloc_or_die(
|
|
||||||
1, sizeof(osd_primary_op_data_t) + sizeof(osd_rmw_stripe_t) * (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pg_it->second.pg_size)
|
|
||||||
);
|
|
||||||
op_data->pg_num = pg_num;
|
|
||||||
op_data->oid = oid;
|
|
||||||
op_data->stripes = ((osd_rmw_stripe_t*)(op_data+1));
|
|
||||||
op_data->scheme = pool_cfg.scheme;
|
|
||||||
op_data->pg_data_size = pg_data_size;
|
|
||||||
cur_op->op_data = op_data;
|
|
||||||
split_stripes(pg_data_size, bs_block_size, (uint32_t)(cur_op->req.rw.offset - oid.stripe), cur_op->req.rw.len, op_data->stripes);
|
|
||||||
pg_it->second.inflight++;
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
static uint64_t* get_object_osd_set(pg_t &pg, object_id &oid, uint64_t *def, pg_osd_set_state_t **object_state)
|
|
||||||
{
|
|
||||||
if (!(pg.state & (PG_HAS_INCOMPLETE | PG_HAS_DEGRADED | PG_HAS_MISPLACED)))
|
|
||||||
{
|
|
||||||
*object_state = NULL;
|
|
||||||
return def;
|
|
||||||
}
|
|
||||||
auto st_it = pg.incomplete_objects.find(oid);
|
|
||||||
if (st_it != pg.incomplete_objects.end())
|
|
||||||
{
|
|
||||||
*object_state = st_it->second;
|
|
||||||
return st_it->second->read_target.data();
|
|
||||||
}
|
|
||||||
st_it = pg.degraded_objects.find(oid);
|
|
||||||
if (st_it != pg.degraded_objects.end())
|
|
||||||
{
|
|
||||||
*object_state = st_it->second;
|
|
||||||
return st_it->second->read_target.data();
|
|
||||||
}
|
|
||||||
st_it = pg.misplaced_objects.find(oid);
|
|
||||||
if (st_it != pg.misplaced_objects.end())
|
|
||||||
{
|
|
||||||
*object_state = st_it->second;
|
|
||||||
return st_it->second->read_target.data();
|
|
||||||
}
|
|
||||||
*object_state = NULL;
|
|
||||||
return def;
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_t::continue_primary_read(osd_op_t *cur_op)
|
|
||||||
{
|
|
||||||
if (!cur_op->op_data && !prepare_primary_rw(cur_op))
|
|
||||||
{
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
osd_primary_op_data_t *op_data = cur_op->op_data;
|
|
||||||
if (op_data->st == 1) goto resume_1;
|
|
||||||
else if (op_data->st == 2) goto resume_2;
|
|
||||||
{
|
|
||||||
auto & pg = pgs[{ .pool_id = INODE_POOL(op_data->oid.inode), .pg_num = op_data->pg_num }];
|
|
||||||
for (int role = 0; role < op_data->pg_data_size; role++)
|
|
||||||
{
|
|
||||||
op_data->stripes[role].read_start = op_data->stripes[role].req_start;
|
|
||||||
op_data->stripes[role].read_end = op_data->stripes[role].req_end;
|
|
||||||
}
|
|
||||||
// Determine version
|
|
||||||
auto vo_it = pg.ver_override.find(op_data->oid);
|
|
||||||
op_data->target_ver = vo_it != pg.ver_override.end() ? vo_it->second : UINT64_MAX;
|
|
||||||
if (pg.state == PG_ACTIVE || op_data->scheme == POOL_SCHEME_REPLICATED)
|
|
||||||
{
|
|
||||||
// Fast happy-path
|
|
||||||
cur_op->buf = alloc_read_buffer(op_data->stripes, op_data->pg_data_size, 0);
|
|
||||||
submit_primary_subops(SUBMIT_READ, op_data->target_ver,
|
|
||||||
(op_data->scheme == POOL_SCHEME_REPLICATED ? pg.pg_size : op_data->pg_data_size), pg.cur_set.data(), cur_op);
|
|
||||||
op_data->st = 1;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// PG may be degraded or have misplaced objects
|
|
||||||
uint64_t* cur_set = get_object_osd_set(pg, op_data->oid, pg.cur_set.data(), &op_data->object_state);
|
|
||||||
if (extend_missing_stripes(op_data->stripes, cur_set, op_data->pg_data_size, pg.pg_size) < 0)
|
|
||||||
{
|
|
||||||
finish_op(cur_op, -EIO);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
// Submit reads
|
|
||||||
op_data->pg_size = pg.pg_size;
|
|
||||||
op_data->scheme = pg.scheme;
|
|
||||||
op_data->degraded = 1;
|
|
||||||
cur_op->buf = alloc_read_buffer(op_data->stripes, pg.pg_size, 0);
|
|
||||||
submit_primary_subops(SUBMIT_READ, op_data->target_ver, pg.pg_size, cur_set, cur_op);
|
|
||||||
op_data->st = 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
resume_1:
|
|
||||||
return;
|
|
||||||
resume_2:
|
|
||||||
if (op_data->errors > 0)
|
|
||||||
{
|
|
||||||
finish_op(cur_op, op_data->epipe > 0 ? -EPIPE : -EIO);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (op_data->degraded)
|
|
||||||
{
|
|
||||||
// Reconstruct missing stripes
|
|
||||||
osd_rmw_stripe_t *stripes = op_data->stripes;
|
|
||||||
if (op_data->scheme == POOL_SCHEME_XOR)
|
|
||||||
{
|
|
||||||
reconstruct_stripes_xor(stripes, op_data->pg_size);
|
|
||||||
}
|
|
||||||
else if (op_data->scheme == POOL_SCHEME_JERASURE)
|
|
||||||
{
|
|
||||||
reconstruct_stripes_jerasure(stripes, op_data->pg_size, op_data->pg_data_size);
|
|
||||||
}
|
|
||||||
for (int role = 0; role < op_data->pg_size; role++)
|
|
||||||
{
|
|
||||||
if (stripes[role].req_end != 0)
|
|
||||||
{
|
|
||||||
// Send buffer in parts to avoid copying
|
|
||||||
cur_op->iov.push_back(
|
|
||||||
stripes[role].read_buf + (stripes[role].req_start - stripes[role].read_start),
|
|
||||||
stripes[role].req_end - stripes[role].req_start
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
cur_op->iov.push_back(cur_op->buf, cur_op->req.rw.len);
|
|
||||||
}
|
|
||||||
finish_op(cur_op, cur_op->req.rw.len);
|
|
||||||
}
|
|
||||||
|
|
||||||
bool osd_t::check_write_queue(osd_op_t *cur_op, pg_t & pg)
|
|
||||||
{
|
|
||||||
osd_primary_op_data_t *op_data = cur_op->op_data;
|
|
||||||
// Check if actions are pending for this object
|
|
||||||
auto act_it = pg.flush_actions.lower_bound((obj_piece_id_t){
|
|
||||||
.oid = op_data->oid,
|
|
||||||
.osd_num = 0,
|
|
||||||
});
|
|
||||||
if (act_it != pg.flush_actions.end() &&
|
|
||||||
act_it->first.oid.inode == op_data->oid.inode &&
|
|
||||||
(act_it->first.oid.stripe & ~STRIPE_MASK) == op_data->oid.stripe)
|
|
||||||
{
|
|
||||||
pg.write_queue.emplace(op_data->oid, cur_op);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
// Check if there are other write requests to the same object
|
|
||||||
auto vo_it = pg.write_queue.find(op_data->oid);
|
|
||||||
if (vo_it != pg.write_queue.end())
|
|
||||||
{
|
|
||||||
op_data->st = 1;
|
|
||||||
pg.write_queue.emplace(op_data->oid, cur_op);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
pg.write_queue.emplace(op_data->oid, cur_op);
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_t::continue_primary_write(osd_op_t *cur_op)
|
|
||||||
{
|
|
||||||
if (!cur_op->op_data && !prepare_primary_rw(cur_op))
|
|
||||||
{
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
osd_primary_op_data_t *op_data = cur_op->op_data;
|
|
||||||
auto & pg = pgs[{ .pool_id = INODE_POOL(op_data->oid.inode), .pg_num = op_data->pg_num }];
|
|
||||||
if (op_data->st == 1) goto resume_1;
|
|
||||||
else if (op_data->st == 2) goto resume_2;
|
|
||||||
else if (op_data->st == 3) goto resume_3;
|
|
||||||
else if (op_data->st == 4) goto resume_4;
|
|
||||||
else if (op_data->st == 5) goto resume_5;
|
|
||||||
else if (op_data->st == 6) goto resume_6;
|
|
||||||
else if (op_data->st == 7) goto resume_7;
|
|
||||||
else if (op_data->st == 8) goto resume_8;
|
|
||||||
else if (op_data->st == 9) goto resume_9;
|
|
||||||
else if (op_data->st == 10) goto resume_10;
|
|
||||||
assert(op_data->st == 0);
|
|
||||||
if (!check_write_queue(cur_op, pg))
|
|
||||||
{
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
resume_1:
|
|
||||||
// Determine blocks to read and write
|
|
||||||
// Missing chunks are allowed to be overwritten even in incomplete objects
|
|
||||||
// FIXME: Allow to do small writes to the old (degraded/misplaced) OSD set for lower performance impact
|
|
||||||
op_data->prev_set = get_object_osd_set(pg, op_data->oid, pg.cur_set.data(), &op_data->object_state);
|
|
||||||
if (op_data->scheme == POOL_SCHEME_REPLICATED)
|
|
||||||
{
|
|
||||||
// Simplified algorithm
|
|
||||||
op_data->stripes[0].write_start = op_data->stripes[0].req_start;
|
|
||||||
op_data->stripes[0].write_end = op_data->stripes[0].req_end;
|
|
||||||
op_data->stripes[0].write_buf = cur_op->buf;
|
|
||||||
if (pg.cur_set.data() != op_data->prev_set && (op_data->stripes[0].write_start != 0 ||
|
|
||||||
op_data->stripes[0].write_end != bs_block_size))
|
|
||||||
{
|
|
||||||
// Object is degraded/misplaced and will be moved to <write_osd_set>
|
|
||||||
op_data->stripes[0].read_start = 0;
|
|
||||||
op_data->stripes[0].read_end = bs_block_size;
|
|
||||||
cur_op->rmw_buf = op_data->stripes[0].read_buf = memalign_or_die(MEM_ALIGNMENT, bs_block_size);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
cur_op->rmw_buf = calc_rmw(cur_op->buf, op_data->stripes, op_data->prev_set,
|
|
||||||
pg.pg_size, op_data->pg_data_size, pg.pg_cursize, pg.cur_set.data(), bs_block_size);
|
|
||||||
if (!cur_op->rmw_buf)
|
|
||||||
{
|
|
||||||
// Refuse partial overwrite of an incomplete object
|
|
||||||
cur_op->reply.hdr.retval = -EINVAL;
|
|
||||||
goto continue_others;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Read required blocks
|
|
||||||
submit_primary_subops(SUBMIT_RMW_READ, UINT64_MAX, pg.pg_size, op_data->prev_set, cur_op);
|
|
||||||
resume_2:
|
|
||||||
op_data->st = 2;
|
|
||||||
return;
|
|
||||||
resume_3:
|
|
||||||
if (op_data->errors > 0)
|
|
||||||
{
|
|
||||||
pg_cancel_write_queue(pg, cur_op, op_data->oid, op_data->epipe > 0 ? -EPIPE : -EIO);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
// Save version override for parallel reads
|
|
||||||
pg.ver_override[op_data->oid] = op_data->fact_ver;
|
|
||||||
if (op_data->scheme == POOL_SCHEME_REPLICATED)
|
|
||||||
{
|
|
||||||
// Only (possibly) copy new data from the request into the recovery buffer
|
|
||||||
if (pg.cur_set.data() != op_data->prev_set && (op_data->stripes[0].write_start != 0 ||
|
|
||||||
op_data->stripes[0].write_end != bs_block_size))
|
|
||||||
{
|
|
||||||
memcpy(
|
|
||||||
op_data->stripes[0].read_buf + op_data->stripes[0].req_start,
|
|
||||||
op_data->stripes[0].write_buf,
|
|
||||||
op_data->stripes[0].req_end - op_data->stripes[0].req_start
|
|
||||||
);
|
|
||||||
op_data->stripes[0].write_buf = op_data->stripes[0].read_buf;
|
|
||||||
op_data->stripes[0].write_start = 0;
|
|
||||||
op_data->stripes[0].write_end = bs_block_size;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// Recover missing stripes, calculate parity
|
|
||||||
if (pg.scheme == POOL_SCHEME_XOR)
|
|
||||||
{
|
|
||||||
calc_rmw_parity_xor(op_data->stripes, pg.pg_size, op_data->prev_set, pg.cur_set.data(), bs_block_size);
|
|
||||||
}
|
|
||||||
else if (pg.scheme == POOL_SCHEME_JERASURE)
|
|
||||||
{
|
|
||||||
calc_rmw_parity_jerasure(op_data->stripes, pg.pg_size, op_data->pg_data_size, op_data->prev_set, pg.cur_set.data(), bs_block_size);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Send writes
|
|
||||||
if ((op_data->fact_ver >> (64-PG_EPOCH_BITS)) < pg.epoch)
|
|
||||||
{
|
|
||||||
op_data->target_ver = ((uint64_t)pg.epoch << (64-PG_EPOCH_BITS)) | 1;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
if ((op_data->fact_ver & (1ul<<(64-PG_EPOCH_BITS) - 1)) == (1ul<<(64-PG_EPOCH_BITS) - 1))
|
|
||||||
{
|
|
||||||
assert(pg.epoch != ((1ul << PG_EPOCH_BITS)-1));
|
|
||||||
pg.epoch++;
|
|
||||||
}
|
|
||||||
op_data->target_ver = op_data->fact_ver + 1;
|
|
||||||
}
|
|
||||||
if (pg.epoch > pg.reported_epoch)
|
|
||||||
{
|
|
||||||
// Report newer epoch before writing
|
|
||||||
// FIXME: We may report only one PG state here...
|
|
||||||
this->pg_state_dirty.insert({ .pool_id = pg.pool_id, .pg_num = pg.pg_num });
|
|
||||||
pg.history_changed = true;
|
|
||||||
report_pg_states();
|
|
||||||
resume_10:
|
|
||||||
if (pg.epoch > pg.reported_epoch)
|
|
||||||
{
|
|
||||||
op_data->st = 10;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
submit_primary_subops(SUBMIT_WRITE, op_data->target_ver, pg.pg_size, pg.cur_set.data(), cur_op);
|
|
||||||
resume_4:
|
|
||||||
op_data->st = 4;
|
|
||||||
return;
|
|
||||||
resume_5:
|
|
||||||
if (op_data->errors > 0)
|
|
||||||
{
|
|
||||||
pg_cancel_write_queue(pg, cur_op, op_data->oid, op_data->epipe > 0 ? -EPIPE : -EIO);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
resume_6:
|
|
||||||
resume_7:
|
|
||||||
if (!remember_unstable_write(cur_op, pg, pg.cur_loc_set, 6))
|
|
||||||
{
|
|
||||||
// FIXME: Check for immediate_commit == IMMEDIATE_SMALL
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (op_data->fact_ver == 1)
|
|
||||||
{
|
|
||||||
// Object is created
|
|
||||||
pg.clean_count++;
|
|
||||||
pg.total_count++;
|
|
||||||
}
|
|
||||||
if (op_data->object_state)
|
|
||||||
{
|
|
||||||
{
|
|
||||||
int recovery_type = op_data->object_state->state & (OBJ_DEGRADED|OBJ_INCOMPLETE) ? 0 : 1;
|
|
||||||
recovery_stat_count[0][recovery_type]++;
|
|
||||||
if (!recovery_stat_count[0][recovery_type])
|
|
||||||
{
|
|
||||||
recovery_stat_count[0][recovery_type]++;
|
|
||||||
recovery_stat_bytes[0][recovery_type] = 0;
|
|
||||||
}
|
|
||||||
for (int role = 0; role < (op_data->scheme == POOL_SCHEME_REPLICATED ? 1 : pg.pg_size); role++)
|
|
||||||
{
|
|
||||||
recovery_stat_bytes[0][recovery_type] += op_data->stripes[role].write_end - op_data->stripes[role].write_start;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (op_data->object_state->state & OBJ_MISPLACED)
|
|
||||||
{
|
|
||||||
// Remove extra chunks
|
|
||||||
submit_primary_del_subops(cur_op, pg.cur_set.data(), pg.pg_size, op_data->object_state->osd_set);
|
|
||||||
if (op_data->n_subops > 0)
|
|
||||||
{
|
|
||||||
resume_8:
|
|
||||||
op_data->st = 8;
|
|
||||||
return;
|
|
||||||
resume_9:
|
|
||||||
if (op_data->errors > 0)
|
|
||||||
{
|
|
||||||
pg_cancel_write_queue(pg, cur_op, op_data->oid, op_data->epipe > 0 ? -EPIPE : -EIO);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Clear object state
|
|
||||||
remove_object_from_state(op_data->oid, op_data->object_state, pg);
|
|
||||||
pg.clean_count++;
|
|
||||||
}
|
|
||||||
cur_op->reply.hdr.retval = cur_op->req.rw.len;
|
|
||||||
continue_others:
|
|
||||||
// Remove version override
|
|
||||||
pg.ver_override.erase(op_data->oid);
|
|
||||||
object_id oid = op_data->oid;
|
|
||||||
finish_op(cur_op, cur_op->reply.hdr.retval);
|
|
||||||
// Continue other write operations to the same object
|
|
||||||
auto next_it = pg.write_queue.find(oid);
|
|
||||||
auto this_it = next_it;
|
|
||||||
if (this_it != pg.write_queue.end() && this_it->second == cur_op)
|
|
||||||
{
|
|
||||||
next_it++;
|
|
||||||
pg.write_queue.erase(this_it);
|
|
||||||
if (next_it != pg.write_queue.end() && next_it->first == oid)
|
|
||||||
{
|
|
||||||
osd_op_t *next_op = next_it->second;
|
|
||||||
continue_primary_write(next_op);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
bool osd_t::remember_unstable_write(osd_op_t *cur_op, pg_t & pg, pg_osd_set_t & loc_set, int base_state)
|
|
||||||
{
|
|
||||||
osd_primary_op_data_t *op_data = cur_op->op_data;
|
|
||||||
if (op_data->st == base_state)
|
|
||||||
{
|
|
||||||
goto resume_6;
|
|
||||||
}
|
|
||||||
else if (op_data->st == base_state+1)
|
|
||||||
{
|
|
||||||
goto resume_7;
|
|
||||||
}
|
|
||||||
// FIXME: Check for immediate_commit == IMMEDIATE_SMALL
|
|
||||||
if (immediate_commit == IMMEDIATE_ALL)
|
|
||||||
{
|
|
||||||
if (op_data->scheme != POOL_SCHEME_REPLICATED)
|
|
||||||
{
|
|
||||||
// Send STABILIZE ops immediately
|
|
||||||
op_data->unstable_write_osds = new std::vector<unstable_osd_num_t>();
|
|
||||||
op_data->unstable_writes = new obj_ver_id[loc_set.size()];
|
|
||||||
{
|
|
||||||
int last_start = 0;
|
|
||||||
for (auto & chunk: loc_set)
|
|
||||||
{
|
|
||||||
op_data->unstable_writes[last_start] = (obj_ver_id){
|
|
||||||
.oid = {
|
|
||||||
.inode = op_data->oid.inode,
|
|
||||||
.stripe = op_data->oid.stripe | chunk.role,
|
|
||||||
},
|
|
||||||
.version = op_data->fact_ver,
|
|
||||||
};
|
|
||||||
op_data->unstable_write_osds->push_back((unstable_osd_num_t){
|
|
||||||
.osd_num = chunk.osd_num,
|
|
||||||
.start = last_start,
|
|
||||||
.len = 1,
|
|
||||||
});
|
|
||||||
last_start++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
submit_primary_stab_subops(cur_op);
|
|
||||||
resume_6:
|
|
||||||
op_data->st = 6;
|
|
||||||
return false;
|
|
||||||
resume_7:
|
|
||||||
// FIXME: Free those in the destructor?
|
|
||||||
delete op_data->unstable_write_osds;
|
|
||||||
delete[] op_data->unstable_writes;
|
|
||||||
op_data->unstable_writes = NULL;
|
|
||||||
op_data->unstable_write_osds = NULL;
|
|
||||||
if (op_data->errors > 0)
|
|
||||||
{
|
|
||||||
pg_cancel_write_queue(pg, cur_op, op_data->oid, op_data->epipe > 0 ? -EPIPE : -EIO);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
if (op_data->scheme != POOL_SCHEME_REPLICATED)
|
|
||||||
{
|
|
||||||
// Remember version as unstable for EC/XOR
|
|
||||||
for (auto & chunk: loc_set)
|
|
||||||
{
|
|
||||||
this->dirty_osds.insert(chunk.osd_num);
|
|
||||||
this->unstable_writes[(osd_object_id_t){
|
|
||||||
.osd_num = chunk.osd_num,
|
|
||||||
.oid = {
|
|
||||||
.inode = op_data->oid.inode,
|
|
||||||
.stripe = op_data->oid.stripe | chunk.role,
|
|
||||||
},
|
|
||||||
}] = op_data->fact_ver;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
// Only remember to sync OSDs for replicated pools
|
|
||||||
for (auto & chunk: loc_set)
|
|
||||||
{
|
|
||||||
this->dirty_osds.insert(chunk.osd_num);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Remember PG as dirty to drop the connection when PG goes offline
|
|
||||||
// (this is required because of the "lazy sync")
|
|
||||||
c_cli.clients[cur_op->peer_fd]->dirty_pgs.insert({ .pool_id = pg.pool_id, .pg_num = pg.pg_num });
|
|
||||||
dirty_pgs.insert({ .pool_id = pg.pool_id, .pg_num = pg.pg_num });
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Save and clear unstable_writes -> SYNC all -> STABLE all
|
|
||||||
void osd_t::continue_primary_sync(osd_op_t *cur_op)
|
|
||||||
{
|
|
||||||
if (!cur_op->op_data)
|
|
||||||
{
|
|
||||||
cur_op->op_data = (osd_primary_op_data_t*)calloc_or_die(1, sizeof(osd_primary_op_data_t));
|
|
||||||
}
|
|
||||||
osd_primary_op_data_t *op_data = cur_op->op_data;
|
|
||||||
if (op_data->st == 1) goto resume_1;
|
|
||||||
else if (op_data->st == 2) goto resume_2;
|
|
||||||
else if (op_data->st == 3) goto resume_3;
|
|
||||||
else if (op_data->st == 4) goto resume_4;
|
|
||||||
else if (op_data->st == 5) goto resume_5;
|
|
||||||
else if (op_data->st == 6) goto resume_6;
|
|
||||||
assert(op_data->st == 0);
|
|
||||||
if (syncs_in_progress.size() > 0)
|
|
||||||
{
|
|
||||||
// Wait for previous syncs, if any
|
|
||||||
// FIXME: We may try to execute the current one in parallel, like in Blockstore, but I'm not sure if it matters at all
|
|
||||||
syncs_in_progress.push_back(cur_op);
|
|
||||||
op_data->st = 1;
|
|
||||||
resume_1:
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
syncs_in_progress.push_back(cur_op);
|
|
||||||
}
|
|
||||||
resume_2:
|
|
||||||
if (dirty_osds.size() == 0)
|
|
||||||
{
|
|
||||||
// Nothing to sync
|
|
||||||
goto finish;
|
|
||||||
}
|
|
||||||
// Save and clear unstable_writes
|
|
||||||
// In theory it is possible to do in on a per-client basis, but this seems to be an unnecessary complication
|
|
||||||
// It would be cool not to copy these here at all, but someone has to deduplicate them by object IDs anyway
|
|
||||||
if (unstable_writes.size() > 0)
|
|
||||||
{
|
|
||||||
op_data->unstable_write_osds = new std::vector<unstable_osd_num_t>();
|
|
||||||
op_data->unstable_writes = new obj_ver_id[this->unstable_writes.size()];
|
|
||||||
osd_num_t last_osd = 0;
|
|
||||||
int last_start = 0, last_end = 0;
|
|
||||||
for (auto it = this->unstable_writes.begin(); it != this->unstable_writes.end(); it++)
|
|
||||||
{
|
|
||||||
if (last_osd != it->first.osd_num)
|
|
||||||
{
|
|
||||||
if (last_osd != 0)
|
|
||||||
{
|
|
||||||
op_data->unstable_write_osds->push_back((unstable_osd_num_t){
|
|
||||||
.osd_num = last_osd,
|
|
||||||
.start = last_start,
|
|
||||||
.len = last_end - last_start,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
last_osd = it->first.osd_num;
|
|
||||||
last_start = last_end;
|
|
||||||
}
|
|
||||||
op_data->unstable_writes[last_end] = (obj_ver_id){
|
|
||||||
.oid = it->first.oid,
|
|
||||||
.version = it->second,
|
|
||||||
};
|
|
||||||
last_end++;
|
|
||||||
}
|
|
||||||
if (last_osd != 0)
|
|
||||||
{
|
|
||||||
op_data->unstable_write_osds->push_back((unstable_osd_num_t){
|
|
||||||
.osd_num = last_osd,
|
|
||||||
.start = last_start,
|
|
||||||
.len = last_end - last_start,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
this->unstable_writes.clear();
|
|
||||||
}
|
|
||||||
{
|
|
||||||
void *dirty_buf = malloc_or_die(sizeof(pool_pg_num_t)*dirty_pgs.size() + sizeof(osd_num_t)*dirty_osds.size());
|
|
||||||
op_data->dirty_pgs = (pool_pg_num_t*)dirty_buf;
|
|
||||||
op_data->dirty_osds = (osd_num_t*)(dirty_buf + sizeof(pool_pg_num_t)*dirty_pgs.size());
|
|
||||||
op_data->dirty_pg_count = dirty_pgs.size();
|
|
||||||
op_data->dirty_osd_count = dirty_osds.size();
|
|
||||||
int dpg = 0;
|
|
||||||
for (auto dirty_pg_num: dirty_pgs)
|
|
||||||
{
|
|
||||||
pgs[dirty_pg_num].inflight++;
|
|
||||||
op_data->dirty_pgs[dpg++] = dirty_pg_num;
|
|
||||||
}
|
|
||||||
dirty_pgs.clear();
|
|
||||||
dpg = 0;
|
|
||||||
for (auto osd_num: dirty_osds)
|
|
||||||
{
|
|
||||||
op_data->dirty_osds[dpg++] = osd_num;
|
|
||||||
}
|
|
||||||
dirty_osds.clear();
|
|
||||||
}
|
|
||||||
if (immediate_commit != IMMEDIATE_ALL)
|
|
||||||
{
|
|
||||||
// SYNC
|
|
||||||
submit_primary_sync_subops(cur_op);
|
|
||||||
resume_3:
|
|
||||||
op_data->st = 3;
|
|
||||||
return;
|
|
||||||
resume_4:
|
|
||||||
if (op_data->errors > 0)
|
|
||||||
{
|
|
||||||
goto resume_6;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (op_data->unstable_writes)
|
|
||||||
{
|
|
||||||
// Stabilize version sets, if any
|
|
||||||
submit_primary_stab_subops(cur_op);
|
|
||||||
resume_5:
|
|
||||||
op_data->st = 5;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
resume_6:
|
|
||||||
if (op_data->errors > 0)
|
|
||||||
{
|
|
||||||
// Return PGs and OSDs back into their dirty sets
|
|
||||||
for (int i = 0; i < op_data->dirty_pg_count; i++)
|
|
||||||
{
|
|
||||||
dirty_pgs.insert(op_data->dirty_pgs[i]);
|
|
||||||
}
|
|
||||||
for (int i = 0; i < op_data->dirty_osd_count; i++)
|
|
||||||
{
|
|
||||||
dirty_osds.insert(op_data->dirty_osds[i]);
|
|
||||||
}
|
|
||||||
if (op_data->unstable_writes)
|
|
||||||
{
|
|
||||||
// Return objects back into the unstable write set
|
|
||||||
for (auto unstable_osd: *(op_data->unstable_write_osds))
|
|
||||||
{
|
|
||||||
for (int i = 0; i < unstable_osd.len; i++)
|
|
||||||
{
|
|
||||||
// Except those from peered PGs
|
|
||||||
auto & w = op_data->unstable_writes[i];
|
|
||||||
pool_pg_num_t wpg = {
|
|
||||||
.pool_id = INODE_POOL(w.oid.inode),
|
|
||||||
.pg_num = map_to_pg(w.oid, st_cli.pool_config.at(INODE_POOL(w.oid.inode)).pg_stripe_size),
|
|
||||||
};
|
|
||||||
if (pgs[wpg].state & PG_ACTIVE)
|
|
||||||
{
|
|
||||||
uint64_t & dest = this->unstable_writes[(osd_object_id_t){
|
|
||||||
.osd_num = unstable_osd.osd_num,
|
|
||||||
.oid = w.oid,
|
|
||||||
}];
|
|
||||||
dest = dest < w.version ? w.version : dest;
|
|
||||||
dirty_pgs.insert(wpg);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (int i = 0; i < op_data->dirty_pg_count; i++)
|
|
||||||
{
|
|
||||||
auto & pg = pgs.at(op_data->dirty_pgs[i]);
|
|
||||||
pg.inflight--;
|
|
||||||
if ((pg.state & PG_STOPPING) && pg.inflight == 0 && !pg.flush_batch)
|
|
||||||
{
|
|
||||||
finish_stop_pg(pg);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// FIXME: Free those in the destructor?
|
|
||||||
free(op_data->dirty_pgs);
|
|
||||||
op_data->dirty_pgs = NULL;
|
|
||||||
op_data->dirty_osds = NULL;
|
|
||||||
if (op_data->unstable_writes)
|
|
||||||
{
|
|
||||||
delete op_data->unstable_write_osds;
|
|
||||||
delete[] op_data->unstable_writes;
|
|
||||||
op_data->unstable_writes = NULL;
|
|
||||||
op_data->unstable_write_osds = NULL;
|
|
||||||
}
|
|
||||||
if (op_data->errors > 0)
|
|
||||||
{
|
|
||||||
finish_op(cur_op, op_data->epipe > 0 ? -EPIPE : -EIO);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
finish:
|
|
||||||
if (cur_op->peer_fd)
|
|
||||||
{
|
|
||||||
auto it = c_cli.clients.find(cur_op->peer_fd);
|
|
||||||
if (it != c_cli.clients.end())
|
|
||||||
it->second->dirty_pgs.clear();
|
|
||||||
}
|
|
||||||
finish_op(cur_op, 0);
|
|
||||||
}
|
|
||||||
assert(syncs_in_progress.front() == cur_op);
|
|
||||||
syncs_in_progress.pop_front();
|
|
||||||
if (syncs_in_progress.size() > 0)
|
|
||||||
{
|
|
||||||
cur_op = syncs_in_progress.front();
|
|
||||||
op_data = cur_op->op_data;
|
|
||||||
op_data->st++;
|
|
||||||
goto resume_2;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Decrement pg_osd_set_state_t's object_count and change PG state accordingly
|
|
||||||
void osd_t::remove_object_from_state(object_id & oid, pg_osd_set_state_t *object_state, pg_t & pg)
|
|
||||||
{
|
|
||||||
if (object_state->state & OBJ_INCOMPLETE)
|
|
||||||
{
|
|
||||||
// Successful write means that object is not incomplete anymore
|
|
||||||
this->incomplete_objects--;
|
|
||||||
pg.incomplete_objects.erase(oid);
|
|
||||||
if (!pg.incomplete_objects.size())
|
|
||||||
{
|
|
||||||
pg.state = pg.state & ~PG_HAS_INCOMPLETE;
|
|
||||||
report_pg_state(pg);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else if (object_state->state & OBJ_DEGRADED)
|
|
||||||
{
|
|
||||||
this->degraded_objects--;
|
|
||||||
pg.degraded_objects.erase(oid);
|
|
||||||
if (!pg.degraded_objects.size())
|
|
||||||
{
|
|
||||||
pg.state = pg.state & ~PG_HAS_DEGRADED;
|
|
||||||
report_pg_state(pg);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else if (object_state->state & OBJ_MISPLACED)
|
|
||||||
{
|
|
||||||
this->misplaced_objects--;
|
|
||||||
pg.misplaced_objects.erase(oid);
|
|
||||||
if (!pg.misplaced_objects.size())
|
|
||||||
{
|
|
||||||
pg.state = pg.state & ~PG_HAS_MISPLACED;
|
|
||||||
report_pg_state(pg);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
throw std::runtime_error("BUG: Invalid object state: "+std::to_string(object_state->state));
|
|
||||||
}
|
|
||||||
object_state->object_count--;
|
|
||||||
if (!object_state->object_count)
|
|
||||||
{
|
|
||||||
pg.state_dict.erase(object_state->osd_set);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_t::continue_primary_del(osd_op_t *cur_op)
|
|
||||||
{
|
|
||||||
if (!cur_op->op_data && !prepare_primary_rw(cur_op))
|
|
||||||
{
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
osd_primary_op_data_t *op_data = cur_op->op_data;
|
|
||||||
auto & pg = pgs[{ .pool_id = INODE_POOL(op_data->oid.inode), .pg_num = op_data->pg_num }];
|
|
||||||
if (op_data->st == 1) goto resume_1;
|
|
||||||
else if (op_data->st == 2) goto resume_2;
|
|
||||||
else if (op_data->st == 3) goto resume_3;
|
|
||||||
else if (op_data->st == 4) goto resume_4;
|
|
||||||
else if (op_data->st == 5) goto resume_5;
|
|
||||||
assert(op_data->st == 0);
|
|
||||||
// Delete is forbidden even in active PGs if they're also degraded or have previous dead OSDs
|
|
||||||
if (pg.state & (PG_DEGRADED | PG_LEFT_ON_DEAD))
|
|
||||||
{
|
|
||||||
finish_op(cur_op, -EBUSY);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (!check_write_queue(cur_op, pg))
|
|
||||||
{
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
resume_1:
|
|
||||||
// Determine which OSDs contain this object and delete it
|
|
||||||
op_data->prev_set = get_object_osd_set(pg, op_data->oid, pg.cur_set.data(), &op_data->object_state);
|
|
||||||
// Submit 1 read to determine the actual version number
|
|
||||||
submit_primary_subops(SUBMIT_RMW_READ, UINT64_MAX, pg.pg_size, op_data->prev_set, cur_op);
|
|
||||||
resume_2:
|
|
||||||
op_data->st = 2;
|
|
||||||
return;
|
|
||||||
resume_3:
|
|
||||||
if (op_data->errors > 0)
|
|
||||||
{
|
|
||||||
pg_cancel_write_queue(pg, cur_op, op_data->oid, op_data->epipe > 0 ? -EPIPE : -EIO);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
// Save version override for parallel reads
|
|
||||||
pg.ver_override[op_data->oid] = op_data->fact_ver;
|
|
||||||
// Submit deletes
|
|
||||||
op_data->fact_ver++;
|
|
||||||
submit_primary_del_subops(cur_op, NULL, 0, op_data->object_state ? op_data->object_state->osd_set : pg.cur_loc_set);
|
|
||||||
resume_4:
|
|
||||||
op_data->st = 4;
|
|
||||||
return;
|
|
||||||
resume_5:
|
|
||||||
if (op_data->errors > 0)
|
|
||||||
{
|
|
||||||
pg_cancel_write_queue(pg, cur_op, op_data->oid, op_data->epipe > 0 ? -EPIPE : -EIO);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
// Remove version override
|
|
||||||
pg.ver_override.erase(op_data->oid);
|
|
||||||
// Adjust PG stats after "instant stabilize", because we need object_state above
|
|
||||||
if (!op_data->object_state)
|
|
||||||
{
|
|
||||||
pg.clean_count--;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
remove_object_from_state(op_data->oid, op_data->object_state, pg);
|
|
||||||
}
|
|
||||||
pg.total_count--;
|
|
||||||
object_id oid = op_data->oid;
|
|
||||||
finish_op(cur_op, cur_op->req.rw.len);
|
|
||||||
// Continue other write operations to the same object
|
|
||||||
auto next_it = pg.write_queue.find(oid);
|
|
||||||
auto this_it = next_it;
|
|
||||||
if (this_it != pg.write_queue.end() && this_it->second == cur_op)
|
|
||||||
{
|
|
||||||
next_it++;
|
|
||||||
pg.write_queue.erase(this_it);
|
|
||||||
if (next_it != pg.write_queue.end() &&
|
|
||||||
next_it->first == oid)
|
|
||||||
{
|
|
||||||
osd_op_t *next_op = next_it->second;
|
|
||||||
continue_primary_write(next_op);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,41 +0,0 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
|
||||||
// License: VNPL-1.0 (see README.md for details)
|
|
||||||
|
|
||||||
#pragma once
|
|
||||||
|
|
||||||
#include "osd.h"
|
|
||||||
#include "osd_rmw.h"
|
|
||||||
|
|
||||||
#define SUBMIT_READ 0
|
|
||||||
#define SUBMIT_RMW_READ 1
|
|
||||||
#define SUBMIT_WRITE 2
|
|
||||||
|
|
||||||
struct unstable_osd_num_t
|
|
||||||
{
|
|
||||||
osd_num_t osd_num;
|
|
||||||
int start, len;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct osd_primary_op_data_t
|
|
||||||
{
|
|
||||||
int st = 0;
|
|
||||||
pg_num_t pg_num;
|
|
||||||
object_id oid;
|
|
||||||
uint64_t target_ver;
|
|
||||||
uint64_t fact_ver = 0;
|
|
||||||
uint64_t scheme = 0;
|
|
||||||
int n_subops = 0, done = 0, errors = 0, epipe = 0;
|
|
||||||
int degraded = 0, pg_size, pg_data_size;
|
|
||||||
osd_rmw_stripe_t *stripes;
|
|
||||||
osd_op_t *subops = NULL;
|
|
||||||
uint64_t *prev_set = NULL;
|
|
||||||
pg_osd_set_state_t *object_state = NULL;
|
|
||||||
|
|
||||||
// for sync. oops, requires freeing
|
|
||||||
std::vector<unstable_osd_num_t> *unstable_write_osds = NULL;
|
|
||||||
pool_pg_num_t *dirty_pgs = NULL;
|
|
||||||
int dirty_pg_count = 0;
|
|
||||||
osd_num_t *dirty_osds = NULL;
|
|
||||||
int dirty_osd_count = 0;
|
|
||||||
obj_ver_id *unstable_writes = NULL;
|
|
||||||
};
|
|
||||||
@@ -0,0 +1,503 @@
|
|||||||
|
# Install as /usr/share/perl5/PVE/Storage/Custom/VitastorPlugin.pm
|
||||||
|
|
||||||
|
# Proxmox Vitastor Driver
|
||||||
|
# Copyright (c) Vitaliy Filippov, 2021+
|
||||||
|
# License: VNPL-1.1 or GNU AGPLv3.0
|
||||||
|
|
||||||
|
package PVE::Storage::Custom::VitastorPlugin;
|
||||||
|
|
||||||
|
use strict;
|
||||||
|
use warnings;
|
||||||
|
|
||||||
|
use JSON;
|
||||||
|
|
||||||
|
use PVE::Storage::Plugin;
|
||||||
|
use PVE::Tools qw(run_command);
|
||||||
|
|
||||||
|
use base qw(PVE::Storage::Plugin);
|
||||||
|
|
||||||
|
sub api
|
||||||
|
{
|
||||||
|
# Trick it :)
|
||||||
|
return PVE::Storage->APIVER;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub run_cli
|
||||||
|
{
|
||||||
|
my ($scfg, $cmd, %args) = @_;
|
||||||
|
my $retval;
|
||||||
|
my $stderr = '';
|
||||||
|
my $errmsg = $args{errmsg} ? $args{errmsg}.": " : "vitastor-cli error: ";
|
||||||
|
my $json = delete $args{json};
|
||||||
|
$json = 1 if !defined $json;
|
||||||
|
my $binary = delete $args{binary};
|
||||||
|
$binary = '/usr/bin/vitastor-cli' if !defined $binary;
|
||||||
|
if (!exists($args{errfunc}))
|
||||||
|
{
|
||||||
|
$args{errfunc} = sub
|
||||||
|
{
|
||||||
|
my $line = shift;
|
||||||
|
print STDERR $line;
|
||||||
|
*STDERR->flush();
|
||||||
|
$stderr .= $line;
|
||||||
|
};
|
||||||
|
}
|
||||||
|
if (!exists($args{outfunc}))
|
||||||
|
{
|
||||||
|
$retval = '';
|
||||||
|
$args{outfunc} = sub { $retval .= shift };
|
||||||
|
if ($json)
|
||||||
|
{
|
||||||
|
unshift @$cmd, '--json';
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if ($scfg->{vitastor_etcd_address})
|
||||||
|
{
|
||||||
|
unshift @$cmd, '--etcd_address', $scfg->{vitastor_etcd_address};
|
||||||
|
}
|
||||||
|
if ($scfg->{vitastor_config_path})
|
||||||
|
{
|
||||||
|
unshift @$cmd, '--config_path', $scfg->{vitastor_config_path};
|
||||||
|
}
|
||||||
|
unshift @$cmd, $binary;
|
||||||
|
eval { run_command($cmd, %args); };
|
||||||
|
if (my $err = $@)
|
||||||
|
{
|
||||||
|
die "Error invoking vitastor-cli: $err";
|
||||||
|
}
|
||||||
|
if (defined $retval)
|
||||||
|
{
|
||||||
|
# untaint
|
||||||
|
$retval =~ /^(.*)$/s;
|
||||||
|
if ($json)
|
||||||
|
{
|
||||||
|
eval { $retval = JSON::decode_json($1); };
|
||||||
|
if ($@)
|
||||||
|
{
|
||||||
|
die "vitastor-cli returned bad JSON: $@";
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
$retval = $1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return $retval;
|
||||||
|
}
|
||||||
|
|
||||||
|
# Configuration
|
||||||
|
|
||||||
|
sub type
|
||||||
|
{
|
||||||
|
return 'vitastor';
|
||||||
|
}
|
||||||
|
|
||||||
|
sub plugindata
|
||||||
|
{
|
||||||
|
return {
|
||||||
|
content => [ { images => 1, rootdir => 1 }, { images => 1 } ],
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
sub properties
|
||||||
|
{
|
||||||
|
return {
|
||||||
|
vitastor_etcd_address => {
|
||||||
|
description => 'IP address(es) of etcd.',
|
||||||
|
type => 'string',
|
||||||
|
format => 'pve-storage-portal-dns-list',
|
||||||
|
},
|
||||||
|
vitastor_etcd_prefix => {
|
||||||
|
description => 'Prefix for Vitastor etcd metadata',
|
||||||
|
type => 'string',
|
||||||
|
},
|
||||||
|
vitastor_config_path => {
|
||||||
|
description => 'Path to Vitastor configuration file',
|
||||||
|
type => 'string',
|
||||||
|
},
|
||||||
|
vitastor_prefix => {
|
||||||
|
description => 'Image name prefix',
|
||||||
|
type => 'string',
|
||||||
|
},
|
||||||
|
vitastor_pool => {
|
||||||
|
description => 'Default pool to use for images',
|
||||||
|
type => 'string',
|
||||||
|
},
|
||||||
|
vitastor_nbd => {
|
||||||
|
description => 'Use kernel NBD devices (slower)',
|
||||||
|
type => 'boolean',
|
||||||
|
},
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
sub options
|
||||||
|
{
|
||||||
|
return {
|
||||||
|
nodes => { optional => 1 },
|
||||||
|
disable => { optional => 1 },
|
||||||
|
vitastor_etcd_address => { optional => 1},
|
||||||
|
vitastor_etcd_prefix => { optional => 1 },
|
||||||
|
vitastor_config_path => { optional => 1 },
|
||||||
|
vitastor_prefix => { optional => 1 },
|
||||||
|
vitastor_pool => {},
|
||||||
|
vitastor_nbd => { optional => 1 },
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
# Storage implementation
|
||||||
|
|
||||||
|
sub parse_volname
|
||||||
|
{
|
||||||
|
my ($class, $volname) = @_;
|
||||||
|
if ($volname =~ m/^((base-(\d+)-\S+)\/)?((?:(base)|(vm))-(\d+)-\S+)$/)
|
||||||
|
{
|
||||||
|
# ($vtype, $name, $vmid, $basename, $basevmid, $isBase, $format)
|
||||||
|
return ('images', $4, $7, $2, $3, $5, 'raw');
|
||||||
|
}
|
||||||
|
die "unable to parse vitastor volume name '$volname'\n";
|
||||||
|
}
|
||||||
|
|
||||||
|
sub _qemu_option
|
||||||
|
{
|
||||||
|
my ($k, $v) = @_;
|
||||||
|
if (defined $v && $v ne "")
|
||||||
|
{
|
||||||
|
$v =~ s/:/\\:/gso;
|
||||||
|
return ":$k=$v";
|
||||||
|
}
|
||||||
|
return "";
|
||||||
|
}
|
||||||
|
|
||||||
|
sub path
|
||||||
|
{
|
||||||
|
my ($class, $scfg, $volname, $storeid, $snapname) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
my ($vtype, $name, $vmid) = $class->parse_volname($volname);
|
||||||
|
$name .= '@'.$snapname if $snapname;
|
||||||
|
if ($scfg->{vitastor_nbd})
|
||||||
|
{
|
||||||
|
my $mapped = run_cli($scfg, [ 'ls' ], binary => '/usr/bin/vitastor-nbd');
|
||||||
|
my ($kerneldev) = grep { $mapped->{$_}->{image} eq $prefix.$name } keys %$mapped;
|
||||||
|
die "Image not mapped via NBD" if !$kerneldev;
|
||||||
|
return ($kerneldev, $vmid, $vtype);
|
||||||
|
}
|
||||||
|
my $path = "vitastor";
|
||||||
|
$path .= _qemu_option('config_path', $scfg->{vitastor_config_path});
|
||||||
|
# FIXME This is the only exception: etcd_address -> etcd_host for qemu
|
||||||
|
$path .= _qemu_option('etcd_host', $scfg->{vitastor_etcd_address});
|
||||||
|
$path .= _qemu_option('etcd_prefix', $scfg->{vitastor_etcd_prefix});
|
||||||
|
$path .= _qemu_option('image', $prefix.$name);
|
||||||
|
return ($path, $vmid, $vtype);
|
||||||
|
}
|
||||||
|
|
||||||
|
sub _find_free_diskname
|
||||||
|
{
|
||||||
|
my ($class, $storeid, $scfg, $vmid, $fmt, $add_fmt_suffix) = @_;
|
||||||
|
my $list = _process_list($scfg, $storeid, run_cli($scfg, [ 'ls' ]));
|
||||||
|
$list = [ map { $_->{name} } @$list ];
|
||||||
|
return PVE::Storage::Plugin::get_next_vm_diskname($list, $storeid, $vmid, undef, $scfg);
|
||||||
|
}
|
||||||
|
|
||||||
|
# Used only in "Create Template" and, in fact, converts a VM into a template
|
||||||
|
# As a consequence, this is always invoked with the VM powered off
|
||||||
|
# So we just rename vm-xxx to base-xxx and make it a readonly base layer
|
||||||
|
sub create_base
|
||||||
|
{
|
||||||
|
my ($class, $storeid, $scfg, $volname) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
|
||||||
|
my ($vtype, $name, $vmid, $basename, $basevmid, $isBase) = $class->parse_volname($volname);
|
||||||
|
die "create_base not possible with base image\n" if $isBase;
|
||||||
|
|
||||||
|
my $info = _process_list($scfg, $storeid, run_cli($scfg, [ 'ls', $prefix.$name ]))->[0];
|
||||||
|
die "image $name does not exist\n" if !$info;
|
||||||
|
|
||||||
|
die "volname '$volname' contains wrong information about parent {$info->{parent}} $basename\n"
|
||||||
|
if $basename && (!$info->{parent} || $info->{parent} ne $basename);
|
||||||
|
|
||||||
|
my $newname = $name;
|
||||||
|
$newname =~ s/^vm-/base-/;
|
||||||
|
|
||||||
|
my $newvolname = $basename ? "$basename/$newname" : "$newname";
|
||||||
|
run_cli($scfg, [ 'modify', '--rename', $prefix.$newname, '--readonly', $prefix.$name ], json => 0);
|
||||||
|
|
||||||
|
return $newvolname;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub clone_image
|
||||||
|
{
|
||||||
|
my ($class, $scfg, $storeid, $volname, $vmid, $snapname) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
|
||||||
|
my $snap = '';
|
||||||
|
$snap = '@'.$snapname if length $snapname;
|
||||||
|
|
||||||
|
my ($vtype, $basename, $basevmid, undef, undef, $isBase) = $class->parse_volname($volname);
|
||||||
|
die "$volname is not a base image and snapname is not provided\n" if !$isBase && !length($snapname);
|
||||||
|
|
||||||
|
my $name = $class->find_free_diskname($storeid, $scfg, $vmid);
|
||||||
|
|
||||||
|
warn "clone $volname: $basename snapname $snap to $name\n";
|
||||||
|
|
||||||
|
my $newvol = "$basename/$name";
|
||||||
|
$newvol = $name if length($snapname);
|
||||||
|
|
||||||
|
run_cli($scfg, [ 'create', '--parent', $prefix.$basename.$snap, $prefix.$name ], json => 0);
|
||||||
|
|
||||||
|
return $newvol;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub alloc_image
|
||||||
|
{
|
||||||
|
# $size is in kb in this method
|
||||||
|
my ($class, $storeid, $scfg, $vmid, $fmt, $name, $size) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
die "illegal name '$name' - should be 'vm-$vmid-*'\n" if $name && $name !~ m/^vm-$vmid-/;
|
||||||
|
$name = $class->find_free_diskname($storeid, $scfg, $vmid) if !$name;
|
||||||
|
run_cli($scfg, [ 'create', '--size', (int(($size+3)/4)*4).'k', '--pool', $scfg->{vitastor_pool}, $prefix.$name ], json => 0);
|
||||||
|
return $name;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub free_image
|
||||||
|
{
|
||||||
|
my ($class, $storeid, $scfg, $volname, $isBase) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
my ($vtype, $name, $vmid, undef, undef, undef) = $class->parse_volname($volname);
|
||||||
|
$class->deactivate_volume($storeid, $scfg, $volname);
|
||||||
|
my $full_list = run_cli($scfg, [ 'ls', '-l' ]);
|
||||||
|
my $list = _process_list($scfg, $storeid, $full_list);
|
||||||
|
# Remove image and all its snapshots
|
||||||
|
my $rm_names = {
|
||||||
|
map { ($prefix.$_->{name} => 1) }
|
||||||
|
grep { $_->{name} eq $name || substr($_->{name}, 0, length($name)+1) eq ($name.'@') }
|
||||||
|
@$list
|
||||||
|
};
|
||||||
|
my $children = [ grep { $_->{parent_name} && $rm_names->{$_->{parent_name}} } @$full_list ];
|
||||||
|
die "Image has children: ".join(', ', map {
|
||||||
|
substr($_->{name}, 0, length $prefix) eq $prefix
|
||||||
|
? substr($_->name, length $prefix)
|
||||||
|
: $_->{name}
|
||||||
|
} @$children)."\n" if @$children;
|
||||||
|
my $to_remove = [ grep { $rm_names->{$_->{name}} } @$full_list ];
|
||||||
|
for my $rmi (@$to_remove)
|
||||||
|
{
|
||||||
|
run_cli($scfg, [ 'rm-data', '--pool', $rmi->{pool_id}, '--inode', $rmi->{inode_num} ], json => 0);
|
||||||
|
}
|
||||||
|
for my $rmi (@$to_remove)
|
||||||
|
{
|
||||||
|
run_cli($scfg, [ 'rm', $rmi->{name} ], json => 0);
|
||||||
|
}
|
||||||
|
return undef;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub _process_list
|
||||||
|
{
|
||||||
|
my ($scfg, $storeid, $result) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
my $list = [];
|
||||||
|
foreach my $el (@$result)
|
||||||
|
{
|
||||||
|
next if !$el->{name} || length($prefix) && substr($el->{name}, 0, length $prefix) ne $prefix;
|
||||||
|
my $name = substr($el->{name}, length $prefix);
|
||||||
|
next if $name =~ /@/;
|
||||||
|
my ($owner) = $name =~ /^(?:vm|base)-(\d+)-/s;
|
||||||
|
next if !defined $owner;
|
||||||
|
my $parent = !defined $el->{parent_name}
|
||||||
|
? undef
|
||||||
|
: ($prefix eq '' || substr($el->{parent_name}, 0, length $prefix) eq $prefix
|
||||||
|
? substr($el->{parent_name}, length $prefix) : '');
|
||||||
|
my $volid = $parent && $parent =~ /^(base-\d+-\S+)$/s
|
||||||
|
? "$storeid:$1/$name" : "$storeid:$name";
|
||||||
|
push @$list, {
|
||||||
|
format => 'raw',
|
||||||
|
volid => $volid,
|
||||||
|
name => $name,
|
||||||
|
size => $el->{size},
|
||||||
|
parent => $parent,
|
||||||
|
vmid => $owner,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
return $list;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub list_images
|
||||||
|
{
|
||||||
|
my ($class, $storeid, $scfg, $vmid, $vollist, $cache) = @_;
|
||||||
|
my $list = _process_list($scfg, $storeid, run_cli($scfg, [ 'ls', '-l' ]));
|
||||||
|
if ($vollist)
|
||||||
|
{
|
||||||
|
my $h = { map { ($_ => 1) } @$vollist };
|
||||||
|
$list = [ grep { $h->{$_->{volid}} } @$list ]
|
||||||
|
}
|
||||||
|
elsif (defined $vmid)
|
||||||
|
{
|
||||||
|
$list = [ grep { $_->{vmid} eq $vmid } @$list ];
|
||||||
|
}
|
||||||
|
return $list;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub status
|
||||||
|
{
|
||||||
|
my ($class, $storeid, $scfg, $cache) = @_;
|
||||||
|
my $stats = [ grep { $_->{name} eq $scfg->{vitastor_pool} } @{ run_cli($scfg, [ 'df' ]) } ]->[0];
|
||||||
|
my $free = $stats ? $stats->{max_available} : 0;
|
||||||
|
my $used = $stats ? $stats->{used_raw}/($stats->{raw_to_usable}||1) : 0;
|
||||||
|
my $total = $free+$used;
|
||||||
|
my $active = $stats ? 1 : 0;
|
||||||
|
return ($total, $free, $used, $active);
|
||||||
|
}
|
||||||
|
|
||||||
|
sub activate_storage
|
||||||
|
{
|
||||||
|
my ($class, $storeid, $scfg, $cache) = @_;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub deactivate_storage
|
||||||
|
{
|
||||||
|
my ($class, $storeid, $scfg, $cache) = @_;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub map_volume
|
||||||
|
{
|
||||||
|
my ($class, $storeid, $scfg, $volname, $snapname) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
|
||||||
|
my ($vtype, $img_name, $vmid) = $class->parse_volname($volname);
|
||||||
|
my $name = $img_name;
|
||||||
|
$name .= '@'.$snapname if $snapname;
|
||||||
|
|
||||||
|
my $mapped = run_cli($scfg, [ 'ls' ], binary => '/usr/bin/vitastor-nbd');
|
||||||
|
my ($kerneldev) = grep { $mapped->{$_}->{image} eq $prefix.$name } keys %$mapped;
|
||||||
|
return $kerneldev if $kerneldev && -b $kerneldev; # already mapped
|
||||||
|
|
||||||
|
$kerneldev = run_cli($scfg, [ 'map', '--image', $prefix.$name ], binary => '/usr/bin/vitastor-nbd', json => 0);
|
||||||
|
return $kerneldev;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub unmap_volume
|
||||||
|
{
|
||||||
|
my ($class, $storeid, $scfg, $volname, $snapname) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
|
||||||
|
return 1 if !$scfg->{vitastor_nbd};
|
||||||
|
|
||||||
|
my ($vtype, $name, $vmid) = $class->parse_volname($volname);
|
||||||
|
$name .= '@'.$snapname if $snapname;
|
||||||
|
|
||||||
|
my $mapped = run_cli($scfg, [ 'ls' ], binary => '/usr/bin/vitastor-nbd');
|
||||||
|
my ($kerneldev) = grep { $mapped->{$_}->{image} eq $prefix.$name } keys %$mapped;
|
||||||
|
if ($kerneldev && -b $kerneldev)
|
||||||
|
{
|
||||||
|
run_cli($scfg, [ 'unmap', $kerneldev ], binary => '/usr/bin/vitastor-nbd', json => 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub activate_volume
|
||||||
|
{
|
||||||
|
my ($class, $storeid, $scfg, $volname, $snapname, $cache) = @_;
|
||||||
|
$class->map_volume($storeid, $scfg, $volname, $snapname) if $scfg->{vitastor_nbd};
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub deactivate_volume
|
||||||
|
{
|
||||||
|
my ($class, $storeid, $scfg, $volname, $snapname, $cache) = @_;
|
||||||
|
$class->unmap_volume($storeid, $scfg, $volname, $snapname);
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub volume_size_info
|
||||||
|
{
|
||||||
|
my ($class, $scfg, $storeid, $volname, $timeout) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
my ($vtype, $name, $vmid) = $class->parse_volname($volname);
|
||||||
|
my $info = _process_list($scfg, $storeid, run_cli($scfg, [ 'ls', $prefix.$name ]))->[0];
|
||||||
|
#return wantarray ? ($size, $format, $used, $parent, $st->ctime) : $size;
|
||||||
|
return $info->{size};
|
||||||
|
}
|
||||||
|
|
||||||
|
sub volume_resize
|
||||||
|
{
|
||||||
|
my ($class, $scfg, $storeid, $volname, $size, $running) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
my ($vtype, $name, $vmid) = $class->parse_volname($volname);
|
||||||
|
# $size is in bytes in this method
|
||||||
|
run_cli($scfg, [ 'modify', '--resize', (int(($size+4095)/4096)*4).'k', $prefix.$name ], json => 0);
|
||||||
|
return undef;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub volume_snapshot
|
||||||
|
{
|
||||||
|
my ($class, $scfg, $storeid, $volname, $snap) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
my ($vtype, $name, $vmid) = $class->parse_volname($volname);
|
||||||
|
run_cli($scfg, [ 'create', '--snapshot', $snap, $prefix.$name ], json => 0);
|
||||||
|
return undef;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub volume_snapshot_rollback
|
||||||
|
{
|
||||||
|
my ($class, $scfg, $storeid, $volname, $snap) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
my ($vtype, $name, $vmid) = $class->parse_volname($volname);
|
||||||
|
run_cli($scfg, [ 'rm', $prefix.$name ], json => 0);
|
||||||
|
run_cli($scfg, [ 'create', '--parent', $prefix.$name.'@'.$snap, $prefix.$name ], json => 0);
|
||||||
|
return undef;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub volume_snapshot_delete
|
||||||
|
{
|
||||||
|
my ($class, $scfg, $storeid, $volname, $snap, $running) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
my ($vtype, $name, $vmid) = $class->parse_volname($volname);
|
||||||
|
run_cli($scfg, [ 'rm', $prefix.$name.'@'.$snap ], json => 0);
|
||||||
|
return undef;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub volume_snapshot_needs_fsfreeze
|
||||||
|
{
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub volume_has_feature
|
||||||
|
{
|
||||||
|
my ($class, $scfg, $feature, $storeid, $volname, $snapname, $running) = @_;
|
||||||
|
my $features = {
|
||||||
|
snapshot => { current => 1, snap => 1 },
|
||||||
|
clone => { base => 1, snap => 1 },
|
||||||
|
template => { current => 1 },
|
||||||
|
copy => { base => 1, current => 1, snap => 1 },
|
||||||
|
sparseinit => { base => 1, current => 1 },
|
||||||
|
rename => { current => 1 },
|
||||||
|
};
|
||||||
|
my ($vtype, $name, $vmid, $basename, $basevmid, $isBase) = $class->parse_volname($volname);
|
||||||
|
my $key = undef;
|
||||||
|
if ($snapname)
|
||||||
|
{
|
||||||
|
$key = 'snap';
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
$key = $isBase ? 'base' : 'current';
|
||||||
|
}
|
||||||
|
return 1 if $features->{$feature}->{$key};
|
||||||
|
return undef;
|
||||||
|
}
|
||||||
|
|
||||||
|
sub rename_volume
|
||||||
|
{
|
||||||
|
my ($class, $scfg, $storeid, $source_volname, $target_vmid, $target_volname) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
my (undef, $source_image, $source_vmid, $base_name, $base_vmid, undef, $format) =
|
||||||
|
$class->parse_volname($source_volname);
|
||||||
|
$target_volname = $class->find_free_diskname($storeid, $scfg, $target_vmid, $format) if !$target_volname;
|
||||||
|
run_cli($scfg, [ 'modify', '--rename', $prefix.$target_volname, $prefix.$source_image ], json => 0);
|
||||||
|
$base_name = $base_name ? "${base_name}/" : '';
|
||||||
|
return "${storeid}:${base_name}${target_volname}";
|
||||||
|
}
|
||||||
|
|
||||||
|
1;
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user