Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
278852b4d5 | ||
|
|
7b11c6e90d | ||
|
|
6251ce8b9a | ||
|
|
d4a42f61cf | ||
|
|
6b003bcc34 | ||
|
|
1c945bcb41 | ||
|
|
716527b184 | ||
|
|
8c0486bd76 | ||
|
|
d2cf271f64 | ||
|
|
d93b488e32 | ||
|
|
fbffec5abb | ||
|
|
b6eb8f2055 | ||
|
|
e373ea2163 | ||
|
|
5e12b4a1a5 | ||
|
|
78b067566f | ||
|
|
cf3abdb9e3 | ||
|
|
826b35b369 | ||
|
|
cdc730314b | ||
|
|
5248d7f324 | ||
|
|
4161f0bd01 | ||
|
|
5627977a9b | ||
|
|
ec9cfa76c1 | ||
|
|
0c88884576 | ||
|
|
ba7637d9ad | ||
|
|
ac1025c7a5 | ||
|
|
2a81cef78a | ||
|
|
bcf6a7c7d1 | ||
|
|
85c4be3957 | ||
|
|
cf2ba05e4b | ||
|
|
04c2f8d408 | ||
|
|
0e528ca8f3 | ||
|
|
c6f733b96a | ||
|
|
938ac09248 | ||
|
|
3becdbf5b9 | ||
|
|
9c49315fdf | ||
|
|
430d3cfb6f | ||
|
|
c491db699c | ||
|
|
e9d053e30f | ||
|
|
1be51f903c | ||
|
|
cd51f14a90 | ||
|
|
9b8107875f | ||
|
|
ca606570f7 | ||
|
|
22d094ccc6 | ||
|
|
fd84d84279 | ||
|
|
af2b1e28e3 | ||
|
|
9dda449f48 | ||
|
|
e6d4b32629 | ||
|
|
4de22a08e2 | ||
|
|
a403de46b3 | ||
|
|
ac20f605f6 | ||
|
|
51ecbadb12 | ||
|
|
6a4627b625 | ||
|
|
fd2b8b8792 | ||
|
|
f3d662bac7 | ||
|
|
553cc8ef87 | ||
|
|
67fdf1142b | ||
|
|
02f6e564a6 | ||
|
|
33d14061d6 | ||
|
|
677e755e4e | ||
|
|
11a972cbfb | ||
|
|
1834743a0e | ||
|
|
213f76c66c | ||
|
|
91698404a7 | ||
|
|
27bd38d95e | ||
|
|
ef0e61be1b | ||
|
|
26fb08d7da | ||
|
|
80fa3094b3 | ||
|
|
df931b1e17 | ||
|
|
906294ae9a | ||
|
|
15f69719e4 | ||
|
|
43aa4cfff6 | ||
|
|
6901227390 | ||
|
|
7f718feaf6 | ||
|
|
236ffbb24e | ||
|
|
0e300f4c50 | ||
|
|
6d82a3daa3 | ||
|
|
c0c01a8e57 | ||
|
|
334755e912 | ||
|
|
c16f955a51 | ||
|
|
c1dc14f5ee | ||
|
|
ec6a70bbd3 | ||
|
|
f236ed895a | ||
|
|
7d70c90196 | ||
|
|
0a04490043 | ||
|
|
dce7ffde6f | ||
|
|
f6bd1ff0e5 | ||
|
|
ac00a06757 | ||
|
|
155cfb3c73 | ||
|
|
126891126a | ||
|
|
547a394be6 | ||
|
|
1a511acead | ||
|
|
5576a0d9ff | ||
|
|
879e9a32d1 | ||
|
|
747fd5c121 | ||
|
|
b4aab7a78e | ||
|
|
de26a995fc | ||
|
|
8418a9ad7b | ||
|
|
27be4ee2fa | ||
|
|
00517e2bac | ||
|
|
e0a2615cbc | ||
|
|
63fe3c323a | ||
|
|
a88465df05 | ||
|
|
ad24be717a | ||
|
|
648e3b12f0 | ||
|
|
a675993c74 | ||
|
|
c9dfd0f67d | ||
|
|
84919a10a9 | ||
|
|
51ae4d6e24 | ||
|
|
572b20fedc | ||
|
|
4e2724b28f | ||
|
|
768b1675f8 | ||
|
|
38fa722725 | ||
|
|
e56d83fb7f | ||
|
|
ff95a85875 | ||
|
|
98203568a8 | ||
|
|
89df98ee08 | ||
|
|
0007a831b6 | ||
|
|
40517c335f | ||
|
|
c9f7308b6a | ||
|
|
85c7e3bde0 | ||
|
|
4fb55b3535 | ||
|
|
912aca11a3 | ||
|
|
7b454bd16c | ||
|
|
a0c8be46a4 | ||
|
|
53b4329fac | ||
|
|
a7f41c4a12 | ||
|
|
5d78057ac3 | ||
|
|
8efc5a353f | ||
|
|
603b26b896 | ||
|
|
a3b0fe0deb | ||
|
|
f504e356d5 | ||
|
|
4ed17b7070 | ||
|
|
1fd2819724 | ||
|
|
dcdabbc1ec | ||
|
|
625d5b7b9e | ||
|
|
9e507fd333 | ||
|
|
c2b5118127 | ||
|
|
4b926e2223 | ||
|
|
a5d9a6996a | ||
|
|
0ee03e7172 | ||
|
|
88b7d9afcd | ||
|
|
f271c8450c | ||
|
|
f78d7d4efc | ||
|
|
fdaf7c88ff | ||
|
|
2fb6eb0c30 | ||
|
|
36d2b56208 | ||
|
|
14b22f2ba9 | ||
|
|
fe8b1fe0cc | ||
|
|
1ec963e468 | ||
|
|
5100f822d8 | ||
|
|
7432494e88 | ||
|
|
d0c0f3ea39 | ||
|
|
f61190f31d | ||
|
|
3dc0ab5c33 | ||
|
|
de96efed2f | ||
|
|
87a5230798 | ||
|
|
0c5e6d4346 | ||
|
|
b278087410 | ||
|
|
a8e821b13b | ||
|
|
caa70317fa | ||
|
|
b8eaaabfe4 | ||
|
|
e4d80c415e | ||
|
|
553191c3ff | ||
|
|
ab385252b5 | ||
|
|
041185c673 | ||
|
|
b03ac80a57 | ||
|
|
2ba56074f9 | ||
|
|
4acfe149cb | ||
|
|
008ed5b269 | ||
|
|
4fffe0f032 | ||
|
|
a76d5ccc0d | ||
|
|
8ed1e180e0 | ||
|
|
8832fc3b14 | ||
|
|
0134934c99 | ||
|
|
2e36f292bd | ||
|
|
bcc6419760 | ||
|
|
dd5941b9a4 | ||
|
|
4005b88865 | ||
|
|
280b5cd675 | ||
|
|
e5c505eaf4 | ||
|
|
c1d244d4f0 | ||
|
|
9b264a212f | ||
|
|
ff7f5cb4f4 | ||
|
|
25ecca7625 | ||
|
|
99c4244004 | ||
|
|
9949b9fb4e | ||
|
|
e6881ad1d5 | ||
|
|
b30635b932 | ||
|
|
0c1154833c | ||
|
|
c227bb05b6 | ||
|
|
dd85315f22 | ||
|
|
47d2f4e0be | ||
|
|
2a5028d17f | ||
|
|
07915c2881 | ||
|
|
79141eb383 | ||
|
|
f7cbb6ed56 | ||
|
|
8f8172db99 | ||
|
|
94be147e80 |
+109
-1
@@ -63,7 +63,7 @@ jobs:
|
|||||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
steps:
|
steps:
|
||||||
# leak sanitizer sometimes crashes
|
# leak sanitizer sometimes crashes
|
||||||
- run: cd /root/vitastor/build && ASAN_OPTIONS=detect_leaks=0 make -j16 test
|
- run: cd /root/vitastor/build && ASAN_OPTIONS=detect_leaks=0 make -j16 build_tests test
|
||||||
|
|
||||||
npm_lint:
|
npm_lint:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
@@ -306,6 +306,78 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_dump_load:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_dump_load.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_dump_load_32k:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: TEST_NAME=32k OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k" OFFSET_ARGS="$OSD_ARGS" /root/vitastor/tests/test_dump_load.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_old_dump_load:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_dump_load.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_dump_load_old_32k:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: TEST_NAME=old_32k OLD=1 OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k" OFFSET_ARGS="$OSD_ARGS" /root/vitastor/tests/test_dump_load.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_old_interrupted_rebalance:
|
test_old_interrupted_rebalance:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -1458,6 +1530,24 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_resize_last:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_resize_last.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_resize_auto:
|
test_resize_auto:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -1494,6 +1584,24 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_old_resize_last:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: OLD=1 /root/vitastor/tests/test_resize_last.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_old_resize_auto:
|
test_old_resize_auto:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
|
|||||||
+7
-7
@@ -1,20 +1,20 @@
|
|||||||
cmake_minimum_required(VERSION 2.8.12)
|
cmake_minimum_required(VERSION 2.8...3.30)
|
||||||
|
|
||||||
project(vitastor)
|
project(vitastor)
|
||||||
|
|
||||||
set(VITASTOR_VERSION "3.0.6")
|
set(VITASTOR_VERSION "3.0.14")
|
||||||
|
|
||||||
include(CTest)
|
include(CTest)
|
||||||
|
|
||||||
add_custom_target(build_tests)
|
add_custom_target(build_tests)
|
||||||
add_custom_target(test
|
set_property(TEST PROPERTY ENVIRONMENT LSAN_OPTIONS=suppressions=${CMAKE_CURRENT_BINARY_DIR}/lsan-suppress.txt)
|
||||||
COMMAND
|
add_test(gen_lsan_suppress
|
||||||
echo leak:tcmalloc > ${CMAKE_CURRENT_BINARY_DIR}/lsan-suppress.txt &&
|
${CMAKE_COMMAND} -E echo leak:tcmalloc > "${CMAKE_CURRENT_BINARY_DIR}/lsan-suppress.txt"
|
||||||
env LSAN_OPTIONS=suppressions=${CMAKE_CURRENT_BINARY_DIR}/lsan-suppress.txt ${CMAKE_CTEST_COMMAND}
|
|
||||||
)
|
)
|
||||||
|
set_tests_properties(gen_lsan_suppress PROPERTIES FIXTURES_SETUP f_lsan_suppress)
|
||||||
|
set_property(TEST PROPERTY FIXTURES_REQUIRED f_lsan_suppress)
|
||||||
# make -j16 -C ../../build test_heap && ../../build/src/test/test_heap
|
# make -j16 -C ../../build test_heap && ../../build/src/test/test_heap
|
||||||
# make -j16 -C ../../build test_heap && rm -f $(find ../../build -name '*.gcda') && ctest -V -T test -T coverage -R heap --test-dir ../../build && (cd ../../build; gcovr -f ../src --html --html-nested -o coverage/index.html; cd ../src/test)
|
# make -j16 -C ../../build test_heap && rm -f $(find ../../build -name '*.gcda') && ctest -V -T test -T coverage -R heap --test-dir ../../build && (cd ../../build; gcovr -f ../src --html --html-nested -o coverage/index.html; cd ../src/test)
|
||||||
# make -j16 -C ../../build test_blockstore && rm -f $(find ../../build -name '*.gcda') && ctest -V -T test -T coverage -R blockstore --test-dir ../../build && (cd ../../build; gcovr -f ../src --html --html-nested -o coverage/index.html; cd ../src/test)
|
# make -j16 -C ../../build test_blockstore && rm -f $(find ../../build -name '*.gcda') && ctest -V -T test -T coverage -R blockstore --test-dir ../../build && (cd ../../build; gcovr -f ../src --html --html-nested -o coverage/index.html; cd ../src/test)
|
||||||
# kcov --include-path=../../../src ../../kcov ./test_blockstore
|
# kcov --include-path=../../../src ../../kcov ./test_blockstore
|
||||||
add_dependencies(test build_tests)
|
|
||||||
add_subdirectory(src)
|
add_subdirectory(src)
|
||||||
|
|||||||
+1
-1
Submodule cpp-btree updated: 8de8b467ac...431d2e1d35
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
VITASTOR_VERSION ?= v3.0.6
|
VITASTOR_VERSION ?= v3.0.14
|
||||||
|
|
||||||
all: build push
|
all: build push
|
||||||
|
|
||||||
|
|||||||
@@ -49,7 +49,7 @@ spec:
|
|||||||
capabilities:
|
capabilities:
|
||||||
add: ["SYS_ADMIN"]
|
add: ["SYS_ADMIN"]
|
||||||
allowPrivilegeEscalation: true
|
allowPrivilegeEscalation: true
|
||||||
image: vitalif/vitastor-csi:v3.0.6
|
image: vitalif/vitastor-csi:v3.0.14
|
||||||
args:
|
args:
|
||||||
- "--node=$(NODE_ID)"
|
- "--node=$(NODE_ID)"
|
||||||
- "--endpoint=$(CSI_ENDPOINT)"
|
- "--endpoint=$(CSI_ENDPOINT)"
|
||||||
|
|||||||
@@ -121,7 +121,7 @@ spec:
|
|||||||
privileged: true
|
privileged: true
|
||||||
capabilities:
|
capabilities:
|
||||||
add: ["SYS_ADMIN"]
|
add: ["SYS_ADMIN"]
|
||||||
image: vitalif/vitastor-csi:v3.0.6
|
image: vitalif/vitastor-csi:v3.0.14
|
||||||
args:
|
args:
|
||||||
- "--node=$(NODE_ID)"
|
- "--node=$(NODE_ID)"
|
||||||
- "--endpoint=$(CSI_ENDPOINT)"
|
- "--endpoint=$(CSI_ENDPOINT)"
|
||||||
|
|||||||
+1
-1
@@ -5,7 +5,7 @@ package vitastor
|
|||||||
|
|
||||||
const (
|
const (
|
||||||
vitastorCSIDriverName = "csi.vitastor.io"
|
vitastorCSIDriverName = "csi.vitastor.io"
|
||||||
vitastorCSIDriverVersion = "3.0.6"
|
vitastorCSIDriverVersion = "3.0.14"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Config struct fills the parameters of request or user input
|
// Config struct fills the parameters of request or user input
|
||||||
|
|||||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
vitastor (3.0.6-1) unstable; urgency=medium
|
vitastor (3.0.14-1) unstable; urgency=medium
|
||||||
|
|
||||||
* Bugfixes
|
* Bugfixes
|
||||||
|
|
||||||
|
|||||||
Vendored
+1
-1
@@ -44,7 +44,7 @@ curl -s https://git.yourcmc.ru/vitalif/antietcd/archive/master.tar.gz | tar -zx
|
|||||||
curl -s https://git.yourcmc.ru/vitalif/tinyraft/archive/master.tar.gz | tar -zx
|
curl -s https://git.yourcmc.ru/vitalif/tinyraft/archive/master.tar.gz | tar -zx
|
||||||
|
|
||||||
cd /root/vitastor/packages/vitastor-$REL
|
cd /root/vitastor/packages/vitastor-$REL
|
||||||
if [[ "$REL" = "trixie" && -e ../vitastor-bookworm/vitastor_$VER.orig.tar.xz ]]; then
|
if [[ ( "$REL" = "trixie" || "$REL" = "resolute" ) && -e ../vitastor-bookworm/vitastor_$VER.orig.tar.xz ]]; then
|
||||||
# Fucking shit, archives differ between bookworm (xz 5.4.1) and trixie (xz 5.8.1)
|
# Fucking shit, archives differ between bookworm (xz 5.4.1) and trixie (xz 5.8.1)
|
||||||
cp ../vitastor-bookworm/vitastor_$VER.orig.tar.xz .
|
cp ../vitastor-bookworm/vitastor_$VER.orig.tar.xz .
|
||||||
else
|
else
|
||||||
|
|||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
VITASTOR_VERSION ?= v3.0.6
|
VITASTOR_VERSION ?= v3.0.14
|
||||||
|
|
||||||
all: build push
|
all: build push
|
||||||
|
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
#
|
#
|
||||||
|
|
||||||
# Desired Vitastor version
|
# Desired Vitastor version
|
||||||
VITASTOR_VERSION=v3.0.6
|
VITASTOR_VERSION=v3.0.14
|
||||||
|
|
||||||
# Additional arguments for all containers
|
# Additional arguments for all containers
|
||||||
# For example, you may want to specify a custom logging driver here
|
# For example, you may want to specify a custom logging driver here
|
||||||
|
|||||||
@@ -70,6 +70,7 @@ with an OSD restart or, for some of them, even without restarting by updating co
|
|||||||
- [use_atomic_flag](#use_atomic_flag)
|
- [use_atomic_flag](#use_atomic_flag)
|
||||||
- [pg_reshard_chunk_size](#pg_reshard_chunk_size)
|
- [pg_reshard_chunk_size](#pg_reshard_chunk_size)
|
||||||
- [pg_reshard_chunk_pause_ms](#pg_reshard_chunk_pause_ms)
|
- [pg_reshard_chunk_pause_ms](#pg_reshard_chunk_pause_ms)
|
||||||
|
- [gc_on_start](#gc_on_start)
|
||||||
|
|
||||||
## bind_address
|
## bind_address
|
||||||
|
|
||||||
@@ -753,3 +754,9 @@ This option sets the maximum number of object is a chunk. Moving 100k objects us
|
|||||||
- Default: 100
|
- Default: 100
|
||||||
|
|
||||||
This option sets the interval between handling two PG count change chunks.
|
This option sets the interval between handling two PG count change chunks.
|
||||||
|
|
||||||
|
## gc_on_start
|
||||||
|
|
||||||
|
- Type: boolean
|
||||||
|
|
||||||
|
Forcibly clean all garbage entries in the new store on every OSD restart.
|
||||||
|
|||||||
@@ -71,6 +71,7 @@
|
|||||||
- [use_atomic_flag](#use_atomic_flag)
|
- [use_atomic_flag](#use_atomic_flag)
|
||||||
- [pg_reshard_chunk_size](#pg_reshard_chunk_size)
|
- [pg_reshard_chunk_size](#pg_reshard_chunk_size)
|
||||||
- [pg_reshard_chunk_pause_ms](#pg_reshard_chunk_pause_ms)
|
- [pg_reshard_chunk_pause_ms](#pg_reshard_chunk_pause_ms)
|
||||||
|
- [gc_on_start](#gc_on_start)
|
||||||
|
|
||||||
## bind_address
|
## bind_address
|
||||||
|
|
||||||
@@ -793,3 +794,9 @@ pg_minsize OSD во время переключений, что может по
|
|||||||
- Значение по умолчанию: 100
|
- Значение по умолчанию: 100
|
||||||
|
|
||||||
Данная опция задаёт интервал между обработкой двух порций изменения числа PG пулов.
|
Данная опция задаёт интервал между обработкой двух порций изменения числа PG пулов.
|
||||||
|
|
||||||
|
## gc_on_start
|
||||||
|
|
||||||
|
- Тип: булево (да/нет)
|
||||||
|
|
||||||
|
Принудительно очищать все мусорные записи в новом хранилище при каждом запуске OSD.
|
||||||
|
|||||||
@@ -938,3 +938,7 @@
|
|||||||
This option sets the interval between handling two PG count change chunks.
|
This option sets the interval between handling two PG count change chunks.
|
||||||
info_ru: |
|
info_ru: |
|
||||||
Данная опция задаёт интервал между обработкой двух порций изменения числа PG пулов.
|
Данная опция задаёт интервал между обработкой двух порций изменения числа PG пулов.
|
||||||
|
- name: gc_on_start
|
||||||
|
type: bool
|
||||||
|
info: Forcibly clean all garbage entries in the new store on every OSD restart.
|
||||||
|
info_ru: Принудительно очищать все мусорные записи в новом хранилище при каждом запуске OSD.
|
||||||
|
|||||||
@@ -26,9 +26,9 @@ at Vitastor Kubernetes operator: https://github.com/Antilles7227/vitastor-operat
|
|||||||
The instruction is very simple.
|
The instruction is very simple.
|
||||||
|
|
||||||
1. Download a Docker image of the desired version: \
|
1. Download a Docker image of the desired version: \
|
||||||
`docker pull vitalif/vitastor:v3.0.6`
|
`docker pull vitalif/vitastor:v3.0.14`
|
||||||
2. Install scripts to the host system: \
|
2. Install scripts to the host system: \
|
||||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.6 install.sh`
|
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.14 install.sh`
|
||||||
3. Reload udev rules: \
|
3. Reload udev rules: \
|
||||||
`udevadm control --reload-rules`
|
`udevadm control --reload-rules`
|
||||||
4. Enable the vitastor-host service: \
|
4. Enable the vitastor-host service: \
|
||||||
|
|||||||
@@ -25,9 +25,9 @@ Vitastor можно установить в Docker/Podman. При этом etcd,
|
|||||||
Инструкция по установке максимально простая.
|
Инструкция по установке максимально простая.
|
||||||
|
|
||||||
1. Скачайте Docker-образ желаемой версии: \
|
1. Скачайте Docker-образ желаемой версии: \
|
||||||
`docker pull vitalif/vitastor:v3.0.6`
|
`docker pull vitalif/vitastor:v3.0.14`
|
||||||
2. Установите скрипты в хост-систему командой: \
|
2. Установите скрипты в хост-систему командой: \
|
||||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.6 install.sh`
|
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.14 install.sh`
|
||||||
3. Перезагрузите правила udev: \
|
3. Перезагрузите правила udev: \
|
||||||
`udevadm control --reload-rules`
|
`udevadm control --reload-rules`
|
||||||
4. Включите сервис vitastor-host: \
|
4. Включите сервис vitastor-host: \
|
||||||
|
|||||||
@@ -17,6 +17,7 @@
|
|||||||
- Debian 10 (Buster): `deb https://vitastor.io/debian buster main`
|
- Debian 10 (Buster): `deb https://vitastor.io/debian buster main`
|
||||||
- Ubuntu 22.04 (Jammy): `deb https://vitastor.io/debian jammy main`
|
- Ubuntu 22.04 (Jammy): `deb https://vitastor.io/debian jammy main`
|
||||||
- Ubuntu 24.04 (Noble): `deb https://vitastor.io/debian noble main`
|
- Ubuntu 24.04 (Noble): `deb https://vitastor.io/debian noble main`
|
||||||
|
- Ubuntu 26.04 (Resolute): `deb https://vitastor.io/debian resolute main`
|
||||||
- Add `-oldstable` to bookworm/bullseye/buster in this line to install the last
|
- Add `-oldstable` to bookworm/bullseye/buster in this line to install the last
|
||||||
stable version from 0.9.x branch instead of 1.x
|
stable version from 0.9.x branch instead of 1.x
|
||||||
- To always prefer vitastor-patched QEMU and Libvirt versions, add the following to `/etc/apt/preferences`:
|
- To always prefer vitastor-patched QEMU and Libvirt versions, add the following to `/etc/apt/preferences`:
|
||||||
|
|||||||
@@ -17,6 +17,7 @@
|
|||||||
- Debian 10 (Buster): `deb https://vitastor.io/debian buster main`
|
- Debian 10 (Buster): `deb https://vitastor.io/debian buster main`
|
||||||
- Ubuntu 22.04 (Jammy): `deb https://vitastor.io/debian jammy main`
|
- Ubuntu 22.04 (Jammy): `deb https://vitastor.io/debian jammy main`
|
||||||
- Ubuntu 24.04 (Noble): `deb https://vitastor.io/debian noble main`
|
- Ubuntu 24.04 (Noble): `deb https://vitastor.io/debian noble main`
|
||||||
|
- Ubuntu 26.04 (Resolute): `deb https://vitastor.io/debian resolute main`
|
||||||
- Добавьте `-oldstable` к слову bookworm/bullseye/buster в этой строке, чтобы
|
- Добавьте `-oldstable` к слову bookworm/bullseye/buster в этой строке, чтобы
|
||||||
установить последнюю стабильную версию из ветки 0.9.x вместо 1.x
|
установить последнюю стабильную версию из ветки 0.9.x вместо 1.x
|
||||||
- Чтобы всегда предпочитались версии пакетов QEMU и Libvirt с патчами Vitastor, добавьте в `/etc/apt/preferences`:
|
- Чтобы всегда предпочитались версии пакетов QEMU и Libvirt с патчами Vitastor, добавьте в `/etc/apt/preferences`:
|
||||||
|
|||||||
@@ -16,8 +16,7 @@
|
|||||||
designated initializers support from C++20
|
designated initializers support from C++20
|
||||||
- CMake
|
- CMake
|
||||||
- jerasure headers and libraries
|
- jerasure headers and libraries
|
||||||
- ISA-L, libibverbs and librdmacm headers and libraries (optional)
|
- ISA-L, libibverbs, librdmacm, libnl3 headers and libraries (optional)
|
||||||
- tcmalloc (google-perftools-dev)
|
|
||||||
|
|
||||||
## Basic instructions
|
## Basic instructions
|
||||||
|
|
||||||
|
|||||||
@@ -16,8 +16,7 @@
|
|||||||
назначенных инициализаторов (designated initializers) из C++20
|
назначенных инициализаторов (designated initializers) из C++20
|
||||||
- CMake
|
- CMake
|
||||||
- Заголовки и библиотеки jerasure
|
- Заголовки и библиотеки jerasure
|
||||||
- Опционально - заголовки и библиотеки ISA-L, libibverbs, librdmacm
|
- Опционально - заголовки и библиотеки ISA-L, libibverbs, librdmacm, libnl3
|
||||||
- tcmalloc (google-perftools-dev)
|
|
||||||
|
|
||||||
## Базовая инструкция
|
## Базовая инструкция
|
||||||
|
|
||||||
|
|||||||
@@ -262,3 +262,4 @@ Options:
|
|||||||
| `--logfile <FILE>` | log to the specified file |
|
| `--logfile <FILE>` | log to the specified file |
|
||||||
| `--enforce 1` | enforce permissions at the server side (no by default) |
|
| `--enforce 1` | enforce permissions at the server side (no by default) |
|
||||||
| `--foreground 1` | stay in foreground, do not daemonize |
|
| `--foreground 1` | stay in foreground, do not daemonize |
|
||||||
|
| `--trace` | trace all NFS requests |
|
||||||
|
|||||||
@@ -274,3 +274,4 @@ VitastorFS из GPUDirect.
|
|||||||
| `--logfile <FILE>` | записывать логи в заданный файл |
|
| `--logfile <FILE>` | записывать логи в заданный файл |
|
||||||
| `--enforce 1` | проверять права доступа на стороне сервера (по умолчанию нет) |
|
| `--enforce 1` | проверять права доступа на стороне сервера (по умолчанию нет) |
|
||||||
| `--foreground 1` | не уходить в фон после запуска |
|
| `--foreground 1` | не уходить в фон после запуска |
|
||||||
|
| `--trace` | логгировать все запросы NFS |
|
||||||
|
|||||||
+1
-1
Submodule json11 updated: fd37016cf8...edcd85b8bd
@@ -112,9 +112,10 @@ function make_cyclic(pgs, parity_space)
|
|||||||
{
|
{
|
||||||
if (parity_space > 1)
|
if (parity_space > 1)
|
||||||
{
|
{
|
||||||
for (const pg in pgs)
|
for (const id in pgs)
|
||||||
{
|
{
|
||||||
for (let i = 1; i < pg.size; i++)
|
const pg = pgs[id];
|
||||||
|
for (let i = 1; i < pg.length; i++)
|
||||||
{
|
{
|
||||||
const cyclic = [ ...pg.slice(i), ...pg.slice(0, i) ];
|
const cyclic = [ ...pg.slice(i), ...pg.slice(0, i) ];
|
||||||
pgs['pg_'+cyclic.join('_')] = cyclic;
|
pgs['pg_'+cyclic.join('_')] = cyclic;
|
||||||
|
|||||||
+2
-2
@@ -627,7 +627,7 @@ class Mon
|
|||||||
if (this.state.pg.history[pool_id] &&
|
if (this.state.pg.history[pool_id] &&
|
||||||
this.state.pg.history[pool_id][pg])
|
this.state.pg.history[pool_id][pg])
|
||||||
{
|
{
|
||||||
pg_history[pg-1] = this.state.pg.history[pool_id][pg];
|
pg_history[pg-1] = JSON.parse(JSON.stringify(this.state.pg.history[pool_id][pg]));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
const real_prev_pgs = [];
|
const real_prev_pgs = [];
|
||||||
@@ -719,7 +719,7 @@ class Mon
|
|||||||
this.next_recheck_timer = null;
|
this.next_recheck_timer = null;
|
||||||
this.next_recheck_at = 0;
|
this.next_recheck_at = 0;
|
||||||
this.schedule_recheck();
|
this.schedule_recheck();
|
||||||
}, now-this.next_recheck_at);
|
}, (this.next_recheck_at-now)*1000);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "vitastor-mon",
|
"name": "vitastor-mon",
|
||||||
"version": "3.0.6",
|
"version": "3.0.14",
|
||||||
"description": "Vitastor SDS monitor service",
|
"description": "Vitastor SDS monitor service",
|
||||||
"main": "mon-main.js",
|
"main": "mon-main.js",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
|
|||||||
+2
-2
@@ -84,7 +84,7 @@ function scale_pg_history(prev_pg_history, prev_pgs, new_pgs)
|
|||||||
finish_pg_history(merged_history[1]);
|
finish_pg_history(merged_history[1]);
|
||||||
for (let i = 0; i < new_pg_count; i++)
|
for (let i = 0; i < new_pg_count; i++)
|
||||||
{
|
{
|
||||||
new_pg_history[i] = { ...merged_history[1] };
|
new_pg_history[i] = JSON.parse(JSON.stringify(merged_history[1]));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Mark history keys for removed PGs as removed
|
// Mark history keys for removed PGs as removed
|
||||||
@@ -102,7 +102,7 @@ function scale_pg_count(prev_pgs, new_pg_count)
|
|||||||
{
|
{
|
||||||
for (let i = prev_pgs.length; i < new_pg_count; i++)
|
for (let i = prev_pgs.length; i < new_pg_count; i++)
|
||||||
{
|
{
|
||||||
prev_pgs[i] = prev_pgs[i % prev_pgs.length];
|
prev_pgs[i] = [ ...prev_pgs[i % prev_pgs.length] ];
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
else if (prev_pgs.length > new_pg_count)
|
else if (prev_pgs.length > new_pg_count)
|
||||||
|
|||||||
@@ -37,6 +37,7 @@ function derive_osd_stats(st, prev, prev_diff)
|
|||||||
const n = c.count - BigInt(pr && pr.count||0);
|
const n = c.count - BigInt(pr && pr.count||0);
|
||||||
diff.recovery_stats[op] = { ...c, bps: n > 0 ? b*1000n/timediff : 0n, iops: n > 0 ? n*1000n/timediff : 0n };
|
diff.recovery_stats[op] = { ...c, bps: n > 0 ? b*1000n/timediff : 0n, iops: n > 0 ? n*1000n/timediff : 0n };
|
||||||
}
|
}
|
||||||
|
diff.inode_stats = {};
|
||||||
for (const pool_id in st.inode_stats||{})
|
for (const pool_id in st.inode_stats||{})
|
||||||
{
|
{
|
||||||
diff.inode_stats[pool_id] = {};
|
diff.inode_stats[pool_id] = {};
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "vitastor",
|
"name": "vitastor",
|
||||||
"version": "3.0.6",
|
"version": "3.0.14",
|
||||||
"description": "Low-level native bindings to Vitastor client library",
|
"description": "Low-level native bindings to Vitastor client library",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"keywords": [
|
"keywords": [
|
||||||
|
|||||||
@@ -50,7 +50,7 @@ from cinder.volume import configuration
|
|||||||
from cinder.volume import driver
|
from cinder.volume import driver
|
||||||
from cinder.volume import volume_utils
|
from cinder.volume import volume_utils
|
||||||
|
|
||||||
VITASTOR_VERSION = '3.0.6'
|
VITASTOR_VERSION = '3.0.14'
|
||||||
|
|
||||||
LOG = logging.getLogger(__name__)
|
LOG = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,637 @@
|
|||||||
|
diff --git a/include/libvirt/libvirt-storage.h b/include/libvirt/libvirt-storage.h
|
||||||
|
index aaad4a3da1..5f5daa8341 100644
|
||||||
|
--- a/include/libvirt/libvirt-storage.h
|
||||||
|
+++ b/include/libvirt/libvirt-storage.h
|
||||||
|
@@ -326,6 +326,7 @@ typedef enum {
|
||||||
|
VIR_CONNECT_LIST_STORAGE_POOLS_ZFS = 1 << 17, /* (Since: 1.2.8) */
|
||||||
|
VIR_CONNECT_LIST_STORAGE_POOLS_VSTORAGE = 1 << 18, /* (Since: 3.1.0) */
|
||||||
|
VIR_CONNECT_LIST_STORAGE_POOLS_ISCSI_DIRECT = 1 << 19, /* (Since: 5.6.0) */
|
||||||
|
+ VIR_CONNECT_LIST_STORAGE_POOLS_VITASTOR = 1 << 20, /* (Since: 5.0.0) */
|
||||||
|
} virConnectListAllStoragePoolsFlags;
|
||||||
|
|
||||||
|
int virConnectListAllStoragePools(virConnectPtr conn,
|
||||||
|
diff --git a/src/conf/domain_conf.c b/src/conf/domain_conf.c
|
||||||
|
index 9ca5c2450c..cc52f00c0c 100644
|
||||||
|
--- a/src/conf/domain_conf.c
|
||||||
|
+++ b/src/conf/domain_conf.c
|
||||||
|
@@ -7453,7 +7453,8 @@ virDomainDiskSourceNetworkParse(xmlNodePtr node,
|
||||||
|
src->configFile = virXPathString("string(./config/@file)", ctxt);
|
||||||
|
|
||||||
|
if (src->protocol == VIR_STORAGE_NET_PROTOCOL_HTTP ||
|
||||||
|
- src->protocol == VIR_STORAGE_NET_PROTOCOL_HTTPS)
|
||||||
|
+ src->protocol == VIR_STORAGE_NET_PROTOCOL_HTTPS ||
|
||||||
|
+ src->protocol == VIR_STORAGE_NET_PROTOCOL_VITASTOR)
|
||||||
|
src->query = virXMLPropString(node, "query");
|
||||||
|
|
||||||
|
if (virDomainStorageNetworkParseHosts(node, ctxt, &src->hosts, &src->nhosts) < 0)
|
||||||
|
@@ -32187,6 +32188,7 @@ virDomainStorageSourceTranslateSourcePool(virStorageSource *src,
|
||||||
|
|
||||||
|
case VIR_STORAGE_POOL_MPATH:
|
||||||
|
case VIR_STORAGE_POOL_RBD:
|
||||||
|
+ case VIR_STORAGE_POOL_VITASTOR:
|
||||||
|
case VIR_STORAGE_POOL_SHEEPDOG:
|
||||||
|
case VIR_STORAGE_POOL_GLUSTER:
|
||||||
|
case VIR_STORAGE_POOL_LAST:
|
||||||
|
diff --git a/src/conf/domain_validate.c b/src/conf/domain_validate.c
|
||||||
|
index 7346a61731..83e94d762e 100644
|
||||||
|
--- a/src/conf/domain_validate.c
|
||||||
|
+++ b/src/conf/domain_validate.c
|
||||||
|
@@ -520,6 +520,7 @@ virDomainDiskDefValidateSourceChainOne(const virStorageSource *src)
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_RBD:
|
||||||
|
break;
|
||||||
|
|
||||||
|
+ case VIR_STORAGE_NET_PROTOCOL_VITASTOR:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_NBD:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_SHEEPDOG:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_GLUSTER:
|
||||||
|
@@ -592,7 +593,7 @@ virDomainDiskDefValidateSourceChainOne(const virStorageSource *src)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
- /* internal snapshots and config files are currently supported only with rbd: */
|
||||||
|
+ /* internal snapshots are currently supported only with rbd: */
|
||||||
|
if (virStorageSourceGetActualType(src) != VIR_STORAGE_TYPE_NETWORK &&
|
||||||
|
src->protocol != VIR_STORAGE_NET_PROTOCOL_RBD) {
|
||||||
|
if (src->snapshot) {
|
||||||
|
@@ -600,10 +601,14 @@ virDomainDiskDefValidateSourceChainOne(const virStorageSource *src)
|
||||||
|
_("<snapshot> element is currently supported only with 'rbd' disks"));
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
-
|
||||||
|
+ }
|
||||||
|
+ /* config files are currently supported only with rbd and vitastor: */
|
||||||
|
+ if (virStorageSourceGetActualType(src) != VIR_STORAGE_TYPE_NETWORK &&
|
||||||
|
+ src->protocol != VIR_STORAGE_NET_PROTOCOL_RBD &&
|
||||||
|
+ src->protocol != VIR_STORAGE_NET_PROTOCOL_VITASTOR) {
|
||||||
|
if (src->configFile) {
|
||||||
|
virReportError(VIR_ERR_XML_ERROR, "%s",
|
||||||
|
- _("<config> element is currently supported only with 'rbd' disks"));
|
||||||
|
+ _("<config> element is currently supported only with 'rbd' and 'vitastor' disks"));
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
diff --git a/src/conf/schemas/domaincommon.rng b/src/conf/schemas/domaincommon.rng
|
||||||
|
index 114dd3f96f..c71f9a3277 100644
|
||||||
|
--- a/src/conf/schemas/domaincommon.rng
|
||||||
|
+++ b/src/conf/schemas/domaincommon.rng
|
||||||
|
@@ -2093,6 +2093,35 @@
|
||||||
|
</element>
|
||||||
|
</define>
|
||||||
|
|
||||||
|
+ <define name="diskSourceNetworkProtocolVitastor">
|
||||||
|
+ <element name="source">
|
||||||
|
+ <interleave>
|
||||||
|
+ <attribute name="protocol">
|
||||||
|
+ <value>vitastor</value>
|
||||||
|
+ </attribute>
|
||||||
|
+ <ref name="diskSourceCommon"/>
|
||||||
|
+ <optional>
|
||||||
|
+ <attribute name="name"/>
|
||||||
|
+ </optional>
|
||||||
|
+ <optional>
|
||||||
|
+ <attribute name="query"/>
|
||||||
|
+ </optional>
|
||||||
|
+ <zeroOrMore>
|
||||||
|
+ <ref name="diskSourceNetworkHost"/>
|
||||||
|
+ </zeroOrMore>
|
||||||
|
+ <optional>
|
||||||
|
+ <element name="config">
|
||||||
|
+ <attribute name="file">
|
||||||
|
+ <ref name="absFilePath"/>
|
||||||
|
+ </attribute>
|
||||||
|
+ <empty/>
|
||||||
|
+ </element>
|
||||||
|
+ </optional>
|
||||||
|
+ <empty/>
|
||||||
|
+ </interleave>
|
||||||
|
+ </element>
|
||||||
|
+ </define>
|
||||||
|
+
|
||||||
|
<define name="diskSourceNetworkProtocolISCSI">
|
||||||
|
<element name="source">
|
||||||
|
<attribute name="protocol">
|
||||||
|
@@ -2443,6 +2472,7 @@
|
||||||
|
<ref name="diskSourceNetworkProtocolSimple"/>
|
||||||
|
<ref name="diskSourceNetworkProtocolVxHS"/>
|
||||||
|
<ref name="diskSourceNetworkProtocolNFS"/>
|
||||||
|
+ <ref name="diskSourceNetworkProtocolVitastor"/>
|
||||||
|
</choice>
|
||||||
|
</define>
|
||||||
|
|
||||||
|
diff --git a/src/conf/storage_conf.c b/src/conf/storage_conf.c
|
||||||
|
index 1dc9365bf2..a8a736be81 100644
|
||||||
|
--- a/src/conf/storage_conf.c
|
||||||
|
+++ b/src/conf/storage_conf.c
|
||||||
|
@@ -56,7 +56,7 @@ VIR_ENUM_IMPL(virStoragePool,
|
||||||
|
"logical", "disk", "iscsi",
|
||||||
|
"iscsi-direct", "scsi", "mpath",
|
||||||
|
"rbd", "sheepdog", "gluster",
|
||||||
|
- "zfs", "vstorage",
|
||||||
|
+ "zfs", "vstorage", "vitastor",
|
||||||
|
);
|
||||||
|
|
||||||
|
VIR_ENUM_IMPL(virStoragePoolFormatFileSystem,
|
||||||
|
@@ -242,6 +242,18 @@ static virStoragePoolTypeInfo poolTypeInfo[] = {
|
||||||
|
.formatToString = virStorageFileFormatTypeToString,
|
||||||
|
}
|
||||||
|
},
|
||||||
|
+ {.poolType = VIR_STORAGE_POOL_VITASTOR,
|
||||||
|
+ .poolOptions = {
|
||||||
|
+ .flags = (VIR_STORAGE_POOL_SOURCE_HOST |
|
||||||
|
+ VIR_STORAGE_POOL_SOURCE_NETWORK |
|
||||||
|
+ VIR_STORAGE_POOL_SOURCE_NAME),
|
||||||
|
+ },
|
||||||
|
+ .volOptions = {
|
||||||
|
+ .defaultFormat = VIR_STORAGE_FILE_RAW,
|
||||||
|
+ .formatFromString = virStorageVolumeFormatFromString,
|
||||||
|
+ .formatToString = virStorageFileFormatTypeToString,
|
||||||
|
+ }
|
||||||
|
+ },
|
||||||
|
{.poolType = VIR_STORAGE_POOL_SHEEPDOG,
|
||||||
|
.poolOptions = {
|
||||||
|
.flags = (VIR_STORAGE_POOL_SOURCE_HOST |
|
||||||
|
@@ -538,6 +550,11 @@ virStoragePoolDefParseSource(xmlXPathContextPtr ctxt,
|
||||||
|
_("element 'name' is mandatory for RBD pool"));
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
+ if (pool_type == VIR_STORAGE_POOL_VITASTOR && source->name == NULL) {
|
||||||
|
+ virReportError(VIR_ERR_XML_ERROR, "%s",
|
||||||
|
+ _("element 'name' is mandatory for Vitastor pool"));
|
||||||
|
+ return -1;
|
||||||
|
+ }
|
||||||
|
|
||||||
|
if (options->formatFromString) {
|
||||||
|
g_autofree char *format = NULL;
|
||||||
|
@@ -1127,6 +1144,7 @@ virStoragePoolDefFormatBuf(virBuffer *buf,
|
||||||
|
/* RBD, Sheepdog, Gluster and Iscsi-direct devices are not local block devs nor
|
||||||
|
* files, so they don't have a target */
|
||||||
|
if (def->type != VIR_STORAGE_POOL_RBD &&
|
||||||
|
+ def->type != VIR_STORAGE_POOL_VITASTOR &&
|
||||||
|
def->type != VIR_STORAGE_POOL_SHEEPDOG &&
|
||||||
|
def->type != VIR_STORAGE_POOL_GLUSTER &&
|
||||||
|
def->type != VIR_STORAGE_POOL_ISCSI_DIRECT) {
|
||||||
|
diff --git a/src/conf/storage_conf.h b/src/conf/storage_conf.h
|
||||||
|
index fc67957cfe..720c07ef74 100644
|
||||||
|
--- a/src/conf/storage_conf.h
|
||||||
|
+++ b/src/conf/storage_conf.h
|
||||||
|
@@ -103,6 +103,7 @@ typedef enum {
|
||||||
|
VIR_STORAGE_POOL_GLUSTER, /* Gluster device */
|
||||||
|
VIR_STORAGE_POOL_ZFS, /* ZFS */
|
||||||
|
VIR_STORAGE_POOL_VSTORAGE, /* Virtuozzo Storage */
|
||||||
|
+ VIR_STORAGE_POOL_VITASTOR, /* Vitastor */
|
||||||
|
|
||||||
|
VIR_STORAGE_POOL_LAST,
|
||||||
|
} virStoragePoolType;
|
||||||
|
@@ -454,6 +455,7 @@ VIR_ENUM_DECL(virStoragePartedFs);
|
||||||
|
VIR_CONNECT_LIST_STORAGE_POOLS_SCSI | \
|
||||||
|
VIR_CONNECT_LIST_STORAGE_POOLS_MPATH | \
|
||||||
|
VIR_CONNECT_LIST_STORAGE_POOLS_RBD | \
|
||||||
|
+ VIR_CONNECT_LIST_STORAGE_POOLS_VITASTOR | \
|
||||||
|
VIR_CONNECT_LIST_STORAGE_POOLS_SHEEPDOG | \
|
||||||
|
VIR_CONNECT_LIST_STORAGE_POOLS_GLUSTER | \
|
||||||
|
VIR_CONNECT_LIST_STORAGE_POOLS_ZFS | \
|
||||||
|
diff --git a/src/conf/storage_source_conf.c b/src/conf/storage_source_conf.c
|
||||||
|
index d7b9bdfecb..38aefd0dd4 100644
|
||||||
|
--- a/src/conf/storage_source_conf.c
|
||||||
|
+++ b/src/conf/storage_source_conf.c
|
||||||
|
@@ -90,6 +90,7 @@ VIR_ENUM_IMPL(virStorageNetProtocol,
|
||||||
|
"ssh",
|
||||||
|
"vxhs",
|
||||||
|
"nfs",
|
||||||
|
+ "vitastor",
|
||||||
|
);
|
||||||
|
|
||||||
|
|
||||||
|
@@ -1317,6 +1318,7 @@ virStorageSourceNetworkDefaultPort(virStorageNetProtocol protocol)
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_GLUSTER:
|
||||||
|
return 24007;
|
||||||
|
|
||||||
|
+ case VIR_STORAGE_NET_PROTOCOL_VITASTOR:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_RBD:
|
||||||
|
/* we don't provide a default for RBD */
|
||||||
|
return 0;
|
||||||
|
diff --git a/src/conf/storage_source_conf.h b/src/conf/storage_source_conf.h
|
||||||
|
index 22c35d420d..f1e32ea83d 100644
|
||||||
|
--- a/src/conf/storage_source_conf.h
|
||||||
|
+++ b/src/conf/storage_source_conf.h
|
||||||
|
@@ -131,6 +131,7 @@ typedef enum {
|
||||||
|
VIR_STORAGE_NET_PROTOCOL_SSH,
|
||||||
|
VIR_STORAGE_NET_PROTOCOL_VXHS,
|
||||||
|
VIR_STORAGE_NET_PROTOCOL_NFS,
|
||||||
|
+ VIR_STORAGE_NET_PROTOCOL_VITASTOR,
|
||||||
|
|
||||||
|
VIR_STORAGE_NET_PROTOCOL_LAST
|
||||||
|
} virStorageNetProtocol;
|
||||||
|
diff --git a/src/conf/virstorageobj.c b/src/conf/virstorageobj.c
|
||||||
|
index 59fa5da372..4739167f5f 100644
|
||||||
|
--- a/src/conf/virstorageobj.c
|
||||||
|
+++ b/src/conf/virstorageobj.c
|
||||||
|
@@ -1438,6 +1438,7 @@ virStoragePoolObjSourceFindDuplicateCb(const void *payload,
|
||||||
|
return 1;
|
||||||
|
break;
|
||||||
|
|
||||||
|
+ case VIR_STORAGE_POOL_VITASTOR:
|
||||||
|
case VIR_STORAGE_POOL_ISCSI_DIRECT:
|
||||||
|
case VIR_STORAGE_POOL_RBD:
|
||||||
|
case VIR_STORAGE_POOL_LAST:
|
||||||
|
@@ -1921,6 +1922,8 @@ virStoragePoolObjMatch(virStoragePoolObj *obj,
|
||||||
|
(obj->def->type == VIR_STORAGE_POOL_MPATH)) ||
|
||||||
|
(MATCH(VIR_CONNECT_LIST_STORAGE_POOLS_RBD) &&
|
||||||
|
(obj->def->type == VIR_STORAGE_POOL_RBD)) ||
|
||||||
|
+ (MATCH(VIR_CONNECT_LIST_STORAGE_POOLS_VITASTOR) &&
|
||||||
|
+ (obj->def->type == VIR_STORAGE_POOL_VITASTOR)) ||
|
||||||
|
(MATCH(VIR_CONNECT_LIST_STORAGE_POOLS_SHEEPDOG) &&
|
||||||
|
(obj->def->type == VIR_STORAGE_POOL_SHEEPDOG)) ||
|
||||||
|
(MATCH(VIR_CONNECT_LIST_STORAGE_POOLS_GLUSTER) &&
|
||||||
|
diff --git a/src/libvirt-storage.c b/src/libvirt-storage.c
|
||||||
|
index db7660aac4..561df34709 100644
|
||||||
|
--- a/src/libvirt-storage.c
|
||||||
|
+++ b/src/libvirt-storage.c
|
||||||
|
@@ -94,6 +94,7 @@ virStoragePoolGetConnect(virStoragePoolPtr pool)
|
||||||
|
* VIR_CONNECT_LIST_STORAGE_POOLS_SCSI
|
||||||
|
* VIR_CONNECT_LIST_STORAGE_POOLS_MPATH
|
||||||
|
* VIR_CONNECT_LIST_STORAGE_POOLS_RBD
|
||||||
|
+ * VIR_CONNECT_LIST_STORAGE_POOLS_VITASTOR
|
||||||
|
* VIR_CONNECT_LIST_STORAGE_POOLS_SHEEPDOG
|
||||||
|
* VIR_CONNECT_LIST_STORAGE_POOLS_GLUSTER
|
||||||
|
* VIR_CONNECT_LIST_STORAGE_POOLS_ZFS
|
||||||
|
diff --git a/src/libxl/libxl_conf.c b/src/libxl/libxl_conf.c
|
||||||
|
index 2b988157fa..9d0eb47b25 100644
|
||||||
|
--- a/src/libxl/libxl_conf.c
|
||||||
|
+++ b/src/libxl/libxl_conf.c
|
||||||
|
@@ -1069,6 +1069,7 @@ libxlMakeNetworkDiskSrcStr(virStorageSource *src,
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_SSH:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_VXHS:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_NFS:
|
||||||
|
+ case VIR_STORAGE_NET_PROTOCOL_VITASTOR:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_LAST:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_NONE:
|
||||||
|
virReportError(VIR_ERR_NO_SUPPORT,
|
||||||
|
diff --git a/src/libxl/xen_xl.c b/src/libxl/xen_xl.c
|
||||||
|
index e72e7d7f44..8482c21805 100644
|
||||||
|
--- a/src/libxl/xen_xl.c
|
||||||
|
+++ b/src/libxl/xen_xl.c
|
||||||
|
@@ -1461,6 +1461,7 @@ xenFormatXLDiskSrcNet(virStorageSource *src)
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_SSH:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_VXHS:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_NFS:
|
||||||
|
+ case VIR_STORAGE_NET_PROTOCOL_VITASTOR:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_LAST:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_NONE:
|
||||||
|
virReportError(VIR_ERR_NO_SUPPORT,
|
||||||
|
diff --git a/src/qemu/qemu_block.c b/src/qemu/qemu_block.c
|
||||||
|
index 9b43279797..459d8e8a65 100644
|
||||||
|
--- a/src/qemu/qemu_block.c
|
||||||
|
+++ b/src/qemu/qemu_block.c
|
||||||
|
@@ -743,6 +743,38 @@ qemuBlockStorageSourceGetRBDProps(virStorageSource *src,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
+static virJSONValue *
|
||||||
|
+qemuBlockStorageSourceGetVitastorProps(virStorageSource *src)
|
||||||
|
+{
|
||||||
|
+ virJSONValue *ret = NULL;
|
||||||
|
+ virStorageNetHostDef *host;
|
||||||
|
+ size_t i;
|
||||||
|
+ g_auto(virBuffer) buf = VIR_BUFFER_INITIALIZER;
|
||||||
|
+ g_autofree char *etcd = NULL;
|
||||||
|
+
|
||||||
|
+ for (i = 0; i < src->nhosts; i++) {
|
||||||
|
+ host = src->hosts + i;
|
||||||
|
+ if ((virStorageNetHostTransport)host->transport != VIR_STORAGE_NET_HOST_TRANS_TCP) {
|
||||||
|
+ return NULL;
|
||||||
|
+ }
|
||||||
|
+ virBufferAsprintf(&buf, i > 0 ? ",%s:%u" : "%s:%u", host->name, host->port);
|
||||||
|
+ }
|
||||||
|
+ if (src->nhosts > 0) {
|
||||||
|
+ etcd = virBufferContentAndReset(&buf);
|
||||||
|
+ }
|
||||||
|
+
|
||||||
|
+ if (virJSONValueObjectAdd(&ret,
|
||||||
|
+ "S:etcd-host", etcd,
|
||||||
|
+ "S:etcd-prefix", src->query,
|
||||||
|
+ "S:config-path", src->configFile,
|
||||||
|
+ "s:image", src->path,
|
||||||
|
+ NULL) < 0)
|
||||||
|
+ return NULL;
|
||||||
|
+
|
||||||
|
+ return ret;
|
||||||
|
+}
|
||||||
|
+
|
||||||
|
+
|
||||||
|
static virJSONValue *
|
||||||
|
qemuBlockStorageSourceGetSshProps(virStorageSource *src)
|
||||||
|
{
|
||||||
|
@@ -1094,6 +1126,12 @@ qemuBlockStorageSourceGetBackendProps(virStorageSource *src,
|
||||||
|
return NULL;
|
||||||
|
break;
|
||||||
|
|
||||||
|
+ case VIR_STORAGE_NET_PROTOCOL_VITASTOR:
|
||||||
|
+ driver = "vitastor";
|
||||||
|
+ if (!(fileprops = qemuBlockStorageSourceGetVitastorProps(src)))
|
||||||
|
+ return NULL;
|
||||||
|
+ break;
|
||||||
|
+
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_SSH:
|
||||||
|
driver = "ssh";
|
||||||
|
if (!(fileprops = qemuBlockStorageSourceGetSshProps(src)))
|
||||||
|
@@ -1997,6 +2035,7 @@ qemuBlockGetBackingStoreString(virStorageSource *src,
|
||||||
|
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_SHEEPDOG:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_RBD:
|
||||||
|
+ case VIR_STORAGE_NET_PROTOCOL_VITASTOR:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_VXHS:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_NFS:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_SSH:
|
||||||
|
@@ -2377,6 +2416,12 @@ qemuBlockStorageSourceCreateGetStorageProps(virStorageSource *src,
|
||||||
|
return -1;
|
||||||
|
break;
|
||||||
|
|
||||||
|
+ case VIR_STORAGE_NET_PROTOCOL_VITASTOR:
|
||||||
|
+ driver = "vitastor";
|
||||||
|
+ if (!(location = qemuBlockStorageSourceGetVitastorProps(src)))
|
||||||
|
+ return -1;
|
||||||
|
+ break;
|
||||||
|
+
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_SSH:
|
||||||
|
if (srcPriv->nbdkitProcess) {
|
||||||
|
/* disk creation not yet supported with nbdkit, and even if it
|
||||||
|
diff --git a/src/qemu/qemu_domain.c b/src/qemu/qemu_domain.c
|
||||||
|
index ac56fc7cb4..9e407b4aab 100644
|
||||||
|
--- a/src/qemu/qemu_domain.c
|
||||||
|
+++ b/src/qemu/qemu_domain.c
|
||||||
|
@@ -4677,7 +4677,8 @@ qemuDomainValidateStorageSource(virStorageSource *src,
|
||||||
|
if (src->query &&
|
||||||
|
(actualType != VIR_STORAGE_TYPE_NETWORK ||
|
||||||
|
(src->protocol != VIR_STORAGE_NET_PROTOCOL_HTTPS &&
|
||||||
|
- src->protocol != VIR_STORAGE_NET_PROTOCOL_HTTP))) {
|
||||||
|
+ src->protocol != VIR_STORAGE_NET_PROTOCOL_HTTP &&
|
||||||
|
+ src->protocol != VIR_STORAGE_NET_PROTOCOL_VITASTOR))) {
|
||||||
|
virReportError(VIR_ERR_CONFIG_UNSUPPORTED, "%s",
|
||||||
|
_("query is supported only with HTTP(S) protocols"));
|
||||||
|
return -1;
|
||||||
|
@@ -9103,6 +9104,7 @@ qemuDomainPrepareStorageSourceTLS(virStorageSource *src,
|
||||||
|
break;
|
||||||
|
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_RBD:
|
||||||
|
+ case VIR_STORAGE_NET_PROTOCOL_VITASTOR:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_SHEEPDOG:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_GLUSTER:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_ISCSI:
|
||||||
|
diff --git a/src/qemu/qemu_snapshot.c b/src/qemu/qemu_snapshot.c
|
||||||
|
index e738afffc3..37d64f469b 100644
|
||||||
|
--- a/src/qemu/qemu_snapshot.c
|
||||||
|
+++ b/src/qemu/qemu_snapshot.c
|
||||||
|
@@ -665,6 +665,7 @@ qemuSnapshotPrepareDiskExternalInactive(virDomainSnapshotDiskDef *snapdisk,
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_NONE:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_NBD:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_RBD:
|
||||||
|
+ case VIR_STORAGE_NET_PROTOCOL_VITASTOR:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_SHEEPDOG:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_GLUSTER:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_ISCSI:
|
||||||
|
@@ -893,6 +894,7 @@ qemuSnapshotPrepareDiskInternal(virDomainDiskDef *disk,
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_NONE:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_NBD:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_RBD:
|
||||||
|
+ case VIR_STORAGE_NET_PROTOCOL_VITASTOR:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_SHEEPDOG:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_GLUSTER:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_ISCSI:
|
||||||
|
diff --git a/src/storage/storage_driver.c b/src/storage/storage_driver.c
|
||||||
|
index e19e032427..59f91f4710 100644
|
||||||
|
--- a/src/storage/storage_driver.c
|
||||||
|
+++ b/src/storage/storage_driver.c
|
||||||
|
@@ -1626,6 +1626,7 @@ storageVolLookupByPathCallback(virStoragePoolObj *obj,
|
||||||
|
|
||||||
|
case VIR_STORAGE_POOL_GLUSTER:
|
||||||
|
case VIR_STORAGE_POOL_RBD:
|
||||||
|
+ case VIR_STORAGE_POOL_VITASTOR:
|
||||||
|
case VIR_STORAGE_POOL_SHEEPDOG:
|
||||||
|
case VIR_STORAGE_POOL_ZFS:
|
||||||
|
case VIR_STORAGE_POOL_LAST:
|
||||||
|
diff --git a/src/storage_file/storage_source_backingstore.c b/src/storage_file/storage_source_backingstore.c
|
||||||
|
index 821378883c..2211f6891b 100644
|
||||||
|
--- a/src/storage_file/storage_source_backingstore.c
|
||||||
|
+++ b/src/storage_file/storage_source_backingstore.c
|
||||||
|
@@ -264,6 +264,75 @@ virStorageSourceParseRBDColonString(const char *rbdstr,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
+static int
|
||||||
|
+virStorageSourceParseVitastorColonString(const char *colonstr,
|
||||||
|
+ virStorageSource *src)
|
||||||
|
+{
|
||||||
|
+ char *p, *e, *next;
|
||||||
|
+ g_autofree char *options = NULL;
|
||||||
|
+
|
||||||
|
+ /* optionally skip the "vitastor:" prefix if provided */
|
||||||
|
+ if (STRPREFIX(colonstr, "vitastor:"))
|
||||||
|
+ colonstr += strlen("vitastor:");
|
||||||
|
+
|
||||||
|
+ options = g_strdup(colonstr);
|
||||||
|
+
|
||||||
|
+ p = options;
|
||||||
|
+ while (*p) {
|
||||||
|
+ /* find : delimiter or end of string */
|
||||||
|
+ for (e = p; *e && *e != ':'; ++e) {
|
||||||
|
+ if (*e == '\\') {
|
||||||
|
+ e++;
|
||||||
|
+ if (*e == '\0')
|
||||||
|
+ break;
|
||||||
|
+ }
|
||||||
|
+ }
|
||||||
|
+ if (*e == '\0') {
|
||||||
|
+ next = e; /* last kv pair */
|
||||||
|
+ } else {
|
||||||
|
+ next = e + 1;
|
||||||
|
+ *e = '\0';
|
||||||
|
+ }
|
||||||
|
+
|
||||||
|
+ if (STRPREFIX(p, "image=")) {
|
||||||
|
+ src->path = g_strdup(p + strlen("image="));
|
||||||
|
+ } else if (STRPREFIX(p, "etcd-prefix=")) {
|
||||||
|
+ src->query = g_strdup(p + strlen("etcd-prefix="));
|
||||||
|
+ } else if (STRPREFIX(p, "config-path=")) {
|
||||||
|
+ src->configFile = g_strdup(p + strlen("config-path="));
|
||||||
|
+ } else if (STRPREFIX(p, "etcd-host=")) {
|
||||||
|
+ char *h, *sep;
|
||||||
|
+
|
||||||
|
+ h = p + strlen("etcd-host=");
|
||||||
|
+ while (h < e) {
|
||||||
|
+ for (sep = h; sep < e; ++sep) {
|
||||||
|
+ if (*sep == '\\' && (sep[1] == ',' ||
|
||||||
|
+ sep[1] == ';' ||
|
||||||
|
+ sep[1] == ' ')) {
|
||||||
|
+ *sep = '\0';
|
||||||
|
+ sep += 2;
|
||||||
|
+ break;
|
||||||
|
+ }
|
||||||
|
+ }
|
||||||
|
+
|
||||||
|
+ if (virStorageSourceRBDAddHost(src, h) < 0)
|
||||||
|
+ return -1;
|
||||||
|
+
|
||||||
|
+ h = sep;
|
||||||
|
+ }
|
||||||
|
+ }
|
||||||
|
+
|
||||||
|
+ p = next;
|
||||||
|
+ }
|
||||||
|
+
|
||||||
|
+ if (!src->path) {
|
||||||
|
+ return -1;
|
||||||
|
+ }
|
||||||
|
+
|
||||||
|
+ return 0;
|
||||||
|
+}
|
||||||
|
+
|
||||||
|
+
|
||||||
|
static int
|
||||||
|
virStorageSourceParseNBDColonString(const char *nbdstr,
|
||||||
|
virStorageSource *src)
|
||||||
|
@@ -379,6 +448,11 @@ virStorageSourceParseBackingColon(virStorageSource *src,
|
||||||
|
return -1;
|
||||||
|
break;
|
||||||
|
|
||||||
|
+ case VIR_STORAGE_NET_PROTOCOL_VITASTOR:
|
||||||
|
+ if (virStorageSourceParseVitastorColonString(path, src) < 0)
|
||||||
|
+ return -1;
|
||||||
|
+ break;
|
||||||
|
+
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_SHEEPDOG:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_LAST:
|
||||||
|
case VIR_STORAGE_NET_PROTOCOL_NONE:
|
||||||
|
@@ -953,6 +1027,54 @@ virStorageSourceParseBackingJSONRBD(virStorageSource *src,
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
+static int
|
||||||
|
+virStorageSourceParseBackingJSONVitastor(virStorageSource *src,
|
||||||
|
+ virJSONValue *json,
|
||||||
|
+ const char *jsonstr G_GNUC_UNUSED,
|
||||||
|
+ int opaque G_GNUC_UNUSED)
|
||||||
|
+{
|
||||||
|
+ const char *filename;
|
||||||
|
+ const char *image = virJSONValueObjectGetString(json, "image");
|
||||||
|
+ const char *conf = virJSONValueObjectGetString(json, "config-path");
|
||||||
|
+ const char *etcd_prefix = virJSONValueObjectGetString(json, "etcd-prefix");
|
||||||
|
+ virJSONValue *servers = virJSONValueObjectGetArray(json, "server");
|
||||||
|
+ size_t nservers;
|
||||||
|
+ size_t i;
|
||||||
|
+
|
||||||
|
+ src->type = VIR_STORAGE_TYPE_NETWORK;
|
||||||
|
+ src->protocol = VIR_STORAGE_NET_PROTOCOL_VITASTOR;
|
||||||
|
+
|
||||||
|
+ /* legacy syntax passed via 'filename' option */
|
||||||
|
+ if ((filename = virJSONValueObjectGetString(json, "filename")))
|
||||||
|
+ return virStorageSourceParseVitastorColonString(filename, src);
|
||||||
|
+
|
||||||
|
+ if (!image) {
|
||||||
|
+ virReportError(VIR_ERR_INVALID_ARG, "%s",
|
||||||
|
+ _("missing image name in Vitastor backing volume "
|
||||||
|
+ "JSON specification"));
|
||||||
|
+ return -1;
|
||||||
|
+ }
|
||||||
|
+
|
||||||
|
+ src->path = g_strdup(image);
|
||||||
|
+ src->configFile = g_strdup(conf);
|
||||||
|
+ src->query = g_strdup(etcd_prefix);
|
||||||
|
+
|
||||||
|
+ if (servers) {
|
||||||
|
+ nservers = virJSONValueArraySize(servers);
|
||||||
|
+
|
||||||
|
+ src->hosts = g_new0(virStorageNetHostDef, nservers);
|
||||||
|
+ src->nhosts = nservers;
|
||||||
|
+
|
||||||
|
+ for (i = 0; i < nservers; i++) {
|
||||||
|
+ if (virStorageSourceParseBackingJSONInetSocketAddress(src->hosts + i,
|
||||||
|
+ virJSONValueArrayGet(servers, i)) < 0)
|
||||||
|
+ return -1;
|
||||||
|
+ }
|
||||||
|
+ }
|
||||||
|
+
|
||||||
|
+ return 0;
|
||||||
|
+}
|
||||||
|
+
|
||||||
|
static int
|
||||||
|
virStorageSourceParseBackingJSONRaw(virStorageSource *src,
|
||||||
|
virJSONValue *json,
|
||||||
|
@@ -1130,6 +1252,7 @@ static const struct virStorageSourceJSONDriverParser jsonParsers[] = {
|
||||||
|
{"sheepdog", false, virStorageSourceParseBackingJSONSheepdog, 0},
|
||||||
|
{"ssh", false, virStorageSourceParseBackingJSONSSH, 0},
|
||||||
|
{"rbd", false, virStorageSourceParseBackingJSONRBD, 0},
|
||||||
|
+ {"vitastor", false, virStorageSourceParseBackingJSONVitastor, 0},
|
||||||
|
{"raw", true, virStorageSourceParseBackingJSONRaw, 0},
|
||||||
|
{"nfs", false, virStorageSourceParseBackingJSONNFS, 0},
|
||||||
|
{"vxhs", false, virStorageSourceParseBackingJSONVxHS, 0},
|
||||||
|
diff --git a/src/test/test_driver.c b/src/test/test_driver.c
|
||||||
|
index 1165689de7..bba846351c 100644
|
||||||
|
--- a/src/test/test_driver.c
|
||||||
|
+++ b/src/test/test_driver.c
|
||||||
|
@@ -7345,6 +7345,7 @@ testStorageVolumeTypeForPool(int pooltype)
|
||||||
|
case VIR_STORAGE_POOL_ISCSI_DIRECT:
|
||||||
|
case VIR_STORAGE_POOL_GLUSTER:
|
||||||
|
case VIR_STORAGE_POOL_RBD:
|
||||||
|
+ case VIR_STORAGE_POOL_VITASTOR:
|
||||||
|
return VIR_STORAGE_VOL_NETWORK;
|
||||||
|
case VIR_STORAGE_POOL_LOGICAL:
|
||||||
|
case VIR_STORAGE_POOL_DISK:
|
||||||
|
diff --git a/tests/storagepoolcapsschemadata/poolcaps-fs.xml b/tests/storagepoolcapsschemadata/poolcaps-fs.xml
|
||||||
|
index eee75af746..8bd0a57bdd 100644
|
||||||
|
--- a/tests/storagepoolcapsschemadata/poolcaps-fs.xml
|
||||||
|
+++ b/tests/storagepoolcapsschemadata/poolcaps-fs.xml
|
||||||
|
@@ -204,4 +204,11 @@
|
||||||
|
</enum>
|
||||||
|
</volOptions>
|
||||||
|
</pool>
|
||||||
|
+ <pool type='vitastor' supported='no'>
|
||||||
|
+ <volOptions>
|
||||||
|
+ <defaultFormat type='raw'/>
|
||||||
|
+ <enum name='targetFormatType'>
|
||||||
|
+ </enum>
|
||||||
|
+ </volOptions>
|
||||||
|
+ </pool>
|
||||||
|
</storagepoolCapabilities>
|
||||||
|
diff --git a/tests/storagepoolcapsschemadata/poolcaps-full.xml b/tests/storagepoolcapsschemadata/poolcaps-full.xml
|
||||||
|
index 805950a937..852df0de16 100644
|
||||||
|
--- a/tests/storagepoolcapsschemadata/poolcaps-full.xml
|
||||||
|
+++ b/tests/storagepoolcapsschemadata/poolcaps-full.xml
|
||||||
|
@@ -204,4 +204,11 @@
|
||||||
|
</enum>
|
||||||
|
</volOptions>
|
||||||
|
</pool>
|
||||||
|
+ <pool type='vitastor' supported='yes'>
|
||||||
|
+ <volOptions>
|
||||||
|
+ <defaultFormat type='raw'/>
|
||||||
|
+ <enum name='targetFormatType'>
|
||||||
|
+ </enum>
|
||||||
|
+ </volOptions>
|
||||||
|
+ </pool>
|
||||||
|
</storagepoolCapabilities>
|
||||||
|
diff --git a/tests/storagepoolxml2argvtest.c b/tests/storagepoolxml2argvtest.c
|
||||||
|
index d5c2531ab8..b19308ac38 100644
|
||||||
|
--- a/tests/storagepoolxml2argvtest.c
|
||||||
|
+++ b/tests/storagepoolxml2argvtest.c
|
||||||
|
@@ -57,6 +57,7 @@ testCompareXMLToArgvFiles(bool shouldFail,
|
||||||
|
case VIR_STORAGE_POOL_GLUSTER:
|
||||||
|
case VIR_STORAGE_POOL_ZFS:
|
||||||
|
case VIR_STORAGE_POOL_VSTORAGE:
|
||||||
|
+ case VIR_STORAGE_POOL_VITASTOR:
|
||||||
|
case VIR_STORAGE_POOL_LAST:
|
||||||
|
default:
|
||||||
|
VIR_TEST_DEBUG("pool type '%s' has no xml2argv test", defTypeStr);
|
||||||
|
diff --git a/tools/virsh-pool.c b/tools/virsh-pool.c
|
||||||
|
index 2010ef1356..072e2ff9e8 100644
|
||||||
|
--- a/tools/virsh-pool.c
|
||||||
|
+++ b/tools/virsh-pool.c
|
||||||
|
@@ -1187,6 +1187,9 @@ cmdPoolList(vshControl *ctl, const vshCmd *cmd G_GNUC_UNUSED)
|
||||||
|
case VIR_STORAGE_POOL_VSTORAGE:
|
||||||
|
flags |= VIR_CONNECT_LIST_STORAGE_POOLS_VSTORAGE;
|
||||||
|
break;
|
||||||
|
+ case VIR_STORAGE_POOL_VITASTOR:
|
||||||
|
+ flags |= VIR_CONNECT_LIST_STORAGE_POOLS_VITASTOR;
|
||||||
|
+ break;
|
||||||
|
case VIR_STORAGE_POOL_LAST:
|
||||||
|
break;
|
||||||
|
}
|
||||||
@@ -1,29 +1,172 @@
|
|||||||
diff --git a/src/client/qemu_driver.c b/src/client/qemu_driver.c
|
diff --git a/block/meson.build b/block/meson.build
|
||||||
index d8356dab..5f4cd50d 100644
|
index 34b1b2a306..24ca0f1e52 100644
|
||||||
--- a/src/client/qemu_driver.c
|
--- a/block/meson.build
|
||||||
+++ b/src/client/qemu_driver.c
|
+++ b/block/meson.build
|
||||||
@@ -974,14 +974,21 @@ static void vitastor_co_read_bitmap_cb(void *opaque, long retval, uint8_t *bitma
|
@@ -114,6 +114,7 @@ foreach m : [
|
||||||
#endif
|
[libnfs, 'nfs', files('nfs.c')],
|
||||||
}
|
[libssh, 'ssh', files('ssh.c')],
|
||||||
|
[rbd, 'rbd', files('rbd.c')],
|
||||||
|
+ [vitastor, 'vitastor', files('vitastor.c')],
|
||||||
|
]
|
||||||
|
if m[0].found()
|
||||||
|
module_ss = ss.source_set()
|
||||||
|
diff --git a/meson.build b/meson.build
|
||||||
|
index 50c774a195..e5c7a3a4b1 100644
|
||||||
|
--- a/meson.build
|
||||||
|
+++ b/meson.build
|
||||||
|
@@ -1652,6 +1652,26 @@ if not get_option('rbd').auto() or have_block
|
||||||
|
endif
|
||||||
|
endif
|
||||||
|
|
||||||
-static int coroutine_fn vitastor_co_block_status(
|
+vitastor = not_found
|
||||||
- BlockDriverState *bs, bool want_zero, int64_t offset, int64_t bytes,
|
+if not get_option('vitastor').auto() or have_block
|
||||||
- int64_t *pnum, int64_t *map, BlockDriverState **file)
|
+ libvitastor_client = cc.find_library('vitastor_client', has_headers: ['vitastor_c.h'],
|
||||||
+static int coroutine_fn vitastor_co_block_status(BlockDriverState *bs,
|
+ required: get_option('vitastor'))
|
||||||
+#if QEMU_VERSION_MAJOR > 10 || QEMU_VERSION_MAJOR == 10 && QEMU_VERSION_MINOR >= 1
|
+ if libvitastor_client.found()
|
||||||
+ unsigned int mode,
|
+ if cc.links('''
|
||||||
+#else
|
+ #include <vitastor_c.h>
|
||||||
+ bool want_zero,
|
+ int main(void) {
|
||||||
+#endif
|
+ vitastor_c_create_qemu(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0);
|
||||||
+ int64_t offset, int64_t bytes, int64_t *pnum, int64_t *map, BlockDriverState **file)
|
+ return 0;
|
||||||
{
|
+ }''', dependencies: libvitastor_client)
|
||||||
// Allocated => return BDRV_BLOCK_DATA|BDRV_BLOCK_OFFSET_VALID
|
+ vitastor = declare_dependency(dependencies: libvitastor_client)
|
||||||
// Not allocated => return 0
|
+ elif get_option('vitastor').enabled()
|
||||||
// Error => return -errno
|
+ error('could not link libvitastor_client')
|
||||||
// Set pnum to length of the extent, `*map` = `offset`, `*file` = `bs`
|
+ else
|
||||||
+#if QEMU_VERSION_MAJOR > 10 || QEMU_VERSION_MAJOR == 10 && QEMU_VERSION_MINOR >= 1
|
+ warning('could not link libvitastor_client, disabling')
|
||||||
+ int want_zero = (mode == BDRV_WANT_PRECISE);
|
+ endif
|
||||||
+#endif
|
+ endif
|
||||||
VitastorRPC task;
|
+endif
|
||||||
VitastorClient *client = bs->opaque;
|
+
|
||||||
uint64_t inode = client->watch ? vitastor_c_inode_get_num(client->watch) : client->inode;
|
glusterfs = not_found
|
||||||
|
glusterfs_ftruncate_has_stat = false
|
||||||
|
glusterfs_iocb_has_stat = false
|
||||||
|
@@ -2547,6 +2567,7 @@ endif
|
||||||
|
config_host_data.set('CONFIG_OPENGL', opengl.found())
|
||||||
|
config_host_data.set('CONFIG_PLUGIN', get_option('plugins'))
|
||||||
|
config_host_data.set('CONFIG_RBD', rbd.found())
|
||||||
|
+config_host_data.set('CONFIG_VITASTOR', vitastor.found())
|
||||||
|
config_host_data.set('CONFIG_RDMA', rdma.found())
|
||||||
|
config_host_data.set('CONFIG_RELOCATABLE', get_option('relocatable'))
|
||||||
|
config_host_data.set('CONFIG_SAFESTACK', get_option('safe_stack'))
|
||||||
|
@@ -4972,6 +4993,7 @@ summary_info += {'fdt support': fdt_opt == 'internal' ? 'internal' : fdt}
|
||||||
|
summary_info += {'libcap-ng support': libcap_ng}
|
||||||
|
summary_info += {'bpf support': libbpf}
|
||||||
|
summary_info += {'rbd support': rbd}
|
||||||
|
+summary_info += {'vitastor support': vitastor}
|
||||||
|
summary_info += {'smartcard support': cacard}
|
||||||
|
summary_info += {'U2F support': u2f}
|
||||||
|
summary_info += {'libusb': libusb}
|
||||||
|
diff --git a/meson_options.txt b/meson_options.txt
|
||||||
|
index fff1521e58..f0844c0e00 100644
|
||||||
|
--- a/meson_options.txt
|
||||||
|
+++ b/meson_options.txt
|
||||||
|
@@ -202,6 +202,8 @@ option('pvg', type: 'feature', value: 'auto',
|
||||||
|
description: 'macOS paravirtualized graphics support')
|
||||||
|
option('rbd', type : 'feature', value : 'auto',
|
||||||
|
description: 'Ceph block device driver')
|
||||||
|
+option('vitastor', type : 'feature', value : 'auto',
|
||||||
|
+ description: 'Vitastor block device driver')
|
||||||
|
option('opengl', type : 'feature', value : 'auto',
|
||||||
|
description: 'OpenGL support')
|
||||||
|
option('rdma', type : 'feature', value : 'auto',
|
||||||
|
diff --git a/qapi/block-core.json b/qapi/block-core.json
|
||||||
|
index dc6eb4ae23..d043f4340e 100644
|
||||||
|
--- a/qapi/block-core.json
|
||||||
|
+++ b/qapi/block-core.json
|
||||||
|
@@ -3280,7 +3280,7 @@
|
||||||
|
'parallels', 'preallocate', 'qcow', 'qcow2', 'qed', 'quorum',
|
||||||
|
'raw', 'rbd',
|
||||||
|
{ 'name': 'replication', 'if': 'CONFIG_REPLICATION' },
|
||||||
|
- 'ssh', 'throttle', 'vdi', 'vhdx',
|
||||||
|
+ 'ssh', 'throttle', 'vdi', 'vhdx', 'vitastor',
|
||||||
|
{ 'name': 'virtio-blk-vfio-pci', 'if': 'CONFIG_BLKIO' },
|
||||||
|
{ 'name': 'virtio-blk-vhost-user', 'if': 'CONFIG_BLKIO' },
|
||||||
|
{ 'name': 'virtio-blk-vhost-vdpa', 'if': 'CONFIG_BLKIO' },
|
||||||
|
@@ -4363,6 +4363,28 @@
|
||||||
|
'*key-secret': 'str',
|
||||||
|
'*server': ['InetSocketAddressBase'] } }
|
||||||
|
|
||||||
|
+##
|
||||||
|
+# @BlockdevOptionsVitastor:
|
||||||
|
+#
|
||||||
|
+# Driver specific block device options for vitastor
|
||||||
|
+#
|
||||||
|
+# @image: Image name
|
||||||
|
+# @inode: Inode number
|
||||||
|
+# @pool: Pool ID
|
||||||
|
+# @size: Desired image size in bytes
|
||||||
|
+# @config-path: Path to Vitastor configuration
|
||||||
|
+# @etcd-host: etcd connection address(es)
|
||||||
|
+# @etcd-prefix: etcd key/value prefix
|
||||||
|
+##
|
||||||
|
+{ 'struct': 'BlockdevOptionsVitastor',
|
||||||
|
+ 'data': { '*inode': 'uint64',
|
||||||
|
+ '*pool': 'uint64',
|
||||||
|
+ '*size': 'uint64',
|
||||||
|
+ '*image': 'str',
|
||||||
|
+ '*config-path': 'str',
|
||||||
|
+ '*etcd-host': 'str',
|
||||||
|
+ '*etcd-prefix': 'str' } }
|
||||||
|
+
|
||||||
|
##
|
||||||
|
# @ReplicationMode:
|
||||||
|
#
|
||||||
|
@@ -4831,6 +4853,7 @@
|
||||||
|
'throttle': 'BlockdevOptionsThrottle',
|
||||||
|
'vdi': 'BlockdevOptionsGenericFormat',
|
||||||
|
'vhdx': 'BlockdevOptionsGenericFormat',
|
||||||
|
+ 'vitastor': 'BlockdevOptionsVitastor',
|
||||||
|
'virtio-blk-vfio-pci':
|
||||||
|
{ 'type': 'BlockdevOptionsVirtioBlkVfioPci',
|
||||||
|
'if': 'CONFIG_BLKIO' },
|
||||||
|
@@ -5304,6 +5327,20 @@
|
||||||
|
'*cluster-size' : 'size',
|
||||||
|
'*encrypt' : 'RbdEncryptionCreateOptions' } }
|
||||||
|
|
||||||
|
+##
|
||||||
|
+# @BlockdevCreateOptionsVitastor:
|
||||||
|
+#
|
||||||
|
+# Driver specific image creation options for Vitastor.
|
||||||
|
+#
|
||||||
|
+# @location: Where to store the new image file. This location cannot
|
||||||
|
+# point to a snapshot.
|
||||||
|
+#
|
||||||
|
+# @size: Size of the virtual disk in bytes
|
||||||
|
+##
|
||||||
|
+{ 'struct': 'BlockdevCreateOptionsVitastor',
|
||||||
|
+ 'data': { 'location': 'BlockdevOptionsVitastor',
|
||||||
|
+ 'size': 'size' } }
|
||||||
|
+
|
||||||
|
##
|
||||||
|
# @BlockdevVmdkSubformat:
|
||||||
|
#
|
||||||
|
@@ -5526,6 +5563,7 @@
|
||||||
|
'ssh': 'BlockdevCreateOptionsSsh',
|
||||||
|
'vdi': 'BlockdevCreateOptionsVdi',
|
||||||
|
'vhdx': 'BlockdevCreateOptionsVhdx',
|
||||||
|
+ 'vitastor': 'BlockdevCreateOptionsVitastor',
|
||||||
|
'vmdk': 'BlockdevCreateOptionsVmdk',
|
||||||
|
'vpc': 'BlockdevCreateOptionsVpc'
|
||||||
|
} }
|
||||||
|
diff --git a/scripts/meson-buildoptions.sh b/scripts/meson-buildoptions.sh
|
||||||
|
index 0ebe6bc52a..2c37ad3892 100644
|
||||||
|
--- a/scripts/meson-buildoptions.sh
|
||||||
|
+++ b/scripts/meson-buildoptions.sh
|
||||||
|
@@ -175,6 +175,7 @@ meson_options_help() {
|
||||||
|
printf "%s\n" ' qga-vss build QGA VSS support (broken with MinGW)'
|
||||||
|
printf "%s\n" ' qpl Query Processing Library support'
|
||||||
|
printf "%s\n" ' rbd Ceph block device driver'
|
||||||
|
+ printf "%s\n" ' vitastor Vitastor block device driver'
|
||||||
|
printf "%s\n" ' rdma Enable RDMA-based migration'
|
||||||
|
printf "%s\n" ' replication replication support'
|
||||||
|
printf "%s\n" ' rust Rust support'
|
||||||
|
@@ -459,6 +460,8 @@ _meson_option_parse() {
|
||||||
|
--disable-qpl) printf "%s" -Dqpl=disabled ;;
|
||||||
|
--enable-rbd) printf "%s" -Drbd=enabled ;;
|
||||||
|
--disable-rbd) printf "%s" -Drbd=disabled ;;
|
||||||
|
+ --enable-vitastor) printf "%s" -Dvitastor=enabled ;;
|
||||||
|
+ --disable-vitastor) printf "%s" -Dvitastor=disabled ;;
|
||||||
|
--enable-rdma) printf "%s" -Drdma=enabled ;;
|
||||||
|
--disable-rdma) printf "%s" -Drdma=disabled ;;
|
||||||
|
--enable-relocatable) printf "%s" -Drelocatable=true ;;
|
||||||
|
|||||||
@@ -0,0 +1,172 @@
|
|||||||
|
diff --git a/block/meson.build b/block/meson.build
|
||||||
|
index 34b1b2a306..24ca0f1e52 100644
|
||||||
|
--- a/block/meson.build
|
||||||
|
+++ b/block/meson.build
|
||||||
|
@@ -114,6 +114,7 @@ foreach m : [
|
||||||
|
[libnfs, 'nfs', files('nfs.c')],
|
||||||
|
[libssh, 'ssh', files('ssh.c')],
|
||||||
|
[rbd, 'rbd', files('rbd.c')],
|
||||||
|
+ [vitastor, 'vitastor', files('vitastor.c')],
|
||||||
|
]
|
||||||
|
if m[0].found()
|
||||||
|
module_ss = ss.source_set()
|
||||||
|
diff --git a/meson.build b/meson.build
|
||||||
|
index d9293294d8..776a5becc6 100644
|
||||||
|
--- a/meson.build
|
||||||
|
+++ b/meson.build
|
||||||
|
@@ -1665,6 +1665,26 @@ if not get_option('rbd').auto() or have_block
|
||||||
|
endif
|
||||||
|
endif
|
||||||
|
|
||||||
|
+vitastor = not_found
|
||||||
|
+if not get_option('vitastor').auto() or have_block
|
||||||
|
+ libvitastor_client = cc.find_library('vitastor_client', has_headers: ['vitastor_c.h'],
|
||||||
|
+ required: get_option('vitastor'))
|
||||||
|
+ if libvitastor_client.found()
|
||||||
|
+ if cc.links('''
|
||||||
|
+ #include <vitastor_c.h>
|
||||||
|
+ int main(void) {
|
||||||
|
+ vitastor_c_create_qemu(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0);
|
||||||
|
+ return 0;
|
||||||
|
+ }''', dependencies: libvitastor_client)
|
||||||
|
+ vitastor = declare_dependency(dependencies: libvitastor_client)
|
||||||
|
+ elif get_option('vitastor').enabled()
|
||||||
|
+ error('could not link libvitastor_client')
|
||||||
|
+ else
|
||||||
|
+ warning('could not link libvitastor_client, disabling')
|
||||||
|
+ endif
|
||||||
|
+ endif
|
||||||
|
+endif
|
||||||
|
+
|
||||||
|
glusterfs = not_found
|
||||||
|
glusterfs_ftruncate_has_stat = false
|
||||||
|
glusterfs_iocb_has_stat = false
|
||||||
|
@@ -2509,6 +2529,7 @@ endif
|
||||||
|
config_host_data.set('CONFIG_OPENGL', opengl.found())
|
||||||
|
config_host_data.set('CONFIG_PLUGIN', get_option('plugins'))
|
||||||
|
config_host_data.set('CONFIG_RBD', rbd.found())
|
||||||
|
+config_host_data.set('CONFIG_VITASTOR', vitastor.found())
|
||||||
|
config_host_data.set('CONFIG_RDMA', rdma.found())
|
||||||
|
config_host_data.set('CONFIG_RELOCATABLE', get_option('relocatable'))
|
||||||
|
config_host_data.set('CONFIG_SAFESTACK', get_option('safe_stack'))
|
||||||
|
@@ -4948,6 +4969,7 @@ summary_info += {'fdt support': fdt_opt == 'internal' ? 'internal' : fdt}
|
||||||
|
summary_info += {'libcap-ng support': libcap_ng}
|
||||||
|
summary_info += {'bpf support': libbpf}
|
||||||
|
summary_info += {'rbd support': rbd}
|
||||||
|
+summary_info += {'vitastor support': vitastor}
|
||||||
|
summary_info += {'smartcard support': cacard}
|
||||||
|
summary_info += {'U2F support': u2f}
|
||||||
|
summary_info += {'libusb': libusb}
|
||||||
|
diff --git a/meson_options.txt b/meson_options.txt
|
||||||
|
index 2836156257..148086cc6f 100644
|
||||||
|
--- a/meson_options.txt
|
||||||
|
+++ b/meson_options.txt
|
||||||
|
@@ -206,6 +206,8 @@ option('pvg', type: 'feature', value: 'auto',
|
||||||
|
description: 'macOS paravirtualized graphics support')
|
||||||
|
option('rbd', type : 'feature', value : 'auto',
|
||||||
|
description: 'Ceph block device driver')
|
||||||
|
+option('vitastor', type : 'feature', value : 'auto',
|
||||||
|
+ description: 'Vitastor block device driver')
|
||||||
|
option('opengl', type : 'feature', value : 'auto',
|
||||||
|
description: 'OpenGL support')
|
||||||
|
option('rdma', type : 'feature', value : 'auto',
|
||||||
|
diff --git a/qapi/block-core.json b/qapi/block-core.json
|
||||||
|
index b82af74256..f25a6f5ce8 100644
|
||||||
|
--- a/qapi/block-core.json
|
||||||
|
+++ b/qapi/block-core.json
|
||||||
|
@@ -3351,7 +3351,7 @@
|
||||||
|
'parallels', 'preallocate', 'qcow', 'qcow2', 'qed', 'quorum',
|
||||||
|
'raw', 'rbd',
|
||||||
|
{ 'name': 'replication', 'if': 'CONFIG_REPLICATION' },
|
||||||
|
- 'ssh', 'throttle', 'vdi', 'vhdx',
|
||||||
|
+ 'ssh', 'throttle', 'vdi', 'vhdx', 'vitastor',
|
||||||
|
{ 'name': 'virtio-blk-vfio-pci', 'if': 'CONFIG_BLKIO' },
|
||||||
|
{ 'name': 'virtio-blk-vhost-user', 'if': 'CONFIG_BLKIO' },
|
||||||
|
{ 'name': 'virtio-blk-vhost-vdpa', 'if': 'CONFIG_BLKIO' },
|
||||||
|
@@ -4434,6 +4434,28 @@
|
||||||
|
'*key-secret': 'str',
|
||||||
|
'*server': ['InetSocketAddressBase'] } }
|
||||||
|
|
||||||
|
+##
|
||||||
|
+# @BlockdevOptionsVitastor:
|
||||||
|
+#
|
||||||
|
+# Driver specific block device options for vitastor
|
||||||
|
+#
|
||||||
|
+# @image: Image name
|
||||||
|
+# @inode: Inode number
|
||||||
|
+# @pool: Pool ID
|
||||||
|
+# @size: Desired image size in bytes
|
||||||
|
+# @config-path: Path to Vitastor configuration
|
||||||
|
+# @etcd-host: etcd connection address(es)
|
||||||
|
+# @etcd-prefix: etcd key/value prefix
|
||||||
|
+##
|
||||||
|
+{ 'struct': 'BlockdevOptionsVitastor',
|
||||||
|
+ 'data': { '*inode': 'uint64',
|
||||||
|
+ '*pool': 'uint64',
|
||||||
|
+ '*size': 'uint64',
|
||||||
|
+ '*image': 'str',
|
||||||
|
+ '*config-path': 'str',
|
||||||
|
+ '*etcd-host': 'str',
|
||||||
|
+ '*etcd-prefix': 'str' } }
|
||||||
|
+
|
||||||
|
##
|
||||||
|
# @ReplicationMode:
|
||||||
|
#
|
||||||
|
@@ -4902,6 +4924,7 @@
|
||||||
|
'throttle': 'BlockdevOptionsThrottle',
|
||||||
|
'vdi': 'BlockdevOptionsGenericFormat',
|
||||||
|
'vhdx': 'BlockdevOptionsGenericFormat',
|
||||||
|
+ 'vitastor': 'BlockdevOptionsVitastor',
|
||||||
|
'virtio-blk-vfio-pci':
|
||||||
|
{ 'type': 'BlockdevOptionsVirtioBlkVfioPci',
|
||||||
|
'if': 'CONFIG_BLKIO' },
|
||||||
|
@@ -5376,6 +5399,20 @@
|
||||||
|
'*cluster-size' : 'size',
|
||||||
|
'*encrypt' : 'RbdEncryptionCreateOptions' } }
|
||||||
|
|
||||||
|
+##
|
||||||
|
+# @BlockdevCreateOptionsVitastor:
|
||||||
|
+#
|
||||||
|
+# Driver specific image creation options for Vitastor.
|
||||||
|
+#
|
||||||
|
+# @location: Where to store the new image file. This location cannot
|
||||||
|
+# point to a snapshot.
|
||||||
|
+#
|
||||||
|
+# @size: Size of the virtual disk in bytes
|
||||||
|
+##
|
||||||
|
+{ 'struct': 'BlockdevCreateOptionsVitastor',
|
||||||
|
+ 'data': { 'location': 'BlockdevOptionsVitastor',
|
||||||
|
+ 'size': 'size' } }
|
||||||
|
+
|
||||||
|
##
|
||||||
|
# @BlockdevVmdkSubformat:
|
||||||
|
#
|
||||||
|
@@ -5598,6 +5635,7 @@
|
||||||
|
'ssh': 'BlockdevCreateOptionsSsh',
|
||||||
|
'vdi': 'BlockdevCreateOptionsVdi',
|
||||||
|
'vhdx': 'BlockdevCreateOptionsVhdx',
|
||||||
|
+ 'vitastor': 'BlockdevCreateOptionsVitastor',
|
||||||
|
'vmdk': 'BlockdevCreateOptionsVmdk',
|
||||||
|
'vpc': 'BlockdevCreateOptionsVpc'
|
||||||
|
} }
|
||||||
|
diff --git a/scripts/meson-buildoptions.sh b/scripts/meson-buildoptions.sh
|
||||||
|
index 3d0d132344..65ee8c855e 100644
|
||||||
|
--- a/scripts/meson-buildoptions.sh
|
||||||
|
+++ b/scripts/meson-buildoptions.sh
|
||||||
|
@@ -177,6 +177,7 @@ meson_options_help() {
|
||||||
|
printf "%s\n" ' qga-vss build QGA VSS support (broken with MinGW)'
|
||||||
|
printf "%s\n" ' qpl Query Processing Library support'
|
||||||
|
printf "%s\n" ' rbd Ceph block device driver'
|
||||||
|
+ printf "%s\n" ' vitastor Vitastor block device driver'
|
||||||
|
printf "%s\n" ' rdma Enable RDMA-based migration'
|
||||||
|
printf "%s\n" ' replication replication support'
|
||||||
|
printf "%s\n" ' rust Rust support'
|
||||||
|
@@ -464,6 +465,8 @@ _meson_option_parse() {
|
||||||
|
--disable-qpl) printf "%s" -Dqpl=disabled ;;
|
||||||
|
--enable-rbd) printf "%s" -Drbd=enabled ;;
|
||||||
|
--disable-rbd) printf "%s" -Drbd=disabled ;;
|
||||||
|
+ --enable-vitastor) printf "%s" -Dvitastor=enabled ;;
|
||||||
|
+ --disable-vitastor) printf "%s" -Dvitastor=disabled ;;
|
||||||
|
--enable-rdma) printf "%s" -Drdma=enabled ;;
|
||||||
|
--disable-rdma) printf "%s" -Drdma=disabled ;;
|
||||||
|
--enable-relocatable) printf "%s" -Drelocatable=true ;;
|
||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 3.0.6
|
Version: 3.0.14
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-3.0.6.el10.tar.gz
|
Source0: vitastor-3.0.14.el10.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: gcc-c++
|
BuildRequires: gcc-c++
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 3.0.6
|
Version: 3.0.14
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-3.0.6.el7.tar.gz
|
Source0: vitastor-3.0.14.el7.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: devtoolset-9-gcc-c++
|
BuildRequires: devtoolset-9-gcc-c++
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 3.0.6
|
Version: 3.0.14
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-3.0.6.el8.tar.gz
|
Source0: vitastor-3.0.14.el8.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: gcc-toolset-9-gcc-c++
|
BuildRequires: gcc-toolset-9-gcc-c++
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 3.0.6
|
Version: 3.0.14
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-3.0.6.el9.tar.gz
|
Source0: vitastor-3.0.14.el9.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: gcc-c++
|
BuildRequires: gcc-c++
|
||||||
|
|||||||
+2
-3
@@ -1,9 +1,8 @@
|
|||||||
cmake_minimum_required(VERSION 2.8.12)
|
cmake_minimum_required(VERSION 2.8...3.30)
|
||||||
|
|
||||||
project(vitastor)
|
project(vitastor)
|
||||||
|
|
||||||
include(GNUInstallDirs)
|
include(GNUInstallDirs)
|
||||||
include(CTest)
|
|
||||||
include(CheckIncludeFile)
|
include(CheckIncludeFile)
|
||||||
|
|
||||||
find_package(PkgConfig)
|
find_package(PkgConfig)
|
||||||
@@ -21,7 +20,7 @@ if("${CMAKE_INSTALL_PREFIX}" MATCHES "^/usr/local/?$")
|
|||||||
endif()
|
endif()
|
||||||
set(ENABLE_COVERAGE false CACHE BOOL "Enable code coverage")
|
set(ENABLE_COVERAGE false CACHE BOOL "Enable code coverage")
|
||||||
|
|
||||||
add_definitions(-DVITASTOR_VERSION="3.0.6")
|
add_definitions(-DVITASTOR_VERSION="3.0.14")
|
||||||
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
|
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
|
||||||
add_link_options(-fno-omit-frame-pointer)
|
add_link_options(-fno-omit-frame-pointer)
|
||||||
if (${WITH_ASAN})
|
if (${WITH_ASAN})
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
cmake_minimum_required(VERSION 2.8.12)
|
cmake_minimum_required(VERSION 2.8...3.30)
|
||||||
|
|
||||||
project(vitastor)
|
project(vitastor)
|
||||||
|
|
||||||
|
|||||||
@@ -228,4 +228,9 @@ public:
|
|||||||
virtual uint64_t get_journal_size() = 0;
|
virtual uint64_t get_journal_size() = 0;
|
||||||
|
|
||||||
virtual uint32_t get_bitmap_granularity() = 0;
|
virtual uint32_t get_bitmap_granularity() = 0;
|
||||||
|
|
||||||
|
virtual uint64_t get_live_entries() = 0;
|
||||||
|
virtual uint64_t get_live_memory() = 0;
|
||||||
|
virtual uint64_t get_garbage_entries() = 0;
|
||||||
|
virtual uint64_t get_garbage_memory() = 0;
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -94,6 +94,9 @@ void blockstore_disk_t::parse_config(std::map<std::string, std::string> & config
|
|||||||
csum_block_size = parse_size(config["csum_block_size"]);
|
csum_block_size = parse_size(config["csum_block_size"]);
|
||||||
discard_on_start = config.find("discard_on_start") != config.end() &&
|
discard_on_start = config.find("discard_on_start") != config.end() &&
|
||||||
(config["discard_on_start"] == "true" || config["discard_on_start"] == "1" || config["discard_on_start"] == "yes");
|
(config["discard_on_start"] == "true" || config["discard_on_start"] == "1" || config["discard_on_start"] == "yes");
|
||||||
|
gc_on_start = config.find("gc_on_start") == config.end() ||
|
||||||
|
(config["gc_on_start"] == "true" || config["gc_on_start"] == "1" || config["gc_on_start"] == "yes");
|
||||||
|
skip_double_claim = (config["skip_double_claim"] == "true" || config["skip_double_claim"] == "1" || config["skip_double_claim"] == "yes");
|
||||||
min_discard_size = parse_size(config["min_discard_size"]);
|
min_discard_size = parse_size(config["min_discard_size"]);
|
||||||
if (!min_discard_size)
|
if (!min_discard_size)
|
||||||
min_discard_size = 1024*1024;
|
min_discard_size = 1024*1024;
|
||||||
@@ -514,7 +517,7 @@ void blockstore_disk_t::close_all()
|
|||||||
|
|
||||||
// Sadly DISCARD only works through ioctl(), but it seems to always block the device queue,
|
// Sadly DISCARD only works through ioctl(), but it seems to always block the device queue,
|
||||||
// so it's not a big deal that we can only run it synchronously.
|
// so it's not a big deal that we can only run it synchronously.
|
||||||
int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_free)
|
int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_used)
|
||||||
{
|
{
|
||||||
if (mock_mode)
|
if (mock_mode)
|
||||||
{
|
{
|
||||||
@@ -525,7 +528,7 @@ int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_free)
|
|||||||
uint64_t discarded = 0;
|
uint64_t discarded = 0;
|
||||||
for (; i <= block_count; i++)
|
for (; i <= block_count; i++)
|
||||||
{
|
{
|
||||||
if (i >= block_count || is_free(i))
|
if (i >= block_count || is_used(i))
|
||||||
{
|
{
|
||||||
if (i > j && (i-j)*data_block_size >= min_discard_size)
|
if (i > j && (i-j)*data_block_size >= min_discard_size)
|
||||||
{
|
{
|
||||||
@@ -538,17 +541,21 @@ int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_free)
|
|||||||
if (range[0] % discard_granularity)
|
if (range[0] % discard_granularity)
|
||||||
range[0] = range[0] + discard_granularity - (range[0] % discard_granularity);
|
range[0] = range[0] + discard_granularity - (range[0] % discard_granularity);
|
||||||
if (range[0] >= range[1])
|
if (range[0] >= range[1])
|
||||||
continue;
|
range[1] = 0;
|
||||||
range[1] -= range[0];
|
else
|
||||||
|
range[1] -= range[0];
|
||||||
}
|
}
|
||||||
r = ioctl(data_fd, BLKDISCARD, &range);
|
if (range[1] > 0)
|
||||||
if (r != 0)
|
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Failed to execute BLKDISCARD %ju+%ju on %s: %s (code %d)\n",
|
r = ioctl(data_fd, BLKDISCARD, &range);
|
||||||
range[0], range[1], data_device.c_str(), strerror(-r), r);
|
if (r != 0)
|
||||||
return -errno;
|
{
|
||||||
|
fprintf(stderr, "Failed to execute BLKDISCARD %ju+%ju on %s: %s (code %d)\n",
|
||||||
|
range[0], range[1], data_device.c_str(), strerror(-r), r);
|
||||||
|
return -errno;
|
||||||
|
}
|
||||||
|
discarded += range[1];
|
||||||
}
|
}
|
||||||
discarded += range[1];
|
|
||||||
}
|
}
|
||||||
j = i+1;
|
j = i+1;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -57,6 +57,10 @@ struct blockstore_disk_t
|
|||||||
bool inmemory_journal = true;
|
bool inmemory_journal = true;
|
||||||
// Data discard granularity and minimum size (for the sake of performance)
|
// Data discard granularity and minimum size (for the sake of performance)
|
||||||
bool discard_on_start = false;
|
bool discard_on_start = false;
|
||||||
|
// GC on start (new store)
|
||||||
|
bool gc_on_start = true;
|
||||||
|
// Skip double claim conflicts on start (new store, temporary until the bug is found)
|
||||||
|
bool skip_double_claim = false;
|
||||||
uint64_t min_discard_size = 1024*1024;
|
uint64_t min_discard_size = 1024*1024;
|
||||||
uint64_t discard_granularity = 0;
|
uint64_t discard_granularity = 0;
|
||||||
|
|
||||||
@@ -79,7 +83,7 @@ struct blockstore_disk_t
|
|||||||
void calc_lengths(bool skip_meta_check = false);
|
void calc_lengths(bool skip_meta_check = false);
|
||||||
void check_lengths();
|
void check_lengths();
|
||||||
void close_all();
|
void close_all();
|
||||||
int trim_data(std::function<bool(uint64_t)> is_free);
|
int trim_data(std::function<bool(uint64_t)> is_used);
|
||||||
|
|
||||||
inline uint64_t dirty_dyn_size(uint64_t offset, uint64_t len)
|
inline uint64_t dirty_dyn_size(uint64_t offset, uint64_t len)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -174,14 +174,18 @@ bool journal_flusher_co::loop()
|
|||||||
else if (wait_state == 19) goto resume_19;
|
else if (wait_state == 19) goto resume_19;
|
||||||
else if (wait_state == 20) goto resume_20;
|
else if (wait_state == 20) goto resume_20;
|
||||||
else if (wait_state == 21) goto resume_21;
|
else if (wait_state == 21) goto resume_21;
|
||||||
|
else if (wait_state == 22) goto resume_22;
|
||||||
|
else if (wait_state == 23) goto resume_23;
|
||||||
|
else if (wait_state == 24) goto resume_24;
|
||||||
|
else if (wait_state == 25) goto resume_25;
|
||||||
resume_0:
|
resume_0:
|
||||||
wait_state = 0;
|
wait_state = 0;
|
||||||
wait_count = 0;
|
wait_count = 0;
|
||||||
cur_oid = {};
|
cur_oid = {};
|
||||||
res = bs->heap->get_next_compact(cur_oid);
|
res = bs->heap->get_next_compact(cur_oid);
|
||||||
// Advance fsynced_lsn every <journal_trim_interval> intent writes
|
|
||||||
if ((bs->intent_write_counter >= bs->journal_trim_interval) && co_id == 0)
|
if ((bs->intent_write_counter >= bs->journal_trim_interval) && co_id == 0)
|
||||||
{
|
{
|
||||||
|
// Advance fsynced_lsn every <journal_trim_interval> intent writes
|
||||||
bs->intent_write_counter = 0;
|
bs->intent_write_counter = 0;
|
||||||
resume_17:
|
resume_17:
|
||||||
resume_18:
|
resume_18:
|
||||||
@@ -196,6 +200,7 @@ resume_21:
|
|||||||
if (res == ENOENT && flusher->force_start > 0 && co_id == 0 &&
|
if (res == ENOENT && flusher->force_start > 0 && co_id == 0 &&
|
||||||
(!bs->dsk.disable_journal_fsync || !bs->dsk.disable_meta_fsync || !bs->dsk.disable_data_fsync))
|
(!bs->dsk.disable_journal_fsync || !bs->dsk.disable_meta_fsync || !bs->dsk.disable_data_fsync))
|
||||||
{
|
{
|
||||||
|
// When under pressure, do an additional fsync to force entries to be marked compactable
|
||||||
flusher->active_flushers++;
|
flusher->active_flushers++;
|
||||||
resume_14:
|
resume_14:
|
||||||
resume_15:
|
resume_15:
|
||||||
@@ -259,11 +264,9 @@ resume_1:
|
|||||||
if (wr->type() == BS_HEAP_SMALL_WRITE ||
|
if (wr->type() == BS_HEAP_SMALL_WRITE ||
|
||||||
wr->type() == BS_HEAP_INTENT_WRITE && bs->dsk.csum_block_size > bs->dsk.bitmap_granularity)
|
wr->type() == BS_HEAP_INTENT_WRITE && bs->dsk.csum_block_size > bs->dsk.bitmap_granularity)
|
||||||
{
|
{
|
||||||
auto res = bs->prepare_read(read_vec, cur_obj, wr, 0, bs->dsk.data_block_size,
|
bs->prepare_read(read_vec, cur_obj, wr, 0, bs->dsk.data_block_size,
|
||||||
wr->type() == BS_HEAP_INTENT_WRITE && bs->dsk.csum_block_size > bs->dsk.bitmap_granularity && !bs->perfect_csum_update
|
wr->type() == BS_HEAP_INTENT_WRITE && bs->dsk.csum_block_size > bs->dsk.bitmap_granularity && !bs->perfect_csum_update
|
||||||
? COPY_BUF_SKIP_CSUM : 0);
|
? COPY_BUF_SKIP_CSUM : 0);
|
||||||
if (res > 0)
|
|
||||||
copy_count++;
|
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
if (!compact_info.compact_lsn)
|
if (!compact_info.compact_lsn)
|
||||||
@@ -273,30 +276,53 @@ resume_1:
|
|||||||
bs->heap->unlock_entry(cur_oid);
|
bs->heap->unlock_entry(cur_oid);
|
||||||
goto resume_0;
|
goto resume_0;
|
||||||
}
|
}
|
||||||
mem_or(new_bmp, compact_info.clean_wr->get_int_bitmap(bs->heap), bs->dsk.clean_entry_bitmap_size);
|
|
||||||
if (!bitmap_copied)
|
|
||||||
{
|
|
||||||
memcpy(new_ext_bmp, compact_info.clean_wr->get_ext_bitmap(bs->heap), bs->dsk.clean_entry_bitmap_size);
|
|
||||||
bitmap_copied = true;
|
|
||||||
}
|
|
||||||
if (bs->dsk.csum_block_size && bs->dsk.csum_block_size <= bs->dsk.bitmap_granularity)
|
|
||||||
{
|
|
||||||
memcpy(new_csums, compact_info.clean_wr->get_checksums(bs->heap), bs->dsk.data_block_size/bs->dsk.csum_block_size * (bs->dsk.data_csum_type & 0xFF));
|
|
||||||
for (size_t i = csum_copy.size(); i > 0; i--)
|
|
||||||
{
|
|
||||||
auto wr = csum_copy[i-1];
|
|
||||||
memcpy(new_csums + wr->small().offset/bs->dsk.csum_block_size*(bs->dsk.data_csum_type & 0xFF),
|
|
||||||
wr->get_checksums(bs->heap), wr->small().len/bs->dsk.csum_block_size*(bs->dsk.data_csum_type & 0xFF));
|
|
||||||
}
|
|
||||||
csum_copy.clear();
|
|
||||||
}
|
|
||||||
clean_loc = compact_info.clean_wr->big_location(bs->heap);
|
|
||||||
flusher->active_flushers++;
|
flusher->active_flushers++;
|
||||||
if (bs->log_level > 10)
|
for (i = 0; i < read_vec.size(); i++)
|
||||||
{
|
{
|
||||||
printf("Compacting %jx:%jx v%ju..v%ju / l%ju..l%ju (%d writes)\n", cur_oid.inode, cur_oid.stripe,
|
if ((read_vec[i].copy_flags & COPY_BUF_JOURNAL) &&
|
||||||
compact_info.clean_wr->version, compact_info.compact_version,
|
!(read_vec[i].copy_flags & COPY_BUF_COALESCED))
|
||||||
compact_info.clean_wr->lsn, compact_info.compact_lsn, copy_count);
|
{
|
||||||
|
copy_count++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (copy_count > 0 && !bs->dsk.disable_data_fsync)
|
||||||
|
{
|
||||||
|
init_fsync_data();
|
||||||
|
}
|
||||||
|
if (compact_info.do_delete)
|
||||||
|
{
|
||||||
|
if (bs->log_level > 10)
|
||||||
|
{
|
||||||
|
printf("Compacting %jx:%jx up to l%ju (delete)\n", cur_oid.inode, cur_oid.stripe, compact_info.compact_lsn);
|
||||||
|
}
|
||||||
|
clean_loc = UINT64_MAX;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
if (bs->log_level > 10)
|
||||||
|
{
|
||||||
|
printf("Compacting %jx:%jx v%ju..v%ju / l%ju..l%ju (%d writes)\n", cur_oid.inode, cur_oid.stripe,
|
||||||
|
compact_info.clean_wr->version, compact_info.compact_version,
|
||||||
|
compact_info.clean_wr->lsn, compact_info.compact_lsn, copy_count);
|
||||||
|
}
|
||||||
|
mem_or(new_bmp, compact_info.clean_wr->get_int_bitmap(bs->heap), bs->dsk.clean_entry_bitmap_size);
|
||||||
|
if (!bitmap_copied)
|
||||||
|
{
|
||||||
|
memcpy(new_ext_bmp, compact_info.clean_wr->get_ext_bitmap(bs->heap), bs->dsk.clean_entry_bitmap_size);
|
||||||
|
bitmap_copied = true;
|
||||||
|
}
|
||||||
|
if (bs->dsk.csum_block_size && bs->dsk.csum_block_size <= bs->dsk.bitmap_granularity)
|
||||||
|
{
|
||||||
|
memcpy(new_csums, compact_info.clean_wr->get_checksums(bs->heap), bs->dsk.data_block_size/bs->dsk.csum_block_size * (bs->dsk.data_csum_type & 0xFF));
|
||||||
|
for (size_t i = csum_copy.size(); i > 0; i--)
|
||||||
|
{
|
||||||
|
auto wr = csum_copy[i-1];
|
||||||
|
memcpy(new_csums + wr->small().offset/bs->dsk.csum_block_size*(bs->dsk.data_csum_type & 0xFF),
|
||||||
|
wr->get_checksums(bs->heap), wr->small().len/bs->dsk.csum_block_size*(bs->dsk.data_csum_type & 0xFF));
|
||||||
|
}
|
||||||
|
csum_copy.clear();
|
||||||
|
}
|
||||||
|
clean_loc = compact_info.clean_wr->big_location(bs->heap);
|
||||||
}
|
}
|
||||||
overwrite_start = overwrite_end = 0;
|
overwrite_start = overwrite_end = 0;
|
||||||
if (read_vec.size() > 0)
|
if (read_vec.size() > 0)
|
||||||
@@ -336,6 +362,13 @@ resume_3:
|
|||||||
if (res == ENOENT || res == EDOM)
|
if (res == ENOENT || res == EDOM)
|
||||||
{
|
{
|
||||||
// Abort compaction
|
// Abort compaction
|
||||||
|
abort_compact:
|
||||||
|
if (copy_count > 0 && !bs->dsk.disable_data_fsync)
|
||||||
|
{
|
||||||
|
cur_sync->member_count--;
|
||||||
|
if (cur_sync->member_count > 0)
|
||||||
|
bs->ringloop->wakeup();
|
||||||
|
}
|
||||||
flusher->flushing.erase(cur_oid);
|
flusher->flushing.erase(cur_oid);
|
||||||
bs->heap->unlock_entry(cur_oid);
|
bs->heap->unlock_entry(cur_oid);
|
||||||
flusher->active_flushers--;
|
flusher->active_flushers--;
|
||||||
@@ -349,10 +382,7 @@ resume_4:
|
|||||||
if (res == ENOENT)
|
if (res == ENOENT)
|
||||||
{
|
{
|
||||||
// Abort compaction
|
// Abort compaction
|
||||||
flusher->flushing.erase(cur_oid);
|
goto abort_compact;
|
||||||
bs->heap->unlock_entry(cur_oid);
|
|
||||||
flusher->active_flushers--;
|
|
||||||
goto resume_0;
|
|
||||||
}
|
}
|
||||||
if (res == EAGAIN)
|
if (res == EAGAIN)
|
||||||
{
|
{
|
||||||
@@ -381,14 +411,14 @@ resume_9:
|
|||||||
for (i = 0; i < read_vec.size(); i++)
|
for (i = 0; i < read_vec.size(); i++)
|
||||||
{
|
{
|
||||||
if ((read_vec[i].copy_flags & COPY_BUF_JOURNAL) &&
|
if ((read_vec[i].copy_flags & COPY_BUF_JOURNAL) &&
|
||||||
!(read_vec[i].copy_flags & COPY_BUF_COALESCED) ||
|
!(read_vec[i].copy_flags & COPY_BUF_COALESCED))
|
||||||
(read_vec[i].copy_flags & COPY_BUF_PADDED)) // FIXME Shit, simplify these flags
|
|
||||||
{
|
{
|
||||||
assert(read_vec[i].buf);
|
assert(read_vec[i].buf);
|
||||||
await_sqe(10);
|
await_sqe(10);
|
||||||
data->iov = (struct iovec){ read_vec[i].buf + (read_vec[i].copy_flags & COPY_BUF_PADDED
|
data->iov = (struct iovec){ read_vec[i].buf + (read_vec[i].copy_flags & COPY_BUF_PADDED
|
||||||
? read_vec[i].offset - read_vec[i].disk_offset : 0), (size_t)read_vec[i].len };
|
? read_vec[i].offset - read_vec[i].disk_offset : 0), (size_t)read_vec[i].len };
|
||||||
data->callback = simple_callback_w;
|
data->callback = simple_callback_w;
|
||||||
|
assert(clean_loc + read_vec[i].offset + data->iov.iov_len <= bs->dsk.block_count*bs->dsk.data_block_size);
|
||||||
io_uring_prep_writev(sqe, bs->dsk.data_fd, &data->iov, 1, bs->dsk.data_offset + clean_loc + read_vec[i].offset);
|
io_uring_prep_writev(sqe, bs->dsk.data_fd, &data->iov, 1, bs->dsk.data_offset + clean_loc + read_vec[i].offset);
|
||||||
wait_count++;
|
wait_count++;
|
||||||
}
|
}
|
||||||
@@ -399,6 +429,17 @@ resume_11:
|
|||||||
wait_state = 11;
|
wait_state = 11;
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
if (copy_count > 0 && !bs->dsk.disable_data_fsync)
|
||||||
|
{
|
||||||
|
resume_22:
|
||||||
|
resume_23:
|
||||||
|
resume_24:
|
||||||
|
resume_25:
|
||||||
|
if (!fsync_data(22))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
// Lock is only needed to prevent freeing the big_write because we overwrite it...
|
// Lock is only needed to prevent freeing the big_write because we overwrite it...
|
||||||
bs->heap->unlock_entry(cur_oid);
|
bs->heap->unlock_entry(cur_oid);
|
||||||
// Mark the object compacted, but don't free and remove small_writes
|
// Mark the object compacted, but don't free and remove small_writes
|
||||||
@@ -408,12 +449,14 @@ resume_11:
|
|||||||
if (!cur_obj)
|
if (!cur_obj)
|
||||||
{
|
{
|
||||||
// Abort compaction
|
// Abort compaction
|
||||||
|
flusher->active_flushers--;
|
||||||
flusher->flushing.erase(cur_oid);
|
flusher->flushing.erase(cur_oid);
|
||||||
goto resume_0;
|
goto resume_0;
|
||||||
}
|
}
|
||||||
if (!calc_block_checksums())
|
if (!calc_block_checksums())
|
||||||
{
|
{
|
||||||
// Abort compaction
|
// Abort compaction
|
||||||
|
flusher->active_flushers--;
|
||||||
flusher->flushing.erase(cur_oid);
|
flusher->flushing.erase(cur_oid);
|
||||||
goto resume_0;
|
goto resume_0;
|
||||||
}
|
}
|
||||||
@@ -422,6 +465,7 @@ resume_11:
|
|||||||
if (res == EBUSY)
|
if (res == EBUSY)
|
||||||
{
|
{
|
||||||
// Abort compaction, object is already overwritten by something else
|
// Abort compaction, object is already overwritten by something else
|
||||||
|
flusher->active_flushers--;
|
||||||
flusher->flushing.erase(cur_oid);
|
flusher->flushing.erase(cur_oid);
|
||||||
goto resume_0;
|
goto resume_0;
|
||||||
}
|
}
|
||||||
@@ -586,13 +630,13 @@ int journal_flusher_co::check_and_punch_checksums()
|
|||||||
bs->heap->calc_block_checksums((uint32_t*)(new_csums+csum_off), vec.buf, punch_bmp, vec.offset, vec.offset+vec.len, true, NULL);
|
bs->heap->calc_block_checksums((uint32_t*)(new_csums+csum_off), vec.buf, punch_bmp, vec.offset, vec.offset+vec.len, true, NULL);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Modified, we should add_punch_holes and then write the block to disk
|
// Modified, we should punch_holes and then write the block to disk
|
||||||
return EBUSY;
|
return EBUSY;
|
||||||
}
|
}
|
||||||
|
|
||||||
bool journal_flusher_co::calc_block_checksums()
|
bool journal_flusher_co::calc_block_checksums()
|
||||||
{
|
{
|
||||||
if (bs->dsk.csum_block_size <= bs->dsk.bitmap_granularity)
|
if (bs->dsk.csum_block_size <= bs->dsk.bitmap_granularity || compact_info.do_delete)
|
||||||
{
|
{
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
@@ -699,6 +743,67 @@ resume_1:
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void journal_flusher_co::init_fsync_data()
|
||||||
|
{
|
||||||
|
cur_sync = flusher->data_syncs.begin();
|
||||||
|
if (cur_sync == flusher->data_syncs.end() || cur_sync->ready_count > 0)
|
||||||
|
{
|
||||||
|
cur_sync = flusher->data_syncs.emplace(cur_sync);
|
||||||
|
}
|
||||||
|
cur_sync->member_count++;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool journal_flusher_co::fsync_data(int wait_base)
|
||||||
|
{
|
||||||
|
if (wait_state == wait_base)
|
||||||
|
goto resume_0;
|
||||||
|
else if (wait_state == wait_base+1)
|
||||||
|
goto resume_1;
|
||||||
|
else if (wait_state == wait_base+2)
|
||||||
|
goto resume_2;
|
||||||
|
else if (wait_state == wait_base+3)
|
||||||
|
goto resume_3;
|
||||||
|
cur_sync->ready_count++;
|
||||||
|
resume_0:
|
||||||
|
if (cur_sync->ready_count < cur_sync->member_count)
|
||||||
|
{
|
||||||
|
wait_state = wait_base;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (!cur_sync->sent)
|
||||||
|
{
|
||||||
|
// Sync batch is ready. Do it.
|
||||||
|
await_sqe(1);
|
||||||
|
data->iov = { 0 };
|
||||||
|
data->callback = simple_callback_w;
|
||||||
|
io_uring_prep_fsync(sqe, bs->dsk.data_fd, IORING_FSYNC_DATASYNC);
|
||||||
|
cur_sync->sent = true;
|
||||||
|
wait_count++;
|
||||||
|
resume_2:
|
||||||
|
if (wait_count > 0)
|
||||||
|
{
|
||||||
|
wait_state = wait_base+2;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
cur_sync->done = true;
|
||||||
|
// Wake up other flushers
|
||||||
|
bs->ringloop->wakeup();
|
||||||
|
}
|
||||||
|
resume_3:
|
||||||
|
if (!cur_sync->done)
|
||||||
|
{
|
||||||
|
wait_state = wait_base+3;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
cur_sync->done_count++;
|
||||||
|
if (cur_sync->done_count >= cur_sync->member_count)
|
||||||
|
{
|
||||||
|
flusher->data_syncs.erase(cur_sync);
|
||||||
|
cur_sync = flusher->data_syncs.end();
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
bool journal_flusher_co::fsync_meta(int wait_base)
|
bool journal_flusher_co::fsync_meta(int wait_base)
|
||||||
{
|
{
|
||||||
if (wait_state == wait_base) goto resume_0;
|
if (wait_state == wait_base) goto resume_0;
|
||||||
|
|||||||
@@ -25,6 +25,15 @@ struct flusher_meta_write_t
|
|||||||
std::map<uint64_t, meta_sector_t>::iterator it;
|
std::map<uint64_t, meta_sector_t>::iterator it;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
struct flusher_data_sync_t
|
||||||
|
{
|
||||||
|
int member_count = 0;
|
||||||
|
int ready_count = 0;
|
||||||
|
int done_count = 0;
|
||||||
|
bool sent = false;
|
||||||
|
bool done = false;
|
||||||
|
};
|
||||||
|
|
||||||
class journal_flusher_t;
|
class journal_flusher_t;
|
||||||
|
|
||||||
// Journal flusher coroutine
|
// Journal flusher coroutine
|
||||||
@@ -58,6 +67,7 @@ class journal_flusher_co
|
|||||||
int i, res;
|
int i, res;
|
||||||
bool read_to_fill_incomplete;
|
bool read_to_fill_incomplete;
|
||||||
int copy_count;
|
int copy_count;
|
||||||
|
std::list<flusher_data_sync_t>::iterator cur_sync;
|
||||||
|
|
||||||
friend class journal_flusher_t;
|
friend class journal_flusher_t;
|
||||||
|
|
||||||
@@ -68,6 +78,8 @@ class journal_flusher_co
|
|||||||
bool calc_block_checksums();
|
bool calc_block_checksums();
|
||||||
bool write_meta_block(int wait_base);
|
bool write_meta_block(int wait_base);
|
||||||
bool read_buffered(int wait_base);
|
bool read_buffered(int wait_base);
|
||||||
|
void init_fsync_data();
|
||||||
|
bool fsync_data(int wait_base);
|
||||||
bool fsync_meta(int wait_base);
|
bool fsync_meta(int wait_base);
|
||||||
bool fsync_buffer(int wait_base);
|
bool fsync_buffer(int wait_base);
|
||||||
bool trim_lsn(int wait_base);
|
bool trim_lsn(int wait_base);
|
||||||
@@ -88,6 +100,7 @@ class journal_flusher_t
|
|||||||
|
|
||||||
robin_hood::unordered_flat_set<object_id> flushing;
|
robin_hood::unordered_flat_set<object_id> flushing;
|
||||||
int active_flushers = 0;
|
int active_flushers = 0;
|
||||||
|
std::list<flusher_data_sync_t> data_syncs;
|
||||||
int wanting_meta_fsync = 0;
|
int wanting_meta_fsync = 0;
|
||||||
bool fsyncing_meta = false;
|
bool fsyncing_meta = false;
|
||||||
int syncing_buffer = 0;
|
int syncing_buffer = 0;
|
||||||
|
|||||||
+555
-252
File diff suppressed because it is too large
Load Diff
@@ -57,11 +57,11 @@ struct __attribute__((__packed__)) heap_entry_t
|
|||||||
inline heap_small_write_t& small() { return *(heap_small_write_t*)this; }
|
inline heap_small_write_t& small() { return *(heap_small_write_t*)this; }
|
||||||
inline heap_big_write_t& big() { return *(heap_big_write_t*)this; }
|
inline heap_big_write_t& big() { return *(heap_big_write_t*)this; }
|
||||||
inline heap_big_intent_t& big_intent() { return *(heap_big_intent_t*)this; }
|
inline heap_big_intent_t& big_intent() { return *(heap_big_intent_t*)this; }
|
||||||
bool is_garbage();
|
bool is_garbage() const;
|
||||||
void set_garbage();
|
void set_garbage();
|
||||||
bool is_overwrite();
|
bool is_overwrite() const;
|
||||||
bool is_compactable();
|
bool is_compactable() const;
|
||||||
bool is_before(heap_entry_t *other);
|
bool is_before(const heap_entry_t *other) const;
|
||||||
uint32_t get_size(blockstore_heap_t *heap);
|
uint32_t get_size(blockstore_heap_t *heap);
|
||||||
uint8_t *get_ext_bitmap(blockstore_heap_t *heap);
|
uint8_t *get_ext_bitmap(blockstore_heap_t *heap);
|
||||||
uint8_t *get_int_bitmap(blockstore_heap_t *heap);
|
uint8_t *get_int_bitmap(blockstore_heap_t *heap);
|
||||||
@@ -117,10 +117,13 @@ struct heap_object_mvcc_t
|
|||||||
|
|
||||||
struct heap_block_info_t
|
struct heap_block_info_t
|
||||||
{
|
{
|
||||||
uint32_t used_space = 0;
|
struct __attribute__((__packed__))
|
||||||
|
{
|
||||||
|
uint32_t used_space = 0;
|
||||||
|
uint32_t garbage_space = 0;
|
||||||
|
};
|
||||||
uint64_t mod_lsn = 0, mod_lsn_to = 0; // only 1 block write of LSN sequence is allowed at a moment
|
uint64_t mod_lsn = 0, mod_lsn_to = 0; // only 1 block write of LSN sequence is allowed at a moment
|
||||||
bool is_writing: 1;
|
bool is_writing = false;
|
||||||
bool has_garbage: 1;
|
|
||||||
std::vector<heap_list_item_t*> entries;
|
std::vector<heap_list_item_t*> entries;
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -155,6 +158,16 @@ struct heap_li_equal
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
struct heap_recheck_state_t
|
||||||
|
{
|
||||||
|
heap_entry_t *obj = NULL;
|
||||||
|
heap_entry_t *next_wr = NULL;
|
||||||
|
size_t total_reads = 0;
|
||||||
|
size_t sent_reads = 0;
|
||||||
|
size_t checked_reads = 0;
|
||||||
|
heap_entry_t *bad_wr = NULL;
|
||||||
|
};
|
||||||
|
|
||||||
using i64hash_t = robin_hood::hash<uint64_t>;
|
using i64hash_t = robin_hood::hash<uint64_t>;
|
||||||
using heap_inode_map_t = robin_hood::unordered_flat_set<heap_list_item_t*, heap_li_hash, heap_li_equal, 88>;
|
using heap_inode_map_t = robin_hood::unordered_flat_set<heap_list_item_t*, heap_li_hash, heap_li_equal, 88>;
|
||||||
using heap_block_index_t = robin_hood::unordered_flat_map<uint64_t,
|
using heap_block_index_t = robin_hood::unordered_flat_map<uint64_t,
|
||||||
@@ -184,6 +197,11 @@ class blockstore_heap_t
|
|||||||
uint64_t buffer_area_used_space = 0;
|
uint64_t buffer_area_used_space = 0;
|
||||||
uint64_t data_used_space = 0;
|
uint64_t data_used_space = 0;
|
||||||
|
|
||||||
|
uint64_t live_entries = 0;
|
||||||
|
uint64_t live_memory = 0;
|
||||||
|
uint64_t garbage_entries = 0;
|
||||||
|
uint64_t garbage_memory = 0;
|
||||||
|
|
||||||
uint64_t next_lsn = 0;
|
uint64_t next_lsn = 0;
|
||||||
uint32_t last_allocated_block = UINT32_MAX;
|
uint32_t last_allocated_block = UINT32_MAX;
|
||||||
heap_mvcc_map_t object_mvcc;
|
heap_mvcc_map_t object_mvcc;
|
||||||
@@ -200,9 +218,12 @@ class blockstore_heap_t
|
|||||||
|
|
||||||
bool marked_used_blocks = false;
|
bool marked_used_blocks = false;
|
||||||
bool recheck_queue_filled = false;
|
bool recheck_queue_filled = false;
|
||||||
std::vector<heap_list_item_t*> loaded_list_items;
|
std::vector<heap_list_item_t*> postponed_items;
|
||||||
|
std::vector<heap_list_item_t*> init_erase_items;
|
||||||
std::set<uint32_t> recheck_modified_blocks;
|
std::set<uint32_t> recheck_modified_blocks;
|
||||||
std::deque<heap_entry_t*> recheck_queue;
|
std::deque<heap_entry_t*> recheck_queue;
|
||||||
|
std::map<heap_entry_t*, heap_recheck_state_t> recheck_states;
|
||||||
|
size_t recheck_pending_reads = 0;
|
||||||
int recheck_in_progress = 0;
|
int recheck_in_progress = 0;
|
||||||
bool in_recheck = false;
|
bool in_recheck = false;
|
||||||
std::function<void(bool is_data, uint64_t offset, uint64_t len, uint8_t* buf, std::function<void()>)> recheck_cb;
|
std::function<void(bool is_data, uint64_t offset, uint64_t len, uint8_t* buf, std::function<void()>)> recheck_cb;
|
||||||
@@ -211,14 +232,22 @@ class blockstore_heap_t
|
|||||||
uint64_t get_pg_id(inode_t inode, uint64_t stripe);
|
uint64_t get_pg_id(inode_t inode, uint64_t stripe);
|
||||||
bool validate_object(heap_entry_t *obj);
|
bool validate_object(heap_entry_t *obj);
|
||||||
void fill_recheck_queue();
|
void fill_recheck_queue();
|
||||||
|
void recheck_drop_entries(heap_entry_t *obj, heap_entry_t *bad_wr);
|
||||||
|
void recheck_start_reads(heap_recheck_state_t *st);
|
||||||
int mark_used_blocks();
|
int mark_used_blocks();
|
||||||
|
void init_free_bad_entry(heap_entry_t *wr);
|
||||||
|
void init_erase_bad_entry(heap_list_item_t *li);
|
||||||
|
bool init_erase_double_claim(heap_list_item_t *prev_li, heap_list_item_t *cur_li);
|
||||||
|
void recheck_full_gc();
|
||||||
void recheck_buffer(heap_entry_t *cwr, uint8_t *buf);
|
void recheck_buffer(heap_entry_t *cwr, uint8_t *buf);
|
||||||
void defragment_block(uint32_t block_num);
|
void defragment_block(uint32_t block_num);
|
||||||
void reshard_add(heap_reshard_state_t *st, heap_list_item_t *li);
|
void reshard_add(heap_reshard_state_t *st, heap_list_item_t *li);
|
||||||
|
|
||||||
void gc_block(heap_block_info_t & inf);
|
void gc_block(heap_block_info_t & inf);
|
||||||
int allocate_entry(uint32_t entry_size, uint32_t *block_num, bool allow_last_free);
|
int allocate_entry(uint32_t entry_size, uint32_t *block_num, bool allow_last_free);
|
||||||
void insert_list_item(heap_list_item_t *li);
|
void insert_list_items(heap_list_item_t** v, size_t count, bool postpone);
|
||||||
|
void remove_list_item(heap_list_item_t *li);
|
||||||
|
void unlink_list_item(heap_list_item_t *li);
|
||||||
int add_entry(uint32_t wr_size, uint32_t *modified_block, bool allow_last_free,
|
int add_entry(uint32_t wr_size, uint32_t *modified_block, bool allow_last_free,
|
||||||
bool explicit_complete, std::function<void(heap_entry_t *wr)> fill_entry);
|
bool explicit_complete, std::function<void(heap_entry_t *wr)> fill_entry);
|
||||||
int add_simple(heap_entry_t *obj, uint64_t version, uint32_t *modified_block, uint32_t entry_type);
|
int add_simple(heap_entry_t *obj, uint64_t version, uint32_t *modified_block, uint32_t entry_type);
|
||||||
@@ -333,7 +362,7 @@ public:
|
|||||||
|
|
||||||
// get metadata block data buffer and used space
|
// get metadata block data buffer and used space
|
||||||
void get_meta_block(uint32_t block_num, uint8_t *buffer);
|
void get_meta_block(uint32_t block_num, uint8_t *buffer);
|
||||||
void fill_block_empty_space(uint8_t *buffer, uint32_t pos);
|
void fill_block_empty_space(uint8_t *buffer, uint64_t pos);
|
||||||
uint32_t get_meta_block_used_space(uint32_t block_num);
|
uint32_t get_meta_block_used_space(uint32_t block_num);
|
||||||
|
|
||||||
// get space usage statistics
|
// get space usage statistics
|
||||||
@@ -345,6 +374,10 @@ public:
|
|||||||
uint32_t get_compact_queue_size();
|
uint32_t get_compact_queue_size();
|
||||||
uint32_t get_to_compact_count();
|
uint32_t get_to_compact_count();
|
||||||
uint64_t get_compacted_count();
|
uint64_t get_compacted_count();
|
||||||
|
uint64_t get_live_entries();
|
||||||
|
uint64_t get_live_memory();
|
||||||
|
uint64_t get_garbage_entries();
|
||||||
|
uint64_t get_garbage_memory();
|
||||||
|
|
||||||
uint64_t entry_pos(uint32_t block_num, uint32_t offset);
|
uint64_t entry_pos(uint32_t block_num, uint32_t offset);
|
||||||
heap_entry_t *entry_from_pos(uint64_t entry_pos, bool allow_unallocated = false);
|
heap_entry_t *entry_from_pos(uint64_t entry_pos, bool allow_unallocated = false);
|
||||||
|
|||||||
@@ -101,6 +101,7 @@ void blockstore_impl_t::loop()
|
|||||||
unsigned initial_ring_space = ringloop->space_left();
|
unsigned initial_ring_space = ringloop->space_left();
|
||||||
int op_idx = 0, new_idx = 0;
|
int op_idx = 0, new_idx = 0;
|
||||||
bool has_unfinished_writes = false;
|
bool has_unfinished_writes = false;
|
||||||
|
bool has_unfinished_sync = false;
|
||||||
for (; op_idx < submit_queue.size(); op_idx++, new_idx++)
|
for (; op_idx < submit_queue.size(); op_idx++, new_idx++)
|
||||||
{
|
{
|
||||||
auto op = submit_queue[op_idx];
|
auto op = submit_queue[op_idx];
|
||||||
@@ -138,7 +139,13 @@ void blockstore_impl_t::loop()
|
|||||||
else if (op->opcode == BS_OP_SYNC)
|
else if (op->opcode == BS_OP_SYNC)
|
||||||
{
|
{
|
||||||
// syncs only completed writes, so doesn't have to be blocked by anything
|
// syncs only completed writes, so doesn't have to be blocked by anything
|
||||||
wr_st = continue_sync(op);
|
if (!has_unfinished_sync)
|
||||||
|
{
|
||||||
|
wr_st = continue_sync(op);
|
||||||
|
has_unfinished_sync = (wr_st != 2);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
wr_st = 0;
|
||||||
}
|
}
|
||||||
else if (op->opcode == BS_OP_STABLE || op->opcode == BS_OP_ROLLBACK)
|
else if (op->opcode == BS_OP_STABLE || op->opcode == BS_OP_ROLLBACK)
|
||||||
{
|
{
|
||||||
@@ -154,9 +161,7 @@ void blockstore_impl_t::loop()
|
|||||||
wr_st = 2;
|
wr_st = 2;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
|
||||||
wr_st = 0;
|
wr_st = 0;
|
||||||
}
|
|
||||||
}
|
}
|
||||||
if (wr_st == 2)
|
if (wr_st == 2)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -117,9 +117,12 @@ public:
|
|||||||
|
|
||||||
journal_flusher_t *flusher;
|
journal_flusher_t *flusher;
|
||||||
int write_iodepth = 0;
|
int write_iodepth = 0;
|
||||||
int inflight_big = 0;
|
|
||||||
int intent_write_counter = 0;
|
int intent_write_counter = 0;
|
||||||
bool fsyncing_data = false;
|
uint64_t data_fsync_next = 0;
|
||||||
|
uint64_t data_fsync_cur = 0;
|
||||||
|
uint64_t data_fsync_sent = 0;
|
||||||
|
uint64_t data_fsync_done = 0;
|
||||||
|
std::deque<bool> data_fsyncs;
|
||||||
|
|
||||||
bool live = false, queue_stall = false;
|
bool live = false, queue_stall = false;
|
||||||
ring_loop_i *ringloop = NULL;
|
ring_loop_i *ringloop = NULL;
|
||||||
@@ -229,4 +232,9 @@ public:
|
|||||||
uint64_t get_free_block_count();
|
uint64_t get_free_block_count();
|
||||||
inline uint32_t get_bitmap_granularity() { return dsk.bitmap_granularity; }
|
inline uint32_t get_bitmap_granularity() { return dsk.bitmap_granularity; }
|
||||||
inline uint64_t get_journal_size() { return dsk.journal_len; }
|
inline uint64_t get_journal_size() { return dsk.journal_len; }
|
||||||
|
|
||||||
|
inline uint64_t get_live_entries() { return heap->get_live_entries(); }
|
||||||
|
inline uint64_t get_live_memory() { return heap->get_live_memory(); }
|
||||||
|
inline uint64_t get_garbage_entries() { return heap->get_garbage_entries(); }
|
||||||
|
inline uint64_t get_garbage_memory() { return heap->get_garbage_memory(); }
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -10,7 +10,6 @@
|
|||||||
#define INIT_META_EMPTY 0
|
#define INIT_META_EMPTY 0
|
||||||
#define INIT_META_READING 1
|
#define INIT_META_READING 1
|
||||||
#define INIT_META_READ_DONE 2
|
#define INIT_META_READ_DONE 2
|
||||||
#define INIT_META_WRITING 3
|
|
||||||
|
|
||||||
#define GET_SQE() \
|
#define GET_SQE() \
|
||||||
sqe = bs->get_sqe();\
|
sqe = bs->get_sqe();\
|
||||||
@@ -23,14 +22,15 @@ blockstore_init_meta::blockstore_init_meta(blockstore_impl_t *bs)
|
|||||||
this->bs = bs;
|
this->bs = bs;
|
||||||
}
|
}
|
||||||
|
|
||||||
void blockstore_init_meta::handle_event(ring_data_t *data, int buf_num)
|
void blockstore_init_meta::handle_event(ring_data_t *data, int buf_num, const char *op)
|
||||||
{
|
{
|
||||||
if (data->res < 0)
|
if (data->res != data->iov.iov_len)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(
|
throw std::runtime_error(strprintf(
|
||||||
std::string("read metadata failed at offset ") + std::to_string(buf_num >= 0 ? bufs[buf_num].offset : last_read_offset) +
|
"%s failed at offset %ju: got %s (code %d), but expected %zu",
|
||||||
std::string(": ") + strerror(-data->res)
|
op, (buf_num >= 0 ? bufs[buf_num].offset : last_read_offset), strerror(-data->res),
|
||||||
);
|
data->res, data->iov.iov_len
|
||||||
|
));
|
||||||
}
|
}
|
||||||
if (buf_num >= 0)
|
if (buf_num >= 0)
|
||||||
{
|
{
|
||||||
@@ -60,7 +60,7 @@ int blockstore_init_meta::loop()
|
|||||||
GET_SQE();
|
GET_SQE();
|
||||||
last_read_offset = 0;
|
last_read_offset = 0;
|
||||||
data->iov = { bs->meta_superblock, (size_t)bs->dsk.meta_block_size };
|
data->iov = { bs->meta_superblock, (size_t)bs->dsk.meta_block_size };
|
||||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "read metadata header"); };
|
||||||
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
||||||
bs->ringloop->submit();
|
bs->ringloop->submit();
|
||||||
submitted++;
|
submitted++;
|
||||||
@@ -72,25 +72,19 @@ resume_1:
|
|||||||
}
|
}
|
||||||
if (is_zero((uint64_t*)bs->meta_superblock, bs->dsk.meta_block_size))
|
if (is_zero((uint64_t*)bs->meta_superblock, bs->dsk.meta_block_size))
|
||||||
{
|
{
|
||||||
{
|
assert(bs->dsk.meta_format == BLOCKSTORE_META_FORMAT_HEAP);
|
||||||
blockstore_meta_header_v3_t *hdr = (blockstore_meta_header_v3_t *)bs->meta_superblock;
|
blockstore_meta_header_v3_t *hdr = (blockstore_meta_header_v3_t *)bs->meta_superblock;
|
||||||
hdr->zero = 0;
|
hdr->zero = 0;
|
||||||
hdr->magic = BLOCKSTORE_META_MAGIC_V1;
|
hdr->magic = BLOCKSTORE_META_MAGIC_V1;
|
||||||
hdr->version = bs->dsk.meta_format;
|
hdr->version = bs->dsk.meta_format;
|
||||||
hdr->meta_block_size = bs->dsk.meta_block_size;
|
hdr->meta_block_size = bs->dsk.meta_block_size;
|
||||||
hdr->data_block_size = bs->dsk.data_block_size;
|
hdr->data_block_size = bs->dsk.data_block_size;
|
||||||
hdr->bitmap_granularity = bs->dsk.bitmap_granularity;
|
hdr->bitmap_granularity = bs->dsk.bitmap_granularity;
|
||||||
if (bs->dsk.meta_format >= BLOCKSTORE_META_FORMAT_V2)
|
hdr->completed_lsn = 0;
|
||||||
{
|
hdr->data_csum_type = bs->dsk.data_csum_type;
|
||||||
hdr->data_csum_type = bs->dsk.data_csum_type;
|
hdr->csum_block_size = bs->dsk.csum_block_size;
|
||||||
hdr->csum_block_size = bs->dsk.csum_block_size;
|
hdr->meta_area_size = bs->dsk.meta_area_size;
|
||||||
}
|
hdr->set_crc32c();
|
||||||
if (bs->dsk.meta_format >= BLOCKSTORE_META_FORMAT_HEAP)
|
|
||||||
{
|
|
||||||
hdr->meta_area_size = bs->dsk.meta_area_size;
|
|
||||||
}
|
|
||||||
hdr->set_crc32c();
|
|
||||||
}
|
|
||||||
if (bs->readonly)
|
if (bs->readonly)
|
||||||
{
|
{
|
||||||
printf("Skipping metadata initialization because blockstore is readonly\n");
|
printf("Skipping metadata initialization because blockstore is readonly\n");
|
||||||
@@ -98,21 +92,8 @@ resume_1:
|
|||||||
else
|
else
|
||||||
{
|
{
|
||||||
printf("Initializing metadata area\n");
|
printf("Initializing metadata area\n");
|
||||||
GET_SQE();
|
|
||||||
last_read_offset = 0;
|
|
||||||
data->iov = (struct iovec){ bs->meta_superblock, (size_t)bs->dsk.meta_block_size };
|
|
||||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
|
||||||
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
|
||||||
bs->ringloop->submit();
|
|
||||||
submitted++;
|
|
||||||
resume_2:
|
|
||||||
if (submitted > 0)
|
|
||||||
{
|
|
||||||
wait_state = 2;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
zero_on_init = true;
|
|
||||||
}
|
}
|
||||||
|
zero_on_init = true;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -153,9 +134,17 @@ resume_1:
|
|||||||
);
|
);
|
||||||
exit(1);
|
exit(1);
|
||||||
}
|
}
|
||||||
|
uint32_t csum = hdr->header_csum;
|
||||||
|
hdr->header_csum = 0;
|
||||||
|
if (crc32c(0, hdr, sizeof(*hdr)) != csum)
|
||||||
|
{
|
||||||
|
printf("Metadata header is corrupt (checksum mismatch).\n");
|
||||||
|
exit(1);
|
||||||
|
}
|
||||||
|
hdr->header_csum = csum;
|
||||||
}
|
}
|
||||||
bs->heap->start_load(((blockstore_meta_header_v3_t *)bs->meta_superblock)->completed_lsn);
|
bs->heap->start_load(((blockstore_meta_header_v3_t *)bs->meta_superblock)->completed_lsn);
|
||||||
if (bs->dsk.inmemory_journal)
|
if (bs->dsk.inmemory_journal && !zero_on_init)
|
||||||
{
|
{
|
||||||
// Read buffer area
|
// Read buffer area
|
||||||
printf("Reading buffered data\n");
|
printf("Reading buffered data\n");
|
||||||
@@ -167,7 +156,7 @@ resume_1:
|
|||||||
bs->buffer_area + md_offset,
|
bs->buffer_area + md_offset,
|
||||||
(size_t)(bs->dsk.journal_len - md_offset < bs->metadata_buf_size ? bs->dsk.journal_len - md_offset : bs->metadata_buf_size),
|
(size_t)(bs->dsk.journal_len - md_offset < bs->metadata_buf_size ? bs->dsk.journal_len - md_offset : bs->metadata_buf_size),
|
||||||
};
|
};
|
||||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "read buffer area"); };
|
||||||
io_uring_prep_readv(sqe, bs->dsk.journal_fd, &data->iov, 1, bs->dsk.journal_offset + md_offset);
|
io_uring_prep_readv(sqe, bs->dsk.journal_fd, &data->iov, 1, bs->dsk.journal_offset + md_offset);
|
||||||
md_offset += data->iov.iov_len;
|
md_offset += data->iov.iov_len;
|
||||||
submitted++;
|
submitted++;
|
||||||
@@ -186,7 +175,7 @@ resume_3:
|
|||||||
next_offset = md_offset;
|
next_offset = md_offset;
|
||||||
// Read the rest of the metadata
|
// Read the rest of the metadata
|
||||||
resume_4:
|
resume_4:
|
||||||
if (next_offset < bs->dsk.meta_area_size && submitted == 0)
|
if (next_offset < bs->dsk.meta_area_size && submitted == 0 && (!zero_on_init || !bs->readonly))
|
||||||
{
|
{
|
||||||
// Submit one read
|
// Submit one read
|
||||||
for (int i = 0; i < 2; i++)
|
for (int i = 0; i < 2; i++)
|
||||||
@@ -203,12 +192,15 @@ resume_4:
|
|||||||
GET_SQE();
|
GET_SQE();
|
||||||
assert(bufs[i].size <= 0x7fffffff);
|
assert(bufs[i].size <= 0x7fffffff);
|
||||||
data->iov = { bufs[i].buf, (size_t)bufs[i].size };
|
data->iov = { bufs[i].buf, (size_t)bufs[i].size };
|
||||||
data->callback = [this, i](ring_data_t *data) { handle_event(data, i); };
|
|
||||||
if (!zero_on_init)
|
if (!zero_on_init)
|
||||||
|
{
|
||||||
|
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "read metadata"); };
|
||||||
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
|
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
|
||||||
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
// Fill metadata with empty block pattern
|
// Fill metadata with empty block pattern
|
||||||
|
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "clear metadata"); };
|
||||||
memset(bufs[i].buf, 0, bufs[i].size);
|
memset(bufs[i].buf, 0, bufs[i].size);
|
||||||
for (uint64_t o = 0; o < bufs[i].size; o += bs->dsk.meta_block_size)
|
for (uint64_t o = 0; o < bufs[i].size; o += bs->dsk.meta_block_size)
|
||||||
bs->heap->fill_block_empty_space(bufs[i].buf + o, 0);
|
bs->heap->fill_block_empty_space(bufs[i].buf + o, 0);
|
||||||
@@ -224,11 +216,14 @@ resume_4:
|
|||||||
if (bufs[i].state == INIT_META_READ_DONE)
|
if (bufs[i].state == INIT_META_READ_DONE)
|
||||||
{
|
{
|
||||||
// Handle result
|
// Handle result
|
||||||
uint64_t loaded = 0;
|
if (!zero_on_init)
|
||||||
int r = bs->heap->load_blocks(bufs[i].offset-bs->dsk.meta_block_size, bufs[i].size, bufs[i].buf, bs->skip_corrupted_meta_entries, loaded);
|
{
|
||||||
if (r != 0)
|
uint64_t loaded = 0;
|
||||||
exit(1);
|
int r = bs->heap->load_blocks(bufs[i].offset-bs->dsk.meta_block_size, bufs[i].size, bufs[i].buf, bs->skip_corrupted_meta_entries, loaded);
|
||||||
entries_loaded += loaded;
|
if (r != 0)
|
||||||
|
exit(1);
|
||||||
|
entries_loaded += loaded;
|
||||||
|
}
|
||||||
bufs[i].state = 0;
|
bufs[i].state = 0;
|
||||||
bs->ringloop->wakeup();
|
bs->ringloop->wakeup();
|
||||||
}
|
}
|
||||||
@@ -238,25 +233,9 @@ resume_4:
|
|||||||
wait_state = 4;
|
wait_state = 4;
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
// metadata read finished
|
// metadata read/clear finished
|
||||||
bs->heap->finish_load();
|
bs->heap->finish_load();
|
||||||
printf("Metadata entries loaded: %ju, used blocks: %ju / %ju\n", entries_loaded, bs->heap->get_data_used_space() / bs->dsk.data_block_size, bs->dsk.block_count);
|
printf("Metadata entries loaded: %ju, rechecking unfinished writes and garbage entries\n", entries_loaded);
|
||||||
if (zero_on_init && !bs->dsk.disable_meta_fsync)
|
|
||||||
{
|
|
||||||
GET_SQE();
|
|
||||||
io_uring_prep_fsync(sqe, bs->dsk.meta_fd, IORING_FSYNC_DATASYNC);
|
|
||||||
last_read_offset = 0;
|
|
||||||
data->iov = { 0 };
|
|
||||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
|
||||||
submitted++;
|
|
||||||
bs->ringloop->submit();
|
|
||||||
resume_5:
|
|
||||||
if (submitted > 0)
|
|
||||||
{
|
|
||||||
wait_state = 5;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// asynchronous recheck
|
// asynchronous recheck
|
||||||
resume_6:
|
resume_6:
|
||||||
wait_state = 6;
|
wait_state = 6;
|
||||||
@@ -293,6 +272,11 @@ resume_7:
|
|||||||
if (bs->readonly)
|
if (bs->readonly)
|
||||||
{
|
{
|
||||||
recheck_mod.clear();
|
recheck_mod.clear();
|
||||||
|
printf("Actual metadata entries: %ju\n", bs->heap->get_live_entries());
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
printf("Actual metadata entries: %ju, clearing garbage in %zu metadata blocks\n", bs->heap->get_live_entries(), recheck_mod.size());
|
||||||
}
|
}
|
||||||
for (i = 0; i < recheck_mod.size(); i++)
|
for (i = 0; i < recheck_mod.size(); i++)
|
||||||
{
|
{
|
||||||
@@ -306,7 +290,7 @@ resume_8:
|
|||||||
uint32_t block_num = recheck_mod[i];
|
uint32_t block_num = recheck_mod[i];
|
||||||
uint64_t block_offset = bs->dsk.meta_offset + (uint64_t)(block_num+1) * bs->dsk.meta_block_size;
|
uint64_t block_offset = bs->dsk.meta_offset + (uint64_t)(block_num+1) * bs->dsk.meta_block_size;
|
||||||
data = ((ring_data_t*)sqe->user_data);
|
data = ((ring_data_t*)sqe->user_data);
|
||||||
uint8_t *buf = (uint8_t*)malloc_or_die(bs->dsk.meta_block_size);
|
uint8_t *buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, bs->dsk.meta_block_size);
|
||||||
bs->heap->get_meta_block(block_num, buf);
|
bs->heap->get_meta_block(block_num, buf);
|
||||||
data->iov = { buf, bs->dsk.meta_block_size };
|
data->iov = { buf, bs->dsk.meta_block_size };
|
||||||
data->callback = [this, buf, block_offset](ring_data_t *data)
|
data->callback = [this, buf, block_offset](ring_data_t *data)
|
||||||
@@ -332,5 +316,47 @@ resume_9:
|
|||||||
}
|
}
|
||||||
free(metadata_buffer);
|
free(metadata_buffer);
|
||||||
metadata_buffer = NULL;
|
metadata_buffer = NULL;
|
||||||
|
do_fsync:
|
||||||
|
if (!bs->dsk.disable_meta_fsync && !bs->readonly)
|
||||||
|
{
|
||||||
|
GET_SQE();
|
||||||
|
io_uring_prep_fsync(sqe, bs->dsk.meta_fd, IORING_FSYNC_DATASYNC);
|
||||||
|
last_read_offset = 0;
|
||||||
|
data->iov = { 0 };
|
||||||
|
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "fsync metadata"); };
|
||||||
|
submitted++;
|
||||||
|
bs->ringloop->submit();
|
||||||
|
resume_5:
|
||||||
|
if (submitted > 0)
|
||||||
|
{
|
||||||
|
wait_state = 5;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (zero_on_init && !header_written && !bs->readonly)
|
||||||
|
{
|
||||||
|
GET_SQE();
|
||||||
|
header_written = true;
|
||||||
|
last_read_offset = 0;
|
||||||
|
data->iov = (struct iovec){ bs->meta_superblock, (size_t)bs->dsk.meta_block_size };
|
||||||
|
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "write metadata header"); };
|
||||||
|
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
||||||
|
bs->ringloop->submit();
|
||||||
|
submitted++;
|
||||||
|
resume_2:
|
||||||
|
if (submitted > 0)
|
||||||
|
{
|
||||||
|
wait_state = 2;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
if (!bs->dsk.disable_meta_fsync)
|
||||||
|
{
|
||||||
|
goto do_fsync;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
printf("Loading finished. Data used: %ju / %ju bytes (%s / %s)\n",
|
||||||
|
bs->heap->get_data_used_space(), bs->dsk.block_count * bs->dsk.data_block_size,
|
||||||
|
format_size(bs->heap->get_data_used_space()).c_str(),
|
||||||
|
format_size(bs->dsk.block_count * bs->dsk.data_block_size).c_str());
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -17,6 +17,7 @@ class blockstore_init_meta
|
|||||||
int wait_state = 0;
|
int wait_state = 0;
|
||||||
int wait_count = 0;
|
int wait_count = 0;
|
||||||
bool zero_on_init = false;
|
bool zero_on_init = false;
|
||||||
|
bool header_written = false;
|
||||||
void *metadata_buffer = NULL;
|
void *metadata_buffer = NULL;
|
||||||
blockstore_init_meta_buf bufs[2] = {};
|
blockstore_init_meta_buf bufs[2] = {};
|
||||||
int submitted = 0;
|
int submitted = 0;
|
||||||
@@ -29,7 +30,7 @@ class blockstore_init_meta
|
|||||||
std::vector<uint32_t> recheck_mod;
|
std::vector<uint32_t> recheck_mod;
|
||||||
int i = 0, j = 0;
|
int i = 0, j = 0;
|
||||||
bool handle_meta_block(uint8_t *buf, uint64_t count, uint64_t done_cnt);
|
bool handle_meta_block(uint8_t *buf, uint64_t count, uint64_t done_cnt);
|
||||||
void handle_event(ring_data_t *data, int buf_num);
|
void handle_event(ring_data_t *data, int buf_num, const char *op);
|
||||||
public:
|
public:
|
||||||
blockstore_init_meta(blockstore_impl_t *bs);
|
blockstore_init_meta(blockstore_impl_t *bs);
|
||||||
int loop();
|
int loop();
|
||||||
|
|||||||
@@ -0,0 +1,113 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
|
#include "blockstore_mock.h"
|
||||||
|
|
||||||
|
blockstore_mock_t::blockstore_mock_t(const blockstore_config_t & config)
|
||||||
|
{
|
||||||
|
}
|
||||||
|
|
||||||
|
void blockstore_mock_t::parse_config(blockstore_config_t & config)
|
||||||
|
{
|
||||||
|
}
|
||||||
|
|
||||||
|
void* blockstore_mock_t::reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit)
|
||||||
|
{
|
||||||
|
return NULL;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool blockstore_mock_t::reshard_continue(void *reshard_state, uint64_t chunk_limit)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
void blockstore_mock_t::loop()
|
||||||
|
{
|
||||||
|
}
|
||||||
|
|
||||||
|
bool blockstore_mock_t::is_started()
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool blockstore_mock_t::is_stalled()
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool blockstore_mock_t::is_safe_to_stop()
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
void blockstore_mock_t::enqueue_op(blockstore_op_t *op)
|
||||||
|
{
|
||||||
|
}
|
||||||
|
|
||||||
|
int blockstore_mock_t::read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version)
|
||||||
|
{
|
||||||
|
return -EIO;
|
||||||
|
}
|
||||||
|
|
||||||
|
const std::map<uint64_t, uint64_t> & blockstore_mock_t::get_inode_space_stats()
|
||||||
|
{
|
||||||
|
return inode_space;
|
||||||
|
}
|
||||||
|
|
||||||
|
void blockstore_mock_t::set_no_inode_stats(const std::vector<uint64_t> & pool_ids)
|
||||||
|
{
|
||||||
|
}
|
||||||
|
|
||||||
|
void blockstore_mock_t::dump_diagnostics()
|
||||||
|
{
|
||||||
|
}
|
||||||
|
|
||||||
|
std::string blockstore_mock_t::get_op_diag(blockstore_op_t *op)
|
||||||
|
{
|
||||||
|
return "";
|
||||||
|
}
|
||||||
|
|
||||||
|
uint32_t blockstore_mock_t::get_block_size()
|
||||||
|
{
|
||||||
|
return block_size;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t blockstore_mock_t::get_block_count()
|
||||||
|
{
|
||||||
|
return block_count;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t blockstore_mock_t::get_free_block_count()
|
||||||
|
{
|
||||||
|
return block_count;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t blockstore_mock_t::get_journal_size()
|
||||||
|
{
|
||||||
|
return 32*1024*1024;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint32_t blockstore_mock_t::get_bitmap_granularity()
|
||||||
|
{
|
||||||
|
return bitmap_granularity;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t blockstore_mock_t::get_live_entries()
|
||||||
|
{
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t blockstore_mock_t::get_live_memory()
|
||||||
|
{
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t blockstore_mock_t::get_garbage_entries()
|
||||||
|
{
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t blockstore_mock_t::get_garbage_memory()
|
||||||
|
{
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include "blockstore.h"
|
||||||
|
|
||||||
|
class blockstore_mock_t: public blockstore_i
|
||||||
|
{
|
||||||
|
public:
|
||||||
|
uint32_t block_size = 128*1024;
|
||||||
|
uint32_t bitmap_granularity = 4096;
|
||||||
|
uint64_t block_count = 100*1024*8;
|
||||||
|
std::map<uint64_t, uint64_t> inode_space;
|
||||||
|
|
||||||
|
blockstore_mock_t(const blockstore_config_t & config);
|
||||||
|
void parse_config(blockstore_config_t & config) override;
|
||||||
|
void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit) override;
|
||||||
|
bool reshard_continue(void *reshard_state, uint64_t chunk_limit) override;
|
||||||
|
void loop() override;
|
||||||
|
bool is_started() override;
|
||||||
|
bool is_stalled() override;
|
||||||
|
bool is_safe_to_stop() override;
|
||||||
|
void enqueue_op(blockstore_op_t *op) override;
|
||||||
|
int read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version = NULL) override;
|
||||||
|
const std::map<uint64_t, uint64_t> & get_inode_space_stats() override;
|
||||||
|
void set_no_inode_stats(const std::vector<uint64_t> & pool_ids) override;
|
||||||
|
void dump_diagnostics() override;
|
||||||
|
std::string get_op_diag(blockstore_op_t *op) override;
|
||||||
|
uint32_t get_block_size() override;
|
||||||
|
uint64_t get_block_count() override;
|
||||||
|
uint64_t get_free_block_count() override;
|
||||||
|
uint64_t get_journal_size() override;
|
||||||
|
uint32_t get_bitmap_granularity() override;
|
||||||
|
uint64_t get_live_entries() override;
|
||||||
|
uint64_t get_live_memory() override;
|
||||||
|
uint64_t get_garbage_entries() override;
|
||||||
|
uint64_t get_garbage_memory() override;
|
||||||
|
};
|
||||||
@@ -462,6 +462,10 @@ int blockstore_impl_t::read_bitmap(object_id oid, uint64_t target_version, void
|
|||||||
{
|
{
|
||||||
if (target_version >= wr->version)
|
if (target_version >= wr->version)
|
||||||
{
|
{
|
||||||
|
if (wr->type() == BS_HEAP_DELETE)
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
found = true;
|
found = true;
|
||||||
if (result_version)
|
if (result_version)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -16,6 +16,7 @@ int blockstore_impl_t::dequeue_stable(blockstore_op_t *op)
|
|||||||
else if (priv->op_state == 5) goto resume_5;
|
else if (priv->op_state == 5) goto resume_5;
|
||||||
assert(!priv->op_state);
|
assert(!priv->op_state);
|
||||||
op->retval = 0;
|
op->retval = 0;
|
||||||
|
PRIV(op)->lsn = 0;
|
||||||
priv->modified_block = priv->modified_block2 = UINT32_MAX;
|
priv->modified_block = priv->modified_block2 = UINT32_MAX;
|
||||||
for (priv->stab_pos = 0; priv->stab_pos < op->len; priv->stab_pos++)
|
for (priv->stab_pos = 0; priv->stab_pos < op->len; priv->stab_pos++)
|
||||||
{
|
{
|
||||||
@@ -27,6 +28,7 @@ int blockstore_impl_t::dequeue_stable(blockstore_op_t *op)
|
|||||||
FINISH_OP(op);
|
FINISH_OP(op);
|
||||||
return 2;
|
return 2;
|
||||||
}
|
}
|
||||||
|
priv->modified_block2 = UINT32_MAX;
|
||||||
int res = op->opcode == BS_OP_STABLE
|
int res = op->opcode == BS_OP_STABLE
|
||||||
? heap->add_commit(obj, v[priv->stab_pos].version, &priv->modified_block2)
|
? heap->add_commit(obj, v[priv->stab_pos].version, &priv->modified_block2)
|
||||||
: heap->add_rollback(obj, v[priv->stab_pos].version, &priv->modified_block2);
|
: heap->add_rollback(obj, v[priv->stab_pos].version, &priv->modified_block2);
|
||||||
@@ -36,6 +38,12 @@ int blockstore_impl_t::dequeue_stable(blockstore_op_t *op)
|
|||||||
FINISH_OP(op);
|
FINISH_OP(op);
|
||||||
return 2;
|
return 2;
|
||||||
}
|
}
|
||||||
|
if (res == ENOENT)
|
||||||
|
{
|
||||||
|
op->retval = -ENOENT;
|
||||||
|
FINISH_OP(op);
|
||||||
|
return 2;
|
||||||
|
}
|
||||||
if (res == ENOSPC)
|
if (res == ENOSPC)
|
||||||
{
|
{
|
||||||
if (!heap->get_to_compact_count())
|
if (!heap->get_to_compact_count())
|
||||||
@@ -45,11 +53,6 @@ int blockstore_impl_t::dequeue_stable(blockstore_op_t *op)
|
|||||||
FINISH_OP(op);
|
FINISH_OP(op);
|
||||||
return 2;
|
return 2;
|
||||||
}
|
}
|
||||||
if (priv->modified_block2 != UINT32_MAX)
|
|
||||||
{
|
|
||||||
priv->stab_pos--;
|
|
||||||
goto resume_1;
|
|
||||||
}
|
|
||||||
priv->wait_for = WAIT_COMPACTION;
|
priv->wait_for = WAIT_COMPACTION;
|
||||||
priv->wait_detail = heap->get_compacted_count();
|
priv->wait_detail = heap->get_compacted_count();
|
||||||
flusher->request_trim();
|
flusher->request_trim();
|
||||||
|
|||||||
@@ -9,6 +9,7 @@ int blockstore_impl_t::continue_sync(blockstore_op_t *op)
|
|||||||
if (!PRIV(op)->op_state)
|
if (!PRIV(op)->op_state)
|
||||||
{
|
{
|
||||||
op->retval = 0;
|
op->retval = 0;
|
||||||
|
PRIV(op)->lsn = 0;
|
||||||
}
|
}
|
||||||
int res = do_sync(op, 0);
|
int res = do_sync(op, 0);
|
||||||
if (res == 2)
|
if (res == 2)
|
||||||
@@ -28,9 +29,12 @@ bool blockstore_impl_t::has_unsynced()
|
|||||||
|
|
||||||
bool blockstore_impl_t::submit_fsyncs(int & wait_count)
|
bool blockstore_impl_t::submit_fsyncs(int & wait_count)
|
||||||
{
|
{
|
||||||
int n = (unsynced_meta_write_count > 0 && !dsk.disable_meta_fsync) +
|
int n = (unsynced_meta_write_count > 0 && !dsk.disable_meta_fsync ? 1 : 0) +
|
||||||
(unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync && dsk.journal_fd != dsk.meta_fd) +
|
(unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync &&
|
||||||
(unsynced_data_write_count > 0 && !dsk.disable_data_fsync && dsk.data_fd != dsk.meta_fd && dsk.data_fd != dsk.journal_fd);
|
(!unsynced_meta_write_count || dsk.journal_fd != dsk.meta_fd) ? 1 : 0) +
|
||||||
|
(unsynced_data_write_count > 0 && !dsk.disable_data_fsync &&
|
||||||
|
(!unsynced_meta_write_count || dsk.data_fd != dsk.meta_fd) &&
|
||||||
|
(!unsynced_buffer_write_count || dsk.data_fd != dsk.journal_fd) ? 1 : 0);
|
||||||
if (ringloop->space_left() < n)
|
if (ringloop->space_left() < n)
|
||||||
{
|
{
|
||||||
return false;
|
return false;
|
||||||
@@ -59,7 +63,8 @@ bool blockstore_impl_t::submit_fsyncs(int & wait_count)
|
|||||||
data->callback = cb;
|
data->callback = cb;
|
||||||
wait_count++;
|
wait_count++;
|
||||||
}
|
}
|
||||||
if (unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync && dsk.meta_fd != dsk.journal_fd)
|
if (unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync &&
|
||||||
|
(!unsynced_meta_write_count || dsk.journal_fd != dsk.meta_fd))
|
||||||
{
|
{
|
||||||
// fsync buffer
|
// fsync buffer
|
||||||
io_uring_sqe *sqe = get_sqe();
|
io_uring_sqe *sqe = get_sqe();
|
||||||
@@ -70,7 +75,9 @@ bool blockstore_impl_t::submit_fsyncs(int & wait_count)
|
|||||||
data->callback = cb;
|
data->callback = cb;
|
||||||
wait_count++;
|
wait_count++;
|
||||||
}
|
}
|
||||||
if (unsynced_data_write_count > 0 && !dsk.disable_data_fsync && dsk.data_fd != dsk.meta_fd && dsk.data_fd != dsk.journal_fd)
|
if (unsynced_data_write_count > 0 && !dsk.disable_data_fsync &&
|
||||||
|
(!unsynced_meta_write_count || dsk.data_fd != dsk.meta_fd) &&
|
||||||
|
(!unsynced_buffer_write_count || dsk.data_fd != dsk.journal_fd))
|
||||||
{
|
{
|
||||||
// fsync data
|
// fsync data
|
||||||
io_uring_sqe *sqe = get_sqe();
|
io_uring_sqe *sqe = get_sqe();
|
||||||
@@ -104,9 +111,11 @@ int blockstore_impl_t::do_sync(blockstore_op_t *op, int base_state)
|
|||||||
unsynced_data_write_count = unsynced_buffer_write_count = unsynced_meta_write_count = 0;
|
unsynced_data_write_count = unsynced_buffer_write_count = unsynced_meta_write_count = 0;
|
||||||
return 2;
|
return 2;
|
||||||
}
|
}
|
||||||
PRIV(op)->modified_block = heap->get_completed_lsn();
|
assert(!PRIV(op)->lsn);
|
||||||
|
PRIV(op)->lsn = heap->get_completed_lsn();
|
||||||
if (!submit_fsyncs(PRIV(op)->pending_ops))
|
if (!submit_fsyncs(PRIV(op)->pending_ops))
|
||||||
{
|
{
|
||||||
|
PRIV(op)->lsn = 0;
|
||||||
PRIV(op)->wait_detail = 1;
|
PRIV(op)->wait_detail = 1;
|
||||||
PRIV(op)->wait_for = WAIT_SQE;
|
PRIV(op)->wait_for = WAIT_SQE;
|
||||||
return 0;
|
return 0;
|
||||||
@@ -118,6 +127,6 @@ resume_1:
|
|||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
resume_2:
|
resume_2:
|
||||||
heap->mark_lsn_fsynced(PRIV(op)->modified_block);
|
heap->mark_lsn_fsynced(PRIV(op)->lsn);
|
||||||
return 2;
|
return 2;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -37,6 +37,7 @@ void blockstore_impl_t::prepare_meta_block_write(uint32_t modified_block)
|
|||||||
heap->complete_block_write(modified_block);
|
heap->complete_block_write(modified_block);
|
||||||
ringloop->wakeup();
|
ringloop->wakeup();
|
||||||
};
|
};
|
||||||
|
assert(((uint64_t)modified_block+2)*dsk.meta_block_size <= dsk.meta_area_size);
|
||||||
io_uring_prep_writev(
|
io_uring_prep_writev(
|
||||||
sqe, dsk.meta_fd, &data->iov, 1, dsk.meta_offset + ((uint64_t)modified_block+1)*dsk.meta_block_size
|
sqe, dsk.meta_fd, &data->iov, 1, dsk.meta_offset + ((uint64_t)modified_block+1)*dsk.meta_block_size
|
||||||
);
|
);
|
||||||
@@ -177,14 +178,18 @@ enospc:
|
|||||||
ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
||||||
data->iov = (struct iovec){ op->buf, op->len };
|
data->iov = (struct iovec){ op->buf, op->len };
|
||||||
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
||||||
|
assert(loc+op->offset+op->len <= dsk.block_count*dsk.data_block_size);
|
||||||
io_uring_prep_writev(sqe, dsk.data_fd, &data->iov, 1, dsk.data_offset + loc + op->offset);
|
io_uring_prep_writev(sqe, dsk.data_fd, &data->iov, 1, dsk.data_offset + loc + op->offset);
|
||||||
|
if (!dsk.disable_data_fsync)
|
||||||
|
{
|
||||||
|
// use PRIV->lsn for fsync_data_id
|
||||||
|
PRIV(op)->lsn = ++data_fsync_next;
|
||||||
|
data_fsyncs.push_back(false);
|
||||||
|
}
|
||||||
PRIV(op)->pending_ops++;
|
PRIV(op)->pending_ops++;
|
||||||
write_iodepth++;
|
write_iodepth++;
|
||||||
if (PRIV(op)->write_type == BS_HEAP_BIG_WRITE)
|
if (PRIV(op)->write_type == BS_HEAP_BIG_WRITE)
|
||||||
{
|
|
||||||
PRIV(op)->op_state = 1;
|
PRIV(op)->op_state = 1;
|
||||||
inflight_big++;
|
|
||||||
}
|
|
||||||
else
|
else
|
||||||
PRIV(op)->op_state = 3;
|
PRIV(op)->op_state = 3;
|
||||||
}
|
}
|
||||||
@@ -264,6 +269,7 @@ enospc:
|
|||||||
BS_SUBMIT_GET_SQE(sqe2, data2);
|
BS_SUBMIT_GET_SQE(sqe2, data2);
|
||||||
data2->iov = (struct iovec){ op->buf, op->len };
|
data2->iov = (struct iovec){ op->buf, op->len };
|
||||||
data2->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
data2->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
||||||
|
assert(loc+op->len <= dsk.journal_len);
|
||||||
io_uring_prep_writev(sqe2, dsk.journal_fd, &data2->iov, 1, dsk.journal_offset + loc);
|
io_uring_prep_writev(sqe2, dsk.journal_fd, &data2->iov, 1, dsk.journal_offset + loc);
|
||||||
PRIV(op)->pending_ops++;
|
PRIV(op)->pending_ops++;
|
||||||
}
|
}
|
||||||
@@ -294,8 +300,6 @@ again:
|
|||||||
goto resume_10;
|
goto resume_10;
|
||||||
else if (op_state == 11)
|
else if (op_state == 11)
|
||||||
goto resume_11;
|
goto resume_11;
|
||||||
else if (op_state == 12)
|
|
||||||
goto resume_12;
|
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
// In progress
|
// In progress
|
||||||
@@ -314,38 +318,44 @@ again:
|
|||||||
resume_2:
|
resume_2:
|
||||||
// We must fsync all big writes to avoid complex write workflows
|
// We must fsync all big writes to avoid complex write workflows
|
||||||
// It's OK for all HDDs and for server SSDs, but slightly worse for desktop SSDs
|
// It's OK for all HDDs and for server SSDs, but slightly worse for desktop SSDs
|
||||||
inflight_big--;
|
|
||||||
if (!dsk.disable_data_fsync)
|
if (!dsk.disable_data_fsync)
|
||||||
{
|
{
|
||||||
// fsync data in a batch
|
// Mark our data write as completed and advance data_fsync_cur
|
||||||
resume_11:
|
data_fsyncs[PRIV(op)->lsn - data_fsync_cur - 1] = true;
|
||||||
if (inflight_big > 0)
|
while (data_fsyncs.size() > 0 && data_fsyncs.front())
|
||||||
|
{
|
||||||
|
data_fsyncs.pop_front();
|
||||||
|
data_fsync_cur++;
|
||||||
|
}
|
||||||
|
PRIV(op)->op_state = 11;
|
||||||
|
// Then wait for all other data writes currently in progress to do less fsync calls
|
||||||
|
// I.e. to fsync data in batches
|
||||||
|
PRIV(op)->lsn = data_fsync_cur + data_fsyncs.size();
|
||||||
|
resume_11:
|
||||||
|
if (data_fsync_cur < PRIV(op)->lsn)
|
||||||
{
|
{
|
||||||
PRIV(op)->op_state = 11;
|
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
if (fsyncing_data)
|
if (PRIV(op)->lsn > data_fsync_sent)
|
||||||
{
|
{
|
||||||
resume_12:
|
BS_SUBMIT_GET_SQE(sqe, data);
|
||||||
if (fsyncing_data)
|
io_uring_prep_fsync(sqe, dsk.data_fd, IORING_FSYNC_DATASYNC);
|
||||||
|
data->iov = { 0 };
|
||||||
|
data->callback = [this, op, fs = data_fsync_cur](ring_data_t *data)
|
||||||
{
|
{
|
||||||
PRIV(op)->op_state = 12;
|
if (fs > data_fsync_done)
|
||||||
return 1;
|
{
|
||||||
}
|
data_fsync_done = fs;
|
||||||
goto resume_4;
|
ringloop->wakeup();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
data_fsync_sent = data_fsync_cur;
|
||||||
}
|
}
|
||||||
fsyncing_data = true;
|
if (PRIV(op)->lsn > data_fsync_done)
|
||||||
BS_SUBMIT_GET_SQE(sqe, data);
|
|
||||||
io_uring_prep_fsync(sqe, dsk.data_fd, IORING_FSYNC_DATASYNC);
|
|
||||||
data->iov = { 0 };
|
|
||||||
data->callback = [this, op](ring_data_t *data)
|
|
||||||
{
|
{
|
||||||
fsyncing_data = false;
|
return 1;
|
||||||
handle_write_event(data, op);
|
}
|
||||||
};
|
PRIV(op)->lsn = 0;
|
||||||
PRIV(op)->pending_ops++;
|
|
||||||
PRIV(op)->op_state = 3;
|
|
||||||
return 1;
|
|
||||||
}
|
}
|
||||||
resume_4:
|
resume_4:
|
||||||
{
|
{
|
||||||
@@ -453,6 +463,7 @@ resume_10:
|
|||||||
BS_SUBMIT_GET_SQE(sqe, data);
|
BS_SUBMIT_GET_SQE(sqe, data);
|
||||||
data->iov = (struct iovec){ op->buf, op->len };
|
data->iov = (struct iovec){ op->buf, op->len };
|
||||||
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
||||||
|
assert(PRIV(op)->location + op->offset <= dsk.block_count*dsk.data_block_size);
|
||||||
io_uring_prep_writev(sqe, dsk.data_fd, &data->iov, 1, dsk.data_offset + PRIV(op)->location + op->offset);
|
io_uring_prep_writev(sqe, dsk.data_fd, &data->iov, 1, dsk.data_offset + PRIV(op)->location + op->offset);
|
||||||
if (dsk.use_atomic_flag)
|
if (dsk.use_atomic_flag)
|
||||||
sqe->rw_flags = RWF_ATOMIC;
|
sqe->rw_flags = RWF_ATOMIC;
|
||||||
|
|||||||
@@ -12,7 +12,7 @@ multilist_alloc_t::multilist_alloc_t(uint32_t count, uint32_t maxn):
|
|||||||
count(count), maxn(maxn)
|
count(count), maxn(maxn)
|
||||||
{
|
{
|
||||||
// not-so-memory-efficient: 16 MB memory per 1 GB buffer space, but buffer spaces are small, so OK
|
// not-so-memory-efficient: 16 MB memory per 1 GB buffer space, but buffer spaces are small, so OK
|
||||||
assert(count > 1 && count < 0x80000000);
|
assert(count > 1 && count < 0x80000000 && count >= maxn);
|
||||||
sizes.resize(count);
|
sizes.resize(count);
|
||||||
nexts.resize(count); // nexts[i] = 0 -> area is used; nexts[i] = 1 -> no next; nexts[i] >= 2 -> next item
|
nexts.resize(count); // nexts[i] = 0 -> area is used; nexts[i] = 1 -> no next; nexts[i] >= 2 -> next item
|
||||||
prevs.resize(count);
|
prevs.resize(count);
|
||||||
@@ -171,7 +171,7 @@ void multilist_alloc_t::print()
|
|||||||
printf("\n");
|
printf("\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
void multilist_alloc_t::use(uint32_t pos, uint32_t size)
|
bool multilist_alloc_t::use(uint32_t pos, uint32_t size)
|
||||||
{
|
{
|
||||||
assert(pos+size <= count && size > 0);
|
assert(pos+size <= count && size > 0);
|
||||||
if (sizes[pos] <= 0)
|
if (sizes[pos] <= 0)
|
||||||
@@ -182,7 +182,8 @@ void multilist_alloc_t::use(uint32_t pos, uint32_t size)
|
|||||||
else
|
else
|
||||||
while (start > 0 && !sizes[start])
|
while (start > 0 && !sizes[start])
|
||||||
start--;
|
start--;
|
||||||
assert(sizes[start] >= size);
|
if (sizes[start] < size+(pos-start))
|
||||||
|
return false;
|
||||||
use_full(start);
|
use_full(start);
|
||||||
uint32_t full = sizes[start];
|
uint32_t full = sizes[start];
|
||||||
sizes[pos-1] = -pos+start;
|
sizes[pos-1] = -pos+start;
|
||||||
@@ -199,7 +200,8 @@ void multilist_alloc_t::use(uint32_t pos, uint32_t size)
|
|||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
assert(sizes[pos] >= size);
|
if (sizes[pos] < size)
|
||||||
|
return false;
|
||||||
use_full(pos);
|
use_full(pos);
|
||||||
if (sizes[pos] > size)
|
if (sizes[pos] > size)
|
||||||
{
|
{
|
||||||
@@ -214,12 +216,13 @@ void multilist_alloc_t::use(uint32_t pos, uint32_t size)
|
|||||||
#ifdef MULTILIST_TRACE
|
#ifdef MULTILIST_TRACE
|
||||||
print();
|
print();
|
||||||
#endif
|
#endif
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
void multilist_alloc_t::use_full(uint32_t pos)
|
void multilist_alloc_t::use_full(uint32_t pos)
|
||||||
{
|
{
|
||||||
uint32_t prevsize = sizes[pos];
|
uint32_t prevsize = sizes[pos];
|
||||||
assert(prevsize);
|
assert(prevsize > 0);
|
||||||
assert(nexts[pos]);
|
assert(nexts[pos]);
|
||||||
uint32_t pi = (prevsize < maxn ? prevsize : maxn)-1;
|
uint32_t pi = (prevsize < maxn ? prevsize : maxn)-1;
|
||||||
if (heads[pi] == pos+1)
|
if (heads[pi] == pos+1)
|
||||||
|
|||||||
@@ -17,7 +17,7 @@ struct multilist_alloc_t
|
|||||||
bool is_free(uint32_t pos);
|
bool is_free(uint32_t pos);
|
||||||
uint32_t find(uint32_t size);
|
uint32_t find(uint32_t size);
|
||||||
void use_full(uint32_t pos);
|
void use_full(uint32_t pos);
|
||||||
void use(uint32_t pos, uint32_t size);
|
bool use(uint32_t pos, uint32_t size);
|
||||||
void do_free(uint32_t pos);
|
void do_free(uint32_t pos);
|
||||||
void free(uint32_t pos);
|
void free(uint32_t pos);
|
||||||
void verify();
|
void verify();
|
||||||
|
|||||||
@@ -141,7 +141,7 @@ struct __attribute__((__packed__)) journal_entry
|
|||||||
inline uint32_t je_crc32(journal_entry *je)
|
inline uint32_t je_crc32(journal_entry *je)
|
||||||
{
|
{
|
||||||
// 0x48674bc7 = crc32(4 zero bytes)
|
// 0x48674bc7 = crc32(4 zero bytes)
|
||||||
return crc32c(0x48674bc7, ((uint8_t*)je)+4, je->size-4);
|
return je->size < 4 ? 0 : crc32c(0x48674bc7, ((uint8_t*)je)+4, je->size-4);
|
||||||
}
|
}
|
||||||
|
|
||||||
// "VITAstor"
|
// "VITAstor"
|
||||||
|
|||||||
+55
-22
@@ -71,6 +71,11 @@ bool journal_flusher_t::is_active()
|
|||||||
return active_flushers > 0 || dequeuing;
|
return active_flushers > 0 || dequeuing;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
size_t journal_flusher_t::get_queue_size()
|
||||||
|
{
|
||||||
|
return flush_queue.size();
|
||||||
|
}
|
||||||
|
|
||||||
void journal_flusher_t::loop()
|
void journal_flusher_t::loop()
|
||||||
{
|
{
|
||||||
target_flusher_count = bs->write_iodepth*2;
|
target_flusher_count = bs->write_iodepth*2;
|
||||||
@@ -384,6 +389,7 @@ stop_flusher:
|
|||||||
wait_state = 0;
|
wait_state = 0;
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
copy_count = 0;
|
||||||
try_trim = true;
|
try_trim = true;
|
||||||
cur.oid = flusher->flush_queue.front();
|
cur.oid = flusher->flush_queue.front();
|
||||||
cur.version = flusher->flush_versions[cur.oid];
|
cur.version = flusher->flush_versions[cur.oid];
|
||||||
@@ -511,6 +517,31 @@ resume_2:
|
|||||||
{
|
{
|
||||||
uo_it->second.was_changed = true;
|
uo_it->second.was_changed = true;
|
||||||
}
|
}
|
||||||
|
if (!bs->journal.inmemory)
|
||||||
|
{
|
||||||
|
// Verify journaled data checksums (but not COALESCED)
|
||||||
|
for (it = v.begin(); it != v.end(); it++)
|
||||||
|
{
|
||||||
|
if (it->copy_flags == COPY_BUF_JOURNAL)
|
||||||
|
{
|
||||||
|
iovec iov = { .iov_base = it->buf, .iov_len = it->len };
|
||||||
|
bs->verify_journal_checksums(
|
||||||
|
it->csum_buf, it->offset, &iov, 1,
|
||||||
|
[&](uint32_t bad_block, uint32_t calc_csum, uint32_t stored_csum)
|
||||||
|
{
|
||||||
|
printf(
|
||||||
|
"Checksum mismatch in object %jx:%jx v%ju in journal at 0x%jx, checksum block #%u: got %08x, expected %08x\n",
|
||||||
|
cur.oid.inode, cur.oid.stripe, cur.version, it->disk_offset,
|
||||||
|
bad_block / bs->dsk.csum_block_size, calc_csum, stored_csum
|
||||||
|
);
|
||||||
|
bad_block += it->offset;
|
||||||
|
assert(!(bad_block % bs->dsk.csum_block_size) && bad_block < bs->dsk.data_block_size);
|
||||||
|
mangle_csum_blocks.insert(bad_block);
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
// Submit data writes
|
// Submit data writes
|
||||||
for (it = v.begin(); it != v.end(); it++)
|
for (it = v.begin(); it != v.end(); it++)
|
||||||
@@ -520,6 +551,7 @@ resume_2:
|
|||||||
await_sqe(15);
|
await_sqe(15);
|
||||||
data->iov = (struct iovec){ it->buf, (size_t)it->len };
|
data->iov = (struct iovec){ it->buf, (size_t)it->len };
|
||||||
data->callback = simple_callback_w;
|
data->callback = simple_callback_w;
|
||||||
|
assert(clean_loc+it->offset+it->len <= bs->dsk.block_count*bs->dsk.data_block_size);
|
||||||
io_uring_prep_writev(
|
io_uring_prep_writev(
|
||||||
sqe, bs->dsk.data_fd, &data->iov, 1, bs->dsk.data_offset + clean_loc + it->offset
|
sqe, bs->dsk.data_fd, &data->iov, 1, bs->dsk.data_offset + clean_loc + it->offset
|
||||||
);
|
);
|
||||||
@@ -633,6 +665,7 @@ resume_2:
|
|||||||
}
|
}
|
||||||
// All done
|
// All done
|
||||||
flusher->active_flushers--;
|
flusher->active_flushers--;
|
||||||
|
copy_count = 0; // used by is_mutated()...
|
||||||
wait_state = 0;
|
wait_state = 0;
|
||||||
goto resume_0;
|
goto resume_0;
|
||||||
}
|
}
|
||||||
@@ -749,6 +782,7 @@ bool journal_flusher_co::write_meta_block(flusher_meta_write_t & meta_block, int
|
|||||||
await_sqe(0);
|
await_sqe(0);
|
||||||
data->iov = (struct iovec){ meta_block.buf, (size_t)bs->dsk.meta_block_size };
|
data->iov = (struct iovec){ meta_block.buf, (size_t)bs->dsk.meta_block_size };
|
||||||
data->callback = simple_callback_w;
|
data->callback = simple_callback_w;
|
||||||
|
assert(bs->dsk.meta_block_size + meta_block.sector + bs->dsk.meta_block_size <= bs->dsk.meta_area_size);
|
||||||
io_uring_prep_writev(
|
io_uring_prep_writev(
|
||||||
sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bs->dsk.meta_block_size + meta_block.sector
|
sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bs->dsk.meta_block_size + meta_block.sector
|
||||||
);
|
);
|
||||||
@@ -813,35 +847,21 @@ bool journal_flusher_co::clear_incomplete_csum_block_bits(int wait_base)
|
|||||||
bs->verify_padded_checksums(new_clean_bitmap, new_clean_bitmap + 2*bs->dsk.clean_entry_bitmap_size,
|
bs->verify_padded_checksums(new_clean_bitmap, new_clean_bitmap + 2*bs->dsk.clean_entry_bitmap_size,
|
||||||
v[i].offset, &iov, 1, [&](uint32_t bad_block, uint32_t calc_csum, uint32_t stored_csum)
|
v[i].offset, &iov, 1, [&](uint32_t bad_block, uint32_t calc_csum, uint32_t stored_csum)
|
||||||
{
|
{
|
||||||
printf("Checksum mismatch in object %jx:%jx v%ju in data area at offset 0x%jx+0x%x: got %08x, expected %08x\n",
|
printf("Checksum mismatch in object %jx:%jx v%ju in data area at offset 0x%jx+0x%x during flush: got %08x, expected %08x\n",
|
||||||
cur.oid.inode, cur.oid.stripe, old_clean_ver, old_clean_loc, bad_block, calc_csum, stored_csum);
|
cur.oid.inode, cur.oid.stripe, old_clean_ver, old_clean_loc, bad_block, calc_csum, stored_csum);
|
||||||
for (uint32_t j = 0; j < bs->dsk.csum_block_size; j += bs->dsk.bitmap_granularity)
|
assert(!(bad_block % bs->dsk.csum_block_size) && bad_block < bs->dsk.data_block_size);
|
||||||
{
|
mangle_csum_blocks.insert(bad_block);
|
||||||
// Simplest method of mangling: flip one byte in every sector
|
|
||||||
((uint8_t*)v[i].buf)[j+bad_block-v[i].offset] ^= 0xff;
|
|
||||||
}
|
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
bs->verify_journal_checksums(v[i].csum_buf, v[i].offset, &iov, 1, [&](uint32_t bad_block, uint32_t calc_csum, uint32_t stored_csum)
|
bs->verify_journal_checksums(v[i].csum_buf, v[i].offset, &iov, 1, [&](uint32_t bad_block, uint32_t calc_csum, uint32_t stored_csum)
|
||||||
{
|
{
|
||||||
printf("Checksum mismatch in object %jx:%jx v%ju in journal at offset 0x%jx+0x%x (block offset 0x%jx): got %08x, expected %08x\n",
|
printf("Checksum mismatch in object %jx:%jx v%ju in journal at offset 0x%jx+0x%x (block offset 0x%jx) during flush: got %08x, expected %08x\n",
|
||||||
cur.oid.inode, cur.oid.stripe, old_clean_ver,
|
cur.oid.inode, cur.oid.stripe, old_clean_ver,
|
||||||
v[i].disk_offset, bad_block, v[i].offset, calc_csum, stored_csum);
|
v[i].disk_offset, bad_block, v[i].offset, calc_csum, stored_csum);
|
||||||
bad_block += (v[i].offset/bs->dsk.csum_block_size) * bs->dsk.csum_block_size;
|
assert(!(bad_block % bs->dsk.csum_block_size) && bad_block < bs->dsk.data_block_size);
|
||||||
uint32_t bad_block_end = bad_block + bs->dsk.csum_block_size + (v[i].offset/bs->dsk.csum_block_size) * bs->dsk.csum_block_size;
|
mangle_csum_blocks.insert(bad_block);
|
||||||
if (bad_block < v[i].offset)
|
|
||||||
bad_block = v[i].offset;
|
|
||||||
if (bad_block_end > v[i].offset+v[i].len)
|
|
||||||
bad_block_end = v[i].offset+v[i].len;
|
|
||||||
bad_block -= v[i].offset;
|
|
||||||
bad_block_end -= v[i].offset;
|
|
||||||
for (uint32_t j = bad_block; j < bad_block_end; j += bs->dsk.bitmap_granularity)
|
|
||||||
{
|
|
||||||
// Simplest method of mangling: flip one byte in every sector
|
|
||||||
((uint8_t*)v[i].buf)[j] ^= 0xff;
|
|
||||||
}
|
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -950,6 +970,11 @@ void journal_flusher_co::calc_block_checksums(uint32_t *new_data_csums, bool ski
|
|||||||
}
|
}
|
||||||
// `v` should contain aligned items, possibly split into pieces
|
// `v` should contain aligned items, possibly split into pieces
|
||||||
assert(!block_done);
|
assert(!block_done);
|
||||||
|
for (uint32_t mangle_block: mangle_csum_blocks)
|
||||||
|
{
|
||||||
|
// Flip 1 bit
|
||||||
|
new_data_csums[mangle_block / bs->dsk.csum_block_size] ^= 1;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void journal_flusher_co::scan_dirty()
|
void journal_flusher_co::scan_dirty()
|
||||||
@@ -1086,7 +1111,8 @@ void journal_flusher_co::scan_dirty()
|
|||||||
last--;
|
last--;
|
||||||
read_to_fill_incomplete = bs->fill_partial_checksum_blocks(
|
read_to_fill_incomplete = bs->fill_partial_checksum_blocks(
|
||||||
v, fulfilled, bmp_ptr, NULL, false, NULL, v[0].offset/bs->dsk.csum_block_size * bs->dsk.csum_block_size,
|
v, fulfilled, bmp_ptr, NULL, false, NULL, v[0].offset/bs->dsk.csum_block_size * bs->dsk.csum_block_size,
|
||||||
((v[last].offset+v[last].len-1) / bs->dsk.csum_block_size + 1) * bs->dsk.csum_block_size
|
((v[last].offset+v[last].len-1) / bs->dsk.csum_block_size + 1) * bs->dsk.csum_block_size,
|
||||||
|
0, bs->dsk.data_block_size
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
else if (fill_incomplete && clean_init_bitmap)
|
else if (fill_incomplete && clean_init_bitmap)
|
||||||
@@ -1116,6 +1142,7 @@ bool journal_flusher_co::read_dirty(int wait_base)
|
|||||||
if (wait_state == wait_base) goto resume_0;
|
if (wait_state == wait_base) goto resume_0;
|
||||||
else if (wait_state == wait_base+1) goto resume_1;
|
else if (wait_state == wait_base+1) goto resume_1;
|
||||||
wait_count = wait_journal_count = 0;
|
wait_count = wait_journal_count = 0;
|
||||||
|
mangle_csum_blocks.clear();
|
||||||
if (bs->journal.inmemory && !read_to_fill_incomplete)
|
if (bs->journal.inmemory && !read_to_fill_incomplete)
|
||||||
{
|
{
|
||||||
// Happy path: nothing to read :)
|
// Happy path: nothing to read :)
|
||||||
@@ -1347,7 +1374,7 @@ bool journal_flusher_co::fsync_batch(bool fsync_meta, int wait_base)
|
|||||||
cur_sync->ready_count++;
|
cur_sync->ready_count++;
|
||||||
flusher->syncing_flushers++;
|
flusher->syncing_flushers++;
|
||||||
resume_1:
|
resume_1:
|
||||||
if (!cur_sync->state)
|
if (cur_sync->state == 0)
|
||||||
{
|
{
|
||||||
if (flusher->syncing_flushers >= flusher->active_flushers || !flusher->flush_queue.size())
|
if (flusher->syncing_flushers >= flusher->active_flushers || !flusher->flush_queue.size())
|
||||||
{
|
{
|
||||||
@@ -1375,6 +1402,12 @@ bool journal_flusher_co::fsync_batch(bool fsync_meta, int wait_base)
|
|||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
else if (cur_sync->state == 1)
|
||||||
|
{
|
||||||
|
// Wait for fsync completion
|
||||||
|
wait_state = wait_base+1;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
flusher->syncing_flushers--;
|
flusher->syncing_flushers--;
|
||||||
cur_sync->ready_count--;
|
cur_sync->ready_count--;
|
||||||
if (cur_sync->ready_count == 0)
|
if (cur_sync->ready_count == 0)
|
||||||
|
|||||||
@@ -66,6 +66,7 @@ class journal_flusher_co
|
|||||||
uint64_t clean_bitmap_offset, clean_bitmap_len;
|
uint64_t clean_bitmap_offset, clean_bitmap_len;
|
||||||
uint8_t *clean_init_dyn_ptr;
|
uint8_t *clean_init_dyn_ptr;
|
||||||
uint8_t *new_clean_bitmap;
|
uint8_t *new_clean_bitmap;
|
||||||
|
std::unordered_set<uint32_t> mangle_csum_blocks;
|
||||||
|
|
||||||
uint64_t new_trim_pos;
|
uint64_t new_trim_pos;
|
||||||
|
|
||||||
@@ -123,6 +124,7 @@ public:
|
|||||||
void loop();
|
void loop();
|
||||||
bool is_trim_wanted() { return trim_wanted; }
|
bool is_trim_wanted() { return trim_wanted; }
|
||||||
bool is_active();
|
bool is_active();
|
||||||
|
size_t get_queue_size();
|
||||||
void mark_trim_possible();
|
void mark_trim_possible();
|
||||||
void request_trim();
|
void request_trim();
|
||||||
void release_trim();
|
void release_trim();
|
||||||
|
|||||||
@@ -6,11 +6,12 @@
|
|||||||
|
|
||||||
namespace v1 {
|
namespace v1 {
|
||||||
|
|
||||||
blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd)
|
blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd, bool mock_mode)
|
||||||
{
|
{
|
||||||
assert(sizeof(blockstore_op_private_t) <= BS_OP_PRIVATE_DATA_SIZE);
|
assert(sizeof(blockstore_op_private_t) <= BS_OP_PRIVATE_DATA_SIZE);
|
||||||
this->tfd = tfd;
|
this->tfd = tfd;
|
||||||
this->ringloop = ringloop;
|
this->ringloop = ringloop;
|
||||||
|
dsk.mock_mode = mock_mode;
|
||||||
ring_consumer.loop = [this]() { loop(); };
|
ring_consumer.loop = [this]() { loop(); };
|
||||||
ringloop->register_consumer(&ring_consumer);
|
ringloop->register_consumer(&ring_consumer);
|
||||||
initialized = 0;
|
initialized = 0;
|
||||||
@@ -35,6 +36,11 @@ blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_i *
|
|||||||
|
|
||||||
blockstore_impl_t::~blockstore_impl_t()
|
blockstore_impl_t::~blockstore_impl_t()
|
||||||
{
|
{
|
||||||
|
for (auto& obj: dirty_db)
|
||||||
|
{
|
||||||
|
if (obj.second.dyn_data)
|
||||||
|
free(obj.second.dyn_data);
|
||||||
|
}
|
||||||
delete data_alloc;
|
delete data_alloc;
|
||||||
delete flusher;
|
delete flusher;
|
||||||
if (zero_object)
|
if (zero_object)
|
||||||
@@ -855,4 +861,29 @@ std::string blockstore_impl_t::get_op_diag(blockstore_op_t *op)
|
|||||||
return std::string(buf);
|
return std::string(buf);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
uint64_t blockstore_impl_t::get_live_entries()
|
||||||
|
{
|
||||||
|
return used_blocks;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t blockstore_impl_t::get_live_memory()
|
||||||
|
{
|
||||||
|
uint64_t used = 0;
|
||||||
|
for (auto & kv: clean_db_shards)
|
||||||
|
{
|
||||||
|
used += kv.second.size() * sizeof(blockstore_clean_db_t::value_type);
|
||||||
|
}
|
||||||
|
return used;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t blockstore_impl_t::get_garbage_entries()
|
||||||
|
{
|
||||||
|
return dirty_db.size();
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t blockstore_impl_t::get_garbage_memory()
|
||||||
|
{
|
||||||
|
return (sizeof(obj_ver_id) + sizeof(dirty_entry) + 32) * dirty_db.size();
|
||||||
|
}
|
||||||
|
|
||||||
} // namespace v1
|
} // namespace v1
|
||||||
|
|||||||
@@ -30,6 +30,8 @@
|
|||||||
|
|
||||||
//#define BLOCKSTORE_DEBUG
|
//#define BLOCKSTORE_DEBUG
|
||||||
|
|
||||||
|
struct bs_test_t;
|
||||||
|
|
||||||
namespace v1 {
|
namespace v1 {
|
||||||
|
|
||||||
#include "journal.h"
|
#include "journal.h"
|
||||||
@@ -96,7 +98,7 @@ struct blockstore_op_private_t
|
|||||||
int op_state;
|
int op_state;
|
||||||
|
|
||||||
// Read
|
// Read
|
||||||
uint64_t clean_block_used;
|
uint64_t clean_loc_used;
|
||||||
std::vector<copy_buffer_t> read_vec;
|
std::vector<copy_buffer_t> read_vec;
|
||||||
|
|
||||||
// Sync, write
|
// Sync, write
|
||||||
@@ -122,6 +124,7 @@ typedef uint64_t pool_pg_id_t;
|
|||||||
|
|
||||||
class blockstore_impl_t: public blockstore_i
|
class blockstore_impl_t: public blockstore_i
|
||||||
{
|
{
|
||||||
|
friend struct ::bs_test_t;
|
||||||
blockstore_disk_t dsk;
|
blockstore_disk_t dsk;
|
||||||
|
|
||||||
/******* OPTIONS *******/
|
/******* OPTIONS *******/
|
||||||
@@ -220,6 +223,7 @@ class blockstore_impl_t: public blockstore_i
|
|||||||
|
|
||||||
// Read
|
// Read
|
||||||
int dequeue_read(blockstore_op_t *read_op);
|
int dequeue_read(blockstore_op_t *read_op);
|
||||||
|
void release_clean(blockstore_op_t *op);
|
||||||
void find_holes(std::vector<copy_buffer_t> & read_vec, uint32_t item_start, uint32_t item_end,
|
void find_holes(std::vector<copy_buffer_t> & read_vec, uint32_t item_start, uint32_t item_end,
|
||||||
std::function<int(int, bool, uint32_t, uint32_t)> callback);
|
std::function<int(int, bool, uint32_t, uint32_t)> callback);
|
||||||
int fulfill_read(blockstore_op_t *read_op,
|
int fulfill_read(blockstore_op_t *read_op,
|
||||||
@@ -230,7 +234,8 @@ class blockstore_impl_t: public blockstore_i
|
|||||||
uint8_t *clean_entry_bitmap, int *dyn_data,
|
uint8_t *clean_entry_bitmap, int *dyn_data,
|
||||||
uint32_t item_start, uint32_t item_end, uint64_t clean_loc, uint64_t clean_ver);
|
uint32_t item_start, uint32_t item_end, uint64_t clean_loc, uint64_t clean_ver);
|
||||||
int fill_partial_checksum_blocks(std::vector<copy_buffer_t> & rv, uint64_t & fulfilled,
|
int fill_partial_checksum_blocks(std::vector<copy_buffer_t> & rv, uint64_t & fulfilled,
|
||||||
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf, uint64_t read_offset, uint64_t read_end);
|
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf,
|
||||||
|
uint32_t read_offset, uint32_t read_end, uint32_t item_start, uint32_t item_end);
|
||||||
int pad_journal_read(std::vector<copy_buffer_t> & rv, copy_buffer_t & cp,
|
int pad_journal_read(std::vector<copy_buffer_t> & rv, copy_buffer_t & cp,
|
||||||
uint64_t dirty_offset, uint64_t dirty_end, uint64_t dirty_loc, uint8_t *csum_ptr, int *dyn_data,
|
uint64_t dirty_offset, uint64_t dirty_end, uint64_t dirty_loc, uint8_t *csum_ptr, int *dyn_data,
|
||||||
uint64_t offset, uint64_t submit_len, uint64_t & blk_begin, uint64_t & blk_end, uint8_t* & blk_buf);
|
uint64_t offset, uint64_t submit_len, uint64_t & blk_begin, uint64_t & blk_end, uint8_t* & blk_buf);
|
||||||
@@ -281,7 +286,7 @@ class blockstore_impl_t: public blockstore_i
|
|||||||
|
|
||||||
public:
|
public:
|
||||||
|
|
||||||
blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd);
|
blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd, bool mock_mode = false);
|
||||||
~blockstore_impl_t();
|
~blockstore_impl_t();
|
||||||
|
|
||||||
void parse_config(blockstore_config_t & config);
|
void parse_config(blockstore_config_t & config);
|
||||||
@@ -332,6 +337,10 @@ public:
|
|||||||
inline uint64_t get_free_block_count() { return dsk.block_count - used_blocks; }
|
inline uint64_t get_free_block_count() { return dsk.block_count - used_blocks; }
|
||||||
inline uint32_t get_bitmap_granularity() { return dsk.disk_alignment; }
|
inline uint32_t get_bitmap_granularity() { return dsk.disk_alignment; }
|
||||||
inline uint64_t get_journal_size() { return dsk.journal_len; }
|
inline uint64_t get_journal_size() { return dsk.journal_len; }
|
||||||
|
uint64_t get_live_entries();
|
||||||
|
uint64_t get_live_memory();
|
||||||
|
uint64_t get_garbage_entries();
|
||||||
|
uint64_t get_garbage_memory();
|
||||||
};
|
};
|
||||||
|
|
||||||
} // namespace v1
|
} // namespace v1
|
||||||
|
|||||||
+83
-68
@@ -1,6 +1,7 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.1 (see README.md for details)
|
// License: VNPL-1.1 (see README.md for details)
|
||||||
|
|
||||||
|
#include "str_util.h"
|
||||||
#include "impl.h"
|
#include "impl.h"
|
||||||
#include "internal.h"
|
#include "internal.h"
|
||||||
|
|
||||||
@@ -30,14 +31,15 @@ blockstore_init_meta::blockstore_init_meta(blockstore_impl_t *bs)
|
|||||||
this->bs = bs;
|
this->bs = bs;
|
||||||
}
|
}
|
||||||
|
|
||||||
void blockstore_init_meta::handle_event(ring_data_t *data, int buf_num)
|
void blockstore_init_meta::handle_event(ring_data_t *data, int buf_num, const char *op)
|
||||||
{
|
{
|
||||||
if (data->res < 0)
|
if (data->res != data->iov.iov_len)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(
|
throw std::runtime_error(strprintf(
|
||||||
std::string("read metadata failed at offset ") + std::to_string(buf_num >= 0 ? bufs[buf_num].offset : last_read_offset) +
|
"%s failed at offset %ju: got %s (code %d), but expected %zu",
|
||||||
std::string(": ") + strerror(-data->res)
|
op, (buf_num >= 0 ? bufs[buf_num].offset : last_read_offset), strerror(-data->res),
|
||||||
);
|
data->res, data->iov.iov_len
|
||||||
|
));
|
||||||
}
|
}
|
||||||
if (buf_num >= 0)
|
if (buf_num >= 0)
|
||||||
{
|
{
|
||||||
@@ -65,10 +67,11 @@ int blockstore_init_meta::loop()
|
|||||||
if (!metadata_buffer)
|
if (!metadata_buffer)
|
||||||
throw std::runtime_error("Failed to allocate metadata read buffer");
|
throw std::runtime_error("Failed to allocate metadata read buffer");
|
||||||
// Read superblock
|
// Read superblock
|
||||||
|
hdr = (blockstore_meta_header_v2_t *)memalign_or_die(MEM_ALIGNMENT, bs->dsk.meta_block_size);
|
||||||
GET_SQE();
|
GET_SQE();
|
||||||
last_read_offset = 0;
|
last_read_offset = 0;
|
||||||
data->iov = { metadata_buffer, (size_t)bs->dsk.meta_block_size };
|
data->iov = { hdr, (size_t)bs->dsk.meta_block_size };
|
||||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "read metadata header"); };
|
||||||
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
||||||
bs->ringloop->submit();
|
bs->ringloop->submit();
|
||||||
submitted++;
|
submitted++;
|
||||||
@@ -78,24 +81,8 @@ resume_1:
|
|||||||
wait_state = 1;
|
wait_state = 1;
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
if (iszero((uint64_t*)metadata_buffer, bs->dsk.meta_block_size / sizeof(uint64_t)))
|
if (iszero((uint64_t*)hdr, bs->dsk.meta_block_size / sizeof(uint64_t)))
|
||||||
{
|
{
|
||||||
{
|
|
||||||
blockstore_meta_header_v2_t *hdr = (blockstore_meta_header_v2_t *)metadata_buffer;
|
|
||||||
hdr->zero = 0;
|
|
||||||
hdr->magic = BLOCKSTORE_META_MAGIC_V1;
|
|
||||||
hdr->version = bs->dsk.meta_format;
|
|
||||||
hdr->meta_block_size = bs->dsk.meta_block_size;
|
|
||||||
hdr->data_block_size = bs->dsk.data_block_size;
|
|
||||||
hdr->bitmap_granularity = bs->dsk.bitmap_granularity;
|
|
||||||
if (bs->dsk.meta_format >= BLOCKSTORE_META_FORMAT_V2)
|
|
||||||
{
|
|
||||||
hdr->data_csum_type = bs->dsk.data_csum_type;
|
|
||||||
hdr->csum_block_size = bs->dsk.csum_block_size;
|
|
||||||
hdr->header_csum = 0;
|
|
||||||
hdr->header_csum = crc32c(0, hdr, sizeof(*hdr));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (bs->readonly)
|
if (bs->readonly)
|
||||||
{
|
{
|
||||||
printf("Skipping metadata initialization because blockstore is readonly\n");
|
printf("Skipping metadata initialization because blockstore is readonly\n");
|
||||||
@@ -103,25 +90,11 @@ resume_1:
|
|||||||
else
|
else
|
||||||
{
|
{
|
||||||
printf("Initializing metadata area\n");
|
printf("Initializing metadata area\n");
|
||||||
GET_SQE();
|
|
||||||
last_read_offset = 0;
|
|
||||||
data->iov = (struct iovec){ metadata_buffer, (size_t)bs->dsk.meta_block_size };
|
|
||||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
|
||||||
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
|
||||||
bs->ringloop->submit();
|
|
||||||
submitted++;
|
|
||||||
resume_3:
|
|
||||||
if (submitted > 0)
|
|
||||||
{
|
|
||||||
wait_state = 3;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
zero_on_init = true;
|
|
||||||
}
|
}
|
||||||
|
zero_on_init = true;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
blockstore_meta_header_v2_t *hdr = (blockstore_meta_header_v2_t *)metadata_buffer;
|
|
||||||
if (hdr->zero != 0 || hdr->magic != BLOCKSTORE_META_MAGIC_V1 || hdr->version < BLOCKSTORE_META_FORMAT_V1)
|
if (hdr->zero != 0 || hdr->magic != BLOCKSTORE_META_MAGIC_V1 || hdr->version < BLOCKSTORE_META_FORMAT_V1)
|
||||||
{
|
{
|
||||||
printf(
|
printf(
|
||||||
@@ -223,12 +196,15 @@ resume_2:
|
|||||||
GET_SQE();
|
GET_SQE();
|
||||||
assert(bufs[i].size <= 0x7fffffff);
|
assert(bufs[i].size <= 0x7fffffff);
|
||||||
data->iov = { bufs[i].buf, (size_t)bufs[i].size };
|
data->iov = { bufs[i].buf, (size_t)bufs[i].size };
|
||||||
data->callback = [this, i](ring_data_t *data) { handle_event(data, i); };
|
|
||||||
if (!zero_on_init)
|
if (!zero_on_init)
|
||||||
|
{
|
||||||
|
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "read metadata"); };
|
||||||
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
|
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
|
||||||
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
// Fill metadata with zeroes
|
// Fill metadata with zeroes
|
||||||
|
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "clear metadata"); };
|
||||||
memset(data->iov.iov_base, 0, data->iov.iov_len);
|
memset(data->iov.iov_base, 0, data->iov.iov_len);
|
||||||
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
|
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
|
||||||
}
|
}
|
||||||
@@ -256,7 +232,7 @@ resume_2:
|
|||||||
GET_SQE();
|
GET_SQE();
|
||||||
assert(bufs[i].size <= 0x7fffffff);
|
assert(bufs[i].size <= 0x7fffffff);
|
||||||
data->iov = { bufs[i].buf, (size_t)bufs[i].size };
|
data->iov = { bufs[i].buf, (size_t)bufs[i].size };
|
||||||
data->callback = [this, i](ring_data_t *data) { handle_event(data, i); };
|
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "write metadata"); };
|
||||||
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
|
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
|
||||||
bs->ringloop->submit();
|
bs->ringloop->submit();
|
||||||
bufs[i].state = INIT_META_WRITING;
|
bufs[i].state = INIT_META_WRITING;
|
||||||
@@ -285,7 +261,7 @@ resume_2:
|
|||||||
GET_SQE();
|
GET_SQE();
|
||||||
last_read_offset = (1+next_offset)*bs->dsk.meta_block_size;
|
last_read_offset = (1+next_offset)*bs->dsk.meta_block_size;
|
||||||
data->iov = { metadata_buffer, (size_t)bs->dsk.meta_block_size };
|
data->iov = { metadata_buffer, (size_t)bs->dsk.meta_block_size };
|
||||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "read metadata"); };
|
||||||
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + (1+next_offset)*bs->dsk.meta_block_size);
|
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + (1+next_offset)*bs->dsk.meta_block_size);
|
||||||
bs->ringloop->submit();
|
bs->ringloop->submit();
|
||||||
submitted++;
|
submitted++;
|
||||||
@@ -302,7 +278,7 @@ resume_5:
|
|||||||
}
|
}
|
||||||
GET_SQE();
|
GET_SQE();
|
||||||
data->iov = { metadata_buffer, (size_t)bs->dsk.meta_block_size };
|
data->iov = { metadata_buffer, (size_t)bs->dsk.meta_block_size };
|
||||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "write metadata"); };
|
||||||
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + (1+next_offset)*bs->dsk.meta_block_size);
|
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + (1+next_offset)*bs->dsk.meta_block_size);
|
||||||
bs->ringloop->submit();
|
bs->ringloop->submit();
|
||||||
submitted++;
|
submitted++;
|
||||||
@@ -317,27 +293,64 @@ resume_6:
|
|||||||
}
|
}
|
||||||
// metadata read finished
|
// metadata read finished
|
||||||
printf("Metadata entries loaded: %ju, free blocks: %ju / %ju\n", entries_loaded, bs->data_alloc->get_free_count(), bs->dsk.block_count);
|
printf("Metadata entries loaded: %ju, free blocks: %ju / %ju\n", entries_loaded, bs->data_alloc->get_free_count(), bs->dsk.block_count);
|
||||||
|
if (zero_on_init && !bs->readonly)
|
||||||
|
{
|
||||||
|
do_fsync:
|
||||||
|
if (!bs->disable_meta_fsync)
|
||||||
|
{
|
||||||
|
GET_SQE();
|
||||||
|
io_uring_prep_fsync(sqe, bs->dsk.meta_fd, IORING_FSYNC_DATASYNC);
|
||||||
|
last_read_offset = 0;
|
||||||
|
data->iov = { 0 };
|
||||||
|
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "fsync metadata"); };
|
||||||
|
submitted++;
|
||||||
|
bs->ringloop->submit();
|
||||||
|
resume_4:
|
||||||
|
if (submitted > 0)
|
||||||
|
{
|
||||||
|
wait_state = 4;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!header_written)
|
||||||
|
{
|
||||||
|
GET_SQE();
|
||||||
|
hdr->zero = 0;
|
||||||
|
hdr->magic = BLOCKSTORE_META_MAGIC_V1;
|
||||||
|
hdr->version = bs->dsk.meta_format;
|
||||||
|
hdr->meta_block_size = bs->dsk.meta_block_size;
|
||||||
|
hdr->data_block_size = bs->dsk.data_block_size;
|
||||||
|
hdr->bitmap_granularity = bs->dsk.bitmap_granularity;
|
||||||
|
if (bs->dsk.meta_format >= BLOCKSTORE_META_FORMAT_V2)
|
||||||
|
{
|
||||||
|
hdr->data_csum_type = bs->dsk.data_csum_type;
|
||||||
|
hdr->csum_block_size = bs->dsk.csum_block_size;
|
||||||
|
hdr->header_csum = 0;
|
||||||
|
hdr->header_csum = crc32c(0, hdr, sizeof(*hdr));
|
||||||
|
}
|
||||||
|
header_written = true;
|
||||||
|
last_read_offset = 0;
|
||||||
|
data->iov = (struct iovec){ hdr, (size_t)bs->dsk.meta_block_size };
|
||||||
|
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "write metadata header"); };
|
||||||
|
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
||||||
|
bs->ringloop->submit();
|
||||||
|
submitted++;
|
||||||
|
resume_3:
|
||||||
|
if (submitted > 0)
|
||||||
|
{
|
||||||
|
wait_state = 3;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
goto do_fsync;
|
||||||
|
}
|
||||||
|
}
|
||||||
if (!bs->inmemory_meta)
|
if (!bs->inmemory_meta)
|
||||||
{
|
{
|
||||||
free(metadata_buffer);
|
free(metadata_buffer);
|
||||||
metadata_buffer = NULL;
|
metadata_buffer = NULL;
|
||||||
}
|
}
|
||||||
if (zero_on_init && !bs->disable_meta_fsync)
|
free(hdr);
|
||||||
{
|
hdr = NULL;
|
||||||
GET_SQE();
|
|
||||||
io_uring_prep_fsync(sqe, bs->dsk.meta_fd, IORING_FSYNC_DATASYNC);
|
|
||||||
last_read_offset = 0;
|
|
||||||
data->iov = { 0 };
|
|
||||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
|
||||||
submitted++;
|
|
||||||
bs->ringloop->submit();
|
|
||||||
resume_4:
|
|
||||||
if (submitted > 0)
|
|
||||||
{
|
|
||||||
wait_state = 4;
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -345,6 +358,8 @@ bool blockstore_init_meta::handle_meta_block(uint8_t *buf, uint64_t entries_per_
|
|||||||
{
|
{
|
||||||
bool updated = false;
|
bool updated = false;
|
||||||
uint64_t max_i = entries_per_block;
|
uint64_t max_i = entries_per_block;
|
||||||
|
if (done_cnt > bs->dsk.block_count)
|
||||||
|
return false;
|
||||||
if (max_i > bs->dsk.block_count-done_cnt)
|
if (max_i > bs->dsk.block_count-done_cnt)
|
||||||
max_i = bs->dsk.block_count-done_cnt;
|
max_i = bs->dsk.block_count-done_cnt;
|
||||||
for (uint64_t i = 0; i < max_i; i++)
|
for (uint64_t i = 0; i < max_i; i++)
|
||||||
@@ -455,21 +470,21 @@ blockstore_init_journal::blockstore_init_journal(blockstore_impl_t *bs)
|
|||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
void blockstore_init_journal::handle_event(ring_data_t *data1)
|
void blockstore_init_journal::handle_event(ring_data_t *data)
|
||||||
{
|
{
|
||||||
if (data1->res <= 0)
|
if (data->res != data->iov.iov_len)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(
|
throw std::runtime_error(strprintf(
|
||||||
std::string("read journal failed at offset ") + std::to_string(journal_pos) +
|
"read journal failed at offset %ju: got %s (code %d), but expected %zu",
|
||||||
std::string(": ") + strerror(-data1->res)
|
journal_pos, strerror(-data->res), data->res, data->iov.iov_len
|
||||||
);
|
));
|
||||||
}
|
}
|
||||||
done.push_back({
|
done.push_back({
|
||||||
.buf = submitted_buf,
|
.buf = submitted_buf,
|
||||||
.pos = journal_pos,
|
.pos = journal_pos,
|
||||||
.len = (uint64_t)data1->res,
|
.len = (uint64_t)data->res,
|
||||||
});
|
});
|
||||||
journal_pos += data1->res;
|
journal_pos += data->res;
|
||||||
if (journal_pos >= bs->journal.len)
|
if (journal_pos >= bs->journal.len)
|
||||||
{
|
{
|
||||||
// Continue from the beginning
|
// Continue from the beginning
|
||||||
|
|||||||
@@ -16,7 +16,9 @@ class blockstore_init_meta
|
|||||||
blockstore_impl_t *bs;
|
blockstore_impl_t *bs;
|
||||||
int wait_state = 0;
|
int wait_state = 0;
|
||||||
bool zero_on_init = false;
|
bool zero_on_init = false;
|
||||||
|
bool header_written = false;
|
||||||
void *metadata_buffer = NULL;
|
void *metadata_buffer = NULL;
|
||||||
|
blockstore_meta_header_v2_t *hdr = NULL;
|
||||||
blockstore_init_meta_buf bufs[2] = {};
|
blockstore_init_meta_buf bufs[2] = {};
|
||||||
int submitted = 0;
|
int submitted = 0;
|
||||||
struct io_uring_sqe *sqe;
|
struct io_uring_sqe *sqe;
|
||||||
@@ -29,7 +31,7 @@ class blockstore_init_meta
|
|||||||
int i = 0, j = 0;
|
int i = 0, j = 0;
|
||||||
std::vector<uint64_t> entries_to_zero;
|
std::vector<uint64_t> entries_to_zero;
|
||||||
bool handle_meta_block(uint8_t *buf, uint64_t count, uint64_t done_cnt);
|
bool handle_meta_block(uint8_t *buf, uint64_t count, uint64_t done_cnt);
|
||||||
void handle_event(ring_data_t *data, int buf_num);
|
void handle_event(ring_data_t *data, int buf_num, const char *op);
|
||||||
public:
|
public:
|
||||||
blockstore_init_meta(blockstore_impl_t *bs);
|
blockstore_init_meta(blockstore_impl_t *bs);
|
||||||
int loop();
|
int loop();
|
||||||
|
|||||||
@@ -193,6 +193,7 @@ void blockstore_impl_t::prepare_journal_sector_write(int cur_sector, blockstore_
|
|||||||
(size_t)journal.block_size
|
(size_t)journal.block_size
|
||||||
};
|
};
|
||||||
data->callback = [this, flush_id = journal.submit_id](ring_data_t *data) { handle_journal_write(data, flush_id); };
|
data->callback = [this, flush_id = journal.submit_id](ring_data_t *data) { handle_journal_write(data, flush_id); };
|
||||||
|
assert(journal.sector_info[cur_sector].offset+journal.block_size <= dsk.journal_len);
|
||||||
io_uring_prep_writev(
|
io_uring_prep_writev(
|
||||||
sqe, dsk.journal_fd, &data->iov, 1, journal.offset + journal.sector_info[cur_sector].offset
|
sqe, dsk.journal_fd, &data->iov, 1, journal.offset + journal.sector_info[cur_sector].offset
|
||||||
);
|
);
|
||||||
|
|||||||
+145
-54
@@ -101,8 +101,8 @@ int blockstore_impl_t::fulfill_read(blockstore_op_t *read_op,
|
|||||||
.copy_flags = COPY_BUF_JOURNAL|COPY_BUF_CSUM_FILL,
|
.copy_flags = COPY_BUF_JOURNAL|COPY_BUF_CSUM_FILL,
|
||||||
.offset = blk_begin,
|
.offset = blk_begin,
|
||||||
.len = blk_end-blk_begin,
|
.len = blk_end-blk_begin,
|
||||||
.csum_buf = (csum + (blk_begin/dsk.csum_block_size -
|
.csum_buf = (!csum ? NULL : (csum + (blk_begin/dsk.csum_block_size -
|
||||||
item_start/dsk.csum_block_size) * (dsk.data_csum_type & 0xFF)),
|
item_start/dsk.csum_block_size) * (dsk.data_csum_type & 0xFF))),
|
||||||
.dyn_data = dyn_data,
|
.dyn_data = dyn_data,
|
||||||
});
|
});
|
||||||
if (dyn_data)
|
if (dyn_data)
|
||||||
@@ -134,7 +134,7 @@ int blockstore_impl_t::fulfill_read(blockstore_op_t *read_op,
|
|||||||
// If we don't track it then we may IN THEORY read another object's data:
|
// If we don't track it then we may IN THEORY read another object's data:
|
||||||
// submit read -> remove the object -> flush remove -> overwrite with another object -> finish read
|
// submit read -> remove the object -> flush remove -> overwrite with another object -> finish read
|
||||||
// Very improbable, but possible
|
// Very improbable, but possible
|
||||||
PRIV(read_op)->clean_block_used = 1;
|
PRIV(read_op)->clean_loc_used = UINT64_MAX;
|
||||||
}
|
}
|
||||||
rv.insert(rv.begin() + pos, el);
|
rv.insert(rv.begin() + pos, el);
|
||||||
fulfilled += el.len;
|
fulfilled += el.len;
|
||||||
@@ -167,7 +167,8 @@ uint8_t* blockstore_impl_t::get_clean_entry_bitmap(uint64_t block_loc, int offse
|
|||||||
}
|
}
|
||||||
|
|
||||||
int blockstore_impl_t::fill_partial_checksum_blocks(std::vector<copy_buffer_t> & rv, uint64_t & fulfilled,
|
int blockstore_impl_t::fill_partial_checksum_blocks(std::vector<copy_buffer_t> & rv, uint64_t & fulfilled,
|
||||||
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf, uint64_t read_offset, uint64_t read_end)
|
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf,
|
||||||
|
uint32_t read_offset, uint32_t read_end, uint32_t item_start, uint32_t item_end)
|
||||||
{
|
{
|
||||||
if (read_end == read_offset)
|
if (read_end == read_offset)
|
||||||
return 0;
|
return 0;
|
||||||
@@ -175,10 +176,38 @@ int blockstore_impl_t::fill_partial_checksum_blocks(std::vector<copy_buffer_t> &
|
|||||||
read_buf -= read_offset;
|
read_buf -= read_offset;
|
||||||
uint32_t last_block = (read_end-1)/dsk.csum_block_size;
|
uint32_t last_block = (read_end-1)/dsk.csum_block_size;
|
||||||
uint32_t start_block = read_offset/dsk.csum_block_size;
|
uint32_t start_block = read_offset/dsk.csum_block_size;
|
||||||
|
uint32_t item_start_block = item_start/dsk.csum_block_size;
|
||||||
uint32_t end_block = 0;
|
uint32_t end_block = 0;
|
||||||
|
auto zero_range = [&](int pos, bool alloc, uint32_t cur_start, uint32_t cur_end)
|
||||||
|
{
|
||||||
|
if (alloc)
|
||||||
|
return 0;
|
||||||
|
copy_buffer_t el = {
|
||||||
|
.copy_flags = COPY_BUF_ZERO,
|
||||||
|
.offset = cur_start,
|
||||||
|
.len = cur_end-cur_start,
|
||||||
|
};
|
||||||
|
rv.insert(rv.begin() + pos, el);
|
||||||
|
if (read_buf)
|
||||||
|
memset(read_buf + el.offset - read_offset, 0, el.len);
|
||||||
|
fulfilled += el.len;
|
||||||
|
return 1;
|
||||||
|
};
|
||||||
|
if (read_offset < item_start)
|
||||||
|
{
|
||||||
|
// Zero-fill the beginning
|
||||||
|
find_holes(rv, read_offset, item_start, zero_range);
|
||||||
|
read_offset = item_start;
|
||||||
|
}
|
||||||
|
if (read_end > item_end)
|
||||||
|
{
|
||||||
|
// Zero-fill the end
|
||||||
|
find_holes(rv, item_end, read_end, zero_range);
|
||||||
|
read_end = item_end;
|
||||||
|
}
|
||||||
while (start_block <= last_block)
|
while (start_block <= last_block)
|
||||||
{
|
{
|
||||||
if (read_range_fulfilled(rv, fulfilled, read_buf, clean_entry_bitmap,
|
if (read_range_fulfilled(rv, fulfilled, read_buf, from_journal ? NULL : clean_entry_bitmap,
|
||||||
start_block*dsk.csum_block_size < read_offset ? read_offset : start_block*dsk.csum_block_size,
|
start_block*dsk.csum_block_size < read_offset ? read_offset : start_block*dsk.csum_block_size,
|
||||||
(start_block+1)*dsk.csum_block_size > read_end ? read_end : (start_block+1)*dsk.csum_block_size))
|
(start_block+1)*dsk.csum_block_size > read_end ? read_end : (start_block+1)*dsk.csum_block_size))
|
||||||
{
|
{
|
||||||
@@ -190,7 +219,7 @@ int blockstore_impl_t::fill_partial_checksum_blocks(std::vector<copy_buffer_t> &
|
|||||||
// Find a sequence of checksum blocks required to be read
|
// Find a sequence of checksum blocks required to be read
|
||||||
end_block = start_block;
|
end_block = start_block;
|
||||||
while ((end_block+1)*dsk.csum_block_size < read_end &&
|
while ((end_block+1)*dsk.csum_block_size < read_end &&
|
||||||
!read_range_fulfilled(rv, fulfilled, read_buf, clean_entry_bitmap,
|
!read_range_fulfilled(rv, fulfilled, read_buf, from_journal ? NULL : clean_entry_bitmap,
|
||||||
(end_block+1)*dsk.csum_block_size < read_offset ? read_offset : (end_block+1)*dsk.csum_block_size,
|
(end_block+1)*dsk.csum_block_size < read_offset ? read_offset : (end_block+1)*dsk.csum_block_size,
|
||||||
(end_block+2)*dsk.csum_block_size > read_end ? read_end : (end_block+2)*dsk.csum_block_size))
|
(end_block+2)*dsk.csum_block_size > read_end ? read_end : (end_block+2)*dsk.csum_block_size))
|
||||||
{
|
{
|
||||||
@@ -202,8 +231,10 @@ int blockstore_impl_t::fill_partial_checksum_blocks(std::vector<copy_buffer_t> &
|
|||||||
.copy_flags = COPY_BUF_CSUM_FILL | (from_journal ? COPY_BUF_JOURNALED_BIG : 0),
|
.copy_flags = COPY_BUF_CSUM_FILL | (from_journal ? COPY_BUF_JOURNALED_BIG : 0),
|
||||||
.offset = start_block*dsk.csum_block_size,
|
.offset = start_block*dsk.csum_block_size,
|
||||||
.len = (end_block-start_block)*dsk.csum_block_size,
|
.len = (end_block-start_block)*dsk.csum_block_size,
|
||||||
// save clean_entry_bitmap if we're reading clean data from the journal
|
// save checksum reference if we're reading clean data from the journal
|
||||||
.csum_buf = from_journal ? clean_entry_bitmap : NULL,
|
.csum_buf = from_journal
|
||||||
|
? clean_entry_bitmap + dsk.clean_entry_bitmap_size + (start_block-item_start_block)*(dsk.data_csum_type & 0xFF)
|
||||||
|
: NULL,
|
||||||
.dyn_data = dyn_data,
|
.dyn_data = dyn_data,
|
||||||
});
|
});
|
||||||
if (dyn_data)
|
if (dyn_data)
|
||||||
@@ -226,6 +257,11 @@ bool blockstore_impl_t::read_range_fulfilled(std::vector<copy_buffer_t> & rv, ui
|
|||||||
{
|
{
|
||||||
if (alloc)
|
if (alloc)
|
||||||
return 0;
|
return 0;
|
||||||
|
if (!clean_entry_bitmap)
|
||||||
|
{
|
||||||
|
all_done = false;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
int diff = 0;
|
int diff = 0;
|
||||||
uint32_t bmp_start = cur_start/dsk.bitmap_granularity;
|
uint32_t bmp_start = cur_start/dsk.bitmap_granularity;
|
||||||
uint32_t bmp_end = cur_end/dsk.bitmap_granularity;
|
uint32_t bmp_end = cur_end/dsk.bitmap_granularity;
|
||||||
@@ -323,7 +359,7 @@ bool blockstore_impl_t::read_checksum_block(blockstore_op_t *op, int rv_pos, uin
|
|||||||
{
|
{
|
||||||
iov[n_iov++] = (struct iovec){ (uint8_t*)op->buf+cur_start-op->offset, lim_end-cur_start };
|
iov[n_iov++] = (struct iovec){ (uint8_t*)op->buf+cur_start-op->offset, lim_end-cur_start };
|
||||||
rv.insert(rv.begin() + pos, (copy_buffer_t){
|
rv.insert(rv.begin() + pos, (copy_buffer_t){
|
||||||
.copy_flags = COPY_BUF_DATA,
|
.copy_flags = COPY_BUF_DATA|COPY_BUF_COALESCED,
|
||||||
.offset = cur_start,
|
.offset = cur_start,
|
||||||
.len = lim_end-cur_start,
|
.len = lim_end-cur_start,
|
||||||
});
|
});
|
||||||
@@ -361,10 +397,10 @@ bool blockstore_impl_t::read_checksum_block(blockstore_op_t *op, int rv_pos, uin
|
|||||||
PRIV(op)->pending_ops++;
|
PRIV(op)->pending_ops++;
|
||||||
io_uring_prep_readv(sqe, submit_fd, iov + n_pos, n_cur, submit_offset + clean_loc + item_start + d_pos);
|
io_uring_prep_readv(sqe, submit_fd, iov + n_pos, n_cur, submit_offset + clean_loc + item_start + d_pos);
|
||||||
data->callback = [this, op](ring_data_t *data) { handle_read_event(data, op); };
|
data->callback = [this, op](ring_data_t *data) { handle_read_event(data, op); };
|
||||||
if (n_pos > 0 || n_pos + IOV_MAX < n_iov)
|
if (n_pos > 0 || n_iov > IOV_MAX)
|
||||||
{
|
{
|
||||||
uint32_t d_len = 0;
|
uint32_t d_len = 0;
|
||||||
for (int i = 0; i < IOV_MAX; i++)
|
for (int i = 0; i < n_cur; i++)
|
||||||
d_len += iov[n_pos+i].iov_len;
|
d_len += iov[n_pos+i].iov_len;
|
||||||
data->iov.iov_len = d_len;
|
data->iov.iov_len = d_len;
|
||||||
d_pos += d_len;
|
d_pos += d_len;
|
||||||
@@ -376,7 +412,7 @@ bool blockstore_impl_t::read_checksum_block(blockstore_op_t *op, int rv_pos, uin
|
|||||||
{
|
{
|
||||||
// Reads running parallel to flushes of the same clean block may read
|
// Reads running parallel to flushes of the same clean block may read
|
||||||
// a mixture of old and new data. So we don't verify checksums for such blocks.
|
// a mixture of old and new data. So we don't verify checksums for such blocks.
|
||||||
PRIV(op)->clean_block_used = 1;
|
PRIV(op)->clean_loc_used = UINT64_MAX;
|
||||||
}
|
}
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
@@ -402,7 +438,7 @@ int blockstore_impl_t::dequeue_read(blockstore_op_t *read_op)
|
|||||||
}
|
}
|
||||||
uint64_t fulfilled = 0;
|
uint64_t fulfilled = 0;
|
||||||
PRIV(read_op)->pending_ops = 0;
|
PRIV(read_op)->pending_ops = 0;
|
||||||
PRIV(read_op)->clean_block_used = 0;
|
PRIV(read_op)->clean_loc_used = 0;
|
||||||
auto & rv = PRIV(read_op)->read_vec;
|
auto & rv = PRIV(read_op)->read_vec;
|
||||||
uint64_t result_version = 0;
|
uint64_t result_version = 0;
|
||||||
if (dirty_found)
|
if (dirty_found)
|
||||||
@@ -515,26 +551,50 @@ int blockstore_impl_t::dequeue_read(blockstore_op_t *read_op)
|
|||||||
return 2;
|
return 2;
|
||||||
undo_read:
|
undo_read:
|
||||||
// need to wait. undo added requests, don't dequeue op
|
// need to wait. undo added requests, don't dequeue op
|
||||||
if (dsk.csum_block_size > dsk.bitmap_granularity)
|
release_clean(read_op);
|
||||||
|
for (auto & vec: rv)
|
||||||
{
|
{
|
||||||
for (auto & vec: rv)
|
if ((vec.copy_flags & COPY_BUF_CSUM_FILL) && vec.buf)
|
||||||
{
|
{
|
||||||
if ((vec.copy_flags & COPY_BUF_CSUM_FILL) && vec.buf)
|
free(vec.buf);
|
||||||
{
|
vec.buf = NULL;
|
||||||
free(vec.buf);
|
}
|
||||||
vec.buf = NULL;
|
if (vec.dyn_data && --(*vec.dyn_data) == 0) // refcount
|
||||||
}
|
{
|
||||||
if (vec.dyn_data && --(*vec.dyn_data) == 0) // refcount
|
free(vec.dyn_data);
|
||||||
{
|
vec.dyn_data = NULL;
|
||||||
free(vec.dyn_data);
|
|
||||||
vec.dyn_data = NULL;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
rv.clear();
|
rv.clear();
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void blockstore_impl_t::release_clean(blockstore_op_t *op)
|
||||||
|
{
|
||||||
|
if (PRIV(op)->clean_loc_used == UINT64_MAX)
|
||||||
|
{
|
||||||
|
PRIV(op)->clean_loc_used = 0;
|
||||||
|
}
|
||||||
|
if (PRIV(op)->clean_loc_used)
|
||||||
|
{
|
||||||
|
// Release clean data block
|
||||||
|
auto uo_it = used_clean_objects.find(PRIV(op)->clean_loc_used - 1);
|
||||||
|
if (uo_it != used_clean_objects.end())
|
||||||
|
{
|
||||||
|
uo_it->second.refs--;
|
||||||
|
if (uo_it->second.refs <= 0)
|
||||||
|
{
|
||||||
|
if (uo_it->second.was_freed)
|
||||||
|
{
|
||||||
|
data_alloc->set((PRIV(op)->clean_loc_used - 1) / dsk.data_block_size, false);
|
||||||
|
}
|
||||||
|
used_clean_objects.erase(uo_it);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
PRIV(op)->clean_loc_used = 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
int blockstore_impl_t::pad_journal_read(std::vector<copy_buffer_t> & rv, copy_buffer_t & cp,
|
int blockstore_impl_t::pad_journal_read(std::vector<copy_buffer_t> & rv, copy_buffer_t & cp,
|
||||||
// FIXME Passing dirty_entry& would be nicer
|
// FIXME Passing dirty_entry& would be nicer
|
||||||
uint64_t dirty_offset, uint64_t dirty_end, uint64_t dirty_loc, uint8_t *csum_ptr, int *dyn_data,
|
uint64_t dirty_offset, uint64_t dirty_end, uint64_t dirty_loc, uint8_t *csum_ptr, int *dyn_data,
|
||||||
@@ -598,11 +658,15 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
|
|||||||
{
|
{
|
||||||
auto & rv = PRIV(read_op)->read_vec;
|
auto & rv = PRIV(read_op)->read_vec;
|
||||||
int req = fill_partial_checksum_blocks(rv, fulfilled, clean_entry_bitmap, dyn_data, from_journal,
|
int req = fill_partial_checksum_blocks(rv, fulfilled, clean_entry_bitmap, dyn_data, from_journal,
|
||||||
(uint8_t*)read_op->buf, read_op->offset, read_op->offset+read_op->len);
|
(uint8_t*)read_op->buf, read_op->offset, read_op->offset+read_op->len, item_start, item_end);
|
||||||
if (!inmemory_meta && !from_journal && req > 0)
|
if (!inmemory_meta && !from_journal && req > 0)
|
||||||
{
|
{
|
||||||
// Read checksums from disk
|
// Read checksums from disk
|
||||||
uint8_t *csum_buf = read_clean_meta_block(read_op, clean_loc, rv.size()-req);
|
uint8_t *csum_buf = read_clean_meta_block(read_op, clean_loc, rv.size()-req);
|
||||||
|
if (!csum_buf)
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
for (int i = req; i > 0; i--)
|
for (int i = req; i > 0; i--)
|
||||||
{
|
{
|
||||||
rv[rv.size()-i].csum_buf = csum_buf;
|
rv[rv.size()-i].csum_buf = csum_buf;
|
||||||
@@ -615,7 +679,7 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
|
|||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
PRIV(read_op)->clean_block_used = req > 0;
|
PRIV(read_op)->clean_loc_used = req > 0 ? UINT64_MAX : 0;
|
||||||
}
|
}
|
||||||
else if (from_journal)
|
else if (from_journal)
|
||||||
{
|
{
|
||||||
@@ -665,6 +729,10 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
|
|||||||
{
|
{
|
||||||
// Read checksums from disk
|
// Read checksums from disk
|
||||||
csum_buf = read_clean_meta_block(read_op, clean_loc, PRIV(read_op)->read_vec.size());
|
csum_buf = read_clean_meta_block(read_op, clean_loc, PRIV(read_op)->read_vec.size());
|
||||||
|
if (!csum_buf)
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
csum_done = true;
|
csum_done = true;
|
||||||
}
|
}
|
||||||
uint8_t *csum = !dsk.csum_block_size ? 0 : (csum_buf + 2*dsk.clean_entry_bitmap_size + bmp_start*(dsk.data_csum_type & 0xFF));
|
uint8_t *csum = !dsk.csum_block_size ? 0 : (csum_buf + 2*dsk.clean_entry_bitmap_size + bmp_start*(dsk.data_csum_type & 0xFF));
|
||||||
@@ -679,13 +747,13 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Increment reference counter if clean data is being read from the disk
|
// Increment reference counter if clean data is being read from the disk
|
||||||
if (PRIV(read_op)->clean_block_used)
|
if (PRIV(read_op)->clean_loc_used == UINT64_MAX)
|
||||||
{
|
{
|
||||||
auto & uo = used_clean_objects[clean_loc];
|
auto & uo = used_clean_objects[clean_loc];
|
||||||
uo.refs++;
|
uo.refs++;
|
||||||
if (dsk.csum_block_size && flusher->is_mutated(clean_loc))
|
if (dsk.csum_block_size && flusher->is_mutated(clean_loc))
|
||||||
uo.was_changed = true;
|
uo.was_changed = true;
|
||||||
PRIV(read_op)->clean_block_used = clean_loc;
|
PRIV(read_op)->clean_loc_used = clean_loc + 1;
|
||||||
}
|
}
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
@@ -725,12 +793,18 @@ bool blockstore_impl_t::verify_padded_checksums(uint8_t *clean_entry_bitmap, uin
|
|||||||
while (pos < iov[i].iov_len)
|
while (pos < iov[i].iov_len)
|
||||||
{
|
{
|
||||||
uint32_t start = pos;
|
uint32_t start = pos;
|
||||||
uint8_t bit = (clean_entry_bitmap[bmp_pos >> 3] >> (bmp_pos & 0x7)) & 1;
|
uint8_t bit = 1;
|
||||||
while (pos < iov[i].iov_len && ((clean_entry_bitmap[bmp_pos >> 3] >> (bmp_pos & 0x7)) & 1) == bit)
|
if (clean_entry_bitmap)
|
||||||
{
|
{
|
||||||
pos += dsk.bitmap_granularity;
|
bit = (clean_entry_bitmap[bmp_pos >> 3] >> (bmp_pos & 0x7)) & 1;
|
||||||
bmp_pos++;
|
while (pos < iov[i].iov_len && ((clean_entry_bitmap[bmp_pos >> 3] >> (bmp_pos & 0x7)) & 1) == bit)
|
||||||
|
{
|
||||||
|
pos += dsk.bitmap_granularity;
|
||||||
|
bmp_pos++;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
else
|
||||||
|
pos = iov[i].iov_len;
|
||||||
uint32_t len = pos-start;
|
uint32_t len = pos-start;
|
||||||
auto buf = (uint8_t*)iov[i].iov_base+start;
|
auto buf = (uint8_t*)iov[i].iov_base+start;
|
||||||
while (block_done+len >= dsk.csum_block_size)
|
while (block_done+len >= dsk.csum_block_size)
|
||||||
@@ -807,7 +881,7 @@ bool blockstore_impl_t::verify_clean_padded_checksums(blockstore_op_t *op, uint6
|
|||||||
{
|
{
|
||||||
uint32_t offset = clean_loc % dsk.data_block_size;
|
uint32_t offset = clean_loc % dsk.data_block_size;
|
||||||
if (from_journal)
|
if (from_journal)
|
||||||
return verify_padded_checksums(dyn_data, dyn_data + dsk.clean_entry_bitmap_size, offset, iov, n_iov, bad_block_cb);
|
return verify_padded_checksums(NULL, dyn_data, offset, iov, n_iov, bad_block_cb);
|
||||||
clean_loc = (clean_loc / dsk.data_block_size) * dsk.data_block_size;
|
clean_loc = (clean_loc / dsk.data_block_size) * dsk.data_block_size;
|
||||||
if (!dyn_data)
|
if (!dyn_data)
|
||||||
{
|
{
|
||||||
@@ -835,7 +909,7 @@ void blockstore_impl_t::handle_read_event(ring_data_t *data, blockstore_op_t *op
|
|||||||
void *meta_block = NULL;
|
void *meta_block = NULL;
|
||||||
if (dsk.csum_block_size > dsk.bitmap_granularity)
|
if (dsk.csum_block_size > dsk.bitmap_granularity)
|
||||||
{
|
{
|
||||||
for (int i = rv.size()-1; i >= 0 && (rv[i].copy_flags & COPY_BUF_CSUM_FILL); i--)
|
for (int i = 0; i < rv.size(); i++)
|
||||||
{
|
{
|
||||||
if (rv[i].copy_flags & COPY_BUF_META_BLOCK)
|
if (rv[i].copy_flags & COPY_BUF_META_BLOCK)
|
||||||
{
|
{
|
||||||
@@ -845,8 +919,41 @@ void blockstore_impl_t::handle_read_event(ring_data_t *data, blockstore_op_t *op
|
|||||||
rv[i].buf = NULL;
|
rv[i].buf = NULL;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
struct iovec *iov = (struct iovec*)((uint8_t*)rv[i].buf + (rv[i].len & 0xFFFFFFFF));
|
if (rv[i].copy_flags & COPY_BUF_ZERO)
|
||||||
int n_iov = rv[i].len >> 32;
|
{
|
||||||
|
// Zero read
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (rv[i].copy_flags & COPY_BUF_COALESCED)
|
||||||
|
{
|
||||||
|
// Sub-block shared with another read. Skip
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if ((rv[i].copy_flags & COPY_BUF_JOURNAL) && journal.inmemory)
|
||||||
|
{
|
||||||
|
// Do not check journal checksums in-memory
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
iovec single_iov = {};
|
||||||
|
iovec *iov = NULL;
|
||||||
|
int n_iov = 0;
|
||||||
|
if (rv[i].copy_flags & COPY_BUF_CSUM_FILL)
|
||||||
|
{
|
||||||
|
// Padded, buffer list passed using a 'creepy way'
|
||||||
|
iov = (struct iovec*)((uint8_t*)rv[i].buf + (rv[i].len & 0xFFFFFFFF));
|
||||||
|
n_iov = rv[i].len >> 32;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// Not padded, buffer is fully within the input buffer
|
||||||
|
assert(op->buf);
|
||||||
|
assert(rv[i].csum_buf);
|
||||||
|
iov = &single_iov;
|
||||||
|
n_iov = 1;
|
||||||
|
assert(rv[i].offset >= op->offset);
|
||||||
|
assert(rv[i].offset + rv[i].len <= op->offset + op->len);
|
||||||
|
single_iov = { .iov_base = op->buf + rv[i].offset - op->offset, .iov_len = rv[i].len };
|
||||||
|
}
|
||||||
bool ok = true;
|
bool ok = true;
|
||||||
if (rv[i].copy_flags & COPY_BUF_JOURNAL)
|
if (rv[i].copy_flags & COPY_BUF_JOURNAL)
|
||||||
{
|
{
|
||||||
@@ -944,23 +1051,7 @@ void blockstore_impl_t::handle_read_event(ring_data_t *data, blockstore_op_t *op
|
|||||||
meta_block = NULL;
|
meta_block = NULL;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (PRIV(op)->clean_block_used)
|
release_clean(op);
|
||||||
{
|
|
||||||
// Release clean data block
|
|
||||||
auto uo_it = used_clean_objects.find(PRIV(op)->clean_block_used);
|
|
||||||
if (uo_it != used_clean_objects.end())
|
|
||||||
{
|
|
||||||
uo_it->second.refs--;
|
|
||||||
if (uo_it->second.refs <= 0)
|
|
||||||
{
|
|
||||||
if (uo_it->second.was_freed)
|
|
||||||
{
|
|
||||||
data_alloc->set(PRIV(op)->clean_block_used, false);
|
|
||||||
}
|
|
||||||
used_clean_objects.erase(uo_it);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (!journal.inmemory)
|
if (!journal.inmemory)
|
||||||
{
|
{
|
||||||
// Release journal sector usage
|
// Release journal sector usage
|
||||||
|
|||||||
@@ -491,7 +491,7 @@ void blockstore_impl_t::mark_stable(obj_ver_id v, bool forget_dirty)
|
|||||||
if (!exists)
|
if (!exists)
|
||||||
{
|
{
|
||||||
uint64_t space_id = dirty_it->first.oid.inode;
|
uint64_t space_id = dirty_it->first.oid.inode;
|
||||||
if (no_inode_stats[dirty_it->first.oid.inode >> (64-POOL_ID_BITS)])
|
if (no_inode_stats.find(dirty_it->first.oid.inode >> (64-POOL_ID_BITS)) != no_inode_stats.end())
|
||||||
space_id = space_id & ~(((uint64_t)1 << (64-POOL_ID_BITS)) - 1);
|
space_id = space_id & ~(((uint64_t)1 << (64-POOL_ID_BITS)) - 1);
|
||||||
inode_space_stats[space_id] += dsk.data_block_size;
|
inode_space_stats[space_id] += dsk.data_block_size;
|
||||||
used_blocks++;
|
used_blocks++;
|
||||||
@@ -501,7 +501,7 @@ void blockstore_impl_t::mark_stable(obj_ver_id v, bool forget_dirty)
|
|||||||
else if (IS_DELETE(dirty_it->second.state))
|
else if (IS_DELETE(dirty_it->second.state))
|
||||||
{
|
{
|
||||||
uint64_t space_id = dirty_it->first.oid.inode;
|
uint64_t space_id = dirty_it->first.oid.inode;
|
||||||
if (no_inode_stats[dirty_it->first.oid.inode >> (64-POOL_ID_BITS)])
|
if (no_inode_stats.find(dirty_it->first.oid.inode >> (64-POOL_ID_BITS)) != no_inode_stats.end())
|
||||||
space_id = space_id & ~(((uint64_t)1 << (64-POOL_ID_BITS)) - 1);
|
space_id = space_id & ~(((uint64_t)1 << (64-POOL_ID_BITS)) - 1);
|
||||||
auto & sp = inode_space_stats[space_id];
|
auto & sp = inode_space_stats[space_id];
|
||||||
if (sp > dsk.data_block_size)
|
if (sp > dsk.data_block_size)
|
||||||
|
|||||||
@@ -368,9 +368,9 @@ int blockstore_impl_t::dequeue_write(blockstore_op_t *op)
|
|||||||
}
|
}
|
||||||
data->iov.iov_len = op->len + stripe_offset + stripe_end; // to check it in the callback
|
data->iov.iov_len = op->len + stripe_offset + stripe_end; // to check it in the callback
|
||||||
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
||||||
io_uring_prep_writev(
|
const uint64_t write_offset = (loc * dsk.data_block_size) + op->offset - stripe_offset;
|
||||||
sqe, dsk.data_fd, PRIV(op)->iov_zerofill, vcnt, dsk.data_offset + (loc * dsk.data_block_size) + op->offset - stripe_offset
|
assert(write_offset+op->len+stripe_offset+stripe_end <= dsk.block_count*dsk.data_block_size);
|
||||||
);
|
io_uring_prep_writev(sqe, dsk.data_fd, PRIV(op)->iov_zerofill, vcnt, dsk.data_offset + write_offset);
|
||||||
PRIV(op)->pending_ops = 1;
|
PRIV(op)->pending_ops = 1;
|
||||||
if (!(dirty_it->second.state & BS_ST_INSTANT))
|
if (!(dirty_it->second.state & BS_ST_INSTANT))
|
||||||
{
|
{
|
||||||
@@ -495,9 +495,8 @@ int blockstore_impl_t::dequeue_write(blockstore_op_t *op)
|
|||||||
.op = op,
|
.op = op,
|
||||||
});
|
});
|
||||||
data2->callback = [this, flush_id = journal.submit_id](ring_data_t *data) { handle_journal_write(data, flush_id); };
|
data2->callback = [this, flush_id = journal.submit_id](ring_data_t *data) { handle_journal_write(data, flush_id); };
|
||||||
io_uring_prep_writev(
|
assert(journal.next_free+op->len <= dsk.journal_len);
|
||||||
sqe2, dsk.journal_fd, &data2->iov, 1, journal.offset + journal.next_free
|
io_uring_prep_writev(sqe2, dsk.journal_fd, &data2->iov, 1, journal.offset + journal.next_free);
|
||||||
);
|
|
||||||
PRIV(op)->pending_ops++;
|
PRIV(op)->pending_ops++;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
|
|||||||
+38
-14
@@ -1,8 +1,24 @@
|
|||||||
cmake_minimum_required(VERSION 2.8.12)
|
cmake_minimum_required(VERSION 2.8...3.30)
|
||||||
|
|
||||||
project(vitastor)
|
project(vitastor)
|
||||||
|
|
||||||
# libvitastor_common.a
|
# libvitastor_common.a
|
||||||
|
add_library(vitastor_common STATIC
|
||||||
|
etcd_state_client.cpp
|
||||||
|
msgr_stop.cpp
|
||||||
|
msgr_op.cpp
|
||||||
|
../../json11/json11.cpp
|
||||||
|
osd_ops.cpp
|
||||||
|
pg_states.cpp
|
||||||
|
../util/allocator.cpp
|
||||||
|
../util/addr_util.cpp
|
||||||
|
../util/timerfd_manager.cpp
|
||||||
|
../util/str_util.cpp
|
||||||
|
../util/json_util.cpp
|
||||||
|
)
|
||||||
|
target_compile_options(vitastor_common PUBLIC -fPIC)
|
||||||
|
|
||||||
|
# libvitastor_net.a
|
||||||
set(MSGR_RDMA "")
|
set(MSGR_RDMA "")
|
||||||
if (IBVERBS_LIBRARIES)
|
if (IBVERBS_LIBRARIES)
|
||||||
set(MSGR_RDMA "msgr_rdma.cpp")
|
set(MSGR_RDMA "msgr_rdma.cpp")
|
||||||
@@ -11,24 +27,32 @@ set(MSGR_RDMACM "")
|
|||||||
if (RDMACM_LIBRARIES)
|
if (RDMACM_LIBRARIES)
|
||||||
set(MSGR_RDMACM "msgr_rdmacm.cpp")
|
set(MSGR_RDMACM "msgr_rdmacm.cpp")
|
||||||
endif (RDMACM_LIBRARIES)
|
endif (RDMACM_LIBRARIES)
|
||||||
add_library(vitastor_common STATIC
|
add_library(vitastor_net STATIC
|
||||||
../util/epoll_manager.cpp etcd_state_client.cpp messenger.cpp ../util/addr_util.cpp
|
../util/epoll_manager.cpp
|
||||||
msgr_stop.cpp msgr_op.cpp msgr_send.cpp msgr_receive.cpp ../util/ringloop.cpp ../../json11/json11.cpp
|
etcd_state_client_http.cpp
|
||||||
http_client.cpp osd_ops.cpp pg_states.cpp ../util/timerfd_manager.cpp ../util/str_util.cpp ../util/json_util.cpp ${MSGR_RDMA} ${MSGR_RDMACM}
|
messenger.cpp
|
||||||
|
msgr_iothread.cpp
|
||||||
|
msgr_send.cpp
|
||||||
|
msgr_receive.cpp
|
||||||
|
../util/ringloop.cpp
|
||||||
|
http_client.cpp
|
||||||
|
${MSGR_RDMA}
|
||||||
|
${MSGR_RDMACM}
|
||||||
)
|
)
|
||||||
target_link_libraries(vitastor_common pthread)
|
target_link_libraries(vitastor_net pthread vitastor_common)
|
||||||
target_compile_options(vitastor_common PUBLIC -fPIC)
|
target_compile_options(vitastor_net PUBLIC -fPIC)
|
||||||
|
|
||||||
# libvitastor_client.so
|
# libvitastor_client.so
|
||||||
add_library(vitastor_client SHARED
|
add_library(vitastor_client SHARED
|
||||||
cluster_client.cpp
|
cluster_client.cpp
|
||||||
|
cluster_client_real.cpp
|
||||||
cluster_client_list.cpp
|
cluster_client_list.cpp
|
||||||
cluster_client_wb.cpp
|
cluster_client_wb.cpp
|
||||||
vitastor_c.cpp
|
vitastor_c.cpp
|
||||||
)
|
)
|
||||||
set_target_properties(vitastor_client PROPERTIES PUBLIC_HEADER "client/vitastor_c.h")
|
set_target_properties(vitastor_client PROPERTIES PUBLIC_HEADER "client/vitastor_c.h")
|
||||||
target_link_libraries(vitastor_client
|
target_link_libraries(vitastor_client
|
||||||
vitastor_common
|
vitastor_net
|
||||||
vitastor_cli
|
vitastor_cli
|
||||||
${LIBURING_LIBRARIES}
|
${LIBURING_LIBRARIES}
|
||||||
${IBVERBS_LIBRARIES}
|
${IBVERBS_LIBRARIES}
|
||||||
@@ -52,9 +76,6 @@ if (${WITH_FIO})
|
|||||||
../util/rw_blocking.cpp
|
../util/rw_blocking.cpp
|
||||||
../util/addr_util.cpp
|
../util/addr_util.cpp
|
||||||
)
|
)
|
||||||
target_link_libraries(fio_vitastor_sec
|
|
||||||
tcmalloc_minimal
|
|
||||||
)
|
|
||||||
endif (${WITH_FIO})
|
endif (${WITH_FIO})
|
||||||
|
|
||||||
# vitastor-nbd
|
# vitastor-nbd
|
||||||
@@ -98,10 +119,13 @@ endif (${WITH_QEMU})
|
|||||||
add_executable(test_cluster_client
|
add_executable(test_cluster_client
|
||||||
EXCLUDE_FROM_ALL
|
EXCLUDE_FROM_ALL
|
||||||
../test/test_cluster_client.cpp
|
../test/test_cluster_client.cpp
|
||||||
pg_states.cpp osd_ops.cpp cluster_client.cpp cluster_client_list.cpp cluster_client_wb.cpp msgr_op.cpp ../test/mock/messenger.cpp msgr_stop.cpp
|
cluster_client.cpp
|
||||||
etcd_state_client.cpp ../util/timerfd_manager.cpp ../util/addr_util.cpp ../util/str_util.cpp ../util/json_util.cpp ../../json11/json11.cpp
|
cluster_client_list.cpp
|
||||||
|
cluster_client_wb.cpp
|
||||||
|
../test/mock/messenger.cpp
|
||||||
|
etcd_state_client_mock.cpp
|
||||||
)
|
)
|
||||||
target_compile_definitions(test_cluster_client PUBLIC -D__MOCK__)
|
target_link_libraries(test_cluster_client vitastor_common ${LIBURING_LIBRARIES})
|
||||||
target_include_directories(test_cluster_client BEFORE PUBLIC ${CMAKE_SOURCE_DIR}/src/test/mock)
|
target_include_directories(test_cluster_client BEFORE PUBLIC ${CMAKE_SOURCE_DIR}/src/test/mock)
|
||||||
add_dependencies(build_tests test_cluster_client)
|
add_dependencies(build_tests test_cluster_client)
|
||||||
add_test(NAME test_cluster_client COMMAND test_cluster_client)
|
add_test(NAME test_cluster_client COMMAND test_cluster_client)
|
||||||
|
|||||||
@@ -11,7 +11,7 @@
|
|||||||
#define TRY_SEND_CONNECTING 1
|
#define TRY_SEND_CONNECTING 1
|
||||||
#define TRY_SEND_OK 2
|
#define TRY_SEND_OK 2
|
||||||
|
|
||||||
cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config)
|
cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config, std::unique_ptr<etcd_state_client_t> st_cli_ptr)
|
||||||
{
|
{
|
||||||
wb = new writeback_cache_t();
|
wb = new writeback_cache_t();
|
||||||
|
|
||||||
@@ -27,7 +27,7 @@ cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd
|
|||||||
msgr.ringloop = ringloop;
|
msgr.ringloop = ringloop;
|
||||||
msgr.repeer_pgs = [this](osd_num_t peer_osd)
|
msgr.repeer_pgs = [this](osd_num_t peer_osd)
|
||||||
{
|
{
|
||||||
if (msgr.osd_peer_fds.find(peer_osd) != msgr.osd_peer_fds.end())
|
if (msgr.osd_peers.find(peer_osd) != msgr.osd_peers.end())
|
||||||
{
|
{
|
||||||
// peer_osd just connected
|
// peer_osd just connected
|
||||||
continue_ops();
|
continue_ops();
|
||||||
@@ -47,29 +47,29 @@ cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd
|
|||||||
msgr.exec_op = [this](osd_op_t *op)
|
msgr.exec_op = [this](osd_op_t *op)
|
||||||
{
|
{
|
||||||
// Garbage in
|
// Garbage in
|
||||||
fprintf(stderr, "Incoming garbage from peer %d\n", op->peer_fd);
|
fprintf(stderr, "Can't handle incoming operation from client %lu\n", op->client_id);
|
||||||
msgr.stop_client(op->peer_fd);
|
msgr.stop_client(op->client_id);
|
||||||
delete op;
|
delete op;
|
||||||
};
|
};
|
||||||
msgr.parse_config(config);
|
msgr.parse_config(config);
|
||||||
|
|
||||||
st_cli.tfd = tfd;
|
st_cli = std::move(st_cli_ptr);
|
||||||
st_cli.on_load_config_hook = [this](json11::Json::object & cfg) { on_load_config_hook(cfg); };
|
st_cli->on_load_config_hook = [this](json11::Json::object & cfg) { on_load_config_hook(cfg); };
|
||||||
st_cli.on_change_osd_state_hook = [this](uint64_t peer_osd) { on_change_osd_state_hook(peer_osd); };
|
st_cli->on_change_osd_state_hook = [this](uint64_t peer_osd) { on_change_osd_state_hook(peer_osd); };
|
||||||
st_cli.on_change_pool_config_hook = [this]() { on_change_pool_config_hook(); };
|
st_cli->on_change_pool_config_hook = [this]() { on_change_pool_config_hook(); };
|
||||||
st_cli.on_change_pg_config_hook = [this]() { on_change_pool_config_hook(); };
|
st_cli->on_change_pg_config_hook = [this]() { on_change_pool_config_hook(); };
|
||||||
st_cli.on_change_pg_state_hook = [this](pool_id_t pool_id, pg_num_t pg_num, osd_num_t prev_primary) { on_change_pg_state_hook(pool_id, pg_num, prev_primary); };
|
st_cli->on_change_pg_state_hook = [this](pool_id_t pool_id, pg_num_t pg_num, osd_num_t prev_primary) { on_change_pg_state_hook(pool_id, pg_num, prev_primary); };
|
||||||
st_cli.on_change_node_placement_hook = [this]() { on_change_node_placement_hook(); };
|
st_cli->on_change_node_placement_hook = [this]() { on_change_node_placement_hook(); };
|
||||||
st_cli.on_load_pgs_hook = [this](bool success) { on_load_pgs_hook(success); };
|
st_cli->on_load_pgs_hook = [this](bool success) { on_load_pgs_hook(success); };
|
||||||
st_cli.on_reload_hook = [this]() { st_cli.load_global_config(); };
|
st_cli->on_reload_hook = [this]() { this->st_cli->load_global_config(); };
|
||||||
|
|
||||||
st_cli.parse_config(config);
|
st_cli->parse_config(config);
|
||||||
st_cli.infinite_start = false;
|
st_cli->infinite_start = false;
|
||||||
if (!config["client_infinite_start"].is_null())
|
if (!config["client_infinite_start"].is_null())
|
||||||
{
|
{
|
||||||
st_cli.infinite_start = config["client_infinite_start"].bool_value();
|
st_cli->infinite_start = config["client_infinite_start"].bool_value();
|
||||||
}
|
}
|
||||||
st_cli.load_global_config();
|
st_cli->load_global_config();
|
||||||
|
|
||||||
scrap_buffer_size = SCRAP_BUFFER_SIZE;
|
scrap_buffer_size = SCRAP_BUFFER_SIZE;
|
||||||
scrap_buffer = malloc_or_die(scrap_buffer_size);
|
scrap_buffer = malloc_or_die(scrap_buffer_size);
|
||||||
@@ -156,7 +156,7 @@ void cluster_client_t::continue_raw_ops(osd_num_t peer_osd)
|
|||||||
{
|
{
|
||||||
auto op = it->second;
|
auto op = it->second;
|
||||||
op->op_type = OSD_OP_OUT;
|
op->op_type = OSD_OP_OUT;
|
||||||
op->peer_fd = msgr.osd_peer_fds.at(peer_osd);
|
op->client_id = msgr.osd_peers.at(peer_osd)->client_id;
|
||||||
msgr.outbox_push(op);
|
msgr.outbox_push(op);
|
||||||
raw_ops.erase(it++);
|
raw_ops.erase(it++);
|
||||||
}
|
}
|
||||||
@@ -469,7 +469,7 @@ void cluster_client_t::on_load_config_hook(json11::Json::object & etcd_global_co
|
|||||||
auto etcd_report_interval = config["etcd_report_interval"].uint64_value();
|
auto etcd_report_interval = config["etcd_report_interval"].uint64_value();
|
||||||
if (!etcd_report_interval)
|
if (!etcd_report_interval)
|
||||||
etcd_report_interval = 5;
|
etcd_report_interval = 5;
|
||||||
client_wait_up_timeout = 1+etcd_report_interval+(st_cli.max_etcd_attempts*(2*st_cli.etcd_quick_timeout)+999)/1000;
|
client_wait_up_timeout = 1+etcd_report_interval+(st_cli->max_etcd_attempts*(2*st_cli->etcd_quick_timeout)+999)/1000;
|
||||||
}
|
}
|
||||||
// log_level
|
// log_level
|
||||||
log_level = config["log_level"].uint64_value();
|
log_level = config["log_level"].uint64_value();
|
||||||
@@ -482,8 +482,8 @@ void cluster_client_t::on_load_config_hook(json11::Json::object & etcd_global_co
|
|||||||
client_hostname = new_hostname;
|
client_hostname = new_hostname;
|
||||||
}
|
}
|
||||||
msgr.parse_config(config);
|
msgr.parse_config(config);
|
||||||
st_cli.parse_config(config);
|
st_cli->parse_config(config);
|
||||||
st_cli.load_pgs();
|
st_cli->load_pgs();
|
||||||
}
|
}
|
||||||
|
|
||||||
osd_num_t cluster_client_t::select_random_osd(const std::vector<osd_num_t> & osds)
|
osd_num_t cluster_client_t::select_random_osd(const std::vector<osd_num_t> & osds)
|
||||||
@@ -492,7 +492,7 @@ osd_num_t cluster_client_t::select_random_osd(const std::vector<osd_num_t> & osd
|
|||||||
int alive_count = 0;
|
int alive_count = 0;
|
||||||
for (auto & osd_num: osds)
|
for (auto & osd_num: osds)
|
||||||
{
|
{
|
||||||
if (!st_cli.peer_states[osd_num].is_null())
|
if (!st_cli->peer_states[osd_num].is_null())
|
||||||
alive_set[alive_count++] = osd_num;
|
alive_set[alive_count++] = osd_num;
|
||||||
}
|
}
|
||||||
if (!alive_count)
|
if (!alive_count)
|
||||||
@@ -509,7 +509,7 @@ osd_num_t cluster_client_t::select_nearest_osd(const std::vector<osd_num_t> & os
|
|||||||
while (self_tree_metrics.find(cur_id) == self_tree_metrics.end())
|
while (self_tree_metrics.find(cur_id) == self_tree_metrics.end())
|
||||||
{
|
{
|
||||||
self_tree_metrics[cur_id] = metric++;
|
self_tree_metrics[cur_id] = metric++;
|
||||||
json11::Json cur_placement = st_cli.node_placement[cur_id];
|
json11::Json cur_placement = st_cli->node_placement[cur_id];
|
||||||
cur_id = cur_placement["parent"].string_value();
|
cur_id = cur_placement["parent"].string_value();
|
||||||
}
|
}
|
||||||
if (cur_id != "")
|
if (cur_id != "")
|
||||||
@@ -529,7 +529,7 @@ osd_num_t cluster_client_t::select_nearest_osd(const std::vector<osd_num_t> & os
|
|||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
auto & peer_state = st_cli.peer_states[osd_num];
|
auto & peer_state = st_cli->peer_states[osd_num];
|
||||||
if (!peer_state.is_null())
|
if (!peer_state.is_null())
|
||||||
{
|
{
|
||||||
metric = self_tree_metrics[""];
|
metric = self_tree_metrics[""];
|
||||||
@@ -539,7 +539,7 @@ osd_num_t cluster_client_t::select_nearest_osd(const std::vector<osd_num_t> & os
|
|||||||
while (seen.find(cur_id) == seen.end())
|
while (seen.find(cur_id) == seen.end())
|
||||||
{
|
{
|
||||||
seen.insert(cur_id);
|
seen.insert(cur_id);
|
||||||
json11::Json cur_placement = st_cli.node_placement[cur_id];
|
json11::Json cur_placement = st_cli->node_placement[cur_id];
|
||||||
std::string cur_parent = cur_placement["parent"].string_value();
|
std::string cur_parent = cur_placement["parent"].string_value();
|
||||||
cur_id = (!first || cur_parent != "" ? cur_parent : peer_state["host"].string_value());
|
cur_id = (!first || cur_parent != "" ? cur_parent : peer_state["host"].string_value());
|
||||||
first = false;
|
first = false;
|
||||||
@@ -564,7 +564,7 @@ osd_num_t cluster_client_t::select_nearest_osd(const std::vector<osd_num_t> & os
|
|||||||
|
|
||||||
void cluster_client_t::on_load_pgs_hook(bool success)
|
void cluster_client_t::on_load_pgs_hook(bool success)
|
||||||
{
|
{
|
||||||
for (auto & pool_item: st_cli.pool_config)
|
for (auto & pool_item: st_cli->pool_config)
|
||||||
{
|
{
|
||||||
pg_counts[pool_item.first] = pool_item.second.real_pg_count;
|
pg_counts[pool_item.first] = pool_item.second.real_pg_count;
|
||||||
}
|
}
|
||||||
@@ -584,13 +584,13 @@ void cluster_client_t::on_load_pgs_hook(bool success)
|
|||||||
|
|
||||||
void cluster_client_t::on_change_pool_config_hook()
|
void cluster_client_t::on_change_pool_config_hook()
|
||||||
{
|
{
|
||||||
for (auto & pool_item: st_cli.pool_config)
|
for (auto & pool_item: st_cli->pool_config)
|
||||||
{
|
{
|
||||||
if (pg_counts[pool_item.first] != pool_item.second.real_pg_count)
|
if (pg_counts[pool_item.first] != pool_item.second.real_pg_count)
|
||||||
{
|
{
|
||||||
if (log_level > 2 && pg_counts[pool_item.first])
|
if (log_level > 2 && pg_counts[pool_item.first])
|
||||||
{
|
{
|
||||||
printf("Pool %u (%s) PG count changed from %lu to %lu\n", pool_item.first, pool_item.second.name.c_str(),
|
fprintf(stderr, "Pool %u (%s) PG count changed from %lu to %lu\n", pool_item.first, pool_item.second.name.c_str(),
|
||||||
pg_counts[pool_item.first], pool_item.second.real_pg_count);
|
pg_counts[pool_item.first], pool_item.second.real_pg_count);
|
||||||
}
|
}
|
||||||
// At this point, all pool operations should have been suspended
|
// At this point, all pool operations should have been suspended
|
||||||
@@ -612,7 +612,7 @@ void cluster_client_t::on_change_pool_config_hook()
|
|||||||
|
|
||||||
void cluster_client_t::on_change_pg_state_hook(pool_id_t pool_id, pg_num_t pg_num, osd_num_t prev_primary)
|
void cluster_client_t::on_change_pg_state_hook(pool_id_t pool_id, pg_num_t pg_num, osd_num_t prev_primary)
|
||||||
{
|
{
|
||||||
auto & pg_cfg = st_cli.pool_config[pool_id].pg_config[pg_num];
|
auto & pg_cfg = st_cli->pool_config[pool_id].pg_config[pg_num];
|
||||||
if (pg_cfg.cur_primary != prev_primary)
|
if (pg_cfg.cur_primary != prev_primary)
|
||||||
{
|
{
|
||||||
// Repeat this PG operations because an OSD which stopped being primary may not fsync operations
|
// Repeat this PG operations because an OSD which stopped being primary may not fsync operations
|
||||||
@@ -630,8 +630,8 @@ bool cluster_client_t::get_immediate_commit(uint64_t inode)
|
|||||||
pool_id_t pool_id = INODE_POOL(inode);
|
pool_id_t pool_id = INODE_POOL(inode);
|
||||||
if (!pool_id)
|
if (!pool_id)
|
||||||
return true;
|
return true;
|
||||||
auto pool_it = st_cli.pool_config.find(pool_id);
|
auto pool_it = st_cli->pool_config.find(pool_id);
|
||||||
if (pool_it == st_cli.pool_config.end())
|
if (pool_it == st_cli->pool_config.end())
|
||||||
return true;
|
return true;
|
||||||
return pool_it->second.immediate_commit == IMMEDIATE_ALL;
|
return pool_it->second.immediate_commit == IMMEDIATE_ALL;
|
||||||
}
|
}
|
||||||
@@ -641,7 +641,7 @@ void cluster_client_t::on_change_osd_state_hook(uint64_t peer_osd)
|
|||||||
osd_tree_metrics.erase(peer_osd);
|
osd_tree_metrics.erase(peer_osd);
|
||||||
if (msgr.wanted_peers.find(peer_osd) != msgr.wanted_peers.end())
|
if (msgr.wanted_peers.find(peer_osd) != msgr.wanted_peers.end())
|
||||||
{
|
{
|
||||||
msgr.connect_peer(peer_osd, st_cli.peer_states[peer_osd]);
|
msgr.connect_peer(peer_osd, st_cli->peer_states[peer_osd]);
|
||||||
continue_lists();
|
continue_lists();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -871,13 +871,13 @@ void cluster_client_t::execute_cas(cluster_op_t *op)
|
|||||||
if (op->retval != expected && op->retval >= 0)
|
if (op->retval != expected && op->retval >= 0)
|
||||||
op->retval = -EIO;
|
op->retval = -EIO;
|
||||||
op->retval = op->retval == -EPIPE ? -EINTR : op->retval;
|
op->retval = op->retval == -EPIPE ? -EINTR : op->retval;
|
||||||
auto peer_it = msgr.osd_peer_fds.find(op->parts[0].osd_num);
|
auto peer_it = msgr.osd_peers.find(op->parts[0].osd_num);
|
||||||
if (op->retval != 0 || (op->flags & OP_IMMEDIATE_COMMIT))
|
if (op->retval != 0 || (op->flags & OP_IMMEDIATE_COMMIT))
|
||||||
{
|
{
|
||||||
auto cb = std::move(op->callback);
|
auto cb = std::move(op->callback);
|
||||||
cb(op);
|
cb(op);
|
||||||
}
|
}
|
||||||
else if (peer_it == msgr.osd_peer_fds.end())
|
else if (peer_it == msgr.osd_peers.end())
|
||||||
{
|
{
|
||||||
// Care must be taken to make sure that the client doesn't reconnect to the OSD
|
// Care must be taken to make sure that the client doesn't reconnect to the OSD
|
||||||
// before executing the previously completed operation callback (!)
|
// before executing the previously completed operation callback (!)
|
||||||
@@ -888,10 +888,10 @@ void cluster_client_t::execute_cas(cluster_op_t *op)
|
|||||||
else
|
else
|
||||||
{
|
{
|
||||||
// CAS writes have a built-in sync
|
// CAS writes have a built-in sync
|
||||||
auto peer_fd = peer_it->second;
|
osd_client_t *cl = peer_it->second;
|
||||||
*part = (osd_op_t){
|
*part = (osd_op_t){
|
||||||
.op_type = OSD_OP_OUT,
|
.op_type = OSD_OP_OUT,
|
||||||
.peer_fd = peer_fd,
|
.client_id = cl->client_id,
|
||||||
.req = {
|
.req = {
|
||||||
.hdr = {
|
.hdr = {
|
||||||
.magic = SECONDARY_OSD_OP_MAGIC,
|
.magic = SECONDARY_OSD_OP_MAGIC,
|
||||||
@@ -936,8 +936,8 @@ bool cluster_client_t::check_rw(cluster_op_t *op)
|
|||||||
cb(op);
|
cb(op);
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
auto pool_it = st_cli.pool_config.find(pool_id);
|
auto pool_it = st_cli->pool_config.find(pool_id);
|
||||||
if (pool_it == st_cli.pool_config.end() || pool_it->second.real_pg_count == 0)
|
if (pool_it == st_cli->pool_config.end() || pool_it->second.real_pg_count == 0)
|
||||||
{
|
{
|
||||||
// Pools are loaded, but this one is unknown
|
// Pools are loaded, but this one is unknown
|
||||||
op->retval = -EINVAL;
|
op->retval = -EINVAL;
|
||||||
@@ -960,8 +960,8 @@ bool cluster_client_t::check_rw(cluster_op_t *op)
|
|||||||
}
|
}
|
||||||
if ((op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) && !(op->flags & OSD_OP_IGNORE_READONLY))
|
if ((op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) && !(op->flags & OSD_OP_IGNORE_READONLY))
|
||||||
{
|
{
|
||||||
auto ino_it = st_cli.inode_config.find(op->inode);
|
auto ino_it = st_cli->inode_config.find(op->inode);
|
||||||
if (ino_it != st_cli.inode_config.end() && ino_it->second.readonly)
|
if (ino_it != st_cli->inode_config.end() && ino_it->second.readonly)
|
||||||
{
|
{
|
||||||
op->retval = -EROFS;
|
op->retval = -EROFS;
|
||||||
auto cb = std::move(op->callback);
|
auto cb = std::move(op->callback);
|
||||||
@@ -972,15 +972,15 @@ bool cluster_client_t::check_rw(cluster_op_t *op)
|
|||||||
op->deoptimise_snapshot = false;
|
op->deoptimise_snapshot = false;
|
||||||
if (enable_writeback && (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP))
|
if (enable_writeback && (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP))
|
||||||
{
|
{
|
||||||
auto ino_it = st_cli.inode_config.find(op->inode);
|
auto ino_it = st_cli->inode_config.find(op->inode);
|
||||||
if (ino_it != st_cli.inode_config.end())
|
if (ino_it != st_cli->inode_config.end())
|
||||||
{
|
{
|
||||||
int chain_size = 0;
|
int chain_size = 0;
|
||||||
while (ino_it != st_cli.inode_config.end() && ino_it->second.parent_id)
|
while (ino_it != st_cli->inode_config.end() && ino_it->second.parent_id)
|
||||||
{
|
{
|
||||||
// Check for loops - FIXME check it in etcd_state_client
|
// Check for loops - FIXME check it in etcd_state_client
|
||||||
if (ino_it->second.parent_id == op->inode ||
|
if (ino_it->second.parent_id == op->inode ||
|
||||||
chain_size > st_cli.inode_config.size())
|
chain_size > st_cli->inode_config.size())
|
||||||
{
|
{
|
||||||
op->retval = -EINVAL;
|
op->retval = -EINVAL;
|
||||||
auto cb = std::move(op->callback);
|
auto cb = std::move(op->callback);
|
||||||
@@ -995,7 +995,7 @@ bool cluster_client_t::check_rw(cluster_op_t *op)
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
chain_size++;
|
chain_size++;
|
||||||
ino_it = st_cli.inode_config.find(ino_it->second.parent_id);
|
ino_it = st_cli->inode_config.find(ino_it->second.parent_id);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1004,17 +1004,17 @@ bool cluster_client_t::check_rw(cluster_op_t *op)
|
|||||||
|
|
||||||
void cluster_client_t::execute_raw(osd_num_t osd_num, osd_op_t *op)
|
void cluster_client_t::execute_raw(osd_num_t osd_num, osd_op_t *op)
|
||||||
{
|
{
|
||||||
auto fd_it = msgr.osd_peer_fds.find(osd_num);
|
auto peer_it = msgr.osd_peers.find(osd_num);
|
||||||
if (fd_it != msgr.osd_peer_fds.end())
|
if (peer_it != msgr.osd_peers.end())
|
||||||
{
|
{
|
||||||
op->op_type = OSD_OP_OUT;
|
op->op_type = OSD_OP_OUT;
|
||||||
op->peer_fd = fd_it->second;
|
op->client_id = peer_it->second->client_id;
|
||||||
msgr.outbox_push(op);
|
msgr.outbox_push(op);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
if (msgr.wanted_peers.find(osd_num) == msgr.wanted_peers.end())
|
if (msgr.wanted_peers.find(osd_num) == msgr.wanted_peers.end())
|
||||||
msgr.connect_peer(osd_num, st_cli.peer_states[osd_num]);
|
msgr.connect_peer(osd_num, st_cli->peer_states[osd_num]);
|
||||||
raw_ops.emplace(osd_num, op);
|
raw_ops.emplace(osd_num, op);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1119,28 +1119,35 @@ resume_2:
|
|||||||
// Finished successfully
|
// Finished successfully
|
||||||
// Even if the PG count has changed in meanwhile we treat it as success
|
// Even if the PG count has changed in meanwhile we treat it as success
|
||||||
// because if some operations were invalid for the new PG count we'd get errors
|
// because if some operations were invalid for the new PG count we'd get errors
|
||||||
|
if (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP)
|
||||||
|
{
|
||||||
|
// Copy part bitmaps only after finishing all part reads
|
||||||
|
for (auto & part: op->parts)
|
||||||
|
if ((part.flags & (PART_SENT|PART_DONE|PART_VALID)) == (PART_SENT|PART_DONE|PART_VALID))
|
||||||
|
copy_part_bitmap(op, &part);
|
||||||
|
}
|
||||||
if (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_CHAIN_BITMAP)
|
if (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_CHAIN_BITMAP)
|
||||||
{
|
{
|
||||||
// Check parent inode
|
// Check parent inode
|
||||||
auto ino_it = st_cli.inode_config.find(op->cur_inode);
|
auto ino_it = st_cli->inode_config.find(op->cur_inode);
|
||||||
// Skip parents from the same pool
|
// Skip parents from the same pool
|
||||||
int skipped = 0;
|
int skipped = 0;
|
||||||
while (!op->deoptimise_snapshot &&
|
while (!op->deoptimise_snapshot &&
|
||||||
ino_it != st_cli.inode_config.end() && ino_it->second.parent_id &&
|
ino_it != st_cli->inode_config.end() && ino_it->second.parent_id &&
|
||||||
INODE_POOL(ino_it->second.parent_id) == INODE_POOL(op->cur_inode))
|
INODE_POOL(ino_it->second.parent_id) == INODE_POOL(op->cur_inode))
|
||||||
{
|
{
|
||||||
// Check for loops - FIXME check it in etcd_state_client
|
// Check for loops - FIXME check it in etcd_state_client
|
||||||
if (ino_it->second.parent_id == op->inode ||
|
if (ino_it->second.parent_id == op->inode ||
|
||||||
skipped > st_cli.inode_config.size())
|
skipped > st_cli->inode_config.size())
|
||||||
{
|
{
|
||||||
op->retval = -EINVAL;
|
op->retval = -EINVAL;
|
||||||
erase_op(op);
|
erase_op(op);
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
skipped++;
|
skipped++;
|
||||||
ino_it = st_cli.inode_config.find(ino_it->second.parent_id);
|
ino_it = st_cli->inode_config.find(ino_it->second.parent_id);
|
||||||
}
|
}
|
||||||
if (ino_it != st_cli.inode_config.end() &&
|
if (ino_it != st_cli->inode_config.end() &&
|
||||||
ino_it->second.parent_id &&
|
ino_it->second.parent_id &&
|
||||||
ino_it->second.parent_id != op->inode)
|
ino_it->second.parent_id != op->inode)
|
||||||
{
|
{
|
||||||
@@ -1154,7 +1161,7 @@ resume_2:
|
|||||||
op->retval = op->len;
|
op->retval = op->len;
|
||||||
if (op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP)
|
if (op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP)
|
||||||
{
|
{
|
||||||
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->inode));
|
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(op->inode));
|
||||||
op->retval = op->len / pool_cfg.bitmap_granularity;
|
op->retval = op->len / pool_cfg.bitmap_granularity;
|
||||||
}
|
}
|
||||||
if (op->flush_id)
|
if (op->flush_id)
|
||||||
@@ -1164,7 +1171,7 @@ resume_2:
|
|||||||
erase_op(op);
|
erase_op(op);
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
else if (op->retval != 0 && !(op->flags & OP_FLUSH_BUFFER) &&
|
else if (op->retval != 0 && op->opcode != OSD_OP_SYNC && !(op->flags & OP_FLUSH_BUFFER) &&
|
||||||
op->retval != -EPIPE && (op->retval != -EIO || !client_eio_retry_interval) && (op->retval != -ENOSPC || !client_retry_enospc))
|
op->retval != -EPIPE && (op->retval != -EIO || !client_eio_retry_interval) && (op->retval != -ENOSPC || !client_retry_enospc))
|
||||||
{
|
{
|
||||||
// Fatal error (neither -EPIPE, -EIO nor -ENOSPC)
|
// Fatal error (neither -EPIPE, -EIO nor -ENOSPC)
|
||||||
@@ -1240,7 +1247,7 @@ void cluster_client_t::slice_rw(cluster_op_t *op)
|
|||||||
{
|
{
|
||||||
// Slice the request into individual object stripe requests
|
// Slice the request into individual object stripe requests
|
||||||
// Primary OSDs still operate individual stripes, but their size is multiplied by PG minsize in case of EC
|
// Primary OSDs still operate individual stripes, but their size is multiplied by PG minsize in case of EC
|
||||||
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->cur_inode));
|
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(op->cur_inode));
|
||||||
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
||||||
uint64_t pg_block_size = pool_cfg.data_block_size * pg_data_size;
|
uint64_t pg_block_size = pool_cfg.data_block_size * pg_data_size;
|
||||||
uint64_t first_stripe = (op->offset / pg_block_size) * pg_block_size;
|
uint64_t first_stripe = (op->offset / pg_block_size) * pg_block_size;
|
||||||
@@ -1339,7 +1346,7 @@ bool cluster_client_t::affects_pg(uint64_t inode, uint64_t offset, uint64_t len,
|
|||||||
{
|
{
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(inode));
|
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(inode));
|
||||||
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
||||||
uint64_t pg_block_size = pool_cfg.data_block_size * pg_data_size;
|
uint64_t pg_block_size = pool_cfg.data_block_size * pg_data_size;
|
||||||
uint64_t first_stripe = (offset / pg_block_size) * pg_block_size;
|
uint64_t first_stripe = (offset / pg_block_size) * pg_block_size;
|
||||||
@@ -1358,7 +1365,7 @@ bool cluster_client_t::affects_pg(uint64_t inode, uint64_t offset, uint64_t len,
|
|||||||
|
|
||||||
bool cluster_client_t::affects_osd(uint64_t inode, uint64_t offset, uint64_t len, osd_num_t osd)
|
bool cluster_client_t::affects_osd(uint64_t inode, uint64_t offset, uint64_t len, osd_num_t osd)
|
||||||
{
|
{
|
||||||
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(inode));
|
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(inode));
|
||||||
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
||||||
uint64_t pg_block_size = pool_cfg.data_block_size * pg_data_size;
|
uint64_t pg_block_size = pool_cfg.data_block_size * pg_data_size;
|
||||||
uint64_t first_stripe = (offset / pg_block_size) * pg_block_size;
|
uint64_t first_stripe = (offset / pg_block_size) * pg_block_size;
|
||||||
@@ -1382,7 +1389,7 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
|
|||||||
init_msgr();
|
init_msgr();
|
||||||
}
|
}
|
||||||
auto part = &op->parts[i];
|
auto part = &op->parts[i];
|
||||||
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->cur_inode));
|
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(op->cur_inode));
|
||||||
auto pg_it = pool_cfg.pg_config.find(part->pg_num);
|
auto pg_it = pool_cfg.pg_config.find(part->pg_num);
|
||||||
if (pg_it != pool_cfg.pg_config.end() &&
|
if (pg_it != pool_cfg.pg_config.end() &&
|
||||||
!pg_it->second.pause && pg_it->second.cur_primary &&
|
!pg_it->second.pause && pg_it->second.cur_primary &&
|
||||||
@@ -1401,10 +1408,10 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
|
|||||||
primary_osd = nearest_osd;
|
primary_osd = nearest_osd;
|
||||||
}
|
}
|
||||||
part->osd_num = primary_osd;
|
part->osd_num = primary_osd;
|
||||||
auto peer_it = msgr.osd_peer_fds.find(primary_osd);
|
auto peer_it = msgr.osd_peers.find(primary_osd);
|
||||||
if (peer_it != msgr.osd_peer_fds.end())
|
if (peer_it != msgr.osd_peers.end())
|
||||||
{
|
{
|
||||||
int peer_fd = peer_it->second;
|
osd_client_t *cl = peer_it->second;
|
||||||
part->flags |= PART_SENT|PART_VALID;
|
part->flags |= PART_SENT|PART_VALID;
|
||||||
op->inflight_count++;
|
op->inflight_count++;
|
||||||
uint64_t pg_bitmap_size = (pool_cfg.data_block_size / pool_cfg.bitmap_granularity / 8) * (
|
uint64_t pg_bitmap_size = (pool_cfg.data_block_size / pool_cfg.bitmap_granularity / 8) * (
|
||||||
@@ -1413,13 +1420,13 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
|
|||||||
uint64_t meta_rev = 0;
|
uint64_t meta_rev = 0;
|
||||||
if (op->opcode != OSD_OP_READ_BITMAP && op->opcode != OSD_OP_DELETE && !op->deoptimise_snapshot)
|
if (op->opcode != OSD_OP_READ_BITMAP && op->opcode != OSD_OP_DELETE && !op->deoptimise_snapshot)
|
||||||
{
|
{
|
||||||
auto ino_it = st_cli.inode_config.find(op->cur_inode);
|
auto ino_it = st_cli->inode_config.find(op->cur_inode);
|
||||||
if (ino_it != st_cli.inode_config.end())
|
if (ino_it != st_cli->inode_config.end())
|
||||||
meta_rev = ino_it->second.mod_revision;
|
meta_rev = ino_it->second.mod_revision;
|
||||||
}
|
}
|
||||||
part->op = (osd_op_t){
|
part->op = (osd_op_t){
|
||||||
.op_type = OSD_OP_OUT,
|
.op_type = OSD_OP_OUT,
|
||||||
.peer_fd = peer_fd,
|
.client_id = cl->client_id,
|
||||||
.req = { .rw = {
|
.req = { .rw = {
|
||||||
.header = {
|
.header = {
|
||||||
.magic = SECONDARY_OSD_OP_MAGIC,
|
.magic = SECONDARY_OSD_OP_MAGIC,
|
||||||
@@ -1446,7 +1453,7 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
|
|||||||
}
|
}
|
||||||
else if (msgr.wanted_peers.find(primary_osd) == msgr.wanted_peers.end())
|
else if (msgr.wanted_peers.find(primary_osd) == msgr.wanted_peers.end())
|
||||||
{
|
{
|
||||||
msgr.connect_peer(primary_osd, st_cli.peer_states[primary_osd]);
|
msgr.connect_peer(primary_osd, st_cli->peer_states[primary_osd]);
|
||||||
return TRY_SEND_CONNECTING;
|
return TRY_SEND_CONNECTING;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1468,8 +1475,8 @@ int cluster_client_t::continue_sync(cluster_op_t *op)
|
|||||||
for (auto do_it = dirty_osds.begin(); do_it != dirty_osds.end(); )
|
for (auto do_it = dirty_osds.begin(); do_it != dirty_osds.end(); )
|
||||||
{
|
{
|
||||||
osd_num_t sync_osd = *do_it;
|
osd_num_t sync_osd = *do_it;
|
||||||
auto peer_it = msgr.osd_peer_fds.find(sync_osd);
|
auto peer_it = msgr.osd_peers.find(sync_osd);
|
||||||
if (peer_it == msgr.osd_peer_fds.end())
|
if (peer_it == msgr.osd_peers.end())
|
||||||
dirty_osds.erase(do_it++);
|
dirty_osds.erase(do_it++);
|
||||||
else
|
else
|
||||||
do_it++;
|
do_it++;
|
||||||
@@ -1522,12 +1529,12 @@ resume_1:
|
|||||||
|
|
||||||
void cluster_client_t::send_sync(cluster_op_t *op, cluster_op_part_t *part)
|
void cluster_client_t::send_sync(cluster_op_t *op, cluster_op_part_t *part)
|
||||||
{
|
{
|
||||||
auto peer_fd = msgr.osd_peer_fds.at(part->osd_num);
|
osd_client_t *cl = msgr.osd_peers.at(part->osd_num);
|
||||||
part->flags |= PART_SENT;
|
part->flags |= PART_SENT;
|
||||||
op->inflight_count++;
|
op->inflight_count++;
|
||||||
part->op = (osd_op_t){
|
part->op = (osd_op_t){
|
||||||
.op_type = OSD_OP_OUT,
|
.op_type = OSD_OP_OUT,
|
||||||
.peer_fd = peer_fd,
|
.client_id = cl->client_id,
|
||||||
.req = {
|
.req = {
|
||||||
.hdr = {
|
.hdr = {
|
||||||
.magic = SECONDARY_OSD_OP_MAGIC,
|
.magic = SECONDARY_OSD_OP_MAGIC,
|
||||||
@@ -1567,10 +1574,10 @@ void cluster_client_t::handle_op_part(cluster_op_part_t *part)
|
|||||||
// Error priority: EIO > ENOSPC > ETIMEDOUT > EPIPE
|
// Error priority: EIO > ENOSPC > ETIMEDOUT > EPIPE
|
||||||
op->retval = part->op.reply.hdr.retval;
|
op->retval = part->op.reply.hdr.retval;
|
||||||
}
|
}
|
||||||
int stop_fd = -1;
|
uint64_t stop_client_id = 0;
|
||||||
if (op->retval != -EINTR && op->retval != -EIO && op->retval != -ENOSPC)
|
if (op->retval != -EINTR && op->retval != -EIO && op->retval != -ENOSPC)
|
||||||
{
|
{
|
||||||
stop_fd = part->op.peer_fd;
|
stop_client_id = part->op.client_id;
|
||||||
if (op->retval != -EPIPE || log_level > 0)
|
if (op->retval != -EPIPE || log_level > 0)
|
||||||
{
|
{
|
||||||
fprintf(
|
fprintf(
|
||||||
@@ -1597,9 +1604,9 @@ void cluster_client_t::handle_op_part(cluster_op_part_t *part)
|
|||||||
op->retry_after = op->retval != -EPIPE ? client_eio_retry_interval : client_retry_interval;
|
op->retry_after = op->retval != -EPIPE ? client_eio_retry_interval : client_retry_interval;
|
||||||
}
|
}
|
||||||
reset_retry_timer(op->retry_after);
|
reset_retry_timer(op->retry_after);
|
||||||
if (stop_fd >= 0)
|
if (stop_client_id)
|
||||||
{
|
{
|
||||||
msgr.stop_client(stop_fd);
|
msgr.stop_client(stop_client_id);
|
||||||
}
|
}
|
||||||
op->inflight_count--;
|
op->inflight_count--;
|
||||||
if (op->inflight_count == 0 && !op->retry_after)
|
if (op->inflight_count == 0 && !op->retry_after)
|
||||||
@@ -1630,13 +1637,6 @@ void cluster_client_t::handle_op_part(cluster_op_part_t *part)
|
|||||||
}
|
}
|
||||||
if (op->inflight_count == 0 && !op->retry_after)
|
if (op->inflight_count == 0 && !op->retry_after)
|
||||||
{
|
{
|
||||||
// Copy part bitmaps only after finishing all part reads
|
|
||||||
if (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP)
|
|
||||||
{
|
|
||||||
for (auto & part: op->parts)
|
|
||||||
if (part.flags == (PART_SENT|PART_VALID|PART_DONE))
|
|
||||||
copy_part_bitmap(op, &part);
|
|
||||||
}
|
|
||||||
if (op->opcode == OSD_OP_SYNC)
|
if (op->opcode == OSD_OP_SYNC)
|
||||||
continue_sync(op);
|
continue_sync(op);
|
||||||
else
|
else
|
||||||
@@ -1648,7 +1648,7 @@ void cluster_client_t::handle_op_part(cluster_op_part_t *part)
|
|||||||
void cluster_client_t::copy_part_bitmap(cluster_op_t *op, cluster_op_part_t *part)
|
void cluster_client_t::copy_part_bitmap(cluster_op_t *op, cluster_op_part_t *part)
|
||||||
{
|
{
|
||||||
// Copy (OR) bitmap
|
// Copy (OR) bitmap
|
||||||
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->cur_inode));
|
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(op->cur_inode));
|
||||||
uint32_t pg_block_size = pool_cfg.data_block_size * (
|
uint32_t pg_block_size = pool_cfg.data_block_size * (
|
||||||
pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks
|
pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks
|
||||||
);
|
);
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
#pragma once
|
#pragma once
|
||||||
|
|
||||||
#include "messenger.h"
|
#include "messenger.h"
|
||||||
#include "etcd_state_client.h"
|
#include "etcd_state_client_http.h"
|
||||||
|
|
||||||
#define DEFAULT_CLIENT_MAX_DIRTY_BYTES 32*1024*1024
|
#define DEFAULT_CLIENT_MAX_DIRTY_BYTES 32*1024*1024
|
||||||
#define DEFAULT_CLIENT_MAX_DIRTY_OPS 1024
|
#define DEFAULT_CLIENT_MAX_DIRTY_OPS 1024
|
||||||
@@ -83,9 +83,6 @@ class writeback_cache_t;
|
|||||||
// FIXME: Split into public and private interfaces
|
// FIXME: Split into public and private interfaces
|
||||||
class __attribute__((visibility("default"))) cluster_client_t
|
class __attribute__((visibility("default"))) cluster_client_t
|
||||||
{
|
{
|
||||||
#ifdef __MOCK__
|
|
||||||
public:
|
|
||||||
#endif
|
|
||||||
timerfd_manager_t *tfd = NULL;
|
timerfd_manager_t *tfd = NULL;
|
||||||
ring_loop_t *ringloop = NULL;
|
ring_loop_t *ringloop = NULL;
|
||||||
|
|
||||||
@@ -134,7 +131,7 @@ public:
|
|||||||
bool msgr_initialized = false;
|
bool msgr_initialized = false;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
etcd_state_client_t st_cli;
|
std::unique_ptr<etcd_state_client_t> st_cli;
|
||||||
|
|
||||||
osd_messenger_t msgr;
|
osd_messenger_t msgr;
|
||||||
void init_msgr();
|
void init_msgr();
|
||||||
@@ -142,7 +139,8 @@ public:
|
|||||||
json11::Json::object cli_config, file_config, etcd_global_config;
|
json11::Json::object cli_config, file_config, etcd_global_config;
|
||||||
json11::Json::object config;
|
json11::Json::object config;
|
||||||
|
|
||||||
cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config);
|
static cluster_client_t* create(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config);
|
||||||
|
cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config, std::unique_ptr<etcd_state_client_t> st_cli);
|
||||||
~cluster_client_t();
|
~cluster_client_t();
|
||||||
void execute(cluster_op_t *op);
|
void execute(cluster_op_t *op);
|
||||||
void execute_raw(osd_num_t osd_num, osd_op_t *op);
|
void execute_raw(osd_num_t osd_num, osd_op_t *op);
|
||||||
@@ -155,15 +153,9 @@ public:
|
|||||||
void list_inode(inode_t inode, uint64_t min_offset, uint64_t max_offset, int max_parallel_pgs, std::function<void(
|
void list_inode(inode_t inode, uint64_t min_offset, uint64_t max_offset, int max_parallel_pgs, std::function<void(
|
||||||
int status, int pgs_left, pg_num_t pg_num, std::set<object_id>&& objects)> pg_callback);
|
int status, int pgs_left, pg_num_t pg_num, std::set<object_id>&& objects)> pg_callback);
|
||||||
|
|
||||||
//inline uint32_t get_bs_bitmap_granularity() { return st_cli.global_bitmap_granularity; }
|
|
||||||
//inline uint64_t get_bs_block_size() { return st_cli.global_block_size; }
|
|
||||||
|
|
||||||
#ifndef __MOCK__
|
|
||||||
protected:
|
protected:
|
||||||
#endif
|
|
||||||
void continue_ops(int time_passed = 0);
|
void continue_ops(int time_passed = 0);
|
||||||
|
|
||||||
protected:
|
|
||||||
bool affects_osd(uint64_t inode, uint64_t offset, uint64_t len, osd_num_t osd);
|
bool affects_osd(uint64_t inode, uint64_t offset, uint64_t len, osd_num_t osd);
|
||||||
bool affects_pg(uint64_t inode, uint64_t offset, uint64_t len, pool_id_t pool_id, pg_num_t pg_num);
|
bool affects_pg(uint64_t inode, uint64_t offset, uint64_t len, pool_id_t pool_id, pg_num_t pg_num);
|
||||||
|
|
||||||
@@ -204,4 +196,5 @@ protected:
|
|||||||
osd_num_t select_nearest_osd(const std::vector<osd_num_t> & osds);
|
osd_num_t select_nearest_osd(const std::vector<osd_num_t> & osds);
|
||||||
|
|
||||||
friend class writeback_cache_t;
|
friend class writeback_cache_t;
|
||||||
|
friend class cluster_client_test_t;
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -63,14 +63,14 @@ void cluster_client_t::list_inode(inode_t inode, uint64_t min_offset, uint64_t m
|
|||||||
{
|
{
|
||||||
init_msgr();
|
init_msgr();
|
||||||
pool_id_t pool_id = INODE_POOL(inode);
|
pool_id_t pool_id = INODE_POOL(inode);
|
||||||
if (!pool_id || st_cli.pool_config.find(pool_id) == st_cli.pool_config.end())
|
if (!pool_id || st_cli->pool_config.find(pool_id) == st_cli->pool_config.end())
|
||||||
{
|
{
|
||||||
if (log_level > 0)
|
if (log_level > 0)
|
||||||
fprintf(stderr, "Pool %u does not exist\n", pool_id);
|
fprintf(stderr, "Pool %u does not exist\n", pool_id);
|
||||||
pg_callback(-EINVAL, 0, 0, std::set<object_id>());
|
pg_callback(-EINVAL, 0, 0, std::set<object_id>());
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
auto pg_stripe_size = st_cli.pool_config.at(pool_id).pg_stripe_size;
|
auto pg_stripe_size = st_cli->pool_config.at(pool_id).pg_stripe_size;
|
||||||
if (min_offset)
|
if (min_offset)
|
||||||
min_offset = (min_offset/pg_stripe_size) * pg_stripe_size;
|
min_offset = (min_offset/pg_stripe_size) * pg_stripe_size;
|
||||||
inode_list_t *lst = new inode_list_t();
|
inode_list_t *lst = new inode_list_t();
|
||||||
@@ -110,13 +110,13 @@ bool cluster_client_t::continue_listing(inode_list_t *lst)
|
|||||||
|
|
||||||
bool cluster_client_t::restart_listing(inode_list_t* lst)
|
bool cluster_client_t::restart_listing(inode_list_t* lst)
|
||||||
{
|
{
|
||||||
auto pool_it = st_cli.pool_config.find(lst->pool_id);
|
auto pool_it = st_cli->pool_config.find(lst->pool_id);
|
||||||
// We want listing to be consistent. To achieve it we should:
|
// We want listing to be consistent. To achieve it we should:
|
||||||
// 1) retry listing of each PG if its state changes
|
// 1) retry listing of each PG if its state changes
|
||||||
// 2) abort listing if PG count changes during listing
|
// 2) abort listing if PG count changes during listing
|
||||||
// 3) ideally, only talk to the primary OSD - this will be done separately
|
// 3) ideally, only talk to the primary OSD - this will be done separately
|
||||||
// So first we add all PGs without checking their state
|
// So first we add all PGs without checking their state
|
||||||
if (pool_it == st_cli.pool_config.end() ||
|
if (pool_it == st_cli->pool_config.end() ||
|
||||||
lst->real_pg_count != pool_it->second.real_pg_count)
|
lst->real_pg_count != pool_it->second.real_pg_count)
|
||||||
{
|
{
|
||||||
for (auto pg: lst->pgs)
|
for (auto pg: lst->pgs)
|
||||||
@@ -136,7 +136,7 @@ bool cluster_client_t::restart_listing(inode_list_t* lst)
|
|||||||
fprintf(stderr, "PG count in pool %u changed during listing\n", lst->pool_id);
|
fprintf(stderr, "PG count in pool %u changed during listing\n", lst->pool_id);
|
||||||
}
|
}
|
||||||
lst->pgs.clear();
|
lst->pgs.clear();
|
||||||
if (pool_it == st_cli.pool_config.end())
|
if (pool_it == st_cli->pool_config.end())
|
||||||
{
|
{
|
||||||
// Unknown pool
|
// Unknown pool
|
||||||
lst->callback(-EINVAL, 0, 0, std::set<object_id>());
|
lst->callback(-EINVAL, 0, 0, std::set<object_id>());
|
||||||
@@ -248,7 +248,7 @@ void cluster_client_t::set_list_retry_timeout(int ms, timespec new_time)
|
|||||||
|
|
||||||
int cluster_client_t::start_pg_listing(inode_list_pg_t *pg)
|
int cluster_client_t::start_pg_listing(inode_list_pg_t *pg)
|
||||||
{
|
{
|
||||||
auto & pool_cfg = st_cli.pool_config.at(pg->lst->pool_id);
|
auto & pool_cfg = st_cli->pool_config.at(pg->lst->pool_id);
|
||||||
auto pg_it = pool_cfg.pg_config.find(pg->pg_num);
|
auto pg_it = pool_cfg.pg_config.find(pg->pg_num);
|
||||||
assert(pg->lst->real_pg_count == pool_cfg.real_pg_count);
|
assert(pg->lst->real_pg_count == pool_cfg.real_pg_count);
|
||||||
if (pg_it == pool_cfg.pg_config.end() ||
|
if (pg_it == pool_cfg.pg_config.end() ||
|
||||||
@@ -277,7 +277,7 @@ int cluster_client_t::start_pg_listing(inode_list_pg_t *pg)
|
|||||||
for (auto peer_it = all_peers.begin(); peer_it != all_peers.end(); )
|
for (auto peer_it = all_peers.begin(); peer_it != all_peers.end(); )
|
||||||
{
|
{
|
||||||
if (*peer_it != pg_it->second.cur_primary &&
|
if (*peer_it != pg_it->second.cur_primary &&
|
||||||
st_cli.peer_states[*peer_it].is_null())
|
st_cli->peer_states[*peer_it].is_null())
|
||||||
{
|
{
|
||||||
pg->inactive_osds.push_back(*peer_it);
|
pg->inactive_osds.push_back(*peer_it);
|
||||||
all_peers.erase(peer_it++);
|
all_peers.erase(peer_it++);
|
||||||
@@ -295,14 +295,14 @@ int cluster_client_t::start_pg_listing(inode_list_pg_t *pg)
|
|||||||
bool conn = true;
|
bool conn = true;
|
||||||
for (osd_num_t peer_osd: all_peers)
|
for (osd_num_t peer_osd: all_peers)
|
||||||
{
|
{
|
||||||
if (msgr.osd_peer_fds.find(peer_osd) == msgr.osd_peer_fds.end())
|
if (msgr.osd_peers.find(peer_osd) == msgr.osd_peers.end())
|
||||||
{
|
{
|
||||||
// Initiate connection
|
// Initiate connection
|
||||||
if (st_cli.peer_states[peer_osd].is_null())
|
if (st_cli->peer_states[peer_osd].is_null())
|
||||||
{
|
{
|
||||||
return LIST_PG_WAIT_ACTIVE;
|
return LIST_PG_WAIT_ACTIVE;
|
||||||
}
|
}
|
||||||
msgr.connect_peer(peer_osd, st_cli.peer_states[peer_osd]);
|
msgr.connect_peer(peer_osd, st_cli->peer_states[peer_osd]);
|
||||||
conn = false;
|
conn = false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -336,11 +336,11 @@ void cluster_client_t::send_list(inode_list_osd_t *cur_list)
|
|||||||
if (!cur_list->pg->inflight_ops)
|
if (!cur_list->pg->inflight_ops)
|
||||||
cur_list->pg->lst->inflight_pgs++;
|
cur_list->pg->lst->inflight_pgs++;
|
||||||
cur_list->pg->inflight_ops++;
|
cur_list->pg->inflight_ops++;
|
||||||
auto & pool_cfg = st_cli.pool_config[cur_list->pg->lst->pool_id];
|
auto & pool_cfg = st_cli->pool_config[cur_list->pg->lst->pool_id];
|
||||||
osd_op_t *op = new osd_op_t();
|
osd_op_t *op = new osd_op_t();
|
||||||
op->op_type = OSD_OP_OUT;
|
op->op_type = OSD_OP_OUT;
|
||||||
// Already checked that it exists above, but anyway
|
// Already checked that it exists above, but anyway
|
||||||
op->peer_fd = msgr.osd_peer_fds.at(cur_list->osd_num);
|
op->client_id = msgr.osd_peers.at(cur_list->osd_num)->client_id;
|
||||||
op->req = (osd_any_op_t){
|
op->req = (osd_any_op_t){
|
||||||
.sec_list = {
|
.sec_list = {
|
||||||
.header = {
|
.header = {
|
||||||
|
|||||||
@@ -0,0 +1,11 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
#include "cluster_client.h"
|
||||||
|
#include "etcd_state_client_http.h"
|
||||||
|
|
||||||
|
cluster_client_t* cluster_client_t::create(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config)
|
||||||
|
{
|
||||||
|
auto st_cli = new etcd_state_client_http_t(tfd);
|
||||||
|
return new cluster_client_t(ringloop, tfd, config, std::unique_ptr<etcd_state_client_t>(st_cli));
|
||||||
|
}
|
||||||
@@ -88,6 +88,11 @@ void writeback_cache_t::copy_write(cluster_op_t *op, int state, uint64_t new_flu
|
|||||||
// ...or just save it for writeback if write buffering is enabled
|
// ...or just save it for writeback if write buffering is enabled
|
||||||
if (op->len == 0)
|
if (op->len == 0)
|
||||||
{
|
{
|
||||||
|
// FIXME: OSD_OP_DELETEs are currently only sent by vitastor-cli rm/rm-data and
|
||||||
|
// actually have len=0, because delete is actually a delete of the full object
|
||||||
|
// containing the requested offset, not a "punch hole" operation. But here, writeback
|
||||||
|
// cache assumes it IS a "punch hole" operation. I should select one of these
|
||||||
|
// approaches and fix everything accordingly when I decide to implement TRIM.
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
auto dirty_it = find_dirty(op->inode, op->offset);
|
auto dirty_it = find_dirty(op->inode, op->offset);
|
||||||
@@ -126,6 +131,7 @@ void writeback_cache_t::copy_write(cluster_op_t *op, int state, uint64_t new_flu
|
|||||||
writeback_bytes -= op->len;
|
writeback_bytes -= op->len;
|
||||||
}
|
}
|
||||||
writeback_queue_size++;
|
writeback_queue_size++;
|
||||||
|
writeback_queue.push_back({ op->inode, new_end });
|
||||||
}
|
}
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
@@ -160,6 +166,7 @@ void writeback_cache_t::copy_write(cluster_op_t *op, int state, uint64_t new_flu
|
|||||||
{
|
{
|
||||||
writeback_queue_size++;
|
writeback_queue_size++;
|
||||||
}
|
}
|
||||||
|
writeback_queue.push_back({ op->inode, new_end });
|
||||||
}
|
}
|
||||||
auto new_dirty_it = dirty_buffers.emplace_hint(dirty_it, (object_id){
|
auto new_dirty_it = dirty_buffers.emplace_hint(dirty_it, (object_id){
|
||||||
.inode = op->inode,
|
.inode = op->inode,
|
||||||
@@ -244,12 +251,13 @@ void writeback_cache_t::copy_write(cluster_op_t *op, int state, uint64_t new_flu
|
|||||||
writeback_queue_size--;
|
writeback_queue_size--;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!is_del)
|
if (!is_del && op->len > 0)
|
||||||
{
|
{
|
||||||
uint64_t pos = 0, len = op->len, iov_idx = 0;
|
uint64_t pos = 0, len = op->len, iov_idx = 0;
|
||||||
while (len > 0 && iov_idx < op->iov.count)
|
while (iov_idx < op->iov.count)
|
||||||
{
|
{
|
||||||
auto & iov = op->iov.buf[iov_idx];
|
auto & iov = op->iov.buf[iov_idx];
|
||||||
|
assert(pos + iov.iov_len <= len);
|
||||||
memcpy(buf + pos, iov.iov_base, iov.iov_len);
|
memcpy(buf + pos, iov.iov_base, iov.iov_len);
|
||||||
pos += iov.iov_len;
|
pos += iov.iov_len;
|
||||||
iov_idx++;
|
iov_idx++;
|
||||||
@@ -443,7 +451,7 @@ void writeback_cache_t::start_writebacks(cluster_client_t *cli, int count)
|
|||||||
started++;
|
started++;
|
||||||
assert(writeback_queue_size > 0);
|
assert(writeback_queue_size > 0);
|
||||||
writeback_queue_size--;
|
writeback_queue_size--;
|
||||||
writeback_bytes -= off - from_it->first.stripe;
|
writeback_bytes -= (is_del ? 0 : off - from_it->first.stripe);
|
||||||
assert(writeback_queue_size > 0 || !writeback_bytes);
|
assert(writeback_queue_size > 0 || !writeback_bytes);
|
||||||
flush_buffers(cli, from_it, to_it);
|
flush_buffers(cli, from_it, to_it);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,13 +1,12 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
#include <assert.h>
|
||||||
|
|
||||||
#include "osd_ops.h"
|
#include "osd_ops.h"
|
||||||
#include "pg_states.h"
|
#include "pg_states.h"
|
||||||
#include "etcd_state_client.h"
|
#include "etcd_state_client.h"
|
||||||
#ifndef __MOCK__
|
|
||||||
#include "addr_util.h"
|
#include "addr_util.h"
|
||||||
#include "http_client.h"
|
|
||||||
#endif
|
|
||||||
#include "str_util.h"
|
#include "str_util.h"
|
||||||
|
|
||||||
etcd_state_client_t::~etcd_state_client_t()
|
etcd_state_client_t::~etcd_state_client_t()
|
||||||
@@ -17,28 +16,8 @@ etcd_state_client_t::~etcd_state_client_t()
|
|||||||
delete watch;
|
delete watch;
|
||||||
}
|
}
|
||||||
watches.clear();
|
watches.clear();
|
||||||
etcd_watches_initialised = -1;
|
|
||||||
#ifndef __MOCK__
|
|
||||||
stop_ws_keepalive();
|
|
||||||
if (etcd_watch_ws)
|
|
||||||
{
|
|
||||||
http_close(etcd_watch_ws);
|
|
||||||
etcd_watch_ws = NULL;
|
|
||||||
}
|
|
||||||
if (keepalive_client)
|
|
||||||
{
|
|
||||||
http_close(keepalive_client);
|
|
||||||
keepalive_client = NULL;
|
|
||||||
}
|
|
||||||
#endif
|
|
||||||
if (load_pgs_timer_id >= 0)
|
|
||||||
{
|
|
||||||
tfd->clear_timer(load_pgs_timer_id);
|
|
||||||
load_pgs_timer_id = -1;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#ifndef __MOCK__
|
|
||||||
etcd_kv_t etcd_state_client_t::parse_etcd_kv(const json11::Json & kv_json)
|
etcd_kv_t etcd_state_client_t::parse_etcd_kv(const json11::Json & kv_json)
|
||||||
{
|
{
|
||||||
etcd_kv_t kv;
|
etcd_kv_t kv;
|
||||||
@@ -72,104 +51,6 @@ std::vector<std::string> etcd_state_client_t::get_addresses()
|
|||||||
return addrs;
|
return addrs;
|
||||||
}
|
}
|
||||||
|
|
||||||
void etcd_state_client_t::etcd_call_oneshot(std::string etcd_address, std::string api, json11::Json payload,
|
|
||||||
int timeout, std::function<void(std::string, json11::Json)> callback)
|
|
||||||
{
|
|
||||||
std::string etcd_api_path;
|
|
||||||
int pos = etcd_address.find('/');
|
|
||||||
if (pos >= 0)
|
|
||||||
{
|
|
||||||
etcd_api_path = etcd_address.substr(pos);
|
|
||||||
etcd_address = etcd_address.substr(0, pos);
|
|
||||||
}
|
|
||||||
std::string req = payload.dump();
|
|
||||||
req = "POST "+etcd_api_path+api+" HTTP/1.1\r\n"
|
|
||||||
"Host: "+etcd_address+"\r\n"
|
|
||||||
"Content-Type: application/json\r\n"
|
|
||||||
"Content-Length: "+std::to_string(req.size())+"\r\n"
|
|
||||||
"Connection: close\r\n"
|
|
||||||
"\r\n"+req;
|
|
||||||
auto http_cli = http_init(tfd);
|
|
||||||
auto cb = [http_cli, callback](const http_response_t *response)
|
|
||||||
{
|
|
||||||
std::string err;
|
|
||||||
json11::Json data;
|
|
||||||
response->parse_json_response(err, data);
|
|
||||||
callback(err, data);
|
|
||||||
http_close(http_cli);
|
|
||||||
};
|
|
||||||
http_request(http_cli, etcd_address, req, { .timeout = timeout }, cb);
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::etcd_call(std::string api, json11::Json payload, int timeout,
|
|
||||||
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
|
|
||||||
{
|
|
||||||
if (!etcd_addresses.size() && !etcd_local.size())
|
|
||||||
{
|
|
||||||
fprintf(stderr, "etcd_address is missing in Vitastor configuration\n");
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
pick_next_etcd();
|
|
||||||
std::string etcd_address = selected_etcd_address;
|
|
||||||
std::string etcd_api_path;
|
|
||||||
int pos = etcd_address.find('/');
|
|
||||||
if (pos >= 0)
|
|
||||||
{
|
|
||||||
etcd_api_path = etcd_address.substr(pos);
|
|
||||||
etcd_address = etcd_address.substr(0, pos);
|
|
||||||
}
|
|
||||||
std::string req = payload.dump();
|
|
||||||
req = "POST "+etcd_api_path+api+" HTTP/1.1\r\n"
|
|
||||||
"Host: "+etcd_address+"\r\n"
|
|
||||||
"Content-Type: application/json\r\n"
|
|
||||||
"Content-Length: "+std::to_string(req.size())+"\r\n"
|
|
||||||
"Connection: keep-alive\r\n"
|
|
||||||
"Keep-Alive: timeout="+std::to_string(etcd_keepalive_timeout)+"\r\n"
|
|
||||||
"\r\n"+req;
|
|
||||||
retries--;
|
|
||||||
auto cb = [this, api, payload, timeout, retries, interval, callback,
|
|
||||||
cur_addr = selected_etcd_address](const http_response_t *response)
|
|
||||||
{
|
|
||||||
std::string err;
|
|
||||||
json11::Json data;
|
|
||||||
response->parse_json_response(err, data);
|
|
||||||
if (err != "")
|
|
||||||
{
|
|
||||||
if (cur_addr == selected_etcd_address)
|
|
||||||
selected_etcd_address = "";
|
|
||||||
if (retries > 0)
|
|
||||||
{
|
|
||||||
if (this->log_level > 0)
|
|
||||||
{
|
|
||||||
fprintf(
|
|
||||||
stderr, "Warning: etcd request failed: %s, retrying %d more times\n",
|
|
||||||
err.c_str(), retries
|
|
||||||
);
|
|
||||||
}
|
|
||||||
if (interval > 0)
|
|
||||||
{
|
|
||||||
// FIXME: Prevent destruction of etcd_state_client if timers or requests are active
|
|
||||||
tfd->set_timer(interval, false, [this, api, payload, timeout, retries, interval, callback](int)
|
|
||||||
{
|
|
||||||
etcd_call(api, payload, timeout, retries, interval, callback);
|
|
||||||
});
|
|
||||||
}
|
|
||||||
else
|
|
||||||
etcd_call(api, payload, timeout, retries, interval, callback);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
callback(err, data);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
callback(err, data);
|
|
||||||
};
|
|
||||||
if (!keepalive_client)
|
|
||||||
{
|
|
||||||
keepalive_client = http_init(tfd);
|
|
||||||
}
|
|
||||||
http_request(keepalive_client, etcd_address, req, { .timeout = timeout, .keepalive = true }, cb);
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::add_etcd_url(std::string addr)
|
void etcd_state_client_t::add_etcd_url(std::string addr)
|
||||||
{
|
{
|
||||||
if (addr.length() > 0)
|
if (addr.length() > 0)
|
||||||
@@ -256,7 +137,6 @@ void etcd_state_client_t::parse_config(const json11::Json & config)
|
|||||||
if (this->etcd_keepalive_timeout < 30)
|
if (this->etcd_keepalive_timeout < 30)
|
||||||
this->etcd_keepalive_timeout = 30;
|
this->etcd_keepalive_timeout = 30;
|
||||||
}
|
}
|
||||||
auto old_etcd_ws_keepalive_interval = this->etcd_ws_keepalive_interval;
|
|
||||||
this->etcd_ws_keepalive_interval = config["etcd_ws_keepalive_interval"].uint64_value();
|
this->etcd_ws_keepalive_interval = config["etcd_ws_keepalive_interval"].uint64_value();
|
||||||
if (this->etcd_ws_keepalive_interval <= 0)
|
if (this->etcd_ws_keepalive_interval <= 0)
|
||||||
{
|
{
|
||||||
@@ -282,294 +162,9 @@ void etcd_state_client_t::parse_config(const json11::Json & config)
|
|||||||
{
|
{
|
||||||
this->etcd_min_reload_interval = 50;
|
this->etcd_min_reload_interval = 50;
|
||||||
}
|
}
|
||||||
if (this->etcd_ws_keepalive_interval != old_etcd_ws_keepalive_interval && ws_keepalive_timer >= 0)
|
|
||||||
{
|
|
||||||
#ifndef __MOCK__
|
|
||||||
stop_ws_keepalive();
|
|
||||||
start_ws_keepalive();
|
|
||||||
#endif
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
void etcd_state_client_t::pick_next_etcd()
|
void etcd_state_client_t::load_global_config(std::function<void(const std::string & error)> cb)
|
||||||
{
|
|
||||||
if (selected_etcd_address != "")
|
|
||||||
return;
|
|
||||||
if (addresses_to_try.size() == 0)
|
|
||||||
{
|
|
||||||
// Prefer local etcd, if any
|
|
||||||
for (int i = 0; i < etcd_local.size(); i++)
|
|
||||||
addresses_to_try.push_back(etcd_local[i]);
|
|
||||||
std::vector<int> ns;
|
|
||||||
for (int i = 0; i < etcd_addresses.size(); i++)
|
|
||||||
ns.push_back(i);
|
|
||||||
if (!rand_initialized)
|
|
||||||
{
|
|
||||||
timespec tv;
|
|
||||||
clock_gettime(CLOCK_REALTIME, &tv);
|
|
||||||
srand48(tv.tv_sec*1000000000 + tv.tv_nsec);
|
|
||||||
rand_initialized = true;
|
|
||||||
}
|
|
||||||
while (ns.size())
|
|
||||||
{
|
|
||||||
int i = lrand48() % ns.size();
|
|
||||||
addresses_to_try.push_back(etcd_addresses[ns[i]]);
|
|
||||||
ns.erase(ns.begin()+i, ns.begin()+i+1);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
selected_etcd_address = addresses_to_try[0];
|
|
||||||
addresses_to_try.erase(addresses_to_try.begin(), addresses_to_try.begin()+1);
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::start_etcd_watcher()
|
|
||||||
{
|
|
||||||
if (!etcd_addresses.size() && !etcd_local.size())
|
|
||||||
{
|
|
||||||
fprintf(stderr, "etcd_address is missing in Vitastor configuration\n");
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
pick_next_etcd();
|
|
||||||
std::string etcd_address = selected_etcd_address;
|
|
||||||
std::string etcd_api_path;
|
|
||||||
int pos = etcd_address.find('/');
|
|
||||||
if (pos >= 0)
|
|
||||||
{
|
|
||||||
etcd_api_path = etcd_address.substr(pos);
|
|
||||||
etcd_address = etcd_address.substr(0, pos);
|
|
||||||
}
|
|
||||||
etcd_watches_initialised = 0;
|
|
||||||
ws_alive = 1;
|
|
||||||
if (etcd_watch_ws)
|
|
||||||
{
|
|
||||||
http_close(etcd_watch_ws);
|
|
||||||
etcd_watch_ws = NULL;
|
|
||||||
}
|
|
||||||
if (this->log_level > 1)
|
|
||||||
{
|
|
||||||
fprintf(stderr, "Trying to connect to etcd websocket at %s, watch from revision %ju/%ju/%ju\n", etcd_address.c_str(),
|
|
||||||
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
|
|
||||||
}
|
|
||||||
etcd_watch_ws = open_websocket(tfd, etcd_address, etcd_api_path+"/watch", etcd_slow_timeout,
|
|
||||||
[this, cur_addr = selected_etcd_address](const http_response_t *msg)
|
|
||||||
{
|
|
||||||
if (msg->body.length())
|
|
||||||
{
|
|
||||||
ws_alive = 1;
|
|
||||||
std::string json_err;
|
|
||||||
json11::Json data = json11::Json::parse(msg->body, json_err);
|
|
||||||
if (json_err != "")
|
|
||||||
{
|
|
||||||
fprintf(stderr, "Bad JSON in etcd event: %s, ignoring event\n", json_err.c_str());
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
uint64_t watch_id = data["result"]["watch_id"].uint64_value();
|
|
||||||
if (data["result"]["created"].bool_value())
|
|
||||||
{
|
|
||||||
if (watch_id == ETCD_CONFIG_WATCH_ID ||
|
|
||||||
watch_id == ETCD_PG_STATE_WATCH_ID ||
|
|
||||||
watch_id == ETCD_OSD_STATE_WATCH_ID)
|
|
||||||
{
|
|
||||||
etcd_watches_initialised++;
|
|
||||||
}
|
|
||||||
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES && this->log_level > 0)
|
|
||||||
{
|
|
||||||
fprintf(stderr, "Successfully subscribed to etcd at %s, revision %ju/%ju/%ju\n", cur_addr.c_str(),
|
|
||||||
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (data["result"]["canceled"].bool_value())
|
|
||||||
{
|
|
||||||
// etcd watch canceled, maybe because the revision was compacted
|
|
||||||
if (data["result"]["compact_revision"].uint64_value())
|
|
||||||
{
|
|
||||||
// we may miss events if we proceed
|
|
||||||
// so we should restart from the beginning if we can
|
|
||||||
if (on_reload_hook != NULL)
|
|
||||||
{
|
|
||||||
// check to not trigger on_reload_hook multiple times
|
|
||||||
if (etcd_watch_ws != NULL)
|
|
||||||
{
|
|
||||||
fprintf(stderr, "Revisions before %ju were compacted by etcd, reloading state\n",
|
|
||||||
data["result"]["compact_revision"].uint64_value());
|
|
||||||
http_close(etcd_watch_ws);
|
|
||||||
etcd_watch_ws = NULL;
|
|
||||||
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = 0;
|
|
||||||
on_reload_hook();
|
|
||||||
}
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
fprintf(stderr, "Revisions before %ju were compacted by etcd, exiting\n",
|
|
||||||
data["result"]["compact_revision"].uint64_value());
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
fprintf(stderr, "Watch canceled by etcd, reason: %s, exiting\n", data["result"]["cancel_reason"].string_value().c_str());
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Save revision only if it's present in the message - because sometimes etcd sends something without a header, like:
|
|
||||||
// {"error": {"grpc_code": 14, "http_code": 503, "http_status": "Service Unavailable", "message": "error reading from server: EOF"}}
|
|
||||||
// Also don't save revision from the initial created: true messages because they always contain the latest revision
|
|
||||||
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES &&
|
|
||||||
!data["result"]["header"]["revision"].is_null() &&
|
|
||||||
!data["result"]["created"].bool_value())
|
|
||||||
{
|
|
||||||
// Restart watchers from the same revision number as in the last received message,
|
|
||||||
// not from the next one to protect against revision being split into multiple messages,
|
|
||||||
// even though etcd guarantees not to do that **within a single watcher** without fragment=true:
|
|
||||||
// https://etcd.io/docs/v3.5/learning/api_guarantees/#watch-apis
|
|
||||||
// Revision contents are ALWAYS split into separate messages for different watchers though!
|
|
||||||
// So generally we have to resume each watcher from its own revision...
|
|
||||||
// Progress messages may have watch_id=-1 if sent on behalf of multiple watchers though.
|
|
||||||
// And antietcd has an advanced semantic which merges the same revision for all watchers
|
|
||||||
// into one message and just omits watch_id.
|
|
||||||
// So we also have to handle the case where watch_id is -1 or not present (0).
|
|
||||||
auto watch_rev = data["result"]["header"]["revision"].uint64_value();
|
|
||||||
if (!watch_id || watch_id == UINT64_MAX)
|
|
||||||
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = watch_rev;
|
|
||||||
else if (watch_id == ETCD_CONFIG_WATCH_ID)
|
|
||||||
etcd_watch_revision_config = watch_rev;
|
|
||||||
else if (watch_id == ETCD_PG_STATE_WATCH_ID)
|
|
||||||
etcd_watch_revision_pg = watch_rev;
|
|
||||||
else if (watch_id == ETCD_OSD_STATE_WATCH_ID)
|
|
||||||
etcd_watch_revision_osd = watch_rev;
|
|
||||||
addresses_to_try.clear();
|
|
||||||
}
|
|
||||||
// First gather all changes into a hash to remove multiple overwrites
|
|
||||||
std::map<std::string, etcd_kv_t> changes;
|
|
||||||
for (auto & ev: data["result"]["events"].array_items())
|
|
||||||
{
|
|
||||||
auto kv = parse_etcd_kv(ev["kv"]);
|
|
||||||
if (kv.key != "")
|
|
||||||
{
|
|
||||||
changes[kv.key] = kv;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (auto & kv: changes)
|
|
||||||
{
|
|
||||||
if (this->log_level > 3)
|
|
||||||
{
|
|
||||||
fprintf(stderr, "Incoming event: %s -> %s\n", kv.first.c_str(), kv.second.value.dump().c_str());
|
|
||||||
}
|
|
||||||
parse_state(kv.second);
|
|
||||||
}
|
|
||||||
// React to changes
|
|
||||||
if (on_change_hook != NULL)
|
|
||||||
{
|
|
||||||
on_change_hook(changes);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (msg->eof)
|
|
||||||
{
|
|
||||||
fprintf(stderr, "Disconnected from etcd %s\n", cur_addr.c_str());
|
|
||||||
if (cur_addr == selected_etcd_address)
|
|
||||||
selected_etcd_address = "";
|
|
||||||
if (etcd_watch_ws)
|
|
||||||
{
|
|
||||||
http_close(etcd_watch_ws);
|
|
||||||
etcd_watch_ws = NULL;
|
|
||||||
}
|
|
||||||
if (etcd_watches_initialised == 0)
|
|
||||||
{
|
|
||||||
// Connection not established, retry in <etcd_quick_timeout>
|
|
||||||
tfd->set_timer(etcd_quick_timeout, false, [this](int)
|
|
||||||
{
|
|
||||||
start_etcd_watcher();
|
|
||||||
});
|
|
||||||
}
|
|
||||||
else if (etcd_watches_initialised > 0)
|
|
||||||
{
|
|
||||||
// Connection was live, retry immediately
|
|
||||||
etcd_watches_initialised = 0;
|
|
||||||
start_etcd_watcher();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
});
|
|
||||||
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
|
||||||
{ "create_request", json11::Json::object {
|
|
||||||
{ "key", base64_encode(etcd_prefix+"/config/") },
|
|
||||||
{ "range_end", base64_encode(etcd_prefix+"/config0") },
|
|
||||||
{ "start_revision", etcd_watch_revision_config },
|
|
||||||
{ "watch_id", ETCD_CONFIG_WATCH_ID },
|
|
||||||
{ "progress_notify", true },
|
|
||||||
} }
|
|
||||||
}).dump());
|
|
||||||
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
|
||||||
{ "create_request", json11::Json::object {
|
|
||||||
{ "key", base64_encode(etcd_prefix+"/osd/state/") },
|
|
||||||
{ "range_end", base64_encode(etcd_prefix+"/osd/state0") },
|
|
||||||
{ "start_revision", etcd_watch_revision_osd },
|
|
||||||
{ "watch_id", ETCD_OSD_STATE_WATCH_ID },
|
|
||||||
{ "progress_notify", true },
|
|
||||||
} }
|
|
||||||
}).dump());
|
|
||||||
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
|
||||||
{ "create_request", json11::Json::object {
|
|
||||||
{ "key", base64_encode(etcd_prefix+"/pg/") },
|
|
||||||
{ "range_end", base64_encode(etcd_prefix+"/pg0") },
|
|
||||||
{ "start_revision", etcd_watch_revision_pg },
|
|
||||||
{ "watch_id", ETCD_PG_STATE_WATCH_ID },
|
|
||||||
{ "progress_notify", true },
|
|
||||||
} }
|
|
||||||
}).dump());
|
|
||||||
// FIXME: Do not watch /pg/history/ at all in client code (not in OSD)
|
|
||||||
if (on_start_watcher_hook)
|
|
||||||
{
|
|
||||||
on_start_watcher_hook(etcd_watch_ws);
|
|
||||||
}
|
|
||||||
start_ws_keepalive();
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::stop_ws_keepalive()
|
|
||||||
{
|
|
||||||
if (ws_keepalive_timer >= 0)
|
|
||||||
{
|
|
||||||
tfd->clear_timer(ws_keepalive_timer);
|
|
||||||
ws_keepalive_timer = -1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::start_ws_keepalive()
|
|
||||||
{
|
|
||||||
if (ws_keepalive_timer < 0)
|
|
||||||
{
|
|
||||||
ws_keepalive_timer = tfd->set_timer(etcd_ws_keepalive_interval*1000, true, [this](int)
|
|
||||||
{
|
|
||||||
if (!etcd_watch_ws || etcd_watches_initialised < ETCD_TOTAL_WATCHES)
|
|
||||||
{
|
|
||||||
// Do nothing
|
|
||||||
}
|
|
||||||
else if (!ws_alive)
|
|
||||||
{
|
|
||||||
if (this->log_level > 0)
|
|
||||||
{
|
|
||||||
fprintf(stderr, "Websocket ping failed, disconnecting from etcd %s\n", selected_etcd_address.c_str());
|
|
||||||
}
|
|
||||||
if (etcd_watch_ws)
|
|
||||||
{
|
|
||||||
http_close(etcd_watch_ws);
|
|
||||||
etcd_watch_ws = NULL;
|
|
||||||
}
|
|
||||||
start_etcd_watcher();
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
ws_alive = 0;
|
|
||||||
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
|
||||||
{ "progress_request", json11::Json::object { } }
|
|
||||||
}).dump());
|
|
||||||
}
|
|
||||||
});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::load_global_config()
|
|
||||||
{
|
{
|
||||||
json11::Json::object req = { { "success", json11::Json::array {
|
json11::Json::object req = { { "success", json11::Json::array {
|
||||||
json11::Json::object {
|
json11::Json::object {
|
||||||
@@ -583,22 +178,12 @@ void etcd_state_client_t::load_global_config()
|
|||||||
} }
|
} }
|
||||||
},
|
},
|
||||||
} } };
|
} } };
|
||||||
etcd_txn(req, etcd_quick_timeout, max_etcd_attempts, 0, [this](std::string err, json11::Json data)
|
etcd_txn(req, etcd_quick_timeout, max_etcd_attempts, 0, [this, cb](std::string err, json11::Json data)
|
||||||
{
|
{
|
||||||
if (err != "")
|
if (err != "")
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Error reading configuration from etcd: %s\n", err.c_str());
|
fprintf(stderr, "Error reading configuration from etcd: %s\n", err.c_str());
|
||||||
if (infinite_start)
|
cb(err);
|
||||||
{
|
|
||||||
tfd->set_timer(etcd_slow_timeout, false, [this](int timer_id)
|
|
||||||
{
|
|
||||||
load_global_config();
|
|
||||||
});
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
exit(1);
|
|
||||||
}
|
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
json11::Json config_kv = data["responses"][0]["response_range"]["kvs"][0];
|
json11::Json config_kv = data["responses"][0]["response_range"]["kvs"][0];
|
||||||
@@ -629,28 +214,12 @@ void etcd_state_client_t::load_global_config()
|
|||||||
parse_state(kv);
|
parse_state(kv);
|
||||||
}
|
}
|
||||||
on_load_config_hook(global_config);
|
on_load_config_hook(global_config);
|
||||||
|
cb("");
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
void etcd_state_client_t::load_pgs()
|
void etcd_state_client_t::load_pgs(std::function<void(const std::string &)> cb)
|
||||||
{
|
{
|
||||||
timespec tv;
|
|
||||||
clock_gettime(CLOCK_REALTIME, &tv);
|
|
||||||
uint64_t ms_passed = (tv.tv_sec-etcd_last_reload.tv_sec)*1000 + (tv.tv_nsec-etcd_last_reload.tv_nsec)/1000000;
|
|
||||||
if (ms_passed < etcd_min_reload_interval)
|
|
||||||
{
|
|
||||||
if (load_pgs_timer_id < 0)
|
|
||||||
{
|
|
||||||
load_pgs_timer_id = tfd->set_timer(etcd_min_reload_interval+50-ms_passed, false, [this](int) { load_pgs(); });
|
|
||||||
}
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
etcd_last_reload = tv;
|
|
||||||
if (load_pgs_timer_id >= 0)
|
|
||||||
{
|
|
||||||
tfd->clear_timer(load_pgs_timer_id);
|
|
||||||
load_pgs_timer_id = -1;
|
|
||||||
}
|
|
||||||
json11::Json::array txn = {
|
json11::Json::array txn = {
|
||||||
json11::Json::object {
|
json11::Json::object {
|
||||||
{ "request_range", json11::Json::object {
|
{ "request_range", json11::Json::object {
|
||||||
@@ -698,16 +267,13 @@ void etcd_state_client_t::load_pgs()
|
|||||||
{
|
{
|
||||||
req["compare"] = checks;
|
req["compare"] = checks;
|
||||||
}
|
}
|
||||||
etcd_txn_slow(req, [this](std::string err, json11::Json data)
|
etcd_txn_slow(req, [this, cb](std::string err, json11::Json data)
|
||||||
{
|
{
|
||||||
if (err != "")
|
if (err != "")
|
||||||
{
|
{
|
||||||
// Retry indefinitely
|
// Retry indefinitely
|
||||||
fprintf(stderr, "Error loading PGs from etcd: %s\n", err.c_str());
|
fprintf(stderr, "Error loading PGs from etcd: %s\n", err.c_str());
|
||||||
tfd->set_timer(etcd_slow_timeout, false, [this](int timer_id)
|
cb(err);
|
||||||
{
|
|
||||||
load_pgs();
|
|
||||||
});
|
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if (!data["succeeded"].bool_value())
|
if (!data["succeeded"].bool_value())
|
||||||
@@ -735,24 +301,9 @@ void etcd_state_client_t::load_pgs()
|
|||||||
}
|
}
|
||||||
clean_nonexistent_pgs();
|
clean_nonexistent_pgs();
|
||||||
on_load_pgs_hook(true);
|
on_load_pgs_hook(true);
|
||||||
start_etcd_watcher();
|
cb("");
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
#else
|
|
||||||
void etcd_state_client_t::parse_config(const json11::Json & config)
|
|
||||||
{
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::load_global_config()
|
|
||||||
{
|
|
||||||
json11::Json::object global_config;
|
|
||||||
on_load_config_hook(global_config);
|
|
||||||
}
|
|
||||||
|
|
||||||
void etcd_state_client_t::load_pgs()
|
|
||||||
{
|
|
||||||
}
|
|
||||||
#endif
|
|
||||||
|
|
||||||
void etcd_state_client_t::reset_pg_exists()
|
void etcd_state_client_t::reset_pg_exists()
|
||||||
{
|
{
|
||||||
@@ -865,7 +416,8 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
|
|||||||
if (pc.pg_size < 1 ||
|
if (pc.pg_size < 1 ||
|
||||||
pool_item.second["pg_size"].uint64_value() < 3 &&
|
pool_item.second["pg_size"].uint64_value() < 3 &&
|
||||||
(pc.scheme == POOL_SCHEME_XOR || pc.scheme == POOL_SCHEME_EC) ||
|
(pc.scheme == POOL_SCHEME_XOR || pc.scheme == POOL_SCHEME_EC) ||
|
||||||
pool_item.second["pg_size"].uint64_value() > 256)
|
// limit is 64 because osd_peering_pg.cpp uses a 64-bit mask for has_roles
|
||||||
|
pool_item.second["pg_size"].uint64_value() > 64)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Pool %u has invalid pg_size, skipping pool\n", pool_id);
|
fprintf(stderr, "Pool %u has invalid pg_size, skipping pool\n", pool_id);
|
||||||
continue;
|
continue;
|
||||||
@@ -1185,7 +737,6 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
|
|||||||
if (i >= pg_state_bit_count)
|
if (i >= pg_state_bit_count)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Unexpected pool %u PG %u state keyword in etcd: %s\n", pool_id, pg_num, e.dump().c_str());
|
fprintf(stderr, "Unexpected pool %u PG %u state keyword in etcd: %s\n", pool_id, pg_num, e.dump().c_str());
|
||||||
return;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!cur_primary || !value["state"].is_array() || !state ||
|
if (!cur_primary || !value["state"].is_array() || !state ||
|
||||||
@@ -1194,7 +745,6 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
|
|||||||
(state & PG_INCOMPLETE) && state != PG_INCOMPLETE && state != (PG_INCOMPLETE|PG_HAS_INVALID))
|
(state & PG_INCOMPLETE) && state != PG_INCOMPLETE && state != (PG_INCOMPLETE|PG_HAS_INVALID))
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Unexpected pool %u PG %u state in etcd: primary=%ju, state=%s\n", pool_id, pg_num, cur_primary, value["state"].dump().c_str());
|
fprintf(stderr, "Unexpected pool %u PG %u state in etcd: primary=%ju, state=%s\n", pool_id, pg_num, cur_primary, value["state"].dump().c_str());
|
||||||
return;
|
|
||||||
}
|
}
|
||||||
pg_cfg.cur_primary = cur_primary;
|
pg_cfg.cur_primary = cur_primary;
|
||||||
pg_cfg.cur_state = state;
|
pg_cfg.cur_state = state;
|
||||||
@@ -1207,8 +757,14 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
|
|||||||
else if (key.substr(0, etcd_prefix.length()+11) == etcd_prefix+"/osd/state/")
|
else if (key.substr(0, etcd_prefix.length()+11) == etcd_prefix+"/osd/state/")
|
||||||
{
|
{
|
||||||
// <etcd_prefix>/osd/state/%d
|
// <etcd_prefix>/osd/state/%d
|
||||||
osd_num_t peer_osd = std::stoull(key.substr(etcd_prefix.length()+11));
|
osd_num_t peer_osd = 0;
|
||||||
if (peer_osd > 0)
|
char null_byte = 0;
|
||||||
|
int scanned = sscanf(key.c_str() + etcd_prefix.length()+11, "%ju%c", &peer_osd, &null_byte);
|
||||||
|
if (scanned != 1 || !peer_osd)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Bad etcd key %s, ignoring\n", key.c_str());
|
||||||
|
}
|
||||||
|
else
|
||||||
{
|
{
|
||||||
if (value.is_object() && value["state"] == "up")
|
if (value.is_object() && value["state"] == "up")
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -103,15 +103,13 @@ protected:
|
|||||||
std::vector<std::string> local_ips;
|
std::vector<std::string> local_ips;
|
||||||
std::vector<std::string> etcd_addresses;
|
std::vector<std::string> etcd_addresses;
|
||||||
std::vector<std::string> etcd_local;
|
std::vector<std::string> etcd_local;
|
||||||
std::string selected_etcd_address;
|
|
||||||
std::vector<std::string> addresses_to_try;
|
|
||||||
std::vector<inode_watch_t*> watches;
|
std::vector<inode_watch_t*> watches;
|
||||||
|
std::set<osd_num_t> seen_peers;
|
||||||
bool new_pg_config = false;
|
bool new_pg_config = false;
|
||||||
int ws_keepalive_timer = -1;
|
|
||||||
int ws_alive = 0;
|
|
||||||
bool rand_initialized = false;
|
|
||||||
void add_etcd_url(std::string);
|
void add_etcd_url(std::string);
|
||||||
void pick_next_etcd();
|
void reset_pg_exists();
|
||||||
|
void clean_nonexistent_pgs();
|
||||||
public:
|
public:
|
||||||
int etcd_keepalive_timeout = 30;
|
int etcd_keepalive_timeout = 30;
|
||||||
int etcd_ws_keepalive_interval = 5;
|
int etcd_ws_keepalive_interval = 5;
|
||||||
@@ -123,21 +121,15 @@ public:
|
|||||||
uint64_t global_block_size = DEFAULT_BLOCK_SIZE;
|
uint64_t global_block_size = DEFAULT_BLOCK_SIZE;
|
||||||
uint32_t global_bitmap_granularity = DEFAULT_BITMAP_GRANULARITY;
|
uint32_t global_bitmap_granularity = DEFAULT_BITMAP_GRANULARITY;
|
||||||
uint32_t global_immediate_commit = IMMEDIATE_NONE;
|
uint32_t global_immediate_commit = IMMEDIATE_NONE;
|
||||||
|
|
||||||
std::string etcd_prefix;
|
std::string etcd_prefix;
|
||||||
int log_level = 0;
|
int log_level = 0;
|
||||||
timerfd_manager_t *tfd = NULL;
|
|
||||||
|
|
||||||
http_co_t *etcd_watch_ws = NULL, *keepalive_client = NULL;
|
|
||||||
int etcd_watches_initialised = 0;
|
|
||||||
uint64_t etcd_watch_revision_config = 0;
|
uint64_t etcd_watch_revision_config = 0;
|
||||||
uint64_t etcd_watch_revision_osd = 0;
|
uint64_t etcd_watch_revision_osd = 0;
|
||||||
uint64_t etcd_watch_revision_pg = 0;
|
uint64_t etcd_watch_revision_pg = 0;
|
||||||
timespec etcd_last_reload = {};
|
|
||||||
int load_pgs_timer_id = -1;
|
|
||||||
std::map<pool_id_t, pool_config_t> pool_config;
|
std::map<pool_id_t, pool_config_t> pool_config;
|
||||||
std::map<osd_num_t, json11::Json> peer_states;
|
std::map<osd_num_t, json11::Json> peer_states;
|
||||||
std::set<osd_num_t> seen_peers;
|
|
||||||
std::map<inode_t, inode_config_t> inode_config;
|
std::map<inode_t, inode_config_t> inode_config;
|
||||||
std::map<std::string, inode_t> inode_by_name;
|
std::map<std::string, inode_t> inode_by_name;
|
||||||
json11::Json node_placement;
|
json11::Json node_placement;
|
||||||
@@ -160,24 +152,22 @@ public:
|
|||||||
json11::Json::object serialize_inode_cfg(inode_config_t *cfg);
|
json11::Json::object serialize_inode_cfg(inode_config_t *cfg);
|
||||||
etcd_kv_t parse_etcd_kv(const json11::Json & kv_json);
|
etcd_kv_t parse_etcd_kv(const json11::Json & kv_json);
|
||||||
std::vector<std::string> get_addresses();
|
std::vector<std::string> get_addresses();
|
||||||
void etcd_call_oneshot(std::string etcd_address, std::string api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback);
|
virtual void etcd_call_oneshot(std::string etcd_address, std::string api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback) = 0;
|
||||||
void etcd_call(std::string api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
|
virtual void etcd_call(std::string api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback) = 0;
|
||||||
void etcd_txn(json11::Json txn, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
|
void etcd_txn(json11::Json txn, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
|
||||||
void etcd_txn_slow(json11::Json txn, std::function<void(std::string, json11::Json)> callback);
|
void etcd_txn_slow(json11::Json txn, std::function<void(std::string, json11::Json)> callback);
|
||||||
void start_etcd_watcher();
|
virtual void etcd_add_watch(json11::Json watch) = 0;
|
||||||
void stop_ws_keepalive();
|
void load_global_config(std::function<void(const std::string &)> cb);
|
||||||
void start_ws_keepalive();
|
virtual void load_global_config() = 0;
|
||||||
void load_global_config();
|
void load_pgs(std::function<void(const std::string &)> cb);
|
||||||
void load_pgs();
|
virtual void load_pgs() = 0;
|
||||||
void reset_pg_exists();
|
|
||||||
void clean_nonexistent_pgs();
|
|
||||||
void parse_state(const etcd_kv_t & kv);
|
void parse_state(const etcd_kv_t & kv);
|
||||||
void parse_config(const json11::Json & config);
|
virtual void parse_config(const json11::Json & config);
|
||||||
void insert_inode_config(const inode_config_t & cfg);
|
void insert_inode_config(const inode_config_t & cfg);
|
||||||
inode_watch_t* watch_inode(std::string name);
|
inode_watch_t* watch_inode(std::string name);
|
||||||
void close_watch(inode_watch_t* watch);
|
void close_watch(inode_watch_t* watch);
|
||||||
int address_count();
|
int address_count();
|
||||||
~etcd_state_client_t();
|
virtual ~etcd_state_client_t();
|
||||||
|
|
||||||
static uint32_t parse_immediate_commit(const std::string & immediate_commit_str, uint32_t default_value);
|
static uint32_t parse_immediate_commit(const std::string & immediate_commit_str, uint32_t default_value);
|
||||||
static uint32_t parse_scheme(const std::string & scheme_str);
|
static uint32_t parse_scheme(const std::string & scheme_str);
|
||||||
|
|||||||
@@ -0,0 +1,487 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
#include "etcd_state_client_http.h"
|
||||||
|
#include "addr_util.h"
|
||||||
|
#include "http_client.h"
|
||||||
|
#include "str_util.h"
|
||||||
|
|
||||||
|
etcd_state_client_http_t::etcd_state_client_http_t(timerfd_manager_t *tfd)
|
||||||
|
{
|
||||||
|
this->tfd = tfd;
|
||||||
|
}
|
||||||
|
|
||||||
|
etcd_state_client_http_t::~etcd_state_client_http_t()
|
||||||
|
{
|
||||||
|
stop_ws_keepalive();
|
||||||
|
if (etcd_watch_ws)
|
||||||
|
{
|
||||||
|
http_close(etcd_watch_ws);
|
||||||
|
etcd_watch_ws = NULL;
|
||||||
|
}
|
||||||
|
if (keepalive_client)
|
||||||
|
{
|
||||||
|
http_close(keepalive_client);
|
||||||
|
keepalive_client = NULL;
|
||||||
|
}
|
||||||
|
if (load_pgs_timer_id >= 0)
|
||||||
|
{
|
||||||
|
tfd->clear_timer(load_pgs_timer_id);
|
||||||
|
load_pgs_timer_id = -1;
|
||||||
|
}
|
||||||
|
etcd_watches_initialised = -1;
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_http_t::etcd_add_watch(json11::Json watch)
|
||||||
|
{
|
||||||
|
if (etcd_watch_ws)
|
||||||
|
{
|
||||||
|
http_post_message(etcd_watch_ws, WS_TEXT, watch.dump());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_http_t::etcd_call_oneshot(std::string etcd_address, std::string api, json11::Json payload,
|
||||||
|
int timeout, std::function<void(std::string, json11::Json)> callback)
|
||||||
|
{
|
||||||
|
std::string etcd_api_path;
|
||||||
|
int pos = etcd_address.find('/');
|
||||||
|
if (pos >= 0)
|
||||||
|
{
|
||||||
|
etcd_api_path = etcd_address.substr(pos);
|
||||||
|
etcd_address = etcd_address.substr(0, pos);
|
||||||
|
}
|
||||||
|
std::string req = payload.dump();
|
||||||
|
req = "POST "+etcd_api_path+api+" HTTP/1.1\r\n"
|
||||||
|
"Host: "+etcd_address+"\r\n"
|
||||||
|
"Content-Type: application/json\r\n"
|
||||||
|
"Content-Length: "+std::to_string(req.size())+"\r\n"
|
||||||
|
"Connection: close\r\n"
|
||||||
|
"\r\n"+req;
|
||||||
|
auto http_cli = http_init(tfd);
|
||||||
|
auto cb = [http_cli, callback](const http_response_t *response)
|
||||||
|
{
|
||||||
|
std::string err;
|
||||||
|
json11::Json data;
|
||||||
|
response->parse_json_response(err, data);
|
||||||
|
callback(err, data);
|
||||||
|
http_close(http_cli);
|
||||||
|
};
|
||||||
|
http_request(http_cli, etcd_address, req, { .timeout = timeout }, cb);
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_http_t::etcd_call(std::string api, json11::Json payload, int timeout,
|
||||||
|
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
|
||||||
|
{
|
||||||
|
if (!etcd_addresses.size() && !etcd_local.size())
|
||||||
|
{
|
||||||
|
fprintf(stderr, "etcd_address is missing in Vitastor configuration\n");
|
||||||
|
exit(1);
|
||||||
|
}
|
||||||
|
pick_next_etcd();
|
||||||
|
std::string etcd_address = selected_etcd_address;
|
||||||
|
std::string etcd_api_path;
|
||||||
|
int pos = etcd_address.find('/');
|
||||||
|
if (pos >= 0)
|
||||||
|
{
|
||||||
|
etcd_api_path = etcd_address.substr(pos);
|
||||||
|
etcd_address = etcd_address.substr(0, pos);
|
||||||
|
}
|
||||||
|
std::string req = payload.dump();
|
||||||
|
req = "POST "+etcd_api_path+api+" HTTP/1.1\r\n"
|
||||||
|
"Host: "+etcd_address+"\r\n"
|
||||||
|
"Content-Type: application/json\r\n"
|
||||||
|
"Content-Length: "+std::to_string(req.size())+"\r\n"
|
||||||
|
"Connection: keep-alive\r\n"
|
||||||
|
"Keep-Alive: timeout="+std::to_string(etcd_keepalive_timeout)+"\r\n"
|
||||||
|
"\r\n"+req;
|
||||||
|
retries--;
|
||||||
|
auto cb = [this, api, payload, timeout, retries, interval, callback,
|
||||||
|
cur_addr = selected_etcd_address](const http_response_t *response)
|
||||||
|
{
|
||||||
|
std::string err;
|
||||||
|
json11::Json data;
|
||||||
|
response->parse_json_response(err, data);
|
||||||
|
if (err != "")
|
||||||
|
{
|
||||||
|
if (cur_addr == selected_etcd_address)
|
||||||
|
selected_etcd_address = "";
|
||||||
|
if (retries > 0)
|
||||||
|
{
|
||||||
|
if (this->log_level > 0)
|
||||||
|
{
|
||||||
|
fprintf(
|
||||||
|
stderr, "Warning: etcd request failed: %s, retrying %d more times\n",
|
||||||
|
err.c_str(), retries
|
||||||
|
);
|
||||||
|
}
|
||||||
|
if (interval > 0)
|
||||||
|
{
|
||||||
|
// FIXME: Prevent destruction of etcd_state_client if timers or requests are active
|
||||||
|
tfd->set_timer(interval, false, [this, api, payload, timeout, retries, interval, callback](int)
|
||||||
|
{
|
||||||
|
etcd_call(api, payload, timeout, retries, interval, callback);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
else
|
||||||
|
etcd_call(api, payload, timeout, retries, interval, callback);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
callback(err, data);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
callback(err, data);
|
||||||
|
};
|
||||||
|
if (!keepalive_client)
|
||||||
|
{
|
||||||
|
keepalive_client = http_init(tfd);
|
||||||
|
}
|
||||||
|
http_request(keepalive_client, etcd_address, req, { .timeout = timeout, .keepalive = true }, cb);
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_http_t::parse_config(const json11::Json & config)
|
||||||
|
{
|
||||||
|
auto old_etcd_ws_keepalive_interval = this->etcd_ws_keepalive_interval;
|
||||||
|
etcd_state_client_t::parse_config(config);
|
||||||
|
if (this->etcd_ws_keepalive_interval != old_etcd_ws_keepalive_interval && ws_keepalive_timer >= 0)
|
||||||
|
{
|
||||||
|
stop_ws_keepalive();
|
||||||
|
start_ws_keepalive();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_http_t::pick_next_etcd()
|
||||||
|
{
|
||||||
|
if (selected_etcd_address != "")
|
||||||
|
return;
|
||||||
|
if (addresses_to_try.size() == 0)
|
||||||
|
{
|
||||||
|
// Prefer local etcd, if any
|
||||||
|
for (int i = 0; i < etcd_local.size(); i++)
|
||||||
|
addresses_to_try.push_back(etcd_local[i]);
|
||||||
|
std::vector<int> ns;
|
||||||
|
for (int i = 0; i < etcd_addresses.size(); i++)
|
||||||
|
ns.push_back(i);
|
||||||
|
if (!rand_initialized)
|
||||||
|
{
|
||||||
|
timespec tv;
|
||||||
|
clock_gettime(CLOCK_REALTIME, &tv);
|
||||||
|
srand48(tv.tv_sec*1000000000 + tv.tv_nsec);
|
||||||
|
rand_initialized = true;
|
||||||
|
}
|
||||||
|
while (ns.size())
|
||||||
|
{
|
||||||
|
int i = lrand48() % ns.size();
|
||||||
|
addresses_to_try.push_back(etcd_addresses[ns[i]]);
|
||||||
|
ns.erase(ns.begin()+i, ns.begin()+i+1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
selected_etcd_address = addresses_to_try[0];
|
||||||
|
addresses_to_try.erase(addresses_to_try.begin(), addresses_to_try.begin()+1);
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_http_t::start_etcd_watcher()
|
||||||
|
{
|
||||||
|
if (!etcd_addresses.size() && !etcd_local.size())
|
||||||
|
{
|
||||||
|
fprintf(stderr, "etcd_address is missing in Vitastor configuration\n");
|
||||||
|
exit(1);
|
||||||
|
}
|
||||||
|
pick_next_etcd();
|
||||||
|
std::string etcd_address = selected_etcd_address;
|
||||||
|
std::string etcd_api_path;
|
||||||
|
int pos = etcd_address.find('/');
|
||||||
|
if (pos >= 0)
|
||||||
|
{
|
||||||
|
etcd_api_path = etcd_address.substr(pos);
|
||||||
|
etcd_address = etcd_address.substr(0, pos);
|
||||||
|
}
|
||||||
|
etcd_watches_initialised = 0;
|
||||||
|
ws_alive = 1;
|
||||||
|
if (etcd_watch_ws)
|
||||||
|
{
|
||||||
|
http_close(etcd_watch_ws);
|
||||||
|
etcd_watch_ws = NULL;
|
||||||
|
}
|
||||||
|
if (this->log_level > 1)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Trying to connect to etcd websocket at %s, watch from revision %ju/%ju/%ju\n", etcd_address.c_str(),
|
||||||
|
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
|
||||||
|
}
|
||||||
|
etcd_watch_ws = open_websocket(tfd, etcd_address, etcd_api_path+"/watch", etcd_slow_timeout,
|
||||||
|
[this, cur_addr = selected_etcd_address](const http_response_t *msg)
|
||||||
|
{
|
||||||
|
if (msg->body.length())
|
||||||
|
{
|
||||||
|
ws_alive = 1;
|
||||||
|
std::string json_err;
|
||||||
|
json11::Json data = json11::Json::parse(msg->body, json_err);
|
||||||
|
if (json_err != "")
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Bad JSON in etcd event: %s, ignoring event\n", json_err.c_str());
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
uint64_t watch_id = data["result"]["watch_id"].uint64_value();
|
||||||
|
if (data["result"]["created"].bool_value())
|
||||||
|
{
|
||||||
|
if (watch_id == ETCD_CONFIG_WATCH_ID ||
|
||||||
|
watch_id == ETCD_PG_STATE_WATCH_ID ||
|
||||||
|
watch_id == ETCD_OSD_STATE_WATCH_ID)
|
||||||
|
{
|
||||||
|
etcd_watches_initialised++;
|
||||||
|
}
|
||||||
|
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES && this->log_level > 0)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Successfully subscribed to etcd at %s, revision %ju/%ju/%ju\n", cur_addr.c_str(),
|
||||||
|
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (data["result"]["canceled"].bool_value())
|
||||||
|
{
|
||||||
|
// etcd watch canceled, maybe because the revision was compacted
|
||||||
|
if (data["result"]["compact_revision"].uint64_value())
|
||||||
|
{
|
||||||
|
// we may miss events if we proceed
|
||||||
|
// so we should restart from the beginning if we can
|
||||||
|
if (on_reload_hook != NULL)
|
||||||
|
{
|
||||||
|
// check to not trigger on_reload_hook multiple times
|
||||||
|
if (etcd_watch_ws != NULL)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Revisions before %ju were compacted by etcd, reloading state\n",
|
||||||
|
data["result"]["compact_revision"].uint64_value());
|
||||||
|
http_close(etcd_watch_ws);
|
||||||
|
etcd_watch_ws = NULL;
|
||||||
|
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = 0;
|
||||||
|
on_reload_hook();
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Revisions before %ju were compacted by etcd, exiting\n",
|
||||||
|
data["result"]["compact_revision"].uint64_value());
|
||||||
|
exit(1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Watch canceled by etcd, reason: %s, exiting\n", data["result"]["cancel_reason"].string_value().c_str());
|
||||||
|
exit(1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Save revision only if it's present in the message - because sometimes etcd sends something without a header, like:
|
||||||
|
// {"error": {"grpc_code": 14, "http_code": 503, "http_status": "Service Unavailable", "message": "error reading from server: EOF"}}
|
||||||
|
// Also don't save revision from the initial created: true messages because they always contain the latest revision
|
||||||
|
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES &&
|
||||||
|
!data["result"]["header"]["revision"].is_null() &&
|
||||||
|
!data["result"]["created"].bool_value())
|
||||||
|
{
|
||||||
|
// Restart watchers from the same revision number as in the last received message,
|
||||||
|
// not from the next one to protect against revision being split into multiple messages,
|
||||||
|
// even though etcd guarantees not to do that **within a single watcher** without fragment=true:
|
||||||
|
// https://etcd.io/docs/v3.5/learning/api_guarantees/#watch-apis
|
||||||
|
// Revision contents are ALWAYS split into separate messages for different watchers though!
|
||||||
|
// So generally we have to resume each watcher from its own revision...
|
||||||
|
// Progress messages may have watch_id=-1 if sent on behalf of multiple watchers though.
|
||||||
|
// And antietcd has an advanced semantic which merges the same revision for all watchers
|
||||||
|
// into one message and just omits watch_id.
|
||||||
|
// So we also have to handle the case where watch_id is -1 or not present (0).
|
||||||
|
auto watch_rev = data["result"]["header"]["revision"].uint64_value();
|
||||||
|
if (!watch_id || watch_id == UINT64_MAX)
|
||||||
|
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = watch_rev;
|
||||||
|
else if (watch_id == ETCD_CONFIG_WATCH_ID)
|
||||||
|
etcd_watch_revision_config = watch_rev;
|
||||||
|
else if (watch_id == ETCD_PG_STATE_WATCH_ID)
|
||||||
|
etcd_watch_revision_pg = watch_rev;
|
||||||
|
else if (watch_id == ETCD_OSD_STATE_WATCH_ID)
|
||||||
|
etcd_watch_revision_osd = watch_rev;
|
||||||
|
addresses_to_try.clear();
|
||||||
|
}
|
||||||
|
// First gather all changes into a hash to remove multiple overwrites
|
||||||
|
std::map<std::string, etcd_kv_t> changes;
|
||||||
|
for (auto & ev: data["result"]["events"].array_items())
|
||||||
|
{
|
||||||
|
auto kv = parse_etcd_kv(ev["kv"]);
|
||||||
|
if (kv.key != "")
|
||||||
|
{
|
||||||
|
changes[kv.key] = kv;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (auto & kv: changes)
|
||||||
|
{
|
||||||
|
if (this->log_level > 3)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Incoming event: %s -> %s\n", kv.first.c_str(), kv.second.value.dump().c_str());
|
||||||
|
}
|
||||||
|
parse_state(kv.second);
|
||||||
|
}
|
||||||
|
// React to changes
|
||||||
|
if (on_change_hook != NULL)
|
||||||
|
{
|
||||||
|
on_change_hook(changes);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (msg->eof)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Disconnected from etcd %s\n", cur_addr.c_str());
|
||||||
|
if (cur_addr == selected_etcd_address)
|
||||||
|
selected_etcd_address = "";
|
||||||
|
if (etcd_watch_ws)
|
||||||
|
{
|
||||||
|
http_close(etcd_watch_ws);
|
||||||
|
etcd_watch_ws = NULL;
|
||||||
|
}
|
||||||
|
if (etcd_watches_initialised == 0)
|
||||||
|
{
|
||||||
|
// Connection not established, retry in <etcd_quick_timeout>
|
||||||
|
tfd->set_timer(etcd_quick_timeout, false, [this](int)
|
||||||
|
{
|
||||||
|
start_etcd_watcher();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
else if (etcd_watches_initialised > 0)
|
||||||
|
{
|
||||||
|
// Connection was live, retry immediately
|
||||||
|
etcd_watches_initialised = 0;
|
||||||
|
start_etcd_watcher();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
||||||
|
{ "create_request", json11::Json::object {
|
||||||
|
{ "key", base64_encode(etcd_prefix+"/config/") },
|
||||||
|
{ "range_end", base64_encode(etcd_prefix+"/config0") },
|
||||||
|
{ "start_revision", etcd_watch_revision_config },
|
||||||
|
{ "watch_id", ETCD_CONFIG_WATCH_ID },
|
||||||
|
{ "progress_notify", true },
|
||||||
|
} }
|
||||||
|
}).dump());
|
||||||
|
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
||||||
|
{ "create_request", json11::Json::object {
|
||||||
|
{ "key", base64_encode(etcd_prefix+"/osd/state/") },
|
||||||
|
{ "range_end", base64_encode(etcd_prefix+"/osd/state0") },
|
||||||
|
{ "start_revision", etcd_watch_revision_osd },
|
||||||
|
{ "watch_id", ETCD_OSD_STATE_WATCH_ID },
|
||||||
|
{ "progress_notify", true },
|
||||||
|
} }
|
||||||
|
}).dump());
|
||||||
|
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
||||||
|
{ "create_request", json11::Json::object {
|
||||||
|
{ "key", base64_encode(etcd_prefix+"/pg/") },
|
||||||
|
{ "range_end", base64_encode(etcd_prefix+"/pg0") },
|
||||||
|
{ "start_revision", etcd_watch_revision_pg },
|
||||||
|
{ "watch_id", ETCD_PG_STATE_WATCH_ID },
|
||||||
|
{ "progress_notify", true },
|
||||||
|
} }
|
||||||
|
}).dump());
|
||||||
|
// FIXME: Do not watch /pg/history/ at all in client code (not in OSD)
|
||||||
|
if (on_start_watcher_hook)
|
||||||
|
{
|
||||||
|
on_start_watcher_hook(etcd_watch_ws);
|
||||||
|
}
|
||||||
|
start_ws_keepalive();
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_http_t::stop_ws_keepalive()
|
||||||
|
{
|
||||||
|
if (ws_keepalive_timer >= 0)
|
||||||
|
{
|
||||||
|
tfd->clear_timer(ws_keepalive_timer);
|
||||||
|
ws_keepalive_timer = -1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_http_t::start_ws_keepalive()
|
||||||
|
{
|
||||||
|
if (ws_keepalive_timer < 0)
|
||||||
|
{
|
||||||
|
ws_keepalive_timer = tfd->set_timer(etcd_ws_keepalive_interval*1000, true, [this](int)
|
||||||
|
{
|
||||||
|
if (!etcd_watch_ws || etcd_watches_initialised < ETCD_TOTAL_WATCHES)
|
||||||
|
{
|
||||||
|
// Do nothing
|
||||||
|
}
|
||||||
|
else if (!ws_alive)
|
||||||
|
{
|
||||||
|
if (this->log_level > 0)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Websocket ping failed, disconnecting from etcd %s\n", selected_etcd_address.c_str());
|
||||||
|
}
|
||||||
|
if (etcd_watch_ws)
|
||||||
|
{
|
||||||
|
http_close(etcd_watch_ws);
|
||||||
|
etcd_watch_ws = NULL;
|
||||||
|
}
|
||||||
|
start_etcd_watcher();
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
ws_alive = 0;
|
||||||
|
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
||||||
|
{ "progress_request", json11::Json::object { } }
|
||||||
|
}).dump());
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_http_t::load_global_config()
|
||||||
|
{
|
||||||
|
etcd_state_client_t::load_global_config([this](const std::string & err)
|
||||||
|
{
|
||||||
|
if (err != "")
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Error reading configuration from etcd: %s\n", err.c_str());
|
||||||
|
if (infinite_start)
|
||||||
|
{
|
||||||
|
tfd->set_timer(etcd_slow_timeout, false, [this](int timer_id)
|
||||||
|
{
|
||||||
|
load_global_config();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
exit(1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_http_t::load_pgs()
|
||||||
|
{
|
||||||
|
timespec tv;
|
||||||
|
clock_gettime(CLOCK_REALTIME, &tv);
|
||||||
|
uint64_t ms_passed = (tv.tv_sec-etcd_last_reload.tv_sec)*1000 + (tv.tv_nsec-etcd_last_reload.tv_nsec)/1000000;
|
||||||
|
if (ms_passed < etcd_min_reload_interval)
|
||||||
|
{
|
||||||
|
if (load_pgs_timer_id < 0)
|
||||||
|
{
|
||||||
|
load_pgs_timer_id = tfd->set_timer(etcd_min_reload_interval+50-ms_passed, false, [this](int) { load_pgs(); });
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
etcd_last_reload = tv;
|
||||||
|
if (load_pgs_timer_id >= 0)
|
||||||
|
{
|
||||||
|
tfd->clear_timer(load_pgs_timer_id);
|
||||||
|
load_pgs_timer_id = -1;
|
||||||
|
}
|
||||||
|
etcd_state_client_t::load_pgs([this](const std::string & err)
|
||||||
|
{
|
||||||
|
if (err != "")
|
||||||
|
{
|
||||||
|
// Retry indefinitely
|
||||||
|
fprintf(stderr, "Error loading PGs from etcd: %s\n", err.c_str());
|
||||||
|
tfd->set_timer(etcd_slow_timeout, false, [this](int timer_id)
|
||||||
|
{
|
||||||
|
load_pgs();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
start_etcd_watcher();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
@@ -0,0 +1,36 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include "etcd_state_client.h"
|
||||||
|
|
||||||
|
struct __attribute__((visibility("default"))) etcd_state_client_http_t: public etcd_state_client_t
|
||||||
|
{
|
||||||
|
protected:
|
||||||
|
timerfd_manager_t *tfd = NULL;
|
||||||
|
std::string selected_etcd_address;
|
||||||
|
std::vector<std::string> addresses_to_try;
|
||||||
|
int ws_keepalive_timer = -1;
|
||||||
|
int ws_alive = 0;
|
||||||
|
bool rand_initialized = false;
|
||||||
|
int etcd_watches_initialised = 0;
|
||||||
|
timespec etcd_last_reload = {};
|
||||||
|
int load_pgs_timer_id = -1;
|
||||||
|
http_co_t *keepalive_client = NULL;
|
||||||
|
|
||||||
|
void pick_next_etcd();
|
||||||
|
void start_etcd_watcher();
|
||||||
|
void stop_ws_keepalive();
|
||||||
|
void start_ws_keepalive();
|
||||||
|
public:
|
||||||
|
http_co_t *etcd_watch_ws = NULL;
|
||||||
|
etcd_state_client_http_t(timerfd_manager_t *tfd);
|
||||||
|
void etcd_call_oneshot(std::string etcd_address, std::string api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback) override;
|
||||||
|
void etcd_call(std::string api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback) override;
|
||||||
|
void etcd_add_watch(json11::Json watch) override;
|
||||||
|
void load_global_config() override;
|
||||||
|
void load_pgs() override;
|
||||||
|
void parse_config(const json11::Json & config) override;
|
||||||
|
~etcd_state_client_http_t();
|
||||||
|
};
|
||||||
@@ -0,0 +1,205 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
#include <assert.h>
|
||||||
|
#include "etcd_state_client_mock.h"
|
||||||
|
#include "str_util.h"
|
||||||
|
|
||||||
|
etcd_state_client_mock_t::etcd_state_client_mock_t()
|
||||||
|
{
|
||||||
|
timespec tv;
|
||||||
|
clock_gettime(CLOCK_REALTIME, &tv);
|
||||||
|
srand48(tv.tv_sec*1000000000 + tv.tv_nsec);
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_mock_t::etcd_add_watch(json11::Json watch)
|
||||||
|
{
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_mock_t::etcd_call_oneshot(std::string etcd_address, std::string api, json11::Json payload,
|
||||||
|
int timeout, std::function<void(std::string, json11::Json)> callback)
|
||||||
|
{
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_mock_t::pause()
|
||||||
|
{
|
||||||
|
paused = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_mock_t::resume()
|
||||||
|
{
|
||||||
|
paused = false;
|
||||||
|
auto queue = std::move(this->queue);
|
||||||
|
for (auto& req: queue)
|
||||||
|
{
|
||||||
|
etcd_call(req.api, req.payload, req.timeout, req.retries, req.interval, req.callback);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_mock_t::set(const std::string& key, json11::Json data, uint64_t mod_revision, uint64_t lease_id)
|
||||||
|
{
|
||||||
|
if (!mod_revision)
|
||||||
|
mod_revision = ++this->mod_revision;
|
||||||
|
this->data[key] = (etcd_mock_key_data_t){ .value = data.dump(), .mod_revision = mod_revision, .lease_id = lease_id };
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_mock_t::etcd_call(std::string api, json11::Json payload, int timeout,
|
||||||
|
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
|
||||||
|
{
|
||||||
|
if (paused)
|
||||||
|
{
|
||||||
|
queue.push_back({ api, payload, timeout, retries, interval, callback });
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
printf("+ etcd: %s\n", api.c_str());
|
||||||
|
if (api == "/kv/txn")
|
||||||
|
{
|
||||||
|
bool ok = true;
|
||||||
|
for (auto& check: payload["compare"].array_items())
|
||||||
|
{
|
||||||
|
auto key = base64_decode(check["key"].string_value());
|
||||||
|
etcd_mock_key_data_t *key_data = data.find(key) != data.end() ? &data.at(key) : NULL;
|
||||||
|
auto target = check["target"].string_value();
|
||||||
|
auto res = check["result"].string_value();
|
||||||
|
assert(res == "LESS" || res == "");
|
||||||
|
bool less = res == "LESS";
|
||||||
|
if (target == "MOD")
|
||||||
|
{
|
||||||
|
uint64_t rev = check["mod_revision"].uint64_value();
|
||||||
|
assert(!less || rev);
|
||||||
|
ok = ok && (less ? (!key_data || key_data->mod_revision < rev) : (key_data && key_data->mod_revision == rev));
|
||||||
|
}
|
||||||
|
else if (target == "CREATE")
|
||||||
|
{
|
||||||
|
uint64_t rev = check["create_revision"].uint64_value();
|
||||||
|
assert(rev == 0 && !less);
|
||||||
|
ok = ok && !key_data;
|
||||||
|
}
|
||||||
|
else if (target == "VERSION")
|
||||||
|
{
|
||||||
|
uint64_t rev = check["version"].uint64_value();
|
||||||
|
assert(rev == 0 && !less);
|
||||||
|
ok = ok && !key_data;
|
||||||
|
}
|
||||||
|
else if (target == "LEASE")
|
||||||
|
{
|
||||||
|
assert(!less);
|
||||||
|
uint64_t lease_id = check["lease"].uint64_value();
|
||||||
|
ok = ok && key_data && key_data->lease_id == lease_id;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
assert(0);
|
||||||
|
}
|
||||||
|
std::map<std::string, etcd_kv_t> changes;
|
||||||
|
bool has_mod = false;
|
||||||
|
for (auto& op: payload[ok ? "success" : "failure"].array_items())
|
||||||
|
{
|
||||||
|
auto& obj = op.object_items();
|
||||||
|
has_mod = has_mod || obj.find("request_put") != obj.end() ||
|
||||||
|
obj.find("request_delete_range") != obj.end();
|
||||||
|
}
|
||||||
|
if (has_mod)
|
||||||
|
{
|
||||||
|
mod_revision++;
|
||||||
|
}
|
||||||
|
json11::Json::array responses;
|
||||||
|
for (auto& op_ptr: payload[ok ? "success" : "failure"].array_items())
|
||||||
|
{
|
||||||
|
auto& op = op_ptr.object_items();
|
||||||
|
if (op.find("request_range") != op.end())
|
||||||
|
{
|
||||||
|
json11::Json::array kvs;
|
||||||
|
auto req = op.at("request_range");
|
||||||
|
auto key = base64_decode(req["key"].string_value());
|
||||||
|
auto range_end = base64_decode(req["range_end"].string_value());
|
||||||
|
auto begin_it = range_end.empty() ? data.find(key) : data.lower_bound(key);
|
||||||
|
auto end_it = range_end.empty() ? (begin_it == data.end() ? begin_it : std::next(begin_it)) : data.lower_bound(range_end);
|
||||||
|
for (auto it = begin_it; it != end_it; it++)
|
||||||
|
{
|
||||||
|
printf("\\- get: %s = %s, rev %ju\n", it->first.c_str(), it->second.value.c_str(), it->second.mod_revision);
|
||||||
|
kvs.push_back(json11::Json::object {
|
||||||
|
{ "key", base64_encode(it->first) },
|
||||||
|
{ "value", base64_encode(it->second.value) },
|
||||||
|
{ "mod_revision", it->second.mod_revision },
|
||||||
|
});
|
||||||
|
}
|
||||||
|
responses.push_back(json11::Json::object {
|
||||||
|
{ "response_range", json11::Json::object{ { "header", json11::Json::object{ { "revision", mod_revision } } }, { "kvs", kvs } } },
|
||||||
|
});
|
||||||
|
}
|
||||||
|
else if (op.find("request_put") != op.end())
|
||||||
|
{
|
||||||
|
auto req = op.at("request_put");
|
||||||
|
auto key = base64_decode(req["key"].string_value());
|
||||||
|
auto value = base64_decode(req["value"].string_value());
|
||||||
|
auto lease_id = req["lease"].uint64_value();
|
||||||
|
printf("\\- put: %s = %s, rev %ju, lease %ju\n", key.c_str(), value.c_str(), mod_revision, lease_id);
|
||||||
|
data[key] = {
|
||||||
|
.value = value,
|
||||||
|
.mod_revision = mod_revision,
|
||||||
|
.lease_id = lease_id,
|
||||||
|
};
|
||||||
|
std::string err;
|
||||||
|
json11::Json json_value = json11::Json::parse(value, err);
|
||||||
|
if (err != "")
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Invalid JSON in etcd key %s during test: %s\n", key.c_str(), value.c_str());
|
||||||
|
exit(1);
|
||||||
|
}
|
||||||
|
changes[key] = { .key = key, .value = json_value, .mod_revision = mod_revision };
|
||||||
|
responses.push_back(json11::Json::object {
|
||||||
|
{ "response_put", json11::Json::object{ { "header", json11::Json::object{ { "revision", mod_revision } } } } },
|
||||||
|
});
|
||||||
|
}
|
||||||
|
else if (op.find("request_delete_range") != op.end())
|
||||||
|
{
|
||||||
|
auto req = op.at("request_delete_range");
|
||||||
|
auto key = base64_decode(req["key"].string_value());
|
||||||
|
auto range_end = base64_decode(req["range_end"].string_value());
|
||||||
|
uint64_t n_del = 0;
|
||||||
|
for (auto it = data.lower_bound(key); it != data.end() && (range_end == "" || it->first < range_end); )
|
||||||
|
{
|
||||||
|
auto & key = it->first;
|
||||||
|
printf("\\- del: %s\n", key.c_str());
|
||||||
|
changes[key] = { .key = key, .mod_revision = mod_revision };
|
||||||
|
n_del++;
|
||||||
|
data.erase(it++);
|
||||||
|
}
|
||||||
|
responses.push_back(json11::Json::object {
|
||||||
|
{ "response_delete_range", json11::Json::object{ { "header", json11::Json::object{ { "revision", mod_revision } } }, { "deleted", n_del } } },
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
callback("", json11::Json::object{
|
||||||
|
{ "header", json11::Json::object{ { "revision", mod_revision } } },
|
||||||
|
{ "succeeded", ok },
|
||||||
|
{ "responses", responses }
|
||||||
|
});
|
||||||
|
// Push changes to watcher
|
||||||
|
if (changes.size())
|
||||||
|
{
|
||||||
|
for (auto & kv: changes)
|
||||||
|
parse_state(kv.second);
|
||||||
|
if (on_change_hook != NULL)
|
||||||
|
on_change_hook(changes);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else if (api == "/lease/grant")
|
||||||
|
{
|
||||||
|
uint64_t lease_id = (((uint64_t)lrand48()) << 32) | lrand48();
|
||||||
|
leases[lease_id] = payload["TTL"].uint64_value();
|
||||||
|
callback("", json11::Json::object{ { "ID", std::to_string(lease_id) } });
|
||||||
|
}
|
||||||
|
else
|
||||||
|
callback("Unsupported", json11::Json());
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_mock_t::load_global_config()
|
||||||
|
{
|
||||||
|
etcd_state_client_t::load_global_config([this](const std::string & err) {});
|
||||||
|
}
|
||||||
|
|
||||||
|
void etcd_state_client_mock_t::load_pgs()
|
||||||
|
{
|
||||||
|
etcd_state_client_t::load_pgs([this](const std::string & err) {});
|
||||||
|
}
|
||||||
@@ -0,0 +1,42 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include "etcd_state_client.h"
|
||||||
|
|
||||||
|
struct etcd_mock_key_data_t
|
||||||
|
{
|
||||||
|
std::string value;
|
||||||
|
uint64_t mod_revision;
|
||||||
|
uint64_t lease_id;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct etcd_mock_request_t
|
||||||
|
{
|
||||||
|
std::string api;
|
||||||
|
json11::Json payload;
|
||||||
|
int timeout;
|
||||||
|
int retries;
|
||||||
|
int interval;
|
||||||
|
std::function<void(std::string, json11::Json)> callback;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct etcd_state_client_mock_t: public etcd_state_client_t
|
||||||
|
{
|
||||||
|
uint64_t mod_revision = 0;
|
||||||
|
bool paused = false;
|
||||||
|
std::vector<etcd_mock_request_t> queue;
|
||||||
|
public:
|
||||||
|
std::map<uint64_t, uint64_t> leases;
|
||||||
|
std::map<std::string, etcd_mock_key_data_t> data;
|
||||||
|
etcd_state_client_mock_t();
|
||||||
|
void set(const std::string& key, json11::Json data, uint64_t mod_revision = 0, uint64_t lease_id = 0);
|
||||||
|
void pause();
|
||||||
|
void resume();
|
||||||
|
void etcd_call_oneshot(std::string etcd_address, std::string api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback) override;
|
||||||
|
void etcd_call(std::string api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback) override;
|
||||||
|
void etcd_add_watch(json11::Json watch) override;
|
||||||
|
void load_global_config() override;
|
||||||
|
void load_pgs() override;
|
||||||
|
};
|
||||||
+70
-168
@@ -15,106 +15,6 @@
|
|||||||
#include "msgr_rdma.h"
|
#include "msgr_rdma.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#include <sys/poll.h>
|
|
||||||
|
|
||||||
msgr_iothread_t::msgr_iothread_t():
|
|
||||||
ring(RINGLOOP_DEFAULT_SIZE, true),
|
|
||||||
thread(&msgr_iothread_t::run, this)
|
|
||||||
{
|
|
||||||
eventfd = ring.register_eventfd();
|
|
||||||
if (eventfd < 0)
|
|
||||||
{
|
|
||||||
throw std::runtime_error(std::string("failed to register eventfd: ") + strerror(-eventfd));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
msgr_iothread_t::~msgr_iothread_t()
|
|
||||||
{
|
|
||||||
stop();
|
|
||||||
}
|
|
||||||
|
|
||||||
void msgr_iothread_t::add_sqe(io_uring_sqe & sqe)
|
|
||||||
{
|
|
||||||
mu.lock();
|
|
||||||
queue.push_back((iothread_sqe_t){ .sqe = sqe, .data = std::move(*(ring_data_t*)sqe.user_data) });
|
|
||||||
if (queue.size() == 1)
|
|
||||||
{
|
|
||||||
cond.notify_all();
|
|
||||||
}
|
|
||||||
mu.unlock();
|
|
||||||
}
|
|
||||||
|
|
||||||
void msgr_iothread_t::stop()
|
|
||||||
{
|
|
||||||
mu.lock();
|
|
||||||
if (stopped)
|
|
||||||
{
|
|
||||||
mu.unlock();
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
stopped = true;
|
|
||||||
if (outer_loop_data)
|
|
||||||
{
|
|
||||||
outer_loop_data->callback = [](ring_data_t*){};
|
|
||||||
}
|
|
||||||
cond.notify_all();
|
|
||||||
close(eventfd);
|
|
||||||
mu.unlock();
|
|
||||||
thread.join();
|
|
||||||
}
|
|
||||||
|
|
||||||
void msgr_iothread_t::add_to_ringloop(ring_loop_t *outer_loop)
|
|
||||||
{
|
|
||||||
assert(!this->outer_loop || this->outer_loop == outer_loop);
|
|
||||||
io_uring_sqe *sqe = outer_loop->get_sqe();
|
|
||||||
assert(sqe != NULL);
|
|
||||||
this->outer_loop = outer_loop;
|
|
||||||
this->outer_loop_data = ((ring_data_t*)sqe->user_data);
|
|
||||||
io_uring_prep_poll_add(sqe, eventfd, POLLIN);
|
|
||||||
outer_loop_data->callback = [this](ring_data_t *data)
|
|
||||||
{
|
|
||||||
if (data->res < 0)
|
|
||||||
{
|
|
||||||
throw std::runtime_error(std::string("eventfd poll failed: ") + strerror(-data->res));
|
|
||||||
}
|
|
||||||
outer_loop_data = NULL;
|
|
||||||
if (stopped)
|
|
||||||
{
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
add_to_ringloop(this->outer_loop);
|
|
||||||
ring.loop();
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
void msgr_iothread_t::run()
|
|
||||||
{
|
|
||||||
while (true)
|
|
||||||
{
|
|
||||||
{
|
|
||||||
std::unique_lock<std::mutex> lk(mu);
|
|
||||||
while (!stopped && !queue.size())
|
|
||||||
cond.wait(lk);
|
|
||||||
if (stopped)
|
|
||||||
return;
|
|
||||||
int i = 0;
|
|
||||||
for (; i < queue.size(); i++)
|
|
||||||
{
|
|
||||||
io_uring_sqe *sqe = ring.get_sqe();
|
|
||||||
if (!sqe)
|
|
||||||
break;
|
|
||||||
ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
|
||||||
*data = std::move(queue[i].data);
|
|
||||||
*sqe = queue[i].sqe;
|
|
||||||
sqe->user_data = (uint64_t)data;
|
|
||||||
}
|
|
||||||
queue.erase(queue.begin(), queue.begin()+i);
|
|
||||||
}
|
|
||||||
// We only want to offload sendmsg/recvmsg. Callbacks will be called in main thread
|
|
||||||
ring.submit();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::init()
|
void osd_messenger_t::init()
|
||||||
{
|
{
|
||||||
#ifdef WITH_RDMACM
|
#ifdef WITH_RDMACM
|
||||||
@@ -173,21 +73,17 @@ void osd_messenger_t::init()
|
|||||||
}
|
}
|
||||||
if (ringloop && iothread_count > 0)
|
if (ringloop && iothread_count > 0)
|
||||||
{
|
{
|
||||||
for (int i = 0; i < iothread_count; i++)
|
init_iothreads();
|
||||||
{
|
|
||||||
auto iot = new msgr_iothread_t();
|
|
||||||
iothreads.push_back(iot);
|
|
||||||
iot->add_to_ringloop(ringloop);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
keepalive_timer_id = tfd->set_timer(1000, true, [this](int)
|
keepalive_timer_id = tfd->set_timer(1000, true, [this](int)
|
||||||
{
|
{
|
||||||
|
std::vector<uint64_t> clients_to_stop;
|
||||||
|
std::vector<osd_op_t*> ops_to_send;
|
||||||
auto cl_it = clients.begin();
|
auto cl_it = clients.begin();
|
||||||
while (cl_it != clients.end())
|
while (cl_it != clients.end())
|
||||||
{
|
{
|
||||||
auto cl = cl_it->second;
|
auto cl = cl_it->second;
|
||||||
cl_it++;
|
cl_it++;
|
||||||
auto peer_fd = cl->peer_fd;
|
|
||||||
if (!cl->osd_num && !cl->in_osd_num || cl->peer_state != PEER_CONNECTED && cl->peer_state != PEER_RDMA)
|
if (!cl->osd_num && !cl->in_osd_num || cl->peer_state != PEER_CONNECTED && cl->peer_state != PEER_RDMA)
|
||||||
{
|
{
|
||||||
// Do not run keepalive on regular clients
|
// Do not run keepalive on regular clients
|
||||||
@@ -199,10 +95,9 @@ void osd_messenger_t::init()
|
|||||||
if (!cl->ping_time_remaining)
|
if (!cl->ping_time_remaining)
|
||||||
{
|
{
|
||||||
// Ping timed out, stop the client
|
// Ping timed out, stop the client
|
||||||
fprintf(stderr, "Ping timed out for OSD %ju (client %d), disconnecting peer\n", cl->in_osd_num ? cl->in_osd_num : cl->osd_num, cl->peer_fd);
|
fprintf(stderr, "Ping timed out for OSD %ju (client %ju), disconnecting peer\n",
|
||||||
stop_client(peer_fd, true);
|
cl->in_osd_num ? cl->in_osd_num : cl->osd_num, cl->client_id);
|
||||||
// Restart iterator because it may be invalidated
|
clients_to_stop.push_back(cl->client_id);
|
||||||
cl_it = clients.upper_bound(peer_fd);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
else if (cl->idle_time_remaining > 0)
|
else if (cl->idle_time_remaining > 0)
|
||||||
@@ -213,37 +108,36 @@ void osd_messenger_t::init()
|
|||||||
// Connection is idle for <osd_idle_time>, send ping
|
// Connection is idle for <osd_idle_time>, send ping
|
||||||
osd_op_t *op = new osd_op_t();
|
osd_op_t *op = new osd_op_t();
|
||||||
op->op_type = OSD_OP_OUT;
|
op->op_type = OSD_OP_OUT;
|
||||||
op->peer_fd = cl->peer_fd;
|
op->client_id = cl->client_id;
|
||||||
op->req = (osd_any_op_t){
|
op->req = (osd_any_op_t){
|
||||||
.hdr = {
|
.hdr = {
|
||||||
.magic = SECONDARY_OSD_OP_MAGIC,
|
.magic = SECONDARY_OSD_OP_MAGIC,
|
||||||
.opcode = OSD_OP_PING,
|
.opcode = OSD_OP_PING,
|
||||||
},
|
},
|
||||||
};
|
};
|
||||||
op->callback = [this, cl](osd_op_t *op)
|
op->callback = [this](osd_op_t *op)
|
||||||
{
|
{
|
||||||
auto cl_it = clients.find(op->peer_fd);
|
auto cl_it = clients.find(op->client_id);
|
||||||
if (cl_it == clients.end() || cl_it->second != cl)
|
if (cl_it == clients.end())
|
||||||
{
|
{
|
||||||
// client is already dropped
|
// client is already dropped
|
||||||
delete op;
|
delete op;
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
int fail_fd = (op->reply.hdr.retval != 0 ? op->peer_fd : -1);
|
auto cl = cl_it->second;
|
||||||
|
uint64_t fail_client_id = (op->reply.hdr.retval != 0 ? op->client_id : 0);
|
||||||
auto fail_osd_num = cl->in_osd_num ? cl->in_osd_num : cl->osd_num;
|
auto fail_osd_num = cl->in_osd_num ? cl->in_osd_num : cl->osd_num;
|
||||||
cl->ping_time_remaining = 0;
|
cl->ping_time_remaining = 0;
|
||||||
delete op;
|
delete op;
|
||||||
if (fail_fd >= 0)
|
if (fail_client_id)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Ping failed for OSD %ju (client %d), disconnecting peer\n", fail_osd_num, fail_fd);
|
fprintf(stderr, "Ping failed for OSD %ju (client %ju), disconnecting peer\n", fail_osd_num, fail_client_id);
|
||||||
stop_client(fail_fd, true);
|
stop_client(fail_client_id);
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
cl->ping_time_remaining = osd_ping_timeout;
|
cl->ping_time_remaining = osd_ping_timeout;
|
||||||
cl->idle_time_remaining = osd_idle_timeout;
|
cl->idle_time_remaining = osd_idle_timeout;
|
||||||
outbox_push(op);
|
ops_to_send.push_back(op);
|
||||||
// Restart iterator because it may be invalidated
|
|
||||||
cl_it = clients.upper_bound(peer_fd);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
@@ -251,6 +145,14 @@ void osd_messenger_t::init()
|
|||||||
cl->idle_time_remaining = osd_idle_timeout;
|
cl->idle_time_remaining = osd_idle_timeout;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
for (uint64_t client_id: clients_to_stop)
|
||||||
|
{
|
||||||
|
stop_client(client_id);
|
||||||
|
}
|
||||||
|
for (osd_op_t *op: ops_to_send)
|
||||||
|
{
|
||||||
|
outbox_push(op);
|
||||||
|
}
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -263,16 +165,9 @@ osd_messenger_t::~osd_messenger_t()
|
|||||||
}
|
}
|
||||||
while (clients.size() > 0)
|
while (clients.size() > 0)
|
||||||
{
|
{
|
||||||
stop_client(clients.begin()->first, true, true);
|
stop_client(clients.begin()->first, true);
|
||||||
}
|
|
||||||
if (iothreads.size())
|
|
||||||
{
|
|
||||||
for (auto iot: iothreads)
|
|
||||||
{
|
|
||||||
delete iot;
|
|
||||||
}
|
|
||||||
iothreads.clear();
|
|
||||||
}
|
}
|
||||||
|
destroy_iothreads();
|
||||||
#ifdef WITH_RDMA
|
#ifdef WITH_RDMA
|
||||||
for (auto rdma_context: rdma_contexts)
|
for (auto rdma_context: rdma_contexts)
|
||||||
{
|
{
|
||||||
@@ -440,7 +335,7 @@ void osd_messenger_t::try_connect_peer(uint64_t peer_osd)
|
|||||||
{
|
{
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if (osd_peer_fds.find(peer_osd) != osd_peer_fds.end())
|
if (osd_peers.find(peer_osd) != osd_peers.end())
|
||||||
{
|
{
|
||||||
wanted_peers.erase(peer_osd);
|
wanted_peers.erase(peer_osd);
|
||||||
return;
|
return;
|
||||||
@@ -467,20 +362,20 @@ void osd_messenger_t::try_connect_peer_tcp(osd_num_t peer_osd, const char *peer_
|
|||||||
#ifdef WITH_RDMACM
|
#ifdef WITH_RDMACM
|
||||||
if (disable_tcp)
|
if (disable_tcp)
|
||||||
{
|
{
|
||||||
on_connect_peer(peer_osd, -EINVAL);
|
on_connect_peer(peer_osd, -EINVAL, 0);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
struct sockaddr_storage addr;
|
struct sockaddr_storage addr;
|
||||||
if (!string_to_addr(peer_host, 0, peer_port, &addr))
|
if (!string_to_addr(peer_host, 0, peer_port, &addr))
|
||||||
{
|
{
|
||||||
on_connect_peer(peer_osd, -EINVAL);
|
on_connect_peer(peer_osd, -EINVAL, 0);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
int peer_fd = socket(addr.ss_family, SOCK_STREAM, 0);
|
int peer_fd = socket(addr.ss_family, SOCK_STREAM, 0);
|
||||||
if (peer_fd < 0)
|
if (peer_fd < 0)
|
||||||
{
|
{
|
||||||
on_connect_peer(peer_osd, -errno);
|
on_connect_peer(peer_osd, -errno, 0);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
fcntl(peer_fd, F_SETFL, fcntl(peer_fd, F_GETFL, 0) | O_NONBLOCK);
|
fcntl(peer_fd, F_SETFL, fcntl(peer_fd, F_GETFL, 0) | O_NONBLOCK);
|
||||||
@@ -488,21 +383,25 @@ void osd_messenger_t::try_connect_peer_tcp(osd_num_t peer_osd, const char *peer_
|
|||||||
if (r < 0 && errno != EINPROGRESS)
|
if (r < 0 && errno != EINPROGRESS)
|
||||||
{
|
{
|
||||||
close(peer_fd);
|
close(peer_fd);
|
||||||
on_connect_peer(peer_osd, -errno);
|
on_connect_peer(peer_osd, -errno, 0);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
clients[peer_fd] = new osd_client_t();
|
const uint64_t client_id = next_client_id++;
|
||||||
|
osd_client_t *cl = new osd_client_t();
|
||||||
if (log_level > 0)
|
if (log_level > 0)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Connecting to OSD %ju at %s:%d (client %d)\n", peer_osd, peer_host, peer_port, peer_fd);
|
fprintf(stderr, "Connecting to OSD %ju at %s:%d (client %ju, FD %d)\n", peer_osd, peer_host, peer_port, client_id, peer_fd);
|
||||||
}
|
}
|
||||||
clients[peer_fd]->peer_addr = addr;
|
cl->client_id = client_id;
|
||||||
clients[peer_fd]->peer_port = peer_port;
|
cl->peer_addr = addr;
|
||||||
clients[peer_fd]->peer_fd = peer_fd;
|
cl->peer_port = peer_port;
|
||||||
clients[peer_fd]->peer_state = PEER_CONNECTING;
|
cl->peer_fd = peer_fd;
|
||||||
clients[peer_fd]->connect_timeout_id = -1;
|
cl->peer_state = PEER_CONNECTING;
|
||||||
clients[peer_fd]->osd_num = peer_osd;
|
cl->connect_timeout_id = -1;
|
||||||
clients[peer_fd]->in_buf = malloc_or_die(receive_buffer_size);
|
cl->osd_num = peer_osd;
|
||||||
|
cl->in_buf = malloc_or_die(receive_buffer_size);
|
||||||
|
clients[client_id] = cl;
|
||||||
|
clients_by_fd[peer_fd] = cl;
|
||||||
tfd->set_fd_handler(peer_fd, true, [this](int peer_fd, int epoll_events)
|
tfd->set_fd_handler(peer_fd, true, [this](int peer_fd, int epoll_events)
|
||||||
{
|
{
|
||||||
// Either OUT (connected) or HUP
|
// Either OUT (connected) or HUP
|
||||||
@@ -510,11 +409,11 @@ void osd_messenger_t::try_connect_peer_tcp(osd_num_t peer_osd, const char *peer_
|
|||||||
});
|
});
|
||||||
if (peer_connect_timeout > 0)
|
if (peer_connect_timeout > 0)
|
||||||
{
|
{
|
||||||
clients[peer_fd]->connect_timeout_id = tfd->set_timer(1000*peer_connect_timeout, false, [this, peer_fd](int timer_id)
|
cl->connect_timeout_id = tfd->set_timer(1000*peer_connect_timeout, false, [this, client_id](int timer_id)
|
||||||
{
|
{
|
||||||
osd_num_t peer_osd = clients.at(peer_fd)->osd_num;
|
osd_num_t peer_osd = clients.at(client_id)->osd_num;
|
||||||
stop_client(peer_fd, true);
|
stop_client(client_id);
|
||||||
on_connect_peer(peer_osd, -EPIPE);
|
on_connect_peer(peer_osd, -EPIPE, 0);
|
||||||
return;
|
return;
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
@@ -522,7 +421,7 @@ void osd_messenger_t::try_connect_peer_tcp(osd_num_t peer_osd, const char *peer_
|
|||||||
|
|
||||||
void osd_messenger_t::handle_connect_epoll(int peer_fd)
|
void osd_messenger_t::handle_connect_epoll(int peer_fd)
|
||||||
{
|
{
|
||||||
auto cl = clients[peer_fd];
|
auto cl = clients_by_fd.at(peer_fd);
|
||||||
if (cl->connect_timeout_id >= 0)
|
if (cl->connect_timeout_id >= 0)
|
||||||
{
|
{
|
||||||
tfd->clear_timer(cl->connect_timeout_id);
|
tfd->clear_timer(cl->connect_timeout_id);
|
||||||
@@ -537,8 +436,8 @@ void osd_messenger_t::handle_connect_epoll(int peer_fd)
|
|||||||
}
|
}
|
||||||
if (result != 0)
|
if (result != 0)
|
||||||
{
|
{
|
||||||
stop_client(peer_fd, true);
|
stop_client(cl->client_id);
|
||||||
on_connect_peer(peer_osd, -result);
|
on_connect_peer(peer_osd, -result, 0);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
int one = 1;
|
int one = 1;
|
||||||
@@ -555,23 +454,23 @@ void osd_messenger_t::handle_connect_epoll(int peer_fd)
|
|||||||
void osd_messenger_t::handle_peer_epoll(int peer_fd, int epoll_events)
|
void osd_messenger_t::handle_peer_epoll(int peer_fd, int epoll_events)
|
||||||
{
|
{
|
||||||
// Mark client as ready (i.e. some data is available)
|
// Mark client as ready (i.e. some data is available)
|
||||||
|
auto cl = clients_by_fd.at(peer_fd);
|
||||||
if (epoll_events & EPOLLRDHUP)
|
if (epoll_events & EPOLLRDHUP)
|
||||||
{
|
{
|
||||||
// Stop client
|
// Stop client
|
||||||
if (log_level > 0)
|
if (log_level > 0)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "[OSD %ju] client %d disconnected\n", this->osd_num, peer_fd);
|
fprintf(stderr, "[OSD %ju] client %ju disconnected\n", this->osd_num, cl->client_id);
|
||||||
}
|
}
|
||||||
stop_client(peer_fd, true);
|
stop_client(cl->client_id);
|
||||||
}
|
}
|
||||||
else if (epoll_events & EPOLLIN)
|
else if (epoll_events & EPOLLIN)
|
||||||
{
|
{
|
||||||
// Mark client as ready (i.e. some data is available)
|
// Mark client as ready (i.e. some data is available)
|
||||||
auto cl = clients[peer_fd];
|
|
||||||
cl->read_ready++;
|
cl->read_ready++;
|
||||||
if (cl->read_ready == 1)
|
if (cl->read_ready == 1)
|
||||||
{
|
{
|
||||||
read_ready_clients.push_back(cl->peer_fd);
|
read_ready_clients.push_back(cl->client_id);
|
||||||
if (ringloop)
|
if (ringloop)
|
||||||
ringloop->wakeup();
|
ringloop->wakeup();
|
||||||
else
|
else
|
||||||
@@ -580,13 +479,13 @@ void osd_messenger_t::handle_peer_epoll(int peer_fd, int epoll_events)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void osd_messenger_t::on_connect_peer(osd_num_t peer_osd, int peer_fd)
|
void osd_messenger_t::on_connect_peer(osd_num_t peer_osd, int errcode, uint64_t client_id)
|
||||||
{
|
{
|
||||||
auto & wp = wanted_peers.at(peer_osd);
|
auto & wp = wanted_peers.at(peer_osd);
|
||||||
wp.connecting = false;
|
wp.connecting = false;
|
||||||
if (peer_fd < 0)
|
if (errcode < 0)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Failed to connect to peer OSD %ju address %s port %d: %s\n", peer_osd, wp.cur_addr.c_str(), wp.cur_port, strerror(-peer_fd));
|
fprintf(stderr, "Failed to connect to peer OSD %ju address %s port %d: %s\n", peer_osd, wp.cur_addr.c_str(), wp.cur_port, strerror(-errcode));
|
||||||
if (wp.address_changed)
|
if (wp.address_changed)
|
||||||
{
|
{
|
||||||
wp.address_changed = false;
|
wp.address_changed = false;
|
||||||
@@ -613,7 +512,7 @@ void osd_messenger_t::on_connect_peer(osd_num_t peer_osd, int peer_fd)
|
|||||||
}
|
}
|
||||||
if (log_level > 0)
|
if (log_level > 0)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "[OSD %ju] Connected with peer OSD %ju (client %d)\n", osd_num, peer_osd, peer_fd);
|
fprintf(stderr, "[OSD %ju] Connected with peer OSD %ju (client %ju)\n", osd_num, peer_osd, client_id);
|
||||||
}
|
}
|
||||||
wanted_peers.erase(peer_osd);
|
wanted_peers.erase(peer_osd);
|
||||||
repeer_pgs(peer_osd);
|
repeer_pgs(peer_osd);
|
||||||
@@ -623,7 +522,7 @@ void osd_messenger_t::check_peer_config(osd_client_t *cl)
|
|||||||
{
|
{
|
||||||
osd_op_t *op = new osd_op_t();
|
osd_op_t *op = new osd_op_t();
|
||||||
op->op_type = OSD_OP_OUT;
|
op->op_type = OSD_OP_OUT;
|
||||||
op->peer_fd = cl->peer_fd;
|
op->client_id = cl->client_id;
|
||||||
op->req = (osd_any_op_t){
|
op->req = (osd_any_op_t){
|
||||||
.show_conf = {
|
.show_conf = {
|
||||||
.header = {
|
.header = {
|
||||||
@@ -647,7 +546,7 @@ void osd_messenger_t::check_peer_config(osd_client_t *cl)
|
|||||||
if (!selected_ctx)
|
if (!selected_ctx)
|
||||||
{
|
{
|
||||||
if (log_level > 0)
|
if (log_level > 0)
|
||||||
fprintf(stderr, "No RDMA context for OSD %ju connection (peer %d), using only TCP\n", cl->osd_num, cl->peer_fd);
|
fprintf(stderr, "No RDMA context for OSD %ju connection (client %ju), using only TCP\n", cl->osd_num, cl->client_id);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -708,8 +607,8 @@ void osd_messenger_t::check_peer_config(osd_client_t *cl)
|
|||||||
if (err)
|
if (err)
|
||||||
{
|
{
|
||||||
osd_num_t peer_osd = cl->osd_num;
|
osd_num_t peer_osd = cl->osd_num;
|
||||||
stop_client(op->peer_fd);
|
stop_client(op->client_id);
|
||||||
on_connect_peer(peer_osd, -EINVAL);
|
on_connect_peer(peer_osd, -EINVAL, 0);
|
||||||
delete op;
|
delete op;
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
@@ -744,8 +643,8 @@ void osd_messenger_t::check_peer_config(osd_client_t *cl)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
osd_peer_fds[cl->osd_num] = cl->peer_fd;
|
osd_peers[cl->osd_num] = cl;
|
||||||
on_connect_peer(cl->osd_num, cl->peer_fd);
|
on_connect_peer(cl->osd_num, 0, cl->client_id);
|
||||||
delete op;
|
delete op;
|
||||||
};
|
};
|
||||||
outbox_push(op);
|
outbox_push(op);
|
||||||
@@ -760,13 +659,16 @@ void osd_messenger_t::accept_connections(int listen_fd)
|
|||||||
while ((peer_fd = accept(listen_fd, (sockaddr*)&addr, &peer_addr_size)) >= 0)
|
while ((peer_fd = accept(listen_fd, (sockaddr*)&addr, &peer_addr_size)) >= 0)
|
||||||
{
|
{
|
||||||
assert(peer_fd != 0);
|
assert(peer_fd != 0);
|
||||||
fprintf(stderr, "[OSD %ju] new client %d: connection from %s\n", this->osd_num, peer_fd,
|
const uint64_t client_id = next_client_id++;
|
||||||
|
fprintf(stderr, "[OSD %ju] new client %ju (FD %d): connection from %s\n", this->osd_num, client_id, peer_fd,
|
||||||
addr_to_string(addr).c_str());
|
addr_to_string(addr).c_str());
|
||||||
fcntl(peer_fd, F_SETFL, fcntl(peer_fd, F_GETFL, 0) | O_NONBLOCK);
|
fcntl(peer_fd, F_SETFL, fcntl(peer_fd, F_GETFL, 0) | O_NONBLOCK);
|
||||||
int one = 1;
|
int one = 1;
|
||||||
setsockopt(peer_fd, SOL_TCP, TCP_NODELAY, &one, sizeof(one));
|
setsockopt(peer_fd, SOL_TCP, TCP_NODELAY, &one, sizeof(one));
|
||||||
auto cl = new osd_client_t();
|
auto cl = new osd_client_t();
|
||||||
clients[peer_fd] = cl;
|
cl->client_id = client_id;
|
||||||
|
clients[cl->client_id] = cl;
|
||||||
|
clients_by_fd[peer_fd] = cl;
|
||||||
cl->is_incoming = true;
|
cl->is_incoming = true;
|
||||||
cl->peer_addr = addr;
|
cl->peer_addr = addr;
|
||||||
cl->peer_addr = addr;
|
cl->peer_addr = addr;
|
||||||
|
|||||||
+28
-52
@@ -12,6 +12,7 @@
|
|||||||
#include <deque>
|
#include <deque>
|
||||||
#include <vector>
|
#include <vector>
|
||||||
|
|
||||||
|
#include "../util/robin_hood.h"
|
||||||
#include "malloc_or_die.h"
|
#include "malloc_or_die.h"
|
||||||
#include "json11/json11.hpp"
|
#include "json11/json11.hpp"
|
||||||
#include "msgr_op.h"
|
#include "msgr_op.h"
|
||||||
@@ -37,6 +38,8 @@
|
|||||||
#define MSGR_SENDP_HDR 1
|
#define MSGR_SENDP_HDR 1
|
||||||
#define MSGR_SENDP_FREE 2
|
#define MSGR_SENDP_FREE 2
|
||||||
|
|
||||||
|
#define MAX_SIMPLE_PAYLOAD_SIZE 1048576
|
||||||
|
|
||||||
struct msgr_sendp_t
|
struct msgr_sendp_t
|
||||||
{
|
{
|
||||||
osd_op_t *op;
|
osd_op_t *op;
|
||||||
@@ -50,6 +53,7 @@ struct msgr_rdma_context_t;
|
|||||||
|
|
||||||
struct osd_client_t
|
struct osd_client_t
|
||||||
{
|
{
|
||||||
|
uint64_t client_id = 0;
|
||||||
int refs = 0;
|
int refs = 0;
|
||||||
|
|
||||||
sockaddr_storage peer_addr = {};
|
sockaddr_storage peer_addr = {};
|
||||||
@@ -85,7 +89,7 @@ struct osd_client_t
|
|||||||
std::vector<osd_op_t*> received_ops;
|
std::vector<osd_op_t*> received_ops;
|
||||||
|
|
||||||
// Outbound operations
|
// Outbound operations
|
||||||
std::map<uint64_t, osd_op_t*> sent_ops;
|
robin_hood::unordered_flat_map<uint64_t, osd_op_t*> sent_ops;
|
||||||
uint64_t send_op_id = 0;
|
uint64_t send_op_id = 0;
|
||||||
|
|
||||||
// PGs dirtied by this client's primary-writes
|
// PGs dirtied by this client's primary-writes
|
||||||
@@ -127,43 +131,7 @@ struct osd_op_stats_t
|
|||||||
uint64_t subop_stat_count[OSD_OP_MAX+1] = { 0 };
|
uint64_t subop_stat_count[OSD_OP_MAX+1] = { 0 };
|
||||||
};
|
};
|
||||||
|
|
||||||
#include <mutex>
|
|
||||||
#include <condition_variable>
|
|
||||||
#include <thread>
|
|
||||||
|
|
||||||
#ifdef __MOCK__
|
|
||||||
class msgr_iothread_t;
|
class msgr_iothread_t;
|
||||||
#else
|
|
||||||
struct iothread_sqe_t
|
|
||||||
{
|
|
||||||
io_uring_sqe sqe;
|
|
||||||
ring_data_t data;
|
|
||||||
};
|
|
||||||
|
|
||||||
class msgr_iothread_t
|
|
||||||
{
|
|
||||||
protected:
|
|
||||||
ring_loop_t ring;
|
|
||||||
ring_loop_t *outer_loop = NULL;
|
|
||||||
ring_data_t *outer_loop_data = NULL;
|
|
||||||
int eventfd = -1;
|
|
||||||
bool stopped = false;
|
|
||||||
std::mutex mu;
|
|
||||||
std::condition_variable cond;
|
|
||||||
std::vector<iothread_sqe_t> queue;
|
|
||||||
std::thread thread;
|
|
||||||
|
|
||||||
void run();
|
|
||||||
public:
|
|
||||||
|
|
||||||
msgr_iothread_t();
|
|
||||||
~msgr_iothread_t();
|
|
||||||
|
|
||||||
void add_sqe(io_uring_sqe & sqe);
|
|
||||||
void stop();
|
|
||||||
void add_to_ringloop(ring_loop_t *outer_loop);
|
|
||||||
};
|
|
||||||
#endif
|
|
||||||
|
|
||||||
#ifdef WITH_RDMA
|
#ifdef WITH_RDMA
|
||||||
struct rdma_event_channel;
|
struct rdma_event_channel;
|
||||||
@@ -201,25 +169,30 @@ protected:
|
|||||||
uint64_t rdma_max_sge = 0, rdma_max_send = 0, rdma_max_recv = 0;
|
uint64_t rdma_max_sge = 0, rdma_max_send = 0, rdma_max_recv = 0;
|
||||||
uint64_t rdma_max_msg = 0;
|
uint64_t rdma_max_msg = 0;
|
||||||
rdma_event_channel *rdmacm_evch = NULL;
|
rdma_event_channel *rdmacm_evch = NULL;
|
||||||
std::map<rdma_cm_id*, osd_client_t*> rdmacm_connections;
|
robin_hood::unordered_flat_map<rdma_cm_id*, osd_client_t*> rdmacm_connections;
|
||||||
std::map<rdma_cm_id*, rdmacm_connecting_t*> rdmacm_connecting;
|
robin_hood::unordered_flat_map<rdma_cm_id*, rdmacm_connecting_t*> rdmacm_connecting;
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
std::vector<msgr_iothread_t*> iothreads;
|
std::vector<msgr_iothread_t*> iothreads;
|
||||||
std::vector<int> read_ready_clients;
|
std::vector<uint64_t> read_ready_clients;
|
||||||
std::vector<int> write_ready_clients;
|
std::vector<uint64_t> write_ready_clients;
|
||||||
// We don't use ringloop->set_immediate here because we may have no ringloop in client :)
|
// We don't use ringloop->set_immediate here because we may have no ringloop in client :)
|
||||||
std::deque<osd_op_t*> set_immediate_ops;
|
std::deque<osd_op_t*> set_immediate_ops;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
timerfd_manager_t *tfd = NULL;
|
timerfd_manager_t *tfd = NULL;
|
||||||
ring_loop_t *ringloop = NULL;
|
ring_loop_i *ringloop = NULL;
|
||||||
bool has_sendmsg_zc = false;
|
bool has_sendmsg_zc = false;
|
||||||
// osd_num_t is only for logging and asserts
|
uint64_t next_client_id = 1;
|
||||||
osd_num_t osd_num;
|
// osd_num = 0 for client messenger, osd_num > 0 for OSD messenger
|
||||||
std::map<int, osd_client_t*> clients;
|
osd_num_t osd_num = 0;
|
||||||
std::map<osd_num_t, osd_wanted_peer_t> wanted_peers;
|
uint32_t clean_entry_bitmap_size = 0;
|
||||||
std::map<uint64_t, int> osd_peer_fds;
|
uint32_t bs_block_size = 0;
|
||||||
|
uint32_t max_write_request_size = 0;
|
||||||
|
robin_hood::unordered_flat_map<uint64_t, osd_client_t*> clients;
|
||||||
|
robin_hood::unordered_flat_map<uint64_t, osd_client_t*> osd_peers;
|
||||||
|
robin_hood::unordered_flat_map<int, osd_client_t*> clients_by_fd;
|
||||||
|
robin_hood::unordered_flat_map<osd_num_t, osd_wanted_peer_t> wanted_peers;
|
||||||
std::vector<std::string> osd_networks;
|
std::vector<std::string> osd_networks;
|
||||||
std::vector<addr_mask_t> osd_network_masks;
|
std::vector<addr_mask_t> osd_network_masks;
|
||||||
std::vector<std::string> osd_cluster_networks;
|
std::vector<std::string> osd_cluster_networks;
|
||||||
@@ -230,9 +203,11 @@ public:
|
|||||||
osd_op_stats_t stats, recovery_stats;
|
osd_op_stats_t stats, recovery_stats;
|
||||||
|
|
||||||
void init();
|
void init();
|
||||||
|
void init_iothreads();
|
||||||
void parse_config(const json11::Json & config);
|
void parse_config(const json11::Json & config);
|
||||||
void connect_peer(uint64_t osd_num, json11::Json peer_state);
|
void connect_peer(uint64_t osd_num, json11::Json peer_state);
|
||||||
void stop_client(int peer_fd, bool force = false, bool force_delete = false);
|
void stop_client(uint64_t client_id, bool force_delete = false);
|
||||||
|
void destroy_client(osd_client_t *cl);
|
||||||
void outbox_push(osd_op_t *cur_op);
|
void outbox_push(osd_op_t *cur_op);
|
||||||
std::function<void(osd_op_t*)> exec_op;
|
std::function<void(osd_op_t*)> exec_op;
|
||||||
std::function<void(osd_num_t)> repeer_pgs;
|
std::function<void(osd_num_t)> repeer_pgs;
|
||||||
@@ -241,6 +216,7 @@ public:
|
|||||||
void read_requests();
|
void read_requests();
|
||||||
void send_replies();
|
void send_replies();
|
||||||
void accept_connections(int listen_fd);
|
void accept_connections(int listen_fd);
|
||||||
|
void destroy_iothreads();
|
||||||
~osd_messenger_t();
|
~osd_messenger_t();
|
||||||
|
|
||||||
static json11::Json::object read_config(const json11::Json & config);
|
static json11::Json::object read_config(const json11::Json & config);
|
||||||
@@ -251,7 +227,7 @@ public:
|
|||||||
|
|
||||||
#ifdef WITH_RDMA
|
#ifdef WITH_RDMA
|
||||||
bool is_rdma_enabled();
|
bool is_rdma_enabled();
|
||||||
bool connect_rdma(int peer_fd, std::string rdma_address, uint64_t client_max_msg);
|
json11::Json connect_rdma(uint64_t client_id, std::string rdma_address, uint64_t client_max_msg);
|
||||||
#endif
|
#endif
|
||||||
#ifdef WITH_RDMACM
|
#ifdef WITH_RDMACM
|
||||||
bool is_use_rdmacm();
|
bool is_use_rdmacm();
|
||||||
@@ -267,7 +243,7 @@ protected:
|
|||||||
void try_connect_peer_tcp(osd_num_t peer_osd, const char *peer_host, int peer_port);
|
void try_connect_peer_tcp(osd_num_t peer_osd, const char *peer_host, int peer_port);
|
||||||
void handle_peer_epoll(int peer_fd, int epoll_events);
|
void handle_peer_epoll(int peer_fd, int epoll_events);
|
||||||
void handle_connect_epoll(int peer_fd);
|
void handle_connect_epoll(int peer_fd);
|
||||||
void on_connect_peer(osd_num_t peer_osd, int peer_fd);
|
void on_connect_peer(osd_num_t peer_osd, int errcode, uint64_t client_id);
|
||||||
void check_peer_config(osd_client_t *cl);
|
void check_peer_config(osd_client_t *cl);
|
||||||
void cancel_osd_ops(osd_client_t *cl);
|
void cancel_osd_ops(osd_client_t *cl);
|
||||||
void cancel_op(osd_op_t *op);
|
void cancel_op(osd_op_t *op);
|
||||||
@@ -278,17 +254,17 @@ protected:
|
|||||||
bool handle_read(int result, osd_client_t *cl);
|
bool handle_read(int result, osd_client_t *cl);
|
||||||
bool handle_read_buffer(osd_client_t *cl, void *curbuf, int remain);
|
bool handle_read_buffer(osd_client_t *cl, void *curbuf, int remain);
|
||||||
bool handle_finished_read(osd_client_t *cl);
|
bool handle_finished_read(osd_client_t *cl);
|
||||||
void handle_op_hdr(osd_client_t *cl);
|
bool handle_op_hdr(osd_client_t *cl);
|
||||||
bool handle_reply_hdr(osd_client_t *cl);
|
bool handle_reply_hdr(osd_client_t *cl);
|
||||||
void handle_reply_ready(osd_op_t *op);
|
void handle_reply_ready(osd_op_t *op);
|
||||||
void handle_immediate_ops();
|
void handle_immediate_ops();
|
||||||
void clear_immediate_ops(int peer_fd);
|
|
||||||
|
|
||||||
#ifdef WITH_RDMA
|
#ifdef WITH_RDMA
|
||||||
void try_send_rdma(osd_client_t *cl);
|
void try_send_rdma(osd_client_t *cl);
|
||||||
bool init_recv_rdma(osd_client_t *cl);
|
bool init_recv_rdma(osd_client_t *cl);
|
||||||
void handle_rdma_events(msgr_rdma_context_t *rdma_context);
|
void handle_rdma_events(msgr_rdma_context_t *rdma_context);
|
||||||
msgr_rdma_context_t* choose_rdma_context(osd_client_t *cl);
|
msgr_rdma_context_t* choose_rdma_context(osd_client_t *cl);
|
||||||
|
void destroy_rdma_conn(msgr_rdma_connection_t *rdma_conn);
|
||||||
#endif
|
#endif
|
||||||
#ifdef WITH_RDMACM
|
#ifdef WITH_RDMACM
|
||||||
void handle_rdmacm_events();
|
void handle_rdmacm_events();
|
||||||
|
|||||||
@@ -0,0 +1,129 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
#include <stdexcept>
|
||||||
|
#include <sys/poll.h>
|
||||||
|
#include <unistd.h>
|
||||||
|
|
||||||
|
#include "messenger.h"
|
||||||
|
#include "msgr_iothread.h"
|
||||||
|
|
||||||
|
msgr_iothread_t::msgr_iothread_t():
|
||||||
|
ring(RINGLOOP_DEFAULT_SIZE, true),
|
||||||
|
thread(&msgr_iothread_t::run, this)
|
||||||
|
{
|
||||||
|
eventfd = ring.register_eventfd();
|
||||||
|
if (eventfd < 0)
|
||||||
|
{
|
||||||
|
throw std::runtime_error(std::string("failed to register eventfd: ") + strerror(-eventfd));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
msgr_iothread_t::~msgr_iothread_t()
|
||||||
|
{
|
||||||
|
stop();
|
||||||
|
}
|
||||||
|
|
||||||
|
void msgr_iothread_t::add_sqe(io_uring_sqe & sqe)
|
||||||
|
{
|
||||||
|
mu.lock();
|
||||||
|
queue.push_back((iothread_sqe_t){ .sqe = sqe, .data = std::move(*(ring_data_t*)sqe.user_data) });
|
||||||
|
if (queue.size() == 1)
|
||||||
|
{
|
||||||
|
cond.notify_all();
|
||||||
|
}
|
||||||
|
mu.unlock();
|
||||||
|
}
|
||||||
|
|
||||||
|
void msgr_iothread_t::stop()
|
||||||
|
{
|
||||||
|
mu.lock();
|
||||||
|
if (stopped)
|
||||||
|
{
|
||||||
|
mu.unlock();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
stopped = true;
|
||||||
|
if (outer_loop_data)
|
||||||
|
{
|
||||||
|
outer_loop_data->callback = [](ring_data_t*){};
|
||||||
|
}
|
||||||
|
cond.notify_all();
|
||||||
|
close(eventfd);
|
||||||
|
mu.unlock();
|
||||||
|
thread.join();
|
||||||
|
}
|
||||||
|
|
||||||
|
void msgr_iothread_t::add_to_ringloop(ring_loop_i *outer_loop)
|
||||||
|
{
|
||||||
|
assert(!this->outer_loop || this->outer_loop == outer_loop);
|
||||||
|
io_uring_sqe *sqe = outer_loop->get_sqe();
|
||||||
|
assert(sqe != NULL);
|
||||||
|
this->outer_loop = outer_loop;
|
||||||
|
this->outer_loop_data = ((ring_data_t*)sqe->user_data);
|
||||||
|
io_uring_prep_poll_add(sqe, eventfd, POLLIN);
|
||||||
|
outer_loop_data->callback = [this](ring_data_t *data)
|
||||||
|
{
|
||||||
|
if (data->res < 0)
|
||||||
|
{
|
||||||
|
throw std::runtime_error(std::string("eventfd poll failed: ") + strerror(-data->res));
|
||||||
|
}
|
||||||
|
outer_loop_data = NULL;
|
||||||
|
if (stopped)
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
add_to_ringloop(this->outer_loop);
|
||||||
|
ring.loop();
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
void msgr_iothread_t::run()
|
||||||
|
{
|
||||||
|
while (true)
|
||||||
|
{
|
||||||
|
{
|
||||||
|
std::unique_lock<std::mutex> lk(mu);
|
||||||
|
while (!stopped && !queue.size())
|
||||||
|
cond.wait(lk);
|
||||||
|
if (stopped)
|
||||||
|
return;
|
||||||
|
int i = 0;
|
||||||
|
for (; i < queue.size(); i++)
|
||||||
|
{
|
||||||
|
io_uring_sqe *sqe = ring.get_sqe();
|
||||||
|
if (!sqe)
|
||||||
|
break;
|
||||||
|
ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
||||||
|
*data = std::move(queue[i].data);
|
||||||
|
*sqe = queue[i].sqe;
|
||||||
|
sqe->user_data = (uint64_t)data;
|
||||||
|
}
|
||||||
|
queue.erase(queue.begin(), queue.begin()+i);
|
||||||
|
}
|
||||||
|
// We only want to offload sendmsg/recvmsg. Callbacks will be called in main thread
|
||||||
|
ring.submit();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void osd_messenger_t::init_iothreads()
|
||||||
|
{
|
||||||
|
for (int i = 0; i < iothread_count; i++)
|
||||||
|
{
|
||||||
|
auto iot = new msgr_iothread_t();
|
||||||
|
iothreads.push_back(iot);
|
||||||
|
iot->add_to_ringloop(ringloop);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void osd_messenger_t::destroy_iothreads()
|
||||||
|
{
|
||||||
|
if (iothreads.size())
|
||||||
|
{
|
||||||
|
for (auto iot: iothreads)
|
||||||
|
{
|
||||||
|
delete iot;
|
||||||
|
}
|
||||||
|
iothreads.clear();
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,38 @@
|
|||||||
|
// Copyright (c) Vitaliy Filippov, 2019+
|
||||||
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
|
#include <mutex>
|
||||||
|
#include <condition_variable>
|
||||||
|
#include <thread>
|
||||||
|
|
||||||
|
#include "ringloop.h"
|
||||||
|
|
||||||
|
struct iothread_sqe_t
|
||||||
|
{
|
||||||
|
io_uring_sqe sqe;
|
||||||
|
ring_data_t data;
|
||||||
|
};
|
||||||
|
|
||||||
|
class msgr_iothread_t
|
||||||
|
{
|
||||||
|
protected:
|
||||||
|
ring_loop_t ring;
|
||||||
|
ring_loop_i *outer_loop = NULL;
|
||||||
|
ring_data_t *outer_loop_data = NULL;
|
||||||
|
int eventfd = -1;
|
||||||
|
bool stopped = false;
|
||||||
|
std::mutex mu;
|
||||||
|
std::condition_variable cond;
|
||||||
|
std::vector<iothread_sqe_t> queue;
|
||||||
|
std::thread thread;
|
||||||
|
|
||||||
|
void run();
|
||||||
|
public:
|
||||||
|
|
||||||
|
msgr_iothread_t();
|
||||||
|
~msgr_iothread_t();
|
||||||
|
|
||||||
|
void add_sqe(io_uring_sqe & sqe);
|
||||||
|
void stop();
|
||||||
|
void add_to_ringloop(ring_loop_i *outer_loop);
|
||||||
|
};
|
||||||
@@ -3,6 +3,7 @@
|
|||||||
|
|
||||||
#include <assert.h>
|
#include <assert.h>
|
||||||
|
|
||||||
|
#include "messenger.h"
|
||||||
#include "msgr_op.h"
|
#include "msgr_op.h"
|
||||||
|
|
||||||
osd_op_t::~osd_op_t()
|
osd_op_t::~osd_op_t()
|
||||||
@@ -38,3 +39,52 @@ bool osd_op_t::is_recovery_related()
|
|||||||
req.hdr.opcode == OSD_OP_SEC_SYNC &&
|
req.hdr.opcode == OSD_OP_SEC_SYNC &&
|
||||||
(req.sec_sync.flags & OSD_OP_RECOVERY_RELATED);
|
(req.sec_sync.flags & OSD_OP_RECOVERY_RELATED);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void osd_messenger_t::measure_exec(osd_op_t *cur_op)
|
||||||
|
{
|
||||||
|
// Measure execution latency
|
||||||
|
if (cur_op->req.hdr.opcode > OSD_OP_MAX)
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (!cur_op->tv_end.tv_sec)
|
||||||
|
{
|
||||||
|
clock_gettime(CLOCK_REALTIME, &cur_op->tv_end);
|
||||||
|
}
|
||||||
|
uint64_t len = 0;
|
||||||
|
if (cur_op->req.hdr.opcode == OSD_OP_READ ||
|
||||||
|
cur_op->req.hdr.opcode == OSD_OP_WRITE ||
|
||||||
|
cur_op->req.hdr.opcode == OSD_OP_SCRUB)
|
||||||
|
{
|
||||||
|
// req.rw.len is internally set to the full object size for scrubs
|
||||||
|
len = cur_op->req.rw.len;
|
||||||
|
}
|
||||||
|
else if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ ||
|
||||||
|
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
|
||||||
|
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE)
|
||||||
|
{
|
||||||
|
len = cur_op->req.sec_rw.len;
|
||||||
|
}
|
||||||
|
inc_op_stats(stats, cur_op->req.hdr.opcode, cur_op->tv_begin, cur_op->tv_end, len);
|
||||||
|
if (cur_op->is_recovery_related())
|
||||||
|
{
|
||||||
|
inc_op_stats(recovery_stats, cur_op->req.hdr.opcode, cur_op->tv_begin, cur_op->tv_end, len);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void osd_messenger_t::inc_op_stats(osd_op_stats_t & stats, uint64_t opcode, timespec & tv_begin, timespec & tv_end, uint64_t len)
|
||||||
|
{
|
||||||
|
uint64_t usecs = (
|
||||||
|
(tv_end.tv_sec - tv_begin.tv_sec)*1000000 +
|
||||||
|
(tv_end.tv_nsec - tv_begin.tv_nsec)/1000
|
||||||
|
);
|
||||||
|
stats.op_stat_count[opcode]++;
|
||||||
|
if (!stats.op_stat_count[opcode])
|
||||||
|
{
|
||||||
|
stats.op_stat_count[opcode] = 1;
|
||||||
|
stats.op_stat_sum[opcode] = 0;
|
||||||
|
stats.op_stat_bytes[opcode] = 0;
|
||||||
|
}
|
||||||
|
stats.op_stat_sum[opcode] += usecs;
|
||||||
|
stats.op_stat_bytes[opcode] += len;
|
||||||
|
}
|
||||||
|
|||||||
@@ -156,7 +156,8 @@ struct __attribute__((visibility("default"))) osd_op_t
|
|||||||
{
|
{
|
||||||
timespec tv_begin = { 0 }, tv_end = { 0 };
|
timespec tv_begin = { 0 }, tv_end = { 0 };
|
||||||
uint64_t op_type = OSD_OP_IN;
|
uint64_t op_type = OSD_OP_IN;
|
||||||
int peer_fd;
|
uint64_t client_id = 0;
|
||||||
|
osd_num_t osd_num = 0;
|
||||||
osd_any_op_t req;
|
osd_any_op_t req;
|
||||||
osd_any_reply_t reply;
|
osd_any_reply_t reply;
|
||||||
blockstore_op_t *bs_op = NULL;
|
blockstore_op_t *bs_op = NULL;
|
||||||
@@ -164,7 +165,7 @@ struct __attribute__((visibility("default"))) osd_op_t
|
|||||||
// bitmap, bitmap_len, bmp_data are only meaningful for reads
|
// bitmap, bitmap_len, bmp_data are only meaningful for reads
|
||||||
void *bitmap = NULL;
|
void *bitmap = NULL;
|
||||||
unsigned bitmap_len = 0;
|
unsigned bitmap_len = 0;
|
||||||
unsigned bmp_data = 0;
|
size_t bmp_data = 0;
|
||||||
void *bitmap_buf = NULL;
|
void *bitmap_buf = NULL;
|
||||||
void *rmw_buf = NULL;
|
void *rmw_buf = NULL;
|
||||||
osd_primary_op_data_t* op_data = NULL;
|
osd_primary_op_data_t* op_data = NULL;
|
||||||
|
|||||||
+47
-19
@@ -187,6 +187,8 @@ std::vector<msgr_rdma_context_t*> msgr_rdma_context_t::create_all(const std::vec
|
|||||||
ibv_device **raw_dev_list = NULL;
|
ibv_device **raw_dev_list = NULL;
|
||||||
ibv_device **dev_list = NULL;
|
ibv_device **dev_list = NULL;
|
||||||
ibv_device *single_list[2] = {};
|
ibv_device *single_list[2] = {};
|
||||||
|
int up_ports = 0;
|
||||||
|
int single_port_num = 0;
|
||||||
|
|
||||||
raw_dev_list = dev_list = ibv_get_device_list(NULL);
|
raw_dev_list = dev_list = ibv_get_device_list(NULL);
|
||||||
if (!dev_list || !*dev_list)
|
if (!dev_list || !*dev_list)
|
||||||
@@ -221,6 +223,7 @@ std::vector<msgr_rdma_context_t*> msgr_rdma_context_t::create_all(const std::vec
|
|||||||
dev_list = single_list;
|
dev_list = single_list;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
retry:
|
||||||
for (int i = 0; dev_list[i]; ++i)
|
for (int i = 0; dev_list[i]; ++i)
|
||||||
{
|
{
|
||||||
auto dev = dev_list[i];
|
auto dev = dev_list[i];
|
||||||
@@ -258,6 +261,9 @@ std::vector<msgr_rdma_context_t*> msgr_rdma_context_t::create_all(const std::vec
|
|||||||
fprintf(stderr, "RDMA device %s port %d GID %d does not exist\n", ibv_get_device_name(dev), port_num, sel_gid_index);
|
fprintf(stderr, "RDMA device %s port %d GID %d does not exist\n", ibv_get_device_name(dev), port_num, sel_gid_index);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
up_ports++;
|
||||||
|
single_port_num = port_num;
|
||||||
|
single_list[0] = dev;
|
||||||
uint32_t port_mtu = sel_mtu ? sel_mtu : ibv_mtu_to_bytes(portinfo.active_mtu);
|
uint32_t port_mtu = sel_mtu ? sel_mtu : ibv_mtu_to_bytes(portinfo.active_mtu);
|
||||||
#ifdef IBV_ADVISE_MR_ADVICE_PREFETCH_NO_FAULT
|
#ifdef IBV_ADVISE_MR_ADVICE_PREFETCH_NO_FAULT
|
||||||
if (sel_gid_index < 0)
|
if (sel_gid_index < 0)
|
||||||
@@ -298,6 +304,14 @@ cleanup_dev:
|
|||||||
ibv_close_device(context);
|
ibv_close_device(context);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (!ret.size() && up_ports == 1 && dev_list != single_list)
|
||||||
|
{
|
||||||
|
// Auto-select the only available device/port if there is only one
|
||||||
|
dev_list = single_list;
|
||||||
|
sel_port_num = single_port_num;
|
||||||
|
goto retry;
|
||||||
|
}
|
||||||
|
|
||||||
cleanup:
|
cleanup:
|
||||||
if (raw_dev_list)
|
if (raw_dev_list)
|
||||||
ibv_free_device_list(raw_dev_list);
|
ibv_free_device_list(raw_dev_list);
|
||||||
@@ -493,7 +507,7 @@ int msgr_rdma_connection_t::connect(msgr_rdma_address_t *dest)
|
|||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
bool osd_messenger_t::connect_rdma(int peer_fd, std::string rdma_address, uint64_t client_max_msg)
|
json11::Json osd_messenger_t::connect_rdma(uint64_t client_id, std::string rdma_address, uint64_t client_max_msg)
|
||||||
{
|
{
|
||||||
// Try to connect to the peer using RDMA
|
// Try to connect to the peer using RDMA
|
||||||
msgr_rdma_address_t addr;
|
msgr_rdma_address_t addr;
|
||||||
@@ -503,13 +517,13 @@ bool osd_messenger_t::connect_rdma(int peer_fd, std::string rdma_address, uint64
|
|||||||
{
|
{
|
||||||
client_max_msg = rdma_max_msg;
|
client_max_msg = rdma_max_msg;
|
||||||
}
|
}
|
||||||
auto cl = clients.at(peer_fd);
|
auto cl = clients.at(client_id);
|
||||||
msgr_rdma_context_t *selected_ctx = choose_rdma_context(cl);
|
msgr_rdma_context_t *selected_ctx = choose_rdma_context(cl);
|
||||||
if (!selected_ctx)
|
if (!selected_ctx)
|
||||||
{
|
{
|
||||||
if (log_level > 0)
|
if (log_level > 0)
|
||||||
fprintf(stderr, "No RDMA context for peer %d, using only TCP\n", cl->peer_fd);
|
fprintf(stderr, "No RDMA context for peer %ju, using only TCP\n", client_id);
|
||||||
return false;
|
return json11::Json();
|
||||||
}
|
}
|
||||||
msgr_rdma_connection_t *rdma_conn = msgr_rdma_connection_t::create(selected_ctx, rdma_max_send, rdma_max_recv, rdma_max_sge, client_max_msg);
|
msgr_rdma_connection_t *rdma_conn = msgr_rdma_connection_t::create(selected_ctx, rdma_max_send, rdma_max_recv, rdma_max_sge, client_max_msg);
|
||||||
if (rdma_conn)
|
if (rdma_conn)
|
||||||
@@ -519,28 +533,30 @@ bool osd_messenger_t::connect_rdma(int peer_fd, std::string rdma_address, uint64
|
|||||||
{
|
{
|
||||||
delete rdma_conn;
|
delete rdma_conn;
|
||||||
fprintf(
|
fprintf(
|
||||||
stderr, "Failed to connect RDMA queue pair to %s (client %d)\n",
|
stderr, "Failed to connect RDMA queue pair to %s (client %ju)\n",
|
||||||
addr.to_string().c_str(), peer_fd
|
addr.to_string().c_str(), client_id
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
// Remember connection, but switch to RDMA only after sending the configuration response
|
// Remember connection, but switch to RDMA only after sending the configuration response
|
||||||
auto cl = clients.at(peer_fd);
|
|
||||||
cl->rdma_conn = rdma_conn;
|
cl->rdma_conn = rdma_conn;
|
||||||
cl->peer_state = PEER_RDMA_CONNECTING;
|
cl->peer_state = PEER_RDMA_CONNECTING;
|
||||||
return true;
|
return json11::Json::object{
|
||||||
|
{"rdma_address", rdma_conn->addr.to_string()},
|
||||||
|
{"rdma_max_msg", rdma_conn->max_msg},
|
||||||
|
};
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return false;
|
return json11::Json();
|
||||||
}
|
}
|
||||||
|
|
||||||
static void try_send_rdma_wr(osd_client_t *cl, ibv_sge *sge, int op_sge)
|
static void try_send_rdma_wr(osd_client_t *cl, ibv_sge *sge, int op_sge)
|
||||||
{
|
{
|
||||||
ibv_send_wr *bad_wr = NULL;
|
ibv_send_wr *bad_wr = NULL;
|
||||||
ibv_send_wr wr = {
|
ibv_send_wr wr = {
|
||||||
.wr_id = (uint64_t)(cl->peer_fd*2+1),
|
.wr_id = cl->client_id,
|
||||||
.sg_list = sge,
|
.sg_list = sge,
|
||||||
.num_sge = op_sge,
|
.num_sge = op_sge,
|
||||||
.opcode = IBV_WR_SEND,
|
.opcode = IBV_WR_SEND,
|
||||||
@@ -599,7 +615,9 @@ void osd_messenger_t::try_send_rdma(osd_client_t *cl)
|
|||||||
while (!rc->send_out_full && copied > 0 && rc->cur_send < rc->max_send)
|
while (!rc->send_out_full && copied > 0 && rc->cur_send < rc->max_send)
|
||||||
{
|
{
|
||||||
dst = (uint8_t*)rc->send_out.buf + rc->send_out_pos;
|
dst = (uint8_t*)rc->send_out.buf + rc->send_out_pos;
|
||||||
dst_len = (rc->send_out_pos < rc->send_out_size ? rc->send_out_size-rc->send_out_pos : rc->send_done_pos-rc->send_out_pos);
|
dst_len = (rc->send_out_pos >= rc->send_done_pos
|
||||||
|
? rc->send_out_size-rc->send_out_pos
|
||||||
|
: rc->send_done_pos-rc->send_out_pos);
|
||||||
if (dst_len > rc->max_msg)
|
if (dst_len > rc->max_msg)
|
||||||
dst_len = rc->max_msg;
|
dst_len = rc->max_msg;
|
||||||
copied = try_send_rdma_copy(cl, dst, dst_len);
|
copied = try_send_rdma_copy(cl, dst, dst_len);
|
||||||
@@ -609,7 +627,7 @@ void osd_messenger_t::try_send_rdma(osd_client_t *cl)
|
|||||||
if (rc->send_out_pos == rc->send_out_size)
|
if (rc->send_out_pos == rc->send_out_size)
|
||||||
rc->send_out_pos = 0;
|
rc->send_out_pos = 0;
|
||||||
assert(rc->send_out_pos < rc->send_out_size);
|
assert(rc->send_out_pos < rc->send_out_size);
|
||||||
if (rc->send_out_pos >= rc->send_done_pos)
|
if (rc->send_out_pos == rc->send_done_pos)
|
||||||
rc->send_out_full = true;
|
rc->send_out_full = true;
|
||||||
ibv_sge sge = {
|
ibv_sge sge = {
|
||||||
.addr = (uintptr_t)dst,
|
.addr = (uintptr_t)dst,
|
||||||
@@ -631,7 +649,7 @@ static void try_recv_rdma_wr(osd_client_t *cl, void *buf)
|
|||||||
};
|
};
|
||||||
ibv_recv_wr *bad_wr = NULL;
|
ibv_recv_wr *bad_wr = NULL;
|
||||||
ibv_recv_wr wr = {
|
ibv_recv_wr wr = {
|
||||||
.wr_id = (uint64_t)(cl->peer_fd*2),
|
.wr_id = cl->client_id,
|
||||||
.sg_list = &sge,
|
.sg_list = &sge,
|
||||||
.num_sge = 1,
|
.num_sge = 1,
|
||||||
};
|
};
|
||||||
@@ -688,8 +706,8 @@ void osd_messenger_t::handle_rdma_events(msgr_rdma_context_t *rdma_context)
|
|||||||
event_count = ibv_poll_cq(rdma_context->cq, RDMA_EVENTS_AT_ONCE, wc);
|
event_count = ibv_poll_cq(rdma_context->cq, RDMA_EVENTS_AT_ONCE, wc);
|
||||||
for (int i = 0; i < event_count; i++)
|
for (int i = 0; i < event_count; i++)
|
||||||
{
|
{
|
||||||
int client_id = wc[i].wr_id >> 1;
|
uint64_t client_id = wc[i].wr_id;
|
||||||
bool is_send = wc[i].wr_id & 1;
|
bool is_send = wc[i].opcode == IBV_WC_SEND;
|
||||||
auto cl_it = clients.find(client_id);
|
auto cl_it = clients.find(client_id);
|
||||||
if (cl_it == clients.end())
|
if (cl_it == clients.end())
|
||||||
{
|
{
|
||||||
@@ -703,14 +721,13 @@ void osd_messenger_t::handle_rdma_events(msgr_rdma_context_t *rdma_context)
|
|||||||
auto rc = cl->rdma_conn;
|
auto rc = cl->rdma_conn;
|
||||||
if (wc[i].status != IBV_WC_SUCCESS)
|
if (wc[i].status != IBV_WC_SUCCESS)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "RDMA work request failed for client %d", client_id);
|
fprintf(stderr, "RDMA work request failed for client %ju", client_id);
|
||||||
if (cl->osd_num)
|
if (cl->osd_num)
|
||||||
{
|
{
|
||||||
fprintf(stderr, " (OSD %ju)", cl->osd_num);
|
fprintf(stderr, " (OSD %ju)", cl->osd_num);
|
||||||
}
|
}
|
||||||
fprintf(stderr, " with status: %s, stopping client\n", ibv_wc_status_str(wc[i].status));
|
fprintf(stderr, " with status: %s, stopping client\n", ibv_wc_status_str(wc[i].status));
|
||||||
stop_client(client_id);
|
stop_client(client_id);
|
||||||
clear_immediate_ops(client_id);
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if (!is_send)
|
if (!is_send)
|
||||||
@@ -721,8 +738,6 @@ void osd_messenger_t::handle_rdma_events(msgr_rdma_context_t *rdma_context)
|
|||||||
rc->cur_recv--;
|
rc->cur_recv--;
|
||||||
if (!handle_read_buffer(cl, rc->recv_buffers[rc->next_recv_buf], wc[i].byte_len))
|
if (!handle_read_buffer(cl, rc->recv_buffers[rc->next_recv_buf], wc[i].byte_len))
|
||||||
{
|
{
|
||||||
// handle_read_buffer may stop the client
|
|
||||||
clear_immediate_ops(client_id);
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
try_recv_rdma_wr(cl, rc->recv_buffers[rc->next_recv_buf]);
|
try_recv_rdma_wr(cl, rc->recv_buffers[rc->next_recv_buf]);
|
||||||
@@ -782,3 +797,16 @@ void osd_messenger_t::handle_rdma_events(msgr_rdma_context_t *rdma_context)
|
|||||||
} while (event_count > 0);
|
} while (event_count > 0);
|
||||||
handle_immediate_ops();
|
handle_immediate_ops();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void osd_messenger_t::destroy_rdma_conn(msgr_rdma_connection_t *rdma_conn)
|
||||||
|
{
|
||||||
|
if (rdma_conn->cmid)
|
||||||
|
{
|
||||||
|
auto rdma_it = rdmacm_connections.find(rdma_conn->cmid);
|
||||||
|
if (rdma_it != rdmacm_connections.end() && rdma_it->second->rdma_conn == rdma_conn)
|
||||||
|
{
|
||||||
|
rdmacm_connections.erase(rdma_it);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
delete rdma_conn;
|
||||||
|
}
|
||||||
|
|||||||
+12
-32
@@ -11,7 +11,7 @@
|
|||||||
struct rdmacm_connecting_t
|
struct rdmacm_connecting_t
|
||||||
{
|
{
|
||||||
rdma_cm_id *cmid = NULL;
|
rdma_cm_id *cmid = NULL;
|
||||||
int peer_fd = -1;
|
uint64_t client_id = 0;
|
||||||
osd_num_t peer_osd = 0;
|
osd_num_t peer_osd = 0;
|
||||||
std::string addr;
|
std::string addr;
|
||||||
sockaddr_storage parsed_addr = {};
|
sockaddr_storage parsed_addr = {};
|
||||||
@@ -117,9 +117,9 @@ void osd_messenger_t::handle_rdmacm_events()
|
|||||||
auto cli_it = rdmacm_connections.find(ev->id);
|
auto cli_it = rdmacm_connections.find(ev->id);
|
||||||
if (cli_it != rdmacm_connections.end())
|
if (cli_it != rdmacm_connections.end())
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Received %s event for peer %d, closing connection\n",
|
fprintf(stderr, "Received %s event for client %ju, closing connection\n",
|
||||||
event_type_name, cli_it->second->peer_fd);
|
event_type_name, cli_it->second->client_id);
|
||||||
stop_client(cli_it->second->peer_fd);
|
stop_client(cli_it->second->client_id);
|
||||||
}
|
}
|
||||||
else if (rdmacm_connecting.find(ev->id) != rdmacm_connecting.end())
|
else if (rdmacm_connecting.find(ev->id) != rdmacm_connecting.end())
|
||||||
{
|
{
|
||||||
@@ -265,14 +265,6 @@ msgr_rdma_context_t* osd_messenger_t::rdmacm_create_qp(rdma_cm_id *cmid)
|
|||||||
|
|
||||||
void osd_messenger_t::rdmacm_accept(rdma_cm_event *ev)
|
void osd_messenger_t::rdmacm_accept(rdma_cm_event *ev)
|
||||||
{
|
{
|
||||||
// Make a fake FD (FIXME: do not use FDs for identifying clients!)
|
|
||||||
int fake_fd = socket(AF_INET, SOCK_STREAM, 0);
|
|
||||||
if (fake_fd < 0)
|
|
||||||
{
|
|
||||||
fprintf(stderr, "Failed to allocate a fake socket for RDMA-CM client: %s (code %d)\n", strerror(errno), errno);
|
|
||||||
rdma_destroy_id(ev->id);
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
auto rdma_context = rdmacm_create_qp(ev->id);
|
auto rdma_context = rdmacm_create_qp(ev->id);
|
||||||
if (!rdma_context)
|
if (!rdma_context)
|
||||||
{
|
{
|
||||||
@@ -297,12 +289,12 @@ void osd_messenger_t::rdmacm_accept(rdma_cm_event *ev)
|
|||||||
// Wait for RDMA_CM_ESTABLISHED, and enable the connection only after it
|
// Wait for RDMA_CM_ESTABLISHED, and enable the connection only after it
|
||||||
auto conn = new rdmacm_connecting_t;
|
auto conn = new rdmacm_connecting_t;
|
||||||
conn->cmid = ev->id;
|
conn->cmid = ev->id;
|
||||||
conn->peer_fd = fake_fd;
|
conn->client_id = next_client_id++;
|
||||||
conn->parsed_addr = *(sockaddr_storage*)rdma_get_peer_addr(ev->id);
|
conn->parsed_addr = *(sockaddr_storage*)rdma_get_peer_addr(ev->id);
|
||||||
conn->rdma_context = rdma_context;
|
conn->rdma_context = rdma_context;
|
||||||
rdmacm_set_conn_timeout(conn);
|
rdmacm_set_conn_timeout(conn);
|
||||||
rdmacm_connecting[ev->id] = conn;
|
rdmacm_connecting[ev->id] = conn;
|
||||||
fprintf(stderr, "[OSD %ju] new client %d: connection from %s via RDMA-CM\n", this->osd_num, conn->peer_fd,
|
fprintf(stderr, "[OSD %ju] new client %ju: connection from %s via RDMA-CM\n", this->osd_num, conn->client_id,
|
||||||
addr_to_string(conn->parsed_addr).c_str());
|
addr_to_string(conn->parsed_addr).c_str());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -332,8 +324,6 @@ void osd_messenger_t::rdmacm_on_connect_peer_error(rdma_cm_id *cmid, int res)
|
|||||||
auto peer_osd = conn->peer_osd;
|
auto peer_osd = conn->peer_osd;
|
||||||
if (conn->timeout_id >= 0)
|
if (conn->timeout_id >= 0)
|
||||||
tfd->clear_timer(conn->timeout_id);
|
tfd->clear_timer(conn->timeout_id);
|
||||||
if (conn->peer_fd >= 0)
|
|
||||||
close(conn->peer_fd);
|
|
||||||
if (conn->rdma_context)
|
if (conn->rdma_context)
|
||||||
conn->rdma_context->reserve_cqe(-rdma_max_send-rdma_max_recv);
|
conn->rdma_context->reserve_cqe(-rdma_max_send-rdma_max_recv);
|
||||||
if (conn->cmid)
|
if (conn->cmid)
|
||||||
@@ -354,7 +344,7 @@ void osd_messenger_t::rdmacm_on_connect_peer_error(rdma_cm_id *cmid, int res)
|
|||||||
else
|
else
|
||||||
{
|
{
|
||||||
// TCP is disabled
|
// TCP is disabled
|
||||||
on_connect_peer(peer_osd, res == 0 ? -EINVAL : (res > 0 ? -res : res));
|
on_connect_peer(peer_osd, res == 0 ? -EINVAL : (res > 0 ? -res : res), 0);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -365,7 +355,7 @@ void osd_messenger_t::rdmacm_try_connect_peer(uint64_t peer_osd, const std::stri
|
|||||||
if (!string_to_addr(addr, false, rdmacm_port, &sa))
|
if (!string_to_addr(addr, false, rdmacm_port, &sa))
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Address %s is invalid\n", addr.c_str());
|
fprintf(stderr, "Address %s is invalid\n", addr.c_str());
|
||||||
on_connect_peer(peer_osd, -EINVAL);
|
on_connect_peer(peer_osd, -EINVAL, 0);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
rdma_cm_id *cmid = NULL;
|
rdma_cm_id *cmid = NULL;
|
||||||
@@ -376,17 +366,7 @@ void osd_messenger_t::rdmacm_try_connect_peer(uint64_t peer_osd, const std::stri
|
|||||||
if (!disable_tcp)
|
if (!disable_tcp)
|
||||||
try_connect_peer_tcp(peer_osd, addr.c_str(), fallback_tcp_port);
|
try_connect_peer_tcp(peer_osd, addr.c_str(), fallback_tcp_port);
|
||||||
else
|
else
|
||||||
on_connect_peer(peer_osd, res);
|
on_connect_peer(peer_osd, res, 0);
|
||||||
return;
|
|
||||||
}
|
|
||||||
// Make a fake FD (FIXME: do not use FDs for identifying clients!)
|
|
||||||
int fake_fd = socket(AF_INET, SOCK_STREAM, 0);
|
|
||||||
if (fake_fd < 0)
|
|
||||||
{
|
|
||||||
int res = -errno;
|
|
||||||
rdma_destroy_id(cmid);
|
|
||||||
// Can't create socket, pointless to try TCP
|
|
||||||
on_connect_peer(peer_osd, res);
|
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if (log_level > 0)
|
if (log_level > 0)
|
||||||
@@ -394,7 +374,7 @@ void osd_messenger_t::rdmacm_try_connect_peer(uint64_t peer_osd, const std::stri
|
|||||||
auto conn = new rdmacm_connecting_t;
|
auto conn = new rdmacm_connecting_t;
|
||||||
rdmacm_connecting[cmid] = conn;
|
rdmacm_connecting[cmid] = conn;
|
||||||
conn->cmid = cmid;
|
conn->cmid = cmid;
|
||||||
conn->peer_fd = fake_fd;
|
conn->client_id = next_client_id++;
|
||||||
conn->peer_osd = peer_osd;
|
conn->peer_osd = peer_osd;
|
||||||
conn->addr = addr;
|
conn->addr = addr;
|
||||||
conn->parsed_addr = sa;
|
conn->parsed_addr = sa;
|
||||||
@@ -511,13 +491,13 @@ void osd_messenger_t::rdmacm_established(rdma_cm_event *ev)
|
|||||||
auto cl = new osd_client_t();
|
auto cl = new osd_client_t();
|
||||||
cl->peer_addr = conn->parsed_addr;
|
cl->peer_addr = conn->parsed_addr;
|
||||||
cl->peer_port = conn->rdmacm_port;
|
cl->peer_port = conn->rdmacm_port;
|
||||||
cl->peer_fd = conn->peer_fd;
|
cl->client_id = conn->client_id;
|
||||||
cl->peer_state = PEER_RDMA;
|
cl->peer_state = PEER_RDMA;
|
||||||
cl->connect_timeout_id = -1;
|
cl->connect_timeout_id = -1;
|
||||||
cl->osd_num = peer_osd;
|
cl->osd_num = peer_osd;
|
||||||
cl->in_buf = malloc_or_die(receive_buffer_size);
|
cl->in_buf = malloc_or_die(receive_buffer_size);
|
||||||
cl->rdma_conn = rc;
|
cl->rdma_conn = rc;
|
||||||
clients[conn->peer_fd] = cl;
|
clients[conn->client_id] = cl;
|
||||||
if (conn->timeout_id >= 0)
|
if (conn->timeout_id >= 0)
|
||||||
{
|
{
|
||||||
tfd->clear_timer(conn->timeout_id);
|
tfd->clear_timer(conn->timeout_id);
|
||||||
|
|||||||
+91
-50
@@ -2,15 +2,16 @@
|
|||||||
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||||
|
|
||||||
#include "messenger.h"
|
#include "messenger.h"
|
||||||
|
#include "msgr_iothread.h"
|
||||||
|
|
||||||
void osd_messenger_t::read_requests()
|
void osd_messenger_t::read_requests()
|
||||||
{
|
{
|
||||||
for (int i = 0; i < read_ready_clients.size(); i++)
|
for (int i = 0; i < read_ready_clients.size(); i++)
|
||||||
{
|
{
|
||||||
int peer_fd = read_ready_clients[i];
|
uint64_t client_id = read_ready_clients[i];
|
||||||
auto cl_it = clients.find(peer_fd);
|
auto cl_it = clients.find(client_id);
|
||||||
if (cl_it == clients.end() || !cl_it->second || cl_it->second->read_msg.msg_iovlen ||
|
if (cl_it == clients.end() || !cl_it->second || cl_it->second->read_msg.msg_iovlen ||
|
||||||
cl_it->second->peer_state == PEER_RDMA || cl_it->second->peer_state == PEER_RDMA_CONNECTING)
|
cl_it->second->peer_state != PEER_CONNECTED)
|
||||||
{
|
{
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -32,7 +33,7 @@ void osd_messenger_t::read_requests()
|
|||||||
cl->refs++;
|
cl->refs++;
|
||||||
if (ringloop && !use_sync_send_recv)
|
if (ringloop && !use_sync_send_recv)
|
||||||
{
|
{
|
||||||
auto iothread = iothreads.size() ? iothreads[peer_fd % iothreads.size()] : NULL;
|
auto iothread = iothreads.size() ? iothreads[cl->peer_fd % iothreads.size()] : NULL;
|
||||||
io_uring_sqe sqe_local;
|
io_uring_sqe sqe_local;
|
||||||
ring_data_t data_local;
|
ring_data_t data_local;
|
||||||
io_uring_sqe* sqe = (iothread ? &sqe_local : ringloop->get_sqe());
|
io_uring_sqe* sqe = (iothread ? &sqe_local : ringloop->get_sqe());
|
||||||
@@ -50,7 +51,7 @@ void osd_messenger_t::read_requests()
|
|||||||
}
|
}
|
||||||
ring_data_t* data = ((ring_data_t*)sqe->user_data);
|
ring_data_t* data = ((ring_data_t*)sqe->user_data);
|
||||||
data->callback = [this, cl](ring_data_t *data) { handle_read(data->res, cl); };
|
data->callback = [this, cl](ring_data_t *data) { handle_read(data->res, cl); };
|
||||||
io_uring_prep_recvmsg(sqe, peer_fd, &cl->read_msg, 0);
|
io_uring_prep_recvmsg(sqe, cl->peer_fd, &cl->read_msg, 0);
|
||||||
if (iothread)
|
if (iothread)
|
||||||
{
|
{
|
||||||
iothread->add_sqe(sqe_local);
|
iothread->add_sqe(sqe_local);
|
||||||
@@ -58,7 +59,7 @@ void osd_messenger_t::read_requests()
|
|||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
int result = recvmsg(peer_fd, &cl->read_msg, 0);
|
int result = recvmsg(cl->peer_fd, &cl->read_msg, 0);
|
||||||
if (result < 0)
|
if (result < 0)
|
||||||
{
|
{
|
||||||
result = -errno;
|
result = -errno;
|
||||||
@@ -73,7 +74,6 @@ void osd_messenger_t::read_requests()
|
|||||||
bool osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
bool osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
||||||
{
|
{
|
||||||
bool ret = false;
|
bool ret = false;
|
||||||
int peer_fd = cl->peer_fd;
|
|
||||||
cl->read_msg.msg_iovlen = 0;
|
cl->read_msg.msg_iovlen = 0;
|
||||||
cl->refs--;
|
cl->refs--;
|
||||||
if (cl->peer_state == PEER_RDMA)
|
if (cl->peer_state == PEER_RDMA)
|
||||||
@@ -84,7 +84,7 @@ bool osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
|||||||
{
|
{
|
||||||
if (cl->refs <= 0)
|
if (cl->refs <= 0)
|
||||||
{
|
{
|
||||||
delete cl;
|
destroy_client(cl);
|
||||||
}
|
}
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
@@ -93,20 +93,20 @@ bool osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
|||||||
// this is a client socket, so don't panic on error. just disconnect it
|
// this is a client socket, so don't panic on error. just disconnect it
|
||||||
if (result != 0)
|
if (result != 0)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Client %d socket read error: %d (%s). Disconnecting client\n", cl->peer_fd, -result, strerror(-result));
|
fprintf(stderr, "Client %ju socket read error: %d (%s). Disconnecting client\n", cl->client_id, -result, strerror(-result));
|
||||||
}
|
}
|
||||||
stop_client(cl->peer_fd);
|
stop_client(cl->client_id);
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
if (result == -EAGAIN || result == -EINTR || result < cl->read_iov.iov_len)
|
if (result == -EAGAIN || result == -EINTR || result < cl->read_iov.iov_len)
|
||||||
{
|
{
|
||||||
cl->read_ready--;
|
cl->read_ready--;
|
||||||
if (cl->read_ready > 0)
|
if (cl->read_ready > 0)
|
||||||
read_ready_clients.push_back(cl->peer_fd);
|
read_ready_clients.push_back(cl->client_id);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
read_ready_clients.push_back(cl->peer_fd);
|
read_ready_clients.push_back(cl->client_id);
|
||||||
}
|
}
|
||||||
if (result > 0)
|
if (result > 0)
|
||||||
{
|
{
|
||||||
@@ -114,7 +114,6 @@ bool osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
|||||||
{
|
{
|
||||||
if (!handle_read_buffer(cl, cl->in_buf, result))
|
if (!handle_read_buffer(cl, cl->in_buf, result))
|
||||||
{
|
{
|
||||||
clear_immediate_ops(peer_fd);
|
|
||||||
handle_immediate_ops();
|
handle_immediate_ops();
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
@@ -128,7 +127,6 @@ bool osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
|||||||
{
|
{
|
||||||
if (!handle_finished_read(cl))
|
if (!handle_finished_read(cl))
|
||||||
{
|
{
|
||||||
clear_immediate_ops(peer_fd);
|
|
||||||
handle_immediate_ops();
|
handle_immediate_ops();
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
@@ -143,26 +141,6 @@ bool osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
|||||||
return ret;
|
return ret;
|
||||||
}
|
}
|
||||||
|
|
||||||
void osd_messenger_t::clear_immediate_ops(int peer_fd)
|
|
||||||
{
|
|
||||||
size_t i = 0, j = 0;
|
|
||||||
while (i < set_immediate_ops.size())
|
|
||||||
{
|
|
||||||
if (set_immediate_ops[i]->peer_fd == peer_fd && set_immediate_ops[i]->op_type == OSD_OP_IN)
|
|
||||||
{
|
|
||||||
delete set_immediate_ops[i];
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
if (i != j)
|
|
||||||
set_immediate_ops[j] = set_immediate_ops[i];
|
|
||||||
j++;
|
|
||||||
}
|
|
||||||
i++;
|
|
||||||
}
|
|
||||||
set_immediate_ops.resize(j);
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::handle_immediate_ops()
|
void osd_messenger_t::handle_immediate_ops()
|
||||||
{
|
{
|
||||||
while (set_immediate_ops.size())
|
while (set_immediate_ops.size())
|
||||||
@@ -171,7 +149,11 @@ void osd_messenger_t::handle_immediate_ops()
|
|||||||
set_immediate_ops.pop_front();
|
set_immediate_ops.pop_front();
|
||||||
if (op->op_type == OSD_OP_IN)
|
if (op->op_type == OSD_OP_IN)
|
||||||
{
|
{
|
||||||
exec_op(op);
|
auto cl_it = clients.find(op->client_id);
|
||||||
|
if (cl_it != clients.end() && cl_it->second->peer_state != PEER_STOPPED)
|
||||||
|
exec_op(op);
|
||||||
|
else
|
||||||
|
delete op;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -189,7 +171,7 @@ bool osd_messenger_t::handle_read_buffer(osd_client_t *cl, void *curbuf, int rem
|
|||||||
if (!cl->read_op)
|
if (!cl->read_op)
|
||||||
{
|
{
|
||||||
cl->read_op = new osd_op_t;
|
cl->read_op = new osd_op_t;
|
||||||
cl->read_op->peer_fd = cl->peer_fd;
|
cl->read_op->client_id = cl->client_id;
|
||||||
cl->read_op->op_type = OSD_OP_IN;
|
cl->read_op->op_type = OSD_OP_IN;
|
||||||
cl->recv_list.push_back(cl->read_op->req.buf, OSD_PACKET_SIZE);
|
cl->recv_list.push_back(cl->read_op->req.buf, OSD_PACKET_SIZE);
|
||||||
cl->read_remaining = OSD_PACKET_SIZE;
|
cl->read_remaining = OSD_PACKET_SIZE;
|
||||||
@@ -243,18 +225,22 @@ bool osd_messenger_t::handle_finished_read(osd_client_t *cl)
|
|||||||
{
|
{
|
||||||
if (cl->read_op->req.hdr.id != cl->read_op_id)
|
if (cl->read_op->req.hdr.id != cl->read_op_id)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Warning: operation sequencing is broken on client %d: expected num %ju, got %ju, stopping client\n", cl->peer_fd, cl->read_op_id, cl->read_op->req.hdr.id);
|
fprintf(stderr, "Warning: operation sequencing is broken on client %ju: expected num %ju, got %ju, stopping client\n", cl->client_id, cl->read_op_id, cl->read_op->req.hdr.id);
|
||||||
stop_client(cl->peer_fd);
|
stop_client(cl->client_id);
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
cl->read_op_id++;
|
cl->read_op_id++;
|
||||||
}
|
}
|
||||||
handle_op_hdr(cl);
|
if (!handle_op_hdr(cl))
|
||||||
|
{
|
||||||
|
stop_client(cl->client_id);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Received garbage: magic=%jx id=%ju opcode=%jx from %d\n", cl->read_op->req.hdr.magic, cl->read_op->req.hdr.id, cl->read_op->req.hdr.opcode, cl->peer_fd);
|
fprintf(stderr, "Received garbage: magic=%jx id=%ju opcode=%jx from client %ju\n", cl->read_op->req.hdr.magic, cl->read_op->req.hdr.id, cl->read_op->req.hdr.opcode, cl->client_id);
|
||||||
stop_client(cl->peer_fd);
|
stop_client(cl->client_id);
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -280,7 +266,7 @@ bool osd_messenger_t::handle_finished_read(osd_client_t *cl)
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
void osd_messenger_t::handle_op_hdr(osd_client_t *cl)
|
bool osd_messenger_t::handle_op_hdr(osd_client_t *cl)
|
||||||
{
|
{
|
||||||
osd_op_t *cur_op = cl->read_op;
|
osd_op_t *cur_op = cl->read_op;
|
||||||
if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ)
|
if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ)
|
||||||
@@ -292,7 +278,16 @@ void osd_messenger_t::handle_op_hdr(osd_client_t *cl)
|
|||||||
{
|
{
|
||||||
if (cur_op->req.sec_rw.attr_len > 0)
|
if (cur_op->req.sec_rw.attr_len > 0)
|
||||||
{
|
{
|
||||||
if (cur_op->req.sec_rw.attr_len > sizeof(unsigned))
|
if (cur_op->req.sec_rw.attr_len > clean_entry_bitmap_size)
|
||||||
|
{
|
||||||
|
if (log_level > 1)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Error: peer %ju secondary write request attr_len too large (%u > %u bytes), stopping\n", cl->client_id,
|
||||||
|
cur_op->req.sec_rw.attr_len, clean_entry_bitmap_size);
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
else if (cur_op->req.sec_rw.attr_len > sizeof(cur_op->bmp_data))
|
||||||
cur_op->bitmap = cur_op->rmw_buf = malloc_or_die(cur_op->req.sec_rw.attr_len);
|
cur_op->bitmap = cur_op->rmw_buf = malloc_or_die(cur_op->req.sec_rw.attr_len);
|
||||||
else
|
else
|
||||||
cur_op->bitmap = &cur_op->bmp_data;
|
cur_op->bitmap = &cur_op->bmp_data;
|
||||||
@@ -300,6 +295,15 @@ void osd_messenger_t::handle_op_hdr(osd_client_t *cl)
|
|||||||
}
|
}
|
||||||
if (cur_op->req.sec_rw.len > 0)
|
if (cur_op->req.sec_rw.len > 0)
|
||||||
{
|
{
|
||||||
|
if (cur_op->req.sec_rw.len > bs_block_size)
|
||||||
|
{
|
||||||
|
if (log_level > 1)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Error: peer %ju secondary write request size too large (%u > %u bytes), stopping\n", cl->client_id,
|
||||||
|
cur_op->req.sec_rw.len, bs_block_size);
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.sec_rw.len);
|
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.sec_rw.len);
|
||||||
cl->recv_list.push_back(cur_op->buf, cur_op->req.sec_rw.len);
|
cl->recv_list.push_back(cur_op->buf, cur_op->req.sec_rw.len);
|
||||||
}
|
}
|
||||||
@@ -310,6 +314,15 @@ void osd_messenger_t::handle_op_hdr(osd_client_t *cl)
|
|||||||
{
|
{
|
||||||
if (cur_op->req.sec_stab.len > 0)
|
if (cur_op->req.sec_stab.len > 0)
|
||||||
{
|
{
|
||||||
|
if (cur_op->req.sec_stab.len > MAX_SIMPLE_PAYLOAD_SIZE)
|
||||||
|
{
|
||||||
|
if (log_level > 1)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Error: peer %ju stabilize request size too large (%lu > %u bytes), stopping\n", cl->client_id,
|
||||||
|
cur_op->req.sec_stab.len, MAX_SIMPLE_PAYLOAD_SIZE);
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.sec_stab.len);
|
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.sec_stab.len);
|
||||||
cl->recv_list.push_back(cur_op->buf, cur_op->req.sec_stab.len);
|
cl->recv_list.push_back(cur_op->buf, cur_op->req.sec_stab.len);
|
||||||
}
|
}
|
||||||
@@ -319,6 +332,15 @@ void osd_messenger_t::handle_op_hdr(osd_client_t *cl)
|
|||||||
{
|
{
|
||||||
if (cur_op->req.sec_read_bmp.len > 0)
|
if (cur_op->req.sec_read_bmp.len > 0)
|
||||||
{
|
{
|
||||||
|
if (cur_op->req.sec_read_bmp.len > MAX_SIMPLE_PAYLOAD_SIZE)
|
||||||
|
{
|
||||||
|
if (log_level > 1)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Error: peer %ju sec_read_bmp request size too large (%lu > %u bytes), stopping\n", cl->client_id,
|
||||||
|
cur_op->req.sec_read_bmp.len, MAX_SIMPLE_PAYLOAD_SIZE);
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.sec_read_bmp.len);
|
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.sec_read_bmp.len);
|
||||||
cl->recv_list.push_back(cur_op->buf, cur_op->req.sec_read_bmp.len);
|
cl->recv_list.push_back(cur_op->buf, cur_op->req.sec_read_bmp.len);
|
||||||
}
|
}
|
||||||
@@ -328,6 +350,15 @@ void osd_messenger_t::handle_op_hdr(osd_client_t *cl)
|
|||||||
{
|
{
|
||||||
if (cur_op->req.rw.len > 0)
|
if (cur_op->req.rw.len > 0)
|
||||||
{
|
{
|
||||||
|
if (cur_op->req.rw.len > max_write_request_size)
|
||||||
|
{
|
||||||
|
if (log_level > 1)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Error: peer %ju write request size too large (%u > %u bytes), stopping\n", cl->client_id,
|
||||||
|
cur_op->req.rw.len, max_write_request_size);
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.rw.len);
|
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.rw.len);
|
||||||
cl->recv_list.push_back(cur_op->buf, cur_op->req.rw.len);
|
cl->recv_list.push_back(cur_op->buf, cur_op->req.rw.len);
|
||||||
}
|
}
|
||||||
@@ -337,6 +368,15 @@ void osd_messenger_t::handle_op_hdr(osd_client_t *cl)
|
|||||||
{
|
{
|
||||||
if (cur_op->req.show_conf.json_len > 0)
|
if (cur_op->req.show_conf.json_len > 0)
|
||||||
{
|
{
|
||||||
|
if (cur_op->req.show_conf.json_len > MAX_SIMPLE_PAYLOAD_SIZE)
|
||||||
|
{
|
||||||
|
if (log_level > 1)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Error: peer %ju show_config request length too large (%lu > %u bytes), stopping\n", cl->client_id,
|
||||||
|
cur_op->req.show_conf.json_len, MAX_SIMPLE_PAYLOAD_SIZE);
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
cur_op->buf = malloc_or_die(cur_op->req.show_conf.json_len+1);
|
cur_op->buf = malloc_or_die(cur_op->req.show_conf.json_len+1);
|
||||||
((uint8_t*)cur_op->buf)[cur_op->req.show_conf.json_len] = 0;
|
((uint8_t*)cur_op->buf)[cur_op->req.show_conf.json_len] = 0;
|
||||||
cl->recv_list.push_back(cur_op->buf, cur_op->req.show_conf.json_len);
|
cl->recv_list.push_back(cur_op->buf, cur_op->req.show_conf.json_len);
|
||||||
@@ -362,16 +402,17 @@ void osd_messenger_t::handle_op_hdr(osd_client_t *cl)
|
|||||||
cl->read_op = NULL;
|
cl->read_op = NULL;
|
||||||
cl->read_state = 0;
|
cl->read_state = 0;
|
||||||
}
|
}
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
bool osd_messenger_t::handle_reply_hdr(osd_client_t *cl)
|
bool osd_messenger_t::handle_reply_hdr(osd_client_t *cl)
|
||||||
{
|
{
|
||||||
auto req_it = cl->sent_ops.find(cl->read_op->req.hdr.id);
|
auto req_it = cl->sent_ops.find(cl->read_op->req.hdr.id);
|
||||||
if (req_it == cl->sent_ops.end())
|
if (req_it == cl->sent_ops.end() || req_it->second->req.hdr.opcode != cl->read_op->req.hdr.opcode)
|
||||||
{
|
{
|
||||||
// Command out of sync. Drop connection
|
// Command out of sync. Drop connection
|
||||||
fprintf(stderr, "Client %d command out of sync: id %ju\n", cl->peer_fd, cl->read_op->req.hdr.id);
|
fprintf(stderr, "Client %ju command out of sync: id %ju\n", cl->client_id, cl->read_op->req.hdr.id);
|
||||||
stop_client(cl->peer_fd);
|
stop_client(cl->client_id);
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
osd_op_t *op = req_it->second;
|
osd_op_t *op = req_it->second;
|
||||||
@@ -385,13 +426,13 @@ bool osd_messenger_t::handle_reply_hdr(osd_client_t *cl)
|
|||||||
if (op->reply.hdr.retval >= 0 && (op->reply.hdr.retval != expected_size || bmp_len > op->bitmap_len))
|
if (op->reply.hdr.retval >= 0 && (op->reply.hdr.retval != expected_size || bmp_len > op->bitmap_len))
|
||||||
{
|
{
|
||||||
// Check reply length to not overflow the buffer
|
// Check reply length to not overflow the buffer
|
||||||
fprintf(stderr, "Client %d read reply of different length: expected %u+%u, got %jd+%u\n",
|
fprintf(stderr, "Client %ju read reply of different length: expected %u+%u, got %jd+%u\n",
|
||||||
cl->peer_fd, expected_size, op->bitmap_len, op->reply.hdr.retval, bmp_len);
|
cl->client_id, expected_size, op->bitmap_len, op->reply.hdr.retval, bmp_len);
|
||||||
cl->sent_ops[op->req.hdr.id] = op;
|
cl->sent_ops[op->req.hdr.id] = op;
|
||||||
stop_client(cl->peer_fd);
|
stop_client(cl->client_id);
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
if (bmp_len > 0)
|
if (op->reply.hdr.retval >= 0 && bmp_len > 0)
|
||||||
{
|
{
|
||||||
assert(op->bitmap);
|
assert(op->bitmap);
|
||||||
cl->recv_list.push_back(op->bitmap, bmp_len);
|
cl->recv_list.push_back(op->bitmap, bmp_len);
|
||||||
|
|||||||
+34
-76
@@ -6,11 +6,18 @@
|
|||||||
#include <sys/epoll.h>
|
#include <sys/epoll.h>
|
||||||
|
|
||||||
#include "messenger.h"
|
#include "messenger.h"
|
||||||
|
#include "msgr_iothread.h"
|
||||||
|
|
||||||
void osd_messenger_t::outbox_push(osd_op_t *cur_op)
|
void osd_messenger_t::outbox_push(osd_op_t *cur_op)
|
||||||
{
|
{
|
||||||
assert(cur_op->peer_fd);
|
assert(cur_op->client_id);
|
||||||
osd_client_t *cl = clients.at(cur_op->peer_fd);
|
auto cl_it = clients.find(cur_op->client_id);
|
||||||
|
if (cl_it == clients.end() || cl_it->second->peer_state == PEER_STOPPED)
|
||||||
|
{
|
||||||
|
delete cur_op;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
osd_client_t *cl = cl_it->second;
|
||||||
if (cur_op->op_type == OSD_OP_OUT)
|
if (cur_op->op_type == OSD_OP_OUT)
|
||||||
{
|
{
|
||||||
clock_gettime(CLOCK_REALTIME, &cur_op->tv_begin);
|
clock_gettime(CLOCK_REALTIME, &cur_op->tv_begin);
|
||||||
@@ -18,8 +25,7 @@ void osd_messenger_t::outbox_push(osd_op_t *cur_op)
|
|||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
// Check that operation actually belongs to this client
|
// Remove the operation from received op list
|
||||||
// FIXME: Review if this is still needed
|
|
||||||
bool found = false;
|
bool found = false;
|
||||||
for (auto it = cl->received_ops.begin(); it != cl->received_ops.end(); it++)
|
for (auto it = cl->received_ops.begin(); it != cl->received_ops.end(); it++)
|
||||||
{
|
{
|
||||||
@@ -30,11 +36,8 @@ void osd_messenger_t::outbox_push(osd_op_t *cur_op)
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!found)
|
// Can't be not found because client IDs are unique
|
||||||
{
|
assert(found);
|
||||||
delete cur_op;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
auto & to_send_list = cl->write_msg.msg_iovlen ? cl->next_send_list : cl->send_list;
|
auto & to_send_list = cl->write_msg.msg_iovlen ? cl->next_send_list : cl->send_list;
|
||||||
auto & to_outbox = cl->write_msg.msg_iovlen ? cl->next_outbox : cl->outbox;
|
auto & to_outbox = cl->write_msg.msg_iovlen ? cl->next_outbox : cl->outbox;
|
||||||
@@ -97,10 +100,15 @@ void osd_messenger_t::outbox_push(osd_op_t *cur_op)
|
|||||||
if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ_BMP)
|
if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ_BMP)
|
||||||
{
|
{
|
||||||
if (cur_op->op_type == OSD_OP_IN && cur_op->reply.hdr.retval > 0)
|
if (cur_op->op_type == OSD_OP_IN && cur_op->reply.hdr.retval > 0)
|
||||||
|
{
|
||||||
to_send_list.push_back((iovec){ .iov_base = cur_op->buf, .iov_len = (size_t)cur_op->reply.hdr.retval });
|
to_send_list.push_back((iovec){ .iov_base = cur_op->buf, .iov_len = (size_t)cur_op->reply.hdr.retval });
|
||||||
|
to_outbox.push_back((msgr_sendp_t){ .op = cur_op, .flags = 0 });
|
||||||
|
}
|
||||||
else if (cur_op->op_type == OSD_OP_OUT && cur_op->req.sec_read_bmp.len > 0)
|
else if (cur_op->op_type == OSD_OP_OUT && cur_op->req.sec_read_bmp.len > 0)
|
||||||
|
{
|
||||||
to_send_list.push_back((iovec){ .iov_base = cur_op->buf, .iov_len = (size_t)cur_op->req.sec_read_bmp.len });
|
to_send_list.push_back((iovec){ .iov_base = cur_op->buf, .iov_len = (size_t)cur_op->req.sec_read_bmp.len });
|
||||||
to_outbox.push_back((msgr_sendp_t){ .op = cur_op, .flags = 0 });
|
to_outbox.push_back((msgr_sendp_t){ .op = cur_op, .flags = 0 });
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if (cur_op->op_type == OSD_OP_IN)
|
if (cur_op->op_type == OSD_OP_IN)
|
||||||
{
|
{
|
||||||
@@ -126,72 +134,22 @@ void osd_messenger_t::outbox_push(osd_op_t *cur_op)
|
|||||||
if ((cl->write_msg.msg_iovlen > 0 || !try_send(cl)) && (cl->write_state == 0))
|
if ((cl->write_msg.msg_iovlen > 0 || !try_send(cl)) && (cl->write_state == 0))
|
||||||
{
|
{
|
||||||
cl->write_state = CL_WRITE_READY;
|
cl->write_state = CL_WRITE_READY;
|
||||||
write_ready_clients.push_back(cur_op->peer_fd);
|
write_ready_clients.push_back(cur_op->client_id);
|
||||||
}
|
}
|
||||||
ringloop->wakeup();
|
ringloop->wakeup();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void osd_messenger_t::inc_op_stats(osd_op_stats_t & stats, uint64_t opcode, timespec & tv_begin, timespec & tv_end, uint64_t len)
|
|
||||||
{
|
|
||||||
uint64_t usecs = (
|
|
||||||
(tv_end.tv_sec - tv_begin.tv_sec)*1000000 +
|
|
||||||
(tv_end.tv_nsec - tv_begin.tv_nsec)/1000
|
|
||||||
);
|
|
||||||
stats.op_stat_count[opcode]++;
|
|
||||||
if (!stats.op_stat_count[opcode])
|
|
||||||
{
|
|
||||||
stats.op_stat_count[opcode] = 1;
|
|
||||||
stats.op_stat_sum[opcode] = 0;
|
|
||||||
stats.op_stat_bytes[opcode] = 0;
|
|
||||||
}
|
|
||||||
stats.op_stat_sum[opcode] += usecs;
|
|
||||||
stats.op_stat_bytes[opcode] += len;
|
|
||||||
}
|
|
||||||
|
|
||||||
void osd_messenger_t::measure_exec(osd_op_t *cur_op)
|
|
||||||
{
|
|
||||||
// Measure execution latency
|
|
||||||
if (cur_op->req.hdr.opcode > OSD_OP_MAX)
|
|
||||||
{
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
if (!cur_op->tv_end.tv_sec)
|
|
||||||
{
|
|
||||||
clock_gettime(CLOCK_REALTIME, &cur_op->tv_end);
|
|
||||||
}
|
|
||||||
uint64_t len = 0;
|
|
||||||
if (cur_op->req.hdr.opcode == OSD_OP_READ ||
|
|
||||||
cur_op->req.hdr.opcode == OSD_OP_WRITE ||
|
|
||||||
cur_op->req.hdr.opcode == OSD_OP_SCRUB)
|
|
||||||
{
|
|
||||||
// req.rw.len is internally set to the full object size for scrubs
|
|
||||||
len = cur_op->req.rw.len;
|
|
||||||
}
|
|
||||||
else if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ ||
|
|
||||||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
|
|
||||||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE)
|
|
||||||
{
|
|
||||||
len = cur_op->req.sec_rw.len;
|
|
||||||
}
|
|
||||||
inc_op_stats(stats, cur_op->req.hdr.opcode, cur_op->tv_begin, cur_op->tv_end, len);
|
|
||||||
if (cur_op->is_recovery_related())
|
|
||||||
{
|
|
||||||
inc_op_stats(recovery_stats, cur_op->req.hdr.opcode, cur_op->tv_begin, cur_op->tv_end, len);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
bool osd_messenger_t::try_send(osd_client_t *cl)
|
bool osd_messenger_t::try_send(osd_client_t *cl)
|
||||||
{
|
{
|
||||||
int peer_fd = cl->peer_fd;
|
if (!cl->send_list.size() || cl->write_msg.msg_iovlen > 0 || cl->peer_state == PEER_STOPPED || cl->peer_fd < 0)
|
||||||
if (!cl->send_list.size() || cl->write_msg.msg_iovlen > 0)
|
|
||||||
{
|
{
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
assert(cl->peer_state != PEER_RDMA);
|
assert(cl->peer_state != PEER_RDMA);
|
||||||
if (ringloop && !use_sync_send_recv)
|
if (ringloop && !use_sync_send_recv)
|
||||||
{
|
{
|
||||||
auto iothread = iothreads.size() ? iothreads[peer_fd % iothreads.size()] : NULL;
|
auto iothread = iothreads.size() ? iothreads[cl->peer_fd % iothreads.size()] : NULL;
|
||||||
io_uring_sqe sqe_local;
|
io_uring_sqe sqe_local;
|
||||||
ring_data_t data_local;
|
ring_data_t data_local;
|
||||||
io_uring_sqe* sqe = (iothread ? &sqe_local : ringloop->get_sqe());
|
io_uring_sqe* sqe = (iothread ? &sqe_local : ringloop->get_sqe());
|
||||||
@@ -218,11 +176,11 @@ bool osd_messenger_t::try_send(osd_client_t *cl)
|
|||||||
}
|
}
|
||||||
if (use_zc)
|
if (use_zc)
|
||||||
{
|
{
|
||||||
io_uring_prep_sendmsg_zc(sqe, peer_fd, &cl->write_msg, MSG_WAITALL);
|
io_uring_prep_sendmsg_zc(sqe, cl->peer_fd, &cl->write_msg, MSG_WAITALL);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
io_uring_prep_sendmsg(sqe, peer_fd, &cl->write_msg, MSG_WAITALL);
|
io_uring_prep_sendmsg(sqe, cl->peer_fd, &cl->write_msg, MSG_WAITALL);
|
||||||
}
|
}
|
||||||
if (iothread)
|
if (iothread)
|
||||||
{
|
{
|
||||||
@@ -234,7 +192,7 @@ bool osd_messenger_t::try_send(osd_client_t *cl)
|
|||||||
cl->write_msg.msg_iov = cl->send_list.data();
|
cl->write_msg.msg_iov = cl->send_list.data();
|
||||||
cl->write_msg.msg_iovlen = cl->send_list.size() < IOV_MAX ? cl->send_list.size() : IOV_MAX;
|
cl->write_msg.msg_iovlen = cl->send_list.size() < IOV_MAX ? cl->send_list.size() : IOV_MAX;
|
||||||
cl->refs++;
|
cl->refs++;
|
||||||
int result = sendmsg(peer_fd, &cl->write_msg, MSG_NOSIGNAL);
|
int result = sendmsg(cl->peer_fd, &cl->write_msg, MSG_NOSIGNAL);
|
||||||
if (result < 0)
|
if (result < 0)
|
||||||
{
|
{
|
||||||
result = -errno;
|
result = -errno;
|
||||||
@@ -249,8 +207,8 @@ void osd_messenger_t::send_replies()
|
|||||||
{
|
{
|
||||||
for (int i = 0; i < write_ready_clients.size(); i++)
|
for (int i = 0; i < write_ready_clients.size(); i++)
|
||||||
{
|
{
|
||||||
int peer_fd = write_ready_clients[i];
|
uint64_t client_id = write_ready_clients[i];
|
||||||
auto cl_it = clients.find(peer_fd);
|
auto cl_it = clients.find(client_id);
|
||||||
if (cl_it != clients.end() && cl_it->second->peer_state != PEER_RDMA && !try_send(cl_it->second))
|
if (cl_it != clients.end() && cl_it->second->peer_state != PEER_RDMA && !try_send(cl_it->second))
|
||||||
{
|
{
|
||||||
write_ready_clients.erase(write_ready_clients.begin(), write_ready_clients.begin() + i);
|
write_ready_clients.erase(write_ready_clients.begin(), write_ready_clients.begin() + i);
|
||||||
@@ -274,15 +232,15 @@ void osd_messenger_t::handle_send(int result, bool prev, bool more, osd_client_t
|
|||||||
{
|
{
|
||||||
if (cl->refs <= 0)
|
if (cl->refs <= 0)
|
||||||
{
|
{
|
||||||
delete cl;
|
destroy_client(cl);
|
||||||
}
|
}
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if (result < 0 && result != -EAGAIN && result != -EINTR)
|
if (result < 0 && result != -EAGAIN && result != -EINTR)
|
||||||
{
|
{
|
||||||
// this is a client socket, so don't panic. just disconnect it
|
// this is a client socket, so don't panic. just disconnect it
|
||||||
fprintf(stderr, "Client %d socket write error: %d (%s). Disconnecting client\n", cl->peer_fd, -result, strerror(-result));
|
fprintf(stderr, "Client %ju socket write error: %d (%s). Disconnecting client\n", cl->client_id, -result, strerror(-result));
|
||||||
stop_client(cl->peer_fd);
|
stop_client(cl->client_id);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if (result >= 0)
|
if (result >= 0)
|
||||||
@@ -326,9 +284,9 @@ void osd_messenger_t::handle_send(int result, bool prev, bool more, osd_client_t
|
|||||||
int expected = cl->send_list.size() < IOV_MAX ? cl->send_list.size() : IOV_MAX;
|
int expected = cl->send_list.size() < IOV_MAX ? cl->send_list.size() : IOV_MAX;
|
||||||
if (done != expected)
|
if (done != expected)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Client %d socket write error: expected to send "
|
fprintf(stderr, "Client %ju socket write error: expected to send "
|
||||||
"%d iovecs with MSG_WAITALL but sent %d. Disconnecting client\n", cl->peer_fd, expected, done);
|
"%d iovecs with MSG_WAITALL but sent %d. Disconnecting client\n", cl->client_id, expected, done);
|
||||||
stop_client(cl->peer_fd);
|
stop_client(cl->client_id);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
cl->zc_free_list.push_back(NULL); // end marker
|
cl->zc_free_list.push_back(NULL); // end marker
|
||||||
@@ -352,7 +310,7 @@ void osd_messenger_t::handle_send(int result, bool prev, bool more, osd_client_t
|
|||||||
// FIXME: Ignore pings during RDMA state transition
|
// FIXME: Ignore pings during RDMA state transition
|
||||||
if (log_level > 0)
|
if (log_level > 0)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Successfully connected with client %d using RDMA\n", cl->peer_fd);
|
fprintf(stderr, "Successfully connected with client %ju using RDMA\n", cl->client_id);
|
||||||
}
|
}
|
||||||
cl->peer_state = PEER_RDMA;
|
cl->peer_state = PEER_RDMA;
|
||||||
// Add the initial receive request
|
// Add the initial receive request
|
||||||
@@ -362,6 +320,6 @@ void osd_messenger_t::handle_send(int result, bool prev, bool more, osd_client_t
|
|||||||
}
|
}
|
||||||
if (cl->write_state != 0)
|
if (cl->write_state != 0)
|
||||||
{
|
{
|
||||||
write_ready_clients.push_back(cl->peer_fd);
|
write_ready_clients.push_back(cl->client_id);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+61
-60
@@ -5,9 +5,6 @@
|
|||||||
#include <assert.h>
|
#include <assert.h>
|
||||||
|
|
||||||
#include "messenger.h"
|
#include "messenger.h"
|
||||||
#ifdef WITH_RDMA
|
|
||||||
#include "msgr_rdma.h"
|
|
||||||
#endif
|
|
||||||
|
|
||||||
void osd_client_t::cancel_ops()
|
void osd_client_t::cancel_ops()
|
||||||
{
|
{
|
||||||
@@ -43,34 +40,40 @@ void osd_op_t::cancel()
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void osd_messenger_t::stop_client(int peer_fd, bool force, bool force_delete)
|
// force_delete means stop the client anyway, even if there are refs to it in the event loop.
|
||||||
|
// the flag should be used in the destructor.
|
||||||
|
// why? - because yes, we could close the FD first and let it fail all requests in the event loop,
|
||||||
|
// but in that case it can be quickly reopened and we can get old failed responses for the new FD.
|
||||||
|
void osd_messenger_t::stop_client(uint64_t client_id, bool force_delete)
|
||||||
{
|
{
|
||||||
assert(peer_fd != 0);
|
auto it = clients.find(client_id);
|
||||||
auto it = clients.find(peer_fd);
|
if (!client_id || it == clients.end())
|
||||||
if (it == clients.end())
|
|
||||||
{
|
{
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
osd_client_t *cl = it->second;
|
osd_client_t *cl = it->second;
|
||||||
// FIXME: This 'force' flag is probably an ugly reenterability hack - check its logic and maybe remove it
|
if (cl->peer_state == PEER_STOPPED)
|
||||||
if (cl->peer_state == PEER_CONNECTING && !force || cl->peer_state == PEER_STOPPED)
|
|
||||||
{
|
{
|
||||||
|
if (force_delete)
|
||||||
|
{
|
||||||
|
destroy_client(cl);
|
||||||
|
}
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
clear_immediate_ops(peer_fd);
|
cl->received_ops.clear();
|
||||||
if (log_level > 0)
|
if (log_level > 0)
|
||||||
{
|
{
|
||||||
if (cl->osd_num)
|
if (cl->osd_num)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "[OSD %ju] Stopping client %d (OSD peer %ju)\n", osd_num, peer_fd, cl->osd_num);
|
fprintf(stderr, "[OSD %ju] Stopping client %ju (OSD peer %ju)\n", osd_num, client_id, cl->osd_num);
|
||||||
}
|
}
|
||||||
else if (cl->in_osd_num)
|
else if (cl->in_osd_num)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "[OSD %ju] Stopping client %d (incoming OSD peer %ju)\n", osd_num, peer_fd, cl->in_osd_num);
|
fprintf(stderr, "[OSD %ju] Stopping client %ju (incoming OSD peer %ju)\n", osd_num, client_id, cl->in_osd_num);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
fprintf(stderr, "[OSD %ju] Stopping client %d (regular client)\n", osd_num, peer_fd);
|
fprintf(stderr, "[OSD %ju] Stopping client %ju (regular client)\n", osd_num, client_id);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// First set state to STOPPED so another stop_client() call doesn't try to free it again
|
// First set state to STOPPED so another stop_client() call doesn't try to free it again
|
||||||
@@ -79,48 +82,18 @@ void osd_messenger_t::stop_client(int peer_fd, bool force, bool force_delete)
|
|||||||
cl->peer_state = PEER_STOPPED;
|
cl->peer_state = PEER_STOPPED;
|
||||||
if (cl->osd_num)
|
if (cl->osd_num)
|
||||||
{
|
{
|
||||||
auto osd_it = osd_peer_fds.find(cl->osd_num);
|
auto osd_it = osd_peers.find(cl->osd_num);
|
||||||
if (osd_it != osd_peer_fds.end() && osd_it->second == cl->peer_fd)
|
if (osd_it != osd_peers.end() && osd_it->second == cl)
|
||||||
{
|
{
|
||||||
// ...and forget OSD peer
|
// ...and forget OSD peer
|
||||||
osd_peer_fds.erase(osd_it);
|
osd_peers.erase(osd_it);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
#ifdef WITH_RDMA
|
|
||||||
if (cl->rdma_conn && cl->rdma_conn->cmid)
|
|
||||||
{
|
|
||||||
auto rdma_it = rdmacm_connections.find(cl->rdma_conn->cmid);
|
|
||||||
if (rdma_it != rdmacm_connections.end() && rdma_it->second == cl)
|
|
||||||
{
|
|
||||||
rdmacm_connections.erase(rdma_it);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
#endif
|
|
||||||
#ifndef __MOCK__
|
|
||||||
// Then remove FD from the eventloop so we don't accidentally read something
|
|
||||||
tfd->set_fd_handler(peer_fd, false, NULL);
|
|
||||||
if (cl->connect_timeout_id >= 0)
|
if (cl->connect_timeout_id >= 0)
|
||||||
{
|
{
|
||||||
tfd->clear_timer(cl->connect_timeout_id);
|
tfd->clear_timer(cl->connect_timeout_id);
|
||||||
cl->connect_timeout_id = -1;
|
cl->connect_timeout_id = -1;
|
||||||
}
|
}
|
||||||
for (auto rit = read_ready_clients.begin(); rit != read_ready_clients.end(); rit++)
|
|
||||||
{
|
|
||||||
if (*rit == peer_fd)
|
|
||||||
{
|
|
||||||
read_ready_clients.erase(rit);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (auto wit = write_ready_clients.begin(); wit != write_ready_clients.end(); wit++)
|
|
||||||
{
|
|
||||||
if (*wit == peer_fd)
|
|
||||||
{
|
|
||||||
write_ready_clients.erase(wit);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
#endif
|
|
||||||
if (cl->in_osd_num && break_pg_locks)
|
if (cl->in_osd_num && break_pg_locks)
|
||||||
{
|
{
|
||||||
// Break PG locks
|
// Break PG locks
|
||||||
@@ -134,19 +107,56 @@ void osd_messenger_t::stop_client(int peer_fd, bool force, bool force_delete)
|
|||||||
// so do not repeer on it.
|
// so do not repeer on it.
|
||||||
repeer_pgs(cl->osd_num);
|
repeer_pgs(cl->osd_num);
|
||||||
}
|
}
|
||||||
// Find the item again because it can be invalidated at this point
|
if (cl->peer_fd >= 0)
|
||||||
it = clients.find(peer_fd);
|
|
||||||
if (it != clients.end())
|
|
||||||
{
|
{
|
||||||
clients.erase(it);
|
int r = shutdown(cl->peer_fd, SHUT_RDWR);
|
||||||
|
if (r != 0 && errno != ENOTCONN)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "[OSD %ju] failed to shutdown a socket: %s (code %d)\n", osd_num, strerror(errno), errno);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
cl->refs--;
|
cl->refs--;
|
||||||
if (cl->refs <= 0 || force_delete)
|
if (cl->refs <= 0 || force_delete)
|
||||||
{
|
{
|
||||||
delete cl;
|
destroy_client(cl);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void osd_messenger_t::destroy_client(osd_client_t *cl)
|
||||||
|
{
|
||||||
|
// Find the item again because it can be invalidated at this point
|
||||||
|
clients.erase(cl->client_id);
|
||||||
|
if (cl->peer_fd >= 0)
|
||||||
|
{
|
||||||
|
tfd->set_fd_handler(cl->peer_fd, false, NULL);
|
||||||
|
for (auto rit = read_ready_clients.begin(); rit != read_ready_clients.end(); rit++)
|
||||||
|
{
|
||||||
|
if (*rit == cl->client_id)
|
||||||
|
{
|
||||||
|
read_ready_clients.erase(rit);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (auto wit = write_ready_clients.begin(); wit != write_ready_clients.end(); wit++)
|
||||||
|
{
|
||||||
|
if (*wit == cl->client_id)
|
||||||
|
{
|
||||||
|
write_ready_clients.erase(wit);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
clients_by_fd.erase(cl->peer_fd);
|
||||||
|
}
|
||||||
|
#ifdef WITH_RDMA
|
||||||
|
if (cl->rdma_conn)
|
||||||
|
{
|
||||||
|
destroy_rdma_conn(cl->rdma_conn);
|
||||||
|
cl->rdma_conn = NULL;
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
delete cl;
|
||||||
|
}
|
||||||
|
|
||||||
osd_client_t::~osd_client_t()
|
osd_client_t::~osd_client_t()
|
||||||
{
|
{
|
||||||
free(in_buf);
|
free(in_buf);
|
||||||
@@ -176,13 +186,4 @@ osd_client_t::~osd_client_t()
|
|||||||
delete op;
|
delete op;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
#ifndef __MOCK__
|
|
||||||
#ifdef WITH_RDMA
|
|
||||||
if (rdma_conn)
|
|
||||||
{
|
|
||||||
delete rdma_conn;
|
|
||||||
rdma_conn = NULL;
|
|
||||||
}
|
|
||||||
#endif
|
|
||||||
#endif
|
|
||||||
}
|
}
|
||||||
|
|||||||
+12
-11
@@ -301,10 +301,9 @@ const char *help_text =
|
|||||||
" --nbd_disconnect_on_close 1\n"
|
" --nbd_disconnect_on_close 1\n"
|
||||||
" Disconnect the nbd device on close by last opener.\n"
|
" Disconnect the nbd device on close by last opener.\n"
|
||||||
#endif
|
#endif
|
||||||
#ifdef NBD_FLAG_READ_ONLY
|
" --readonly\n"
|
||||||
" --nbd_ro 1\n"
|
" --nbd_ro 1\n"
|
||||||
" Set device into read only mode.\n"
|
" Set device into read only mode.\n"
|
||||||
#endif
|
|
||||||
"\n"
|
"\n"
|
||||||
"vitastor-nbd netlink-unmap /dev/nbdN\n"
|
"vitastor-nbd netlink-unmap /dev/nbdN\n"
|
||||||
" Unmap a device using netlink interface. Works with both netlink and ioctl mapped devices.\n"
|
" Unmap a device using netlink interface. Works with both netlink and ioctl mapped devices.\n"
|
||||||
@@ -348,6 +347,7 @@ protected:
|
|||||||
int read_ready = 0;
|
int read_ready = 0;
|
||||||
msghdr read_msg = { 0 }, send_msg = { 0 };
|
msghdr read_msg = { 0 }, send_msg = { 0 };
|
||||||
iovec read_iov = { 0 };
|
iovec read_iov = { 0 };
|
||||||
|
bool stop = false;
|
||||||
|
|
||||||
std::string logfile = "/dev/null";
|
std::string logfile = "/dev/null";
|
||||||
|
|
||||||
@@ -514,7 +514,7 @@ help:
|
|||||||
// Create client
|
// Create client
|
||||||
ringloop = new ring_loop_t(RINGLOOP_DEFAULT_SIZE);
|
ringloop = new ring_loop_t(RINGLOOP_DEFAULT_SIZE);
|
||||||
epmgr = new epoll_manager_t(ringloop);
|
epmgr = new epoll_manager_t(ringloop);
|
||||||
cli = new cluster_client_t(ringloop, epmgr->tfd, cfg);
|
cli = cluster_client_t::create(ringloop, epmgr->tfd, cfg);
|
||||||
if (!inode)
|
if (!inode)
|
||||||
{
|
{
|
||||||
// Load image metadata
|
// Load image metadata
|
||||||
@@ -525,7 +525,7 @@ help:
|
|||||||
break;
|
break;
|
||||||
ringloop->wait();
|
ringloop->wait();
|
||||||
}
|
}
|
||||||
watch = cli->st_cli.watch_inode(image_name);
|
watch = cli->st_cli->watch_inode(image_name);
|
||||||
device_size = watch->cfg.size;
|
device_size = watch->cfg.size;
|
||||||
if (!watch->cfg.num || !device_size)
|
if (!watch->cfg.num || !device_size)
|
||||||
{
|
{
|
||||||
@@ -581,10 +581,8 @@ help:
|
|||||||
}
|
}
|
||||||
uint64_t flags = NBD_FLAG_SEND_FLUSH;
|
uint64_t flags = NBD_FLAG_SEND_FLUSH;
|
||||||
uint64_t cflags = 0;
|
uint64_t cflags = 0;
|
||||||
#ifdef NBD_FLAG_READ_ONLY
|
if (!cfg["readonly"].is_null() || !cfg["nbd_ro"].is_null())
|
||||||
if (!cfg["nbd_ro"].is_null())
|
|
||||||
flags |= NBD_FLAG_READ_ONLY;
|
flags |= NBD_FLAG_READ_ONLY;
|
||||||
#endif
|
|
||||||
#ifdef NBD_CFLAG_DESTROY_ON_DISCONNECT
|
#ifdef NBD_CFLAG_DESTROY_ON_DISCONNECT
|
||||||
if (!cfg["nbd_destroy_on_disconnect"].is_null())
|
if (!cfg["nbd_destroy_on_disconnect"].is_null())
|
||||||
cflags |= NBD_CFLAG_DESTROY_ON_DISCONNECT;
|
cflags |= NBD_CFLAG_DESTROY_ON_DISCONNECT;
|
||||||
@@ -620,7 +618,10 @@ help:
|
|||||||
if (!cfg["dev_num"].is_null())
|
if (!cfg["dev_num"].is_null())
|
||||||
{
|
{
|
||||||
int r;
|
int r;
|
||||||
if ((r = run_nbd(sockfd, cfg["dev_num"].int64_value(), device_size, NBD_FLAG_SEND_FLUSH, nbd_timeout, bg)) != 0)
|
uint64_t flags = NBD_FLAG_SEND_FLUSH;
|
||||||
|
if (!cfg["readonly"].is_null())
|
||||||
|
flags |= NBD_FLAG_READ_ONLY;
|
||||||
|
if ((r = run_nbd(sockfd, cfg["dev_num"].int64_value(), device_size, flags, nbd_timeout, bg)) != 0)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "run_nbd: %s\n", strerror(-r));
|
fprintf(stderr, "run_nbd: %s\n", strerror(-r));
|
||||||
exit(1);
|
exit(1);
|
||||||
@@ -678,8 +679,7 @@ help:
|
|||||||
};
|
};
|
||||||
ringloop->register_consumer(&consumer);
|
ringloop->register_consumer(&consumer);
|
||||||
// Add FD to epoll
|
// Add FD to epoll
|
||||||
bool stop = false;
|
epmgr->tfd->set_fd_handler(sockfd[0], false, [this](int peer_fd, int epoll_events)
|
||||||
epmgr->tfd->set_fd_handler(sockfd[0], false, [this, &stop](int peer_fd, int epoll_events)
|
|
||||||
{
|
{
|
||||||
if (epoll_events & EPOLLRDHUP)
|
if (epoll_events & EPOLLRDHUP)
|
||||||
{
|
{
|
||||||
@@ -1118,7 +1118,8 @@ protected:
|
|||||||
{
|
{
|
||||||
// Disconnect
|
// Disconnect
|
||||||
close(nbd_fd);
|
close(nbd_fd);
|
||||||
exit(0);
|
stop = true;
|
||||||
|
return;
|
||||||
}
|
}
|
||||||
if (be32toh(cur_req.magic) != NBD_REQUEST_MAGIC ||
|
if (be32toh(cur_req.magic) != NBD_REQUEST_MAGIC ||
|
||||||
req_type != NBD_CMD_READ && req_type != NBD_CMD_WRITE && req_type != NBD_CMD_FLUSH)
|
req_type != NBD_CMD_READ && req_type != NBD_CMD_WRITE && req_type != NBD_CMD_FLUSH)
|
||||||
|
|||||||
@@ -705,6 +705,54 @@ static void vitastor_close(BlockDriverState *bs)
|
|||||||
client->last_bitmap = NULL;
|
client->last_bitmap = NULL;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Unregister all event sources from the current AioContext. Called by the
|
||||||
|
// block layer before bs is moved to a different AioContext (e.g. during live
|
||||||
|
// migration, drain, dataplane switching). The block layer guarantees that no
|
||||||
|
// requests are in flight at this point.
|
||||||
|
static void vitastor_detach_aio_context(BlockDriverState *bs)
|
||||||
|
{
|
||||||
|
VitastorClient *client = bs->opaque;
|
||||||
|
int i;
|
||||||
|
#if defined VITASTOR_C_API_VERSION && VITASTOR_C_API_VERSION >= 2
|
||||||
|
if (client->uring_eventfd >= 0)
|
||||||
|
{
|
||||||
|
universal_aio_set_fd_handler(client->ctx, client->uring_eventfd, NULL, NULL, NULL);
|
||||||
|
// Wait until any scheduled B/H is processed before switching contexts:
|
||||||
|
// it would otherwise fire on the old context with stale state.
|
||||||
|
if (client->bh_uring_scheduled)
|
||||||
|
{
|
||||||
|
BDRV_POLL_WHILE(bs, client->bh_uring_scheduled);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
for (i = 0; i < client->fd_count; i++)
|
||||||
|
{
|
||||||
|
universal_aio_set_fd_handler(client->ctx, client->fds[i]->fd, NULL, NULL, NULL);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// (Re-)register all event sources on the new AioContext.
|
||||||
|
static void vitastor_attach_aio_context(BlockDriverState *bs, AioContext *new_ctx)
|
||||||
|
{
|
||||||
|
VitastorClient *client = bs->opaque;
|
||||||
|
int i;
|
||||||
|
client->ctx = new_ctx;
|
||||||
|
#if defined VITASTOR_C_API_VERSION && VITASTOR_C_API_VERSION >= 2
|
||||||
|
if (client->uring_eventfd >= 0)
|
||||||
|
{
|
||||||
|
universal_aio_set_fd_handler(new_ctx, client->uring_eventfd, vitastor_uring_handler, NULL, client);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
for (i = 0; i < client->fd_count; i++)
|
||||||
|
{
|
||||||
|
VitastorFdData *fdd = client->fds[i];
|
||||||
|
universal_aio_set_fd_handler(new_ctx, fdd->fd,
|
||||||
|
fdd->fd_read ? vitastor_aio_fd_read : NULL,
|
||||||
|
fdd->fd_write ? vitastor_aio_fd_write : NULL,
|
||||||
|
fdd);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#if QEMU_VERSION_MAJOR >= 3 || QEMU_VERSION_MAJOR == 2 && QEMU_VERSION_MINOR >= 2
|
#if QEMU_VERSION_MAJOR >= 3 || QEMU_VERSION_MAJOR == 2 && QEMU_VERSION_MINOR >= 2
|
||||||
static void vitastor_refresh_filename(BlockDriverState *bs)
|
static void vitastor_refresh_filename(BlockDriverState *bs)
|
||||||
{
|
{
|
||||||
@@ -1049,7 +1097,7 @@ static int coroutine_fn vitastor_co_block_status(BlockDriverState *bs,
|
|||||||
{
|
{
|
||||||
// Get larger allocated extents, possibly with false positives
|
// Get larger allocated extents, possibly with false positives
|
||||||
uint64_t bmp_pos = (offset-task.offset) / task.bitmap_granularity;
|
uint64_t bmp_pos = (offset-task.offset) / task.bitmap_granularity;
|
||||||
uint64_t bmp_end = (offset+bytes-task.offset) / task.bitmap_granularity - bmp_pos;
|
uint64_t bmp_end = (offset+bytes-task.offset) / task.bitmap_granularity;
|
||||||
while (bmp_pos < bmp_end)
|
while (bmp_pos < bmp_end)
|
||||||
{
|
{
|
||||||
if (!(bmp_pos & 7) && bmp_end >= bmp_pos+8)
|
if (!(bmp_pos & 7) && bmp_end >= bmp_pos+8)
|
||||||
@@ -1188,6 +1236,9 @@ static BlockDriver bdrv_vitastor = {
|
|||||||
#endif
|
#endif
|
||||||
.bdrv_close = vitastor_close,
|
.bdrv_close = vitastor_close,
|
||||||
|
|
||||||
|
.bdrv_detach_aio_context = vitastor_detach_aio_context,
|
||||||
|
.bdrv_attach_aio_context = vitastor_attach_aio_context,
|
||||||
|
|
||||||
// Option list for the create operation
|
// Option list for the create operation
|
||||||
#if QEMU_VERSION_MAJOR >= 3 || QEMU_VERSION_MAJOR == 2 && QEMU_VERSION_MINOR > 0
|
#if QEMU_VERSION_MAJOR >= 3 || QEMU_VERSION_MAJOR == 2 && QEMU_VERSION_MINOR > 0
|
||||||
.create_opts = &vitastor_create_opts,
|
.create_opts = &vitastor_create_opts,
|
||||||
|
|||||||
@@ -62,6 +62,13 @@ const char *help_text =
|
|||||||
"All usual Vitastor config options like --config_path <path_to_config> may also be specified in CLI.\n"
|
"All usual Vitastor config options like --config_path <path_to_config> may also be specified in CLI.\n"
|
||||||
;
|
;
|
||||||
|
|
||||||
|
struct ublk_request
|
||||||
|
{
|
||||||
|
uint64_t ublk_cmd;
|
||||||
|
int index;
|
||||||
|
int result;
|
||||||
|
};
|
||||||
|
|
||||||
class ublk_server
|
class ublk_server
|
||||||
{
|
{
|
||||||
protected:
|
protected:
|
||||||
@@ -241,7 +248,7 @@ help:
|
|||||||
|
|
||||||
// Create client
|
// Create client
|
||||||
epmgr = new epoll_manager_t(ringloop);
|
epmgr = new epoll_manager_t(ringloop);
|
||||||
cli = new cluster_client_t(ringloop, epmgr->tfd, cfg);
|
cli = cluster_client_t::create(ringloop, epmgr->tfd, cfg);
|
||||||
|
|
||||||
// cli->config contains merged config
|
// cli->config contains merged config
|
||||||
if (!cfg["queue_depth"].is_null())
|
if (!cfg["queue_depth"].is_null())
|
||||||
@@ -273,7 +280,7 @@ help:
|
|||||||
}
|
}
|
||||||
if (!inode)
|
if (!inode)
|
||||||
{
|
{
|
||||||
watch = cli->st_cli.watch_inode(image_name);
|
watch = cli->st_cli->watch_inode(image_name);
|
||||||
device_size = watch->cfg.size;
|
device_size = watch->cfg.size;
|
||||||
if (!watch->cfg.num || !device_size)
|
if (!watch->cfg.num || !device_size)
|
||||||
{
|
{
|
||||||
@@ -282,9 +289,9 @@ help:
|
|||||||
exit(1);
|
exit(1);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
const bool writeback = cli->get_immediate_commit(inode);
|
const bool writeback = !cli->get_immediate_commit(inode ? inode : watch->cfg.num);
|
||||||
auto pool_it = cli->st_cli.pool_config.find(INODE_POOL(inode ? inode : watch->cfg.num));
|
auto pool_it = cli->st_cli->pool_config.find(INODE_POOL(inode ? inode : watch->cfg.num));
|
||||||
if (pool_it == cli->st_cli.pool_config.end())
|
if (pool_it == cli->st_cli->pool_config.end())
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Pool %u does not exist\n", INODE_POOL(inode ? inode : watch->cfg.num));
|
fprintf(stderr, "Pool %u does not exist\n", INODE_POOL(inode ? inode : watch->cfg.num));
|
||||||
exit(1);
|
exit(1);
|
||||||
@@ -331,6 +338,11 @@ help:
|
|||||||
daemonize_fork(notifyfd);
|
daemonize_fork(notifyfd);
|
||||||
close(notifyfd[0]);
|
close(notifyfd[0]);
|
||||||
}
|
}
|
||||||
|
consumer.loop = [this]()
|
||||||
|
{
|
||||||
|
submit_postponed();
|
||||||
|
};
|
||||||
|
ringloop->register_consumer(&consumer);
|
||||||
start_device(recover);
|
start_device(recover);
|
||||||
if (pidfile != "")
|
if (pidfile != "")
|
||||||
write_pid();
|
write_pid();
|
||||||
@@ -350,6 +362,7 @@ help:
|
|||||||
ringloop->wait();
|
ringloop->wait();
|
||||||
}
|
}
|
||||||
cli->flush();
|
cli->flush();
|
||||||
|
ringloop->unregister_consumer(&consumer);
|
||||||
delete cli;
|
delete cli;
|
||||||
delete epmgr;
|
delete epmgr;
|
||||||
cli = NULL;
|
cli = NULL;
|
||||||
@@ -553,6 +566,8 @@ protected:
|
|||||||
ublksrv_ctrl_dev_info ublk_dev = {};
|
ublksrv_ctrl_dev_info ublk_dev = {};
|
||||||
ublksrv_io_desc *ublk_queue = NULL;
|
ublksrv_io_desc *ublk_queue = NULL;
|
||||||
std::vector<uint8_t*> buffers;
|
std::vector<uint8_t*> buffers;
|
||||||
|
ring_consumer_t consumer;
|
||||||
|
std::vector<ublk_request> postponed_requests;
|
||||||
|
|
||||||
void open_control()
|
void open_control()
|
||||||
{
|
{
|
||||||
@@ -734,9 +749,19 @@ protected:
|
|||||||
ctrl_fd = -1;
|
ctrl_fd = -1;
|
||||||
}
|
}
|
||||||
|
|
||||||
void submit_request(uint64_t ublk_cmd, int i, int res)
|
bool submit_request(uint64_t ublk_cmd, int i, int res)
|
||||||
{
|
{
|
||||||
io_uring_sqe *sqe = ringloop->get_sqe();
|
io_uring_sqe *sqe = ringloop->get_sqe();
|
||||||
|
if (!sqe)
|
||||||
|
{
|
||||||
|
// Handle full io_uring by postponing the request
|
||||||
|
postponed_requests.push_back((ublk_request){
|
||||||
|
.ublk_cmd = ublk_cmd,
|
||||||
|
.index = i,
|
||||||
|
.result = res,
|
||||||
|
});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
ring_data_t* data = ((ring_data_t*)sqe->user_data);
|
ring_data_t* data = ((ring_data_t*)sqe->user_data);
|
||||||
sqe->fd = cdev_fd;
|
sqe->fd = cdev_fd;
|
||||||
sqe->opcode = IORING_OP_URING_CMD;
|
sqe->opcode = IORING_OP_URING_CMD;
|
||||||
@@ -750,6 +775,22 @@ protected:
|
|||||||
cmd->addr = (uint64_t)buffers[i];
|
cmd->addr = (uint64_t)buffers[i];
|
||||||
cmd->result = res;
|
cmd->result = res;
|
||||||
data->callback = [this, i](ring_data_t *data) { exec_request(data->res, i); };
|
data->callback = [this, i](ring_data_t *data) { exec_request(data->res, i); };
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
void submit_postponed()
|
||||||
|
{
|
||||||
|
int sent = 0;
|
||||||
|
while (postponed_requests.size())
|
||||||
|
{
|
||||||
|
ublk_request r = postponed_requests.back();
|
||||||
|
postponed_requests.pop_back();
|
||||||
|
if (!submit_request(r.ublk_cmd, r.index, r.result))
|
||||||
|
break;
|
||||||
|
sent++;
|
||||||
|
}
|
||||||
|
if (sent)
|
||||||
|
ringloop->submit();
|
||||||
}
|
}
|
||||||
|
|
||||||
void exec_request(int res, int i)
|
void exec_request(int res, int i)
|
||||||
@@ -864,6 +905,11 @@ protected:
|
|||||||
int sync_ublk_cmd(uint32_t cmd_op, void *addr, uint32_t len, uint16_t dev_path_len = 0, uint64_t data0 = 0)
|
int sync_ublk_cmd(uint32_t cmd_op, void *addr, uint32_t len, uint16_t dev_path_len = 0, uint64_t data0 = 0)
|
||||||
{
|
{
|
||||||
io_uring_sqe *sqe = ringloop->get_sqe();
|
io_uring_sqe *sqe = ringloop->get_sqe();
|
||||||
|
if (!sqe)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Error: io_uring is full when trying to execute a control command\n");
|
||||||
|
exit(1);
|
||||||
|
}
|
||||||
sqe->fd = ctrl_fd;
|
sqe->fd = ctrl_fd;
|
||||||
sqe->opcode = IORING_OP_URING_CMD;
|
sqe->opcode = IORING_OP_URING_CMD;
|
||||||
sqe->ioprio = 0;
|
sqe->ioprio = 0;
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ includedir=${prefix}/@CMAKE_INSTALL_INCLUDEDIR@
|
|||||||
|
|
||||||
Name: Vitastor
|
Name: Vitastor
|
||||||
Description: Vitastor client library
|
Description: Vitastor client library
|
||||||
Version: 3.0.6
|
Version: 3.0.14
|
||||||
Libs: -L${libdir} -lvitastor_client
|
Libs: -L${libdir} -lvitastor_client
|
||||||
Cflags: -I${includedir}
|
Cflags: -I${includedir}
|
||||||
|
|
||||||
|
|||||||
+13
-13
@@ -103,7 +103,7 @@ vitastor_c *vitastor_c_create_qemu(QEMUSetFDHandler *aio_set_fd_handler, void *a
|
|||||||
rdma_device, rdma_port_num, rdma_gid_index, rdma_mtu, log_level
|
rdma_device, rdma_port_num, rdma_gid_index, rdma_mtu, log_level
|
||||||
);
|
);
|
||||||
auto self = vitastor_c_create_qemu_common(aio_set_fd_handler, aio_context);
|
auto self = vitastor_c_create_qemu_common(aio_set_fd_handler, aio_context);
|
||||||
self->cli = new cluster_client_t(NULL, self->tfd, cfg_json);
|
self->cli = cluster_client_t::create(NULL, self->tfd, cfg_json);
|
||||||
return self;
|
return self;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -126,7 +126,7 @@ vitastor_c *vitastor_c_create_qemu_uring(QEMUSetFDHandler *aio_set_fd_handler, v
|
|||||||
);
|
);
|
||||||
auto self = vitastor_c_create_qemu_common(aio_set_fd_handler, aio_context);
|
auto self = vitastor_c_create_qemu_common(aio_set_fd_handler, aio_context);
|
||||||
self->ringloop = ringloop;
|
self->ringloop = ringloop;
|
||||||
self->cli = new cluster_client_t(self->ringloop, self->tfd, cfg_json);
|
self->cli = cluster_client_t::create(self->ringloop, self->tfd, cfg_json);
|
||||||
ringloop->loop();
|
ringloop->loop();
|
||||||
return self;
|
return self;
|
||||||
}
|
}
|
||||||
@@ -150,7 +150,7 @@ vitastor_c *vitastor_c_create_uring(const char *config_path, const char *etcd_ho
|
|||||||
vitastor_c *self = new vitastor_c;
|
vitastor_c *self = new vitastor_c;
|
||||||
self->ringloop = ringloop;
|
self->ringloop = ringloop;
|
||||||
self->epmgr = new epoll_manager_t(self->ringloop);
|
self->epmgr = new epoll_manager_t(self->ringloop);
|
||||||
self->cli = new cluster_client_t(self->ringloop, self->epmgr->tfd, cfg_json);
|
self->cli = cluster_client_t::create(self->ringloop, self->epmgr->tfd, cfg_json);
|
||||||
ringloop->loop();
|
ringloop->loop();
|
||||||
return self;
|
return self;
|
||||||
}
|
}
|
||||||
@@ -191,7 +191,7 @@ vitastor_c *vitastor_c_create_uring_json(const char **options, int options_len)
|
|||||||
vitastor_c *self = new vitastor_c;
|
vitastor_c *self = new vitastor_c;
|
||||||
self->ringloop = ringloop;
|
self->ringloop = ringloop;
|
||||||
self->epmgr = new epoll_manager_t(self->ringloop);
|
self->epmgr = new epoll_manager_t(self->ringloop);
|
||||||
self->cli = new cluster_client_t(self->ringloop, self->epmgr->tfd, cfg_json);
|
self->cli = cluster_client_t::create(self->ringloop, self->epmgr->tfd, cfg_json);
|
||||||
ringloop->loop();
|
ringloop->loop();
|
||||||
return self;
|
return self;
|
||||||
}
|
}
|
||||||
@@ -206,7 +206,7 @@ vitastor_c *vitastor_c_create_epoll_json(const char **options, int options_len)
|
|||||||
json11::Json cfg_json(cfg);
|
json11::Json cfg_json(cfg);
|
||||||
vitastor_c *self = new vitastor_c;
|
vitastor_c *self = new vitastor_c;
|
||||||
self->epmgr = new epoll_manager_t(NULL);
|
self->epmgr = new epoll_manager_t(NULL);
|
||||||
self->cli = new cluster_client_t(NULL, self->epmgr->tfd, cfg_json);
|
self->cli = cluster_client_t::create(NULL, self->epmgr->tfd, cfg_json);
|
||||||
return self;
|
return self;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -396,7 +396,7 @@ void vitastor_c_watch_inode(vitastor_c *client, char *image, VitastorIOHandler c
|
|||||||
{
|
{
|
||||||
client->cli->on_ready([=]()
|
client->cli->on_ready([=]()
|
||||||
{
|
{
|
||||||
auto watch = client->cli->st_cli.watch_inode(std::string(image));
|
auto watch = client->cli->st_cli->watch_inode(std::string(image));
|
||||||
cb(opaque, (long)watch);
|
cb(opaque, (long)watch);
|
||||||
});
|
});
|
||||||
if (client->ringloop)
|
if (client->ringloop)
|
||||||
@@ -407,7 +407,7 @@ void vitastor_c_watch_inode(vitastor_c *client, char *image, VitastorIOHandler c
|
|||||||
|
|
||||||
void vitastor_c_close_watch(vitastor_c *client, void *handle)
|
void vitastor_c_close_watch(vitastor_c *client, void *handle)
|
||||||
{
|
{
|
||||||
client->cli->st_cli.close_watch((inode_watch_t*)handle);
|
client->cli->st_cli->close_watch((inode_watch_t*)handle);
|
||||||
}
|
}
|
||||||
|
|
||||||
uint64_t vitastor_c_inode_get_size(void *handle)
|
uint64_t vitastor_c_inode_get_size(void *handle)
|
||||||
@@ -424,8 +424,8 @@ uint64_t vitastor_c_inode_get_num(void *handle)
|
|||||||
|
|
||||||
uint32_t vitastor_c_inode_get_block_size(vitastor_c *client, uint64_t inode_num)
|
uint32_t vitastor_c_inode_get_block_size(vitastor_c *client, uint64_t inode_num)
|
||||||
{
|
{
|
||||||
auto pool_it = client->cli->st_cli.pool_config.find(INODE_POOL(inode_num));
|
auto pool_it = client->cli->st_cli->pool_config.find(INODE_POOL(inode_num));
|
||||||
if (pool_it == client->cli->st_cli.pool_config.end())
|
if (pool_it == client->cli->st_cli->pool_config.end())
|
||||||
return 0;
|
return 0;
|
||||||
auto & pool_cfg = pool_it->second;
|
auto & pool_cfg = pool_it->second;
|
||||||
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
||||||
@@ -434,8 +434,8 @@ uint32_t vitastor_c_inode_get_block_size(vitastor_c *client, uint64_t inode_num)
|
|||||||
|
|
||||||
uint32_t vitastor_c_inode_get_bitmap_granularity(vitastor_c *client, uint64_t inode_num)
|
uint32_t vitastor_c_inode_get_bitmap_granularity(vitastor_c *client, uint64_t inode_num)
|
||||||
{
|
{
|
||||||
auto pool_it = client->cli->st_cli.pool_config.find(INODE_POOL(inode_num));
|
auto pool_it = client->cli->st_cli->pool_config.find(INODE_POOL(inode_num));
|
||||||
if (pool_it == client->cli->st_cli.pool_config.end())
|
if (pool_it == client->cli->st_cli->pool_config.end())
|
||||||
return 0;
|
return 0;
|
||||||
// FIXME: READ_BITMAP may fails if parent bitmap granularity differs from inode bitmap granularity
|
// FIXME: READ_BITMAP may fails if parent bitmap granularity differs from inode bitmap granularity
|
||||||
return pool_it->second.bitmap_granularity;
|
return pool_it->second.bitmap_granularity;
|
||||||
@@ -471,8 +471,8 @@ uint64_t vitastor_c_inode_get_mod_revision(void *handle)
|
|||||||
|
|
||||||
uint32_t vitastor_c_inode_get_immediate_commit(vitastor_c *client, uint64_t inode_num)
|
uint32_t vitastor_c_inode_get_immediate_commit(vitastor_c *client, uint64_t inode_num)
|
||||||
{
|
{
|
||||||
auto pool_it = client->cli->st_cli.pool_config.find(INODE_POOL(inode_num));
|
auto pool_it = client->cli->st_cli->pool_config.find(INODE_POOL(inode_num));
|
||||||
if (pool_it == client->cli->st_cli.pool_config.end())
|
if (pool_it == client->cli->st_cli->pool_config.end())
|
||||||
return 0;
|
return 0;
|
||||||
return pool_it->second.immediate_commit;
|
return pool_it->second.immediate_commit;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
cmake_minimum_required(VERSION 2.8.12)
|
cmake_minimum_required(VERSION 2.8...3.30)
|
||||||
|
|
||||||
project(vitastor)
|
project(vitastor)
|
||||||
|
|
||||||
@@ -27,6 +27,7 @@ add_library(vitastor_cli STATIC
|
|||||||
cli_pool_ls.cpp
|
cli_pool_ls.cpp
|
||||||
cli_pool_modify.cpp
|
cli_pool_modify.cpp
|
||||||
cli_pool_rm.cpp
|
cli_pool_rm.cpp
|
||||||
|
cli_raw_ls.cpp
|
||||||
)
|
)
|
||||||
target_compile_options(vitastor_cli PUBLIC -fPIC)
|
target_compile_options(vitastor_cli PUBLIC -fPIC)
|
||||||
|
|
||||||
|
|||||||
+11
-1
@@ -126,6 +126,11 @@ static const char* help_text =
|
|||||||
" --min-offset, --max-offset\n"
|
" --min-offset, --max-offset\n"
|
||||||
" Restrict listing to specific offsets inside inodes.\n"
|
" Restrict listing to specific offsets inside inodes.\n"
|
||||||
"\n"
|
"\n"
|
||||||
|
"vitastor-cli raw-ls [OPTIONS]\n"
|
||||||
|
" Find object(s) in the cluster using raw secondary listing operations. Options:\n"
|
||||||
|
" [--min_inode NUM] [--max_inode NUM] [--offset NUM] [--pg_num NUM] [--pg_count COUNT]\n"
|
||||||
|
" [--pg_stripe_size NUM] [--osds 1,2,3,...]\n"
|
||||||
|
"\n"
|
||||||
"vitastor-cli fix [--objects <objects>] [--bad-osds <osds>] [--part <part>] [--check no]\n"
|
"vitastor-cli fix [--objects <objects>] [--bad-osds <osds>] [--part <part>] [--check no]\n"
|
||||||
" Fix inconsistent objects in the cluster by deleting some copies.\n"
|
" Fix inconsistent objects in the cluster by deleting some copies.\n"
|
||||||
" --objects <objects>\n"
|
" --objects <objects>\n"
|
||||||
@@ -459,6 +464,11 @@ static int run(cli_tool_t *p, json11::Json::object cfg)
|
|||||||
// Describe unclean objects
|
// Describe unclean objects
|
||||||
action_cb = p->start_describe(cfg);
|
action_cb = p->start_describe(cfg);
|
||||||
}
|
}
|
||||||
|
else if (cmd[0] == "raw-ls")
|
||||||
|
{
|
||||||
|
// Run raw listings
|
||||||
|
action_cb = p->start_raw_ls(cfg);
|
||||||
|
}
|
||||||
else if (cmd[0] == "fix")
|
else if (cmd[0] == "fix")
|
||||||
{
|
{
|
||||||
// Fix inconsistent objects (by deleting some copies)
|
// Fix inconsistent objects (by deleting some copies)
|
||||||
@@ -545,7 +555,7 @@ static int run(cli_tool_t *p, json11::Json::object cfg)
|
|||||||
json11::Json cfg_j = cfg;
|
json11::Json cfg_j = cfg;
|
||||||
p->ringloop = new ring_loop_t(RINGLOOP_DEFAULT_SIZE);
|
p->ringloop = new ring_loop_t(RINGLOOP_DEFAULT_SIZE);
|
||||||
p->epmgr = new epoll_manager_t(p->ringloop);
|
p->epmgr = new epoll_manager_t(p->ringloop);
|
||||||
p->cli = new cluster_client_t(p->ringloop, p->epmgr->tfd, cfg_j);
|
p->cli = cluster_client_t::create(p->ringloop, p->epmgr->tfd, cfg_j);
|
||||||
p->loop_and_wait(action_cb, [&](const cli_result_t & r)
|
p->loop_and_wait(action_cb, [&](const cli_result_t & r)
|
||||||
{
|
{
|
||||||
result = r;
|
result = r;
|
||||||
|
|||||||
@@ -62,6 +62,7 @@ public:
|
|||||||
std::function<bool(cli_result_t &)> start_fix(json11::Json);
|
std::function<bool(cli_result_t &)> start_fix(json11::Json);
|
||||||
std::function<bool(cli_result_t &)> start_flatten(json11::Json);
|
std::function<bool(cli_result_t &)> start_flatten(json11::Json);
|
||||||
std::function<bool(cli_result_t &)> start_ls(json11::Json);
|
std::function<bool(cli_result_t &)> start_ls(json11::Json);
|
||||||
|
std::function<bool(cli_result_t &)> start_raw_ls(json11::Json cfg);
|
||||||
std::function<bool(cli_result_t &)> start_merge(json11::Json);
|
std::function<bool(cli_result_t &)> start_merge(json11::Json);
|
||||||
std::function<bool(cli_result_t &)> start_modify(json11::Json);
|
std::function<bool(cli_result_t &)> start_modify(json11::Json);
|
||||||
std::function<bool(cli_result_t &)> start_modify_osd(json11::Json);
|
std::function<bool(cli_result_t &)> start_modify_osd(json11::Json);
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user