Compare commits
66
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c6af10a444 | ||
|
|
bc24936969 | ||
|
|
b6c854c860 | ||
|
|
0ce6a61fcb | ||
|
|
4498d0665a | ||
|
|
55cce76d36 | ||
|
|
70c0cd2d5f | ||
|
|
455dc9cb1c | ||
|
|
9243aed615 | ||
|
|
3dd0e4daca | ||
|
|
3779e2670d | ||
|
|
0d92e00e41 | ||
|
|
0d1eeb4310 | ||
|
|
b21c92cb7a | ||
|
|
9db0980fc1 | ||
|
|
4a8ee088c1 | ||
|
|
f2585bcd02 | ||
|
|
72d1447eec | ||
|
|
c67db91305 | ||
|
|
75dfafa0cf | ||
|
|
b1d9098904 | ||
|
|
f676d262ec | ||
|
|
954505b5fd | ||
|
|
c137860f18 | ||
|
|
f45fb987d1 | ||
|
|
d744bd43ae | ||
|
|
8da99ee24a | ||
|
|
3c3a58aa1e | ||
|
|
40d4162249 | ||
|
|
026aaf59e7 | ||
|
|
cb40ab3dfd | ||
|
|
459a412f91 | ||
|
|
2b52769d64 | ||
|
|
ff83e12b79 | ||
|
|
4340b2b7bc | ||
|
|
6be0be02fa | ||
|
|
a31a1c0ab1 | ||
|
|
1c18daa644 | ||
|
|
b367c81f44 | ||
|
|
197ef6c171 | ||
|
|
4a907f09df | ||
|
|
360ccbaa4d | ||
|
|
d960816a9b | ||
|
|
3dabffb1be | ||
|
|
5ba63a2fa5 | ||
|
|
26be916e3f | ||
|
|
eb1dbd7459 | ||
|
|
13842ecd90 | ||
|
|
27e2c08e38 | ||
|
|
f9975311ea | ||
|
|
652ca3f1c3 | ||
|
|
c81cbcf69e | ||
|
|
9e28c4c6e3 | ||
|
|
e14be3fdad | ||
|
|
2e5d9cb238 | ||
|
|
376226552a | ||
|
|
6bdd260b50 | ||
|
|
b72cfbad66 | ||
|
|
792efa673e | ||
|
|
f3c5c3b776 | ||
|
|
5e26376f1f | ||
|
|
55047c10fb | ||
|
|
e13edcdc2f | ||
|
|
aaf54f264d | ||
|
|
7e879bea9d | ||
|
|
af34be70ab |
@@ -360,78 +360,6 @@ jobs:
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_dump_load:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 3
|
||||
run: /root/vitastor/tests/test_dump_load.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_dump_load_32k:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 3
|
||||
run: TEST_NAME=32k OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k" OFFSET_ARGS="$OSD_ARGS" /root/vitastor/tests/test_dump_load.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_old_dump_load:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 3
|
||||
run: OLD=1 /root/vitastor/tests/test_dump_load.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_dump_load_old_32k:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 3
|
||||
run: TEST_NAME=old_32k OLD=1 OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k" OFFSET_ARGS="$OSD_ARGS" /root/vitastor/tests/test_dump_load.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_old_interrupted_rebalance:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
@@ -1656,24 +1584,6 @@ jobs:
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_resize_last:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 3
|
||||
run: /root/vitastor/tests/test_resize_last.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_resize_auto:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
@@ -1710,24 +1620,6 @@ jobs:
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_old_resize_last:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||
steps:
|
||||
- name: Run test
|
||||
id: test
|
||||
timeout-minutes: 3
|
||||
run: OLD=1 /root/vitastor/tests/test_resize_last.sh
|
||||
- name: Print logs
|
||||
if: always() && steps.test.outcome == 'failure'
|
||||
run: |
|
||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||
echo "-------- $i --------"
|
||||
cat $i
|
||||
echo ""
|
||||
done
|
||||
|
||||
test_old_resize_auto:
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
|
||||
+1
-1
@@ -2,7 +2,7 @@ cmake_minimum_required(VERSION 2.8...3.30)
|
||||
|
||||
project(vitastor)
|
||||
|
||||
set(VITASTOR_VERSION "3.0.15")
|
||||
set(VITASTOR_VERSION "3.0.12")
|
||||
|
||||
include(CTest)
|
||||
|
||||
|
||||
+1
-1
Submodule cpp-btree updated: 431d2e1d35...ebe44c9b66
+1
-1
@@ -1,4 +1,4 @@
|
||||
VITASTOR_VERSION ?= v3.0.15
|
||||
VITASTOR_VERSION ?= v3.0.12
|
||||
|
||||
all: build push
|
||||
|
||||
|
||||
@@ -49,7 +49,7 @@ spec:
|
||||
capabilities:
|
||||
add: ["SYS_ADMIN"]
|
||||
allowPrivilegeEscalation: true
|
||||
image: vitalif/vitastor-csi:v3.0.15
|
||||
image: vitalif/vitastor-csi:v3.0.12
|
||||
args:
|
||||
- "--node=$(NODE_ID)"
|
||||
- "--endpoint=$(CSI_ENDPOINT)"
|
||||
|
||||
@@ -121,7 +121,7 @@ spec:
|
||||
privileged: true
|
||||
capabilities:
|
||||
add: ["SYS_ADMIN"]
|
||||
image: vitalif/vitastor-csi:v3.0.15
|
||||
image: vitalif/vitastor-csi:v3.0.12
|
||||
args:
|
||||
- "--node=$(NODE_ID)"
|
||||
- "--endpoint=$(CSI_ENDPOINT)"
|
||||
|
||||
+1
-1
@@ -5,7 +5,7 @@ package vitastor
|
||||
|
||||
const (
|
||||
vitastorCSIDriverName = "csi.vitastor.io"
|
||||
vitastorCSIDriverVersion = "3.0.15"
|
||||
vitastorCSIDriverVersion = "3.0.12"
|
||||
)
|
||||
|
||||
// Config struct fills the parameters of request or user input
|
||||
|
||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
||||
vitastor (3.0.15-1) unstable; urgency=medium
|
||||
vitastor (3.0.12-1) unstable; urgency=medium
|
||||
|
||||
* Bugfixes
|
||||
|
||||
|
||||
Vendored
-1
@@ -11,7 +11,6 @@ override_dh_install:
|
||||
cp -v node-binding/package.json node-binding/index.js node-binding/addon.cc node-binding/addon.h node-binding/client.cc node-binding/client.h debian/tmp/usr/lib/x86_64-linux-gnu/nodejs/vitastor
|
||||
cp -v node-binding/build/Release/addon.node debian/tmp/usr/lib/x86_64-linux-gnu/nodejs/vitastor/build/Release
|
||||
dh_install
|
||||
cd debian/vitastor-mon/usr/lib/vitastor/mon && npm install --production
|
||||
|
||||
override_dh_installdeb:
|
||||
cat debian/fio_version >> debian/vitastor-fio.substvars
|
||||
|
||||
Vendored
+6
@@ -37,6 +37,12 @@ rm -rf a b
|
||||
|
||||
echo "dep:fio=$FIO" > debian/fio_version
|
||||
|
||||
cd /root/vitastor/packages/vitastor-$REL/vitastor-$VER
|
||||
mkdir mon/node_modules
|
||||
cd mon/node_modules
|
||||
curl -s https://git.yourcmc.ru/vitalif/antietcd/archive/master.tar.gz | tar -zx
|
||||
curl -s https://git.yourcmc.ru/vitalif/tinyraft/archive/master.tar.gz | tar -zx
|
||||
|
||||
cd /root/vitastor/packages/vitastor-$REL
|
||||
if [[ ( "$REL" = "trixie" || "$REL" = "resolute" ) && -e ../vitastor-bookworm/vitastor_$VER.orig.tar.xz ]]; then
|
||||
# Fucking shit, archives differ between bookworm (xz 5.4.1) and trixie (xz 5.8.1)
|
||||
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
VITASTOR_VERSION ?= v3.0.15
|
||||
VITASTOR_VERSION ?= v3.0.12
|
||||
|
||||
all: build push
|
||||
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
#
|
||||
|
||||
# Desired Vitastor version
|
||||
VITASTOR_VERSION=v3.0.15
|
||||
VITASTOR_VERSION=v3.0.12
|
||||
|
||||
# Additional arguments for all containers
|
||||
# For example, you may want to specify a custom logging driver here
|
||||
|
||||
@@ -50,9 +50,6 @@ or antietcd_data_dir options). All other antietcd parameters
|
||||
cluster, cluster_key, persist_filter, stale_read can also be set in
|
||||
Vitastor configuration with `antietcd_` prefix.
|
||||
|
||||
See also: [antietcd_cert](security.en.md#antietcd_cert),
|
||||
[antietcd_key](security.en.md#antietcd_key) and [etcd_proxy](security.en.md#etcd_proxyurls).
|
||||
|
||||
You can dump/load data to or from antietcd using Antietcd `anticli` tool:
|
||||
|
||||
```
|
||||
|
||||
@@ -50,9 +50,6 @@ antietcd_data_file или antietcd_data_dir). Все остальные пара
|
||||
node_id, cluster, cluster_key, persist_filter, stale_read также можно задавать
|
||||
в конфигурации Vitastor с префиксом `antietcd_`.
|
||||
|
||||
Смотрите также настройки [antietcd_cert](security.ru.md#antietcd_cert),
|
||||
[antietcd_key](security.ru.md#antietcd_key) и [etcd_proxy](security.ru.md#etcd_proxyurls).
|
||||
|
||||
Вы можете выгружать/загружать данные в или из antietcd с помощью его инструмента
|
||||
`anticli`:
|
||||
|
||||
|
||||
+30
-186
@@ -10,36 +10,13 @@ These parameters affect your Vitastor installation security and apply to OSDs, m
|
||||
|
||||
Most of them can be set in /etc/vitastor/vitastor.conf and in etcd, but don't support online modification.
|
||||
|
||||
All certificate and private key parameters mentioned may contain a path to a PEM file or just
|
||||
a PEM string with certificate or a private key. In the latter case, the string must begin with
|
||||
"-----BEGIN CERTIFICATE-----" or "-----BEGIN PRIVATE KEY-----".
|
||||
|
||||
- [use_perms](#use_perms)
|
||||
- [cert](#cert)
|
||||
- [pkey](#pkey)
|
||||
- [etcd_ca](#etcd_ca)
|
||||
- [client_ca](#client_ca)
|
||||
- [osd_ca](#osd_ca)
|
||||
- [mon_ca](#mon_ca)
|
||||
- [antietcd_cert](#antietcd_cert)
|
||||
- [antietcd_key](#antietcd_key)
|
||||
- [etcd_proxy.urls](#etcd_proxyurls)
|
||||
- [etcd_proxy.cert](#etcd_proxycert)
|
||||
- [etcd_proxy.key](#etcd_proxykey)
|
||||
- [etcd_proxy.ca](#etcd_proxyca)
|
||||
- [osd_cert](#osd_cert)
|
||||
- [osd_pkey](#osd_pkey)
|
||||
- [api_cert](#api_cert)
|
||||
- [api_pkey](#api_pkey)
|
||||
- [etcd_client_cert](#etcd_client_cert)
|
||||
- [etcd_client_key](#etcd_client_key)
|
||||
- [etcd_ca](#etcd_ca)
|
||||
- [osd_etcd_client_cert](#osd_etcd_client_cert)
|
||||
- [osd_etcd_client_key](#osd_etcd_client_key)
|
||||
- [mon_etcd_client_cert](#mon_etcd_client_cert)
|
||||
- [mon_etcd_client_key](#mon_etcd_client_key)
|
||||
- [proto_checksums](#proto_checksums)
|
||||
- [force_proto_checksums](#force_proto_checksums)
|
||||
- [max_cipher_pool_size](#max_cipher_pool_size)
|
||||
- [vault_url](#vault_url)
|
||||
- [vault_secret_api_path](#vault_secret_api_path)
|
||||
- [vault_client_cert](#vault_client_cert)
|
||||
@@ -48,195 +25,54 @@ a PEM string with certificate or a private key. In the latter case, the string m
|
||||
- [vault_timeout_ms](#vault_timeout_ms)
|
||||
- [vault_error_timeout_sec](#vault_error_timeout_sec)
|
||||
- [vault_refresh_leeway_sec](#vault_refresh_leeway_sec)
|
||||
|
||||
## use_perms
|
||||
|
||||
- Type: boolean
|
||||
- Default: false
|
||||
|
||||
Enable client permissions in a Vitastor cluster, including Antietcd built into the Monitor.
|
||||
Requires configured encryption. Also note that separate Antietcd requires separate configuration
|
||||
to use permissions (see [security documentation](../intro/security.en.md) for details).
|
||||
|
||||
## cert
|
||||
|
||||
- Type: string
|
||||
|
||||
Client certificate of the current Vitastor user. Required for Vitastor protocol encryption.
|
||||
Must be signed with [client_ca](#client_ca). Also used as the client certificate for etcd/Antietcd
|
||||
connections by default.
|
||||
|
||||
## pkey
|
||||
|
||||
- Type: string
|
||||
|
||||
Private key of the current Vitastor user.
|
||||
|
||||
## etcd_ca
|
||||
|
||||
- Type: string
|
||||
|
||||
Trusted TLS CA to verify etcd server certificate. Or just the etcd server's
|
||||
certificate itself - it's fine to use it for etcd_ca.
|
||||
|
||||
## client_ca
|
||||
|
||||
- Type: string
|
||||
|
||||
Trusted TLS CA to verify Vitastor client certificates.
|
||||
Mandatory for Vitastor protocol encryption.
|
||||
|
||||
## osd_ca
|
||||
|
||||
- Type: string
|
||||
|
||||
Trusted TLS CA to verify Vitastor OSD certificates. Also mandatory for Vitastor protocol
|
||||
encryption. Must be different from client_ca. May be equal to osd_cert - different OSDs
|
||||
don't require separate certificates at the moment because their permissions don't differ.
|
||||
|
||||
## mon_ca
|
||||
|
||||
- Type: string
|
||||
|
||||
Trusted TLS CA to verify Vitastor Monitor certificates. Used only for separate Antietcd,
|
||||
not required when a monitor built-in Antietcd is used. May be equal to mon_client_etcd_cert.
|
||||
|
||||
## antietcd_cert
|
||||
|
||||
- Type: string
|
||||
|
||||
Server TLS certificate for Antietcd built into the Monitor.
|
||||
|
||||
## antietcd_key
|
||||
|
||||
- Type: string
|
||||
|
||||
Private key for antietcd_cert.
|
||||
|
||||
## etcd_proxy.urls
|
||||
|
||||
- Type: string or array of strings
|
||||
|
||||
etcd URLs for Antietcd etcd proxy mode.
|
||||
See [Mon as Etcd proxy](../intro/security.en.md#mon-as-etcd-proxy) for details.
|
||||
|
||||
## etcd_proxy.cert
|
||||
|
||||
- Type: string
|
||||
|
||||
Client certificate for Antietcd connections to etcd in proxy mode.
|
||||
|
||||
## etcd_proxy.key
|
||||
|
||||
- Type: string
|
||||
|
||||
Private key for etcd_proxy.cert.
|
||||
|
||||
## etcd_proxy.ca
|
||||
|
||||
- Type: string
|
||||
|
||||
Trusted TLS CA to verify etcd server certificate when connecting to it from Antietcd.
|
||||
|
||||
## osd_cert
|
||||
|
||||
- Type: string
|
||||
|
||||
Vitastor OSD server certificate. Required for Vitastor protocol encryption. May be equal
|
||||
to [osd_ca](#osd_ca) - all OSDs share the same permission set for now. Also used as the client
|
||||
certificate for connections from OSD to etcd/Antietcd by default.
|
||||
|
||||
## osd_pkey
|
||||
|
||||
- Type: string
|
||||
|
||||
Private key for osd_cert.
|
||||
|
||||
## api_cert
|
||||
|
||||
- Type: string
|
||||
|
||||
Server TLS certificate for [vitastor-cli serve](../usage/cli.en.md#serve) API server.
|
||||
|
||||
## api_pkey
|
||||
|
||||
- Type: string
|
||||
|
||||
Private key for api_cert.
|
||||
- [max_cipher_pool_size](#max_cipher_pool_size)
|
||||
|
||||
## etcd_client_cert
|
||||
|
||||
- Type: string
|
||||
|
||||
Client TLS certificate to use for connections from Vitastor clients to etcd/Antietcd if you don't want
|
||||
to use the common client certificate [cert](#cert).
|
||||
Client TLS certificate to use for Vitastor client (not OSD and not monitor)
|
||||
etcd https connections. May be path to a file or just a PEM string with certificate.
|
||||
In the latter case, string must begin with "-----BEGIN CERTIFICATE-----".
|
||||
|
||||
## etcd_client_key
|
||||
|
||||
- Type: string
|
||||
|
||||
Private key for etcd_client_cert.
|
||||
Private key for etcd_client_cert (also a file or a PEM string).
|
||||
|
||||
## etcd_ca
|
||||
|
||||
- Type: string
|
||||
|
||||
Trusted TLS CA to verify etcd server certificate. May be path to a file,
|
||||
directory or just a PEM string with certificate.
|
||||
|
||||
## osd_etcd_client_cert
|
||||
|
||||
- Type: string
|
||||
|
||||
Client TLS certificate to use for connections from Vitastor OSDs to etcd/Antietcd if you don't want
|
||||
to use the common OSD certificate [osd_cert](#osd_cert).
|
||||
Same as [etcd_client_cert](#etcd_client_cert), but only for OSDs.
|
||||
OSDs, clients and monitors should have different permissions, so they should
|
||||
use different certificates.
|
||||
|
||||
## osd_etcd_client_key
|
||||
|
||||
- Type: string
|
||||
|
||||
Private key for osd_etcd_client_cert.
|
||||
Same as [etcd_client_key](#etcd_client_key), but only for OSDs.
|
||||
|
||||
## mon_etcd_client_cert
|
||||
|
||||
- Type: string
|
||||
|
||||
Client TLS certificate to use for connections from Vitastor Monitors to etcd/Antietcd - required
|
||||
if you don't use the built-in Antietcd. In case you use it Monitor has direct access to Antietcd data
|
||||
and doesn't require any connection.
|
||||
Same as [etcd_client_cert](#etcd_client_cert), but only for Vitastor monitors.
|
||||
|
||||
## mon_etcd_client_key
|
||||
|
||||
- Type: string
|
||||
|
||||
Private key for mon_etcd_client_cert.
|
||||
|
||||
## proto_checksums
|
||||
|
||||
- Type: string
|
||||
- Default: payload
|
||||
|
||||
One of "full", "payload", "gcm", "none":
|
||||
- "full" means calculate and verify transport level checksums from the full message data
|
||||
including the header - recommended for unencrypted setups.
|
||||
- "payload" enables checksums only for the actual read/write data, but skips them for message
|
||||
headers - recommended for encrypted setups because headers are already protected by AES-GCM.
|
||||
- "gcm" disables checksums and enables AES-GCM encryption of the whole messages including headers
|
||||
and data - AES-GCM already includes MAC which is actually a stronger checksum. This option is
|
||||
slower and is only recommended for untrusted networks.
|
||||
- "none" disables transport level checksums at all.
|
||||
|
||||
## force_proto_checksums
|
||||
|
||||
- Type: string
|
||||
|
||||
To allow older clients to connect to a Vitastor cluster with enabled checksums, Vitastor OSDs
|
||||
allow clients to downgrade their proto_checksums by default. force_proto_checksums sets the
|
||||
minimum security level allowed for connecting clients. When encryption is disabled, default
|
||||
force_proto_checksums is none and clients without checksums are allowed. With enabled
|
||||
encryption, force_proto_checksums becomes "payload" by default to block unauthenticated data
|
||||
on the transport level.
|
||||
|
||||
## max_cipher_pool_size
|
||||
|
||||
- Type: integer
|
||||
- Default: 256
|
||||
|
||||
Maximum number of OpenSSL cipher contexts cached in OSD memory, counted separately
|
||||
for each cipher and for encryption/decryption. Probably doesn't require modification.
|
||||
Same as [etcd_client_key](#etcd_client_key), but only for Vitastor monitors.
|
||||
|
||||
## vault_url
|
||||
|
||||
@@ -267,14 +103,14 @@ Vault v1 secret API mount path to use.
|
||||
|
||||
- Type: string
|
||||
|
||||
Client TLS certificate to use for Vault connections if you don't want to use the common Vitastor
|
||||
client certificate [cert](#cert) which is also used for Vault connections by default.
|
||||
Client TLS certificate to use for Vault connections. Just like [etcd_client_cert](#etcd_client_cert),
|
||||
may be path to a file or just a certificate in PEM string.
|
||||
|
||||
## vault_client_key
|
||||
|
||||
- Type: string
|
||||
|
||||
Private key for the vault_client_cert certificate.
|
||||
Private key for vault_client_cert (also a file or a PEM string).
|
||||
|
||||
## vault_ca
|
||||
|
||||
@@ -304,3 +140,11 @@ Time (in seconds) to wait before retrying after receiving an error from Vault.
|
||||
|
||||
Extra time (in seconds) before real Vault token lease_timeout to refresh it, just
|
||||
in case of system clock drift.
|
||||
|
||||
## max_cipher_pool_size
|
||||
|
||||
- Type: integer
|
||||
- Default: 256
|
||||
|
||||
Maximum number of OpenSSL cipher contexts cached in OSD memory, counted separately
|
||||
for each cipher and for encryption/decryption. Probably doesn't require modification.
|
||||
|
||||
+28
-186
@@ -12,36 +12,13 @@ OSD, мониторами и клиентами.
|
||||
Большая их часть может задаваться в /etc/vitastor/vitastor.conf и в etcd, но не
|
||||
поддерживает онлайн-изменение.
|
||||
|
||||
Все параметры сертификатов и закрытых ключей могут быть путём к файлу или просто
|
||||
строкой с сертификатом в формате PEM. В последнем случае строка должна начинаться с
|
||||
"-----BEGIN CERTIFICATE-----" или "-----BEGIN PRIVATE KEY-----".
|
||||
|
||||
- [use_perms](#use_perms)
|
||||
- [cert](#cert)
|
||||
- [pkey](#pkey)
|
||||
- [etcd_ca](#etcd_ca)
|
||||
- [client_ca](#client_ca)
|
||||
- [osd_ca](#osd_ca)
|
||||
- [mon_ca](#mon_ca)
|
||||
- [antietcd_cert](#antietcd_cert)
|
||||
- [antietcd_key](#antietcd_key)
|
||||
- [etcd_proxy.urls](#etcd_proxyurls)
|
||||
- [etcd_proxy.cert](#etcd_proxycert)
|
||||
- [etcd_proxy.key](#etcd_proxykey)
|
||||
- [etcd_proxy.ca](#etcd_proxyca)
|
||||
- [osd_cert](#osd_cert)
|
||||
- [osd_pkey](#osd_pkey)
|
||||
- [api_cert](#api_cert)
|
||||
- [api_pkey](#api_pkey)
|
||||
- [etcd_client_cert](#etcd_client_cert)
|
||||
- [etcd_client_key](#etcd_client_key)
|
||||
- [etcd_ca](#etcd_ca)
|
||||
- [osd_etcd_client_cert](#osd_etcd_client_cert)
|
||||
- [osd_etcd_client_key](#osd_etcd_client_key)
|
||||
- [mon_etcd_client_cert](#mon_etcd_client_cert)
|
||||
- [mon_etcd_client_key](#mon_etcd_client_key)
|
||||
- [proto_checksums](#proto_checksums)
|
||||
- [force_proto_checksums](#force_proto_checksums)
|
||||
- [max_cipher_pool_size](#max_cipher_pool_size)
|
||||
- [vault_url](#vault_url)
|
||||
- [vault_secret_api_path](#vault_secret_api_path)
|
||||
- [vault_client_cert](#vault_client_cert)
|
||||
@@ -50,199 +27,56 @@ OSD, мониторами и клиентами.
|
||||
- [vault_timeout_ms](#vault_timeout_ms)
|
||||
- [vault_error_timeout_sec](#vault_error_timeout_sec)
|
||||
- [vault_refresh_leeway_sec](#vault_refresh_leeway_sec)
|
||||
- [max_cipher_pool_size](#max_cipher_pool_size)
|
||||
|
||||
## use_perms
|
||||
|
||||
- Тип: булево (да/нет)
|
||||
- Значение по умолчанию: false
|
||||
|
||||
Включает клиентские привилегии в кластере Vitastor, в том числе во встроенном в мониторе Antietcd.
|
||||
Требует настроенного шифрования протокола. Также обратите внимание, что отдельно установленный Antietcd
|
||||
требует отдельной настройки привилегий (подробности смотрите в [документации безопасности](../intro/security.ru.md)).
|
||||
|
||||
## cert
|
||||
## etcd_client_cert
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Клиентский сертификат текущего пользователя Vitastor. Требуется для шифрования протокола Vitastor.
|
||||
Должен быть подписан [client_ca](#client_ca). Также по умолчанию используется как клиентский
|
||||
сертификат для подключения к etcd/Antietcd и Vault.
|
||||
Клиентский TLS сертификат для https-подключений к etcd для клиентов Vitastor
|
||||
(не OSD и не мониторов). Может быть путём к файлу или просто строкой с
|
||||
сертификатом в формате PEM. В последнем случае строка должна начинаться с
|
||||
"-----BEGIN CERTIFICATE-----".
|
||||
|
||||
## pkey
|
||||
## etcd_client_key
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Закрытый ключ текущего пользователя Vitastor.
|
||||
Закрытый ключ для сертификата etcd_client_cert (также путь к файлу или PEM строка).
|
||||
|
||||
## etcd_ca
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Доверенный корневой TLS-сертификат для проверки сертификата сервера etcd.
|
||||
Либо же просто сам сертификат сервера etcd - его можно использовать как etcd_ca.
|
||||
|
||||
## client_ca
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Доверенный TLS-сертификат для проверки сертификатов клиентов Vitastor.
|
||||
Требуется для шифрования протокола Vitastor.
|
||||
|
||||
## osd_ca
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Доверенный TLS-сертификат для проверки сертификатов OSD Vitastor. Также обязателен
|
||||
для шифрования протокола Vitastor. Должен отличаться от client_ca. Может быть равен
|
||||
osd_cert - разные OSD не требуют разных сертификатов, потому что на данный момент
|
||||
привилегии разных OSD никак не отличаются.
|
||||
|
||||
## mon_ca
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Доверенный TLS-сертификат для проверки сертификатов мониторов Vitastor. Используется
|
||||
только отдельно установленным Antietcd, не требуется при использовании встроенного в монитор
|
||||
Antietcd. Может быть равен mon_client_etcd_cert.
|
||||
|
||||
## antietcd_cert
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Серверный TLS-сертификат для Antietcd, встроенного в монитор.
|
||||
|
||||
## antietcd_key
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Закрытый ключ для сертификата antietcd_cert.
|
||||
|
||||
## etcd_proxy.urls
|
||||
|
||||
- Тип: строка или массив строк
|
||||
|
||||
Адреса etcd для режима Antietcd etcd-прокси.
|
||||
Смотрите подробности в разделе [Mon в роли Etcd proxy](../intro/security.ru.md#mon-в-роли-etcd-proxy).
|
||||
|
||||
## etcd_proxy.cert
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Клиентский сертификат для подключений от Antietcd к etcd в режиме прокси.
|
||||
|
||||
## etcd_proxy.key
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Закрытый ключ для сертификата etcd_proxy.cert.
|
||||
|
||||
## etcd_proxy.ca
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Доверенный TLS-сертификат для проверки сертификата сервера etcd при подключениях от Antietcd.
|
||||
|
||||
## osd_cert
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Сертификат сервера Vitastor OSD. Требуется для шифрования протокола Vitastor. Может быть равен
|
||||
[osd_ca](#osd_ca) - все OSD на данный момент имеют одинаковые привилегии. Также по умолчанию
|
||||
используется как клиентский сертификат для подключения от OSD к etcd/Antietcd.
|
||||
|
||||
## osd_pkey
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Закрытый ключ для сертификата osd_cert.
|
||||
|
||||
## api_cert
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Серверный TLS-сертификат для API-сервера [vitastor-cli serve](../usage/cli.ru.md#serve).
|
||||
|
||||
## api_pkey
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Закрытый ключ для сертификата api_cert.
|
||||
|
||||
## etcd_client_cert
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Клиентский TLS сертификат для подключений от клиентов Vitastor к etcd/Antietcd, если вы не хотите
|
||||
использовать общий клиентский сертификат [cert](#cert).
|
||||
|
||||
## etcd_client_key
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Закрытый ключ для сертификата etcd_client_cert.
|
||||
Может быть путём к файлу, директории или просто строкой с сертификатом в
|
||||
формате PEM.
|
||||
|
||||
## osd_etcd_client_cert
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Клиентский TLS сертификат для подключений от Vitastor OSD к etcd/Antietcd, если вы не хотите
|
||||
использовать общий сертификат OSD [osd_cert](#osd_cert).
|
||||
Аналогично [etcd_client_cert](#etcd_client_cert), но только для OSD.
|
||||
OSD, клиенты и мониторы должны иметь разные привилегии, поэтому они должны
|
||||
использовать разные сертификаты.
|
||||
|
||||
## osd_etcd_client_key
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Закрытый ключ для сертификата osd_etcd_client_cert.
|
||||
Аналогично [etcd_client_key](#etcd_client_key), но только для OSD.
|
||||
|
||||
## mon_etcd_client_cert
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Клиентский TLS сертификат для подключений от мониторов Vitastor к etcd/Antietcd - требуется, если
|
||||
вы не используете встроенный в монитор Antietcd. Если вы используете его, то монитор и так имеет
|
||||
прямой доступ к данным Antietcd и не требует никаких соединений.
|
||||
Аналогично [etcd_client_cert](#etcd_client_cert), но только для мониторов Vitastor.
|
||||
|
||||
## mon_etcd_client_key
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Закрытый ключ для сертификата mon_etcd_client_cert.
|
||||
|
||||
## proto_checksums
|
||||
|
||||
- Тип: строка
|
||||
- Значение по умолчанию: payload
|
||||
|
||||
Одно из значений "full", "payload", "gcm" и "none":
|
||||
- "full" означает расчёт и проверку контрольных сумм на транспортном уровне от полных сообщений,
|
||||
включая их заголовки и данные - рекомендуется для кластеров без шифрования.
|
||||
- "payload" включает контрольные суммы только для данных сообщений, но пропускает заголовки -
|
||||
такая настройка рекомендуется для кластеров с включённым шифрованием, потому что в них заголовки
|
||||
и так защищены шифрованием AES-GCM.
|
||||
- "gcm" отключает контрольные суммы и включает шифрование полных сообщений включая заголовки и
|
||||
данные - AES-GCM уже включает в себя MAC, который по сути является криптостойкой контрольной
|
||||
суммой. Такая настройка медленнее и рекомендуется только для недоверенных сетей.
|
||||
- "none" полностью отключает контрольные суммы на транспортном уровне.
|
||||
|
||||
## force_proto_checksums
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Чтобы старые клиенты Vitastor могли подключаться к кластеру с включёнными контрольными
|
||||
суммами, Vitastor OSD по умолчанию разрешают клиентам отключать контрольные суммы
|
||||
данных (proto_checksums). Настройка force_proto_checksums задаёт минимальный уровень
|
||||
безопасности, разрешённый для подключающихся клиентов. Когда шифрование отключено,
|
||||
force_proto_checksums по умолчанию равно none и подключения клиентов без контрольных
|
||||
сумм разрешаются. При включённом шифровании значение по умолчанию force_proto_checksums
|
||||
становится "payload", чтобы блокировать подключения с неаутентифицированными данными.
|
||||
|
||||
## max_cipher_pool_size
|
||||
|
||||
- Тип: целое число
|
||||
- Значение по умолчанию: 256
|
||||
|
||||
Максимальное количество кэшируемых в памяти OSD контекстов шифра OpenSSL, учитываемое
|
||||
отдельно для каждого шифра и для шифрования и расшифровки. Вряд ли требует изменения.
|
||||
Аналогично [etcd_client_key](#etcd_client_key), но только для мониторов Vitastor.
|
||||
|
||||
## vault_url
|
||||
|
||||
@@ -272,14 +106,14 @@ force_proto_checksums по умолчанию равно none и подключ
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Клиентский TLS сертификат для подключений к Vault на тот случай, если вы не хотите использовать
|
||||
общий сертификат клиента Vitastor [cert](#cert), используемый для подключений к Vault по умолчанию.
|
||||
Клиентский TLS сертификат для подключений к Vault. Как и [etcd_client_cert](#etcd_client_cert),
|
||||
может быть путём к файлу или просто PEM-строкой с сертификатом.
|
||||
|
||||
## vault_client_key
|
||||
|
||||
- Тип: строка
|
||||
|
||||
Закрытый ключ для сертификата vault_client_cert.
|
||||
Закрытый ключ для сертификата vault_client_cert (также путь к файлу или PEM строка).
|
||||
|
||||
## vault_ca
|
||||
|
||||
@@ -310,3 +144,11 @@ force_proto_checksums по умолчанию равно none и подключ
|
||||
|
||||
Зазор времени (в секундах), чтобы обновлять токены Vault чуть раньше их реального
|
||||
lease_timeout, на случай "ухода" системных часов.
|
||||
|
||||
## max_cipher_pool_size
|
||||
|
||||
- Тип: целое число
|
||||
- Значение по умолчанию: 256
|
||||
|
||||
Максимальное количество кэшируемых в памяти OSD контекстов шифра OpenSSL, учитываемое
|
||||
отдельно для каждого шифра и для шифрования и расшифровки. Вряд ли требует изменения.
|
||||
|
||||
@@ -64,7 +64,7 @@ for (const file of params_files)
|
||||
let out = '\n';
|
||||
for (const c of cfg)
|
||||
{
|
||||
out += `\n- [${c.name}](#${c.name.replace(/\./g, '')})`;
|
||||
out += `\n- [${c.name}](#${c.name})`;
|
||||
}
|
||||
for (const c of cfg)
|
||||
{
|
||||
|
||||
@@ -21,9 +21,6 @@
|
||||
cluster, cluster_key, persist_filter, stale_read can also be set in
|
||||
Vitastor configuration with `antietcd_` prefix.
|
||||
|
||||
See also: [antietcd_cert](security.en.md#antietcd_cert),
|
||||
[antietcd_key](security.en.md#antietcd_key) and [etcd_proxy](security.en.md#etcd_proxyurls).
|
||||
|
||||
You can dump/load data to or from antietcd using Antietcd `anticli` tool:
|
||||
|
||||
```
|
||||
@@ -50,9 +47,6 @@
|
||||
node_id, cluster, cluster_key, persist_filter, stale_read также можно задавать
|
||||
в конфигурации Vitastor с префиксом `antietcd_`.
|
||||
|
||||
Смотрите также настройки [antietcd_cert](security.ru.md#antietcd_cert),
|
||||
[antietcd_key](security.ru.md#antietcd_key) и [etcd_proxy](security.ru.md#etcd_proxyurls).
|
||||
|
||||
Вы можете выгружать/загружать данные в или из antietcd с помощью его инструмента
|
||||
`anticli`:
|
||||
|
||||
|
||||
@@ -3,7 +3,3 @@
|
||||
These parameters affect your Vitastor installation security and apply to OSDs, monitors and clients.
|
||||
|
||||
Most of them can be set in /etc/vitastor/vitastor.conf and in etcd, but don't support online modification.
|
||||
|
||||
All certificate and private key parameters mentioned may contain a path to a PEM file or just
|
||||
a PEM string with certificate or a private key. In the latter case, the string must begin with
|
||||
"-----BEGIN CERTIFICATE-----" or "-----BEGIN PRIVATE KEY-----".
|
||||
|
||||
@@ -5,7 +5,3 @@ OSD, мониторами и клиентами.
|
||||
|
||||
Большая их часть может задаваться в /etc/vitastor/vitastor.conf и в etcd, но не
|
||||
поддерживает онлайн-изменение.
|
||||
|
||||
Все параметры сертификатов и закрытых ключей могут быть путём к файлу или просто
|
||||
строкой с сертификатом в формате PEM. В последнем случае строка должна начинаться с
|
||||
"-----BEGIN CERTIFICATE-----" или "-----BEGIN PRIVATE KEY-----".
|
||||
|
||||
+45
-190
@@ -1,203 +1,49 @@
|
||||
- name: use_perms
|
||||
type: bool
|
||||
default: false
|
||||
info: |
|
||||
Enable client permissions in a Vitastor cluster, including Antietcd built into the Monitor.
|
||||
Requires configured encryption. Also note that separate Antietcd requires separate configuration
|
||||
to use permissions (see [security documentation](../intro/security.en.md) for details).
|
||||
info_ru: |
|
||||
Включает клиентские привилегии в кластере Vitastor, в том числе во встроенном в мониторе Antietcd.
|
||||
Требует настроенного шифрования протокола. Также обратите внимание, что отдельно установленный Antietcd
|
||||
требует отдельной настройки привилегий (подробности смотрите в [документации безопасности](../intro/security.ru.md)).
|
||||
- name: cert
|
||||
type: string
|
||||
info: |
|
||||
Client certificate of the current Vitastor user. Required for Vitastor protocol encryption.
|
||||
Must be signed with [client_ca](#client_ca). Also used as the client certificate for etcd/Antietcd
|
||||
connections by default.
|
||||
info_ru: |
|
||||
Клиентский сертификат текущего пользователя Vitastor. Требуется для шифрования протокола Vitastor.
|
||||
Должен быть подписан [client_ca](#client_ca). Также по умолчанию используется как клиентский
|
||||
сертификат для подключения к etcd/Antietcd и Vault.
|
||||
- name: pkey
|
||||
type: string
|
||||
info: Private key of the current Vitastor user.
|
||||
info_ru: Закрытый ключ текущего пользователя Vitastor.
|
||||
- name: etcd_ca
|
||||
type: string
|
||||
info: |
|
||||
Trusted TLS CA to verify etcd server certificate. Or just the etcd server's
|
||||
certificate itself - it's fine to use it for etcd_ca.
|
||||
info_ru: |
|
||||
Доверенный корневой TLS-сертификат для проверки сертификата сервера etcd.
|
||||
Либо же просто сам сертификат сервера etcd - его можно использовать как etcd_ca.
|
||||
- name: client_ca
|
||||
type: string
|
||||
info: |
|
||||
Trusted TLS CA to verify Vitastor client certificates.
|
||||
Mandatory for Vitastor protocol encryption.
|
||||
info_ru: |
|
||||
Доверенный TLS-сертификат для проверки сертификатов клиентов Vitastor.
|
||||
Требуется для шифрования протокола Vitastor.
|
||||
- name: osd_ca
|
||||
type: string
|
||||
info: |
|
||||
Trusted TLS CA to verify Vitastor OSD certificates. Also mandatory for Vitastor protocol
|
||||
encryption. Must be different from client_ca. May be equal to osd_cert - different OSDs
|
||||
don't require separate certificates at the moment because their permissions don't differ.
|
||||
info_ru: |
|
||||
Доверенный TLS-сертификат для проверки сертификатов OSD Vitastor. Также обязателен
|
||||
для шифрования протокола Vitastor. Должен отличаться от client_ca. Может быть равен
|
||||
osd_cert - разные OSD не требуют разных сертификатов, потому что на данный момент
|
||||
привилегии разных OSD никак не отличаются.
|
||||
- name: mon_ca
|
||||
type: string
|
||||
info: |
|
||||
Trusted TLS CA to verify Vitastor Monitor certificates. Used only for separate Antietcd,
|
||||
not required when a monitor built-in Antietcd is used. May be equal to mon_client_etcd_cert.
|
||||
info_ru: |
|
||||
Доверенный TLS-сертификат для проверки сертификатов мониторов Vitastor. Используется
|
||||
только отдельно установленным Antietcd, не требуется при использовании встроенного в монитор
|
||||
Antietcd. Может быть равен mon_client_etcd_cert.
|
||||
- name: antietcd_cert
|
||||
type: string
|
||||
info: Server TLS certificate for Antietcd built into the Monitor.
|
||||
info_ru: Серверный TLS-сертификат для Antietcd, встроенного в монитор.
|
||||
- name: antietcd_key
|
||||
type: string
|
||||
info: Private key for antietcd_cert.
|
||||
info_ru: Закрытый ключ для сертификата antietcd_cert.
|
||||
- name: etcd_proxy.urls
|
||||
type: string or array of strings
|
||||
type_ru: строка или массив строк
|
||||
info: |
|
||||
etcd URLs for Antietcd etcd proxy mode.
|
||||
See [Mon as Etcd proxy](../intro/security.en.md#mon-as-etcd-proxy) for details.
|
||||
info_ru: |
|
||||
Адреса etcd для режима Antietcd etcd-прокси.
|
||||
Смотрите подробности в разделе [Mon в роли Etcd proxy](../intro/security.ru.md#mon-в-роли-etcd-proxy).
|
||||
- name: etcd_proxy.cert
|
||||
type: string
|
||||
info: Client certificate for Antietcd connections to etcd in proxy mode.
|
||||
info_ru: Клиентский сертификат для подключений от Antietcd к etcd в режиме прокси.
|
||||
- name: etcd_proxy.key
|
||||
type: string
|
||||
info: Private key for etcd_proxy.cert.
|
||||
info_ru: Закрытый ключ для сертификата etcd_proxy.cert.
|
||||
- name: etcd_proxy.ca
|
||||
type: string
|
||||
info: Trusted TLS CA to verify etcd server certificate when connecting to it from Antietcd.
|
||||
info_ru: Доверенный TLS-сертификат для проверки сертификата сервера etcd при подключениях от Antietcd.
|
||||
- name: osd_cert
|
||||
type: string
|
||||
info: |
|
||||
Vitastor OSD server certificate. Required for Vitastor protocol encryption. May be equal
|
||||
to [osd_ca](#osd_ca) - all OSDs share the same permission set for now. Also used as the client
|
||||
certificate for connections from OSD to etcd/Antietcd by default.
|
||||
info_ru: |
|
||||
Сертификат сервера Vitastor OSD. Требуется для шифрования протокола Vitastor. Может быть равен
|
||||
[osd_ca](#osd_ca) - все OSD на данный момент имеют одинаковые привилегии. Также по умолчанию
|
||||
используется как клиентский сертификат для подключения от OSD к etcd/Antietcd.
|
||||
- name: osd_pkey
|
||||
type: string
|
||||
info: Private key for osd_cert.
|
||||
info_ru: Закрытый ключ для сертификата osd_cert.
|
||||
- name: api_cert
|
||||
type: string
|
||||
info: Server TLS certificate for [vitastor-cli serve](../usage/cli.en.md#serve) API server.
|
||||
info_ru: Серверный TLS-сертификат для API-сервера [vitastor-cli serve](../usage/cli.ru.md#serve).
|
||||
- name: api_pkey
|
||||
type: string
|
||||
info: Private key for api_cert.
|
||||
info_ru: Закрытый ключ для сертификата api_cert.
|
||||
- name: etcd_client_cert
|
||||
type: string
|
||||
info: |
|
||||
Client TLS certificate to use for connections from Vitastor clients to etcd/Antietcd if you don't want
|
||||
to use the common client certificate [cert](#cert).
|
||||
Client TLS certificate to use for Vitastor client (not OSD and not monitor)
|
||||
etcd https connections. May be path to a file or just a PEM string with certificate.
|
||||
In the latter case, string must begin with "-----BEGIN CERTIFICATE-----".
|
||||
info_ru: |
|
||||
Клиентский TLS сертификат для подключений от клиентов Vitastor к etcd/Antietcd, если вы не хотите
|
||||
использовать общий клиентский сертификат [cert](#cert).
|
||||
Клиентский TLS сертификат для https-подключений к etcd для клиентов Vitastor
|
||||
(не OSD и не мониторов). Может быть путём к файлу или просто строкой с
|
||||
сертификатом в формате PEM. В последнем случае строка должна начинаться с
|
||||
"-----BEGIN CERTIFICATE-----".
|
||||
- name: etcd_client_key
|
||||
type: string
|
||||
info: Private key for etcd_client_cert.
|
||||
info_ru: Закрытый ключ для сертификата etcd_client_cert.
|
||||
info: Private key for etcd_client_cert (also a file or a PEM string).
|
||||
info_ru: Закрытый ключ для сертификата etcd_client_cert (также путь к файлу или PEM строка).
|
||||
- name: etcd_ca
|
||||
type: string
|
||||
info: |
|
||||
Trusted TLS CA to verify etcd server certificate. May be path to a file,
|
||||
directory or just a PEM string with certificate.
|
||||
info_ru: |
|
||||
Доверенный корневой TLS-сертификат для проверки сертификата сервера etcd.
|
||||
Может быть путём к файлу, директории или просто строкой с сертификатом в
|
||||
формате PEM.
|
||||
- name: osd_etcd_client_cert
|
||||
type: string
|
||||
info: |
|
||||
Client TLS certificate to use for connections from Vitastor OSDs to etcd/Antietcd if you don't want
|
||||
to use the common OSD certificate [osd_cert](#osd_cert).
|
||||
Same as [etcd_client_cert](#etcd_client_cert), but only for OSDs.
|
||||
OSDs, clients and monitors should have different permissions, so they should
|
||||
use different certificates.
|
||||
info_ru: |
|
||||
Клиентский TLS сертификат для подключений от Vitastor OSD к etcd/Antietcd, если вы не хотите
|
||||
использовать общий сертификат OSD [osd_cert](#osd_cert).
|
||||
Аналогично [etcd_client_cert](#etcd_client_cert), но только для OSD.
|
||||
OSD, клиенты и мониторы должны иметь разные привилегии, поэтому они должны
|
||||
использовать разные сертификаты.
|
||||
- name: osd_etcd_client_key
|
||||
type: string
|
||||
info: Private key for osd_etcd_client_cert.
|
||||
info_ru: Закрытый ключ для сертификата osd_etcd_client_cert.
|
||||
info: Same as [etcd_client_key](#etcd_client_key), but only for OSDs.
|
||||
info_ru: Аналогично [etcd_client_key](#etcd_client_key), но только для OSD.
|
||||
- name: mon_etcd_client_cert
|
||||
type: string
|
||||
info: |
|
||||
Client TLS certificate to use for connections from Vitastor Monitors to etcd/Antietcd - required
|
||||
if you don't use the built-in Antietcd. In case you use it Monitor has direct access to Antietcd data
|
||||
and doesn't require any connection.
|
||||
info_ru: |
|
||||
Клиентский TLS сертификат для подключений от мониторов Vitastor к etcd/Antietcd - требуется, если
|
||||
вы не используете встроенный в монитор Antietcd. Если вы используете его, то монитор и так имеет
|
||||
прямой доступ к данным Antietcd и не требует никаких соединений.
|
||||
info: Same as [etcd_client_cert](#etcd_client_cert), but only for Vitastor monitors.
|
||||
info_ru: Аналогично [etcd_client_cert](#etcd_client_cert), но только для мониторов Vitastor.
|
||||
- name: mon_etcd_client_key
|
||||
type: string
|
||||
info: Private key for mon_etcd_client_cert.
|
||||
info_ru: Закрытый ключ для сертификата mon_etcd_client_cert.
|
||||
- name: proto_checksums
|
||||
type: string
|
||||
default: payload
|
||||
info: |
|
||||
One of "full", "payload", "gcm", "none":
|
||||
- "full" means calculate and verify transport level checksums from the full message data
|
||||
including the header - recommended for unencrypted setups.
|
||||
- "payload" enables checksums only for the actual read/write data, but skips them for message
|
||||
headers - recommended for encrypted setups because headers are already protected by AES-GCM.
|
||||
- "gcm" disables checksums and enables AES-GCM encryption of the whole messages including headers
|
||||
and data - AES-GCM already includes MAC which is actually a stronger checksum. This option is
|
||||
slower and is only recommended for untrusted networks.
|
||||
- "none" disables transport level checksums at all.
|
||||
info_ru: |
|
||||
Одно из значений "full", "payload", "gcm" и "none":
|
||||
- "full" означает расчёт и проверку контрольных сумм на транспортном уровне от полных сообщений,
|
||||
включая их заголовки и данные - рекомендуется для кластеров без шифрования.
|
||||
- "payload" включает контрольные суммы только для данных сообщений, но пропускает заголовки -
|
||||
такая настройка рекомендуется для кластеров с включённым шифрованием, потому что в них заголовки
|
||||
и так защищены шифрованием AES-GCM.
|
||||
- "gcm" отключает контрольные суммы и включает шифрование полных сообщений включая заголовки и
|
||||
данные - AES-GCM уже включает в себя MAC, который по сути является криптостойкой контрольной
|
||||
суммой. Такая настройка медленнее и рекомендуется только для недоверенных сетей.
|
||||
- "none" полностью отключает контрольные суммы на транспортном уровне.
|
||||
- name: force_proto_checksums
|
||||
type: string
|
||||
info: |
|
||||
To allow older clients to connect to a Vitastor cluster with enabled checksums, Vitastor OSDs
|
||||
allow clients to downgrade their proto_checksums by default. force_proto_checksums sets the
|
||||
minimum security level allowed for connecting clients. When encryption is disabled, default
|
||||
force_proto_checksums is none and clients without checksums are allowed. With enabled
|
||||
encryption, force_proto_checksums becomes "payload" by default to block unauthenticated data
|
||||
on the transport level.
|
||||
info_ru: |
|
||||
Чтобы старые клиенты Vitastor могли подключаться к кластеру с включёнными контрольными
|
||||
суммами, Vitastor OSD по умолчанию разрешают клиентам отключать контрольные суммы
|
||||
данных (proto_checksums). Настройка force_proto_checksums задаёт минимальный уровень
|
||||
безопасности, разрешённый для подключающихся клиентов. Когда шифрование отключено,
|
||||
force_proto_checksums по умолчанию равно none и подключения клиентов без контрольных
|
||||
сумм разрешаются. При включённом шифровании значение по умолчанию force_proto_checksums
|
||||
становится "payload", чтобы блокировать подключения с неаутентифицированными данными.
|
||||
- name: max_cipher_pool_size
|
||||
type: int
|
||||
default: 256
|
||||
info: |
|
||||
Maximum number of OpenSSL cipher contexts cached in OSD memory, counted separately
|
||||
for each cipher and for encryption/decryption. Probably doesn't require modification.
|
||||
info_ru: |
|
||||
Максимальное количество кэшируемых в памяти OSD контекстов шифра OpenSSL, учитываемое
|
||||
отдельно для каждого шифра и для шифрования и расшифровки. Вряд ли требует изменения.
|
||||
info: Same as [etcd_client_key](#etcd_client_key), but only for Vitastor monitors.
|
||||
info_ru: Аналогично [etcd_client_key](#etcd_client_key), но только для мониторов Vitastor.
|
||||
- name: vault_url
|
||||
type: string
|
||||
info: |
|
||||
@@ -235,15 +81,15 @@
|
||||
- name: vault_client_cert
|
||||
type: string
|
||||
info: |
|
||||
Client TLS certificate to use for Vault connections if you don't want to use the common Vitastor
|
||||
client certificate [cert](#cert) which is also used for Vault connections by default.
|
||||
Client TLS certificate to use for Vault connections. Just like [etcd_client_cert](#etcd_client_cert),
|
||||
may be path to a file or just a certificate in PEM string.
|
||||
info_ru: |
|
||||
Клиентский TLS сертификат для подключений к Vault на тот случай, если вы не хотите использовать
|
||||
общий сертификат клиента Vitastor [cert](#cert), используемый для подключений к Vault по умолчанию.
|
||||
Клиентский TLS сертификат для подключений к Vault. Как и [etcd_client_cert](#etcd_client_cert),
|
||||
может быть путём к файлу или просто PEM-строкой с сертификатом.
|
||||
- name: vault_client_key
|
||||
type: string
|
||||
info: Private key for the vault_client_cert certificate.
|
||||
info_ru: Закрытый ключ для сертификата vault_client_cert.
|
||||
info: Private key for vault_client_cert (also a file or a PEM string).
|
||||
info_ru: Закрытый ключ для сертификата vault_client_cert (также путь к файлу или PEM строка).
|
||||
- name: vault_ca
|
||||
type: string
|
||||
info: |
|
||||
@@ -274,3 +120,12 @@
|
||||
info_ru: |
|
||||
Зазор времени (в секундах), чтобы обновлять токены Vault чуть раньше их реального
|
||||
lease_timeout, на случай "ухода" системных часов.
|
||||
- name: max_cipher_pool_size
|
||||
type: int
|
||||
default: 256
|
||||
info: |
|
||||
Maximum number of OpenSSL cipher contexts cached in OSD memory, counted separately
|
||||
for each cipher and for encryption/decryption. Probably doesn't require modification.
|
||||
info_ru: |
|
||||
Максимальное количество кэшируемых в памяти OSD контекстов шифра OpenSSL, учитываемое
|
||||
отдельно для каждого шифра и для шифрования и расшифровки. Вряд ли требует изменения.
|
||||
|
||||
@@ -26,9 +26,9 @@ at Vitastor Kubernetes operator: https://github.com/Antilles7227/vitastor-operat
|
||||
The instruction is very simple.
|
||||
|
||||
1. Download a Docker image of the desired version: \
|
||||
`docker pull vitalif/vitastor:v3.0.15`
|
||||
`docker pull vitalif/vitastor:v3.0.12`
|
||||
2. Install scripts to the host system: \
|
||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.15 install.sh`
|
||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.12 install.sh`
|
||||
3. Reload udev rules: \
|
||||
`udevadm control --reload-rules`
|
||||
4. Enable the vitastor-host service: \
|
||||
|
||||
@@ -25,9 +25,9 @@ Vitastor можно установить в Docker/Podman. При этом etcd,
|
||||
Инструкция по установке максимально простая.
|
||||
|
||||
1. Скачайте Docker-образ желаемой версии: \
|
||||
`docker pull vitalif/vitastor:v3.0.15`
|
||||
`docker pull vitalif/vitastor:v3.0.12`
|
||||
2. Установите скрипты в хост-систему командой: \
|
||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.15 install.sh`
|
||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.12 install.sh`
|
||||
3. Перезагрузите правила udev: \
|
||||
`udevadm control --reload-rules`
|
||||
4. Включите сервис vitastor-host: \
|
||||
|
||||
@@ -41,16 +41,12 @@
|
||||
## Configure monitors
|
||||
|
||||
On the monitor hosts:
|
||||
- Create minimal configuration in `/etc/vitastor/vitastor.conf`:
|
||||
- Put identical etcd_address into `/etc/vitastor/vitastor.conf`. Example:
|
||||
```
|
||||
{
|
||||
"etcd_address": ["http://10.200.1.10:2379","http://10.200.1.11:2379","http://10.200.1.12:2379"],
|
||||
"osd_network": "10.200.1.0/24",
|
||||
"use_perms": false
|
||||
"etcd_address": ["10.200.1.10:2379","10.200.1.11:2379","10.200.1.12:2379"]
|
||||
}
|
||||
```
|
||||
- Note that you can enable encryption by using `https://` and `use_perms` option.
|
||||
[Details](security.en.md#quick-setup) about encryption setup with make-etcd.
|
||||
- Create systemd units for etcd by running: `/usr/lib/vitastor/mon/make-etcd`
|
||||
Or, if you installed Vitastor in Docker, run `systemctl start vitastor-host; docker exec vitastor make-etcd`.
|
||||
- Start etcd and monitors: `systemctl enable --now vitastor-etcd vitastor-mon`
|
||||
|
||||
@@ -41,23 +41,25 @@
|
||||
## Настройте мониторы
|
||||
|
||||
На хостах, выделенных под мониторы:
|
||||
- Создайте минимальную конфигурацию в `/etc/vitastor/vitastor.conf`:
|
||||
- Пропишите одинаковые etcd_address в `/etc/vitastor/vitastor.conf`. Например:
|
||||
```
|
||||
{
|
||||
"etcd_address": ["http://10.200.1.10:2379","http://10.200.1.11:2379","http://10.200.1.12:2379"],
|
||||
"osd_network": "10.200.1.0/24",
|
||||
"use_perms": false
|
||||
"etcd_address": ["10.200.1.10:2379","10.200.1.11:2379","10.200.1.12:2379"]
|
||||
}
|
||||
```
|
||||
- Обратите внимание, что с помощью схемы `https://` и опции `use_perms` можно включить шифрование.
|
||||
[Подробно](security.ru.md#быстрая-настройка) о настройке шифрования через make-etcd.
|
||||
- Инициализируйте сервисы etcd, запустив `/usr/lib/vitastor/mon/make-etcd`.\
|
||||
Либо, если вы установили Vitastor в Docker, запустите `systemctl start vitastor-host; docker exec vitastor make-etcd`.
|
||||
- Запустите etcd и мониторы: `systemctl enable --now vitastor-etcd vitastor-mon`
|
||||
|
||||
## Настройте OSD
|
||||
|
||||
- Создайте/скопируйте с узлов с мониторами файл конфигурации `/etc/vitastor/vitastor.conf`.
|
||||
- Пропишите etcd_address и [osd_network](../config/network.ru.md#osd_network) в `/etc/vitastor/vitastor.conf`. Например:
|
||||
```
|
||||
{
|
||||
"etcd_address": ["10.200.1.10:2379","10.200.1.11:2379","10.200.1.12:2379"],
|
||||
"osd_network": "10.200.1.0/24"
|
||||
}
|
||||
```
|
||||
- Инициализуйте OSD:
|
||||
- Только SSD или только HDD: `vitastor-disk prepare /dev/sdXXX [/dev/sdYYY ...]`.
|
||||
Если вы используете десктопные SSD без конденсаторов, добавьте опцию `--disable_data_fsync off`,
|
||||
|
||||
@@ -1,657 +0,0 @@
|
||||
[Documentation](../../README.md#documentation) → Introduction → Security in Vitastor
|
||||
|
||||
-----
|
||||
|
||||
[Читать на русском](security.ru.md)
|
||||
|
||||
# Security in Vitastor
|
||||
|
||||
- [Overview](#overview)
|
||||
- [Quick setup](#quick-setup)
|
||||
- Principles of operation
|
||||
- [etcd transport encryption (TLS)](#etcd-transport-encryption-tls)
|
||||
- [OSD transport encryption (AES-GCM)](#osd-transport-encryption-aes-gcm)
|
||||
- [End-to-end image data encryption (AES-XTS)](#end-to-end-image-data-encryption-aes-xts)
|
||||
- [Certificate-based authentication](#certificate-based-authentication)
|
||||
- [Users and access rights](#users-and-access-rights)
|
||||
- [etcd privileges](#etcd-privileges)
|
||||
- Manual setup
|
||||
- [Configuring OSD transport encryption](#configuring-osd-transport-encryption)
|
||||
- etcd/Antietcd setup options
|
||||
- [Mon with embedded Antietcd](#mon-with-embedded-antietcd)
|
||||
- [Mon as an Etcd proxy](#mon-as-an-etcd-proxy)
|
||||
- [Mon with a separate Antietcd Proxy](#mon-with-a-separate-antietcd-proxy)
|
||||
- [Standalone Antietcd without etcd](#standalone-antietcd-without-etcd)
|
||||
- [Vault/OpenBao setup](#vaultopenbao-setup)
|
||||
- [Vault setup example](#vault-setup-example)
|
||||
- Lists of allowed operations
|
||||
- [etcd data access rights](#etcd-data-access-rights)
|
||||
- [OSD data access rights](#osd-data-access-rights)
|
||||
- [API access rights](#api-access-rights)
|
||||
- [Encryption performance](#encryption-performance)
|
||||
|
||||
## Overview
|
||||
|
||||
Starting from version 3.1.0, Vitastor provides full data protection:
|
||||
control plane protection (etcd), data plane protection (OSDs), and end-to-end data encryption.
|
||||
|
||||
- Control plane protection:
|
||||
- etcd transport encryption (TLS)
|
||||
- Authentication via client TLS (X.509) certificates
|
||||
- Access control of clients to etcd data
|
||||
- Data plane protection:
|
||||
- Full AES-GCM encryption of OSD transport (similar to TLS, but faster)
|
||||
- Alternatively, AES-GCM encryption of just operation headers with data checksums using a secret "salt"
|
||||
- Authentication via client TLS (X.509) certificates
|
||||
- Access control of clients on the OSD side
|
||||
- End-to-end encryption:
|
||||
- Data is encrypted using AES-XTS on the client side, the Vitastor cluster has no access to plaintext data
|
||||
- AES-XTS keys can be stored in etcd or in an external Vault/OpenBao
|
||||
|
||||
All features are optional and disabled in the simplest configuration. By default, only
|
||||
transport-level data checksums ([proto_checksums](../config/security.en.md#proto_checksums)=payload)
|
||||
are enabled for clients that support them (>= 3.1.0). For older clients, connections
|
||||
without data checksums are allowed by default ([force_proto_checksums](../config/security.en.md#force_proto_checksums) is empty).
|
||||
|
||||
For a quick setup, jump to the [Quick setup](#quick-setup) section.
|
||||
|
||||
Descriptions of all security-related parameters can be found [here](../config/security.en.md).
|
||||
|
||||
## Quick setup
|
||||
|
||||
For a quick setup, use the `/usr/lib/vitastor/mon/make-etcd` script:
|
||||
|
||||
1. Log in to the node where the first monitor and etcd will be located.
|
||||
2. Create `/etc/vitastor/vitastor.conf` with minimal parameters: etcd_address,
|
||||
osd_network and, if you want to enable privileges, use_perms (note `https://`
|
||||
in etcd addresses):
|
||||
```
|
||||
{
|
||||
"etcd_address": ["https://10.0.0.10:2379","https://10.0.0.11:2379","https://10.0.0.12:2379"],
|
||||
"osd_network": "10.0.0.0/24",
|
||||
"use_perms": true
|
||||
}
|
||||
```
|
||||
3. Run `/usr/lib/vitastor/mon/make-etcd` without parameters or with the `--antietcd-only`
|
||||
parameter if you want to initialize the cluster with Antietcd only, without etcd.
|
||||
4. The script will generate all necessary certificates and offer to copy them to the other
|
||||
monitor nodes (agree!).
|
||||
5. Log in to all other monitor nodes and repeat the `/usr/lib/vitastor/mon/make-etcd` call there.
|
||||
6. If you also have nodes with OSDs only (without monitors), run the following command to
|
||||
copy only the required configuration to these nodes:
|
||||
```
|
||||
/usr/lib/vitastor/mon/make-etcd --copy-to-osd osdnode1,osdnode2,...
|
||||
```
|
||||
|
||||
After that, you can proceed with OSD initialization.
|
||||
|
||||
If you want to understand the setup in more detail, read the [Principles of operation](#principles-of-operation)
|
||||
and [Manual setup](#manual-setup) sections below.
|
||||
|
||||
## Principles of operation
|
||||
|
||||
### etcd transport encryption (TLS)
|
||||
|
||||
Possible setups:
|
||||
- Without encryption (http)
|
||||
- With encryption (https)
|
||||
- With encryption and client certificate authentication. Either the same certificate
|
||||
used for authentication on the OSD side (`cert`+`pkey` / `osd_cert`+`osd_pkey`)
|
||||
is used, or a separately specified certificate (`etcd_client_cert`+`etcd_client_key`).
|
||||
|
||||
### OSD transport encryption (AES-GCM)
|
||||
|
||||
Possible setups:
|
||||
- Unencrypted transport without checksums: `proto_checksums=none`.
|
||||
- Unencrypted transport with data checksums: `proto_checksums=payload` (may be omitted,
|
||||
this is the default value). It's allowed to disable checksums on the client side, or
|
||||
use an older client that does not support checksums. If you want to block connections
|
||||
from clients without checksums, use the option `force_proto_checksums=payload`.
|
||||
- Header-only encryption with data checksums: activated when the options
|
||||
`cert`, `pkey`, `osd_ca` are set on the client side and `osd_cert`, `osd_pkey`, `osd_ca`, `client_ca`
|
||||
on the OSD side, with `proto_checksums=payload`. In this mode, disabling checksums on the client
|
||||
side is forbidden by default, i.e. `force_proto_checksums=payload` is used.
|
||||
- Full transport encryption of all traffic: same as the previous option, but with `proto_checksums=gcm`.
|
||||
In this case, clients are by default allowed to downgrade to checksums only, but this
|
||||
can also be forbidden via `force_proto_checksums=gcm`. This is the slowest setup and
|
||||
it's only recommended for insecure (public) networks. In particular, full traffic
|
||||
encryption together with end-to-end AES-XTS image encryption encrypts data twice.
|
||||
|
||||
Encryption uses the AES-256-GCM algorithm and a custom simplified key exchange protocol,
|
||||
fully analogous to TLS 1.3 ECDHE.
|
||||
|
||||
### End-to-end image data encryption (AES-XTS)
|
||||
|
||||
The Vitastor client supports encrypting each image's data with its own key. In this case,
|
||||
data is encrypted by the client before sending it to OSDs and OSDs can't see it in plain.
|
||||
The encryption key can be changed when cloning/creating image snapshots. For example,
|
||||
you can make a base VM image (say, Debian Linux) unencrypted, but have encrypted client VM
|
||||
images inheriting from it.
|
||||
|
||||
Image encryption keys can be stored in etcd or in an external Vault. In the latter case,
|
||||
etcd only stores key IDs and Vitastor cluster can't decrypt the data at all. To use
|
||||
Vault, create an image with the `--enc_key vault:ID` option, specify vault_url and vault_ca
|
||||
options in the configuration, create accounts for all clients in Vault, and grant them access
|
||||
to the required v1 secrets.
|
||||
|
||||
Once again, if AES-XTS is used together with full traffic encryption (`proto_checksums=gcm`),
|
||||
image data is encrypted twice — first with AES-XTS, and then with AES-GCM. Use it only if
|
||||
you are completely paranoid :-).
|
||||
|
||||
### Certificate-based authentication
|
||||
|
||||
When encryption is enabled, Vitastor clients, OSDs, and monitors authenticate via certificates
|
||||
for both etcd (Antietcd) and OSD connections.
|
||||
|
||||
Separate certificates must be used for OSDs and monitors — either self-signed, or signed
|
||||
by separate CAs (`osd_ca` and `mon_ca`). All OSDs can use the same certificate, and all
|
||||
monitors can also use the same certificate, since the privileges of different OSDs or
|
||||
different monitors do not differ (theoretically, one could differentiate OSD certificates
|
||||
by pool, but there has been no need for this so far).
|
||||
|
||||
Also, a monitor certificate may not be needed at all if Antietcd is embedded into the monitor
|
||||
itself. In this case, the monitor already has access to all etcd data directly in memory.
|
||||
|
||||
### Users and access rights
|
||||
|
||||
When transport encryption is disabled, Vitastor operates without access control, i.e.,
|
||||
any cluster client has full access to both the management layer and the data layer. This
|
||||
option is suitable for dedicated trusted storage networks.
|
||||
|
||||
When OSD transport encryption is enabled (at least for headers), you can enable access
|
||||
rights by turning on the `use_perms=true` option. When this option is enabled, each user
|
||||
can perform only the operations that they are permitted, and even OSDs and monitors are
|
||||
also forbidden from performing "unnecessary" operations.
|
||||
|
||||
Each user (or administrator) must have their own certificate signed by a common root
|
||||
certificate for clients (`client_ca`), with a Common Name equal to the user name.
|
||||
Privilege settings are stored in etcd. OSDs and monitors don't need user accounts;
|
||||
they authenticate via separate certificates.
|
||||
|
||||
User privileges are stored in etcd data under the keys `/vitastor/config/user/<name>`.
|
||||
The following is defined per user in this key:
|
||||
- Type:
|
||||
- Client (`type=client` or omitted) — can only read and modify explicitly permitted images.
|
||||
- Administrator (`type=admin`) — can read and modify all images, and also administer the
|
||||
cluster: view overall statistics and status, create and delete OSDs, etc.
|
||||
- List of group names the user is a member of.
|
||||
|
||||
Images have the following properties:
|
||||
- Owner (owner) — the user name that is allowed to both read and modify the image
|
||||
- Owner group (owner_group) — the owner group name
|
||||
- Reader group (reader_group) — the name of the group of users allowed to read the image
|
||||
|
||||
And there is also a property on the pool:
|
||||
- Creator group (creator_group) — the name of the group of users allowed to create images in the pool
|
||||
|
||||
For the list of allowed operations on image data on the OSD side, see the
|
||||
[OSD data access rights](#osd-data-access-rights) section.
|
||||
|
||||
### etcd privileges
|
||||
|
||||
etcd privileges are implemented through Antietcd in all modes of operation.
|
||||
|
||||
Built-in etcd privileges are not supported due to numerous inconveniences:
|
||||
- Certificate-based authentication does not work at all in etcd's REST interface,
|
||||
- Privileges are stored separately from k/v data and cannot participate in transactions,
|
||||
- Only the administrator (root) can change privileges,
|
||||
- There is no support for filtering range read responses by privileges.
|
||||
|
||||
If etcd is used, Antietcd acts as a filtering proxy and can be embedded in the Vitastor
|
||||
monitor or run separately. In this case, etcd must allow incoming connections only from
|
||||
Antietcd, and all other components must connect to Antietcd.
|
||||
|
||||
If Antietcd runs as a part of the Vitastor monitor, it is sufficient to enable the option
|
||||
`use_perms=true` and set the required certificates. If Antietcd is run separately, privileges
|
||||
have to be enabled separately using Antietcd options. For more details on the setup, see
|
||||
the [etcd/Antietcd setup options](#etcdantietcd-setup-options) section.
|
||||
|
||||
For the list of allowed operations with etcd data, see the
|
||||
[etcd data access rights](#etcd-data-access-rights) section.
|
||||
|
||||
## Manual setup
|
||||
|
||||
### Configuring OSD transport encryption
|
||||
|
||||
You need 2 certificates: one for OSDs and one for signing all client certificates.
|
||||
For OSDs, you can use a self-signed certificate (osd_ca.crt) or a separate certificate (osd.crt)
|
||||
signed by a trusted osd_ca.crt certificate. For clients, you must use separate certificates
|
||||
signed by a common trusted (client_ca.crt).
|
||||
|
||||
Add to the Vitastor configuration on OSD servers:
|
||||
- use_perms: true
|
||||
- osd_ca: osd_ca.crt
|
||||
- client_ca: client_ca.crt
|
||||
- osd_cert: osd_ca.crt
|
||||
- osd_pkey: osd_ca.key
|
||||
|
||||
On the client side:
|
||||
- use_perms: true
|
||||
- cert: client.crt
|
||||
- pkey: client.key
|
||||
|
||||
### etcd/Antietcd setup options
|
||||
|
||||
The following configuration options are available:
|
||||
|
||||
#### Mon with embedded Antietcd
|
||||
|
||||
The simplest option. You need 1 certificate for Antietcd (antietcd.crt), plus root
|
||||
certificates for OSDs and clients.
|
||||
|
||||
Vitastor settings (`/etc/vitastor/vitastor.conf`):
|
||||
- etcd_address: [ "http://mon1:2379", ... ] (addresses of your monitors with port 2379)
|
||||
- use_perms: true
|
||||
- use_antietcd: true
|
||||
- antietcd_cert: antietcd.crt
|
||||
- antietcd_key: antietcd.key
|
||||
- etcd_ca: antietcd.crt
|
||||
- osd_ca: osd_ca.crt
|
||||
- client_ca: client_ca.crt
|
||||
|
||||
#### Mon as an Etcd proxy
|
||||
|
||||
If you want to enable privileges, but stay on etcd, you can use etcd proxy mode.
|
||||
|
||||
You will need 2 separate certificates: one for etcd (etcd.crt) and one for antietcd (antietcd.crt).
|
||||
The etcd client port must be different from the standard 2379 — for example, you can pick 2381.
|
||||
OSD and client certificates are also needed.
|
||||
|
||||
Vitastor settings:
|
||||
- etcd_address: [ "http://mon1:2379", ... ] (addresses of your monitors with port 2379)
|
||||
- use_perms: true
|
||||
- use_antietcd: true
|
||||
- etcd_proxy:
|
||||
```
|
||||
{
|
||||
"urls": [ "http://mon1:2381", ... ], // addresses of your etcd with port 2381
|
||||
"cert": "antietcd.crt",
|
||||
"key": "antietcd.key",
|
||||
"ca": "etcd.crt"
|
||||
}
|
||||
```
|
||||
- antietcd_cert: antietcd.crt
|
||||
- antietcd_key: antietcd.key
|
||||
- etcd_ca: antietcd.crt
|
||||
- osd_ca: osd_ca.crt
|
||||
- client_ca: client_ca.crt
|
||||
|
||||
etcd command-line options:
|
||||
```
|
||||
--advertise-client-urls=https://<ADDRESS>:2381 --listen-client-urls=https://<ADDRESS>:2381 \
|
||||
--client-cert-auth --cert-file=etcd.crt --key-file=etcd.key --trusted-ca-file=antietcd.crt \
|
||||
--peer-client-cert-auth --peer-cert-file=etcd.crt --peer-key-file=etcd.key --peer-trusted-ca-file=etcd.crt
|
||||
```
|
||||
|
||||
#### Mon with a separate Antietcd Proxy
|
||||
|
||||
If in addition to the previous option you want to offload Antietcd from the Vitastor monitor's
|
||||
tasks, you can run it separately.
|
||||
|
||||
Similar to the previous option, 2 certificates are needed: one for etcd and one for antietcd,
|
||||
plus separate certificates for clients, OSDs, and monitors will be needed.
|
||||
|
||||
Vitastor settings:
|
||||
- etcd_address: [ "http://mon1:2379", ... ] (addresses of your monitors with port 2379)
|
||||
- use_perms: true
|
||||
- use_antietcd: false
|
||||
- etcd_ca: antietcd.crt
|
||||
- osd_ca: osd_ca.crt
|
||||
- client_ca: client_ca.crt
|
||||
- mon_etcd_client_cert: mon_ca.crt
|
||||
- mon_etcd_client_key: mon_ca.key
|
||||
|
||||
Antietcd command-line options:
|
||||
```
|
||||
--port 2379 \
|
||||
--client_cert_auth 1 --auth_filter vitastor_auth_filter.js --etcd_proxy url1,url2,... \
|
||||
--cert antietcd.crt --key antietcd.key --ca client_ca.crt --osd_ca osd_ca.crt --mon_ca mon_ca.crt \
|
||||
--etcd_cert antietcd.crt --etcd_key antietcd.key --etcd_ca etcd.crt
|
||||
```
|
||||
|
||||
etcd command-line options (same as in the previous option):
|
||||
```
|
||||
--advertise-client-urls=https://<ADDRESS>:2381 --listen-client-urls=https://<ADDRESS>:2381 \
|
||||
--client-cert-auth --cert-file=etcd.crt --key-file=etcd.key --trusted-ca-file=antietcd.crt \
|
||||
--peer-client-cert-auth --peer-cert-file=etcd.crt --peer-key-file=etcd.key --peer-trusted-ca-file=etcd.crt
|
||||
```
|
||||
|
||||
#### Standalone Antietcd without etcd
|
||||
|
||||
Same as the previous option, but etcd and its certificate are not needed:
|
||||
|
||||
Vitastor settings (same as in the previous option):
|
||||
- etcd_address: [ "http://mon1:2379", ... ] (addresses of your monitors with port 2379)
|
||||
- use_perms: true
|
||||
- use_antietcd: false
|
||||
- etcd_ca: antietcd.crt
|
||||
- osd_ca: osd_ca.crt
|
||||
- client_ca: client_ca.crt
|
||||
- mon_etcd_client_cert: mon_ca.crt
|
||||
- mon_etcd_client_key: mon_ca.key
|
||||
|
||||
Antietcd command-line options:
|
||||
```
|
||||
--port 2379 \
|
||||
--client_cert_auth 1 --auth_filter vitastor_auth_filter.js \
|
||||
--persist_filter vitastor_persist_filter.js \
|
||||
--cert antietcd.crt --key antietcd.key --ca client_ca.crt --osd_ca osd_ca.crt --mon_ca mon_ca.crt
|
||||
```
|
||||
|
||||
### Vault/OpenBao setup
|
||||
|
||||
To use Vault, each client that needs to get image keys from Vault needs a Vault account.
|
||||
Vitastor only supports client certificate-based authentication, so all client certificates
|
||||
(`cert`+`pkey`) must be registered in Vault, and they must be granted access to the
|
||||
corresponding secrets (v1 secrets API is supported).
|
||||
|
||||
The required format of a Vault secret is a single `key` field as a hexadecimal string.
|
||||
The AES-256-XTS algorithm is used, so the key length is 64 bytes, i.e., the string must
|
||||
consist of 128 hexadecimal digits.
|
||||
|
||||
To connect to Vault, set the following settings in Vitastor.conf:
|
||||
- `vault_url` — Vault address (e.g., `https://vault:8200`)
|
||||
- `vault_ca` — Vault's own certificate
|
||||
|
||||
After that, if you create an image (`vitastor-cli create`) with the option `--enc_key vault:<ID>`,
|
||||
Vitastor clients will first contact Vault to obtain a token at `/v1/auth/cert/login`,
|
||||
and then request the actual secret from Vault at `/v1/secret/<ID>`.
|
||||
|
||||
#### Vault setup example
|
||||
|
||||
Step-by-step instructions for setting up a test Vault using OpenBao as an example:
|
||||
|
||||
1. If TLS is not yet configured, generate a self-signed TLS certificate for Vault:
|
||||
```
|
||||
openssl req -days 3650 -x509 -addext basicConstraints=critical,CA:TRUE,pathlen:1 --addext subjectAltName=DNS:vault \
|
||||
-new -newkey rsa:4096 -nodes -keyout /etc/openbao/vault.key -out /etc/openbao/vault.crt
|
||||
```
|
||||
Configure it in `/etc/openbao/openbao.hcl`:
|
||||
```
|
||||
listener "tcp" {
|
||||
address = "0.0.0.0:8200"
|
||||
tls_cert_file = "/etc/openbao/vault.crt"
|
||||
tls_key_file = "/etc/openbao/vault.key"
|
||||
}
|
||||
```
|
||||
And restart OpenBao (`systemctl restart openbao`).
|
||||
2. Copy Vault's TLS certificate for Vitastor:
|
||||
```
|
||||
cp /etc/openbao/vault.crt /etc/vitastor/vault.crt
|
||||
```
|
||||
Transfer it to all client nodes and specify it in `/etc/vitastor/vitastor.conf`:
|
||||
```
|
||||
{
|
||||
...
|
||||
"vault_url": "http://vault:8200",
|
||||
"vault_ca": "/etc/vitastor/vault.crt"
|
||||
}
|
||||
```
|
||||
3. Check Vault status:
|
||||
```
|
||||
bao status -ca-cert /etc/openbao/vault.crt -address=https://vault:8200
|
||||
```
|
||||
4. Initialize Vault in test mode from 1 node (with 1 key share):
|
||||
```
|
||||
bao operator init -n 1 -t 1 -ca-cert /etc/openbao/vault.crt -address=https://vault:8200
|
||||
```
|
||||
5. Unseal Vault:
|
||||
```
|
||||
bao operator unseal -ca-cert /etc/openbao/vault.crt -address=https://vault:8200
|
||||
```
|
||||
6. Enable certificate-based authentication:
|
||||
```
|
||||
bao auth enable -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 cert
|
||||
```
|
||||
7. Enable v1 secrets:
|
||||
```
|
||||
bao secrets enable -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 -path=secret kv-v1
|
||||
```
|
||||
8. Create a test secret:
|
||||
```
|
||||
bao kv put -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 secret/vitastor/testimg3 key=$(openssl rand -hex 64)
|
||||
```
|
||||
9. Generate a signed certificate for a Vitastor user (on a machine where you have `client_ca.crt` and `client_ca.key`):
|
||||
```
|
||||
openssl req -subj '/CN=testimg3' -nodes -new -keyout testimg3.key -out testimg3.csr
|
||||
openssl x509 -req -days 3650 -CA client_ca.crt -CAkey client_ca.key -CAcreateserial -in testimg3.csr -out testimg3.crt
|
||||
rm testimg3.csr
|
||||
```
|
||||
10. Create a user in Vault and grant it access to the secret:
|
||||
```
|
||||
cat >testimg3.policy <<EOF
|
||||
path "/secret/vitastor/testimg3" {
|
||||
capabilities = ["read"]
|
||||
}
|
||||
EOF
|
||||
|
||||
bao policy write -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 testimg3 testimg3.policy
|
||||
|
||||
bao write -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 auth/cert/certs/testimg3 \
|
||||
certificate=@testimg3.crt display_name=testimg3 token_ttl=24h token_policies=testimg3
|
||||
```
|
||||
11. Test access to the secret:
|
||||
```
|
||||
curl --cacert /etc/vitastor/vault.crt --cert testimg3.crt --key testimg3.key \
|
||||
--json '{}' https://vault:8200/v1/auth/cert/login
|
||||
```
|
||||
A token will be printed, substitute it into the following request:
|
||||
```
|
||||
curl --cacert /etc/vitastor/vault.crt --cert testimg3.crt --key testimg3.key \
|
||||
-H 'X-Vault-Token: <RECEIVED TOKEN>' https://vault:8200/v1/secret/vitastor/testimg3
|
||||
```
|
||||
12. Create an image in Vitastor with the given secret (as an administrator or someone who
|
||||
has the right to create images in your pool):
|
||||
```
|
||||
vitastor-cli create -s 100G --enc_key vault:vitastor/testimg3 --owner testimg3 testimg3
|
||||
```
|
||||
13. Test access to the image as user testimg3:
|
||||
```
|
||||
vitastor-cli --cert testimg3.crt --pkey testimg3.key dd if=/dev/urandom oimg=testimg3 bs=1M count=100
|
||||
```
|
||||
|
||||
## Lists of allowed operations
|
||||
|
||||
### etcd data access rights
|
||||
|
||||
Below, all key names are given without the common prefix `/vitastor`.
|
||||
|
||||
Allowed operations with keys in Antietcd for clients (`type=client`):
|
||||
- Read-only:
|
||||
- Always allowed:
|
||||
- `/config/global`
|
||||
- `/config/node_placement`
|
||||
- `/config/pools`
|
||||
- `/pg/config`
|
||||
- `/osd/state/*`
|
||||
- `/pg/state/*`
|
||||
- `/index/maxid/*`
|
||||
- For images [readable by the user](#users-and-access-rights):
|
||||
- `/config/inode/*`
|
||||
- `/index/image/*`
|
||||
- `/inode/stats/*`
|
||||
- Read and write:
|
||||
- For pools in which the user can create images:
|
||||
- `/index/maxid/*`
|
||||
- For images owned by the user:
|
||||
- `/config/inode/*`
|
||||
- `/index/image/*`
|
||||
|
||||
Allowed operations with keys in Antietcd for administrators (`type=admin`):
|
||||
- Read:
|
||||
- `/stats`
|
||||
- `/mon/*`
|
||||
- `/pg/*`
|
||||
- `/pgstats/*`
|
||||
- `/inode/stats/*`
|
||||
- `/pool/stats/*`
|
||||
- Read and write:
|
||||
- `/config/*`
|
||||
- `/osd/*`
|
||||
- `/index/*`
|
||||
- `/pg/history/*`
|
||||
|
||||
Allowed operations with keys in etcd for OSDs:
|
||||
- Read:
|
||||
- `/pg/config`
|
||||
- `/config/*`
|
||||
- Read and write:
|
||||
- `/osd/*`
|
||||
- `/pg/state/*`
|
||||
- `/pg/history/*`
|
||||
- `/pgstats/*`
|
||||
|
||||
Allowed operations with keys in etcd for monitors:
|
||||
- Read:
|
||||
- `/config/*`
|
||||
- `/osd/*`
|
||||
- `/pgstats/*`
|
||||
- Read and write:
|
||||
- `/pg/config`
|
||||
- `/stats`
|
||||
- `/history/last_clean_pgs`
|
||||
- `/mon/*`
|
||||
- `/pg/history/*`
|
||||
- `/inode/stats/*`
|
||||
- `/pool/stats/*`
|
||||
|
||||
### OSD data access rights
|
||||
|
||||
When the `use_perms` option and encryption are enabled, OSDs authenticate clients via
|
||||
certificates and allow each client only what is allowed by the access control model.
|
||||
|
||||
Client operations:
|
||||
- READ — allowed for images the user has read access to.
|
||||
- WRITE, DELETE, SCRUB — allowed for images the user has write access to.
|
||||
- SYNC — the operation is not tied to an image and is always allowed.
|
||||
- DESCRIBE — the operation is allowed only for administrators (used by the commands
|
||||
`vitastor-cli describe` and `fix`).
|
||||
- PING — the operation is always allowed.
|
||||
- SHOW_CONFIG — the operation is always allowed, however, if the client presents
|
||||
itself as an OSD in it, then it is verified that it uses a certificate signed by `osd_ca`.
|
||||
- SEC_LIST (listing) — allowed for other OSDs and administrators with any parameters,
|
||||
and for regular clients only allowed for requests limited to an image the user has
|
||||
read access to.
|
||||
|
||||
Cluster operations — allowed only for other OSDs:
|
||||
- SEC_READ
|
||||
- SEC_WRITE
|
||||
- SEC_WRITE_STABLE
|
||||
- SEC_SYNC
|
||||
- SEC_STABILIZE
|
||||
- SEC_ROLLBACK
|
||||
- SEC_DELETE
|
||||
- SEC_READ_BMP
|
||||
- SEC_LOCK
|
||||
|
||||
### API access rights
|
||||
|
||||
[vitastor-cli serve](../usage/cli.en.md#serve) also supports client authentication
|
||||
via certificates. Only certificates signed by `client_ca` are accepted. A separate
|
||||
certificate `server_cert` with the key `server_pkey` is used as the server certificate.
|
||||
|
||||
For `vitastor-cli serve` to work correctly, it itself must use a certificate
|
||||
(`cert`+`pkey`) of a user with administrator rights (`type=admin`) to access Vitastor.
|
||||
|
||||
Regular clients, when accessing the API, are only allowed API operations on images
|
||||
available to them either for reading (for reads) or for writing (for modification).
|
||||
All other API calls are allowed only for administrators.
|
||||
|
||||
List of allowed API operations:
|
||||
|
||||
Clients (users with `type=client`) are allowed the following operations:
|
||||
- image/list — for images the user can read.
|
||||
- image/create — for pools in which the user is allowed to create images, or for
|
||||
creating snapshots of images owned by the user.
|
||||
- image/delete, image/flatten, image/modify — for images owned by the user.
|
||||
|
||||
All other operations are allowed only for administrators (`type=admin`).
|
||||
|
||||
## Encryption performance
|
||||
|
||||
You may wonder — how fast is all this wonderful encryption?
|
||||
|
||||
The answer is — it depends heavily on the CPU. On modern processors (with AVX512 with VAES
|
||||
support) it is very fast — AES encryption speed can reach 10-20 GB/s and above. This
|
||||
primarily concerns the CPU of client machines, because end-to-end encryption is performed
|
||||
entirely on the client, and client uses its signle thread for transport encryption too,
|
||||
while there are many OSDs on the server side, and it is easier to add resources there.
|
||||
|
||||
On older processors, the speed is noticeably worse — for example, on a Xeon E5 v4 it is
|
||||
only 3 GB/s.
|
||||
|
||||
You can evaluate the performance of your processors using the `vitastor-cli cpubench` command.
|
||||
|
||||
Example output (💪 AMD EPYC 9575F):
|
||||
|
||||
```
|
||||
$ vitastor-cli cpubench
|
||||
Vitastor transport encryption benchmark (AES-256-GCM, AES-256-XTS and xxhash3)
|
||||
|
||||
Warmup...
|
||||
|
||||
No transport encryption, data checksums enabled, e2e unencrypted image
|
||||
xxhash3 1 M block... 209000 iterations in 2001 ms = 104447.78 MB/s
|
||||
xxhash3 4 K block... 37000000 iterations in 2022 ms = 71479.35 MB/s
|
||||
|
||||
Header encryption with payload checksums, e2e unencrypted image
|
||||
AES-256-GCM encrypt header + xxhash3 1 M block... 210000 iterations in 2015 ms = 104218.36 MB/s
|
||||
AES-256-GCM encrypt header + xxhash3 4 K block... 26000000 iterations in 2073 ms = 48993.01 MB/s
|
||||
|
||||
Full transport encryption, e2e unencrypted image
|
||||
AES-256-GCM encrypt header and 1 M block... 54000 iterations in 2000 ms = 27000.00 MB/s
|
||||
AES-256-GCM encrypt header and 4 K block... 11700000 iterations in 2014 ms = 22692.71 MB/s
|
||||
|
||||
No transport encryption, no checksums, e2e encrypted image
|
||||
AES-256-XTS encrypt 1 M block... 50000 iterations in 2039 ms = 24521.82 MB/s
|
||||
AES-256-XTS encrypt 4 K block... 12600000 iterations in 2009 ms = 24499.13 MB/s
|
||||
|
||||
No transport encryption, e2e encrypted image, data checksums enabled
|
||||
AES-256-XTS encrypt + xxhash3 1 M block... 40000 iterations in 2013 ms = 19870.84 MB/s
|
||||
AES-256-XTS encrypt + xxhash3 4 K block... 10200000 iterations in 2011 ms = 19812.90 MB/s
|
||||
|
||||
Header encryption with payload checksums, e2e encrypted image
|
||||
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 1 M block... 40000 iterations in 2014 ms = 19860.97 MB/s
|
||||
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 4 K block... 8700000 iterations in 2011 ms = 16899.24 MB/s
|
||||
|
||||
Full transport encryption, e2e encrypted image
|
||||
AES-256-XTS + AES-256-GCM encrypt 1 M block... 26000 iterations in 2062 ms = 12609.12 MB/s
|
||||
AES-256-XTS + AES-256-GCM encrypt 4 K block... 6300000 iterations in 2006 ms = 12267.88 MB/s
|
||||
```
|
||||
|
||||
And here is Xeon E5-2680v4:
|
||||
|
||||
```
|
||||
$ vitastor-cli cpubench
|
||||
Vitastor transport encryption benchmark (AES-256-GCM, AES-256-XTS and xxhash3)
|
||||
|
||||
Warmup...
|
||||
|
||||
No transport encryption, data checksums enabled, e2e unencrypted image
|
||||
xxhash3 1 M block... 62000 iterations in 2021 ms = 30677.88 MB/s
|
||||
xxhash3 4 K block... 12400000 iterations in 2006 ms = 24146.31 MB/s
|
||||
|
||||
Header encryption with payload checksums, e2e unencrypted image
|
||||
AES-256-GCM encrypt header + xxhash3 1 M block... 62000 iterations in 2027 ms = 30587.07 MB/s
|
||||
AES-256-GCM encrypt header + xxhash3 4 K block... 6800000 iterations in 2011 ms = 13208.60 MB/s
|
||||
|
||||
Full transport encryption, e2e unencrypted image
|
||||
AES-256-GCM encrypt header and 1 M block... 7000 iterations in 2317 ms = 3021.15 MB/s
|
||||
AES-256-GCM encrypt header and 4 K block... 1500000 iterations in 2102 ms = 2787.52 MB/s
|
||||
|
||||
No transport encryption, no checksums, e2e encrypted image
|
||||
AES-256-XTS encrypt 1 M block... 7000 iterations in 2317 ms = 3021.15 MB/s
|
||||
AES-256-XTS encrypt 4 K block... 1600000 iterations in 2088 ms = 2993.30 MB/s
|
||||
|
||||
No transport encryption, e2e encrypted image, data checksums enabled
|
||||
AES-256-XTS encrypt + xxhash3 1 M block... 6000 iterations in 2188 ms = 2742.23 MB/s
|
||||
AES-256-XTS encrypt + xxhash3 4 K block... 1400000 iterations in 2053 ms = 2663.78 MB/s
|
||||
|
||||
Header encryption with payload checksums, e2e encrypted image
|
||||
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 1 M block... 6000 iterations in 2190 ms = 2739.73 MB/s
|
||||
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 4 K block... 1300000 iterations in 2101 ms = 2417.00 MB/s
|
||||
|
||||
Full transport encryption, e2e encrypted image
|
||||
AES-256-XTS + AES-256-GCM encrypt 1 M block... 4000 iterations in 2666 ms = 1500.38 MB/s
|
||||
AES-256-XTS + AES-256-GCM encrypt 4 K block... 800000 iterations in 2113 ms = 1478.94 MB/s
|
||||
```
|
||||
+225
-448
@@ -1,105 +1,56 @@
|
||||
[Документация](../../README-ru.md#документация) → Введение → Безопасность в Vitastor
|
||||
[Документация](../../README-ru.md#документация) → Безопасность
|
||||
|
||||
-----
|
||||
|
||||
[Read in English](security.en.md)
|
||||
|
||||
# Безопасность в Vitastor
|
||||
# Оглавление
|
||||
|
||||
⚠️ Предупреждение: детальное описание настроек безопасности достаточно длинное.
|
||||
|
||||
Если не боитесь - читайте [Подробное описание](#подробное-описание).
|
||||
|
||||
Если хотите просто быстро настроить Vitastor с шифрованием - читайте начало статьи.
|
||||
|
||||
- [Обзор](#обзор)
|
||||
- [Быстрая настройка](#быстрая-настройка)
|
||||
- Принципы работы
|
||||
- [Шифрование соединений с etcd (TLS)](#шифрование-соединений-с-etcd-tls)
|
||||
- [Шифрование соединений с OSD (AES-GCM)](#шифрование-соединений-с-osd-aes-gcm)
|
||||
- [Сквозное шифрование данных образов (AES-XTS)](#сквозное-шифрование-данных-образов-aes-xts)
|
||||
- [Аутентификация по сертификатам](#аутентификация-по-сертификатам)
|
||||
- [Пользователи и права доступа](#пользователи-и-права-доступа)
|
||||
- [Привилегии etcd](#привилегии-etcd)
|
||||
- Ручная настройка
|
||||
- [Настройка шифрования соединений OSD](#настройка-шифрования-соединений-osd)
|
||||
- Варианты настройки etcd/Antietcd
|
||||
- [Mon со встроенным Antietcd](#mon-со-встроенным-antietcd)
|
||||
- [Mon в роли Etcd proxy](#mon-в-роли-etcd-proxy)
|
||||
- [Mon с отдельным Antietcd Proxy](#mon-с-отдельным-antietcd-proxy)
|
||||
- [Отдельный Antietcd без etcd](#отдельный-antietcd-без-etcd)
|
||||
- [Настройка Vault/OpenBao](#настройка-vaultopenbao)
|
||||
- [Пример настройки Vault](#пример-настройки-vault)
|
||||
- Списки разрешённых операций
|
||||
- [Права доступа к данным etcd](#права-доступа-к-данным-etcd)
|
||||
- [Права доступа к данным OSD](#права-доступа-к-данным-osd)
|
||||
- [Права доступа к API](#права-доступа-к-api)
|
||||
- [Производительность шифрования](#производительность-шифрования)
|
||||
-
|
||||
|
||||
## Обзор
|
||||
# Быстрая настройка
|
||||
|
||||
Начиная с версии 3.1.0, Vitastor предоставляет полную защиту данных: защиту слоя
|
||||
управления (etcd), защиту слоя данных (OSD) и сквозное шифрование данных.
|
||||
|
||||
- Защита слоя управления:
|
||||
- Шифрование соединений с etcd (TLS)
|
||||
- Аутентификация по клиентским TLS (X.509) сертификатам
|
||||
- Разграничение прав доступа клиентов к данным etcd
|
||||
- Защита слоя данных:
|
||||
- Либо полное AES-GCM шифрование соединений с OSD (аналогично TLS, но быстрее)
|
||||
- Либо шифрование AES-GCM только заголовков команд с контрольными суммами данных с секретной "солью"
|
||||
- Аутентификация по клиентским TLS (X.509) сертификатам
|
||||
- Разграничение прав доступа клиентов на стороне OSD
|
||||
- Сквозное шифрование:
|
||||
- Данные шифруются AES-XTS на стороне клиента, кластер Vitastor не имеет доступа к открытым данным
|
||||
- Ключи AES-XTS могут храниться в etcd или во внешнем Vault/OpenBao
|
||||
|
||||
Все функции опциональны и в простейшем варианте настройки выключены. По умолчанию включены
|
||||
только контрольные суммы данных на транспортном уровне ([proto_checksums](../config/security.ru.md#proto_checksums)=payload) для
|
||||
поддерживающих их клиентов (>= 3.1.0). Для более старых клиентов по умолчанию разрешены
|
||||
соединения без контрольных сумм данных ([force_proto_checksums](../config/security.ru.md#force_proto_checksums) пусто).
|
||||
# Пользовательские сценарии
|
||||
|
||||
Для быстрой настройки перейдите к разделу [Быстрая настройка](#быстрая-настройка).
|
||||
Зачем всё это нужно вам?
|
||||
|
||||
Описания всех параметров, связанных с безопасностью, читайте [здесь](../config/security.ru.md).
|
||||
|
||||
## Быстрая настройка
|
||||
|
||||
Для быстрой настройки используйте скрипт `/usr/lib/vitastor/mon/make-etcd`:
|
||||
# Подробное описание
|
||||
|
||||
1. Зайдите на узел, на котором будет располагаться первый монитор и etcd.
|
||||
2. Создайте там минимальный `/etc/vitastor/vitastor.conf` с параметрами etcd_address,
|
||||
osd_network и, если хотите включить привилегии - use_perms (обратите внимание на `https://`
|
||||
в адресах etcd):
|
||||
```
|
||||
{
|
||||
"etcd_address": ["https://10.0.0.10:2379","https://10.0.0.11:2379","https://10.0.0.12:2379"],
|
||||
"osd_network": "10.0.0.0/24",
|
||||
"use_perms": true
|
||||
}
|
||||
```
|
||||
3. Запустите `/usr/lib/vitastor/mon/make-etcd` без параметров или с параметром `--antietcd-only`,
|
||||
если хотите инициализировать кластер только с Antietcd без etcd.
|
||||
4. Скрипт сгенерирует все необходимые сертификаты и предложит скопировать их на остальные узлы
|
||||
мониторов (соглашайтесь!).
|
||||
5. Зайдите на все остальные узлы мониторов и повторите там вызов `/usr/lib/vitastor/mon/make-etcd`.
|
||||
6. Если у вас будут узлы только с OSD без мониторов, выполните следующую команду, чтобы скопировать
|
||||
только нужную конфигурацию на эти узлы:
|
||||
```
|
||||
/usr/lib/vitastor/mon/make-etcd --copy-to-osd osdnode1,osdnode2,...
|
||||
```
|
||||
Начиная с версии 3.1.0, в Vitastor есть следующие функции:
|
||||
1. Шифрование соединений с etcd (TLS)
|
||||
2. Шифрование соединений с OSD (AES-GCM) - по выбору либо только заголовков, либо и заголовков, и данных
|
||||
3. Сквозное шифрование данных образов (AES-XTS)
|
||||
4. Хранения ключей шифрования AES-XTS во внешнем Vault
|
||||
5. Контрольных сумм данных на транспортном уровне с секретной "солью"
|
||||
6. Аутентификация с помощью TLS (X.509) сертификатов и закрытых ключей
|
||||
7. Разграничение прав доступа клиентов к данным etcd
|
||||
8. Разграничение прав доступа клиентов к данным самих образов (на стороне OSD)
|
||||
|
||||
После этого можете переходить к инициализации OSD.
|
||||
По умолчанию шифрование, аутентификация и авторизация отключены, но, начиная с 3.1.0,
|
||||
используются контрольные суммы данных на транспортном уровне (`proto_checksums=payload`).
|
||||
|
||||
Если хотите разобраться в настройке подробнее, читайте далее разделы [Принципы работы](#принципы-работы)
|
||||
и [Ручная настройка](#ручная-настройка).
|
||||
|
||||
## Принципы работы
|
||||
|
||||
### Шифрование соединений с etcd (TLS)
|
||||
## Шифрование соединений с etcd (TLS)
|
||||
|
||||
Варианты настройки:
|
||||
- Без шифрования (http)
|
||||
- С шифрованием (https)
|
||||
- С шифрованием и аутентификацией по клиентским сертификатам. Используется либо тот
|
||||
же сертификат, что используется для аутентификации на стороне OSD (`cert`+`pkey` / `osd_cert`+`osd_pkey`),
|
||||
либо отдельно указанный сертификат (`etcd_client_cert`+`etcd_client_key`)
|
||||
- С клиентским сертификатом, но при выключенной авторизации (`use_auth=false`) - используется
|
||||
отдельный сертификат и ключ: `etcd_client_cert`, `etcd_client_key`
|
||||
- С клиентским сертификатом, при включённой аутентификации на уровне OSD - используется общий
|
||||
сертификат и ключ: для OSD - `osd_cert` и `osd_pkey`, для клиентов - `cert` и `pkey`
|
||||
|
||||
### Шифрование соединений с OSD (AES-GCM)
|
||||
## Шифрование соединений с OSD (AES-GCM)
|
||||
|
||||
Варианты настройки:
|
||||
- Без шифрования и без контрольных сумм: `proto_checksums=none`.
|
||||
@@ -114,31 +65,46 @@
|
||||
отключение контрольных сумм на уровне клиента, то есть используется `force_proto_checksums=payload`.
|
||||
- С полным шифрованием всего трафика: аналогично прошлому варианту, но с `proto_checksums=gcm`.
|
||||
Клиенту при этом по умолчанию разрешается понизить уровень защиты до контрольных сумм, но
|
||||
это тоже можно запретить через `force_proto_checksums=gcm`. Данный вариант самый медленный и
|
||||
рекомендуется только для небезопасных (публичных) сетей. В том числе потому, что при использовании
|
||||
и полного шифрования трафика, и сквозного шифрования образов AES-XTS, данные шифруются дважды.
|
||||
это тоже можно запретить через `force_proto_checksums=gcm`. Данный вариант не является рекомендуемым,
|
||||
так как добавлен в первую очередь для возможной поддержки небезопасных (публичных) сетей и
|
||||
больше всего снижает производительность. В частности, если одновременно использовать полное
|
||||
шифрование трафика и сквозное шифрование образов AES-XTS, то данные будут шифроваться дважды.
|
||||
|
||||
Для шифрования используется алгоритм AES-256-GCM и собственный упрощённый протокол согласования
|
||||
ключей, полностью аналогичный TLS 1.3 ECDHE.
|
||||
|
||||
### Сквозное шифрование данных образов (AES-XTS)
|
||||
## Сквозное шифрование данных образов (AES-XTS)
|
||||
|
||||
Клиент Vitastor поддерживает шифрование данных каждого образа своим ключом. В этом случае на OSD
|
||||
уходят уже зашифрованные данные и сами OSD не видят исходные данные клиента. При этом ключ можно
|
||||
менять при клонировании/создании снимков образов. Например, можно сделать базовый образ ВМ
|
||||
(условный Debian Linux) нешифрованным, но наследовать от него шифрованные образы клиентских ВМ.
|
||||
уходят уже зашифрованные данные и сами OSD не видят настоящее содержимое образов. Разные ключи
|
||||
в том числе могут иметь разные снимки или клоны одного и того же образа. Например, можно сделать
|
||||
базовый образ ВМ (условный Debian Linux) нешифрованным, но наследовать от него шифрованные образы
|
||||
клиентских ВМ.
|
||||
|
||||
Ключи шифрования образов могут храниться либо в etcd, либо во внешнем Vault. Во втором случае
|
||||
в etcd хранятся только ID ключей, а Vitastor вообще не имеет доступа к данным образов. Для
|
||||
использования Vault нужно создать образ с опцией `--enc_key vault:ID`, в конфигурации указать
|
||||
опции vault_url, и vault_ca, создать всем клиентам учётные записи в Vault и дать им доступ
|
||||
к требуемым секретам v1.
|
||||
использования Vault нужно создать образ с опцией `--enc_key vault:ID`, а в конфигурации указать
|
||||
опции:
|
||||
- vault_url
|
||||
- vault_ca
|
||||
- vault_client_cert
|
||||
- vault_client_key
|
||||
|
||||
Ещё раз повторимся, что если AES-XTS используется с полным шифрованием трафика (`proto_checksums=gcm`),
|
||||
то данные образов шифруются дважды - сначала AES-XTS, а потом AES-GCM. Можете использовать,
|
||||
только если вы совсем параноик :-).
|
||||
|
||||
### Аутентификация по сертификатам
|
||||
## Производительность шифрования
|
||||
|
||||
У вас может возникнуть вопрос - а как быстро всё это прекрасное шифрование работает?
|
||||
|
||||
Ответ - скорость сильно зависит от процессора. Складывается она из нескольких вещей:
|
||||
|
||||
-
|
||||
|
||||
TODO: vitastor-cli bench.
|
||||
|
||||
## Аутентификация по сертификатам
|
||||
|
||||
При включённом шифровании клиенты, OSD и мониторы Vitastor аутентифицируются по сертификатам
|
||||
как при соединениях с etcd (Antietcd), так и с OSD.
|
||||
@@ -152,24 +118,14 @@
|
||||
Также сертификат монитора может быть вообще не нужен, если Antietcd встраивается в сам монитор.
|
||||
В этом случае монитор и так имеет доступ ко всем данным etcd прямо в памяти.
|
||||
|
||||
### Пользователи и права доступа
|
||||
Каждый клиент должен иметь свой сертификат, подписанный общим корневым сертификатом
|
||||
для клиентов (`client_ca`). Common Name сертификата должно равняться имени пользователя.
|
||||
|
||||
При отключённом шифровании трафика Vitastor работает без разграничения прав доступа, то есть,
|
||||
любой клиент кластера имеет полный доступ как к слою управлению, так и к слою данных. Такой
|
||||
вариант подходит для выделенных доверенных сетей хранения.
|
||||
|
||||
При включённом шифровании трафика OSD (хотя бы заголовков) есть возможность задействовать
|
||||
права доступа, включив опцию `use_perms=true`. При включённой опции каждый пользователь может
|
||||
выполнять только те операции, которые ему разрешены, и даже OSD и мониторам также запрещены
|
||||
"лишние" операции.
|
||||
|
||||
Каждый пользователь (или администратор) должен иметь свой сертификат, подписанный общим
|
||||
корневым сертификатом для клиентов (`client_ca`), с Common Name, равным имени пользователя.
|
||||
Настройки привилегий же хранятся в etcd. Для OSD и мониторов учётные записи не нужны,
|
||||
они аутентифицируются по отдельным сертификатам.
|
||||
## Модель прав доступа
|
||||
|
||||
Привилегии пользователей хранятся в данных etcd в ключах `/vitastor/config/user/<имя>`.
|
||||
В этом ключе для каждого пользователя задаётся:
|
||||
|
||||
У пользователя есть 2 свойства:
|
||||
- Тип:
|
||||
- Клиент (`type=client` или не указано) - может читать и модифицировать только явным образом
|
||||
разрешённые образы.
|
||||
@@ -177,285 +133,67 @@
|
||||
кластер: смотреть общую статистику и состояние, создавать и удалять OSD и так далее.
|
||||
- Список имён групп, членом которых пользователь является.
|
||||
|
||||
У образов есть следующие свойства:
|
||||
У образов есть 3 свойства:
|
||||
- Владелец (owner) - имя пользователя, которому разрешено и читать, и менять образ
|
||||
- Группа владельцев (owner_group) - имя группы владельцев
|
||||
- Группа читателей (reader_group) - имя группы пользователей, которым разрешено читать образ
|
||||
- Группа читатетей (reader_group) - имя группы пользователей, которым разрешено читать образ
|
||||
|
||||
И также есть свойство у пула:
|
||||
У пулов есть 1 свойство:
|
||||
- Группа создателей (creator_group) - имя группы пользователей, которым разрешено создавать образы в пуле
|
||||
|
||||
Перечень разрешённых операций с данными образов на стороне OSD смотрите в разделе
|
||||
[Права доступа к данным OSD](#права-доступа-к-данным-osd).
|
||||
## Права доступа к данным etcd
|
||||
|
||||
### Привилегии etcd
|
||||
Привилегии реализуются через Antietcd во всех режимах работы. Если используется etcd, то
|
||||
Antietcd выступает в роли фильтрующего прокси, при этом он может быть встроен в монитор
|
||||
Vitastor или запущен отдельно. В этом случае etcd должен разрешать входящие подключения
|
||||
только от Antietcd, а все остальные компоненты должны соединяться с Antietcd.
|
||||
|
||||
Привилегии etcd реализуются через Antietcd во всех режимах работы.
|
||||
Если же используется Antietcd, то привилегии реализуются в нём самом.
|
||||
|
||||
Встроенные привилегии etcd не поддерживаются по причине их многочисленных неудобств:
|
||||
- Аутентификация по сертификатам вообще не работает в REST интерфейсе etcd,
|
||||
- Привилегии хранятся отдельно от k/v данных и не могут участвовать в транзакциях,
|
||||
- Менять привилегии может только администратор (root),
|
||||
- Нет поддержки фильтрации диапазонных ответов чтения по привилегиям.
|
||||
Если используется встроенный в монитор Antietcd, то привилегии включаются либо параметром
|
||||
`use_auth: true`, либо, если этот параметр не указан - включается автоматически, если задан
|
||||
любой из параметров `client_ca`, `osd_ca`, `mon_ca`. При этом монитор требует указания
|
||||
параметров `client_ca` и `osd_ca`, а если не используется режим проксирования в etcd -
|
||||
также `antietcd_server_ca`, чтобы Antietcd мог отличать кластерные соединения от клиентских.
|
||||
|
||||
Если используется etcd, то Antietcd выступает в роли фильтрующего прокси, при этом он
|
||||
может быть встроен в монитор Vitastor или запущен отдельно. В этом случае etcd должен
|
||||
разрешать входящие подключения только от Antietcd, а все остальные компоненты должны
|
||||
соединяться с Antietcd.
|
||||
Если используется отдельно стоящий Antietcd, привилегии нужно включать явным образом.
|
||||
|
||||
Если Antietcd запускается в составе монитора Vitastor, то достаточно включить опцию
|
||||
`use_perms=true` и задать нужные сертификаты. Если Antietcd запускается отдельно, то
|
||||
привилегии нужно включать отдельно опциями Antietcd. Подробнее о настройке смотрите
|
||||
раздел [Варианты настройки etcd/Antietcd](#варианты-настройки-etcdantietcd).
|
||||
Встроенные привилегии etcd не поддерживаются по причине их многочисленных недоработок:
|
||||
- Аутентификация по сертификатам не работает в REST интерфейсе etcd,
|
||||
- Привилегии хранятся отдельно от k/v и не могут участвовать в транзакциях,
|
||||
- Менять привилегии может только администратор (root)
|
||||
- Нет поддержки фильтрации ответов чтения по привилегиям.
|
||||
|
||||
Перечень разрешённых операций с данными etcd смотрите в разделе
|
||||
[Права доступа к данным etcd](#права-доступа-к-данным-etcd).
|
||||
Подробный список привилегий на ключи в etcd [смотрите ниже](#привилегии-etcd).
|
||||
|
||||
## Ручная настройка
|
||||
## Права доступа к данным OSD
|
||||
|
||||
### Настройка шифрования соединений OSD
|
||||
Регулируется опцией `use_auth`, либо, если она не указана, включается автоматически,
|
||||
если используется шифрование, то есть, если заданы опции `osd_ca` и `client_ca`.
|
||||
|
||||
Вам нужно 2 сертификата: один для OSD и один для подписи сертификатов всех клиентов.
|
||||
Для OSD можно использовать самоподписанный сертификат (osd_ca.crt) или отдельный сертификат (osd.crt),
|
||||
подписанный доверенным сертификатом osd_ca.crt. Для клиентов нужно использовать отдельные
|
||||
сертификаты, подписанные общим доверенным (client_ca.crt).
|
||||
OSD аутентифицирует клиентов по сертификатам и разрешает каждому клиенту только
|
||||
то, что ему разрешено согласно модели прав доступа.
|
||||
|
||||
В конфигурацию Vitastor на серверах OSD нужно добавить:
|
||||
- use_perms: true
|
||||
- osd_ca: osd_ca.crt
|
||||
- client_ca: client_ca.crt
|
||||
- osd_cert: osd_ca.crt
|
||||
- osd_pkey: osd_ca.key
|
||||
Подробный список разрешаемых OSD операций [смотрите ниже](#привилегии-osd).
|
||||
|
||||
На стороне клиентов:
|
||||
- use_perms: true
|
||||
- cert: client.crt
|
||||
- pkey: client.key
|
||||
## Права доступа к API
|
||||
|
||||
### Варианты настройки etcd/Antietcd
|
||||
[vitastor-cli serve](../usage/cli.ru.md#serve) также поддерживает клиентскую
|
||||
аутентификацию по сертификатам. Принимаются только сертификаты, подписанные
|
||||
`client_ca`. В качестве серверного сертификата используется отдельный сертификат
|
||||
`server_cert` с ключом `server_key`.
|
||||
|
||||
Доступны следующие варианты настройки:
|
||||
При этом для корректной работы `vitastor-cli serve` он сам должен использовать
|
||||
для доступа в Vitastor сертификат (`cert`+`pkey`) пользователя с правами
|
||||
администратора (`type=admin`).
|
||||
|
||||
#### Mon со встроенным Antietcd
|
||||
Обычным клиентам при доступе к API разрешаются только API-операции с образами,
|
||||
доступными им либо на чтение (для чтения), либо на запись (для модификации).
|
||||
Все остальные API-вызовы разрешаются только для администраторов.
|
||||
|
||||
Самый простой вариант. Вам нужен 1 сертификат для Antietcd (antietcd.crt), плюс
|
||||
корневые сертификаты для OSD и клиентов.
|
||||
Подробный список разрешаемых API операций [смотрите ниже](#привилегии-api).
|
||||
|
||||
Настройки Vitastor (`/etc/vitastor/vitastor.conf`):
|
||||
- etcd_address: [ "http://mon1:2379", ... ] (адреса ваших мониторов с портом 2379)
|
||||
- use_perms: true
|
||||
- use_antietcd: true
|
||||
- antietcd_cert: antietcd.crt
|
||||
- antietcd_key: antietcd.key
|
||||
- etcd_ca: antietcd.crt
|
||||
- osd_ca: osd_ca.crt
|
||||
- client_ca: client_ca.crt
|
||||
|
||||
#### Mon в роли Etcd proxy
|
||||
|
||||
Если вы хотите включить привилегии, но остаться на etcd, можно задействовать режим etcd proxy.
|
||||
|
||||
Вам понадобится 2 отдельных сертификата: один для etcd (etcd.crt) и один для antietcd (antietcd.crt).
|
||||
Клиентский порт etcd должен отличаться от стандартного 2379, например, можно выбрать 2381.
|
||||
Также нужны сертификаты OSD и клиентов.
|
||||
|
||||
Настройки Vitastor:
|
||||
- etcd_address: [ "http://mon1:2379", ... ] (адреса ваших мониторов с портом 2379)
|
||||
- use_perms: true
|
||||
- use_antietcd: true
|
||||
- etcd_proxy:
|
||||
```
|
||||
{
|
||||
"urls": [ "http://mon1:2381", ... ], // адреса ваших etcd с портом 2381
|
||||
"cert": "antietcd.crt",
|
||||
"key": "antietcd.key",
|
||||
"ca": "etcd.crt"
|
||||
}
|
||||
```
|
||||
- antietcd_cert: antietcd.crt
|
||||
- antietcd_key: antietcd.key
|
||||
- etcd_ca: antietcd.crt
|
||||
- osd_ca: osd_ca.crt
|
||||
- client_ca: client_ca.crt
|
||||
|
||||
Опции командной строки etcd:
|
||||
```
|
||||
--advertise-client-urls=https://<АДРЕС>:2381 --listen-client-urls=https://<АДРЕС>:2381 \
|
||||
--client-cert-auth --cert-file=etcd.crt --key-file=etcd.key --trusted-ca-file=antietcd.crt \
|
||||
--peer-client-cert-auth --peer-cert-file=etcd.crt --peer-key-file=etcd.key --peer-trusted-ca-file=etcd.crt
|
||||
```
|
||||
|
||||
#### Mon с отдельным Antietcd Proxy
|
||||
|
||||
Если в дополнение к предыдущему варианту вы хотите разгрузить Antietcd от задач монитора Vitastor,
|
||||
можно запустить его отдельно.
|
||||
|
||||
Аналогично предыдущему варианту нужно 2 сертификата: один для etcd и один для antietcd, плюс понадобятся
|
||||
отдельные сертификаты для клиентов, OSD и монитора.
|
||||
|
||||
Настройки Vitastor:
|
||||
- etcd_address: [ "http://mon1:2379", ... ] (адреса ваших мониторов с портом 2379)
|
||||
- use_perms: true
|
||||
- use_antietcd: false
|
||||
- etcd_ca: antietcd.crt
|
||||
- osd_ca: osd_ca.crt
|
||||
- client_ca: client_ca.crt
|
||||
- mon_etcd_client_cert: mon_ca.crt
|
||||
- mon_etcd_client_key: mon_ca.key
|
||||
|
||||
Опции командной строки Antietcd:
|
||||
```
|
||||
--port 2379 \
|
||||
--client_cert_auth 1 --auth_filter vitastor_auth_filter.js --etcd_proxy url1,url2,... \
|
||||
--cert antietcd.crt --key antietcd.key --ca client_ca.crt --osd_ca osd_ca.crt --mon_ca mon_ca.crt \
|
||||
--etcd_cert antietcd.crt --etcd_key antietcd.key --etcd_ca etcd.crt
|
||||
```
|
||||
|
||||
Опции командной строки etcd (не отличаются от предыдущего варианта):
|
||||
```
|
||||
--advertise-client-urls=https://<АДРЕС>:2381 --listen-client-urls=https://<АДРЕС>:2381 \
|
||||
--client-cert-auth --cert-file=etcd.crt --key-file=etcd.key --trusted-ca-file=antietcd.crt \
|
||||
--peer-client-cert-auth --peer-cert-file=etcd.crt --peer-key-file=etcd.key --peer-trusted-ca-file=etcd.crt
|
||||
```
|
||||
|
||||
#### Отдельный Antietcd без etcd
|
||||
|
||||
Аналогично предыдущему варианту, но etcd и его сертификат не нужны:
|
||||
|
||||
Настройки Vitastor (не отличаются от предыдущего варианта):
|
||||
- etcd_address: [ "http://mon1:2379", ... ] (адреса ваших мониторов с портом 2379)
|
||||
- use_perms: true
|
||||
- use_antietcd: false
|
||||
- etcd_ca: antietcd.crt
|
||||
- osd_ca: osd_ca.crt
|
||||
- client_ca: client_ca.crt
|
||||
- mon_etcd_client_cert: mon_ca.crt
|
||||
- mon_etcd_client_key: mon_ca.key
|
||||
|
||||
Опции командной строки Antietcd:
|
||||
```
|
||||
--port 2379 \
|
||||
--client_cert_auth 1 --auth_filter vitastor_auth_filter.js \
|
||||
--persist_filter vitastor_persist_filter.js \
|
||||
--cert antietcd.crt --key antietcd.key --ca client_ca.crt --osd_ca osd_ca.crt --mon_ca mon_ca.crt
|
||||
```
|
||||
|
||||
### Настройка Vault/OpenBao
|
||||
|
||||
Для использования Vault каждому клиенту, который будет получать из Vault ключи
|
||||
образов, нужна учётная запись в Vault. Vitastor поддерживает только аутентификацию
|
||||
по клиентским сертификатам, так что все сертификаты клиентов (`cert`+`pkey`) должны
|
||||
быть зарегистрированы в Vault и им должен быть дан доступ к соответствующим секретам
|
||||
(поддерживается API секретов v1).
|
||||
|
||||
Требуемый формат секрета Vault - одно поле `key` в формате шестнадцатеричной строки.
|
||||
Используется алгоритм AES-256-XTS, так что длина ключа - 64 байта, то есть строка
|
||||
должна состоять из 128 шестнадцатеричных цифр.
|
||||
|
||||
Для подключения Vault включите следующие настройки в Vitastor.conf:
|
||||
- `vault_url` - адрес Vault (например, `https://vault:8200`)
|
||||
- `vault_ca` - сертификат самого Vault
|
||||
|
||||
После этого, если создать образ (`vitastor-cli create`) с опцией `--enc_key vault:<ID>`,
|
||||
то для получения ключа клиенты Vitastor сначала обратятся к Vault для получения токена
|
||||
по адресу `/v1/auth/cert/login`, а потом запросят из Vault сам секрет по адресу `/v1/secret/<ID>`.
|
||||
|
||||
#### Пример настройки Vault
|
||||
|
||||
Пошаговая инструкция для настройки тестового Vault на примере OpenBao:
|
||||
|
||||
1. Если ещё не настроен TLS, генерируем самоподписанный TLS сертификат для Vault:
|
||||
```
|
||||
openssl req -days 3650 -x509 -addext basicConstraints=critical,CA:TRUE,pathlen:1 --addext subjectAltName=DNS:vault \
|
||||
-new -newkey rsa:4096 -nodes -keyout /etc/openbao/vault.key -out /etc/openbao/vault.crt
|
||||
```
|
||||
Настраиваем его в `/etc/openbao/openbao.hcl`:
|
||||
```
|
||||
listener "tcp" {
|
||||
address = "0.0.0.0:8200"
|
||||
tls_cert_file = "/etc/openbao/vault.crt"
|
||||
tls_key_file = "/etc/openbao/vault.key"
|
||||
}
|
||||
```
|
||||
И перезапускаем OpenBao (`systemctl restart openbao`).
|
||||
2. Копируем TLS сертификат Vault для Vitastor:
|
||||
```
|
||||
cp /etc/openbao/vault.crt /etc/vitastor/vault.crt
|
||||
```
|
||||
Переносим его на все клиентские ноды и прописываем в `/etc/vitastor/vitastor.conf`:
|
||||
```
|
||||
{
|
||||
...
|
||||
"vault_url": "http://vault:8200",
|
||||
"vault_ca": "/etc/vitastor/vault.crt"
|
||||
}
|
||||
```
|
||||
3. Проверяем статус Vault:
|
||||
```
|
||||
bao status -ca-cert /etc/openbao/vault.crt -address=https://vault:8200
|
||||
```
|
||||
4. Инициализируем Vault в тестовом режиме из 1 ноды (с 1 частью ключа):
|
||||
```
|
||||
bao operator init -n 1 -t 1 -ca-cert /etc/openbao/vault.crt -address=https://vault:8200
|
||||
```
|
||||
5. Разблокируем Vault:
|
||||
```
|
||||
bao operator unseal -ca-cert /etc/openbao/vault.crt -address=https://vault:8200
|
||||
```
|
||||
6. Включаем аутентификацию по сертификатам:
|
||||
```
|
||||
bao auth enable -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 cert
|
||||
```
|
||||
7. Включаем секреты v1:
|
||||
```
|
||||
bao secrets enable -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 -path=secret kv-v1
|
||||
```
|
||||
8. Создаём тестовый секрет:
|
||||
```
|
||||
bao kv put -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 secret/vitastor/testimg3 key=$(openssl rand -hex 64)
|
||||
```
|
||||
9. Генерируем подписанный сертификат для пользователя Vitastor (там, где у вас есть `client_ca.crt` и `client_ca.key`):
|
||||
```
|
||||
openssl req -subj '/CN=testimg3' -nodes -new -keyout testimg3.key -out testimg3.csr
|
||||
openssl x509 -req -days 3650 -CA client_ca.crt -CAkey client_ca.key -CAcreateserial -in testimg3.csr -out testimg3.crt
|
||||
rm testimg3.csr
|
||||
```
|
||||
10. Создаём пользователя в Vault и даём ему доступ к секрету:
|
||||
```
|
||||
cat >testimg3.policy <<EOF
|
||||
path "/secret/vitastor/testimg3" {
|
||||
capabilities = ["read"]
|
||||
}
|
||||
EOF
|
||||
|
||||
bao policy write -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 testimg3 testimg3.policy
|
||||
|
||||
bao write -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 auth/cert/certs/testimg3 \
|
||||
certificate=@testimg3.crt display_name=testimg3 token_ttl=24h token_policies=testimg3
|
||||
```
|
||||
11. Тестируем доступ к секрету:
|
||||
```
|
||||
curl --cacert /etc/vitastor/vault.crt --cert testimg3.crt --key testimg3.key \
|
||||
--json '{}' https://vault:8200/v1/auth/cert/login
|
||||
```
|
||||
Будет выведен токен, подставляем его в следующий запрос:
|
||||
```
|
||||
curl --cacert /etc/vitastor/vault.crt --cert testimg3.crt --key testimg3.key \
|
||||
-H 'X-Vault-Token: <ПОЛУЧЕННЫЙ ТОКЕН>' https://vault:8200/v1/secret/vitastor/testimg3
|
||||
```
|
||||
12. Создаём образ в Vitastor с заданным секретом (от имени администратора или того, кто имеет
|
||||
право создавать образы в вашем пуле):
|
||||
```
|
||||
vitastor-cli create -s 100G --enc_key vault:vitastor/testimg3 --owner testimg3 testimg3
|
||||
```
|
||||
13. Тестируем доступ к образу от имени пользователя testimg3:
|
||||
```
|
||||
vitastor-cli --cert testimg3.crt --pkey testimg3.key dd if=/dev/urandom oimg=testimg3 bs=1M count=100
|
||||
```
|
||||
|
||||
## Списки разрешённых операций
|
||||
|
||||
### Права доступа к данным etcd
|
||||
## Привилегии etcd
|
||||
|
||||
Ниже все названия ключей приведены без общего префикса `/vitastor`.
|
||||
|
||||
@@ -469,7 +207,7 @@
|
||||
- `/osd/state/*`
|
||||
- `/pg/state/*`
|
||||
- `/index/maxid/*`
|
||||
- Для образов, которые [может читать пользователь](#пользователи-и-права-доступа):
|
||||
- Для образов, которые [может читать пользователь](#модель-прав-доступа):
|
||||
- `/config/inode/*`
|
||||
- `/index/image/*`
|
||||
- `/inode/stats/*`
|
||||
@@ -518,10 +256,7 @@
|
||||
- `/inode/stats/*`
|
||||
- `/pool/stats/*`
|
||||
|
||||
### Права доступа к данным OSD
|
||||
|
||||
При включённой опции `use_perms` и шифровании OSD аутентифицирует клиентов по сертификатам
|
||||
и разрешает каждому клиенту только то, что ему разрешено согласно модели прав доступа.
|
||||
## Привилегии OSD
|
||||
|
||||
Клиентские операции:
|
||||
- READ - разрешено для образов, доступных пользователю на чтение.
|
||||
@@ -547,22 +282,7 @@
|
||||
- SEC_READ_BMP
|
||||
- SEC_LOCK
|
||||
|
||||
### Права доступа к API
|
||||
|
||||
[vitastor-cli serve](../usage/cli.ru.md#serve) также поддерживает клиентскую
|
||||
аутентификацию по сертификатам. Принимаются только сертификаты, подписанные
|
||||
`client_ca`. В качестве серверного сертификата используется отдельный сертификат
|
||||
`server_cert` с ключом `server_pkey`.
|
||||
|
||||
При этом для корректной работы `vitastor-cli serve` он сам должен использовать
|
||||
для доступа в Vitastor сертификат (`cert`+`pkey`) пользователя с правами
|
||||
администратора (`type=admin`).
|
||||
|
||||
Обычным клиентам при доступе к API разрешаются только API-операции с образами,
|
||||
доступными им либо на чтение (для чтения), либо на запись (для модификации).
|
||||
Все остальные API-вызовы разрешаются только для администраторов.
|
||||
|
||||
Список разрешённых операций API:
|
||||
## Привилегии API
|
||||
|
||||
Клиентам (пользователям с `type=client`) разрешаются операции:
|
||||
- image/list - для образов, которые пользователь может читать.
|
||||
@@ -572,91 +292,148 @@
|
||||
|
||||
Все остальные операции разрешаются только администраторам (`type=admin`).
|
||||
|
||||
## Производительность шифрования
|
||||
|
||||
У вас может возникнуть вопрос - а как быстро всё это прекрасное шифрование работает?
|
||||
|
||||
Ответ - сильно зависит от процессора. На современных процессорах (при наличии AVX512 с VAES)
|
||||
очень быстро - скорость шифрования AES может составлять 10-20 Гбайт/с и выше. В первую очередь
|
||||
подразумевается CPU клиентских машин, потому что сквозное шифрование выполняется целиком на
|
||||
клиенте, а транспортное хоть также и затрагивает OSD, но у клиента поток один, а OSD на стороне
|
||||
сервера много и добавить там ресурсов легче.
|
||||
|
||||
На более старых процессорах скорость заметно хуже, например, на Xeon E5 v4 она составляет
|
||||
буквально 3 Гбайт/с.
|
||||
|
||||
Вы можете оценить производительность своих процессоров с помощью команды `vitastor-cli cpubench`.
|
||||
|
||||
Пример вывода (💪 AMD EPYC 9575F):
|
||||
|
||||
```
|
||||
$ vitastor-cli cpubench
|
||||
Vitastor transport encryption benchmark (AES-256-GCM, AES-256-XTS and xxhash3)
|
||||
|
||||
Warmup...
|
||||
|
||||
No transport encryption, data checksums enabled, e2e unencrypted image
|
||||
xxhash3 1 M block... 209000 iterations in 2001 ms = 104447.78 MB/s
|
||||
xxhash3 4 K block... 37000000 iterations in 2022 ms = 71479.35 MB/s
|
||||
|
||||
Header encryption with payload checksums, e2e unencrypted image
|
||||
AES-256-GCM encrypt header + xxhash3 1 M block... 210000 iterations in 2015 ms = 104218.36 MB/s
|
||||
AES-256-GCM encrypt header + xxhash3 4 K block... 26000000 iterations in 2073 ms = 48993.01 MB/s
|
||||
|
||||
Full transport encryption, e2e unencrypted image
|
||||
AES-256-GCM encrypt header and 1 M block... 54000 iterations in 2000 ms = 27000.00 MB/s
|
||||
AES-256-GCM encrypt header and 4 K block... 11700000 iterations in 2014 ms = 22692.71 MB/s
|
||||
|
||||
No transport encryption, no checksums, e2e encrypted image
|
||||
AES-256-XTS encrypt 1 M block... 50000 iterations in 2039 ms = 24521.82 MB/s
|
||||
AES-256-XTS encrypt 4 K block... 12600000 iterations in 2009 ms = 24499.13 MB/s
|
||||
|
||||
No transport encryption, e2e encrypted image, data checksums enabled
|
||||
AES-256-XTS encrypt + xxhash3 1 M block... 40000 iterations in 2013 ms = 19870.84 MB/s
|
||||
AES-256-XTS encrypt + xxhash3 4 K block... 10200000 iterations in 2011 ms = 19812.90 MB/s
|
||||
|
||||
Header encryption with payload checksums, e2e encrypted image
|
||||
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 1 M block... 40000 iterations in 2014 ms = 19860.97 MB/s
|
||||
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 4 K block... 8700000 iterations in 2011 ms = 16899.24 MB/s
|
||||
|
||||
Full transport encryption, e2e encrypted image
|
||||
AES-256-XTS + AES-256-GCM encrypt 1 M block... 26000 iterations in 2062 ms = 12609.12 MB/s
|
||||
AES-256-XTS + AES-256-GCM encrypt 4 K block... 6300000 iterations in 2006 ms = 12267.88 MB/s
|
||||
```
|
||||
|
||||
А вот Xeon E5-2680v4:
|
||||
|
||||
```
|
||||
$ vitastor-cli cpubench
|
||||
Vitastor transport encryption benchmark (AES-256-GCM, AES-256-XTS and xxhash3)
|
||||
|
||||
Warmup...
|
||||
Таким образом, доступны следующие варианты настройки:
|
||||
|
||||
No transport encryption, data checksums enabled, e2e unencrypted image
|
||||
xxhash3 1 M block... 62000 iterations in 2021 ms = 30677.88 MB/s
|
||||
xxhash3 4 K block... 12400000 iterations in 2006 ms = 24146.31 MB/s
|
||||
### Mon в роли Etcd proxy
|
||||
|
||||
Header encryption with payload checksums, e2e unencrypted image
|
||||
AES-256-GCM encrypt header + xxhash3 1 M block... 62000 iterations in 2027 ms = 30587.07 MB/s
|
||||
AES-256-GCM encrypt header + xxhash3 4 K block... 6800000 iterations in 2011 ms = 13208.60 MB/s
|
||||
Mon
|
||||
- use_antietcd: true
|
||||
- etcd_proxy = {
|
||||
urls: [],
|
||||
cert = <antietcd.pem>,
|
||||
key,
|
||||
ca = <etcd.pem>,
|
||||
}
|
||||
- antietcd_cert = antietcd.pem
|
||||
- antietcd_key
|
||||
|
||||
Full transport encryption, e2e unencrypted image
|
||||
AES-256-GCM encrypt header and 1 M block... 7000 iterations in 2317 ms = 3021.15 MB/s
|
||||
AES-256-GCM encrypt header and 4 K block... 1500000 iterations in 2102 ms = 2787.52 MB/s
|
||||
etcd
|
||||
--client-cert-auth --cert-file=etcd.pem --key-file=etcd.key --trusted-ca-file=antietcd.pem \
|
||||
--peer-client-cert-auth --peer-cert-file=etcd.pem --peer-key-file=etcd.key --peer-trusted-ca-file=etcd.pem
|
||||
|
||||
No transport encryption, no checksums, e2e encrypted image
|
||||
AES-256-XTS encrypt 1 M block... 7000 iterations in 2317 ms = 3021.15 MB/s
|
||||
AES-256-XTS encrypt 4 K block... 1600000 iterations in 2088 ms = 2993.30 MB/s
|
||||
### Mon с отдельным Antietcd Proxy
|
||||
|
||||
No transport encryption, e2e encrypted image, data checksums enabled
|
||||
AES-256-XTS encrypt + xxhash3 1 M block... 6000 iterations in 2188 ms = 2742.23 MB/s
|
||||
AES-256-XTS encrypt + xxhash3 4 K block... 1400000 iterations in 2053 ms = 2663.78 MB/s
|
||||
Mon
|
||||
- use_antietcd: false
|
||||
- etcd_ca = antietcd.pem
|
||||
|
||||
Header encryption with payload checksums, e2e encrypted image
|
||||
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 1 M block... 6000 iterations in 2190 ms = 2739.73 MB/s
|
||||
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 4 K block... 1300000 iterations in 2101 ms = 2417.00 MB/s
|
||||
Antietcd
|
||||
--client_cert_auth 1 --auth_filter vitastor_auth_filter.js --etcd_proxy url1,url2,... \
|
||||
--cert antietcd.pem --key antietcd.key --ca client_ca.pem --osd_ca osd_ca.pem \
|
||||
--etcd_cert antietcd.pem --etcd_key antietcd.key --etcd_ca etcd.pem
|
||||
|
||||
Full transport encryption, e2e encrypted image
|
||||
AES-256-XTS + AES-256-GCM encrypt 1 M block... 4000 iterations in 2666 ms = 1500.38 MB/s
|
||||
AES-256-XTS + AES-256-GCM encrypt 4 K block... 800000 iterations in 2113 ms = 1478.94 MB/s
|
||||
```
|
||||
etcd
|
||||
--client-cert-auth --cert-file=etcd.pem --key-file=etcd.key --trusted-ca-file=antietcd.pem \
|
||||
--peer-client-cert-auth --peer-cert-file=etcd.pem --peer-key-file=etcd.key --peer-trusted-ca-file=etcd.pem
|
||||
|
||||
### Mon со встроенным Antietcd
|
||||
|
||||
Mon
|
||||
- use_antietcd: true
|
||||
- use_auth: true
|
||||
- antietcd_cert = antietcd.pem
|
||||
- antietcd_key
|
||||
|
||||
### Отдельный Antietcd
|
||||
|
||||
Mon
|
||||
- use_antietcd: false
|
||||
- etcd_ca = antietcd.pem
|
||||
|
||||
Antietcd
|
||||
--client_cert_auth 1 --auth_filter vitastor_auth_filter.js
|
||||
|
||||
## Варианты настройки
|
||||
|
||||
### Настройка по умолчанию
|
||||
|
||||
Используются только контрольные суммы данных на транспортном уровне. Соединения с etcd не шифруются.
|
||||
Аутентификация и авторизация не используется, любой клиент имеет доступ ко всем данным кластера.
|
||||
|
||||
Аналог настройки:
|
||||
- proto_checksums: payload
|
||||
|
||||
### Полная защита
|
||||
|
||||
Везде
|
||||
- osd_ca
|
||||
- client_ca
|
||||
- etcd_ca = antietcd.pem
|
||||
|
||||
OSD
|
||||
- osd_cert
|
||||
- osd_pkey
|
||||
|
||||
Клиент
|
||||
- cert
|
||||
- pkey
|
||||
|
||||
### Только защита etcd
|
||||
|
||||
- etcd_ca
|
||||
- etcd_cert
|
||||
- etcd_key
|
||||
|
||||
### antietcd и только защита antietcd
|
||||
|
||||
- etcd_ca
|
||||
- etcd_cert
|
||||
- etcd_key
|
||||
- use_antietcd: true
|
||||
- antietcd_cert = etcd_ca
|
||||
- antietcd_key
|
||||
- antietcd_ca = etcd_cert
|
||||
|
||||
### Полное шифрование протокола, включая данные
|
||||
|
||||
Внимание: если включить этот вариант защиты и при этом
|
||||
|
||||
### Только контрольные суммы на транспортном уровне, без шифрования
|
||||
|
||||
## Настройка Vault/OpenBao
|
||||
|
||||
openssl req -days 3650 -x509 -addext basicConstraints=critical,CA:TRUE,pathlen:1 --addext subjectAltName=DNS:vault \
|
||||
-new -newkey rsa:4096 -nodes -keyout vault.key -out vault.crt
|
||||
|
||||
bao status -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200
|
||||
|
||||
bao operator init -n 1 -t 1 -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200
|
||||
|
||||
bao operator unseal -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200
|
||||
|
||||
bao auth enable -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200 cert
|
||||
|
||||
bao secrets enable -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200 -path=secret kv-v1
|
||||
|
||||
bao kv put -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200 secret/vitastor/testimg3 key=$(openssl rand -hex 64)
|
||||
|
||||
cat >testimg3.policy <<EOF
|
||||
path "/secret/vitastor/testimg3" {
|
||||
capabilities = ["read"]
|
||||
}
|
||||
EOF
|
||||
|
||||
bao policy write -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200 testimg3 testimg3.policy
|
||||
|
||||
bao write -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200 auth/cert/certs/testimg3 certificate=@testimg3.crt display_name=testimg3 token_ttl=24h token_policies=testimg3
|
||||
|
||||
curl --cacert /etc/vitastor/vault.crt --cert testimg3.crt --key testimg3.key --json '{}' https://vault:8200/v1/auth/cert/login
|
||||
|
||||
curl --cacert /etc/vitastor/vault.crt --cert testimg3.crt --key testimg3.key -H 'X-Vault-Token: s.Qkrm78BeK7Rqdz5MA3eJZNbu' https://vault:8200/v1/secret/vitastor/testimg3
|
||||
|
||||
+22
-18
@@ -28,38 +28,32 @@ class AntiEtcdAdapter
|
||||
is_local['::'] = true;
|
||||
is_local[''] = true;
|
||||
// split :, 3 -> <schema>:<//ip>:<port>
|
||||
const selected = [];
|
||||
for (let i = 0; i < cluster.length; i++)
|
||||
const cluster_local = cluster.map(s =>
|
||||
{
|
||||
const m = /^(https?:\/\/)?(?:\[(.*)\]|([^\[\:]+))(?::(\d+))?$/.exec(cluster[i]);
|
||||
if (!m)
|
||||
continue;
|
||||
const ip = m[3] || m[2];
|
||||
const port = m[4] || 2379;
|
||||
if (is_local[ip] && (!cfg_port || port == cfg_port))
|
||||
selected.push({ idx: i, ip, port });
|
||||
}
|
||||
const m = /^https?:\/\/(?:\[(.*)\]|([^\[\:]+))(?::(\d+))?$/.exec(s);
|
||||
return [ m[2] || m[1], m[3] || 2379 ];
|
||||
});
|
||||
const selected = cluster_local.filter(ip => is_local[ip[0]] && (!cfg_port || ip[1] == cfg_port));
|
||||
if (selected.length > 1)
|
||||
{
|
||||
console.error('More than 1 etcd_address matches local IPs, please specify port');
|
||||
console.error('More than 1 etcd_address matches local IPs: '+(selected.join(', '))+', please specify port');
|
||||
process.exit(1);
|
||||
}
|
||||
else if (selected.length == 1)
|
||||
{
|
||||
const antietcd_config = {
|
||||
ip: selected[0].ip,
|
||||
port: selected[0].port,
|
||||
ip: selected[0][0],
|
||||
port: selected[0][1],
|
||||
cert: config.antietcd_cert,
|
||||
key: config.antietcd_key,
|
||||
ca: config.client_ca,
|
||||
data: config.antietcd_data_file || ((config.antietcd_data_dir || '/var/lib/vitastor') + '/mon_'+selected[0].port+'.json.gz'),
|
||||
ca: config.antietcd_ca,
|
||||
data: config.antietcd_data_file || ((config.antietcd_data_dir || '/var/lib/vitastor') + '/mon_'+selected[0][1]+'.json.gz'),
|
||||
persist_filter: vitastor_persist_filter({ vitastor_prefix: config.etcd_prefix || '/vitastor' }),
|
||||
node_id: cluster[selected[0].idx].replace(/^(https?:\/\/)/, ''), // same as in <cluster> below
|
||||
node_id: selected[0][0]+':'+selected[0][1], // node_id = ip:port
|
||||
cluster: (cluster.length == 1 ? null : cluster.reduce((a, c) => { a[c.replace(/^(https?:\/\/)/, '')] = c; return a; }, {})),
|
||||
cluster_key: (config.etcd_prefix || '/vitastor'),
|
||||
stale_read: 1,
|
||||
log_level: 1,
|
||||
logs: { cluster: true },
|
||||
};
|
||||
if (config.etcd_proxy)
|
||||
{
|
||||
@@ -78,13 +72,23 @@ class AntiEtcdAdapter
|
||||
delete antietcd_config.cluster;
|
||||
delete antietcd_config.cluster_key;
|
||||
}
|
||||
if (config.use_perms)
|
||||
const use_auth = config.use_auth || config.use_auth == null && config.client_ca;
|
||||
if (use_auth)
|
||||
{
|
||||
antietcd_config.client_cert_auth = true;
|
||||
antietcd_config.auth_filter = vitastor_auth_filter;
|
||||
antietcd_config.ca = config.client_ca;
|
||||
antietcd_config.osd_ca = config.osd_ca;
|
||||
antietcd_config.mon_ca = config.mon_ca;
|
||||
if (!config.etcd_proxy)
|
||||
{
|
||||
antietcd_config.peer_ca = config.antietcd_server_ca;
|
||||
if (!config.antietcd_server_ca || config.antietcd_server_ca == config.client_ca)
|
||||
{
|
||||
console.error('Secure setup requires separate antietcd_server_ca (for signing antietcd server certificates) and client_ca (for signing client certificates)');
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const key in config)
|
||||
{
|
||||
|
||||
@@ -112,10 +112,9 @@ function make_cyclic(pgs, parity_space)
|
||||
{
|
||||
if (parity_space > 1)
|
||||
{
|
||||
for (const id in pgs)
|
||||
for (const pg in pgs)
|
||||
{
|
||||
const pg = pgs[id];
|
||||
for (let i = 1; i < pg.length; i++)
|
||||
for (let i = 1; i < pg.size; i++)
|
||||
{
|
||||
const cyclic = [ ...pg.slice(i), ...pg.slice(0, i) ];
|
||||
pgs['pg_'+cyclic.join('_')] = cyclic;
|
||||
|
||||
+2
-2
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "vitastor-mon",
|
||||
"version": "3.0.15",
|
||||
"version": "3.0.12",
|
||||
"description": "Vitastor SDS monitor service",
|
||||
"main": "mon-main.js",
|
||||
"scripts": {
|
||||
@@ -9,7 +9,7 @@
|
||||
"author": "Vitaliy Filippov",
|
||||
"license": "UNLICENSED",
|
||||
"dependencies": {
|
||||
"antietcd": "^1.3.1",
|
||||
"antietcd": "^1.2.4",
|
||||
"sprintf-js": "^1.1.2",
|
||||
"ws": "^7.2.5"
|
||||
},
|
||||
|
||||
+76
-126
@@ -6,39 +6,38 @@
|
||||
const child_process = require('child_process');
|
||||
const fs = require('fs');
|
||||
const os = require('os');
|
||||
const path = require('path');
|
||||
const readline = require('readline');
|
||||
|
||||
run().catch(e => { console.error(e); process.exit(1); });
|
||||
|
||||
const help_text = `Initialize a Vitastor cluster (etcd, vitastor.conf and TLS certificates)
|
||||
(c) Vitaliy Filippov, 2026+ (MIT)
|
||||
(c) Vitaliy Filippov, 2019+ (MIT)
|
||||
|
||||
USAGE:
|
||||
1) Create a minimal vitastor.conf with etcd_address, osd_network and (optionally) use_perms.
|
||||
Non-encrypted: {"etcd_address":["http://10.0.0.10:2379","http://10.0.0.11:2379","http://10.0.0.12:2379"],"osd_network":"10.0.0.0/24"}
|
||||
Encrypted: {"etcd_address":["https://10.0.0.10:2379","https://10.0.0.11:2379","https://10.0.0.12:2379"],"use_perms":true,"osd_network":"10.0.0.0/24"}
|
||||
(Note https:// etcd URLs!)
|
||||
2) Run: ${process.argv[1]} [./vitastor.conf] [--antietcd-only]
|
||||
1) Create a minimal vitastor.conf with etcd_address, osd_network and (optionally) use_auth.
|
||||
Example: {"etcd_address":["http://10.0.0.10:2379","http://10.0.0.11:2379","http://10.0.0.12:2379"],"use_auth":false,"osd_network":"10.0.0.0/24"}
|
||||
Or: {"etcd_address":["https://10.0.0.10:2379","https://10.0.0.11:2379","https://10.0.0.12:2379"],"use_auth":true,"osd_network":"10.0.0.0/24"}
|
||||
2) Run: ${process.argv[1]} [./vitastor.conf]
|
||||
You can run it on etcd/monitor nodes or on an external node.
|
||||
It configures etcd, generates TLS certificates (on the first or external node), copies
|
||||
them to other etcd/monitor nodes, and updates vitastor.conf with TLS options.
|
||||
It configures etcd, generates TLS certificates (on the first or external node), copies them
|
||||
to other etcd/monitor nodes, and updates vitastor.conf with TLS options.
|
||||
3) If you have OSD-only nodes, run:
|
||||
${process.argv[1]} --copy-to-osd-node NODE_NAME ./vitastor.conf
|
||||
It copies vitastor.conf and required TLS certificates to that node.
|
||||
|
||||
OPTIONS:
|
||||
--antietcd-only
|
||||
disable etcd (proxy or direct mode), use only antietcd
|
||||
--gen-certs
|
||||
force certificate generation even if it's not the first node
|
||||
--no-certs
|
||||
disable certificate generation
|
||||
--copy yes|no|ask
|
||||
copy vitastor.conf and TLS certificates for monitor&etcd to monitor nodes using scp
|
||||
(default is ask)
|
||||
--no-copy
|
||||
do not copy initial certificates to other nodes
|
||||
--copy-to-osd-node NODE[,NODE2,...]
|
||||
copy vitastor.conf and TLS certificates for OSDs to NODES using scp
|
||||
copy vitastor.conf and TLS certificates required for OSDs to NODES using scp
|
||||
--copy-to-mon-node NODE[,NODE2,...]
|
||||
copy vitastor.conf and TLS certificates required for monitor and etcd to NODES using scp
|
||||
--copy-to-client-node NODE[,NODE2,...]
|
||||
copy vitastor.conf and TLS certificates required for clients to NODES using scp
|
||||
`;
|
||||
|
||||
async function run()
|
||||
@@ -46,35 +45,21 @@ async function run()
|
||||
let config_path = '/etc/vitastor/vitastor.conf';
|
||||
let config_dir = '/etc/vitastor/';
|
||||
let gen_certs = 'auto';
|
||||
let antietcd_only = false;
|
||||
let copy = 'ask';
|
||||
let copy_to_osd = null;
|
||||
let copy_initial = true;
|
||||
let copy_to_osd =
|
||||
for (let i = 2; i < process.argv.length; i++)
|
||||
{
|
||||
const arg = process.argv[i];
|
||||
if (arg == '-h' || arg == '--help')
|
||||
{
|
||||
console.log(help_text);
|
||||
process.exit(0);
|
||||
}
|
||||
else if (arg == '--gen-certs')
|
||||
{
|
||||
gen_certs = true;
|
||||
}
|
||||
else if (arg == '--no-certs')
|
||||
{
|
||||
gen_certs = false;
|
||||
}
|
||||
else if (arg == '--antietcd-only')
|
||||
{
|
||||
antietcd_only = true;
|
||||
}
|
||||
else if (arg == '--copy-to-osd-node' && i < process.argv.length-1)
|
||||
else if (arg == '--only-certs')
|
||||
{
|
||||
i++;
|
||||
copy_to_osd = process.argv[i].split(/,/);
|
||||
gen_certs =
|
||||
}
|
||||
else if (arg == '--copy' && i < process.argv.length-1)
|
||||
else if (arg == '--copy')
|
||||
{
|
||||
i++;
|
||||
copy = process.argv[i];
|
||||
@@ -92,7 +77,6 @@ async function run()
|
||||
else
|
||||
{
|
||||
config_path = arg;
|
||||
config_dir = path.dirname(arg);
|
||||
}
|
||||
}
|
||||
if (!fs.existsSync(config_path))
|
||||
@@ -116,22 +100,18 @@ async function run()
|
||||
port: s[3],
|
||||
}));
|
||||
const tls = etcds.filter(e => e.scheme === 'https').length > 0;
|
||||
const use_perms = tls && config.use_perms;
|
||||
const use_auth = tls && config.use_auth;
|
||||
const num = select_local_etcd(etcds);
|
||||
if (copy_to_osd)
|
||||
{
|
||||
copy_to_osd_nodes(copy_to_osd, config_dir, use_perms, antietcd_only);
|
||||
process.exit(0);
|
||||
}
|
||||
if (tls)
|
||||
{
|
||||
const etcd_ca = config_dir+'/'+path.basename(config.etcd_ca);
|
||||
if (gen_certs === true)
|
||||
if (gen_certs === 'yes')
|
||||
{
|
||||
gen_certs = true;
|
||||
console.log('Certificate generation is requested explicitly, generating');
|
||||
}
|
||||
else if (gen_certs === false)
|
||||
else if (gen_certs === 'no')
|
||||
{
|
||||
gen_certs = false;
|
||||
console.log('Certificate generation is disabled explicitly, skipping');
|
||||
}
|
||||
else if (num < 0)
|
||||
@@ -139,10 +119,10 @@ async function run()
|
||||
gen_certs = true;
|
||||
console.log('No matching IPs in etcd_address from '+config_path+', only generating certificates');
|
||||
}
|
||||
else if (config.etcd_ca && fs.existsSync(etcd_ca))
|
||||
else if (fs.existsSync("/etc/vitastor/etcd.crt"))
|
||||
{
|
||||
gen_certs = false;
|
||||
console.log(etcd_ca+' already exists, assuming certificates are already generated');
|
||||
console.log('/etc/vitastor/etcd.crt already exists, assuming certificates are already generated');
|
||||
}
|
||||
else if (num === 0)
|
||||
{
|
||||
@@ -151,25 +131,24 @@ async function run()
|
||||
}
|
||||
else
|
||||
{
|
||||
console.log('This is monitor node '+(num+1)+', '+etcd_ca+' does not exist, please copy certificates to this node');
|
||||
console.log('This is monitor node '+(num+1)+', /etc/vitastor/etcd.crt does not exist, please copy certificates to this node');
|
||||
process.exit(1);
|
||||
}
|
||||
await write_auth_config(config, config_path, etcds, use_perms, antietcd_only);
|
||||
if (gen_certs)
|
||||
{
|
||||
if (copy === 'ask')
|
||||
copy = await ask_copy('Copy certificates and vitastor.conf to other nodes after generation?');
|
||||
copy = (copy === 'y' || copy === 'yes');
|
||||
await make_certs(config_dir, copy, etcds, use_perms, antietcd_only);
|
||||
await make_certs(dir, copy);
|
||||
}
|
||||
await write_auth_config(config, config_path);
|
||||
}
|
||||
if (num < 0)
|
||||
{
|
||||
console.log('No matching IPs in etcd_address from '+config_path);
|
||||
process.exit(tls && gen_certs ? 0 : 1);
|
||||
}
|
||||
await configure_etcd(etcds, num, tls, use_perms);
|
||||
await enable_mon();
|
||||
await configure_etcd();
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
@@ -181,98 +160,83 @@ async function ask_copy(question)
|
||||
prompt: '> ',
|
||||
});
|
||||
let copy;
|
||||
while (copy != 'y' && copy != 'n' && copy != 'yes' && copy != 'no')
|
||||
while (true)
|
||||
{
|
||||
if (copy)
|
||||
console.log('Please type "yes" or "no"');
|
||||
copy = await new Promise(ok => rl.question(question, ok));
|
||||
if (copy != 'y' && copy != 'n' && copy != 'yes' && copy != 'no')
|
||||
console.log('Please type "yes" or "no"');
|
||||
else
|
||||
break;
|
||||
}
|
||||
return copy;
|
||||
}
|
||||
|
||||
async function copy_to_osd_nodes(to, dir, use_perms, antietcd_only)
|
||||
{
|
||||
const osd_to_copy = [ 'vitastor.conf' ];
|
||||
if (!antietcd_only && !use_perms)
|
||||
osd_to_copy.push('etcd_ca.crt');
|
||||
else
|
||||
osd_to_copy.push('antietcd_ca.crt');
|
||||
if (use_perms)
|
||||
osd_to_copy.push('osd.crt', 'osd.key', 'client_ca.crt');
|
||||
console.warn('Copying configuration to OSD nodes '+to.join(', '));
|
||||
for (const node of to)
|
||||
await system("scp "+dir+osd_to_copy.join(" "+dir)+" root@"+node+":/etc/vitastor/");
|
||||
}
|
||||
|
||||
async function make_certs(dir, copy, etcds, use_perms, antietcd_only)
|
||||
async function make_certs(dir, copy)
|
||||
{
|
||||
console.log(`-----
|
||||
Generating certificates in ${dir}
|
||||
-----
|
||||
`);
|
||||
const to_copy = [ 'vitastor.conf' ];
|
||||
const osd_to_copy = [ 'vitastor.conf' ];
|
||||
if (!antietcd_only)
|
||||
{
|
||||
await make_ca("/O=Vitastor etcd CA", dir+"etcd_ca");
|
||||
await make_signed("/CN=Vitastor etcd", dir+"etcd", dir+"etcd_ca", etcds.map(e => "IP:"+e.ip).join(','));
|
||||
to_copy.push('etcd_ca.crt', 'etcd.crt', 'etcd.key');
|
||||
if (!use_perms)
|
||||
osd_to_copy.push('etcd_ca.crt');
|
||||
}
|
||||
if (use_perms || antietcd_only)
|
||||
await make_ca("/O=Vitastor etcd CA", dir+"etcd_ca");
|
||||
await make_signed("/CN=Vitastor etcd", dir+"etcd", dir+"etcd_ca", etcds.map(e => "IP:"+e.ip).join(','));
|
||||
if (use_auth)
|
||||
{
|
||||
await make_ca("/O=Vitastor Antietcd CA", dir+"antietcd_ca");
|
||||
await make_signed("/CN=Vitastor Antietcd", dir+"antietcd", dir+"antietcd_ca", etcds.map(e => "IP:"+e.ip).join(','));
|
||||
to_copy.push('antietcd_ca.crt', 'antietcd.crt', 'antietcd.key');
|
||||
osd_to_copy.push('antietcd_ca.crt');
|
||||
}
|
||||
if (use_perms)
|
||||
{
|
||||
await make_ca("/CN=Vitastor OSD", dir+"osd");
|
||||
await make_ca("/O=Vitastor Client CA", dir+"client_ca");
|
||||
await make_signed("/CN=admin", dir+"admin", dir+"client_ca");
|
||||
to_copy.push('osd.crt', 'osd.key', 'client_ca.crt');
|
||||
osd_to_copy.push('osd.crt', 'osd.key', 'client_ca.crt');
|
||||
}
|
||||
console.log(`-----
|
||||
if (use_auth)
|
||||
{
|
||||
console.log(`-----
|
||||
Certificates generated, commands to copy them:
|
||||
- Monitor+OSD node:
|
||||
cd ${dir} && scp ${to_copy.join(' ')} root@NODE:/etc/vitastor/
|
||||
cd ${dir} && scp antietcd_ca.crt antietcd.crt antietcd.key osd.crt osd.key client_ca.crt etcd_ca.crt etcd.crt etcd.key root@NODE:/etc/vitastor/
|
||||
- Monitor node:
|
||||
cd ${dir} && scp ${to_copy.filter(f => f != 'osd.key').join(' ')} root@NODE:/etc/vitastor/
|
||||
cd ${dir} && scp antietcd_ca.crt antietcd.crt antietcd.key osd.crt client_ca.crt etcd_ca.crt etcd.crt etcd.key root@NODE:/etc/vitastor/
|
||||
- OSD node:
|
||||
cd ${dir} && scp ${osd_to_copy.join(' ')} root@NODE:/etc/vitastor/
|
||||
cd ${dir} && scp antietcd_ca.crt osd.crt osd.key client_ca.crt root@NODE:/etc/vitastor/
|
||||
-----
|
||||
`);
|
||||
}
|
||||
else
|
||||
{
|
||||
console.log(`-----
|
||||
Certificates generated, commands to copy them:
|
||||
- Monitor node:
|
||||
cd ${dir} && scp etcd_ca.crt etcd.crt etcd.key root@NODE:/etc/vitastor/
|
||||
-----
|
||||
`);
|
||||
}
|
||||
if (copy)
|
||||
{
|
||||
const to_copy = use_auth
|
||||
? [ "antietcd_ca.crt", "antietcd.crt", "antietcd.key", "osd.crt", "osd.key", "client_ca.crt", "etcd_ca.crt", "etcd.crt", "etcd.key" ]
|
||||
: [ "etcd_ca.crt", "etcd.crt", "etcd.key" ];
|
||||
for (const node of etcds)
|
||||
{
|
||||
await system("scp "+dir+to_copy.join(" "+dir)+" root@"+node.ip+":/etc/vitastor/");
|
||||
await system("scp "+dir+to_copy.join(" "+dir)+" root@"+node.ip+"/etc/vitastor/");
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
console.warn('Certificates generated in '+dir+', please copy them to other nodes');
|
||||
console.warn('Certificates generated in /etc/vitastor, please copy them to other nodes');
|
||||
}
|
||||
}
|
||||
|
||||
async function write_auth_config(config, config_path, etcds, use_perms, antietcd_only)
|
||||
async function write_auth_config(config, config_path)
|
||||
{
|
||||
const auth = {};
|
||||
if (use_perms)
|
||||
if (use_auth)
|
||||
{
|
||||
auth["use_antietcd"] = true;
|
||||
if (!antietcd_only)
|
||||
{
|
||||
auth["etcd_proxy"] = {
|
||||
urls: etcds.map(e => e.ip+':2381'),
|
||||
cert: "/etc/vitastor/antietcd.crt",
|
||||
key: "/etc/vitastor/antietcd.key",
|
||||
ca: "/etc/vitastor/etcd_ca.crt",
|
||||
};
|
||||
}
|
||||
auth["etcd_proxy"] = {
|
||||
urls: etcds.map(e => e.ip+':2381'),
|
||||
cert: "/etc/vitastor/antietcd.crt",
|
||||
key: "/etc/vitastor/antietcd.key",
|
||||
ca: "/etc/vitastor/etcd_ca.crt",
|
||||
};
|
||||
auth["antietcd_cert"] = "/etc/vitastor/antietcd.crt";
|
||||
auth["antietcd_key"] = "/etc/vitastor/antietcd.key";
|
||||
auth["etcd_ca"] = "/etc/vitastor/antietcd_ca.crt";
|
||||
@@ -285,17 +249,7 @@ async function write_auth_config(config, config_path, etcds, use_perms, antietcd
|
||||
}
|
||||
else
|
||||
{
|
||||
if (antietcd_only)
|
||||
{
|
||||
auth["use_antietcd"] = true;
|
||||
auth["antietcd_cert"] = "/etc/vitastor/antietcd.crt";
|
||||
auth["antietcd_key"] = "/etc/vitastor/antietcd.key";
|
||||
auth["etcd_ca"] = "/etc/vitastor/antietcd_ca.crt";
|
||||
}
|
||||
else
|
||||
{
|
||||
auth["etcd_ca"] = "/etc/vitastor/etcd.crt";
|
||||
}
|
||||
auth["etcd_ca"] = "/etc/vitastor/etcd.crt";
|
||||
}
|
||||
for (const k in auth)
|
||||
{
|
||||
@@ -317,7 +271,7 @@ Updating ${config_path}
|
||||
fs.writeFileSync(config_path, JSON.stringify(config, 0, 4));
|
||||
}
|
||||
|
||||
async function configure_etcd(etcds, num, tls, use_perms)
|
||||
async configure_etcd()
|
||||
{
|
||||
const in_docker = fs.existsSync("/etc/vitastor/etcd.conf") &&
|
||||
fs.existsSync("/etc/vitastor/docker.conf");
|
||||
@@ -334,8 +288,8 @@ async function configure_etcd(etcds, num, tls, use_perms)
|
||||
const etcd_url = etcds[num].scheme + '://' + etcds[num].addr;
|
||||
const options = {
|
||||
name: 'etcd'+etcds[num].ip.replace(/[^0-9a-z_]/ig, '_'),
|
||||
advertise_client_urls: etcd_url+':'+(use_perms ? 2381 : 2379),
|
||||
listen_client_urls: etcd_url+':'+(use_perms ? 2381 : 2379),
|
||||
advertise_client_urls: etcd_url+':'+(use_auth ? 2381 : 2379),
|
||||
listen_client_urls: etcd_url+':'+(use_auth ? 2381 : 2379),
|
||||
initial_advertise_peer_urls: etcd_url+':2380',
|
||||
listen_peer_urls: etcd_url+':2380',
|
||||
initial_cluster_token: 'vitastor-etcd-1',
|
||||
@@ -351,14 +305,14 @@ async function configure_etcd(etcds, num, tls, use_perms)
|
||||
{
|
||||
options['cert_file'] = '/etc/vitastor/etcd.crt';
|
||||
options['key_file'] = '/etc/vitastor/etcd.key';
|
||||
if (use_perms)
|
||||
if (use_auth)
|
||||
{
|
||||
options['client_cert_auth'] = '1';
|
||||
options['trusted_ca_file'] = '/etc/vitastor/antietcd.crt';
|
||||
}
|
||||
options['peer_cert_file'] = '/etc/vitastor/etcd.crt';
|
||||
options['peer_key_file'] = '/etc/vitastor/etcd.key';
|
||||
if (use_perms)
|
||||
if (use_auth)
|
||||
{
|
||||
options['peer_client_cert_auth'] = '1';
|
||||
options['peer_trusted_ca_file'] = '/etc/vitastor/etcd.crt';
|
||||
@@ -379,7 +333,8 @@ async function configure_etcd(etcds, num, tls, use_perms)
|
||||
}
|
||||
await system(`mkdir -p /var/lib/etcd/vitastor`);
|
||||
fs.writeFileSync(
|
||||
"/etc/systemd/system/vitastor-etcd.service", `[Unit]
|
||||
"/etc/systemd/system/vitastor-etcd.service",
|
||||
`[Unit]
|
||||
Description=etcd for vitastor
|
||||
After=network-online.target local-fs.target time-sync.target
|
||||
Wants=network-online.target local-fs.target time-sync.target
|
||||
@@ -408,11 +363,6 @@ WantedBy=multi-user.target
|
||||
await system(`systemctl enable --now vitastor-etcd`);
|
||||
}
|
||||
|
||||
async function enable_mon()
|
||||
{
|
||||
await system(`systemctl enable --now vitastor-mon`);
|
||||
}
|
||||
|
||||
function replace_env(text, key, value)
|
||||
{
|
||||
let found = false;
|
||||
|
||||
@@ -68,10 +68,10 @@ class VitastorAuthFilter
|
||||
|
||||
async init()
|
||||
{
|
||||
if (!this.cfg.cert || !this.cfg.key || !this.cfg.ca || !this.cfg.osd_ca || !this.cfg.client_cert_auth)
|
||||
if (!this.cfg.cert || !this.cfg.key || !this.cfg.osd_ca || !this.cfg.etcd_proxy && !this.cfg.peer_ca || !this.cfg.client_cert_auth)
|
||||
{
|
||||
throw new Error('Authenticated Vitastor setups require enabled client_cert_auth, cert, key'+
|
||||
' and separate ca (client CA), osd_ca and optionally mon_ca');
|
||||
' and separate ca (client CA), osd_ca'+(this.cfg.etcd_proxy ? '' : ', peer_ca')+' and optionally mon_ca');
|
||||
}
|
||||
this.osd_ca = await this.antietcd.readPEM(this.cfg.osd_ca);
|
||||
this.osd_ca_obj = new X509Certificate(this.osd_ca);
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "vitastor",
|
||||
"version": "3.0.15",
|
||||
"version": "3.0.12",
|
||||
"description": "Low-level native bindings to Vitastor client library",
|
||||
"main": "index.js",
|
||||
"keywords": [
|
||||
|
||||
+10
-45
@@ -366,38 +366,15 @@ sub map_volume
|
||||
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||
|
||||
my ($vtype, $img_name, $vmid) = $class->parse_volname($volname);
|
||||
my $name = $prefix.$img_name;
|
||||
my $name = $img_name;
|
||||
$name .= '@'.$snapname if $snapname;
|
||||
|
||||
my $mapped = run_cli($scfg, [ 'ls' ], binary => '/usr/bin/vitastor-nbd');
|
||||
my ($kerneldev) = grep {
|
||||
$mapped->{$_} && $mapped->{$_}->{image} && $mapped->{$_}->{image} eq $name
|
||||
} keys %$mapped;
|
||||
my ($kerneldev) = grep { $mapped->{$_}->{image} eq $prefix.$name } keys %$mapped;
|
||||
return $kerneldev if $kerneldev && -b $kerneldev; # already mapped
|
||||
|
||||
if ($kerneldev && -b $kerneldev)
|
||||
{
|
||||
my $size = `/usr/sbin/blockdev --getsize64 $kerneldev`;
|
||||
return $kerneldev if $size && $size > 0;
|
||||
}
|
||||
|
||||
my $map_out = run_cli($scfg, [ 'map', '--image', $name ], binary => '/usr/bin/vitastor-nbd', json => 0);
|
||||
$map_out =~ s/^\s+|\s+$//gso;
|
||||
|
||||
# Wait until the device is started
|
||||
for (my $i = 0; $i < 100; $i++)
|
||||
{
|
||||
$mapped = run_cli($scfg, [ 'ls' ], binary => '/usr/bin/vitastor-nbd');
|
||||
($kerneldev) = grep { $mapped->{$_} && $mapped->{$_}->{image} && $mapped->{$_}->{image} eq $name } keys %$mapped;
|
||||
if ($kerneldev && -b $kerneldev)
|
||||
{
|
||||
my $size = `/usr/sbin/blockdev --getsize64 $kerneldev`;
|
||||
return $kerneldev if $size && $size > 0;
|
||||
}
|
||||
select(undef, undef, undef, 0.1);
|
||||
}
|
||||
|
||||
die "Failed to map Vitastor image $name via NBD".
|
||||
($map_out ? ", vitastor-nbd map returned '$map_out'" : "")."\n";
|
||||
$kerneldev = run_cli($scfg, [ 'map', '--image', $prefix.$name ], binary => '/usr/bin/vitastor-nbd', json => 0);
|
||||
return $kerneldev;
|
||||
}
|
||||
|
||||
sub unmap_volume
|
||||
@@ -406,19 +383,13 @@ sub unmap_volume
|
||||
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||
|
||||
my ($vtype, $name, $vmid) = $class->parse_volname($volname);
|
||||
$name = $prefix.$name;
|
||||
$name .= '@'.$snapname if $snapname;
|
||||
|
||||
my $mapped = run_cli($scfg, [ 'ls' ], binary => '/usr/bin/vitastor-nbd');
|
||||
|
||||
my @kerneldevs = grep {
|
||||
$mapped->{$_} && $mapped->{$_}->{image} && $mapped->{$_}->{image} eq $name
|
||||
} keys %$mapped;
|
||||
|
||||
for my $kerneldev (@kerneldevs)
|
||||
my ($kerneldev) = grep { $mapped->{$_}->{image} eq $prefix.$name } keys %$mapped;
|
||||
if ($kerneldev && -b $kerneldev)
|
||||
{
|
||||
next if !$kerneldev || !-b $kerneldev;
|
||||
eval { run_cli($scfg, [ 'unmap', $kerneldev ], binary => '/usr/bin/vitastor-nbd', json => 0); };
|
||||
warn "Failed to unmap Vitastor image $name from $kerneldev: $@" if $@;
|
||||
run_cli($scfg, [ 'unmap', $kerneldev ], binary => '/usr/bin/vitastor-nbd', json => 0);
|
||||
}
|
||||
|
||||
return 1;
|
||||
@@ -434,13 +405,7 @@ sub activate_volume
|
||||
sub deactivate_volume
|
||||
{
|
||||
my ($class, $storeid, $scfg, $volname, $snapname, $cache) = @_;
|
||||
|
||||
# Even with vitastor_nbd=0, Proxmox may call map_volume() for special
|
||||
# volumes like tpmstate0 because swtpm needs a local file/block path.
|
||||
# Therefore, always try to unmap an existing NBD mapping here.
|
||||
# unmap_volume() is a no-op if the volume is not currently mapped.
|
||||
$class->unmap_volume($storeid, $scfg, $volname, $snapname);
|
||||
|
||||
$class->unmap_volume($storeid, $scfg, $volname, $snapname) if $scfg->{vitastor_nbd};
|
||||
return 1;
|
||||
}
|
||||
|
||||
|
||||
@@ -50,7 +50,7 @@ from cinder.volume import configuration
|
||||
from cinder.volume import driver
|
||||
from cinder.volume import volume_utils
|
||||
|
||||
VITASTOR_VERSION = '3.0.15'
|
||||
VITASTOR_VERSION = '3.0.12'
|
||||
|
||||
LOG = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
Name: vitastor
|
||||
Version: 3.0.15
|
||||
Version: 3.0.12
|
||||
Release: 1%{?dist}
|
||||
Summary: Vitastor, a fast software-defined clustered block storage
|
||||
|
||||
License: Vitastor Network Public License 1.1
|
||||
URL: https://vitastor.io/
|
||||
Source0: vitastor-3.0.15.el10.tar.gz
|
||||
Source0: vitastor-3.0.12.el10.tar.gz
|
||||
|
||||
BuildRequires: gperftools-devel
|
||||
BuildRequires: gcc-c++
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
Name: vitastor
|
||||
Version: 3.0.15
|
||||
Version: 3.0.12
|
||||
Release: 1%{?dist}
|
||||
Summary: Vitastor, a fast software-defined clustered block storage
|
||||
|
||||
License: Vitastor Network Public License 1.1
|
||||
URL: https://vitastor.io/
|
||||
Source0: vitastor-3.0.15.el7.tar.gz
|
||||
Source0: vitastor-3.0.12.el7.tar.gz
|
||||
|
||||
BuildRequires: gperftools-devel
|
||||
BuildRequires: devtoolset-9-gcc-c++
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
Name: vitastor
|
||||
Version: 3.0.15
|
||||
Version: 3.0.12
|
||||
Release: 1%{?dist}
|
||||
Summary: Vitastor, a fast software-defined clustered block storage
|
||||
|
||||
License: Vitastor Network Public License 1.1
|
||||
URL: https://vitastor.io/
|
||||
Source0: vitastor-3.0.15.el8.tar.gz
|
||||
Source0: vitastor-3.0.12.el8.tar.gz
|
||||
|
||||
BuildRequires: gperftools-devel
|
||||
BuildRequires: gcc-toolset-9-gcc-c++
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
Name: vitastor
|
||||
Version: 3.0.15
|
||||
Version: 3.0.12
|
||||
Release: 1%{?dist}
|
||||
Summary: Vitastor, a fast software-defined clustered block storage
|
||||
|
||||
License: Vitastor Network Public License 1.1
|
||||
URL: https://vitastor.io/
|
||||
Source0: vitastor-3.0.15.el9.tar.gz
|
||||
Source0: vitastor-3.0.12.el9.tar.gz
|
||||
|
||||
BuildRequires: gperftools-devel
|
||||
BuildRequires: gcc-c++
|
||||
|
||||
+1
-1
@@ -20,7 +20,7 @@ if("${CMAKE_INSTALL_PREFIX}" MATCHES "^/usr/local/?$")
|
||||
endif()
|
||||
set(ENABLE_COVERAGE false CACHE BOOL "Enable code coverage")
|
||||
|
||||
add_definitions(-DVITASTOR_VERSION="3.0.15")
|
||||
add_definitions(-DVITASTOR_VERSION="3.0.12")
|
||||
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
|
||||
add_link_options(-fno-omit-frame-pointer)
|
||||
if (${WITH_ASAN})
|
||||
|
||||
@@ -521,7 +521,7 @@ void blockstore_disk_t::close_all()
|
||||
|
||||
// Sadly DISCARD only works through ioctl(), but it seems to always block the device queue,
|
||||
// so it's not a big deal that we can only run it synchronously.
|
||||
int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_used)
|
||||
int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_free)
|
||||
{
|
||||
if (mock_mode)
|
||||
{
|
||||
@@ -532,7 +532,7 @@ int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_used)
|
||||
uint64_t discarded = 0;
|
||||
for (; i <= block_count; i++)
|
||||
{
|
||||
if (i >= block_count || is_used(i))
|
||||
if (i >= block_count || is_free(i))
|
||||
{
|
||||
if (i > j && (i-j)*data_block_size >= min_discard_size)
|
||||
{
|
||||
@@ -545,21 +545,17 @@ int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_used)
|
||||
if (range[0] % discard_granularity)
|
||||
range[0] = range[0] + discard_granularity - (range[0] % discard_granularity);
|
||||
if (range[0] >= range[1])
|
||||
range[1] = 0;
|
||||
else
|
||||
range[1] -= range[0];
|
||||
continue;
|
||||
range[1] -= range[0];
|
||||
}
|
||||
if (range[1] > 0)
|
||||
r = ioctl(data_fd, BLKDISCARD, &range);
|
||||
if (r != 0)
|
||||
{
|
||||
r = ioctl(data_fd, BLKDISCARD, &range);
|
||||
if (r != 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to execute BLKDISCARD %ju+%ju on %s: %s (code %d)\n",
|
||||
range[0], range[1], data_device.c_str(), strerror(-r), r);
|
||||
return -errno;
|
||||
}
|
||||
discarded += range[1];
|
||||
fprintf(stderr, "Failed to execute BLKDISCARD %ju+%ju on %s: %s (code %d)\n",
|
||||
range[0], range[1], data_device.c_str(), strerror(-r), r);
|
||||
return -errno;
|
||||
}
|
||||
discarded += range[1];
|
||||
}
|
||||
j = i+1;
|
||||
}
|
||||
|
||||
@@ -84,7 +84,7 @@ struct blockstore_disk_t
|
||||
void calc_lengths(bool skip_meta_check = false);
|
||||
void check_lengths();
|
||||
void close_all();
|
||||
int trim_data(std::function<bool(uint64_t)> is_used);
|
||||
int trim_data(std::function<bool(uint64_t)> is_free);
|
||||
|
||||
inline uint64_t dirty_dyn_size(uint64_t offset, uint64_t len)
|
||||
{
|
||||
|
||||
@@ -289,41 +289,30 @@ resume_1:
|
||||
{
|
||||
init_fsync_data();
|
||||
}
|
||||
if (compact_info.do_delete)
|
||||
if (bs->log_level > 10)
|
||||
{
|
||||
if (bs->log_level > 10)
|
||||
{
|
||||
printf("Compacting %jx:%jx up to l%ju (delete)\n", cur_oid.inode, cur_oid.stripe, compact_info.compact_lsn);
|
||||
}
|
||||
clean_loc = UINT64_MAX;
|
||||
printf("Compacting %jx:%jx v%ju..v%ju / l%ju..l%ju (%d writes)\n", cur_oid.inode, cur_oid.stripe,
|
||||
compact_info.clean_wr->version, compact_info.compact_version,
|
||||
compact_info.clean_wr->lsn, compact_info.compact_lsn, copy_count);
|
||||
}
|
||||
else
|
||||
mem_or(new_bmp, compact_info.clean_wr->get_int_bitmap(bs->heap), bs->dsk.clean_entry_bitmap_size);
|
||||
if (!bitmap_copied)
|
||||
{
|
||||
if (bs->log_level > 10)
|
||||
{
|
||||
printf("Compacting %jx:%jx v%ju..v%ju / l%ju..l%ju (%d writes)\n", cur_oid.inode, cur_oid.stripe,
|
||||
compact_info.clean_wr->version, compact_info.compact_version,
|
||||
compact_info.clean_wr->lsn, compact_info.compact_lsn, copy_count);
|
||||
}
|
||||
mem_or(new_bmp, compact_info.clean_wr->get_int_bitmap(bs->heap), bs->dsk.clean_entry_bitmap_size);
|
||||
if (!bitmap_copied)
|
||||
{
|
||||
memcpy(new_ext_bmp, compact_info.clean_wr->get_ext_bitmap(bs->heap), bs->dsk.clean_entry_bitmap_size);
|
||||
bitmap_copied = true;
|
||||
}
|
||||
if (bs->dsk.csum_block_size && bs->dsk.csum_block_size <= bs->dsk.bitmap_granularity)
|
||||
{
|
||||
memcpy(new_csums, compact_info.clean_wr->get_checksums(bs->heap), bs->dsk.data_block_size/bs->dsk.csum_block_size * (bs->dsk.data_csum_type & 0xFF));
|
||||
for (size_t i = csum_copy.size(); i > 0; i--)
|
||||
{
|
||||
auto wr = csum_copy[i-1];
|
||||
memcpy(new_csums + wr->small().offset/bs->dsk.csum_block_size*(bs->dsk.data_csum_type & 0xFF),
|
||||
wr->get_checksums(bs->heap), wr->small().len/bs->dsk.csum_block_size*(bs->dsk.data_csum_type & 0xFF));
|
||||
}
|
||||
csum_copy.clear();
|
||||
}
|
||||
clean_loc = compact_info.clean_wr->big_location(bs->heap);
|
||||
memcpy(new_ext_bmp, compact_info.clean_wr->get_ext_bitmap(bs->heap), bs->dsk.clean_entry_bitmap_size);
|
||||
bitmap_copied = true;
|
||||
}
|
||||
if (bs->dsk.csum_block_size && bs->dsk.csum_block_size <= bs->dsk.bitmap_granularity)
|
||||
{
|
||||
memcpy(new_csums, compact_info.clean_wr->get_checksums(bs->heap), bs->dsk.data_block_size/bs->dsk.csum_block_size * (bs->dsk.data_csum_type & 0xFF));
|
||||
for (size_t i = csum_copy.size(); i > 0; i--)
|
||||
{
|
||||
auto wr = csum_copy[i-1];
|
||||
memcpy(new_csums + wr->small().offset/bs->dsk.csum_block_size*(bs->dsk.data_csum_type & 0xFF),
|
||||
wr->get_checksums(bs->heap), wr->small().len/bs->dsk.csum_block_size*(bs->dsk.data_csum_type & 0xFF));
|
||||
}
|
||||
csum_copy.clear();
|
||||
}
|
||||
clean_loc = compact_info.clean_wr->big_location(bs->heap);
|
||||
overwrite_start = overwrite_end = 0;
|
||||
if (read_vec.size() > 0)
|
||||
{
|
||||
@@ -636,7 +625,7 @@ int journal_flusher_co::check_and_punch_checksums()
|
||||
|
||||
bool journal_flusher_co::calc_block_checksums()
|
||||
{
|
||||
if (bs->dsk.csum_block_size <= bs->dsk.bitmap_granularity || compact_info.do_delete)
|
||||
if (bs->dsk.csum_block_size <= bs->dsk.bitmap_granularity)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -23,7 +23,7 @@
|
||||
|
||||
#define HEAP_INFLIGHT_DONE 1
|
||||
#define HEAP_INFLIGHT_COMPACTABLE 2
|
||||
#define HEAP_INFLIGHT_OVERWRITE 4
|
||||
#define HEAP_INFLIGHT_COMPACTED 4
|
||||
#define HEAP_INFLIGHT_GC 8
|
||||
#define HEAP_INFLIGHT_EXPLICIT 16
|
||||
|
||||
@@ -123,8 +123,6 @@ uint32_t heap_entry_t::get_size(blockstore_heap_t *heap)
|
||||
}
|
||||
if (type() == BS_HEAP_SMALL_WRITE || type() == BS_HEAP_INTENT_WRITE)
|
||||
{
|
||||
if (size < sizeof(heap_small_write_t))
|
||||
return heap->get_small_entry_size(0, 0);
|
||||
return heap->get_small_entry_size(small().offset, small().len);
|
||||
}
|
||||
return heap->get_simple_entry_size();
|
||||
@@ -376,11 +374,14 @@ corrupted_object:
|
||||
return EDOM;
|
||||
}
|
||||
}
|
||||
if (wr->size != wr->get_size(this))
|
||||
if (((wr->entry_type & BS_HEAP_TYPE) == BS_HEAP_SMALL_WRITE ||
|
||||
(wr->entry_type & BS_HEAP_TYPE) == BS_HEAP_INTENT_WRITE) &&
|
||||
wr->size < sizeof(heap_small_write_t))
|
||||
{
|
||||
// Check entry size
|
||||
fprintf(stderr, "Error: entry %jx:%jx v%ju has invalid size in metadata block %u at %u (%u != %u bytes)\n",
|
||||
wr->inode, wr->stripe, wr->version, block_num, block_offset, wr->size, wr->get_size(this));
|
||||
// Small writes require accessing offset & len to calculate correct length,
|
||||
// so require at least sizeof(heap_small_write_t) for them
|
||||
fprintf(stderr, "Error: entry %jx:%jx v%ju has invalid size in metadata block %u at %u (%u < min %zu bytes)\n",
|
||||
wr->inode, wr->stripe, wr->version, block_num, block_offset, wr->size, sizeof(heap_small_write_t));
|
||||
goto corrupted_object;
|
||||
}
|
||||
if (wr->entry_type == BS_HEAP_COMMIT && !wr->version)
|
||||
@@ -569,7 +570,7 @@ void blockstore_heap_t::finish_load()
|
||||
size_t s = 0, e, n = postponed_items.size();
|
||||
for (e = 1; e <= n; e++)
|
||||
{
|
||||
if (e >= n || postponed_items[e]->entry.inode != postponed_items[s]->entry.inode ||
|
||||
if (e >= n || postponed_items[e]->entry.inode != postponed_items[s]->entry.inode &&
|
||||
postponed_items[e]->entry.stripe != postponed_items[s]->entry.stripe)
|
||||
{
|
||||
insert_list_items(postponed_items.data()+s, e-s, false);
|
||||
@@ -692,14 +693,10 @@ int blockstore_heap_t::mark_used_blocks()
|
||||
}
|
||||
use_data(wr->inode, wr->big_location(this));
|
||||
}
|
||||
if (wr->is_compactable())
|
||||
if (wr->is_compactable() && !added)
|
||||
{
|
||||
to_compact_count++;
|
||||
if (!added)
|
||||
{
|
||||
compact_queue.push_back((object_id){ .inode = wr->inode, .stripe = wr->stripe });
|
||||
added = true;
|
||||
}
|
||||
compact_queue.push_back((object_id){ .inode = wr->inode, .stripe = wr->stripe });
|
||||
added = true;
|
||||
}
|
||||
if (wr->is_overwrite())
|
||||
{
|
||||
@@ -709,11 +706,6 @@ int blockstore_heap_t::mark_used_blocks()
|
||||
});
|
||||
}
|
||||
}
|
||||
for (auto li: init_erase_items)
|
||||
{
|
||||
unlink_list_item(li);
|
||||
}
|
||||
init_erase_items.clear();
|
||||
if (dsk->gc_on_start)
|
||||
{
|
||||
recheck_full_gc();
|
||||
@@ -749,6 +741,7 @@ void blockstore_heap_t::init_erase_bad_entry(heap_list_item_t *li)
|
||||
inf.garbage_space -= (li->entry.is_garbage() ? li->entry.size : 0);
|
||||
});
|
||||
recheck_modified_blocks.insert(li->block_num);
|
||||
unlink_list_item(li);
|
||||
}
|
||||
|
||||
bool blockstore_heap_t::init_erase_double_claim(heap_list_item_t *prev_li, heap_list_item_t *cur_li)
|
||||
@@ -803,8 +796,6 @@ bool blockstore_heap_t::init_erase_double_claim(heap_list_item_t *prev_li, heap_
|
||||
overwritten = erase_li->entry.is_overwrite();
|
||||
}
|
||||
init_erase_bad_entry(erase_li);
|
||||
// Can't erase (mutate map) while iterating, so postpone it
|
||||
init_erase_items.push_back(erase_li);
|
||||
erase_li = prev_erase_li;
|
||||
}
|
||||
}
|
||||
@@ -818,8 +809,6 @@ bool blockstore_heap_t::init_erase_double_claim(heap_list_item_t *prev_li, heap_
|
||||
auto next_erase_li = erase_li->next;
|
||||
init_free_bad_entry(&erase_li->entry);
|
||||
init_erase_bad_entry(erase_li);
|
||||
// Can't erase (mutate map) while iterating, so postpone it
|
||||
init_erase_items.push_back(erase_li);
|
||||
erase_li = next_erase_li;
|
||||
}
|
||||
erase_li = cur_li;
|
||||
@@ -828,8 +817,6 @@ bool blockstore_heap_t::init_erase_double_claim(heap_list_item_t *prev_li, heap_
|
||||
{
|
||||
auto prev_erase_li = erase_li->prev;
|
||||
init_erase_bad_entry(erase_li);
|
||||
// Can't erase (mutate map) while iterating, so postpone it
|
||||
init_erase_items.push_back(erase_li);
|
||||
erase_li = prev_erase_li;
|
||||
}
|
||||
}
|
||||
@@ -900,7 +887,6 @@ void blockstore_heap_t::recheck_drop_entries(heap_entry_t *obj, heap_entry_t *ba
|
||||
auto prev = li->prev;
|
||||
assert(li->entry.type() == bad_wr->type());
|
||||
init_erase_bad_entry(li);
|
||||
unlink_list_item(li);
|
||||
li = prev;
|
||||
}
|
||||
}
|
||||
@@ -1182,7 +1168,7 @@ bool blockstore_heap_t::calc_block_checksums(uint32_t *block_csums, uint8_t *bit
|
||||
while (pos < end && pos < block_end && !(bitmap[pos/dsk->bitmap_granularity/8] & (1 << ((pos/dsk->bitmap_granularity) % 8))))
|
||||
pos += dsk->bitmap_granularity;
|
||||
// zero padding at the beginning or at the end of the block is not counted
|
||||
if (pos > prev && prev > blk_start && pos < block_end)
|
||||
if (pos > prev && prev > 0 && pos < block_end)
|
||||
{
|
||||
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
|
||||
{
|
||||
@@ -1627,7 +1613,7 @@ int blockstore_heap_t::add_entry(uint32_t wr_size, uint32_t *modified_block,
|
||||
// Remember the object as dirty and remove older entries when this block is written and fsynced
|
||||
push_inflight_lsn(next_lsn, new_wr,
|
||||
(explicit_complete ? HEAP_INFLIGHT_EXPLICIT : 0) |
|
||||
(new_wr->is_overwrite() ? HEAP_INFLIGHT_OVERWRITE : 0) |
|
||||
(new_wr->is_overwrite() ? HEAP_INFLIGHT_COMPACTED : 0) |
|
||||
(new_wr->is_compactable() ? HEAP_INFLIGHT_COMPACTABLE : 0));
|
||||
insert_list_items(&li, 1, false);
|
||||
li->block_num = block_num;
|
||||
@@ -1661,22 +1647,10 @@ int blockstore_heap_t::add_small_write(object_id oid, heap_entry_t **obj_ptr, ui
|
||||
wr->small().location = location;
|
||||
if (bitmap)
|
||||
memcpy(wr->get_ext_bitmap(this), bitmap, dsk->clean_entry_bitmap_size);
|
||||
else if (obj)
|
||||
memcpy(wr->get_ext_bitmap(this), obj->get_ext_bitmap(this), dsk->clean_entry_bitmap_size);
|
||||
else
|
||||
{
|
||||
bool found = false;
|
||||
iterate_with_stable(obj, UINT64_MAX, [&](heap_entry_t *old_wr, bool stable)
|
||||
{
|
||||
if (old_wr->get_ext_bitmap(this))
|
||||
{
|
||||
found = true;
|
||||
memcpy(wr->get_ext_bitmap(this), old_wr->get_ext_bitmap(this), dsk->clean_entry_bitmap_size);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
if (!found)
|
||||
memset(wr->get_ext_bitmap(this), 0, dsk->clean_entry_bitmap_size);
|
||||
}
|
||||
memset(wr->get_ext_bitmap(this), 0, dsk->clean_entry_bitmap_size);
|
||||
calc_checksums(wr, (uint8_t*)data, true);
|
||||
*obj_ptr = wr;
|
||||
});
|
||||
@@ -2140,10 +2114,7 @@ void blockstore_heap_t::iterate_with_stable(heap_entry_t *obj, uint64_t max_lsn,
|
||||
{
|
||||
if (old_wr->type() == BS_HEAP_ROLLBACK)
|
||||
{
|
||||
if (rollback_version > old_wr->version)
|
||||
{
|
||||
rollback_version = old_wr->version;
|
||||
}
|
||||
rollback_version = old_wr->version;
|
||||
}
|
||||
else if (old_wr->type() == BS_HEAP_COMMIT)
|
||||
{
|
||||
@@ -2195,10 +2166,7 @@ heap_compact_t blockstore_heap_t::iterate_compaction(heap_entry_t *obj, uint64_t
|
||||
res.compact_lsn = wr->lsn;
|
||||
res.compact_version = wr->version;
|
||||
}
|
||||
if (rollback_version > wr->version)
|
||||
{
|
||||
rollback_version = wr->version;
|
||||
}
|
||||
rollback_version = wr->version;
|
||||
continue;
|
||||
}
|
||||
if (wr->type() == BS_HEAP_COMMIT && wr->lsn <= fsynced_lsn)
|
||||
@@ -2376,7 +2344,7 @@ void blockstore_heap_t::use_data(inode_t inode, uint64_t location)
|
||||
{
|
||||
auto sh_it = pool_shard_settings.find(INODE_POOL(inode));
|
||||
if (sh_it != pool_shard_settings.end() && sh_it->second.no_inode_stats)
|
||||
inode = INODE_WITH_POOL(INODE_POOL(inode), 0);
|
||||
inode = (INODE_POOL(inode) << POOL_ID_BITS);
|
||||
assert(!data_alloc->get(location / dsk->data_block_size));
|
||||
data_alloc->set(location / dsk->data_block_size, true);
|
||||
inode_space_stats[inode] += dsk->data_block_size;
|
||||
@@ -2387,7 +2355,7 @@ void blockstore_heap_t::free_data(inode_t inode, uint64_t location)
|
||||
{
|
||||
auto sh_it = pool_shard_settings.find(INODE_POOL(inode));
|
||||
if (sh_it != pool_shard_settings.end() && sh_it->second.no_inode_stats)
|
||||
inode = INODE_WITH_POOL(INODE_POOL(inode), 0);
|
||||
inode = (INODE_POOL(inode) << POOL_ID_BITS);
|
||||
assert(data_alloc->get(location / dsk->data_block_size));
|
||||
data_alloc->set(location / dsk->data_block_size, false);
|
||||
auto sp_it = inode_space_stats.find(inode);
|
||||
@@ -2424,8 +2392,7 @@ void blockstore_heap_t::use_buffer_area(inode_t inode, uint64_t location, uint64
|
||||
return;
|
||||
}
|
||||
assert(!(size % dsk->bitmap_granularity));
|
||||
bool ok = buffer_alloc->use(location / dsk->bitmap_granularity, size / dsk->bitmap_granularity);
|
||||
assert(ok);
|
||||
buffer_alloc->use(location / dsk->bitmap_granularity, size / dsk->bitmap_granularity);
|
||||
buffer_area_used_space += size;
|
||||
}
|
||||
|
||||
@@ -2467,7 +2434,7 @@ void blockstore_heap_t::get_meta_block(uint32_t block_num, uint8_t *buffer)
|
||||
}
|
||||
}
|
||||
|
||||
void blockstore_heap_t::fill_block_empty_space(uint8_t *buffer, uint64_t pos)
|
||||
void blockstore_heap_t::fill_block_empty_space(uint8_t *buffer, uint32_t pos)
|
||||
{
|
||||
if (pos > dsk->meta_block_size)
|
||||
{
|
||||
@@ -2555,7 +2522,7 @@ uint64_t blockstore_heap_t::get_garbage_memory()
|
||||
void blockstore_heap_t::push_inflight_lsn(uint64_t lsn, heap_entry_t *wr, uint64_t flags)
|
||||
{
|
||||
uint64_t next_inf = first_inflight_lsn + inflight_lsn.size();
|
||||
if (flags & (HEAP_INFLIGHT_COMPACTABLE|HEAP_INFLIGHT_OVERWRITE))
|
||||
if (flags & (HEAP_INFLIGHT_COMPACTABLE|HEAP_INFLIGHT_COMPACTED))
|
||||
{
|
||||
to_compact_count++;
|
||||
}
|
||||
@@ -2622,7 +2589,7 @@ void blockstore_heap_t::mark_lsn_fsynced(uint64_t lsn)
|
||||
void blockstore_heap_t::apply_inflight(heap_inflight_lsn_t & inflight)
|
||||
{
|
||||
auto wr = inflight.wr;
|
||||
if (inflight.flags & HEAP_INFLIGHT_OVERWRITE)
|
||||
if (inflight.flags & HEAP_INFLIGHT_COMPACTED)
|
||||
{
|
||||
// Mark previous entries as garbage, sequentially
|
||||
mark_garbage_up_to(wr);
|
||||
|
||||
@@ -220,7 +220,6 @@ class blockstore_heap_t
|
||||
bool marked_used_blocks = false;
|
||||
bool recheck_queue_filled = false;
|
||||
std::vector<heap_list_item_t*> postponed_items;
|
||||
std::vector<heap_list_item_t*> init_erase_items;
|
||||
std::set<uint32_t> recheck_modified_blocks;
|
||||
std::deque<heap_entry_t*> recheck_queue;
|
||||
std::map<heap_entry_t*, heap_recheck_state_t> recheck_states;
|
||||
@@ -363,7 +362,7 @@ public:
|
||||
|
||||
// get metadata block data buffer and used space
|
||||
void get_meta_block(uint32_t block_num, uint8_t *buffer);
|
||||
void fill_block_empty_space(uint8_t *buffer, uint64_t pos);
|
||||
void fill_block_empty_space(uint8_t *buffer, uint32_t pos);
|
||||
uint32_t get_meta_block_used_space(uint32_t block_num);
|
||||
|
||||
// get space usage statistics
|
||||
|
||||
@@ -117,12 +117,9 @@ public:
|
||||
|
||||
journal_flusher_t *flusher;
|
||||
int write_iodepth = 0;
|
||||
int inflight_big = 0;
|
||||
int intent_write_counter = 0;
|
||||
uint64_t data_fsync_next = 0;
|
||||
uint64_t data_fsync_cur = 0;
|
||||
uint64_t data_fsync_sent = 0;
|
||||
uint64_t data_fsync_done = 0;
|
||||
std::deque<bool> data_fsyncs;
|
||||
bool fsyncing_data = false;
|
||||
|
||||
bool live = false, queue_stall = false;
|
||||
ring_loop_i *ringloop = NULL;
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
#define INIT_META_EMPTY 0
|
||||
#define INIT_META_READING 1
|
||||
#define INIT_META_READ_DONE 2
|
||||
#define INIT_META_WRITING 3
|
||||
|
||||
#define GET_SQE() \
|
||||
sqe = bs->get_sqe();\
|
||||
@@ -22,15 +23,14 @@ blockstore_init_meta::blockstore_init_meta(blockstore_impl_t *bs)
|
||||
this->bs = bs;
|
||||
}
|
||||
|
||||
void blockstore_init_meta::handle_event(ring_data_t *data, int buf_num, const char *op)
|
||||
void blockstore_init_meta::handle_event(ring_data_t *data, int buf_num)
|
||||
{
|
||||
if (data->res != data->iov.iov_len)
|
||||
if (data->res < 0)
|
||||
{
|
||||
throw std::runtime_error(strprintf(
|
||||
"%s failed at offset %ju: got %s (code %d), but expected %zu",
|
||||
op, (buf_num >= 0 ? bufs[buf_num].offset : last_read_offset), strerror(-data->res),
|
||||
data->res, data->iov.iov_len
|
||||
));
|
||||
throw std::runtime_error(
|
||||
std::string("read metadata failed at offset ") + std::to_string(buf_num >= 0 ? bufs[buf_num].offset : last_read_offset) +
|
||||
std::string(": ") + strerror(-data->res)
|
||||
);
|
||||
}
|
||||
if (buf_num >= 0)
|
||||
{
|
||||
@@ -60,7 +60,7 @@ int blockstore_init_meta::loop()
|
||||
GET_SQE();
|
||||
last_read_offset = 0;
|
||||
data->iov = { bs->meta_superblock, (size_t)bs->dsk.meta_block_size };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "read metadata header"); };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
||||
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
||||
bs->ringloop->submit();
|
||||
submitted++;
|
||||
@@ -72,19 +72,25 @@ resume_1:
|
||||
}
|
||||
if (is_zero((uint64_t*)bs->meta_superblock, bs->dsk.meta_block_size))
|
||||
{
|
||||
assert(bs->dsk.meta_format == BLOCKSTORE_META_FORMAT_HEAP);
|
||||
blockstore_meta_header_v3_t *hdr = (blockstore_meta_header_v3_t *)bs->meta_superblock;
|
||||
hdr->zero = 0;
|
||||
hdr->magic = BLOCKSTORE_META_MAGIC_V1;
|
||||
hdr->version = bs->dsk.meta_format;
|
||||
hdr->meta_block_size = bs->dsk.meta_block_size;
|
||||
hdr->data_block_size = bs->dsk.data_block_size;
|
||||
hdr->bitmap_granularity = bs->dsk.bitmap_granularity;
|
||||
hdr->completed_lsn = 0;
|
||||
hdr->data_csum_type = bs->dsk.data_csum_type;
|
||||
hdr->csum_block_size = bs->dsk.csum_block_size;
|
||||
hdr->meta_area_size = bs->dsk.meta_area_size;
|
||||
hdr->set_crc32c();
|
||||
{
|
||||
blockstore_meta_header_v3_t *hdr = (blockstore_meta_header_v3_t *)bs->meta_superblock;
|
||||
hdr->zero = 0;
|
||||
hdr->magic = BLOCKSTORE_META_MAGIC_V1;
|
||||
hdr->version = bs->dsk.meta_format;
|
||||
hdr->meta_block_size = bs->dsk.meta_block_size;
|
||||
hdr->data_block_size = bs->dsk.data_block_size;
|
||||
hdr->bitmap_granularity = bs->dsk.bitmap_granularity;
|
||||
if (bs->dsk.meta_format >= BLOCKSTORE_META_FORMAT_V2)
|
||||
{
|
||||
hdr->data_csum_type = bs->dsk.data_csum_type;
|
||||
hdr->csum_block_size = bs->dsk.csum_block_size;
|
||||
}
|
||||
if (bs->dsk.meta_format >= BLOCKSTORE_META_FORMAT_HEAP)
|
||||
{
|
||||
hdr->meta_area_size = bs->dsk.meta_area_size;
|
||||
}
|
||||
hdr->set_crc32c();
|
||||
}
|
||||
if (bs->readonly)
|
||||
{
|
||||
printf("Skipping metadata initialization because blockstore is readonly\n");
|
||||
@@ -92,8 +98,21 @@ resume_1:
|
||||
else
|
||||
{
|
||||
printf("Initializing metadata area\n");
|
||||
GET_SQE();
|
||||
last_read_offset = 0;
|
||||
data->iov = (struct iovec){ bs->meta_superblock, (size_t)bs->dsk.meta_block_size };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
||||
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
||||
bs->ringloop->submit();
|
||||
submitted++;
|
||||
resume_2:
|
||||
if (submitted > 0)
|
||||
{
|
||||
wait_state = 2;
|
||||
return 1;
|
||||
}
|
||||
zero_on_init = true;
|
||||
}
|
||||
zero_on_init = true;
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -144,7 +163,7 @@ resume_1:
|
||||
hdr->header_csum = csum;
|
||||
}
|
||||
bs->heap->start_load(((blockstore_meta_header_v3_t *)bs->meta_superblock)->completed_lsn);
|
||||
if (bs->dsk.inmemory_journal && !zero_on_init)
|
||||
if (bs->dsk.inmemory_journal)
|
||||
{
|
||||
// Read buffer area
|
||||
printf("Reading buffered data\n");
|
||||
@@ -156,7 +175,7 @@ resume_1:
|
||||
bs->buffer_area + md_offset,
|
||||
(size_t)(bs->dsk.journal_len - md_offset < bs->metadata_buf_size ? bs->dsk.journal_len - md_offset : bs->metadata_buf_size),
|
||||
};
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "read buffer area"); };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
||||
io_uring_prep_readv(sqe, bs->dsk.journal_fd, &data->iov, 1, bs->dsk.journal_offset + md_offset);
|
||||
md_offset += data->iov.iov_len;
|
||||
submitted++;
|
||||
@@ -175,7 +194,7 @@ resume_3:
|
||||
next_offset = md_offset;
|
||||
// Read the rest of the metadata
|
||||
resume_4:
|
||||
if (next_offset < bs->dsk.meta_area_size && submitted == 0 && (!zero_on_init || !bs->readonly))
|
||||
if (next_offset < bs->dsk.meta_area_size && submitted == 0)
|
||||
{
|
||||
// Submit one read
|
||||
for (int i = 0; i < 2; i++)
|
||||
@@ -192,15 +211,12 @@ resume_4:
|
||||
GET_SQE();
|
||||
assert(bufs[i].size <= 0x7fffffff);
|
||||
data->iov = { bufs[i].buf, (size_t)bufs[i].size };
|
||||
data->callback = [this, i](ring_data_t *data) { handle_event(data, i); };
|
||||
if (!zero_on_init)
|
||||
{
|
||||
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "read metadata"); };
|
||||
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
|
||||
}
|
||||
else
|
||||
{
|
||||
// Fill metadata with empty block pattern
|
||||
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "clear metadata"); };
|
||||
memset(bufs[i].buf, 0, bufs[i].size);
|
||||
for (uint64_t o = 0; o < bufs[i].size; o += bs->dsk.meta_block_size)
|
||||
bs->heap->fill_block_empty_space(bufs[i].buf + o, 0);
|
||||
@@ -216,14 +232,11 @@ resume_4:
|
||||
if (bufs[i].state == INIT_META_READ_DONE)
|
||||
{
|
||||
// Handle result
|
||||
if (!zero_on_init)
|
||||
{
|
||||
uint64_t loaded = 0;
|
||||
int r = bs->heap->load_blocks(bufs[i].offset-bs->dsk.meta_block_size, bufs[i].size, bufs[i].buf, bs->skip_corrupted_meta_entries, loaded);
|
||||
if (r != 0)
|
||||
exit(1);
|
||||
entries_loaded += loaded;
|
||||
}
|
||||
uint64_t loaded = 0;
|
||||
int r = bs->heap->load_blocks(bufs[i].offset-bs->dsk.meta_block_size, bufs[i].size, bufs[i].buf, bs->skip_corrupted_meta_entries, loaded);
|
||||
if (r != 0)
|
||||
exit(1);
|
||||
entries_loaded += loaded;
|
||||
bufs[i].state = 0;
|
||||
bs->ringloop->wakeup();
|
||||
}
|
||||
@@ -233,7 +246,7 @@ resume_4:
|
||||
wait_state = 4;
|
||||
return 1;
|
||||
}
|
||||
// metadata read/clear finished
|
||||
// metadata read finished
|
||||
bs->heap->finish_load();
|
||||
printf("Metadata entries loaded: %ju, rechecking unfinished writes and garbage entries\n", entries_loaded);
|
||||
// asynchronous recheck
|
||||
@@ -316,14 +329,13 @@ resume_9:
|
||||
}
|
||||
free(metadata_buffer);
|
||||
metadata_buffer = NULL;
|
||||
do_fsync:
|
||||
if (!bs->dsk.disable_meta_fsync && !bs->readonly)
|
||||
{
|
||||
GET_SQE();
|
||||
io_uring_prep_fsync(sqe, bs->dsk.meta_fd, IORING_FSYNC_DATASYNC);
|
||||
last_read_offset = 0;
|
||||
data->iov = { 0 };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "fsync metadata"); };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
||||
submitted++;
|
||||
bs->ringloop->submit();
|
||||
resume_5:
|
||||
@@ -333,27 +345,6 @@ do_fsync:
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
if (zero_on_init && !header_written && !bs->readonly)
|
||||
{
|
||||
GET_SQE();
|
||||
header_written = true;
|
||||
last_read_offset = 0;
|
||||
data->iov = (struct iovec){ bs->meta_superblock, (size_t)bs->dsk.meta_block_size };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "write metadata header"); };
|
||||
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
||||
bs->ringloop->submit();
|
||||
submitted++;
|
||||
resume_2:
|
||||
if (submitted > 0)
|
||||
{
|
||||
wait_state = 2;
|
||||
return 1;
|
||||
}
|
||||
if (!bs->dsk.disable_meta_fsync)
|
||||
{
|
||||
goto do_fsync;
|
||||
}
|
||||
}
|
||||
printf("Loading finished. Data used: %ju / %ju bytes (%s / %s)\n",
|
||||
bs->heap->get_data_used_space(), bs->dsk.block_count * bs->dsk.data_block_size,
|
||||
format_size(bs->heap->get_data_used_space()).c_str(),
|
||||
|
||||
@@ -17,7 +17,6 @@ class blockstore_init_meta
|
||||
int wait_state = 0;
|
||||
int wait_count = 0;
|
||||
bool zero_on_init = false;
|
||||
bool header_written = false;
|
||||
void *metadata_buffer = NULL;
|
||||
blockstore_init_meta_buf bufs[2] = {};
|
||||
int submitted = 0;
|
||||
@@ -30,7 +29,7 @@ class blockstore_init_meta
|
||||
std::vector<uint32_t> recheck_mod;
|
||||
int i = 0, j = 0;
|
||||
bool handle_meta_block(uint8_t *buf, uint64_t count, uint64_t done_cnt);
|
||||
void handle_event(ring_data_t *data, int buf_num, const char *op);
|
||||
void handle_event(ring_data_t *data, int buf_num);
|
||||
public:
|
||||
blockstore_init_meta(blockstore_impl_t *bs);
|
||||
int loop();
|
||||
|
||||
@@ -1,113 +0,0 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#include "blockstore_mock.h"
|
||||
|
||||
blockstore_mock_t::blockstore_mock_t(const blockstore_config_t & config)
|
||||
{
|
||||
}
|
||||
|
||||
void blockstore_mock_t::parse_config(blockstore_config_t & config)
|
||||
{
|
||||
}
|
||||
|
||||
void* blockstore_mock_t::reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit)
|
||||
{
|
||||
return NULL;
|
||||
}
|
||||
|
||||
bool blockstore_mock_t::reshard_continue(void *reshard_state, uint64_t chunk_limit)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
void blockstore_mock_t::loop()
|
||||
{
|
||||
}
|
||||
|
||||
bool blockstore_mock_t::is_started()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
bool blockstore_mock_t::is_stalled()
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
bool blockstore_mock_t::is_safe_to_stop()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
void blockstore_mock_t::enqueue_op(blockstore_op_t *op)
|
||||
{
|
||||
}
|
||||
|
||||
int blockstore_mock_t::read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version)
|
||||
{
|
||||
return -EIO;
|
||||
}
|
||||
|
||||
const std::map<uint64_t, uint64_t> & blockstore_mock_t::get_inode_space_stats()
|
||||
{
|
||||
return inode_space;
|
||||
}
|
||||
|
||||
void blockstore_mock_t::set_no_inode_stats(const std::vector<uint64_t> & pool_ids)
|
||||
{
|
||||
}
|
||||
|
||||
void blockstore_mock_t::dump_diagnostics()
|
||||
{
|
||||
}
|
||||
|
||||
std::string blockstore_mock_t::get_op_diag(blockstore_op_t *op)
|
||||
{
|
||||
return "";
|
||||
}
|
||||
|
||||
uint32_t blockstore_mock_t::get_block_size()
|
||||
{
|
||||
return block_size;
|
||||
}
|
||||
|
||||
uint64_t blockstore_mock_t::get_block_count()
|
||||
{
|
||||
return block_count;
|
||||
}
|
||||
|
||||
uint64_t blockstore_mock_t::get_free_block_count()
|
||||
{
|
||||
return block_count;
|
||||
}
|
||||
|
||||
uint64_t blockstore_mock_t::get_journal_size()
|
||||
{
|
||||
return 32*1024*1024;
|
||||
}
|
||||
|
||||
uint32_t blockstore_mock_t::get_bitmap_granularity()
|
||||
{
|
||||
return bitmap_granularity;
|
||||
}
|
||||
|
||||
uint64_t blockstore_mock_t::get_live_entries()
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint64_t blockstore_mock_t::get_live_memory()
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint64_t blockstore_mock_t::get_garbage_entries()
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint64_t blockstore_mock_t::get_garbage_memory()
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -1,39 +0,0 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "blockstore.h"
|
||||
|
||||
class blockstore_mock_t: public blockstore_i
|
||||
{
|
||||
public:
|
||||
uint32_t block_size = 128*1024;
|
||||
uint32_t bitmap_granularity = 4096;
|
||||
uint64_t block_count = 100*1024*8;
|
||||
std::map<uint64_t, uint64_t> inode_space;
|
||||
|
||||
blockstore_mock_t(const blockstore_config_t & config);
|
||||
void parse_config(blockstore_config_t & config) override;
|
||||
void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit) override;
|
||||
bool reshard_continue(void *reshard_state, uint64_t chunk_limit) override;
|
||||
void loop() override;
|
||||
bool is_started() override;
|
||||
bool is_stalled() override;
|
||||
bool is_safe_to_stop() override;
|
||||
void enqueue_op(blockstore_op_t *op) override;
|
||||
int read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version = NULL) override;
|
||||
const std::map<uint64_t, uint64_t> & get_inode_space_stats() override;
|
||||
void set_no_inode_stats(const std::vector<uint64_t> & pool_ids) override;
|
||||
void dump_diagnostics() override;
|
||||
std::string get_op_diag(blockstore_op_t *op) override;
|
||||
uint32_t get_block_size() override;
|
||||
uint64_t get_block_count() override;
|
||||
uint64_t get_free_block_count() override;
|
||||
uint64_t get_journal_size() override;
|
||||
uint32_t get_bitmap_granularity() override;
|
||||
uint64_t get_live_entries() override;
|
||||
uint64_t get_live_memory() override;
|
||||
uint64_t get_garbage_entries() override;
|
||||
uint64_t get_garbage_memory() override;
|
||||
};
|
||||
@@ -28,7 +28,6 @@ int blockstore_impl_t::dequeue_stable(blockstore_op_t *op)
|
||||
FINISH_OP(op);
|
||||
return 2;
|
||||
}
|
||||
priv->modified_block2 = UINT32_MAX;
|
||||
int res = op->opcode == BS_OP_STABLE
|
||||
? heap->add_commit(obj, v[priv->stab_pos].version, &priv->modified_block2)
|
||||
: heap->add_rollback(obj, v[priv->stab_pos].version, &priv->modified_block2);
|
||||
@@ -53,6 +52,11 @@ int blockstore_impl_t::dequeue_stable(blockstore_op_t *op)
|
||||
FINISH_OP(op);
|
||||
return 2;
|
||||
}
|
||||
if (priv->modified_block2 != UINT32_MAX)
|
||||
{
|
||||
priv->stab_pos--;
|
||||
goto resume_1;
|
||||
}
|
||||
priv->wait_for = WAIT_COMPACTION;
|
||||
priv->wait_detail = heap->get_compacted_count();
|
||||
flusher->request_trim();
|
||||
|
||||
@@ -29,12 +29,9 @@ bool blockstore_impl_t::has_unsynced()
|
||||
|
||||
bool blockstore_impl_t::submit_fsyncs(int & wait_count)
|
||||
{
|
||||
int n = (unsynced_meta_write_count > 0 && !dsk.disable_meta_fsync ? 1 : 0) +
|
||||
(unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync &&
|
||||
(!unsynced_meta_write_count || dsk.journal_fd != dsk.meta_fd) ? 1 : 0) +
|
||||
(unsynced_data_write_count > 0 && !dsk.disable_data_fsync &&
|
||||
(!unsynced_meta_write_count || dsk.data_fd != dsk.meta_fd) &&
|
||||
(!unsynced_buffer_write_count || dsk.data_fd != dsk.journal_fd) ? 1 : 0);
|
||||
int n = (unsynced_meta_write_count > 0 && !dsk.disable_meta_fsync) +
|
||||
(unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync && dsk.journal_fd != dsk.meta_fd) +
|
||||
(unsynced_data_write_count > 0 && !dsk.disable_data_fsync && dsk.data_fd != dsk.meta_fd && dsk.data_fd != dsk.journal_fd);
|
||||
if (ringloop->space_left() < n)
|
||||
{
|
||||
return false;
|
||||
@@ -63,8 +60,7 @@ bool blockstore_impl_t::submit_fsyncs(int & wait_count)
|
||||
data->callback = cb;
|
||||
wait_count++;
|
||||
}
|
||||
if (unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync &&
|
||||
(!unsynced_meta_write_count || dsk.journal_fd != dsk.meta_fd))
|
||||
if (unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync && dsk.meta_fd != dsk.journal_fd)
|
||||
{
|
||||
// fsync buffer
|
||||
io_uring_sqe *sqe = get_sqe();
|
||||
@@ -75,9 +71,7 @@ bool blockstore_impl_t::submit_fsyncs(int & wait_count)
|
||||
data->callback = cb;
|
||||
wait_count++;
|
||||
}
|
||||
if (unsynced_data_write_count > 0 && !dsk.disable_data_fsync &&
|
||||
(!unsynced_meta_write_count || dsk.data_fd != dsk.meta_fd) &&
|
||||
(!unsynced_buffer_write_count || dsk.data_fd != dsk.journal_fd))
|
||||
if (unsynced_data_write_count > 0 && !dsk.disable_data_fsync && dsk.data_fd != dsk.meta_fd && dsk.data_fd != dsk.journal_fd)
|
||||
{
|
||||
// fsync data
|
||||
io_uring_sqe *sqe = get_sqe();
|
||||
@@ -115,7 +109,6 @@ int blockstore_impl_t::do_sync(blockstore_op_t *op, int base_state)
|
||||
PRIV(op)->lsn = heap->get_completed_lsn();
|
||||
if (!submit_fsyncs(PRIV(op)->pending_ops))
|
||||
{
|
||||
PRIV(op)->lsn = 0;
|
||||
PRIV(op)->wait_detail = 1;
|
||||
PRIV(op)->wait_for = WAIT_SQE;
|
||||
return 0;
|
||||
|
||||
@@ -180,16 +180,13 @@ enospc:
|
||||
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
|
||||
assert(loc+op->offset+op->len <= dsk.block_count*dsk.data_block_size);
|
||||
io_uring_prep_writev(sqe, dsk.data_fd, &data->iov, 1, dsk.data_offset + loc + op->offset);
|
||||
if (!dsk.disable_data_fsync)
|
||||
{
|
||||
// use PRIV->lsn for fsync_data_id
|
||||
PRIV(op)->lsn = ++data_fsync_next;
|
||||
data_fsyncs.push_back(false);
|
||||
}
|
||||
PRIV(op)->pending_ops++;
|
||||
write_iodepth++;
|
||||
if (PRIV(op)->write_type == BS_HEAP_BIG_WRITE)
|
||||
{
|
||||
PRIV(op)->op_state = 1;
|
||||
inflight_big++;
|
||||
}
|
||||
else
|
||||
PRIV(op)->op_state = 3;
|
||||
}
|
||||
@@ -300,6 +297,8 @@ again:
|
||||
goto resume_10;
|
||||
else if (op_state == 11)
|
||||
goto resume_11;
|
||||
else if (op_state == 12)
|
||||
goto resume_12;
|
||||
else
|
||||
{
|
||||
// In progress
|
||||
@@ -318,44 +317,38 @@ again:
|
||||
resume_2:
|
||||
// We must fsync all big writes to avoid complex write workflows
|
||||
// It's OK for all HDDs and for server SSDs, but slightly worse for desktop SSDs
|
||||
inflight_big--;
|
||||
if (!dsk.disable_data_fsync)
|
||||
{
|
||||
// Mark our data write as completed and advance data_fsync_cur
|
||||
data_fsyncs[PRIV(op)->lsn - data_fsync_cur - 1] = true;
|
||||
while (data_fsyncs.size() > 0 && data_fsyncs.front())
|
||||
{
|
||||
data_fsyncs.pop_front();
|
||||
data_fsync_cur++;
|
||||
}
|
||||
PRIV(op)->op_state = 11;
|
||||
// Then wait for all other data writes currently in progress to do less fsync calls
|
||||
// I.e. to fsync data in batches
|
||||
PRIV(op)->lsn = data_fsync_cur + data_fsyncs.size();
|
||||
// fsync data in a batch
|
||||
resume_11:
|
||||
if (data_fsync_cur < PRIV(op)->lsn)
|
||||
if (inflight_big > 0)
|
||||
{
|
||||
PRIV(op)->op_state = 11;
|
||||
return 1;
|
||||
}
|
||||
if (PRIV(op)->lsn > data_fsync_sent)
|
||||
if (fsyncing_data)
|
||||
{
|
||||
BS_SUBMIT_GET_SQE(sqe, data);
|
||||
io_uring_prep_fsync(sqe, dsk.data_fd, IORING_FSYNC_DATASYNC);
|
||||
data->iov = { 0 };
|
||||
data->callback = [this, op, fs = data_fsync_cur](ring_data_t *data)
|
||||
resume_12:
|
||||
if (fsyncing_data)
|
||||
{
|
||||
if (fs > data_fsync_done)
|
||||
{
|
||||
data_fsync_done = fs;
|
||||
ringloop->wakeup();
|
||||
}
|
||||
};
|
||||
data_fsync_sent = data_fsync_cur;
|
||||
PRIV(op)->op_state = 12;
|
||||
return 1;
|
||||
}
|
||||
goto resume_4;
|
||||
}
|
||||
if (PRIV(op)->lsn > data_fsync_done)
|
||||
fsyncing_data = true;
|
||||
BS_SUBMIT_GET_SQE(sqe, data);
|
||||
io_uring_prep_fsync(sqe, dsk.data_fd, IORING_FSYNC_DATASYNC);
|
||||
data->iov = { 0 };
|
||||
data->callback = [this, op](ring_data_t *data)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
PRIV(op)->lsn = 0;
|
||||
fsyncing_data = false;
|
||||
handle_write_event(data, op);
|
||||
};
|
||||
PRIV(op)->pending_ops++;
|
||||
PRIV(op)->op_state = 3;
|
||||
return 1;
|
||||
}
|
||||
resume_4:
|
||||
{
|
||||
|
||||
@@ -12,7 +12,7 @@ multilist_alloc_t::multilist_alloc_t(uint32_t count, uint32_t maxn):
|
||||
count(count), maxn(maxn)
|
||||
{
|
||||
// not-so-memory-efficient: 16 MB memory per 1 GB buffer space, but buffer spaces are small, so OK
|
||||
assert(count > 1 && count < 0x80000000 && count >= maxn);
|
||||
assert(count > 1 && count < 0x80000000);
|
||||
sizes.resize(count);
|
||||
nexts.resize(count); // nexts[i] = 0 -> area is used; nexts[i] = 1 -> no next; nexts[i] >= 2 -> next item
|
||||
prevs.resize(count);
|
||||
@@ -171,7 +171,7 @@ void multilist_alloc_t::print()
|
||||
printf("\n");
|
||||
}
|
||||
|
||||
bool multilist_alloc_t::use(uint32_t pos, uint32_t size)
|
||||
void multilist_alloc_t::use(uint32_t pos, uint32_t size)
|
||||
{
|
||||
assert(pos+size <= count && size > 0);
|
||||
if (sizes[pos] <= 0)
|
||||
@@ -182,8 +182,7 @@ bool multilist_alloc_t::use(uint32_t pos, uint32_t size)
|
||||
else
|
||||
while (start > 0 && !sizes[start])
|
||||
start--;
|
||||
if (sizes[start] < size+(pos-start))
|
||||
return false;
|
||||
assert(sizes[start] >= size);
|
||||
use_full(start);
|
||||
uint32_t full = sizes[start];
|
||||
sizes[pos-1] = -pos+start;
|
||||
@@ -200,8 +199,7 @@ bool multilist_alloc_t::use(uint32_t pos, uint32_t size)
|
||||
}
|
||||
else
|
||||
{
|
||||
if (sizes[pos] < size)
|
||||
return false;
|
||||
assert(sizes[pos] >= size);
|
||||
use_full(pos);
|
||||
if (sizes[pos] > size)
|
||||
{
|
||||
@@ -216,13 +214,12 @@ bool multilist_alloc_t::use(uint32_t pos, uint32_t size)
|
||||
#ifdef MULTILIST_TRACE
|
||||
print();
|
||||
#endif
|
||||
return true;
|
||||
}
|
||||
|
||||
void multilist_alloc_t::use_full(uint32_t pos)
|
||||
{
|
||||
uint32_t prevsize = sizes[pos];
|
||||
assert(prevsize > 0);
|
||||
assert(prevsize);
|
||||
assert(nexts[pos]);
|
||||
uint32_t pi = (prevsize < maxn ? prevsize : maxn)-1;
|
||||
if (heads[pi] == pos+1)
|
||||
|
||||
@@ -17,7 +17,7 @@ struct multilist_alloc_t
|
||||
bool is_free(uint32_t pos);
|
||||
uint32_t find(uint32_t size);
|
||||
void use_full(uint32_t pos);
|
||||
bool use(uint32_t pos, uint32_t size);
|
||||
void use(uint32_t pos, uint32_t size);
|
||||
void do_free(uint32_t pos);
|
||||
void free(uint32_t pos);
|
||||
void verify();
|
||||
|
||||
+22
-53
@@ -71,11 +71,6 @@ bool journal_flusher_t::is_active()
|
||||
return active_flushers > 0 || dequeuing;
|
||||
}
|
||||
|
||||
size_t journal_flusher_t::get_queue_size()
|
||||
{
|
||||
return flush_queue.size();
|
||||
}
|
||||
|
||||
void journal_flusher_t::loop()
|
||||
{
|
||||
target_flusher_count = bs->write_iodepth*2;
|
||||
@@ -389,7 +384,6 @@ stop_flusher:
|
||||
wait_state = 0;
|
||||
return true;
|
||||
}
|
||||
copy_count = 0;
|
||||
try_trim = true;
|
||||
cur.oid = flusher->flush_queue.front();
|
||||
cur.version = flusher->flush_versions[cur.oid];
|
||||
@@ -517,31 +511,6 @@ resume_2:
|
||||
{
|
||||
uo_it->second.was_changed = true;
|
||||
}
|
||||
if (!bs->journal.inmemory)
|
||||
{
|
||||
// Verify journaled data checksums (but not COALESCED)
|
||||
for (it = v.begin(); it != v.end(); it++)
|
||||
{
|
||||
if (it->copy_flags == COPY_BUF_JOURNAL)
|
||||
{
|
||||
iovec iov = { .iov_base = it->buf, .iov_len = it->len };
|
||||
bs->verify_journal_checksums(
|
||||
it->csum_buf, it->offset, &iov, 1,
|
||||
[&](uint32_t bad_block, uint32_t calc_csum, uint32_t stored_csum)
|
||||
{
|
||||
printf(
|
||||
"Checksum mismatch in object %jx:%jx v%ju in journal at 0x%jx, checksum block #%u: got %08x, expected %08x\n",
|
||||
cur.oid.inode, cur.oid.stripe, cur.version, it->disk_offset,
|
||||
bad_block / bs->dsk.csum_block_size, calc_csum, stored_csum
|
||||
);
|
||||
bad_block += it->offset;
|
||||
assert(!(bad_block % bs->dsk.csum_block_size) && bad_block < bs->dsk.data_block_size);
|
||||
mangle_csum_blocks.insert(bad_block);
|
||||
}
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Submit data writes
|
||||
for (it = v.begin(); it != v.end(); it++)
|
||||
@@ -665,7 +634,6 @@ resume_2:
|
||||
}
|
||||
// All done
|
||||
flusher->active_flushers--;
|
||||
copy_count = 0; // used by is_mutated()...
|
||||
wait_state = 0;
|
||||
goto resume_0;
|
||||
}
|
||||
@@ -847,21 +815,35 @@ bool journal_flusher_co::clear_incomplete_csum_block_bits(int wait_base)
|
||||
bs->verify_padded_checksums(new_clean_bitmap, new_clean_bitmap + 2*bs->dsk.clean_entry_bitmap_size,
|
||||
v[i].offset, &iov, 1, [&](uint32_t bad_block, uint32_t calc_csum, uint32_t stored_csum)
|
||||
{
|
||||
printf("Checksum mismatch in object %jx:%jx v%ju in data area at offset 0x%jx+0x%x during flush: got %08x, expected %08x\n",
|
||||
printf("Checksum mismatch in object %jx:%jx v%ju in data area at offset 0x%jx+0x%x: got %08x, expected %08x\n",
|
||||
cur.oid.inode, cur.oid.stripe, old_clean_ver, old_clean_loc, bad_block, calc_csum, stored_csum);
|
||||
assert(!(bad_block % bs->dsk.csum_block_size) && bad_block < bs->dsk.data_block_size);
|
||||
mangle_csum_blocks.insert(bad_block);
|
||||
for (uint32_t j = 0; j < bs->dsk.csum_block_size; j += bs->dsk.bitmap_granularity)
|
||||
{
|
||||
// Simplest method of mangling: flip one byte in every sector
|
||||
((uint8_t*)v[i].buf)[j+bad_block-v[i].offset] ^= 0xff;
|
||||
}
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
bs->verify_journal_checksums(v[i].csum_buf, v[i].offset, &iov, 1, [&](uint32_t bad_block, uint32_t calc_csum, uint32_t stored_csum)
|
||||
{
|
||||
printf("Checksum mismatch in object %jx:%jx v%ju in journal at offset 0x%jx+0x%x (block offset 0x%jx) during flush: got %08x, expected %08x\n",
|
||||
printf("Checksum mismatch in object %jx:%jx v%ju in journal at offset 0x%jx+0x%x (block offset 0x%jx): got %08x, expected %08x\n",
|
||||
cur.oid.inode, cur.oid.stripe, old_clean_ver,
|
||||
v[i].disk_offset, bad_block, v[i].offset, calc_csum, stored_csum);
|
||||
assert(!(bad_block % bs->dsk.csum_block_size) && bad_block < bs->dsk.data_block_size);
|
||||
mangle_csum_blocks.insert(bad_block);
|
||||
bad_block += (v[i].offset/bs->dsk.csum_block_size) * bs->dsk.csum_block_size;
|
||||
uint32_t bad_block_end = bad_block + bs->dsk.csum_block_size + (v[i].offset/bs->dsk.csum_block_size) * bs->dsk.csum_block_size;
|
||||
if (bad_block < v[i].offset)
|
||||
bad_block = v[i].offset;
|
||||
if (bad_block_end > v[i].offset+v[i].len)
|
||||
bad_block_end = v[i].offset+v[i].len;
|
||||
bad_block -= v[i].offset;
|
||||
bad_block_end -= v[i].offset;
|
||||
for (uint32_t j = bad_block; j < bad_block_end; j += bs->dsk.bitmap_granularity)
|
||||
{
|
||||
// Simplest method of mangling: flip one byte in every sector
|
||||
((uint8_t*)v[i].buf)[j] ^= 0xff;
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -970,11 +952,6 @@ void journal_flusher_co::calc_block_checksums(uint32_t *new_data_csums, bool ski
|
||||
}
|
||||
// `v` should contain aligned items, possibly split into pieces
|
||||
assert(!block_done);
|
||||
for (uint32_t mangle_block: mangle_csum_blocks)
|
||||
{
|
||||
// Flip 1 bit
|
||||
new_data_csums[mangle_block / bs->dsk.csum_block_size] ^= 1;
|
||||
}
|
||||
}
|
||||
|
||||
void journal_flusher_co::scan_dirty()
|
||||
@@ -1111,8 +1088,7 @@ void journal_flusher_co::scan_dirty()
|
||||
last--;
|
||||
read_to_fill_incomplete = bs->fill_partial_checksum_blocks(
|
||||
v, fulfilled, bmp_ptr, NULL, false, NULL, v[0].offset/bs->dsk.csum_block_size * bs->dsk.csum_block_size,
|
||||
((v[last].offset+v[last].len-1) / bs->dsk.csum_block_size + 1) * bs->dsk.csum_block_size,
|
||||
0, bs->dsk.data_block_size
|
||||
((v[last].offset+v[last].len-1) / bs->dsk.csum_block_size + 1) * bs->dsk.csum_block_size
|
||||
);
|
||||
}
|
||||
else if (fill_incomplete && clean_init_bitmap)
|
||||
@@ -1142,7 +1118,6 @@ bool journal_flusher_co::read_dirty(int wait_base)
|
||||
if (wait_state == wait_base) goto resume_0;
|
||||
else if (wait_state == wait_base+1) goto resume_1;
|
||||
wait_count = wait_journal_count = 0;
|
||||
mangle_csum_blocks.clear();
|
||||
if (bs->journal.inmemory && !read_to_fill_incomplete)
|
||||
{
|
||||
// Happy path: nothing to read :)
|
||||
@@ -1374,7 +1349,7 @@ bool journal_flusher_co::fsync_batch(bool fsync_meta, int wait_base)
|
||||
cur_sync->ready_count++;
|
||||
flusher->syncing_flushers++;
|
||||
resume_1:
|
||||
if (cur_sync->state == 0)
|
||||
if (!cur_sync->state)
|
||||
{
|
||||
if (flusher->syncing_flushers >= flusher->active_flushers || !flusher->flush_queue.size())
|
||||
{
|
||||
@@ -1402,12 +1377,6 @@ bool journal_flusher_co::fsync_batch(bool fsync_meta, int wait_base)
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if (cur_sync->state == 1)
|
||||
{
|
||||
// Wait for fsync completion
|
||||
wait_state = wait_base+1;
|
||||
return false;
|
||||
}
|
||||
flusher->syncing_flushers--;
|
||||
cur_sync->ready_count--;
|
||||
if (cur_sync->ready_count == 0)
|
||||
|
||||
@@ -66,7 +66,6 @@ class journal_flusher_co
|
||||
uint64_t clean_bitmap_offset, clean_bitmap_len;
|
||||
uint8_t *clean_init_dyn_ptr;
|
||||
uint8_t *new_clean_bitmap;
|
||||
std::unordered_set<uint32_t> mangle_csum_blocks;
|
||||
|
||||
uint64_t new_trim_pos;
|
||||
|
||||
@@ -124,7 +123,6 @@ public:
|
||||
void loop();
|
||||
bool is_trim_wanted() { return trim_wanted; }
|
||||
bool is_active();
|
||||
size_t get_queue_size();
|
||||
void mark_trim_possible();
|
||||
void request_trim();
|
||||
void release_trim();
|
||||
|
||||
@@ -6,12 +6,11 @@
|
||||
|
||||
namespace v1 {
|
||||
|
||||
blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd, bool mock_mode)
|
||||
blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd)
|
||||
{
|
||||
assert(sizeof(blockstore_op_private_t) <= BS_OP_PRIVATE_DATA_SIZE);
|
||||
this->tfd = tfd;
|
||||
this->ringloop = ringloop;
|
||||
dsk.mock_mode = mock_mode;
|
||||
ring_consumer.loop = [this]() { loop(); };
|
||||
ringloop->register_consumer(&ring_consumer);
|
||||
initialized = 0;
|
||||
@@ -36,11 +35,6 @@ blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_i *
|
||||
|
||||
blockstore_impl_t::~blockstore_impl_t()
|
||||
{
|
||||
for (auto& obj: dirty_db)
|
||||
{
|
||||
if (obj.second.dyn_data)
|
||||
free(obj.second.dyn_data);
|
||||
}
|
||||
delete data_alloc;
|
||||
delete flusher;
|
||||
if (zero_object)
|
||||
|
||||
@@ -30,8 +30,6 @@
|
||||
|
||||
//#define BLOCKSTORE_DEBUG
|
||||
|
||||
struct bs_test_t;
|
||||
|
||||
namespace v1 {
|
||||
|
||||
#include "journal.h"
|
||||
@@ -98,7 +96,7 @@ struct blockstore_op_private_t
|
||||
int op_state;
|
||||
|
||||
// Read
|
||||
uint64_t clean_loc_used;
|
||||
uint64_t clean_block_used;
|
||||
std::vector<copy_buffer_t> read_vec;
|
||||
|
||||
// Sync, write
|
||||
@@ -124,7 +122,6 @@ typedef uint64_t pool_pg_id_t;
|
||||
|
||||
class blockstore_impl_t: public blockstore_i
|
||||
{
|
||||
friend struct ::bs_test_t;
|
||||
blockstore_disk_t dsk;
|
||||
|
||||
/******* OPTIONS *******/
|
||||
@@ -223,7 +220,6 @@ class blockstore_impl_t: public blockstore_i
|
||||
|
||||
// Read
|
||||
int dequeue_read(blockstore_op_t *read_op);
|
||||
void release_clean(blockstore_op_t *op);
|
||||
void find_holes(std::vector<copy_buffer_t> & read_vec, uint32_t item_start, uint32_t item_end,
|
||||
std::function<int(int, bool, uint32_t, uint32_t)> callback);
|
||||
int fulfill_read(blockstore_op_t *read_op,
|
||||
@@ -234,8 +230,7 @@ class blockstore_impl_t: public blockstore_i
|
||||
uint8_t *clean_entry_bitmap, int *dyn_data,
|
||||
uint32_t item_start, uint32_t item_end, uint64_t clean_loc, uint64_t clean_ver);
|
||||
int fill_partial_checksum_blocks(std::vector<copy_buffer_t> & rv, uint64_t & fulfilled,
|
||||
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf,
|
||||
uint32_t read_offset, uint32_t read_end, uint32_t item_start, uint32_t item_end);
|
||||
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf, uint64_t read_offset, uint64_t read_end);
|
||||
int pad_journal_read(std::vector<copy_buffer_t> & rv, copy_buffer_t & cp,
|
||||
uint64_t dirty_offset, uint64_t dirty_end, uint64_t dirty_loc, uint8_t *csum_ptr, int *dyn_data,
|
||||
uint64_t offset, uint64_t submit_len, uint64_t & blk_begin, uint64_t & blk_end, uint8_t* & blk_buf);
|
||||
@@ -286,7 +281,7 @@ class blockstore_impl_t: public blockstore_i
|
||||
|
||||
public:
|
||||
|
||||
blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd, bool mock_mode = false);
|
||||
blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd);
|
||||
~blockstore_impl_t();
|
||||
|
||||
void parse_config(blockstore_config_t & config);
|
||||
|
||||
+68
-83
@@ -1,7 +1,6 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#include "str_util.h"
|
||||
#include "impl.h"
|
||||
#include "internal.h"
|
||||
|
||||
@@ -31,15 +30,14 @@ blockstore_init_meta::blockstore_init_meta(blockstore_impl_t *bs)
|
||||
this->bs = bs;
|
||||
}
|
||||
|
||||
void blockstore_init_meta::handle_event(ring_data_t *data, int buf_num, const char *op)
|
||||
void blockstore_init_meta::handle_event(ring_data_t *data, int buf_num)
|
||||
{
|
||||
if (data->res != data->iov.iov_len)
|
||||
if (data->res < 0)
|
||||
{
|
||||
throw std::runtime_error(strprintf(
|
||||
"%s failed at offset %ju: got %s (code %d), but expected %zu",
|
||||
op, (buf_num >= 0 ? bufs[buf_num].offset : last_read_offset), strerror(-data->res),
|
||||
data->res, data->iov.iov_len
|
||||
));
|
||||
throw std::runtime_error(
|
||||
std::string("read metadata failed at offset ") + std::to_string(buf_num >= 0 ? bufs[buf_num].offset : last_read_offset) +
|
||||
std::string(": ") + strerror(-data->res)
|
||||
);
|
||||
}
|
||||
if (buf_num >= 0)
|
||||
{
|
||||
@@ -67,11 +65,10 @@ int blockstore_init_meta::loop()
|
||||
if (!metadata_buffer)
|
||||
throw std::runtime_error("Failed to allocate metadata read buffer");
|
||||
// Read superblock
|
||||
hdr = (blockstore_meta_header_v2_t *)memalign_or_die(MEM_ALIGNMENT, bs->dsk.meta_block_size);
|
||||
GET_SQE();
|
||||
last_read_offset = 0;
|
||||
data->iov = { hdr, (size_t)bs->dsk.meta_block_size };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "read metadata header"); };
|
||||
data->iov = { metadata_buffer, (size_t)bs->dsk.meta_block_size };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
||||
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
||||
bs->ringloop->submit();
|
||||
submitted++;
|
||||
@@ -81,8 +78,24 @@ resume_1:
|
||||
wait_state = 1;
|
||||
return 1;
|
||||
}
|
||||
if (iszero((uint64_t*)hdr, bs->dsk.meta_block_size / sizeof(uint64_t)))
|
||||
if (iszero((uint64_t*)metadata_buffer, bs->dsk.meta_block_size / sizeof(uint64_t)))
|
||||
{
|
||||
{
|
||||
blockstore_meta_header_v2_t *hdr = (blockstore_meta_header_v2_t *)metadata_buffer;
|
||||
hdr->zero = 0;
|
||||
hdr->magic = BLOCKSTORE_META_MAGIC_V1;
|
||||
hdr->version = bs->dsk.meta_format;
|
||||
hdr->meta_block_size = bs->dsk.meta_block_size;
|
||||
hdr->data_block_size = bs->dsk.data_block_size;
|
||||
hdr->bitmap_granularity = bs->dsk.bitmap_granularity;
|
||||
if (bs->dsk.meta_format >= BLOCKSTORE_META_FORMAT_V2)
|
||||
{
|
||||
hdr->data_csum_type = bs->dsk.data_csum_type;
|
||||
hdr->csum_block_size = bs->dsk.csum_block_size;
|
||||
hdr->header_csum = 0;
|
||||
hdr->header_csum = crc32c(0, hdr, sizeof(*hdr));
|
||||
}
|
||||
}
|
||||
if (bs->readonly)
|
||||
{
|
||||
printf("Skipping metadata initialization because blockstore is readonly\n");
|
||||
@@ -90,11 +103,25 @@ resume_1:
|
||||
else
|
||||
{
|
||||
printf("Initializing metadata area\n");
|
||||
GET_SQE();
|
||||
last_read_offset = 0;
|
||||
data->iov = (struct iovec){ metadata_buffer, (size_t)bs->dsk.meta_block_size };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
||||
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
||||
bs->ringloop->submit();
|
||||
submitted++;
|
||||
resume_3:
|
||||
if (submitted > 0)
|
||||
{
|
||||
wait_state = 3;
|
||||
return 1;
|
||||
}
|
||||
zero_on_init = true;
|
||||
}
|
||||
zero_on_init = true;
|
||||
}
|
||||
else
|
||||
{
|
||||
blockstore_meta_header_v2_t *hdr = (blockstore_meta_header_v2_t *)metadata_buffer;
|
||||
if (hdr->zero != 0 || hdr->magic != BLOCKSTORE_META_MAGIC_V1 || hdr->version < BLOCKSTORE_META_FORMAT_V1)
|
||||
{
|
||||
printf(
|
||||
@@ -196,15 +223,12 @@ resume_2:
|
||||
GET_SQE();
|
||||
assert(bufs[i].size <= 0x7fffffff);
|
||||
data->iov = { bufs[i].buf, (size_t)bufs[i].size };
|
||||
data->callback = [this, i](ring_data_t *data) { handle_event(data, i); };
|
||||
if (!zero_on_init)
|
||||
{
|
||||
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "read metadata"); };
|
||||
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
|
||||
}
|
||||
else
|
||||
{
|
||||
// Fill metadata with zeroes
|
||||
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "clear metadata"); };
|
||||
memset(data->iov.iov_base, 0, data->iov.iov_len);
|
||||
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
|
||||
}
|
||||
@@ -232,7 +256,7 @@ resume_2:
|
||||
GET_SQE();
|
||||
assert(bufs[i].size <= 0x7fffffff);
|
||||
data->iov = { bufs[i].buf, (size_t)bufs[i].size };
|
||||
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "write metadata"); };
|
||||
data->callback = [this, i](ring_data_t *data) { handle_event(data, i); };
|
||||
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
|
||||
bs->ringloop->submit();
|
||||
bufs[i].state = INIT_META_WRITING;
|
||||
@@ -261,7 +285,7 @@ resume_2:
|
||||
GET_SQE();
|
||||
last_read_offset = (1+next_offset)*bs->dsk.meta_block_size;
|
||||
data->iov = { metadata_buffer, (size_t)bs->dsk.meta_block_size };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "read metadata"); };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
||||
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + (1+next_offset)*bs->dsk.meta_block_size);
|
||||
bs->ringloop->submit();
|
||||
submitted++;
|
||||
@@ -278,7 +302,7 @@ resume_5:
|
||||
}
|
||||
GET_SQE();
|
||||
data->iov = { metadata_buffer, (size_t)bs->dsk.meta_block_size };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "write metadata"); };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
||||
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + (1+next_offset)*bs->dsk.meta_block_size);
|
||||
bs->ringloop->submit();
|
||||
submitted++;
|
||||
@@ -293,64 +317,27 @@ resume_6:
|
||||
}
|
||||
// metadata read finished
|
||||
printf("Metadata entries loaded: %ju, free blocks: %ju / %ju\n", entries_loaded, bs->data_alloc->get_free_count(), bs->dsk.block_count);
|
||||
if (zero_on_init && !bs->readonly)
|
||||
{
|
||||
do_fsync:
|
||||
if (!bs->disable_meta_fsync)
|
||||
{
|
||||
GET_SQE();
|
||||
io_uring_prep_fsync(sqe, bs->dsk.meta_fd, IORING_FSYNC_DATASYNC);
|
||||
last_read_offset = 0;
|
||||
data->iov = { 0 };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "fsync metadata"); };
|
||||
submitted++;
|
||||
bs->ringloop->submit();
|
||||
resume_4:
|
||||
if (submitted > 0)
|
||||
{
|
||||
wait_state = 4;
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
if (!header_written)
|
||||
{
|
||||
GET_SQE();
|
||||
hdr->zero = 0;
|
||||
hdr->magic = BLOCKSTORE_META_MAGIC_V1;
|
||||
hdr->version = bs->dsk.meta_format;
|
||||
hdr->meta_block_size = bs->dsk.meta_block_size;
|
||||
hdr->data_block_size = bs->dsk.data_block_size;
|
||||
hdr->bitmap_granularity = bs->dsk.bitmap_granularity;
|
||||
if (bs->dsk.meta_format >= BLOCKSTORE_META_FORMAT_V2)
|
||||
{
|
||||
hdr->data_csum_type = bs->dsk.data_csum_type;
|
||||
hdr->csum_block_size = bs->dsk.csum_block_size;
|
||||
hdr->header_csum = 0;
|
||||
hdr->header_csum = crc32c(0, hdr, sizeof(*hdr));
|
||||
}
|
||||
header_written = true;
|
||||
last_read_offset = 0;
|
||||
data->iov = (struct iovec){ hdr, (size_t)bs->dsk.meta_block_size };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "write metadata header"); };
|
||||
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
|
||||
bs->ringloop->submit();
|
||||
submitted++;
|
||||
resume_3:
|
||||
if (submitted > 0)
|
||||
{
|
||||
wait_state = 3;
|
||||
return 1;
|
||||
}
|
||||
goto do_fsync;
|
||||
}
|
||||
}
|
||||
if (!bs->inmemory_meta)
|
||||
{
|
||||
free(metadata_buffer);
|
||||
metadata_buffer = NULL;
|
||||
}
|
||||
free(hdr);
|
||||
hdr = NULL;
|
||||
if (zero_on_init && !bs->disable_meta_fsync)
|
||||
{
|
||||
GET_SQE();
|
||||
io_uring_prep_fsync(sqe, bs->dsk.meta_fd, IORING_FSYNC_DATASYNC);
|
||||
last_read_offset = 0;
|
||||
data->iov = { 0 };
|
||||
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
|
||||
submitted++;
|
||||
bs->ringloop->submit();
|
||||
resume_4:
|
||||
if (submitted > 0)
|
||||
{
|
||||
wait_state = 4;
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -358,8 +345,6 @@ bool blockstore_init_meta::handle_meta_block(uint8_t *buf, uint64_t entries_per_
|
||||
{
|
||||
bool updated = false;
|
||||
uint64_t max_i = entries_per_block;
|
||||
if (done_cnt > bs->dsk.block_count)
|
||||
return false;
|
||||
if (max_i > bs->dsk.block_count-done_cnt)
|
||||
max_i = bs->dsk.block_count-done_cnt;
|
||||
for (uint64_t i = 0; i < max_i; i++)
|
||||
@@ -470,21 +455,21 @@ blockstore_init_journal::blockstore_init_journal(blockstore_impl_t *bs)
|
||||
};
|
||||
}
|
||||
|
||||
void blockstore_init_journal::handle_event(ring_data_t *data)
|
||||
void blockstore_init_journal::handle_event(ring_data_t *data1)
|
||||
{
|
||||
if (data->res != data->iov.iov_len)
|
||||
if (data1->res <= 0)
|
||||
{
|
||||
throw std::runtime_error(strprintf(
|
||||
"read journal failed at offset %ju: got %s (code %d), but expected %zu",
|
||||
journal_pos, strerror(-data->res), data->res, data->iov.iov_len
|
||||
));
|
||||
throw std::runtime_error(
|
||||
std::string("read journal failed at offset ") + std::to_string(journal_pos) +
|
||||
std::string(": ") + strerror(-data1->res)
|
||||
);
|
||||
}
|
||||
done.push_back({
|
||||
.buf = submitted_buf,
|
||||
.pos = journal_pos,
|
||||
.len = (uint64_t)data->res,
|
||||
.len = (uint64_t)data1->res,
|
||||
});
|
||||
journal_pos += data->res;
|
||||
journal_pos += data1->res;
|
||||
if (journal_pos >= bs->journal.len)
|
||||
{
|
||||
// Continue from the beginning
|
||||
|
||||
@@ -16,9 +16,7 @@ class blockstore_init_meta
|
||||
blockstore_impl_t *bs;
|
||||
int wait_state = 0;
|
||||
bool zero_on_init = false;
|
||||
bool header_written = false;
|
||||
void *metadata_buffer = NULL;
|
||||
blockstore_meta_header_v2_t *hdr = NULL;
|
||||
blockstore_init_meta_buf bufs[2] = {};
|
||||
int submitted = 0;
|
||||
struct io_uring_sqe *sqe;
|
||||
@@ -31,7 +29,7 @@ class blockstore_init_meta
|
||||
int i = 0, j = 0;
|
||||
std::vector<uint64_t> entries_to_zero;
|
||||
bool handle_meta_block(uint8_t *buf, uint64_t count, uint64_t done_cnt);
|
||||
void handle_event(ring_data_t *data, int buf_num, const char *op);
|
||||
void handle_event(ring_data_t *data, int buf_num);
|
||||
public:
|
||||
blockstore_init_meta(blockstore_impl_t *bs);
|
||||
int loop();
|
||||
|
||||
+54
-145
@@ -101,8 +101,8 @@ int blockstore_impl_t::fulfill_read(blockstore_op_t *read_op,
|
||||
.copy_flags = COPY_BUF_JOURNAL|COPY_BUF_CSUM_FILL,
|
||||
.offset = blk_begin,
|
||||
.len = blk_end-blk_begin,
|
||||
.csum_buf = (!csum ? NULL : (csum + (blk_begin/dsk.csum_block_size -
|
||||
item_start/dsk.csum_block_size) * (dsk.data_csum_type & 0xFF))),
|
||||
.csum_buf = (csum + (blk_begin/dsk.csum_block_size -
|
||||
item_start/dsk.csum_block_size) * (dsk.data_csum_type & 0xFF)),
|
||||
.dyn_data = dyn_data,
|
||||
});
|
||||
if (dyn_data)
|
||||
@@ -134,7 +134,7 @@ int blockstore_impl_t::fulfill_read(blockstore_op_t *read_op,
|
||||
// If we don't track it then we may IN THEORY read another object's data:
|
||||
// submit read -> remove the object -> flush remove -> overwrite with another object -> finish read
|
||||
// Very improbable, but possible
|
||||
PRIV(read_op)->clean_loc_used = UINT64_MAX;
|
||||
PRIV(read_op)->clean_block_used = 1;
|
||||
}
|
||||
rv.insert(rv.begin() + pos, el);
|
||||
fulfilled += el.len;
|
||||
@@ -167,8 +167,7 @@ uint8_t* blockstore_impl_t::get_clean_entry_bitmap(uint64_t block_loc, int offse
|
||||
}
|
||||
|
||||
int blockstore_impl_t::fill_partial_checksum_blocks(std::vector<copy_buffer_t> & rv, uint64_t & fulfilled,
|
||||
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf,
|
||||
uint32_t read_offset, uint32_t read_end, uint32_t item_start, uint32_t item_end)
|
||||
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf, uint64_t read_offset, uint64_t read_end)
|
||||
{
|
||||
if (read_end == read_offset)
|
||||
return 0;
|
||||
@@ -176,38 +175,10 @@ int blockstore_impl_t::fill_partial_checksum_blocks(std::vector<copy_buffer_t> &
|
||||
read_buf -= read_offset;
|
||||
uint32_t last_block = (read_end-1)/dsk.csum_block_size;
|
||||
uint32_t start_block = read_offset/dsk.csum_block_size;
|
||||
uint32_t item_start_block = item_start/dsk.csum_block_size;
|
||||
uint32_t end_block = 0;
|
||||
auto zero_range = [&](int pos, bool alloc, uint32_t cur_start, uint32_t cur_end)
|
||||
{
|
||||
if (alloc)
|
||||
return 0;
|
||||
copy_buffer_t el = {
|
||||
.copy_flags = COPY_BUF_ZERO,
|
||||
.offset = cur_start,
|
||||
.len = cur_end-cur_start,
|
||||
};
|
||||
rv.insert(rv.begin() + pos, el);
|
||||
if (read_buf)
|
||||
memset(read_buf + el.offset - read_offset, 0, el.len);
|
||||
fulfilled += el.len;
|
||||
return 1;
|
||||
};
|
||||
if (read_offset < item_start)
|
||||
{
|
||||
// Zero-fill the beginning
|
||||
find_holes(rv, read_offset, item_start, zero_range);
|
||||
read_offset = item_start;
|
||||
}
|
||||
if (read_end > item_end)
|
||||
{
|
||||
// Zero-fill the end
|
||||
find_holes(rv, item_end, read_end, zero_range);
|
||||
read_end = item_end;
|
||||
}
|
||||
while (start_block <= last_block)
|
||||
{
|
||||
if (read_range_fulfilled(rv, fulfilled, read_buf, from_journal ? NULL : clean_entry_bitmap,
|
||||
if (read_range_fulfilled(rv, fulfilled, read_buf, clean_entry_bitmap,
|
||||
start_block*dsk.csum_block_size < read_offset ? read_offset : start_block*dsk.csum_block_size,
|
||||
(start_block+1)*dsk.csum_block_size > read_end ? read_end : (start_block+1)*dsk.csum_block_size))
|
||||
{
|
||||
@@ -219,7 +190,7 @@ int blockstore_impl_t::fill_partial_checksum_blocks(std::vector<copy_buffer_t> &
|
||||
// Find a sequence of checksum blocks required to be read
|
||||
end_block = start_block;
|
||||
while ((end_block+1)*dsk.csum_block_size < read_end &&
|
||||
!read_range_fulfilled(rv, fulfilled, read_buf, from_journal ? NULL : clean_entry_bitmap,
|
||||
!read_range_fulfilled(rv, fulfilled, read_buf, clean_entry_bitmap,
|
||||
(end_block+1)*dsk.csum_block_size < read_offset ? read_offset : (end_block+1)*dsk.csum_block_size,
|
||||
(end_block+2)*dsk.csum_block_size > read_end ? read_end : (end_block+2)*dsk.csum_block_size))
|
||||
{
|
||||
@@ -231,10 +202,8 @@ int blockstore_impl_t::fill_partial_checksum_blocks(std::vector<copy_buffer_t> &
|
||||
.copy_flags = COPY_BUF_CSUM_FILL | (from_journal ? COPY_BUF_JOURNALED_BIG : 0),
|
||||
.offset = start_block*dsk.csum_block_size,
|
||||
.len = (end_block-start_block)*dsk.csum_block_size,
|
||||
// save checksum reference if we're reading clean data from the journal
|
||||
.csum_buf = from_journal
|
||||
? clean_entry_bitmap + dsk.clean_entry_bitmap_size + (start_block-item_start_block)*(dsk.data_csum_type & 0xFF)
|
||||
: NULL,
|
||||
// save clean_entry_bitmap if we're reading clean data from the journal
|
||||
.csum_buf = from_journal ? clean_entry_bitmap : NULL,
|
||||
.dyn_data = dyn_data,
|
||||
});
|
||||
if (dyn_data)
|
||||
@@ -257,11 +226,6 @@ bool blockstore_impl_t::read_range_fulfilled(std::vector<copy_buffer_t> & rv, ui
|
||||
{
|
||||
if (alloc)
|
||||
return 0;
|
||||
if (!clean_entry_bitmap)
|
||||
{
|
||||
all_done = false;
|
||||
return 0;
|
||||
}
|
||||
int diff = 0;
|
||||
uint32_t bmp_start = cur_start/dsk.bitmap_granularity;
|
||||
uint32_t bmp_end = cur_end/dsk.bitmap_granularity;
|
||||
@@ -359,7 +323,7 @@ bool blockstore_impl_t::read_checksum_block(blockstore_op_t *op, int rv_pos, uin
|
||||
{
|
||||
iov[n_iov++] = (struct iovec){ (uint8_t*)op->buf+cur_start-op->offset, lim_end-cur_start };
|
||||
rv.insert(rv.begin() + pos, (copy_buffer_t){
|
||||
.copy_flags = COPY_BUF_DATA|COPY_BUF_COALESCED,
|
||||
.copy_flags = COPY_BUF_DATA,
|
||||
.offset = cur_start,
|
||||
.len = lim_end-cur_start,
|
||||
});
|
||||
@@ -397,10 +361,10 @@ bool blockstore_impl_t::read_checksum_block(blockstore_op_t *op, int rv_pos, uin
|
||||
PRIV(op)->pending_ops++;
|
||||
io_uring_prep_readv(sqe, submit_fd, iov + n_pos, n_cur, submit_offset + clean_loc + item_start + d_pos);
|
||||
data->callback = [this, op](ring_data_t *data) { handle_read_event(data, op); };
|
||||
if (n_pos > 0 || n_iov > IOV_MAX)
|
||||
if (n_pos > 0 || n_pos + IOV_MAX < n_iov)
|
||||
{
|
||||
uint32_t d_len = 0;
|
||||
for (int i = 0; i < n_cur; i++)
|
||||
for (int i = 0; i < IOV_MAX; i++)
|
||||
d_len += iov[n_pos+i].iov_len;
|
||||
data->iov.iov_len = d_len;
|
||||
d_pos += d_len;
|
||||
@@ -412,7 +376,7 @@ bool blockstore_impl_t::read_checksum_block(blockstore_op_t *op, int rv_pos, uin
|
||||
{
|
||||
// Reads running parallel to flushes of the same clean block may read
|
||||
// a mixture of old and new data. So we don't verify checksums for such blocks.
|
||||
PRIV(op)->clean_loc_used = UINT64_MAX;
|
||||
PRIV(op)->clean_block_used = 1;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -438,7 +402,7 @@ int blockstore_impl_t::dequeue_read(blockstore_op_t *read_op)
|
||||
}
|
||||
uint64_t fulfilled = 0;
|
||||
PRIV(read_op)->pending_ops = 0;
|
||||
PRIV(read_op)->clean_loc_used = 0;
|
||||
PRIV(read_op)->clean_block_used = 0;
|
||||
auto & rv = PRIV(read_op)->read_vec;
|
||||
uint64_t result_version = 0;
|
||||
if (dirty_found)
|
||||
@@ -551,50 +515,26 @@ int blockstore_impl_t::dequeue_read(blockstore_op_t *read_op)
|
||||
return 2;
|
||||
undo_read:
|
||||
// need to wait. undo added requests, don't dequeue op
|
||||
release_clean(read_op);
|
||||
for (auto & vec: rv)
|
||||
if (dsk.csum_block_size > dsk.bitmap_granularity)
|
||||
{
|
||||
if ((vec.copy_flags & COPY_BUF_CSUM_FILL) && vec.buf)
|
||||
for (auto & vec: rv)
|
||||
{
|
||||
free(vec.buf);
|
||||
vec.buf = NULL;
|
||||
}
|
||||
if (vec.dyn_data && --(*vec.dyn_data) == 0) // refcount
|
||||
{
|
||||
free(vec.dyn_data);
|
||||
vec.dyn_data = NULL;
|
||||
if ((vec.copy_flags & COPY_BUF_CSUM_FILL) && vec.buf)
|
||||
{
|
||||
free(vec.buf);
|
||||
vec.buf = NULL;
|
||||
}
|
||||
if (vec.dyn_data && --(*vec.dyn_data) == 0) // refcount
|
||||
{
|
||||
free(vec.dyn_data);
|
||||
vec.dyn_data = NULL;
|
||||
}
|
||||
}
|
||||
}
|
||||
rv.clear();
|
||||
return 0;
|
||||
}
|
||||
|
||||
void blockstore_impl_t::release_clean(blockstore_op_t *op)
|
||||
{
|
||||
if (PRIV(op)->clean_loc_used == UINT64_MAX)
|
||||
{
|
||||
PRIV(op)->clean_loc_used = 0;
|
||||
}
|
||||
if (PRIV(op)->clean_loc_used)
|
||||
{
|
||||
// Release clean data block
|
||||
auto uo_it = used_clean_objects.find(PRIV(op)->clean_loc_used - 1);
|
||||
if (uo_it != used_clean_objects.end())
|
||||
{
|
||||
uo_it->second.refs--;
|
||||
if (uo_it->second.refs <= 0)
|
||||
{
|
||||
if (uo_it->second.was_freed)
|
||||
{
|
||||
data_alloc->set((PRIV(op)->clean_loc_used - 1) / dsk.data_block_size, false);
|
||||
}
|
||||
used_clean_objects.erase(uo_it);
|
||||
}
|
||||
}
|
||||
PRIV(op)->clean_loc_used = 0;
|
||||
}
|
||||
}
|
||||
|
||||
int blockstore_impl_t::pad_journal_read(std::vector<copy_buffer_t> & rv, copy_buffer_t & cp,
|
||||
// FIXME Passing dirty_entry& would be nicer
|
||||
uint64_t dirty_offset, uint64_t dirty_end, uint64_t dirty_loc, uint8_t *csum_ptr, int *dyn_data,
|
||||
@@ -658,15 +598,11 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
|
||||
{
|
||||
auto & rv = PRIV(read_op)->read_vec;
|
||||
int req = fill_partial_checksum_blocks(rv, fulfilled, clean_entry_bitmap, dyn_data, from_journal,
|
||||
(uint8_t*)read_op->buf, read_op->offset, read_op->offset+read_op->len, item_start, item_end);
|
||||
(uint8_t*)read_op->buf, read_op->offset, read_op->offset+read_op->len);
|
||||
if (!inmemory_meta && !from_journal && req > 0)
|
||||
{
|
||||
// Read checksums from disk
|
||||
uint8_t *csum_buf = read_clean_meta_block(read_op, clean_loc, rv.size()-req);
|
||||
if (!csum_buf)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
for (int i = req; i > 0; i--)
|
||||
{
|
||||
rv[rv.size()-i].csum_buf = csum_buf;
|
||||
@@ -679,7 +615,7 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
|
||||
return false;
|
||||
}
|
||||
}
|
||||
PRIV(read_op)->clean_loc_used = req > 0 ? UINT64_MAX : 0;
|
||||
PRIV(read_op)->clean_block_used = req > 0;
|
||||
}
|
||||
else if (from_journal)
|
||||
{
|
||||
@@ -729,10 +665,6 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
|
||||
{
|
||||
// Read checksums from disk
|
||||
csum_buf = read_clean_meta_block(read_op, clean_loc, PRIV(read_op)->read_vec.size());
|
||||
if (!csum_buf)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
csum_done = true;
|
||||
}
|
||||
uint8_t *csum = !dsk.csum_block_size ? 0 : (csum_buf + 2*dsk.clean_entry_bitmap_size + bmp_start*(dsk.data_csum_type & 0xFF));
|
||||
@@ -747,13 +679,13 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
|
||||
}
|
||||
}
|
||||
// Increment reference counter if clean data is being read from the disk
|
||||
if (PRIV(read_op)->clean_loc_used == UINT64_MAX)
|
||||
if (PRIV(read_op)->clean_block_used)
|
||||
{
|
||||
auto & uo = used_clean_objects[clean_loc];
|
||||
uo.refs++;
|
||||
if (dsk.csum_block_size && flusher->is_mutated(clean_loc))
|
||||
uo.was_changed = true;
|
||||
PRIV(read_op)->clean_loc_used = clean_loc + 1;
|
||||
PRIV(read_op)->clean_block_used = clean_loc;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -793,18 +725,12 @@ bool blockstore_impl_t::verify_padded_checksums(uint8_t *clean_entry_bitmap, uin
|
||||
while (pos < iov[i].iov_len)
|
||||
{
|
||||
uint32_t start = pos;
|
||||
uint8_t bit = 1;
|
||||
if (clean_entry_bitmap)
|
||||
uint8_t bit = (clean_entry_bitmap[bmp_pos >> 3] >> (bmp_pos & 0x7)) & 1;
|
||||
while (pos < iov[i].iov_len && ((clean_entry_bitmap[bmp_pos >> 3] >> (bmp_pos & 0x7)) & 1) == bit)
|
||||
{
|
||||
bit = (clean_entry_bitmap[bmp_pos >> 3] >> (bmp_pos & 0x7)) & 1;
|
||||
while (pos < iov[i].iov_len && ((clean_entry_bitmap[bmp_pos >> 3] >> (bmp_pos & 0x7)) & 1) == bit)
|
||||
{
|
||||
pos += dsk.bitmap_granularity;
|
||||
bmp_pos++;
|
||||
}
|
||||
pos += dsk.bitmap_granularity;
|
||||
bmp_pos++;
|
||||
}
|
||||
else
|
||||
pos = iov[i].iov_len;
|
||||
uint32_t len = pos-start;
|
||||
auto buf = (uint8_t*)iov[i].iov_base+start;
|
||||
while (block_done+len >= dsk.csum_block_size)
|
||||
@@ -881,7 +807,7 @@ bool blockstore_impl_t::verify_clean_padded_checksums(blockstore_op_t *op, uint6
|
||||
{
|
||||
uint32_t offset = clean_loc % dsk.data_block_size;
|
||||
if (from_journal)
|
||||
return verify_padded_checksums(NULL, dyn_data, offset, iov, n_iov, bad_block_cb);
|
||||
return verify_padded_checksums(dyn_data, dyn_data + dsk.clean_entry_bitmap_size, offset, iov, n_iov, bad_block_cb);
|
||||
clean_loc = (clean_loc / dsk.data_block_size) * dsk.data_block_size;
|
||||
if (!dyn_data)
|
||||
{
|
||||
@@ -909,7 +835,7 @@ void blockstore_impl_t::handle_read_event(ring_data_t *data, blockstore_op_t *op
|
||||
void *meta_block = NULL;
|
||||
if (dsk.csum_block_size > dsk.bitmap_granularity)
|
||||
{
|
||||
for (int i = 0; i < rv.size(); i++)
|
||||
for (int i = rv.size()-1; i >= 0 && (rv[i].copy_flags & COPY_BUF_CSUM_FILL); i--)
|
||||
{
|
||||
if (rv[i].copy_flags & COPY_BUF_META_BLOCK)
|
||||
{
|
||||
@@ -919,41 +845,8 @@ void blockstore_impl_t::handle_read_event(ring_data_t *data, blockstore_op_t *op
|
||||
rv[i].buf = NULL;
|
||||
continue;
|
||||
}
|
||||
if (rv[i].copy_flags & COPY_BUF_ZERO)
|
||||
{
|
||||
// Zero read
|
||||
continue;
|
||||
}
|
||||
if (rv[i].copy_flags & COPY_BUF_COALESCED)
|
||||
{
|
||||
// Sub-block shared with another read. Skip
|
||||
continue;
|
||||
}
|
||||
if ((rv[i].copy_flags & COPY_BUF_JOURNAL) && journal.inmemory)
|
||||
{
|
||||
// Do not check journal checksums in-memory
|
||||
continue;
|
||||
}
|
||||
iovec single_iov = {};
|
||||
iovec *iov = NULL;
|
||||
int n_iov = 0;
|
||||
if (rv[i].copy_flags & COPY_BUF_CSUM_FILL)
|
||||
{
|
||||
// Padded, buffer list passed using a 'creepy way'
|
||||
iov = (struct iovec*)((uint8_t*)rv[i].buf + (rv[i].len & 0xFFFFFFFF));
|
||||
n_iov = rv[i].len >> 32;
|
||||
}
|
||||
else
|
||||
{
|
||||
// Not padded, buffer is fully within the input buffer
|
||||
assert(op->buf);
|
||||
assert(rv[i].csum_buf);
|
||||
iov = &single_iov;
|
||||
n_iov = 1;
|
||||
assert(rv[i].offset >= op->offset);
|
||||
assert(rv[i].offset + rv[i].len <= op->offset + op->len);
|
||||
single_iov = { .iov_base = op->buf + rv[i].offset - op->offset, .iov_len = rv[i].len };
|
||||
}
|
||||
struct iovec *iov = (struct iovec*)((uint8_t*)rv[i].buf + (rv[i].len & 0xFFFFFFFF));
|
||||
int n_iov = rv[i].len >> 32;
|
||||
bool ok = true;
|
||||
if (rv[i].copy_flags & COPY_BUF_JOURNAL)
|
||||
{
|
||||
@@ -1051,7 +944,23 @@ void blockstore_impl_t::handle_read_event(ring_data_t *data, blockstore_op_t *op
|
||||
meta_block = NULL;
|
||||
}
|
||||
}
|
||||
release_clean(op);
|
||||
if (PRIV(op)->clean_block_used)
|
||||
{
|
||||
// Release clean data block
|
||||
auto uo_it = used_clean_objects.find(PRIV(op)->clean_block_used);
|
||||
if (uo_it != used_clean_objects.end())
|
||||
{
|
||||
uo_it->second.refs--;
|
||||
if (uo_it->second.refs <= 0)
|
||||
{
|
||||
if (uo_it->second.was_freed)
|
||||
{
|
||||
data_alloc->set(PRIV(op)->clean_block_used, false);
|
||||
}
|
||||
used_clean_objects.erase(uo_it);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!journal.inmemory)
|
||||
{
|
||||
// Release journal sector usage
|
||||
|
||||
@@ -491,7 +491,7 @@ void blockstore_impl_t::mark_stable(obj_ver_id v, bool forget_dirty)
|
||||
if (!exists)
|
||||
{
|
||||
uint64_t space_id = dirty_it->first.oid.inode;
|
||||
if (no_inode_stats.find(dirty_it->first.oid.inode >> (64-POOL_ID_BITS)) != no_inode_stats.end())
|
||||
if (no_inode_stats[dirty_it->first.oid.inode >> (64-POOL_ID_BITS)])
|
||||
space_id = space_id & ~(((uint64_t)1 << (64-POOL_ID_BITS)) - 1);
|
||||
inode_space_stats[space_id] += dsk.data_block_size;
|
||||
used_blocks++;
|
||||
@@ -501,7 +501,7 @@ void blockstore_impl_t::mark_stable(obj_ver_id v, bool forget_dirty)
|
||||
else if (IS_DELETE(dirty_it->second.state))
|
||||
{
|
||||
uint64_t space_id = dirty_it->first.oid.inode;
|
||||
if (no_inode_stats.find(dirty_it->first.oid.inode >> (64-POOL_ID_BITS)) != no_inode_stats.end())
|
||||
if (no_inode_stats[dirty_it->first.oid.inode >> (64-POOL_ID_BITS)])
|
||||
space_id = space_id & ~(((uint64_t)1 << (64-POOL_ID_BITS)) - 1);
|
||||
auto & sp = inode_space_stats[space_id];
|
||||
if (sp > dsk.data_block_size)
|
||||
|
||||
+16
-58
@@ -3,27 +3,6 @@ cmake_minimum_required(VERSION 2.8...3.30)
|
||||
project(vitastor)
|
||||
|
||||
# libvitastor_common.a
|
||||
add_library(vitastor_common STATIC
|
||||
etcd_state_client.cpp
|
||||
msgr_stop.cpp
|
||||
msgr_op.cpp
|
||||
../../json11/json11.cpp
|
||||
osd_ops.cpp
|
||||
pg_states.cpp
|
||||
msgr_encrypt.cpp
|
||||
msgr_handshake.cpp
|
||||
../util/allocator.cpp
|
||||
../util/addr_util.cpp
|
||||
../util/timerfd_manager.cpp
|
||||
../util/str_util.cpp
|
||||
../util/json_util.cpp
|
||||
../util/xxh_x86dispatch.c
|
||||
../util/openssl_util.cpp
|
||||
)
|
||||
target_compile_options(vitastor_common PUBLIC -fPIC)
|
||||
target_link_libraries(vitastor_common ${OPENSSL_LIBRARIES} ${ISAL_CRYPTO_LIBRARIES})
|
||||
|
||||
# libvitastor_net.a
|
||||
set(MSGR_RDMA "")
|
||||
if (IBVERBS_LIBRARIES)
|
||||
set(MSGR_RDMA "msgr_rdma.cpp")
|
||||
@@ -32,32 +11,25 @@ set(MSGR_RDMACM "")
|
||||
if (RDMACM_LIBRARIES)
|
||||
set(MSGR_RDMACM "msgr_rdmacm.cpp")
|
||||
endif (RDMACM_LIBRARIES)
|
||||
add_library(vitastor_net STATIC
|
||||
../util/epoll_manager.cpp
|
||||
etcd_state_client_http.cpp
|
||||
messenger.cpp
|
||||
msgr_iothread.cpp
|
||||
msgr_send.cpp
|
||||
msgr_receive.cpp
|
||||
msgr_encrypt.cpp
|
||||
../util/ringloop.cpp
|
||||
http_client.cpp
|
||||
${MSGR_RDMA}
|
||||
${MSGR_RDMACM}
|
||||
add_library(vitastor_common STATIC
|
||||
../util/epoll_manager.cpp etcd_state_client.cpp messenger.cpp msgr_iothread.cpp ../util/addr_util.cpp ../util/xxh_x86dispatch.c ../util/openssl_util.cpp
|
||||
msgr_encrypt.cpp msgr_handshake.cpp msgr_stop.cpp msgr_op.cpp msgr_send.cpp msgr_receive.cpp ../util/ringloop.cpp ../../json11/json11.cpp
|
||||
http_client.cpp osd_ops.cpp pg_states.cpp ../util/timerfd_manager.cpp ../util/str_util.cpp ../util/json_util.cpp ${MSGR_RDMA} ${MSGR_RDMACM}
|
||||
)
|
||||
target_link_libraries(vitastor_net pthread vitastor_common ${CARES_LIBRARIES})
|
||||
target_compile_options(vitastor_net PUBLIC -fPIC)
|
||||
target_link_libraries(vitastor_common pthread ${OPENSSL_LIBRARIES} ${CARES_LIBRARIES} ${ISAL_CRYPTO_LIBRARIES})
|
||||
target_compile_options(vitastor_common PUBLIC -fPIC)
|
||||
|
||||
# libvitastor_client_int.a
|
||||
add_library(vitastor_client_int STATIC
|
||||
# libvitastor_client.so
|
||||
add_library(vitastor_client SHARED
|
||||
cluster_client.cpp
|
||||
cluster_client_real.cpp
|
||||
cluster_client_list.cpp
|
||||
cluster_client_wb.cpp
|
||||
cluster_client_icache.cpp
|
||||
vitastor_c.cpp
|
||||
)
|
||||
target_link_libraries(vitastor_client_int
|
||||
vitastor_net
|
||||
set_target_properties(vitastor_client PROPERTIES PUBLIC_HEADER "client/vitastor_c.h")
|
||||
target_link_libraries(vitastor_client
|
||||
vitastor_common
|
||||
vitastor_cli
|
||||
${LIBURING_LIBRARIES}
|
||||
${IBVERBS_LIBRARIES}
|
||||
@@ -65,16 +37,6 @@ target_link_libraries(vitastor_client_int
|
||||
${OPENSSL_LIBRARIES}
|
||||
${ISAL_CRYPTO_LIBRARIES}
|
||||
)
|
||||
target_compile_options(vitastor_client_int PUBLIC -fPIC)
|
||||
|
||||
# libvitastor_client.so
|
||||
add_library(vitastor_client SHARED
|
||||
vitastor_c.cpp
|
||||
)
|
||||
set_target_properties(vitastor_client PROPERTIES PUBLIC_HEADER "client/vitastor_c.h")
|
||||
target_link_libraries(vitastor_client
|
||||
vitastor_client_int
|
||||
)
|
||||
set_target_properties(vitastor_client PROPERTIES VERSION ${VITASTOR_VERSION} SOVERSION 0)
|
||||
configure_file(vitastor.pc.in vitastor.pc @ONLY)
|
||||
|
||||
@@ -136,15 +98,11 @@ endif (${WITH_QEMU})
|
||||
add_executable(test_cluster_client
|
||||
EXCLUDE_FROM_ALL
|
||||
../test/test_cluster_client.cpp
|
||||
cluster_client.cpp
|
||||
cluster_client_list.cpp
|
||||
cluster_client_wb.cpp
|
||||
cluster_client_icache.cpp
|
||||
../test/mock/messenger.cpp
|
||||
../test/mock/vault.cpp
|
||||
etcd_state_client_mock.cpp
|
||||
pg_states.cpp osd_ops.cpp cluster_client.cpp cluster_client_list.cpp cluster_client_wb.cpp cluster_client_icache.cpp msgr_op.cpp ../test/mock/messenger.cpp msgr_stop.cpp msgr_encrypt.cpp
|
||||
etcd_state_client.cpp ../util/timerfd_manager.cpp ../util/addr_util.cpp ../util/str_util.cpp ../util/json_util.cpp ../util/xxh_x86dispatch.c ../util/openssl_util.cpp ../../json11/json11.cpp
|
||||
)
|
||||
target_link_libraries(test_cluster_client vitastor_common ${LIBURING_LIBRARIES} ${OPENSSL_LIBRARIES} ${ISAL_CRYPTO_LIBRARIES})
|
||||
target_link_libraries(test_cluster_client ${LIBURING_LIBRARIES} ${OPENSSL_LIBRARIES} ${ISAL_CRYPTO_LIBRARIES})
|
||||
target_compile_definitions(test_cluster_client PUBLIC -D__MOCK__)
|
||||
target_include_directories(test_cluster_client BEFORE PUBLIC ${CMAKE_SOURCE_DIR}/src/test/mock)
|
||||
add_dependencies(build_tests test_cluster_client)
|
||||
add_test(NAME test_cluster_client COMMAND test_cluster_client)
|
||||
|
||||
@@ -11,7 +11,7 @@
|
||||
#define TRY_SEND_CONNECTING 1
|
||||
#define TRY_SEND_OK 2
|
||||
|
||||
cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config, std::unique_ptr<etcd_state_client_t> st_cli_ptr)
|
||||
cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config)
|
||||
{
|
||||
wb = new writeback_cache_t();
|
||||
|
||||
@@ -53,24 +53,24 @@ cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd
|
||||
};
|
||||
msgr.parse_config(config, true);
|
||||
|
||||
st_cli = std::move(st_cli_ptr);
|
||||
st_cli->on_load_config_hook = [this](json11::Json::object & cfg) { on_load_config_hook(cfg); };
|
||||
st_cli->on_change_osd_state_hook = [this](uint64_t peer_osd) { on_change_osd_state_hook(peer_osd); };
|
||||
st_cli->on_change_pool_config_hook = [this]() { on_change_pool_config_hook(); };
|
||||
st_cli->on_change_pg_config_hook = [this]() { on_change_pool_config_hook(); };
|
||||
st_cli->on_change_pg_state_hook = [this](pool_id_t pool_id, pg_num_t pg_num, osd_num_t prev_primary) { on_change_pg_state_hook(pool_id, pg_num, prev_primary); };
|
||||
st_cli->on_change_node_placement_hook = [this]() { on_change_node_placement_hook(); };
|
||||
st_cli->on_load_pgs_hook = [this](bool success) { on_load_pgs_hook(success); };
|
||||
st_cli->on_reload_hook = [this]() { this->st_cli->load_global_config(); };
|
||||
st_cli->on_inode_change_hook = [this](uint64_t inode, bool removed) { on_change_inode_hook(inode, removed); };
|
||||
st_cli.tfd = tfd;
|
||||
st_cli.on_load_config_hook = [this](json11::Json::object & cfg) { on_load_config_hook(cfg); };
|
||||
st_cli.on_change_osd_state_hook = [this](uint64_t peer_osd) { on_change_osd_state_hook(peer_osd); };
|
||||
st_cli.on_change_pool_config_hook = [this]() { on_change_pool_config_hook(); };
|
||||
st_cli.on_change_pg_config_hook = [this]() { on_change_pool_config_hook(); };
|
||||
st_cli.on_change_pg_state_hook = [this](pool_id_t pool_id, pg_num_t pg_num, osd_num_t prev_primary) { on_change_pg_state_hook(pool_id, pg_num, prev_primary); };
|
||||
st_cli.on_change_node_placement_hook = [this]() { on_change_node_placement_hook(); };
|
||||
st_cli.on_load_pgs_hook = [this](bool success) { on_load_pgs_hook(success); };
|
||||
st_cli.on_reload_hook = [this]() { st_cli.load_global_config(); };
|
||||
st_cli.on_inode_change_hook = [this](uint64_t inode, bool removed) { on_change_inode_hook(inode, removed); };
|
||||
|
||||
st_cli->parse_config(config);
|
||||
st_cli->infinite_start = false;
|
||||
st_cli.parse_config(config);
|
||||
st_cli.infinite_start = false;
|
||||
if (!config["client_infinite_start"].is_null())
|
||||
{
|
||||
st_cli->infinite_start = config["client_infinite_start"].bool_value();
|
||||
st_cli.infinite_start = config["client_infinite_start"].bool_value();
|
||||
}
|
||||
st_cli->load_global_config();
|
||||
st_cli.load_global_config();
|
||||
}
|
||||
|
||||
cluster_client_t::~cluster_client_t()
|
||||
@@ -467,7 +467,7 @@ void cluster_client_t::on_load_config_hook(json11::Json::object & etcd_global_co
|
||||
auto etcd_report_interval = config["etcd_report_interval"].uint64_value();
|
||||
if (!etcd_report_interval)
|
||||
etcd_report_interval = 5;
|
||||
client_wait_up_timeout = 1+etcd_report_interval+(st_cli->max_etcd_attempts*(2*st_cli->etcd_quick_timeout)+999)/1000;
|
||||
client_wait_up_timeout = 1+etcd_report_interval+(st_cli.max_etcd_attempts*(2*st_cli.etcd_quick_timeout)+999)/1000;
|
||||
}
|
||||
// log_level
|
||||
log_level = config["log_level"].uint64_value();
|
||||
@@ -482,8 +482,8 @@ void cluster_client_t::on_load_config_hook(json11::Json::object & etcd_global_co
|
||||
// vault
|
||||
vault_parse_config();
|
||||
msgr.parse_config(config, false);
|
||||
st_cli->parse_config(config);
|
||||
st_cli->load_pgs();
|
||||
st_cli.parse_config(config);
|
||||
st_cli.load_pgs();
|
||||
}
|
||||
|
||||
osd_num_t cluster_client_t::select_random_osd(const std::vector<osd_num_t> & osds)
|
||||
@@ -492,7 +492,7 @@ osd_num_t cluster_client_t::select_random_osd(const std::vector<osd_num_t> & osd
|
||||
int alive_count = 0;
|
||||
for (auto & osd_num: osds)
|
||||
{
|
||||
if (!st_cli->peer_states[osd_num].is_null())
|
||||
if (!st_cli.peer_states[osd_num].is_null())
|
||||
alive_set[alive_count++] = osd_num;
|
||||
}
|
||||
if (!alive_count)
|
||||
@@ -509,7 +509,7 @@ osd_num_t cluster_client_t::select_nearest_osd(const std::vector<osd_num_t> & os
|
||||
while (self_tree_metrics.find(cur_id) == self_tree_metrics.end())
|
||||
{
|
||||
self_tree_metrics[cur_id] = metric++;
|
||||
json11::Json cur_placement = st_cli->node_placement[cur_id];
|
||||
json11::Json cur_placement = st_cli.node_placement[cur_id];
|
||||
cur_id = cur_placement["parent"].string_value();
|
||||
}
|
||||
if (cur_id != "")
|
||||
@@ -529,7 +529,7 @@ osd_num_t cluster_client_t::select_nearest_osd(const std::vector<osd_num_t> & os
|
||||
}
|
||||
else
|
||||
{
|
||||
auto & peer_state = st_cli->peer_states[osd_num];
|
||||
auto & peer_state = st_cli.peer_states[osd_num];
|
||||
if (!peer_state.is_null())
|
||||
{
|
||||
metric = self_tree_metrics[""];
|
||||
@@ -539,7 +539,7 @@ osd_num_t cluster_client_t::select_nearest_osd(const std::vector<osd_num_t> & os
|
||||
while (seen.find(cur_id) == seen.end())
|
||||
{
|
||||
seen.insert(cur_id);
|
||||
json11::Json cur_placement = st_cli->node_placement[cur_id];
|
||||
json11::Json cur_placement = st_cli.node_placement[cur_id];
|
||||
std::string cur_parent = cur_placement["parent"].string_value();
|
||||
cur_id = (!first || cur_parent != "" ? cur_parent : peer_state["host"].string_value());
|
||||
first = false;
|
||||
@@ -564,7 +564,7 @@ osd_num_t cluster_client_t::select_nearest_osd(const std::vector<osd_num_t> & os
|
||||
|
||||
void cluster_client_t::on_load_pgs_hook(bool success)
|
||||
{
|
||||
for (auto & pool_item: st_cli->pool_config)
|
||||
for (auto & pool_item: st_cli.pool_config)
|
||||
{
|
||||
pg_counts[pool_item.first] = pool_item.second.real_pg_count;
|
||||
}
|
||||
@@ -584,7 +584,7 @@ void cluster_client_t::on_load_pgs_hook(bool success)
|
||||
|
||||
void cluster_client_t::on_change_pool_config_hook()
|
||||
{
|
||||
for (auto & pool_item: st_cli->pool_config)
|
||||
for (auto & pool_item: st_cli.pool_config)
|
||||
{
|
||||
if (pg_counts[pool_item.first] != pool_item.second.real_pg_count)
|
||||
{
|
||||
@@ -615,7 +615,7 @@ void cluster_client_t::on_change_pool_config_hook()
|
||||
|
||||
void cluster_client_t::on_change_pg_state_hook(pool_id_t pool_id, pg_num_t pg_num, osd_num_t prev_primary)
|
||||
{
|
||||
auto & pg_cfg = st_cli->pool_config[pool_id].pg_config[pg_num];
|
||||
auto & pg_cfg = st_cli.pool_config[pool_id].pg_config[pg_num];
|
||||
if (pg_cfg.cur_primary != prev_primary)
|
||||
{
|
||||
// Repeat this PG operations because an OSD which stopped being primary may not fsync operations
|
||||
@@ -633,8 +633,8 @@ bool cluster_client_t::get_immediate_commit(uint64_t inode)
|
||||
pool_id_t pool_id = INODE_POOL(inode);
|
||||
if (!pool_id)
|
||||
return true;
|
||||
auto pool_it = st_cli->pool_config.find(pool_id);
|
||||
if (pool_it == st_cli->pool_config.end())
|
||||
auto pool_it = st_cli.pool_config.find(pool_id);
|
||||
if (pool_it == st_cli.pool_config.end())
|
||||
return true;
|
||||
return pool_it->second.immediate_commit == IMMEDIATE_ALL;
|
||||
}
|
||||
@@ -644,7 +644,7 @@ void cluster_client_t::on_change_osd_state_hook(uint64_t peer_osd)
|
||||
osd_tree_metrics.erase(peer_osd);
|
||||
if (msgr.wanted_peers.find(peer_osd) != msgr.wanted_peers.end())
|
||||
{
|
||||
msgr.connect_peer(peer_osd, st_cli->peer_states[peer_osd]);
|
||||
msgr.connect_peer(peer_osd, st_cli.peer_states[peer_osd]);
|
||||
continue_lists();
|
||||
}
|
||||
}
|
||||
@@ -943,8 +943,8 @@ bool cluster_client_t::check_rw(cluster_op_t *op)
|
||||
cb(op);
|
||||
return false;
|
||||
}
|
||||
auto pool_it = st_cli->pool_config.find(pool_id);
|
||||
if (pool_it == st_cli->pool_config.end() || pool_it->second.real_pg_count == 0)
|
||||
auto pool_it = st_cli.pool_config.find(pool_id);
|
||||
if (pool_it == st_cli.pool_config.end() || pool_it->second.real_pg_count == 0)
|
||||
{
|
||||
// Pools are loaded, but this one is unknown
|
||||
op->retval = -EINVAL;
|
||||
@@ -1057,7 +1057,7 @@ void cluster_client_t::execute_raw(osd_num_t osd_num, osd_op_t *op)
|
||||
else
|
||||
{
|
||||
if (msgr.wanted_peers.find(osd_num) == msgr.wanted_peers.end())
|
||||
msgr.connect_peer(osd_num, st_cli->peer_states[osd_num]);
|
||||
msgr.connect_peer(osd_num, st_cli.peer_states[osd_num]);
|
||||
raw_ops.emplace(osd_num, op);
|
||||
}
|
||||
}
|
||||
@@ -1206,7 +1206,7 @@ resume_2:
|
||||
op->retval = op->len;
|
||||
if (op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP)
|
||||
{
|
||||
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(op->inode));
|
||||
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->inode));
|
||||
op->retval = op->len / pool_cfg.bitmap_granularity;
|
||||
}
|
||||
if (op->flush_id)
|
||||
@@ -1286,7 +1286,7 @@ void cluster_client_t::slice_rw(cluster_op_t *op)
|
||||
{
|
||||
// Slice the request into individual object stripe requests
|
||||
// Primary OSDs still operate individual stripes, but their size is multiplied by PG minsize in case of EC
|
||||
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(op->cur_inode));
|
||||
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->cur_inode));
|
||||
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
||||
uint64_t pg_block_size = pool_cfg.data_block_size * pg_data_size;
|
||||
uint64_t first_stripe = (op->offset / pg_block_size) * pg_block_size;
|
||||
@@ -1389,7 +1389,7 @@ bool cluster_client_t::affects_pg(uint64_t inode, uint64_t offset, uint64_t len,
|
||||
{
|
||||
return false;
|
||||
}
|
||||
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(inode));
|
||||
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(inode));
|
||||
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
||||
uint64_t pg_block_size = pool_cfg.data_block_size * pg_data_size;
|
||||
uint64_t first_stripe = (offset / pg_block_size) * pg_block_size;
|
||||
@@ -1408,7 +1408,7 @@ bool cluster_client_t::affects_pg(uint64_t inode, uint64_t offset, uint64_t len,
|
||||
|
||||
bool cluster_client_t::affects_osd(uint64_t inode, uint64_t offset, uint64_t len, osd_num_t osd)
|
||||
{
|
||||
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(inode));
|
||||
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(inode));
|
||||
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
||||
uint64_t pg_block_size = pool_cfg.data_block_size * pg_data_size;
|
||||
uint64_t first_stripe = (offset / pg_block_size) * pg_block_size;
|
||||
@@ -1432,7 +1432,7 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
|
||||
init_msgr();
|
||||
}
|
||||
auto part = &op->parts[i];
|
||||
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(op->cur_inode));
|
||||
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->cur_inode));
|
||||
auto pg_it = pool_cfg.pg_config.find(part->pg_num);
|
||||
if (pg_it != pool_cfg.pg_config.end() &&
|
||||
!pg_it->second.pause && pg_it->second.cur_primary &&
|
||||
@@ -1466,8 +1466,8 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
|
||||
uint64_t meta_rev = 0;
|
||||
if (op->opcode != OSD_OP_READ_BITMAP && op->opcode != OSD_OP_DELETE && !op->deoptimise_snapshot)
|
||||
{
|
||||
auto ino_it = st_cli->inode_config.find(op->cur_inode);
|
||||
if (ino_it != st_cli->inode_config.end())
|
||||
auto ino_it = st_cli.inode_config.find(op->cur_inode);
|
||||
if (ino_it != st_cli.inode_config.end())
|
||||
meta_rev = ino_it->second.mod_revision;
|
||||
}
|
||||
part->op = (osd_op_t){
|
||||
@@ -1501,7 +1501,7 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
|
||||
}
|
||||
else if (msgr.wanted_peers.find(primary_osd) == msgr.wanted_peers.end())
|
||||
{
|
||||
msgr.connect_peer(primary_osd, st_cli->peer_states[primary_osd]);
|
||||
msgr.connect_peer(primary_osd, st_cli.peer_states[primary_osd]);
|
||||
return TRY_SEND_CONNECTING;
|
||||
}
|
||||
}
|
||||
@@ -1694,7 +1694,7 @@ void cluster_client_t::handle_op_part(cluster_op_part_t *part)
|
||||
void cluster_client_t::copy_part_bitmap(cluster_op_t *op, cluster_op_part_t *part)
|
||||
{
|
||||
// Copy (OR) bitmap
|
||||
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(op->cur_inode));
|
||||
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->cur_inode));
|
||||
uint32_t pg_block_size = pool_cfg.data_block_size * (
|
||||
pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks
|
||||
);
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
#pragma once
|
||||
|
||||
#include "messenger.h"
|
||||
#include "etcd_state_client_http.h"
|
||||
#include "etcd_state_client.h"
|
||||
#include "../util/robin_hood.h"
|
||||
|
||||
#define DEFAULT_CLIENT_MAX_DIRTY_BYTES 32*1024*1024
|
||||
@@ -176,7 +176,7 @@ class __attribute__((visibility("default"))) cluster_client_t
|
||||
bool msgr_initialized = false;
|
||||
|
||||
public:
|
||||
std::unique_ptr<etcd_state_client_t> st_cli;
|
||||
etcd_state_client_t st_cli;
|
||||
|
||||
osd_messenger_t msgr;
|
||||
void init_msgr();
|
||||
@@ -184,8 +184,7 @@ public:
|
||||
json11::Json::object cli_config, file_config, etcd_global_config;
|
||||
json11::Json::object config;
|
||||
|
||||
static cluster_client_t* create(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config);
|
||||
cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config, std::unique_ptr<etcd_state_client_t> st_cli);
|
||||
cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config);
|
||||
~cluster_client_t();
|
||||
void execute(cluster_op_t *op);
|
||||
void execute_raw(osd_num_t osd_num, osd_op_t *op);
|
||||
|
||||
@@ -4,8 +4,14 @@
|
||||
#include <stdexcept>
|
||||
#include <assert.h>
|
||||
#include "cluster_client_impl.h"
|
||||
#include "http_client.h"
|
||||
#include "str_util.h"
|
||||
|
||||
#define VAULT_KEY_NOT_LOADED 0
|
||||
#define VAULT_KEY_LOADING 1
|
||||
#define VAULT_KEY_LOADED 2
|
||||
#define VAULT_KEY_ERROR 3
|
||||
|
||||
inode_cache_t::~inode_cache_t()
|
||||
{
|
||||
if (key_data)
|
||||
@@ -16,15 +22,24 @@ inode_cache_t::~inode_cache_t()
|
||||
}
|
||||
}
|
||||
|
||||
void cluster_client_t::vault_destroy()
|
||||
{
|
||||
if (vault_http_ctx)
|
||||
{
|
||||
#ifndef __MOCK__
|
||||
http_destroy(vault_http_cli);
|
||||
http_context_destroy(vault_http_ctx);
|
||||
vault_http_cli = NULL;
|
||||
vault_http_ctx = NULL;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
void cluster_client_t::vault_parse_config()
|
||||
{
|
||||
vault_url = config["vault_url"].string_value();
|
||||
vault_client_cert = config["vault_client_cert"].string_value();
|
||||
if (vault_client_cert.empty())
|
||||
vault_client_cert = config["cert"].string_value();
|
||||
vault_client_key = config["vault_client_key"].string_value();
|
||||
if (vault_client_key.empty())
|
||||
vault_client_key = config["pkey"].string_value();
|
||||
vault_ca = config["vault_ca"].string_value();
|
||||
vault_secret_api_path = "/v1/secret/";
|
||||
if (config["vault_secret_api_path"].is_string())
|
||||
@@ -76,14 +91,14 @@ std::shared_ptr<inode_cache_t> cluster_client_t::inode_cache_get(inode_t ino)
|
||||
return icache_it->second;
|
||||
}
|
||||
// Fill inode cache
|
||||
auto ino_it = st_cli->inode_config.find(ino);
|
||||
if (ino_it == st_cli->inode_config.end())
|
||||
auto ino_it = st_cli.inode_config.find(ino);
|
||||
if (ino_it == st_cli.inode_config.end())
|
||||
{
|
||||
inode_cache[ino] = NULL;
|
||||
return NULL;
|
||||
}
|
||||
auto pool_it = st_cli->pool_config.find(INODE_POOL(ino));
|
||||
if (pool_it == st_cli->pool_config.end())
|
||||
auto pool_it = st_cli.pool_config.find(INODE_POOL(ino));
|
||||
if (pool_it == st_cli.pool_config.end())
|
||||
{
|
||||
inode_cache[ino] = NULL;
|
||||
return NULL;
|
||||
@@ -110,11 +125,11 @@ std::shared_ptr<inode_cache_t> cluster_client_t::inode_cache_get(inode_t ino)
|
||||
break;
|
||||
}
|
||||
seen.insert(parent_id);
|
||||
ino_it = st_cli->inode_config.find(parent_id);
|
||||
ino_it = st_cli.inode_config.find(parent_id);
|
||||
if (INODE_POOL(parent_id) == INODE_POOL(ino))
|
||||
{
|
||||
icache->chain.push_back(parent_id);
|
||||
if (ino_it == st_cli->inode_config.end())
|
||||
if (ino_it == st_cli.inode_config.end())
|
||||
chain_cfg.push_back(NULL);
|
||||
else
|
||||
{
|
||||
@@ -125,7 +140,7 @@ std::shared_ptr<inode_cache_t> cluster_client_t::inode_cache_get(inode_t ino)
|
||||
}
|
||||
else if (!icache->other_pool_parent_id)
|
||||
icache->other_pool_parent_id = parent_id;
|
||||
if (ino_it == st_cli->inode_config.end())
|
||||
if (ino_it == st_cli.inode_config.end())
|
||||
break;
|
||||
parent_id = ino_it->second.parent_id;
|
||||
}
|
||||
@@ -208,6 +223,113 @@ std::shared_ptr<inode_cache_t> cluster_client_t::inode_cache_get(inode_t ino)
|
||||
return icache;
|
||||
}
|
||||
|
||||
#ifndef __MOCK__
|
||||
bool cluster_client_t::vault_check_token()
|
||||
{
|
||||
timespec now;
|
||||
clock_gettime(CLOCK_REALTIME, &now);
|
||||
if (!vault_token_expire.tv_sec || vault_token_expire.tv_sec < now.tv_sec)
|
||||
{
|
||||
vault_loading = true;
|
||||
http_json_post(
|
||||
vault_http_cli, vault_url+"/v1/auth/cert/login", json11::Json::object{}, "",
|
||||
(http_options_t){ .timeout = (int)vault_timeout_ms, .keepalive = true },
|
||||
[this](http_message_t *response)
|
||||
{
|
||||
clock_gettime(CLOCK_REALTIME, &vault_token_expire);
|
||||
vault_loading = false;
|
||||
std::string err;
|
||||
json11::Json data;
|
||||
response->parse_json_response(err, data);
|
||||
if (err != "")
|
||||
{
|
||||
vault_token_expire.tv_sec += vault_error_timeout_sec;
|
||||
fprintf(stderr, "Vault request failed: %s\n", err.c_str());
|
||||
}
|
||||
else
|
||||
{
|
||||
uint64_t ttl = data["auth"]["lease_duration"].uint64_value();
|
||||
vault_token = data["auth"]["client_token"].string_value();
|
||||
if (vault_token.empty() || !ttl)
|
||||
{
|
||||
vault_token_expire.tv_sec += vault_error_timeout_sec;
|
||||
fprintf(stderr, "No token or lease_duration in Vault response: %s\n", data.dump().c_str());
|
||||
}
|
||||
else
|
||||
{
|
||||
if (ttl < vault_refresh_leeway_sec)
|
||||
vault_token_expire.tv_sec += ttl/2;
|
||||
else
|
||||
vault_token_expire.tv_sec += ttl - vault_refresh_leeway_sec;
|
||||
}
|
||||
}
|
||||
vault_load_keys();
|
||||
}
|
||||
);
|
||||
return false;
|
||||
}
|
||||
if (vault_token.empty())
|
||||
{
|
||||
// Auth error happened, mark all loads as failed
|
||||
for (auto & key_id: vault_key_load_queue)
|
||||
{
|
||||
auto & k = vault_keys[key_id];
|
||||
k.key_state = VAULT_KEY_ERROR;
|
||||
}
|
||||
vault_key_load_queue.clear();
|
||||
auto ops = std::move(key_wait_ops);
|
||||
for (cluster_op_t *op: ops)
|
||||
inode_cache.erase(op->inode);
|
||||
for (cluster_op_t *op: ops)
|
||||
execute_internal(op);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
|
||||
void cluster_client_t::vault_load_keys()
|
||||
{
|
||||
if (vault_loading || !vault_key_load_queue.size())
|
||||
{
|
||||
return;
|
||||
}
|
||||
#ifdef __MOCK__
|
||||
vault_loading = true;
|
||||
#else
|
||||
if (!vault_http_ctx)
|
||||
{
|
||||
std::string error;
|
||||
vault_http_ctx = http_context_init(tfd, vault_client_cert, vault_client_key, vault_ca, true, error);
|
||||
if (!vault_http_ctx)
|
||||
{
|
||||
fprintf(stderr, "Failed to initialize HTTP context for Vault: %s\n", error.c_str());
|
||||
exit(1);
|
||||
}
|
||||
vault_http_cli = http_init(vault_http_ctx);
|
||||
}
|
||||
if (!vault_check_token())
|
||||
{
|
||||
return;
|
||||
}
|
||||
std::string key_id = vault_key_load_queue[0];
|
||||
vault_key_load_queue.erase(vault_key_load_queue.begin());
|
||||
vault_loading = true;
|
||||
http_get(
|
||||
vault_http_cli, vault_url+vault_secret_api_path+key_id.substr(strlen(VAULT_KEY_PREFIX)), "X-Vault-Token: "+vault_token+"\r\n",
|
||||
(http_options_t){ .timeout = (int)vault_timeout_ms, .keepalive = true },
|
||||
[this, key_id](http_message_t *response)
|
||||
{
|
||||
vault_loading = false;
|
||||
std::string err;
|
||||
json11::Json data;
|
||||
response->parse_json_response(err, data);
|
||||
vault_parse_secret(key_id, err, data);
|
||||
}
|
||||
);
|
||||
#endif
|
||||
}
|
||||
|
||||
void cluster_client_t::vault_parse_secret(const std::string & key_id, const std::string & err, json11::Json data)
|
||||
{
|
||||
vault_loading = false;
|
||||
|
||||
@@ -17,11 +17,6 @@
|
||||
#define OP_FLUSH_BUFFER 0x02
|
||||
#define OP_IMMEDIATE_COMMIT 0x04
|
||||
|
||||
#define VAULT_KEY_NOT_LOADED 0
|
||||
#define VAULT_KEY_LOADING 1
|
||||
#define VAULT_KEY_LOADED 2
|
||||
#define VAULT_KEY_ERROR 3
|
||||
|
||||
struct cluster_buffer_t
|
||||
{
|
||||
uint8_t *buf;
|
||||
|
||||
@@ -63,14 +63,14 @@ void cluster_client_t::list_inode(inode_t inode, uint64_t min_offset, uint64_t m
|
||||
{
|
||||
init_msgr();
|
||||
pool_id_t pool_id = INODE_POOL(inode);
|
||||
if (!pool_id || st_cli->pool_config.find(pool_id) == st_cli->pool_config.end())
|
||||
if (!pool_id || st_cli.pool_config.find(pool_id) == st_cli.pool_config.end())
|
||||
{
|
||||
if (log_level > 0)
|
||||
fprintf(stderr, "Pool %u does not exist\n", pool_id);
|
||||
pg_callback(-EINVAL, 0, 0, std::set<object_id>());
|
||||
return;
|
||||
}
|
||||
auto pg_stripe_size = st_cli->pool_config.at(pool_id).pg_stripe_size;
|
||||
auto pg_stripe_size = st_cli.pool_config.at(pool_id).pg_stripe_size;
|
||||
if (min_offset)
|
||||
min_offset = (min_offset/pg_stripe_size) * pg_stripe_size;
|
||||
inode_list_t *lst = new inode_list_t();
|
||||
@@ -110,13 +110,13 @@ bool cluster_client_t::continue_listing(inode_list_t *lst)
|
||||
|
||||
bool cluster_client_t::restart_listing(inode_list_t* lst)
|
||||
{
|
||||
auto pool_it = st_cli->pool_config.find(lst->pool_id);
|
||||
auto pool_it = st_cli.pool_config.find(lst->pool_id);
|
||||
// We want listing to be consistent. To achieve it we should:
|
||||
// 1) retry listing of each PG if its state changes
|
||||
// 2) abort listing if PG count changes during listing
|
||||
// 3) ideally, only talk to the primary OSD - this will be done separately
|
||||
// So first we add all PGs without checking their state
|
||||
if (pool_it == st_cli->pool_config.end() ||
|
||||
if (pool_it == st_cli.pool_config.end() ||
|
||||
lst->real_pg_count != pool_it->second.real_pg_count)
|
||||
{
|
||||
for (auto pg: lst->pgs)
|
||||
@@ -136,7 +136,7 @@ bool cluster_client_t::restart_listing(inode_list_t* lst)
|
||||
fprintf(stderr, "PG count in pool %u changed during listing\n", lst->pool_id);
|
||||
}
|
||||
lst->pgs.clear();
|
||||
if (pool_it == st_cli->pool_config.end())
|
||||
if (pool_it == st_cli.pool_config.end())
|
||||
{
|
||||
// Unknown pool
|
||||
lst->callback(-EINVAL, 0, 0, std::set<object_id>());
|
||||
@@ -248,7 +248,7 @@ void cluster_client_t::set_list_retry_timeout(int ms, timespec new_time)
|
||||
|
||||
int cluster_client_t::start_pg_listing(inode_list_pg_t *pg)
|
||||
{
|
||||
auto & pool_cfg = st_cli->pool_config.at(pg->lst->pool_id);
|
||||
auto & pool_cfg = st_cli.pool_config.at(pg->lst->pool_id);
|
||||
auto pg_it = pool_cfg.pg_config.find(pg->pg_num);
|
||||
assert(pg->lst->real_pg_count == pool_cfg.real_pg_count);
|
||||
if (pg_it == pool_cfg.pg_config.end() ||
|
||||
@@ -277,7 +277,7 @@ int cluster_client_t::start_pg_listing(inode_list_pg_t *pg)
|
||||
for (auto peer_it = all_peers.begin(); peer_it != all_peers.end(); )
|
||||
{
|
||||
if (*peer_it != pg_it->second.cur_primary &&
|
||||
st_cli->peer_states[*peer_it].is_null())
|
||||
st_cli.peer_states[*peer_it].is_null())
|
||||
{
|
||||
pg->inactive_osds.push_back(*peer_it);
|
||||
all_peers.erase(peer_it++);
|
||||
@@ -298,11 +298,11 @@ int cluster_client_t::start_pg_listing(inode_list_pg_t *pg)
|
||||
if (msgr.osd_peers.find(peer_osd) == msgr.osd_peers.end())
|
||||
{
|
||||
// Initiate connection
|
||||
if (st_cli->peer_states[peer_osd].is_null())
|
||||
if (st_cli.peer_states[peer_osd].is_null())
|
||||
{
|
||||
return LIST_PG_WAIT_ACTIVE;
|
||||
}
|
||||
msgr.connect_peer(peer_osd, st_cli->peer_states[peer_osd]);
|
||||
msgr.connect_peer(peer_osd, st_cli.peer_states[peer_osd]);
|
||||
conn = false;
|
||||
}
|
||||
}
|
||||
@@ -336,7 +336,7 @@ void cluster_client_t::send_list(inode_list_osd_t *cur_list)
|
||||
if (!cur_list->pg->inflight_ops)
|
||||
cur_list->pg->lst->inflight_pgs++;
|
||||
cur_list->pg->inflight_ops++;
|
||||
auto & pool_cfg = st_cli->pool_config[cur_list->pg->lst->pool_id];
|
||||
auto & pool_cfg = st_cli.pool_config[cur_list->pg->lst->pool_id];
|
||||
osd_op_t *op = new osd_op_t();
|
||||
op->op_type = OSD_OP_OUT;
|
||||
// Already checked that it exists above, but anyway
|
||||
|
||||
@@ -1,125 +0,0 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||
|
||||
#include "cluster_client.h"
|
||||
#include "cluster_client_impl.h"
|
||||
#include "etcd_state_client_http.h"
|
||||
#include "http_client.h"
|
||||
|
||||
cluster_client_t* cluster_client_t::create(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config)
|
||||
{
|
||||
auto st_cli = new etcd_state_client_http_t(tfd);
|
||||
return new cluster_client_t(ringloop, tfd, config, std::unique_ptr<etcd_state_client_t>(st_cli));
|
||||
}
|
||||
|
||||
bool cluster_client_t::vault_check_token()
|
||||
{
|
||||
timespec now;
|
||||
clock_gettime(CLOCK_REALTIME, &now);
|
||||
if (!vault_token_expire.tv_sec || vault_token_expire.tv_sec < now.tv_sec)
|
||||
{
|
||||
vault_loading = true;
|
||||
http_json_post(
|
||||
vault_http_cli, vault_url+"/v1/auth/cert/login", json11::Json::object{}, "",
|
||||
(http_options_t){ .timeout = (int)vault_timeout_ms, .keepalive = true },
|
||||
[this](http_message_t *response)
|
||||
{
|
||||
clock_gettime(CLOCK_REALTIME, &vault_token_expire);
|
||||
vault_loading = false;
|
||||
std::string err;
|
||||
json11::Json data;
|
||||
response->parse_json_response(err, data);
|
||||
if (err != "")
|
||||
{
|
||||
vault_token_expire.tv_sec += vault_error_timeout_sec;
|
||||
fprintf(stderr, "Vault request failed: %s\n", err.c_str());
|
||||
}
|
||||
else
|
||||
{
|
||||
uint64_t ttl = data["auth"]["lease_duration"].uint64_value();
|
||||
vault_token = data["auth"]["client_token"].string_value();
|
||||
if (vault_token.empty() || !ttl)
|
||||
{
|
||||
vault_token_expire.tv_sec += vault_error_timeout_sec;
|
||||
fprintf(stderr, "No token or lease_duration in Vault response: %s\n", data.dump().c_str());
|
||||
}
|
||||
else
|
||||
{
|
||||
if (ttl < vault_refresh_leeway_sec)
|
||||
vault_token_expire.tv_sec += ttl/2;
|
||||
else
|
||||
vault_token_expire.tv_sec += ttl - vault_refresh_leeway_sec;
|
||||
}
|
||||
}
|
||||
vault_load_keys();
|
||||
}
|
||||
);
|
||||
return false;
|
||||
}
|
||||
if (vault_token.empty())
|
||||
{
|
||||
// Auth error happened, mark all loads as failed
|
||||
for (auto & key_id: vault_key_load_queue)
|
||||
{
|
||||
auto & k = vault_keys[key_id];
|
||||
k.key_state = VAULT_KEY_ERROR;
|
||||
}
|
||||
vault_key_load_queue.clear();
|
||||
auto ops = std::move(key_wait_ops);
|
||||
for (cluster_op_t *op: ops)
|
||||
inode_cache.erase(op->inode);
|
||||
for (cluster_op_t *op: ops)
|
||||
execute_internal(op);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void cluster_client_t::vault_destroy()
|
||||
{
|
||||
if (vault_http_ctx)
|
||||
{
|
||||
http_destroy(vault_http_cli);
|
||||
http_context_destroy(vault_http_ctx);
|
||||
vault_http_cli = NULL;
|
||||
vault_http_ctx = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
void cluster_client_t::vault_load_keys()
|
||||
{
|
||||
if (vault_loading || !vault_key_load_queue.size())
|
||||
{
|
||||
return;
|
||||
}
|
||||
if (!vault_http_ctx)
|
||||
{
|
||||
std::string error;
|
||||
vault_http_ctx = http_context_init(tfd, vault_client_cert, vault_client_key, vault_ca, true, error);
|
||||
if (!vault_http_ctx)
|
||||
{
|
||||
fprintf(stderr, "Failed to initialize HTTP context for Vault: %s\n", error.c_str());
|
||||
exit(1);
|
||||
}
|
||||
vault_http_cli = http_init(vault_http_ctx);
|
||||
}
|
||||
if (!vault_check_token())
|
||||
{
|
||||
return;
|
||||
}
|
||||
std::string key_id = vault_key_load_queue[0];
|
||||
vault_key_load_queue.erase(vault_key_load_queue.begin());
|
||||
vault_loading = true;
|
||||
http_get(
|
||||
vault_http_cli, vault_url+vault_secret_api_path+key_id.substr(strlen(VAULT_KEY_PREFIX)), "X-Vault-Token: "+vault_token+"\r\n",
|
||||
(http_options_t){ .timeout = (int)vault_timeout_ms, .keepalive = true },
|
||||
[this, key_id](http_message_t *response)
|
||||
{
|
||||
vault_loading = false;
|
||||
std::string err;
|
||||
json11::Json data;
|
||||
response->parse_json_response(err, data);
|
||||
vault_parse_secret(key_id, err, data);
|
||||
}
|
||||
);
|
||||
}
|
||||
@@ -131,7 +131,6 @@ void writeback_cache_t::copy_write(cluster_op_t *op, int state, uint64_t new_flu
|
||||
writeback_bytes -= op->len;
|
||||
}
|
||||
writeback_queue_size++;
|
||||
writeback_queue.push_back({ op->inode, new_end });
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -166,7 +165,6 @@ void writeback_cache_t::copy_write(cluster_op_t *op, int state, uint64_t new_flu
|
||||
{
|
||||
writeback_queue_size++;
|
||||
}
|
||||
writeback_queue.push_back({ op->inode, new_end });
|
||||
}
|
||||
auto new_dirty_it = dirty_buffers.emplace_hint(dirty_it, (object_id){
|
||||
.inode = op->inode,
|
||||
|
||||
@@ -1,12 +1,16 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||
|
||||
#include <assert.h>
|
||||
#include "malloc_or_die.h"
|
||||
#include "osd_ops.h"
|
||||
#include "msgr_op.h"
|
||||
#include "pg_states.h"
|
||||
#include "etcd_state_client.h"
|
||||
#ifndef __MOCK__
|
||||
#include "addr_util.h"
|
||||
#include "http_client.h"
|
||||
#endif
|
||||
#include "str_util.h"
|
||||
#include "json_util.h"
|
||||
|
||||
@@ -17,8 +21,33 @@ etcd_state_client_t::~etcd_state_client_t()
|
||||
delete watch;
|
||||
}
|
||||
watches.clear();
|
||||
etcd_watches_initialised = -1;
|
||||
#ifndef __MOCK__
|
||||
stop_ws_keepalive();
|
||||
if (etcd_watch_ws)
|
||||
{
|
||||
http_destroy(etcd_watch_ws);
|
||||
etcd_watch_ws = NULL;
|
||||
}
|
||||
if (keepalive_client)
|
||||
{
|
||||
http_destroy(keepalive_client);
|
||||
keepalive_client = NULL;
|
||||
}
|
||||
if (http_ctx)
|
||||
{
|
||||
http_context_destroy(http_ctx);
|
||||
http_ctx = NULL;
|
||||
}
|
||||
#endif
|
||||
if (load_pgs_timer_id >= 0)
|
||||
{
|
||||
tfd->clear_timer(load_pgs_timer_id);
|
||||
load_pgs_timer_id = -1;
|
||||
}
|
||||
}
|
||||
|
||||
#ifndef __MOCK__
|
||||
etcd_kv_t etcd_state_client_t::parse_etcd_kv(const json11::Json & kv_json)
|
||||
{
|
||||
etcd_kv_t kv;
|
||||
@@ -92,6 +121,99 @@ bool etcd_state_client_t::check_image_perm(const std::shared_ptr<user_info_t> &
|
||||
return write ? (perm_item.perm == user_perm_t::OWNER) : (perm_item.perm != user_perm_t::DENY);
|
||||
}
|
||||
|
||||
http_context_t *etcd_state_client_t::get_http_ctx()
|
||||
{
|
||||
if (!http_ctx)
|
||||
{
|
||||
std::string error;
|
||||
http_ctx = http_context_init(tfd, etcd_client_cert, etcd_client_key, etcd_ca, true, error);
|
||||
if (!http_ctx)
|
||||
{
|
||||
fprintf(stderr, "Failed to initialize HTTP context: %s\n", error.c_str());
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
return http_ctx;
|
||||
}
|
||||
|
||||
void etcd_state_client_t::etcd_call_oneshot(const std::string & etcd_url, const std::string & api, json11::Json payload,
|
||||
int timeout, std::function<void(std::string, json11::Json)> callback)
|
||||
{
|
||||
auto http_cli = http_init(get_http_ctx());
|
||||
http_json_post(http_cli, etcd_url+api, payload, "", { .timeout = timeout }, [http_cli, callback](http_message_t *response)
|
||||
{
|
||||
std::string err;
|
||||
json11::Json data;
|
||||
response->parse_json_response(err, data);
|
||||
callback(err, data);
|
||||
http_destroy(http_cli);
|
||||
});
|
||||
}
|
||||
|
||||
void etcd_state_client_t::etcd_call(const std::string & api, json11::Json payload, int timeout,
|
||||
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
|
||||
{
|
||||
pick_next_etcd([=]()
|
||||
{
|
||||
etcd_call_selected(api, payload, timeout, retries, interval, callback);
|
||||
});
|
||||
}
|
||||
|
||||
void etcd_state_client_t::etcd_call_selected(const std::string & api, json11::Json payload, int timeout,
|
||||
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
|
||||
{
|
||||
const auto & url = selected_etcd_url;
|
||||
std::string req = payload.dump();
|
||||
req = "POST "+url.path+api+" HTTP/1.1\r\n"
|
||||
"Host: "+url.hostname+"\r\n"
|
||||
"Content-Type: application/json\r\n"
|
||||
"Content-Length: "+std::to_string(req.size())+"\r\n"
|
||||
"Connection: keep-alive\r\n"
|
||||
"Keep-Alive: timeout="+std::to_string(etcd_keepalive_timeout)+"\r\n"
|
||||
"\r\n"+req;
|
||||
retries--;
|
||||
auto cb = [this, api, payload, timeout, retries, interval, callback,
|
||||
cur_addr = url.addr](http_message_t *response)
|
||||
{
|
||||
std::string err;
|
||||
json11::Json data;
|
||||
response->parse_json_response(err, data);
|
||||
if (err != "")
|
||||
{
|
||||
if (cur_addr == selected_etcd_url.addr)
|
||||
selected_etcd_url = (http_url_t){};
|
||||
if (retries > 0)
|
||||
{
|
||||
if (this->log_level > 0)
|
||||
{
|
||||
fprintf(
|
||||
stderr, "Warning: etcd request failed: %s, retrying %d more times\n",
|
||||
err.c_str(), retries
|
||||
);
|
||||
}
|
||||
if (interval > 0)
|
||||
{
|
||||
// FIXME: Prevent destruction of etcd_state_client if timers or requests are active
|
||||
tfd->set_timer(interval, false, [this, api, payload, timeout, retries, interval, callback](int)
|
||||
{
|
||||
etcd_call(api, payload, timeout, retries, interval, callback);
|
||||
});
|
||||
}
|
||||
else
|
||||
etcd_call(api, payload, timeout, retries, interval, callback);
|
||||
}
|
||||
else
|
||||
callback(err, data);
|
||||
}
|
||||
else
|
||||
callback(err, data);
|
||||
};
|
||||
if (!keepalive_client)
|
||||
keepalive_client = http_init(get_http_ctx());
|
||||
http_request(keepalive_client, url.addr, req, { .timeout = timeout, .keepalive = true, .ssl = url.ssl }, cb);
|
||||
}
|
||||
|
||||
|
||||
void etcd_state_client_t::add_etcd_url(std::string etcd_address)
|
||||
{
|
||||
if (etcd_address.size() > 0)
|
||||
@@ -202,6 +324,7 @@ void etcd_state_client_t::parse_config(const json11::Json & config)
|
||||
if (this->etcd_keepalive_timeout < 30)
|
||||
this->etcd_keepalive_timeout = 30;
|
||||
}
|
||||
auto old_etcd_ws_keepalive_interval = this->etcd_ws_keepalive_interval;
|
||||
this->etcd_ws_keepalive_interval = config["etcd_ws_keepalive_interval"].uint64_value();
|
||||
if (this->etcd_ws_keepalive_interval <= 0)
|
||||
{
|
||||
@@ -227,9 +350,347 @@ void etcd_state_client_t::parse_config(const json11::Json & config)
|
||||
{
|
||||
this->etcd_min_reload_interval = 50;
|
||||
}
|
||||
if (this->etcd_ws_keepalive_interval != old_etcd_ws_keepalive_interval && ws_keepalive_timer >= 0)
|
||||
{
|
||||
#ifndef __MOCK__
|
||||
stop_ws_keepalive();
|
||||
start_ws_keepalive();
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
void etcd_state_client_t::load_global_config(std::function<void(const std::string & error)> cb)
|
||||
void etcd_state_client_t::pick_next_etcd(std::function<void()> cb)
|
||||
{
|
||||
if (!etcd_addresses.size() && !etcd_local.size())
|
||||
{
|
||||
fprintf(stderr, "etcd_address is missing in Vitastor configuration\n");
|
||||
exit(1);
|
||||
}
|
||||
if (selected_etcd_url.addr != "")
|
||||
{
|
||||
cb();
|
||||
return;
|
||||
}
|
||||
if (etcd_urls_to_try.size() != 0)
|
||||
{
|
||||
selected_etcd_url = std::move(etcd_urls_to_try[0]);
|
||||
etcd_urls_to_try.erase(etcd_urls_to_try.begin());
|
||||
cb();
|
||||
return;
|
||||
}
|
||||
on_resolve_queue.push_back(std::move(cb));
|
||||
if (on_resolve_queue.size() > 1)
|
||||
{
|
||||
// Already resolving
|
||||
return;
|
||||
}
|
||||
assert(!resolve_count);
|
||||
local_to_try = 0;
|
||||
for (auto & url: etcd_local_addr_urls)
|
||||
{
|
||||
// Prefer local IPs, if any
|
||||
etcd_urls_to_try.push_back(url);
|
||||
local_to_try++;
|
||||
}
|
||||
for (auto & url: etcd_nonlocal_addr_urls)
|
||||
{
|
||||
etcd_urls_to_try.push_back(url);
|
||||
}
|
||||
resolve_count++;
|
||||
for (auto & url: etcd_name_urls)
|
||||
{
|
||||
resolve_count++;
|
||||
http_resolve(get_http_ctx(), url.ssl, url.addr, [this, url](const std::string & error, const std::vector<std::string>& addresses)
|
||||
{
|
||||
if (error != "")
|
||||
fprintf(stderr, "Error resolving %s: %s\n", url.addr.c_str(), error.c_str());
|
||||
for (auto & addr: addresses)
|
||||
{
|
||||
auto url_copy = url;
|
||||
url_copy.addr = addr;
|
||||
if (local_ips.find(addr) != local_ips.end())
|
||||
{
|
||||
etcd_urls_to_try.insert(etcd_urls_to_try.begin(), std::move(url_copy));
|
||||
local_to_try++;
|
||||
}
|
||||
else
|
||||
etcd_urls_to_try.push_back(std::move(url_copy));
|
||||
}
|
||||
resolve_count--;
|
||||
if (!resolve_count)
|
||||
pick_next_etcd_on_resolve();
|
||||
});
|
||||
}
|
||||
resolve_count--;
|
||||
if (!resolve_count)
|
||||
{
|
||||
pick_next_etcd_on_resolve();
|
||||
}
|
||||
}
|
||||
|
||||
void etcd_state_client_t::pick_next_etcd_on_resolve()
|
||||
{
|
||||
if (!etcd_urls_to_try.size())
|
||||
{
|
||||
fprintf(stderr, "None of etcd_address could be resolved\n");
|
||||
exit(1);
|
||||
}
|
||||
if (!rand_initialized)
|
||||
{
|
||||
timespec tv;
|
||||
clock_gettime(CLOCK_REALTIME, &tv);
|
||||
srand48(tv.tv_sec*1000000000 + tv.tv_nsec);
|
||||
rand_initialized = true;
|
||||
}
|
||||
// Shuffle addresses
|
||||
for (size_t i = etcd_urls_to_try.size()-1; i > local_to_try; i--)
|
||||
{
|
||||
size_t j = local_to_try + lrand48() % (i - local_to_try);
|
||||
if (j != i)
|
||||
std::swap(etcd_urls_to_try[i], etcd_urls_to_try[j]);
|
||||
}
|
||||
selected_etcd_url = std::move(etcd_urls_to_try[0]);
|
||||
etcd_urls_to_try.erase(etcd_urls_to_try.begin());
|
||||
auto cbs = std::move(on_resolve_queue);
|
||||
for (auto cb: cbs)
|
||||
{
|
||||
cb();
|
||||
}
|
||||
}
|
||||
|
||||
void etcd_state_client_t::start_etcd_watcher()
|
||||
{
|
||||
pick_next_etcd([this]()
|
||||
{
|
||||
start_etcd_watcher_selected();
|
||||
});
|
||||
}
|
||||
|
||||
void etcd_state_client_t::start_etcd_watcher_selected()
|
||||
{
|
||||
const auto & url = selected_etcd_url;
|
||||
etcd_watches_initialised = 0;
|
||||
ws_alive = 1;
|
||||
if (this->log_level > 1)
|
||||
{
|
||||
fprintf(stderr, "Trying to connect to etcd websocket at %s%s%s (hostname %s), watch from revision %ju/%ju/%ju\n",
|
||||
url.ssl ? "https://" : "http://", url.addr.c_str(), url.path.c_str(), url.hostname.c_str(),
|
||||
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
|
||||
}
|
||||
if (!etcd_watch_ws)
|
||||
etcd_watch_ws = http_init(get_http_ctx());
|
||||
else
|
||||
http_close(etcd_watch_ws);
|
||||
open_websocket(etcd_watch_ws, url.addr, url.hostname, url.path+"/watch", { .timeout = etcd_slow_timeout, .ssl = url.ssl },
|
||||
[this, cur_addr = url.addr](http_message_t *msg)
|
||||
{
|
||||
if (msg->body.length())
|
||||
{
|
||||
ws_alive = 1;
|
||||
std::string json_err;
|
||||
json11::Json data = json11::Json::parse(msg->body, json_err);
|
||||
if (json_err != "")
|
||||
{
|
||||
fprintf(stderr, "Bad JSON in etcd event: %s, ignoring event\n", json_err.c_str());
|
||||
}
|
||||
else
|
||||
{
|
||||
uint64_t watch_id = data["result"]["watch_id"].uint64_value();
|
||||
if (data["result"]["created"].bool_value())
|
||||
{
|
||||
if (watch_id == ETCD_CONFIG_WATCH_ID ||
|
||||
watch_id == ETCD_PG_STATE_WATCH_ID ||
|
||||
watch_id == ETCD_OSD_STATE_WATCH_ID)
|
||||
{
|
||||
etcd_watches_initialised++;
|
||||
}
|
||||
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES && this->log_level > 0)
|
||||
{
|
||||
fprintf(stderr, "Successfully subscribed to etcd at %s, revision %ju/%ju/%ju\n", cur_addr.c_str(),
|
||||
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
|
||||
}
|
||||
}
|
||||
if (data["result"]["canceled"].bool_value())
|
||||
{
|
||||
// etcd watch canceled, maybe because the revision was compacted
|
||||
if (data["result"]["compact_revision"].uint64_value())
|
||||
{
|
||||
// we may miss events if we proceed
|
||||
// so we should restart from the beginning if we can
|
||||
if (on_reload_hook != NULL)
|
||||
{
|
||||
// check to not trigger on_reload_hook multiple times
|
||||
if (etcd_watch_ws != NULL)
|
||||
{
|
||||
fprintf(stderr, "Revisions before %ju were compacted by etcd, reloading state\n",
|
||||
data["result"]["compact_revision"].uint64_value());
|
||||
http_close(etcd_watch_ws);
|
||||
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = 0;
|
||||
on_reload_hook();
|
||||
}
|
||||
return;
|
||||
}
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "Revisions before %ju were compacted by etcd, exiting\n",
|
||||
data["result"]["compact_revision"].uint64_value());
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "Watch canceled by etcd, reason: %s, exiting\n", data["result"]["cancel_reason"].string_value().c_str());
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
// Save revision only if it's present in the message - because sometimes etcd sends something without a header, like:
|
||||
// {"error": {"grpc_code": 14, "http_code": 503, "http_status": "Service Unavailable", "message": "error reading from server: EOF"}}
|
||||
// Also don't save revision from the initial created: true messages because they always contain the latest revision
|
||||
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES &&
|
||||
!data["result"]["header"]["revision"].is_null() &&
|
||||
!data["result"]["created"].bool_value())
|
||||
{
|
||||
// Restart watchers from the same revision number as in the last received message,
|
||||
// not from the next one to protect against revision being split into multiple messages,
|
||||
// even though etcd guarantees not to do that **within a single watcher** without fragment=true:
|
||||
// https://etcd.io/docs/v3.5/learning/api_guarantees/#watch-apis
|
||||
// Revision contents are ALWAYS split into separate messages for different watchers though!
|
||||
// So generally we have to resume each watcher from its own revision...
|
||||
// Progress messages may have watch_id=-1 if sent on behalf of multiple watchers though.
|
||||
// And antietcd has an advanced semantic which merges the same revision for all watchers
|
||||
// into one message and just omits watch_id.
|
||||
// So we also have to handle the case where watch_id is -1 or not present (0).
|
||||
auto watch_rev = data["result"]["header"]["revision"].uint64_value();
|
||||
if (!watch_id || watch_id == UINT64_MAX)
|
||||
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = watch_rev;
|
||||
else if (watch_id == ETCD_CONFIG_WATCH_ID)
|
||||
etcd_watch_revision_config = watch_rev;
|
||||
else if (watch_id == ETCD_PG_STATE_WATCH_ID)
|
||||
etcd_watch_revision_pg = watch_rev;
|
||||
else if (watch_id == ETCD_OSD_STATE_WATCH_ID)
|
||||
etcd_watch_revision_osd = watch_rev;
|
||||
etcd_urls_to_try.clear();
|
||||
}
|
||||
// First gather all changes into a hash to remove multiple overwrites
|
||||
std::map<std::string, etcd_kv_t> changes;
|
||||
for (auto & ev: data["result"]["events"].array_items())
|
||||
{
|
||||
auto kv = parse_etcd_kv(ev["kv"]);
|
||||
if (kv.key != "")
|
||||
{
|
||||
changes[kv.key] = kv;
|
||||
}
|
||||
}
|
||||
for (auto & kv: changes)
|
||||
{
|
||||
if (this->log_level > 3)
|
||||
{
|
||||
fprintf(stderr, "Incoming event: %s -> %s\n", kv.first.c_str(), kv.second.value.dump().c_str());
|
||||
}
|
||||
parse_state(kv.second);
|
||||
}
|
||||
// React to changes
|
||||
if (on_change_hook != NULL)
|
||||
{
|
||||
on_change_hook(changes);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (msg->eof)
|
||||
{
|
||||
fprintf(stderr, "Disconnected from etcd %s\n", cur_addr.c_str());
|
||||
if (cur_addr == selected_etcd_url.addr)
|
||||
selected_etcd_url = (http_url_t){};
|
||||
if (etcd_watches_initialised == 0)
|
||||
{
|
||||
// Connection not established, retry in <etcd_quick_timeout>
|
||||
tfd->set_timer(etcd_quick_timeout, false, [this](int)
|
||||
{
|
||||
start_etcd_watcher();
|
||||
});
|
||||
}
|
||||
else if (etcd_watches_initialised > 0)
|
||||
{
|
||||
// Connection was live, retry immediately
|
||||
etcd_watches_initialised = 0;
|
||||
start_etcd_watcher();
|
||||
}
|
||||
}
|
||||
});
|
||||
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
||||
{ "create_request", json11::Json::object {
|
||||
{ "key", base64_encode(etcd_prefix+"/config/") },
|
||||
{ "range_end", base64_encode(etcd_prefix+"/config0") },
|
||||
{ "start_revision", etcd_watch_revision_config },
|
||||
{ "watch_id", ETCD_CONFIG_WATCH_ID },
|
||||
{ "progress_notify", true },
|
||||
} }
|
||||
}).dump());
|
||||
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
||||
{ "create_request", json11::Json::object {
|
||||
{ "key", base64_encode(etcd_prefix+"/osd/state/") },
|
||||
{ "range_end", base64_encode(etcd_prefix+"/osd/state0") },
|
||||
{ "start_revision", etcd_watch_revision_osd },
|
||||
{ "watch_id", ETCD_OSD_STATE_WATCH_ID },
|
||||
{ "progress_notify", true },
|
||||
} }
|
||||
}).dump());
|
||||
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
||||
{ "create_request", json11::Json::object {
|
||||
{ "key", base64_encode(etcd_prefix+"/pg/") },
|
||||
{ "range_end", base64_encode(etcd_prefix+"/pg0") },
|
||||
{ "start_revision", etcd_watch_revision_pg },
|
||||
{ "watch_id", ETCD_PG_STATE_WATCH_ID },
|
||||
{ "progress_notify", true },
|
||||
} }
|
||||
}).dump());
|
||||
// FIXME: Do not watch /pg/history/ at all in client code (not in OSD)
|
||||
if (on_start_watcher_hook)
|
||||
{
|
||||
on_start_watcher_hook(etcd_watch_ws);
|
||||
}
|
||||
start_ws_keepalive();
|
||||
}
|
||||
|
||||
void etcd_state_client_t::stop_ws_keepalive()
|
||||
{
|
||||
if (ws_keepalive_timer >= 0)
|
||||
{
|
||||
tfd->clear_timer(ws_keepalive_timer);
|
||||
ws_keepalive_timer = -1;
|
||||
}
|
||||
}
|
||||
|
||||
void etcd_state_client_t::start_ws_keepalive()
|
||||
{
|
||||
if (ws_keepalive_timer < 0)
|
||||
{
|
||||
ws_keepalive_timer = tfd->set_timer(etcd_ws_keepalive_interval*1000, true, [this](int)
|
||||
{
|
||||
if (!etcd_watch_ws || etcd_watches_initialised < ETCD_TOTAL_WATCHES)
|
||||
{
|
||||
// Do nothing
|
||||
}
|
||||
else if (!ws_alive)
|
||||
{
|
||||
if (this->log_level > 0)
|
||||
{
|
||||
fprintf(stderr, "Websocket ping failed, disconnecting from etcd %s\n", selected_etcd_url.addr.c_str());
|
||||
}
|
||||
start_etcd_watcher();
|
||||
}
|
||||
else
|
||||
{
|
||||
ws_alive = 0;
|
||||
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
||||
{ "progress_request", json11::Json::object { } }
|
||||
}).dump());
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
void etcd_state_client_t::load_global_config()
|
||||
{
|
||||
json11::Json::object req = { { "success", json11::Json::array {
|
||||
json11::Json::object {
|
||||
@@ -243,12 +704,22 @@ void etcd_state_client_t::load_global_config(std::function<void(const std::strin
|
||||
} }
|
||||
},
|
||||
} } };
|
||||
etcd_txn(req, etcd_quick_timeout, max_etcd_attempts, 0, [this, cb](std::string err, json11::Json data)
|
||||
etcd_txn(req, etcd_quick_timeout, max_etcd_attempts, 0, [this](std::string err, json11::Json data)
|
||||
{
|
||||
if (err != "")
|
||||
{
|
||||
fprintf(stderr, "Error reading configuration from etcd: %s\n", err.c_str());
|
||||
cb(err);
|
||||
if (infinite_start)
|
||||
{
|
||||
tfd->set_timer(etcd_slow_timeout, false, [this](int timer_id)
|
||||
{
|
||||
load_global_config();
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
exit(1);
|
||||
}
|
||||
return;
|
||||
}
|
||||
json11::Json config_kv = data["responses"][0]["response_range"]["kvs"][0];
|
||||
@@ -279,12 +750,28 @@ void etcd_state_client_t::load_global_config(std::function<void(const std::strin
|
||||
parse_state(kv);
|
||||
}
|
||||
on_load_config_hook(global_config);
|
||||
cb("");
|
||||
});
|
||||
}
|
||||
|
||||
void etcd_state_client_t::load_pgs(std::function<void(const std::string &)> cb)
|
||||
void etcd_state_client_t::load_pgs()
|
||||
{
|
||||
timespec tv;
|
||||
clock_gettime(CLOCK_REALTIME, &tv);
|
||||
uint64_t ms_passed = (tv.tv_sec-etcd_last_reload.tv_sec)*1000 + (tv.tv_nsec-etcd_last_reload.tv_nsec)/1000000;
|
||||
if (ms_passed < etcd_min_reload_interval)
|
||||
{
|
||||
if (load_pgs_timer_id < 0)
|
||||
{
|
||||
load_pgs_timer_id = tfd->set_timer(etcd_min_reload_interval+50-ms_passed, false, [this](int) { load_pgs(); });
|
||||
}
|
||||
return;
|
||||
}
|
||||
etcd_last_reload = tv;
|
||||
if (load_pgs_timer_id >= 0)
|
||||
{
|
||||
tfd->clear_timer(load_pgs_timer_id);
|
||||
load_pgs_timer_id = -1;
|
||||
}
|
||||
json11::Json::array txn = {
|
||||
json11::Json::object {
|
||||
{ "request_range", json11::Json::object {
|
||||
@@ -322,13 +809,16 @@ void etcd_state_client_t::load_pgs(std::function<void(const std::string &)> cb)
|
||||
{
|
||||
req["compare"] = checks;
|
||||
}
|
||||
etcd_txn_slow(req, [this, cb](std::string err, json11::Json data)
|
||||
etcd_txn_slow(req, [this](std::string err, json11::Json data)
|
||||
{
|
||||
if (err != "")
|
||||
{
|
||||
// Retry indefinitely
|
||||
fprintf(stderr, "Error loading PGs from etcd: %s\n", err.c_str());
|
||||
cb(err);
|
||||
tfd->set_timer(etcd_slow_timeout, false, [this](int timer_id)
|
||||
{
|
||||
load_pgs();
|
||||
});
|
||||
return;
|
||||
}
|
||||
if (!data["succeeded"].bool_value())
|
||||
@@ -356,9 +846,24 @@ void etcd_state_client_t::load_pgs(std::function<void(const std::string &)> cb)
|
||||
}
|
||||
clean_nonexistent_pgs();
|
||||
on_load_pgs_hook(true);
|
||||
cb("");
|
||||
start_etcd_watcher();
|
||||
});
|
||||
}
|
||||
#else
|
||||
void etcd_state_client_t::parse_config(const json11::Json & config)
|
||||
{
|
||||
}
|
||||
|
||||
void etcd_state_client_t::load_global_config()
|
||||
{
|
||||
json11::Json::object global_config;
|
||||
on_load_config_hook(global_config);
|
||||
}
|
||||
|
||||
void etcd_state_client_t::load_pgs()
|
||||
{
|
||||
}
|
||||
#endif
|
||||
|
||||
void etcd_state_client_t::reset_pg_exists()
|
||||
{
|
||||
@@ -471,8 +976,7 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
|
||||
if (pc.pg_size < 1 ||
|
||||
pool_item.second["pg_size"].uint64_value() < 3 &&
|
||||
(pc.scheme == POOL_SCHEME_XOR || pc.scheme == POOL_SCHEME_EC) ||
|
||||
// limit is 64 because osd_peering_pg.cpp uses a 64-bit mask for has_roles
|
||||
pool_item.second["pg_size"].uint64_value() > 64)
|
||||
pool_item.second["pg_size"].uint64_value() > 256)
|
||||
{
|
||||
fprintf(stderr, "Pool %u has invalid pg_size, skipping pool\n", pool_id);
|
||||
continue;
|
||||
@@ -814,14 +1318,8 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
|
||||
else if (key.substr(0, etcd_prefix.length()+11) == etcd_prefix+"/osd/state/")
|
||||
{
|
||||
// <etcd_prefix>/osd/state/%d
|
||||
osd_num_t peer_osd = 0;
|
||||
char null_byte = 0;
|
||||
int scanned = sscanf(key.c_str() + etcd_prefix.length()+11, "%ju%c", &peer_osd, &null_byte);
|
||||
if (scanned != 1 || !peer_osd)
|
||||
{
|
||||
fprintf(stderr, "Bad etcd key %s, ignoring\n", key.c_str());
|
||||
}
|
||||
else
|
||||
osd_num_t peer_osd = std::stoull(key.substr(etcd_prefix.length()+11));
|
||||
if (peer_osd > 0)
|
||||
{
|
||||
if (value.is_object() && value["state"] == "up")
|
||||
{
|
||||
|
||||
@@ -136,6 +136,7 @@ struct user_info_t
|
||||
};
|
||||
|
||||
struct http_co_t;
|
||||
struct http_context_t;
|
||||
|
||||
struct __attribute__((visibility("default"))) etcd_state_client_t
|
||||
{
|
||||
@@ -146,13 +147,21 @@ protected:
|
||||
std::vector<http_url_t> etcd_local_addr_urls;
|
||||
std::vector<http_url_t> etcd_nonlocal_addr_urls;
|
||||
std::vector<http_url_t> etcd_name_urls;
|
||||
size_t local_to_try = 0;
|
||||
std::vector<http_url_t> etcd_urls_to_try;
|
||||
http_url_t selected_etcd_url;
|
||||
size_t resolve_count = 0;
|
||||
std::vector<inode_watch_t*> watches;
|
||||
std::set<osd_num_t> seen_peers;
|
||||
std::vector<std::function<void()>> on_resolve_queue;
|
||||
bool new_pg_config = false;
|
||||
|
||||
int ws_keepalive_timer = -1;
|
||||
int ws_alive = 0;
|
||||
bool rand_initialized = false;
|
||||
void add_etcd_url(std::string);
|
||||
void reset_pg_exists();
|
||||
void clean_nonexistent_pgs();
|
||||
void pick_next_etcd(std::function<void()> cb);
|
||||
void pick_next_etcd_on_resolve();
|
||||
void etcd_call_selected(const std::string & api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
|
||||
void start_etcd_watcher_selected();
|
||||
public:
|
||||
int etcd_keepalive_timeout = 30;
|
||||
int etcd_ws_keepalive_interval = 5;
|
||||
@@ -171,13 +180,19 @@ public:
|
||||
std::string etcd_client_key;
|
||||
std::string etcd_ca;
|
||||
int log_level = 0;
|
||||
timerfd_manager_t *tfd = NULL;
|
||||
|
||||
http_context_t *http_ctx = NULL;
|
||||
http_co_t *etcd_watch_ws = NULL, *keepalive_client = NULL;
|
||||
int etcd_watches_initialised = 0;
|
||||
uint64_t etcd_watch_revision_config = 0;
|
||||
uint64_t etcd_watch_revision_osd = 0;
|
||||
uint64_t etcd_watch_revision_pg = 0;
|
||||
|
||||
timespec etcd_last_reload = {};
|
||||
int load_pgs_timer_id = -1;
|
||||
std::map<pool_id_t, pool_config_t> pool_config;
|
||||
std::map<osd_num_t, json11::Json> peer_states;
|
||||
std::set<osd_num_t> seen_peers;
|
||||
std::map<inode_t, inode_config_t> inode_config;
|
||||
std::map<std::string, inode_t> inode_by_name;
|
||||
robin_hood::unordered_flat_map<std::string, std::shared_ptr<user_info_t>> user_info;
|
||||
@@ -205,23 +220,25 @@ public:
|
||||
std::vector<std::string> get_addresses();
|
||||
std::shared_ptr<user_info_t> get_user(const std::string & username);
|
||||
bool check_image_perm(const std::shared_ptr<user_info_t> & user_info, inode_t inode_num, bool write);
|
||||
virtual void etcd_call_oneshot(const std::string & etcd_address, const std::string & api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback) = 0;
|
||||
virtual void etcd_call(const std::string & api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback) = 0;
|
||||
http_context_t *get_http_ctx();
|
||||
void etcd_call_oneshot(const std::string & etcd_address, const std::string & api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback);
|
||||
void etcd_call(const std::string & api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
|
||||
void etcd_txn(json11::Json txn, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
|
||||
void etcd_txn_slow(json11::Json txn, std::function<void(std::string, json11::Json)> callback);
|
||||
virtual void etcd_add_watch(json11::Json watch) = 0;
|
||||
virtual std::string get_username() = 0;
|
||||
void load_global_config(std::function<void(const std::string &)> cb);
|
||||
virtual void load_global_config() = 0;
|
||||
void load_pgs(std::function<void(const std::string &)> cb);
|
||||
virtual void load_pgs() = 0;
|
||||
void start_etcd_watcher();
|
||||
void stop_ws_keepalive();
|
||||
void start_ws_keepalive();
|
||||
void load_global_config();
|
||||
void load_pgs();
|
||||
void reset_pg_exists();
|
||||
void clean_nonexistent_pgs();
|
||||
void parse_state(const etcd_kv_t & kv);
|
||||
virtual void parse_config(const json11::Json & config);
|
||||
void parse_config(const json11::Json & config);
|
||||
void insert_inode_config(const inode_config_t & cfg);
|
||||
inode_watch_t* watch_inode(std::string name);
|
||||
void close_watch(inode_watch_t* watch);
|
||||
int address_count();
|
||||
virtual ~etcd_state_client_t();
|
||||
~etcd_state_client_t();
|
||||
|
||||
static uint32_t parse_immediate_commit(const std::string & immediate_commit_str, uint32_t default_value);
|
||||
static uint32_t parse_scheme(const std::string & scheme_str);
|
||||
|
||||
@@ -1,545 +0,0 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||
|
||||
#include <assert.h>
|
||||
#include "etcd_state_client_http.h"
|
||||
#include "addr_util.h"
|
||||
#include "http_client.h"
|
||||
#include "str_util.h"
|
||||
|
||||
etcd_state_client_http_t::etcd_state_client_http_t(timerfd_manager_t *tfd)
|
||||
{
|
||||
this->tfd = tfd;
|
||||
}
|
||||
|
||||
etcd_state_client_http_t::~etcd_state_client_http_t()
|
||||
{
|
||||
stop_ws_keepalive();
|
||||
if (etcd_watch_ws)
|
||||
{
|
||||
http_destroy(etcd_watch_ws);
|
||||
etcd_watch_ws = NULL;
|
||||
}
|
||||
if (keepalive_client)
|
||||
{
|
||||
http_destroy(keepalive_client);
|
||||
keepalive_client = NULL;
|
||||
}
|
||||
if (load_pgs_timer_id >= 0)
|
||||
{
|
||||
tfd->clear_timer(load_pgs_timer_id);
|
||||
load_pgs_timer_id = -1;
|
||||
}
|
||||
if (http_ctx)
|
||||
{
|
||||
http_context_destroy(http_ctx);
|
||||
http_ctx = NULL;
|
||||
}
|
||||
etcd_watches_initialised = -1;
|
||||
}
|
||||
|
||||
void etcd_state_client_http_t::etcd_add_watch(json11::Json watch)
|
||||
{
|
||||
if (etcd_watch_ws)
|
||||
{
|
||||
http_post_message(etcd_watch_ws, WS_TEXT, watch.dump());
|
||||
}
|
||||
}
|
||||
|
||||
std::string etcd_state_client_http_t::get_username()
|
||||
{
|
||||
return http_context_get_ssl_cn(get_http_ctx());
|
||||
}
|
||||
|
||||
http_context_t *etcd_state_client_http_t::get_http_ctx()
|
||||
{
|
||||
if (!http_ctx)
|
||||
{
|
||||
std::string error;
|
||||
http_ctx = http_context_init(tfd, etcd_client_cert, etcd_client_key, etcd_ca, true, error);
|
||||
if (!http_ctx)
|
||||
{
|
||||
fprintf(stderr, "Failed to initialize HTTP context: %s\n", error.c_str());
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
return http_ctx;
|
||||
}
|
||||
|
||||
void etcd_state_client_http_t::etcd_call_oneshot(const std::string & etcd_url, const std::string & api, json11::Json payload,
|
||||
int timeout, std::function<void(std::string, json11::Json)> callback)
|
||||
{
|
||||
auto http_cli = http_init(get_http_ctx());
|
||||
http_json_post(http_cli, etcd_url+api, payload, "", { .timeout = timeout }, [http_cli, callback](http_message_t *response)
|
||||
{
|
||||
std::string err;
|
||||
json11::Json data;
|
||||
response->parse_json_response(err, data);
|
||||
callback(err, data);
|
||||
http_destroy(http_cli);
|
||||
});
|
||||
}
|
||||
|
||||
void etcd_state_client_http_t::etcd_call(const std::string & api, json11::Json payload, int timeout,
|
||||
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
|
||||
{
|
||||
pick_next_etcd([=]()
|
||||
{
|
||||
etcd_call_selected(api, payload, timeout, retries, interval, callback);
|
||||
});
|
||||
}
|
||||
|
||||
void etcd_state_client_http_t::etcd_call_selected(const std::string & api, json11::Json payload, int timeout,
|
||||
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
|
||||
{
|
||||
const auto & url = selected_etcd_url;
|
||||
std::string req = payload.dump();
|
||||
req = "POST "+url.path+api+" HTTP/1.1\r\n"
|
||||
"Host: "+url.hostname+"\r\n"
|
||||
"Content-Type: application/json\r\n"
|
||||
"Content-Length: "+std::to_string(req.size())+"\r\n"
|
||||
"Connection: keep-alive\r\n"
|
||||
"Keep-Alive: timeout="+std::to_string(etcd_keepalive_timeout)+"\r\n"
|
||||
"\r\n"+req;
|
||||
retries--;
|
||||
auto cb = [this, api, payload, timeout, retries, interval, callback,
|
||||
cur_addr = url.addr](http_message_t *response)
|
||||
{
|
||||
std::string err;
|
||||
json11::Json data;
|
||||
response->parse_json_response(err, data);
|
||||
if (err != "")
|
||||
{
|
||||
if (cur_addr == selected_etcd_url.addr)
|
||||
selected_etcd_url = (http_url_t){};
|
||||
if (retries > 0)
|
||||
{
|
||||
if (this->log_level > 0)
|
||||
{
|
||||
fprintf(
|
||||
stderr, "Warning: etcd request failed: %s, retrying %d more times\n",
|
||||
err.c_str(), retries
|
||||
);
|
||||
}
|
||||
if (interval > 0)
|
||||
{
|
||||
// FIXME: Prevent destruction of etcd_state_client if timers or requests are active
|
||||
tfd->set_timer(interval, false, [this, api, payload, timeout, retries, interval, callback](int)
|
||||
{
|
||||
etcd_call(api, payload, timeout, retries, interval, callback);
|
||||
});
|
||||
}
|
||||
else
|
||||
etcd_call(api, payload, timeout, retries, interval, callback);
|
||||
}
|
||||
else
|
||||
callback(err, data);
|
||||
}
|
||||
else
|
||||
callback(err, data);
|
||||
};
|
||||
if (!keepalive_client)
|
||||
keepalive_client = http_init(get_http_ctx());
|
||||
http_request(keepalive_client, url.addr, req, { .timeout = timeout, .keepalive = true, .ssl = url.ssl }, cb);
|
||||
}
|
||||
|
||||
void etcd_state_client_http_t::parse_config(const json11::Json & config)
|
||||
{
|
||||
auto old_etcd_ws_keepalive_interval = this->etcd_ws_keepalive_interval;
|
||||
etcd_state_client_t::parse_config(config);
|
||||
if (this->etcd_ws_keepalive_interval != old_etcd_ws_keepalive_interval && ws_keepalive_timer >= 0)
|
||||
{
|
||||
stop_ws_keepalive();
|
||||
start_ws_keepalive();
|
||||
}
|
||||
}
|
||||
|
||||
void etcd_state_client_http_t::pick_next_etcd(std::function<void()> cb)
|
||||
{
|
||||
if (!etcd_addresses.size() && !etcd_local.size())
|
||||
{
|
||||
fprintf(stderr, "etcd_address is missing in Vitastor configuration\n");
|
||||
exit(1);
|
||||
}
|
||||
if (selected_etcd_url.addr != "")
|
||||
{
|
||||
cb();
|
||||
return;
|
||||
}
|
||||
if (etcd_urls_to_try.size() != 0)
|
||||
{
|
||||
selected_etcd_url = std::move(etcd_urls_to_try[0]);
|
||||
etcd_urls_to_try.erase(etcd_urls_to_try.begin());
|
||||
cb();
|
||||
return;
|
||||
}
|
||||
on_resolve_queue.push_back(std::move(cb));
|
||||
if (on_resolve_queue.size() > 1)
|
||||
{
|
||||
// Already resolving
|
||||
return;
|
||||
}
|
||||
assert(!resolve_count);
|
||||
local_to_try = 0;
|
||||
for (auto & url: etcd_local_addr_urls)
|
||||
{
|
||||
// Prefer local IPs, if any
|
||||
etcd_urls_to_try.push_back(url);
|
||||
local_to_try++;
|
||||
}
|
||||
for (auto & url: etcd_nonlocal_addr_urls)
|
||||
{
|
||||
etcd_urls_to_try.push_back(url);
|
||||
}
|
||||
resolve_count++;
|
||||
for (auto & url: etcd_name_urls)
|
||||
{
|
||||
resolve_count++;
|
||||
http_resolve(get_http_ctx(), url.ssl, url.addr, [this, url](const std::string & error, const std::vector<std::string>& addresses)
|
||||
{
|
||||
if (error != "")
|
||||
fprintf(stderr, "Error resolving %s: %s\n", url.addr.c_str(), error.c_str());
|
||||
for (auto & addr: addresses)
|
||||
{
|
||||
auto url_copy = url;
|
||||
url_copy.addr = addr;
|
||||
if (local_ips.find(addr) != local_ips.end())
|
||||
{
|
||||
etcd_urls_to_try.insert(etcd_urls_to_try.begin(), std::move(url_copy));
|
||||
local_to_try++;
|
||||
}
|
||||
else
|
||||
etcd_urls_to_try.push_back(std::move(url_copy));
|
||||
}
|
||||
resolve_count--;
|
||||
if (!resolve_count)
|
||||
pick_next_etcd_on_resolve();
|
||||
});
|
||||
}
|
||||
resolve_count--;
|
||||
if (!resolve_count)
|
||||
{
|
||||
pick_next_etcd_on_resolve();
|
||||
}
|
||||
}
|
||||
|
||||
void etcd_state_client_http_t::pick_next_etcd_on_resolve()
|
||||
{
|
||||
if (!etcd_urls_to_try.size())
|
||||
{
|
||||
fprintf(stderr, "None of etcd_address could be resolved\n");
|
||||
exit(1);
|
||||
}
|
||||
if (!rand_initialized)
|
||||
{
|
||||
timespec tv;
|
||||
clock_gettime(CLOCK_REALTIME, &tv);
|
||||
srand48(tv.tv_sec*1000000000 + tv.tv_nsec);
|
||||
rand_initialized = true;
|
||||
}
|
||||
// Shuffle addresses
|
||||
for (size_t i = etcd_urls_to_try.size()-1; i > local_to_try; i--)
|
||||
{
|
||||
size_t j = local_to_try + lrand48() % (i - local_to_try);
|
||||
if (j != i)
|
||||
std::swap(etcd_urls_to_try[i], etcd_urls_to_try[j]);
|
||||
}
|
||||
selected_etcd_url = std::move(etcd_urls_to_try[0]);
|
||||
etcd_urls_to_try.erase(etcd_urls_to_try.begin());
|
||||
auto cbs = std::move(on_resolve_queue);
|
||||
for (auto cb: cbs)
|
||||
{
|
||||
cb();
|
||||
}
|
||||
}
|
||||
|
||||
void etcd_state_client_http_t::start_etcd_watcher()
|
||||
{
|
||||
pick_next_etcd([this]()
|
||||
{
|
||||
start_etcd_watcher_selected();
|
||||
});
|
||||
}
|
||||
|
||||
void etcd_state_client_http_t::start_etcd_watcher_selected()
|
||||
{
|
||||
const auto & url = selected_etcd_url;
|
||||
etcd_watches_initialised = 0;
|
||||
ws_alive = 1;
|
||||
if (this->log_level > 1)
|
||||
{
|
||||
fprintf(stderr, "Trying to connect to etcd websocket at %s%s%s (hostname %s), watch from revision %ju/%ju/%ju\n",
|
||||
url.ssl ? "https://" : "http://", url.addr.c_str(), url.path.c_str(), url.hostname.c_str(),
|
||||
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
|
||||
}
|
||||
if (!etcd_watch_ws)
|
||||
etcd_watch_ws = http_init(get_http_ctx());
|
||||
else
|
||||
http_close(etcd_watch_ws);
|
||||
open_websocket(etcd_watch_ws, url.addr, url.hostname, url.path+"/watch", { .timeout = etcd_slow_timeout, .ssl = url.ssl },
|
||||
[this, cur_addr = url.addr](http_message_t *msg)
|
||||
{
|
||||
if (msg->body.length())
|
||||
{
|
||||
ws_alive = 1;
|
||||
std::string json_err;
|
||||
json11::Json data = json11::Json::parse(msg->body, json_err);
|
||||
if (json_err != "")
|
||||
{
|
||||
fprintf(stderr, "Bad JSON in etcd event: %s, ignoring event\n", json_err.c_str());
|
||||
}
|
||||
else
|
||||
{
|
||||
uint64_t watch_id = data["result"]["watch_id"].uint64_value();
|
||||
if (data["result"]["created"].bool_value())
|
||||
{
|
||||
if (watch_id == ETCD_CONFIG_WATCH_ID ||
|
||||
watch_id == ETCD_PG_STATE_WATCH_ID ||
|
||||
watch_id == ETCD_OSD_STATE_WATCH_ID)
|
||||
{
|
||||
etcd_watches_initialised++;
|
||||
}
|
||||
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES && this->log_level > 0)
|
||||
{
|
||||
fprintf(stderr, "Successfully subscribed to etcd at %s, revision %ju/%ju/%ju\n", cur_addr.c_str(),
|
||||
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
|
||||
}
|
||||
}
|
||||
if (data["result"]["canceled"].bool_value())
|
||||
{
|
||||
// etcd watch canceled, maybe because the revision was compacted
|
||||
if (data["result"]["compact_revision"].uint64_value())
|
||||
{
|
||||
// we may miss events if we proceed
|
||||
// so we should restart from the beginning if we can
|
||||
if (on_reload_hook != NULL)
|
||||
{
|
||||
// check to not trigger on_reload_hook multiple times
|
||||
if (etcd_watch_ws != NULL)
|
||||
{
|
||||
fprintf(stderr, "Revisions before %ju were compacted by etcd, reloading state\n",
|
||||
data["result"]["compact_revision"].uint64_value());
|
||||
http_close(etcd_watch_ws);
|
||||
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = 0;
|
||||
on_reload_hook();
|
||||
}
|
||||
return;
|
||||
}
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "Revisions before %ju were compacted by etcd, exiting\n",
|
||||
data["result"]["compact_revision"].uint64_value());
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "Watch canceled by etcd, reason: %s, exiting\n", data["result"]["cancel_reason"].string_value().c_str());
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
// Save revision only if it's present in the message - because sometimes etcd sends something without a header, like:
|
||||
// {"error": {"grpc_code": 14, "http_code": 503, "http_status": "Service Unavailable", "message": "error reading from server: EOF"}}
|
||||
// Also don't save revision from the initial created: true messages because they always contain the latest revision
|
||||
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES &&
|
||||
!data["result"]["header"]["revision"].is_null() &&
|
||||
!data["result"]["created"].bool_value())
|
||||
{
|
||||
// Restart watchers from the same revision number as in the last received message,
|
||||
// not from the next one to protect against revision being split into multiple messages,
|
||||
// even though etcd guarantees not to do that **within a single watcher** without fragment=true:
|
||||
// https://etcd.io/docs/v3.5/learning/api_guarantees/#watch-apis
|
||||
// Revision contents are ALWAYS split into separate messages for different watchers though!
|
||||
// So generally we have to resume each watcher from its own revision...
|
||||
// Progress messages may have watch_id=-1 if sent on behalf of multiple watchers though.
|
||||
// And antietcd has an advanced semantic which merges the same revision for all watchers
|
||||
// into one message and just omits watch_id.
|
||||
// So we also have to handle the case where watch_id is -1 or not present (0).
|
||||
auto watch_rev = data["result"]["header"]["revision"].uint64_value();
|
||||
if (!watch_id || watch_id == UINT64_MAX)
|
||||
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = watch_rev;
|
||||
else if (watch_id == ETCD_CONFIG_WATCH_ID)
|
||||
etcd_watch_revision_config = watch_rev;
|
||||
else if (watch_id == ETCD_PG_STATE_WATCH_ID)
|
||||
etcd_watch_revision_pg = watch_rev;
|
||||
else if (watch_id == ETCD_OSD_STATE_WATCH_ID)
|
||||
etcd_watch_revision_osd = watch_rev;
|
||||
etcd_urls_to_try.clear();
|
||||
}
|
||||
// First gather all changes into a hash to remove multiple overwrites
|
||||
std::map<std::string, etcd_kv_t> changes;
|
||||
for (auto & ev: data["result"]["events"].array_items())
|
||||
{
|
||||
auto kv = parse_etcd_kv(ev["kv"]);
|
||||
if (kv.key != "")
|
||||
{
|
||||
changes[kv.key] = kv;
|
||||
}
|
||||
}
|
||||
for (auto & kv: changes)
|
||||
{
|
||||
if (this->log_level > 3)
|
||||
{
|
||||
fprintf(stderr, "Incoming event: %s -> %s\n", kv.first.c_str(), kv.second.value.dump().c_str());
|
||||
}
|
||||
parse_state(kv.second);
|
||||
}
|
||||
// React to changes
|
||||
if (on_change_hook != NULL)
|
||||
{
|
||||
on_change_hook(changes);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (msg->eof)
|
||||
{
|
||||
fprintf(stderr, "Disconnected from etcd %s\n", cur_addr.c_str());
|
||||
if (cur_addr == selected_etcd_url.addr)
|
||||
selected_etcd_url = (http_url_t){};
|
||||
if (etcd_watches_initialised == 0)
|
||||
{
|
||||
// Connection not established, retry in <etcd_quick_timeout>
|
||||
tfd->set_timer(etcd_quick_timeout, false, [this](int)
|
||||
{
|
||||
start_etcd_watcher();
|
||||
});
|
||||
}
|
||||
else if (etcd_watches_initialised > 0)
|
||||
{
|
||||
// Connection was live, retry immediately
|
||||
etcd_watches_initialised = 0;
|
||||
start_etcd_watcher();
|
||||
}
|
||||
}
|
||||
});
|
||||
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
||||
{ "create_request", json11::Json::object {
|
||||
{ "key", base64_encode(etcd_prefix+"/config/") },
|
||||
{ "range_end", base64_encode(etcd_prefix+"/config0") },
|
||||
{ "start_revision", etcd_watch_revision_config },
|
||||
{ "watch_id", ETCD_CONFIG_WATCH_ID },
|
||||
{ "progress_notify", true },
|
||||
} }
|
||||
}).dump());
|
||||
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
||||
{ "create_request", json11::Json::object {
|
||||
{ "key", base64_encode(etcd_prefix+"/osd/state/") },
|
||||
{ "range_end", base64_encode(etcd_prefix+"/osd/state0") },
|
||||
{ "start_revision", etcd_watch_revision_osd },
|
||||
{ "watch_id", ETCD_OSD_STATE_WATCH_ID },
|
||||
{ "progress_notify", true },
|
||||
} }
|
||||
}).dump());
|
||||
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
||||
{ "create_request", json11::Json::object {
|
||||
{ "key", base64_encode(etcd_prefix+"/pg/") },
|
||||
{ "range_end", base64_encode(etcd_prefix+"/pg0") },
|
||||
{ "start_revision", etcd_watch_revision_pg },
|
||||
{ "watch_id", ETCD_PG_STATE_WATCH_ID },
|
||||
{ "progress_notify", true },
|
||||
} }
|
||||
}).dump());
|
||||
// FIXME: Do not watch /pg/history/ at all in client code (not in OSD)
|
||||
if (on_start_watcher_hook)
|
||||
{
|
||||
on_start_watcher_hook(etcd_watch_ws);
|
||||
}
|
||||
start_ws_keepalive();
|
||||
}
|
||||
|
||||
void etcd_state_client_http_t::stop_ws_keepalive()
|
||||
{
|
||||
if (ws_keepalive_timer >= 0)
|
||||
{
|
||||
tfd->clear_timer(ws_keepalive_timer);
|
||||
ws_keepalive_timer = -1;
|
||||
}
|
||||
}
|
||||
|
||||
void etcd_state_client_http_t::start_ws_keepalive()
|
||||
{
|
||||
if (ws_keepalive_timer < 0)
|
||||
{
|
||||
ws_keepalive_timer = tfd->set_timer(etcd_ws_keepalive_interval*1000, true, [this](int)
|
||||
{
|
||||
if (!etcd_watch_ws || etcd_watches_initialised < ETCD_TOTAL_WATCHES)
|
||||
{
|
||||
// Do nothing
|
||||
}
|
||||
else if (!ws_alive)
|
||||
{
|
||||
if (this->log_level > 0)
|
||||
{
|
||||
fprintf(stderr, "Websocket ping failed, disconnecting from etcd %s\n", selected_etcd_url.addr.c_str());
|
||||
}
|
||||
start_etcd_watcher();
|
||||
}
|
||||
else
|
||||
{
|
||||
ws_alive = 0;
|
||||
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
|
||||
{ "progress_request", json11::Json::object { } }
|
||||
}).dump());
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
void etcd_state_client_http_t::load_global_config()
|
||||
{
|
||||
etcd_state_client_t::load_global_config([this](const std::string & err)
|
||||
{
|
||||
if (err != "")
|
||||
{
|
||||
fprintf(stderr, "Error reading configuration from etcd: %s\n", err.c_str());
|
||||
if (infinite_start)
|
||||
{
|
||||
tfd->set_timer(etcd_slow_timeout, false, [this](int timer_id)
|
||||
{
|
||||
load_global_config();
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void etcd_state_client_http_t::load_pgs()
|
||||
{
|
||||
timespec tv;
|
||||
clock_gettime(CLOCK_REALTIME, &tv);
|
||||
uint64_t ms_passed = (tv.tv_sec-etcd_last_reload.tv_sec)*1000 + (tv.tv_nsec-etcd_last_reload.tv_nsec)/1000000;
|
||||
if (ms_passed < etcd_min_reload_interval)
|
||||
{
|
||||
if (load_pgs_timer_id < 0)
|
||||
{
|
||||
load_pgs_timer_id = tfd->set_timer(etcd_min_reload_interval+50-ms_passed, false, [this](int) { load_pgs(); });
|
||||
}
|
||||
return;
|
||||
}
|
||||
etcd_last_reload = tv;
|
||||
if (load_pgs_timer_id >= 0)
|
||||
{
|
||||
tfd->clear_timer(load_pgs_timer_id);
|
||||
load_pgs_timer_id = -1;
|
||||
}
|
||||
etcd_state_client_t::load_pgs([this](const std::string & err)
|
||||
{
|
||||
if (err != "")
|
||||
{
|
||||
// Retry indefinitely
|
||||
fprintf(stderr, "Error loading PGs from etcd: %s\n", err.c_str());
|
||||
tfd->set_timer(etcd_slow_timeout, false, [this](int timer_id)
|
||||
{
|
||||
load_pgs();
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
start_etcd_watcher();
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -1,50 +0,0 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "etcd_state_client.h"
|
||||
|
||||
struct http_context_t;
|
||||
|
||||
struct __attribute__((visibility("default"))) etcd_state_client_http_t: public etcd_state_client_t
|
||||
{
|
||||
protected:
|
||||
timerfd_manager_t *tfd = NULL;
|
||||
int ws_keepalive_timer = -1;
|
||||
int ws_alive = 0;
|
||||
bool rand_initialized = false;
|
||||
int etcd_watches_initialised = 0;
|
||||
timespec etcd_last_reload = {};
|
||||
int load_pgs_timer_id = -1;
|
||||
http_co_t *keepalive_client = NULL;
|
||||
http_co_t *etcd_watch_ws = NULL;
|
||||
http_context_t *http_ctx = NULL;
|
||||
size_t local_to_try = 0;
|
||||
std::vector<http_url_t> etcd_urls_to_try;
|
||||
http_url_t selected_etcd_url;
|
||||
size_t resolve_count = 0;
|
||||
std::vector<std::function<void()>> on_resolve_queue;
|
||||
|
||||
void pick_next_etcd(std::function<void()> cb);
|
||||
void pick_next_etcd_on_resolve();
|
||||
void start_etcd_watcher();
|
||||
void start_etcd_watcher_selected();
|
||||
void stop_ws_keepalive();
|
||||
void start_ws_keepalive();
|
||||
http_context_t *get_http_ctx();
|
||||
public:
|
||||
etcd_state_client_http_t(timerfd_manager_t *tfd);
|
||||
void etcd_call_oneshot(const std::string & etcd_url, const std::string & api, json11::Json payload,
|
||||
int timeout, std::function<void(std::string, json11::Json)> callback) override;
|
||||
void etcd_call(const std::string & api, json11::Json payload, int timeout,
|
||||
int retries, int interval, std::function<void(std::string, json11::Json)> callback) override;
|
||||
void etcd_call_selected(const std::string & api, json11::Json payload, int timeout,
|
||||
int retries, int interval, std::function<void(std::string, json11::Json)> callback);
|
||||
void etcd_add_watch(json11::Json watch) override;
|
||||
std::string get_username() override;
|
||||
void load_global_config() override;
|
||||
void load_pgs() override;
|
||||
void parse_config(const json11::Json & config) override;
|
||||
~etcd_state_client_http_t();
|
||||
};
|
||||
@@ -1,210 +0,0 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||
|
||||
#include <assert.h>
|
||||
#include "etcd_state_client_mock.h"
|
||||
#include "str_util.h"
|
||||
|
||||
etcd_state_client_mock_t::etcd_state_client_mock_t()
|
||||
{
|
||||
timespec tv;
|
||||
clock_gettime(CLOCK_REALTIME, &tv);
|
||||
srand48(tv.tv_sec*1000000000 + tv.tv_nsec);
|
||||
}
|
||||
|
||||
void etcd_state_client_mock_t::etcd_add_watch(json11::Json watch)
|
||||
{
|
||||
}
|
||||
|
||||
std::string etcd_state_client_mock_t::get_username()
|
||||
{
|
||||
return username;
|
||||
}
|
||||
|
||||
void etcd_state_client_mock_t::etcd_call_oneshot(const std::string & etcd_address, const std::string & api, json11::Json payload,
|
||||
int timeout, std::function<void(std::string, json11::Json)> callback)
|
||||
{
|
||||
}
|
||||
|
||||
void etcd_state_client_mock_t::pause()
|
||||
{
|
||||
paused = true;
|
||||
}
|
||||
|
||||
void etcd_state_client_mock_t::resume()
|
||||
{
|
||||
paused = false;
|
||||
auto queue = std::move(this->queue);
|
||||
for (auto& req: queue)
|
||||
{
|
||||
etcd_call(req.api, req.payload, req.timeout, req.retries, req.interval, req.callback);
|
||||
}
|
||||
}
|
||||
|
||||
void etcd_state_client_mock_t::set(const std::string& key, json11::Json data, uint64_t mod_revision, uint64_t lease_id)
|
||||
{
|
||||
if (!mod_revision)
|
||||
mod_revision = ++this->mod_revision;
|
||||
this->data[key] = (etcd_mock_key_data_t){ .value = data.dump(), .mod_revision = mod_revision, .lease_id = lease_id };
|
||||
}
|
||||
|
||||
void etcd_state_client_mock_t::etcd_call(const std::string & api, json11::Json payload, int timeout,
|
||||
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
|
||||
{
|
||||
if (paused)
|
||||
{
|
||||
queue.push_back({ api, payload, timeout, retries, interval, callback });
|
||||
return;
|
||||
}
|
||||
printf("+ etcd: %s\n", api.c_str());
|
||||
if (api == "/kv/txn")
|
||||
{
|
||||
bool ok = true;
|
||||
for (auto& check: payload["compare"].array_items())
|
||||
{
|
||||
auto key = base64_decode(check["key"].string_value());
|
||||
etcd_mock_key_data_t *key_data = data.find(key) != data.end() ? &data.at(key) : NULL;
|
||||
auto target = check["target"].string_value();
|
||||
auto res = check["result"].string_value();
|
||||
assert(res == "LESS" || res == "");
|
||||
bool less = res == "LESS";
|
||||
if (target == "MOD")
|
||||
{
|
||||
uint64_t rev = check["mod_revision"].uint64_value();
|
||||
assert(!less || rev);
|
||||
ok = ok && (less ? (!key_data || key_data->mod_revision < rev) : (key_data && key_data->mod_revision == rev));
|
||||
}
|
||||
else if (target == "CREATE")
|
||||
{
|
||||
uint64_t rev = check["create_revision"].uint64_value();
|
||||
assert(rev == 0 && !less);
|
||||
ok = ok && !key_data;
|
||||
}
|
||||
else if (target == "VERSION")
|
||||
{
|
||||
uint64_t rev = check["version"].uint64_value();
|
||||
assert(rev == 0 && !less);
|
||||
ok = ok && !key_data;
|
||||
}
|
||||
else if (target == "LEASE")
|
||||
{
|
||||
assert(!less);
|
||||
uint64_t lease_id = check["lease"].uint64_value();
|
||||
ok = ok && key_data && key_data->lease_id == lease_id;
|
||||
}
|
||||
else
|
||||
assert(0);
|
||||
}
|
||||
std::map<std::string, etcd_kv_t> changes;
|
||||
bool has_mod = false;
|
||||
for (auto& op: payload[ok ? "success" : "failure"].array_items())
|
||||
{
|
||||
auto& obj = op.object_items();
|
||||
has_mod = has_mod || obj.find("request_put") != obj.end() ||
|
||||
obj.find("request_delete_range") != obj.end();
|
||||
}
|
||||
if (has_mod)
|
||||
{
|
||||
mod_revision++;
|
||||
}
|
||||
json11::Json::array responses;
|
||||
for (auto& op_ptr: payload[ok ? "success" : "failure"].array_items())
|
||||
{
|
||||
auto& op = op_ptr.object_items();
|
||||
if (op.find("request_range") != op.end())
|
||||
{
|
||||
json11::Json::array kvs;
|
||||
auto req = op.at("request_range");
|
||||
auto key = base64_decode(req["key"].string_value());
|
||||
auto range_end = base64_decode(req["range_end"].string_value());
|
||||
auto begin_it = range_end.empty() ? data.find(key) : data.lower_bound(key);
|
||||
auto end_it = range_end.empty() ? (begin_it == data.end() ? begin_it : std::next(begin_it)) : data.lower_bound(range_end);
|
||||
for (auto it = begin_it; it != end_it; it++)
|
||||
{
|
||||
printf("\\- get: %s = %s, rev %ju\n", it->first.c_str(), it->second.value.c_str(), it->second.mod_revision);
|
||||
kvs.push_back(json11::Json::object {
|
||||
{ "key", base64_encode(it->first) },
|
||||
{ "value", base64_encode(it->second.value) },
|
||||
{ "mod_revision", it->second.mod_revision },
|
||||
});
|
||||
}
|
||||
responses.push_back(json11::Json::object {
|
||||
{ "response_range", json11::Json::object{ { "header", json11::Json::object{ { "revision", mod_revision } } }, { "kvs", kvs } } },
|
||||
});
|
||||
}
|
||||
else if (op.find("request_put") != op.end())
|
||||
{
|
||||
auto req = op.at("request_put");
|
||||
auto key = base64_decode(req["key"].string_value());
|
||||
auto value = base64_decode(req["value"].string_value());
|
||||
auto lease_id = req["lease"].uint64_value();
|
||||
printf("\\- put: %s = %s, rev %ju, lease %ju\n", key.c_str(), value.c_str(), mod_revision, lease_id);
|
||||
data[key] = {
|
||||
.value = value,
|
||||
.mod_revision = mod_revision,
|
||||
.lease_id = lease_id,
|
||||
};
|
||||
std::string err;
|
||||
json11::Json json_value = json11::Json::parse(value, err);
|
||||
if (err != "")
|
||||
{
|
||||
fprintf(stderr, "Invalid JSON in etcd key %s during test: %s\n", key.c_str(), value.c_str());
|
||||
exit(1);
|
||||
}
|
||||
changes[key] = { .key = key, .value = json_value, .mod_revision = mod_revision };
|
||||
responses.push_back(json11::Json::object {
|
||||
{ "response_put", json11::Json::object{ { "header", json11::Json::object{ { "revision", mod_revision } } } } },
|
||||
});
|
||||
}
|
||||
else if (op.find("request_delete_range") != op.end())
|
||||
{
|
||||
auto req = op.at("request_delete_range");
|
||||
auto key = base64_decode(req["key"].string_value());
|
||||
auto range_end = base64_decode(req["range_end"].string_value());
|
||||
uint64_t n_del = 0;
|
||||
for (auto it = data.lower_bound(key); it != data.end() && (range_end == "" || it->first < range_end); )
|
||||
{
|
||||
auto & key = it->first;
|
||||
printf("\\- del: %s\n", key.c_str());
|
||||
changes[key] = { .key = key, .mod_revision = mod_revision };
|
||||
n_del++;
|
||||
data.erase(it++);
|
||||
}
|
||||
responses.push_back(json11::Json::object {
|
||||
{ "response_delete_range", json11::Json::object{ { "header", json11::Json::object{ { "revision", mod_revision } } }, { "deleted", n_del } } },
|
||||
});
|
||||
}
|
||||
}
|
||||
callback("", json11::Json::object{
|
||||
{ "header", json11::Json::object{ { "revision", mod_revision } } },
|
||||
{ "succeeded", ok },
|
||||
{ "responses", responses }
|
||||
});
|
||||
// Push changes to watcher
|
||||
if (changes.size())
|
||||
{
|
||||
for (auto & kv: changes)
|
||||
parse_state(kv.second);
|
||||
if (on_change_hook != NULL)
|
||||
on_change_hook(changes);
|
||||
}
|
||||
}
|
||||
else if (api == "/lease/grant")
|
||||
{
|
||||
uint64_t lease_id = (((uint64_t)lrand48()) << 32) | lrand48();
|
||||
leases[lease_id] = payload["TTL"].uint64_value();
|
||||
callback("", json11::Json::object{ { "ID", std::to_string(lease_id) } });
|
||||
}
|
||||
else
|
||||
callback("Unsupported", json11::Json());
|
||||
}
|
||||
|
||||
void etcd_state_client_mock_t::load_global_config()
|
||||
{
|
||||
etcd_state_client_t::load_global_config([this](const std::string & err) {});
|
||||
}
|
||||
|
||||
void etcd_state_client_mock_t::load_pgs()
|
||||
{
|
||||
etcd_state_client_t::load_pgs([this](const std::string & err) {});
|
||||
}
|
||||
@@ -1,44 +0,0 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "etcd_state_client.h"
|
||||
|
||||
struct etcd_mock_key_data_t
|
||||
{
|
||||
std::string value;
|
||||
uint64_t mod_revision;
|
||||
uint64_t lease_id;
|
||||
};
|
||||
|
||||
struct etcd_mock_request_t
|
||||
{
|
||||
std::string api;
|
||||
json11::Json payload;
|
||||
int timeout;
|
||||
int retries;
|
||||
int interval;
|
||||
std::function<void(std::string, json11::Json)> callback;
|
||||
};
|
||||
|
||||
struct etcd_state_client_mock_t: public etcd_state_client_t
|
||||
{
|
||||
uint64_t mod_revision = 0;
|
||||
bool paused = false;
|
||||
std::vector<etcd_mock_request_t> queue;
|
||||
public:
|
||||
std::map<uint64_t, uint64_t> leases;
|
||||
std::map<std::string, etcd_mock_key_data_t> data;
|
||||
std::string username;
|
||||
etcd_state_client_mock_t();
|
||||
void set(const std::string& key, json11::Json data, uint64_t mod_revision = 0, uint64_t lease_id = 0);
|
||||
void pause();
|
||||
void resume();
|
||||
void etcd_call_oneshot(const std::string & etcd_address, const std::string & api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback) override;
|
||||
void etcd_call(const std::string & api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback) override;
|
||||
void etcd_add_watch(json11::Json watch) override;
|
||||
std::string get_username() override;
|
||||
void load_global_config() override;
|
||||
void load_pgs() override;
|
||||
};
|
||||
@@ -318,7 +318,7 @@ public:
|
||||
|
||||
#ifdef WITH_RDMA
|
||||
bool is_rdma_enabled();
|
||||
json11::Json connect_rdma(uint64_t client_id, std::string rdma_address, uint64_t client_max_msg);
|
||||
bool connect_rdma(uint64_t client_id, std::string rdma_address, uint64_t client_max_msg);
|
||||
#endif
|
||||
#ifdef WITH_RDMACM
|
||||
bool is_use_rdmacm();
|
||||
|
||||
@@ -427,37 +427,35 @@ void osd_messenger_t::op_encrypted_copy_buf(osd_client_t *cl, uint8_t *enc_buf,
|
||||
assert(cl->write_op->enc->key_chain[0]);
|
||||
cl->xts_enc_ctx->start(cl, cl->write_op->enc->key_chain[0], cl->write_op->req.rw.offset, cl->write_op->enc->bitmap_granularity);
|
||||
}
|
||||
size_t old_out = done_enc;
|
||||
while (done_plain < plain_len && done_enc < enc_len)
|
||||
{
|
||||
size_t done_in = 0;
|
||||
size_t done_out = 0;
|
||||
cl->xts_enc_ctx->update(plain+done_plain, plain_len-done_plain, enc_buf+done_enc, enc_len-done_enc, done_in, done_out);
|
||||
if (cl->write_csum_state && done_out > 0)
|
||||
XXH3_64bits_update(cl->write_csum_state, enc_buf+done_enc, done_out);
|
||||
done_enc += done_out;
|
||||
cl->write_op_pos += done_in;
|
||||
done_plain += done_in;
|
||||
}
|
||||
if (cl->write_csum_state && done_enc > old_out)
|
||||
XXH3_64bits_update(cl->write_csum_state, enc_buf+old_out, done_enc-old_out);
|
||||
}
|
||||
|
||||
void osd_messenger_t::op_decrypted_copy_buf(osd_client_t *cl, uint8_t *enc_buf, size_t enc_len, uint8_t *plain, size_t plain_len, size_t & done_plain, size_t & done_enc)
|
||||
{
|
||||
op_decrypt_start(cl);
|
||||
size_t old_in = done_enc;
|
||||
while (done_plain < plain_len && done_enc < enc_len)
|
||||
{
|
||||
size_t done_in = 0;
|
||||
size_t done_out = 0;
|
||||
// plain == NULL means skip output
|
||||
cl->xts_dec_ctx->update(enc_buf+done_enc, enc_len-done_enc, plain ? plain+done_plain : NULL, plain_len-done_plain, done_in, done_out);
|
||||
if (cl->read_csum_state && done_in > 0)
|
||||
XXH3_64bits_update(cl->read_csum_state, enc_buf+done_enc, done_in);
|
||||
done_enc += done_in;
|
||||
cl->read_op_pos += done_out;
|
||||
cl->read_op_inline_decrypt_in += done_in;
|
||||
done_plain += done_out;
|
||||
}
|
||||
if (cl->read_csum_state && done_enc > old_in)
|
||||
XXH3_64bits_update(cl->read_csum_state, enc_buf+old_in, done_enc-old_in);
|
||||
}
|
||||
|
||||
void osd_messenger_t::op_decrypt_start(osd_client_t* cl)
|
||||
|
||||
@@ -3,7 +3,6 @@
|
||||
|
||||
#include <assert.h>
|
||||
|
||||
#include "messenger.h"
|
||||
#include "msgr_op.h"
|
||||
|
||||
osd_op_t::~osd_op_t()
|
||||
@@ -39,52 +38,3 @@ bool osd_op_t::is_recovery_related()
|
||||
req.hdr.opcode == OSD_OP_SEC_SYNC &&
|
||||
(req.sec_sync.flags & OSD_OP_RECOVERY_RELATED);
|
||||
}
|
||||
|
||||
void osd_messenger_t::measure_exec(osd_op_t *cur_op)
|
||||
{
|
||||
// Measure execution latency
|
||||
if (cur_op->req.hdr.opcode > OSD_OP_MAX)
|
||||
{
|
||||
return;
|
||||
}
|
||||
if (!cur_op->tv_end.tv_sec)
|
||||
{
|
||||
clock_gettime(CLOCK_REALTIME, &cur_op->tv_end);
|
||||
}
|
||||
uint64_t len = 0;
|
||||
if (cur_op->req.hdr.opcode == OSD_OP_READ ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_WRITE ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SCRUB)
|
||||
{
|
||||
// req.rw.len is internally set to the full object size for scrubs
|
||||
len = cur_op->req.rw.len;
|
||||
}
|
||||
else if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE)
|
||||
{
|
||||
len = cur_op->req.sec_rw.len;
|
||||
}
|
||||
inc_op_stats(stats, cur_op->req.hdr.opcode, cur_op->tv_begin, cur_op->tv_end, len);
|
||||
if (cur_op->is_recovery_related())
|
||||
{
|
||||
inc_op_stats(recovery_stats, cur_op->req.hdr.opcode, cur_op->tv_begin, cur_op->tv_end, len);
|
||||
}
|
||||
}
|
||||
|
||||
void osd_messenger_t::inc_op_stats(osd_op_stats_t & stats, uint64_t opcode, timespec & tv_begin, timespec & tv_end, uint64_t len)
|
||||
{
|
||||
uint64_t usecs = (
|
||||
(tv_end.tv_sec - tv_begin.tv_sec)*1000000 +
|
||||
(tv_end.tv_nsec - tv_begin.tv_nsec)/1000
|
||||
);
|
||||
stats.op_stat_count[opcode]++;
|
||||
if (!stats.op_stat_count[opcode])
|
||||
{
|
||||
stats.op_stat_count[opcode] = 1;
|
||||
stats.op_stat_sum[opcode] = 0;
|
||||
stats.op_stat_bytes[opcode] = 0;
|
||||
}
|
||||
stats.op_stat_sum[opcode] += usecs;
|
||||
stats.op_stat_bytes[opcode] += len;
|
||||
}
|
||||
|
||||
@@ -507,7 +507,7 @@ int msgr_rdma_connection_t::connect(msgr_rdma_address_t *dest)
|
||||
return 0;
|
||||
}
|
||||
|
||||
json11::Json osd_messenger_t::connect_rdma(uint64_t client_id, std::string rdma_address, uint64_t client_max_msg)
|
||||
bool osd_messenger_t::connect_rdma(uint64_t client_id, std::string rdma_address, uint64_t client_max_msg)
|
||||
{
|
||||
// Try to connect to the peer using RDMA
|
||||
msgr_rdma_address_t addr;
|
||||
@@ -523,7 +523,7 @@ json11::Json osd_messenger_t::connect_rdma(uint64_t client_id, std::string rdma_
|
||||
{
|
||||
if (log_level > 0)
|
||||
fprintf(stderr, "No RDMA context for peer %ju, using only TCP\n", client_id);
|
||||
return json11::Json();
|
||||
return false;
|
||||
}
|
||||
msgr_rdma_connection_t *rdma_conn = msgr_rdma_connection_t::create(selected_ctx, rdma_max_send, rdma_max_recv, rdma_max_sge, client_max_msg);
|
||||
if (rdma_conn)
|
||||
@@ -542,14 +542,11 @@ json11::Json osd_messenger_t::connect_rdma(uint64_t client_id, std::string rdma_
|
||||
// Remember connection, but switch to RDMA only after sending the configuration response
|
||||
cl->rdma_conn = rdma_conn;
|
||||
cl->peer_state = PEER_RDMA_CONNECTING;
|
||||
return json11::Json::object{
|
||||
{"rdma_address", rdma_conn->addr.to_string()},
|
||||
{"rdma_max_msg", rdma_conn->max_msg},
|
||||
};
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return json11::Json();
|
||||
return false;
|
||||
}
|
||||
|
||||
static void try_send_rdma_wr(osd_client_t *cl, ibv_sge *sge, int op_sge)
|
||||
|
||||
@@ -486,6 +486,55 @@ void osd_messenger_t::outbox_push(osd_op_t *cur_op)
|
||||
}
|
||||
}
|
||||
|
||||
void osd_messenger_t::inc_op_stats(osd_op_stats_t & stats, uint64_t opcode, timespec & tv_begin, timespec & tv_end, uint64_t len)
|
||||
{
|
||||
uint64_t usecs = (
|
||||
(tv_end.tv_sec - tv_begin.tv_sec)*1000000 +
|
||||
(tv_end.tv_nsec - tv_begin.tv_nsec)/1000
|
||||
);
|
||||
stats.op_stat_count[opcode]++;
|
||||
if (!stats.op_stat_count[opcode])
|
||||
{
|
||||
stats.op_stat_count[opcode] = 1;
|
||||
stats.op_stat_sum[opcode] = 0;
|
||||
stats.op_stat_bytes[opcode] = 0;
|
||||
}
|
||||
stats.op_stat_sum[opcode] += usecs;
|
||||
stats.op_stat_bytes[opcode] += len;
|
||||
}
|
||||
|
||||
void osd_messenger_t::measure_exec(osd_op_t *cur_op)
|
||||
{
|
||||
// Measure execution latency
|
||||
if (cur_op->req.hdr.opcode > OSD_OP_MAX)
|
||||
{
|
||||
return;
|
||||
}
|
||||
if (!cur_op->tv_end.tv_sec)
|
||||
{
|
||||
clock_gettime(CLOCK_REALTIME, &cur_op->tv_end);
|
||||
}
|
||||
uint64_t len = 0;
|
||||
if (cur_op->req.hdr.opcode == OSD_OP_READ ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_WRITE ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SCRUB)
|
||||
{
|
||||
// req.rw.len is internally set to the full object size for scrubs
|
||||
len = cur_op->req.rw.len;
|
||||
}
|
||||
else if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
|
||||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE)
|
||||
{
|
||||
len = cur_op->req.sec_rw.len;
|
||||
}
|
||||
inc_op_stats(stats, cur_op->req.hdr.opcode, cur_op->tv_begin, cur_op->tv_end, len);
|
||||
if (cur_op->is_recovery_related())
|
||||
{
|
||||
inc_op_stats(recovery_stats, cur_op->req.hdr.opcode, cur_op->tv_begin, cur_op->tv_end, len);
|
||||
}
|
||||
}
|
||||
|
||||
bool osd_messenger_t::try_send(osd_client_t *cl)
|
||||
{
|
||||
if (cl->peer_state == PEER_STOPPED || cl->peer_fd < 0)
|
||||
|
||||
+11
-12
@@ -301,9 +301,10 @@ const char *help_text =
|
||||
" --nbd_disconnect_on_close 1\n"
|
||||
" Disconnect the nbd device on close by last opener.\n"
|
||||
#endif
|
||||
" --readonly\n"
|
||||
#ifdef NBD_FLAG_READ_ONLY
|
||||
" --nbd_ro 1\n"
|
||||
" Set device into read only mode.\n"
|
||||
#endif
|
||||
"\n"
|
||||
"vitastor-nbd netlink-unmap /dev/nbdN\n"
|
||||
" Unmap a device using netlink interface. Works with both netlink and ioctl mapped devices.\n"
|
||||
@@ -347,7 +348,6 @@ protected:
|
||||
int read_ready = 0;
|
||||
msghdr read_msg = { 0 }, send_msg = { 0 };
|
||||
iovec read_iov = { 0 };
|
||||
bool stop = false;
|
||||
|
||||
std::string logfile = "/dev/null";
|
||||
|
||||
@@ -514,7 +514,7 @@ help:
|
||||
// Create client
|
||||
ringloop = new ring_loop_t(RINGLOOP_DEFAULT_SIZE);
|
||||
epmgr = new epoll_manager_t(ringloop);
|
||||
cli = cluster_client_t::create(ringloop, epmgr->tfd, cfg);
|
||||
cli = new cluster_client_t(ringloop, epmgr->tfd, cfg);
|
||||
if (!inode)
|
||||
{
|
||||
// Load image metadata
|
||||
@@ -525,7 +525,7 @@ help:
|
||||
break;
|
||||
ringloop->wait();
|
||||
}
|
||||
watch = cli->st_cli->watch_inode(image_name);
|
||||
watch = cli->st_cli.watch_inode(image_name);
|
||||
device_size = watch->cfg.size;
|
||||
if (!watch->cfg.num || !device_size)
|
||||
{
|
||||
@@ -581,8 +581,10 @@ help:
|
||||
}
|
||||
uint64_t flags = NBD_FLAG_SEND_FLUSH;
|
||||
uint64_t cflags = 0;
|
||||
if (!cfg["readonly"].is_null() || !cfg["nbd_ro"].is_null())
|
||||
#ifdef NBD_FLAG_READ_ONLY
|
||||
if (!cfg["nbd_ro"].is_null())
|
||||
flags |= NBD_FLAG_READ_ONLY;
|
||||
#endif
|
||||
#ifdef NBD_CFLAG_DESTROY_ON_DISCONNECT
|
||||
if (!cfg["nbd_destroy_on_disconnect"].is_null())
|
||||
cflags |= NBD_CFLAG_DESTROY_ON_DISCONNECT;
|
||||
@@ -618,10 +620,7 @@ help:
|
||||
if (!cfg["dev_num"].is_null())
|
||||
{
|
||||
int r;
|
||||
uint64_t flags = NBD_FLAG_SEND_FLUSH;
|
||||
if (!cfg["readonly"].is_null())
|
||||
flags |= NBD_FLAG_READ_ONLY;
|
||||
if ((r = run_nbd(sockfd, cfg["dev_num"].int64_value(), device_size, flags, nbd_timeout, bg)) != 0)
|
||||
if ((r = run_nbd(sockfd, cfg["dev_num"].int64_value(), device_size, NBD_FLAG_SEND_FLUSH, nbd_timeout, bg)) != 0)
|
||||
{
|
||||
fprintf(stderr, "run_nbd: %s\n", strerror(-r));
|
||||
exit(1);
|
||||
@@ -679,7 +678,8 @@ help:
|
||||
};
|
||||
ringloop->register_consumer(&consumer);
|
||||
// Add FD to epoll
|
||||
epmgr->tfd->set_fd_handler(sockfd[0], false, [this](int peer_fd, int epoll_events)
|
||||
bool stop = false;
|
||||
epmgr->tfd->set_fd_handler(sockfd[0], false, [this, &stop](int peer_fd, int epoll_events)
|
||||
{
|
||||
if (epoll_events & EPOLLRDHUP)
|
||||
{
|
||||
@@ -1118,8 +1118,7 @@ protected:
|
||||
{
|
||||
// Disconnect
|
||||
close(nbd_fd);
|
||||
stop = true;
|
||||
return;
|
||||
exit(0);
|
||||
}
|
||||
if (be32toh(cur_req.magic) != NBD_REQUEST_MAGIC ||
|
||||
req_type != NBD_CMD_READ && req_type != NBD_CMD_WRITE && req_type != NBD_CMD_FLUSH)
|
||||
|
||||
@@ -705,54 +705,6 @@ static void vitastor_close(BlockDriverState *bs)
|
||||
client->last_bitmap = NULL;
|
||||
}
|
||||
|
||||
// Unregister all event sources from the current AioContext. Called by the
|
||||
// block layer before bs is moved to a different AioContext (e.g. during live
|
||||
// migration, drain, dataplane switching). The block layer guarantees that no
|
||||
// requests are in flight at this point.
|
||||
static void vitastor_detach_aio_context(BlockDriverState *bs)
|
||||
{
|
||||
VitastorClient *client = bs->opaque;
|
||||
int i;
|
||||
#if defined VITASTOR_C_API_VERSION && VITASTOR_C_API_VERSION >= 2
|
||||
if (client->uring_eventfd >= 0)
|
||||
{
|
||||
universal_aio_set_fd_handler(client->ctx, client->uring_eventfd, NULL, NULL, NULL);
|
||||
// Wait until any scheduled B/H is processed before switching contexts:
|
||||
// it would otherwise fire on the old context with stale state.
|
||||
if (client->bh_uring_scheduled)
|
||||
{
|
||||
BDRV_POLL_WHILE(bs, client->bh_uring_scheduled);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
for (i = 0; i < client->fd_count; i++)
|
||||
{
|
||||
universal_aio_set_fd_handler(client->ctx, client->fds[i]->fd, NULL, NULL, NULL);
|
||||
}
|
||||
}
|
||||
|
||||
// (Re-)register all event sources on the new AioContext.
|
||||
static void vitastor_attach_aio_context(BlockDriverState *bs, AioContext *new_ctx)
|
||||
{
|
||||
VitastorClient *client = bs->opaque;
|
||||
int i;
|
||||
client->ctx = new_ctx;
|
||||
#if defined VITASTOR_C_API_VERSION && VITASTOR_C_API_VERSION >= 2
|
||||
if (client->uring_eventfd >= 0)
|
||||
{
|
||||
universal_aio_set_fd_handler(new_ctx, client->uring_eventfd, vitastor_uring_handler, NULL, client);
|
||||
}
|
||||
#endif
|
||||
for (i = 0; i < client->fd_count; i++)
|
||||
{
|
||||
VitastorFdData *fdd = client->fds[i];
|
||||
universal_aio_set_fd_handler(new_ctx, fdd->fd,
|
||||
fdd->fd_read ? vitastor_aio_fd_read : NULL,
|
||||
fdd->fd_write ? vitastor_aio_fd_write : NULL,
|
||||
fdd);
|
||||
}
|
||||
}
|
||||
|
||||
#if QEMU_VERSION_MAJOR >= 3 || QEMU_VERSION_MAJOR == 2 && QEMU_VERSION_MINOR >= 2
|
||||
static void vitastor_refresh_filename(BlockDriverState *bs)
|
||||
{
|
||||
@@ -880,25 +832,6 @@ static int vitastor_refresh_limits(BlockDriverState *bs)
|
||||
// return 0;
|
||||
//}
|
||||
|
||||
// Move the running coroutine to the BlockDriverState's home AioContext.
|
||||
//
|
||||
// The block-coroutine-wrapper generator sets poll_state.ctx to
|
||||
// qemu_get_current_aio_context() in the sync wrappers (bdrv_flush(),
|
||||
// bdrv_pread() etc.). When bdrv_flush_all() runs under BQL from outside the
|
||||
// bs's iothread (e.g. on the migration thread inside do_vm_stop()), that is
|
||||
// the main AioContext, not the iothread that actually owns the bs. The
|
||||
// coroutine then runs on the wrong context while completions are delivered on
|
||||
// the iothread, and racing aio_co_schedule() vs. qemu_aio_coroutine_enter()
|
||||
// on the same coroutine triggers "Co-routine was already scheduled in
|
||||
// aio_co_schedule" and aborts the process (observed during live migration).
|
||||
//
|
||||
// aio_co_reschedule_self() is a no-op when we are already on the target ctx.
|
||||
#if QEMU_VERSION_MAJOR > 5 || QEMU_VERSION_MAJOR == 5 && QEMU_VERSION_MINOR >= 2
|
||||
#define vitastor_co_pin_to_bs_ctx(bs) aio_co_reschedule_self(bdrv_get_aio_context(bs))
|
||||
#else
|
||||
#define vitastor_co_pin_to_bs_ctx(bs) ((void)0)
|
||||
#endif
|
||||
|
||||
static void vitastor_co_init_task(BlockDriverState *bs, VitastorRPC *task)
|
||||
{
|
||||
*task = (VitastorRPC) {
|
||||
@@ -957,7 +890,6 @@ static int coroutine_fn vitastor_co_preadv(BlockDriverState *bs,
|
||||
{
|
||||
VitastorClient *client = bs->opaque;
|
||||
VitastorRPC task;
|
||||
vitastor_co_pin_to_bs_ctx(bs);
|
||||
vitastor_co_init_task(bs, &task);
|
||||
task.iov = iov;
|
||||
|
||||
@@ -986,7 +918,6 @@ static int coroutine_fn vitastor_co_pwritev(BlockDriverState *bs,
|
||||
{
|
||||
VitastorClient *client = bs->opaque;
|
||||
VitastorRPC task;
|
||||
vitastor_co_pin_to_bs_ctx(bs);
|
||||
vitastor_co_init_task(bs, &task);
|
||||
task.iov = iov;
|
||||
|
||||
@@ -1060,7 +991,6 @@ static int coroutine_fn vitastor_co_block_status(BlockDriverState *bs,
|
||||
#endif
|
||||
VitastorRPC task;
|
||||
VitastorClient *client = bs->opaque;
|
||||
vitastor_co_pin_to_bs_ctx(bs);
|
||||
uint64_t inode = client->watch ? vitastor_c_inode_get_num(client->watch) : client->inode;
|
||||
uint8_t bit = 0;
|
||||
if (client->last_bitmap && client->last_bitmap_inode == inode &&
|
||||
@@ -1173,7 +1103,6 @@ static int coroutine_fn vitastor_co_flush(BlockDriverState *bs)
|
||||
{
|
||||
VitastorClient *client = bs->opaque;
|
||||
VitastorRPC task;
|
||||
vitastor_co_pin_to_bs_ctx(bs);
|
||||
vitastor_co_init_task(bs, &task);
|
||||
|
||||
qemu_mutex_lock(&client->mutex);
|
||||
@@ -1259,11 +1188,6 @@ static BlockDriver bdrv_vitastor = {
|
||||
#endif
|
||||
.bdrv_close = vitastor_close,
|
||||
|
||||
// Re-register fd handlers when the bs is moved to a different AioContext
|
||||
// (live migration, drain, iothread reassignment).
|
||||
.bdrv_detach_aio_context = vitastor_detach_aio_context,
|
||||
.bdrv_attach_aio_context = vitastor_attach_aio_context,
|
||||
|
||||
// Option list for the create operation
|
||||
#if QEMU_VERSION_MAJOR >= 3 || QEMU_VERSION_MAJOR == 2 && QEMU_VERSION_MINOR > 0
|
||||
.create_opts = &vitastor_create_opts,
|
||||
|
||||
@@ -62,13 +62,6 @@ const char *help_text =
|
||||
"All usual Vitastor config options like --config_path <path_to_config> may also be specified in CLI.\n"
|
||||
;
|
||||
|
||||
struct ublk_request
|
||||
{
|
||||
uint64_t ublk_cmd;
|
||||
int index;
|
||||
int result;
|
||||
};
|
||||
|
||||
class ublk_server
|
||||
{
|
||||
protected:
|
||||
@@ -248,7 +241,7 @@ help:
|
||||
|
||||
// Create client
|
||||
epmgr = new epoll_manager_t(ringloop);
|
||||
cli = cluster_client_t::create(ringloop, epmgr->tfd, cfg);
|
||||
cli = new cluster_client_t(ringloop, epmgr->tfd, cfg);
|
||||
|
||||
// cli->config contains merged config
|
||||
if (!cfg["queue_depth"].is_null())
|
||||
@@ -280,7 +273,7 @@ help:
|
||||
}
|
||||
if (!inode)
|
||||
{
|
||||
watch = cli->st_cli->watch_inode(image_name);
|
||||
watch = cli->st_cli.watch_inode(image_name);
|
||||
device_size = watch->cfg.size;
|
||||
if (!watch->cfg.num || !device_size)
|
||||
{
|
||||
@@ -289,9 +282,9 @@ help:
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
const bool writeback = !cli->get_immediate_commit(inode ? inode : watch->cfg.num);
|
||||
auto pool_it = cli->st_cli->pool_config.find(INODE_POOL(inode ? inode : watch->cfg.num));
|
||||
if (pool_it == cli->st_cli->pool_config.end())
|
||||
const bool writeback = !cli->get_immediate_commit(inode);
|
||||
auto pool_it = cli->st_cli.pool_config.find(INODE_POOL(inode ? inode : watch->cfg.num));
|
||||
if (pool_it == cli->st_cli.pool_config.end())
|
||||
{
|
||||
fprintf(stderr, "Pool %u does not exist\n", INODE_POOL(inode ? inode : watch->cfg.num));
|
||||
exit(1);
|
||||
@@ -338,11 +331,6 @@ help:
|
||||
daemonize_fork(notifyfd);
|
||||
close(notifyfd[0]);
|
||||
}
|
||||
consumer.loop = [this]()
|
||||
{
|
||||
submit_postponed();
|
||||
};
|
||||
ringloop->register_consumer(&consumer);
|
||||
start_device(recover);
|
||||
if (pidfile != "")
|
||||
write_pid();
|
||||
@@ -362,7 +350,6 @@ help:
|
||||
ringloop->wait();
|
||||
}
|
||||
cli->flush();
|
||||
ringloop->unregister_consumer(&consumer);
|
||||
delete cli;
|
||||
delete epmgr;
|
||||
cli = NULL;
|
||||
@@ -566,8 +553,6 @@ protected:
|
||||
ublksrv_ctrl_dev_info ublk_dev = {};
|
||||
ublksrv_io_desc *ublk_queue = NULL;
|
||||
std::vector<uint8_t*> buffers;
|
||||
ring_consumer_t consumer;
|
||||
std::vector<ublk_request> postponed_requests;
|
||||
|
||||
void open_control()
|
||||
{
|
||||
@@ -749,19 +734,9 @@ protected:
|
||||
ctrl_fd = -1;
|
||||
}
|
||||
|
||||
bool submit_request(uint64_t ublk_cmd, int i, int res)
|
||||
void submit_request(uint64_t ublk_cmd, int i, int res)
|
||||
{
|
||||
io_uring_sqe *sqe = ringloop->get_sqe();
|
||||
if (!sqe)
|
||||
{
|
||||
// Handle full io_uring by postponing the request
|
||||
postponed_requests.push_back((ublk_request){
|
||||
.ublk_cmd = ublk_cmd,
|
||||
.index = i,
|
||||
.result = res,
|
||||
});
|
||||
return false;
|
||||
}
|
||||
ring_data_t* data = ((ring_data_t*)sqe->user_data);
|
||||
sqe->fd = cdev_fd;
|
||||
sqe->opcode = IORING_OP_URING_CMD;
|
||||
@@ -775,22 +750,6 @@ protected:
|
||||
cmd->addr = (uint64_t)buffers[i];
|
||||
cmd->result = res;
|
||||
data->callback = [this, i](ring_data_t *data) { exec_request(data->res, i); };
|
||||
return true;
|
||||
}
|
||||
|
||||
void submit_postponed()
|
||||
{
|
||||
int sent = 0;
|
||||
while (postponed_requests.size())
|
||||
{
|
||||
ublk_request r = postponed_requests.back();
|
||||
postponed_requests.pop_back();
|
||||
if (!submit_request(r.ublk_cmd, r.index, r.result))
|
||||
break;
|
||||
sent++;
|
||||
}
|
||||
if (sent)
|
||||
ringloop->submit();
|
||||
}
|
||||
|
||||
void exec_request(int res, int i)
|
||||
@@ -905,11 +864,6 @@ protected:
|
||||
int sync_ublk_cmd(uint32_t cmd_op, void *addr, uint32_t len, uint16_t dev_path_len = 0, uint64_t data0 = 0)
|
||||
{
|
||||
io_uring_sqe *sqe = ringloop->get_sqe();
|
||||
if (!sqe)
|
||||
{
|
||||
fprintf(stderr, "Error: io_uring is full when trying to execute a control command\n");
|
||||
exit(1);
|
||||
}
|
||||
sqe->fd = ctrl_fd;
|
||||
sqe->opcode = IORING_OP_URING_CMD;
|
||||
sqe->ioprio = 0;
|
||||
|
||||
@@ -6,7 +6,7 @@ includedir=${prefix}/@CMAKE_INSTALL_INCLUDEDIR@
|
||||
|
||||
Name: Vitastor
|
||||
Description: Vitastor client library
|
||||
Version: 3.0.15
|
||||
Version: 3.0.12
|
||||
Libs: -L${libdir} -lvitastor_client
|
||||
Cflags: -I${includedir}
|
||||
|
||||
|
||||
+13
-13
@@ -103,7 +103,7 @@ vitastor_c *vitastor_c_create_qemu(QEMUSetFDHandler *aio_set_fd_handler, void *a
|
||||
rdma_device, rdma_port_num, rdma_gid_index, rdma_mtu, log_level
|
||||
);
|
||||
auto self = vitastor_c_create_qemu_common(aio_set_fd_handler, aio_context);
|
||||
self->cli = cluster_client_t::create(NULL, self->tfd, cfg_json);
|
||||
self->cli = new cluster_client_t(NULL, self->tfd, cfg_json);
|
||||
return self;
|
||||
}
|
||||
|
||||
@@ -126,7 +126,7 @@ vitastor_c *vitastor_c_create_qemu_uring(QEMUSetFDHandler *aio_set_fd_handler, v
|
||||
);
|
||||
auto self = vitastor_c_create_qemu_common(aio_set_fd_handler, aio_context);
|
||||
self->ringloop = ringloop;
|
||||
self->cli = cluster_client_t::create(self->ringloop, self->tfd, cfg_json);
|
||||
self->cli = new cluster_client_t(self->ringloop, self->tfd, cfg_json);
|
||||
ringloop->loop();
|
||||
return self;
|
||||
}
|
||||
@@ -150,7 +150,7 @@ vitastor_c *vitastor_c_create_uring(const char *config_path, const char *etcd_ho
|
||||
vitastor_c *self = new vitastor_c;
|
||||
self->ringloop = ringloop;
|
||||
self->epmgr = new epoll_manager_t(self->ringloop);
|
||||
self->cli = cluster_client_t::create(self->ringloop, self->epmgr->tfd, cfg_json);
|
||||
self->cli = new cluster_client_t(self->ringloop, self->epmgr->tfd, cfg_json);
|
||||
ringloop->loop();
|
||||
return self;
|
||||
}
|
||||
@@ -191,7 +191,7 @@ vitastor_c *vitastor_c_create_uring_json(const char **options, int options_len)
|
||||
vitastor_c *self = new vitastor_c;
|
||||
self->ringloop = ringloop;
|
||||
self->epmgr = new epoll_manager_t(self->ringloop);
|
||||
self->cli = cluster_client_t::create(self->ringloop, self->epmgr->tfd, cfg_json);
|
||||
self->cli = new cluster_client_t(self->ringloop, self->epmgr->tfd, cfg_json);
|
||||
ringloop->loop();
|
||||
return self;
|
||||
}
|
||||
@@ -206,7 +206,7 @@ vitastor_c *vitastor_c_create_epoll_json(const char **options, int options_len)
|
||||
json11::Json cfg_json(cfg);
|
||||
vitastor_c *self = new vitastor_c;
|
||||
self->epmgr = new epoll_manager_t(NULL);
|
||||
self->cli = cluster_client_t::create(NULL, self->epmgr->tfd, cfg_json);
|
||||
self->cli = new cluster_client_t(NULL, self->epmgr->tfd, cfg_json);
|
||||
return self;
|
||||
}
|
||||
|
||||
@@ -396,7 +396,7 @@ void vitastor_c_watch_inode(vitastor_c *client, char *image, VitastorIOHandler c
|
||||
{
|
||||
client->cli->on_ready([=]()
|
||||
{
|
||||
auto watch = client->cli->st_cli->watch_inode(std::string(image));
|
||||
auto watch = client->cli->st_cli.watch_inode(std::string(image));
|
||||
cb(opaque, (long)watch);
|
||||
});
|
||||
if (client->ringloop)
|
||||
@@ -407,7 +407,7 @@ void vitastor_c_watch_inode(vitastor_c *client, char *image, VitastorIOHandler c
|
||||
|
||||
void vitastor_c_close_watch(vitastor_c *client, void *handle)
|
||||
{
|
||||
client->cli->st_cli->close_watch((inode_watch_t*)handle);
|
||||
client->cli->st_cli.close_watch((inode_watch_t*)handle);
|
||||
}
|
||||
|
||||
uint64_t vitastor_c_inode_get_size(void *handle)
|
||||
@@ -424,8 +424,8 @@ uint64_t vitastor_c_inode_get_num(void *handle)
|
||||
|
||||
uint32_t vitastor_c_inode_get_block_size(vitastor_c *client, uint64_t inode_num)
|
||||
{
|
||||
auto pool_it = client->cli->st_cli->pool_config.find(INODE_POOL(inode_num));
|
||||
if (pool_it == client->cli->st_cli->pool_config.end())
|
||||
auto pool_it = client->cli->st_cli.pool_config.find(INODE_POOL(inode_num));
|
||||
if (pool_it == client->cli->st_cli.pool_config.end())
|
||||
return 0;
|
||||
auto & pool_cfg = pool_it->second;
|
||||
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
||||
@@ -434,8 +434,8 @@ uint32_t vitastor_c_inode_get_block_size(vitastor_c *client, uint64_t inode_num)
|
||||
|
||||
uint32_t vitastor_c_inode_get_bitmap_granularity(vitastor_c *client, uint64_t inode_num)
|
||||
{
|
||||
auto pool_it = client->cli->st_cli->pool_config.find(INODE_POOL(inode_num));
|
||||
if (pool_it == client->cli->st_cli->pool_config.end())
|
||||
auto pool_it = client->cli->st_cli.pool_config.find(INODE_POOL(inode_num));
|
||||
if (pool_it == client->cli->st_cli.pool_config.end())
|
||||
return 0;
|
||||
// FIXME: READ_BITMAP may fails if parent bitmap granularity differs from inode bitmap granularity
|
||||
return pool_it->second.bitmap_granularity;
|
||||
@@ -471,8 +471,8 @@ uint64_t vitastor_c_inode_get_mod_revision(void *handle)
|
||||
|
||||
uint32_t vitastor_c_inode_get_immediate_commit(vitastor_c *client, uint64_t inode_num)
|
||||
{
|
||||
auto pool_it = client->cli->st_cli->pool_config.find(INODE_POOL(inode_num));
|
||||
if (pool_it == client->cli->st_cli->pool_config.end())
|
||||
auto pool_it = client->cli->st_cli.pool_config.find(INODE_POOL(inode_num));
|
||||
if (pool_it == client->cli->st_cli.pool_config.end())
|
||||
return 0;
|
||||
return pool_it->second.immediate_commit;
|
||||
}
|
||||
|
||||
@@ -15,7 +15,6 @@ add_custom_command(
|
||||
add_library(vitastor_cli STATIC
|
||||
cli_common.cpp
|
||||
cli_alloc_osd.cpp
|
||||
cli_cpubench.cpp
|
||||
cli_describe.cpp
|
||||
cli_fix.cpp
|
||||
cli_ls.cpp
|
||||
@@ -51,5 +50,5 @@ add_executable(vitastor-cli
|
||||
cli.cpp
|
||||
)
|
||||
target_link_libraries(vitastor-cli
|
||||
vitastor_client_int
|
||||
vitastor_client
|
||||
)
|
||||
|
||||
+3
-10
@@ -260,15 +260,12 @@ static const char* help_text =
|
||||
"vitastor-cli rm-user|remove-user|delete-user <username>\n"
|
||||
" Remove a user.\n"
|
||||
"\n"
|
||||
"vitastor-cli cpubench [--json]\n"
|
||||
" Run CPU crypto performance tests: AES-256-GCM, AES-256-XTS and xxhash3.\n"
|
||||
"\n"
|
||||
"vitastor-cli serve\n"
|
||||
" Start HTTP server able to handle CLI commands over a REST API. Options:\n"
|
||||
" --bind_address ADDR Specify server IP address or addresses, separated by space. Default is 127.0.0.1.\n"
|
||||
" --port 8080 Specify server port.\n"
|
||||
" --api_cert FILE Path to server TLS certificate file (PEM format).\n"
|
||||
" --api_pkey FILE Path to server TLS private key file.\n"
|
||||
" --server_cert FILE Path to server TLS certificate file (PEM format).\n"
|
||||
" --server_key FILE Path to server TLS private key file.\n"
|
||||
" --client_ca FILE Path to file with TLS CA certificates used to validate client connections.\n"
|
||||
"\n"
|
||||
"Use vitastor-cli --help <command> for command details or vitastor-cli --help --all for all details.\n"
|
||||
@@ -632,10 +629,6 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
|
||||
// Start HTTP server
|
||||
action_cb = start_serve(cfg);
|
||||
}
|
||||
else if (cmd[0] == "cpubench")
|
||||
{
|
||||
action_cb = start_cpubench(cfg);
|
||||
}
|
||||
else
|
||||
{
|
||||
result = { .err = EOPNOTSUPP, .text = "unknown command: "+cmd[0].string_value() };
|
||||
@@ -655,7 +648,7 @@ static int run(cli_tool_t *p, json11::Json::object cfg)
|
||||
json11::Json cfg_j = cfg;
|
||||
p->ringloop = new ring_loop_t(RINGLOOP_DEFAULT_SIZE);
|
||||
p->epmgr = new epoll_manager_t(p->ringloop);
|
||||
p->cli = cluster_client_t::create(p->ringloop, p->epmgr->tfd, cfg_j);
|
||||
p->cli = new cluster_client_t(p->ringloop, p->epmgr->tfd, cfg_j);
|
||||
p->loop_and_wait(action_cb, [&](const cli_result_t & r)
|
||||
{
|
||||
result = r;
|
||||
|
||||
@@ -67,7 +67,6 @@ public:
|
||||
|
||||
std::function<bool(cli_result_t &)> start(json11::Json::object cfg, cli_result_t & result);
|
||||
std::function<bool(cli_result_t &)> start_alloc_osd(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_cpubench(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_create(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_dd(json11::Json);
|
||||
std::function<bool(cli_result_t &)> start_describe(json11::Json);
|
||||
|
||||
@@ -35,7 +35,7 @@ struct alloc_osd_t
|
||||
{ "target", "VERSION" },
|
||||
{ "version", 0 },
|
||||
{ "key", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/osd/stats/"+std::to_string(new_id)
|
||||
parent->cli->st_cli.etcd_prefix+"/osd/stats/"+std::to_string(new_id)
|
||||
) },
|
||||
},
|
||||
} },
|
||||
@@ -43,7 +43,7 @@ struct alloc_osd_t
|
||||
json11::Json::object {
|
||||
{ "request_put", json11::Json::object {
|
||||
{ "key", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/osd/stats/"+std::to_string(new_id)
|
||||
parent->cli->st_cli.etcd_prefix+"/osd/stats/"+std::to_string(new_id)
|
||||
) },
|
||||
{ "value", base64_encode("{}") },
|
||||
} },
|
||||
@@ -52,8 +52,8 @@ struct alloc_osd_t
|
||||
{ "failure", json11::Json::array {
|
||||
json11::Json::object {
|
||||
{ "request_range", json11::Json::object {
|
||||
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/osd/stats/") },
|
||||
{ "range_end", base64_encode(parent->cli->st_cli->etcd_prefix+"/osd/stats0") },
|
||||
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/osd/stats/") },
|
||||
{ "range_end", base64_encode(parent->cli->st_cli.etcd_prefix+"/osd/stats0") },
|
||||
{ "keys_only", true },
|
||||
} },
|
||||
},
|
||||
|
||||
+17
-17
@@ -17,8 +17,8 @@ bool cli_tool_t::check_image_perm(const inode_config_t & cfg, bool write)
|
||||
|
||||
json11::Json::object cli_tool_t::format_image(const inode_config_t & cfg)
|
||||
{
|
||||
auto pool_it = cli->st_cli->pool_config.find(INODE_POOL(cfg.num));
|
||||
bool good_pool = pool_it != cli->st_cli->pool_config.end();
|
||||
auto pool_it = cli->st_cli.pool_config.find(INODE_POOL(cfg.num));
|
||||
bool good_pool = pool_it != cli->st_cli.pool_config.end();
|
||||
auto img = json11::Json::object {
|
||||
{ "name", cfg.name },
|
||||
{ "size", cfg.size },
|
||||
@@ -50,8 +50,8 @@ json11::Json::object cli_tool_t::format_image(const inode_config_t & cfg)
|
||||
}
|
||||
if (cfg.parent_id)
|
||||
{
|
||||
auto parent_it = cli->st_cli->inode_config.find(cfg.parent_id);
|
||||
if (parent_it != cli->st_cli->inode_config.end())
|
||||
auto parent_it = cli->st_cli.inode_config.find(cfg.parent_id);
|
||||
if (parent_it != cli->st_cli.inode_config.end())
|
||||
{
|
||||
img["parent_name"] = parent_it->second.name;
|
||||
}
|
||||
@@ -64,8 +64,8 @@ json11::Json::object cli_tool_t::format_image(const inode_config_t & cfg)
|
||||
|
||||
void cli_tool_t::change_parent(inode_t cur, inode_t new_parent, cli_result_t *result)
|
||||
{
|
||||
auto cur_cfg_it = cli->st_cli->inode_config.find(cur);
|
||||
if (cur_cfg_it == cli->st_cli->inode_config.end())
|
||||
auto cur_cfg_it = cli->st_cli.inode_config.find(cur);
|
||||
if (cur_cfg_it == cli->st_cli.inode_config.end())
|
||||
{
|
||||
char buf[128];
|
||||
snprintf(buf, 128, "Inode 0x%jx disappeared", cur);
|
||||
@@ -74,13 +74,13 @@ void cli_tool_t::change_parent(inode_t cur, inode_t new_parent, cli_result_t *re
|
||||
}
|
||||
inode_config_t new_cfg = cur_cfg_it->second;
|
||||
std::string cur_name = new_cfg.name;
|
||||
std::string cur_cfg_key = base64_encode(cli->st_cli->etcd_prefix+
|
||||
std::string cur_cfg_key = base64_encode(cli->st_cli.etcd_prefix+
|
||||
"/config/inode/"+std::to_string(INODE_POOL(cur))+
|
||||
"/"+std::to_string(INODE_NO_POOL(cur)));
|
||||
new_cfg.parent_id = new_parent;
|
||||
json11::Json::object cur_cfg_json = cli->st_cli->serialize_inode_cfg(&new_cfg);
|
||||
json11::Json::object cur_cfg_json = cli->st_cli.serialize_inode_cfg(&new_cfg);
|
||||
waiting++;
|
||||
cli->st_cli->etcd_txn_slow(json11::Json::object {
|
||||
cli->st_cli.etcd_txn_slow(json11::Json::object {
|
||||
{ "compare", json11::Json::array {
|
||||
json11::Json::object {
|
||||
{ "target", "MOD" },
|
||||
@@ -109,8 +109,8 @@ void cli_tool_t::change_parent(inode_t cur, inode_t new_parent, cli_result_t *re
|
||||
}
|
||||
else if (new_parent)
|
||||
{
|
||||
auto new_parent_it = cli->st_cli->inode_config.find(new_parent);
|
||||
std::string new_parent_name = new_parent_it != cli->st_cli->inode_config.end()
|
||||
auto new_parent_it = cli->st_cli.inode_config.find(new_parent);
|
||||
std::string new_parent_name = new_parent_it != cli->st_cli.inode_config.end()
|
||||
? new_parent_it->second.name : "<unknown>";
|
||||
*result = (cli_result_t){
|
||||
.text = "Parent of layer "+cur_name+" (inode "+std::to_string(INODE_NO_POOL(cur))+
|
||||
@@ -133,7 +133,7 @@ void cli_tool_t::change_parent(inode_t cur, inode_t new_parent, cli_result_t *re
|
||||
void cli_tool_t::etcd_txn(json11::Json txn)
|
||||
{
|
||||
waiting++;
|
||||
cli->st_cli->etcd_txn_slow(txn, [this](std::string err, json11::Json res)
|
||||
cli->st_cli.etcd_txn_slow(txn, [this](std::string err, json11::Json res)
|
||||
{
|
||||
waiting--;
|
||||
if (err != "")
|
||||
@@ -147,7 +147,7 @@ void cli_tool_t::etcd_txn(json11::Json txn)
|
||||
|
||||
inode_config_t* cli_tool_t::get_inode_cfg(const std::string & name)
|
||||
{
|
||||
for (auto & ic: cli->st_cli->inode_config)
|
||||
for (auto & ic: cli->st_cli.inode_config)
|
||||
{
|
||||
if (ic.second.name == name)
|
||||
{
|
||||
@@ -231,11 +231,11 @@ void cli_tool_t::iterate_kvs_1(json11::Json kvs, const std::string & prefix, std
|
||||
bool is_pool = prefix == "/pool/stats/";
|
||||
for (auto & kv_item: kvs.array_items())
|
||||
{
|
||||
auto kv = cli->st_cli->parse_etcd_kv(kv_item);
|
||||
auto kv = cli->st_cli.parse_etcd_kv(kv_item);
|
||||
uint64_t num = 0;
|
||||
char null_byte = 0;
|
||||
// OSD or pool number
|
||||
int scanned = sscanf(kv.key.substr(cli->st_cli->etcd_prefix.size() + prefix.size()).c_str(), "%ju%c", &num, &null_byte);
|
||||
int scanned = sscanf(kv.key.substr(cli->st_cli.etcd_prefix.size() + prefix.size()).c_str(), "%ju%c", &num, &null_byte);
|
||||
if (scanned != 1 || !num || is_pool && num >= POOL_ID_MAX)
|
||||
{
|
||||
fprintf(stderr, "Invalid key in etcd: %s\n", kv.key.c_str());
|
||||
@@ -250,12 +250,12 @@ void cli_tool_t::iterate_kvs_2(json11::Json kvs, const std::string & prefix, std
|
||||
bool is_inode = prefix == "/config/inode/" || prefix == "/inode/stats/";
|
||||
for (auto & kv_item: kvs.array_items())
|
||||
{
|
||||
auto kv = cli->st_cli->parse_etcd_kv(kv_item);
|
||||
auto kv = cli->st_cli.parse_etcd_kv(kv_item);
|
||||
pool_id_t pool_id = 0;
|
||||
uint64_t num = 0;
|
||||
char null_byte = 0;
|
||||
// pool+pg or pool+inode
|
||||
int scanned = sscanf(kv.key.substr(cli->st_cli->etcd_prefix.size() + prefix.size()).c_str(),
|
||||
int scanned = sscanf(kv.key.substr(cli->st_cli.etcd_prefix.size() + prefix.size()).c_str(),
|
||||
"%u/%ju%c", &pool_id, &num, &null_byte);
|
||||
if (scanned != 2 || !pool_id || is_inode && INODE_POOL(num) || !is_inode && num >= UINT32_MAX)
|
||||
{
|
||||
|
||||
@@ -1,374 +0,0 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#include <openssl/rand.h>
|
||||
#include "cli.h"
|
||||
#include "messenger.h"
|
||||
#include "msgr_encrypt.h"
|
||||
#include "msgr_op.h"
|
||||
#include "str_util.h"
|
||||
#include "xxhash.h"
|
||||
#include "xxh_x86dispatch.h"
|
||||
|
||||
// Prevent the compiler from proving that memory is unused.
|
||||
static inline void clobber_memory(const void* p, size_t n)
|
||||
{
|
||||
#if defined(__GNUC__) || defined(__clang__)
|
||||
__asm__ __volatile__("" : : "r"(p), "m"(*(const char(*)[1])p) : "memory");
|
||||
#else
|
||||
(void)p; (void)n;
|
||||
#endif
|
||||
}
|
||||
|
||||
void init_gcm(osd_client_t *cl)
|
||||
{
|
||||
cl->my_key.resize(AES_256_GCM_KEY_SIZE+AES_256_GCM_IV_SIZE);
|
||||
RAND_bytes(cl->my_key.data(), AES_256_GCM_KEY_SIZE+AES_256_GCM_IV_SIZE);
|
||||
#ifdef WITH_ISAL_CRYPTO
|
||||
cl->enc_ctx = (isal_gcm_context_data*)malloc_or_die(sizeof(isal_gcm_context_data));
|
||||
isal_aes_gcm_pre_256(cl->my_key.data(), &cl->my_key_isal);
|
||||
#else
|
||||
cl->enc_ctx = EVP_CIPHER_CTX_new();
|
||||
assert(cl->enc_ctx);
|
||||
int r = EVP_EncryptInit_ex(cl->enc_ctx, EVP_aes_256_gcm(), NULL, NULL, NULL);
|
||||
if (r != 1)
|
||||
{
|
||||
fprintf(stderr, "EncryptInit error: ");
|
||||
ERR_print_errors_fp(stderr);
|
||||
abort();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void init_gcm_round(osd_client_t *cl)
|
||||
{
|
||||
#ifdef WITH_ISAL_CRYPTO
|
||||
int r = isal_aes_gcm_init_256(&cl->my_key_isal, cl->enc_ctx, cl->my_key.data() + AES_256_GCM_KEY_SIZE, NULL, 0);
|
||||
if (r != 0)
|
||||
{
|
||||
fprintf(stderr, "isal_aes_gcm_init_256 error %d\n", r);
|
||||
abort();
|
||||
}
|
||||
#else
|
||||
int r = EVP_EncryptInit_ex(cl->enc_ctx, NULL, NULL, (uint8_t*)cl->my_key.data(), cl->my_key.data() + AES_256_GCM_KEY_SIZE);
|
||||
if (r != 1)
|
||||
{
|
||||
fprintf(stderr, "EncryptInit error: ");
|
||||
ERR_print_errors_fp(stderr);
|
||||
abort();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void finalize_gcm_round(osd_client_t *cl)
|
||||
{
|
||||
uint8_t tag[16];
|
||||
#ifdef WITH_ISAL_CRYPTO
|
||||
int r = isal_aes_gcm_enc_256_finalize(&cl->my_key_isal, cl->enc_ctx, tag, 16);
|
||||
assert(!r);
|
||||
#else
|
||||
int actual_out = 0;
|
||||
int r = EVP_EncryptFinal_ex(cl->enc_ctx, NULL, &actual_out);
|
||||
if (r != 1)
|
||||
{
|
||||
fprintf(stderr, "EncryptFinal error: ");
|
||||
ERR_print_errors_fp(stderr);
|
||||
abort();
|
||||
}
|
||||
assert(actual_out == 0);
|
||||
r = EVP_CIPHER_CTX_ctrl(cl->enc_ctx, EVP_CTRL_GCM_GET_TAG, 16, tag);
|
||||
assert(r == 1);
|
||||
#endif
|
||||
clobber_memory(&tag, sizeof(tag));
|
||||
}
|
||||
|
||||
void encrypt_gcm(osd_client_t *cl, uint8_t *in_buf, uint8_t *out_buf, size_t bufsize)
|
||||
{
|
||||
#ifdef WITH_ISAL_CRYPTO
|
||||
int r = isal_aes_gcm_enc_256_update(&cl->my_key_isal, cl->enc_ctx, out_buf, in_buf, bufsize);
|
||||
assert(!r);
|
||||
#else
|
||||
int actual_out;
|
||||
if (EVP_EncryptUpdate(cl->enc_ctx, out_buf, &actual_out, in_buf, bufsize) != 1)
|
||||
{
|
||||
fprintf(stderr, "EncryptUpdate error: ");
|
||||
ERR_print_errors_fp(stderr);
|
||||
abort();
|
||||
}
|
||||
assert(actual_out == bufsize);
|
||||
#endif
|
||||
}
|
||||
|
||||
void bench_aes_xts(uint64_t millis, size_t bufsize, int csum_status, bool quiet, int json)
|
||||
{
|
||||
size_t check_interval = 100;
|
||||
if (!quiet && !json)
|
||||
{
|
||||
printf("%s %s block... ",
|
||||
csum_status == MSGR_CSUM_GCM ? "AES-256-XTS + AES-256-GCM encrypt" :
|
||||
(csum_status == MSGR_CSUM_PAYLOAD ? "AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3" :
|
||||
(csum_status == MSGR_CSUM_FULL ? "AES-256-XTS encrypt + xxhash3" : "AES-256-XTS encrypt")),
|
||||
format_size(bufsize).c_str());
|
||||
}
|
||||
uint8_t xts_key[AES_256_XTS_KEY_SIZE];
|
||||
RAND_bytes(xts_key, AES_256_XTS_KEY_SIZE);
|
||||
uint8_t *in_buf = (uint8_t*)malloc_or_die(bufsize);
|
||||
uint8_t *out_buf = (uint8_t*)malloc_or_die(bufsize);
|
||||
XXH3_state_t *hash_state = NULL;
|
||||
osd_client_t *cl = new osd_client_t();
|
||||
cl->proto_csum_status = csum_status;
|
||||
if (csum_status == MSGR_CSUM_GCM || csum_status == MSGR_CSUM_PAYLOAD)
|
||||
{
|
||||
init_gcm(cl);
|
||||
}
|
||||
if (csum_status == MSGR_CSUM_PAYLOAD || csum_status == MSGR_CSUM_FULL)
|
||||
{
|
||||
hash_state = XXH3_createState();
|
||||
}
|
||||
op_aes_xts_encrypt_t enc;
|
||||
timespec tv_begin, tv_end;
|
||||
clock_gettime(CLOCK_REALTIME, &tv_begin);
|
||||
uint64_t iters = 0;
|
||||
while (true)
|
||||
{
|
||||
if (csum_status == MSGR_CSUM_GCM)
|
||||
{
|
||||
init_gcm_round(cl);
|
||||
}
|
||||
if (csum_status == MSGR_CSUM_PAYLOAD || csum_status == MSGR_CSUM_FULL)
|
||||
{
|
||||
XXH3_64bits_reset(hash_state);
|
||||
}
|
||||
if (csum_status == MSGR_CSUM_PAYLOAD)
|
||||
{
|
||||
init_gcm_round(cl);
|
||||
encrypt_gcm(cl, in_buf, out_buf, OSD_PACKET_SIZE);
|
||||
}
|
||||
enc.start(cl, xts_key, 0, 4096);
|
||||
size_t done_in = 0, done_out = 0;
|
||||
while (done_in < bufsize || done_out < bufsize)
|
||||
{
|
||||
enc.update(in_buf, bufsize, out_buf, bufsize, done_in, done_out);
|
||||
}
|
||||
if (csum_status == MSGR_CSUM_PAYLOAD || csum_status == MSGR_CSUM_FULL)
|
||||
{
|
||||
XXH3_64bits_update(hash_state, in_buf, bufsize);
|
||||
}
|
||||
if (csum_status == MSGR_CSUM_GCM)
|
||||
{
|
||||
finalize_gcm_round(cl);
|
||||
}
|
||||
else if (csum_status)
|
||||
{
|
||||
uint64_t hash = XXH3_64bits_digest(hash_state);
|
||||
clobber_memory(&hash, sizeof(hash));
|
||||
if (csum_status == MSGR_CSUM_PAYLOAD)
|
||||
{
|
||||
encrypt_gcm(cl, (uint8_t*)&hash, out_buf+OSD_PACKET_SIZE, sizeof(hash));
|
||||
finalize_gcm_round(cl);
|
||||
}
|
||||
}
|
||||
if (!(iters % check_interval))
|
||||
{
|
||||
clock_gettime(CLOCK_REALTIME, &tv_end);
|
||||
uint64_t passed = tv_end.tv_sec*1000 - tv_begin.tv_sec*1000 + tv_end.tv_nsec/1000000 - tv_begin.tv_nsec/1000000;
|
||||
if (passed >= millis)
|
||||
break;
|
||||
else if (passed < 10)
|
||||
check_interval *= 10;
|
||||
}
|
||||
iters++;
|
||||
}
|
||||
uint64_t result_ms = tv_end.tv_sec*1000 - tv_begin.tv_sec*1000 + tv_end.tv_nsec/1000000 - tv_begin.tv_nsec/1000000;
|
||||
double result_mbps = 1000.0 * bufsize / 1048576 * iters / result_ms;
|
||||
if (!quiet)
|
||||
{
|
||||
if (json)
|
||||
{
|
||||
printf(
|
||||
"%s{ \"xxhash3\": %s, \"aes-256-xts\": true, \"aes-256-gcm\": %s, \"bufsize\": %zu, \"iters\": %ju, \"ms\": %ju, \"mbps\": %.2f }",
|
||||
json == 1 ? "" : ",\n ",
|
||||
csum_status == MSGR_CSUM_PAYLOAD || csum_status == MSGR_CSUM_FULL ? "true" : "false",
|
||||
csum_status == MSGR_CSUM_PAYLOAD ? "\"header\"" : (csum_status == MSGR_CSUM_GCM ? "\"full\"" : "\"none\""),
|
||||
bufsize, iters, result_ms, result_mbps
|
||||
);
|
||||
}
|
||||
else
|
||||
printf("%ju iterations in %ju ms = %.2f MB/s\n", iters, result_ms, result_mbps);
|
||||
}
|
||||
if (csum_status == MSGR_CSUM_PAYLOAD || csum_status == MSGR_CSUM_FULL)
|
||||
{
|
||||
XXH3_freeState(hash_state);
|
||||
hash_state = NULL;
|
||||
}
|
||||
delete cl;
|
||||
free(out_buf);
|
||||
free(in_buf);
|
||||
}
|
||||
|
||||
void bench_aes_gcm(uint64_t millis, size_t bufsize, bool with_csum, int json)
|
||||
{
|
||||
size_t check_interval = 100;
|
||||
if (!json)
|
||||
{
|
||||
printf("%s %s block... ",
|
||||
with_csum ? "AES-256-GCM encrypt header + xxhash3" : "AES-256-GCM encrypt header and",
|
||||
format_size(bufsize).c_str());
|
||||
}
|
||||
uint8_t *in_buf = (uint8_t*)malloc_or_die(bufsize);
|
||||
uint8_t *out_buf = (uint8_t*)malloc_or_die(bufsize);
|
||||
XXH3_state_t *hash_state = NULL;
|
||||
osd_client_t *cl = new osd_client_t();
|
||||
init_gcm(cl);
|
||||
if (with_csum)
|
||||
{
|
||||
hash_state = XXH3_createState();
|
||||
}
|
||||
timespec tv_begin, tv_end;
|
||||
clock_gettime(CLOCK_REALTIME, &tv_begin);
|
||||
uint64_t iters = 0;
|
||||
while (true)
|
||||
{
|
||||
init_gcm_round(cl);
|
||||
encrypt_gcm(cl, in_buf, out_buf, OSD_PACKET_SIZE);
|
||||
if (with_csum)
|
||||
{
|
||||
XXH3_64bits_reset(hash_state);
|
||||
XXH3_64bits_update(hash_state, in_buf, bufsize);
|
||||
uint64_t hash = XXH3_64bits_digest(hash_state);
|
||||
encrypt_gcm(cl, (uint8_t*)&hash, out_buf+OSD_PACKET_SIZE, sizeof(hash));
|
||||
}
|
||||
else
|
||||
{
|
||||
encrypt_gcm(cl, in_buf, out_buf, bufsize);
|
||||
}
|
||||
finalize_gcm_round(cl);
|
||||
if (!(iters % check_interval))
|
||||
{
|
||||
clock_gettime(CLOCK_REALTIME, &tv_end);
|
||||
uint64_t passed = tv_end.tv_sec*1000 - tv_begin.tv_sec*1000 + tv_end.tv_nsec/1000000 - tv_begin.tv_nsec/1000000;
|
||||
if (passed >= millis)
|
||||
break;
|
||||
else if (passed < 10)
|
||||
check_interval *= 10;
|
||||
}
|
||||
iters++;
|
||||
}
|
||||
uint64_t result_ms = tv_end.tv_sec*1000 - tv_begin.tv_sec*1000 + tv_end.tv_nsec/1000000 - tv_begin.tv_nsec/1000000;
|
||||
double result_mbps = 1000.0 * bufsize / 1048576 * iters / result_ms;
|
||||
if (json)
|
||||
{
|
||||
printf(
|
||||
"%s{ \"xxhash3\": %s, \"aes-256-xts\": false, \"aes-256-gcm\": %s, \"bufsize\": %zu, \"iters\": %ju, \"ms\": %ju, \"mbps\": %.2f }",
|
||||
json == 1 ? "" : ",\n ",
|
||||
with_csum ? "true" : "false",
|
||||
with_csum ? "\"header\"" : "\"full\"",
|
||||
bufsize, iters, result_ms, result_mbps
|
||||
);
|
||||
}
|
||||
else
|
||||
{
|
||||
printf("%ju iterations in %ju ms = %.2f MB/s\n", iters, result_ms, result_mbps);
|
||||
}
|
||||
if (with_csum)
|
||||
{
|
||||
XXH3_freeState(hash_state);
|
||||
hash_state = NULL;
|
||||
}
|
||||
delete cl;
|
||||
free(out_buf);
|
||||
free(in_buf);
|
||||
}
|
||||
|
||||
void bench_xxh(uint64_t millis, size_t bufsize, int json)
|
||||
{
|
||||
size_t check_interval = 100;
|
||||
if (!json)
|
||||
{
|
||||
printf("xxhash3 %s block... ", format_size(bufsize).c_str());
|
||||
}
|
||||
uint8_t *in_buf = (uint8_t*)malloc_or_die(bufsize);
|
||||
XXH3_state_t *hash_state = XXH3_createState();
|
||||
timespec tv_begin, tv_end;
|
||||
clock_gettime(CLOCK_REALTIME, &tv_begin);
|
||||
uint64_t iters = 0;
|
||||
while (true)
|
||||
{
|
||||
XXH3_64bits_reset(hash_state);
|
||||
XXH3_64bits_update(hash_state, in_buf, bufsize);
|
||||
uint64_t hash = XXH3_64bits_digest(hash_state);
|
||||
clobber_memory(&hash, sizeof(hash));
|
||||
if (!(iters % check_interval))
|
||||
{
|
||||
clock_gettime(CLOCK_REALTIME, &tv_end);
|
||||
uint64_t passed = tv_end.tv_sec*1000 - tv_begin.tv_sec*1000 + tv_end.tv_nsec/1000000 - tv_begin.tv_nsec/1000000;
|
||||
if (passed >= millis)
|
||||
break;
|
||||
else if (passed < 10)
|
||||
check_interval *= 10;
|
||||
}
|
||||
iters++;
|
||||
}
|
||||
uint64_t result_ms = tv_end.tv_sec*1000 - tv_begin.tv_sec*1000 + tv_end.tv_nsec/1000000 - tv_begin.tv_nsec/1000000;
|
||||
double result_mbps = 1000.0 * bufsize / 1048576 * iters / result_ms;
|
||||
if (json)
|
||||
{
|
||||
printf(
|
||||
"%s{ \"xxhash3\": true, \"aes-256-xts\": false, \"aes-256-gcm\": \"none\", \"bufsize\": %zu, \"iters\": %ju, \"ms\": %ju, \"mbps\": %.2f }",
|
||||
json == 1 ? "" : ",\n ",
|
||||
bufsize, iters, result_ms, result_mbps
|
||||
);
|
||||
}
|
||||
else
|
||||
{
|
||||
printf("%ju iterations in %ju ms = %.2f MB/s\n", iters, result_ms, result_mbps);
|
||||
}
|
||||
XXH3_freeState(hash_state);
|
||||
hash_state = NULL;
|
||||
free(in_buf);
|
||||
}
|
||||
|
||||
// Run hardware performance tests (for now, only encryption-related)
|
||||
std::function<bool(cli_result_t &)> cli_tool_t::start_cpubench(json11::Json cfg)
|
||||
{
|
||||
int json = !!json_output;
|
||||
if (!json_output)
|
||||
{
|
||||
printf("Vitastor transport encryption benchmark (AES-256-GCM, AES-256-XTS and xxhash3)\n");
|
||||
printf("\nWarmup...\n");
|
||||
}
|
||||
else
|
||||
printf("[\n ");
|
||||
bench_aes_xts(1000, 1048576, 0, true, false);
|
||||
if (!json_output)
|
||||
printf("\nNo transport encryption, data checksums enabled, e2e unencrypted image\n");
|
||||
bench_xxh(2000, 1048576, json ? json++ : 0);
|
||||
bench_xxh(2000, 4096, json ? json++ : 0);
|
||||
if (!json_output)
|
||||
printf("\nHeader encryption with payload checksums, e2e unencrypted image\n");
|
||||
bench_aes_gcm(2000, 1048576, true, json ? json++ : 0);
|
||||
bench_aes_gcm(2000, 4096, true, json ? json++ : 0);
|
||||
if (!json_output)
|
||||
printf("\nFull transport encryption, e2e unencrypted image\n");
|
||||
bench_aes_gcm(2000, 1048576, false, json ? json++ : 0);
|
||||
bench_aes_gcm(2000, 4096, false, json ? json++ : 0);
|
||||
if (!json_output)
|
||||
printf("\nNo transport encryption, no checksums, e2e encrypted image\n");
|
||||
bench_aes_xts(2000, 1048576, 0, false, json ? json++ : 0);
|
||||
bench_aes_xts(2000, 4096, 0, false, json ? json++ : 0);
|
||||
if (!json_output)
|
||||
printf("\nNo transport encryption, e2e encrypted image, data checksums enabled\n");
|
||||
bench_aes_xts(2000, 1048576, MSGR_CSUM_FULL, false, json ? json++ : 0);
|
||||
bench_aes_xts(2000, 4096, MSGR_CSUM_FULL, false, json ? json++ : 0);
|
||||
if (!json_output)
|
||||
printf("\nHeader encryption with payload checksums, e2e encrypted image\n");
|
||||
bench_aes_xts(2000, 1048576, MSGR_CSUM_PAYLOAD, false, json ? json++ : 0);
|
||||
bench_aes_xts(2000, 4096, MSGR_CSUM_PAYLOAD, false, json ? json++ : 0);
|
||||
if (!json_output)
|
||||
printf("\nFull transport encryption, e2e encrypted image\n");
|
||||
bench_aes_xts(2000, 1048576, MSGR_CSUM_GCM, false, json ? json++ : 0);
|
||||
bench_aes_xts(2000, 4096, MSGR_CSUM_GCM, false, json ? json++ : 0);
|
||||
if (json_output)
|
||||
printf("\n]\n");
|
||||
return NULL;
|
||||
}
|
||||
+34
-34
@@ -53,7 +53,7 @@ struct image_creator_t
|
||||
|
||||
void loop()
|
||||
{
|
||||
auto & pools = parent->cli->st_cli->pool_config;
|
||||
auto & pools = parent->cli->st_cli.pool_config;
|
||||
if (state >= 1)
|
||||
goto resume_1;
|
||||
if (image_name == "")
|
||||
@@ -125,8 +125,8 @@ struct image_creator_t
|
||||
{
|
||||
return true;
|
||||
}
|
||||
auto pool_it = parent->cli->st_cli->pool_config.find(new_pool_id);
|
||||
if (pool_it == parent->cli->st_cli->pool_config.end() ||
|
||||
auto pool_it = parent->cli->st_cli.pool_config.find(new_pool_id);
|
||||
if (pool_it == parent->cli->st_cli.pool_config.end() ||
|
||||
(pool_it->second.creator_group == "" || parent->user->groups.find(pool_it->second.creator_group) == parent->user->groups.end()))
|
||||
{
|
||||
result = (cli_result_t){ .err = EACCES, .text = "Pool image create permission denied" };
|
||||
@@ -142,7 +142,7 @@ struct image_creator_t
|
||||
goto resume_2;
|
||||
else if (state == 3)
|
||||
goto resume_3;
|
||||
for (auto & ic: parent->cli->st_cli->inode_config)
|
||||
for (auto & ic: parent->cli->st_cli.inode_config)
|
||||
{
|
||||
if (ic.second.name == image_name)
|
||||
{
|
||||
@@ -222,7 +222,7 @@ resume_3:
|
||||
} while (!parent->etcd_result["succeeded"].bool_value());
|
||||
// Save into inode_config for library users to be able to take it from there immediately
|
||||
new_cfg.mod_revision = parent->etcd_result["header"]["revision"].uint64_value();
|
||||
parent->cli->st_cli->insert_inode_config(new_cfg);
|
||||
parent->cli->st_cli.insert_inode_config(new_cfg);
|
||||
auto img = parent->format_image(new_cfg);
|
||||
result = (cli_result_t){
|
||||
.err = 0,
|
||||
@@ -240,8 +240,8 @@ resume_3:
|
||||
goto resume_3;
|
||||
else if (state == 4)
|
||||
goto resume_4;
|
||||
// FIXME: take all info from etcd requests, not mixed with st_cli->inode_config
|
||||
for (auto & ic: parent->cli->st_cli->inode_config)
|
||||
// FIXME: take all info from etcd requests, not mixed with st_cli.inode_config
|
||||
for (auto & ic: parent->cli->st_cli.inode_config)
|
||||
{
|
||||
if (ic.second.name == image_name+"@"+new_snap)
|
||||
{
|
||||
@@ -307,10 +307,10 @@ resume_4:
|
||||
} while (!parent->etcd_result["succeeded"].bool_value());
|
||||
// Save into inode_config for library users to be able to take it from there immediately
|
||||
new_cfg.mod_revision = parent->etcd_result["header"]["revision"].uint64_value();
|
||||
parent->cli->st_cli->insert_inode_config(new_cfg);
|
||||
parent->cli->st_cli.insert_inode_config(new_cfg);
|
||||
{
|
||||
auto new_pool_it = parent->cli->st_cli->pool_config.find(new_pool_id);
|
||||
new_pool_name = new_pool_it != parent->cli->st_cli->pool_config.end() ? new_pool_it->second.name : "";
|
||||
auto new_pool_it = parent->cli->st_cli.pool_config.find(new_pool_id);
|
||||
new_pool_name = new_pool_it != parent->cli->st_cli.pool_config.end() ? new_pool_it->second.name : "";
|
||||
}
|
||||
result = (cli_result_t){
|
||||
.err = 0,
|
||||
@@ -337,7 +337,7 @@ resume_4:
|
||||
return json11::Json::object {
|
||||
{ "request_range", json11::Json::object {
|
||||
{ "key", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/index/maxid/"+std::to_string(new_pool_id)
|
||||
parent->cli->st_cli.etcd_prefix+"/index/maxid/"+std::to_string(new_pool_id)
|
||||
) },
|
||||
} },
|
||||
};
|
||||
@@ -349,13 +349,13 @@ resume_4:
|
||||
max_id_mod_rev = 0;
|
||||
if (response["response_range"]["kvs"].array_items().size() > 0)
|
||||
{
|
||||
auto kv = parent->cli->st_cli->parse_etcd_kv(response["response_range"]["kvs"][0]);
|
||||
auto kv = parent->cli->st_cli.parse_etcd_kv(response["response_range"]["kvs"][0]);
|
||||
new_id = 1+INODE_NO_POOL(kv.value.uint64_value());
|
||||
max_id_mod_rev = kv.mod_revision;
|
||||
}
|
||||
// Also check existing inodes - for the case when some inodes are created without changing /index/maxid
|
||||
auto ino_it = parent->cli->st_cli->inode_config.lower_bound(INODE_WITH_POOL(new_pool_id+1, 0));
|
||||
if (ino_it != parent->cli->st_cli->inode_config.begin())
|
||||
auto ino_it = parent->cli->st_cli.inode_config.lower_bound(INODE_WITH_POOL(new_pool_id+1, 0));
|
||||
if (ino_it != parent->cli->st_cli.inode_config.begin())
|
||||
{
|
||||
ino_it--;
|
||||
if (INODE_POOL(ino_it->first) == new_pool_id && new_id < 1+INODE_NO_POOL(ino_it->first))
|
||||
@@ -374,7 +374,7 @@ resume_4:
|
||||
json11::Json::object {
|
||||
{ "request_range", json11::Json::object {
|
||||
{ "key", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/index/image/"+image_name
|
||||
parent->cli->st_cli.etcd_prefix+"/index/image/"+image_name
|
||||
) },
|
||||
} },
|
||||
},
|
||||
@@ -395,7 +395,7 @@ resume_2:
|
||||
idx_mod_rev = 0;
|
||||
if (parent->etcd_result["responses"][1]["response_range"]["kvs"].array_items().size() == 0)
|
||||
{
|
||||
for (auto & ic: parent->cli->st_cli->inode_config)
|
||||
for (auto & ic: parent->cli->st_cli.inode_config)
|
||||
{
|
||||
if (ic.second.name == image_name)
|
||||
{
|
||||
@@ -411,7 +411,7 @@ resume_2:
|
||||
{
|
||||
// FIXME: Parse kvs in etcd_state_client automatically
|
||||
{
|
||||
auto kv = parent->cli->st_cli->parse_etcd_kv(parent->etcd_result["responses"][1]["response_range"]["kvs"][0]);
|
||||
auto kv = parent->cli->st_cli.parse_etcd_kv(parent->etcd_result["responses"][1]["response_range"]["kvs"][0]);
|
||||
old_id = INODE_NO_POOL(kv.value["id"].uint64_value());
|
||||
old_pool_id = (pool_id_t)kv.value["pool_id"].uint64_value();
|
||||
idx_mod_rev = kv.mod_revision;
|
||||
@@ -427,7 +427,7 @@ resume_2:
|
||||
json11::Json::object {
|
||||
{ "request_range", json11::Json::object {
|
||||
{ "key", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/config/inode/"+
|
||||
parent->cli->st_cli.etcd_prefix+"/config/inode/"+
|
||||
std::to_string(old_pool_id)+"/"+std::to_string(old_id)
|
||||
) },
|
||||
} },
|
||||
@@ -445,8 +445,8 @@ resume_3:
|
||||
return;
|
||||
}
|
||||
{
|
||||
auto kv = parent->cli->st_cli->parse_etcd_kv(parent->etcd_result["responses"][0]["response_range"]["kvs"][0]);
|
||||
cur_cfg = parent->cli->st_cli->deserialize_inode_cfg(INODE_WITH_POOL(old_pool_id, old_id), kv.value, kv.mod_revision);
|
||||
auto kv = parent->cli->st_cli.parse_etcd_kv(parent->etcd_result["responses"][0]["response_range"]["kvs"][0]);
|
||||
cur_cfg = parent->cli->st_cli.deserialize_inode_cfg(INODE_WITH_POOL(old_pool_id, old_id), kv.value, kv.mod_revision);
|
||||
size = cur_cfg.size;
|
||||
}
|
||||
}
|
||||
@@ -474,7 +474,7 @@ resume_3:
|
||||
{
|
||||
new_cfg.enc_key = cur_cfg.enc_key;
|
||||
}
|
||||
new_cfg.owner = parent->cli->st_cli->get_username();
|
||||
new_cfg.owner = http_context_get_ssl_cn(parent->cli->st_cli.get_http_ctx());
|
||||
if (!new_owner.empty())
|
||||
{
|
||||
new_cfg.owner = new_owner;
|
||||
@@ -492,7 +492,7 @@ resume_3:
|
||||
{ "target", "VERSION" },
|
||||
{ "version", 0 },
|
||||
{ "key", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/config/inode/"+
|
||||
parent->cli->st_cli.etcd_prefix+"/config/inode/"+
|
||||
std::to_string(new_pool_id)+"/"+std::to_string(new_id)
|
||||
) },
|
||||
},
|
||||
@@ -500,31 +500,31 @@ resume_3:
|
||||
{ "target", "VERSION" },
|
||||
{ "version", 0 },
|
||||
{ "key", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/index/image/"+image_name+
|
||||
parent->cli->st_cli.etcd_prefix+"/index/image/"+image_name+
|
||||
(new_snap != "" ? "@"+new_snap : "")
|
||||
) },
|
||||
},
|
||||
json11::Json::object {
|
||||
{ "target", "MOD" },
|
||||
{ "mod_revision", max_id_mod_rev },
|
||||
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/index/maxid/"+std::to_string(new_pool_id)) },
|
||||
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/index/maxid/"+std::to_string(new_pool_id)) },
|
||||
},
|
||||
};
|
||||
json11::Json::array success = json11::Json::array {
|
||||
json11::Json::object {
|
||||
{ "request_put", json11::Json::object {
|
||||
{ "key", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/config/inode/"+
|
||||
parent->cli->st_cli.etcd_prefix+"/config/inode/"+
|
||||
std::to_string(new_pool_id)+"/"+std::to_string(new_id)
|
||||
) },
|
||||
{ "value", base64_encode(
|
||||
json11::Json(parent->cli->st_cli->serialize_inode_cfg(&new_cfg)).dump()
|
||||
json11::Json(parent->cli->st_cli.serialize_inode_cfg(&new_cfg)).dump()
|
||||
) },
|
||||
} },
|
||||
},
|
||||
json11::Json::object {
|
||||
{ "request_put", json11::Json::object {
|
||||
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/index/image/"+image_name) },
|
||||
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/index/image/"+image_name) },
|
||||
{ "value", base64_encode(json11::Json(json11::Json::object{
|
||||
{ "id", new_id },
|
||||
{ "pool_id", (uint64_t)new_pool_id },
|
||||
@@ -534,7 +534,7 @@ resume_3:
|
||||
json11::Json::object {
|
||||
{ "request_put", json11::Json::object {
|
||||
{ "key", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/index/maxid/"+
|
||||
parent->cli->st_cli.etcd_prefix+"/index/maxid/"+
|
||||
std::to_string(new_pool_id)
|
||||
) },
|
||||
{ "value", base64_encode(std::to_string(new_id)) }
|
||||
@@ -545,7 +545,7 @@ resume_3:
|
||||
json11::Json::object {
|
||||
{ "request_range", json11::Json::object {
|
||||
{ "key", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/index/image/"+
|
||||
parent->cli->st_cli.etcd_prefix+"/index/image/"+
|
||||
image_name+(new_snap != "" ? "@"+new_snap : "")
|
||||
) },
|
||||
} },
|
||||
@@ -560,29 +560,29 @@ resume_3:
|
||||
{ "target", "MOD" },
|
||||
{ "mod_revision", cur_cfg.mod_revision },
|
||||
{ "key", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/config/inode/"+
|
||||
parent->cli->st_cli.etcd_prefix+"/config/inode/"+
|
||||
std::to_string(old_pool_id)+"/"+std::to_string(old_id)
|
||||
) },
|
||||
});
|
||||
checks.push_back(json11::Json::object {
|
||||
{ "target", "MOD" },
|
||||
{ "mod_revision", idx_mod_rev },
|
||||
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/index/image/"+image_name) }
|
||||
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/index/image/"+image_name) }
|
||||
});
|
||||
success.push_back(json11::Json::object {
|
||||
{ "request_put", json11::Json::object {
|
||||
{ "key", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/config/inode/"+
|
||||
parent->cli->st_cli.etcd_prefix+"/config/inode/"+
|
||||
std::to_string(old_pool_id)+"/"+std::to_string(old_id)
|
||||
) },
|
||||
{ "value", base64_encode(
|
||||
json11::Json(parent->cli->st_cli->serialize_inode_cfg(&snap_cfg)).dump()
|
||||
json11::Json(parent->cli->st_cli.serialize_inode_cfg(&snap_cfg)).dump()
|
||||
) },
|
||||
} },
|
||||
});
|
||||
success.push_back(json11::Json::object {
|
||||
{ "request_put", json11::Json::object {
|
||||
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/index/image/"+image_name+"@"+new_snap) },
|
||||
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/index/image/"+image_name+"@"+new_snap) },
|
||||
{ "value", base64_encode(json11::Json(json11::Json::object{
|
||||
{ "id", old_id },
|
||||
{ "pool_id", (uint64_t)old_pool_id },
|
||||
|
||||
+20
-26
@@ -52,19 +52,19 @@ struct dd_in_info_t
|
||||
in_seekable = true;
|
||||
if (iimg != "")
|
||||
{
|
||||
iwatch = parent->cli->st_cli->watch_inode(iimg);
|
||||
iwatch = parent->cli->st_cli.watch_inode(iimg);
|
||||
if (!iwatch->cfg.num)
|
||||
{
|
||||
result = (cli_result_t){ .err = ENOENT, .text = "Image "+iimg+" does not exist" };
|
||||
parent->cli->st_cli->close_watch(iwatch);
|
||||
parent->cli->st_cli.close_watch(iwatch);
|
||||
iwatch = NULL;
|
||||
return;
|
||||
}
|
||||
auto pool_it = parent->cli->st_cli->pool_config.find(INODE_POOL(iwatch->cfg.num));
|
||||
if (pool_it == parent->cli->st_cli->pool_config.end())
|
||||
auto pool_it = parent->cli->st_cli.pool_config.find(INODE_POOL(iwatch->cfg.num));
|
||||
if (pool_it == parent->cli->st_cli.pool_config.end())
|
||||
{
|
||||
result = (cli_result_t){ .err = ENOENT, .text = "Pool of image "+iimg+" does not exist" };
|
||||
parent->cli->st_cli->close_watch(iwatch);
|
||||
parent->cli->st_cli.close_watch(iwatch);
|
||||
iwatch = NULL;
|
||||
return;
|
||||
}
|
||||
@@ -131,7 +131,7 @@ struct dd_in_info_t
|
||||
{
|
||||
if (iimg != "")
|
||||
{
|
||||
parent->cli->st_cli->close_watch(iwatch);
|
||||
parent->cli->st_cli.close_watch(iwatch);
|
||||
iwatch = NULL;
|
||||
}
|
||||
else if (ifile != "")
|
||||
@@ -163,11 +163,11 @@ struct dd_out_info_t
|
||||
|
||||
pool_config_t *find_pool(cli_tool_t *parent, const std::string & name)
|
||||
{
|
||||
if (name == "" && parent->cli->st_cli->pool_config.size() == 1)
|
||||
if (name == "" && parent->cli->st_cli.pool_config.size() == 1)
|
||||
{
|
||||
return &parent->cli->st_cli->pool_config.begin()->second;
|
||||
return &parent->cli->st_cli.pool_config.begin()->second;
|
||||
}
|
||||
for (auto & pp: parent->cli->st_cli->pool_config)
|
||||
for (auto & pp: parent->cli->st_cli.pool_config)
|
||||
{
|
||||
if (pp.second.name == name)
|
||||
{
|
||||
@@ -186,14 +186,14 @@ struct dd_out_info_t
|
||||
if (oimg != "")
|
||||
{
|
||||
out_seekable = true;
|
||||
owatch = parent->cli->st_cli->watch_inode(oimg);
|
||||
owatch = parent->cli->st_cli.watch_inode(oimg);
|
||||
if (owatch->cfg.num)
|
||||
{
|
||||
auto pool_it = parent->cli->st_cli->pool_config.find(INODE_POOL(owatch->cfg.num));
|
||||
if (pool_it == parent->cli->st_cli->pool_config.end())
|
||||
auto pool_it = parent->cli->st_cli.pool_config.find(INODE_POOL(owatch->cfg.num));
|
||||
if (pool_it == parent->cli->st_cli.pool_config.end())
|
||||
{
|
||||
result = (cli_result_t){ .err = ENOENT, .text = "Pool of image "+oimg+" does not exist" };
|
||||
parent->cli->st_cli->close_watch(owatch);
|
||||
parent->cli->st_cli.close_watch(owatch);
|
||||
owatch = NULL;
|
||||
return true;
|
||||
}
|
||||
@@ -209,7 +209,7 @@ struct dd_out_info_t
|
||||
else
|
||||
{
|
||||
result = (cli_result_t){ .err = ENOENT, .text = "Pool to create output image "+oimg+" is not specified" };
|
||||
parent->cli->st_cli->close_watch(owatch);
|
||||
parent->cli->st_cli.close_watch(owatch);
|
||||
owatch = NULL;
|
||||
return true;
|
||||
}
|
||||
@@ -224,14 +224,14 @@ struct dd_out_info_t
|
||||
if (!out_create)
|
||||
{
|
||||
result = (cli_result_t){ .err = ENOENT, .text = "Image "+oimg+" does not exist" };
|
||||
parent->cli->st_cli->close_watch(owatch);
|
||||
parent->cli->st_cli.close_watch(owatch);
|
||||
owatch = NULL;
|
||||
return true;
|
||||
}
|
||||
if (!out_size)
|
||||
{
|
||||
result = (cli_result_t){ .err = ENOENT, .text = "Input size is unknown, specify size to create output image "+oimg };
|
||||
parent->cli->st_cli->close_watch(owatch);
|
||||
parent->cli->st_cli.close_watch(owatch);
|
||||
owatch = NULL;
|
||||
return true;
|
||||
}
|
||||
@@ -247,7 +247,7 @@ struct dd_out_info_t
|
||||
if (!out_size)
|
||||
{
|
||||
result = (cli_result_t){ .err = ENOENT, .text = "Input size is unknown, specify size to truncate output image" };
|
||||
parent->cli->st_cli->close_watch(owatch);
|
||||
parent->cli->st_cli.close_watch(owatch);
|
||||
owatch = NULL;
|
||||
return true;
|
||||
}
|
||||
@@ -275,7 +275,7 @@ resume_1:
|
||||
sub_cb = NULL;
|
||||
if (result.err)
|
||||
{
|
||||
parent->cli->st_cli->close_watch(owatch);
|
||||
parent->cli->st_cli.close_watch(owatch);
|
||||
owatch = NULL;
|
||||
return true;
|
||||
}
|
||||
@@ -324,14 +324,8 @@ resume_2:
|
||||
cluster_op_t *sync_op = new cluster_op_t;
|
||||
sync_op->opcode = OSD_OP_SYNC;
|
||||
parent->waiting++;
|
||||
sync_op->callback = [this, parent](cluster_op_t *sync_op)
|
||||
sync_op->callback = [parent](cluster_op_t *sync_op)
|
||||
{
|
||||
if (sync_op->retval != 0 && !result.err)
|
||||
{
|
||||
// Just in case, actually OP_SYNC can't fail
|
||||
result.err = -sync_op->retval;
|
||||
result.text = "Failed to sync "+oimg+": "+std::string(strerror(result.err));
|
||||
}
|
||||
parent->waiting--;
|
||||
delete sync_op;
|
||||
parent->ringloop->wakeup();
|
||||
@@ -360,7 +354,7 @@ resume_2:
|
||||
{
|
||||
if (oimg != "")
|
||||
{
|
||||
parent->cli->st_cli->close_watch(owatch);
|
||||
parent->cli->st_cli.close_watch(owatch);
|
||||
owatch = NULL;
|
||||
}
|
||||
else
|
||||
|
||||
@@ -72,7 +72,7 @@ struct cli_describe_t
|
||||
only_pool = pool_id;
|
||||
if (!only_pool && pool_name != "")
|
||||
{
|
||||
for (auto & pp: parent->cli->st_cli->pool_config)
|
||||
for (auto & pp: parent->cli->st_cli.pool_config)
|
||||
{
|
||||
if (pp.second.name == pool_name)
|
||||
{
|
||||
@@ -153,7 +153,7 @@ struct cli_describe_t
|
||||
{
|
||||
uint64_t min_pool = min_inode >> (64-POOL_ID_BITS);
|
||||
uint64_t max_pool = max_inode >> (64-POOL_ID_BITS);
|
||||
for (auto & pp: parent->cli->st_cli->pool_config)
|
||||
for (auto & pp: parent->cli->st_cli.pool_config)
|
||||
{
|
||||
if (pp.first >= min_pool && (!max_pool || pp.first <= max_pool))
|
||||
{
|
||||
|
||||
+2
-2
@@ -134,8 +134,8 @@ struct cli_fix_t
|
||||
return;
|
||||
}
|
||||
auto & obj = objects[processed_count++];
|
||||
auto pool_cfg_it = parent->cli->st_cli->pool_config.find(INODE_POOL(obj.inode));
|
||||
if (pool_cfg_it == parent->cli->st_cli->pool_config.end())
|
||||
auto pool_cfg_it = parent->cli->st_cli.pool_config.find(INODE_POOL(obj.inode));
|
||||
if (pool_cfg_it == parent->cli->st_cli.pool_config.end())
|
||||
{
|
||||
fprintf(stderr, "Object %jx:%jx is from unknown pool\n", obj.inode, obj.stripe);
|
||||
continue;
|
||||
|
||||
@@ -47,8 +47,8 @@ struct snap_flattener_t
|
||||
chain_list.push_back(cur->num);
|
||||
while (cur->parent_id != 0 && cur->parent_id != target_cfg->num)
|
||||
{
|
||||
auto it = parent->cli->st_cli->inode_config.find(cur->parent_id);
|
||||
if (it == parent->cli->st_cli->inode_config.end())
|
||||
auto it = parent->cli->st_cli.inode_config.find(cur->parent_id);
|
||||
if (it == parent->cli->st_cli.inode_config.end())
|
||||
{
|
||||
result = (cli_result_t){
|
||||
.err = ENOENT,
|
||||
@@ -103,9 +103,9 @@ struct snap_flattener_t
|
||||
{ "from", top_parent_name },
|
||||
{ "to", target_name },
|
||||
{ "target", target_name },
|
||||
{ "delete_source", false },
|
||||
{ "delete-source", false },
|
||||
{ "cas", use_cas },
|
||||
{ "fsync_interval", fsync_interval },
|
||||
{ "fsync-interval", fsync_interval },
|
||||
});
|
||||
// Wait for it
|
||||
resume_1:
|
||||
|
||||
+17
-17
@@ -38,7 +38,7 @@ struct image_lister_t
|
||||
{
|
||||
if (list_pool_name != "")
|
||||
{
|
||||
for (auto & ic: parent->cli->st_cli->pool_config)
|
||||
for (auto & ic: parent->cli->st_cli.pool_config)
|
||||
{
|
||||
if (ic.second.name == list_pool_name)
|
||||
{
|
||||
@@ -54,11 +54,11 @@ struct image_lister_t
|
||||
}
|
||||
}
|
||||
auto begin_it = list_pool_id
|
||||
? parent->cli->st_cli->inode_config.lower_bound(INODE_WITH_POOL(list_pool_id, 0))
|
||||
: parent->cli->st_cli->inode_config.begin();
|
||||
? parent->cli->st_cli.inode_config.lower_bound(INODE_WITH_POOL(list_pool_id, 0))
|
||||
: parent->cli->st_cli.inode_config.begin();
|
||||
auto end_it = list_pool_id
|
||||
? parent->cli->st_cli->inode_config.lower_bound(INODE_WITH_POOL(list_pool_id+1, 0))
|
||||
: parent->cli->st_cli->inode_config.end();
|
||||
? parent->cli->st_cli.inode_config.lower_bound(INODE_WITH_POOL(list_pool_id+1, 0))
|
||||
: parent->cli->st_cli.inode_config.end();
|
||||
for (auto it = begin_it; it != end_it; it++)
|
||||
{
|
||||
if (!parent->check_image_perm(it->second, false))
|
||||
@@ -81,21 +81,21 @@ struct image_lister_t
|
||||
json11::Json::object {
|
||||
{ "request_range", (list_pool_id
|
||||
? json11::Json::object {
|
||||
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/pool/stats/"+std::to_string(list_pool_id)) },
|
||||
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/pool/stats/"+std::to_string(list_pool_id)) },
|
||||
}
|
||||
: json11::Json::object {
|
||||
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/pool/stats/") },
|
||||
{ "range_end", base64_encode(parent->cli->st_cli->etcd_prefix+"/pool/stats0") },
|
||||
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/pool/stats/") },
|
||||
{ "range_end", base64_encode(parent->cli->st_cli.etcd_prefix+"/pool/stats0") },
|
||||
}) },
|
||||
},
|
||||
json11::Json::object {
|
||||
{ "request_range", json11::Json::object {
|
||||
{ "key", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/inode/stats"+
|
||||
parent->cli->st_cli.etcd_prefix+"/inode/stats"+
|
||||
(list_pool_id ? "/"+std::to_string(list_pool_id) : "")+"/"
|
||||
) },
|
||||
{ "range_end", base64_encode(
|
||||
parent->cli->st_cli->etcd_prefix+"/inode/stats"+
|
||||
parent->cli->st_cli.etcd_prefix+"/inode/stats"+
|
||||
(list_pool_id ? "/"+std::to_string(list_pool_id) : "")+"0"
|
||||
) },
|
||||
} },
|
||||
@@ -117,11 +117,11 @@ resume_1:
|
||||
std::map<pool_id_t, uint64_t> pool_pg_real_size;
|
||||
for (auto & kv_item: space_info["responses"][0]["response_range"]["kvs"].array_items())
|
||||
{
|
||||
auto kv = parent->cli->st_cli->parse_etcd_kv(kv_item);
|
||||
auto kv = parent->cli->st_cli.parse_etcd_kv(kv_item);
|
||||
// pool ID
|
||||
pool_id_t pool_id;
|
||||
char null_byte = 0;
|
||||
int scanned = sscanf(kv.key.substr(parent->cli->st_cli->etcd_prefix.length()).c_str(), "/pool/stats/%u%c", &pool_id, &null_byte);
|
||||
int scanned = sscanf(kv.key.substr(parent->cli->st_cli.etcd_prefix.length()).c_str(), "/pool/stats/%u%c", &pool_id, &null_byte);
|
||||
if (scanned != 1 || !pool_id || pool_id >= POOL_ID_MAX)
|
||||
{
|
||||
fprintf(stderr, "Invalid key in etcd: %s\n", kv.key.c_str());
|
||||
@@ -132,12 +132,12 @@ resume_1:
|
||||
}
|
||||
for (auto & kv_item: space_info["responses"][1]["response_range"]["kvs"].array_items())
|
||||
{
|
||||
auto kv = parent->cli->st_cli->parse_etcd_kv(kv_item);
|
||||
auto kv = parent->cli->st_cli.parse_etcd_kv(kv_item);
|
||||
// pool ID & inode number
|
||||
pool_id_t pool_id;
|
||||
inode_t only_inode_num;
|
||||
char null_byte = 0;
|
||||
int scanned = sscanf(kv.key.substr(parent->cli->st_cli->etcd_prefix.length()).c_str(),
|
||||
int scanned = sscanf(kv.key.substr(parent->cli->st_cli.etcd_prefix.length()).c_str(),
|
||||
"/inode/stats/%u/%ju%c", &pool_id, &only_inode_num, &null_byte);
|
||||
if (scanned != 2 || !pool_id || pool_id >= POOL_ID_MAX || INODE_POOL(only_inode_num) != 0)
|
||||
{
|
||||
@@ -152,8 +152,8 @@ resume_1:
|
||||
continue;
|
||||
}
|
||||
// save stats
|
||||
auto pool_it = parent->cli->st_cli->pool_config.find(pool_id);
|
||||
if (pool_it != parent->cli->st_cli->pool_config.end())
|
||||
auto pool_it = parent->cli->st_cli.pool_config.find(pool_id);
|
||||
if (pool_it != parent->cli->st_cli.pool_config.end())
|
||||
{
|
||||
auto & pool_cfg = pool_it->second;
|
||||
used_size = used_size / (pool_pg_real_size[pool_id] ? pool_pg_real_size[pool_id] : 1)
|
||||
@@ -166,7 +166,7 @@ resume_1:
|
||||
{ "size", 0 },
|
||||
{ "readonly", false },
|
||||
{ "pool_id", (uint64_t)INODE_POOL(inode_num) },
|
||||
{ "pool_name", pool_it != parent->cli->st_cli->pool_config.end()
|
||||
{ "pool_name", pool_it != parent->cli->st_cli.pool_config.end()
|
||||
? (pool_it->second.name == "" ? "<Unnamed>" : pool_it->second.name) : "?" },
|
||||
{ "inode_num", INODE_NO_POOL(inode_num) },
|
||||
{ "inode_id", inode_num },
|
||||
|
||||
+7
-13
@@ -110,8 +110,8 @@ struct snap_merger_t
|
||||
cur->parent_id != to_cfg->num &&
|
||||
cur->parent_id != 0)
|
||||
{
|
||||
auto it = parent->cli->st_cli->inode_config.find(cur->parent_id);
|
||||
if (it == parent->cli->st_cli->inode_config.end())
|
||||
auto it = parent->cli->st_cli.inode_config.find(cur->parent_id);
|
||||
if (it == parent->cli->st_cli.inode_config.end())
|
||||
{
|
||||
result = (cli_result_t){
|
||||
.err = ENOENT,
|
||||
@@ -166,7 +166,7 @@ struct snap_merger_t
|
||||
//
|
||||
// <from> - <layer 1> - <target> - <to>
|
||||
// \- <layer 2> <---------X-------- NOT ALLOWED
|
||||
for (auto & ic: parent->cli->st_cli->inode_config)
|
||||
for (auto & ic: parent->cli->st_cli.inode_config)
|
||||
{
|
||||
auto it = sources.find(ic.second.num);
|
||||
if (it == sources.end() && ic.second.parent_id != 0)
|
||||
@@ -182,7 +182,7 @@ struct snap_merger_t
|
||||
.text = "Layers at or above "+(check_delete_source ? from_name : target_name)+
|
||||
", but below "+to_name+" are not allowed to have other children, but "+
|
||||
ic.second.name+" is a child of "+
|
||||
parent->cli->st_cli->inode_config.at(ic.second.parent_id).name,
|
||||
parent->cli->st_cli.inode_config.at(ic.second.parent_id).name,
|
||||
};
|
||||
state = 100;
|
||||
return;
|
||||
@@ -213,7 +213,7 @@ struct snap_merger_t
|
||||
|
||||
uint64_t get_block_size(inode_t inode, uint32_t *bitmap_granularity)
|
||||
{
|
||||
auto & pool_cfg = parent->cli->st_cli->pool_config.at(INODE_POOL(inode));
|
||||
auto & pool_cfg = parent->cli->st_cli.pool_config.at(INODE_POOL(inode));
|
||||
uint64_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
|
||||
if (bitmap_granularity)
|
||||
*bitmap_granularity = pool_cfg.bitmap_granularity;
|
||||
@@ -251,8 +251,6 @@ struct snap_merger_t
|
||||
goto resume_100;
|
||||
// Get parents and so on
|
||||
start_merge();
|
||||
if (state == 100)
|
||||
return;
|
||||
// First list lower layers
|
||||
list_errcode.clear();
|
||||
list_layers(true);
|
||||
@@ -420,7 +418,7 @@ struct snap_merger_t
|
||||
}
|
||||
if (!pgs_left)
|
||||
{
|
||||
auto & name = parent->cli->st_cli->inode_config.at(src).name;
|
||||
auto & name = parent->cli->st_cli.inode_config.at(src).name;
|
||||
if (list_errcode.find(src) != list_errcode.end())
|
||||
{
|
||||
fprintf(stderr, "Failed to get listing of layer %s (inode %ju in pool %u): %s (code %d)\n",
|
||||
@@ -614,10 +612,8 @@ struct snap_merger_t
|
||||
subop->offset = offset;
|
||||
subop->len = 0;
|
||||
subop->flags = OSD_OP_IGNORE_READONLY | OSD_OP_WAIT_UP_TIMEOUT;
|
||||
in_flight++;
|
||||
subop->callback = [this](cluster_op_t *subop)
|
||||
subop->callback = [](cluster_op_t *subop)
|
||||
{
|
||||
in_flight--;
|
||||
if (subop->retval != 0)
|
||||
{
|
||||
fprintf(stderr, "error deleting from layer 0x%jx at offset %jx: %s", subop->inode, subop->offset, strerror(-subop->retval));
|
||||
@@ -644,10 +640,8 @@ struct snap_merger_t
|
||||
uint64_t to = last_written_offset;
|
||||
cluster_op_t *subop = new cluster_op_t;
|
||||
subop->opcode = OSD_OP_SYNC;
|
||||
in_flight++;
|
||||
subop->callback = [this, to](cluster_op_t *subop)
|
||||
{
|
||||
in_flight--;
|
||||
delete subop;
|
||||
// We can now delete source data between <from> and <to>
|
||||
// But to do this we have to keep all object lists in memory :-(
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user