Compare commits

..
9 Commits
Author SHA1 Message Date
Vitaliy Filippov 4acfe149cb Release 3.0.10
Important bug fixes (new store):
- Fix OSDs possibly refusing to start with "write metadata failed at offset xxx: Invalid argument"
  (fix buffer alignment during initial garbage collection)
- Rollback change from 3.0.4 - on-disk garbage entries are not skipped on start again. This change
  doesn't have any impact normally, but OSDs originally running 3.0.0-3.0.2 and then upgraded
  to 3.0.9 may hit a bug where 3.0.9 refuses to start due to entries marked as garbage too
  early and flushed to disk in 3.0.0-3.0.2.

Other changes:
- Auto-select the only RDMA device/port if there is only one
- Rollback one 3.0.9 change - there was no actual use-after-free :)
  (the problem was only relevant to an unstable development version)
- Fix `vitastor-nfs --trace` option
- Fix an unintended 1 second sleep in vitastor-cli rm-data
- Fix inode statistics not being cleared for a deleted pool
- Fix print to stdout in client
2026-04-27 13:53:48 +03:00
Vitaliy Filippov 008ed5b269 Rollback 25ecca7625 - there was no actual use-after-free :) 2026-04-22 01:35:38 +03:00
Vitaliy Filippov 4fffe0f032 Fix vitastor-nfs trace option 2026-04-20 21:51:47 +03:00
Vitaliy Filippov a76d5ccc0d Wakeup callers in rm-data 2026-04-17 13:53:32 +03:00
Vitaliy Filippov 8ed1e180e0 Clear inode_stats in mon 2026-04-14 02:44:46 +03:00
Vitaliy Filippov 8832fc3b14 Fix print to stdout in client 2026-04-14 02:43:50 +03:00
Vitaliy Filippov 0134934c99 Do not skip garbage entries on start (rollback change from 3.0.4) 2026-04-11 12:05:02 +03:00
Vitaliy Filippov 2e36f292bd Fix buffer alignment during metadata clearing on init 2026-04-10 21:56:12 +03:00
Vitaliy Filippov bcc6419760 Auto-select the only RDMA device/port if there is only one 2026-04-08 15:46:10 +03:00
144 changed files with 1126 additions and 16412 deletions
-126
View File
@@ -234,60 +234,6 @@ jobs:
echo ""
done
test_etcd_fail_https:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 10
run: ETCD_SCHEME=https /root/vitastor/tests/test_etcd_fail.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_etcd_fail_https_antietcd:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 10
run: ETCD_SCHEME=https ANTIETCD=1 /root/vitastor/tests/test_etcd_fail.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_snapshot_https:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: ETCD_SCHEME=https /root/vitastor/tests/test_snapshot.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_interrupted_rebalance:
runs-on: ubuntu-latest
needs: build
@@ -702,24 +648,6 @@ jobs:
echo ""
done
test_snapshot_chain_encrypted:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: ENCRYPTED=1 /root/vitastor/tests/test_snapshot_chain.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_old_snapshot_chain:
runs-on: ubuntu-latest
needs: build
@@ -1296,24 +1224,6 @@ jobs:
echo ""
done
test_checksum_xxhash:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: TEST_NAME=xxhash OSD_ARGS="--data_csum_type xxh3_32" /root/vitastor/tests/test_checksum.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_old_checksum:
runs-on: ubuntu-latest
needs: build
@@ -2142,39 +2052,3 @@ jobs:
echo ""
done
test_write_encrypted:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: /root/vitastor/tests/test_write_encrypted.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_write_encrypted_ec:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: SCHEME=ec /root/vitastor/tests/test_write_encrypted.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
-8
View File
@@ -38,14 +38,6 @@ for my $line (<>)
{
$test_name .= '_antietcd';
}
elsif ($1 eq 'ETCD_SCHEME' && $2 eq 'https')
{
$test_name .= '_https';
}
elsif ($1 eq 'ENCRYPTED')
{
$test_name .= '_encrypted';
}
elsif ($1 eq 'OLD')
{
$test_name =~ s/^test_/test_old_/s;
-1
View File
@@ -3,4 +3,3 @@
package-lock.json
fio
qemu
node_modules
+1 -1
View File
@@ -2,7 +2,7 @@ cmake_minimum_required(VERSION 2.8...3.30)
project(vitastor)
set(VITASTOR_VERSION "3.0.9")
set(VITASTOR_VERSION "3.0.10")
include(CTest)
-1
View File
@@ -62,7 +62,6 @@ Vitastor поддерживает QEMU-драйвер, протоколы UBLK,
- [Дисковые параметры OSD](docs/config/layout-osd.ru.md)
- [Прочие параметры OSD](docs/config/osd.ru.md)
- [Параметры мониторов](docs/config/monitor.ru.md)
- [Безопасность](docs/config/security.ru.md)
- [Настройки пулов](docs/config/pool.ru.md)
- [Метаданные образов в etcd](docs/config/inode.ru.md)
- Использование
-1
View File
@@ -62,7 +62,6 @@ Read more details in the documentation. You can start from here: [Quick Start](d
- [OSD Disk Layout](docs/config/layout-osd.en.md)
- [OSD Runtime Parameters](docs/config/osd.en.md)
- [Monitor](docs/config/monitor.en.md)
- [Security](docs/config/security.en.md)
- [Pool configuration](docs/config/pool.en.md)
- [Image metadata in etcd](docs/config/inode.en.md)
- Usage
+1 -1
View File
@@ -1,4 +1,4 @@
VITASTOR_VERSION ?= v3.0.9
VITASTOR_VERSION ?= v3.0.10
all: build push
+1 -1
View File
@@ -49,7 +49,7 @@ spec:
capabilities:
add: ["SYS_ADMIN"]
allowPrivilegeEscalation: true
image: vitalif/vitastor-csi:v3.0.9
image: vitalif/vitastor-csi:v3.0.10
args:
- "--node=$(NODE_ID)"
- "--endpoint=$(CSI_ENDPOINT)"
+1 -1
View File
@@ -121,7 +121,7 @@ spec:
privileged: true
capabilities:
add: ["SYS_ADMIN"]
image: vitalif/vitastor-csi:v3.0.9
image: vitalif/vitastor-csi:v3.0.10
args:
- "--node=$(NODE_ID)"
- "--endpoint=$(CSI_ENDPOINT)"
+1 -1
View File
@@ -5,7 +5,7 @@ package vitastor
const (
vitastorCSIDriverName = "csi.vitastor.io"
vitastorCSIDriverVersion = "3.0.9"
vitastorCSIDriverVersion = "3.0.10"
)
// Config struct fills the parameters of request or user input
+1 -1
View File
@@ -1,4 +1,4 @@
vitastor (3.0.9-1) unstable; urgency=medium
vitastor (3.0.10-1) unstable; urgency=medium
* Bugfixes
+1 -1
View File
@@ -3,7 +3,7 @@ Section: admin
Priority: optional
Maintainer: Vitaliy Filippov <vitalif@yourcmc.ru>
Build-Depends: debhelper, g++ (>= 8), libstdc++6 (>= 8),
linux-libc-dev, libgoogle-perftools-dev, libjerasure-dev, libgf-complete-dev, libc-ares-dev,
linux-libc-dev, libgoogle-perftools-dev, libjerasure-dev, libgf-complete-dev,
libibverbs-dev, librdmacm-dev, libisal-dev, cmake, pkg-config, libnl-3-dev, libnl-genl-3-dev,
node-bindings <!nocheck>, node-gyp, node-nan
Standards-Version: 4.5.0
+1 -1
View File
@@ -25,7 +25,7 @@ RUN set -e -x; \
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
RUN apt-get update && \
apt-get -y install fio libgoogle-perftools-dev devscripts libjerasure-dev cmake libc-ares-dev \
apt-get -y install fio libgoogle-perftools-dev devscripts libjerasure-dev cmake \
libibverbs-dev librdmacm-dev libisal-dev libnl-3-dev libnl-genl-3-dev curl nodejs npm node-nan node-bindings && \
apt-get -y build-dep fio && \
apt-get --download-only source fio
+1 -1
View File
@@ -1,4 +1,4 @@
VITASTOR_VERSION ?= v3.0.9
VITASTOR_VERSION ?= v3.0.10
all: build push
+1 -1
View File
@@ -4,7 +4,7 @@
#
# Desired Vitastor version
VITASTOR_VERSION=v3.0.9
VITASTOR_VERSION=v3.0.10
# Additional arguments for all containers
# For example, you may want to specify a custom logging driver here
-1
View File
@@ -38,4 +38,3 @@ In the future, additional configuration methods may be added:
- [OSD Disk Layout](config/layout-osd.en.md)
- [OSD Runtime Parameters](config/osd.en.md)
- [Monitor](config/monitor.en.md)
- [Security Parameters](config/security.en.md)
-1
View File
@@ -41,4 +41,3 @@
- [Дисковые параметры OSD](config/layout-osd.ru.md)
- [Прочие параметры OSD](config/osd.ru.md)
- [Параметры мониторов](config/monitor.ru.md)
- [Параметры безопасности](config/security.ru.md)
+2 -8
View File
@@ -198,14 +198,8 @@ put a modified value into etcd key /vitastor/config/global.
- Type: string
- Default: none
Data and metadata checksum type to use. May be "crc32c", "xxh3_32" or "none".
Select crc32c or xxh3_32 and set csum_block_size to enable data checksums.
Both crc32c and xxh3_32 are almost equally fast, xxh3_32 is safer. xxh3_32 is
the xxhash3 algorithm truncated from 64 to 32 bits (which is still a good hash).
Note that enabled data checksums either increase memory usage or reduce
performance. Check details in [csum_block_size](#csum_block_size) description.
Data checksum type to use. May be "crc32c" or "none". Set to "crc32c" to
enable data checksums.
## csum_block_size
+2 -6
View File
@@ -209,12 +209,8 @@ journal_block_size и meta_block_size. Однако на данный момен
- Тип: строка
- Значение по умолчанию: none
Тип используемых OSD контрольных сумм данных и метаданных. Может быть "crc32c",
"xxh3_32" или "none". Выберите crc32c или xxh3_32 и установите csum_block_size,
чтобы включить контрольные суммы данных.
И crc32c, и xxh3_32 примерно одинаково быстры, xxh3_32 надёжней. xxh3_32 - это
алгоритм xxhash3, обрезанный с 64 до 32 бит (это всё равно хороший хеш).
Тип используемых OSD контрольных сумм данных. Может быть "crc32c" или "none".
Установите в "crc32c", чтобы включить расчёт и проверку контрольных сумм данных.
Следует понимать, что контрольные суммы в зависимости от размера блока их
расчёта либо увеличивают потребление памяти, либо снижают производительность.
-150
View File
@@ -1,150 +0,0 @@
[Documentation](../../README.md#documentation) → [Configuration](../config.en.md) → Security Parameters
-----
[Читать на русском](security.ru.md)
# Security Parameters
These parameters affect your Vitastor installation security and apply to OSDs, monitors and clients.
Most of them can be set in /etc/vitastor/vitastor.conf and in etcd, but don't support online modification.
- [etcd_client_cert](#etcd_client_cert)
- [etcd_client_key](#etcd_client_key)
- [etcd_ca](#etcd_ca)
- [osd_etcd_client_cert](#osd_etcd_client_cert)
- [osd_etcd_client_key](#osd_etcd_client_key)
- [mon_etcd_client_cert](#mon_etcd_client_cert)
- [mon_etcd_client_key](#mon_etcd_client_key)
- [vault_url](#vault_url)
- [vault_secret_api_path](#vault_secret_api_path)
- [vault_client_cert](#vault_client_cert)
- [vault_client_key](#vault_client_key)
- [vault_ca](#vault_ca)
- [vault_timeout_ms](#vault_timeout_ms)
- [vault_error_timeout_sec](#vault_error_timeout_sec)
- [vault_refresh_leeway_sec](#vault_refresh_leeway_sec)
- [max_aes_xts_pool_size](#max_aes_xts_pool_size)
## etcd_client_cert
- Type: string
Client TLS certificate to use for Vitastor client (not OSD and not monitor)
etcd https connections. May be path to a file or just a PEM string with certificate.
In the latter case, string must begin with "-----BEGIN CERTIFICATE-----".
## etcd_client_key
- Type: string
Private key for etcd_client_cert (also a file or a PEM string).
## etcd_ca
- Type: string
Trusted TLS CA to verify etcd server certificate. May be path to a file,
directory or just a PEM string with certificate.
## osd_etcd_client_cert
- Type: string
Same as [etcd_client_cert](#etcd_client_cert), but only for OSDs.
OSDs, clients and monitors should have different permissions, so they should
use different certificates.
## osd_etcd_client_key
- Type: string
Same as [etcd_client_key](#etcd_client_key), but only for OSDs.
## mon_etcd_client_cert
- Type: string
Same as [etcd_client_cert](#etcd_client_cert), but only for Vitastor monitors.
## mon_etcd_client_key
- Type: string
Same as [etcd_client_key](#etcd_client_key), but only for Vitastor monitors.
## vault_url
- Type: string
Vault base URL.
Vitastor clients support AES-256-XTS image data encryption with different per-image keys.
Encryption is performed by the client, OSDs don't have access to decrypted data.
Encryption keys may be stored in etcd or, for the increased security level, in an external
[HashiCorp Vault](https://developer.hashicorp.com/vault/) or [OpenBao](https://openbao.org/)
instance.
Vitastor clients use [v1 k/v secrets engine](https://openbao.org/api-docs/secret/kv/kv-v1/)
and [TLS authentication engine](https://openbao.org/api-docs/auth/cert/) in Vault.
In that case, only key IDs are stored in etcd.
## vault_secret_api_path
- Type: string
- Default: /v1/secret/
Vault v1 secret API mount path to use.
## vault_client_cert
- Type: string
Client TLS certificate to use for Vault connections. Just like [etcd_client_cert](#etcd_client_cert),
may be path to a file or just a certificate in PEM string.
## vault_client_key
- Type: string
Private key for vault_client_cert (also a file or a PEM string).
## vault_ca
- Type: string
Trusted TLS CA to verify Vault server certificate. May be path to a file,
directory or just a PEM string with certificate.
## vault_timeout_ms
- Type: integer
- Default: 5000
Timeout for Vault requests in milliseconds.
## vault_error_timeout_sec
- Type: integer
- Default: 60
Time (in seconds) to wait before retrying after receiving an error from Vault.
## vault_refresh_leeway_sec
- Type: integer
- Default: 60
Extra time (in seconds) before real Vault token lease_timeout to refresh it, just
in case of system clock drift.
## max_aes_xts_pool_size
- Type: integer
- Default: 256
Maximum number of OpenSSL encryption contexts cached in OSD memory. Probably
doesn't require modification.
-154
View File
@@ -1,154 +0,0 @@
[Документация](../../README-ru.md#документация) → [Конфигурация](../config.ru.md) → Параметры безопасности
-----
[Read in English](security.en.md)
# Параметры безопасности
Данные параметры затрагивают безопасность инсталляций Vitastor и используются
OSD, мониторами и клиентами.
Большая их часть может задаваться в /etc/vitastor/vitastor.conf и в etcd, но не
поддерживает онлайн-изменение.
- [etcd_client_cert](#etcd_client_cert)
- [etcd_client_key](#etcd_client_key)
- [etcd_ca](#etcd_ca)
- [osd_etcd_client_cert](#osd_etcd_client_cert)
- [osd_etcd_client_key](#osd_etcd_client_key)
- [mon_etcd_client_cert](#mon_etcd_client_cert)
- [mon_etcd_client_key](#mon_etcd_client_key)
- [vault_url](#vault_url)
- [vault_secret_api_path](#vault_secret_api_path)
- [vault_client_cert](#vault_client_cert)
- [vault_client_key](#vault_client_key)
- [vault_ca](#vault_ca)
- [vault_timeout_ms](#vault_timeout_ms)
- [vault_error_timeout_sec](#vault_error_timeout_sec)
- [vault_refresh_leeway_sec](#vault_refresh_leeway_sec)
- [max_aes_xts_pool_size](#max_aes_xts_pool_size)
## etcd_client_cert
- Тип: строка
Клиентский TLS сертификат для https-подключений к etcd для клиентов Vitastor
(не OSD и не мониторов). Может быть путём к файлу или просто строкой с
сертификатом в формате PEM. В последнем случае строка должна начинаться с
"-----BEGIN CERTIFICATE-----".
## etcd_client_key
- Тип: строка
Закрытый ключ для сертификата etcd_client_cert (также путь к файлу или PEM строка).
## etcd_ca
- Тип: строка
Доверенный корневой TLS-сертификат для проверки сертификата сервера etcd.
Может быть путём к файлу, директории или просто строкой с сертификатом в
формате PEM.
## osd_etcd_client_cert
- Тип: строка
Аналогично [etcd_client_cert](#etcd_client_cert), но только для OSD.
OSD, клиенты и мониторы должны иметь разные привилегии, поэтому они должны
использовать разные сертификаты.
## osd_etcd_client_key
- Тип: строка
Аналогично [etcd_client_key](#etcd_client_key), но только для OSD.
## mon_etcd_client_cert
- Тип: строка
Аналогично [etcd_client_cert](#etcd_client_cert), но только для мониторов Vitastor.
## mon_etcd_client_key
- Тип: строка
Аналогично [etcd_client_key](#etcd_client_key), но только для мониторов Vitastor.
## vault_url
- Тип: строка
Базовый адрес Vault.
Клиенты Vitastor поддерживают AES-256-XTS шифрование данных образов с отдельными ключами на
каждый образ. Данные шифруются клиентами, OSD не имеют доступа к незашифрованным данным.
Ключи шифрования могут храниться в etcd или, для повышенного уровня безопасности, во внешнем
[HashiCorp Vault](https://developer.hashicorp.com/vault/) или [OpenBao](https://openbao.org/).
Клиенты Vitastor используют [движок секретов v1](https://openbao.org/api-docs/secret/kv/kv-v1/)
и [TLS-аутентификацию](https://openbao.org/api-docs/auth/cert/) в Vault.
В этом случае, только ID ключей хранятся в etcd.
## vault_secret_api_path
- Тип: строка
- Значение по умолчанию: /v1/secret/
Путь к API секретов v1 для использования клиентами.
## vault_client_cert
- Тип: строка
Клиентский TLS сертификат для подключений к Vault. Как и [etcd_client_cert](#etcd_client_cert),
может быть путём к файлу или просто PEM-строкой с сертификатом.
## vault_client_key
- Тип: строка
Закрытый ключ для сертификата vault_client_cert (также путь к файлу или PEM строка).
## vault_ca
- Тип: строка
Доверенный корневой TLS-сертификат для проверки сертификата сервера Vault.
Может быть путём к файлу, директории или просто строкой с сертификатом в
формате PEM.
## vault_timeout_ms
- Тип: целое число
- Значение по умолчанию: 5000
Максимально время выполнения Vault-запросов в миллисекундах.
## vault_error_timeout_sec
- Тип: целое число
- Значение по умолчанию: 60
Время (в секундах) для ожидания перед повторной попыткой при получении ошибки от Vault.
## vault_refresh_leeway_sec
- Тип: целое число
- Значение по умолчанию: 60
Зазор времени (в секундах), чтобы обновлять токены Vault чуть раньше их реального
lease_timeout, на случай "ухода" системных часов.
## max_aes_xts_pool_size
- Тип: целое число
- Значение по умолчанию: 256
Максимальное количество кэшируемых в памяти OSD контекстов шифрования OpenSSL.
Вряд ли требует изменения.
-2
View File
@@ -44,8 +44,6 @@
{{../../config/monitor.en.md|indent=2}}
{{../../config/security.en.md|indent=2}}
{{../../config/pool.en.md|indent=2}}
{{../../config/inode.en.md|indent=2}}
-2
View File
@@ -44,8 +44,6 @@
{{../../config/monitor.ru.md|indent=2}}
{{../../config/security.ru.md|indent=2}}
{{../../config/pool.ru.md|indent=2}}
{{../../config/inode.ru.md|indent=2}}
+4 -14
View File
@@ -233,21 +233,11 @@
type: string
default: none
info: |
Data and metadata checksum type to use. May be "crc32c", "xxh3_32" or "none".
Select crc32c or xxh3_32 and set csum_block_size to enable data checksums.
Both crc32c and xxh3_32 are almost equally fast, xxh3_32 is safer. xxh3_32 is
the xxhash3 algorithm truncated from 64 to 32 bits (which is still a good hash).
Note that enabled data checksums either increase memory usage or reduce
performance. Check details in [csum_block_size](#csum_block_size) description.
Data checksum type to use. May be "crc32c" or "none". Set to "crc32c" to
enable data checksums.
info_ru: |
Тип используемых OSD контрольных сумм данных и метаданных. Может быть "crc32c",
"xxh3_32" или "none". Выберите crc32c или xxh3_32 и установите csum_block_size,
чтобы включить контрольные суммы данных.
И crc32c, и xxh3_32 примерно одинаково быстры, xxh3_32 надёжней. xxh3_32 - это
алгоритм xxhash3, обрезанный с 64 до 32 бит (это всё равно хороший хеш).
Тип используемых OSD контрольных сумм данных. Может быть "crc32c" или "none".
Установите в "crc32c", чтобы включить расчёт и проверку контрольных сумм данных.
Следует понимать, что контрольные суммы в зависимости от размера блока их
расчёта либо увеличивают потребление памяти, либо снижают производительность.
-5
View File
@@ -1,5 +0,0 @@
{
"dependencies": {
"yaml": "^2.8.2"
}
}
-5
View File
@@ -1,5 +0,0 @@
# Security Parameters
These parameters affect your Vitastor installation security and apply to OSDs, monitors and clients.
Most of them can be set in /etc/vitastor/vitastor.conf and in etcd, but don't support online modification.
-7
View File
@@ -1,7 +0,0 @@
# Параметры безопасности
Данные параметры затрагивают безопасность инсталляций Vitastor и используются
OSD, мониторами и клиентами.
Большая их часть может задаваться в /etc/vitastor/vitastor.conf и в etcd, но не
поддерживает онлайн-изменение.
-131
View File
@@ -1,131 +0,0 @@
- name: etcd_client_cert
type: string
info: |
Client TLS certificate to use for Vitastor client (not OSD and not monitor)
etcd https connections. May be path to a file or just a PEM string with certificate.
In the latter case, string must begin with "-----BEGIN CERTIFICATE-----".
info_ru: |
Клиентский TLS сертификат для https-подключений к etcd для клиентов Vitastor
(не OSD и не мониторов). Может быть путём к файлу или просто строкой с
сертификатом в формате PEM. В последнем случае строка должна начинаться с
"-----BEGIN CERTIFICATE-----".
- name: etcd_client_key
type: string
info: Private key for etcd_client_cert (also a file or a PEM string).
info_ru: Закрытый ключ для сертификата etcd_client_cert (также путь к файлу или PEM строка).
- name: etcd_ca
type: string
info: |
Trusted TLS CA to verify etcd server certificate. May be path to a file,
directory or just a PEM string with certificate.
info_ru: |
Доверенный корневой TLS-сертификат для проверки сертификата сервера etcd.
Может быть путём к файлу, директории или просто строкой с сертификатом в
формате PEM.
- name: osd_etcd_client_cert
type: string
info: |
Same as [etcd_client_cert](#etcd_client_cert), but only for OSDs.
OSDs, clients and monitors should have different permissions, so they should
use different certificates.
info_ru: |
Аналогично [etcd_client_cert](#etcd_client_cert), но только для OSD.
OSD, клиенты и мониторы должны иметь разные привилегии, поэтому они должны
использовать разные сертификаты.
- name: osd_etcd_client_key
type: string
info: Same as [etcd_client_key](#etcd_client_key), but only for OSDs.
info_ru: Аналогично [etcd_client_key](#etcd_client_key), но только для OSD.
- name: mon_etcd_client_cert
type: string
info: Same as [etcd_client_cert](#etcd_client_cert), but only for Vitastor monitors.
info_ru: Аналогично [etcd_client_cert](#etcd_client_cert), но только для мониторов Vitastor.
- name: mon_etcd_client_key
type: string
info: Same as [etcd_client_key](#etcd_client_key), but only for Vitastor monitors.
info_ru: Аналогично [etcd_client_key](#etcd_client_key), но только для мониторов Vitastor.
- name: vault_url
type: string
info: |
Vault base URL.
Vitastor clients support AES-256-XTS image data encryption with different per-image keys.
Encryption is performed by the client, OSDs don't have access to decrypted data.
Encryption keys may be stored in etcd or, for the increased security level, in an external
[HashiCorp Vault](https://developer.hashicorp.com/vault/) or [OpenBao](https://openbao.org/)
instance.
Vitastor clients use [v1 k/v secrets engine](https://openbao.org/api-docs/secret/kv/kv-v1/)
and [TLS authentication engine](https://openbao.org/api-docs/auth/cert/) in Vault.
In that case, only key IDs are stored in etcd.
info_ru: |
Базовый адрес Vault.
Клиенты Vitastor поддерживают AES-256-XTS шифрование данных образов с отдельными ключами на
каждый образ. Данные шифруются клиентами, OSD не имеют доступа к незашифрованным данным.
Ключи шифрования могут храниться в etcd или, для повышенного уровня безопасности, во внешнем
[HashiCorp Vault](https://developer.hashicorp.com/vault/) или [OpenBao](https://openbao.org/).
Клиенты Vitastor используют [движок секретов v1](https://openbao.org/api-docs/secret/kv/kv-v1/)
и [TLS-аутентификацию](https://openbao.org/api-docs/auth/cert/) в Vault.
В этом случае, только ID ключей хранятся в etcd.
- name: vault_secret_api_path
type: string
default: /v1/secret/
info: Vault v1 secret API mount path to use.
info_ru: Путь к API секретов v1 для использования клиентами.
- name: vault_client_cert
type: string
info: |
Client TLS certificate to use for Vault connections. Just like [etcd_client_cert](#etcd_client_cert),
may be path to a file or just a certificate in PEM string.
info_ru: |
Клиентский TLS сертификат для подключений к Vault. Как и [etcd_client_cert](#etcd_client_cert),
может быть путём к файлу или просто PEM-строкой с сертификатом.
- name: vault_client_key
type: string
info: Private key for vault_client_cert (also a file or a PEM string).
info_ru: Закрытый ключ для сертификата vault_client_cert (также путь к файлу или PEM строка).
- name: vault_ca
type: string
info: |
Trusted TLS CA to verify Vault server certificate. May be path to a file,
directory or just a PEM string with certificate.
info_ru: |
Доверенный корневой TLS-сертификат для проверки сертификата сервера Vault.
Может быть путём к файлу, директории или просто строкой с сертификатом в
формате PEM.
- name: vault_timeout_ms
type: int
default: 5000
info: Timeout for Vault requests in milliseconds.
info_ru: Максимально время выполнения Vault-запросов в миллисекундах.
- name: vault_error_timeout_sec
type: int
default: 60
info: |
Time (in seconds) to wait before retrying after receiving an error from Vault.
info_ru: |
Время (в секундах) для ожидания перед повторной попыткой при получении ошибки от Vault.
- name: vault_refresh_leeway_sec
type: int
default: 60
info: |
Extra time (in seconds) before real Vault token lease_timeout to refresh it, just
in case of system clock drift.
info_ru: |
Зазор времени (в секундах), чтобы обновлять токены Vault чуть раньше их реального
lease_timeout, на случай "ухода" системных часов.
- name: max_aes_xts_pool_size
type: int
default: 256
info: |
Maximum number of OpenSSL encryption contexts cached in OSD memory. Probably
doesn't require modification.
info_ru: |
Максимальное количество кэшируемых в памяти OSD контекстов шифрования OpenSSL.
Вряд ли требует изменения.
+2 -2
View File
@@ -26,9 +26,9 @@ at Vitastor Kubernetes operator: https://github.com/Antilles7227/vitastor-operat
The instruction is very simple.
1. Download a Docker image of the desired version: \
`docker pull vitalif/vitastor:v3.0.9`
`docker pull vitalif/vitastor:v3.0.10`
2. Install scripts to the host system: \
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.9 install.sh`
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.10 install.sh`
3. Reload udev rules: \
`udevadm control --reload-rules`
4. Enable the vitastor-host service: \
+2 -2
View File
@@ -25,9 +25,9 @@ Vitastor можно установить в Docker/Podman. При этом etcd,
Инструкция по установке максимально простая.
1. Скачайте Docker-образ желаемой версии: \
`docker pull vitalif/vitastor:v3.0.9`
`docker pull vitalif/vitastor:v3.0.10`
2. Установите скрипты в хост-систему командой: \
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.9 install.sh`
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.10 install.sh`
3. Перезагрузите правила udev: \
`udevadm control --reload-rules`
4. Включите сервис vitastor-host: \
+2 -2
View File
@@ -15,8 +15,8 @@
- gcc and g++ 8 or newer, clang 10 or newer, or other compiler with C++11 plus
designated initializers support from C++20
- CMake
- jerasure, c-ares headers and libraries
- ISA-L, libibverbs, librdmacm, libnl3 headers and libraries (optional)
- jerasure headers and libraries
- ISA-L, libibverbs and librdmacm headers and libraries (optional)
- tcmalloc (google-perftools-dev)
## Basic instructions
+2 -2
View File
@@ -15,8 +15,8 @@
- gcc и g++ >= 8, либо clang >= 10, либо другой компилятор с поддержкой C++11 плюс
назначенных инициализаторов (designated initializers) из C++20
- CMake
- Заголовки и библиотеки jerasure, c-ares
- Опционально - заголовки и библиотеки ISA-L, libibverbs, librdmacm, libnl3
- Заголовки и библиотеки jerasure
- Опционально - заголовки и библиотеки ISA-L, libibverbs, librdmacm
- tcmalloc (google-perftools-dev)
## Базовая инструкция
-2
View File
@@ -41,8 +41,6 @@
- [Built-in Prometheus metric exporter](../config/monitor.en.md#enable_prometheus)
- [NFS RDMA support](../usage/nfs.en.md#rdma) (probably also usable for GPUDirect)
- [S3](../installation/s3.en.md)
- [TLS support for etcd connections](../config/security.en.md)
- [AES-256-XTS image encryption](../usage/cli.en.md#create) and [Vault support](../config/security.en.md#vault_url) for key storage
## Plugins and tools
-2
View File
@@ -43,8 +43,6 @@
- [Встроенный Prometheus-экспортер метрик](../config/monitor.ru.md#enable_prometheus)
- [Поддержка NFS RDMA](../usage/nfs.ru.md#rdma) (вероятно, также подходящая для GPUDirect)
- [S3](../installation/s3.ru.md)
- [Поддержка TLS-соединений с etcd](../config/security.ru.md)
- [AES-256-XTS шифрование данных](../usage/cli.ru.md#create) и [поддержка Vault](../config/security.ru.md#vault_url) для хранения ключей
## Драйверы и инструменты
+7 -21
View File
@@ -125,31 +125,18 @@ bench-kaveri kaveri 10 G 10 G 0 B/s 0 0 0 us 0 B/s 0
## create
`vitastor-cli create -s|--size SIZE [OPTIONS] <name>`
`vitastor-cli create -s|--size <size> [-p|--pool <id|name>] [--parent <parent_name>[@<snapshot>]] <name>`
Create an image. Options:
* `-s|--size SIZE` - New image size in bytes or with a K/M/G/T unit suffix.
* `-p|--pool POOL` - Specify pool for the new image (may be omitted if there is only 1 pool).
* `--parent PARENT` - Create a copy-on-write image clone based on PARENT (or PARENT@SNAPSHOT).
If parent is not a snapshot, it must be a read-only image.
* `--enc-key random` - Generate a new random AES-256-XTS encryption key for the new image.
* `--enc-key HEX` - Set a specified AES-256-XTS key (64 bytes in hex) for the new image.
* `--enc-key vault:ID` - Use an encryption key from an external Vault secret with specified ID.
Create an image. You may use K/M/G/T suffixes for `<size>`. If `--parent` is specified,
a copy-on-write image clone is created. Parent must be a snapshot (readonly image).
Pool must be specified if there is more than one pool.
```
vitastor-cli create --snapshot <snapshot> [OPTIONS] <image>
vitastor-cli snap-create [OPTIONS] <image>@<snapshot>
vitastor-cli create --snapshot <snapshot> [-p|--pool <id|name>] <image>
vitastor-cli snap-create [-p|--pool <id|name>] <image>@<snapshot>
```
Create a snapshot of image `<image>`. May be used live if only a single writer is active.
Options:
* `-p|--pool POOL` - Move image to pool POOL, leaving the snapshot in the old pool.
* `--enc-key random` - Change image encryption key to a new random AES-256-XTS key.
* `--enc-key KEY` - Change image encryption key to a specified key, Vault key or to an empty key.
By default, the image retains its old encryption key when taking a snapshot.
Create a snapshot of image `<name>` (either form can be used). May be used live if only a single writer is active.
See also about [how to export snapshots](qemu.en.md#exporting-snapshots).
@@ -164,7 +151,6 @@ You should resize file system in the image, if present, before shrinking it.
* `--deleted 1|0` - Set/clear 'deleted image' flag (set automatically during unfinished deletes).
* `-f|--force` - Proceed with shrinking or setting readwrite flag even if the image has children.
* `--down-ok` - Proceed with shrinking even if some data will be left on unavailable OSDs.
* `--enc-key HEX` - Change image encryption key (allowed only with `--force`).
## dd
+8 -22
View File
@@ -127,32 +127,19 @@ bench-kaveri kaveri 10 G 10 G 0 B/s 0 0 0 us 0 B/s 0
## create
`vitastor-cli create -s|--size SIZE [ОПЦИИ] <name>`
`vitastor-cli create -s|--size <size> [-p|--pool <id|name>] [--parent <parent_name>[@<snapshot>]] <name>`
Создать образ. Опции:
* `-s|--size SIZE` - Размер нового образа в байтах или с суффиксом K/M/G/T (кило/мега/гига/терабайт).
* `-p|--pool POOL` - Создать образ в заданном пуле (можно не указывать, если пул всего один).
* `--parent PARENT` - Создать легковесный клон на основе образа `PARENT` или снимка `PARENT@SNAP`.
Если `PARENT` - не снимок, он должен быть помечен как образ только для чтения.
* `--enc-key random` - Сгенерировать случайный ключ шифрования AES-256-XTS для нового образа.
* `--enc-key HEX` - Установить заданный ключ AES-256-XTS (64 байта в hex) для нового образа.
* `--enc-key vault:ID` - Использовать ключ из внешнего секрета с заданным ID из Vault.
Создать образ. Для размера `<size>` можно использовать суффиксы K/M/G/T (килобайт-мегабайт-гигабайт-терабайт).
Если указана опция `--parent`, создаётся клон образа. Родитель `<parent_name>[@<snapshot>]` должен быть
снимком (или просто немодифицируемым образом). Пул обязательно указывать, если в кластере больше одного пула.
```
vitastor-cli create --snapshot <snapshot> [ОПЦИИ] <image>
vitastor-cli snap-create [ОПЦИИ] <image>@<snapshot>
vitastor-cli create --snapshot <snapshot> [-p|--pool <id|name>] <image>
vitastor-cli snap-create [-p|--pool <id|name>] <image>@<snapshot>
```
Создать снимок образа `<image>` (можно использовать любую форму команды).
Снимок можно создавать без остановки клиентов, если пишущих клиентов не больше одного.
Опции:
* `-p|--pool POOL` - Переместить образ в пул POOL, оставив снимок в старом пуле.
* `--enc-key random` - Изменить ключ шифрования образа на новый случайный ключ AES-256-XTS.
* `--enc-key KEY` - Изменить ключ шифрования образа на заданный ключ, ключ из Vault или пустой ключ.
По умолчанию шифрованные образы сохраняют старый ключ при снятии снимка.
Создать снимок образа `<name>` (можно использовать любую форму команды). Снимок можно создавать без остановки
клиентов, если пишущий клиент максимум 1.
Смотрите также информацию о том, [как экспортировать снимки](qemu.ru.md#экспорт-снимков).
@@ -169,7 +156,6 @@ vitastor-cli snap-create [ОПЦИИ] <image>@<snapshot>
* `--deleted 1|0` - Установить/снять флаг "образ удалён" (устанавливается при незавершённом удалении).
* `-f|--force` - Разрешить уменьшение или перевод в чтение-запись образа, у которого есть клоны.
* `--down-ok` - Разрешить уменьшение, даже если часть данных останется неудалённой на недоступных OSD.
* `--enc-key HEX` - Изменить ключ шифрования образа (разрешено только с `--force`).
## dd
+1
View File
@@ -262,3 +262,4 @@ Options:
| `--logfile <FILE>` | log to the specified file |
| `--enforce 1` | enforce permissions at the server side (no by default) |
| `--foreground 1` | stay in foreground, do not daemonize |
| `--trace` | trace all NFS requests |
+1
View File
@@ -274,3 +274,4 @@ VitastorFS из GPUDirect.
| `--logfile <FILE>` | записывать логи в заданный файл |
| `--enforce 1` | проверять права доступа на стороне сервера (по умолчанию нет) |
| `--foreground 1` | не уходить в фон после запуска |
| `--trace` | логгировать все запросы NFS |
+8 -23
View File
@@ -18,7 +18,7 @@ class AntiEtcdAdapter
cluster = cluster ? (''+(cluster||'')).split(/,+/) : [];
cluster = Object.keys(cluster.reduce((a, url) =>
{
a[url.toLowerCase().replace(/^(https?:\/\/)?(.*?)(\/.*)?$/, (m, m1, m2) => (m1||'http://')+m2)] = true;
a[url.toLowerCase().replace(/^(https?:\/\/)/, '').replace(/\/.*$/, '')] = true;
return a;
}, {}));
const cfg_port = config.antietcd_port;
@@ -26,8 +26,7 @@ class AntiEtcdAdapter
is_local['0.0.0.0'] = true;
is_local['::'] = true;
is_local[''] = true;
// split :, 3 -> <schema>:<//ip>:<port>
const selected = cluster.map(s => s.split(':', 3)).filter(ip => is_local[ip[1].substr(2)] && (!cfg_port || ip[2] == cfg_port));
const selected = cluster.map(s => s.split(':', 2)).filter(ip => is_local[ip[0]] && (!cfg_port || ip[1] == cfg_port));
if (selected.length > 1)
{
console.error('More than 1 etcd_address matches local IPs, please specify port');
@@ -36,30 +35,16 @@ class AntiEtcdAdapter
else if (selected.length == 1)
{
const antietcd_config = {
ip: selected[0][1].substr(2),
port: selected[0][2],
cert: config.antietcd_cert,
key: config.antietcd_key,
ca: config.etcd_ca,
data: config.antietcd_data_file || ((config.antietcd_data_dir || '/var/lib/vitastor') + '/mon_'+selected[0][2]+'.json.gz'),
ip: selected[0][0],
port: selected[0][1],
data: config.antietcd_data_file || ((config.antietcd_data_dir || '/var/lib/vitastor') + '/mon_'+selected[0][1]+'.json.gz'),
persist_filter: vitastor_persist_filter({ vitastor_prefix: config.etcd_prefix || '/vitastor' }),
node_id: selected[0][1].substr(2)+':'+selected[0][2], // node_id = ip:port
cluster: (cluster.length == 1 ? null : cluster.reduce((a, c) => { a[c.replace(/^(https?:\/\/)/, '')] = c; return a; }, {})),
node_id: selected[0][0]+':'+selected[0][1], // node_id = ip:port
cluster: (cluster.length == 1 ? null : cluster.reduce((a, c) => { a[c] = "http://"+c; return a; }, {})),
cluster_key: (config.etcd_prefix || '/vitastor'),
stale_read: 1,
log_level: 1,
};
if (config.use_auth)
{
antietcd_config.client_cert_auth = true;
antietcd_config.auth_filter = require('./vitastor_auth_filter.js');
antietcd_config.peer_ca = config.antietcd_server_ca;
if (!config.antietcd_server_ca || config.antietcd_server_ca == config.etcd_ca)
{
console.error('Secure setup requires separate antietcd_server_ca (for signing antietcd server certificates) and etcd_ca (for signing client certificates)');
process.exit(1);
}
}
for (const key in config)
{
if (key.substr(0, 9) === 'antietcd_')
@@ -184,7 +169,7 @@ class AntiEtcdAdapter
await new Promise(ok => setTimeout(ok, timeout-(Date.now()-prev)));
}
prev = Date.now();
const res = await this.antietcd.api(path.replace(/^\/+/, '').replace(/\/+$/, '').replace(/\/+/g, '_'), body, { username: 'root' });
const res = await this.antietcd.api(path.replace(/^\/+/, '').replace(/\/+$/, '').replace(/\/+/g, '_'), body);
if (res.error)
{
console.error('Failed to query antietcd '+path+' (retry '+retry+'/'+retries+'): '+res.error);
+6 -27
View File
@@ -1,9 +1,7 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 (see README.md for details)
const fs = require('fs');
const http = require('http');
const https = require('https');
const WebSocket = require('ws');
const { b64, local_ips } = require('./utils.js');
@@ -17,30 +15,11 @@ class EtcdAdapter
this.ws = null;
this.ws_alive = false;
this.ws_keepalive_timer = null;
this.opts = {};
}
parse_config(config)
{
this.parse_etcd_addresses(config.etcd_address||config.etcd_url);
if (config.mon_etcd_client_cert || config.etcd_client_cert)
{
this.opts.cert = config.mon_etcd_client_cert || config.etcd_client_cert;
if (this.opts.cert.substr(0, 5) != '-----')
this.opts.cert = fs.readFileSync(this.opts.cert, { encoding: 'utf-8' });
}
if (config.mon_etcd_client_key || config.etcd_client_key)
{
this.opts.key = config.mon_etcd_client_key || config.etcd_client_key;
if (this.opts.key.substr(0, 5) != '-----')
this.opts.key = fs.readFileSync(this.opts.key, { encoding: 'utf-8' });
}
if (config.etcd_ca)
{
this.opts.ca = config.etcd_ca;
if (this.opts.ca.substr(0, 5) != '-----')
this.opts.ca = fs.readFileSync(this.opts.ca, { encoding: 'utf-8' });
}
}
parse_etcd_addresses(addrs)
@@ -60,7 +39,7 @@ class EtcdAdapter
for (let url of addrs)
{
let scheme = 'http';
url = url.trim().replace(/^(https?):\/\//i, (m, m1) => { scheme = m1.toLowerCase(); return ''; });
url = url.trim().replace(/^(https?):\/\//, (m, m1) => { scheme = m1; return ''; });
const slash = url.indexOf('/');
const colon = url.indexOf(':');
const is_local = is_local_ip[colon >= 0 ? url.substr(0, colon) : (slash >= 0 ? url.substr(0, slash) : url)];
@@ -151,7 +130,7 @@ class EtcdAdapter
}
ok(false);
}, this.mon.config.etcd_mon_timeout);
this.ws = new WebSocket(base+'/watch', this.opts);
this.ws = new WebSocket(base+'/watch');
this.ws_used_url = cur_addr;
const fail = () =>
{
@@ -293,7 +272,7 @@ class EtcdAdapter
{
throw new Error(MON_STOPPED);
}
const res = await POST(base+path, body, timeout, this.opts);
const res = await POST(base+path, body, timeout);
if (this.mon.stopped)
{
throw new Error(MON_STOPPED);
@@ -319,7 +298,7 @@ class EtcdAdapter
}
}
function POST(url, body, timeout, opts)
function POST(url, body, timeout)
{
return new Promise(ok =>
{
@@ -331,10 +310,10 @@ function POST(url, body, timeout, opts)
req = null;
ok({ error: 'timeout' });
}, timeout) : null;
let req = (url.substr(0, 5) == 'https' ? https : http).request(url, { method: 'POST', headers: {
let req = http.request(url, { method: 'POST', headers: {
'Content-Type': 'application/json',
'Content-Length': body_text.length,
}, ...(opts||{}) }, (res) =>
} }, (res) =>
{
if (!req)
{
+1 -22
View File
@@ -16,7 +16,6 @@ const etcd_allow = new RegExp('^'+[
'config/pools',
'config/osd/[1-9]\\d*',
'config/pgs', // old name
'config/user/.*',
'pg/config',
'config/inode/[1-9]\\d*/[1-9]\\d*',
'osd/state/[1-9]\\d*',
@@ -46,14 +45,7 @@ const etcd_tree = {
config_path: "/etc/vitastor/vitastor.conf",
etcd_prefix: "/vitastor",
// etcd connection - configurable online
etcd_address: "http://10.0.115.10:2379/v3",
etcd_client_cert: "",
etcd_client_key: "",
osd_etcd_client_cert: "",
osd_etcd_client_key: "",
mon_etcd_client_cert: "",
mon_etcd_client_key: "",
etcd_ca: "",
etcd_address: "10.0.115.10:2379/v3",
// mon
etcd_mon_ttl: 5, // min: 1
etcd_mon_timeout: 1000, // ms. min: 0
@@ -209,8 +201,6 @@ const etcd_tree = {
primary_affinity_tags?: 'nvme' | [ 'nvme', ... ],
// scrub interval
scrub_interval?: '30d',
// users allowed to create images in this pool
creator_group?: '',
},
...
}, */
@@ -227,21 +217,10 @@ const etcd_tree = {
parent_id?: <inode_t>,
readonly?: boolean,
deleted?: boolean,
enc_key?: string,
owner?: string,
owner_group?: string,
reader_group?: string,
}
}
}, */
inode: {},
/* user: {
<username>: {
type: 'osd'|'mon'|'admin'|'client',
groups: string[],
},
}, */
user: {},
},
osd: {
state: {
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "vitastor-mon",
"version": "3.0.9",
"version": "3.0.10",
"description": "Vitastor SDS monitor service",
"main": "mon-main.js",
"scripts": {
+1
View File
@@ -37,6 +37,7 @@ function derive_osd_stats(st, prev, prev_diff)
const n = c.count - BigInt(pr && pr.count||0);
diff.recovery_stats[op] = { ...c, bps: n > 0 ? b*1000n/timediff : 0n, iops: n > 0 ? n*1000n/timediff : 0n };
}
diff.inode_stats = {};
for (const pool_id in st.inode_stats||{})
{
diff.inode_stats[pool_id] = {};
-471
View File
@@ -1,471 +0,0 @@
// AntiEtcd authentication filter for Vitastor
// (c) Vitaliy Filippov, 2026
// License: Mozilla Public License 2.0 or Vitastor Network Public License 1.1
// Permissions are based on:
// 1. Users.
// Stored in /vitastor/config/user/<username>.
// Has 2 properties:
// - type, one of: osd, mon, admin, client.
// osd, mon types should be used by OSDs/monitors.
// admin should be used for administrative access from vitastor-cli.
// client should be used for regular clients.
// - groups, a list of group names the user is included in.
// 2. Images.
// Stored in /vitastor/config/inode/<pool>/<inode>. Has the following properties:
// - owner (user name)
// - owner_group (group name)
// - reader_group
const static_perms = {
invalid: {
keys: {},
prefixes: {},
},
osd: {
keys: { '/pg/config': false },
prefixes: { '/osd/': true, '/pg/state/': true, '/pg/history/': true, '/pgstats/': true },
},
mon: {
keys: { '/pg/config': true, '/stats': true, '/history/last_clean_pgs': true },
prefixes: {
'/config/': false, '/osd/': false, '/mon/': true, '/pg/history/': true,
'/pgstats/': false, '/inode/stats/': true, '/pool/stats/': true,
},
},
admin: {
keys: { '/stats': false },
prefixes: {
'/config/': true, '/osd/': true, '/index/': true, '/pg/history/': true,
'/mon/': false, '/pg/': false, '/pgstats/': false, '/inode/stats/': false, '/pool/stats/': false,
},
},
client: {
keys: { '/config/global': false, '/config/node_placement': false, '/config/pools': false, '/pg/config': false },
prefixes: { '/osd/stats/': false, '/pg/state/': false, '/index/maxid/': false },
},
};
const api_perms = {
osd: { lease_grant: true, lease_revoke: true, lease_keepalive: true },
mon: { lease_grant: true, lease_revoke: true, lease_keepalive: true },
admin: { maintenance_status: true },
client: {},
};
class VitastorAuthFilter
{
constructor(antietcd)
{
this.cfg = antietcd.cfg;
this.antietcd = antietcd;
this.prefix = this.cfg.vitastor_prefix || '/vitastor';
this.prefix_parts = this.prefix.split('/');
}
_get(path, decode)
{
let cur = this.antietcd.etctree.state;
path = path instanceof Array ? path : path.split('/');
for (const p of path)
{
if (!cur.children)
{
return null;
}
cur = cur.children[p];
if (!cur)
{
return null;
}
}
if (decode)
{
return this._decode(path, cur.value);
}
return cur;
}
_decode(path, cur)
{
if (!cur)
{
return null;
}
if (cur)
{
try
{
cur = JSON.parse(cur);
}
catch (e)
{
console.warn('Invalid JSON in '+(path instanceof Array ? path.join('/') : path)+': '+e);
}
}
return cur;
}
// userInfo: { name: string, type: string, perms: static_perms[type], groups: { [string]: true } }
_check_compare(check, userInfo, checked)
{
let key = String(check.key);
if (key.substr(0, this.prefix.length) !== this.prefix)
{
return false;
}
key = key.substr(this.prefix.length);
if (key in userInfo.perms.keys)
{
return true;
}
for (const pfx in userInfo.perms.prefixes)
{
if (key.substr(0, pfx.length) == pfx)
{
return true;
}
}
if (userInfo.type == 'client')
{
// Image permissions
if (key.substr(0, 14) == '/config/inode/')
{
// Allowed to check that a key does not exist
if (check.target == 'VERSION' && check.version == 0)
{
checked['M'+key] = true;
return true;
}
else if (check.target == 'MOD')
{
const data = this._get(check.key);
if (!data || data.mod_revision != check.mod_revision)
{
// Break check to trigger CAS failure
check.mod_revision = '18446744073709551615'; // UINT64_MAX
return true;
}
const inode = this._decode(check.key, data.value);
if (inode && (inode.owner_group && userInfo.groups[inode.owner_group] ||
inode.owner === userInfo.name))
{
checked['M'+key] = true;
return true;
}
}
return false;
}
if (key.substr(0, 13) == '/index/image/')
{
// Allowed to check that a key does not exist
if (check.target == 'VERSION' && check.version == 0)
{
checked['M'+key] = true;
return true;
}
else if (check.target == 'MOD')
{
let data = this._get(check.key);
if (!data || data.mod_revision != check.mod_revision)
{
// Break check to trigger CAS failure
check.mod_revision = '18446744073709551615'; // UINT64_MAX
return true;
}
data = this._decode(check.key, data.value);
if (data)
{
const inode = this._get([ ...this.prefix_parts, 'config', 'inode', data.pool_id, data.id ], true);
if (inode && (inode.owner_group && userInfo.groups[inode.owner_group] ||
inode.owner === userInfo.name))
{
checked['M'+key] = true;
return true;
}
}
}
return false;
}
if (key.substr(0, 13) == '/index/maxid/')
{
const pool_id = key.substr(13);
const pool_cfg = this._get([ ...this.prefix_parts, 'config', 'pools' ], true);
if (!pool_cfg || !pool_cfg[pool_id] || !pool_cfg[pool_id].creator_group || !userInfo.groups[pool_cfg[pool_id].creator_group])
{
return false;
}
if (check.target == 'VERSION' && check.version == 0)
{
checked['I'+parseInt(key.substr(13))+'_0'] = true;
return true;
}
else if (check.target == 'MOD')
{
const data = this._get(check.key);
if (!data || data.mod_revision != check.mod_revision)
{
// Break check to trigger CAS failure
check.mod_revision = '18446744073709551615'; // UINT64_MAX
return true;
}
checked['I'+parseInt(key.substr(13))+'_'+data.value] = true;
return true;
}
return false;
}
}
return false;
}
_check_read(kv, userInfo)
{
let key = String(kv.key);
if (key.substr(0, this.prefix.length) !== this.prefix)
{
return false;
}
key = key.substr(this.prefix.length);
if (key in userInfo.perms.keys)
{
return true;
}
for (const pfx in userInfo.perms.prefixes)
{
if (key.substr(0, pfx.length) == pfx)
{
return true;
}
}
if (userInfo.type == 'client')
{
// Image permissions
if (key.substr(0, 14) == '/config/inode/')
{
const inode = this._decode(kv.key, kv.value);
if (inode && (inode.reader_group && userInfo.groups[inode.reader_group] ||
inode.owner_group && userInfo.groups[inode.owner_group] ||
inode.owner === userInfo.name))
{
return true;
}
return false;
}
if (key.substr(0, 13) == '/index/image/')
{
const data = this._decode(kv.key, kv.value);
const inode = this._get([ ...this.prefix_parts, 'config', 'inode', data.pool_id, data.id ], true);
if (inode && (inode.reader_group && userInfo.groups[inode.reader_group] ||
inode.owner_group && userInfo.groups[inode.owner_group] ||
inode.owner === userInfo.name))
{
return true;
}
return false;
}
}
return false;
}
_check_write(put, userInfo, checked)
{
let key = String(put.key);
if (key.substr(0, this.prefix.length) !== this.prefix)
{
return false;
}
key = key.substr(this.prefix.length);
if (userInfo.perms.keys[key])
{
return true;
}
for (const pfx in userInfo.perms.prefixes)
{
if (userInfo.perms.prefixes[pfx] && key.substr(0, pfx.length) == pfx)
{
return true;
}
}
if (checked && userInfo.type == 'client')
{
if (key.substr(0, 13) == '/index/maxid/' &&
checked['I'+parseInt(key.substr(13))+'_'+(put.value-1)])
{
// Allowed to increment maxid
return true;
}
if (checked['M'+key])
{
// Allowed to modify known images with CAS checks
return true;
}
}
return false;
}
_check_req(req, userInfo, checked)
{
let r;
if ((r = (req.request_range || req.requestRange)))
{
// All range queries are allowed, but responses are filtered - it's simpler
}
else if ((r = (req.request_put || req.requestPut)))
{
if (!this._check_write(r, userInfo, checked))
return false;
}
else if ((r = (req.request_delete_range || req.requestDeleteRange)))
{
if (!r.range_end || r.range_end === r.key)
{
if (!this._check_write({ key: r.key }, userInfo))
return false;
}
else
{
// All keys in range must satisfy prefix
r.range_end = String(r.range_end);
if (r.key.length != r.range_end.length ||
r.key[r.key.length-1] != '/' ||
r.range_end[r.range_end.length-1] != '0')
{
return false;
}
let key = r.key.substr(this.prefix.length);
let found = false;
for (const pfx in userInfo.perms.prefixes)
{
if (userInfo.perms.prefixes[pfx] && key.substr(0, pfx.length) == pfx)
{
found = true;
break;
}
}
if (!found)
return false;
}
}
return true;
}
_get_user(username)
{
if (!username)
{
return null;
}
let userInfo = this._get([ ...this.prefix_parts, 'config', 'user', username ], true);
if (!userInfo)
{
userInfo = { type: 'client' };
}
userInfo.perms = static_perms[userInfo.type] || static_perms['invalid'];
userInfo.name = username;
if (userInfo.groups instanceof Array)
{
userInfo.groups = userInfo.groups.reduce((a, c) => { a[c] = true; return a; }, {});
}
else
{
userInfo.groups = {};
}
return userInfo;
}
filter_api(username, api/*, data*/)
{
if (username === 'root')
{
return true;
}
const userInfo = this._get([ ...this.prefix_parts, 'config', 'user', username ], true);
return userInfo && api_perms[userInfo.type] && api_perms[userInfo.type][api];
}
filter_txn(username, txn)
{
if (username === 'root')
{
return true;
}
const userInfo = this._get_user(username);
if (!userInfo)
{
return null;
}
const checked = {};
if (txn.compare)
{
for (const check of txn.compare)
{
if (!this._check_compare(check, userInfo, checked))
return null;
}
}
// Special transactions:
// 1. create image: create config/inode and index/image, increment index/maxid/<pool> (with CAS)
// 2. create snapshot: same as create image but also rename previous to @snap
if (txn.success)
{
for (const req of txn.success)
{
if (!this._check_req(req, userInfo, checked))
return null;
}
}
if (txn.failure)
{
for (const req of txn.failure)
{
if (!this._check_req(req, userInfo, null))
return null;
}
}
return txn;
}
filter_txn_response(username, txn, res)
{
if (!res.responses || username === 'root')
{
return;
}
const userInfo = this._get_user(username);
if (!userInfo)
{
for (const resp of res.responses)
{
if (resp.response_range && resp.response_range.kvs)
{
resp.response_range.kvs = [];
}
}
return;
}
for (const resp of res.responses)
{
if (resp.response_range && resp.response_range.kvs)
{
resp.response_range.kvs = resp.response_range.kvs.filter(kv => this._check_read(kv, userInfo));
}
}
}
filter_watch_message(username, msg)
{
if (!msg.result || !msg.result.events || username === 'root')
{
return;
}
const userInfo = this._get_user(username);
if (!userInfo)
{
msg.result.events = [];
return;
}
msg.result.events = msg.result.events.filter(ev => this._check_read(ev.kv, userInfo));
}
}
module.exports = VitastorAuthFilter;
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "vitastor",
"version": "3.0.9",
"version": "3.0.10",
"description": "Low-level native bindings to Vitastor client library",
"main": "index.js",
"keywords": [
+1 -1
View File
@@ -50,7 +50,7 @@ from cinder.volume import configuration
from cinder.volume import driver
from cinder.volume import volume_utils
VITASTOR_VERSION = '3.0.9'
VITASTOR_VERSION = '3.0.10'
LOG = logging.getLogger(__name__)
+1 -1
View File
@@ -11,7 +11,7 @@ WORKDIR /root
RUN sed -i 's/enabled=0/enabled=1/' /etc/yum.repos.d/*.repo
RUN dnf -y install epel-release dnf-plugins-core
RUN dnf -y install https://vitastor.io/rpms/centos/10/vitastor-release-1.0-1.el10.noarch.rpm
RUN dnf -y install gcc-c++ gperftools-devel fio nodejs rpm-build jerasure-devel isa-l-devel gf-complete-devel rdma-core-devel cmake libnl3-devel c-ares-devel
RUN dnf -y install gcc-c++ gperftools-devel fio nodejs rpm-build jerasure-devel isa-l-devel gf-complete-devel rdma-core-devel cmake libnl3-devel
RUN dnf download --source fio
RUN rpm --nomd5 -i fio*.src.rpm
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --spec fio.spec
+2 -3
View File
@@ -1,11 +1,11 @@
Name: vitastor
Version: 3.0.9
Version: 3.0.10
Release: 1%{?dist}
Summary: Vitastor, a fast software-defined clustered block storage
License: Vitastor Network Public License 1.1
URL: https://vitastor.io/
Source0: vitastor-3.0.9.el10.tar.gz
Source0: vitastor-3.0.10.el10.tar.gz
BuildRequires: gperftools-devel
BuildRequires: gcc-c++
@@ -16,7 +16,6 @@ BuildRequires: gf-complete-devel
BuildRequires: rdma-core-devel
BuildRequires: cmake
BuildRequires: libnl3-devel
BuildRequires: c-ares-devel
Requires: vitastor-osd = %{version}-%{release}
Requires: vitastor-mon = %{version}-%{release}
Requires: vitastor-client = %{version}-%{release}
+1 -1
View File
@@ -15,7 +15,7 @@ RUN yum -y --enablerepo=extras install centos-release-scl epel-release yum-utils
RUN perl -i -pe 's!mirrorlist=!#mirrorlist=!s; s!#\s*baseurl=http://mirror.centos.org!baseurl=http://vault.centos.org!' /etc/yum.repos.d/CentOS-SCLo-scl*.repo
RUN yum -y install https://vitastor.io/rpms/centos/7/vitastor-release-1.0-1.el7.noarch.rpm
RUN yum -y install devtoolset-9-gcc-c++ devtoolset-9-libatomic-devel gcc make cmake gperftools-devel \
fio rh-nodejs12 jerasure-devel libisa-l-devel gf-complete-devel rdma-core-devel libnl3-devel c-ares-devel
fio rh-nodejs12 jerasure-devel libisa-l-devel gf-complete-devel rdma-core-devel libnl3-devel
RUN yumdownloader --disablerepo=centos-sclo-rh --source fio
RUN rpm --nomd5 -i fio*.src.rpm
RUN rm -f /etc/yum.repos.d/CentOS-Media.repo
+2 -3
View File
@@ -1,11 +1,11 @@
Name: vitastor
Version: 3.0.9
Version: 3.0.10
Release: 1%{?dist}
Summary: Vitastor, a fast software-defined clustered block storage
License: Vitastor Network Public License 1.1
URL: https://vitastor.io/
Source0: vitastor-3.0.9.el7.tar.gz
Source0: vitastor-3.0.10.el7.tar.gz
BuildRequires: gperftools-devel
BuildRequires: devtoolset-9-gcc-c++
@@ -17,7 +17,6 @@ BuildRequires: gf-complete-devel
BuildRequires: rdma-core-devel
BuildRequires: cmake3
BuildRequires: libnl3-devel
BuildRequires: c-ares-devel
Requires: vitastor-osd = %{version}-%{release}
Requires: vitastor-mon = %{version}-%{release}
Requires: vitastor-client = %{version}-%{release}
+1 -1
View File
@@ -13,7 +13,7 @@ RUN dnf -y install centos-release-advanced-virtualization epel-release dnf-plugi
RUN sed -i 's/^mirrorlist=/#mirrorlist=/; s!#baseurl=.*!baseurl=http://vault.centos.org/centos/8.4.2105/virt/$basearch/$avdir/!; s!^baseurl=.*Source/.*!baseurl=http://vault.centos.org/centos/8.4.2105/virt/Source/advanced-virtualization/!' /etc/yum.repos.d/CentOS-Advanced-Virtualization.repo
RUN yum -y install https://vitastor.io/rpms/centos/8/vitastor-release-1.0-1.el8.noarch.rpm
RUN dnf -y install gcc-toolset-9 gcc-toolset-9-gcc-c++ gperftools-devel \
fio nodejs rpm-build jerasure-devel libisa-l-devel gf-complete-devel libibverbs-devel libarchive cmake libnl3-devel c-ares-devel
fio nodejs rpm-build jerasure-devel libisa-l-devel gf-complete-devel libibverbs-devel libarchive cmake libnl3-devel
RUN dnf download --source fio
RUN rpm --nomd5 -i fio*.src.rpm
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --enablerepo=powertools --spec fio.spec
+2 -3
View File
@@ -1,11 +1,11 @@
Name: vitastor
Version: 3.0.9
Version: 3.0.10
Release: 1%{?dist}
Summary: Vitastor, a fast software-defined clustered block storage
License: Vitastor Network Public License 1.1
URL: https://vitastor.io/
Source0: vitastor-3.0.9.el8.tar.gz
Source0: vitastor-3.0.10.el8.tar.gz
BuildRequires: gperftools-devel
BuildRequires: gcc-toolset-9-gcc-c++
@@ -16,7 +16,6 @@ BuildRequires: gf-complete-devel
BuildRequires: rdma-core-devel
BuildRequires: cmake
BuildRequires: libnl3-devel
BuildRequires: c-ares-devel
Requires: vitastor-osd = %{version}-%{release}
Requires: vitastor-mon = %{version}-%{release}
Requires: vitastor-client = %{version}-%{release}
+1 -1
View File
@@ -10,7 +10,7 @@ WORKDIR /root
RUN sed -i 's/enabled=0/enabled=1/' /etc/yum.repos.d/*.repo
RUN dnf -y install epel-release dnf-plugins-core
RUN dnf -y install https://vitastor.io/rpms/centos/9/vitastor-release-1.0-1.el9.noarch.rpm
RUN dnf -y install gcc-c++ gperftools-devel fio nodejs rpm-build jerasure-devel libisa-l-devel gf-complete-devel rdma-core-devel libarchive cmake libnl3-devel c-ares-devel
RUN dnf -y install gcc-c++ gperftools-devel fio nodejs rpm-build jerasure-devel libisa-l-devel gf-complete-devel rdma-core-devel libarchive cmake libnl3-devel
RUN dnf download --source fio
RUN rpm --nomd5 -i fio*.src.rpm
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --spec fio.spec
+2 -3
View File
@@ -1,11 +1,11 @@
Name: vitastor
Version: 3.0.9
Version: 3.0.10
Release: 1%{?dist}
Summary: Vitastor, a fast software-defined clustered block storage
License: Vitastor Network Public License 1.1
URL: https://vitastor.io/
Source0: vitastor-3.0.9.el9.tar.gz
Source0: vitastor-3.0.10.el9.tar.gz
BuildRequires: gperftools-devel
BuildRequires: gcc-c++
@@ -16,7 +16,6 @@ BuildRequires: gf-complete-devel
BuildRequires: rdma-core-devel
BuildRequires: cmake
BuildRequires: libnl3-devel
BuildRequires: c-ares-devel
Requires: vitastor-osd = %{version}-%{release}
Requires: vitastor-mon = %{version}-%{release}
Requires: vitastor-client = %{version}-%{release}
+1 -9
View File
@@ -20,7 +20,7 @@ if("${CMAKE_INSTALL_PREFIX}" MATCHES "^/usr/local/?$")
endif()
set(ENABLE_COVERAGE false CACHE BOOL "Enable code coverage")
add_definitions(-DVITASTOR_VERSION="3.0.9")
add_definitions(-DVITASTOR_VERSION="3.0.10")
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
add_link_options(-fno-omit-frame-pointer)
if (${WITH_ASAN})
@@ -74,14 +74,6 @@ if (RDMACM_LIBRARIES)
add_definitions(-DWITH_RDMACM)
endif (RDMACM_LIBRARIES)
find_package(OpenSSL REQUIRED)
if (OPENSSL_FOUND)
add_definitions(-DWITH_OPENSSL)
endif (OPENSSL_FOUND)
pkg_check_modules(CARES REQUIRED libcares)
include_directories(${CARES_INCLUDE_DIRS})
if (${WITH_SYSTEM_LIBURING})
pkg_check_modules(LIBURING REQUIRED liburing>=2.10)
include_directories(${LIBURING_INCLUDE_DIRS})
+1 -5
View File
@@ -83,17 +83,13 @@ void blockstore_disk_t::parse_config(std::map<std::string, std::string> & config
{
data_csum_type = BLOCKSTORE_CSUM_CRC32C;
}
else if (config["data_csum_type"] == "xxh3_32")
{
data_csum_type = BLOCKSTORE_CSUM_XXH3_32;
}
else if (config["data_csum_type"] == "" || config["data_csum_type"] == "none")
{
data_csum_type = BLOCKSTORE_CSUM_NONE;
}
else
{
throw std::runtime_error("data_csum_type="+config["data_csum_type"]+" is unsupported, only \"crc32c\", \"xxh3_32\" and \"none\" are supported");
throw std::runtime_error("data_csum_type="+config["data_csum_type"]+" is unsupported, only \"crc32c\" and \"none\" are supported");
}
csum_block_size = parse_size(config["csum_block_size"]);
discard_on_start = config.find("discard_on_start") != config.end() &&
-1
View File
@@ -16,7 +16,6 @@
#define BLOCKSTORE_CSUM_NONE 0
// Lower byte of checksum type is its length
#define BLOCKSTORE_CSUM_CRC32C 0x104
#define BLOCKSTORE_CSUM_XXH3_32 0x204
#define MOCK_DATA_FD 1000
#define MOCK_META_FD 1001
+32 -115
View File
@@ -12,7 +12,6 @@
#include "blockstore_heap.h"
#include "../util/allocator.h"
#include "../util/crc32c.h"
#include "../util/xxh_x86dispatch.h"
#include "../util/malloc_or_die.h"
#define BS_HEAP_FREE_MVCC 1
@@ -65,19 +64,19 @@ uint32_t blockstore_heap_t::get_simple_entry_size()
uint32_t blockstore_heap_t::get_big_entry_size()
{
return sizeof(heap_big_write_t) + dsk->clean_entry_bitmap_size*2 +
(!dsk->csum_block_size ? 0 : dsk->data_block_size/dsk->csum_block_size * (dsk->data_csum_type & 0xFF));
(!dsk->data_csum_type ? 0 : dsk->data_block_size/dsk->csum_block_size * (dsk->data_csum_type & 0xFF));
}
uint32_t blockstore_heap_t::get_big_intent_entry_size()
{
return sizeof(heap_big_intent_t) + dsk->clean_entry_bitmap_size*2 +
(!dsk->csum_block_size ? 4 : dsk->data_block_size/dsk->csum_block_size * (dsk->data_csum_type & 0xFF));
(!dsk->data_csum_type ? 4 : dsk->data_block_size/dsk->csum_block_size * (dsk->data_csum_type & 0xFF));
}
uint32_t blockstore_heap_t::get_small_entry_size(uint32_t offset, uint32_t len)
{
return sizeof(heap_small_write_t) + dsk->clean_entry_bitmap_size +
(!dsk->csum_block_size ? 4 : (dsk->data_csum_type & 0xFF) *
(!dsk->data_csum_type ? 4 : (dsk->data_csum_type & 0xFF) *
((offset+len+dsk->csum_block_size-1)/dsk->csum_block_size - offset/dsk->csum_block_size));
}
@@ -92,7 +91,7 @@ uint32_t blockstore_heap_t::get_csum_size(heap_entry_t *wr)
uint32_t blockstore_heap_t::get_csum_size(uint32_t entry_type, uint32_t offset, uint32_t len)
{
if (!dsk->csum_block_size)
if (!dsk->data_csum_type)
{
return 0;
}
@@ -215,24 +214,15 @@ void heap_entry_t::set_big_location(blockstore_heap_t *heap, uint64_t location)
big().block_num = location / heap->dsk->data_block_size;
}
uint32_t heap_entry_t::calc_checksum(blockstore_disk_t *dsk)
uint32_t heap_entry_t::calc_crc32c()
{
auto old_checksum = checksum;
checksum = 0;
uint32_t res = 0;
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
res = (uint32_t)XXH3_64bits(this, size);
else
res = ::crc32c(0, (uint8_t*)this, size);
checksum = old_checksum;
auto old_crc32c = crc32c;
crc32c = 0;
uint32_t res = ::crc32c(0, (uint8_t*)this, size);
crc32c = old_crc32c;
return res;
}
uint32_t heap_entry_t::calc_checksum(blockstore_heap_t *heap)
{
return calc_checksum(heap->dsk);
}
uint64_t blockstore_heap_t::get_pg_id(inode_t inode, uint64_t stripe)
{
uint64_t pg_num = 0;
@@ -344,19 +334,12 @@ corrupted_block:
block_num, block_offset, wr->size, sizeof(heap_entry_t));
goto corrupted_block;
}
if (wr->is_garbage())
{
// Garbage collection is only performed when writing new entries into the block
// because it needs a fake LSN and modified blocks require consecutive modified LSNs
// That's why garbage entries may persist on disk
if (log_level > 5)
{
fprintf(stderr, "Notice: skipping garbage entry %jx:%jx v%ju l%ju in metadata block %u at %u\n",
wr->inode, wr->stripe, wr->version, wr->lsn, block_num, block_offset);
}
block_offset += wr->size;
continue;
}
// Garbage collection is only performed when writing new entries into the block
// because it needs a fake LSN and modified blocks require consecutive modified LSNs
// At the same time, further modifications _after_ putting new entries into the block,
// but _before_ writing it, may mark some entries in it as garbage. That's why garbage
// entries may still be present on disk.
wr->entry_type &= ~BS_HEAP_GARBAGE;
if ((wr->entry_type & BS_HEAP_TYPE) < BS_HEAP_BIG_WRITE ||
(wr->entry_type & BS_HEAP_TYPE) > BS_HEAP_ROLLBACK ||
(wr->entry_type & ~(BS_HEAP_TYPE|BS_HEAP_STABLE)) ||
@@ -397,12 +380,12 @@ corrupted_object:
goto corrupted_object;
}
// Verify crc
uint32_t expected_checksum = wr->calc_checksum(this);
if (wr->checksum != expected_checksum)
uint32_t expected_crc32c = wr->calc_crc32c();
if (wr->crc32c != expected_crc32c)
{
fprintf(stderr, "Error: entry %jx:%jx v%ju l%ju in metadata block %u at %u is corrupt (checksum mismatch: expected %08x, got %08x). ",
fprintf(stderr, "Error: entry %jx:%jx v%ju l%ju in metadata block %u at %u is corrupt (crc32c mismatch: expected %08x, got %08x). ",
wr->inode, wr->stripe, wr->version, wr->lsn,
block_num, block_offset, expected_checksum, wr->checksum);
block_num, block_offset, expected_crc32c, wr->crc32c);
goto corrupted_object;
}
// Verify offset & len
@@ -951,11 +934,7 @@ bool blockstore_heap_t::calc_checksums(heap_entry_t *wr, uint8_t *data, bool set
len = wr->big_intent().len;
else
assert(0);
uint32_t real_csum = 0;
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
real_csum = (uint32_t)XXH3_64bits(data, len);
else
real_csum = crc32c(0, data, len);
uint32_t real_csum = crc32c(0, data, len);
if (set)
{
*wr_csum = real_csum;
@@ -1005,26 +984,11 @@ static uint32_t crc32c_iter(uint32_t prev_crc, const std::function<uint8_t*(uint
return prev_crc;
}
static void xxh3_iter(XXH3_state_t* xxh3_state, const std::function<uint8_t*(uint32_t start, uint32_t & len)> & next, uint32_t pos, uint32_t size)
{
uint32_t cur_len = 0;
while (size > 0)
{
uint8_t *data = next(pos, cur_len);
assert(data);
cur_len = (cur_len < size ? cur_len : size);
XXH3_64bits_update(xxh3_state, data, cur_len);
pos += cur_len;
size -= cur_len;
}
}
bool blockstore_heap_t::calc_block_checksums(uint32_t *block_csums, uint8_t *bitmap,
uint32_t start, uint32_t end, std::function<uint8_t*(uint32_t start, uint32_t & len)> next,
bool set, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb)
{
bool res = true;
XXH3_state_t* xxh3_state = NULL;
uint32_t pos = start;
uint32_t block_end = (start/dsk->csum_block_size + 1)*dsk->csum_block_size;
uint32_t block_crc = 0;
@@ -1041,89 +1005,42 @@ bool blockstore_heap_t::calc_block_checksums(uint32_t *block_csums, uint8_t *bit
pos += dsk->bitmap_granularity;
// zero padding at the beginning or at the end of the block is not counted
if (pos > prev && prev > 0 && pos < block_end)
{
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
{
if (!xxh3_state)
{
xxh3_state = XXH3_createState();
XXH3_64bits_reset(xxh3_state);
}
uint32_t zeropad = pos-prev;
while (zeropad > 0)
{
uint32_t zerolen = zeropad > 4096 ? 4096 : zeropad;
XXH3_64bits_update(xxh3_state, zero_page, zerolen);
zeropad -= zerolen;
}
}
else
block_crc = crc32c_pad(block_crc, NULL, 0, pos-prev, 0);
}
block_crc = crc32c_pad(block_crc, NULL, 0, pos-prev, 0);
prev = pos;
while (pos < end && pos < block_end && (bitmap[pos/dsk->bitmap_granularity/8] & (1 << ((pos/dsk->bitmap_granularity) % 8))))
pos += dsk->bitmap_granularity;
if (pos > prev)
{
isset = true;
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
{
if (!xxh3_state)
{
xxh3_state = XXH3_createState();
XXH3_64bits_reset(xxh3_state);
}
xxh3_iter(xxh3_state, next, prev, pos-prev);
}
else
block_crc = crc32c_iter(block_crc, next, prev, pos-prev);
block_crc = crc32c_iter(block_crc, next, prev, pos-prev);
}
prev = pos;
}
}
else
{
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
{
if (!xxh3_state)
{
xxh3_state = XXH3_createState();
XXH3_64bits_reset(xxh3_state);
}
xxh3_iter(xxh3_state, next, pos, (end > block_end ? block_end : end)-pos);
}
else
block_crc = crc32c_iter(block_crc, next, pos, (end > block_end ? block_end : end)-pos);
block_crc = crc32c_iter(block_crc, next, pos, (end > block_end ? block_end : end)-pos);
pos = (end > block_end ? block_end : end);
isset = true;
}
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32 && xxh3_state)
{
block_crc = (uint32_t)XXH3_64bits_digest(xxh3_state);
XXH3_64bits_reset(xxh3_state);
}
if (set)
{
*block_csums = block_crc;
}
else if (isset && block_crc != *block_csums)
{
res = false;
if (bad_block_cb)
{
bad_block_cb(blk_start, *block_csums, block_crc);
res = false;
}
else
break;
return false;
}
block_end += dsk->csum_block_size;
block_crc = 0;
block_csums++;
}
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32 && xxh3_state)
{
block_crc = (uint32_t)XXH3_64bits_digest(xxh3_state);
XXH3_freeState(xxh3_state);
xxh3_state = NULL;
}
return res;
}
@@ -1482,7 +1399,7 @@ int blockstore_heap_t::add_entry(uint32_t wr_size, uint32_t *modified_block,
insert_list_item(li);
li->block_num = block_num;
new_wr->size = wr_size;
new_wr->checksum = new_wr->calc_checksum(this);
new_wr->crc32c = new_wr->calc_crc32c();
return 0;
}
@@ -1542,7 +1459,7 @@ int blockstore_heap_t::add_big_write(object_id oid, heap_entry_t *old_head, bool
memset(wr->get_ext_bitmap(this), 0, dsk->clean_entry_bitmap_size);
memset(wr->get_int_bitmap(this), 0, dsk->clean_entry_bitmap_size);
bitmap_set(wr->get_int_bitmap(this), offset, len, dsk->bitmap_granularity);
if (dsk->csum_block_size)
if (dsk->data_csum_type)
{
memset(wr->get_checksums(this), 0, get_csum_size(wr));
calc_checksums(wr, (uint8_t*)data, true, offset, len);
@@ -1571,7 +1488,7 @@ int blockstore_heap_t::add_redirect_intent(object_id oid, heap_entry_t **obj_ptr
memset(wr->get_ext_bitmap(this), 0, dsk->clean_entry_bitmap_size);
memset(wr->get_int_bitmap(this), 0, dsk->clean_entry_bitmap_size);
bitmap_set(wr->get_int_bitmap(this), offset, len, dsk->bitmap_granularity);
if (dsk->csum_block_size)
if (dsk->data_csum_type)
memset(wr->get_checksums(this), 0, get_csum_size(wr));
calc_checksums(wr, (uint8_t*)data, true);
*obj_ptr = wr;
@@ -1609,7 +1526,7 @@ int blockstore_heap_t::add_big_intent(object_id oid, heap_entry_t **obj_ptr, uin
memcpy(wr->get_ext_bitmap(this), obj->get_ext_bitmap(this), dsk->clean_entry_bitmap_size);
memcpy(wr->get_int_bitmap(this), obj->get_int_bitmap(this), dsk->clean_entry_bitmap_size);
bitmap_set(wr->get_int_bitmap(this), offset, len, dsk->bitmap_granularity);
if (dsk->csum_block_size)
if (dsk->data_csum_type)
{
if (checksums)
memcpy(wr->get_checksums(this), checksums, get_csum_size(wr));
@@ -1656,7 +1573,7 @@ int blockstore_heap_t::add_compact(heap_entry_t *obj, uint64_t compact_version,
new_wr->set_big_location(this, compact_location);
memcpy(new_wr->get_int_bitmap(this), new_int_bitmap, dsk->clean_entry_bitmap_size);
memcpy(new_wr->get_ext_bitmap(this), new_ext_bitmap, dsk->clean_entry_bitmap_size);
if (dsk->csum_block_size && new_csums)
if (dsk->data_csum_type && new_csums)
memcpy(new_wr->get_checksums(this), new_csums, dsk->data_block_size/dsk->csum_block_size*(dsk->data_csum_type & 0xFF));
});
}
+4 -5
View File
@@ -43,7 +43,7 @@ struct __attribute__((__packed__)) heap_entry_t
{
uint16_t size;
uint16_t entry_type;
uint32_t checksum;
uint32_t crc32c;
uint64_t lsn;
uint64_t inode;
uint64_t stripe;
@@ -69,8 +69,7 @@ struct __attribute__((__packed__)) heap_entry_t
uint32_t *get_checksum(blockstore_heap_t *heap);
uint64_t big_location(blockstore_heap_t *heap);
void set_big_location(blockstore_heap_t *heap, uint64_t location);
uint32_t calc_checksum(blockstore_heap_t *heap);
uint32_t calc_checksum(blockstore_disk_t *dsk);
uint32_t calc_crc32c();
};
struct __attribute__((__packed__)) heap_small_write_t
@@ -81,7 +80,7 @@ struct __attribute__((__packed__)) heap_small_write_t
uint32_t offset;
uint32_t len;
// Also includes 1 bitmap and 1 checksum after the bitmap if block checksums are disabled
// Also includes 1 bitmap and 1 crc32c after the bitmap if checksums are disabled
};
struct __attribute__((__packed__)) heap_big_write_t
@@ -99,7 +98,7 @@ struct __attribute__((__packed__)) heap_big_intent_t
uint32_t offset;
uint32_t len;
// Also includes 2 bitmaps and 1 checksums if block checksums are disabled
// Also includes 2 bitmaps and 1 crc32c if checksums are disabled
};
struct __attribute__((__packed__)) heap_list_item_t
+1 -1
View File
@@ -311,7 +311,7 @@ resume_8:
uint32_t block_num = recheck_mod[i];
uint64_t block_offset = bs->dsk.meta_offset + (uint64_t)(block_num+1) * bs->dsk.meta_block_size;
data = ((ring_data_t*)sqe->user_data);
uint8_t *buf = (uint8_t*)malloc_or_die(bs->dsk.meta_block_size);
uint8_t *buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, bs->dsk.meta_block_size);
bs->heap->get_meta_block(block_num, buf);
data->iov = { buf, bs->dsk.meta_block_size };
data->callback = [this, buf, block_offset](ring_data_t *data)
+5 -8
View File
@@ -12,11 +12,11 @@ if (RDMACM_LIBRARIES)
set(MSGR_RDMACM "msgr_rdmacm.cpp")
endif (RDMACM_LIBRARIES)
add_library(vitastor_common STATIC
../util/epoll_manager.cpp etcd_state_client.cpp messenger.cpp ../util/addr_util.cpp ../util/xxh_x86dispatch.c
msgr_encrypt.cpp msgr_stop.cpp msgr_op.cpp msgr_send.cpp msgr_receive.cpp ../util/ringloop.cpp ../../json11/json11.cpp
../util/epoll_manager.cpp etcd_state_client.cpp messenger.cpp ../util/addr_util.cpp
msgr_stop.cpp msgr_op.cpp msgr_send.cpp msgr_receive.cpp ../util/ringloop.cpp ../../json11/json11.cpp
http_client.cpp osd_ops.cpp pg_states.cpp ../util/timerfd_manager.cpp ../util/str_util.cpp ../util/json_util.cpp ${MSGR_RDMA} ${MSGR_RDMACM}
)
target_link_libraries(vitastor_common pthread ${OPENSSL_LIBRARIES} ${CARES_LIBRARIES})
target_link_libraries(vitastor_common pthread)
target_compile_options(vitastor_common PUBLIC -fPIC)
# libvitastor_client.so
@@ -24,7 +24,6 @@ add_library(vitastor_client SHARED
cluster_client.cpp
cluster_client_list.cpp
cluster_client_wb.cpp
cluster_client_icache.cpp
vitastor_c.cpp
)
set_target_properties(vitastor_client PROPERTIES PUBLIC_HEADER "client/vitastor_c.h")
@@ -34,7 +33,6 @@ target_link_libraries(vitastor_client
${LIBURING_LIBRARIES}
${IBVERBS_LIBRARIES}
${RDMACM_LIBRARIES}
${OPENSSL_LIBRARIES}
)
set_target_properties(vitastor_client PROPERTIES VERSION ${VITASTOR_VERSION} SOVERSION 0)
configure_file(vitastor.pc.in vitastor.pc @ONLY)
@@ -100,10 +98,9 @@ endif (${WITH_QEMU})
add_executable(test_cluster_client
EXCLUDE_FROM_ALL
../test/test_cluster_client.cpp
pg_states.cpp osd_ops.cpp cluster_client.cpp cluster_client_list.cpp cluster_client_wb.cpp cluster_client_icache.cpp msgr_op.cpp ../test/mock/messenger.cpp msgr_stop.cpp msgr_encrypt.cpp
etcd_state_client.cpp ../util/timerfd_manager.cpp ../util/addr_util.cpp ../util/str_util.cpp ../util/json_util.cpp ../util/xxh_x86dispatch.c ../../json11/json11.cpp
pg_states.cpp osd_ops.cpp cluster_client.cpp cluster_client_list.cpp cluster_client_wb.cpp msgr_op.cpp ../test/mock/messenger.cpp msgr_stop.cpp
etcd_state_client.cpp ../util/timerfd_manager.cpp ../util/addr_util.cpp ../util/str_util.cpp ../util/json_util.cpp ../../json11/json11.cpp
)
target_link_libraries(test_cluster_client ${OPENSSL_LIBRARIES})
target_compile_definitions(test_cluster_client PUBLIC -D__MOCK__)
target_include_directories(test_cluster_client BEFORE PUBLIC ${CMAKE_SOURCE_DIR}/src/test/mock)
add_dependencies(build_tests test_cluster_client)
+62 -107
View File
@@ -62,7 +62,6 @@ cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd
st_cli.on_change_node_placement_hook = [this]() { on_change_node_placement_hook(); };
st_cli.on_load_pgs_hook = [this](bool success) { on_load_pgs_hook(success); };
st_cli.on_reload_hook = [this]() { st_cli.load_global_config(); };
st_cli.on_inode_change_hook = [this](uint64_t inode, bool removed) { on_change_inode_hook(inode, removed); };
st_cli.parse_config(config);
st_cli.infinite_start = false;
@@ -71,11 +70,13 @@ cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd
st_cli.infinite_start = config["client_infinite_start"].bool_value();
}
st_cli.load_global_config();
scrap_buffer_size = SCRAP_BUFFER_SIZE;
scrap_buffer = malloc_or_die(scrap_buffer_size);
}
cluster_client_t::~cluster_client_t()
{
vault_destroy();
if (retry_timeout_id >= 0)
{
tfd->clear_timer(retry_timeout_id);
@@ -93,6 +94,7 @@ cluster_client_t::~cluster_client_t()
{
ringloop->unregister_consumer(&consumer);
}
free(scrap_buffer);
delete wb;
wb = NULL;
}
@@ -479,8 +481,6 @@ void cluster_client_t::on_load_config_hook(json11::Json::object & etcd_global_co
self_tree_metrics.clear();
client_hostname = new_hostname;
}
// vault
vault_parse_config();
msgr.parse_config(config);
st_cli.parse_config(config);
st_cli.load_pgs();
@@ -590,7 +590,7 @@ void cluster_client_t::on_change_pool_config_hook()
{
if (log_level > 2 && pg_counts[pool_item.first])
{
printf("Pool %u (%s) PG count changed from %lu to %lu\n", pool_item.first, pool_item.second.name.c_str(),
fprintf(stderr, "Pool %u (%s) PG count changed from %lu to %lu\n", pool_item.first, pool_item.second.name.c_str(),
pg_counts[pool_item.first], pool_item.second.real_pg_count);
}
// At this point, all pool operations should have been suspended
@@ -607,9 +607,6 @@ void cluster_client_t::on_change_pool_config_hook()
pg_counts[pool_item.first] = pool_item.second.real_pg_count;
}
}
inode_cache.clear();
inode_cache_children.clear();
vault_keys.clear();
continue_ops();
}
@@ -676,10 +673,6 @@ bool cluster_client_t::flush()
{
if (!ringloop)
{
if (vault_loading)
{
return false;
}
if (wb->writeback_queue.size())
{
wb->start_writebacks(this, 0);
@@ -702,7 +695,7 @@ bool cluster_client_t::flush()
sync_done = true;
};
execute(sync);
while (!sync_done || vault_loading)
while (!sync_done)
{
ringloop->loop();
if (!sync_done)
@@ -965,40 +958,10 @@ bool cluster_client_t::check_rw(cluster_op_t *op)
{
op->flags |= OP_IMMEDIATE_COMMIT;
}
bool searched = false;
std::shared_ptr<inode_cache_t> icache;
if (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_WRITE)
{
if (!searched)
{
icache = inode_cache_get(op->inode);
searched = true;
}
if (icache && icache->has_parent_loop && op->opcode == OSD_OP_READ)
{
op->retval = -EINVAL;
auto cb = std::move(op->callback);
cb(op);
return false;
}
if (icache && icache->op_enc)
{
// Use shared_ptr aliasing to attach op_enc to the inode cache entry
op->enc = std::shared_ptr<osd_op_enc_t>(icache, icache->op_enc);
}
else
op->enc.reset();
}
else
op->enc.reset();
if ((op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) && !(op->flags & OSD_OP_IGNORE_READONLY))
{
if (!searched)
{
icache = inode_cache_get(op->inode);
searched = true;
}
if (icache && icache->readonly)
auto ino_it = st_cli.inode_config.find(op->inode);
if (ino_it != st_cli.inode_config.end() && ino_it->second.readonly)
{
op->retval = -EROFS;
auto cb = std::move(op->callback);
@@ -1009,39 +972,33 @@ bool cluster_client_t::check_rw(cluster_op_t *op)
op->deoptimise_snapshot = false;
if (enable_writeback && (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP))
{
if (!searched)
auto ino_it = st_cli.inode_config.find(op->inode);
if (ino_it != st_cli.inode_config.end())
{
icache = inode_cache_get(op->inode);
searched = true;
}
if (icache)
{
for (auto & parent: icache->chain)
int chain_size = 0;
while (ino_it != st_cli.inode_config.end() && ino_it->second.parent_id)
{
if (INODE_POOL(parent) == INODE_POOL(op->inode) && wb->has_inode(parent))
// Check for loops - FIXME check it in etcd_state_client
if (ino_it->second.parent_id == op->inode ||
chain_size > st_cli.inode_config.size())
{
op->retval = -EINVAL;
auto cb = std::move(op->callback);
cb(op);
return false;
}
if (INODE_POOL(ino_it->second.parent_id) == INODE_POOL(ino_it->first) &&
wb->has_inode(ino_it->second.parent_id))
{
// Deoptimise reads - we have dirty data for one of the parent layer(s).
op->deoptimise_snapshot = true;
break;
}
chain_size++;
ino_it = st_cli.inode_config.find(ino_it->second.parent_id);
}
}
}
if (icache && icache->err_code)
{
if (icache->err_code == EPERM)
{
op->retval = -EPERM;
auto cb = std::move(op->callback);
cb(op);
return false;
}
else if (icache->err_code == EAGAIN)
{
key_wait_ops.push_back(op);
return false;
}
}
return true;
}
@@ -1164,33 +1121,31 @@ resume_2:
// because if some operations were invalid for the new PG count we'd get errors
if (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_CHAIN_BITMAP)
{
uint64_t next_inode = 0;
auto icache = inode_cache_get(op->cur_inode);
if (icache)
// Check parent inode
auto ino_it = st_cli.inode_config.find(op->cur_inode);
// Skip parents from the same pool
int skipped = 0;
while (!op->deoptimise_snapshot &&
ino_it != st_cli.inode_config.end() && ino_it->second.parent_id &&
INODE_POOL(ino_it->second.parent_id) == INODE_POOL(op->cur_inode))
{
if (icache->has_parent_loop)
// Check for loops - FIXME check it in etcd_state_client
if (ino_it->second.parent_id == op->inode ||
skipped > st_cli.inode_config.size())
{
op->retval = -EINVAL;
erase_op(op);
return 1;
}
if (op->deoptimise_snapshot)
{
if (icache->chain.size() > 1)
next_inode = icache->chain[1];
}
else
{
if (icache->other_pool_parent_id)
next_inode = icache->other_pool_parent_id;
}
skipped++;
ino_it = st_cli.inode_config.find(ino_it->second.parent_id);
}
if (next_inode)
if (ino_it != st_cli.inode_config.end() &&
ino_it->second.parent_id &&
ino_it->second.parent_id != op->inode)
{
// Continue reading from the parent inode
icache = inode_cache_get(next_inode);
op->cur_inode = next_inode;
op->enc = (icache && icache->op_enc ? std::shared_ptr<osd_op_enc_t>(icache, icache->op_enc) : nullptr);
op->cur_inode = ino_it->second.parent_id;
op->parts.clear();
op->done_count = 0;
goto resume_0;
@@ -1241,7 +1196,7 @@ resume_2:
return 0;
}
static void add_iov(int size, int skip, cluster_op_t *op, int &iov_idx, size_t &iov_pos, osd_op_buf_list_t &iov)
static void add_iov(int size, bool skip, cluster_op_t *op, int &iov_idx, size_t &iov_pos, osd_op_buf_list_t &iov, void *scrap, int scrap_len)
{
int left = size;
while (left > 0 && iov_idx < op->iov.count)
@@ -1249,7 +1204,7 @@ static void add_iov(int size, int skip, cluster_op_t *op, int &iov_idx, size_t &
int cur_left = op->iov.buf[iov_idx].iov_len - iov_pos;
if (cur_left < left)
{
if (skip == 0)
if (!skip)
{
iov.push_back((uint8_t*)op->iov.buf[iov_idx].iov_base + iov_pos, cur_left);
}
@@ -1259,7 +1214,7 @@ static void add_iov(int size, int skip, cluster_op_t *op, int &iov_idx, size_t &
}
else
{
if (skip == 0)
if (!skip)
{
iov.push_back((uint8_t*)op->iov.buf[iov_idx].iov_base + iov_pos, left);
}
@@ -1268,10 +1223,16 @@ static void add_iov(int size, int skip, cluster_op_t *op, int &iov_idx, size_t &
}
}
assert(left == 0);
if (skip == 1)
if (skip && scrap_len > 0)
{
// data read into a NULL buffer will be discarded by messenger
iov.push_back(NULL, size);
// All skipped ranges are read into the same useless buffer
left = size;
while (left > 0)
{
int cur_left = scrap_len < left ? scrap_len : left;
iov.push_back(scrap, cur_left);
left -= cur_left;
}
}
}
@@ -1291,11 +1252,7 @@ void cluster_client_t::slice_rw(cluster_op_t *op)
// Allocate memory for the bitmap
unsigned object_bitmap_size = ((op->len / pool_cfg.bitmap_granularity + 7) / 8);
object_bitmap_size = (object_bitmap_size < 8 ? 8 : object_bitmap_size);
unsigned bitmap_mem = object_bitmap_size +
op->parts.size() * pg_data_size *
(pool_cfg.data_block_size / pool_cfg.bitmap_granularity / 8
// read chain info - 1 byte per block
+ (op->enc ? op->len/pool_cfg.bitmap_granularity : 0));
unsigned bitmap_mem = object_bitmap_size + (pool_cfg.data_block_size / pool_cfg.bitmap_granularity / 8 * pg_data_size) * op->parts.size();
if (!op->bitmap_buf || op->bitmap_buf_size < bitmap_mem)
{
op->bitmap_buf = realloc_or_die(op->bitmap_buf, bitmap_mem);
@@ -1337,10 +1294,10 @@ void cluster_client_t::slice_rw(cluster_op_t *op)
{
begin = cur;
// Just advance iov_idx & iov_pos
add_iov(cur-prev, 2, op, iov_idx, iov_pos, op->parts[i].iov);
add_iov(cur-prev, true, op, iov_idx, iov_pos, op->parts[i].iov, NULL, 0);
}
else
add_iov(cur-prev, skip_prev ? 1 : 0, op, iov_idx, iov_pos, op->parts[i].iov);
add_iov(cur-prev, skip_prev, op, iov_idx, iov_pos, op->parts[i].iov, scrap_buffer, scrap_buffer_size);
}
skip_prev = skip;
prev = cur;
@@ -1351,11 +1308,11 @@ void cluster_client_t::slice_rw(cluster_op_t *op)
if (skip_prev)
{
// Just advance iov_idx & iov_pos
add_iov(end-prev, 2, op, iov_idx, iov_pos, op->parts[i].iov);
add_iov(end-prev, true, op, iov_idx, iov_pos, op->parts[i].iov, NULL, 0);
end = prev;
}
else
add_iov(cur-prev, skip_prev ? 1 : 0, op, iov_idx, iov_pos, op->parts[i].iov);
add_iov(cur-prev, skip_prev, op, iov_idx, iov_pos, op->parts[i].iov, scrap_buffer, scrap_buffer_size);
if (end == begin)
{
op->done_count++;
@@ -1364,7 +1321,7 @@ void cluster_client_t::slice_rw(cluster_op_t *op)
}
else if (op->opcode != OSD_OP_READ_BITMAP && op->opcode != OSD_OP_READ_CHAIN_BITMAP && op->opcode != OSD_OP_DELETE)
{
add_iov(end-begin, 0, op, iov_idx, iov_pos, op->parts[i].iov);
add_iov(end-begin, false, op, iov_idx, iov_pos, op->parts[i].iov, NULL, 0);
}
op->parts[i].parent = op;
op->parts[i].offset = begin;
@@ -1450,9 +1407,9 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
osd_client_t *cl = peer_it->second;
part->flags |= PART_SENT|PART_VALID;
op->inflight_count++;
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
uint64_t pg_bitmap_size = pg_data_size * (pool_cfg.data_block_size / pool_cfg.bitmap_granularity / 8
+ (op->opcode == OSD_OP_READ && op->enc ? pool_cfg.data_block_size/pool_cfg.bitmap_granularity : 0));
uint64_t pg_bitmap_size = (pool_cfg.data_block_size / pool_cfg.bitmap_granularity / 8) * (
pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks
);
uint64_t meta_rev = 0;
if (op->opcode != OSD_OP_READ_BITMAP && op->opcode != OSD_OP_DELETE && !op->deoptimise_snapshot)
{
@@ -1471,7 +1428,6 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
.inode = op->cur_inode,
.offset = part->offset,
.len = part->len,
.flags = op->opcode == OSD_OP_READ && op->enc && !op->deoptimise_snapshot ? OSD_OP_RETURN_CHAIN : 0,
.meta_revision = meta_rev,
.version = op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE ? op->version : 0,
} },
@@ -1479,7 +1435,6 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
? (uint8_t*)op->part_bitmaps + pg_bitmap_size*i : NULL),
.bitmap_len = (unsigned)(op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP
? pg_bitmap_size : 0),
.enc = op->enc,
.callback = cb ? cb : [this, part](osd_op_t *op_part)
{
handle_op_part(part);
+4 -58
View File
@@ -5,7 +5,6 @@
#include "messenger.h"
#include "etcd_state_client.h"
#include "../util/robin_hood.h"
#define DEFAULT_CLIENT_MAX_DIRTY_BYTES 32*1024*1024
#define DEFAULT_CLIENT_MAX_DIRTY_OPS 1024
@@ -72,7 +71,6 @@ protected:
cluster_op_t *prev = NULL, *next = NULL;
int prev_wait = 0;
uint64_t flush_id = 0;
std::shared_ptr<osd_op_enc_t> enc;
friend class cluster_client_t;
friend class writeback_cache_t;
};
@@ -82,25 +80,6 @@ struct inode_list_osd_t;
struct inode_list_pg_t;
class writeback_cache_t;
struct inode_cache_t
{
std::vector<inode_t> chain;
uint8_t *key_data = NULL;
osd_op_enc_t *op_enc = NULL;
bool readonly = false;
bool has_parent_loop = false;
inode_t other_pool_parent_id = 0;
int err_code = 0;
~inode_cache_t();
};
struct vault_load_key_t
{
int key_state = 0;
std::string key;
};
// FIXME: Split into public and private interfaces
class __attribute__((visibility("default"))) cluster_client_t
{
@@ -110,8 +89,8 @@ public:
timerfd_manager_t *tfd = NULL;
ring_loop_t *ringloop = NULL;
// config:
std::map<pool_id_t, uint64_t> pg_counts;
std::map<pool_pg_num_t, osd_num_t> pg_primary;
// client_max_dirty_* is actually "max unsynced", for the case when immediate_commit is off
uint64_t client_max_dirty_bytes = 0;
uint64_t client_max_dirty_ops = 0;
@@ -123,23 +102,12 @@ public:
uint64_t client_max_writeback_iodepth = 0;
std::string conf_hostname;
std::string vault_url;
std::string vault_client_cert;
std::string vault_client_key;
std::string vault_ca;
std::string vault_secret_api_path;
uint64_t vault_timeout_ms = 0;
uint64_t vault_error_timeout_sec = 0;
uint64_t vault_refresh_leeway_sec = 0;
int log_level = 0;
int client_retry_interval = 50; // ms
int client_eio_retry_interval = 1000; // ms
bool client_retry_enospc = true;
int client_wait_up_timeout = 16; // sec (for listings)
// state:
std::string client_hostname;
std::map<std::string, int> self_tree_metrics;
std::map<osd_num_t, int> osd_tree_metrics;
@@ -147,28 +115,15 @@ public:
int retry_timeout_id = -1;
int retry_timeout_duration = 0;
std::vector<cluster_op_t*> offline_ops;
std::vector<cluster_op_t*> key_wait_ops;
cluster_op_t *op_queue_head = NULL, *op_queue_tail = NULL;
writeback_cache_t *wb = NULL;
std::set<osd_num_t> dirty_osds;
uint64_t dirty_bytes = 0, dirty_ops = 0;
// inodes require some extra state for read/write, it's stored here.
// moreover, robin_hood access is slightly faster than std::map :)
robin_hood::unordered_flat_map<inode_t, std::shared_ptr<inode_cache_t>> inode_cache;
std::set<std::pair<inode_t, inode_t>> inode_cache_children;
http_context_t *vault_http_ctx = NULL;
http_co_t *vault_http_cli = NULL;
bool vault_loading = false;
std::string vault_token;
bool vault_auth_error = false;
timespec vault_token_expire = {};
std::vector<std::string> vault_key_load_queue;
std::map<std::string, vault_load_key_t> vault_keys;
void *scrap_buffer = NULL;
unsigned scrap_buffer_size = 0;
bool pgs_loaded = false;
std::map<pool_id_t, uint64_t> pg_counts;
ring_consumer_t consumer;
std::vector<std::function<void(void)>> on_ready_hooks;
int list_retry_timeout_id = -1;
@@ -208,13 +163,6 @@ protected:
#endif
void continue_ops(int time_passed = 0);
std::shared_ptr<inode_cache_t> inode_cache_get(inode_t ino);
void vault_parse_config();
bool vault_check_token();
void vault_load_keys();
void vault_destroy();
void vault_parse_secret(const std::string & key_id, const std::string & err, json11::Json data);
protected:
bool affects_osd(uint64_t inode, uint64_t offset, uint64_t len, osd_num_t osd);
bool affects_pg(uint64_t inode, uint64_t offset, uint64_t len, pool_id_t pool_id, pg_num_t pg_num);
@@ -225,7 +173,6 @@ protected:
void on_change_pg_state_hook(pool_id_t pool_id, pg_num_t pg_num, osd_num_t prev_primary);
void on_change_osd_state_hook(uint64_t peer_osd);
void on_change_node_placement_hook();
void on_change_inode_hook(uint64_t inode, bool removed);
void execute_internal(cluster_op_t *op);
void execute_cas(cluster_op_t *op);
@@ -242,7 +189,6 @@ protected:
void erase_op(cluster_op_t *op);
void calc_wait(cluster_op_t *op);
void inc_wait(uint64_t opcode, uint64_t flags, cluster_op_t *next, int inc);
void continue_lists();
bool continue_listing(inode_list_t *lst);
bool restart_listing(inode_list_t* lst);
-367
View File
@@ -1,367 +0,0 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
#include <stdexcept>
#include <assert.h>
#include "cluster_client_impl.h"
#include "http_client.h"
#include "str_util.h"
#define VAULT_KEY_NOT_LOADED 0
#define VAULT_KEY_LOADING 1
#define VAULT_KEY_LOADED 2
#define VAULT_KEY_ERROR 3
inode_cache_t::~inode_cache_t()
{
if (key_data)
{
free(key_data);
key_data = NULL;
op_enc = NULL;
}
}
void cluster_client_t::vault_destroy()
{
if (vault_http_ctx)
{
#ifndef __MOCK__
http_destroy(vault_http_cli);
http_context_destroy(vault_http_ctx);
vault_http_cli = NULL;
vault_http_ctx = NULL;
#endif
}
}
void cluster_client_t::vault_parse_config()
{
vault_url = config["vault_url"].string_value();
vault_client_cert = config["vault_client_cert"].string_value();
vault_client_key = config["vault_client_key"].string_value();
vault_ca = config["vault_ca"].string_value();
vault_secret_api_path = "/v1/secret/";
if (config["vault_secret_api_path"].is_string())
vault_secret_api_path = config["vault_secret_api_path"].string_value();
vault_timeout_ms = config["vault_timeout_ms"].uint64_value();
if (!vault_timeout_ms)
vault_timeout_ms = 5000;
vault_error_timeout_sec = config["vault_error_timeout_sec"].uint64_value();
if (!vault_error_timeout_sec)
vault_error_timeout_sec = 60;
vault_refresh_leeway_sec = config["vault_refresh_leeway_sec"].uint64_value();
if (!vault_refresh_leeway_sec)
vault_refresh_leeway_sec = 60;
}
// FIXME: Rework client API by adding open/close and cache inode information in the "FD" (maybe)
void cluster_client_t::on_change_inode_hook(uint64_t inode, bool removed)
{
std::vector<inode_t> children = { inode };
for (size_t i = 0; i < children.size(); i++)
{
auto it = inode_cache_children.lower_bound(std::make_pair(children[i], (inode_t)0));
while (it != inode_cache_children.end() && it->first == children[i])
{
children.push_back(it->second);
it++;
}
}
for (auto & inode: children)
{
auto it = inode_cache.find(inode);
if (it != inode_cache.end())
{
auto icache = it->second;
for (auto & parent: icache->chain)
{
inode_cache_children.erase(std::make_pair(parent, inode));
}
inode_cache.erase(it);
}
}
}
std::shared_ptr<inode_cache_t> cluster_client_t::inode_cache_get(inode_t ino)
{
auto icache_it = inode_cache.find(ino);
if (icache_it != inode_cache.end())
{
return icache_it->second;
}
// Fill inode cache
auto ino_it = st_cli.inode_config.find(ino);
if (ino_it == st_cli.inode_config.end())
{
inode_cache[ino] = NULL;
return NULL;
}
auto pool_it = st_cli.pool_config.find(INODE_POOL(ino));
if (pool_it == st_cli.pool_config.end())
{
inode_cache[ino] = NULL;
return NULL;
}
auto & inode_cfg = ino_it->second;
auto & pool_cfg = pool_it->second;
std::shared_ptr<inode_cache_t> icache = std::make_shared<inode_cache_t>();
icache->readonly = inode_cfg.readonly;
icache->chain.push_back(ino);
std::vector<inode_config_t*> chain_cfg;
// FIXME: Allow unencrypted read & write when all chain is encrypted with the same key
int enc_key_count = !inode_cfg.enc_key.empty() ? 1 : 0;
if (inode_cfg.parent_id)
{
// Check for loops and cache the chain
robin_hood::unordered_flat_set<inode_t> seen;
seen.insert(ino);
uint64_t parent_id = inode_cfg.parent_id;
while (parent_id)
{
if (seen.find(parent_id) != seen.end())
{
icache->has_parent_loop = true;
break;
}
seen.insert(parent_id);
ino_it = st_cli.inode_config.find(parent_id);
if (INODE_POOL(parent_id) == INODE_POOL(ino))
{
icache->chain.push_back(parent_id);
if (ino_it == st_cli.inode_config.end())
chain_cfg.push_back(NULL);
else
{
chain_cfg.push_back(&ino_it->second);
if (!ino_it->second.enc_key.empty())
enc_key_count++;
}
}
else if (!icache->other_pool_parent_id)
icache->other_pool_parent_id = parent_id;
if (ino_it == st_cli.inode_config.end())
break;
parent_id = ino_it->second.parent_id;
}
}
// Check external keys and wait for loading, if required
if (enc_key_count)
{
for (size_t i = 0; i <= chain_cfg.size(); i++)
{
inode_config_t *cfg = !i ? &inode_cfg : chain_cfg[i-1];
if (cfg && cfg->enc_key.substr(0, strlen(VAULT_KEY_PREFIX)) == VAULT_KEY_PREFIX)
{
auto & ik = vault_keys[inode_cfg.enc_key];
if (ik.key_state == VAULT_KEY_ERROR || vault_url.empty())
{
icache->err_code = EPERM;
enc_key_count = 0;
}
else if (ik.key_state == VAULT_KEY_NOT_LOADED)
{
ik.key_state = VAULT_KEY_LOADING;
vault_key_load_queue.push_back(inode_cfg.enc_key);
vault_load_keys();
icache->err_code = EAGAIN;
enc_key_count = 0;
}
else if (ik.key_state == VAULT_KEY_LOADING)
{
icache->err_code = EAGAIN;
enc_key_count = 0;
}
else
{
assert(ik.key_state == VAULT_KEY_LOADED);
}
}
}
}
// Generate encryption key chain, if applicable
if (enc_key_count)
{
uint8_t *key_data = (uint8_t*)malloc_or_die(
AES_256_XTS_KEY_SIZE * enc_key_count +
sizeof(uint8_t*) * icache->chain.size() +
sizeof(osd_op_enc_t)
);
uint8_t **keys = (uint8_t**)(key_data + AES_256_XTS_KEY_SIZE * enc_key_count);
osd_op_enc_t *enc = (osd_op_enc_t*)((uint8_t*)keys + sizeof(uint8_t*)*icache->chain.size());
size_t key_pos = 0;
for (size_t i = 0; i <= chain_cfg.size(); i++)
{
inode_config_t *cfg = !i ? &inode_cfg : chain_cfg[i-1];
if (cfg && !cfg->enc_key.empty())
{
const auto & key = cfg->enc_key.substr(0, strlen(VAULT_KEY_PREFIX)) == VAULT_KEY_PREFIX
? vault_keys.at(cfg->enc_key).key
: cfg->enc_key;
assert(key_pos < AES_256_XTS_KEY_SIZE * enc_key_count);
assert(key.size() == 2*AES_256_XTS_KEY_SIZE);
keys[i] = key_data + key_pos;
fromhexstr(key, AES_256_XTS_KEY_SIZE, key_data + key_pos);
key_pos += AES_256_XTS_KEY_SIZE;
}
else
keys[i] = NULL;
}
enc->key_chain = keys;
enc->chain_size = icache->chain.size();
enc->read_chain_bitmap_pos = pool_cfg.data_block_size/pool_cfg.bitmap_granularity/8;
enc->bitmap_granularity = pool_cfg.bitmap_granularity;
icache->key_data = key_data;
icache->op_enc = enc;
}
inode_cache[ino] = icache;
for (auto & parent: icache->chain)
{
if (parent != ino)
inode_cache_children.insert(std::make_pair(parent, ino));
}
return icache;
}
#ifndef __MOCK__
bool cluster_client_t::vault_check_token()
{
timespec now;
clock_gettime(CLOCK_REALTIME, &now);
if (!vault_token_expire.tv_sec || vault_token_expire.tv_sec < now.tv_sec)
{
vault_loading = true;
http_json_post(
vault_http_cli, vault_url+"/v1/auth/cert/login", json11::Json::object{}, "",
(http_options_t){ .timeout = (int)vault_timeout_ms, .keepalive = true },
[this](http_message_t *response)
{
clock_gettime(CLOCK_REALTIME, &vault_token_expire);
vault_loading = false;
std::string err;
json11::Json data;
response->parse_json_response(err, data);
if (err != "")
{
vault_token_expire.tv_sec += vault_error_timeout_sec;
fprintf(stderr, "Vault request failed: %s\n", err.c_str());
}
else
{
uint64_t ttl = data["auth"]["lease_duration"].uint64_value();
vault_token = data["auth"]["client_token"].string_value();
if (vault_token.empty() || !ttl)
{
vault_token_expire.tv_sec += vault_error_timeout_sec;
fprintf(stderr, "No token or lease_duration in Vault response: %s\n", data.dump().c_str());
}
else
{
if (ttl < vault_refresh_leeway_sec)
vault_token_expire.tv_sec += ttl/2;
else
vault_token_expire.tv_sec += ttl - vault_refresh_leeway_sec;
}
}
vault_load_keys();
}
);
return false;
}
if (vault_token.empty())
{
// Auth error happened, mark all loads as failed
for (auto & key_id: vault_key_load_queue)
{
auto & k = vault_keys[key_id];
k.key_state = VAULT_KEY_ERROR;
}
vault_key_load_queue.clear();
auto ops = std::move(key_wait_ops);
for (cluster_op_t *op: ops)
inode_cache.erase(op->inode);
for (cluster_op_t *op: ops)
execute_internal(op);
return false;
}
return true;
}
#endif
void cluster_client_t::vault_load_keys()
{
if (vault_loading || !vault_key_load_queue.size())
{
return;
}
#ifdef __MOCK__
vault_loading = true;
#else
if (!vault_http_ctx)
{
std::string error;
vault_http_ctx = http_context_init(tfd, vault_client_cert, vault_client_key, vault_ca, true, error);
if (!vault_http_ctx)
{
fprintf(stderr, "Failed to initialize HTTP context for Vault: %s\n", error.c_str());
exit(1);
}
vault_http_cli = http_init(vault_http_ctx);
}
if (!vault_check_token())
{
return;
}
std::string key_id = vault_key_load_queue[0];
vault_key_load_queue.erase(vault_key_load_queue.begin());
vault_loading = true;
http_get(
vault_http_cli, vault_url+vault_secret_api_path+key_id.substr(strlen(VAULT_KEY_PREFIX)), "X-Vault-Token: "+vault_token+"\r\n",
(http_options_t){ .timeout = (int)vault_timeout_ms, .keepalive = true },
[this, key_id](http_message_t *response)
{
vault_loading = false;
std::string err;
json11::Json data;
response->parse_json_response(err, data);
vault_parse_secret(key_id, err, data);
}
);
#endif
}
void cluster_client_t::vault_parse_secret(const std::string & key_id, const std::string & err, json11::Json data)
{
vault_loading = false;
auto & k = vault_keys[key_id];
if (err != "")
{
k.key_state = VAULT_KEY_ERROR;
fprintf(stderr, "Vault %s%s%s request failed: %s\n", vault_url.c_str(),
vault_secret_api_path.c_str(), key_id.c_str()+strlen(VAULT_KEY_PREFIX), err.c_str());
}
else
{
auto hexkey = data["data"]["key"].string_value();
if (hexkey.empty() || !ishexstr(hexkey) || hexkey.size() != 2*AES_256_XTS_KEY_SIZE)
{
k.key_state = VAULT_KEY_ERROR;
fprintf(stderr, "Vault /v1/secret/%s request failed: 'key' is empty or has invalid format\n", key_id.c_str());
}
else
{
k.key_state = VAULT_KEY_LOADED;
k.key = hexkey;
}
}
if (vault_key_load_queue.empty())
{
auto ops = std::move(key_wait_ops);
for (cluster_op_t *op: ops)
inode_cache.erase(op->inode);
for (cluster_op_t *op: ops)
execute_internal(op);
}
else
vault_load_keys();
}
+1
View File
@@ -5,6 +5,7 @@
#include "cluster_client.h"
#define SCRAP_BUFFER_SIZE 4*1024*1024
#define PART_SENT 1
#define PART_DONE 2
#define PART_ERROR 4
+166 -289
View File
@@ -1,10 +1,7 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
#include <assert.h>
#include "malloc_or_die.h"
#include "osd_ops.h"
#include "msgr_op.h"
#include "pg_states.h"
#include "etcd_state_client.h"
#ifndef __MOCK__
@@ -25,19 +22,14 @@ etcd_state_client_t::~etcd_state_client_t()
stop_ws_keepalive();
if (etcd_watch_ws)
{
http_destroy(etcd_watch_ws);
http_close(etcd_watch_ws);
etcd_watch_ws = NULL;
}
if (keepalive_client)
{
http_destroy(keepalive_client);
http_close(keepalive_client);
keepalive_client = NULL;
}
if (http_ctx)
{
http_context_destroy(http_ctx);
http_ctx = NULL;
}
#endif
if (load_pgs_timer_id >= 0)
{
@@ -80,51 +72,55 @@ std::vector<std::string> etcd_state_client_t::get_addresses()
return addrs;
}
http_context_t *etcd_state_client_t::get_http_ctx()
{
if (!http_ctx)
{
std::string error;
http_ctx = http_context_init(tfd, etcd_client_cert, etcd_client_key, etcd_ca, true, error);
if (!http_ctx)
{
fprintf(stderr, "Failed to initialize HTTP context: %s\n", error.c_str());
exit(1);
}
}
return http_ctx;
}
void etcd_state_client_t::etcd_call_oneshot(const std::string & etcd_url, const std::string & api, json11::Json payload,
void etcd_state_client_t::etcd_call_oneshot(std::string etcd_address, std::string api, json11::Json payload,
int timeout, std::function<void(std::string, json11::Json)> callback)
{
auto http_cli = http_init(get_http_ctx());
http_json_post(http_cli, etcd_url+api, payload, "", { .timeout = timeout }, [http_cli, callback](http_message_t *response)
std::string etcd_api_path;
int pos = etcd_address.find('/');
if (pos >= 0)
{
etcd_api_path = etcd_address.substr(pos);
etcd_address = etcd_address.substr(0, pos);
}
std::string req = payload.dump();
req = "POST "+etcd_api_path+api+" HTTP/1.1\r\n"
"Host: "+etcd_address+"\r\n"
"Content-Type: application/json\r\n"
"Content-Length: "+std::to_string(req.size())+"\r\n"
"Connection: close\r\n"
"\r\n"+req;
auto http_cli = http_init(tfd);
auto cb = [http_cli, callback](const http_response_t *response)
{
std::string err;
json11::Json data;
response->parse_json_response(err, data);
callback(err, data);
http_destroy(http_cli);
});
http_close(http_cli);
};
http_request(http_cli, etcd_address, req, { .timeout = timeout }, cb);
}
void etcd_state_client_t::etcd_call(const std::string & api, json11::Json payload, int timeout,
void etcd_state_client_t::etcd_call(std::string api, json11::Json payload, int timeout,
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
{
pick_next_etcd([=]()
if (!etcd_addresses.size() && !etcd_local.size())
{
etcd_call_selected(api, payload, timeout, retries, interval, callback);
});
}
void etcd_state_client_t::etcd_call_selected(const std::string & api, json11::Json payload, int timeout,
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
{
const auto & url = selected_etcd_url;
fprintf(stderr, "etcd_address is missing in Vitastor configuration\n");
exit(1);
}
pick_next_etcd();
std::string etcd_address = selected_etcd_address;
std::string etcd_api_path;
int pos = etcd_address.find('/');
if (pos >= 0)
{
etcd_api_path = etcd_address.substr(pos);
etcd_address = etcd_address.substr(0, pos);
}
std::string req = payload.dump();
req = "POST "+url.path+api+" HTTP/1.1\r\n"
"Host: "+url.hostname+"\r\n"
req = "POST "+etcd_api_path+api+" HTTP/1.1\r\n"
"Host: "+etcd_address+"\r\n"
"Content-Type: application/json\r\n"
"Content-Length: "+std::to_string(req.size())+"\r\n"
"Connection: keep-alive\r\n"
@@ -132,15 +128,15 @@ void etcd_state_client_t::etcd_call_selected(const std::string & api, json11::Js
"\r\n"+req;
retries--;
auto cb = [this, api, payload, timeout, retries, interval, callback,
cur_addr = url.addr](http_message_t *response)
cur_addr = selected_etcd_address](const http_response_t *response)
{
std::string err;
json11::Json data;
response->parse_json_response(err, data);
if (err != "")
{
if (cur_addr == selected_etcd_url.addr)
selected_etcd_url = (http_url_t){};
if (cur_addr == selected_etcd_address)
selected_etcd_address = "";
if (retries > 0)
{
if (this->log_level > 0)
@@ -168,58 +164,54 @@ void etcd_state_client_t::etcd_call_selected(const std::string & api, json11::Js
callback(err, data);
};
if (!keepalive_client)
keepalive_client = http_init(get_http_ctx());
http_request(keepalive_client, url.addr, req, { .timeout = timeout, .keepalive = true, .ssl = url.ssl }, cb);
{
keepalive_client = http_init(tfd);
}
http_request(keepalive_client, etcd_address, req, { .timeout = timeout, .keepalive = true }, cb);
}
void etcd_state_client_t::add_etcd_url(std::string etcd_address)
void etcd_state_client_t::add_etcd_url(std::string addr)
{
if (etcd_address.size() > 0)
if (addr.length() > 0)
{
if (strtolower(addr.substr(0, 7)) == "http://")
addr = addr.substr(7);
else if (strtolower(addr.substr(0, 8)) == "https://")
{
fprintf(stderr, "HTTPS is unsupported for etcd. Either use plain HTTP or setup a local proxy for etcd interaction\n");
exit(1);
}
if (!local_ips.size())
{
// Fill local_ips
for (auto & ip: getifaddr_list(std::vector<addr_mask_t>(), true))
local_ips.insert(ip);
}
std::string etcd_api_path;
bool ssl = false;
if (etcd_address.substr(0, 8) == "https://")
{
ssl = true;
etcd_address = etcd_address.substr(8);
}
else if (etcd_address.substr(0, 7) == "http://")
etcd_address = etcd_address.substr(7);
auto pos = etcd_address.find('/');
if (pos != std::string::npos)
{
etcd_api_path = etcd_address.substr(pos);
etcd_address = etcd_address.substr(0, pos);
}
local_ips = getifaddr_list(std::vector<addr_mask_t>(), true);
std::string check_addr;
int pos = addr.find('/');
int pos2 = addr.find(':');
if (pos2 >= 0)
check_addr = addr.substr(0, pos2);
else if (pos >= 0)
check_addr = addr.substr(0, pos);
else
etcd_api_path = "/v3";
pos = etcd_address.find(':');
auto check_addr = (pos != std::string::npos ? etcd_address.substr(0, pos) : etcd_address);
bool is_local = local_ips.find(check_addr) != local_ips.end();
auto & to = (is_local ? etcd_local : etcd_addresses);
check_addr = (ssl ? "https://" : "http://") + etcd_address + etcd_api_path;
size_t i;
check_addr = addr;
if (pos == std::string::npos)
addr += "/v3";
bool local = false;
int i;
for (i = 0; i < local_ips.size(); i++)
{
if (local_ips[i] == check_addr)
{
local = true;
break;
}
}
auto & to = local ? this->etcd_local : this->etcd_addresses;
for (i = 0; i < to.size(); i++)
{
if (to[i] == check_addr)
if (to[i] == addr)
break;
}
if (i >= to.size())
{
to.push_back(check_addr);
// Check if it's a domain name
sockaddr_storage ss;
bool is_name = !is_local && !string_to_addr(etcd_address, true, 0, &ss);
auto & to_addr = (is_local ? etcd_local_addr_urls : (is_name ? etcd_name_urls : etcd_nonlocal_addr_urls));
to_addr.push_back((http_url_t){ .ssl = ssl, .addr = etcd_address, .hostname = etcd_address, .path = etcd_api_path });
}
to.push_back(addr);
}
}
@@ -227,9 +219,6 @@ void etcd_state_client_t::parse_config(const json11::Json & config)
{
this->etcd_local.clear();
this->etcd_addresses.clear();
this->etcd_local_addr_urls.clear();
this->etcd_nonlocal_addr_urls.clear();
this->etcd_name_urls.clear();
if (config["etcd_address"].is_string())
{
std::string ea = config["etcd_address"].string_value();
@@ -250,19 +239,7 @@ void etcd_state_client_t::parse_config(const json11::Json & config)
add_etcd_url(ea.string_value());
}
}
if (this->osd_num)
{
this->etcd_client_cert = config["osd_etcd_client_cert"].string_value();
this->etcd_client_key = config["osd_etcd_client_key"].string_value();
}
else
{
this->etcd_client_cert = config["etcd_client_cert"].string_value();
this->etcd_client_key = config["etcd_client_key"].string_value();
}
this->etcd_ca = config["etcd_ca"].string_value();
this->etcd_prefix = config["etcd_prefix"].string_value();
this->use_auth = config["use_auth"].bool_value();
if (this->etcd_prefix == "")
{
this->etcd_prefix = "/vitastor";
@@ -314,130 +291,66 @@ void etcd_state_client_t::parse_config(const json11::Json & config)
}
}
void etcd_state_client_t::pick_next_etcd(std::function<void()> cb)
void etcd_state_client_t::pick_next_etcd()
{
if (selected_etcd_address != "")
return;
if (addresses_to_try.size() == 0)
{
// Prefer local etcd, if any
for (int i = 0; i < etcd_local.size(); i++)
addresses_to_try.push_back(etcd_local[i]);
std::vector<int> ns;
for (int i = 0; i < etcd_addresses.size(); i++)
ns.push_back(i);
if (!rand_initialized)
{
timespec tv;
clock_gettime(CLOCK_REALTIME, &tv);
srand48(tv.tv_sec*1000000000 + tv.tv_nsec);
rand_initialized = true;
}
while (ns.size())
{
int i = lrand48() % ns.size();
addresses_to_try.push_back(etcd_addresses[ns[i]]);
ns.erase(ns.begin()+i, ns.begin()+i+1);
}
}
selected_etcd_address = addresses_to_try[0];
addresses_to_try.erase(addresses_to_try.begin(), addresses_to_try.begin()+1);
}
void etcd_state_client_t::start_etcd_watcher()
{
if (!etcd_addresses.size() && !etcd_local.size())
{
fprintf(stderr, "etcd_address is missing in Vitastor configuration\n");
exit(1);
}
if (selected_etcd_url.addr != "")
pick_next_etcd();
std::string etcd_address = selected_etcd_address;
std::string etcd_api_path;
int pos = etcd_address.find('/');
if (pos >= 0)
{
cb();
return;
etcd_api_path = etcd_address.substr(pos);
etcd_address = etcd_address.substr(0, pos);
}
if (etcd_urls_to_try.size() != 0)
{
selected_etcd_url = std::move(etcd_urls_to_try[0]);
etcd_urls_to_try.erase(etcd_urls_to_try.begin());
cb();
return;
}
on_resolve_queue.push_back(std::move(cb));
if (on_resolve_queue.size() > 1)
{
// Already resolving
return;
}
assert(!resolve_count);
local_to_try = 0;
for (auto & url: etcd_local_addr_urls)
{
// Prefer local IPs, if any
etcd_urls_to_try.push_back(url);
local_to_try++;
}
for (auto & url: etcd_nonlocal_addr_urls)
{
etcd_urls_to_try.push_back(url);
}
resolve_count++;
for (auto & url: etcd_name_urls)
{
resolve_count++;
http_resolve(get_http_ctx(), url.ssl, url.addr, [this, url](const std::string & error, const std::vector<std::string>& addresses)
{
if (error != "")
fprintf(stderr, "Error resolving %s: %s\n", url.addr.c_str(), error.c_str());
for (auto & addr: addresses)
{
auto url_copy = url;
url_copy.addr = addr;
if (local_ips.find(addr) != local_ips.end())
{
etcd_urls_to_try.insert(etcd_urls_to_try.begin(), std::move(url_copy));
local_to_try++;
}
else
etcd_urls_to_try.push_back(std::move(url_copy));
}
resolve_count--;
if (!resolve_count)
pick_next_etcd_on_resolve();
});
}
resolve_count--;
if (!resolve_count)
{
pick_next_etcd_on_resolve();
}
}
void etcd_state_client_t::pick_next_etcd_on_resolve()
{
if (!etcd_urls_to_try.size())
{
fprintf(stderr, "None of etcd_address could be resolved\n");
exit(1);
}
if (!rand_initialized)
{
timespec tv;
clock_gettime(CLOCK_REALTIME, &tv);
srand48(tv.tv_sec*1000000000 + tv.tv_nsec);
rand_initialized = true;
}
// Shuffle addresses
for (size_t i = etcd_urls_to_try.size()-1; i > local_to_try; i--)
{
size_t j = local_to_try + lrand48() % (i - local_to_try);
if (j != i)
std::swap(etcd_urls_to_try[i], etcd_urls_to_try[j]);
}
selected_etcd_url = std::move(etcd_urls_to_try[0]);
etcd_urls_to_try.erase(etcd_urls_to_try.begin());
auto cbs = std::move(on_resolve_queue);
for (auto cb: cbs)
{
cb();
}
}
void etcd_state_client_t::start_etcd_watcher()
{
pick_next_etcd([this]()
{
start_etcd_watcher_selected();
});
}
void etcd_state_client_t::start_etcd_watcher_selected()
{
const auto & url = selected_etcd_url;
etcd_watches_initialised = 0;
ws_alive = 1;
if (etcd_watch_ws)
{
http_close(etcd_watch_ws);
etcd_watch_ws = NULL;
}
if (this->log_level > 1)
{
fprintf(stderr, "Trying to connect to etcd websocket at %s%s%s (hostname %s), watch from revision %ju/%ju/%ju\n",
url.ssl ? "https://" : "http://", url.addr.c_str(), url.path.c_str(), url.hostname.c_str(),
fprintf(stderr, "Trying to connect to etcd websocket at %s, watch from revision %ju/%ju/%ju\n", etcd_address.c_str(),
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
}
if (!etcd_watch_ws)
etcd_watch_ws = http_init(get_http_ctx());
else
http_close(etcd_watch_ws);
open_websocket(etcd_watch_ws, url.addr, url.hostname, url.path+"/watch", { .timeout = etcd_slow_timeout, .ssl = url.ssl },
[this, cur_addr = url.addr](http_message_t *msg)
etcd_watch_ws = open_websocket(tfd, etcd_address, etcd_api_path+"/watch", etcd_slow_timeout,
[this, cur_addr = selected_etcd_address](const http_response_t *msg)
{
if (msg->body.length())
{
@@ -480,6 +393,7 @@ void etcd_state_client_t::start_etcd_watcher_selected()
fprintf(stderr, "Revisions before %ju were compacted by etcd, reloading state\n",
data["result"]["compact_revision"].uint64_value());
http_close(etcd_watch_ws);
etcd_watch_ws = NULL;
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = 0;
on_reload_hook();
}
@@ -524,7 +438,7 @@ void etcd_state_client_t::start_etcd_watcher_selected()
etcd_watch_revision_pg = watch_rev;
else if (watch_id == ETCD_OSD_STATE_WATCH_ID)
etcd_watch_revision_osd = watch_rev;
etcd_urls_to_try.clear();
addresses_to_try.clear();
}
// First gather all changes into a hash to remove multiple overwrites
std::map<std::string, etcd_kv_t> changes;
@@ -554,8 +468,13 @@ void etcd_state_client_t::start_etcd_watcher_selected()
if (msg->eof)
{
fprintf(stderr, "Disconnected from etcd %s\n", cur_addr.c_str());
if (cur_addr == selected_etcd_url.addr)
selected_etcd_url = (http_url_t){};
if (cur_addr == selected_etcd_address)
selected_etcd_address = "";
if (etcd_watch_ws)
{
http_close(etcd_watch_ws);
etcd_watch_ws = NULL;
}
if (etcd_watches_initialised == 0)
{
// Connection not established, retry in <etcd_quick_timeout>
@@ -630,7 +549,12 @@ void etcd_state_client_t::start_ws_keepalive()
{
if (this->log_level > 0)
{
fprintf(stderr, "Websocket ping failed, disconnecting from etcd %s\n", selected_etcd_url.addr.c_str());
fprintf(stderr, "Websocket ping failed, disconnecting from etcd %s\n", selected_etcd_address.c_str());
}
if (etcd_watch_ws)
{
http_close(etcd_watch_ws);
etcd_watch_ws = NULL;
}
start_etcd_watcher();
}
@@ -1018,8 +942,6 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
pc.used_for_app = "fs:"+pc.used_for_app;
else
pc.used_for_app = pool_item.second["used_for_app"].as_string();
// Create group permission
pc.creator_group = pool_item.second["creator_group"].string_value();
// Local Read Configuration
std::string local_reads = pool_item.second["local_reads"].string_value();
if (local_reads == "nearest")
@@ -1343,7 +1265,33 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
}
else
{
insert_inode_config(deserialize_inode_cfg(inode_num, kv.value, kv.mod_revision));
inode_t parent_inode_num = value["parent_id"].uint64_value();
if (parent_inode_num && !(parent_inode_num >> (64-POOL_ID_BITS)))
{
uint64_t parent_pool_id = value["parent_pool"].uint64_value();
if (!parent_pool_id)
parent_inode_num |= pool_id << (64-POOL_ID_BITS);
else if (parent_pool_id >= POOL_ID_MAX)
{
fprintf(
stderr, "Inode %ju/%ju parent_pool value is invalid, ignoring parent setting\n",
inode_num >> (64-POOL_ID_BITS), inode_num & (((uint64_t)1 << (64-POOL_ID_BITS)) - 1)
);
parent_inode_num = 0;
}
else
parent_inode_num |= parent_pool_id << (64-POOL_ID_BITS);
}
insert_inode_config((inode_config_t){
.num = inode_num,
.name = value["name"].string_value(),
.size = value["size"].uint64_value(),
.parent_id = parent_inode_num,
.readonly = value["readonly"].bool_value(),
.deleted = value["deleted"].bool_value(),
.meta = value["meta"],
.mod_revision = kv.mod_revision,
});
}
}
}
@@ -1354,14 +1302,6 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
if (on_change_node_placement_hook)
on_change_node_placement_hook();
}
else if (use_auth && key.substr(0, etcd_prefix.length()+13) == etcd_prefix+"/config/user/")
{
// <etcd_prefix>/config/user/<username>
if (!value.is_object())
user_info.erase(key.substr(etcd_prefix.length()+13));
else
user_info[key.substr(etcd_prefix.length()+13)] = value;
}
}
uint32_t etcd_state_client_t::parse_immediate_commit(const std::string & immediate_commit_str, uint32_t default_value)
@@ -1440,10 +1380,6 @@ json11::Json::object etcd_state_client_t::serialize_inode_cfg(inode_config_t *cf
new_cfg["parent_pool"] = (uint64_t)INODE_POOL(cfg->parent_id);
new_cfg["parent_id"] = (uint64_t)INODE_NO_POOL(cfg->parent_id);
}
if (!cfg->enc_key.empty())
{
new_cfg["enc_key"] = cfg->enc_key;
}
if (cfg->readonly)
{
new_cfg["readonly"] = true;
@@ -1452,18 +1388,6 @@ json11::Json::object etcd_state_client_t::serialize_inode_cfg(inode_config_t *cf
{
new_cfg["deleted"] = true;
}
if (!cfg->owner.empty())
{
new_cfg["owner"] = cfg->owner;
}
if (!cfg->owner_group.empty())
{
new_cfg["owner_group"] = cfg->owner_group;
}
if (!cfg->reader_group.empty())
{
new_cfg["reader_group"] = cfg->reader_group;
}
if (cfg->meta.is_object())
{
new_cfg["meta"] = cfg->meta;
@@ -1471,53 +1395,6 @@ json11::Json::object etcd_state_client_t::serialize_inode_cfg(inode_config_t *cf
return new_cfg;
}
inode_config_t etcd_state_client_t::deserialize_inode_cfg(uint64_t inode_num, json11::Json value, uint64_t mod_revision)
{
inode_t parent_inode_num = value["parent_id"].uint64_value();
if (parent_inode_num && !INODE_POOL(parent_inode_num))
{
uint64_t parent_pool_id = value["parent_pool"].uint64_value();
if (!parent_pool_id)
parent_inode_num = INODE_WITH_POOL(INODE_POOL(inode_num), parent_inode_num);
else if (parent_pool_id >= POOL_ID_MAX)
{
fprintf(
stderr, "Inode %u/%ju parent_pool value is invalid, ignoring parent setting\n",
INODE_POOL(inode_num), INODE_NO_POOL(inode_num)
);
parent_inode_num = 0;
}
else
parent_inode_num |= parent_pool_id << (64-POOL_ID_BITS);
}
std::string enc_key;
if (!value["enc_key"].is_null())
{
enc_key = value["enc_key"].string_value();
if (enc_key.substr(0, strlen(VAULT_KEY_PREFIX)) != VAULT_KEY_PREFIX &&
(enc_key.size() != 2*AES_256_XTS_KEY_SIZE || !ishexstr(enc_key)))
{
enc_key = "";
fprintf(stderr, "Inode %u/%ju has invalid enc_key, should be %u bit hex string or Vault key reference\n",
INODE_POOL(inode_num), INODE_NO_POOL(inode_num), AES_256_XTS_KEY_SIZE);
}
}
return (inode_config_t){
.num = inode_num,
.name = value["name"].string_value(),
.size = value["size"].uint64_value(),
.parent_id = parent_inode_num,
.readonly = value["readonly"].bool_value(),
.deleted = value["deleted"].bool_value(),
.enc_key = std::move(enc_key),
.owner = value["owner"].string_value(),
.owner_group = value["owner_group"].string_value(),
.reader_group = value["reader_group"].string_value(),
.meta = value["meta"],
.mod_revision = mod_revision,
};
}
int etcd_state_client_t::address_count()
{
return etcd_addresses.size() + etcd_local.size();
+7 -41
View File
@@ -4,7 +4,6 @@
#pragma once
#include <set>
#include <memory>
#include "json11/json11.hpp"
#include "object_id.h"
@@ -20,8 +19,6 @@
#define MAX_DATA_BLOCK_SIZE 128*1024*1024
#define DEFAULT_BITMAP_GRANULARITY 4096
#define VAULT_KEY_PREFIX "vault:"
#ifndef IMMEDIATE_NONE
#define IMMEDIATE_NONE 0
#define IMMEDIATE_SMALL 1
@@ -69,7 +66,6 @@ struct pool_config_t
std::map<pg_num_t, pg_config_t> pg_config;
uint64_t scrub_interval = 0;
std::string used_for_app;
std::string creator_group;
int backfillfull = 0;
int local_reads = 0;
@@ -87,9 +83,6 @@ struct inode_config_t
inode_t parent_id = 0;
bool readonly = false;
bool deleted = false;
std::string enc_key;
// Permissions
std::string owner, owner_group, reader_group;
// Arbitrary metadata
json11::Json meta;
// Change revision of the metadata in etcd
@@ -102,41 +95,23 @@ struct inode_watch_t
inode_config_t cfg = {};
};
struct http_url_t
{
bool ssl;
std::string addr;
std::string hostname;
std::string path;
};
struct http_co_t;
struct http_context_t;
struct __attribute__((visibility("default"))) etcd_state_client_t
{
protected:
std::set<std::string> local_ips;
std::vector<std::string> etcd_local;
std::vector<std::string> local_ips;
std::vector<std::string> etcd_addresses;
std::vector<http_url_t> etcd_local_addr_urls;
std::vector<http_url_t> etcd_nonlocal_addr_urls;
std::vector<http_url_t> etcd_name_urls;
size_t local_to_try = 0;
std::vector<http_url_t> etcd_urls_to_try;
http_url_t selected_etcd_url;
size_t resolve_count = 0;
std::vector<std::string> etcd_local;
std::string selected_etcd_address;
std::vector<std::string> addresses_to_try;
std::vector<inode_watch_t*> watches;
std::vector<std::function<void()>> on_resolve_queue;
bool new_pg_config = false;
int ws_keepalive_timer = -1;
int ws_alive = 0;
bool rand_initialized = false;
void add_etcd_url(std::string);
void pick_next_etcd(std::function<void()> cb);
void pick_next_etcd_on_resolve();
void etcd_call_selected(const std::string & api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
void start_etcd_watcher_selected();
void pick_next_etcd();
public:
int etcd_keepalive_timeout = 30;
int etcd_ws_keepalive_interval = 5;
@@ -145,20 +120,14 @@ public:
int etcd_slow_timeout = 5000;
int etcd_min_reload_interval = 1000;
bool infinite_start = true;
bool use_auth = false;
uint64_t global_block_size = DEFAULT_BLOCK_SIZE;
uint32_t global_bitmap_granularity = DEFAULT_BITMAP_GRANULARITY;
uint32_t global_immediate_commit = IMMEDIATE_NONE;
uint64_t osd_num = 0;
std::string etcd_prefix;
std::string etcd_client_cert;
std::string etcd_client_key;
std::string etcd_ca;
int log_level = 0;
timerfd_manager_t *tfd = NULL;
http_context_t *http_ctx = NULL;
http_co_t *etcd_watch_ws = NULL, *keepalive_client = NULL;
int etcd_watches_initialised = 0;
uint64_t etcd_watch_revision_config = 0;
@@ -171,7 +140,6 @@ public:
std::set<osd_num_t> seen_peers;
std::map<inode_t, inode_config_t> inode_config;
std::map<std::string, inode_t> inode_by_name;
std::map<std::string, json11::Json> user_info;
json11::Json node_placement;
std::function<void(std::map<std::string, etcd_kv_t> &)> on_change_hook;
@@ -190,12 +158,10 @@ public:
std::function<void(http_co_t *)> on_start_watcher_hook;
json11::Json::object serialize_inode_cfg(inode_config_t *cfg);
inode_config_t deserialize_inode_cfg(uint64_t inode_num, json11::Json value, uint64_t mod_revision);
etcd_kv_t parse_etcd_kv(const json11::Json & kv_json);
std::vector<std::string> get_addresses();
http_context_t *get_http_ctx();
void etcd_call_oneshot(const std::string & etcd_address, const std::string & api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback);
void etcd_call(const std::string & api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
void etcd_call_oneshot(std::string etcd_address, std::string api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback);
void etcd_call(std::string api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
void etcd_txn(json11::Json txn, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
void etcd_txn_slow(json11::Json txn, std::function<void(std::string, json11::Json)> callback);
void start_etcd_watcher();
File diff suppressed because it is too large Load Diff
+5 -38
View File
@@ -8,10 +8,6 @@
#include <functional>
#include "json11/json11.hpp"
#ifdef WITH_OPENSSL
#include <openssl/types.h>
#endif
#define WS_CONTINUATION 0
#define WS_TEXT 1
#define WS_BINARY 2
@@ -21,19 +17,14 @@
class timerfd_manager_t;
#pragma GCC visibility push(default)
struct http_options_t
{
int timeout;
bool want_streaming;
bool keepalive;
bool ssl;
};
struct http_context_t;
struct http_message_t
struct http_response_t
{
std::string error;
@@ -50,34 +41,10 @@ struct http_message_t
// Opened websocket or keepalive HTTP connection
struct http_co_t;
http_context_t* http_context_init(timerfd_manager_t *tfd, const std::string & ssl_cert, const std::string & ssl_key,
const std::string & ssl_ca, bool verify_peer, std::string & error);
std::string http_context_get_ssl_cn(http_context_t *ctx);
void http_resolve(http_context_t *ctx, bool ssl, std::string host,
std::function<void(const std::string & error, const std::vector<std::string> & addrs)> cb);
void http_context_destroy(http_context_t *ctx);
http_co_t* http_init(http_context_t *ctx = NULL);
void open_websocket(http_co_t *handler, const std::string & addr, const std::string & hostname, const std::string & path,
const http_options_t & options, std::function<void(http_message_t *msg)> on_message);
http_co_t* http_init(timerfd_manager_t *tfd);
http_co_t* open_websocket(timerfd_manager_t *tfd, const std::string & host, const std::string & path,
int timeout, std::function<void(const http_response_t *msg)> on_message);
void http_request(http_co_t *handler, const std::string & host, const std::string & request,
const http_options_t & options, std::function<void(http_message_t *response)> response_callback);
void http_get(http_co_t *handler, const std::string & url, const std::string & headers,
const http_options_t & options, std::function<void(http_message_t *response)> response_callback);
void http_json_post(http_co_t *handler, const std::string & url, json11::Json body, const std::string & headers,
const http_options_t & options, std::function<void(http_message_t *response)> response_callback);
const http_options_t & options, std::function<void(const http_response_t *response)> response_callback);
void http_post_message(http_co_t *handler, uint8_t type, const std::string & msg);
void http_serve(http_co_t *handler, int peer_fd, const http_options_t & options,
std::function<void(http_message_t *msg)> request_callback);
void http_reply(http_co_t *handler, const std::string & reply);
void http_close(http_co_t *co);
void http_destroy(http_co_t *co);
#pragma GCC visibility pop
#ifdef WITH_OPENSSL
bool openssl_ctx_add_ca(SSL_CTX *ssl_ctx, const std::string & file_or_pem);
bool openssl_ctx_use_ca(SSL_CTX *ssl_ctx, const std::string & file_or_pem);
std::string openssl_get_cn(X509 *x509);
bool openssl_ctx_use_cert(SSL_CTX *ssl_ctx, const std::string & file_or_pem, std::string & common_name);
bool openssl_ctx_use_key(SSL_CTX *ssl_ctx, const std::string & file_or_pem);
#endif
+3 -127
View File
@@ -14,13 +14,6 @@
#ifdef WITH_RDMA
#include "msgr_rdma.h"
#endif
#include "http_client.h"
#ifdef WITH_OPENSSL
#include <openssl/bio.h>
#include <openssl/err.h>
#include <openssl/pem.h>
#include <openssl/ssl.h>
#endif
#include <sys/poll.h>
@@ -124,43 +117,6 @@ void msgr_iothread_t::run()
void osd_messenger_t::init()
{
if (!tls_cert.empty() || !tls_key.empty() || !osd_tls_ca.empty() || !client_tls_ca.empty())
{
// Initialize TLS context
// FIXME: require OpenSSL
#ifndef WITH_OPENSSL
fprintf(stderr, "Vitastor is built without OpenSSL support\n");
exit(1);
#else
if (tls_cert.empty() || tls_key.empty() || osd_tls_ca.empty() || osd_num && client_tls_ca.empty())
{
if (osd_num)
fprintf(stderr, "Vitastor OSD TLS requires osd_tls_cert, osd_tls_key, osd_tls_ca, client_tls_ca\n");
else
fprintf(stderr, "Vitastor client TLS requires tls_cert, tls_key and osd_tls_ca\n");
exit(1);
}
else
{
ssl_ctx = SSL_CTX_new(TLS_method());
SSL_CTX_set_verify(ssl_ctx, SSL_VERIFY_PEER, NULL);
bool ok = SSL_CTX_set_min_proto_version(ssl_ctx, TLS1_3_VERSION);
ok = ok && openssl_ctx_add_ca(ssl_ctx, osd_tls_ca);
if (osd_num)
{
// OSD uses 2 separate root certificates to distinguish between clients and peer OSDs
ok = ok && openssl_ctx_add_ca(ssl_ctx, client_tls_ca);
}
ok = ok && openssl_ctx_use_cert(ssl_ctx, tls_cert, tls_cn);
ok = ok && openssl_ctx_use_key(ssl_ctx, tls_key);
if (!ok)
{
SSL_CTX_free(ssl_ctx);
ssl_ctx = NULL;
}
}
#endif
}
#ifdef WITH_RDMACM
if (use_rdmacm)
{
@@ -338,21 +294,6 @@ osd_messenger_t::~osd_messenger_t()
rdma_destroy_event_channel(rdmacm_evch);
rdmacm_evch = NULL;
}
#endif
for (auto encrypt_ctx: encrypt_ctx_pool)
{
destroy_aes_xts_encrypt(encrypt_ctx);
}
for (auto decrypt_ctx: decrypt_ctx_pool)
{
destroy_aes_xts_decrypt(decrypt_ctx);
}
#ifdef WITH_OPENSSL
if (ssl_ctx)
{
SSL_CTX_free(ssl_ctx);
ssl_ctx = NULL;
}
#endif
}
@@ -388,30 +329,6 @@ void osd_messenger_t::parse_config(const json11::Json & config)
if (!this->rdma_max_msg || this->rdma_max_msg > 128*1024*1024)
this->rdma_max_msg = 129*1024;
#endif
this->max_aes_xts_pool_size = config["max_aes_xts_pool_size"].uint64_value();
if (!this->max_aes_xts_pool_size)
this->max_aes_xts_pool_size = 256;
if (config["proto_checksums"].is_null())
this->use_proto_checksums = MSGR_CSUM_PAYLOAD;
else if (config["proto_checksums"].is_bool())
this->use_proto_checksums = config["proto_checksums"].bool_value() ? MSGR_CSUM_FULL : 0;
else if (config["proto_checksums"].string_value() != "")
this->use_proto_checksums = config["proto_checksums"].string_value() == "full" ? MSGR_CSUM_FULL : MSGR_CSUM_PAYLOAD;
else
this->use_proto_checksums = 0;
if (!osd_num)
{
tls_cert = config["tls_cert"].string_value();
tls_key = config["tls_key"].string_value();
osd_tls_ca = config["osd_tls_ca"].string_value();
}
else
{
tls_cert = config["osd_tls_cert"].string_value();
tls_key = config["osd_tls_key"].string_value();
osd_tls_ca = config["osd_tls_ca"].string_value();
client_tls_ca = config["client_tls_ca"].string_value();
}
if (!osd_num)
this->iothread_count = (uint32_t)config["client_iothread_count"].uint64_value();
else
@@ -593,7 +510,7 @@ void osd_messenger_t::try_connect_peer_tcp(osd_num_t peer_osd, const char *peer_
cl->peer_state = PEER_CONNECTING;
cl->connect_timeout_id = -1;
cl->osd_num = peer_osd;
cl->in_buf = (uint8_t*)malloc_or_die(receive_buffer_size);
cl->in_buf = malloc_or_die(receive_buffer_size);
clients[client_id] = cl;
clients_by_fd[peer_fd] = cl;
tfd->set_fd_handler(peer_fd, true, [this](int peer_fd, int epoll_events)
@@ -642,10 +559,6 @@ void osd_messenger_t::handle_connect_epoll(int peer_fd)
handle_peer_epoll(peer_fd, epoll_events);
});
// Check OSD number
if (!tls_cert.empty())
{
ssl_init(cl, false);
}
check_peer_config(cl);
}
@@ -735,12 +648,7 @@ void osd_messenger_t::check_peer_config(osd_client_t *cl)
// Inform that we're OSD <osd_num>
payload["osd_num"] = osd_num;
}
auto features = json11::Json::object{ { "check_sequencing", true } };
if (use_proto_checksums)
{
features["proto_checksums"] = use_proto_checksums;
}
payload["features"] = features;
payload["features"] = json11::Json::object{ { "check_sequencing", true } };
#ifdef WITH_RDMA
if (!use_rdmacm && rdma_contexts.size())
{
@@ -815,14 +723,6 @@ void osd_messenger_t::check_peer_config(osd_client_t *cl)
delete op;
return;
}
if (use_proto_checksums)
{
auto peer_csums = config["features"]["proto_checksums"].uint64_value();
if (peer_csums == MSGR_CSUM_FULL && use_proto_checksums == MSGR_CSUM_FULL)
cl->proto_csum_status = MSGR_CSUM_FULL;
else if (peer_csums && use_proto_checksums)
cl->proto_csum_status = MSGR_CSUM_PAYLOAD;
}
#ifdef WITH_RDMA
if (!use_rdmacm && cl->rdma_conn && config["rdma_address"].is_string())
{
@@ -886,11 +786,7 @@ void osd_messenger_t::accept_connections(int listen_fd)
cl->peer_port = ntohs(((sockaddr_in*)&addr)->sin_port);
cl->peer_fd = peer_fd;
cl->peer_state = PEER_CONNECTED;
cl->in_buf = (uint8_t*)malloc_or_die(receive_buffer_size);
if (!tls_cert.empty())
{
ssl_init(cl, true);
}
cl->in_buf = malloc_or_die(receive_buffer_size);
// Add FD to epoll
tfd->set_fd_handler(peer_fd, false, [this](int peer_fd, int epoll_events)
{
@@ -905,26 +801,6 @@ void osd_messenger_t::accept_connections(int listen_fd)
}
}
void osd_messenger_t::ssl_init(osd_client_t *cl, bool server_mode)
{
#ifdef WITH_OPENSSL
cl->write_to_ssl = BIO_new(BIO_s_mem());
cl->read_from_ssl = BIO_new(BIO_s_mem());
cl->ssl_cli = SSL_new(ssl_ctx);
if (server_mode)
{
SSL_set_accept_state(cl->ssl_cli);
}
else
{
SSL_set_connect_state(cl->ssl_cli);
}
SSL_set_bio(cl->ssl_cli, cl->write_to_ssl, cl->read_from_ssl);
bool ok = ssl_do_handshake(cl);
assert(ok);
#endif
}
#ifdef WITH_RDMA
msgr_rdma_context_t* osd_messenger_t::choose_rdma_context(osd_client_t *cl)
{
+15 -93
View File
@@ -12,11 +12,6 @@
#include <deque>
#include <vector>
#ifdef WITH_OPENSSL
#include <openssl/types.h>
#endif
#include "../util/xxh_x86dispatch.h"
#include "../util/robin_hood.h"
#include "malloc_or_die.h"
#include "json11/json11.hpp"
@@ -36,14 +31,13 @@
#define PEER_RDMA 4
#define PEER_STOPPED 5
#define MSGR_CSUM_PAYLOAD 1
#define MSGR_CSUM_FULL 2
#define MSGR_CSUM_NEG 4
#define VITASTOR_CONFIG_PATH "/etc/vitastor/vitastor.conf"
#define DEFAULT_MIN_ZEROCOPY_SEND_SIZE 32*1024
#define MSGR_SENDP_HDR 1
#define MSGR_SENDP_FREE 2
struct msgr_sendp_t
{
osd_op_t *op;
@@ -55,17 +49,6 @@ struct msgr_rdma_connection_t;
struct msgr_rdma_context_t;
#endif
struct op_aes_xts_encrypt_t;
struct op_aes_xts_decrypt_t;
void destroy_aes_xts_encrypt(op_aes_xts_encrypt_t *encrypt_ctx);
void destroy_aes_xts_decrypt(op_aes_xts_decrypt_t *decrypt_ctx);
struct __attribute__((__packed__)) msgr_tls_record_hdr_t
{
uint8_t encrypted;
uint32_t size;
};
struct osd_client_t
{
uint64_t client_id = 0;
@@ -82,40 +65,23 @@ struct osd_client_t
osd_num_t in_osd_num = 0;
bool is_incoming = false;
uint8_t *in_buf = NULL;
void *in_buf = NULL;
#ifdef WITH_RDMA
msgr_rdma_connection_t *rdma_conn = NULL;
#endif
#ifdef WITH_OPENSSL
SSL *ssl_cli = NULL;
BIO *write_to_ssl = NULL;
// FIXME: use custom bio to avoid 1 more memory copy?
BIO *read_from_ssl = NULL;
uint8_t *ssl_out_buf = NULL;
size_t ssl_out_buf_size = 0, ssl_out_buf_cap = 0;
bool ssl_handshake_done = false;
bool ssl_want_write = false;
#endif
// Read state
int read_ready = 0;
osd_op_t *read_op = NULL;
size_t read_op_size = 0;
size_t read_op_pos = 0;
iovec read_iov = { 0 };
msghdr read_msg = { 0 };
std::vector<iovec> recv_list;
size_t recv_list_size = 0;
int read_remaining = 0;
int read_state = 0;
osd_op_buf_list_t recv_list;
uint64_t read_op_id = 1;
bool check_sequencing = false;
bool enable_pg_locks = false;
op_aes_xts_decrypt_t *decrypt_ctx = NULL;
size_t read_op_inline_decrypt_pos = 0;
size_t read_op_inline_decrypt_in = 0;
int proto_csum_status = 0;
XXH3_state_t* read_csum_state = NULL;
// Incoming operations
std::vector<osd_op_t*> received_ops;
@@ -128,17 +94,11 @@ struct osd_client_t
std::set<pool_pg_num_t> dirty_pgs;
// Write state
std::deque<osd_op_t *> write_ops;
osd_op_t *write_op = NULL;
size_t write_op_pos = 0;
msghdr write_msg = { 0 };
int write_state = 0;
std::vector<iovec> send_list;
size_t send_list_size = 0;
std::deque<osd_op_t*> send_free_ops;
std::vector<iovec> send_list, next_send_list;
std::vector<msgr_sendp_t> outbox, next_outbox;
std::vector<osd_op_t*> zc_free_list;
op_aes_xts_encrypt_t *encrypt_ctx = NULL;
XXH3_state_t* write_csum_state = NULL;
~osd_client_t();
void cancel_ops();
@@ -230,12 +190,6 @@ protected:
bool use_sync_send_recv = false;
int min_zerocopy_send_size = DEFAULT_MIN_ZEROCOPY_SEND_SIZE;
int iothread_count = 0;
int max_aes_xts_pool_size = 256;
std::string tls_cert;
std::string tls_key;
std::string osd_tls_ca;
std::string client_tls_ca;
#ifdef WITH_RDMA
bool use_rdma = true;
@@ -253,28 +207,12 @@ protected:
robin_hood::unordered_flat_map<rdma_cm_id*, rdmacm_connecting_t*> rdmacm_connecting;
#endif
#ifdef WITH_OPENSSL
SSL_CTX *ssl_ctx = NULL;
std::string tls_cn;
void ssl_init(osd_client_t *cl, bool server_mode);
bool ssl_do_handshake(osd_client_t *cl);
bool ssl_do_encrypt(osd_client_t *cl);
size_t ssl_do_encrypt_to(osd_client_t *cl, uint8_t *buf, size_t size);
bool ssl_op_write_buf(osd_client_t *cl, uint8_t *src, size_t src_len, bool skip_csum, size_t & from, size_t & done);
size_t ssl_op_copy_to(osd_client_t *cl, uint8_t *dst, size_t dst_len);
void ssl_op_get_write_buffers(osd_client_t *cl, std::vector<iovec> & lst);
#endif
std::vector<msgr_iothread_t*> iothreads;
std::vector<uint64_t> read_ready_clients;
std::vector<uint64_t> write_ready_clients;
// We don't use ringloop->set_immediate here because we may have no ringloop in client :)
std::deque<osd_op_t*> set_immediate_ops;
std::vector<op_aes_xts_encrypt_t*> encrypt_ctx_pool;
std::vector<op_aes_xts_decrypt_t*> decrypt_ctx_pool;
public:
timerfd_manager_t *tfd = NULL;
ring_loop_t *ringloop = NULL;
@@ -292,7 +230,6 @@ public:
std::vector<addr_mask_t> osd_cluster_network_masks;
std::vector<std::string> all_osd_networks;
std::vector<addr_mask_t> all_osd_network_masks;
int use_proto_checksums = 0;
// op statistics
osd_op_stats_t stats, recovery_stats;
@@ -342,32 +279,17 @@ protected:
bool try_send(osd_client_t *cl);
void handle_send(int result, bool prev, bool more, osd_client_t *cl);
size_t op_copy_to(osd_client_t *cl, uint8_t *dst, size_t dst_len);
void op_get_write_buffers(osd_client_t *cl, std::vector<iovec> & lst);
void next_write_op(osd_client_t *cl);
bool op_write_buf(osd_client_t *cl, uint8_t *src, size_t src_len, uint8_t *dst, size_t dst_len, bool skip_csum, size_t & from, size_t & done);
bool op_copy_data_to(osd_client_t *cl, uint8_t *dst, size_t dst_len, size_t & from, size_t & done);
void handle_read(int result, osd_client_t *cl);
bool handle_read_buffer(osd_client_t *cl, uint8_t *curbuf, size_t bufsize);
bool handle_hdr(osd_client_t *cl);
bool allocate_op_buffers(osd_client_t *cl);
bool allocate_reply_buffers(osd_client_t *cl, osd_op_t *op);
bool op_copy_from(osd_client_t *cl, uint8_t *src, size_t src_len, size_t & done);
void op_get_read_buffers(osd_client_t *cl, std::vector<iovec> & lst);
void op_alloc_temp_buffers(osd_op_t *op, int i);
bool handle_finished_op(osd_client_t *cl);
bool handle_read(int result, osd_client_t *cl);
bool handle_read_buffer(osd_client_t *cl, void *curbuf, int remain);
bool handle_finished_read(osd_client_t *cl);
void handle_op_hdr(osd_client_t *cl);
bool handle_reply_hdr(osd_client_t *cl);
void handle_reply_ready(osd_op_t *op);
void handle_immediate_ops();
bool op_encrypted_copy_data_to(osd_client_t* cl, uint8_t *buf, size_t len, size_t from, size_t & done);
bool op_decrypted_copy_data_from(osd_client_t* cl, uint8_t *buf, size_t len, size_t from, size_t & done);
void op_decrypt_start(osd_client_t* cl);
void op_decrypt_inline(osd_client_t* cl);
void op_decrypt_free(osd_client_t* cl);
#ifdef WITH_RDMA
void try_send_rdma(osd_client_t *cl);
int try_send_rdma_copy(osd_client_t *cl, uint8_t *dst, int dst_len);
bool init_recv_rdma(osd_client_t *cl);
void handle_rdma_events(msgr_rdma_context_t *rdma_context);
msgr_rdma_context_t* choose_rdma_context(osd_client_t *cl);
-484
View File
@@ -1,484 +0,0 @@
// Copyright (c) Vitaliy Filippov, 2026+
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
#include <assert.h>
#include "etcd_state_client.h"
#include "messenger.h"
#include "msgr_encrypt.h"
op_aes_xts_encrypt_t::op_aes_xts_encrypt_t()
{
#ifdef WITH_OPENSSL
if (!(ctx = EVP_CIPHER_CTX_new()))
{
ERR_print_errors_fp(stderr);
abort();
}
EVP_CIPHER_CTX_set_padding(ctx, 0);
if (EVP_EncryptInit_ex(ctx, EVP_aes_256_xts(), NULL, NULL, NULL) != 1)
{
ERR_print_errors_fp(stderr);
abort();
}
#else
fprintf(stderr, "Error: Vitastor is built without encryption support\n");
abort();
#endif
}
op_aes_xts_encrypt_t::~op_aes_xts_encrypt_t()
{
assert(!encrypted);
#ifdef WITH_OPENSSL
EVP_CIPHER_CTX_free(ctx);
#endif
if (tmp)
free(tmp);
}
void op_aes_xts_encrypt_t::start(uint8_t *key, uint64_t start_offset, size_t block_size)
{
assert(!encrypted);
this->start_offset = start_offset;
this->key = key;
this->block_size = block_size;
this->offset = 0;
this->encrypted = false;
this->tmp_pos = 0;
if (tmp && tmp_size != block_size)
{
free(tmp);
tmp = NULL;
tmp_size = 0;
}
#ifdef WITH_OPENSSL
if (EVP_EncryptInit_ex(ctx, NULL, NULL, key, NULL) != 1)
{
ERR_print_errors_fp(stderr);
abort();
}
#endif
}
void op_aes_xts_encrypt_t::encrypt_block(uint8_t *in, uint8_t *out)
{
#ifdef WITH_OPENSSL
uint8_t iv[16] = { 0 };
*((uint64_t*)iv) = start_offset + offset - offset%block_size;
if (EVP_EncryptInit_ex(ctx, NULL, NULL, NULL, iv) != 1)
{
ERR_print_errors_fp(stderr);
abort();
}
int actual_out = 0;
if (EVP_EncryptUpdate(ctx, out, &actual_out, in, block_size) != 1)
{
ERR_print_errors_fp(stderr);
abort();
}
assert(actual_out == block_size);
#endif
}
void op_aes_xts_encrypt_t::update(uint8_t *in, size_t max_in, uint8_t *out, size_t max_out, size_t & done_in, size_t & done_out)
{
// Fucking AES-XTS implementations (all of them) don't have streaming support,
// crafting IV to resume encryption is slow, so we have to accumulate a full block
// and encrypt it at once :-(
// And then we have to support consuming it in parts because it's simpler for the
// higher layers.
if (encrypted)
{
// Copy accumulated and encrypted output
assert(tmp);
if (max_out > block_size - tmp_pos)
max_out = block_size - tmp_pos;
memcpy(out, tmp + tmp_pos, max_out);
done_out += max_out;
tmp_pos += max_out;
if (tmp_pos >= block_size)
encrypted = false;
}
else if (max_in < block_size - offset%block_size)
{
// Just accumulate input
if (!tmp)
{
tmp = (uint8_t*)malloc_or_die(block_size);
tmp_size = block_size;
}
memcpy(tmp + offset%block_size, in, max_in);
done_in += max_in;
offset += max_in;
}
else if (max_out < block_size)
{
// Accumulate and encrypt input in <tmp>, then copy part of it to <out>
if (!tmp)
{
tmp = (uint8_t*)malloc_or_die(block_size);
tmp_size = block_size;
}
max_in = block_size - offset%block_size;
memcpy(tmp + offset%block_size, in, max_in);
encrypt_block(tmp, tmp);
encrypted = true;
memcpy(out, tmp, max_out);
tmp_pos = max_out;
done_in += max_in;
offset += max_in;
done_out += max_out;
}
else if (!(offset%block_size))
{
// Full block - simplest case
encrypt_block(in, out);
done_in += block_size;
offset += block_size;
done_out += block_size;
}
else
{
// Accumulate input and encrypt directly to <output>
assert(tmp);
max_in = block_size - offset%block_size;
memcpy(tmp + offset%block_size, in, max_in);
encrypt_block(tmp, out);
done_in += max_in;
offset += max_in;
done_out += block_size;
}
}
void destroy_aes_xts_encrypt(op_aes_xts_encrypt_t *encrypt_ctx)
{
delete encrypt_ctx;
}
op_aes_xts_decrypt_t::op_aes_xts_decrypt_t()
{
#ifdef WITH_OPENSSL
if (!(ctx = EVP_CIPHER_CTX_new()))
{
ERR_print_errors_fp(stderr);
abort();
}
EVP_CIPHER_CTX_set_padding(ctx, 0);
if (EVP_DecryptInit_ex(ctx, EVP_aes_256_xts(), NULL, NULL, NULL) != 1)
{
ERR_print_errors_fp(stderr);
abort();
}
#else
fprintf(stderr, "Error: Vitastor is built without encryption support\n");
abort();
#endif
}
op_aes_xts_decrypt_t::~op_aes_xts_decrypt_t()
{
assert(!decrypted);
#ifdef WITH_OPENSSL
EVP_CIPHER_CTX_free(ctx);
#endif
if (tmp)
free(tmp);
}
void op_aes_xts_decrypt_t::start(uint8_t **key_chain, size_t chain_size, uint8_t *key_indexes, uint64_t start_offset, size_t block_size)
{
assert(!decrypted);
this->start_offset = start_offset;
this->key_chain = chain_size > 1 ? key_chain : 0;
this->chain_size = chain_size > 1 ? chain_size : 0;
this->key_indexes = chain_size > 1 ? key_indexes : NULL;
assert(chain_size <= 1 || key_indexes != NULL);
this->block_size = block_size;
this->offset = 0;
this->tmp_pos = 0;
if (tmp && tmp_size != block_size)
{
free(tmp);
tmp = NULL;
tmp_size = 0;
}
#ifdef WITH_OPENSSL
if (chain_size == 1 && key_chain[0] && EVP_DecryptInit_ex(ctx, NULL, NULL, key_chain[0], NULL) != 1)
{
ERR_print_errors_fp(stderr);
abort();
}
#endif
}
void op_aes_xts_decrypt_t::decrypt_block(uint8_t *in, uint8_t *out)
{
uint8_t *key = NULL;
if (chain_size)
{
assert(key_indexes[offset/block_size] < chain_size);
key = key_chain[key_indexes[offset/block_size]];
if (!key)
{
if (in != out)
memcpy(out, in, block_size);
return;
}
}
#ifdef WITH_OPENSSL
uint8_t iv[16] = { 0 };
*((uint64_t*)iv) = start_offset + offset - offset%block_size;
if (EVP_DecryptInit_ex(ctx, NULL, NULL, key, iv) != 1)
{
ERR_print_errors_fp(stderr);
abort();
}
int actual_out = 0;
if (EVP_DecryptUpdate(ctx, out, &actual_out, in, block_size) != 1)
{
ERR_print_errors_fp(stderr);
abort();
}
assert(actual_out == block_size);
#endif
}
// out may be NULL, in this case all input is still decrypted to calculate checksums,
// but part of it is skipped and not copied to out
void op_aes_xts_decrypt_t::update(uint8_t *in, size_t max_in, uint8_t *out, size_t max_out, size_t & done_in, size_t & done_out)
{
// Fucking AES-XTS implementations (all of them) don't have streaming support,
// crafting IV to resume decryption is slow, so we have to accumulate a full block
// and decrypt it at once :-(
// And then we have to support consuming it in parts because clients sometimes need
// fragmented output.
if (decrypted)
{
// Copy accumulated and decrypted output
assert(tmp);
if (max_out > block_size - tmp_pos)
max_out = block_size - tmp_pos;
if (out)
memcpy(out, tmp + tmp_pos, max_out);
done_out += max_out;
tmp_pos += max_out;
if (tmp_pos >= block_size)
decrypted = false;
}
else if (max_in < block_size - offset%block_size)
{
// Just accumulate input
if (!tmp)
{
tmp = (uint8_t*)malloc_or_die(block_size);
tmp_size = block_size;
}
memcpy(tmp + offset%block_size, in, max_in);
done_in += max_in;
offset += max_in;
}
else if (max_out < block_size || !out)
{
// Accumulate and decrypt input in <tmp>, then copy part of it to <out>
if (!tmp)
{
tmp = (uint8_t*)malloc_or_die(block_size);
tmp_size = block_size;
}
max_in = block_size - offset%block_size;
memcpy(tmp + offset%block_size, in, max_in);
decrypt_block(tmp, tmp);
decrypted = true;
if (out)
memcpy(out, tmp, max_out);
tmp_pos = max_out;
done_in += max_in;
offset += max_in;
done_out += max_out;
}
else if (!(offset%block_size))
{
// Full block - simplest case
if (out)
decrypt_block(in, out);
done_in += block_size;
offset += block_size;
done_out += block_size;
}
else
{
// Accumulate input and decrypt directly to <output>
assert(tmp);
max_in = block_size - offset%block_size;
memcpy(tmp + offset%block_size, in, max_in);
assert(out);
decrypt_block(tmp, out);
done_in += max_in;
offset += max_in;
done_out += block_size;
}
}
void destroy_aes_xts_decrypt(op_aes_xts_decrypt_t *decrypt_ctx)
{
delete decrypt_ctx;
}
bool osd_messenger_t::op_encrypted_copy_data_to(osd_client_t* cl, uint8_t *enc_buf, size_t enc_len, size_t from, size_t & done)
{
auto op = cl->write_op;
auto & op_pos = cl->write_op_pos;
assert(op->req.hdr.opcode == OSD_OP_WRITE);
if (!from)
{
if (!cl->encrypt_ctx)
{
if (encrypt_ctx_pool.size())
{
cl->encrypt_ctx = encrypt_ctx_pool.back();
encrypt_ctx_pool.pop_back();
}
else
cl->encrypt_ctx = new op_aes_xts_encrypt_t();
}
assert(op->enc->key_chain[0]);
cl->encrypt_ctx->start(op->enc->key_chain[0], op->req.rw.offset, op->enc->bitmap_granularity);
}
for (int i = 0; i < op->iov.count; i++)
{
uint8_t *plain = (uint8_t*)op->iov.buf[i].iov_base;
size_t plain_len = op->iov.buf[i].iov_len;
while (from < plain_len || cl->encrypt_ctx->has_buffered())
{
if (done >= enc_len)
return false;
size_t done_in = 0;
size_t done_out = 0;
cl->encrypt_ctx->update(plain+from, plain_len-from, enc_buf+done, enc_len-done, done_in, done_out);
if (cl->write_csum_state && done_out > 0)
XXH3_64bits_update(cl->write_csum_state, enc_buf+done, done_out);
done += done_out;
op_pos += done_in;
from += done_in;
}
from -= plain_len;
}
if (cl->encrypt_ctx)
{
if (encrypt_ctx_pool.size() > max_aes_xts_pool_size)
delete cl->encrypt_ctx;
else
encrypt_ctx_pool.push_back(cl->encrypt_ctx);
cl->encrypt_ctx = NULL;
}
return true;
}
bool osd_messenger_t::op_decrypted_copy_data_from(osd_client_t* cl, uint8_t *enc_buf, size_t enc_len, size_t from, size_t & done)
{
op_decrypt_start(cl);
auto op = cl->read_op;
assert(op->req.hdr.opcode == OSD_OP_READ);
for (int i = 0; i < op->iov.count; i++)
{
uint8_t *plain = (uint8_t*)op->iov.buf[i].iov_base;
size_t plain_len = op->iov.buf[i].iov_len;
while (from < plain_len)
{
if (done >= enc_len)
return false;
size_t done_in = 0;
size_t done_out = 0;
// plain == NULL means skip output
cl->decrypt_ctx->update(enc_buf+done, enc_len-done, plain ? plain+from : NULL, plain_len-from, done_in, done_out);
if (cl->read_csum_state && done_in > 0)
XXH3_64bits_update(cl->read_csum_state, enc_buf+done, done_in);
done += done_in;
cl->read_op_pos += done_out;
cl->read_op_inline_decrypt_in += done_in;
from += done_out;
if (!done_out)
return false;
}
from -= plain_len;
}
op_decrypt_free(cl);
return true;
}
void osd_messenger_t::op_decrypt_start(osd_client_t* cl)
{
if (!cl->decrypt_ctx)
{
if (decrypt_ctx_pool.size())
{
cl->decrypt_ctx = decrypt_ctx_pool.back();
decrypt_ctx_pool.pop_back();
}
else
cl->decrypt_ctx = new op_aes_xts_decrypt_t();
auto & enc = cl->read_op->enc;
cl->decrypt_ctx->start(enc->key_chain, enc->chain_size,
(cl->read_op->req.rw.flags & OSD_OP_RETURN_CHAIN) ? (uint8_t*)cl->read_op->bitmap + enc->read_chain_bitmap_pos : 0,
cl->read_op->req.rw.offset, enc->bitmap_granularity);
}
}
void osd_messenger_t::op_decrypt_inline(osd_client_t* cl)
{
op_decrypt_start(cl);
osd_op_t *op = cl->read_op;
size_t from_in = cl->read_op_inline_decrypt_in;
int i = 0;
while (i < op->iov.count && from_in >= op->iov.buf[i].iov_len)
{
from_in -= op->iov.buf[i].iov_len;
i++;
}
size_t from_out = cl->read_op_inline_decrypt_pos - OSD_PACKET_SIZE - op->reply.rw.bitmap_len;
int j = 0;
while (j < op->iov.count && from_out >= op->iov.buf[j].iov_len)
{
from_out -= op->iov.buf[j].iov_len;
j++;
}
while (i < op->iov.count && j < op->iov.count)
{
uint8_t *in = (uint8_t*)op->iov.buf[i].iov_base + from_in;
size_t in_len = op->iov.buf[i].iov_len - from_in;
uint8_t *out = (uint8_t*)op->iov.buf[j].iov_base + from_out;
size_t out_len = op->iov.buf[j].iov_len - from_out;
size_t done_in = 0;
size_t done_out = 0;
cl->decrypt_ctx->update(in, in_len, out, out_len, done_in, done_out);
if (done_in >= in_len)
{
i++;
from_in = 0;
}
else
from_in += done_in;
if (done_out >= out_len)
{
j++;
from_out = 0;
}
else
from_out += done_out;
}
assert(j >= op->iov.count);
op_decrypt_free(cl);
}
void osd_messenger_t::op_decrypt_free(osd_client_t* cl)
{
if (cl->decrypt_ctx)
{
if (decrypt_ctx_pool.size() > max_aes_xts_pool_size)
delete cl->decrypt_ctx;
else
decrypt_ctx_pool.push_back(cl->decrypt_ctx);
cl->decrypt_ctx = NULL;
}
}
-68
View File
@@ -1,68 +0,0 @@
// Copyright (c) Vitaliy Filippov, 2026+
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
#include <stdint.h>
#include "../util/xxh_x86dispatch.h"
// WITH_OPENSSL is left to possibly support other crypto libraries
#ifdef WITH_OPENSSL
#include <openssl/conf.h>
#include <openssl/evp.h>
#include <openssl/err.h>
#endif
class op_aes_xts_encrypt_t
{
#ifdef WITH_OPENSSL
EVP_CIPHER_CTX *ctx = NULL;
#endif
uint64_t start_offset = 0;
uint8_t *key = NULL;
size_t offset = 0;
size_t block_size = 0;
uint8_t *tmp = NULL;
size_t tmp_size = 0;
size_t tmp_pos = 0;
bool encrypted = false;
void encrypt_block(uint8_t *in, uint8_t *out);
public:
op_aes_xts_encrypt_t();
~op_aes_xts_encrypt_t();
inline bool has_buffered() { return encrypted; };
void start(uint8_t *key, uint64_t start_offset, size_t block_size);
void update(uint8_t *in, size_t max_in, uint8_t *out, size_t max_out, size_t & done_in, size_t & done_out);
};
void destroy_aes_xts_encrypt(op_aes_xts_encrypt_t *encrypt_ctx);
class op_aes_xts_decrypt_t
{
#ifdef WITH_OPENSSL
EVP_CIPHER_CTX *ctx = NULL;
#endif
uint64_t start_offset = 0;
uint8_t **key_chain = NULL;
size_t chain_size = 0;
uint8_t *key_indexes = NULL;
size_t offset = 0;
size_t block_size = 0;
uint8_t *tmp = NULL;
size_t tmp_size = 0;
size_t tmp_pos = 0;
bool decrypted = false;
void decrypt_block(uint8_t *in, uint8_t *out);
public:
op_aes_xts_decrypt_t();
~op_aes_xts_decrypt_t();
inline bool has_buffered() { return decrypted; };
void start(uint8_t **key_chain, size_t chain_size, uint8_t *key_indexes, uint64_t start_offset, size_t block_size);
void update(uint8_t *in, size_t max_in, uint8_t *out, size_t max_out, size_t & done_in, size_t & done_out);
};
void destroy_aes_xts_decrypt(op_aes_xts_decrypt_t *decrypt_ctx);
+1 -8
View File
@@ -8,6 +8,7 @@
osd_op_t::~osd_op_t()
{
assert(!bs_op);
assert(!op_data);
if (bitmap_buf)
{
free(bitmap_buf);
@@ -22,14 +23,6 @@ osd_op_t::~osd_op_t()
// So we don't reuse it, but free it every time
free(buf);
}
if (enc_buf)
{
free(enc_buf);
}
if (op_data)
{
free(op_data);
}
}
bool osd_op_t::is_recovery_related()
-21
View File
@@ -3,8 +3,6 @@
#pragma once
#include <memory>
#include <sys/uio.h>
#include <stdint.h>
#include <stdio.h>
@@ -18,8 +16,6 @@
#define OSD_OP_INLINE_BUF_COUNT 16
#define AES_256_XTS_KEY_SIZE 64
// Kind of a vector with small-list-optimisation
struct osd_op_buf_list_t
{
@@ -156,19 +152,6 @@ struct blockstore_op_t;
struct osd_primary_op_data_t;
struct osd_op_enc_t
{
// Keys may contain more information in the future, like encryption algorithm and key ID
// In this case, key_chain will become inode_key_t* with inode_key_t also being a structure
// Currently all keys are required to be 512 bit (64 byte) long, for AES-256-XTS
// Raw pointers are convenient for messenger code; external users may use shared_ptr aliasing
// to implement complex freeing of osd_op_enc_t along with their external inode cache info
uint8_t** key_chain = NULL;
size_t chain_size = 0;
uint32_t read_chain_bitmap_pos = 0;
uint32_t bitmap_granularity = 0;
};
struct __attribute__((visibility("default"))) osd_op_t
{
timespec tv_begin = { 0 }, tv_end = { 0 };
@@ -184,9 +167,6 @@ struct __attribute__((visibility("default"))) osd_op_t
unsigned bmp_data = 0;
void *bitmap_buf = NULL;
void *rmw_buf = NULL;
std::shared_ptr<osd_op_enc_t> enc;
uint8_t *enc_buf = NULL;
uint64_t csum = 0; // network layer checksum
osd_primary_op_data_t* op_data = NULL;
std::function<void(osd_op_t*)> callback;
@@ -196,5 +176,4 @@ struct __attribute__((visibility("default"))) osd_op_t
void cancel();
bool is_recovery_related();
uint64_t calc_data_checksum();
};
+53 -25
View File
@@ -568,24 +568,23 @@ static void try_send_rdma_wr(osd_client_t *cl, ibv_sge *sge, int op_sge)
cl->rdma_conn->cur_send++;
}
int osd_messenger_t::try_send_rdma_copy(osd_client_t *cl, uint8_t *dst, int dst_len)
static int try_send_rdma_copy(osd_client_t *cl, uint8_t *dst, int dst_len)
{
auto rc = cl->rdma_conn;
int total_dst_len = dst_len;
while (dst_len > 0 && (cl->write_op || cl->write_ops.size()))
while (dst_len > 0 && rc->send_pos < cl->send_list.size())
{
next_write_op(cl);
osd_op_t *op = cl->write_op;
size_t copied = op_copy_to(cl, dst, dst_len);
if (!copied)
iovec & iov = cl->send_list[rc->send_pos];
uint32_t len = (uint32_t)(iov.iov_len-rc->send_buf_pos < dst_len
? iov.iov_len-rc->send_buf_pos : dst_len);
memcpy(dst, (uint8_t*)iov.iov_base+rc->send_buf_pos, len);
dst += len;
dst_len -= len;
rc->send_buf_pos += len;
if (rc->send_buf_pos >= iov.iov_len)
{
break;
}
dst += copied;
dst_len -= copied;
if (!cl->write_op && op->op_type == OSD_OP_IN)
{
// this is a reply, free the op after sending it
cl->send_free_ops.push_back(op);
rc->send_pos++;
rc->send_buf_pos = 0;
}
}
return total_dst_len-dst_len;
@@ -632,7 +631,6 @@ void osd_messenger_t::try_send_rdma(osd_client_t *cl)
};
try_send_rdma_wr(cl, &sge, 1);
rc->send_sizes.push_back(copied);
cl->send_free_ops.push_back(NULL); // end marker
}
}
}
@@ -729,6 +727,9 @@ void osd_messenger_t::handle_rdma_events(msgr_rdma_context_t *rdma_context)
}
if (!is_send)
{
// Reset OSD ping state - client is obviously alive
cl->ping_time_remaining = 0;
cl->idle_time_remaining = osd_idle_timeout;
rc->cur_recv--;
if (!handle_read_buffer(cl, rc->recv_buffers[rc->next_recv_buf], wc[i].byte_len))
{
@@ -740,27 +741,54 @@ void osd_messenger_t::handle_rdma_events(msgr_rdma_context_t *rdma_context)
else
{
rc->cur_send--;
// byte_len is not filled for send operations
uint64_t sent_size = rc->send_sizes.front();
rc->send_sizes.pop_front();
uint64_t sent_size = rc->send_sizes.at(0);
rc->send_sizes.erase(rc->send_sizes.begin(), rc->send_sizes.begin()+1);
rc->send_done_pos += sent_size;
rc->send_out_full = false;
if (rc->send_done_pos == rc->send_out_size)
rc->send_done_pos = 0;
assert(rc->send_done_pos < rc->send_out_size);
while (cl->send_free_ops.front())
int send_pos = 0, send_buf_pos = 0;
while (sent_size > 0)
{
delete cl->send_free_ops.front();
cl->send_free_ops.pop_front();
if (sent_size >= cl->send_list.at(send_pos).iov_len)
{
sent_size -= cl->send_list[send_pos].iov_len;
send_pos++;
}
else
{
send_buf_pos = sent_size;
sent_size = 0;
}
}
cl->send_free_ops.pop_front();
if ((cl->proto_csum_status & MSGR_CSUM_NEG) && !cl->write_op && !cl->write_ops.size())
assert(rc->send_pos >= send_pos);
if (rc->send_pos == send_pos)
{
// Checksums negotiated, enable
cl->proto_csum_status = cl->proto_csum_status & (~MSGR_CSUM_NEG);
rc->send_buf_pos -= send_buf_pos;
}
rc->send_pos -= send_pos;
for (int i = 0; i < send_pos; i++)
{
if (cl->outbox[i].flags & MSGR_SENDP_FREE)
{
// Reply fully sent
delete cl->outbox[i].op;
}
}
if (send_pos > 0)
{
cl->send_list.erase(cl->send_list.begin(), cl->send_list.begin()+send_pos);
cl->outbox.erase(cl->outbox.begin(), cl->outbox.begin()+send_pos);
}
if (send_buf_pos > 0)
{
cl->send_list[0].iov_base = (uint8_t*)cl->send_list[0].iov_base + send_buf_pos;
cl->send_list[0].iov_len -= send_buf_pos;
}
try_send_rdma(cl);
}
}
} while (event_count > 0);
handle_immediate_ops();
}
+2 -5
View File
@@ -8,11 +8,8 @@
#include <infiniband/verbs.h>
#include <string>
#include <vector>
#include <deque>
#include "addr_util.h"
struct osd_op_t;
struct msgr_rdma_address_t
{
ibv_gid gid;
@@ -75,9 +72,9 @@ struct msgr_rdma_connection_t
int cur_send = 0, cur_recv = 0;
int send_pos = 0, send_buf_pos = 0;
int next_recv_buf = 0;
std::vector<uint8_t*> recv_buffers;
std::vector<void*> recv_buffers;
msgr_rdma_buf_t recv_buf;
std::deque<uint64_t> send_sizes;
std::vector<uint64_t> send_sizes;
msgr_rdma_buf_t send_out;
int send_out_pos = 0, send_done_pos = 0, send_out_size = 0;
bool send_out_full = false;
+1 -1
View File
@@ -495,7 +495,7 @@ void osd_messenger_t::rdmacm_established(rdma_cm_event *ev)
cl->peer_state = PEER_RDMA;
cl->connect_timeout_id = -1;
cl->osd_num = peer_osd;
cl->in_buf = (uint8_t*)malloc_or_die(receive_buffer_size);
cl->in_buf = malloc_or_die(receive_buffer_size);
cl->rdma_conn = rc;
clients[conn->client_id] = cl;
if (conn->timeout_id >= 0)
+205 -466
View File
@@ -1,8 +1,6 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
#define _XOPEN_SOURCE
#include <limits.h>
#include "messenger.h"
void osd_messenger_t::read_requests()
@@ -17,11 +15,7 @@ void osd_messenger_t::read_requests()
continue;
}
auto cl = cl_it->second;
if (cl->read_op && cl->read_op_size-(cl->read_op_pos-OSD_PACKET_SIZE) >= receive_buffer_size)
{
op_get_read_buffers(cl, cl->recv_list);
}
if (!cl->recv_list.size())
if (cl->read_remaining < receive_buffer_size)
{
cl->read_iov.iov_base = cl->in_buf;
cl->read_iov.iov_len = receive_buffer_size;
@@ -31,11 +25,10 @@ void osd_messenger_t::read_requests()
else
{
cl->read_iov.iov_base = 0;
cl->read_iov.iov_len = 0;
cl->read_msg.msg_iov = cl->recv_list.data();
cl->read_msg.msg_iovlen = cl->recv_list.size();
cl->read_iov.iov_len = cl->read_remaining;
cl->read_msg.msg_iov = cl->recv_list.get_iovec();
cl->read_msg.msg_iovlen = cl->recv_list.get_size();
}
assert(!cl->read_op || cl->read_op_pos < OSD_PACKET_SIZE || cl->read_op_size >= (cl->read_op_pos-OSD_PACKET_SIZE));
cl->refs++;
if (ringloop && !use_sync_send_recv)
{
@@ -57,7 +50,7 @@ void osd_messenger_t::read_requests()
}
ring_data_t* data = ((ring_data_t*)sqe->user_data);
data->callback = [this, cl](ring_data_t *data) { handle_read(data->res, cl); };
io_uring_prep_recvmsg(sqe, cl->peer_fd, &cl->read_msg, cl->recv_list.size() ? MSG_WAITALL : 0);
io_uring_prep_recvmsg(sqe, cl->peer_fd, &cl->read_msg, 0);
if (iothread)
{
iothread->add_sqe(sqe_local);
@@ -75,15 +68,16 @@ void osd_messenger_t::read_requests()
}
}
read_ready_clients.clear();
handle_immediate_ops();
}
void osd_messenger_t::handle_read(int result, osd_client_t *cl)
bool osd_messenger_t::handle_read(int result, osd_client_t *cl)
{
bool ret = false;
cl->read_msg.msg_iovlen = 0;
cl->refs--;
if (cl->peer_state == PEER_RDMA)
{
return;
return true;
}
if (cl->peer_state == PEER_STOPPED)
{
@@ -91,7 +85,7 @@ void osd_messenger_t::handle_read(int result, osd_client_t *cl)
{
destroy_client(cl);
}
return;
return false;
}
if (result <= 0 && result != -EAGAIN && result != -EINTR)
{
@@ -101,56 +95,9 @@ void osd_messenger_t::handle_read(int result, osd_client_t *cl)
fprintf(stderr, "Client %ju socket read error: %d (%s). Disconnecting client\n", cl->client_id, -result, strerror(-result));
}
stop_client(cl->client_id);
out_wakeup:
if (set_immediate_ops.size())
ringloop->wakeup();
return;
return false;
}
bool full_read = false;
if (result > 0)
{
if (cl->read_iov.iov_base == cl->in_buf)
{
full_read = result >= cl->read_iov.iov_len;
if (!handle_read_buffer(cl, cl->in_buf, result))
goto out_wakeup;
}
else
{
// Reset OSD ping state
cl->ping_time_remaining = 0;
cl->idle_time_remaining = osd_idle_timeout;
// Long data
size_t i = 0;
while (i < cl->recv_list.size() && result >= cl->recv_list[i].iov_len)
{
if (cl->read_csum_state && cl->recv_list[i].iov_len > 0 &&
i != cl->recv_list.size()-1) // skip the checksum itself
{
XXH3_64bits_update(cl->read_csum_state, cl->recv_list[i].iov_base, cl->recv_list[i].iov_len);
}
result -= cl->recv_list[i].iov_len;
i++;
}
if (i < cl->recv_list.size())
{
cl->recv_list[i].iov_base += result;
cl->recv_list[i].iov_len -= result;
}
else
{
full_read = true;
}
cl->recv_list.erase(cl->recv_list.begin(), cl->recv_list.begin()+i);
if (!cl->recv_list.size())
{
if (!handle_finished_op(cl))
goto out_wakeup;
}
}
}
cl->read_msg.msg_iovlen = 0;
if (result == -EAGAIN || result == -EINTR || !full_read)
if (result == -EAGAIN || result == -EINTR || result < cl->read_iov.iov_len)
{
cl->read_ready--;
if (cl->read_ready > 0)
@@ -160,7 +107,37 @@ out_wakeup:
{
read_ready_clients.push_back(cl->client_id);
}
goto out_wakeup;
if (result > 0)
{
if (cl->read_iov.iov_base == cl->in_buf)
{
if (!handle_read_buffer(cl, cl->in_buf, result))
{
handle_immediate_ops();
return false;
}
}
else
{
// Long data
cl->read_remaining -= result;
cl->recv_list.eat(result);
if (cl->recv_list.done >= cl->recv_list.count)
{
if (!handle_finished_read(cl))
{
handle_immediate_ops();
return false;
}
}
}
if (result >= cl->read_iov.iov_len)
{
ret = true;
}
}
handle_immediate_ops();
return ret;
}
void osd_messenger_t::handle_immediate_ops()
@@ -185,109 +162,113 @@ void osd_messenger_t::handle_immediate_ops()
}
}
bool osd_messenger_t::handle_read_buffer(osd_client_t *cl, uint8_t *curbuf, size_t bufsize)
bool osd_messenger_t::handle_read_buffer(osd_client_t *cl, void *curbuf, int remain)
{
// Reset OSD ping state
cl->ping_time_remaining = 0;
cl->idle_time_remaining = osd_idle_timeout;
// Compose operation(s) from the buffer
size_t done = 0;
while (done < bufsize)
while (remain > 0)
{
if (!cl->read_op)
{
cl->read_op = new osd_op_t;
cl->read_op->client_id = cl->client_id;
cl->read_op->op_type = OSD_OP_IN;
cl->read_op_pos = 0;
cl->read_op_size = 0;
cl->read_op_inline_decrypt_in = 0;
cl->read_op_inline_decrypt_pos = (size_t)-1;
if (cl->proto_csum_status == MSGR_CSUM_FULL || cl->proto_csum_status == MSGR_CSUM_PAYLOAD)
cl->recv_list.push_back(cl->read_op->req.buf, OSD_PACKET_SIZE);
cl->read_remaining = OSD_PACKET_SIZE;
cl->read_state = CL_READ_HDR;
}
while (cl->recv_list.done < cl->recv_list.count && remain > 0)
{
iovec* cur = cl->recv_list.get_iovec();
if (cur->iov_len > remain)
{
if (!cl->read_csum_state)
cl->read_csum_state = XXH3_createState();
XXH3_64bits_reset(cl->read_csum_state);
memcpy(cur->iov_base, curbuf, remain);
cl->read_remaining -= remain;
cur->iov_len -= remain;
cur->iov_base = (uint8_t*)cur->iov_base + remain;
remain = 0;
}
else
{
memcpy(cur->iov_base, curbuf, cur->iov_len);
curbuf = (uint8_t*)curbuf + cur->iov_len;
cl->read_remaining -= cur->iov_len;
remain -= cur->iov_len;
cur->iov_len = 0;
cl->recv_list.done++;
}
}
if (cl->read_op_pos < OSD_PACKET_SIZE)
if (cl->recv_list.done >= cl->recv_list.count)
{
int len = OSD_PACKET_SIZE - cl->read_op_pos;
if (len > bufsize-done)
len = bufsize-done;
memcpy(cl->read_op->req.buf + cl->read_op_pos, curbuf+done, len);
done += len;
cl->read_op_pos += len;
if (cl->read_op_pos < OSD_PACKET_SIZE)
return true;
if (!handle_hdr(cl))
if (!handle_finished_read(cl))
{
stop_client(cl->client_id);
return false;
}
}
if (!op_copy_from(cl, curbuf, bufsize, done))
{
return false;
}
}
return true;
}
bool osd_messenger_t::handle_hdr(osd_client_t *cl)
bool osd_messenger_t::handle_finished_read(osd_client_t *cl)
{
if (cl->proto_csum_status == MSGR_CSUM_FULL)
// Reset OSD ping state
cl->ping_time_remaining = 0;
cl->idle_time_remaining = osd_idle_timeout;
cl->recv_list.reset();
if (cl->read_state == CL_READ_HDR)
{
XXH3_64bits_update(cl->read_csum_state, cl->read_op->req.buf, OSD_PACKET_SIZE);
}
if (cl->read_op->req.hdr.magic == SECONDARY_OSD_REPLY_MAGIC)
{
auto req_it = cl->sent_ops.find(cl->read_op->req.hdr.id);
if (req_it == cl->sent_ops.end())
if (cl->read_op->req.hdr.magic == SECONDARY_OSD_REPLY_MAGIC)
return handle_reply_hdr(cl);
else if (cl->read_op->req.hdr.magic == SECONDARY_OSD_OP_MAGIC)
{
// Command out of sync. Drop connection
fprintf(stderr, "Client %ju command out of sync: id %ju\n", cl->client_id, cl->read_op->req.hdr.id);
return false;
}
osd_op_t *op = req_it->second;
memcpy(op->reply.buf, cl->read_op->req.buf, OSD_PACKET_SIZE);
if (!allocate_reply_buffers(cl, op))
{
return false;
}
cl->sent_ops.erase(req_it);
delete cl->read_op;
cl->read_op = op;
}
else if (cl->read_op->req.hdr.magic == SECONDARY_OSD_OP_MAGIC)
{
if (cl->check_sequencing)
{
if (cl->read_op->req.hdr.id != cl->read_op_id)
if (cl->check_sequencing)
{
fprintf(stderr, "Warning: operation sequencing is broken on client %d: expected num %ju, got %ju, stopping client\n", cl->peer_fd, cl->read_op_id, cl->read_op->req.hdr.id);
return false;
if (cl->read_op->req.hdr.id != cl->read_op_id)
{
fprintf(stderr, "Warning: operation sequencing is broken on client %ju: expected num %ju, got %ju, stopping client\n", cl->client_id, cl->read_op_id, cl->read_op->req.hdr.id);
stop_client(cl->client_id);
return false;
}
cl->read_op_id++;
}
cl->read_op_id++;
handle_op_hdr(cl);
}
if (!allocate_op_buffers(cl))
else
{
fprintf(stderr, "Received garbage: magic=%jx id=%ju opcode=%jx from client %ju\n", cl->read_op->req.hdr.magic, cl->read_op->req.hdr.id, cl->read_op->req.hdr.opcode, cl->client_id);
stop_client(cl->client_id);
return false;
}
}
else if (cl->read_state == CL_READ_DATA)
{
// Operation is ready
cl->received_ops.push_back(cl->read_op);
set_immediate_ops.push_back(cl->read_op);
cl->read_op = NULL;
cl->read_state = 0;
}
else if (cl->read_state == CL_READ_REPLY_DATA)
{
// Reply is ready
handle_reply_ready(cl->read_op);
cl->read_op = NULL;
cl->read_state = 0;
}
else
{
fprintf(stderr, "Received garbage: magic=%jx id=%ju opcode=%jx from client %ju\n", cl->read_op->req.hdr.magic, cl->read_op->req.hdr.id, cl->read_op->req.hdr.opcode, cl->client_id);
return false;
assert(0);
}
return true;
}
bool osd_messenger_t::allocate_op_buffers(osd_client_t *cl)
void osd_messenger_t::handle_op_hdr(osd_client_t *cl)
{
osd_op_t *cur_op = cl->read_op;
cl->read_op_size = 0;
if (cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ)
{
cl->read_remaining = 0;
}
else if (cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE)
{
if (cur_op->req.sec_rw.attr_len > 0)
@@ -296,12 +277,14 @@ bool osd_messenger_t::allocate_op_buffers(osd_client_t *cl)
cur_op->bitmap = cur_op->rmw_buf = malloc_or_die(cur_op->req.sec_rw.attr_len);
else
cur_op->bitmap = &cur_op->bmp_data;
cl->recv_list.push_back(cur_op->bitmap, cur_op->req.sec_rw.attr_len);
}
if (cur_op->req.sec_rw.len > 0)
{
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.sec_rw.len);
cl->recv_list.push_back(cur_op->buf, cur_op->req.sec_rw.len);
}
cl->read_op_size = cur_op->req.sec_rw.len + cur_op->req.sec_rw.attr_len;
cl->read_remaining = cur_op->req.sec_rw.len + cur_op->req.sec_rw.attr_len;
}
else if (cur_op->req.hdr.opcode == OSD_OP_SEC_STABILIZE ||
cur_op->req.hdr.opcode == OSD_OP_SEC_ROLLBACK)
@@ -309,24 +292,27 @@ bool osd_messenger_t::allocate_op_buffers(osd_client_t *cl)
if (cur_op->req.sec_stab.len > 0)
{
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.sec_stab.len);
cl->recv_list.push_back(cur_op->buf, cur_op->req.sec_stab.len);
}
cl->read_op_size = cur_op->req.sec_stab.len;
cl->read_remaining = cur_op->req.sec_stab.len;
}
else if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ_BMP)
{
if (cur_op->req.sec_read_bmp.len > 0)
{
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.sec_read_bmp.len);
cl->recv_list.push_back(cur_op->buf, cur_op->req.sec_read_bmp.len);
}
cl->read_op_size = cur_op->req.sec_read_bmp.len;
cl->read_remaining = cur_op->req.sec_read_bmp.len;
}
else if (cur_op->req.hdr.opcode == OSD_OP_WRITE)
{
if (cur_op->req.rw.len > 0)
{
cur_op->buf = memalign_or_die(MEM_ALIGNMENT, cur_op->req.rw.len);
cl->recv_list.push_back(cur_op->buf, cur_op->req.rw.len);
}
cl->read_op_size = cur_op->req.rw.len;
cl->read_remaining = cur_op->req.rw.len;
}
else if (cur_op->req.hdr.opcode == OSD_OP_SHOW_CONFIG)
{
@@ -334,20 +320,44 @@ bool osd_messenger_t::allocate_op_buffers(osd_client_t *cl)
{
cur_op->buf = malloc_or_die(cur_op->req.show_conf.json_len+1);
((uint8_t*)cur_op->buf)[cur_op->req.show_conf.json_len] = 0;
cl->recv_list.push_back(cur_op->buf, cur_op->req.show_conf.json_len);
}
cl->read_op_size = cur_op->req.show_conf.json_len;
cl->read_remaining = cur_op->req.show_conf.json_len;
}
if (cl->proto_csum_status == MSGR_CSUM_FULL ||
cl->read_op_size > 0 && cl->proto_csum_status == MSGR_CSUM_PAYLOAD)
/*else if (cur_op->req.hdr.opcode == OSD_OP_READ ||
cur_op->req.hdr.opcode == OSD_OP_SCRUB ||
cur_op->req.hdr.opcode == OSD_OP_DESCRIBE)
{
cl->read_op_size += 8;
cl->read_remaining = 0;
}*/
if (cl->read_remaining > 0)
{
// Read data
cl->read_state = CL_READ_DATA;
}
else
{
// Operation is ready
cl->received_ops.push_back(cur_op);
set_immediate_ops.push_back(cur_op);
cl->read_op = NULL;
cl->read_state = 0;
}
return true;
}
bool osd_messenger_t::allocate_reply_buffers(osd_client_t *cl, osd_op_t *op)
bool osd_messenger_t::handle_reply_hdr(osd_client_t *cl)
{
cl->read_op_size = 0;
auto req_it = cl->sent_ops.find(cl->read_op->req.hdr.id);
if (req_it == cl->sent_ops.end())
{
// Command out of sync. Drop connection
fprintf(stderr, "Client %ju command out of sync: id %ju\n", cl->client_id, cl->read_op->req.hdr.id);
stop_client(cl->client_id);
return false;
}
osd_op_t *op = req_it->second;
memcpy(op->reply.buf, cl->read_op->req.buf, OSD_PACKET_SIZE);
cl->sent_ops.erase(req_it);
if (op->reply.hdr.opcode == OSD_OP_SEC_READ || op->reply.hdr.opcode == OSD_OP_READ)
{
// Read data. In this case we assume that the buffer is preallocated by the caller (!)
@@ -358,368 +368,97 @@ bool osd_messenger_t::allocate_reply_buffers(osd_client_t *cl, osd_op_t *op)
// Check reply length to not overflow the buffer
fprintf(stderr, "Client %ju read reply of different length: expected %u+%u, got %jd+%u\n",
cl->client_id, expected_size, op->bitmap_len, op->reply.hdr.retval, bmp_len);
cl->sent_ops[op->req.hdr.id] = op;
stop_client(cl->client_id);
return false;
}
if (bmp_len > 0)
{
assert(op->bitmap);
cl->read_op_size += bmp_len;
cl->recv_list.push_back(op->bitmap, bmp_len);
cl->read_remaining += bmp_len;
}
if (op->reply.hdr.retval > 0)
{
assert(op->iov.count > 0);
cl->read_op_size += op->reply.hdr.retval;
cl->recv_list.append(op->iov);
cl->read_remaining += op->reply.hdr.retval;
}
if (cl->read_remaining == 0)
{
goto reuse;
}
delete cl->read_op;
cl->read_op = op;
cl->read_state = CL_READ_REPLY_DATA;
}
else if (op->reply.hdr.opcode == OSD_OP_SEC_LIST && op->reply.hdr.retval > 0)
{
assert(!op->iov.count);
cl->read_op_size = sizeof(obj_ver_id) * op->reply.hdr.retval;
op->buf = memalign_or_die(MEM_ALIGNMENT, cl->read_op_size);
delete cl->read_op;
cl->read_op = op;
cl->read_state = CL_READ_REPLY_DATA;
cl->read_remaining = sizeof(obj_ver_id) * op->reply.hdr.retval;
op->buf = memalign_or_die(MEM_ALIGNMENT, cl->read_remaining);
cl->recv_list.push_back(op->buf, cl->read_remaining);
}
else if (op->reply.hdr.opcode == OSD_OP_SEC_READ_BMP && op->reply.hdr.retval > 0)
{
assert(!op->iov.count);
cl->read_op_size = op->reply.hdr.retval;
delete cl->read_op;
cl->read_op = op;
cl->read_state = CL_READ_REPLY_DATA;
cl->read_remaining = op->reply.hdr.retval;
free(op->buf);
op->buf = memalign_or_die(MEM_ALIGNMENT, cl->read_op_size);
op->buf = memalign_or_die(MEM_ALIGNMENT, cl->read_remaining);
cl->recv_list.push_back(op->buf, cl->read_remaining);
}
else if (op->reply.hdr.opcode == OSD_OP_SHOW_CONFIG && op->reply.hdr.retval > 0)
{
cl->read_op_size = op->reply.hdr.retval;
delete cl->read_op;
cl->read_op = op;
cl->read_state = CL_READ_REPLY_DATA;
cl->read_remaining = op->reply.hdr.retval;
free(op->buf);
op->buf = malloc_or_die(op->reply.hdr.retval);
cl->recv_list.push_back(op->buf, op->reply.hdr.retval);
}
else if (op->reply.hdr.opcode == OSD_OP_DESCRIBE && op->reply.describe.result_bytes > 0)
{
cl->read_op_size = op->reply.describe.result_bytes;
delete cl->read_op;
cl->read_op = op;
cl->read_state = CL_READ_REPLY_DATA;
cl->read_remaining = op->reply.describe.result_bytes;
free(op->buf);
op->buf = malloc_or_die(op->reply.describe.result_bytes);
cl->recv_list.push_back(op->buf, op->reply.describe.result_bytes);
}
if (cl->proto_csum_status == MSGR_CSUM_FULL ||
cl->read_op_size > 0 && cl->proto_csum_status == MSGR_CSUM_PAYLOAD)
else
{
cl->read_op_size += 8;
reuse:
// It's fine to reuse cl->read_op for the next reply
handle_reply_ready(op);
cl->recv_list.push_back(cl->read_op->req.buf, OSD_PACKET_SIZE);
cl->read_remaining = OSD_PACKET_SIZE;
cl->read_state = CL_READ_HDR;
}
return true;
}
bool osd_messenger_t::op_copy_from(osd_client_t *cl, uint8_t *src, size_t src_len, size_t & done)
void osd_messenger_t::handle_reply_ready(osd_op_t *op)
{
osd_op_t *op = cl->read_op;
size_t from = cl->read_op_pos-OSD_PACKET_SIZE;
auto op_read_buf = [&](uint8_t *dst, size_t dst_len, bool skip_csum = false)
// Measure subop latency
timespec tv_end;
clock_gettime(CLOCK_REALTIME, &tv_end);
stats.subop_stat_count[op->req.hdr.opcode]++;
if (!stats.subop_stat_count[op->req.hdr.opcode])
{
if (from < dst_len)
{
size_t n = dst_len-from;
if (n > src_len-done)
n = src_len-done;
if (cl->read_csum_state && !skip_csum)
{
// it may be skipped if !dst but checksum is still calculated
XXH3_64bits_update(cl->read_csum_state, src+done, n);
}
if (dst)
memcpy(dst+from, src+done, n);
else
assert(!this->osd_num); // NULL buffers are only used by clients
done += n;
cl->read_op_pos += n;
from += n;
if (from < dst_len)
return false;
from = 0;
}
else
from -= dst_len;
return true;
};
if (op->op_type == OSD_OP_IN)
{
if (op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE)
{
if (!op_read_buf((uint8_t*)op->bitmap, op->req.sec_rw.attr_len))
return true;
if (!op_read_buf((uint8_t*)op->buf, op->req.sec_rw.len))
return true;
}
else if (op->req.hdr.opcode == OSD_OP_SEC_STABILIZE ||
op->req.hdr.opcode == OSD_OP_SEC_ROLLBACK)
{
if (!op_read_buf((uint8_t*)op->buf, op->req.sec_stab.len))
return true;
}
else if (op->req.hdr.opcode == OSD_OP_SEC_READ_BMP)
{
if (!op_read_buf((uint8_t*)op->buf, op->req.sec_read_bmp.len))
return true;
}
else if (op->req.hdr.opcode == OSD_OP_WRITE)
{
if (!op_read_buf((uint8_t*)op->buf, op->req.rw.len))
return true;
}
else if (op->req.hdr.opcode == OSD_OP_SHOW_CONFIG)
{
if (!op_read_buf((uint8_t*)op->buf, op->req.show_conf.json_len))
return true;
}
}
else
{
if (op->reply.hdr.opcode == OSD_OP_SEC_READ)
{
if (op->reply.sec_rw.attr_len > 0)
{
if (!op_read_buf((uint8_t*)op->bitmap, op->reply.sec_rw.attr_len))
return true;
}
if (op->reply.hdr.retval > 0)
{
for (int i = 0; i < op->iov.count; i++)
if (!op_read_buf((uint8_t*)op->iov.buf[i].iov_base, op->iov.buf[i].iov_len))
return true;
}
}
else if (op->reply.hdr.opcode == OSD_OP_READ)
{
if (op->reply.rw.bitmap_len > 0)
{
if (!op_read_buf((uint8_t*)op->bitmap, op->reply.rw.bitmap_len))
return true;
}
if (op->reply.hdr.retval > 0)
{
if (op->enc)
{
if (!op_decrypted_copy_data_from(cl, src, src_len, from, done))
return true;
}
else
{
for (int i = 0; i < op->iov.count; i++)
if (!op_read_buf((uint8_t*)op->iov.buf[i].iov_base, op->iov.buf[i].iov_len))
return true;
}
}
}
else if (op->reply.hdr.opcode == OSD_OP_SEC_LIST && op->reply.hdr.retval > 0)
{
if (!op_read_buf((uint8_t*)op->buf, sizeof(obj_ver_id) * op->reply.hdr.retval))
return true;
}
else if ((op->reply.hdr.opcode == OSD_OP_SEC_READ_BMP ||
op->reply.hdr.opcode == OSD_OP_SHOW_CONFIG) && op->reply.hdr.retval > 0)
{
if (!op_read_buf((uint8_t*)op->buf, op->reply.hdr.retval))
return true;
}
else if (op->reply.hdr.opcode == OSD_OP_DESCRIBE && op->reply.describe.result_bytes > 0)
{
if (!op_read_buf((uint8_t*)op->buf, op->reply.describe.result_bytes))
return true;
}
}
if (cl->proto_csum_status == MSGR_CSUM_FULL ||
cl->read_op_size > 0 && cl->proto_csum_status == MSGR_CSUM_PAYLOAD)
{
if (!op_read_buf((uint8_t*)&op->csum, 8, true))
return true;
}
return handle_finished_op(cl);
}
void osd_messenger_t::op_get_read_buffers(osd_client_t *cl, std::vector<iovec> & lst)
{
osd_op_t *op = cl->read_op;
size_t from = cl->read_op_pos-OSD_PACKET_SIZE;
size_t done = 0;
auto op_read_buf = [&](uint8_t *dst, size_t dst_len)
{
if (lst.size() >= IOV_MAX)
return false;
if (from < dst_len)
{
lst.push_back((iovec){ .iov_base = dst+from, .iov_len = dst_len-from });
cl->read_op_pos += dst_len-from;
done += dst_len-from;
from = 0;
}
else
from -= dst_len;
return true;
};
if (op->op_type == OSD_OP_IN)
{
if (op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE)
{
if (!op_read_buf((uint8_t*)op->bitmap, op->req.sec_rw.attr_len))
return;
if (!op_read_buf((uint8_t*)op->buf, op->req.sec_rw.len))
return;
}
else if (op->req.hdr.opcode == OSD_OP_SEC_STABILIZE ||
op->req.hdr.opcode == OSD_OP_SEC_ROLLBACK)
{
if (!op_read_buf((uint8_t*)op->buf, op->req.sec_stab.len))
return;
}
else if (op->req.hdr.opcode == OSD_OP_SEC_READ_BMP)
{
if (!op_read_buf((uint8_t*)op->buf, op->req.sec_read_bmp.len))
return;
}
else if (op->req.hdr.opcode == OSD_OP_WRITE)
{
if (!op_read_buf((uint8_t*)op->buf, op->req.rw.len))
return;
}
else if (op->req.hdr.opcode == OSD_OP_SHOW_CONFIG)
{
if (!op_read_buf((uint8_t*)op->buf, op->req.show_conf.json_len))
return;
}
}
else
{
if (op->reply.hdr.opcode == OSD_OP_SEC_READ)
{
if (op->reply.sec_rw.attr_len > 0)
{
if (!op_read_buf((uint8_t*)op->bitmap, op->reply.sec_rw.attr_len))
return;
}
if (op->reply.hdr.retval > 0)
{
for (int i = 0; i < op->iov.count; i++)
if (!op_read_buf((uint8_t*)op->iov.buf[i].iov_base, op->iov.buf[i].iov_len))
return;
}
}
else if (op->reply.hdr.opcode == OSD_OP_READ)
{
if (op->reply.rw.bitmap_len > 0)
{
if (!op_read_buf((uint8_t*)op->bitmap, op->reply.rw.bitmap_len))
return;
}
if (op->reply.hdr.retval > 0)
{
if (op->enc)
{
cl->read_op_inline_decrypt_pos = cl->read_op_pos;
cl->read_op_pos = cl->read_op_inline_decrypt_in + OSD_PACKET_SIZE + op->reply.rw.bitmap_len;
from = cl->read_op_inline_decrypt_in;
}
for (int i = 0; i < op->iov.count; i++)
{
if (!op->iov.buf[i].iov_base)
{
// When we recvmsg directly into the operation without copying,
// we need some place for all buffers, so we allocate temporary
// buffers for all skipped parts
op_alloc_temp_buffers(op, i);
}
if (!op_read_buf((uint8_t*)op->iov.buf[i].iov_base, op->iov.buf[i].iov_len))
return;
}
}
}
else if (op->reply.hdr.opcode == OSD_OP_SEC_LIST && op->reply.hdr.retval > 0)
{
if (!op_read_buf((uint8_t*)op->buf, sizeof(obj_ver_id) * op->reply.hdr.retval))
return;
}
else if ((op->reply.hdr.opcode == OSD_OP_SEC_READ_BMP ||
op->reply.hdr.opcode == OSD_OP_SHOW_CONFIG) && op->reply.hdr.retval > 0)
{
if (!op_read_buf((uint8_t*)op->buf, op->reply.hdr.retval))
return;
}
else if (op->reply.hdr.opcode == OSD_OP_DESCRIBE && op->reply.describe.result_bytes > 0)
{
if (!op_read_buf((uint8_t*)op->buf, op->reply.describe.result_bytes))
return;
}
}
if (cl->proto_csum_status == MSGR_CSUM_FULL ||
cl->read_op_size > 0 && cl->proto_csum_status == MSGR_CSUM_PAYLOAD)
{
if (!op_read_buf((uint8_t*)&op->csum, 8))
return;
}
}
void osd_messenger_t::op_alloc_temp_buffers(osd_op_t *op, int i)
{
size_t total_skip = 0;
for (int j = i; j < op->iov.count; j++)
{
if (!op->iov.buf[j].iov_base)
{
total_skip += op->iov.buf[j].iov_len;
}
}
assert(total_skip);
assert(!op->rmw_buf);
op->rmw_buf = malloc_or_die(total_skip);
total_skip = 0;
for (int j = i; j < op->iov.count; j++)
{
if (!op->iov.buf[j].iov_base)
{
op->iov.buf[j].iov_base = (uint8_t*)op->rmw_buf + total_skip;
total_skip += op->iov.buf[j].iov_len;
}
}
}
bool osd_messenger_t::handle_finished_op(osd_client_t *cl)
{
osd_op_t *op = cl->read_op;
if (cl->proto_csum_status == MSGR_CSUM_FULL ||
cl->read_op_size > 0 && cl->proto_csum_status == MSGR_CSUM_PAYLOAD)
{
uint64_t real_csum = XXH3_64bits_digest(cl->read_csum_state);
if (op->csum != real_csum)
{
fprintf(stderr, "Client %ju checksum mismatch for received data: expected %016jx, got %016jx, disconnecting client\n",
cl->client_id, op->csum, real_csum);
stop_client(cl->client_id);
return false;
}
}
if (op->op_type == OSD_OP_IN)
{
// Operation is ready
cl->received_ops.push_back(op);
}
else
{
// Inline decryption
if (cl->read_op_inline_decrypt_pos != (size_t)-1)
{
op_decrypt_inline(cl);
cl->read_op_inline_decrypt_pos = (size_t)-1;
}
// Measure subop (outbound op) latency
timespec tv_end;
clock_gettime(CLOCK_REALTIME, &tv_end);
stats.subop_stat_count[op->req.hdr.opcode]++;
if (!stats.subop_stat_count[op->req.hdr.opcode])
{
stats.subop_stat_count[op->req.hdr.opcode]++;
stats.subop_stat_sum[op->req.hdr.opcode] = 0;
}
stats.subop_stat_sum[op->req.hdr.opcode] += (
(tv_end.tv_sec - op->tv_begin.tv_sec)*1000000 +
(tv_end.tv_nsec - op->tv_begin.tv_nsec)/1000
);
stats.subop_stat_sum[op->req.hdr.opcode] = 0;
}
stats.subop_stat_sum[op->req.hdr.opcode] += (
(tv_end.tv_sec - op->tv_begin.tv_sec)*1000000 +
(tv_end.tv_nsec - op->tv_begin.tv_nsec)/1000
);
set_immediate_ops.push_back(op);
cl->read_op = NULL;
return true;
}
+122 -552
View File
@@ -7,13 +7,6 @@
#include "messenger.h"
#ifdef WITH_OPENSSL
#include <openssl/bio.h>
#include <openssl/err.h>
#include <openssl/pem.h>
#include <openssl/ssl.h>
#endif
void osd_messenger_t::outbox_push(osd_op_t *cur_op)
{
assert(cur_op->client_id);
@@ -28,7 +21,6 @@ void osd_messenger_t::outbox_push(osd_op_t *cur_op)
{
clock_gettime(CLOCK_REALTIME, &cur_op->tv_begin);
cur_op->req.hdr.id = ++cl->send_op_id;
cl->sent_ops[cur_op->req.hdr.id] = cur_op;
}
else
{
@@ -45,9 +37,77 @@ void osd_messenger_t::outbox_push(osd_op_t *cur_op)
}
// Can't be not found because client IDs are unique
assert(found);
measure_exec(cur_op);
}
cl->write_ops.push_back(cur_op);
auto & to_send_list = cl->write_msg.msg_iovlen ? cl->next_send_list : cl->send_list;
auto & to_outbox = cl->write_msg.msg_iovlen ? cl->next_outbox : cl->outbox;
if (cur_op->op_type == OSD_OP_IN)
{
measure_exec(cur_op);
to_send_list.push_back((iovec){ .iov_base = cur_op->reply.buf, .iov_len = OSD_PACKET_SIZE });
}
else
{
to_send_list.push_back((iovec){ .iov_base = cur_op->req.buf, .iov_len = OSD_PACKET_SIZE });
cl->sent_ops[cur_op->req.hdr.id] = cur_op;
}
to_outbox.push_back((msgr_sendp_t){ .op = cur_op, .flags = MSGR_SENDP_HDR });
// Bitmap
if (cur_op->op_type == OSD_OP_IN &&
cur_op->req.hdr.opcode == OSD_OP_SEC_READ &&
cur_op->reply.sec_rw.attr_len > 0)
{
to_send_list.push_back((iovec){
.iov_base = cur_op->bitmap,
.iov_len = cur_op->reply.sec_rw.attr_len,
});
to_outbox.push_back((msgr_sendp_t){ .op = cur_op, .flags = 0 });
}
else if (cur_op->op_type == OSD_OP_OUT &&
(cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE || cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE) &&
cur_op->req.sec_rw.attr_len > 0)
{
to_send_list.push_back((iovec){
.iov_base = cur_op->bitmap,
.iov_len = cur_op->req.sec_rw.attr_len,
});
to_outbox.push_back((msgr_sendp_t){ .op = cur_op, .flags = 0 });
}
// Operation data
if ((cur_op->op_type == OSD_OP_IN
? (cur_op->req.hdr.opcode == OSD_OP_READ ||
cur_op->req.hdr.opcode == OSD_OP_SEC_READ ||
cur_op->req.hdr.opcode == OSD_OP_SEC_LIST ||
cur_op->req.hdr.opcode == OSD_OP_SHOW_CONFIG ||
cur_op->req.hdr.opcode == OSD_OP_DESCRIBE)
: (cur_op->req.hdr.opcode == OSD_OP_WRITE ||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE ||
cur_op->req.hdr.opcode == OSD_OP_SEC_STABILIZE ||
cur_op->req.hdr.opcode == OSD_OP_SEC_ROLLBACK ||
cur_op->req.hdr.opcode == OSD_OP_SHOW_CONFIG)) && cur_op->iov.count > 0)
{
for (int i = 0; i < cur_op->iov.count; i++)
{
if (cur_op->iov.buf[i].iov_len > 0)
{
assert(cur_op->iov.buf[i].iov_base);
to_send_list.push_back(cur_op->iov.buf[i]);
to_outbox.push_back((msgr_sendp_t){ .op = cur_op, .flags = 0 });
}
}
}
if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ_BMP)
{
if (cur_op->op_type == OSD_OP_IN && cur_op->reply.hdr.retval > 0)
to_send_list.push_back((iovec){ .iov_base = cur_op->buf, .iov_len = (size_t)cur_op->reply.hdr.retval });
else if (cur_op->op_type == OSD_OP_OUT && cur_op->req.sec_read_bmp.len > 0)
to_send_list.push_back((iovec){ .iov_base = cur_op->buf, .iov_len = (size_t)cur_op->req.sec_read_bmp.len });
to_outbox.push_back((msgr_sendp_t){ .op = cur_op, .flags = 0 });
}
if (cur_op->op_type == OSD_OP_IN)
{
to_outbox[to_outbox.size()-1].flags |= MSGR_SENDP_FREE;
}
#ifdef WITH_RDMA
if (cl->peer_state == PEER_RDMA)
{
@@ -58,7 +118,7 @@ void osd_messenger_t::outbox_push(osd_op_t *cur_op)
if (!ringloop)
{
// FIXME: It's worse because it doesn't allow batching
while (cl->write_op || cl->write_ops.size())
while (cl->outbox.size())
{
try_send(cl);
}
@@ -123,105 +183,13 @@ void osd_messenger_t::measure_exec(osd_op_t *cur_op)
}
}
bool osd_messenger_t::ssl_do_handshake(osd_client_t *cl)
{
int r = SSL_do_handshake(cl->ssl_cli);
if (r > 0)
{
cl->ssl_handshake_done = true;
}
else
{
r = SSL_get_error(cl->ssl_cli, r);
if (r != SSL_ERROR_WANT_WRITE && r != SSL_ERROR_WANT_READ)
{
fprintf(stderr, "Client %ju TLS handshake error: %s, stopping client\n", cl->client_id, ERR_error_string(ERR_get_error(), NULL));
stop_client(cl->client_id);
return false;
}
if (r == SSL_ERROR_WANT_WRITE)
{
cl->ssl_want_write = true;
}
}
return true;
}
bool osd_messenger_t::ssl_do_encrypt(osd_client_t *cl)
{
if (cl->send_list.size() >= IOV_MAX || !cl->ssl_want_write)
return false;
size_t prev_size = cl->ssl_out_buf_size;
while (true)
{
size_t min_cap = cl->ssl_out_buf_size*2;
if (min_cap < 16384)
min_cap = 16384;
if (cl->ssl_out_buf_cap < min_cap)
{
uint8_t *old_buf = cl->ssl_out_buf;
uint8_t *old_end = old_buf + cl->ssl_out_buf_cap;
cl->ssl_out_buf = (uint8_t*)realloc_or_die(cl->ssl_out_buf, min_cap);
cl->ssl_out_buf_cap = min_cap;
for (auto & iov: cl->send_list)
{
if (iov.iov_base >= old_buf && iov.iov_base < old_end)
iov.iov_base = cl->ssl_out_buf + ((uint8_t*)iov.iov_base - old_buf);
}
}
bool full_read = (ssl_do_encrypt_to(cl, cl->ssl_out_buf+cl->ssl_out_buf_size,
cl->ssl_out_buf_cap-cl->ssl_out_buf_size) == cl->ssl_out_buf_cap-cl->ssl_out_buf_size);
if (!full_read)
break;
}
if (cl->ssl_out_buf_size > prev_size)
cl->send_list.push_back((iovec){ .iov_base = cl->ssl_out_buf+prev_size, .iov_len = cl->ssl_out_buf_size-prev_size });
return true;
}
size_t osd_messenger_t::ssl_do_encrypt_to(osd_client_t *cl, uint8_t *buf, size_t size)
{
if (size < sizeof(msgr_tls_record_hdr_t))
return 0;
int r = BIO_read(cl->read_from_ssl, buf+sizeof(msgr_tls_record_hdr_t), size-sizeof(msgr_tls_record_hdr_t));
if (r > 0)
{
if (r < size-sizeof(msgr_tls_record_hdr_t))
cl->ssl_want_write = false;
msgr_tls_record_hdr_t *hdr = (msgr_tls_record_hdr_t*)buf;
hdr->encrypted = 1;
hdr->size = r;
return r+sizeof(msgr_tls_record_hdr_t);
}
return 0;
}
bool osd_messenger_t::try_send(osd_client_t *cl)
{
if (!cl->write_op && !cl->write_ops.size() && !cl->ssl_want_write ||
cl->write_msg.msg_iovlen > 0 || cl->peer_state == PEER_STOPPED || cl->peer_fd < 0)
if (!cl->send_list.size() || cl->write_msg.msg_iovlen > 0 || cl->peer_state == PEER_STOPPED || cl->peer_fd < 0)
{
return true;
}
assert(cl->peer_state != PEER_RDMA);
if (cl->ssl_cli && !cl->ssl_handshake_done)
{
bool ok = ssl_do_encrypt(cl);
assert(ok && cl->send_list.size() > 0);
}
else
{
while ((cl->write_op || cl->write_ops.size()) && cl->send_list.size() < IOV_MAX)
{
next_write_op(cl);
osd_op_t *op = cl->write_op;
op_get_write_buffers(cl, cl->send_list);
if (!cl->write_op && op->op_type == OSD_OP_IN)
{
cl->send_free_ops.push_back(op);
}
}
}
if (ringloop && !use_sync_send_recv)
{
auto iothread = iothreads.size() ? iothreads[cl->peer_fd % iothreads.size()] : NULL;
@@ -234,24 +202,20 @@ bool osd_messenger_t::try_send(osd_client_t *cl)
data_local = {};
}
if (!sqe)
{
return false;
}
cl->send_list_size = 0;
for (auto & iov: cl->send_list)
{
cl->send_list_size += iov.iov_len;
}
cl->write_msg.msg_iov = cl->send_list.data();
cl->write_msg.msg_iovlen = cl->send_list.size() < IOV_MAX ? cl->send_list.size() : IOV_MAX;
cl->refs++;
ring_data_t* data = ((ring_data_t*)sqe->user_data);
data->callback = [this, cl](ring_data_t *data) { handle_send(data->res, data->prev, data->more, cl); };
bool use_zc = has_sendmsg_zc && min_zerocopy_send_size >= 0;
if (use_zc && min_zerocopy_send_size > 0 &&
cl->send_list_size/cl->write_msg.msg_iovlen < min_zerocopy_send_size)
if (use_zc && min_zerocopy_send_size > 0)
{
use_zc = false;
size_t avg_size = 0;
for (size_t i = 0; i < cl->write_msg.msg_iovlen; i++)
avg_size += cl->write_msg.msg_iov[i].iov_len;
if (avg_size/cl->write_msg.msg_iovlen < min_zerocopy_send_size)
use_zc = false;
}
if (use_zc)
{
@@ -282,21 +246,6 @@ bool osd_messenger_t::try_send(osd_client_t *cl)
return true;
}
void osd_messenger_t::next_write_op(osd_client_t *cl)
{
if (!cl->write_op)
{
cl->write_op = cl->write_ops.front();
cl->write_ops.pop_front();
if (cl->proto_csum_status == MSGR_CSUM_FULL || cl->proto_csum_status == MSGR_CSUM_PAYLOAD)
{
if (!cl->write_csum_state)
cl->write_csum_state = XXH3_createState();
XXH3_64bits_reset(cl->write_csum_state);
}
}
}
void osd_messenger_t::send_replies()
{
for (int i = 0; i < write_ready_clients.size(); i++)
@@ -317,7 +266,6 @@ void osd_messenger_t::handle_send(int result, bool prev, bool more, osd_client_t
if (!prev)
{
cl->write_msg.msg_iovlen = 0;
cl->send_list.clear();
}
if (!more)
{
@@ -350,32 +298,57 @@ void osd_messenger_t::handle_send(int result, bool prev, bool more, osd_client_t
cl->zc_free_list.erase(cl->zc_free_list.begin(), cl->zc_free_list.begin()+i+1);
return;
}
if (cl->send_list_size > result)
int done = 0;
while (result > 0 && done < cl->send_list.size())
{
fprintf(stderr, "Client %ju socket write error: expected to send "
"%zu bytes with MSG_WAITALL but sent %u. Disconnecting client\n", cl->client_id, cl->send_list_size, result);
stop_client(cl->client_id);
return;
}
for (auto op: cl->send_free_ops)
{
if (more)
cl->zc_free_list.push_back(op);
iovec & iov = cl->send_list[done];
if (iov.iov_len <= result)
{
if (cl->outbox[done].flags & MSGR_SENDP_FREE)
{
// Reply fully sent
if (more)
cl->zc_free_list.push_back(cl->outbox[done].op);
else
delete cl->outbox[done].op;
}
result -= iov.iov_len;
done++;
}
else
delete op;
{
iov.iov_len -= result;
iov.iov_base = (uint8_t*)iov.iov_base + result;
break;
}
}
cl->ssl_out_buf_size = 0;
if (more)
cl->zc_free_list.push_back(NULL); // end marker
cl->send_free_ops.clear();
cl->write_state = cl->write_op || cl->write_ops.size() ? CL_WRITE_READY : 0;
if ((cl->proto_csum_status & MSGR_CSUM_NEG) && !cl->write_op && !cl->write_ops.size())
{
// Checksums negotiated, enable
cl->proto_csum_status = cl->proto_csum_status & (~MSGR_CSUM_NEG);
int expected = cl->send_list.size() < IOV_MAX ? cl->send_list.size() : IOV_MAX;
if (done != expected)
{
fprintf(stderr, "Client %ju socket write error: expected to send "
"%d iovecs with MSG_WAITALL but sent %d. Disconnecting client\n", cl->client_id, expected, done);
stop_client(cl->client_id);
return;
}
cl->zc_free_list.push_back(NULL); // end marker
}
if (done > 0)
{
cl->send_list.erase(cl->send_list.begin(), cl->send_list.begin()+done);
cl->outbox.erase(cl->outbox.begin(), cl->outbox.begin()+done);
}
if (cl->next_send_list.size())
{
cl->send_list.insert(cl->send_list.end(), cl->next_send_list.begin(), cl->next_send_list.end());
cl->outbox.insert(cl->outbox.end(), cl->next_outbox.begin(), cl->next_outbox.end());
cl->next_send_list.clear();
cl->next_outbox.clear();
}
cl->write_state = cl->outbox.size() > 0 ? CL_WRITE_READY : 0;
#ifdef WITH_RDMA
if (cl->rdma_conn && !cl->write_op && !cl->write_ops.size() && cl->peer_state == PEER_RDMA_CONNECTING)
if (cl->rdma_conn && !cl->outbox.size() && cl->peer_state == PEER_RDMA_CONNECTING)
{
// FIXME: Ignore pings during RDMA state transition
if (log_level > 0)
@@ -393,406 +366,3 @@ void osd_messenger_t::handle_send(int result, bool prev, bool more, osd_client_t
write_ready_clients.push_back(cl->client_id);
}
}
static inline bool op_write_headers(osd_op_t *op, std::function<bool(uint8_t*, size_t, bool)> op_write_buf, bool skip_hdr_csum)
{
if (!op_write_buf((op->op_type == OSD_OP_IN ? op->reply.buf : op->req.buf), OSD_PACKET_SIZE, skip_hdr_csum))
{
return false;
}
// Bitmap
if (op->op_type == OSD_OP_IN &&
op->req.hdr.opcode == OSD_OP_SEC_READ &&
op->reply.sec_rw.attr_len > 0)
{
if (!op_write_buf((uint8_t*)op->bitmap, op->reply.sec_rw.attr_len, false))
return false;
}
else if (op->op_type == OSD_OP_OUT &&
(op->req.hdr.opcode == OSD_OP_SEC_WRITE || op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE) &&
op->req.sec_rw.attr_len > 0)
{
if (!op_write_buf((uint8_t*)op->bitmap, op->req.sec_rw.attr_len, false))
return false;
}
if (op->req.hdr.opcode == OSD_OP_SEC_READ_BMP)
{
if (op->op_type == OSD_OP_IN && op->reply.hdr.retval > 0)
{
if (!op_write_buf((uint8_t*)op->buf, (size_t)op->reply.hdr.retval, false))
return false;
}
else if (op->op_type == OSD_OP_OUT && op->req.sec_read_bmp.len > 0)
{
if (!op_write_buf((uint8_t*)op->buf, (size_t)op->req.sec_read_bmp.len, false))
return false;
}
}
return true;
}
static inline bool op_has_data(osd_op_t *op)
{
return (op->op_type == OSD_OP_IN
? (op->req.hdr.opcode == OSD_OP_READ ||
op->req.hdr.opcode == OSD_OP_SEC_READ ||
op->req.hdr.opcode == OSD_OP_SEC_LIST ||
op->req.hdr.opcode == OSD_OP_SHOW_CONFIG ||
op->req.hdr.opcode == OSD_OP_DESCRIBE)
: (op->req.hdr.opcode == OSD_OP_WRITE ||
op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE ||
op->req.hdr.opcode == OSD_OP_SEC_STABILIZE ||
op->req.hdr.opcode == OSD_OP_SEC_ROLLBACK ||
op->req.hdr.opcode == OSD_OP_SHOW_CONFIG)) && op->iov.count > 0;
}
static inline bool op_has_data_for_ssl(osd_op_t *op)
{
return (op->op_type == OSD_OP_IN
? (op->req.hdr.opcode == OSD_OP_SEC_LIST ||
op->req.hdr.opcode == OSD_OP_SHOW_CONFIG ||
op->req.hdr.opcode == OSD_OP_DESCRIBE)
: (op->req.hdr.opcode == OSD_OP_SEC_STABILIZE ||
op->req.hdr.opcode == OSD_OP_SEC_ROLLBACK ||
op->req.hdr.opcode == OSD_OP_SHOW_CONFIG)) && op->iov.count > 0;
}
static inline bool op_has_data_for_nonssl(osd_op_t *op)
{
return (op->op_type == OSD_OP_IN
? (op->req.hdr.opcode == OSD_OP_READ ||
op->req.hdr.opcode == OSD_OP_SEC_READ)
: (op->req.hdr.opcode == OSD_OP_WRITE ||
op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE)) && op->iov.count > 0;
}
bool osd_messenger_t::ssl_op_write_buf(osd_client_t *cl, uint8_t *src, size_t src_len, bool skip_csum, size_t & from, size_t & done)
{
if (from < src_len)
{
size_t n = src_len-from;
int ok = SSL_write_ex(cl->ssl_cli, src+from, n, &n);
if (ok)
{
cl->ssl_want_write = true;
if (cl->write_csum_state && !skip_csum)
XXH3_64bits_update(cl->write_csum_state, src+from, n);
done += n;
cl->write_op_pos += n;
from += n;
if (from < src_len)
return false;
from = 0;
}
else
{
int res = SSL_get_error(cl->ssl_cli, ok);
if (res == SSL_ERROR_WANT_WRITE || res == 0)
cl->ssl_want_write = true;
else if (res == SSL_ERROR_ZERO_RETURN)
{
fprintf(stderr, "Client %ju TLS disconnected\n", cl->client_id);
stop_client(cl->client_id);
}
else if (res != SSL_ERROR_WANT_READ)
{
fprintf(stderr, "Client %ju TLS write error: %s. Disconnecting client\n", cl->client_id, ERR_error_string(ERR_get_error(), NULL));
stop_client(cl->client_id);
}
return false;
}
}
else
from -= src_len;
return true;
}
bool osd_messenger_t::op_write_buf(osd_client_t *cl, uint8_t *src, size_t src_len, uint8_t *dst, size_t dst_len, bool skip_csum, size_t & from, size_t & done)
{
if (from < src_len)
{
size_t n = src_len-from;
if (n > dst_len-done)
n = dst_len-done;
if (cl->write_csum_state && !skip_csum)
XXH3_64bits_update(cl->write_csum_state, src+from, n);
memcpy(dst+done, src+from, n);
done += n;
cl->write_op_pos += n;
from += n;
if (from < src_len)
return false;
from = 0;
}
else
from -= src_len;
return true;
}
size_t osd_messenger_t::ssl_op_copy_to(osd_client_t *cl, uint8_t *dst, size_t dst_len)
{
size_t done = 0;
size_t from = cl->write_op_pos;
// Encrypt headers and data except read/write data
auto to_ssl = [&](uint8_t *src, size_t src_len, bool skip_csum)
{
return ssl_op_write_buf(cl, src, src_len, skip_csum, from, done);
};
int i = 0;
bool full_hdr = false;
bool full_op = false;
do
{
if (!full_hdr)
full_hdr = op_write_headers(cl->write_op, to_ssl, cl->proto_csum_status != MSGR_CSUM_FULL);
if (full_hdr && op_has_data_for_ssl(cl->write_op))
{
for (; i < cl->write_op->iov.count; i++)
if (!ssl_op_write_buf(cl, (uint8_t*)cl->write_op->iov.buf[i].iov_base, cl->write_op->iov.buf[i].iov_len, false, from, done))
break;
full_op = true;
}
if (cl->peer_state == PEER_STOPPED)
return 0;
if (cl->ssl_want_write)
{
auto ssl_written = ssl_do_encrypt_to(cl, dst+done, dst_len-done);
if (!ssl_written)
return done;
done += ssl_written;
}
} while (!full_op);
// Non-TLS-encrypted operation data
if (op_has_data_for_nonssl(cl->write_op))
{
if (!op_copy_data_to(cl, dst, dst_len, from, done))
return done;
}
// TLS-encrypted checksum (uh oh...)
if (cl->proto_csum_status == MSGR_CSUM_FULL ||
cl->proto_csum_status == MSGR_CSUM_PAYLOAD && cl->write_op_pos > OSD_PACKET_SIZE)
{
if (!from)
cl->write_op->csum = XXH3_64bits_digest(cl->write_csum_state);
if (!ssl_op_write_buf(cl, (uint8_t*)&cl->write_op->csum, 8, true, from, done))
return done;
if (cl->ssl_want_write)
{
auto ssl_written = ssl_do_encrypt_to(cl, dst+done, dst_len-done);
if (!ssl_written)
return done;
done += ssl_written;
}
}
cl->write_op = NULL;
cl->write_op_pos = 0;
return done;
}
size_t osd_messenger_t::op_copy_to(osd_client_t *cl, uint8_t *dst, size_t dst_len)
{
if (cl->ssl_cli)
{
return ssl_op_copy_to(cl, dst, dst_len);
}
size_t done = 0;
size_t from = cl->write_op_pos;
auto op_write_buf = [&](uint8_t *src, size_t src_len, bool skip_csum)
{
return this->op_write_buf(cl, src, src_len, dst, dst_len, skip_csum, from, done);
};
// Header
if (!op_write_headers(cl->write_op, op_write_buf, cl->proto_csum_status != MSGR_CSUM_FULL))
{
return done;
}
// Operation data
if (op_has_data(cl->write_op))
{
if (!op_copy_data_to(cl, dst, dst_len, from, done))
return done;
}
if (cl->proto_csum_status == MSGR_CSUM_FULL ||
cl->proto_csum_status == MSGR_CSUM_PAYLOAD && cl->write_op_pos > OSD_PACKET_SIZE)
{
if (!from)
cl->write_op->csum = XXH3_64bits_digest(cl->write_csum_state);
if (!op_write_buf((uint8_t*)&cl->write_op->csum, 8, true))
return done;
}
cl->write_op = NULL;
cl->write_op_pos = 0;
return done;
}
bool osd_messenger_t::op_copy_data_to(osd_client_t *cl, uint8_t *dst, size_t dst_len, size_t & from, size_t & done)
{
if (cl->write_op->enc)
{
if (!op_encrypted_copy_data_to(cl, dst, dst_len, from, done))
{
return false;
}
}
else
{
for (int i = 0; i < cl->write_op->iov.count; i++)
{
auto & iov = cl->write_op->iov.buf[i];
if (!op_write_buf(cl, (uint8_t*)iov.iov_base, iov.iov_len, dst, dst_len, false, from, done))
return false;
}
}
return true;
}
void osd_messenger_t::op_get_write_buffers(osd_client_t *cl, std::vector<iovec> & lst)
{
size_t from = cl->write_op_pos;
auto op_write_buf = [&](uint8_t *src, size_t src_len, bool skip_csum)
{
if (lst.size() >= IOV_MAX)
return false;
if (from < src_len)
{
if (cl->write_csum_state && !skip_csum)
XXH3_64bits_update(cl->write_csum_state, src+from, src_len-from);
lst.push_back((iovec){ .iov_base = src+from, .iov_len = src_len-from });
cl->write_op_pos += src_len-from;
from = 0;
}
else
from -= src_len;
return true;
};
// Header
if (!op_write_headers(cl->write_op, op_write_buf, cl->proto_csum_status != MSGR_CSUM_FULL))
{
return;
}
// Operation data
if (op_has_data(cl->write_op))
{
if (cl->write_op->enc)
{
if (lst.size() >= IOV_MAX)
return;
// No way except to allocate a temporary buffer and encrypt data to it
assert(cl->write_op->req.hdr.opcode == OSD_OP_WRITE);
size_t remsize = cl->write_op->req.rw.len - from + (from % 16);
assert(remsize > 0);
assert(!cl->write_op->enc_buf);
cl->write_op->enc_buf = (uint8_t*)malloc_or_die(remsize);
size_t done = 0;
bool end = op_encrypted_copy_data_to(cl, cl->write_op->enc_buf, remsize, from, done);
assert(end);
lst.push_back((iovec){ .iov_base = cl->write_op->enc_buf, .iov_len = remsize });
}
else
{
for (int i = 0; i < cl->write_op->iov.count; i++)
{
if (!op_write_buf((uint8_t*)cl->write_op->iov.buf[i].iov_base, cl->write_op->iov.buf[i].iov_len, false))
return;
}
}
}
if (cl->proto_csum_status == MSGR_CSUM_FULL ||
cl->proto_csum_status == MSGR_CSUM_PAYLOAD && cl->write_op_pos > OSD_PACKET_SIZE)
{
if (!from)
cl->write_op->csum = XXH3_64bits_digest(cl->write_csum_state);
if (!op_write_buf((uint8_t*)&cl->write_op->csum, 8, true))
return;
}
cl->write_op = NULL;
cl->write_op_pos = 0;
}
void osd_messenger_t::ssl_op_get_write_buffers(osd_client_t *cl, std::vector<iovec> & lst)
{
size_t done = 0;
size_t from = cl->write_op_pos;
auto op_write_buf = [&](uint8_t *src, size_t src_len, bool skip_csum)
{
if (lst.size() >= IOV_MAX)
return false;
if (from < src_len)
{
if (cl->write_csum_state && !skip_csum)
XXH3_64bits_update(cl->write_csum_state, src+from, src_len-from);
lst.push_back((iovec){ .iov_base = src+from, .iov_len = src_len-from });
cl->write_op_pos += src_len-from;
from = 0;
}
else
from -= src_len;
return true;
};
// Encrypt headers and data except read/write data
auto to_ssl = [&](uint8_t *src, size_t src_len, bool skip_csum)
{
return ssl_op_write_buf(cl, src, src_len, skip_csum, from, done);
};
int i = 0;
bool full_hdr = false;
bool full_op = false;
do
{
if (!full_hdr)
full_hdr = op_write_headers(cl->write_op, to_ssl, cl->proto_csum_status != MSGR_CSUM_FULL);
if (full_hdr && op_has_data_for_ssl(cl->write_op))
{
for (; i < cl->write_op->iov.count; i++)
if (!ssl_op_write_buf(cl, (uint8_t*)cl->write_op->iov.buf[i].iov_base, cl->write_op->iov.buf[i].iov_len, false, from, done))
break;
full_op = true;
}
if (cl->peer_state == PEER_STOPPED)
return;
if (!ssl_do_encrypt(cl))
return;
} while (!full_op);
// Non-TLS-encrypted operation data
if (op_has_data_for_nonssl(cl->write_op))
{
if (cl->write_op->enc)
{
if (lst.size() >= IOV_MAX)
return;
// No way except to allocate a temporary buffer and encrypt data to it
assert(cl->write_op->req.hdr.opcode == OSD_OP_WRITE);
size_t remsize = cl->write_op->req.rw.len - from + (from % 16);
assert(remsize > 0);
assert(!cl->write_op->enc_buf);
cl->write_op->enc_buf = (uint8_t*)malloc_or_die(remsize);
size_t done = 0;
bool end = op_encrypted_copy_data_to(cl, cl->write_op->enc_buf, remsize, from, done);
assert(end);
lst.push_back((iovec){ .iov_base = cl->write_op->enc_buf, .iov_len = remsize });
}
else
{
for (int i = 0; i < cl->write_op->iov.count; i++)
{
if (!op_write_buf((uint8_t*)cl->write_op->iov.buf[i].iov_base, cl->write_op->iov.buf[i].iov_len, false))
return;
}
}
}
// TLS-encrypted checksum (uh oh...)
if (cl->proto_csum_status == MSGR_CSUM_FULL ||
cl->proto_csum_status == MSGR_CSUM_PAYLOAD && cl->write_op_pos > OSD_PACKET_SIZE)
{
if (!from)
cl->write_op->csum = XXH3_64bits_digest(cl->write_csum_state);
if (!ssl_op_write_buf(cl, (uint8_t*)&cl->write_op->csum, 8, true, from, done))
return;
if (!ssl_do_encrypt(cl))
return;
}
cl->write_op = NULL;
cl->write_op_pos = 0;
}
-54
View File
@@ -5,16 +5,9 @@
#include <assert.h>
#include "messenger.h"
#include "../util/xxh_x86dispatch.h"
#ifdef WITH_RDMA
#include "msgr_rdma.h"
#endif
#ifdef WITH_OPENSSL
#include <openssl/bio.h>
#include <openssl/err.h>
#include <openssl/pem.h>
#include <openssl/ssl.h>
#endif
void osd_client_t::cancel_ops()
{
@@ -86,22 +79,6 @@ void osd_messenger_t::stop_client(uint64_t client_id, bool force_delete)
fprintf(stderr, "[OSD %ju] Stopping client %ju (regular client)\n", osd_num, client_id);
}
}
if (cl->encrypt_ctx)
{
if (encrypt_ctx_pool.size() > max_aes_xts_pool_size)
destroy_aes_xts_encrypt(cl->encrypt_ctx);
else
encrypt_ctx_pool.push_back(cl->encrypt_ctx);
cl->encrypt_ctx = NULL;
}
if (cl->decrypt_ctx)
{
if (decrypt_ctx_pool.size() > max_aes_xts_pool_size)
destroy_aes_xts_decrypt(cl->decrypt_ctx);
else
decrypt_ctx_pool.push_back(cl->decrypt_ctx);
cl->decrypt_ctx = NULL;
}
// First set state to STOPPED so another stop_client() call doesn't try to free it again
cl->refs++;
int prev_state = cl->peer_state;
@@ -212,13 +189,6 @@ osd_client_t::~osd_client_t()
}
// Cancel outbound ops
cancel_ops();
for (osd_op_t *op: send_free_ops)
{
if (op)
{
delete op;
}
}
for (osd_op_t *op: zc_free_list)
{
if (op)
@@ -234,29 +204,5 @@ osd_client_t::~osd_client_t()
rdma_conn = NULL;
}
#endif
#endif
if (read_csum_state)
{
XXH3_freeState(read_csum_state);
read_csum_state = NULL;
}
if (write_csum_state)
{
XXH3_freeState(write_csum_state);
write_csum_state = NULL;
}
#ifdef WITH_OPENSSL
if (ssl_cli)
{
SSL_free(ssl_cli);
ssl_cli = NULL;
write_to_ssl = NULL;
read_from_ssl = NULL;
}
if (ssl_out_buf)
{
free(ssl_out_buf);
ssl_out_buf = NULL;
}
#endif
}
+2 -4
View File
@@ -37,7 +37,6 @@
#define OSD_OP_RECOVERY_RELATED (uint32_t)1
#define OSD_OP_IGNORE_PG_LOCK (uint32_t)2
#define OSD_OP_RETURN_CHAIN (uint32_t)4
// Memory alignment for direct I/O (usually 512 bytes)
#ifndef DIRECT_IO_ALIGNMENT
@@ -229,10 +228,9 @@ struct __attribute__((__packed__)) osd_op_rw_t
uint64_t offset;
// length. 0 means to read all bitmaps of the specified range, but no data.
uint32_t len;
// flags
// OSD_OP_RETURN_CHAIN for chained reads: return parent number in chain for each block
// flags (for future)
uint32_t flags;
// inode metadata revision for chained reads
// inode metadata revision
uint64_t meta_revision;
// object version for atomic "CAS" (compare-and-set) writes
// writes and deletes fail with -EINTR if object version differs from (version-1)
+1 -1
View File
@@ -6,7 +6,7 @@ includedir=${prefix}/@CMAKE_INSTALL_INCLUDEDIR@
Name: Vitastor
Description: Vitastor client library
Version: 3.0.9
Version: 3.0.10
Libs: -L${libdir} -lvitastor_client
Cflags: -I${includedir}
+1 -15
View File
@@ -2,19 +2,11 @@ cmake_minimum_required(VERSION 2.8...3.30)
project(vitastor)
set(OPENAPI_JSON_H "${CMAKE_CURRENT_BINARY_DIR}/openapi.json.h")
add_custom_command(
OUTPUT ${OPENAPI_JSON_H}
COMMAND ${CMAKE_COMMAND} -E echo const char* openapi_description = R\\\"json\\\( > ${OPENAPI_JSON_H}
COMMAND ${CMAKE_COMMAND} -E cat ${CMAKE_CURRENT_SOURCE_DIR}/openapi.json >> ${OPENAPI_JSON_H}
COMMAND ${CMAKE_COMMAND} -E echo "\\)json\\\"\\;" >> ${OPENAPI_JSON_H}
DEPENDS openapi.json
)
# libvitastor_cli.a
add_library(vitastor_cli STATIC
cli_common.cpp
cli_alloc_osd.cpp
cli_status.cpp
cli_describe.cpp
cli_fix.cpp
cli_ls.cpp
@@ -22,7 +14,6 @@ add_library(vitastor_cli STATIC
cli_dd.cpp
cli_modify.cpp
cli_modify_osd.cpp
cli_modify_user.cpp
cli_osd_tree.cpp
cli_pg_ls.cpp
cli_flatten.cpp
@@ -36,13 +27,8 @@ add_library(vitastor_cli STATIC
cli_pool_ls.cpp
cli_pool_modify.cpp
cli_pool_rm.cpp
cli_serve.cpp
cli_status.cpp
cli_user_ls.cpp
${OPENAPI_JSON_H}
)
target_compile_options(vitastor_cli PUBLIC -fPIC)
target_include_directories(vitastor_cli PRIVATE ${CMAKE_CURRENT_BINARY_DIR})
# vitastor-cli
add_executable(vitastor-cli
+45 -139
View File
@@ -37,42 +37,24 @@ static const char* help_text =
" --sort FIELD Sort by specified field (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)\n"
" -r|--reverse Sort in descending order\n"
" -n|--count N Only list first N items\n"
" --ids ID1,ID2 Only list images with specified full IDs\n"
" --tree Show image snapshot/clone tree\n"
"\n"
"vitastor-cli create -s|--size SIZE [OPTIONS] <name>\n"
" Create an image. Options:\n"
" -s|--size SIZE New image size in bytes or with a K/M/G/T unit suffix.\n"
" -p|--pool POOL Specify pool for the new image (may be omitted if there is only 1 pool).\n"
" --parent PARENT Create a copy-on-write image clone based on PARENT (or PARENT@SNAPSHOT).\n"
" If parent is not a snapshot, it must be a read-only image.\n"
" --enc-key random Generate a new random AES-256-XTS encryption key for the new image.\n"
" --enc-key HEX Set a specified AES-256-XTS key (64 bytes in hex) for the new image.\n"
" --enc-key vault:ID Use an encryption key from an external Vault secret with specified ID.\n"
" --owner username Set owner (default is the current user from TLS certificate).\n"
" --owner_group name Set owner group name.\n"
" --reader_group rdr Set reader group name.\n"
"vitastor-cli create -s|--size <size> [-p|--pool <id|name>] [--parent <parent_name>[@<snapshot>]] <name>\n"
" Create an image. You may use K/M/G/T suffixes for <size>. If --parent is specified,\n"
" a copy-on-write image clone is created. Parent must be a snapshot (readonly image).\n"
" Pool must be specified if there is more than one pool.\n"
"\n"
"vitastor-cli create --snapshot <snapshot> [OPTIONS] <image>\n"
"vitastor-cli snap-create [OPTIONS] <image>@<snapshot>\n"
" Create a snapshot of image <image>. May be used live if only a single writer is active.\n"
" Options:\n"
" -p|--pool POOL Move image to pool POOL, leaving the snapshot in the old pool.\n"
" --enc-key random Change image encryption key to a new random AES-256-XTS key.\n"
" --enc-key KEY Change image encryption key to a specified key, Vault key or to an empty key.\n"
" By default, the image retains its old key when taking a snapshot.\n"
"vitastor-cli create --snapshot <snapshot> [-p|--pool <id|name>] <image>\n"
"vitastor-cli snap-create [-p|--pool <id|name>] <image>@<snapshot>\n"
" Create a snapshot of image <name>. May be used live if only a single writer is active.\n"
"\n"
"vitastor-cli modify <name> [--rename <new-name>] [--resize <size>] [--readonly | --readwrite] [-f|--force] [--down-ok]\n"
" Rename, resize image or change its readonly status. Images with children can't be made read-write.\n"
" If the new size is smaller than the old size, extra data will be purged.\n"
" You should resize file system in the image, if present, before shrinking it.\n"
" --deleted 1|0 Set/clear 'deleted image' flag (set automatically during unfinished deletes).\n"
" -f|--force Proceed with shrinking or setting readwrite flag even if the image has children.\n"
" --down-ok Proceed with shrinking even if some data will be left on unavailable OSDs.\n"
" --enc-key HEX Change image encryption key (allowed only with --force).\n"
" --owner username Change image owner.\n"
" --owner_group name Change image owner group name.\n"
" --reader_group rdr Change image reader group name.\n"
" --deleted 1|0 Set/clear 'deleted image' flag (set automatically during unfinished deletes).\n"
" -f|--force Proceed with shrinking or setting readwrite flag even if the image has children.\n"
" --down-ok Proceed with shrinking even if some data will be left on unavailable OSDs.\n"
"\n"
"vitastor-cli dd [iimg=<image> | if=<file>] [oimg=<image> | of=<file>] [bs=1M]\n"
" [count=N] [seek/oseek=N] [skip/iseek=M] [iodepth=N] [status=progress]\n"
@@ -211,7 +193,6 @@ static const char* help_text =
" --used_for_app s3:<name> Mark pool as used for S3 location with name <name>\n"
" --pg_stripe_size <number> Increase object grouping stripe\n"
" --max_osd_combinations 10000 Maximum number of random combinations for LP solver input\n"
" --creator_group <group> User group allowed to create images in this pool.\n"
" --wait Wait for the new pool to come online\n"
" -f|--force Do not check that cluster has enough OSDs to create the pool\n"
" Examples:\n"
@@ -223,7 +204,7 @@ static const char* help_text =
" [-s|--pg_size <number>] [--pg_minsize <number>] [-n|--pg_count <count>]\n"
" [--failure_domain <level>] [--root_node <node>] [--osd_tags <tags>] [--used_for_app <type>:<name>]\n"
" [--max_osd_combinations <number>] [--primary_affinity_tags <tags>] [--scrub_interval <time>]\n"
" [--level_placement <rules>] [--raw_placement <rules>] [--creator_group <group>]\n"
" [--level_placement <rules>] [--raw_placement <rules>]\n"
" Non-modifiable parameters (changing them WILL lead to data loss):\n"
" [--block_size <size>] [--bitmap_granularity <size>]\n"
" [--immediate_commit <all|small|none>] [--pg_stripe_size <size>]\n"
@@ -236,7 +217,7 @@ static const char* help_text =
"vitastor-cli rm-pool|pool-rm [--force] <id|name>\n"
" Remove a pool. Refuses to remove pools with images without --force.\n"
"\n"
"vitastor-cli ls-pools|pool-ls|ls-pool|pools [-l] [--detail] [--sort FIELD] [-r] [-n N] [<glob> ...]\n"
"vitastor-cli ls-pools|pool-ls|ls-pool|pools [-l] [--detail] [--sort FIELD] [-r] [-n N] [--stats] [<glob> ...]\n"
" List pools (only matching <glob> patterns if passed).\n"
" -l|--long Also report I/O statistics\n"
" --detail Use list format (not table), show all details\n"
@@ -244,25 +225,6 @@ static const char* help_text =
" -r|--reverse Sort in descending order\n"
" -n|--count N Only list first N items\n"
"\n"
"vitastor-cli ls-users|user-ls|ls-user|list-users [<name> ...]\n"
" List users (only with specified names if passed).\n"
"\n"
"vitastor-cli modify-user --type <type> --groups group1,group2,... <username>\n"
" Create or update user permissions. User names match CN of their certificates.\n"
" --type TYPE Set user type: client, admin, mon or osd. Default is client.\n"
" --groups GROUPS Set user's groups.\n"
"\n"
"vitastor-cli rm-user|remove-user|delete-user <username>\n"
" Remove a user.\n"
"\n"
"vitastor-cli serve\n"
" Start HTTP server able to handle CLI commands over a REST API. Options:\n"
" --bind_address ADDR Specify server IP address or addresses, separated by space. Default is 127.0.0.1.\n"
" --port 8080 Specify server port.\n"
" --ssl_cert FILE Path to server SSL certificate file (PEM format).\n"
" --ssl_key FILE Path to server SSL private key file.\n"
" --ssl_ca FILE Path to file with SSL CA certificates used to validate client connections.\n"
"\n"
"Use vitastor-cli --help <command> for command details or vitastor-cli --help --all for all details.\n"
"\n"
"GLOBAL OPTIONS:\n"
@@ -357,24 +319,27 @@ static json11::Json::object parse_args(int narg, const char *args[])
return cfg;
}
std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg, cli_result_t & result)
static int run(cli_tool_t *p, json11::Json::object cfg)
{
cli_result_t result = {};
p->is_command_line = true;
p->parse_config(cfg);
json11::Json::array cmd = cfg["command"].array_items();
cfg.erase("command");
std::function<bool(cli_result_t &)> action_cb;
if (!cmd.size())
{
result = { .err = EOPNOTSUPP, .text = "command is missing" };
result = { .err = EINVAL, .text = "command is missing" };
}
else if (cmd[0] == "status")
{
// Show cluster status
action_cb = start_status(cfg);
action_cb = p->start_status(cfg);
}
else if (cmd[0] == "df")
{
// Show pool space stats
action_cb = start_pool_ls(cfg);
action_cb = p->start_pool_ls(cfg);
}
else if (cmd[0] == "ls")
{
@@ -384,7 +349,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
cmd.erase(cmd.begin(), cmd.begin()+1);
cfg["names"] = cmd;
}
action_cb = start_ls(cfg);
action_cb = p->start_ls(cfg);
}
else if (cmd[0] == "snap-create")
{
@@ -399,7 +364,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
{
cfg["image"] = name.substr(0, pos);
cfg["snapshot"] = name.substr(pos + 1);
action_cb = start_create(cfg);
action_cb = p->start_create(cfg);
}
}
else if (cmd[0] == "create")
@@ -409,7 +374,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
{
cfg["image"] = cmd[1];
}
action_cb = start_create(cfg);
action_cb = p->start_create(cfg);
}
else if (cmd[0] == "modify")
{
@@ -418,12 +383,12 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
{
cfg["image"] = cmd[1];
}
action_cb = start_modify(cfg);
action_cb = p->start_modify(cfg);
}
else if (cmd[0] == "rm-data")
{
// Delete inode data
action_cb = start_rm_data(cfg);
action_cb = p->start_rm_data(cfg);
}
else if (cmd[0] == "rm-osd")
{
@@ -433,7 +398,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
cmd.erase(cmd.begin(), cmd.begin()+1);
cfg["osd_id"] = cmd;
}
action_cb = start_rm_osd(cfg);
action_cb = p->start_rm_osd(cfg);
}
else if (cmd[0] == "merge-data")
{
@@ -444,7 +409,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
if (cmd.size() > 2)
cfg["to"] = cmd[2];
}
action_cb = start_merge(cfg);
action_cb = p->start_merge(cfg);
}
else if (cmd[0] == "flatten")
{
@@ -453,7 +418,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
{
cfg["image"] = cmd[1];
}
action_cb = start_flatten(cfg);
action_cb = p->start_flatten(cfg);
}
else if (cmd[0] == "dd")
{
@@ -467,31 +432,16 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
cfg[arg.substr(0, p)] = arg.substr(p+1);
}
}
action_cb = start_dd(cfg);
action_cb = p->start_dd(cfg);
}
else if (cmd[0] == "rm")
{
// Remove multiple snapshots and rebase their children
if (cfg["names"].is_array())
{
cfg["globs"] = cfg["names"];
cfg.erase("names");
cfg["exact"] = true;
cfg["matching"] = false;
action_cb = start_rm_wildcard(cfg);
}
else if (cfg["matching"].is_array())
{
cfg["globs"] = cfg["matching"];
cfg["exact"] = false;
cfg["matching"] = true;
action_cb = start_rm_wildcard(cfg);
}
else if (cfg["exact"].bool_value() || cfg["matching"].bool_value())
if (cfg["exact"].bool_value() || cfg["matching"].bool_value())
{
cmd.erase(cmd.begin(), cmd.begin()+1);
cfg["globs"] = cmd;
action_cb = start_rm_wildcard(cfg);
action_cb = p->start_rm_wildcard(cfg);
}
else
{
@@ -501,41 +451,41 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
if (cmd.size() > 2)
cfg["to"] = cmd[2];
}
action_cb = start_rm(cfg);
action_cb = p->start_rm(cfg);
}
}
else if (cmd[0] == "describe")
{
// Describe unclean objects
action_cb = start_describe(cfg);
action_cb = p->start_describe(cfg);
}
else if (cmd[0] == "fix")
{
// Fix inconsistent objects (by deleting some copies)
action_cb = start_fix(cfg);
action_cb = p->start_fix(cfg);
}
else if (cmd[0] == "alloc-osd")
{
// Allocate a new OSD number
action_cb = start_alloc_osd(cfg);
action_cb = p->start_alloc_osd(cfg);
}
else if (cmd[0] == "osd-tree")
{
// Print OSD tree
cfg["as_tree"] = true;
action_cb = start_osd_tree(cfg);
action_cb = p->start_osd_tree(cfg);
}
else if (cmd[0] == "osds" || cmd[0] == "ls-osds" || cmd[0] == "ls-osd" || cmd[0] == "osd-ls")
{
// Print OSD list
action_cb = start_osd_tree(cfg);
cfg["flat"] = true;
action_cb = p->start_osd_tree(cfg);
}
else if (cmd[0] == "modify-osd")
{
// Modify OSD configuration
if (cmd.size() > 1)
cfg["osd_num"] = cmd[1];
action_cb = start_modify_osd(cfg);
action_cb = p->start_modify_osd(cfg);
}
else if (cmd[0] == "pg-list" || cmd[0] == "pg-ls" || cmd[0] == "list-pg" || cmd[0] == "ls-pg" || cmd[0] == "ls-pgs" || cmd[0] == "pgs")
{
@@ -545,7 +495,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
cmd.erase(cmd.begin(), cmd.begin()+1);
cfg["pg_state"] = cmd;
}
action_cb = start_pg_list(cfg);
action_cb = p->start_pg_list(cfg);
}
else if (cmd[0] == "create-pool" || cmd[0] == "pool-create")
{
@@ -554,16 +504,16 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
{
cfg["name"] = cmd[1];
}
action_cb = start_pool_create(cfg);
action_cb = p->start_pool_create(cfg);
}
else if (cmd[0] == "modify-pool" || cmd[0] == "pool-modify")
{
// Modify existing pool
if (cmd.size() > 1)
{
cfg["pool"] = cmd[1];
cfg["old_name"] = cmd[1];
}
action_cb = start_pool_modify(cfg);
action_cb = p->start_pool_modify(cfg);
}
else if (cmd[0] == "rm-pool" || cmd[0] == "pool-rm")
{
@@ -572,7 +522,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
{
cfg["pool"] = cmd[1];
}
action_cb = start_pool_rm(cfg);
action_cb = p->start_pool_rm(cfg);
}
else if (cmd[0] == "ls-pool" || cmd[0] == "pool-ls" || cmd[0] == "ls-pools" || cmd[0] == "pools")
{
@@ -583,55 +533,12 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
cmd.erase(cmd.begin(), cmd.begin()+1);
cfg["names"] = cmd;
}
action_cb = start_pool_ls(cfg);
}
else if (cmd[0] == "user-ls" || cmd[0] == "ls-user" || cmd[0] == "ls-users" || cmd[0] == "list-users")
{
// List users
if (cmd.size() > 1)
{
cmd.erase(cmd.begin(), cmd.begin()+1);
cfg["names"] = cmd;
}
action_cb = start_user_ls(cfg);
}
else if (cmd[0] == "modify-user" || cmd[0] == "user-modify")
{
// Create/update user
if (cmd.size() > 1)
{
cfg["name"] = cmd[1];
}
action_cb = start_modify_user(cfg);
}
else if (cmd[0] == "rm-user" || cmd[0] == "remove-user" || cmd[0] == "delete-user")
{
// Remove user
if (cmd.size() > 1)
{
cfg["name"] = cmd[1];
}
cfg["remove"] = true;
action_cb = start_modify_user(cfg);
}
else if (cmd[0] == "serve")
{
// Start HTTP server
action_cb = start_serve(cfg);
action_cb = p->start_pool_ls(cfg);
}
else
{
result = { .err = EOPNOTSUPP, .text = "unknown command: "+cmd[0].string_value() };
result = { .err = EINVAL, .text = "unknown command: "+cmd[0].string_value() };
}
return action_cb;
}
static int run(cli_tool_t *p, json11::Json::object cfg)
{
cli_result_t result = {};
p->is_command_line = true;
p->parse_config(cfg);
auto action_cb = p->start(cfg, result);
if (action_cb != NULL)
{
// Create client
@@ -643,7 +550,6 @@ static int run(cli_tool_t *p, json11::Json::object cfg)
{
result = r;
action_cb = NULL;
p->ringloop->submit();
});
// Loop until it completes
while (action_cb != NULL)
+1 -19
View File
@@ -9,7 +9,6 @@
#include "object_id.h"
#include "ringloop.h"
#include <functional>
#include <set>
struct rm_inode_t;
struct snap_merger_t;
@@ -27,13 +26,6 @@ struct cli_result_t
json11::Json data;
};
struct cli_user_t
{
std::string name;
std::string type;
std::set<std::string> groups;
};
class cli_tool_t
{
public:
@@ -45,8 +37,6 @@ public:
bool is_command_line = false;
bool color = false;
std::unique_ptr<cli_user_t> user; // for http mode
ring_loop_t *ringloop = NULL;
epoll_manager_t *epmgr = NULL;
cluster_client_t *cli = NULL;
@@ -56,24 +46,18 @@ public:
json11::Json etcd_result;
void parse_config(json11::Json::object & cfg);
void parse_api_opts(json11::Json::object & cfg);
json11::Json parse_tags(std::string tags);
json11::Json::object format_image(const inode_config_t & cfg);
void change_parent(inode_t cur, inode_t new_parent, cli_result_t *result);
inode_config_t* get_inode_cfg(const std::string & name);
bool check_image_perm(const inode_config_t & cfg, bool write);
friend struct rm_inode_t;
friend struct snap_merger_t;
friend struct snap_flattener_t;
friend struct snap_remover_t;
std::function<bool(cli_result_t &)> start(json11::Json::object cfg, cli_result_t & result);
std::function<bool(cli_result_t &)> start_alloc_osd(json11::Json);
std::function<bool(cli_result_t &)> start_create(json11::Json);
std::function<bool(cli_result_t &)> start_dd(json11::Json);
std::function<bool(cli_result_t &)> start_describe(json11::Json);
std::function<bool(cli_result_t &)> start_fix(json11::Json);
std::function<bool(cli_result_t &)> start_flatten(json11::Json);
@@ -81,7 +65,6 @@ public:
std::function<bool(cli_result_t &)> start_merge(json11::Json);
std::function<bool(cli_result_t &)> start_modify(json11::Json);
std::function<bool(cli_result_t &)> start_modify_osd(json11::Json);
std::function<bool(cli_result_t &)> start_modify_user(json11::Json);
std::function<bool(cli_result_t &)> start_osd_tree(json11::Json);
std::function<bool(cli_result_t &)> start_pg_list(json11::Json);
std::function<bool(cli_result_t &)> start_pool_create(json11::Json);
@@ -92,9 +75,8 @@ public:
std::function<bool(cli_result_t &)> start_rm_data(json11::Json);
std::function<bool(cli_result_t &)> start_rm_osd(json11::Json);
std::function<bool(cli_result_t &)> start_rm_wildcard(json11::Json);
std::function<bool(cli_result_t &)> start_serve(json11::Json);
std::function<bool(cli_result_t &)> start_status(json11::Json);
std::function<bool(cli_result_t &)> start_user_ls(json11::Json);
std::function<bool(cli_result_t &)> start_dd(json11::Json);
// Should be called like loop_and_wait(start_status(), <completion callback>)
void loop_and_wait(std::function<bool(cli_result_t &)> loop_cb, std::function<void(const cli_result_t &)> complete_cb);
+7 -67
View File
@@ -6,62 +6,6 @@
#include "cluster_client.h"
#include "cli.h"
bool cli_tool_t::check_image_perm(const inode_config_t & cfg, bool write)
{
return !user ||
user->type == "admin" ||
user->name == cfg.owner ||
cfg.owner_group != "" && user->groups.find(cfg.owner_group) != user->groups.end() ||
!write && cfg.reader_group != "" && user->groups.find(cfg.reader_group) != user->groups.end();
}
json11::Json::object cli_tool_t::format_image(const inode_config_t & cfg)
{
auto pool_it = cli->st_cli.pool_config.find(INODE_POOL(cfg.num));
bool good_pool = pool_it != cli->st_cli.pool_config.end();
auto img = json11::Json::object {
{ "name", cfg.name },
{ "size", cfg.size },
{ "inode_id", cfg.num },
{ "inode_num", INODE_NO_POOL(cfg.num) },
{ "pool_id", (uint64_t)INODE_POOL(cfg.num) },
{ "pool_name", good_pool ? pool_it->second.name : "? (ID:"+std::to_string(INODE_POOL(cfg.num))+")" },
{ "readonly", cfg.readonly },
{ "deleted", cfg.deleted },
};
if (cfg.owner != "")
{
img["owner"] = cfg.owner;
}
if (cfg.owner_group != "")
{
img["owner_group"] = cfg.owner_group;
}
if (cfg.reader_group != "")
{
img["reader_group"] = cfg.reader_group;
}
if (!cfg.enc_key.empty())
{
img["encrypted"] = true;
// Only show Vault key IDs
if (cfg.enc_key.substr(0, strlen(VAULT_KEY_PREFIX)) == VAULT_KEY_PREFIX)
img["enc_key_id"] = cfg.enc_key;
}
if (cfg.parent_id)
{
auto parent_it = cli->st_cli.inode_config.find(cfg.parent_id);
if (parent_it != cli->st_cli.inode_config.end())
{
img["parent_name"] = parent_it->second.name;
}
img["parent_inode_id"] = cfg.parent_id;
img["parent_inode_num"] = INODE_NO_POOL(cfg.parent_id);
img["parent_pool_id"] = (uint64_t)INODE_POOL(cfg.parent_id);
}
return img;
}
void cli_tool_t::change_parent(inode_t cur, inode_t new_parent, cli_result_t *result)
{
auto cur_cfg_it = cli->st_cli.inode_config.find(cur);
@@ -157,16 +101,6 @@ inode_config_t* cli_tool_t::get_inode_cfg(const std::string & name)
return NULL;
}
void cli_tool_t::parse_api_opts(json11::Json::object & cfg)
{
iodepth = cfg["iodepth"].uint64_value();
if (!iodepth)
iodepth = 32;
parallel_osds = cfg["parallel_osds"].uint64_value();
if (!parallel_osds)
parallel_osds = 4;
}
void cli_tool_t::parse_config(json11::Json::object & cfg)
{
for (auto kv_it = cfg.begin(); kv_it != cfg.end();)
@@ -187,10 +121,15 @@ void cli_tool_t::parse_config(json11::Json::object & cfg)
else
color = isatty(1);
json_output = cfg["json"].bool_value();
iodepth = cfg["iodepth"].uint64_value();
if (!iodepth)
iodepth = 32;
parallel_osds = cfg["parallel_osds"].uint64_value();
if (!parallel_osds)
parallel_osds = 4;
log_level = cfg["log_level"].int64_value();
progress = cfg["progress"].uint64_value() ? true : false;
list_first = cfg["wait_list"].uint64_value() ? true : false;
parse_api_opts(cfg);
}
struct cli_result_looper_t
@@ -214,6 +153,7 @@ void cli_tool_t::loop_and_wait(std::function<bool(cli_result_t &)> loop_cb, std:
ringloop->unregister_consumer(&looper->consumer);
looper->loop_cb = NULL;
looper->complete_cb(looper->result);
ringloop->submit();
delete looper;
return;
}
+44 -132
View File
@@ -1,13 +1,8 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 (see README.md for details)
#ifdef WITH_OPENSSL
#include <openssl/rand.h>
#endif
#include <ctype.h>
#include "cli.h"
#include "http_client.h"
#include "cluster_client.h"
#include "str_util.h"
@@ -34,15 +29,11 @@ struct image_creator_t
uint64_t size = 0;
bool force = false;
bool force_size = false;
std::string enc_key;
bool set_key = false;
std::string new_owner, new_owner_group, new_reader_group;
pool_id_t old_pool_id = 0;
inode_t new_parent_id = 0;
inode_t new_id = 0, old_id = 0;
uint64_t max_id_mod_rev = 0, idx_mod_rev = 0;
inode_config_t cur_cfg;
uint64_t max_id_mod_rev = 0, cfg_mod_rev = 0, idx_mod_rev = 0;
inode_config_t new_cfg;
int state = 0;
@@ -73,8 +64,7 @@ struct image_creator_t
}
if (new_pool_id)
{
auto pool_it = pools.find(new_pool_id);
if (pool_it == pools.end())
if (pools.find(new_pool_id) == pools.end())
{
result = (cli_result_t){ .err = ENOENT, .text = "Pool "+std::to_string(new_pool_id)+" does not exist" };
state = 100;
@@ -121,23 +111,6 @@ struct image_creator_t
create_snapshot();
}
bool check_pool_permission()
{
if (!parent->user || parent->user->type == "admin")
{
return true;
}
auto pool_it = parent->cli->st_cli.pool_config.find(new_pool_id);
if (pool_it == parent->cli->st_cli.pool_config.end() ||
(pool_it->second.creator_group == "" || parent->user->groups.find(pool_it->second.creator_group) == parent->user->groups.end()))
{
result = (cli_result_t){ .err = EACCES, .text = "Pool image create permission denied" };
state = 100;
return false;
}
return true;
}
void create_image()
{
if (state == 2)
@@ -177,10 +150,6 @@ struct image_creator_t
state = 100;
return;
}
if (!check_pool_permission())
{
return;
}
if (!size && !force_size)
{
result = (cli_result_t){ .err = EINVAL, .text = "Image size is missing" };
@@ -225,11 +194,15 @@ resume_3:
// Save into inode_config for library users to be able to take it from there immediately
new_cfg.mod_revision = parent->etcd_result["header"]["revision"].uint64_value();
parent->cli->st_cli.insert_inode_config(new_cfg);
auto img = parent->format_image(new_cfg);
result = (cli_result_t){
.err = 0,
.text = "Image "+image_name+" created",
.data = img,
.data = json11::Json::object {
{ "name", image_name },
{ "pool", new_pool_name },
{ "parent", new_parent },
{ "size", size },
}
};
state = 100;
}
@@ -260,7 +233,7 @@ resume_3:
}
do
{
// In addition to next_id, get: cur_cfg, old_id, old_pool_id, size, idx_mod_rev
// In addition to next_id, get: size, old_id, old_pool_id, new_parent, cfg_mod_rev, idx_mod_rev
resume_2:
resume_3:
get_image_details();
@@ -272,22 +245,11 @@ resume_3:
state = 100;
return;
}
if (!parent->check_image_perm(cur_cfg, true))
{
result = (cli_result_t){ .err = EACCES, .text = "Image permission denied" };
state = 100;
return;
}
if (!new_pool_id)
{
// Create snapshot in the same pool by default
new_pool_id = old_pool_id;
}
// Verify pool permissions if the pool is different from the original
if (new_pool_id != old_pool_id && !check_pool_permission())
{
return;
}
attempt_create();
state = 4;
resume_4:
@@ -310,23 +272,13 @@ resume_4:
// Save into inode_config for library users to be able to take it from there immediately
new_cfg.mod_revision = parent->etcd_result["header"]["revision"].uint64_value();
parent->cli->st_cli.insert_inode_config(new_cfg);
{
auto new_pool_it = parent->cli->st_cli.pool_config.find(new_pool_id);
new_pool_name = new_pool_it != parent->cli->st_cli.pool_config.end() ? new_pool_it->second.name : "";
}
result = (cli_result_t){
.err = 0,
.text = "Snapshot "+image_name+"@"+new_snap+" created",
.data = json11::Json::object {
{ "inode_id", INODE_WITH_POOL(new_pool_id, new_id) },
{ "inode_num", new_id },
{ "name", image_name },
{ "pool_id", (uint64_t)new_pool_id },
{ "pool_name", new_pool_name },
{ "parent_name", image_name+"@"+new_snap },
{ "parent_inode_id", INODE_WITH_POOL(old_pool_id, old_id) },
{ "parent_inode_num", old_id },
{ "parent_pool_id", (uint64_t)old_pool_id },
{ "name", image_name+"@"+new_snap },
{ "pool", (uint64_t)new_pool_id },
{ "parent", new_parent },
{ "size", size },
}
};
@@ -371,6 +323,17 @@ resume_4:
goto resume_2;
else if (state == 3)
goto resume_3;
if (!new_pool_id)
{
for (auto & ic: parent->cli->st_cli.inode_config)
{
if (ic.second.name == image_name)
{
new_pool_id = INODE_POOL(ic.first);
break;
}
}
}
parent->etcd_txn(json11::Json::object { { "success", json11::Json::array {
get_next_id(),
json11::Json::object {
@@ -394,7 +357,7 @@ resume_2:
extract_next_id(parent->etcd_result["responses"][0]);
old_id = 0;
old_pool_id = 0;
idx_mod_rev = 0;
cfg_mod_rev = idx_mod_rev = 0;
if (parent->etcd_result["responses"][1]["response_range"]["kvs"].array_items().size() == 0)
{
for (auto & ic: parent->cli->st_cli.inode_config)
@@ -403,8 +366,9 @@ resume_2:
{
old_id = INODE_NO_POOL(ic.first);
old_pool_id = INODE_POOL(ic.first);
cur_cfg = ic.second;
size = ic.second.size;
new_parent_id = ic.second.parent_id;
cfg_mod_rev = ic.second.mod_revision;
break;
}
}
@@ -448,14 +412,16 @@ resume_3:
}
{
auto kv = parent->cli->st_cli.parse_etcd_kv(parent->etcd_result["responses"][0]["response_range"]["kvs"][0]);
cur_cfg = parent->cli->st_cli.deserialize_inode_cfg(INODE_WITH_POOL(old_pool_id, old_id), kv.value, kv.mod_revision);
size = cur_cfg.size;
size = kv.value["size"].uint64_value();
new_parent_id = kv.value["parent_id"].uint64_value();
uint64_t parent_pool_id = kv.value["parent_pool"].uint64_value();
if (new_parent_id)
{
new_parent_id = INODE_WITH_POOL(parent_pool_id ? parent_pool_id : old_pool_id, new_parent_id);
}
cfg_mod_rev = kv.mod_revision;
}
}
if (!new_pool_id)
{
new_pool_id = old_pool_id;
}
}
void attempt_create()
@@ -468,27 +434,6 @@ resume_3:
.readonly = false,
.meta = new_meta,
};
if (set_key)
{
new_cfg.enc_key = enc_key;
}
else if (new_snap != "")
{
new_cfg.enc_key = cur_cfg.enc_key;
}
new_cfg.owner = http_context_get_ssl_cn(parent->cli->st_cli.get_http_ctx());
if (!new_owner.empty())
{
new_cfg.owner = new_owner;
}
if (!new_owner_group.empty())
{
new_cfg.owner_group = new_owner_group;
}
if (!new_reader_group.empty())
{
new_cfg.reader_group = new_reader_group;
}
json11::Json::array checks = json11::Json::array {
json11::Json::object {
{ "target", "VERSION" },
@@ -555,12 +500,16 @@ resume_3:
};
if (new_snap != "")
{
inode_config_t snap_cfg = cur_cfg;
snap_cfg.name = image_name+"@"+new_snap;
snap_cfg.readonly = true;
inode_config_t snap_cfg = {
.num = INODE_WITH_POOL(old_pool_id, old_id),
.name = image_name+"@"+new_snap,
.size = size,
.parent_id = new_parent_id,
.readonly = true,
};
checks.push_back(json11::Json::object {
{ "target", "MOD" },
{ "mod_revision", cur_cfg.mod_revision },
{ "mod_revision", cfg_mod_rev },
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/config/inode/"+
std::to_string(old_pool_id)+"/"+std::to_string(old_id)
@@ -605,16 +554,8 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_create(json11::Json cfg)
auto image_creator = new image_creator_t();
image_creator->parent = this;
image_creator->image_name = cfg["image"].string_value();
if (!cfg["pool"].is_null())
{
image_creator->new_pool_id = cfg["pool"].uint64_value();
image_creator->new_pool_name = cfg["pool"].string_value();
}
else
{
image_creator->new_pool_id = cfg["pool_id"].uint64_value();
image_creator->new_pool_name = cfg["pool_name"].string_value();
}
image_creator->new_pool_id = cfg["pool"].uint64_value();
image_creator->new_pool_name = cfg["pool"].string_value();
image_creator->force = cfg["force"].bool_value();
image_creator->force_size = cfg["force_size"].bool_value();
if (cfg["image_meta"].is_object())
@@ -625,35 +566,6 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_create(json11::Json cfg)
{
image_creator->new_snap = cfg["snapshot"].string_value();
}
if (!cfg["enc_key"].is_null())
{
image_creator->set_key = true;
#ifdef WITH_OPENSSL
if (image_creator->enc_key == "random")
{
uint8_t newkey[64];
RAND_bytes(newkey, 64);
image_creator->enc_key = tohexstr(newkey, 64);
}
#endif
else
{
image_creator->enc_key = cfg["enc_key"].string_value();
if (image_creator->enc_key != "" &&
image_creator->enc_key.substr(0, strlen(VAULT_KEY_PREFIX)) != VAULT_KEY_PREFIX &&
(!ishexstr(image_creator->enc_key) || image_creator->enc_key.size() != 128))
{
return [](cli_result_t & result)
{
result = (cli_result_t){ .err = EINVAL, .text = "Encryption key is not a 512-bit hex string, not \"\" and not \"random\"" };
return true;
};
}
}
}
image_creator->new_owner = cfg["owner"].string_value();
image_creator->new_owner_group = cfg["owner_group"].string_value();
image_creator->new_reader_group = cfg["reader_group"].string_value();
image_creator->new_parent = cfg["parent"].string_value();
if (!cfg["size"].is_null())
{
+1 -1
View File
@@ -864,7 +864,7 @@ resume_2:
// Copy data
if (iinfo.in_seekable && iseek >= iinfo.in_size)
{
result = (cli_result_t){ .err = EINVAL, .text = "Input seek position is beyond end of input" };
result = (cli_result_t){ .err = -EINVAL, .text = "Input seek position is beyond end of input" };
goto close_end;
}
if (!iinfo.iwatch && !iinfo.in_seekable && iseek)
+3 -31
View File
@@ -57,24 +57,12 @@ struct cli_describe_t
void parse_options(json11::Json cfg)
{
uint64_t pool_id;
std::string pool_name;
if (!cfg["pool"].is_null())
{
pool_id = cfg["pool"].uint64_value();
pool_name = pool_id ? "" : cfg["pool"].string_value();
}
else
{
pool_id = cfg["pool_id"].uint64_value();
pool_name = pool_id ? "" : cfg["pool_name"].string_value();
}
only_pool = pool_id;
if (!only_pool && pool_name != "")
only_pool = cfg["pool"].uint64_value();
if (!only_pool && cfg["pool"].is_string())
{
for (auto & pp: parent->cli->st_cli.pool_config)
{
if (pp.second.name == pool_name)
if (pp.second.name == cfg["pool"].string_value())
{
only_pool = pp.first;
break;
@@ -118,22 +106,6 @@ struct cli_describe_t
if (cfg["object_state"].string_value().find("misplaced") != std::string::npos)
object_state |= OBJ_MISPLACED;
}
else if (!object_state && cfg["object_state"].is_array())
{
for (auto & st: cfg["object_state"].array_items())
{
if (st == "inconsistent")
object_state |= OBJ_INCONSISTENT;
else if (st == "corrupted")
object_state |= OBJ_CORRUPTED;
else if (st == "incomplete")
object_state |= OBJ_INCOMPLETE;
else if (st == "degraded")
object_state |= OBJ_DEGRADED;
else if (st == "misplaced")
object_state |= OBJ_MISPLACED;
}
}
}
void loop()
-6
View File
@@ -35,12 +35,6 @@ struct snap_flattener_t
state = 100;
return;
}
if (!parent->check_image_perm(*target_cfg, true))
{
result = (cli_result_t){ .err = EACCES, .text = "Image permission denied" };
state = 100;
return;
}
target_id = target_cfg->num;
std::vector<inode_t> chain_list;
inode_config_t *cur = target_cfg;
+41 -63
View File
@@ -17,7 +17,6 @@ struct image_lister_t
std::string list_pool_name;
std::string sort_field;
std::set<std::string> only_names;
std::vector<uint64_t> only_ids;
bool reverse = false;
bool exact = false;
bool tree = false;
@@ -53,19 +52,35 @@ struct image_lister_t
return;
}
}
auto begin_it = list_pool_id
? parent->cli->st_cli.inode_config.lower_bound(INODE_WITH_POOL(list_pool_id, 0))
: parent->cli->st_cli.inode_config.begin();
auto end_it = list_pool_id
? parent->cli->st_cli.inode_config.lower_bound(INODE_WITH_POOL(list_pool_id+1, 0))
: parent->cli->st_cli.inode_config.end();
for (auto it = begin_it; it != end_it; it++)
for (auto & ic: parent->cli->st_cli.inode_config)
{
if (!parent->check_image_perm(it->second, false))
if (list_pool_id && INODE_POOL(ic.second.num) != list_pool_id)
{
continue;
}
stats[it->second.num] = parent->format_image(it->second);
auto pool_it = parent->cli->st_cli.pool_config.find(INODE_POOL(ic.second.num));
bool good_pool = pool_it != parent->cli->st_cli.pool_config.end();
auto item = json11::Json::object {
{ "name", ic.second.name },
{ "size", ic.second.size },
{ "used_size", 0 },
{ "readonly", ic.second.readonly },
{ "pool_id", (uint64_t)INODE_POOL(ic.second.num) },
{ "pool_name", good_pool ? pool_it->second.name : "? (ID:"+std::to_string(INODE_POOL(ic.second.num))+")" },
{ "inode_num", INODE_NO_POOL(ic.second.num) },
{ "inode_id", ic.second.num },
{ "deleted", ic.second.deleted },
};
if (ic.second.parent_id)
{
auto p_it = parent->cli->st_cli.inode_config.find(ic.second.parent_id);
item["parent_name"] = p_it != parent->cli->st_cli.inode_config.end()
? p_it->second.name : "";
item["parent_pool_id"] = (uint64_t)INODE_POOL(ic.second.parent_id);
item["parent_inode_num"] = INODE_NO_POOL(ic.second.parent_id);
item["parent_inode_id"] = ic.second.parent_id;
}
stats[ic.second.num] = item;
}
}
@@ -112,7 +127,6 @@ resume_1:
state = 100;
return;
}
// FIXME: Do not always read everything
space_info = parent->etcd_result;
std::map<pool_id_t, uint64_t> pool_pg_real_size;
for (auto & kv_item: space_info["responses"][0]["response_range"]["kvs"].array_items())
@@ -146,11 +160,6 @@ resume_1:
}
inode_t inode_num = INODE_WITH_POOL(pool_id, only_inode_num);
uint64_t used_size = kv.value["raw_used"].uint64_value();
auto stat_it = stats.find(inode_num);
if (parent->user && parent->user->type != "admin" && stat_it == stats.end())
{
continue;
}
// save stats
auto pool_it = parent->cli->st_cli.pool_config.find(pool_id);
if (pool_it != parent->cli->st_cli.pool_config.end())
@@ -159,6 +168,7 @@ resume_1:
used_size = used_size / (pool_pg_real_size[pool_id] ? pool_pg_real_size[pool_id] : 1)
* (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
}
auto stat_it = stats.find(inode_num);
if (stat_it == stats.end())
{
stats[inode_num] = json11::Json::object {
@@ -192,33 +202,20 @@ resume_1:
json11::Json::array to_list()
{
json11::Json::array list;
if (only_ids.size())
for (auto & kv: stats)
{
for (auto & id: only_ids)
if (!only_names.size())
{
if (stats.find(id) != stats.end())
{
list.push_back(stats[id]);
}
list.push_back(kv.second);
}
}
else
{
for (auto & kv: stats)
else
{
if (!only_names.size())
for (auto & glob: only_names)
{
list.push_back(kv.second);
}
else
{
for (auto & glob: only_names)
if (exact ? (kv.second["name"].string_value() == glob) : stupid_glob(kv.second["name"].string_value(), glob))
{
if (exact ? (kv.second["name"].string_value() == glob) : stupid_glob(kv.second["name"].string_value(), glob))
{
list.push_back(kv.second);
break;
}
list.push_back(kv.second);
break;
}
}
}
@@ -374,7 +371,7 @@ resume_1:
}
}
cols.push_back(json11::Json::object{
{ "key", "flags" },
{ "key", "ro" },
{ "title", "FLAGS" },
{ "right", true },
});
@@ -402,14 +399,8 @@ resume_1:
kv.second["delete_q"] = format_q(kv.second["delete_queue"].number_value());
}
kv.second["size_fmt"] = format_size(kv.second["size"].uint64_value());
std::string flags;
if (kv.second["deleted"].bool_value())
flags += "DEL";
if (kv.second["readonly"].bool_value())
flags += (flags.empty() ? "RO" : ",RO");
if (kv.second["encrypted"].bool_value())
flags += (flags.empty() ? "ENC" : ",ENC");
kv.second["flags"] = flags;
kv.second["ro"] = kv.second["deleted"].bool_value() ? "DEL" :
(kv.second["readonly"].bool_value() ? "RO" : "-");
}
result.text = print_table(tree ? to_tree(to_list()) : to_list(), cols, parent->color);
state = 100;
@@ -579,30 +570,17 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_ls(json11::Json cfg)
lister->parent = this;
lister->exact = cfg["exact"].bool_value();
lister->tree = cfg["tree"].bool_value();
if (!cfg["pool"].is_null())
{
lister->list_pool_id = cfg["pool"].uint64_value();
lister->list_pool_name = lister->list_pool_id ? "" : cfg["pool"].as_string();
}
else
{
lister->list_pool_id = cfg["pool_id"].uint64_value();
lister->list_pool_name = lister->list_pool_id ? "" : cfg["pool_name"].string_value();
}
lister->list_pool_id = cfg["pool"].uint64_value();
lister->list_pool_name = lister->list_pool_id ? "" : cfg["pool"].as_string();
lister->show_stats = cfg["long"].bool_value();
lister->show_delete = cfg["del"].bool_value();
lister->sort_field = cfg["sort"].string_value() != "" ? cfg["sort"].string_value() : "name";
lister->reverse = cfg["reverse"].bool_value();
lister->max_count = cfg["count"].uint64_value();
if (cfg["names"].is_string())
lister->only_names.insert(cfg["names"].string_value());
for (auto & item: cfg["names"].array_items())
{
lister->only_names.insert(item.string_value());
if (cfg["ids"].is_string())
for (auto & item: explode(",", cfg["ids"].string_value(), true))
lister->only_ids.push_back(stoull_full(item));
for (auto & item: cfg["ids"].array_items())
lister->only_ids.push_back(item.uint64_value());
}
return [lister](cli_result_t & result)
{
lister->loop();
+1 -1
View File
@@ -374,7 +374,7 @@ struct snap_merger_t
result = (cli_result_t){ .text = "Done, layers from "+from_name+" to "+to_name+" merged into "+target_name, .data = json11::Json::object {
{ "from", from_name },
{ "to", to_name },
{ "target", target_name },
{ "into", target_name },
}};
state = 100;
resume_100:
+10 -46
View File
@@ -17,9 +17,6 @@ struct image_changer_t
bool force_size = false, inc_size = false;
bool set_readonly = false, set_readwrite = false, force = false;
bool set_deleted = false, new_deleted = false;
bool set_key = false;
std::string enc_key;
json11::Json new_owner, new_owner_group, new_reader_group;
bool down_ok = false;
// interval between fsyncs
int fsync_interval = 128;
@@ -77,12 +74,6 @@ struct image_changer_t
state = 100;
return;
}
if (!parent->check_image_perm(cfg, true))
{
result = (cli_result_t){ .err = EACCES, .text = "Image permission denied" };
state = 100;
return;
}
for (auto & ic: parent->cli->st_cli.inode_config)
{
if (ic.second.parent_id == inode_num)
@@ -97,7 +88,10 @@ struct image_changer_t
(!new_size && !force_size || cfg.size == new_size || cfg.size >= new_size && inc_size) &&
(new_name == "" || new_name == image_name))
{
result = (cli_result_t){ .err = 0, .text = "No change", .data = parent->format_image(cfg) };
result = (cli_result_t){ .err = 0, .text = "No change", .data = json11::Json::object {
{ "error_code", 0 },
{ "error_text", "No change" },
}};
state = 100;
return;
}
@@ -158,36 +152,6 @@ resume_1:
{
cfg.name = new_name;
}
if (new_owner.is_string())
{
cfg.owner = new_owner.string_value();
}
if (new_owner_group.is_string())
{
cfg.owner_group = new_owner_group.string_value();
}
if (new_reader_group.is_string())
{
cfg.reader_group = new_reader_group.string_value();
}
if (set_key)
{
if (!force)
{
result = (cli_result_t){ .err = EINVAL, .text = "Changing image encryption key is only allowed with --force" };
state = 100;
return;
}
if (enc_key != "" &&
enc_key.substr(0, strlen(VAULT_KEY_PREFIX)) != VAULT_KEY_PREFIX &&
(!ishexstr(enc_key) || enc_key.size() != 128))
{
result = (cli_result_t){ .err = EINVAL, .text = "Encryption key is not a 512-bit hex string and not \"\"" };
state = 100;
return;
}
cfg.enc_key = enc_key;
}
{
std::string cur_cfg_key = base64_encode(parent->cli->st_cli.etcd_prefix+
"/config/inode/"+std::to_string(INODE_POOL(inode_num))+
@@ -271,7 +235,12 @@ resume_2:
result = (cli_result_t){
.err = 0,
.text = "Image "+image_name+" modified",
.data = parent->format_image(cfg)
.data = json11::Json::object {
{ "name", image_name },
{ "inode", INODE_NO_POOL(inode_num) },
{ "pool", (uint64_t)INODE_POOL(inode_num) },
{ "size", new_size },
}
};
state = 100;
}
@@ -292,14 +261,9 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_modify(json11::Json cfg)
changer->set_deleted = !cfg["deleted"].is_null();
changer->new_deleted = json_is_true(cfg["deleted"]);
changer->fsync_interval = cfg["fsync_interval"].uint64_value();
changer->enc_key = cfg["enc_key"].string_value();
changer->set_key = cfg["enc_key"].is_string();
if (!changer->fsync_interval)
changer->fsync_interval = 128;
changer->down_ok = cfg["down_ok"].bool_value();
changer->new_owner = cfg["owner"];
changer->new_owner_group = cfg["owner_group"];
changer->new_reader_group = cfg["reader_group"];
// FIXME Check that the image doesn't have children when shrinking
return [changer](cli_result_t & result)
{
-161
View File
@@ -1,161 +0,0 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 (see README.md for details)
#include "cli.h"
#include "cluster_client.h"
#include "str_util.h"
// Create/update/delete a user
struct cli_modify_user_t
{
cli_tool_t *parent;
std::string user_name;
std::string user_type;
json11::Json groups;
bool del = false;
int state = 0;
cli_result_t result;
etcd_kv_t kv;
json11::Json::object new_cfg;
bool is_done()
{
return state == 100;
}
void loop()
{
if (state == 1)
goto resume_1;
else if (state == 2)
goto resume_2;
if (user_type != "client" && user_type != "admin" && user_type != "mon" && user_type != "osd")
{
result = (cli_result_t){ .err = EINVAL, .text = "Unknown user type: "+user_type };
state = 100;
return;
}
if (groups.is_string())
{
groups = groups == "" ? std::vector<std::string>() : explode(",", groups.string_value(), true);
}
else if (groups.is_array())
{
for (auto & gr: groups.array_items())
{
if (!gr.is_string())
{
result = (cli_result_t){ .err = EINVAL, .text = "Group names must be strings" };
state = 100;
return;
}
}
}
else if (!groups.is_null())
{
result = (cli_result_t){ .err = EINVAL, .text = "Group names must be strings" };
state = 100;
return;
}
if (user_name == "")
{
result = (cli_result_t){ .err = EINVAL, .text = "User name must not be empty" };
state = 100;
return;
}
parent->etcd_txn(json11::Json::object {
{ "success", json11::Json::array { json11::Json::object {
{ "request_range", json11::Json::object {
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/config/user/"+user_name) },
} },
} } }
});
state = 1;
resume_1:
if (parent->waiting > 0)
return;
if (parent->etcd_err.err)
{
result = parent->etcd_err;
state = 100;
return;
}
kv = parent->cli->st_cli.parse_etcd_kv(parent->etcd_result["responses"][0]["response_range"]["kvs"][0]);
while (true)
{
new_cfg = kv.value.object_items();
if (!kv.mod_revision && del)
{
result = (cli_result_t){ .err = ENOENT, .text = "User "+user_name+" does not exist" };
state = 100;
break;
}
if (!groups.is_null())
new_cfg["groups"] = groups;
if (user_type != "")
new_cfg["type"] = user_type;
if (!new_cfg["type"].is_string())
new_cfg["type"] = "client";
parent->etcd_txn(json11::Json::object {
{ "compare", json11::Json::array { json11::Json::object {
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/config/user/"+user_name) },
{ "target", kv.mod_revision ? "MOD" : "VERSION" },
{ kv.mod_revision ? "mod_revision" : "version", kv.mod_revision },
} } },
{ "success", json11::Json::array {
del ? json11::Json::object { { "request_delete_range", json11::Json::object {
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/config/user/"+user_name) },
} } } : json11::Json::object { { "request_put", json11::Json::object {
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/config/user/"+user_name) },
{ "value", base64_encode(json11::Json(new_cfg).dump()) }
} } },
} },
{ "failure", json11::Json::array { json11::Json::object {
{ "request_range", json11::Json::object {
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/config/user/"+user_name) },
} },
} } },
});
state = 2;
resume_2:
if (parent->waiting > 0)
return;
if (parent->etcd_err.err)
{
result = parent->etcd_err;
state = 100;
return;
}
if (parent->etcd_result["succeeded"].bool_value())
break;
kv = parent->cli->st_cli.parse_etcd_kv(parent->etcd_result["responses"][0]["response_range"]["kvs"][0]);
}
state = 100;
result.text = del ? "User "+user_name+" removed" : "User "+user_name+" modified";
new_cfg["name"] = user_name;
result.data = del ? json11::Json::object{ { "ok", true } } : new_cfg;
}
};
std::function<bool(cli_result_t &)> cli_tool_t::start_modify_user(json11::Json cfg)
{
auto creator = new cli_modify_user_t();
creator->parent = this;
creator->user_name = cfg["name"].string_value();
creator->user_type = cfg["type"].string_value();
creator->groups = cfg["groups"];
creator->del = cfg["remove"].bool_value();
return [creator](cli_result_t & result)
{
creator->loop();
if (creator->is_done())
{
result = creator->result;
delete creator;
return true;
}
return false;
};
}
+18 -26
View File
@@ -41,7 +41,7 @@ struct osd_tree_printer_t
{
cli_tool_t *parent;
json11::Json cfg;
bool as_tree = false;
bool flat = false;
bool show_stats = false;
int state = 0;
@@ -209,14 +209,11 @@ resume_1:
for (int i = 1; i < node_seq.size(); i++)
{
auto & node = placement_tree->nodes.at(node_seq[i]);
if (as_tree)
{
fmt_items.push_back(json11::Json::object{
{ "type", node.level },
{ "name", node.name },
{ "parent", node.parent },
});
}
fmt_items.push_back(json11::Json::object{
{ "type", node.level },
{ "name", node.name },
{ "parent", node.parent },
});
for (uint64_t osd_num: node.child_osds)
{
auto & osd = placement_tree->osds.at(osd_num);
@@ -224,22 +221,17 @@ resume_1:
{ "type", "osd" },
{ "name", osd.num },
{ "parent", node.name },
{ "up", osd.up },
{ "up", osd.up ? "up" : "down" },
{ "size", osd.size },
{ "free", osd.free },
{ "reweight", osd.reweight },
{ "noout", osd.noout },
{ "tags", osd.tags },
{ "data_block_size", (uint64_t)osd.block_size },
{ "bitmap_granularity", (uint64_t)osd.bitmap_granularity },
{ "immediate_commit", osd.immediate_commit == IMMEDIATE_NONE ? "none" : (osd.immediate_commit == IMMEDIATE_ALL ? "all" : "small") },
{ "block", (uint64_t)osd.block_size },
{ "bitmap", (uint64_t)osd.bitmap_granularity },
{ "commit", osd.immediate_commit == IMMEDIATE_NONE ? "none" : (osd.immediate_commit == IMMEDIATE_ALL ? "all" : "small") },
{ "op_stats", osd_stats[osd_num]["op_stats"] },
};
if (show_stats)
{
json_osd["op_stats"] = osd_stats[osd_num]["op_stats"];
json_osd["subop_stats"] = osd_stats[osd_num]["subop_stats"];
json_osd["recovery_stats"] = osd_stats[osd_num]["recovery_stats"];
}
if (osd_stats[osd_num]["slow_ops_primary"].uint64_value() > 0)
{
json_osd["slow_ops_primary"] = osd_stats[osd_num]["slow_ops_primary"];
@@ -257,7 +249,7 @@ resume_1:
for (int i = 1; i < node_seq.size(); i++)
{
auto & node = placement_tree->nodes.at(node_seq[i]);
if (as_tree)
if (!flat)
{
fmt_items.push_back(json11::Json::object{
{ "type", str_repeat(" ", indents[i]) + node.level },
@@ -265,7 +257,7 @@ resume_1:
});
}
std::string parent = node.name;
if (!as_tree)
if (flat)
{
auto cur = &placement_tree->nodes.at(node.name);
while (cur->parent != "" && cur->parent != node.name)
@@ -278,7 +270,7 @@ resume_1:
{
auto & osd = placement_tree->osds.at(osd_num);
auto fmt = json11::Json::object{
{ "type", (!as_tree ? "osd" : str_repeat(" ", indents[i]+1) + "osd") },
{ "type", (flat ? "osd" : str_repeat(" ", indents[i]+1) + "osd") },
{ "name", osd.num },
{ "parent", parent },
{ "up", osd.up ? "up" : "down" },
@@ -308,7 +300,7 @@ resume_1:
}
}
json11::Json::array cols;
if (as_tree)
if (!flat)
{
cols.push_back(json11::Json::object{
{ "key", "type" },
@@ -317,9 +309,9 @@ resume_1:
}
cols.push_back(json11::Json::object{
{ "key", "name" },
{ "title", !as_tree ? "OSD" : "NAME" },
{ "title", flat ? "OSD" : "NAME" },
});
if (!as_tree)
if (flat)
{
cols.push_back(json11::Json::object{
{ "key", "parent" },
@@ -422,7 +414,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_osd_tree(json11::Json cfg)
auto osd_tree_printer = new osd_tree_printer_t();
osd_tree_printer->parent = this;
osd_tree_printer->cfg = cfg;
osd_tree_printer->as_tree = cfg["as_tree"].bool_value();
osd_tree_printer->flat = cfg["flat"].bool_value();
osd_tree_printer->show_stats = cfg["long"].bool_value();
return [osd_tree_printer](cli_result_t & result)
{
+2 -8
View File
@@ -282,16 +282,10 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_pg_list(json11::Json cfg)
{
auto pg_lister = new pg_lister_t();
pg_lister->parent = this;
if (!cfg["pool"].is_null())
{
if (cfg["pool"].uint64_value())
pg_lister->pool_id = cfg["pool"].uint64_value();
pg_lister->pool_name = pg_lister->pool_id ? "" : cfg["pool"].string_value();
}
else
{
pg_lister->pool_id = cfg["pool_id"].uint64_value();
pg_lister->pool_name = pg_lister->pool_id ? "" : cfg["pool_name"].string_value();
}
pg_lister->pool_name = cfg["pool"].string_value();
for (auto & st: cfg["pg_state"].array_items())
pg_lister->pg_state.push_back(st.string_value());
if (cfg["pg_state"].is_string())
+1 -1
View File
@@ -91,7 +91,7 @@ std::string validate_pool_config(json11::Json::object & new_cfg, json11::Json ol
}
else if (key == "name" || key == "scheme" || key == "immediate_commit" ||
key == "failure_domain" || key == "root_node" || key == "scrub_interval" || key == "used_for_app" ||
key == "used_for_fs" || key == "raw_placement" || key == "local_reads" || key == "creator_group")
key == "used_for_fs" || key == "raw_placement" || key == "local_reads")
{
if (!value.is_string())
{
+1 -1
View File
@@ -213,7 +213,7 @@ resume_3:
if (failure_domain != "osd")
pool_err += "\n- different parent '"+failure_domain+"' nodes";
result = (cli_result_t){
.err = EBUSY,
.err = EINVAL,
.text = pool_err,
};
state = 100;
+6 -7
View File
@@ -206,7 +206,7 @@ resume_1:
{ "space_efficiency", pool_stats[pool_cfg.id]["space_efficiency"].number_value() },
{ "pg_real_size", pool_stats[pool_cfg.id]["pg_real_size"].uint64_value() },
{ "osd_count", (uint64_t)pg_per_osd.size() },
{ "backfillfull", !!pool_cfg.backfillfull },
{ "backfillfull", pool_cfg.backfillfull },
};
}
// Include full pool config
@@ -546,10 +546,6 @@ resume_3:
{ "write_fmt", "Write" },
{ "delete_fmt", "Delete" },
};
if (sort_field == "osd_tags" || sort_field == "primary_affinity_tags")
{
sort_field += "_fmt";
}
auto list = to_list();
size_t title_len = 0;
for (auto & item: list)
@@ -670,12 +666,15 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_pool_ls(json11::Json cfg)
lister->show_stats = cfg["long"].bool_value();
lister->detailed = cfg["detail"].bool_value();
lister->sort_field = cfg["sort"].string_value();
if ((lister->sort_field == "osd_tags") ||
(lister->sort_field == "primary_affinity_tags" ))
lister->sort_field = lister->sort_field + "_fmt";
lister->reverse = cfg["reverse"].bool_value();
lister->max_count = cfg["count"].uint64_value();
if (cfg["names"].is_string())
lister->only_names.insert(cfg["names"].string_value());
for (auto & item: cfg["names"].array_items())
{
lister->only_names.insert(item.string_value());
}
return [lister](cli_result_t & result)
{
lister->loop();

Some files were not shown because too many files have changed in this diff Show More