Compare commits

..
Author SHA1 Message Date
Vitaliy Filippov 68985bee0a WIP RDMA credits 2025-12-12 18:11:36 +00:00
236 changed files with 2465 additions and 17211 deletions
+7 -8
View File
@@ -1,29 +1,28 @@
FROM node:16-bookworm FROM node:16-bullseye
WORKDIR /root WORKDIR /root
ADD ./docker/etc/apt/trusted.gpg.d /etc/apt/trusted.gpg.d ADD ./docker/vitastor.gpg /etc/apt/trusted.gpg.d
RUN echo 'deb http://deb.debian.org/debian bookworm-backports main' >> /etc/apt/sources.list; \ RUN echo 'deb http://deb.debian.org/debian bullseye-backports main' >> /etc/apt/sources.list; \
echo 'deb http://vitastor.io/debian bookworm main' >> /etc/apt/sources.list; \ echo 'deb http://vitastor.io/debian bullseye main' >> /etc/apt/sources.list; \
echo >> /etc/apt/preferences; \ echo >> /etc/apt/preferences; \
echo 'Package: *' >> /etc/apt/preferences; \ echo 'Package: *' >> /etc/apt/preferences; \
echo 'Pin: release n=bookworm-backports' >> /etc/apt/preferences; \ echo 'Pin: release a=bullseye-backports' >> /etc/apt/preferences; \
echo 'Pin-Priority: 500' >> /etc/apt/preferences; \ echo 'Pin-Priority: 500' >> /etc/apt/preferences; \
echo >> /etc/apt/preferences; \ echo >> /etc/apt/preferences; \
echo 'Package: *' >> /etc/apt/preferences; \ echo 'Package: *' >> /etc/apt/preferences; \
echo 'Pin: origin "vitastor.io"' >> /etc/apt/preferences; \ echo 'Pin: origin "vitastor.io"' >> /etc/apt/preferences; \
echo 'Pin-Priority: 1000' >> /etc/apt/preferences; \ echo 'Pin-Priority: 1000' >> /etc/apt/preferences; \
perl -i -pe 's/Types: deb$/Types: deb deb-src/' /etc/apt/sources.list.d/debian.sources; \
grep '^deb ' /etc/apt/sources.list | perl -pe 's/^deb/deb-src/' >> /etc/apt/sources.list; \ grep '^deb ' /etc/apt/sources.list | perl -pe 's/^deb/deb-src/' >> /etc/apt/sources.list; \
echo 'APT::Install-Recommends false;' >> /etc/apt/apt.conf; \ echo 'APT::Install-Recommends false;' >> /etc/apt/apt.conf; \
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
RUN apt-get update RUN apt-get update
RUN apt-get -y install etcd qemu-system-x86 qemu-block-extra qemu-utils fio libasan8 \ RUN apt-get -y install etcd qemu-system-x86 qemu-block-extra qemu-utils fio libasan5 \
libgoogle-perftools-dev devscripts libjerasure-dev cmake libibverbs-dev libisal-dev libgoogle-perftools-dev devscripts libjerasure-dev cmake libibverbs-dev libisal-dev
RUN apt-get -y build-dep fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'` RUN apt-get -y build-dep fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
RUN apt-get update && apt-get -y install jq lp-solve sudo nfs-common fdisk parted libc-ares-dev udev RUN apt-get update && apt-get -y install jq lp-solve sudo nfs-common fdisk parted
RUN apt-get --download-only source fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'` RUN apt-get --download-only source fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
RUN set -ex; \ RUN set -ex; \
-252
View File
@@ -234,60 +234,6 @@ jobs:
echo "" echo ""
done done
test_etcd_fail_https:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 10
run: ETCD_SCHEME=https /root/vitastor/tests/test_etcd_fail.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_etcd_fail_https_antietcd:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 10
run: ETCD_SCHEME=https ANTIETCD=1 /root/vitastor/tests/test_etcd_fail.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_snapshot_https:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: ETCD_SCHEME=https /root/vitastor/tests/test_snapshot.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_interrupted_rebalance: test_interrupted_rebalance:
runs-on: ubuntu-latest runs-on: ubuntu-latest
needs: build needs: build
@@ -468,24 +414,6 @@ jobs:
echo "" echo ""
done done
test_level_placement:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: /root/vitastor/tests/test_level_placement.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_snapshot: test_snapshot:
runs-on: ubuntu-latest runs-on: ubuntu-latest
needs: build needs: build
@@ -702,24 +630,6 @@ jobs:
echo "" echo ""
done done
test_snapshot_chain_encrypted:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: ENCRYPTED=1 /root/vitastor/tests/test_snapshot_chain.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_old_snapshot_chain: test_old_snapshot_chain:
runs-on: ubuntu-latest runs-on: ubuntu-latest
needs: build needs: build
@@ -1278,96 +1188,6 @@ jobs:
echo "" echo ""
done done
test_checksum:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: /root/vitastor/tests/test_checksum.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_checksum_xxhash:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: TEST_NAME=xxhash OSD_ARGS="--data_csum_type xxh3_32" /root/vitastor/tests/test_checksum.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_old_checksum:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: OLD=1 /root/vitastor/tests/test_checksum.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_corrupt_all:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: /root/vitastor/tests/test_corrupt_all.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_old_corrupt_all:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: OLD=1 /root/vitastor/tests/test_corrupt_all.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_reweight_half: test_reweight_half:
runs-on: ubuntu-latest runs-on: ubuntu-latest
needs: build needs: build
@@ -1980,24 +1800,6 @@ jobs:
echo "" echo ""
done done
test_old_partwr_csum:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: OLD=1 /root/vitastor/tests/test_partwr_csum.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_heal_old_csum_32k_dmj: test_heal_old_csum_32k_dmj:
runs-on: ubuntu-latest runs-on: ubuntu-latest
needs: build needs: build
@@ -2124,57 +1926,3 @@ jobs:
echo "" echo ""
done done
test_nfs_unaligned_append:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: /root/vitastor/tests/test_nfs_unaligned_append.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_write_encrypted:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: /root/vitastor/tests/test_write_encrypted.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_write_encrypted_ec:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: SCHEME=ec /root/vitastor/tests/test_write_encrypted.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
-8
View File
@@ -38,14 +38,6 @@ for my $line (<>)
{ {
$test_name .= '_antietcd'; $test_name .= '_antietcd';
} }
elsif ($1 eq 'ETCD_SCHEME' && $2 eq 'https')
{
$test_name .= '_https';
}
elsif ($1 eq 'ENCRYPTED')
{
$test_name .= '_encrypted';
}
elsif ($1 eq 'OLD') elsif ($1 eq 'OLD')
{ {
$test_name =~ s/^test_/test_old_/s; $test_name =~ s/^test_/test_old_/s;
-1
View File
@@ -3,4 +3,3 @@
package-lock.json package-lock.json
fio fio
qemu qemu
node_modules
+1 -1
View File
@@ -2,7 +2,7 @@ cmake_minimum_required(VERSION 2.8.12)
project(vitastor) project(vitastor)
set(VITASTOR_VERSION "3.0.5") set(VITASTOR_VERSION "3.0.0")
include(CTest) include(CTest)
-1
View File
@@ -62,7 +62,6 @@ Vitastor поддерживает QEMU-драйвер, протоколы UBLK,
- [Дисковые параметры OSD](docs/config/layout-osd.ru.md) - [Дисковые параметры OSD](docs/config/layout-osd.ru.md)
- [Прочие параметры OSD](docs/config/osd.ru.md) - [Прочие параметры OSD](docs/config/osd.ru.md)
- [Параметры мониторов](docs/config/monitor.ru.md) - [Параметры мониторов](docs/config/monitor.ru.md)
- [Безопасность](docs/config/security.ru.md)
- [Настройки пулов](docs/config/pool.ru.md) - [Настройки пулов](docs/config/pool.ru.md)
- [Метаданные образов в etcd](docs/config/inode.ru.md) - [Метаданные образов в etcd](docs/config/inode.ru.md)
- Использование - Использование
-1
View File
@@ -62,7 +62,6 @@ Read more details in the documentation. You can start from here: [Quick Start](d
- [OSD Disk Layout](docs/config/layout-osd.en.md) - [OSD Disk Layout](docs/config/layout-osd.en.md)
- [OSD Runtime Parameters](docs/config/osd.en.md) - [OSD Runtime Parameters](docs/config/osd.en.md)
- [Monitor](docs/config/monitor.en.md) - [Monitor](docs/config/monitor.en.md)
- [Security](docs/config/security.en.md)
- [Pool configuration](docs/config/pool.en.md) - [Pool configuration](docs/config/pool.en.md)
- [Image metadata in etcd](docs/config/inode.en.md) - [Image metadata in etcd](docs/config/inode.en.md)
- Usage - Usage
+7 -7
View File
@@ -1,5 +1,5 @@
# Compile stage # Compile stage
FROM golang:trixie AS build FROM golang:bookworm AS build
ADD go.sum go.mod /app/ ADD go.sum go.mod /app/
RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go mod download -x RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go mod download -x
@@ -9,7 +9,7 @@ RUN perl -i -e '$/ = undef; while(<>) { s/\n\s*(\{\s*\n)/$1\n/g; s/\}(\s*\n\s*)e
CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o vitastor-csi CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o vitastor-csi
# Final stage # Final stage
FROM debian:trixie FROM debian:bookworm
LABEL maintainers="Vitaliy Filippov <vitalif@yourcmc.ru>" LABEL maintainers="Vitaliy Filippov <vitalif@yourcmc.ru>"
LABEL description="Vitastor CSI Driver" LABEL description="Vitastor CSI Driver"
@@ -25,20 +25,20 @@ RUN apt-get update && \
# NFS mount dependencies # NFS mount dependencies
nfs-common netbase \ nfs-common netbase \
# dependencies of qemu-storage-daemon # dependencies of qemu-storage-daemon
libaio1t64 libc6 libfuse3-4 libglib2.0-0t64 libgmp10 libgnutls30t64 \ libnuma1 liburing2 libglib2.0-0 libfuse3-3 libaio1 libzstd1 libnettle8 \
libhogweed6t64 libnettle8t64 libnuma1 libselinux1 liburing2 libzstd1 zlib1g && \ libgmp10 libhogweed6 libp11-kit0 libidn2-0 libunistring2 libtasn1-6 libpcre2-8-0 libffi8 && \
apt-get clean && \ apt-get clean && \
(echo options nbd nbds_max=128 > /etc/modprobe.d/nbd.conf) (echo options nbd nbds_max=128 > /etc/modprobe.d/nbd.conf)
COPY --from=build /app/vitastor-csi /bin/ COPY --from=build /app/vitastor-csi /bin/
RUN (echo deb http://vitastor.io/debian trixie main > /etc/apt/sources.list.d/vitastor.list) && \ RUN (echo deb http://vitastor.io/debian bookworm main > /etc/apt/sources.list.d/vitastor.list) && \
((echo 'Package: *'; echo 'Pin: origin "vitastor.io"'; echo 'Pin-Priority: 1000') > /etc/apt/preferences.d/vitastor.pref) && \ ((echo 'Package: *'; echo 'Pin: origin "vitastor.io"'; echo 'Pin-Priority: 1000') > /etc/apt/preferences.d/vitastor.pref) && \
wget -q -O /etc/apt/trusted.gpg.d/vitastor.gpg https://vitastor.io/debian/pubkey.gpg && \ wget -q -O /etc/apt/trusted.gpg.d/vitastor.gpg https://vitastor.io/debian/pubkey.gpg && \
apt-get update && \ apt-get update && \
apt-get install -y vitastor-client ibverbs-providers && \ apt-get install -y vitastor-client ibverbs-providers && \
wget https://vitastor.io/archive/qemu/qemu-trixie-10.0.2%2Bds-2%2Bvitastor1/qemu-utils_10.0.2%2Bds-2%2Bvitastor1_amd64.deb && \ wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
wget https://vitastor.io/archive/qemu/qemu-trixie-10.0.2%2Bds-2%2Bvitastor1/qemu-block-extra_10.0.2%2Bds-2%2Bvitastor1_amd64.deb && \ wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
dpkg -x qemu-utils*.deb tmp1 && \ dpkg -x qemu-utils*.deb tmp1 && \
dpkg -x qemu-block-extra*.deb tmp1 && \ dpkg -x qemu-block-extra*.deb tmp1 && \
cp -a tmp1/usr/bin/qemu-storage-daemon /usr/bin/ && \ cp -a tmp1/usr/bin/qemu-storage-daemon /usr/bin/ && \
+4 -4
View File
@@ -1,5 +1,5 @@
# Compile stage # Compile stage
FROM golang:trixie AS build FROM golang:bookworm AS build
ADD go.sum go.mod /app/ ADD go.sum go.mod /app/
RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go mod download -x RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go mod download -x
@@ -9,7 +9,7 @@ RUN perl -i -e '$/ = undef; while(<>) { s/\n\s*(\{\s*\n)/$1\n/g; s/\}(\s*\n\s*)e
CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o vitastor-csi CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o vitastor-csi
# Final stage # Final stage
FROM debian:trixie FROM debian:bookworm
LABEL maintainers="Vitaliy Filippov <vitalif@yourcmc.ru>" LABEL maintainers="Vitaliy Filippov <vitalif@yourcmc.ru>"
LABEL description="Vitastor CSI Driver" LABEL description="Vitastor CSI Driver"
@@ -36,8 +36,8 @@ ADD deb /deb
RUN apt-get update && \ RUN apt-get update && \
apt-get -y install /deb/vitastor-client_*.deb && \ apt-get -y install /deb/vitastor-client_*.deb && \
wget https://vitastor.io/archive/qemu/qemu-trixie-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \ wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
wget https://vitastor.io/archive/qemu/qemu-trixie-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \ wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
dpkg -x qemu-utils*.deb tmp1 && \ dpkg -x qemu-utils*.deb tmp1 && \
dpkg -x qemu-block-extra*.deb tmp1 && \ dpkg -x qemu-block-extra*.deb tmp1 && \
cp -a tmp1/usr/bin/qemu-storage-daemon /usr/bin/ && \ cp -a tmp1/usr/bin/qemu-storage-daemon /usr/bin/ && \
+1 -1
View File
@@ -1,4 +1,4 @@
VITASTOR_VERSION ?= v3.0.5 VITASTOR_VERSION ?= v3.0.0
all: build push all: build push
+1 -1
View File
@@ -49,7 +49,7 @@ spec:
capabilities: capabilities:
add: ["SYS_ADMIN"] add: ["SYS_ADMIN"]
allowPrivilegeEscalation: true allowPrivilegeEscalation: true
image: vitalif/vitastor-csi:v3.0.5 image: vitalif/vitastor-csi:v3.0.0
args: args:
- "--node=$(NODE_ID)" - "--node=$(NODE_ID)"
- "--endpoint=$(CSI_ENDPOINT)" - "--endpoint=$(CSI_ENDPOINT)"
+1 -1
View File
@@ -121,7 +121,7 @@ spec:
privileged: true privileged: true
capabilities: capabilities:
add: ["SYS_ADMIN"] add: ["SYS_ADMIN"]
image: vitalif/vitastor-csi:v3.0.5 image: vitalif/vitastor-csi:v3.0.0
args: args:
- "--node=$(NODE_ID)" - "--node=$(NODE_ID)"
- "--endpoint=$(CSI_ENDPOINT)" - "--endpoint=$(CSI_ENDPOINT)"
+1 -1
View File
@@ -5,7 +5,7 @@ package vitastor
const ( const (
vitastorCSIDriverName = "csi.vitastor.io" vitastorCSIDriverName = "csi.vitastor.io"
vitastorCSIDriverVersion = "3.0.5" vitastorCSIDriverVersion = "3.0.0"
) )
// Config struct fills the parameters of request or user input // Config struct fills the parameters of request or user input
-5
View File
@@ -1,5 +0,0 @@
#!/bin/bash
# 26.04 Resolute Raccoon
docker build --build-arg DISTRO=ubuntu --build-arg REL=resolute -t vitastor-buildenv:resolute -f vitastor-buildenv.Dockerfile .
docker run -it --rm -e REL=resolute -v `dirname $0`/../:/root/vitastor vitastor-buildenv:resolute /root/vitastor/debian/vitastor-build.sh
+1 -1
View File
@@ -1,4 +1,4 @@
vitastor (3.0.5-1) unstable; urgency=medium vitastor (3.0.0-1) unstable; urgency=medium
* Bugfixes * Bugfixes
+1 -1
View File
@@ -3,7 +3,7 @@ Section: admin
Priority: optional Priority: optional
Maintainer: Vitaliy Filippov <vitalif@yourcmc.ru> Maintainer: Vitaliy Filippov <vitalif@yourcmc.ru>
Build-Depends: debhelper, g++ (>= 8), libstdc++6 (>= 8), Build-Depends: debhelper, g++ (>= 8), libstdc++6 (>= 8),
linux-libc-dev, libgoogle-perftools-dev, libjerasure-dev, libgf-complete-dev, libc-ares-dev, linux-libc-dev, libgoogle-perftools-dev, libjerasure-dev, libgf-complete-dev,
libibverbs-dev, librdmacm-dev, libisal-dev, cmake, pkg-config, libnl-3-dev, libnl-genl-3-dev, libibverbs-dev, librdmacm-dev, libisal-dev, cmake, pkg-config, libnl-3-dev, libnl-genl-3-dev,
node-bindings <!nocheck>, node-gyp, node-nan node-bindings <!nocheck>, node-gyp, node-nan
Standards-Version: 4.5.0 Standards-Version: 4.5.0
+1 -1
View File
@@ -25,7 +25,7 @@ RUN set -e -x; \
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
RUN apt-get update && \ RUN apt-get update && \
apt-get -y install fio libgoogle-perftools-dev devscripts libjerasure-dev cmake libc-ares-dev \ apt-get -y install fio libgoogle-perftools-dev devscripts libjerasure-dev cmake \
libibverbs-dev librdmacm-dev libisal-dev libnl-3-dev libnl-genl-3-dev curl nodejs npm node-nan node-bindings && \ libibverbs-dev librdmacm-dev libisal-dev libnl-3-dev libnl-genl-3-dev curl nodejs npm node-nan node-bindings && \
apt-get -y build-dep fio && \ apt-get -y build-dep fio && \
apt-get --download-only source fio apt-get --download-only source fio
+1 -1
View File
@@ -1,6 +1,6 @@
# Build Docker image with Vitastor packages # Build Docker image with Vitastor packages
FROM debian:trixie FROM debian:bookworm
ADD etc/apt /etc/apt/ ADD etc/apt /etc/apt/
RUN apt-get update && apt-get -y install vitastor ibverbs-providers udev systemd qemu-system-x86 qemu-system-common qemu-block-extra qemu-utils jq nfs-common && apt-get clean RUN apt-get update && apt-get -y install vitastor ibverbs-providers udev systemd qemu-system-x86 qemu-system-common qemu-block-extra qemu-utils jq nfs-common && apt-get clean
+1 -1
View File
@@ -1,4 +1,4 @@
VITASTOR_VERSION ?= v3.0.5 VITASTOR_VERSION ?= v3.0.0
all: build push all: build push
+1 -1
View File
@@ -1,3 +1,3 @@
Package: * Package: *
Pin: release n=trixie-backports Pin: release n=bookworm-backports
Pin-Priority: 500 Pin-Priority: 500
+2 -2
View File
@@ -1,2 +1,2 @@
deb http://vitastor.io/debian trixie main deb http://vitastor.io/debian bookworm main
#deb http://http.debian.net/debian/ trixie-backports main deb http://http.debian.net/debian/ bookworm-backports main
@@ -7,7 +7,7 @@ PartOf=vitastor.target
[Service] [Service]
Restart=always Restart=always
EnvironmentFile=/etc/vitastor/docker.conf EnvironmentFile=/etc/vitastor/docker.conf
ExecStart=bash -c 'docker run --rm -i -v /etc/vitastor:/etc/vitastor -v /dev:/dev -v /run:/run -e SYSTEMD_IN_CHROOT=0 \ ExecStart=bash -c 'docker run --rm -i -v /etc/vitastor:/etc/vitastor -v /dev:/dev -v /run:/run \
--security-opt seccomp=unconfined --privileged --pid=host --log-driver none --network host --name vitastor vitastor:$VITASTOR_VERSION \ --security-opt seccomp=unconfined --privileged --pid=host --log-driver none --network host --name vitastor vitastor:$VITASTOR_VERSION \
sleep.sh' sleep.sh'
ExecStartPost=udevadm trigger ExecStartPost=udevadm trigger
+1 -1
View File
@@ -4,7 +4,7 @@
# #
# Desired Vitastor version # Desired Vitastor version
VITASTOR_VERSION=v3.0.5 VITASTOR_VERSION=v3.0.0
# Additional arguments for all containers # Additional arguments for all containers
# For example, you may want to specify a custom logging driver here # For example, you may want to specify a custom logging driver here
+3 -2
View File
@@ -2,7 +2,8 @@
set -e set -e
cp -urv /etc/systemd/system/vitastor* /host-etc/systemd/system/ cp -urv /etc/default /host-etc/
cp -urv /etc/udev/rules.d /host-etc/udev/ cp -urv /etc/systemd /host-etc/
cp -urv /etc/udev /host-etc/
cp -urnv /etc/vitastor /host-etc/ cp -urnv /etc/vitastor /host-etc/
cp -urnv /opt/scripts/* /host-bin/ cp -urnv /opt/scripts/* /host-bin/
-1
View File
@@ -38,4 +38,3 @@ In the future, additional configuration methods may be added:
- [OSD Disk Layout](config/layout-osd.en.md) - [OSD Disk Layout](config/layout-osd.en.md)
- [OSD Runtime Parameters](config/osd.en.md) - [OSD Runtime Parameters](config/osd.en.md)
- [Monitor](config/monitor.en.md) - [Monitor](config/monitor.en.md)
- [Security Parameters](config/security.en.md)
-1
View File
@@ -41,4 +41,3 @@
- [Дисковые параметры OSD](config/layout-osd.ru.md) - [Дисковые параметры OSD](config/layout-osd.ru.md)
- [Прочие параметры OSD](config/osd.ru.md) - [Прочие параметры OSD](config/osd.ru.md)
- [Параметры мониторов](config/monitor.ru.md) - [Параметры мониторов](config/monitor.ru.md)
- [Параметры безопасности](config/security.ru.md)
+2 -8
View File
@@ -198,14 +198,8 @@ put a modified value into etcd key /vitastor/config/global.
- Type: string - Type: string
- Default: none - Default: none
Data and metadata checksum type to use. May be "crc32c", "xxh3_32" or "none". Data checksum type to use. May be "crc32c" or "none". Set to "crc32c" to
Select crc32c or xxh3_32 and set csum_block_size to enable data checksums. enable data checksums.
Both crc32c and xxh3_32 are almost equally fast, xxh3_32 is safer. xxh3_32 is
the xxhash3 algorithm truncated from 64 to 32 bits (which is still a good hash).
Note that enabled data checksums either increase memory usage or reduce
performance. Check details in [csum_block_size](#csum_block_size) description.
## csum_block_size ## csum_block_size
+2 -6
View File
@@ -209,12 +209,8 @@ journal_block_size и meta_block_size. Однако на данный момен
- Тип: строка - Тип: строка
- Значение по умолчанию: none - Значение по умолчанию: none
Тип используемых OSD контрольных сумм данных и метаданных. Может быть "crc32c", Тип используемых OSD контрольных сумм данных. Может быть "crc32c" или "none".
"xxh3_32" или "none". Выберите crc32c или xxh3_32 и установите csum_block_size, Установите в "crc32c", чтобы включить расчёт и проверку контрольных сумм данных.
чтобы включить контрольные суммы данных.
И crc32c, и xxh3_32 примерно одинаково быстры, xxh3_32 надёжней. xxh3_32 - это
алгоритм xxhash3, обрезанный с 64 до 32 бит (это всё равно хороший хеш).
Следует понимать, что контрольные суммы в зависимости от размера блока их Следует понимать, что контрольные суммы в зависимости от размера блока их
расчёта либо увеличивают потребление памяти, либо снижают производительность. расчёта либо увеличивают потребление памяти, либо снижают производительность.
+28 -17
View File
@@ -22,6 +22,7 @@ between clients, OSDs and etcd.
- [rdma_max_msg](#rdma_max_msg) - [rdma_max_msg](#rdma_max_msg)
- [rdma_max_recv](#rdma_max_recv) - [rdma_max_recv](#rdma_max_recv)
- [rdma_max_send](#rdma_max_send) - [rdma_max_send](#rdma_max_send)
- [rdma_odp](#rdma_odp)
- [peer_connect_interval](#peer_connect_interval) - [peer_connect_interval](#peer_connect_interval)
- [peer_connect_timeout](#peer_connect_timeout) - [peer_connect_timeout](#peer_connect_timeout)
- [osd_idle_timeout](#osd_idle_timeout) - [osd_idle_timeout](#osd_idle_timeout)
@@ -101,6 +102,11 @@ found or if `osd_network` is not specified. Auto-selection is also
unsupported with old libibverbs < v32, like in Debian 10 Buster or unsupported with old libibverbs < v32, like in Debian 10 Buster or
CentOS 7. CentOS 7.
Vitastor supports all adapters, even ones without ODP support, like
Mellanox ConnectX-3 and non-Mellanox cards. Versions up to Vitastor
1.2.0 required ODP which is only present in Mellanox ConnectX >= 4.
See also [rdma_odp](#rdma_odp).
Run `ibv_devinfo -v` as root to list available RDMA devices and their Run `ibv_devinfo -v` as root to list available RDMA devices and their
features. features.
@@ -110,23 +116,6 @@ the manual of your network vendor for details about setting up the switch
for RoCEv2 correctly. Usually it means setting up Lossless Ethernet with for RoCEv2 correctly. Usually it means setting up Lossless Ethernet with
PFC (Priority Flow Control) and ECN (Explicit Congestion Notification). PFC (Priority Flow Control) and ECN (Explicit Congestion Notification).
Vitastor supports all adapters, even ones without ODP (On-Demand Paging)
support, like Mellanox ConnectX-3 and non-Mellanox cards. ODP is only present
in Mellanox ConnectX >= 4 adapters and allows to skip memory registration
for RDMA and thus, in theory, avoid memory copying.
Versions up to Vitastor 1.2.0 required ODP, then it was disabled by default,
but it was still supported up to 3.0.3. Now ODP support is removed because it
actually only hurts performance: an example 3-node cluster with 8 NVMe in each
node and 2*25 GBit/s ConnectX-6 RDMA network pushed 3950000 read iops without
ODP, but only 239000 iops with ODP.
This happens because Mellanox ODP implementation seems to be based on
message retransmissions when the adapter doesn't know about the buffer yet -
it likely uses standard "RNR retransmissions" (RNR = receiver not ready)
which is generally slow in RDMA/RoCE networks. Here's a presentation about
it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
## rdma_port_num ## rdma_port_num
- Type: integer - Type: integer
@@ -198,6 +187,28 @@ less than `rdma_max_recv` so the receiving side doesn't run out of buffers.
Doesn't affect memory usage - additional memory isn't allocated for send Doesn't affect memory usage - additional memory isn't allocated for send
operations. operations.
## rdma_odp
- Type: boolean
- Default: false
Use RDMA with On-Demand Paging. ODP is currently only available on Mellanox
ConnectX-4 and newer adapters. ODP allows to not register memory explicitly
for RDMA adapter to be able to use it. This, in turn, allows to skip memory
copying during sending. One would think this should improve performance, but
**in reality** RDMA performance with ODP is **drastically** worse. Example
3-node cluster with 8 NVMe in each node and 2*25 GBit/s ConnectX-6 RDMA network
without ODP pushes 3950000 read iops, but only 239000 iops with ODP...
This happens because Mellanox ODP implementation seems to be based on
message retransmissions when the adapter doesn't know about the buffer yet -
it likely uses standard "RNR retransmissions" (RNR = receiver not ready)
which is generally slow in RDMA/RoCE networks. Here's a presentation about
it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
ODP support is retained in the code just in case a good ODP implementation
appears one day.
## peer_connect_interval ## peer_connect_interval
- Type: seconds - Type: seconds
+30 -18
View File
@@ -22,6 +22,7 @@
- [rdma_max_msg](#rdma_max_msg) - [rdma_max_msg](#rdma_max_msg)
- [rdma_max_recv](#rdma_max_recv) - [rdma_max_recv](#rdma_max_recv)
- [rdma_max_send](#rdma_max_send) - [rdma_max_send](#rdma_max_send)
- [rdma_odp](#rdma_odp)
- [peer_connect_interval](#peer_connect_interval) - [peer_connect_interval](#peer_connect_interval)
- [peer_connect_timeout](#peer_connect_timeout) - [peer_connect_timeout](#peer_connect_timeout)
- [osd_idle_timeout](#osd_idle_timeout) - [osd_idle_timeout](#osd_idle_timeout)
@@ -100,6 +101,12 @@ RoCEv1/RoCEv2, и даже позволяет полностью отключи
не задана. Также автовыбор не поддерживается со старыми версиями библиотеки не задана. Также автовыбор не поддерживается со старыми версиями библиотеки
libibverbs < v32, например в Debian 10 Buster или CentOS 7. libibverbs < v32, например в Debian 10 Buster или CentOS 7.
Vitastor поддерживает все модели адаптеров, включая те, у которых
нет поддержки ODP, то есть вы можете использовать RDMA с ConnectX-3 и
картами производства не Mellanox. Версии Vitastor до 1.2.0 включительно
требовали ODP, который есть только на Mellanox ConnectX 4 и более новых.
См. также [rdma_odp](#rdma_odp).
Запустите `ibv_devinfo -v` от имени суперпользователя, чтобы посмотреть Запустите `ibv_devinfo -v` от имени суперпользователя, чтобы посмотреть
список доступных RDMA-устройств, их параметры и возможности. список доступных RDMA-устройств, их параметры и возможности.
@@ -110,24 +117,6 @@ libibverbs < v32, например в Debian 10 Buster или CentOS 7.
подразумевает настройку сети без потерь на основе PFC (Priority Flow подразумевает настройку сети без потерь на основе PFC (Priority Flow
Control) и ECN (Explicit Congestion Notification). Control) и ECN (Explicit Congestion Notification).
Vitastor поддерживает все модели адаптеров, включая те, у которых нет
поддержки ODP (On-Demand Paging), например, ConnectX-3 и карты производства
не Mellanox. Функция ODP доступна только на адаптерах Mellanox ConnectX-4 и
более новых и позволяет не регистрировать память для её использования RDMA-картой,
благодаря чему в теории можно избежать лишних копирований памяти.
Версии Vitastor до 1.2.0 включительно требовали ODP, потом функция был отключена
по умолчанию, но поддерживалась вплоть до версии 3.0.3. Сейчас поддержка ODP
полностью удалена, так как на самом деле она только портит производительность:
например, на 3-узловом кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на
чтение с RDMA без ODP удаётся снять 3950000 iops, а с ODP - всего 239000 iops.
Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и
основана на повторной передаче сообщений, когда карте не известен буфер -
вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready).
А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука.
Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
## rdma_port_num ## rdma_port_num
- Тип: целое число - Тип: целое число
@@ -203,6 +192,29 @@ OSD в любом случае согласовывают реальное зн
Не влияет на потребление памяти - дополнительная память на операции отправки Не влияет на потребление памяти - дополнительная память на операции отправки
не выделяется. не выделяется.
## rdma_odp
- Тип: булево (да/нет)
- Значение по умолчанию: false
Использовать RDMA с On-Demand Paging. ODP - функция, доступная пока что
исключительно на адаптерах Mellanox ConnectX-4 и более новых. ODP позволяет
не регистрировать память для её использования RDMA-картой. Благодаря этому
можно не копировать данные при отправке их в сеть и, казалось бы, это должно
улучшать производительность - но **по факту** получается так, что
производительность только ухудшается, причём сильно. Пример - на 3-узловом
кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на чтение с RDMA без ODP
удаётся снять 3950000 iops, а с ODP - всего 239000 iops...
Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и
основана на повторной передаче сообщений, когда карте не известен буфер -
вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready).
А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука.
Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
Возможность использования ODP сохранена в коде на случай, если вдруг в один
прекрасный день появится хорошая реализация ODP.
## peer_connect_interval ## peer_connect_interval
- Тип: секунды - Тип: секунды
+16 -63
View File
@@ -38,7 +38,6 @@ with an OSD restart or, for some of them, even without restarting by updating co
- [journal_io](#journal_io) - [journal_io](#journal_io)
- [journal_sector_buffer_count](#journal_sector_buffer_count) - [journal_sector_buffer_count](#journal_sector_buffer_count)
- [journal_no_same_sector_overwrites](#journal_no_same_sector_overwrites) - [journal_no_same_sector_overwrites](#journal_no_same_sector_overwrites)
- [skip_corrupted_meta_entries](#skip_corrupted_meta_entries)
- [throttle_small_writes](#throttle_small_writes) - [throttle_small_writes](#throttle_small_writes)
- [throttle_target_iops](#throttle_target_iops) - [throttle_target_iops](#throttle_target_iops)
- [throttle_target_mbs](#throttle_target_mbs) - [throttle_target_mbs](#throttle_target_mbs)
@@ -68,8 +67,6 @@ with an OSD restart or, for some of them, even without restarting by updating co
- [pg_lock_retry_interval_ms](#pg_lock_retry_interval_ms) - [pg_lock_retry_interval_ms](#pg_lock_retry_interval_ms)
- [atomic_write_size](#atomic_write_size) - [atomic_write_size](#atomic_write_size)
- [use_atomic_flag](#use_atomic_flag) - [use_atomic_flag](#use_atomic_flag)
- [pg_reshard_chunk_size](#pg_reshard_chunk_size)
- [pg_reshard_chunk_pause_ms](#pg_reshard_chunk_pause_ms)
## bind_address ## bind_address
@@ -280,19 +277,13 @@ Maximum number of journal flushers (see above min_flusher_count).
- Type: boolean - Type: boolean
- Default: true - Default: true
Only for the old store ([meta_format](layout-osd.en.md#meta_format) 2). This parameter makes Vitastor always keep metadata area of the block device
in memory. It's required for good performance because it allows to avoid
This parameter makes Vitastor keep a copy of metadata area in memory as it is additional read-modify-write cycles during metadata modifications. Metadata
on disk, in addition to the metadata database. When the option is enabled, every area size is currently roughly 224 MB per 1 TB of data. You can turn it off
metadata entry is effectively stored in RAM twice. It's required for good performance to reduce memory usage by this value, but it will hurt performance. This
because it allows to avoid additional read-modify-write cycles during metadata restriction is likely to be removed in the future along with the upgrade
modifications. Metadata area size with the old store is roughly 224 MB per 1 TB of the metadata storage scheme.
of data. You can turn the option off to reduce memory usage by this value, but
it will reduce performance.
For the new store ([meta_format](layout-osd.en.md#meta_format) 3), the option
may be changed in the future to support operation without loading full metadata
database in memory.
## inmemory_journal ## inmemory_journal
@@ -371,8 +362,6 @@ blocks. The only situation when you should increase it to a larger value
is when you enable journal_no_same_sector_overwrites. In this case set is when you enable journal_no_same_sector_overwrites. In this case set
it to, for example, 1024. it to, for example, 1024.
Not applicable to the new store ([meta_format](layout-osd.en.md#meta_format) 3).
## journal_no_same_sector_overwrites ## journal_no_same_sector_overwrites
- Type: boolean - Type: boolean
@@ -386,17 +375,6 @@ journal after writing it instead of possibly overwriting it the second time.
Most (99%) other SSDs don't need this option. Most (99%) other SSDs don't need this option.
Not applicable to the new store ([meta_format](layout-osd.en.md#meta_format) 3).
## skip_corrupted_meta_entries
- Type: boolean
- Default: false
Only for the new store ([meta_format](layout-osd.en.md#meta_format) 3).
Allow OSD to start when some metadata entries or blocks are corrupted by
skipping them. Should be only used as an emergency measure.
## throttle_small_writes ## throttle_small_writes
- Type: boolean - Type: boolean
@@ -704,10 +682,7 @@ with replicated pools and reach the best possible write performance.
Default value is auto-detected during OSD initialization from Default value is auto-detected during OSD initialization from
`/sys/block/xx/queue/atomic_write_max_bytes` or assumed to be 4096 bytes `/sys/block/xx/queue/atomic_write_max_bytes` or assumed to be 4096 bytes
because all known disks support 4 KB atomic writes. Auto-detection is only used for because all known disks support 4 KB atomic writes.
NVMe disks because SAS disks require the explicit WRITE ATOMIC command which requires
RWF_ATOMIC (see below [#use_atomic_flag]) but that flag works incorrectly in current
Linux versions.
You can also check if your NVMe drives support atomic writes by running You can also check if your NVMe drives support atomic writes by running
the command `nvme id-ctrl /dev/nvme0n1 | grep awupf`. If the reported value, the command `nvme id-ctrl /dev/nvme0n1 | grep awupf`. If the reported value,
@@ -722,34 +697,12 @@ reducing Write Amplification and improving write performance up to 2 times.
- Type: boolean - Type: boolean
This option controls whether Vitastor OSDs use RWF_ATOMIC write flag with atomic writes. This option controls whether the Vitastor OSD uses RWF_ATOMIC write flag with atomic
This flag is supported since Linux 6.11 and adds some safety to atomic writes - the kernel writes. This flag is only supported on Linux kernel since 6.11. Atomic writes are
guarantees to not fragment write requests with it and also to check them against the actual generally only safe to use with this flag because it tells the kernel to never fragment
device atomic write capabilities. write requests and also to check the write against the actual atomic write capabilities
of the device.
However, the option is disabled by default because the flag is currently UNUSABLE - Linux This option is enabled by default when atomic_write_size is set to a value larger than 4 KB.
incorrectly requires writes with that flag to be of power-of-2 length and length-aligned. You can disable it if you're sure that your disks support atomic writes and you want to
I.e., for example, 12 KB writes and not-8-KB aligned 8 KB writes are forbidden by the kernel, bypass the Linux atomic write checks.
even though the NVMe specification allows them.
For NVMe disks with `scheduler=none` writes aren't fragmented anyway so it's not a big deal.
However, you can rebuild your kernel with [this patch](../../patches/linux-fix-atomic-write-checks.diff)
and turn this option on. It will make your atomic writes a bit safer.
## pg_reshard_chunk_size
- Type: integer
- Default: 100000
Pool PG count change is a CPU-intensive operation because OSDs store the full object database
in memory and have to move all entries between old and new PGs. Thus it's performed in chunks,
with pauses between chunks to prevent blocking OSD's event loop and other clients' operations.
This option sets the maximum number of object is a chunk. Moving 100k objects usually takes
50-100ms. Chunk size equal to 0 means unlimited.
## pg_reshard_chunk_pause_ms
- Type: milliseconds
- Default: 100
This option sets the interval between handling two PG count change chunks.
+14 -64
View File
@@ -39,7 +39,6 @@
- [journal_io](#journal_io) - [journal_io](#journal_io)
- [journal_sector_buffer_count](#journal_sector_buffer_count) - [journal_sector_buffer_count](#journal_sector_buffer_count)
- [journal_no_same_sector_overwrites](#journal_no_same_sector_overwrites) - [journal_no_same_sector_overwrites](#journal_no_same_sector_overwrites)
- [skip_corrupted_meta_entries](#skip_corrupted_meta_entries)
- [throttle_small_writes](#throttle_small_writes) - [throttle_small_writes](#throttle_small_writes)
- [throttle_target_iops](#throttle_target_iops) - [throttle_target_iops](#throttle_target_iops)
- [throttle_target_mbs](#throttle_target_mbs) - [throttle_target_mbs](#throttle_target_mbs)
@@ -69,8 +68,6 @@
- [pg_lock_retry_interval_ms](#pg_lock_retry_interval_ms) - [pg_lock_retry_interval_ms](#pg_lock_retry_interval_ms)
- [atomic_write_size](#atomic_write_size) - [atomic_write_size](#atomic_write_size)
- [use_atomic_flag](#use_atomic_flag) - [use_atomic_flag](#use_atomic_flag)
- [pg_reshard_chunk_size](#pg_reshard_chunk_size)
- [pg_reshard_chunk_pause_ms](#pg_reshard_chunk_pause_ms)
## bind_address ## bind_address
@@ -288,19 +285,13 @@ Flusher - это микро-поток (корутина), которая коп
- Тип: булево (да/нет) - Тип: булево (да/нет)
- Значение по умолчанию: true - Значение по умолчанию: true
Только для старого хранилища ([meta_format](layout-osd.en.md#meta_format) 2). Данный параметр заставляет Vitastor всегда держать область метаданных диска
в памяти. Это нужно, чтобы избегать дополнительных операций чтения с диска
Данный параметр заставляет Vitastor всегда держать копию области метаданных при записи. Размер области метаданных на данный момент составляет примерно
в памяти в том же виде, как она лежит на диске, в дополнение к БД метаданных. 224 МБ на 1 ТБ данных. При включении потребление памяти снизится примерно
То есть, с включённой опцией каждая запись метаданных хранится в памяти дважды. на эту величину, но при этом также снизится и производительность. В будущем,
Это нужно, чтобы избегать дополнительных операций чтения с диска при записи. после обновления схемы хранения метаданных, это ограничение, скорее всего,
Размер области метаданных в старом хранилище составляет примерно 224 МБ на будет ликвидировано.
1 ТБ данных. Вы можете отключить опцию, чтобы снизить потребление памяти
примерно на эту величину, но при этом также снизится и производительность.
Для нового хранилища ([meta_format](layout-osd.en.md#meta_format) 3) опция,
возможно, будет переработана в будущем для поддержки работы без полной
загрузки метаданных в памяти.
## inmemory_journal ## inmemory_journal
@@ -383,8 +374,6 @@ fsync небезопасным даже с режимом "directsync".
нужно менять - это если вы включаете journal_no_same_sector_overwrites. В нужно менять - это если вы включаете journal_no_same_sector_overwrites. В
этом случае установите данный параметр, например, в 1024. этом случае установите данный параметр, например, в 1024.
Неприменимо к новому хранилищу ([meta_format](layout-osd.en.md#meta_format) 3).
## journal_no_same_sector_overwrites ## journal_no_same_sector_overwrites
- Тип: булево (да/нет) - Тип: булево (да/нет)
@@ -400,18 +389,6 @@ fsync небезопасным даже с режимом "directsync".
Почти все другие SSD (99% моделей) не требуют данной опции. Почти все другие SSD (99% моделей) не требуют данной опции.
Неприменимо к новому хранилищу ([meta_format](layout-osd.en.md#meta_format) 3).
## skip_corrupted_meta_entries
- Тип: булево (да/нет)
- Значение по умолчанию: false
Только для нового хранилища ([meta_format](layout-osd.en.md#meta_format) 3).
Разрешить OSD запускаться, даже если часть блоков или записей метаданных
повреждена, пропуская их. Опция предназначена для использования только в
целях аварийного восстановления.
## throttle_small_writes ## throttle_small_writes
- Тип: булево (да/нет) - Тип: булево (да/нет)
@@ -740,9 +717,6 @@ pg_minsize OSD во время переключений, что может по
Значение по умолчанию авто-определяется во время инициализации OSD из Значение по умолчанию авто-определяется во время инициализации OSD из
`/sys/block/xx/queue/atomic_write_max_bytes` либо принимается равным 4096, `/sys/block/xx/queue/atomic_write_max_bytes` либо принимается равным 4096,
так как все известные диски поддерживают атомарную запись 4 КБ блоков. так как все известные диски поддерживают атомарную запись 4 КБ блоков.
Автоопределение применяется только для NVMe-дисков, так как SAS диски требуют
использования отдельной команды WRITE ATOMIC, а для неё нужен флаг RWF_ATOMIC
(см. ниже [#use_atomic_flag]), а он в текущих версиях Linux работает некорректно.
Вы также можете проверить, поддерживают ли ваши NVMe-диски атомарную запись, Вы также можете проверить, поддерживают ли ваши NVMe-диски атомарную запись,
с помощью команды `nvme id-ctrl /dev/nvme0n1 | grep awupf`. Если значение awupf с помощью команды `nvme id-ctrl /dev/nvme0n1 | grep awupf`. Если значение awupf
@@ -761,35 +735,11 @@ pg_minsize OSD во время переключений, что может по
- Тип: булево (да/нет) - Тип: булево (да/нет)
Данная опция контролирует использование Vitastor OSD флага RWF_ATOMIC при атомарной записи Данная опция контролирует использование Vitastor OSD флага RWF_ATOMIC при атомарной записи
блоков. Этот флаг поддерживается, начиная с версии ядра Linux 6.11 и добавляет немного корректности блоков. Этот флаг поддерживается только в ядрах Linux начиная с 6.11. Атомарная запись
атомарным записям - ядро гарантирует отсутствие фрагментации запросов записи с этим флагом и является безопасной только при использовании этого флага, так как он сообщает ядру о том,
проверяет их на соответствие реальным возможностям устройства. что запрос записи нельзя фрагментировать и о том, что запрос нужно проверить на соответствие
реальным возможностям атомарной записи устройства.
Однако, данная опция по умолчанию отключена, так как флаг в текущих версиях Linux работает Опция включается по умолчанию, когда atomic_write_size устанавливается в значение больше 4 КБ.
абсолютно НЕКОРРЕКТНО - при нём Linux требует, чтобы запросы записи имели длину, равную Вы можете явно отключить её, если уверены, что ваши диски поддерживают атомарную запись и
степени двойки и были выровнены на эту длину. То есть, например, 12 КБ запросы записи, а также хотите обойти проверки уровня ядра.
8 КБ запросы записи по не-кратному 8 КБ смещению запрещаются ядром, хотя спецификация NVMe их
разрешает.
Для NVMe-дисков с `scheduler=none` запросы записи и так не фрагментируются, так что это не так
уж и важно, однако вы можете пересобрать своё ядро с [этим патчем](../../patches/linux-fix-atomic-write-checks.diff)
и включить данную опцию. Это сделает вашу атомарную запись капельку безопаснее.
## pg_reshard_chunk_size
- Тип: целое число
- Значение по умолчанию: 100000
Изменение числа PG в пуле заметно загружает процессор, так как OSD хранят полную базу данных
объектов в памяти и им приходится перемещать все записи объектов между старыми и новыми PG.
Поэтому изменение применяется порциями, с паузами между порциями, чтобы не блокировать обработку
событий OSD и операции остальных клиентов. Данная опция задаёт максимальное число объектов
в порции. Перемещение 100 тысяч объектов (значение по умолчанию) обычно занимает порядка
50-100 миллисекунд. Значение опции 0 отключает лимит размера порции.
## pg_reshard_chunk_pause_ms
- Тип: миллисекунды
- Значение по умолчанию: 100
Данная опция задаёт интервал между обработкой двух порций изменения числа PG пулов.
-150
View File
@@ -1,150 +0,0 @@
[Documentation](../../README.md#documentation) → [Configuration](../config.en.md) → Security Parameters
-----
[Читать на русском](security.ru.md)
# Security Parameters
These parameters affect your Vitastor installation security and apply to OSDs, monitors and clients.
Most of them can be set in /etc/vitastor/vitastor.conf and in etcd, but don't support online modification.
- [etcd_client_cert](#etcd_client_cert)
- [etcd_client_key](#etcd_client_key)
- [etcd_ca](#etcd_ca)
- [osd_etcd_client_cert](#osd_etcd_client_cert)
- [osd_etcd_client_key](#osd_etcd_client_key)
- [mon_etcd_client_cert](#mon_etcd_client_cert)
- [mon_etcd_client_key](#mon_etcd_client_key)
- [vault_url](#vault_url)
- [vault_secret_api_path](#vault_secret_api_path)
- [vault_client_cert](#vault_client_cert)
- [vault_client_key](#vault_client_key)
- [vault_ca](#vault_ca)
- [vault_timeout_ms](#vault_timeout_ms)
- [vault_error_timeout_sec](#vault_error_timeout_sec)
- [vault_refresh_leeway_sec](#vault_refresh_leeway_sec)
- [max_aes_xts_pool_size](#max_aes_xts_pool_size)
## etcd_client_cert
- Type: string
Client TLS certificate to use for Vitastor client (not OSD and not monitor)
etcd https connections. May be path to a file or just a PEM string with certificate.
In the latter case, string must begin with "-----BEGIN CERTIFICATE-----".
## etcd_client_key
- Type: string
Private key for etcd_client_cert (also a file or a PEM string).
## etcd_ca
- Type: string
Trusted TLS CA to verify etcd server certificate. May be path to a file,
directory or just a PEM string with certificate.
## osd_etcd_client_cert
- Type: string
Same as [etcd_client_cert](#etcd_client_cert), but only for OSDs.
OSDs, clients and monitors should have different permissions, so they should
use different certificates.
## osd_etcd_client_key
- Type: string
Same as [etcd_client_key](#etcd_client_key), but only for OSDs.
## mon_etcd_client_cert
- Type: string
Same as [etcd_client_cert](#etcd_client_cert), but only for Vitastor monitors.
## mon_etcd_client_key
- Type: string
Same as [etcd_client_key](#etcd_client_key), but only for Vitastor monitors.
## vault_url
- Type: string
Vault base URL.
Vitastor clients support AES-256-XTS image data encryption with different per-image keys.
Encryption is performed by the client, OSDs don't have access to decrypted data.
Encryption keys may be stored in etcd or, for the increased security level, in an external
[HashiCorp Vault](https://developer.hashicorp.com/vault/) or [OpenBao](https://openbao.org/)
instance.
Vitastor clients use [v1 k/v secrets engine](https://openbao.org/api-docs/secret/kv/kv-v1/)
and [TLS authentication engine](https://openbao.org/api-docs/auth/cert/) in Vault.
In that case, only key IDs are stored in etcd.
## vault_secret_api_path
- Type: string
- Default: /v1/secret/
Vault v1 secret API mount path to use.
## vault_client_cert
- Type: string
Client TLS certificate to use for Vault connections. Just like [etcd_client_cert](#etcd_client_cert),
may be path to a file or just a certificate in PEM string.
## vault_client_key
- Type: string
Private key for vault_client_cert (also a file or a PEM string).
## vault_ca
- Type: string
Trusted TLS CA to verify Vault server certificate. May be path to a file,
directory or just a PEM string with certificate.
## vault_timeout_ms
- Type: integer
- Default: 5000
Timeout for Vault requests in milliseconds.
## vault_error_timeout_sec
- Type: integer
- Default: 60
Time (in seconds) to wait before retrying after receiving an error from Vault.
## vault_refresh_leeway_sec
- Type: integer
- Default: 60
Extra time (in seconds) before real Vault token lease_timeout to refresh it, just
in case of system clock drift.
## max_aes_xts_pool_size
- Type: integer
- Default: 256
Maximum number of OpenSSL encryption contexts cached in OSD memory. Probably
doesn't require modification.
-154
View File
@@ -1,154 +0,0 @@
[Документация](../../README-ru.md#документация) → [Конфигурация](../config.ru.md) → Параметры безопасности
-----
[Read in English](security.en.md)
# Параметры безопасности
Данные параметры затрагивают безопасность инсталляций Vitastor и используются
OSD, мониторами и клиентами.
Большая их часть может задаваться в /etc/vitastor/vitastor.conf и в etcd, но не
поддерживает онлайн-изменение.
- [etcd_client_cert](#etcd_client_cert)
- [etcd_client_key](#etcd_client_key)
- [etcd_ca](#etcd_ca)
- [osd_etcd_client_cert](#osd_etcd_client_cert)
- [osd_etcd_client_key](#osd_etcd_client_key)
- [mon_etcd_client_cert](#mon_etcd_client_cert)
- [mon_etcd_client_key](#mon_etcd_client_key)
- [vault_url](#vault_url)
- [vault_secret_api_path](#vault_secret_api_path)
- [vault_client_cert](#vault_client_cert)
- [vault_client_key](#vault_client_key)
- [vault_ca](#vault_ca)
- [vault_timeout_ms](#vault_timeout_ms)
- [vault_error_timeout_sec](#vault_error_timeout_sec)
- [vault_refresh_leeway_sec](#vault_refresh_leeway_sec)
- [max_aes_xts_pool_size](#max_aes_xts_pool_size)
## etcd_client_cert
- Тип: строка
Клиентский TLS сертификат для https-подключений к etcd для клиентов Vitastor
(не OSD и не мониторов). Может быть путём к файлу или просто строкой с
сертификатом в формате PEM. В последнем случае строка должна начинаться с
"-----BEGIN CERTIFICATE-----".
## etcd_client_key
- Тип: строка
Закрытый ключ для сертификата etcd_client_cert (также путь к файлу или PEM строка).
## etcd_ca
- Тип: строка
Доверенный корневой TLS-сертификат для проверки сертификата сервера etcd.
Может быть путём к файлу, директории или просто строкой с сертификатом в
формате PEM.
## osd_etcd_client_cert
- Тип: строка
Аналогично [etcd_client_cert](#etcd_client_cert), но только для OSD.
OSD, клиенты и мониторы должны иметь разные привилегии, поэтому они должны
использовать разные сертификаты.
## osd_etcd_client_key
- Тип: строка
Аналогично [etcd_client_key](#etcd_client_key), но только для OSD.
## mon_etcd_client_cert
- Тип: строка
Аналогично [etcd_client_cert](#etcd_client_cert), но только для мониторов Vitastor.
## mon_etcd_client_key
- Тип: строка
Аналогично [etcd_client_key](#etcd_client_key), но только для мониторов Vitastor.
## vault_url
- Тип: строка
Базовый адрес Vault.
Клиенты Vitastor поддерживают AES-256-XTS шифрование данных образов с отдельными ключами на
каждый образ. Данные шифруются клиентами, OSD не имеют доступа к незашифрованным данным.
Ключи шифрования могут храниться в etcd или, для повышенного уровня безопасности, во внешнем
[HashiCorp Vault](https://developer.hashicorp.com/vault/) или [OpenBao](https://openbao.org/).
Клиенты Vitastor используют [движок секретов v1](https://openbao.org/api-docs/secret/kv/kv-v1/)
и [TLS-аутентификацию](https://openbao.org/api-docs/auth/cert/) в Vault.
В этом случае, только ID ключей хранятся в etcd.
## vault_secret_api_path
- Тип: строка
- Значение по умолчанию: /v1/secret/
Путь к API секретов v1 для использования клиентами.
## vault_client_cert
- Тип: строка
Клиентский TLS сертификат для подключений к Vault. Как и [etcd_client_cert](#etcd_client_cert),
может быть путём к файлу или просто PEM-строкой с сертификатом.
## vault_client_key
- Тип: строка
Закрытый ключ для сертификата vault_client_cert (также путь к файлу или PEM строка).
## vault_ca
- Тип: строка
Доверенный корневой TLS-сертификат для проверки сертификата сервера Vault.
Может быть путём к файлу, директории или просто строкой с сертификатом в
формате PEM.
## vault_timeout_ms
- Тип: целое число
- Значение по умолчанию: 5000
Максимально время выполнения Vault-запросов в миллисекундах.
## vault_error_timeout_sec
- Тип: целое число
- Значение по умолчанию: 60
Время (в секундах) для ожидания перед повторной попыткой при получении ошибки от Vault.
## vault_refresh_leeway_sec
- Тип: целое число
- Значение по умолчанию: 60
Зазор времени (в секундах), чтобы обновлять токены Vault чуть раньше их реального
lease_timeout, на случай "ухода" системных часов.
## max_aes_xts_pool_size
- Тип: целое число
- Значение по умолчанию: 256
Максимальное количество кэшируемых в памяти OSD контекстов шифрования OpenSSL.
Вряд ли требует изменения.
-2
View File
@@ -44,8 +44,6 @@
{{../../config/monitor.en.md|indent=2}} {{../../config/monitor.en.md|indent=2}}
{{../../config/security.en.md|indent=2}}
{{../../config/pool.en.md|indent=2}} {{../../config/pool.en.md|indent=2}}
{{../../config/inode.en.md|indent=2}} {{../../config/inode.en.md|indent=2}}
-2
View File
@@ -44,8 +44,6 @@
{{../../config/monitor.ru.md|indent=2}} {{../../config/monitor.ru.md|indent=2}}
{{../../config/security.ru.md|indent=2}}
{{../../config/pool.ru.md|indent=2}} {{../../config/pool.ru.md|indent=2}}
{{../../config/inode.ru.md|indent=2}} {{../../config/inode.ru.md|indent=2}}
+4 -14
View File
@@ -233,21 +233,11 @@
type: string type: string
default: none default: none
info: | info: |
Data and metadata checksum type to use. May be "crc32c", "xxh3_32" or "none". Data checksum type to use. May be "crc32c" or "none". Set to "crc32c" to
Select crc32c or xxh3_32 and set csum_block_size to enable data checksums. enable data checksums.
Both crc32c and xxh3_32 are almost equally fast, xxh3_32 is safer. xxh3_32 is
the xxhash3 algorithm truncated from 64 to 32 bits (which is still a good hash).
Note that enabled data checksums either increase memory usage or reduce
performance. Check details in [csum_block_size](#csum_block_size) description.
info_ru: | info_ru: |
Тип используемых OSD контрольных сумм данных и метаданных. Может быть "crc32c", Тип используемых OSD контрольных сумм данных. Может быть "crc32c" или "none".
"xxh3_32" или "none". Выберите crc32c или xxh3_32 и установите csum_block_size, Установите в "crc32c", чтобы включить расчёт и проверку контрольных сумм данных.
чтобы включить контрольные суммы данных.
И crc32c, и xxh3_32 примерно одинаково быстры, xxh3_32 надёжней. xxh3_32 - это
алгоритм xxhash3, обрезанный с 64 до 32 бит (это всё равно хороший хеш).
Следует понимать, что контрольные суммы в зависимости от размера блока их Следует понимать, что контрольные суммы в зависимости от размера блока их
расчёта либо увеличивают потребление памяти, либо снижают производительность. расчёта либо увеличивают потребление памяти, либо снижают производительность.
+50 -35
View File
@@ -84,6 +84,11 @@
unsupported with old libibverbs < v32, like in Debian 10 Buster or unsupported with old libibverbs < v32, like in Debian 10 Buster or
CentOS 7. CentOS 7.
Vitastor supports all adapters, even ones without ODP support, like
Mellanox ConnectX-3 and non-Mellanox cards. Versions up to Vitastor
1.2.0 required ODP which is only present in Mellanox ConnectX >= 4.
See also [rdma_odp](#rdma_odp).
Run `ibv_devinfo -v` as root to list available RDMA devices and their Run `ibv_devinfo -v` as root to list available RDMA devices and their
features. features.
@@ -92,23 +97,6 @@
the manual of your network vendor for details about setting up the switch the manual of your network vendor for details about setting up the switch
for RoCEv2 correctly. Usually it means setting up Lossless Ethernet with for RoCEv2 correctly. Usually it means setting up Lossless Ethernet with
PFC (Priority Flow Control) and ECN (Explicit Congestion Notification). PFC (Priority Flow Control) and ECN (Explicit Congestion Notification).
Vitastor supports all adapters, even ones without ODP (On-Demand Paging)
support, like Mellanox ConnectX-3 and non-Mellanox cards. ODP is only present
in Mellanox ConnectX >= 4 adapters and allows to skip memory registration
for RDMA and thus, in theory, avoid memory copying.
Versions up to Vitastor 1.2.0 required ODP, then it was disabled by default,
but it was still supported up to 3.0.3. Now ODP support is removed because it
actually only hurts performance: an example 3-node cluster with 8 NVMe in each
node and 2*25 GBit/s ConnectX-6 RDMA network pushed 3950000 read iops without
ODP, but only 239000 iops with ODP.
This happens because Mellanox ODP implementation seems to be based on
message retransmissions when the adapter doesn't know about the buffer yet -
it likely uses standard "RNR retransmissions" (RNR = receiver not ready)
which is generally slow in RDMA/RoCE networks. Here's a presentation about
it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
info_ru: | info_ru: |
Название RDMA-устройства для связи с Vitastor OSD (например, "rocep5s0f0"). Название RDMA-устройства для связи с Vitastor OSD (например, "rocep5s0f0").
Если не указано, Vitastor попробует найти RoCE-устройство, соответствующее Если не указано, Vitastor попробует найти RoCE-устройство, соответствующее
@@ -117,6 +105,12 @@
не задана. Также автовыбор не поддерживается со старыми версиями библиотеки не задана. Также автовыбор не поддерживается со старыми версиями библиотеки
libibverbs < v32, например в Debian 10 Buster или CentOS 7. libibverbs < v32, например в Debian 10 Buster или CentOS 7.
Vitastor поддерживает все модели адаптеров, включая те, у которых
нет поддержки ODP, то есть вы можете использовать RDMA с ConnectX-3 и
картами производства не Mellanox. Версии Vitastor до 1.2.0 включительно
требовали ODP, который есть только на Mellanox ConnectX 4 и более новых.
См. также [rdma_odp](#rdma_odp).
Запустите `ibv_devinfo -v` от имени суперпользователя, чтобы посмотреть Запустите `ibv_devinfo -v` от имени суперпользователя, чтобы посмотреть
список доступных RDMA-устройств, их параметры и возможности. список доступных RDMA-устройств, их параметры и возможности.
@@ -126,24 +120,6 @@
коммутатора для RoCEv2 ищите в документации производителя. Обычно это коммутатора для RoCEv2 ищите в документации производителя. Обычно это
подразумевает настройку сети без потерь на основе PFC (Priority Flow подразумевает настройку сети без потерь на основе PFC (Priority Flow
Control) и ECN (Explicit Congestion Notification). Control) и ECN (Explicit Congestion Notification).
Vitastor поддерживает все модели адаптеров, включая те, у которых нет
поддержки ODP (On-Demand Paging), например, ConnectX-3 и карты производства
не Mellanox. Функция ODP доступна только на адаптерах Mellanox ConnectX-4 и
более новых и позволяет не регистрировать память для её использования RDMA-картой,
благодаря чему в теории можно избежать лишних копирований памяти.
Версии Vitastor до 1.2.0 включительно требовали ODP, потом функция был отключена
по умолчанию, но поддерживалась вплоть до версии 3.0.3. Сейчас поддержка ODP
полностью удалена, так как на самом деле она только портит производительность:
например, на 3-узловом кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на
чтение с RDMA без ODP удаётся снять 3950000 iops, а с ODP - всего 239000 iops.
Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и
основана на повторной передаче сообщений, когда карте не известен буфер -
вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready).
А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука.
Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
- name: rdma_port_num - name: rdma_port_num
type: int type: int
info: | info: |
@@ -242,6 +218,45 @@
у принимающей стороны в процессе работы не заканчивались буферы на приём. у принимающей стороны в процессе работы не заканчивались буферы на приём.
Не влияет на потребление памяти - дополнительная память на операции отправки Не влияет на потребление памяти - дополнительная память на операции отправки
не выделяется. не выделяется.
- name: rdma_odp
type: bool
default: false
online: false
info: |
Use RDMA with On-Demand Paging. ODP is currently only available on Mellanox
ConnectX-4 and newer adapters. ODP allows to not register memory explicitly
for RDMA adapter to be able to use it. This, in turn, allows to skip memory
copying during sending. One would think this should improve performance, but
**in reality** RDMA performance with ODP is **drastically** worse. Example
3-node cluster with 8 NVMe in each node and 2*25 GBit/s ConnectX-6 RDMA network
without ODP pushes 3950000 read iops, but only 239000 iops with ODP...
This happens because Mellanox ODP implementation seems to be based on
message retransmissions when the adapter doesn't know about the buffer yet -
it likely uses standard "RNR retransmissions" (RNR = receiver not ready)
which is generally slow in RDMA/RoCE networks. Here's a presentation about
it from ISPASS-2021 conference: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
ODP support is retained in the code just in case a good ODP implementation
appears one day.
info_ru: |
Использовать RDMA с On-Demand Paging. ODP - функция, доступная пока что
исключительно на адаптерах Mellanox ConnectX-4 и более новых. ODP позволяет
не регистрировать память для её использования RDMA-картой. Благодаря этому
можно не копировать данные при отправке их в сеть и, казалось бы, это должно
улучшать производительность - но **по факту** получается так, что
производительность только ухудшается, причём сильно. Пример - на 3-узловом
кластере с 8 NVMe в каждом узле и сетью 2*25 Гбит/с на чтение с RDMA без ODP
удаётся снять 3950000 iops, а с ODP - всего 239000 iops...
Это происходит из-за того, что реализация ODP у Mellanox неоптимальная и
основана на повторной передаче сообщений, когда карте не известен буфер -
вероятно, на стандартных "RNR retransmission" (RNR = receiver not ready).
А данные повторные передачи в RDMA/RoCE - всегда очень медленная штука.
Презентация на эту тему с конференции ISPASS-2021: https://tkygtr6.github.io/pub/ISPASS21_slides.pdf
Возможность использования ODP сохранена в коде на случай, если вдруг в один
прекрасный день появится хорошая реализация ODP.
- name: peer_connect_interval - name: peer_connect_interval
type: sec type: sec
min: 1 min: 1
+30 -100
View File
@@ -253,33 +253,21 @@
type: bool type: bool
default: true default: true
info: | info: |
Only for the old store ([meta_format](layout-osd.en.md#meta_format) 2). This parameter makes Vitastor always keep metadata area of the block device
in memory. It's required for good performance because it allows to avoid
This parameter makes Vitastor keep a copy of metadata area in memory as it is additional read-modify-write cycles during metadata modifications. Metadata
on disk, in addition to the metadata database. When the option is enabled, every area size is currently roughly 224 MB per 1 TB of data. You can turn it off
metadata entry is effectively stored in RAM twice. It's required for good performance to reduce memory usage by this value, but it will hurt performance. This
because it allows to avoid additional read-modify-write cycles during metadata restriction is likely to be removed in the future along with the upgrade
modifications. Metadata area size with the old store is roughly 224 MB per 1 TB of the metadata storage scheme.
of data. You can turn the option off to reduce memory usage by this value, but
it will reduce performance.
For the new store ([meta_format](layout-osd.en.md#meta_format) 3), the option
may be changed in the future to support operation without loading full metadata
database in memory.
info_ru: | info_ru: |
Только для старого хранилища ([meta_format](layout-osd.en.md#meta_format) 2). Данный параметр заставляет Vitastor всегда держать область метаданных диска
в памяти. Это нужно, чтобы избегать дополнительных операций чтения с диска
Данный параметр заставляет Vitastor всегда держать копию области метаданных при записи. Размер области метаданных на данный момент составляет примерно
в памяти в том же виде, как она лежит на диске, в дополнение к БД метаданных. 224 МБ на 1 ТБ данных. При включении потребление памяти снизится примерно
То есть, с включённой опцией каждая запись метаданных хранится в памяти дважды. на эту величину, но при этом также снизится и производительность. В будущем,
Это нужно, чтобы избегать дополнительных операций чтения с диска при записи. после обновления схемы хранения метаданных, это ограничение, скорее всего,
Размер области метаданных в старом хранилище составляет примерно 224 МБ на будет ликвидировано.
1 ТБ данных. Вы можете отключить опцию, чтобы снизить потребление памяти
примерно на эту величину, но при этом также снизится и производительность.
Для нового хранилища ([meta_format](layout-osd.en.md#meta_format) 3) опция,
возможно, будет переработана в будущем для поддержки работы без полной
загрузки метаданных в памяти.
- name: inmemory_journal - name: inmemory_journal
type: bool type: bool
default: true default: true
@@ -398,15 +386,11 @@
blocks. The only situation when you should increase it to a larger value blocks. The only situation when you should increase it to a larger value
is when you enable journal_no_same_sector_overwrites. In this case set is when you enable journal_no_same_sector_overwrites. In this case set
it to, for example, 1024. it to, for example, 1024.
Not applicable to the new store ([meta_format](layout-osd.en.md#meta_format) 3).
info_ru: | info_ru: |
Максимальное число буферов, разрешённых для использования под записываемые Максимальное число буферов, разрешённых для использования под записываемые
в журнал блоки метаданных. Единственная ситуация, в которой этот параметр в журнал блоки метаданных. Единственная ситуация, в которой этот параметр
нужно менять - это если вы включаете journal_no_same_sector_overwrites. В нужно менять - это если вы включаете journal_no_same_sector_overwrites. В
этом случае установите данный параметр, например, в 1024. этом случае установите данный параметр, например, в 1024.
Неприменимо к новому хранилищу ([meta_format](layout-osd.en.md#meta_format) 3).
- name: journal_no_same_sector_overwrites - name: journal_no_same_sector_overwrites
type: bool type: bool
default: false default: false
@@ -418,8 +402,6 @@
journal after writing it instead of possibly overwriting it the second time. journal after writing it instead of possibly overwriting it the second time.
Most (99%) other SSDs don't need this option. Most (99%) other SSDs don't need this option.
Not applicable to the new store ([meta_format](layout-osd.en.md#meta_format) 3).
info_ru: | info_ru: |
Включайте данную опцию для SSD вроде Intel D3-S4510 и D3-S4610, которые Включайте данную опцию для SSD вроде Intel D3-S4510 и D3-S4610, которые
ОЧЕНЬ не любят, когда ПО перезаписывает один и тот же сектор несколько раз ОЧЕНЬ не любят, когда ПО перезаписывает один и тот же сектор несколько раз
@@ -430,20 +412,6 @@
самого сектора. самого сектора.
Почти все другие SSD (99% моделей) не требуют данной опции. Почти все другие SSD (99% моделей) не требуют данной опции.
Неприменимо к новому хранилищу ([meta_format](layout-osd.en.md#meta_format) 3).
- name: skip_corrupted_meta_entries
type: bool
default: false
info: |
Only for the new store ([meta_format](layout-osd.en.md#meta_format) 3).
Allow OSD to start when some metadata entries or blocks are corrupted by
skipping them. Should be only used as an emergency measure.
info_ru: |
Только для нового хранилища ([meta_format](layout-osd.en.md#meta_format) 3).
Разрешить OSD запускаться, даже если часть блоков или записей метаданных
повреждена, пропуская их. Опция предназначена для использования только в
целях аварийного восстановления.
- name: throttle_small_writes - name: throttle_small_writes
type: bool type: bool
default: false default: false
@@ -845,10 +813,7 @@
Default value is auto-detected during OSD initialization from Default value is auto-detected during OSD initialization from
`/sys/block/xx/queue/atomic_write_max_bytes` or assumed to be 4096 bytes `/sys/block/xx/queue/atomic_write_max_bytes` or assumed to be 4096 bytes
because all known disks support 4 KB atomic writes. Auto-detection is only used for because all known disks support 4 KB atomic writes.
NVMe disks because SAS disks require the explicit WRITE ATOMIC command which requires
RWF_ATOMIC (see below [#use_atomic_flag]) but that flag works incorrectly in current
Linux versions.
You can also check if your NVMe drives support atomic writes by running You can also check if your NVMe drives support atomic writes by running
the command `nvme id-ctrl /dev/nvme0n1 | grep awupf`. If the reported value, the command `nvme id-ctrl /dev/nvme0n1 | grep awupf`. If the reported value,
@@ -869,9 +834,6 @@
Значение по умолчанию авто-определяется во время инициализации OSD из Значение по умолчанию авто-определяется во время инициализации OSD из
`/sys/block/xx/queue/atomic_write_max_bytes` либо принимается равным 4096, `/sys/block/xx/queue/atomic_write_max_bytes` либо принимается равным 4096,
так как все известные диски поддерживают атомарную запись 4 КБ блоков. так как все известные диски поддерживают атомарную запись 4 КБ блоков.
Автоопределение применяется только для NVMe-дисков, так как SAS диски требуют
использования отдельной команды WRITE ATOMIC, а для неё нужен флаг RWF_ATOMIC
(см. ниже [#use_atomic_flag]), а он в текущих версиях Linux работает некорректно.
Вы также можете проверить, поддерживают ли ваши NVMe-диски атомарную запись, Вы также можете проверить, поддерживают ли ваши NVMe-диски атомарную запись,
с помощью команды `nvme id-ctrl /dev/nvme0n1 | grep awupf`. Если значение awupf с помощью команды `nvme id-ctrl /dev/nvme0n1 | grep awupf`. Если значение awupf
@@ -887,54 +849,22 @@
- name: use_atomic_flag - name: use_atomic_flag
type: bool type: bool
info: | info: |
This option controls whether Vitastor OSDs use RWF_ATOMIC write flag with atomic writes. This option controls whether the Vitastor OSD uses RWF_ATOMIC write flag with atomic
This flag is supported since Linux 6.11 and adds some safety to atomic writes - the kernel writes. This flag is only supported on Linux kernel since 6.11. Atomic writes are
guarantees to not fragment write requests with it and also to check them against the actual generally only safe to use with this flag because it tells the kernel to never fragment
device atomic write capabilities. write requests and also to check the write against the actual atomic write capabilities
of the device.
However, the option is disabled by default because the flag is currently UNUSABLE - Linux This option is enabled by default when atomic_write_size is set to a value larger than 4 KB.
incorrectly requires writes with that flag to be of power-of-2 length and length-aligned. You can disable it if you're sure that your disks support atomic writes and you want to
I.e., for example, 12 KB writes and not-8-KB aligned 8 KB writes are forbidden by the kernel, bypass the Linux atomic write checks.
even though the NVMe specification allows them.
For NVMe disks with `scheduler=none` writes aren't fragmented anyway so it's not a big deal.
However, you can rebuild your kernel with [this patch](../../patches/linux-fix-atomic-write-checks.diff)
and turn this option on. It will make your atomic writes a bit safer.
info_ru: | info_ru: |
Данная опция контролирует использование Vitastor OSD флага RWF_ATOMIC при атомарной записи Данная опция контролирует использование Vitastor OSD флага RWF_ATOMIC при атомарной записи
блоков. Этот флаг поддерживается, начиная с версии ядра Linux 6.11 и добавляет немного корректности блоков. Этот флаг поддерживается только в ядрах Linux начиная с 6.11. Атомарная запись
атомарным записям - ядро гарантирует отсутствие фрагментации запросов записи с этим флагом и является безопасной только при использовании этого флага, так как он сообщает ядру о том,
проверяет их на соответствие реальным возможностям устройства. что запрос записи нельзя фрагментировать и о том, что запрос нужно проверить на соответствие
реальным возможностям атомарной записи устройства.
Однако, данная опция по умолчанию отключена, так как флаг в текущих версиях Linux работает Опция включается по умолчанию, когда atomic_write_size устанавливается в значение больше 4 КБ.
абсолютно НЕКОРРЕКТНО - при нём Linux требует, чтобы запросы записи имели длину, равную Вы можете явно отключить её, если уверены, что ваши диски поддерживают атомарную запись и
степени двойки и были выровнены на эту длину. То есть, например, 12 КБ запросы записи, а также хотите обойти проверки уровня ядра.
8 КБ запросы записи по не-кратному 8 КБ смещению запрещаются ядром, хотя спецификация NVMe их
разрешает.
Для NVMe-дисков с `scheduler=none` запросы записи и так не фрагментируются, так что это не так
уж и важно, однако вы можете пересобрать своё ядро с [этим патчем](../../patches/linux-fix-atomic-write-checks.diff)
и включить данную опцию. Это сделает вашу атомарную запись капельку безопаснее.
- name: pg_reshard_chunk_size
type: int
default: 100000
info: |
Pool PG count change is a CPU-intensive operation because OSDs store the full object database
in memory and have to move all entries between old and new PGs. Thus it's performed in chunks,
with pauses between chunks to prevent blocking OSD's event loop and other clients' operations.
This option sets the maximum number of object is a chunk. Moving 100k objects usually takes
50-100ms. Chunk size equal to 0 means unlimited.
info_ru: |
Изменение числа PG в пуле заметно загружает процессор, так как OSD хранят полную базу данных
объектов в памяти и им приходится перемещать все записи объектов между старыми и новыми PG.
Поэтому изменение применяется порциями, с паузами между порциями, чтобы не блокировать обработку
событий OSD и операции остальных клиентов. Данная опция задаёт максимальное число объектов
в порции. Перемещение 100 тысяч объектов (значение по умолчанию) обычно занимает порядка
50-100 миллисекунд. Значение опции 0 отключает лимит размера порции.
- name: pg_reshard_chunk_pause_ms
type: ms
default: 100
info: |
This option sets the interval between handling two PG count change chunks.
info_ru: |
Данная опция задаёт интервал между обработкой двух порций изменения числа PG пулов.
-5
View File
@@ -1,5 +0,0 @@
{
"dependencies": {
"yaml": "^2.8.2"
}
}
-5
View File
@@ -1,5 +0,0 @@
# Security Parameters
These parameters affect your Vitastor installation security and apply to OSDs, monitors and clients.
Most of them can be set in /etc/vitastor/vitastor.conf and in etcd, but don't support online modification.
-7
View File
@@ -1,7 +0,0 @@
# Параметры безопасности
Данные параметры затрагивают безопасность инсталляций Vitastor и используются
OSD, мониторами и клиентами.
Большая их часть может задаваться в /etc/vitastor/vitastor.conf и в etcd, но не
поддерживает онлайн-изменение.
-131
View File
@@ -1,131 +0,0 @@
- name: etcd_client_cert
type: string
info: |
Client TLS certificate to use for Vitastor client (not OSD and not monitor)
etcd https connections. May be path to a file or just a PEM string with certificate.
In the latter case, string must begin with "-----BEGIN CERTIFICATE-----".
info_ru: |
Клиентский TLS сертификат для https-подключений к etcd для клиентов Vitastor
(не OSD и не мониторов). Может быть путём к файлу или просто строкой с
сертификатом в формате PEM. В последнем случае строка должна начинаться с
"-----BEGIN CERTIFICATE-----".
- name: etcd_client_key
type: string
info: Private key for etcd_client_cert (also a file or a PEM string).
info_ru: Закрытый ключ для сертификата etcd_client_cert (также путь к файлу или PEM строка).
- name: etcd_ca
type: string
info: |
Trusted TLS CA to verify etcd server certificate. May be path to a file,
directory or just a PEM string with certificate.
info_ru: |
Доверенный корневой TLS-сертификат для проверки сертификата сервера etcd.
Может быть путём к файлу, директории или просто строкой с сертификатом в
формате PEM.
- name: osd_etcd_client_cert
type: string
info: |
Same as [etcd_client_cert](#etcd_client_cert), but only for OSDs.
OSDs, clients and monitors should have different permissions, so they should
use different certificates.
info_ru: |
Аналогично [etcd_client_cert](#etcd_client_cert), но только для OSD.
OSD, клиенты и мониторы должны иметь разные привилегии, поэтому они должны
использовать разные сертификаты.
- name: osd_etcd_client_key
type: string
info: Same as [etcd_client_key](#etcd_client_key), but only for OSDs.
info_ru: Аналогично [etcd_client_key](#etcd_client_key), но только для OSD.
- name: mon_etcd_client_cert
type: string
info: Same as [etcd_client_cert](#etcd_client_cert), but only for Vitastor monitors.
info_ru: Аналогично [etcd_client_cert](#etcd_client_cert), но только для мониторов Vitastor.
- name: mon_etcd_client_key
type: string
info: Same as [etcd_client_key](#etcd_client_key), but only for Vitastor monitors.
info_ru: Аналогично [etcd_client_key](#etcd_client_key), но только для мониторов Vitastor.
- name: vault_url
type: string
info: |
Vault base URL.
Vitastor clients support AES-256-XTS image data encryption with different per-image keys.
Encryption is performed by the client, OSDs don't have access to decrypted data.
Encryption keys may be stored in etcd or, for the increased security level, in an external
[HashiCorp Vault](https://developer.hashicorp.com/vault/) or [OpenBao](https://openbao.org/)
instance.
Vitastor clients use [v1 k/v secrets engine](https://openbao.org/api-docs/secret/kv/kv-v1/)
and [TLS authentication engine](https://openbao.org/api-docs/auth/cert/) in Vault.
In that case, only key IDs are stored in etcd.
info_ru: |
Базовый адрес Vault.
Клиенты Vitastor поддерживают AES-256-XTS шифрование данных образов с отдельными ключами на
каждый образ. Данные шифруются клиентами, OSD не имеют доступа к незашифрованным данным.
Ключи шифрования могут храниться в etcd или, для повышенного уровня безопасности, во внешнем
[HashiCorp Vault](https://developer.hashicorp.com/vault/) или [OpenBao](https://openbao.org/).
Клиенты Vitastor используют [движок секретов v1](https://openbao.org/api-docs/secret/kv/kv-v1/)
и [TLS-аутентификацию](https://openbao.org/api-docs/auth/cert/) в Vault.
В этом случае, только ID ключей хранятся в etcd.
- name: vault_secret_api_path
type: string
default: /v1/secret/
info: Vault v1 secret API mount path to use.
info_ru: Путь к API секретов v1 для использования клиентами.
- name: vault_client_cert
type: string
info: |
Client TLS certificate to use for Vault connections. Just like [etcd_client_cert](#etcd_client_cert),
may be path to a file or just a certificate in PEM string.
info_ru: |
Клиентский TLS сертификат для подключений к Vault. Как и [etcd_client_cert](#etcd_client_cert),
может быть путём к файлу или просто PEM-строкой с сертификатом.
- name: vault_client_key
type: string
info: Private key for vault_client_cert (also a file or a PEM string).
info_ru: Закрытый ключ для сертификата vault_client_cert (также путь к файлу или PEM строка).
- name: vault_ca
type: string
info: |
Trusted TLS CA to verify Vault server certificate. May be path to a file,
directory or just a PEM string with certificate.
info_ru: |
Доверенный корневой TLS-сертификат для проверки сертификата сервера Vault.
Может быть путём к файлу, директории или просто строкой с сертификатом в
формате PEM.
- name: vault_timeout_ms
type: int
default: 5000
info: Timeout for Vault requests in milliseconds.
info_ru: Максимально время выполнения Vault-запросов в миллисекундах.
- name: vault_error_timeout_sec
type: int
default: 60
info: |
Time (in seconds) to wait before retrying after receiving an error from Vault.
info_ru: |
Время (в секундах) для ожидания перед повторной попыткой при получении ошибки от Vault.
- name: vault_refresh_leeway_sec
type: int
default: 60
info: |
Extra time (in seconds) before real Vault token lease_timeout to refresh it, just
in case of system clock drift.
info_ru: |
Зазор времени (в секундах), чтобы обновлять токены Vault чуть раньше их реального
lease_timeout, на случай "ухода" системных часов.
- name: max_aes_xts_pool_size
type: int
default: 256
info: |
Maximum number of OpenSSL encryption contexts cached in OSD memory. Probably
doesn't require modification.
info_ru: |
Максимальное количество кэшируемых в памяти OSD контекстов шифрования OpenSSL.
Вряд ли требует изменения.
+3 -27
View File
@@ -26,37 +26,13 @@ at Vitastor Kubernetes operator: https://github.com/Antilles7227/vitastor-operat
The instruction is very simple. The instruction is very simple.
1. Download a Docker image of the desired version: \ 1. Download a Docker image of the desired version: \
`docker pull vitalif/vitastor:v3.0.5` `docker pull vitalif/vitastor:v3.0.0`
2. Install scripts to the host system: \ 2. Install scripts to the host system: \
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.5 install.sh` `docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.0 install.sh`
3. Reload udev rules: \ 3. Reload udev rules: \
`udevadm control --reload-rules` `udevadm control --reload-rules`
4. Enable the vitastor-host service: \
`systemctl enable --now vitastor-host`
After these steps, you can return to [Quick Start](../intro/quickstart.en.md). And you can return to [Quick Start](../intro/quickstart.en.md).
## Podman
If you use Podman, run the following commands as root before installing Vitastor containers:
```
ln -s podman /usr/bin/docker
mkdir -p /etc/systemd/system/systemd-udevd.service.d
cat >/etc/systemd/system/systemd-udevd.service.d/override.conf <<EOF
[Service]
CapabilityBoundingSet=~
SystemCallFilter=@mount capset
EOF
systemctl daemon-reload
systemctl restart systemd-udevd
```
Without it, udev fails to do calls into a Podman container and Vitastor disk detection doesn't work.
## Upgrading Containers ## Upgrading Containers
+2 -27
View File
@@ -25,39 +25,14 @@ Vitastor можно установить в Docker/Podman. При этом etcd,
Инструкция по установке максимально простая. Инструкция по установке максимально простая.
1. Скачайте Docker-образ желаемой версии: \ 1. Скачайте Docker-образ желаемой версии: \
`docker pull vitalif/vitastor:v3.0.5` `docker pull vitalif/vitastor:v3.0.0`
2. Установите скрипты в хост-систему командой: \ 2. Установите скрипты в хост-систему командой: \
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.5 install.sh` `docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.0 install.sh`
3. Перезагрузите правила udev: \ 3. Перезагрузите правила udev: \
`udevadm control --reload-rules` `udevadm control --reload-rules`
4. Включите сервис vitastor-host: \
`systemctl enable --now vitastor-host`
После этого вы можете возвращаться к разделу [Быстрый старт](../intro/quickstart.ru.md). После этого вы можете возвращаться к разделу [Быстрый старт](../intro/quickstart.ru.md).
## Podman
Если вы используете Podman, перед установкой контейнеров Vitastor выполните следующие
команды от имени суперпользователя:
```
ln -s podman /usr/bin/docker
mkdir -p /etc/systemd/system/systemd-udevd.service.d
cat >/etc/systemd/system/systemd-udevd.service.d/override.conf <<EOF
[Service]
CapabilityBoundingSet=~
SystemCallFilter=@mount capset
EOF
systemctl daemon-reload
systemctl restart systemd-udevd
```
Без этих настроек udev не может делать вызовы внутрь Podman-контейнеров и определение дисков Vitastor не работает.
## Обновление контейнеров ## Обновление контейнеров
Сначала обязательно проверьте раздел [Обновление Vitastor](../usage/admin.ru.md#обновление-vitastor), Сначала обязательно проверьте раздел [Обновление Vitastor](../usage/admin.ru.md#обновление-vitastor),
+1 -3
View File
@@ -33,17 +33,15 @@
- CentOS 7: `yum install https://vitastor.io/rpms/centos/7/vitastor-release.rpm` - CentOS 7: `yum install https://vitastor.io/rpms/centos/7/vitastor-release.rpm`
- CentOS 8: `dnf install https://vitastor.io/rpms/centos/8/vitastor-release.rpm` - CentOS 8: `dnf install https://vitastor.io/rpms/centos/8/vitastor-release.rpm`
- AlmaLinux 9 and other RHEL 9 clones (Rocky, Oracle...): `dnf install https://vitastor.io/rpms/centos/9/vitastor-release.rpm` - AlmaLinux 9 and other RHEL 9 clones (Rocky, Oracle...): `dnf install https://vitastor.io/rpms/centos/9/vitastor-release.rpm`
- AlmaLinux 10 and other RHEL 10 clones: `dnf install https://vitastor.io/rpms/centos/10/vitastor-release.rpm`
- Enable EPEL: `yum/dnf install epel-release` - Enable EPEL: `yum/dnf install epel-release`
- Enable additional CentOS repositories: - Enable additional CentOS repositories:
- CentOS 7: `yum install centos-release-scl` - CentOS 7: `yum install centos-release-scl`
- CentOS 8: `dnf install centos-release-advanced-virtualization` - CentOS 8: `dnf install centos-release-advanced-virtualization`
- RHEL 9/10 clones: not required - RHEL 9 clones: not required
- Enable elrepo-kernel: - Enable elrepo-kernel:
- CentOS 7: `yum install https://www.elrepo.org/elrepo-release-7.el7.elrepo.noarch.rpm` - CentOS 7: `yum install https://www.elrepo.org/elrepo-release-7.el7.elrepo.noarch.rpm`
- CentOS 8: `dnf install https://www.elrepo.org/elrepo-release-8.el8.elrepo.noarch.rpm` - CentOS 8: `dnf install https://www.elrepo.org/elrepo-release-8.el8.elrepo.noarch.rpm`
- RHEL 9 clones: `dnf install https://www.elrepo.org/elrepo-release-9.el9.elrepo.noarch.rpm` - RHEL 9 clones: `dnf install https://www.elrepo.org/elrepo-release-9.el9.elrepo.noarch.rpm`
- RHEL 10 clones: not required
- Install packages: `yum/dnf install vitastor lpsolve etcd kernel-ml qemu-kvm` - Install packages: `yum/dnf install vitastor lpsolve etcd kernel-ml qemu-kvm`
## Installation requirements ## Installation requirements
+1 -3
View File
@@ -33,17 +33,15 @@
- CentOS 7: `yum install https://vitastor.io/rpms/centos/7/vitastor-release.rpm` - CentOS 7: `yum install https://vitastor.io/rpms/centos/7/vitastor-release.rpm`
- CentOS 8: `dnf install https://vitastor.io/rpms/centos/8/vitastor-release.rpm` - CentOS 8: `dnf install https://vitastor.io/rpms/centos/8/vitastor-release.rpm`
- AlmaLinux 9 и другие клоны RHEL 9 (Rocky, Oracle...): `dnf install https://vitastor.io/rpms/centos/9/vitastor-release.rpm` - AlmaLinux 9 и другие клоны RHEL 9 (Rocky, Oracle...): `dnf install https://vitastor.io/rpms/centos/9/vitastor-release.rpm`
- AlmaLinux 10 и другие клоны RHEL 10: `dnf install https://vitastor.io/rpms/centos/10/vitastor-release.rpm`
- Включите EPEL: `yum/dnf install epel-release` - Включите EPEL: `yum/dnf install epel-release`
- Включите дополнительные репозитории CentOS: - Включите дополнительные репозитории CentOS:
- CentOS 7: `yum install centos-release-scl` - CentOS 7: `yum install centos-release-scl`
- CentOS 8: `dnf install centos-release-advanced-virtualization` - CentOS 8: `dnf install centos-release-advanced-virtualization`
- Клоны RHEL 9/10: не нужно - Клоны RHEL 9: не нужно
- Включите elrepo-kernel: - Включите elrepo-kernel:
- CentOS 7: `yum install https://www.elrepo.org/elrepo-release-7.el7.elrepo.noarch.rpm` - CentOS 7: `yum install https://www.elrepo.org/elrepo-release-7.el7.elrepo.noarch.rpm`
- CentOS 8: `dnf install https://www.elrepo.org/elrepo-release-8.el8.elrepo.noarch.rpm` - CentOS 8: `dnf install https://www.elrepo.org/elrepo-release-8.el8.elrepo.noarch.rpm`
- Клоны RHEL 9: `dnf install https://www.elrepo.org/elrepo-release-9.el9.elrepo.noarch.rpm` - Клоны RHEL 9: `dnf install https://www.elrepo.org/elrepo-release-9.el9.elrepo.noarch.rpm`
- Клоны RHEL 10: не нужно
- Установите пакеты: `yum/dnf install vitastor lpsolve etcd kernel-ml qemu-kvm` - Установите пакеты: `yum/dnf install vitastor lpsolve etcd kernel-ml qemu-kvm`
## Установочные требования ## Установочные требования
+1 -1
View File
@@ -6,7 +6,7 @@
# Proxmox VE # Proxmox VE
To enable Vitastor support in Proxmox Virtual Environment (6.4-9.x are supported): To enable Vitastor support in Proxmox Virtual Environment (6.4-8.x are supported):
- Add the corresponding Vitastor Debian repository into sources.list on Proxmox hosts: - Add the corresponding Vitastor Debian repository into sources.list on Proxmox hosts:
trixie for 9.0+, bookworm for 8.1+, pve8.0 for 8.0, bullseye for 7.4, pve7.3 for 7.3, pve7.2 for 7.2, pve7.1 for 7.1, buster for 6.4 trixie for 9.0+, bookworm for 8.1+, pve8.0 for 8.0, bullseye for 7.4, pve7.3 for 7.3, pve7.2 for 7.2, pve7.1 for 7.1, buster for 6.4
+1 -1
View File
@@ -6,7 +6,7 @@
# Proxmox VE # Proxmox VE
Чтобы подключить Vitastor к Proxmox Virtual Environment (поддерживаются версии 6.4-9.x): Чтобы подключить Vitastor к Proxmox Virtual Environment (поддерживаются версии 6.4-8.x):
- Добавьте соответствующий Debian-репозиторий Vitastor в sources.list на хостах Proxmox: - Добавьте соответствующий Debian-репозиторий Vitastor в sources.list на хостах Proxmox:
trixie для 9.0+, bookworm для 8.1+, pve8.0 для 8.0, bullseye для 7.4, pve7.3 для 7.3, pve7.2 для 7.2, pve7.1 для 7.1, buster для 6.4 trixie для 9.0+, bookworm для 8.1+, pve8.0 для 8.0, bullseye для 7.4, pve7.3 для 7.3, pve7.2 для 7.2, pve7.1 для 7.1, buster для 6.4
+2 -2
View File
@@ -15,8 +15,8 @@
- gcc and g++ 8 or newer, clang 10 or newer, or other compiler with C++11 plus - gcc and g++ 8 or newer, clang 10 or newer, or other compiler with C++11 plus
designated initializers support from C++20 designated initializers support from C++20
- CMake - CMake
- jerasure, c-ares headers and libraries - jerasure headers and libraries
- ISA-L, libibverbs, librdmacm, libnl3 headers and libraries (optional) - ISA-L, libibverbs and librdmacm headers and libraries (optional)
- tcmalloc (google-perftools-dev) - tcmalloc (google-perftools-dev)
## Basic instructions ## Basic instructions
+2 -2
View File
@@ -15,8 +15,8 @@
- gcc и g++ >= 8, либо clang >= 10, либо другой компилятор с поддержкой C++11 плюс - gcc и g++ >= 8, либо clang >= 10, либо другой компилятор с поддержкой C++11 плюс
назначенных инициализаторов (designated initializers) из C++20 назначенных инициализаторов (designated initializers) из C++20
- CMake - CMake
- Заголовки и библиотеки jerasure, c-ares - Заголовки и библиотеки jerasure
- Опционально - заголовки и библиотеки ISA-L, libibverbs, librdmacm, libnl3 - Опционально - заголовки и библиотеки ISA-L, libibverbs, librdmacm
- tcmalloc (google-perftools-dev) - tcmalloc (google-perftools-dev)
## Базовая инструкция ## Базовая инструкция
-2
View File
@@ -41,8 +41,6 @@
- [Built-in Prometheus metric exporter](../config/monitor.en.md#enable_prometheus) - [Built-in Prometheus metric exporter](../config/monitor.en.md#enable_prometheus)
- [NFS RDMA support](../usage/nfs.en.md#rdma) (probably also usable for GPUDirect) - [NFS RDMA support](../usage/nfs.en.md#rdma) (probably also usable for GPUDirect)
- [S3](../installation/s3.en.md) - [S3](../installation/s3.en.md)
- [TLS support for etcd connections](../config/security.en.md)
- [AES-256-XTS image encryption](../usage/cli.en.md#create) and [Vault support](../config/security.en.md#vault_url) for key storage
## Plugins and tools ## Plugins and tools
-2
View File
@@ -43,8 +43,6 @@
- [Встроенный Prometheus-экспортер метрик](../config/monitor.ru.md#enable_prometheus) - [Встроенный Prometheus-экспортер метрик](../config/monitor.ru.md#enable_prometheus)
- [Поддержка NFS RDMA](../usage/nfs.ru.md#rdma) (вероятно, также подходящая для GPUDirect) - [Поддержка NFS RDMA](../usage/nfs.ru.md#rdma) (вероятно, также подходящая для GPUDirect)
- [S3](../installation/s3.ru.md) - [S3](../installation/s3.ru.md)
- [Поддержка TLS-соединений с etcd](../config/security.ru.md)
- [AES-256-XTS шифрование данных](../usage/cli.ru.md#create) и [поддержка Vault](../config/security.ru.md#vault_url) для хранения ключей
## Драйверы и инструменты ## Драйверы и инструменты
+7 -21
View File
@@ -125,31 +125,18 @@ bench-kaveri kaveri 10 G 10 G 0 B/s 0 0 0 us 0 B/s 0
## create ## create
`vitastor-cli create -s|--size SIZE [OPTIONS] <name>` `vitastor-cli create -s|--size <size> [-p|--pool <id|name>] [--parent <parent_name>[@<snapshot>]] <name>`
Create an image. Options: Create an image. You may use K/M/G/T suffixes for `<size>`. If `--parent` is specified,
a copy-on-write image clone is created. Parent must be a snapshot (readonly image).
* `-s|--size SIZE` - New image size in bytes or with a K/M/G/T unit suffix. Pool must be specified if there is more than one pool.
* `-p|--pool POOL` - Specify pool for the new image (may be omitted if there is only 1 pool).
* `--parent PARENT` - Create a copy-on-write image clone based on PARENT (or PARENT@SNAPSHOT).
If parent is not a snapshot, it must be a read-only image.
* `--enc-key random` - Generate a new random AES-256-XTS encryption key for the new image.
* `--enc-key HEX` - Set a specified AES-256-XTS key (64 bytes in hex) for the new image.
* `--enc-key vault:ID` - Use an encryption key from an external Vault secret with specified ID.
``` ```
vitastor-cli create --snapshot <snapshot> [OPTIONS] <image> vitastor-cli create --snapshot <snapshot> [-p|--pool <id|name>] <image>
vitastor-cli snap-create [OPTIONS] <image>@<snapshot> vitastor-cli snap-create [-p|--pool <id|name>] <image>@<snapshot>
``` ```
Create a snapshot of image `<image>`. May be used live if only a single writer is active. Create a snapshot of image `<name>` (either form can be used). May be used live if only a single writer is active.
Options:
* `-p|--pool POOL` - Move image to pool POOL, leaving the snapshot in the old pool.
* `--enc-key random` - Change image encryption key to a new random AES-256-XTS key.
* `--enc-key KEY` - Change image encryption key to a specified key, Vault key or to an empty key.
By default, the image retains its old encryption key when taking a snapshot.
See also about [how to export snapshots](qemu.en.md#exporting-snapshots). See also about [how to export snapshots](qemu.en.md#exporting-snapshots).
@@ -164,7 +151,6 @@ You should resize file system in the image, if present, before shrinking it.
* `--deleted 1|0` - Set/clear 'deleted image' flag (set automatically during unfinished deletes). * `--deleted 1|0` - Set/clear 'deleted image' flag (set automatically during unfinished deletes).
* `-f|--force` - Proceed with shrinking or setting readwrite flag even if the image has children. * `-f|--force` - Proceed with shrinking or setting readwrite flag even if the image has children.
* `--down-ok` - Proceed with shrinking even if some data will be left on unavailable OSDs. * `--down-ok` - Proceed with shrinking even if some data will be left on unavailable OSDs.
* `--enc-key HEX` - Change image encryption key (allowed only with `--force`).
## dd ## dd
+8 -22
View File
@@ -127,32 +127,19 @@ bench-kaveri kaveri 10 G 10 G 0 B/s 0 0 0 us 0 B/s 0
## create ## create
`vitastor-cli create -s|--size SIZE [ОПЦИИ] <name>` `vitastor-cli create -s|--size <size> [-p|--pool <id|name>] [--parent <parent_name>[@<snapshot>]] <name>`
Создать образ. Опции: Создать образ. Для размера `<size>` можно использовать суффиксы K/M/G/T (килобайт-мегабайт-гигабайт-терабайт).
Если указана опция `--parent`, создаётся клон образа. Родитель `<parent_name>[@<snapshot>]` должен быть
* `-s|--size SIZE` - Размер нового образа в байтах или с суффиксом K/M/G/T (кило/мега/гига/терабайт). снимком (или просто немодифицируемым образом). Пул обязательно указывать, если в кластере больше одного пула.
* `-p|--pool POOL` - Создать образ в заданном пуле (можно не указывать, если пул всего один).
* `--parent PARENT` - Создать легковесный клон на основе образа `PARENT` или снимка `PARENT@SNAP`.
Если `PARENT` - не снимок, он должен быть помечен как образ только для чтения.
* `--enc-key random` - Сгенерировать случайный ключ шифрования AES-256-XTS для нового образа.
* `--enc-key HEX` - Установить заданный ключ AES-256-XTS (64 байта в hex) для нового образа.
* `--enc-key vault:ID` - Использовать ключ из внешнего секрета с заданным ID из Vault.
``` ```
vitastor-cli create --snapshot <snapshot> [ОПЦИИ] <image> vitastor-cli create --snapshot <snapshot> [-p|--pool <id|name>] <image>
vitastor-cli snap-create [ОПЦИИ] <image>@<snapshot> vitastor-cli snap-create [-p|--pool <id|name>] <image>@<snapshot>
``` ```
Создать снимок образа `<image>` (можно использовать любую форму команды). Создать снимок образа `<name>` (можно использовать любую форму команды). Снимок можно создавать без остановки
Снимок можно создавать без остановки клиентов, если пишущих клиентов не больше одного. клиентов, если пишущий клиент максимум 1.
Опции:
* `-p|--pool POOL` - Переместить образ в пул POOL, оставив снимок в старом пуле.
* `--enc-key random` - Изменить ключ шифрования образа на новый случайный ключ AES-256-XTS.
* `--enc-key KEY` - Изменить ключ шифрования образа на заданный ключ, ключ из Vault или пустой ключ.
По умолчанию шифрованные образы сохраняют старый ключ при снятии снимка.
Смотрите также информацию о том, [как экспортировать снимки](qemu.ru.md#экспорт-снимков). Смотрите также информацию о том, [как экспортировать снимки](qemu.ru.md#экспорт-снимков).
@@ -169,7 +156,6 @@ vitastor-cli snap-create [ОПЦИИ] <image>@<snapshot>
* `--deleted 1|0` - Установить/снять флаг "образ удалён" (устанавливается при незавершённом удалении). * `--deleted 1|0` - Установить/снять флаг "образ удалён" (устанавливается при незавершённом удалении).
* `-f|--force` - Разрешить уменьшение или перевод в чтение-запись образа, у которого есть клоны. * `-f|--force` - Разрешить уменьшение или перевод в чтение-запись образа, у которого есть клоны.
* `--down-ok` - Разрешить уменьшение, даже если часть данных останется неудалённой на недоступных OSD. * `--down-ok` - Разрешить уменьшение, даже если часть данных останется неудалённой на недоступных OSD.
* `--enc-key HEX` - Изменить ключ шифрования образа (разрешено только с `--force`).
## dd ## dd
-2
View File
@@ -95,8 +95,6 @@ Options (single-device mode):
Options (both modes): Options (both modes):
``` ```
--tags tag1,tag2 Set new OSD tag(s)
--weight <number> Set new OSD weight (between 0 to 1)
--journal_size 1G/32M Set journal size (area or partition size) --journal_size 1G/32M Set journal size (area or partition size)
--block_size 1M/128k Set blockstore object size --block_size 1M/128k Set blockstore object size
--bitmap_granularity 4k Set bitmap granularity --bitmap_granularity 4k Set bitmap granularity
-2
View File
@@ -96,8 +96,6 @@ vitastor-disk - инструмент командной строки для уп
Опции для обоих режимов: Опции для обоих режимов:
``` ```
--tags tag1,tag2 Задать теги для новых OSD
--weight <number> Задать вес для новых OSD (от 0 до 1)
--journal_size 1G/32M Задать размер журнала (области или раздела журнала) --journal_size 1G/32M Задать размер журнала (области или раздела журнала)
--block_size 1M/128k Задать размер объекта хранилища --block_size 1M/128k Задать размер объекта хранилища
--bitmap_granularity 4k Задать гранулярность битовых карт --bitmap_granularity 4k Задать гранулярность битовых карт
+7 -11
View File
@@ -18,7 +18,7 @@ class AntiEtcdAdapter
cluster = cluster ? (''+(cluster||'')).split(/,+/) : []; cluster = cluster ? (''+(cluster||'')).split(/,+/) : [];
cluster = Object.keys(cluster.reduce((a, url) => cluster = Object.keys(cluster.reduce((a, url) =>
{ {
a[url.toLowerCase().replace(/^(https?:\/\/)?(.*?)(\/.*)?$/, (m, m1, m2) => (m1||'http://')+m2)] = true; a[url.toLowerCase().replace(/^(https?:\/\/)/, '').replace(/\/.*$/, '')] = true;
return a; return a;
}, {})); }, {}));
const cfg_port = config.antietcd_port; const cfg_port = config.antietcd_port;
@@ -26,8 +26,7 @@ class AntiEtcdAdapter
is_local['0.0.0.0'] = true; is_local['0.0.0.0'] = true;
is_local['::'] = true; is_local['::'] = true;
is_local[''] = true; is_local[''] = true;
// split :, 3 -> <schema>:<//ip>:<port> const selected = cluster.map(s => s.split(':', 2)).filter(ip => is_local[ip[0]] && (!cfg_port || ip[1] == cfg_port));
const selected = cluster.map(s => s.split(':', 3)).filter(ip => is_local[ip[1].substr(2)] && (!cfg_port || ip[2] == cfg_port));
if (selected.length > 1) if (selected.length > 1)
{ {
console.error('More than 1 etcd_address matches local IPs, please specify port'); console.error('More than 1 etcd_address matches local IPs, please specify port');
@@ -36,15 +35,12 @@ class AntiEtcdAdapter
else if (selected.length == 1) else if (selected.length == 1)
{ {
const antietcd_config = { const antietcd_config = {
ip: selected[0][1].substr(2), ip: selected[0][0],
port: selected[0][2], port: selected[0][1],
cert: config.antietcd_cert, data: config.antietcd_data_file || ((config.antietcd_data_dir || '/var/lib/vitastor') + '/mon_'+selected[0][1]+'.json.gz'),
key: config.antietcd_key,
ca: config.etcd_ca,
data: config.antietcd_data_file || ((config.antietcd_data_dir || '/var/lib/vitastor') + '/mon_'+selected[0][2]+'.json.gz'),
persist_filter: vitastor_persist_filter({ vitastor_prefix: config.etcd_prefix || '/vitastor' }), persist_filter: vitastor_persist_filter({ vitastor_prefix: config.etcd_prefix || '/vitastor' }),
node_id: selected[0][1].substr(2)+':'+selected[0][2], // node_id = ip:port node_id: selected[0][0]+':'+selected[0][1], // node_id = ip:port
cluster: (cluster.length == 1 ? null : cluster.reduce((a, c) => { a[c.replace(/^(https?:\/\/)/, '')] = c; return a; }, {})), cluster: (cluster.length == 1 ? null : cluster.reduce((a, c) => { a[c] = "http://"+c; return a; }, {})),
cluster_key: (config.etcd_prefix || '/vitastor'), cluster_key: (config.etcd_prefix || '/vitastor'),
stale_read: 1, stale_read: 1,
log_level: 1, log_level: 1,
+6 -27
View File
@@ -1,9 +1,7 @@
// Copyright (c) Vitaliy Filippov, 2019+ // Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 (see README.md for details) // License: VNPL-1.1 (see README.md for details)
const fs = require('fs');
const http = require('http'); const http = require('http');
const https = require('https');
const WebSocket = require('ws'); const WebSocket = require('ws');
const { b64, local_ips } = require('./utils.js'); const { b64, local_ips } = require('./utils.js');
@@ -17,30 +15,11 @@ class EtcdAdapter
this.ws = null; this.ws = null;
this.ws_alive = false; this.ws_alive = false;
this.ws_keepalive_timer = null; this.ws_keepalive_timer = null;
this.opts = {};
} }
parse_config(config) parse_config(config)
{ {
this.parse_etcd_addresses(config.etcd_address||config.etcd_url); this.parse_etcd_addresses(config.etcd_address||config.etcd_url);
if (config.mon_etcd_client_cert || config.etcd_client_cert)
{
this.opts.cert = config.mon_etcd_client_cert || config.etcd_client_cert;
if (this.opts.cert.substr(0, 5) != '-----')
this.opts.cert = fs.readFileSync(this.opts.cert, { encoding: 'utf-8' });
}
if (config.mon_etcd_client_key || config.etcd_client_key)
{
this.opts.key = config.mon_etcd_client_key || config.etcd_client_key;
if (this.opts.key.substr(0, 5) != '-----')
this.opts.key = fs.readFileSync(this.opts.key, { encoding: 'utf-8' });
}
if (config.etcd_ca)
{
this.opts.ca = config.etcd_ca;
if (this.opts.ca.substr(0, 5) != '-----')
this.opts.ca = fs.readFileSync(this.opts.ca, { encoding: 'utf-8' });
}
} }
parse_etcd_addresses(addrs) parse_etcd_addresses(addrs)
@@ -60,7 +39,7 @@ class EtcdAdapter
for (let url of addrs) for (let url of addrs)
{ {
let scheme = 'http'; let scheme = 'http';
url = url.trim().replace(/^(https?):\/\//i, (m, m1) => { scheme = m1.toLowerCase(); return ''; }); url = url.trim().replace(/^(https?):\/\//, (m, m1) => { scheme = m1; return ''; });
const slash = url.indexOf('/'); const slash = url.indexOf('/');
const colon = url.indexOf(':'); const colon = url.indexOf(':');
const is_local = is_local_ip[colon >= 0 ? url.substr(0, colon) : (slash >= 0 ? url.substr(0, slash) : url)]; const is_local = is_local_ip[colon >= 0 ? url.substr(0, colon) : (slash >= 0 ? url.substr(0, slash) : url)];
@@ -151,7 +130,7 @@ class EtcdAdapter
} }
ok(false); ok(false);
}, this.mon.config.etcd_mon_timeout); }, this.mon.config.etcd_mon_timeout);
this.ws = new WebSocket(base+'/watch', this.opts); this.ws = new WebSocket(base+'/watch');
this.ws_used_url = cur_addr; this.ws_used_url = cur_addr;
const fail = () => const fail = () =>
{ {
@@ -293,7 +272,7 @@ class EtcdAdapter
{ {
throw new Error(MON_STOPPED); throw new Error(MON_STOPPED);
} }
const res = await POST(base+path, body, timeout, this.opts); const res = await POST(base+path, body, timeout);
if (this.mon.stopped) if (this.mon.stopped)
{ {
throw new Error(MON_STOPPED); throw new Error(MON_STOPPED);
@@ -319,7 +298,7 @@ class EtcdAdapter
} }
} }
function POST(url, body, timeout, opts) function POST(url, body, timeout)
{ {
return new Promise(ok => return new Promise(ok =>
{ {
@@ -331,10 +310,10 @@ function POST(url, body, timeout, opts)
req = null; req = null;
ok({ error: 'timeout' }); ok({ error: 'timeout' });
}, timeout) : null; }, timeout) : null;
let req = (url.substr(0, 5) == 'https' ? https : http).request(url, { method: 'POST', headers: { let req = http.request(url, { method: 'POST', headers: {
'Content-Type': 'application/json', 'Content-Type': 'application/json',
'Content-Length': body_text.length, 'Content-Length': body_text.length,
}, ...(opts||{}) }, (res) => } }, (res) =>
{ {
if (!req) if (!req)
{ {
+1 -11
View File
@@ -45,14 +45,7 @@ const etcd_tree = {
config_path: "/etc/vitastor/vitastor.conf", config_path: "/etc/vitastor/vitastor.conf",
etcd_prefix: "/vitastor", etcd_prefix: "/vitastor",
// etcd connection - configurable online // etcd connection - configurable online
etcd_address: "http://10.0.115.10:2379/v3", etcd_address: "10.0.115.10:2379/v3",
etcd_client_cert: "",
etcd_client_key: "",
osd_etcd_client_cert: "",
osd_etcd_client_key: "",
mon_etcd_client_cert: "",
mon_etcd_client_key: "",
etcd_ca: "",
// mon // mon
etcd_mon_ttl: 5, // min: 1 etcd_mon_ttl: 5, // min: 1
etcd_mon_timeout: 1000, // ms. min: 0 etcd_mon_timeout: 1000, // ms. min: 0
@@ -224,8 +217,6 @@ const etcd_tree = {
parent_id?: <inode_t>, parent_id?: <inode_t>,
readonly?: boolean, readonly?: boolean,
deleted?: boolean, deleted?: boolean,
enc_key?: string,
meta?: any,
} }
} }
}, */ }, */
@@ -392,7 +383,6 @@ const etcd_tree = {
/* <name>: { /* <name>: {
id: uint64_t, id: uint64_t,
pool_id: uint64_t, pool_id: uint64_t,
// ...plus a copy of everything from config/inode/x/y
}, */ }, */
}, },
maxid: { maxid: {
+1 -1
View File
@@ -16,7 +16,7 @@ async function create_http_server(cfg, handler)
}; };
if (cfg.mon_https_ca) if (cfg.mon_https_ca)
{ {
tls.ca = await fsp.readFile(cfg.mon_https_ca); tls.mon_https_ca = await fsp.readFile(cfg.mon_https_ca);
} }
if (cfg.mon_https_client_auth) if (cfg.mon_https_client_auth)
{ {
+5 -8
View File
@@ -10,19 +10,16 @@ const NO_OSD = 'Z';
async function lp_solve(text) async function lp_solve(text)
{ {
const cp = child_process.spawn('lp_solve'); const cp = child_process.spawn('lp_solve');
let stdout = '', stderr = '', finish_cb, finished = 0; let stdout = '', stderr = '', finish_cb;
cp.stdout.on('data', buf => stdout += buf.toString()); cp.stdout.on('data', buf => stdout += buf.toString());
cp.stderr.on('data', buf => stderr += buf.toString()); cp.stderr.on('data', buf => stderr += buf.toString());
cp.stdout.on('end', () => finish_cb()); cp.on('exit', () => finish_cb && finish_cb());
cp.stderr.on('end', () => finish_cb());
cp.stdin.write(text); cp.stdin.write(text);
cp.stdin.end(); cp.stdin.end();
await new Promise(ok => (finish_cb = () => if (cp.exitCode == null)
{ {
finished++; await new Promise(ok => finish_cb = ok);
if (finished == 2) }
ok();
}));
if (!stdout.trim()) if (!stdout.trim())
{ {
return null; return null;
+1 -1
View File
@@ -87,7 +87,7 @@ function make_hier_tree(global_config, tree)
tree[''] = { children: [] }; tree[''] = { children: [] };
for (const node_id in tree) for (const node_id in tree)
{ {
if (node_id === '') if (node_id === '' || !(tree[node_id].children||[]).length && (tree[node_id].size||0) <= 0)
{ {
continue; continue;
} }
+2 -2
View File
@@ -1,6 +1,6 @@
{ {
"name": "vitastor-mon", "name": "vitastor-mon",
"version": "3.0.5", "version": "3.0.0",
"description": "Vitastor SDS monitor service", "description": "Vitastor SDS monitor service",
"main": "mon-main.js", "main": "mon-main.js",
"scripts": { "scripts": {
@@ -9,7 +9,7 @@
"author": "Vitaliy Filippov", "author": "Vitaliy Filippov",
"license": "UNLICENSED", "license": "UNLICENSED",
"dependencies": { "dependencies": {
"antietcd": "^1.2.4", "antietcd": "^1.1.3",
"sprintf-js": "^1.1.2", "sprintf-js": "^1.1.2",
"ws": "^7.2.5" "ws": "^7.2.5"
}, },
+2 -16
View File
@@ -52,7 +52,6 @@ function recheck_primary(state, global_config, up_osds, osd_tree)
continue; continue;
} }
const aff_osds = get_affinity_osds(pool_cfg, up_osds, osd_tree); const aff_osds = get_affinity_osds(pool_cfg, up_osds, osd_tree);
let paused = false;
for (let pg_num = 1; pg_num <= pool_cfg.pg_count; pg_num++) for (let pg_num = 1; pg_num <= pool_cfg.pg_count; pg_num++)
{ {
if (!state.pg.config.items[pool_id]) if (!state.pg.config.items[pool_id])
@@ -75,19 +74,6 @@ function recheck_primary(state, global_config, up_osds, osd_tree)
); );
new_pg_config.items[pool_id][pg_num].primary = new_primary; new_pg_config.items[pool_id][pg_num].primary = new_primary;
} }
paused = paused || !!pg_cfg.pause;
}
}
if (paused)
{
if (!new_pg_config)
{
new_pg_config = JSON.parse(JSON.stringify(state.pg.config));
}
console.log(`Resuming paused pool ${pool_id}`);
for (const pg in new_pg_config.items[pool_id])
{
delete new_pg_config.items[pool_id][pg].pause;
} }
} }
} }
@@ -192,10 +178,10 @@ async function generate_pool_pgs(state, global_config, pool_id, osd_tree, levels
const rules = use_rules ? get_pg_rules(pool_id, pool_cfg, global_config.placement_levels) : null; const rules = use_rules ? get_pg_rules(pool_id, pool_cfg, global_config.placement_levels) : null;
const folded = fold_failure_domains(Object.values(pool_tree), use_rules ? rules : [ [ [ pool_cfg.failure_domain ] ] ]); const folded = fold_failure_domains(Object.values(pool_tree), use_rules ? rules : [ [ [ pool_cfg.failure_domain ] ] ]);
// FIXME: Remove/merge make_hier_tree() step somewhere, however it's needed to remove empty nodes // FIXME: Remove/merge make_hier_tree() step somewhere, however it's needed to remove empty nodes
const folded_tree = make_hier_tree(global_config, folded.nodes.reduce((a, c) => { a[c.id] = c; return a; }, {})); const folded_tree = make_hier_tree(global_config, folded.nodes);
const old_pg_count = prev_pgs.length; const old_pg_count = prev_pgs.length;
const optimize_cfg = { const optimize_cfg = {
osd_weights: folded.nodes.reduce((a, c) => { if (/^\d+$/.exec(c.id) && c.size != null) { a[c.id] = c.size||0; } return a; }, {}), osd_weights: folded.nodes.reduce((a, c) => { if (Number(c.id)) { a[c.id] = c.size; } return a; }, {}),
combinator: use_rules combinator: use_rules
// new algorithm: // new algorithm:
? new RuleCombinator(folded_tree, rules, pool_cfg.max_osd_combinations) ? new RuleCombinator(folded_tree, rules, pool_cfg.max_osd_combinations)
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "vitastor", "name": "vitastor",
"version": "3.0.5", "version": "3.0.0",
"description": "Low-level native bindings to Vitastor client library", "description": "Low-level native bindings to Vitastor client library",
"main": "index.js", "main": "index.js",
"keywords": [ "keywords": [
+232 -30
View File
@@ -50,7 +50,7 @@ from cinder.volume import configuration
from cinder.volume import driver from cinder.volume import driver
from cinder.volume import volume_utils from cinder.volume import volume_utils
VITASTOR_VERSION = '3.0.5' VITASTOR_VERSION = '3.0.0'
LOG = logging.getLogger(__name__) LOG = logging.getLogger(__name__)
@@ -275,7 +275,7 @@ class VitastorDriver(driver.CloneableImageVD,
LOG.exception('error getting vitastor pool stats: '+str(e)) LOG.exception('error getting vitastor pool stats: '+str(e))
self._stats = stats self._stats = stats
def get_volume_stats(self, refresh=False): def get_volume_stats(self, refresh=False):
"""Get volume stats. """Get volume stats.
If 'refresh' is True, run update the stats first. If 'refresh' is True, run update the stats first.
@@ -291,14 +291,6 @@ class VitastorDriver(driver.CloneableImageVD,
else: else:
return (1 + resp['kvs'][0]['value'], resp['kvs'][0]['mod_revision']) return (1 + resp['kvs'][0]['value'], resp['kvs'][0]['mod_revision'])
def _cli(self, descr, *args):
args = [ 'vitastor-cli', *args, *(self._vitastor_args()) ]
try:
self._execute(*args)
except processutils.ProcessExecutionError as exc:
LOG.error("Failed to "+descr+": "+exc)
raise exception.VolumeBackendAPIException(data = exc.stderr)
def create_volume(self, volume): def create_volume(self, volume):
"""Creates a logical volume.""" """Creates a logical volume."""
@@ -310,7 +302,7 @@ class VitastorDriver(driver.CloneableImageVD,
LOG.debug("creating volume '%s'", vol_name) LOG.debug("creating volume '%s'", vol_name)
self._cli('create volume', 'create', vol_name, '--size', size) self._create_image(vol_name, { 'size': size })
if volume.encryption_key_id: if volume.encryption_key_id:
self._create_encrypted_volume(volume, volume.obj_context) self._create_encrypted_volume(volume, volume.obj_context)
@@ -354,7 +346,7 @@ class VitastorDriver(driver.CloneableImageVD,
snap_name = utils.convert_str(snapshot.name) snap_name = utils.convert_str(snapshot.name)
if snap_name.find('@') >= 0 or snap_name.find('/') >= 0: if snap_name.find('@') >= 0 or snap_name.find('/') >= 0:
raise exception.VolumeBackendAPIException(data = '@ and / are forbidden in volume and snapshot names') raise exception.VolumeBackendAPIException(data = '@ and / are forbidden in volume and snapshot names')
self._cli('create snapshot', 'snap-create', vol_name+'@'+snap_name) self._create_snapshot(vol_name, vol_name+'@'+snap_name)
def snapshot_revert_use_temp_snapshot(self): def snapshot_revert_use_temp_snapshot(self):
"""Disable the use of a temporary snapshot on revert.""" """Disable the use of a temporary snapshot on revert."""
@@ -367,8 +359,21 @@ class VitastorDriver(driver.CloneableImageVD,
snap_name = utils.convert_str(snapshot.name) snap_name = utils.convert_str(snapshot.name)
# Delete the image and recreate it from the snapshot # Delete the image and recreate it from the snapshot
self._cli('delete image', 'rm', vol_name) args = [ 'vitastor-cli', 'rm', vol_name, *(self._vitastor_args()) ]
self._cli('recreate image', 'create', '--parent', vol_name+'@'+snap_name, vol_name) try:
self._execute(*args)
except processutils.ProcessExecutionError as exc:
LOG.error("Failed to delete image "+vol_name+": "+exc)
raise exception.VolumeBackendAPIException(data = exc.stderr)
args = [
'vitastor-cli', 'create', '--parent', vol_name+'@'+snap_name,
vol_name, *(self._vitastor_args())
]
try:
self._execute(*args)
except processutils.ProcessExecutionError as exc:
LOG.error("Failed to recreate image "+vol_name+" from "+vol_name+"@"+snap_name+": "+exc)
raise exception.VolumeBackendAPIException(data = exc.stderr)
def delete_snapshot(self, snapshot): def delete_snapshot(self, snapshot):
"""Deletes a snapshot.""" """Deletes a snapshot."""
@@ -376,7 +381,15 @@ class VitastorDriver(driver.CloneableImageVD,
vol_name = utils.convert_str(snapshot.volume_name) vol_name = utils.convert_str(snapshot.volume_name)
snap_name = utils.convert_str(snapshot.name) snap_name = utils.convert_str(snapshot.name)
self._cli('remove snapshot', 'rm', vol_name+'@'+snap_name) args = [
'vitastor-cli', 'rm', vol_name+'@'+snap_name,
*(self._vitastor_args())
]
try:
self._execute(*args)
except processutils.ProcessExecutionError as exc:
LOG.error("Failed to remove snapshot "+vol_name+'@'+snap_name+": "+exc)
raise exception.VolumeBackendAPIException(data = exc.stderr)
def _child_count(self, parents): def _child_count(self, parents):
children = 0 children = 0
@@ -414,7 +427,13 @@ class VitastorDriver(driver.CloneableImageVD,
if src_vref.admin_metadata.get('readonly') == 'True': if src_vref.admin_metadata.get('readonly') == 'True':
# source volume is a volume-image cache entry or other readonly volume # source volume is a volume-image cache entry or other readonly volume
# clone without intermediate snapshot # clone without intermediate snapshot
self._cli('create clone', 'create', '--parent', src_name, '--size', size, dest_name) src = self._get_image(src_name)
LOG.debug("creating image '%s' from '%s'", dest_name, src_name)
new_cfg = self._create_image(dest_name, {
'size': size,
'parent_id': src['idx']['id'],
'parent_pool_id': src['idx']['pool_id'],
})
return {} return {}
clone_snap = "%s@%s.clone_snap" % (src_name, dest_name) clone_snap = "%s@%s.clone_snap" % (src_name, dest_name)
@@ -427,12 +446,15 @@ class VitastorDriver(driver.CloneableImageVD,
clone_snap = dest_name clone_snap = dest_name
make_img = False make_img = False
LOG.debug("creating snapshot '%s'", clone_snap) LOG.debug("creating layer '%s' under '%s'", clone_snap, src_name)
self._cli('create base snapshot', 'snap-create', '--allow-existing', '1', clone_snap) new_cfg = self._create_snapshot(src_name, clone_snap, True)
if make_img: if make_img:
# Then create a clone from it # Then create a clone from it
self._cli('create clone', 'create', '--parent', clone_snap, '--size', size, dest_name) new_cfg = self._create_image(dest_name, {
'size': size,
'parent_id': new_cfg['parent_id'],
'parent_pool_id': new_cfg['parent_pool_id'],
})
return {} return {}
@@ -442,8 +464,7 @@ class VitastorDriver(driver.CloneableImageVD,
vol_name = utils.convert_str(volume.name) vol_name = utils.convert_str(volume.name)
snap_name = utils.convert_str(snapshot.name) snap_name = utils.convert_str(snapshot.name)
src_snap = 'volume-'+snapshot.volume_id+'@'+snap_name snap = self._get_image('volume-'+snapshot.volume_id+'@'+snap_name)
snap = self._get_image(src_snap)
if not snap: if not snap:
raise exception.SnapshotNotFound(snapshot_id = snap_name) raise exception.SnapshotNotFound(snapshot_id = snap_name)
snap_inode_id = int(resp['responses'][0]['kvs'][0]['value']['id']) snap_inode_id = int(resp['responses'][0]['kvs'][0]['value']['id'])
@@ -452,8 +473,12 @@ class VitastorDriver(driver.CloneableImageVD,
size = snap['cfg']['size'] size = snap['cfg']['size']
if int(volume.size): if int(volume.size):
size = int(volume.size) * units.Gi size = int(volume.size) * units.Gi
new_cfg = self._create_image(vol_name, {
'size': size,
'parent_id': snap['idx']['id'],
'parent_pool_id': snap['idx']['pool_id'],
})
self._cli('create clone', 'create', vol_name, '--size', size, '--parent', src_snap)
return {} return {}
def _vitastor_args(self): def _vitastor_args(self):
@@ -480,7 +505,49 @@ class VitastorDriver(driver.CloneableImageVD,
"""Deletes a logical volume.""" """Deletes a logical volume."""
vol_name = utils.convert_str(volume.name) vol_name = utils.convert_str(volume.name)
self._cli('delete volume', 'rm', '--matching', vol_name, vol_name+'@*', '--progress', '0')
# Find the volume and all its snapshots
range_end = b'index/image/' + vol_name.encode('utf-8')
range_end = range_end[0 : len(range_end)-1] + six.int2byte(range_end[len(range_end)-1] + 1)
resp = self._etcd_txn({ 'success': [
{ 'request_range': { 'key': 'index/image/'+vol_name, 'range_end': range_end } },
] })
if len(resp['responses'][0]['kvs']) == 0:
# already deleted
LOG.info("volume %s no longer exists in backend", vol_name)
return
layers = resp['responses'][0]['kvs']
layer_ids = {}
for kv in layers:
inode_id = int(kv['value']['id'])
pool_id = int(kv['value']['pool_id'])
inode_pool_id = (pool_id << 48) | (inode_id & 0xffffffffffff)
layer_ids[inode_pool_id] = True
# Check if the volume has clones and raise 'busy' if so
children = self._child_count(layer_ids)
if children > 0:
raise exception.VolumeIsBusy(volume_name = vol_name)
# Clear data
for kv in layers:
args = [
'vitastor-cli', 'rm-data', '--pool', str(kv['value']['pool_id']),
'--inode', str(kv['value']['id']), '--progress', '0',
*(self._vitastor_args())
]
try:
self._execute(*args)
except processutils.ProcessExecutionError as exc:
LOG.error("Failed to remove layer "+kv['key']+": "+exc)
raise exception.VolumeBackendAPIException(data = exc.stderr)
# Delete all layers from etcd
requests = []
for kv in layers:
requests.append({ 'request_delete_range': { 'key': kv['key'] } })
requests.append({ 'request_delete_range': { 'key': 'config/inode/'+str(kv['value']['pool_id'])+'/'+str(kv['value']['id']) } })
self._etcd_txn({ 'success': requests })
def retype(self, context, volume, new_type, diff, host): def retype(self, context, volume, new_type, diff, host):
"""Change extra type specifications for a volume.""" """Change extra type specifications for a volume."""
@@ -500,6 +567,98 @@ class VitastorDriver(driver.CloneableImageVD,
"""Removes an export for a logical volume.""" """Removes an export for a logical volume."""
pass pass
def _create_image(self, vol_name, cfg):
pool_s = str(self.cfg['pool_id'])
image_id = 0
while image_id == 0:
# check if the image already exists and find a free ID
resp = self._etcd_txn({ 'success': [
{ 'request_range': { 'key': 'index/image/'+vol_name } },
{ 'request_range': { 'key': 'index/maxid/'+pool_s } },
] })
if len(resp['responses'][0]['kvs']) > 0:
# already exists
raise exception.VolumeBackendAPIException(data = 'Volume '+vol_name+' already exists')
image_id, id_mod = self._next_id(resp['responses'][1])
# try to create the image
resp = self._etcd_txn({ 'compare': [
{ 'target': 'MOD', 'mod_revision': id_mod, 'key': 'index/maxid/'+pool_s },
{ 'target': 'VERSION', 'version': 0, 'key': 'index/image/'+vol_name },
{ 'target': 'VERSION', 'version': 0, 'key': 'config/inode/'+pool_s+'/'+str(image_id) },
], 'success': [
{ 'request_put': { 'key': 'index/maxid/'+pool_s, 'value': image_id } },
{ 'request_put': { 'key': 'index/image/'+vol_name, 'value': json.dumps({
'id': image_id, 'pool_id': self.cfg['pool_id']
}) } },
{ 'request_put': { 'key': 'config/inode/'+pool_s+'/'+str(image_id), 'value': json.dumps({
**cfg, 'name': vol_name,
}) } },
] })
if not resp.get('succeeded'):
# repeat
image_id = 0
def _create_snapshot(self, vol_name, snap_vol_name, allow_existing = False):
while True:
# check if the image already exists and snapshot doesn't
resp = self._etcd_txn({ 'success': [
{ 'request_range': { 'key': 'index/image/'+vol_name } },
{ 'request_range': { 'key': 'index/image/'+snap_vol_name } },
] })
if len(resp['responses'][0]['kvs']) == 0:
raise exception.VolumeBackendAPIException(data = 'Volume '+vol_name+' does not exist')
if len(resp['responses'][1]['kvs']) > 0:
if allow_existing:
snap_idx = resp['responses'][1]['kvs'][0]['value']
resp = self._etcd_txn({ 'success': [
{ 'request_range': { 'key': 'config/inode/'+str(snap_idx['pool_id'])+'/'+str(snap_idx['id']) } },
] })
if len(resp['responses'][0]['kvs']) == 0:
raise exception.VolumeBackendAPIException(data =
'Volume '+snap_vol_name+' is already indexed, but does not exist'
)
return resp['responses'][0]['kvs'][0]['value']
raise exception.VolumeBackendAPIException(
data = 'Volume '+snap_vol_name+' already exists'
)
vol_idx = resp['responses'][0]['kvs'][0]['value']
vol_idx_mod = resp['responses'][0]['kvs'][0]['mod_revision']
# get image inode config and find a new ID
resp = self._etcd_txn({ 'success': [
{ 'request_range': { 'key': 'config/inode/'+str(vol_idx['pool_id'])+'/'+str(vol_idx['id']) } },
{ 'request_range': { 'key': 'index/maxid/'+str(self.cfg['pool_id']) } },
] })
if len(resp['responses'][0]['kvs']) == 0:
raise exception.VolumeBackendAPIException(data = 'Volume '+vol_name+' does not exist')
vol_cfg = resp['responses'][0]['kvs'][0]['value']
vol_mod = resp['responses'][0]['kvs'][0]['mod_revision']
new_id, id_mod = self._next_id(resp['responses'][1])
# try to redirect image to the new inode
new_cfg = {
**vol_cfg, 'name': vol_name, 'parent_id': vol_idx['id'], 'parent_pool_id': vol_idx['pool_id']
}
resp = self._etcd_txn({ 'compare': [
{ 'target': 'MOD', 'mod_revision': vol_idx_mod, 'key': 'index/image/'+vol_name },
{ 'target': 'MOD', 'mod_revision': vol_mod, 'key': 'config/inode/'+str(vol_idx['pool_id'])+'/'+str(vol_idx['id']) },
{ 'target': 'MOD', 'mod_revision': id_mod, 'key': 'index/maxid/'+str(self.cfg['pool_id']) },
{ 'target': 'VERSION', 'version': 0, 'key': 'index/image/'+snap_vol_name },
{ 'target': 'VERSION', 'version': 0, 'key': 'config/inode/'+str(self.cfg['pool_id'])+'/'+str(new_id) },
], 'success': [
{ 'request_put': { 'key': 'index/maxid/'+str(self.cfg['pool_id']), 'value': new_id } },
{ 'request_put': { 'key': 'index/image/'+vol_name, 'value': json.dumps({
'id': new_id, 'pool_id': self.cfg['pool_id']
}) } },
{ 'request_put': { 'key': 'config/inode/'+str(self.cfg['pool_id'])+'/'+str(new_id), 'value': json.dumps(new_cfg) } },
{ 'request_put': { 'key': 'index/image/'+snap_vol_name, 'value': json.dumps({
'id': vol_idx['id'], 'pool_id': vol_idx['pool_id']
}) } },
{ 'request_put': { 'key': 'config/inode/'+str(vol_idx['pool_id'])+'/'+str(vol_idx['id']), 'value': json.dumps({
**vol_cfg, 'name': snap_vol_name, 'readonly': True
}) } }
] })
if resp.get('succeeded'):
return new_cfg
def initialize_connection(self, volume, connector): def initialize_connection(self, volume, connector):
data = { data = {
'driver_volume_type': 'vitastor', 'driver_volume_type': 'vitastor',
@@ -538,9 +697,13 @@ class VitastorDriver(driver.CloneableImageVD,
size = int(volume.size) * units.Gi size = int(volume.size) * units.Gi
dest_name = utils.convert_str(volume.name) dest_name = utils.convert_str(volume.name)
# Find or create the base snapshot # Find or create the base snapshot
self._cli('create base snapshot', 'create', '--allow-existing', '1', base_vol.name+'@.clone_snap') snap_cfg = self._create_snapshot(base_vol.name, base_vol.name+'@.clone_snap', True)
# Then create a clone from it # Then create a clone from it
self._cli('create clone', 'create', dest_name, '--size', size, '--parent', base_vol.name+'@.clone_snap') new_cfg = self._create_image(dest_name, {
'size': size,
'parent_id': snap_cfg['parent_id'],
'parent_pool_id': snap_cfg['parent_pool_id'],
})
return ({}, True) return ({}, True)
return ({}, False) return ({}, False)
@@ -607,8 +770,26 @@ class VitastorDriver(driver.CloneableImageVD,
def extend_volume(self, volume, new_size): def extend_volume(self, volume, new_size):
"""Extend an existing volume.""" """Extend an existing volume."""
vol_name = utils.convert_str(volume.name) vol_name = utils.convert_str(volume.name)
size = int(new_size) * units.Gi while True:
self._cli('extend volume', 'modify', vol_name, '--resize', new_size) vol = self._get_image(vol_name)
if not vol:
raise exception.VolumeBackendAPIException(data = 'Volume '+vol_name+' does not exist')
# change size
size = int(new_size) * units.Gi
if size == vol['cfg']['size']:
break
resp = self._etcd_txn({ 'compare': [ {
'target': 'MOD',
'mod_revision': vol['cfg_mod'],
'key': 'config/inode/'+str(vol['idx']['pool_id'])+'/'+str(vol['idx']['id']),
} ], 'success': [
{ 'request_put': {
'key': 'config/inode/'+str(vol['idx']['pool_id'])+'/'+str(vol['idx']['id']),
'value': json.dumps({ **vol['cfg'], 'size': size }),
} },
] })
if resp.get('succeeded'):
break
LOG.debug( LOG.debug(
"Extend volume from %(old_size)s GB to %(new_size)s GB.", "Extend volume from %(old_size)s GB to %(new_size)s GB.",
{'old_size': volume.size, 'new_size': new_size} {'old_size': volume.size, 'new_size': new_size}
@@ -681,7 +862,28 @@ class VitastorDriver(driver.CloneableImageVD,
""" """
from_name = self._get_existing_name(existing_ref) from_name = self._get_existing_name(existing_ref)
to_name = utils.convert_str(volume.name) to_name = utils.convert_str(volume.name)
self._cli('rename', 'modify', from_name, '--rename', to_name) self._rename(from_name, to_name)
def _rename(self, from_name, to_name):
while True:
vol = self._get_image(from_name)
if not vol:
raise exception.VolumeBackendAPIException(data = 'Volume '+from_name+' does not exist')
to = self._get_image(to_name)
if to:
raise exception.VolumeBackendAPIException(data = 'Volume '+to_name+' already exists')
resp = self._etcd_txn({ 'compare': [
{ 'target': 'MOD', 'mod_revision': vol['idx_mod'], 'key': 'index/image/'+vol['cfg']['name'] },
{ 'target': 'MOD', 'mod_revision': vol['cfg_mod'], 'key': 'config/inode/'+str(vol['idx']['pool_id'])+'/'+str(vol['idx']['id']) },
{ 'target': 'VERSION', 'version': 0, 'key': 'index/image/'+to_name },
], 'success': [
{ 'request_delete_range': { 'key': 'index/image/'+vol['cfg']['name'] } },
{ 'request_put': { 'key': 'index/image/'+to_name, 'value': json.dumps(vol['idx']) } },
{ 'request_put': { 'key': 'config/inode/'+str(vol['idx']['pool_id'])+'/'+str(vol['idx']['id']),
'value': json.dumps({ **vol['cfg'], 'name': to_name }) } },
] })
if resp.get('succeeded'):
break
def unmanage(self, volume): def unmanage(self, volume):
pass pass
@@ -754,7 +956,7 @@ class VitastorDriver(driver.CloneableImageVD,
snap_name = self._get_existing_name(existing_ref) snap_name = self._get_existing_name(existing_ref)
from_name = vol_name+'@'+snap_name from_name = vol_name+'@'+snap_name
to_name = vol_name+'@'+utils.convert_str(snapshot.name) to_name = vol_name+'@'+utils.convert_str(snapshot.name)
self._cli('rename', 'modify', from_name, '--rename', to_name) self._rename(from_name, to_name)
def unmanage_snapshot(self, snapshot): def unmanage_snapshot(self, snapshot):
"""Removes the specified snapshot from Cinder management.""" """Removes the specified snapshot from Cinder management."""
@@ -1,39 +0,0 @@
From 98d3f68a40130c438854f61db6025f9e9b099cb6 Mon Sep 17 00:00:00 2001
From: Vitaliy Filippov <vitalifster@gmail.com>
Date: Sat, 20 Dec 2025 14:44:35 +0300
Subject: [PATCH] Do not require atomic writes to be power of 2 sized and
aligned on length boundary
It contradicts NVMe specification where alignment is only required when atomic
write boundary (NABSPF/NABO) is set and highly limits usage of NVMe atomic writes
Signed-off-by: Vitaliy Filippov <vitalifster@gmail.com>
---
fs/read_write.c | 8 --------
1 file changed, 8 deletions(-)
diff --git a/fs/read_write.c b/fs/read_write.c
index 833bae068770..5467d710108d 100644
--- a/fs/read_write.c
+++ b/fs/read_write.c
@@ -1802,17 +1802,9 @@ int generic_file_rw_checks(struct file *file_in, struct file *file_out)
int generic_atomic_write_valid(struct kiocb *iocb, struct iov_iter *iter)
{
- size_t len = iov_iter_count(iter);
-
if (!iter_is_ubuf(iter))
return -EINVAL;
- if (!is_power_of_2(len))
- return -EINVAL;
-
- if (!IS_ALIGNED(iocb->ki_pos, len))
- return -EINVAL;
-
if (!(iocb->ki_flags & IOCB_DIRECT))
return -EOPNOTSUPP;
--
2.51.0
+1 -1
View File
@@ -21,7 +21,7 @@ rpmbuild -bp fio.spec
cd $VITASTOR cd $VITASTOR
VER=$(grep ^Version: rpm/vitastor-$REL.spec | awk '{print $2}') VER=$(grep ^Version: rpm/vitastor-$REL.spec | awk '{print $2}')
rm -rf fio rm -rf fio
ln -s $(ls -d ~/rpmbuild/BUILD/fio*/ | grep -v SPECPARTS) fio ln -s ~/rpmbuild/BUILD/fio*/ fio
sh copy-fio-includes.sh sh copy-fio-includes.sh
rm fio rm fio
mv fio-copy fio mv fio-copy fio
-17
View File
@@ -1,17 +0,0 @@
# Build packages for AlmaLinux 10 inside a container
# cd ..
# docker pull --platform=linux/amd64/v2 quay.io/almalinuxorg/almalinux:10
# docker build -t vitastor-buildenv:el10 -f rpm/vitastor-el10.Dockerfile .
# docker run -i --rm -v ./:/root/vitastor vitastor-buildenv:el10 /root/vitastor/rpm/vitastor-build.sh
FROM quay.io/almalinuxorg/almalinux:10
WORKDIR /root
RUN sed -i 's/enabled=0/enabled=1/' /etc/yum.repos.d/*.repo
RUN dnf -y install epel-release dnf-plugins-core
RUN dnf -y install https://vitastor.io/rpms/centos/10/vitastor-release-1.0-1.el10.noarch.rpm
RUN dnf -y install gcc-c++ gperftools-devel fio nodejs rpm-build jerasure-devel isa-l-devel gf-complete-devel rdma-core-devel cmake libnl3-devel c-ares-devel
RUN dnf download --source fio
RUN rpm --nomd5 -i fio*.src.rpm
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --spec fio.spec
-199
View File
@@ -1,199 +0,0 @@
Name: vitastor
Version: 3.0.5
Release: 1%{?dist}
Summary: Vitastor, a fast software-defined clustered block storage
License: Vitastor Network Public License 1.1
URL: https://vitastor.io/
Source0: vitastor-3.0.5.el10.tar.gz
BuildRequires: gperftools-devel
BuildRequires: gcc-c++
BuildRequires: nodejs >= 10
BuildRequires: jerasure-devel
BuildRequires: isa-l-devel
BuildRequires: gf-complete-devel
BuildRequires: rdma-core-devel
BuildRequires: cmake
BuildRequires: libnl3-devel
BuildRequires: c-ares-devel
Requires: vitastor-osd = %{version}-%{release}
Requires: vitastor-mon = %{version}-%{release}
Requires: vitastor-client = %{version}-%{release}
Requires: vitastor-client-devel = %{version}-%{release}
Requires: vitastor-fio = %{version}-%{release}
%description
Vitastor is a small, simple and fast clustered block storage (storage for VM drives),
architecturally similar to Ceph which means strong consistency, primary-replication,
symmetric clustering and automatic data distribution over any number of drives of any
size with configurable redundancy (replication or erasure codes/XOR).
%package -n vitastor-osd
Summary: Vitastor - OSD
Requires: vitastor-client = %{version}-%{release}
Requires: util-linux
Requires: parted
%description -n vitastor-osd
Vitastor object storage daemon, i.e. server program that stores data.
%package -n vitastor-mon
Summary: Vitastor - monitor
Requires: nodejs >= 10
Requires: lpsolve
%description -n vitastor-mon
Vitastor monitor, i.e. server program responsible for watching cluster state and
scheduling cluster-level operations.
%package -n vitastor-client
Summary: Vitastor - client
%description -n vitastor-client
Vitastor client library and command-line interface.
%package -n vitastor-client-devel
Summary: Vitastor - development files
Group: Development/Libraries
Requires: vitastor-client = %{version}-%{release}
%description -n vitastor-client-devel
Vitastor library headers for development.
%package -n vitastor-fio
Summary: Vitastor - fio drivers
Group: Development/Libraries
Requires: vitastor-client = %{version}-%{release}
Requires: fio = 3.36-5.el10
%description -n vitastor-fio
Vitastor fio drivers for benchmarking.
%package -n vitastor-opennebula
Summary: Vitastor for OpenNebula
Group: Development/Libraries
Requires: vitastor-client
Requires: jq
Requires: python3-lxml
Requires: patch
Requires: qemu-kvm-block-vitastor
%description -n vitastor-opennebula
Vitastor storage plugin for OpenNebula.
%prep
%setup -q
%build
%cmake
%cmake_build
%install
rm -rf $RPM_BUILD_ROOT
%cmake_install
cd mon
npm install --production
cd ..
mkdir -p %buildroot/usr/lib/vitastor
cp -r mon %buildroot/usr/lib/vitastor
mv %buildroot/usr/lib/vitastor/mon/scripts/make-etcd %buildroot/usr/lib/vitastor/mon/
mkdir -p %buildroot/lib/systemd/system
cp mon/scripts/vitastor.target mon/scripts/vitastor-mon.service mon/scripts/vitastor-osd@.service %buildroot/lib/systemd/system
mkdir -p %buildroot/lib/udev/rules.d
cp mon/scripts/90-vitastor.rules %buildroot/lib/udev/rules.d
mkdir -p %buildroot/var/lib/one
cp -r opennebula/remotes %buildroot/var/lib/one
cp opennebula/install.sh %buildroot/var/lib/one/remotes/datastore/vitastor/
mkdir -p %buildroot/etc/
cp -r opennebula/sudoers.d %buildroot/etc/
%files
%doc GPL-2.0.txt VNPL-1.1.txt README.md README-ru.md
%files -n vitastor-osd
%_bindir/vitastor-osd
%_bindir/vitastor-disk
%_bindir/vitastor-dump-journal
/lib/systemd/system/vitastor-osd@.service
/lib/systemd/system/vitastor.target
/lib/udev/rules.d/90-vitastor.rules
%pre -n vitastor-osd
groupadd -r -f vitastor 2>/dev/null ||:
useradd -r -g vitastor -s /sbin/nologin -c "Vitastor daemons" -M -d /nonexistent vitastor 2>/dev/null ||:
install -o vitastor -g vitastor -d /var/log/vitastor
mkdir -p /etc/vitastor
%files -n vitastor-mon
/usr/lib/vitastor/mon
/lib/systemd/system/vitastor-mon.service
%pre -n vitastor-mon
groupadd -r -f vitastor 2>/dev/null ||:
useradd -r -g vitastor -s /sbin/nologin -c "Vitastor daemons" -M -d /nonexistent vitastor 2>/dev/null ||:
mkdir -p /etc/vitastor
mkdir -p /var/lib/vitastor
chown vitastor:vitastor /var/lib/vitastor
%files -n vitastor-client
%_bindir/vitastor-nbd
%_bindir/vitastor-ublk
%_bindir/vitastor-nfs
%_bindir/vitastor-cli
%_bindir/vitastor-rm
%_bindir/vitastor-kv
%_bindir/vitastor-kv-stress
%_bindir/vita
%_libdir/libvitastor_client.so*
%_libdir/libvitastor_kv.so*
%files -n vitastor-client-devel
%_includedir/vitastor_c.h
%_includedir/vitastor_kv.h
%_libdir/pkgconfig
%files -n vitastor-fio
%_libdir/libfio_vitastor.so
%_libdir/libfio_vitastor_blk.so
%_libdir/libfio_vitastor_sec.so
%files -n vitastor-opennebula
/var/lib/one
/etc/sudoers.d/opennebula-vitastor
%triggerin -n vitastor-opennebula -- opennebula
[ $2 = 0 ] || exit 0
/var/lib/one/remotes/datastore/vitastor/install.sh
# Turn off the brp-python-bytecompile script
%global __os_install_post %(echo '%{__os_install_post}' | sed -e 's!/usr/lib[^[:space:]]*/brp-python-bytecompile[[:space:]].*$!!g')
%changelog
+1 -1
View File
@@ -15,7 +15,7 @@ RUN yum -y --enablerepo=extras install centos-release-scl epel-release yum-utils
RUN perl -i -pe 's!mirrorlist=!#mirrorlist=!s; s!#\s*baseurl=http://mirror.centos.org!baseurl=http://vault.centos.org!' /etc/yum.repos.d/CentOS-SCLo-scl*.repo RUN perl -i -pe 's!mirrorlist=!#mirrorlist=!s; s!#\s*baseurl=http://mirror.centos.org!baseurl=http://vault.centos.org!' /etc/yum.repos.d/CentOS-SCLo-scl*.repo
RUN yum -y install https://vitastor.io/rpms/centos/7/vitastor-release-1.0-1.el7.noarch.rpm RUN yum -y install https://vitastor.io/rpms/centos/7/vitastor-release-1.0-1.el7.noarch.rpm
RUN yum -y install devtoolset-9-gcc-c++ devtoolset-9-libatomic-devel gcc make cmake gperftools-devel \ RUN yum -y install devtoolset-9-gcc-c++ devtoolset-9-libatomic-devel gcc make cmake gperftools-devel \
fio rh-nodejs12 jerasure-devel libisa-l-devel gf-complete-devel rdma-core-devel libnl3-devel c-ares-devel fio rh-nodejs12 jerasure-devel libisa-l-devel gf-complete-devel rdma-core-devel libnl3-devel
RUN yumdownloader --disablerepo=centos-sclo-rh --source fio RUN yumdownloader --disablerepo=centos-sclo-rh --source fio
RUN rpm --nomd5 -i fio*.src.rpm RUN rpm --nomd5 -i fio*.src.rpm
RUN rm -f /etc/yum.repos.d/CentOS-Media.repo RUN rm -f /etc/yum.repos.d/CentOS-Media.repo
+2 -3
View File
@@ -1,11 +1,11 @@
Name: vitastor Name: vitastor
Version: 3.0.5 Version: 3.0.0
Release: 1%{?dist} Release: 1%{?dist}
Summary: Vitastor, a fast software-defined clustered block storage Summary: Vitastor, a fast software-defined clustered block storage
License: Vitastor Network Public License 1.1 License: Vitastor Network Public License 1.1
URL: https://vitastor.io/ URL: https://vitastor.io/
Source0: vitastor-3.0.5.el7.tar.gz Source0: vitastor-3.0.0.el7.tar.gz
BuildRequires: gperftools-devel BuildRequires: gperftools-devel
BuildRequires: devtoolset-9-gcc-c++ BuildRequires: devtoolset-9-gcc-c++
@@ -17,7 +17,6 @@ BuildRequires: gf-complete-devel
BuildRequires: rdma-core-devel BuildRequires: rdma-core-devel
BuildRequires: cmake3 BuildRequires: cmake3
BuildRequires: libnl3-devel BuildRequires: libnl3-devel
BuildRequires: c-ares-devel
Requires: vitastor-osd = %{version}-%{release} Requires: vitastor-osd = %{version}-%{release}
Requires: vitastor-mon = %{version}-%{release} Requires: vitastor-mon = %{version}-%{release}
Requires: vitastor-client = %{version}-%{release} Requires: vitastor-client = %{version}-%{release}
+1 -1
View File
@@ -13,7 +13,7 @@ RUN dnf -y install centos-release-advanced-virtualization epel-release dnf-plugi
RUN sed -i 's/^mirrorlist=/#mirrorlist=/; s!#baseurl=.*!baseurl=http://vault.centos.org/centos/8.4.2105/virt/$basearch/$avdir/!; s!^baseurl=.*Source/.*!baseurl=http://vault.centos.org/centos/8.4.2105/virt/Source/advanced-virtualization/!' /etc/yum.repos.d/CentOS-Advanced-Virtualization.repo RUN sed -i 's/^mirrorlist=/#mirrorlist=/; s!#baseurl=.*!baseurl=http://vault.centos.org/centos/8.4.2105/virt/$basearch/$avdir/!; s!^baseurl=.*Source/.*!baseurl=http://vault.centos.org/centos/8.4.2105/virt/Source/advanced-virtualization/!' /etc/yum.repos.d/CentOS-Advanced-Virtualization.repo
RUN yum -y install https://vitastor.io/rpms/centos/8/vitastor-release-1.0-1.el8.noarch.rpm RUN yum -y install https://vitastor.io/rpms/centos/8/vitastor-release-1.0-1.el8.noarch.rpm
RUN dnf -y install gcc-toolset-9 gcc-toolset-9-gcc-c++ gperftools-devel \ RUN dnf -y install gcc-toolset-9 gcc-toolset-9-gcc-c++ gperftools-devel \
fio nodejs rpm-build jerasure-devel libisa-l-devel gf-complete-devel libibverbs-devel libarchive cmake libnl3-devel c-ares-devel fio nodejs rpm-build jerasure-devel libisa-l-devel gf-complete-devel libibverbs-devel libarchive cmake libnl3-devel
RUN dnf download --source fio RUN dnf download --source fio
RUN rpm --nomd5 -i fio*.src.rpm RUN rpm --nomd5 -i fio*.src.rpm
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --enablerepo=powertools --spec fio.spec RUN cd ~/rpmbuild/SPECS && dnf builddep -y --enablerepo=powertools --spec fio.spec
+2 -3
View File
@@ -1,11 +1,11 @@
Name: vitastor Name: vitastor
Version: 3.0.5 Version: 3.0.0
Release: 1%{?dist} Release: 1%{?dist}
Summary: Vitastor, a fast software-defined clustered block storage Summary: Vitastor, a fast software-defined clustered block storage
License: Vitastor Network Public License 1.1 License: Vitastor Network Public License 1.1
URL: https://vitastor.io/ URL: https://vitastor.io/
Source0: vitastor-3.0.5.el8.tar.gz Source0: vitastor-3.0.0.el8.tar.gz
BuildRequires: gperftools-devel BuildRequires: gperftools-devel
BuildRequires: gcc-toolset-9-gcc-c++ BuildRequires: gcc-toolset-9-gcc-c++
@@ -16,7 +16,6 @@ BuildRequires: gf-complete-devel
BuildRequires: rdma-core-devel BuildRequires: rdma-core-devel
BuildRequires: cmake BuildRequires: cmake
BuildRequires: libnl3-devel BuildRequires: libnl3-devel
BuildRequires: c-ares-devel
Requires: vitastor-osd = %{version}-%{release} Requires: vitastor-osd = %{version}-%{release}
Requires: vitastor-mon = %{version}-%{release} Requires: vitastor-mon = %{version}-%{release}
Requires: vitastor-client = %{version}-%{release} Requires: vitastor-client = %{version}-%{release}
+1 -1
View File
@@ -10,7 +10,7 @@ WORKDIR /root
RUN sed -i 's/enabled=0/enabled=1/' /etc/yum.repos.d/*.repo RUN sed -i 's/enabled=0/enabled=1/' /etc/yum.repos.d/*.repo
RUN dnf -y install epel-release dnf-plugins-core RUN dnf -y install epel-release dnf-plugins-core
RUN dnf -y install https://vitastor.io/rpms/centos/9/vitastor-release-1.0-1.el9.noarch.rpm RUN dnf -y install https://vitastor.io/rpms/centos/9/vitastor-release-1.0-1.el9.noarch.rpm
RUN dnf -y install gcc-c++ gperftools-devel fio nodejs rpm-build jerasure-devel libisa-l-devel gf-complete-devel rdma-core-devel libarchive cmake libnl3-devel c-ares-devel RUN dnf -y install gcc-c++ gperftools-devel fio nodejs rpm-build jerasure-devel libisa-l-devel gf-complete-devel rdma-core-devel libarchive cmake libnl3-devel
RUN dnf download --source fio RUN dnf download --source fio
RUN rpm --nomd5 -i fio*.src.rpm RUN rpm --nomd5 -i fio*.src.rpm
RUN cd ~/rpmbuild/SPECS && dnf builddep -y --spec fio.spec RUN cd ~/rpmbuild/SPECS && dnf builddep -y --spec fio.spec
+2 -3
View File
@@ -1,11 +1,11 @@
Name: vitastor Name: vitastor
Version: 3.0.5 Version: 3.0.0
Release: 1%{?dist} Release: 1%{?dist}
Summary: Vitastor, a fast software-defined clustered block storage Summary: Vitastor, a fast software-defined clustered block storage
License: Vitastor Network Public License 1.1 License: Vitastor Network Public License 1.1
URL: https://vitastor.io/ URL: https://vitastor.io/
Source0: vitastor-3.0.5.el9.tar.gz Source0: vitastor-3.0.0.el9.tar.gz
BuildRequires: gperftools-devel BuildRequires: gperftools-devel
BuildRequires: gcc-c++ BuildRequires: gcc-c++
@@ -16,7 +16,6 @@ BuildRequires: gf-complete-devel
BuildRequires: rdma-core-devel BuildRequires: rdma-core-devel
BuildRequires: cmake BuildRequires: cmake
BuildRequires: libnl3-devel BuildRequires: libnl3-devel
BuildRequires: c-ares-devel
Requires: vitastor-osd = %{version}-%{release} Requires: vitastor-osd = %{version}-%{release}
Requires: vitastor-mon = %{version}-%{release} Requires: vitastor-mon = %{version}-%{release}
Requires: vitastor-client = %{version}-%{release} Requires: vitastor-client = %{version}-%{release}
+1 -9
View File
@@ -21,7 +21,7 @@ if("${CMAKE_INSTALL_PREFIX}" MATCHES "^/usr/local/?$")
endif() endif()
set(ENABLE_COVERAGE false CACHE BOOL "Enable code coverage") set(ENABLE_COVERAGE false CACHE BOOL "Enable code coverage")
add_definitions(-DVITASTOR_VERSION="3.0.5") add_definitions(-DVITASTOR_VERSION="3.0.0")
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src) add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
add_link_options(-fno-omit-frame-pointer) add_link_options(-fno-omit-frame-pointer)
if (${WITH_ASAN}) if (${WITH_ASAN})
@@ -75,14 +75,6 @@ if (RDMACM_LIBRARIES)
add_definitions(-DWITH_RDMACM) add_definitions(-DWITH_RDMACM)
endif (RDMACM_LIBRARIES) endif (RDMACM_LIBRARIES)
find_package(OpenSSL REQUIRED)
if (OPENSSL_FOUND)
add_definitions(-DWITH_OPENSSL)
endif (OPENSSL_FOUND)
pkg_check_modules(CARES REQUIRED libcares)
include_directories(${CARES_INCLUDE_DIRS})
if (${WITH_SYSTEM_LIBURING}) if (${WITH_SYSTEM_LIBURING})
pkg_check_modules(LIBURING REQUIRED liburing>=2.10) pkg_check_modules(LIBURING REQUIRED liburing>=2.10)
include_directories(${LIBURING_INCLUDE_DIRS}) include_directories(${LIBURING_INCLUDE_DIRS})
+1 -1
View File
@@ -4,7 +4,7 @@ project(vitastor)
# libvitastor_blk.a # libvitastor_blk.a
add_library(vitastor_blk STATIC add_library(vitastor_blk STATIC
../util/allocator.cpp ../util/crc32c.c ../util/xxhash.c ../util/ringloop.cpp ../util/allocator.cpp ../util/crc32c.c ../util/ringloop.cpp
multilist.cpp blockstore_heap.cpp blockstore_disk.cpp multilist.cpp blockstore_heap.cpp blockstore_disk.cpp
blockstore.cpp blockstore_impl.cpp blockstore_init.cpp blockstore_open.cpp blockstore.cpp blockstore_impl.cpp blockstore_init.cpp blockstore_open.cpp
blockstore_flush.cpp blockstore_read.cpp blockstore_stable.cpp blockstore_sync.cpp blockstore_write.cpp blockstore_flush.cpp blockstore_read.cpp blockstore_stable.cpp blockstore_sync.cpp blockstore_write.cpp
-5
View File
@@ -183,11 +183,6 @@ public:
// Update configuration // Update configuration
virtual void parse_config(blockstore_config_t & config) = 0; virtual void parse_config(blockstore_config_t & config) = 0;
// Reshard database for a pool in chunks
// MUST be called only when nobody makes any modifications to the DB for this pool
virtual void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit) = 0;
virtual bool reshard_continue(void *reshard_state, uint64_t chunk_limit) = 0;
// Event loop // Event loop
virtual void loop() = 0; virtual void loop() = 0;
+2 -10
View File
@@ -83,17 +83,13 @@ void blockstore_disk_t::parse_config(std::map<std::string, std::string> & config
{ {
data_csum_type = BLOCKSTORE_CSUM_CRC32C; data_csum_type = BLOCKSTORE_CSUM_CRC32C;
} }
else if (config["data_csum_type"] == "xxh3_32")
{
data_csum_type = BLOCKSTORE_CSUM_XXH3_32;
}
else if (config["data_csum_type"] == "" || config["data_csum_type"] == "none") else if (config["data_csum_type"] == "" || config["data_csum_type"] == "none")
{ {
data_csum_type = BLOCKSTORE_CSUM_NONE; data_csum_type = BLOCKSTORE_CSUM_NONE;
} }
else else
{ {
throw std::runtime_error("data_csum_type="+config["data_csum_type"]+" is unsupported, only \"crc32c\", \"xxh3_32\" and \"none\" are supported"); throw std::runtime_error("data_csum_type="+config["data_csum_type"]+" is unsupported, only \"crc32c\" and \"none\" are supported");
} }
csum_block_size = parse_size(config["csum_block_size"]); csum_block_size = parse_size(config["csum_block_size"]);
discard_on_start = config.find("discard_on_start") != config.end() && discard_on_start = config.find("discard_on_start") != config.end() &&
@@ -175,10 +171,6 @@ void blockstore_disk_t::parse_config(std::map<std::string, std::string> & config
{ {
throw std::runtime_error("Data block size must be a multiple of sparse write tracking granularity"); throw std::runtime_error("Data block size must be a multiple of sparse write tracking granularity");
} }
if (data_block_size / bitmap_granularity < 8)
{
throw std::runtime_error("Data block size must be at least bitmap_granularity*8");
}
if (!data_csum_type) if (!data_csum_type)
{ {
csum_block_size = 0; csum_block_size = 0;
@@ -267,7 +259,7 @@ void blockstore_disk_t::calc_lengths(bool skip_meta_check)
} }
// required metadata size // required metadata size
block_count = data_len / data_block_size; block_count = data_len / data_block_size;
clean_entry_bitmap_size = (data_block_size / bitmap_granularity + 7) / 8; clean_entry_bitmap_size = data_block_size / bitmap_granularity / 8;
clean_dyn_size = clean_entry_bitmap_size*2 + (csum_block_size clean_dyn_size = clean_entry_bitmap_size*2 + (csum_block_size
? data_block_size/csum_block_size*(data_csum_type & 0xFF) : 0); ? data_block_size/csum_block_size*(data_csum_type & 0xFF) : 0);
recalc: recalc:
+3 -4
View File
@@ -16,7 +16,6 @@
#define BLOCKSTORE_CSUM_NONE 0 #define BLOCKSTORE_CSUM_NONE 0
// Lower byte of checksum type is its length // Lower byte of checksum type is its length
#define BLOCKSTORE_CSUM_CRC32C 0x104 #define BLOCKSTORE_CSUM_CRC32C 0x104
#define BLOCKSTORE_CSUM_XXH3_32 0x204
#define MOCK_DATA_FD 1000 #define MOCK_DATA_FD 1000
#define MOCK_META_FD 1001 #define MOCK_META_FD 1001
@@ -27,14 +26,14 @@ class allocator_t;
struct blockstore_disk_t struct blockstore_disk_t
{ {
std::string data_device, meta_device, journal_device; std::string data_device, meta_device, journal_device;
uint64_t data_block_size; uint32_t data_block_size;
uint64_t cfg_journal_size, cfg_data_size; uint64_t cfg_journal_size, cfg_data_size;
// Required write alignment and journal/metadata/data areas' location alignment // Required write alignment and journal/metadata/data areas' location alignment
uint32_t disk_alignment = 4096; uint32_t disk_alignment = 4096;
// Journal block size - minimum_io_size of the journal device is the best choice // Journal block size - minimum_io_size of the journal device is the best choice
uint64_t journal_block_size = 4096; uint32_t journal_block_size = 4096;
// Metadata block size - minimum_io_size of the metadata device is the best choice // Metadata block size - minimum_io_size of the metadata device is the best choice
uint64_t meta_block_size = 4096; uint32_t meta_block_size = 4096;
// Atomic write size of the data block device // Atomic write size of the data block device
uint32_t atomic_write_size = 4096; uint32_t atomic_write_size = 4096;
// Whether we should set RWF_ATOMIC on atomic writes // Whether we should set RWF_ATOMIC on atomic writes
+1
View File
@@ -58,6 +58,7 @@ class journal_flusher_co
int i, res; int i, res;
bool read_to_fill_incomplete; bool read_to_fill_incomplete;
int copy_count; int copy_count;
bool do_repeat = false;
friend class journal_flusher_t; friend class journal_flusher_t;
+161 -379
View File
@@ -12,7 +12,6 @@
#include "blockstore_heap.h" #include "blockstore_heap.h"
#include "../util/allocator.h" #include "../util/allocator.h"
#include "../util/crc32c.h" #include "../util/crc32c.h"
#include "../util/xxhash.h"
#include "../util/malloc_or_die.h" #include "../util/malloc_or_die.h"
#define BS_HEAP_FREE_MVCC 1 #define BS_HEAP_FREE_MVCC 1
@@ -30,15 +29,6 @@
#define IMAP_MALLOC_LOW_BITS ((size_t)0x0F) #define IMAP_MALLOC_LOW_BITS ((size_t)0x0F)
#define IMAP_MAX_LOW 16 #define IMAP_MAX_LOW 16
void inode_map_put(void* & inode_idx, heap_list_item_t* li);
void inode_map_get(void *inode_idx, heap_inode_map_t::iterator & li_it, heap_list_item_t* & li, uint64_t stripe);
void inode_map_free(void* inode_idx);
bool inode_map_is_big(void* & inode_idx);
void inode_map_iterate(void* & inode_idx, std::function<void(heap_list_item_t*)> cb);
void inode_map_replace(void* & inode_idx, const heap_inode_map_t::iterator & li_it, heap_list_item_t* new_li);
void inode_map_erase(robin_hood::unordered_flat_map<inode_t, void*, i64hash_t> & pg_idx, void* & inode_idx,
const heap_inode_map_t::iterator & li_it, heap_list_item_t* li);
static inline heap_list_item_t *list_item(heap_entry_t *wr) static inline heap_list_item_t *list_item(heap_entry_t *wr)
{ {
return (heap_list_item_t*)((uint8_t*)wr - offsetof(struct heap_list_item_t, entry)); return (heap_list_item_t*)((uint8_t*)wr - offsetof(struct heap_list_item_t, entry));
@@ -63,19 +53,19 @@ uint32_t blockstore_heap_t::get_simple_entry_size()
uint32_t blockstore_heap_t::get_big_entry_size() uint32_t blockstore_heap_t::get_big_entry_size()
{ {
return sizeof(heap_big_write_t) + dsk->clean_entry_bitmap_size*2 + return sizeof(heap_big_write_t) + dsk->clean_entry_bitmap_size*2 +
(!dsk->csum_block_size ? 0 : dsk->data_block_size/dsk->csum_block_size * (dsk->data_csum_type & 0xFF)); (!dsk->data_csum_type ? 0 : dsk->data_block_size/dsk->csum_block_size * (dsk->data_csum_type & 0xFF));
} }
uint32_t blockstore_heap_t::get_big_intent_entry_size() uint32_t blockstore_heap_t::get_big_intent_entry_size()
{ {
return sizeof(heap_big_intent_t) + dsk->clean_entry_bitmap_size*2 + return sizeof(heap_big_intent_t) + dsk->clean_entry_bitmap_size*2 +
(!dsk->csum_block_size ? 4 : dsk->data_block_size/dsk->csum_block_size * (dsk->data_csum_type & 0xFF)); (!dsk->data_csum_type ? 4 : dsk->data_block_size/dsk->csum_block_size * (dsk->data_csum_type & 0xFF));
} }
uint32_t blockstore_heap_t::get_small_entry_size(uint32_t offset, uint32_t len) uint32_t blockstore_heap_t::get_small_entry_size(uint32_t offset, uint32_t len)
{ {
return sizeof(heap_small_write_t) + dsk->clean_entry_bitmap_size + return sizeof(heap_small_write_t) + dsk->clean_entry_bitmap_size +
(!dsk->csum_block_size ? 4 : (dsk->data_csum_type & 0xFF) * (!dsk->data_csum_type ? 4 : (dsk->data_csum_type & 0xFF) *
((offset+len+dsk->csum_block_size-1)/dsk->csum_block_size - offset/dsk->csum_block_size)); ((offset+len+dsk->csum_block_size-1)/dsk->csum_block_size - offset/dsk->csum_block_size));
} }
@@ -90,7 +80,7 @@ uint32_t blockstore_heap_t::get_csum_size(heap_entry_t *wr)
uint32_t blockstore_heap_t::get_csum_size(uint32_t entry_type, uint32_t offset, uint32_t len) uint32_t blockstore_heap_t::get_csum_size(uint32_t entry_type, uint32_t offset, uint32_t len)
{ {
if (!dsk->csum_block_size) if (!dsk->data_csum_type)
{ {
return 0; return 0;
} }
@@ -213,24 +203,15 @@ void heap_entry_t::set_big_location(blockstore_heap_t *heap, uint64_t location)
big().block_num = location / heap->dsk->data_block_size; big().block_num = location / heap->dsk->data_block_size;
} }
uint32_t heap_entry_t::calc_checksum(blockstore_disk_t *dsk) uint32_t heap_entry_t::calc_crc32c()
{ {
auto old_checksum = checksum; auto old_crc32c = crc32c;
checksum = 0; crc32c = 0;
uint32_t res = 0; uint32_t res = ::crc32c(0, (uint8_t*)this, size);
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32) crc32c = old_crc32c;
res = (uint32_t)XXH3_64bits(this, size);
else
res = ::crc32c(0, (uint8_t*)this, size);
checksum = old_checksum;
return res; return res;
} }
uint32_t heap_entry_t::calc_checksum(blockstore_heap_t *heap)
{
return calc_checksum(heap->dsk);
}
uint64_t blockstore_heap_t::get_pg_id(inode_t inode, uint64_t stripe) uint64_t blockstore_heap_t::get_pg_id(inode_t inode, uint64_t stripe)
{ {
uint64_t pg_num = 0; uint64_t pg_num = 0;
@@ -311,13 +292,12 @@ int blockstore_heap_t::read_blocks(uint64_t disk_offset, uint64_t disk_size, uin
heap_entry_t *wr = (heap_entry_t*)data; heap_entry_t *wr = (heap_entry_t*)data;
if (wr->size > dsk->meta_block_size-block_offset) if (wr->size > dsk->meta_block_size-block_offset)
{ {
fprintf(stderr, "Error: entry is too large in metadata block %u at %u (%u > max %ju bytes). ", fprintf(stderr, "Error: entry is too large in metadata block %u at %u (%u > max %u bytes). ",
block_num, block_offset, wr->size, dsk->meta_block_size-block_offset); block_num, block_offset, wr->size, dsk->meta_block_size-block_offset);
corrupted_block: corrupted_block:
if (allow_corrupted) if (allow_corrupted)
{ {
fprintf(stderr, "Metadata block is corrupted, skipping\n"); fprintf(stderr, "Metadata block is corrupted, skipping\n");
recheck_modified_blocks.insert(block_num);
break; break;
} }
else else
@@ -342,19 +322,7 @@ corrupted_block:
block_num, block_offset, wr->size, sizeof(heap_entry_t)); block_num, block_offset, wr->size, sizeof(heap_entry_t));
goto corrupted_block; goto corrupted_block;
} }
if (wr->is_garbage()) wr->entry_type &= ~BS_HEAP_GARBAGE;
{
// Garbage collection is only performed when writing new entries into the block
// because it needs a fake LSN and modified blocks require consecutive modified LSNs
// That's why garbage entries may persist on disk
if (log_level > 5)
{
fprintf(stderr, "Notice: skipping garbage entry %jx:%jx v%ju l%ju in metadata block %u at %u\n",
wr->inode, wr->stripe, wr->version, wr->lsn, block_num, block_offset);
}
block_offset += wr->size;
continue;
}
if ((wr->entry_type & BS_HEAP_TYPE) < BS_HEAP_BIG_WRITE || if ((wr->entry_type & BS_HEAP_TYPE) < BS_HEAP_BIG_WRITE ||
(wr->entry_type & BS_HEAP_TYPE) > BS_HEAP_ROLLBACK || (wr->entry_type & BS_HEAP_TYPE) > BS_HEAP_ROLLBACK ||
(wr->entry_type & ~(BS_HEAP_TYPE|BS_HEAP_STABLE)) || (wr->entry_type & ~(BS_HEAP_TYPE|BS_HEAP_STABLE)) ||
@@ -368,7 +336,6 @@ corrupted_object:
if (allow_corrupted) if (allow_corrupted)
{ {
fprintf(stderr, "Entry is corrupted, skipping\n"); fprintf(stderr, "Entry is corrupted, skipping\n");
recheck_modified_blocks.insert(block_num);
block_offset += wr->size; block_offset += wr->size;
continue; continue;
} }
@@ -384,7 +351,7 @@ corrupted_object:
{ {
// Small writes require accessing offset & len to calculate correct length, // Small writes require accessing offset & len to calculate correct length,
// so require at least sizeof(heap_small_write_t) for them // so require at least sizeof(heap_small_write_t) for them
fprintf(stderr, "Error: entry %jx:%jx v%ju has invalid size in metadata block %u at %u (%u < min %zu bytes)\n", fprintf(stderr, "Error: entry %jx:%jx v%ju has invalid size in metadata block %u at %u (%u < min %zu bytes). Metadata is corrupted, aborting\n",
wr->inode, wr->stripe, wr->version, block_num, block_offset, wr->size, sizeof(heap_small_write_t)); wr->inode, wr->stripe, wr->version, block_num, block_offset, wr->size, sizeof(heap_small_write_t));
goto corrupted_object; goto corrupted_object;
} }
@@ -395,12 +362,12 @@ corrupted_object:
goto corrupted_object; goto corrupted_object;
} }
// Verify crc // Verify crc
uint32_t expected_checksum = wr->calc_checksum(this); uint32_t expected_crc32c = wr->calc_crc32c();
if (wr->checksum != expected_checksum) if (wr->crc32c != expected_crc32c)
{ {
fprintf(stderr, "Error: entry %jx:%jx v%ju l%ju in metadata block %u at %u is corrupt (checksum mismatch: expected %08x, got %08x). ", fprintf(stderr, "Error: entry %jx:%jx v%ju in metadata block %u at %u is corrupt (crc32c mismatch: expected %08x, got %08x). Metadata is corrupted, aborting\n",
wr->inode, wr->stripe, wr->version, wr->lsn, wr->inode, wr->stripe, wr->version,
block_num, block_offset, expected_checksum, wr->checksum); block_num, block_offset, expected_crc32c, wr->crc32c);
goto corrupted_object; goto corrupted_object;
} }
// Verify offset & len // Verify offset & len
@@ -409,7 +376,7 @@ corrupted_object:
wr->small().offset % dsk->bitmap_granularity || wr->small().offset % dsk->bitmap_granularity ||
wr->small().len % dsk->bitmap_granularity)) wr->small().len % dsk->bitmap_granularity))
{ {
fprintf(stderr, "Error: %s entry %jx:%jx v%ju has invalid offset/length: %u/%u. Metadata is incompatible with current parameters. ", fprintf(stderr, "Error: %s entry %jx:%jx v%ju has invalid offset/length: %u/%u. Metadata is incompatible with current parameters, aborting\n",
wr->type() == BS_HEAP_SMALL_WRITE ? "small_write" : "intent_write", wr->type() == BS_HEAP_SMALL_WRITE ? "small_write" : "intent_write",
wr->inode, wr->stripe, wr->version, wr->small().offset, wr->small().len); wr->inode, wr->stripe, wr->version, wr->small().offset, wr->small().len);
goto corrupted_object; goto corrupted_object;
@@ -419,7 +386,7 @@ corrupted_object:
wr->big_intent().offset % dsk->bitmap_granularity || wr->big_intent().offset % dsk->bitmap_granularity ||
wr->big_intent().len % dsk->bitmap_granularity)) wr->big_intent().len % dsk->bitmap_granularity))
{ {
fprintf(stderr, "Error: big_intent entry %jx:%jx v%ju has invalid offset/length: %u/%u. Metadata is incompatible with current parameters. ", fprintf(stderr, "Error: big_intent entry %jx:%jx v%ju has invalid offset/length: %u/%u. Metadata is incompatible with current parameters, aborting\n",
wr->inode, wr->stripe, wr->version, wr->big_intent().offset, wr->big_intent().len); wr->inode, wr->stripe, wr->version, wr->big_intent().offset, wr->big_intent().len);
goto corrupted_object; goto corrupted_object;
} }
@@ -446,7 +413,7 @@ int blockstore_heap_t::load_blocks(uint64_t disk_offset, uint64_t size, uint8_t
next_lsn = wr->lsn; next_lsn = wr->lsn;
} }
entries_loaded++; entries_loaded++;
loaded_list_items.push_back(li); insert_list_item(li);
modify_alloc(block_num, [&](heap_block_info_t & inf) modify_alloc(block_num, [&](heap_block_info_t & inf)
{ {
if (!inf.entries.size()) if (!inf.entries.size())
@@ -482,22 +449,19 @@ bool blockstore_heap_t::validate_object(heap_entry_t *obj)
next_wr = wr; next_wr = wr;
if (wr->type() == BS_HEAP_ROLLBACK) if (wr->type() == BS_HEAP_ROLLBACK)
{ {
if (commit_wr && wr->version > commit_wr->version)
{
// rollback may not come before commit with a smaller version
fprintf(stderr, "Error: rollback entry %jx:%jx v%ju l%ju comes before a commit entry v%ju l%ju\n",
wr->inode, wr->stripe, wr->version, wr->lsn, commit_wr->version, commit_wr->lsn);
return false;
}
rollback_wr = wr; rollback_wr = wr;
continue; continue;
} }
if (wr->type() == BS_HEAP_COMMIT) if (wr->type() == BS_HEAP_COMMIT)
{ {
if (commit_wr && wr->version > commit_wr->version) commit_wr = wr;
{
// commit may not come before commit with a smaller version
fprintf(stderr, "Error: commit entry %jx:%jx v%ju l%ju comes before a commit entry v%ju l%ju\n",
wr->inode, wr->stripe, wr->version, wr->lsn, commit_wr->version, commit_wr->lsn);
return false;
}
if (!commit_wr)
{
commit_wr = wr;
}
continue; continue;
} }
if (wr->entry_type & BS_HEAP_STABLE) if (wr->entry_type & BS_HEAP_STABLE)
@@ -550,23 +514,6 @@ bool blockstore_heap_t::validate_object(heap_entry_t *obj)
return true; return true;
} }
void blockstore_heap_t::finish_load()
{
if (loaded_list_items.size())
{
// Sort everything and load in correct order
std::sort(loaded_list_items.begin(), loaded_list_items.end(), [this](const heap_list_item_t* a, const heap_list_item_t* b)
{
return a->entry.lsn < b->entry.lsn;
});
for (auto & li: loaded_list_items)
{
insert_list_item(li);
}
loaded_list_items.clear();
}
}
void blockstore_heap_t::fill_recheck_queue() void blockstore_heap_t::fill_recheck_queue()
{ {
for (auto & pgp: block_index) for (auto & pgp: block_index)
@@ -711,8 +658,7 @@ void blockstore_heap_t::recheck_buffer(heap_entry_t *cwr, uint8_t *buf)
else if (!calc_checksums(cwr, buf, false)) else if (!calc_checksums(cwr, buf, false))
{ {
// write entry is invalid, erase it and mark newer entries with garbage bit // write entry is invalid, erase it and mark newer entries with garbage bit
auto & pg_idx = block_index[get_pg_id(cwr->inode, cwr->stripe)]; auto & inode_idx = block_index[get_pg_id(cwr->inode, cwr->stripe)][cwr->inode];
auto & inode_idx = pg_idx[cwr->inode];
heap_inode_map_t::iterator li_it; heap_inode_map_t::iterator li_it;
heap_list_item_t *li = NULL; heap_list_item_t *li = NULL;
inode_map_get(inode_idx, li_it, li, cwr->stripe); inode_map_get(inode_idx, li_it, li, cwr->stripe);
@@ -738,7 +684,7 @@ void blockstore_heap_t::recheck_buffer(heap_entry_t *cwr, uint8_t *buf)
{ {
fprintf(stderr, "Notice: the whole object %jx:%jx only has unfinished writes, rolling back\n", fprintf(stderr, "Notice: the whole object %jx:%jx only has unfinished writes, rolling back\n",
cwr->inode, cwr->stripe); cwr->inode, cwr->stripe);
inode_map_erase(pg_idx, inode_idx, li_it, li); inode_map_erase(inode_idx, li_it, li);
} }
free_entry(li); free_entry(li);
} }
@@ -753,7 +699,6 @@ bool blockstore_heap_t::recheck_small_writes(std::function<void(bool is_data, ui
} }
if (!recheck_queue_filled) if (!recheck_queue_filled)
{ {
finish_load();
fill_recheck_queue(); fill_recheck_queue();
recheck_queue_filled = true; recheck_queue_filled = true;
} }
@@ -845,7 +790,7 @@ std::vector<uint32_t> blockstore_heap_t::get_recheck_modified_blocks()
return modified; return modified;
} }
int blockstore_heap_t::finish_recheck() int blockstore_heap_t::finish_load(bool allow_corrupted)
{ {
if (!marked_used_blocks) if (!marked_used_blocks)
{ {
@@ -882,17 +827,14 @@ bool blockstore_heap_t::calc_checksums(heap_entry_t *wr, uint8_t *data, bool set
{ {
return true; return true;
} }
uint32_t len = 0;
if (wr->type() == BS_HEAP_SMALL_WRITE || wr->type() == BS_HEAP_INTENT_WRITE) if (wr->type() == BS_HEAP_SMALL_WRITE || wr->type() == BS_HEAP_INTENT_WRITE)
len = wr->small().len; len = wr->small().len;
else if (wr->type() == BS_HEAP_BIG_INTENT) else if (wr->type() == BS_HEAP_BIG_INTENT)
len = wr->big_intent().len; len = wr->big_intent().len;
else else
assert(0); assert(0);
uint32_t real_csum = 0; uint32_t real_csum = crc32c(0, data, len);
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
real_csum = (uint32_t)XXH3_64bits(data, len);
else
real_csum = crc32c(0, data, len);
if (set) if (set)
{ {
*wr_csum = real_csum; *wr_csum = real_csum;
@@ -902,14 +844,13 @@ bool blockstore_heap_t::calc_checksums(heap_entry_t *wr, uint8_t *data, bool set
} }
if (wr->type() == BS_HEAP_BIG_WRITE) if (wr->type() == BS_HEAP_BIG_WRITE)
{ {
assert(offset != UINT32_MAX && len != UINT32_MAX);
return calc_block_checksums((uint32_t*)(wr->get_checksums(this) + offset/dsk->csum_block_size * (dsk->data_csum_type & 0xFF)), return calc_block_checksums((uint32_t*)(wr->get_checksums(this) + offset/dsk->csum_block_size * (dsk->data_csum_type & 0xFF)),
data, wr->get_int_bitmap(this), offset, offset+len, set, NULL); data, wr->get_int_bitmap(this), offset, offset+len, set, NULL);
} }
if (wr->type() == BS_HEAP_BIG_INTENT) if (wr->type() == BS_HEAP_BIG_INTENT)
{ {
auto & bi = wr->big_intent(); auto & bi = wr->big_intent();
return calc_block_checksums((uint32_t*)(wr->get_checksums(this) + bi.offset/dsk->csum_block_size * (dsk->data_csum_type & 0xFF)), return calc_block_checksums((uint32_t*)(wr->get_checksums(this) + offset/dsk->csum_block_size * (dsk->data_csum_type & 0xFF)),
data, wr->get_int_bitmap(this), bi.offset, bi.offset+bi.len, set, NULL); data, wr->get_int_bitmap(this), bi.offset, bi.offset+bi.len, set, NULL);
} }
assert(wr->type() == BS_HEAP_SMALL_WRITE || wr->type() == BS_HEAP_INTENT_WRITE); assert(wr->type() == BS_HEAP_SMALL_WRITE || wr->type() == BS_HEAP_INTENT_WRITE);
@@ -942,26 +883,11 @@ static uint32_t crc32c_iter(uint32_t prev_crc, const std::function<uint8_t*(uint
return prev_crc; return prev_crc;
} }
static void xxh3_iter(XXH3_state_t* xxh3_state, const std::function<uint8_t*(uint32_t start, uint32_t & len)> & next, uint32_t pos, uint32_t size)
{
uint32_t cur_len = 0;
while (size > 0)
{
uint8_t *data = next(pos, cur_len);
assert(data);
cur_len = (cur_len < size ? cur_len : size);
XXH3_64bits_update(xxh3_state, data, cur_len);
pos += cur_len;
size -= cur_len;
}
}
bool blockstore_heap_t::calc_block_checksums(uint32_t *block_csums, uint8_t *bitmap, bool blockstore_heap_t::calc_block_checksums(uint32_t *block_csums, uint8_t *bitmap,
uint32_t start, uint32_t end, std::function<uint8_t*(uint32_t start, uint32_t & len)> next, uint32_t start, uint32_t end, std::function<uint8_t*(uint32_t start, uint32_t & len)> next,
bool set, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb) bool set, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb)
{ {
bool res = true; bool res = true;
XXH3_state_t* xxh3_state = NULL;
uint32_t pos = start; uint32_t pos = start;
uint32_t block_end = (start/dsk->csum_block_size + 1)*dsk->csum_block_size; uint32_t block_end = (start/dsk->csum_block_size + 1)*dsk->csum_block_size;
uint32_t block_crc = 0; uint32_t block_crc = 0;
@@ -978,214 +904,83 @@ bool blockstore_heap_t::calc_block_checksums(uint32_t *block_csums, uint8_t *bit
pos += dsk->bitmap_granularity; pos += dsk->bitmap_granularity;
// zero padding at the beginning or at the end of the block is not counted // zero padding at the beginning or at the end of the block is not counted
if (pos > prev && prev > 0 && pos < block_end) if (pos > prev && prev > 0 && pos < block_end)
{ block_crc = crc32c_pad(block_crc, NULL, 0, pos-prev, 0);
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
{
if (!xxh3_state)
{
xxh3_state = XXH3_createState();
XXH3_64bits_reset(xxh3_state);
}
uint32_t zeropad = pos-prev;
while (zeropad > 0)
{
uint32_t zerolen = zeropad > 4096 ? 4096 : zeropad;
XXH3_64bits_update(xxh3_state, zero_page, zerolen);
zeropad -= zerolen;
}
}
else
block_crc = crc32c_pad(block_crc, NULL, 0, pos-prev, 0);
}
prev = pos; prev = pos;
while (pos < end && pos < block_end && (bitmap[pos/dsk->bitmap_granularity/8] & (1 << ((pos/dsk->bitmap_granularity) % 8)))) while (pos < end && pos < block_end && (bitmap[pos/dsk->bitmap_granularity/8] & (1 << ((pos/dsk->bitmap_granularity) % 8))))
pos += dsk->bitmap_granularity; pos += dsk->bitmap_granularity;
if (pos > prev) if (pos > prev)
{ {
isset = true; isset = true;
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32) block_crc = crc32c_iter(block_crc, next, prev, pos-prev);
{
if (!xxh3_state)
{
xxh3_state = XXH3_createState();
XXH3_64bits_reset(xxh3_state);
}
xxh3_iter(xxh3_state, next, prev, pos-prev);
}
else
block_crc = crc32c_iter(block_crc, next, prev, pos-prev);
} }
prev = pos; prev = pos;
} }
} }
else else
{ {
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32) block_crc = crc32c_iter(block_crc, next, pos, (end > block_end ? block_end : end)-pos);
{
if (!xxh3_state)
{
xxh3_state = XXH3_createState();
XXH3_64bits_reset(xxh3_state);
}
xxh3_iter(xxh3_state, next, pos, (end > block_end ? block_end : end)-pos);
}
else
block_crc = crc32c_iter(block_crc, next, pos, (end > block_end ? block_end : end)-pos);
pos = (end > block_end ? block_end : end); pos = (end > block_end ? block_end : end);
isset = true; isset = true;
} }
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32 && xxh3_state)
{
block_crc = (uint32_t)XXH3_64bits_digest(xxh3_state);
XXH3_64bits_reset(xxh3_state);
}
if (set) if (set)
{ {
*block_csums = block_crc; *block_csums = block_crc;
} }
else if (isset && block_crc != *block_csums) else if (isset && block_crc != *block_csums)
{ {
res = false;
if (bad_block_cb) if (bad_block_cb)
{
bad_block_cb(blk_start, *block_csums, block_crc); bad_block_cb(blk_start, *block_csums, block_crc);
res = false;
}
else else
break; return false;
} }
block_end += dsk->csum_block_size; block_end += dsk->csum_block_size;
block_crc = 0; block_crc = 0;
block_csums++; block_csums++;
} }
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32 && xxh3_state)
{
block_crc = (uint32_t)XXH3_64bits_digest(xxh3_state);
XXH3_freeState(xxh3_state);
xxh3_state = NULL;
}
return res; return res;
} }
struct heap_reshard_state_t void blockstore_heap_t::reshard(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size)
{
int state = 0;
uint64_t pool_id = 0;
uint32_t old_pg_count = 0;
uint32_t pg_count = 0;
uint32_t pg_stripe_size = 0;
uint64_t chunk_size = 0;
heap_block_index_t new_shards;
heap_block_index_t old_shards;
heap_block_index_t::iterator sh_it;
robin_hood::unordered_flat_map<inode_t, void*, i64hash_t>::iterator inode_it;
heap_inode_map_t *stripe_map = NULL;
heap_inode_map_t::iterator stripe_it;
void add(heap_list_item_t *li);
bool run(uint64_t chunk_limit);
};
void heap_reshard_state_t::add(heap_list_item_t *li)
{
// like map_to_pg()
uint64_t pg_num = (li->entry.stripe / pg_stripe_size) % pg_count + 1;
uint64_t shard_id = (pool_id << (64-POOL_ID_BITS)) | pg_num;
inode_map_put(new_shards[shard_id][li->entry.inode], li);
chunk_size++;
}
bool heap_reshard_state_t::run(uint64_t chunk_limit)
{
chunk_size = 0;
if (state == 1)
goto resume_1;
else if (state == 2)
goto resume_2;
sh_it = old_shards.begin();
for (; sh_it != old_shards.end(); sh_it++)
{
inode_it = sh_it->second.begin();
for (; inode_it != sh_it->second.end(); inode_it++)
{
if (!inode_map_is_big(inode_it->second))
{
if (chunk_limit > 0 && chunk_size >= chunk_limit)
{
state = 1;
return false;
}
resume_1:
inode_map_iterate(inode_it->second, [&](heap_list_item_t *li) { add(li); });
}
else
{
stripe_map = (heap_inode_map_t*)inode_it->second;
stripe_it = stripe_map->begin();
for (; stripe_it != stripe_map->end(); stripe_it++)
{
if (chunk_limit > 0 && chunk_size >= chunk_limit)
{
state = 2;
return false;
}
resume_2:
add(*stripe_it);
}
}
inode_map_free(inode_it->second);
}
}
return true;
}
void* blockstore_heap_t::reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit)
{ {
auto & pool_settings = pool_shard_settings[pool]; auto & pool_settings = pool_shard_settings[pool];
if (pool_settings.pg_count == pg_count && pool_settings.pg_stripe_size == pg_stripe_size) if (pool_settings.pg_count == pg_count && pool_settings.pg_stripe_size == pg_stripe_size)
{ {
return NULL; return;
} }
heap_reshard_state_t *st = new heap_reshard_state_t; uint32_t old_pg_count = !pool_settings.pg_count ? 1 : pool_settings.pg_count;
st->pool_id = (uint64_t)pool; uint64_t pool_id = (uint64_t)pool;
st->pg_count = pg_count; heap_block_index_t new_shards;
st->pg_stripe_size = pg_stripe_size; for (uint32_t pg_num = 0; pg_num <= old_pg_count; pg_num++)
st->old_pg_count = !pool_settings.pg_count ? 1 : pool_settings.pg_count;
for (uint32_t pg_num = 0; pg_num <= st->old_pg_count; pg_num++)
{ {
auto sh_it = block_index.find((st->pool_id << (64-POOL_ID_BITS)) | pg_num); auto sh_it = block_index.find((pool_id << (64-POOL_ID_BITS)) | pg_num);
if (sh_it != block_index.end()) if (sh_it == block_index.end())
{ {
st->old_shards[pg_num] = std::move(sh_it->second); continue;
block_index.erase(sh_it);
} }
for (auto & inode_pair: sh_it->second)
{
inode_map_iterate(inode_pair.second, [&](heap_list_item_t *li)
{
// like map_to_pg()
uint64_t pg_num = (li->entry.stripe / pg_stripe_size) % pg_count + 1;
uint64_t shard_id = (pool_id << (64-POOL_ID_BITS)) | pg_num;
inode_map_put(new_shards[shard_id][li->entry.inode], li);
});
inode_map_free(inode_pair.second);
}
block_index.erase(sh_it);
} }
bool finished = reshard_continue(st, chunk_limit); for (auto sh_it = new_shards.begin(); sh_it != new_shards.end(); sh_it++)
return finished ? NULL : st;
}
bool blockstore_heap_t::reshard_continue(void *reshard_state, uint64_t chunk_limit)
{
heap_reshard_state_t *st = (heap_reshard_state_t*)reshard_state;
if (!st->run(chunk_limit))
{
return false;
}
for (auto sh_it = st->new_shards.begin(); sh_it != st->new_shards.end(); sh_it++)
{ {
block_index[sh_it->first] = std::move(sh_it->second); block_index[sh_it->first] = std::move(sh_it->second);
} }
pool_shard_settings[st->pool_id] = (pool_shard_settings_t){ pool_settings = (pool_shard_settings_t){
.pg_count = st->pg_count, .pg_count = pg_count,
.pg_stripe_size = st->pg_stripe_size, .pg_stripe_size = pg_stripe_size,
}; };
delete st;
return true;
}
bool blockstore_heap_t::reshard_check(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size)
{
auto set_it = pool_shard_settings.find(pool);
return (set_it != pool_shard_settings.end() &&
set_it->second.pg_count == pg_count &&
set_it->second.pg_stripe_size == pg_stripe_size);
} }
heap_entry_t *blockstore_heap_t::lock_and_read_entry(object_id oid) heap_entry_t *blockstore_heap_t::lock_and_read_entry(object_id oid)
@@ -1200,6 +995,27 @@ heap_entry_t *blockstore_heap_t::lock_and_read_entry(object_id oid)
return obj; return obj;
} }
heap_entry_t *blockstore_heap_t::read_locked_entry(object_id oid, uint64_t lsn)
{
auto obj = read_entry(oid);
assert(obj);
for (auto wr = obj; wr; wr = prev(wr))
{
if (wr->is_overwrite())
{
if (lsn == wr->lsn)
{
return obj;
}
else
{
obj = prev(wr);
}
}
}
return NULL;
}
bool blockstore_heap_t::unlock_entry(object_id oid) bool blockstore_heap_t::unlock_entry(object_id oid)
{ {
auto mvcc_it = object_mvcc.find(oid); auto mvcc_it = object_mvcc.find(oid);
@@ -1236,35 +1052,6 @@ heap_entry_t *blockstore_heap_t::read_entry(object_id oid)
return &li->entry; return &li->entry;
} }
void blockstore_heap_t::gc_block(heap_block_info_t & inf)
{
if (inf.has_garbage)
{
size_t i = 0, j = 0;
for (; i < inf.entries.size(); i++)
{
if (inf.entries[i]->entry.is_garbage())
{
// old entry invalidated by a newer one, mark it as freeable on block write
// assign a 'virtual' LSN to track GC completion
assert(!inf.mod_lsn_to || inf.mod_lsn_to == next_lsn);
uint64_t gc_lsn = ++next_lsn;
inf.mod_lsn = inf.mod_lsn ? inf.mod_lsn : gc_lsn;
inf.mod_lsn_to = gc_lsn;
push_inflight_lsn(gc_lsn, &inf.entries[i]->entry, HEAP_INFLIGHT_GC);
}
else
{
if (j != i)
inf.entries[j] = inf.entries[i];
j++;
}
}
inf.entries.resize(j);
inf.has_garbage = false;
}
}
int blockstore_heap_t::allocate_entry(uint32_t entry_size, uint32_t *block_num, bool allow_last_free) int blockstore_heap_t::allocate_entry(uint32_t entry_size, uint32_t *block_num, bool allow_last_free)
{ {
if (last_allocated_block != UINT32_MAX) if (last_allocated_block != UINT32_MAX)
@@ -1331,7 +1118,31 @@ int blockstore_heap_t::allocate_entry(uint32_t entry_size, uint32_t *block_num,
} }
// Write into the same block // Write into the same block
auto & inf = block_info.at(last_allocated_block); auto & inf = block_info.at(last_allocated_block);
gc_block(inf); if (inf.has_garbage)
{
size_t i = 0, j = 0;
for (; i < inf.entries.size(); i++)
{
if (inf.entries[i]->entry.is_garbage())
{
// old entry invalidated by a newer one, mark it as freeable on block write
// assign a 'virtual' LSN to track GC completion
assert(!inf.mod_lsn_to || inf.mod_lsn_to == next_lsn);
uint64_t gc_lsn = ++next_lsn;
inf.mod_lsn = inf.mod_lsn ? inf.mod_lsn : gc_lsn;
inf.mod_lsn_to = gc_lsn;
push_inflight_lsn(gc_lsn, &inf.entries[i]->entry, HEAP_INFLIGHT_GC);
}
else
{
if (j != i)
inf.entries[j] = inf.entries[i];
j++;
}
}
inf.entries.resize(j);
inf.has_garbage = false;
}
*block_num = last_allocated_block; *block_num = last_allocated_block;
modify_alloc(last_allocated_block, [&](heap_block_info_t & inf) modify_alloc(last_allocated_block, [&](heap_block_info_t & inf)
{ {
@@ -1412,7 +1223,7 @@ int blockstore_heap_t::add_entry(uint32_t wr_size, uint32_t *modified_block,
insert_list_item(li); insert_list_item(li);
li->block_num = block_num; li->block_num = block_num;
new_wr->size = wr_size; new_wr->size = wr_size;
new_wr->checksum = new_wr->calc_checksum(this); new_wr->crc32c = new_wr->calc_crc32c();
return 0; return 0;
} }
@@ -1472,7 +1283,7 @@ int blockstore_heap_t::add_big_write(object_id oid, heap_entry_t *old_head, bool
memset(wr->get_ext_bitmap(this), 0, dsk->clean_entry_bitmap_size); memset(wr->get_ext_bitmap(this), 0, dsk->clean_entry_bitmap_size);
memset(wr->get_int_bitmap(this), 0, dsk->clean_entry_bitmap_size); memset(wr->get_int_bitmap(this), 0, dsk->clean_entry_bitmap_size);
bitmap_set(wr->get_int_bitmap(this), offset, len, dsk->bitmap_granularity); bitmap_set(wr->get_int_bitmap(this), offset, len, dsk->bitmap_granularity);
if (dsk->csum_block_size) if (dsk->data_csum_type)
{ {
memset(wr->get_checksums(this), 0, get_csum_size(wr)); memset(wr->get_checksums(this), 0, get_csum_size(wr));
calc_checksums(wr, (uint8_t*)data, true, offset, len); calc_checksums(wr, (uint8_t*)data, true, offset, len);
@@ -1501,9 +1312,9 @@ int blockstore_heap_t::add_redirect_intent(object_id oid, heap_entry_t **obj_ptr
memset(wr->get_ext_bitmap(this), 0, dsk->clean_entry_bitmap_size); memset(wr->get_ext_bitmap(this), 0, dsk->clean_entry_bitmap_size);
memset(wr->get_int_bitmap(this), 0, dsk->clean_entry_bitmap_size); memset(wr->get_int_bitmap(this), 0, dsk->clean_entry_bitmap_size);
bitmap_set(wr->get_int_bitmap(this), offset, len, dsk->bitmap_granularity); bitmap_set(wr->get_int_bitmap(this), offset, len, dsk->bitmap_granularity);
if (dsk->csum_block_size) if (dsk->data_csum_type)
memset(wr->get_checksums(this), 0, get_csum_size(wr)); memset(wr->get_checksums(this), 0, get_csum_size(wr));
calc_checksums(wr, (uint8_t*)data, true); calc_checksums(wr, (uint8_t*)data, true, offset, len);
*obj_ptr = wr; *obj_ptr = wr;
}); });
} }
@@ -1539,14 +1350,14 @@ int blockstore_heap_t::add_big_intent(object_id oid, heap_entry_t **obj_ptr, uin
memcpy(wr->get_ext_bitmap(this), obj->get_ext_bitmap(this), dsk->clean_entry_bitmap_size); memcpy(wr->get_ext_bitmap(this), obj->get_ext_bitmap(this), dsk->clean_entry_bitmap_size);
memcpy(wr->get_int_bitmap(this), obj->get_int_bitmap(this), dsk->clean_entry_bitmap_size); memcpy(wr->get_int_bitmap(this), obj->get_int_bitmap(this), dsk->clean_entry_bitmap_size);
bitmap_set(wr->get_int_bitmap(this), offset, len, dsk->bitmap_granularity); bitmap_set(wr->get_int_bitmap(this), offset, len, dsk->bitmap_granularity);
if (dsk->csum_block_size) if (dsk->data_csum_type)
{ {
if (checksums) if (checksums)
memcpy(wr->get_checksums(this), checksums, get_csum_size(wr)); memcpy(wr->get_checksums(this), checksums, dsk->clean_entry_bitmap_size);
else else
{ {
memcpy(wr->get_checksums(this), obj->get_checksums(this), get_csum_size(wr)); memcpy(wr->get_checksums(this), obj->get_checksums(this), dsk->clean_entry_bitmap_size);
calc_checksums(wr, (uint8_t*)data, true); calc_checksums(wr, (uint8_t*)data, true, offset, len);
} }
} }
else else
@@ -1586,7 +1397,7 @@ int blockstore_heap_t::add_compact(heap_entry_t *obj, uint64_t compact_version,
new_wr->set_big_location(this, compact_location); new_wr->set_big_location(this, compact_location);
memcpy(new_wr->get_int_bitmap(this), new_int_bitmap, dsk->clean_entry_bitmap_size); memcpy(new_wr->get_int_bitmap(this), new_int_bitmap, dsk->clean_entry_bitmap_size);
memcpy(new_wr->get_ext_bitmap(this), new_ext_bitmap, dsk->clean_entry_bitmap_size); memcpy(new_wr->get_ext_bitmap(this), new_ext_bitmap, dsk->clean_entry_bitmap_size);
if (dsk->csum_block_size && new_csums) if (dsk->data_csum_type && new_csums)
memcpy(new_wr->get_checksums(this), new_csums, dsk->data_block_size/dsk->csum_block_size*(dsk->data_csum_type & 0xFF)); memcpy(new_wr->get_checksums(this), new_csums, dsk->data_block_size/dsk->csum_block_size*(dsk->data_csum_type & 0xFF));
}); });
} }
@@ -1667,7 +1478,7 @@ int blockstore_heap_t::add_commit(heap_entry_t *obj, uint64_t version, uint32_t
} }
if (!uncommitted) if (!uncommitted)
{ {
return 0; return EBUSY;
} }
return add_simple(obj, version, modified_block, BS_HEAP_COMMIT); return add_simple(obj, version, modified_block, BS_HEAP_COMMIT);
} }
@@ -1677,32 +1488,23 @@ int blockstore_heap_t::add_rollback(heap_entry_t *obj, uint64_t version, uint32_
heap_entry_t *wr = obj; heap_entry_t *wr = obj;
bool found_uncommitted = false; bool found_uncommitted = false;
uint64_t commit_version = 0; uint64_t commit_version = 0;
uint64_t rollback_version = UINT64_MAX; while (wr && !wr->is_overwrite())
while (wr)
{ {
if (wr->type() == BS_HEAP_ROLLBACK) if (wr->type() == BS_HEAP_ROLLBACK)
{ {
if (wr->version <= version) auto rollback_version = wr->version;
{
// All previous writes are already rolled back, stop
break;
}
rollback_version = wr->version;
wr = prev(wr); wr = prev(wr);
while (wr->version > rollback_version)
{
assert(!(wr->entry_type & BS_HEAP_STABLE));
wr = prev(wr);
}
continue; continue;
} }
if (wr->type() == BS_HEAP_COMMIT) if (wr->type() == BS_HEAP_COMMIT)
{ {
if (commit_version < wr->version) if (commit_version < wr->version)
{
commit_version = wr->version; commit_version = wr->version;
}
wr = prev(wr);
continue;
}
if (wr->version > rollback_version)
{
// Already rolled back, skip
wr = prev(wr); wr = prev(wr);
continue; continue;
} }
@@ -1713,10 +1515,14 @@ int blockstore_heap_t::add_rollback(heap_entry_t *obj, uint64_t version, uint32_
{ {
return EBUSY; return EBUSY;
} }
else else if (wr->version == version)
{ {
break; break;
} }
else if (wr->version < version)
{
return ENOENT;
}
} }
else if (wr->version > version) else if (wr->version > version)
{ {
@@ -1897,9 +1703,9 @@ void blockstore_heap_t::iterate_with_stable(heap_entry_t *obj, uint64_t max_lsn,
} }
else else
{ {
// 1) 1 2 3 ROLLBACK(2) COMMIT(3) -> 3 is unstable // 1) 1 2 3 ROLLBACK(2) COMMIT(3) -> impossible
// 2) 1 2 3 4 ROLLBACK(3) COMMIT(2) -> OK // 2) 1 2 3 4 ROLLBACK(3) COMMIT(2) -> OK
// 3) 1 2 3 ROLLBACK(2) 3 COMMIT(3) -> first 3 is unstable // 3) 1 2 3 ROLLBACK(2) 3 COMMIT(3) -> first 3 shouldn't be treated as stable
// 4) 1 2 3 COMMIT(3) ROLLBACK(2) -> impossible // 4) 1 2 3 COMMIT(3) ROLLBACK(2) -> impossible
// I.e. a rollback always has version >= previous commit // I.e. a rollback always has version >= previous commit
// 5) 1 2 3 4 5 ROLLBACK(4) 5 ROLLBACK(3) // 5) 1 2 3 4 5 ROLLBACK(4) 5 ROLLBACK(3)
@@ -2056,21 +1862,24 @@ int blockstore_heap_t::list_objects(uint32_t pg_num, object_id min_oid, object_i
return; return;
} }
uint64_t stable_version = 0; uint64_t stable_version = 0;
iterate_with_stable(obj, UINT64_MAX, [&](heap_entry_t* wr, bool stable) auto first_wr = obj;
for (auto wr = first_wr; wr; wr = prev(wr))
{ {
if (stable) if ((wr->entry_type & BS_HEAP_STABLE) || wr->type() == BS_HEAP_COMMIT || wr->type() == BS_HEAP_ROLLBACK)
{ {
stable_version = wr->version; stable_version = wr->version;
return false; break;
} }
if (unstable_size >= unstable_alloc) else
{ {
unstable_alloc = (!unstable_alloc ? 128 : unstable_alloc*2); if (unstable_size >= unstable_alloc)
unstable = (obj_ver_id*)realloc_or_die(unstable, sizeof(obj_ver_id) * unstable_alloc); {
unstable_alloc = (!unstable_alloc ? 128 : unstable_alloc*2);
unstable = (obj_ver_id*)realloc_or_die(unstable, sizeof(obj_ver_id) * unstable_alloc);
}
unstable[unstable_size++] = (obj_ver_id){ .oid = oid, .version = wr->version };
} }
unstable[unstable_size++] = (obj_ver_id){ .oid = oid, .version = wr->version }; }
return true;
});
if (stable_version) if (stable_version)
{ {
if (res_size >= res_alloc) if (res_size >= res_alloc)
@@ -2132,13 +1941,7 @@ void blockstore_heap_t::free_data(inode_t inode, uint64_t location)
inode = (INODE_POOL(inode) << POOL_ID_BITS); inode = (INODE_POOL(inode) << POOL_ID_BITS);
assert(data_alloc->get(location / dsk->data_block_size)); assert(data_alloc->get(location / dsk->data_block_size));
data_alloc->set(location / dsk->data_block_size, false); data_alloc->set(location / dsk->data_block_size, false);
auto sp_it = inode_space_stats.find(inode); inode_space_stats[inode] -= dsk->data_block_size;
if (sp_it != inode_space_stats.end())
{
sp_it->second -= dsk->data_block_size;
if (sp_it->second == 0)
inode_space_stats.erase(sp_it);
}
data_used_space -= dsk->data_block_size; data_used_space -= dsk->data_block_size;
} }
@@ -2367,15 +2170,12 @@ void blockstore_heap_t::apply_inflight(heap_inflight_lsn_t & inflight)
} }
if (!next) if (!next)
{ {
// The last freed entry must be a deletion
assert(!prev); assert(!prev);
assert(wr->entry_type == BS_HEAP_DELETE|BS_HEAP_STABLE); auto & inode_idx = block_index[get_pg_id(wr->inode, wr->stripe)][wr->inode];
auto & pg_idx = block_index[get_pg_id(wr->inode, wr->stripe)];
auto & inode_idx = pg_idx[wr->inode];
heap_inode_map_t::iterator li_it; heap_inode_map_t::iterator li_it;
heap_list_item_t *old_li = NULL; heap_list_item_t *old_li = NULL;
inode_map_get(inode_idx, li_it, old_li, wr->stripe); inode_map_get(inode_idx, li_it, old_li, wr->stripe);
inode_map_erase(pg_idx, inode_idx, li_it, old_li); inode_map_erase(inode_idx, li_it, old_li);
} }
else else
{ {
@@ -2390,15 +2190,6 @@ void blockstore_heap_t::apply_inflight(heap_inflight_lsn_t & inflight)
} }
} }
bool blockstore_heap_t::is_lsn_completed(uint64_t lsn)
{
if (lsn <= completed_lsn)
return true;
assert(lsn-first_inflight_lsn < inflight_lsn.size());
auto it = inflight_lsn.begin() + (lsn-first_inflight_lsn);
return (it->flags & HEAP_INFLIGHT_DONE);
}
uint64_t blockstore_heap_t::get_completed_lsn() uint64_t blockstore_heap_t::get_completed_lsn()
{ {
return completed_lsn; return completed_lsn;
@@ -2423,7 +2214,7 @@ void blockstore_heap_t::set_no_inode_stats(const std::vector<uint64_t> & pool_id
{ {
// Recalculate if changed // Recalculate if changed
if (ps.second.no_inode_stats == 2 || ps.second.no_inode_stats == 1) if (ps.second.no_inode_stats == 2 || ps.second.no_inode_stats == 1)
recalc_inode_space_stats(ps.first, ps.second.no_inode_stats == 2); recalc_inode_space_stats(ps.first, ps.second.no_inode_stats == 1);
ps.second.no_inode_stats &= 1; ps.second.no_inode_stats &= 1;
} }
} }
@@ -2434,8 +2225,8 @@ void blockstore_heap_t::recalc_inode_space_stats(uint64_t pool_id, bool per_inod
auto sp_begin = inode_space_stats.lower_bound((pool_id << (64-POOL_ID_BITS))); auto sp_begin = inode_space_stats.lower_bound((pool_id << (64-POOL_ID_BITS)));
auto sp_end = inode_space_stats.lower_bound(((pool_id+1) << (64-POOL_ID_BITS))); auto sp_end = inode_space_stats.lower_bound(((pool_id+1) << (64-POOL_ID_BITS)));
inode_space_stats.erase(sp_begin, sp_end); inode_space_stats.erase(sp_begin, sp_end);
uint32_t pg_count = ps.pg_count; uint32_t pg_count = ps.pg_count ? ps.pg_count : 1;
for (uint32_t pg_num = pg_count ? 1 : 0; pg_num <= pg_count; pg_num++) for (uint32_t pg_num = 1; pg_num <= pg_count; pg_num++)
{ {
auto & pg_idx = block_index[(pool_id << (64-POOL_ID_BITS)) | pg_num]; auto & pg_idx = block_index[(pool_id << (64-POOL_ID_BITS)) | pg_num];
for (auto & ip: pg_idx) for (auto & ip: pg_idx)
@@ -2468,15 +2259,12 @@ void blockstore_heap_t::recalc_inode_space_stats(uint64_t pool_id, bool per_inod
// This is some really crazy shit but it seems to work well :) // This is some really crazy shit but it seems to work well :)
// At the same time it has almost zero overhead and works just as fast for fat inodes. // At the same time it has almost zero overhead and works just as fast for fat inodes.
void inode_map_get(void *inode_idx, heap_inode_map_t::iterator & li_it, heap_list_item_t* & li, uint64_t stripe) void blockstore_heap_t::inode_map_get(void *inode_idx, heap_inode_map_t::iterator & li_it, heap_list_item_t* & li, uint64_t stripe)
{ {
size_t map_n = ((size_t)inode_idx & IMAP_MALLOC_LOW_BITS); size_t map_n = ((size_t)inode_idx & IMAP_MALLOC_LOW_BITS);
if (!map_n) if (!map_n)
{ {
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Warray-bounds"
li_it = ((heap_inode_map_t*)inode_idx)->find(list_item_key(&stripe)); li_it = ((heap_inode_map_t*)inode_idx)->find(list_item_key(&stripe));
#pragma GCC diagnostic pop
li = li_it != ((heap_inode_map_t*)inode_idx)->end() ? *li_it : NULL; li = li_it != ((heap_inode_map_t*)inode_idx)->end() ? *li_it : NULL;
} }
else if (map_n == 1) else if (map_n == 1)
@@ -2498,7 +2286,7 @@ void inode_map_get(void *inode_idx, heap_inode_map_t::iterator & li_it, heap_lis
} }
} }
void inode_map_free(void* inode_idx) void blockstore_heap_t::inode_map_free(void* inode_idx)
{ {
size_t n = ((size_t)inode_idx & IMAP_MALLOC_LOW_BITS); size_t n = ((size_t)inode_idx & IMAP_MALLOC_LOW_BITS);
if (!n) if (!n)
@@ -2511,12 +2299,7 @@ void inode_map_free(void* inode_idx)
} }
} }
bool inode_map_is_big(void* & inode_idx) void blockstore_heap_t::inode_map_iterate(void* & inode_idx, std::function<void(heap_list_item_t*)> cb)
{
return !((size_t)inode_idx & IMAP_MALLOC_LOW_BITS);
}
void inode_map_iterate(void* & inode_idx, std::function<void(heap_list_item_t*)> cb)
{ {
size_t n = ((size_t)inode_idx & IMAP_MALLOC_LOW_BITS); size_t n = ((size_t)inode_idx & IMAP_MALLOC_LOW_BITS);
if (!n) if (!n)
@@ -2543,7 +2326,7 @@ void inode_map_iterate(void* & inode_idx, std::function<void(heap_list_item_t*)>
} }
} }
void inode_map_put(void* & inode_idx, heap_list_item_t* li) void blockstore_heap_t::inode_map_put(void* & inode_idx, heap_list_item_t* li)
{ {
if (!inode_idx) if (!inode_idx)
{ {
@@ -2612,7 +2395,7 @@ void inode_map_put(void* & inode_idx, heap_list_item_t* li)
} }
} }
void inode_map_replace(void* & inode_idx, const heap_inode_map_t::iterator & li_it, heap_list_item_t* new_li) void blockstore_heap_t::inode_map_replace(void* & inode_idx, const heap_inode_map_t::iterator & li_it, heap_list_item_t* new_li)
{ {
size_t map_n = ((size_t)inode_idx & IMAP_MALLOC_LOW_BITS); size_t map_n = ((size_t)inode_idx & IMAP_MALLOC_LOW_BITS);
if (!map_n) if (!map_n)
@@ -2638,8 +2421,7 @@ void inode_map_replace(void* & inode_idx, const heap_inode_map_t::iterator & li_
} }
} }
void inode_map_erase(robin_hood::unordered_flat_map<inode_t, void*, i64hash_t> & pg_idx, void* & inode_idx, void blockstore_heap_t::inode_map_erase(void* & inode_idx, const heap_inode_map_t::iterator & li_it, heap_list_item_t* li)
const heap_inode_map_t::iterator & li_it, heap_list_item_t* li)
{ {
size_t map_n = ((size_t)inode_idx & IMAP_MALLOC_LOW_BITS); size_t map_n = ((size_t)inode_idx & IMAP_MALLOC_LOW_BITS);
if (!map_n) if (!map_n)
@@ -2664,7 +2446,7 @@ void inode_map_erase(robin_hood::unordered_flat_map<inode_t, void*, i64hash_t> &
else if (map_n == 1) else if (map_n == 1)
{ {
// Erase // Erase
pg_idx.erase(li->entry.inode); block_index[get_pg_id(li->entry.inode, li->entry.stripe)].erase(li->entry.inode);
} }
else else
{ {
+18 -19
View File
@@ -43,7 +43,7 @@ struct __attribute__((__packed__)) heap_entry_t
{ {
uint16_t size; uint16_t size;
uint16_t entry_type; uint16_t entry_type;
uint32_t checksum; uint32_t crc32c;
uint64_t lsn; uint64_t lsn;
uint64_t inode; uint64_t inode;
uint64_t stripe; uint64_t stripe;
@@ -69,8 +69,7 @@ struct __attribute__((__packed__)) heap_entry_t
uint32_t *get_checksum(blockstore_heap_t *heap); uint32_t *get_checksum(blockstore_heap_t *heap);
uint64_t big_location(blockstore_heap_t *heap); uint64_t big_location(blockstore_heap_t *heap);
void set_big_location(blockstore_heap_t *heap, uint64_t location); void set_big_location(blockstore_heap_t *heap, uint64_t location);
uint32_t calc_checksum(blockstore_heap_t *heap); uint32_t calc_crc32c();
uint32_t calc_checksum(blockstore_disk_t *dsk);
}; };
struct __attribute__((__packed__)) heap_small_write_t struct __attribute__((__packed__)) heap_small_write_t
@@ -81,7 +80,7 @@ struct __attribute__((__packed__)) heap_small_write_t
uint32_t offset; uint32_t offset;
uint32_t len; uint32_t len;
// Also includes 1 bitmap and 1 checksum after the bitmap if block checksums are disabled // Also includes 1 bitmap and 1 crc32c after the bitmap if checksums are disabled
}; };
struct __attribute__((__packed__)) heap_big_write_t struct __attribute__((__packed__)) heap_big_write_t
@@ -99,7 +98,7 @@ struct __attribute__((__packed__)) heap_big_intent_t
uint32_t offset; uint32_t offset;
uint32_t len; uint32_t len;
// Also includes 2 bitmaps and 1 checksums if block checksums are disabled // Also includes 2 bitmaps and 1 crc32c if checksums are disabled
}; };
struct __attribute__((__packed__)) heap_list_item_t struct __attribute__((__packed__)) heap_list_item_t
@@ -138,8 +137,6 @@ struct heap_compact_t
bool do_delete; bool do_delete;
}; };
struct heap_reshard_state_t;
struct heap_li_hash struct heap_li_hash
{ {
size_t operator()(const heap_list_item_t* li) const noexcept size_t operator()(const heap_list_item_t* li) const noexcept
@@ -164,7 +161,7 @@ using heap_mvcc_map_t = robin_hood::unordered_flat_map<object_id, heap_object_mv
class blockstore_heap_t class blockstore_heap_t
{ {
friend struct heap_entry_t; friend class heap_entry_t;
blockstore_disk_t *dsk = NULL; blockstore_disk_t *dsk = NULL;
uint8_t* buffer_area = NULL; uint8_t* buffer_area = NULL;
@@ -201,7 +198,6 @@ class blockstore_heap_t
bool marked_used_blocks = false; bool marked_used_blocks = false;
bool recheck_queue_filled = false; bool recheck_queue_filled = false;
std::vector<heap_list_item_t*> loaded_list_items;
std::set<uint32_t> recheck_modified_blocks; std::set<uint32_t> recheck_modified_blocks;
std::deque<heap_entry_t*> recheck_queue; std::deque<heap_entry_t*> recheck_queue;
int recheck_in_progress = 0; int recheck_in_progress = 0;
@@ -209,15 +205,20 @@ class blockstore_heap_t
std::function<void(bool is_data, uint64_t offset, uint64_t len, uint8_t* buf, std::function<void()>)> recheck_cb; std::function<void(bool is_data, uint64_t offset, uint64_t len, uint8_t* buf, std::function<void()>)> recheck_cb;
int recheck_queue_depth = 0; int recheck_queue_depth = 0;
void inode_map_put(void* & inode_idx, heap_list_item_t* li);
void inode_map_get(void *inode_idx, heap_inode_map_t::iterator & li_it, heap_list_item_t* & li, uint64_t stripe);
void inode_map_free(void* inode_idx);
void inode_map_iterate(void* & inode_idx, std::function<void(heap_list_item_t*)> cb);
void inode_map_replace(void* & inode_idx, const heap_inode_map_t::iterator & li_it, heap_list_item_t* new_li);
void inode_map_erase(void* & inode_idx, const heap_inode_map_t::iterator & li_it, heap_list_item_t* li);
uint64_t get_pg_id(inode_t inode, uint64_t stripe); uint64_t get_pg_id(inode_t inode, uint64_t stripe);
bool validate_object(heap_entry_t *obj); bool validate_object(heap_entry_t *obj);
void fill_recheck_queue(); void fill_recheck_queue();
int mark_used_blocks(); int mark_used_blocks();
void recheck_buffer(heap_entry_t *cwr, uint8_t *buf); void recheck_buffer(heap_entry_t *cwr, uint8_t *buf);
void defragment_block(uint32_t block_num); void defragment_block(uint32_t block_num);
void reshard_add(heap_reshard_state_t *st, heap_list_item_t *li);
void gc_block(heap_block_info_t & inf);
int allocate_entry(uint32_t entry_size, uint32_t *block_num, bool allow_last_free); int allocate_entry(uint32_t entry_size, uint32_t *block_num, bool allow_last_free);
void insert_list_item(heap_list_item_t *li); void insert_list_item(heap_list_item_t *li);
int add_entry(uint32_t wr_size, uint32_t *modified_block, bool allow_last_free, int add_entry(uint32_t wr_size, uint32_t *modified_block, bool allow_last_free,
@@ -240,29 +241,28 @@ public:
std::function<void(uint32_t, uint32_t, uint8_t*)> handle_block); std::function<void(uint32_t, uint32_t, uint8_t*)> handle_block);
int load_blocks(uint64_t disk_offset, uint64_t size, uint8_t *buf, int load_blocks(uint64_t disk_offset, uint64_t size, uint8_t *buf,
bool allow_corrupted, uint64_t &entries_loaded); bool allow_corrupted, uint64_t &entries_loaded);
// finish loading - should be called after load_blocks // finish loading
void finish_load(); int finish_load(bool allow_corrupted = false);
// get blocks which are modified during loading and should be written to the disk // get blocks which are modified during loading and should be written to the disk
// before finishing initialization if not R/O // before finishing initialization if not R/O
std::vector<uint32_t> get_recheck_modified_blocks(); std::vector<uint32_t> get_recheck_modified_blocks();
// recheck small write data after reading the database from disk // recheck small write data after reading the database from disk
bool recheck_small_writes(std::function<void(bool is_data, uint64_t offset, uint64_t len, uint8_t* buf, std::function<void()>)> read_buffer, int queue_depth); bool recheck_small_writes(std::function<void(bool is_data, uint64_t offset, uint64_t len, uint8_t* buf, std::function<void()>)> read_buffer, int queue_depth);
int finish_recheck();
// reshard database according to the pool's PG count // reshard database according to the pool's PG count
void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit); void reshard(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size);
bool reshard_continue(void* reshard_state, uint64_t chunk_limit);
bool reshard_check(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size);
void set_no_inode_stats(const std::vector<uint64_t> & pool_ids); void set_no_inode_stats(const std::vector<uint64_t> & pool_ids);
void recalc_inode_space_stats(uint64_t pool_id, bool per_inode); void recalc_inode_space_stats(uint64_t pool_id, bool per_inode);
// read an object entry and lock it against removal // read an object entry and lock it against removal
// in the future, may become asynchronous // in the future, may become asynchronous
heap_entry_t *lock_and_read_entry(object_id oid); heap_entry_t *lock_and_read_entry(object_id oid);
// re-read a locked object entry with the given lsn (pointer may be invalidated)
heap_entry_t *read_locked_entry(object_id oid, uint64_t lsn);
// read an object entry without locking it // read an object entry without locking it
heap_entry_t *read_entry(object_id oid); heap_entry_t *read_entry(object_id oid);
// unlock an entry // unlock an entry
bool unlock_entry(object_id oid); bool unlock_entry(object_id oid);
// set or verify checksums in a write request // set or verify checksums in a write request
bool calc_checksums(heap_entry_t *wr, uint8_t *data, bool set, uint32_t offset = UINT32_MAX, uint32_t len = UINT32_MAX); bool calc_checksums(heap_entry_t *wr, uint8_t *data, bool set, uint32_t offset = 0, uint32_t len = 0);
// set or verify raw block checksums // set or verify raw block checksums
bool calc_block_checksums(uint32_t *block_csums, uint8_t *data, uint8_t *bitmap, uint32_t start, uint32_t end, bool calc_block_checksums(uint32_t *block_csums, uint8_t *data, uint8_t *bitmap, uint32_t start, uint32_t end,
bool set, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb); bool set, std::function<void(uint32_t, uint32_t, uint32_t)> bad_block_cb);
@@ -314,7 +314,6 @@ public:
void start_block_write(uint32_t block_num); void start_block_write(uint32_t block_num);
void complete_block_write(uint32_t block_num); void complete_block_write(uint32_t block_num);
void complete_lsn_write(uint64_t lsn); void complete_lsn_write(uint64_t lsn);
bool is_lsn_completed(uint64_t lsn);
uint64_t get_completed_lsn(); uint64_t get_completed_lsn();
uint64_t get_fsynced_lsn(); uint64_t get_fsynced_lsn();
void mark_lsn_fsynced(uint64_t lsn); void mark_lsn_fsynced(uint64_t lsn);
+8 -20
View File
@@ -23,7 +23,6 @@ blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_i *
dsk.open_meta(); dsk.open_meta();
dsk.open_journal(); dsk.open_journal();
dsk.calc_lengths(); dsk.calc_lengths();
dsk.check_lengths();
} }
catch (std::exception & e) catch (std::exception & e)
{ {
@@ -32,13 +31,16 @@ blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_i *
} }
meta_superblock = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.meta_block_size); meta_superblock = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.meta_block_size);
memset(meta_superblock, 0, dsk.meta_block_size); memset(meta_superblock, 0, dsk.meta_block_size);
}
void blockstore_impl_t::init()
{
flusher = new journal_flusher_t(this); flusher = new journal_flusher_t(this);
if (dsk.inmemory_journal) if (dsk.inmemory_journal)
{ {
buffer_area = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.journal_len); buffer_area = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.journal_len);
} }
heap = new blockstore_heap_t(&dsk, buffer_area, log_level); heap = new blockstore_heap_t(&dsk, buffer_area, log_level);
ringloop->wakeup();
} }
blockstore_impl_t::~blockstore_impl_t() blockstore_impl_t::~blockstore_impl_t()
@@ -193,12 +195,12 @@ void blockstore_impl_t::loop()
heap->start_block_write(block_num); heap->start_block_write(block_num);
mb.sent = true; mb.sent = true;
} }
pending_modified_blocks.clear();
int ret = ringloop->submit(); int ret = ringloop->submit();
if (ret < 0) if (ret < 0)
{ {
throw std::runtime_error(std::string("io_uring_submit: ") + strerror(-ret)); throw std::runtime_error(std::string("io_uring_submit: ") + strerror(-ret));
} }
pending_modified_blocks.clear();
if ((initial_ring_space - ringloop->space_left()) > 0) if ((initial_ring_space - ringloop->space_left()) > 0)
{ {
live = true; live = true;
@@ -323,13 +325,9 @@ void blockstore_impl_t::process_list(blockstore_op_t *op)
FINISH_OP(op); FINISH_OP(op);
return; return;
} }
// Check if the DB is sharded correctly // Check if the DB needs resharding
if (!heap->reshard_check(INODE_POOL(min_inode), pg_count, pg_stripe_size)) // (we don't know about PGs from the beginning, we only create "shards" here)
{ heap->reshard(INODE_POOL(min_inode), pg_count, pg_stripe_size);
op->retval = -EAGAIN;
FINISH_OP(op);
return;
}
obj_ver_id *result = NULL; obj_ver_id *result = NULL;
size_t stable_count = 0, unstable_count = 0; size_t stable_count = 0, unstable_count = 0;
int res = heap->list_objects(list_pg, op->min_oid, op->max_oid, &result, &stable_count, &unstable_count); int res = heap->list_objects(list_pg, op->min_oid, op->max_oid, &result, &stable_count, &unstable_count);
@@ -396,13 +394,3 @@ std::string blockstore_impl_t::get_op_diag(blockstore_op_t *op)
snprintf(buf, sizeof(buf), "state=%d", priv->op_state); snprintf(buf, sizeof(buf), "state=%d", priv->op_state);
return std::string(buf); return std::string(buf);
} }
void* blockstore_impl_t::reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit)
{
return heap->reshard_start(pool, pg_count, pg_stripe_size, chunk_limit);
}
bool blockstore_impl_t::reshard_continue(void *reshard_state, uint64_t chunk_limit)
{
return heap->reshard_continue(reshard_state, chunk_limit);
}
+1 -4
View File
@@ -78,7 +78,6 @@ public:
// Suitable only for server SSDs with capacitors, requires disabled data and journal fsyncs // Suitable only for server SSDs with capacitors, requires disabled data and journal fsyncs
int immediate_commit = IMMEDIATE_NONE; int immediate_commit = IMMEDIATE_NONE;
bool inmemory_meta = false; bool inmemory_meta = false;
bool skip_corrupted_meta_entries = false;
uint32_t meta_write_recheck_parallelism = 0; uint32_t meta_write_recheck_parallelism = 0;
// Maximum and minimum flusher count // Maximum and minimum flusher count
unsigned max_flusher_count = 0, min_flusher_count = 0; unsigned max_flusher_count = 0, min_flusher_count = 0;
@@ -143,6 +142,7 @@ public:
int metadata_buf_size; int metadata_buf_size;
blockstore_init_meta* metadata_init_reader; blockstore_init_meta* metadata_init_reader;
void init();
void check_wait(blockstore_op_t *op); void check_wait(blockstore_op_t *op);
void init_op(blockstore_op_t *op); void init_op(blockstore_op_t *op);
@@ -190,9 +190,6 @@ public:
void parse_config(blockstore_config_t & config); void parse_config(blockstore_config_t & config);
void parse_config(blockstore_config_t & config, bool init); void parse_config(blockstore_config_t & config, bool init);
void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit);
bool reshard_continue(void *reshard_state, uint64_t chunk_limit);
// Event loop // Event loop
void loop(); void loop();
+14 -6
View File
@@ -72,6 +72,7 @@ resume_1:
} }
if (is_zero((uint64_t*)bs->meta_superblock, bs->dsk.meta_block_size)) if (is_zero((uint64_t*)bs->meta_superblock, bs->dsk.meta_block_size))
{ {
bs->dsk.check_lengths();
{ {
blockstore_meta_header_v3_t *hdr = (blockstore_meta_header_v3_t *)bs->meta_superblock; blockstore_meta_header_v3_t *hdr = (blockstore_meta_header_v3_t *)bs->meta_superblock;
hdr->zero = 0; hdr->zero = 0;
@@ -140,12 +141,12 @@ resume_1:
hdr->bitmap_granularity != bs->dsk.bitmap_granularity || hdr->bitmap_granularity != bs->dsk.bitmap_granularity ||
hdr->data_csum_type != bs->dsk.data_csum_type || hdr->data_csum_type != bs->dsk.data_csum_type ||
hdr->csum_block_size != bs->dsk.csum_block_size || hdr->csum_block_size != bs->dsk.csum_block_size ||
hdr->meta_area_size != bs->dsk.meta_area_size) hdr->meta_area_size > bs->dsk.meta_area_size)
{ {
printf( printf(
"Configuration stored in metadata superblock" "Configuration stored in metadata superblock"
" (meta_block_size=%u, data_block_size=%u, bitmap_granularity=%u, data_csum_type=%u, csum_block_size=%u, meta_area_size=%ju)" " (meta_block_size=%u, data_block_size=%u, bitmap_granularity=%u, data_csum_type=%u, csum_block_size=%u, meta_area_size=%ju)"
" differs from OSD configuration (%ju/%ju/%u, %u/%u, %ju).\n", " differs from OSD configuration (%u/%u/%u, %u/%u, %ju).\n",
hdr->meta_block_size, hdr->data_block_size, hdr->bitmap_granularity, hdr->meta_block_size, hdr->data_block_size, hdr->bitmap_granularity,
hdr->data_csum_type, hdr->csum_block_size, hdr->meta_area_size, hdr->data_csum_type, hdr->csum_block_size, hdr->meta_area_size,
bs->dsk.meta_block_size, bs->dsk.data_block_size, bs->dsk.bitmap_granularity, bs->dsk.meta_block_size, bs->dsk.data_block_size, bs->dsk.bitmap_granularity,
@@ -153,7 +154,15 @@ resume_1:
); );
exit(1); exit(1);
} }
bs->dsk.meta_area_size = hdr->meta_area_size;
if (bs->dsk.meta_format != hdr->version)
{
bs->dsk.meta_format = hdr->version;
bs->dsk.calc_lengths();
}
bs->dsk.check_lengths();
} }
bs->init();
bs->heap->start_load(((blockstore_meta_header_v3_t *)bs->meta_superblock)->completed_lsn); bs->heap->start_load(((blockstore_meta_header_v3_t *)bs->meta_superblock)->completed_lsn);
if (bs->dsk.inmemory_journal) if (bs->dsk.inmemory_journal)
{ {
@@ -225,7 +234,7 @@ resume_4:
{ {
// Handle result // Handle result
uint64_t loaded = 0; uint64_t loaded = 0;
int r = bs->heap->load_blocks(bufs[i].offset-bs->dsk.meta_block_size, bufs[i].size, bufs[i].buf, bs->skip_corrupted_meta_entries, loaded); int r = bs->heap->load_blocks(bufs[i].offset-bs->dsk.meta_block_size, bufs[i].size, bufs[i].buf, false, loaded);
if (r != 0) if (r != 0)
exit(1); exit(1);
entries_loaded += loaded; entries_loaded += loaded;
@@ -239,7 +248,6 @@ resume_4:
return 1; return 1;
} }
// metadata read finished // metadata read finished
bs->heap->finish_load();
printf("Metadata entries loaded: %ju, used blocks: %ju / %ju\n", entries_loaded, bs->heap->get_data_used_space() / bs->dsk.data_block_size, bs->dsk.block_count); printf("Metadata entries loaded: %ju, used blocks: %ju / %ju\n", entries_loaded, bs->heap->get_data_used_space() / bs->dsk.data_block_size, bs->dsk.block_count);
if (zero_on_init && !bs->dsk.disable_meta_fsync) if (zero_on_init && !bs->dsk.disable_meta_fsync)
{ {
@@ -270,7 +278,7 @@ resume_6:
} }
GET_SQE(); GET_SQE();
data->iov = (iovec){ buf, len }; data->iov = (iovec){ buf, len };
data->callback = [offset, cb](ring_data_t *data) data->callback = [this, offset, cb](ring_data_t *data)
{ {
if (data->res < 0) if (data->res < 0)
{ {
@@ -285,7 +293,7 @@ resume_6:
}, bs->meta_write_recheck_parallelism); }, bs->meta_write_recheck_parallelism);
return 1; return 1;
resume_7: resume_7:
if (bs->heap->finish_recheck() != 0) if (bs->heap->finish_load() != 0)
{ {
exit(1); exit(1);
} }
-1
View File
@@ -28,7 +28,6 @@ void blockstore_impl_t::parse_config(blockstore_config_t & config, bool init)
throttle_target_parallelism = strtoull(config["throttle_target_parallelism"].c_str(), NULL, 10); throttle_target_parallelism = strtoull(config["throttle_target_parallelism"].c_str(), NULL, 10);
throttle_threshold_us = strtoull(config["throttle_threshold_us"].c_str(), NULL, 10); throttle_threshold_us = strtoull(config["throttle_threshold_us"].c_str(), NULL, 10);
perfect_csum_update = config["perfect_csum_update"] == "true" || config["perfect_csum_update"] == "1" || config["perfect_csum_update"] == "yes"; perfect_csum_update = config["perfect_csum_update"] == "true" || config["perfect_csum_update"] == "1" || config["perfect_csum_update"] == "yes";
skip_corrupted_meta_entries = config["skip_corrupted_meta_entries"] == "true" || config["skip_corrupted_meta_entries"] == "1" || config["skip_corrupted_meta_entries"] == "yes";
if (config["autosync_writes"] != "") if (config["autosync_writes"] != "")
{ {
autosync_writes = strtoull(config["autosync_writes"].c_str(), NULL, 10); autosync_writes = strtoull(config["autosync_writes"].c_str(), NULL, 10);
+13 -29
View File
@@ -22,7 +22,7 @@ int blockstore_impl_t::dequeue_read(blockstore_op_t *op)
uint64_t result_version = 0; uint64_t result_version = 0;
bool found = false; bool found = false;
uint32_t skip_csum = 0; uint32_t skip_csum = 0;
uint32_t blk_start = op->offset, blk_end = op->offset+op->len; uint32_t blk_start = 0, blk_end = 0;
bool need_skip = dsk.csum_block_size > dsk.bitmap_granularity && !perfect_csum_update; bool need_skip = dsk.csum_block_size > dsk.bitmap_granularity && !perfect_csum_update;
if (need_skip) if (need_skip)
{ {
@@ -32,28 +32,12 @@ int blockstore_impl_t::dequeue_read(blockstore_op_t *op)
if (blk_end % dsk.csum_block_size) if (blk_end % dsk.csum_block_size)
blk_end += dsk.csum_block_size - (blk_end % dsk.csum_block_size); blk_end += dsk.csum_block_size - (blk_end % dsk.csum_block_size);
} }
bool need_wait = false;
heap->iterate_with_stable(obj, obj->lsn, [&](heap_entry_t *wr, bool stable) heap->iterate_with_stable(obj, obj->lsn, [&](heap_entry_t *wr, bool stable)
{ {
if (wr->type() == BS_HEAP_DELETE) if (wr->type() == BS_HEAP_DELETE)
{ {
return false; return false;
} }
if (!heap->is_lsn_completed(wr->lsn))
{
if (wr->type() == BS_HEAP_BIG_INTENT && wr->big_intent().offset < blk_end && wr->big_intent().offset+wr->big_intent().len > blk_start ||
wr->type() == BS_HEAP_INTENT_WRITE && wr->small().offset < blk_end && wr->small().offset+wr->small().len > blk_start)
{
// Wait until intent write is completed
need_wait = true;
return false;
}
else if (wr->type() == BS_HEAP_SMALL_WRITE && wr->small().offset < blk_end && wr->small().offset+wr->small().len > blk_start)
{
// Skip entry and read the previous one
return true;
}
}
if (op->version >= wr->version && !found) if (op->version >= wr->version && !found)
{ {
found = true; found = true;
@@ -63,6 +47,12 @@ int blockstore_impl_t::dequeue_read(blockstore_op_t *op)
memcpy(op->bitmap, wr->get_ext_bitmap(heap), dsk.clean_entry_bitmap_size); memcpy(op->bitmap, wr->get_ext_bitmap(heap), dsk.clean_entry_bitmap_size);
} }
} }
if (need_skip && wr->lsn < heap->get_completed_lsn() &&
(wr->type() == BS_HEAP_BIG_INTENT && wr->big_intent().offset < blk_end && wr->big_intent().offset+wr->big_intent().len > blk_start ||
wr->type() == BS_HEAP_INTENT_WRITE && wr->small().offset < blk_end && wr->small().offset+wr->small().len > blk_start))
{
skip_csum = COPY_BUF_SKIP_CSUM;
}
if (op->version >= wr->version) if (op->version >= wr->version)
{ {
fulfilled += prepare_read(PRIV(op)->read_vec, obj, wr, op->offset, op->offset+op->len, fulfilled += prepare_read(PRIV(op)->read_vec, obj, wr, op->offset, op->offset+op->len,
@@ -75,23 +65,13 @@ int blockstore_impl_t::dequeue_read(blockstore_op_t *op)
return false; return false;
} }
} }
if (need_skip && wr->type() == BS_HEAP_SMALL_WRITE && if (need_skip && (wr->type() == BS_HEAP_SMALL_WRITE || wr->type() == BS_HEAP_INTENT_WRITE) &&
wr->small().offset < blk_end && wr->small().offset+wr->small().len > blk_start) wr->small().offset < blk_end && wr->small().offset+wr->small().len > blk_start)
{ {
// Small write may mutate big write checksums during flush
skip_csum = COPY_BUF_SKIP_CSUM; skip_csum = COPY_BUF_SKIP_CSUM;
} }
return true; return true;
}); });
if (need_wait)
{
undo_wait:
// Need to wait. undo added requests, unlock lsn
heap->unlock_entry(op->oid);
free_read_buffers(rv);
rv.clear();
return 0;
}
if (!found) if (!found)
{ {
// May happen if there are entries but all of them are > requested version // May happen if there are entries but all of them are > requested version
@@ -104,7 +84,11 @@ undo_wait:
assert(fulfilled == op->len); assert(fulfilled == op->len);
if (!fulfill_read(op)) if (!fulfill_read(op))
{ {
goto undo_wait; // Need to wait. undo added requests, unlock lsn
heap->unlock_entry(op->oid);
free_read_buffers(rv);
rv.clear();
return 0;
} }
op->version = result_version; op->version = result_version;
if (!PRIV(op)->pending_ops) if (!PRIV(op)->pending_ops)
+1 -1
View File
@@ -57,9 +57,9 @@ int blockstore_impl_t::dequeue_stable(blockstore_op_t *op)
} }
assert(res == 0); assert(res == 0);
} }
resume_1:
if (priv->modified_block != UINT32_MAX && priv->modified_block2 != priv->modified_block) if (priv->modified_block != UINT32_MAX && priv->modified_block2 != priv->modified_block)
{ {
resume_1:
BS_SUBMIT_CHECK_SQES(1); BS_SUBMIT_CHECK_SQES(1);
prepare_meta_block_write(priv->modified_block); prepare_meta_block_write(priv->modified_block);
resume_2: resume_2:
+7 -10
View File
@@ -22,23 +22,21 @@ void blockstore_impl_t::prepare_meta_block_write(uint32_t modified_block)
ring_data_t *data = ((ring_data_t*)sqe->user_data); ring_data_t *data = ((ring_data_t*)sqe->user_data);
uint8_t *buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.meta_block_size); uint8_t *buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.meta_block_size);
data->iov = (struct iovec){ buf, (size_t)dsk.meta_block_size }; data->iov = (struct iovec){ buf, (size_t)dsk.meta_block_size };
data->callback = [this, modified_block](ring_data_t *data) data->callback = [this, modified_block, buf](ring_data_t *data)
{ {
free(buf);
live = true; live = true;
if (data->res != data->iov.iov_len) if (data->res != data->iov.iov_len)
{ {
// FIXME: our state becomes corrupted after a write error. maybe do something better than just die // FIXME: our state becomes corrupted after a write error. maybe do something better than just die
disk_error_abort("data write", data->res, data->iov.iov_len); disk_error_abort("data write", data->res, data->iov.iov_len);
} }
auto it = modified_blocks.find(modified_block); modified_blocks.erase(modified_block);
assert(it != modified_blocks.end());
free(it->second.buf);
modified_blocks.erase(it);
heap->complete_block_write(modified_block); heap->complete_block_write(modified_block);
ringloop->wakeup(); ringloop->wakeup();
}; };
io_uring_prep_writev( io_uring_prep_writev(
sqe, dsk.meta_fd, &data->iov, 1, dsk.meta_offset + ((uint64_t)modified_block+1)*dsk.meta_block_size sqe, dsk.meta_fd, &data->iov, 1, dsk.meta_offset + (modified_block+1)*dsk.meta_block_size
); );
unsynced_meta_write_count++; unsynced_meta_write_count++;
pending_modified_blocks.push_back(modified_block); pending_modified_blocks.push_back(modified_block);
@@ -251,12 +249,13 @@ enospc:
goto enospc; goto enospc;
assert(res == 0); assert(res == 0);
PRIV(op)->lsn = obj->lsn; PRIV(op)->lsn = obj->lsn;
if (op->len)
heap->use_buffer_area(op->oid.inode, loc, op->len);
prepare_meta_block_write(PRIV(op)->modified_block); prepare_meta_block_write(PRIV(op)->modified_block);
PRIV(op)->pending_ops++; PRIV(op)->pending_ops++;
if (op->len > 0) if (op->len > 0)
{ {
// Prepare buffered data write // Prepare buffered data write
heap->use_buffer_area(op->oid.inode, loc, op->len);
if (dsk.inmemory_journal) if (dsk.inmemory_journal)
{ {
memcpy((uint8_t*)buffer_area + loc, op->buf, op->len); memcpy((uint8_t*)buffer_area + loc, op->buf, op->len);
@@ -349,7 +348,6 @@ resume_12:
} }
resume_4: resume_4:
{ {
BS_SUBMIT_CHECK_SQES(1);
auto obj = heap->read_entry(op->oid); auto obj = heap->read_entry(op->oid);
int res = 0; int res = 0;
if (PRIV(op)->write_type == _REDIRECT_INTENT) if (PRIV(op)->write_type == _REDIRECT_INTENT)
@@ -406,12 +404,11 @@ resume_6:
if (ref_us > exec_us + throttle_threshold_us) if (ref_us > exec_us + throttle_threshold_us)
{ {
// Pause reply // Pause reply
PRIV(op)->pending_ops++;
PRIV(op)->op_state = 7; PRIV(op)->op_state = 7;
// Remember that the timer can in theory be called right here // Remember that the timer can in theory be called right here
tfd->set_timer_us(ref_us-exec_us, false, [this, op](int timer_id) tfd->set_timer_us(ref_us-exec_us, false, [this, op](int timer_id)
{ {
PRIV(op)->pending_ops--; PRIV(op)->op_state = 8;
ringloop->wakeup(); ringloop->wakeup();
}); });
return 1; return 1;
+16 -64
View File
@@ -407,77 +407,32 @@ blockstore_clean_db_t& blockstore_impl_t::clean_db_shard(object_id oid)
return clean_db_shards[(pool_id << (64-POOL_ID_BITS)) | pg_num]; return clean_db_shards[(pool_id << (64-POOL_ID_BITS)) | pg_num];
} }
struct bs_reshard_state_t void blockstore_impl_t::reshard_clean_db(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size)
{ {
int state = 0; uint64_t pool_id = (uint64_t)pool;
uint64_t pool_id = 0;
uint32_t pg_count = 0;
uint32_t pg_stripe_size = 0;
uint64_t chunk_size = 0;
std::map<pool_pg_id_t, blockstore_clean_db_t> old_shards;
std::map<pool_pg_id_t, blockstore_clean_db_t> new_shards; std::map<pool_pg_id_t, blockstore_clean_db_t> new_shards;
std::map<pool_pg_id_t, blockstore_clean_db_t>::iterator sh_it; auto sh_it = clean_db_shards.lower_bound((pool_id << (64-POOL_ID_BITS)));
blockstore_clean_db_t::iterator obj_it;
};
void* blockstore_impl_t::reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit)
{
auto & settings = clean_db_settings[pool];
if (settings.pg_count == pg_count && settings.pg_stripe_size == pg_stripe_size)
{
return NULL;
}
bs_reshard_state_t *st = new bs_reshard_state_t;
st->state = 0;
st->pool_id = pool;
st->pg_count = pg_count;
st->pg_stripe_size = pg_stripe_size;
auto sh_it = clean_db_shards.lower_bound((st->pool_id << (64-POOL_ID_BITS)));
while (sh_it != clean_db_shards.end() && while (sh_it != clean_db_shards.end() &&
(sh_it->first >> (64-POOL_ID_BITS)) == st->pool_id) (sh_it->first >> (64-POOL_ID_BITS)) == pool_id)
{ {
st->old_shards[sh_it->first] = std::move(sh_it->second); for (auto & pair: sh_it->second)
{
// like map_to_pg()
uint64_t pg_num = (pair.first.stripe / pg_stripe_size) % pg_count + 1;
uint64_t shard_id = (pool_id << (64-POOL_ID_BITS)) | pg_num;
new_shards[shard_id][pair.first] = pair.second;
}
clean_db_shards.erase(sh_it++); clean_db_shards.erase(sh_it++);
} }
bool finished = reshard_continue(st, chunk_limit); for (sh_it = new_shards.begin(); sh_it != new_shards.end(); sh_it++)
return finished ? NULL : st;
}
bool blockstore_impl_t::reshard_continue(void *reshard_state, uint64_t chunk_limit)
{
bs_reshard_state_t *st = (bs_reshard_state_t*)reshard_state;
uint64_t chunk_size = 0;
if (st->state == 1)
goto resume_1;
for (st->sh_it = st->old_shards.begin(); st->sh_it != st->old_shards.end(); )
{
for (st->obj_it = st->sh_it->second.begin(); st->obj_it != st->sh_it->second.end(); st->obj_it++)
{
if (chunk_limit > 0 && chunk_size >= chunk_limit)
{
st->state = 1;
return false;
}
resume_1:
// like map_to_pg()
uint64_t pg_num = (st->obj_it->first.stripe / st->pg_stripe_size) % st->pg_count + 1;
uint64_t shard_id = (st->pool_id << (64-POOL_ID_BITS)) | pg_num;
st->new_shards[shard_id][st->obj_it->first] = st->obj_it->second;
chunk_size++;
}
st->old_shards.erase(st->sh_it++);
}
for (auto sh_it = st->new_shards.begin(); sh_it != st->new_shards.end(); sh_it++)
{ {
auto & to = clean_db_shards[sh_it->first]; auto & to = clean_db_shards[sh_it->first];
to.swap(sh_it->second); to.swap(sh_it->second);
} }
clean_db_settings[st->pool_id] = (pool_shard_settings_t){ clean_db_settings[pool_id] = (pool_shard_settings_t){
.pg_count = st->pg_count, .pg_count = pg_count,
.pg_stripe_size = st->pg_stripe_size, .pg_stripe_size = pg_stripe_size,
}; };
delete st;
return true;
} }
void blockstore_impl_t::process_list(blockstore_op_t *op) void blockstore_impl_t::process_list(blockstore_op_t *op)
@@ -510,10 +465,7 @@ void blockstore_impl_t::process_list(blockstore_op_t *op)
sh_it->second.pg_count != pg_count || sh_it->second.pg_count != pg_count ||
sh_it->second.pg_stripe_size != pg_stripe_size) sh_it->second.pg_stripe_size != pg_stripe_size)
{ {
// Sharding mismatch reshard_clean_db(pool_id, pg_count, pg_stripe_size);
op->retval = -EAGAIN;
FINISH_OP(op);
return;
} }
first_shard = last_shard = ((uint64_t)pool_id << (64-POOL_ID_BITS)) | list_pg; first_shard = last_shard = ((uint64_t)pool_id << (64-POOL_ID_BITS)) | list_pg;
} }
+1 -4
View File
@@ -202,6 +202,7 @@ class blockstore_impl_t: public blockstore_i
uint8_t* get_clean_entry_bitmap(uint64_t block_loc, int offset); uint8_t* get_clean_entry_bitmap(uint64_t block_loc, int offset);
blockstore_clean_db_t& clean_db_shard(object_id oid); blockstore_clean_db_t& clean_db_shard(object_id oid);
void reshard_clean_db(pool_id_t pool_id, uint32_t pg_count, uint32_t pg_stripe_size);
void recalc_inode_space_stats(uint64_t pool_id, bool per_inode); void recalc_inode_space_stats(uint64_t pool_id, bool per_inode);
// Journaling // Journaling
@@ -287,10 +288,6 @@ public:
void parse_config(blockstore_config_t & config); void parse_config(blockstore_config_t & config);
void parse_config(blockstore_config_t & config, bool init); void parse_config(blockstore_config_t & config, bool init);
// Reshard database for a pool
void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit);
bool reshard_continue(void *reshard_state, uint64_t chunk_limit);
// Event loop // Event loop
void loop(); void loop();
+1 -1
View File
@@ -189,7 +189,7 @@ resume_1:
printf( printf(
"Configuration stored in metadata superblock" "Configuration stored in metadata superblock"
" (meta_block_size=%u, data_block_size=%u, bitmap_granularity=%u, data_csum_type=%u, csum_block_size=%u)" " (meta_block_size=%u, data_block_size=%u, bitmap_granularity=%u, data_csum_type=%u, csum_block_size=%u)"
" differs from OSD configuration (%ju/%ju/%u, %u/%u).\n", " differs from OSD configuration (%u/%u/%u, %u/%u).\n",
hdr->meta_block_size, hdr->data_block_size, hdr->bitmap_granularity, hdr->meta_block_size, hdr->data_block_size, hdr->bitmap_granularity,
hdr->data_csum_type, hdr->csum_block_size, hdr->data_csum_type, hdr->csum_block_size,
bs->dsk.meta_block_size, bs->dsk.data_block_size, bs->dsk.bitmap_granularity, bs->dsk.meta_block_size, bs->dsk.data_block_size, bs->dsk.bitmap_granularity,
+2 -1
View File
@@ -620,7 +620,8 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
else if (from_journal) else if (from_journal)
{ {
// Don't scan bitmap - journal writes don't have holes (internal bitmap)! // Don't scan bitmap - journal writes don't have holes (internal bitmap)!
uint8_t *csum = !dsk.csum_block_size ? 0 : (clean_entry_bitmap + dsk.clean_entry_bitmap_size); uint8_t *csum = !dsk.csum_block_size ? 0 : (clean_entry_bitmap + dsk.clean_entry_bitmap_size +
item_start/dsk.csum_block_size*(dsk.data_csum_type & 0xFF));
if (!fulfill_read(read_op, fulfilled, item_start, item_end, if (!fulfill_read(read_op, fulfilled, item_start, item_end,
(BS_ST_BIG_WRITE | BS_ST_STABLE), 0, clean_loc + item_start, 0, csum, dyn_data)) (BS_ST_BIG_WRITE | BS_ST_STABLE), 0, clean_loc + item_start, 0, csum, dyn_data))
{ {
+1 -1
View File
@@ -183,7 +183,7 @@ bool blockstore_impl_t::enqueue_write(blockstore_op_t *op)
uint32_t end = (op->offset+op->len-1) / dsk.csum_block_size; uint32_t end = (op->offset+op->len-1) / dsk.csum_block_size;
auto fn = state & BS_ST_BIG_WRITE ? crc32c_pad : crc32c_nopad; auto fn = state & BS_ST_BIG_WRITE ? crc32c_pad : crc32c_nopad;
if (start == end) if (start == end)
data_csums[0] = fn(0, op->buf, op->len, op->offset - start*dsk.csum_block_size, (end+1)*dsk.csum_block_size - (op->offset+op->len)); data_csums[0] = fn(0, op->buf, op->len, op->offset - start*dsk.csum_block_size, end*dsk.csum_block_size - (op->offset+op->len));
else else
{ {
// First block // First block
+3 -6
View File
@@ -13,10 +13,10 @@ if (RDMACM_LIBRARIES)
endif (RDMACM_LIBRARIES) endif (RDMACM_LIBRARIES)
add_library(vitastor_common STATIC add_library(vitastor_common STATIC
../util/epoll_manager.cpp etcd_state_client.cpp messenger.cpp ../util/addr_util.cpp ../util/epoll_manager.cpp etcd_state_client.cpp messenger.cpp ../util/addr_util.cpp
msgr_encrypt.cpp msgr_stop.cpp msgr_op.cpp msgr_send.cpp msgr_receive.cpp ../util/ringloop.cpp ../../json11/json11.cpp msgr_stop.cpp msgr_op.cpp msgr_send.cpp msgr_receive.cpp ../util/ringloop.cpp ../../json11/json11.cpp
http_client.cpp osd_ops.cpp pg_states.cpp ../util/timerfd_manager.cpp ../util/str_util.cpp ../util/json_util.cpp ${MSGR_RDMA} ${MSGR_RDMACM} http_client.cpp osd_ops.cpp pg_states.cpp ../util/timerfd_manager.cpp ../util/str_util.cpp ../util/json_util.cpp ${MSGR_RDMA} ${MSGR_RDMACM}
) )
target_link_libraries(vitastor_common pthread ${OPENSSL_LIBRARIES} ${CARES_LIBRARIES}) target_link_libraries(vitastor_common pthread)
target_compile_options(vitastor_common PUBLIC -fPIC) target_compile_options(vitastor_common PUBLIC -fPIC)
# libvitastor_client.so # libvitastor_client.so
@@ -24,7 +24,6 @@ add_library(vitastor_client SHARED
cluster_client.cpp cluster_client.cpp
cluster_client_list.cpp cluster_client_list.cpp
cluster_client_wb.cpp cluster_client_wb.cpp
cluster_client_icache.cpp
vitastor_c.cpp vitastor_c.cpp
) )
set_target_properties(vitastor_client PROPERTIES PUBLIC_HEADER "client/vitastor_c.h") set_target_properties(vitastor_client PROPERTIES PUBLIC_HEADER "client/vitastor_c.h")
@@ -34,7 +33,6 @@ target_link_libraries(vitastor_client
${LIBURING_LIBRARIES} ${LIBURING_LIBRARIES}
${IBVERBS_LIBRARIES} ${IBVERBS_LIBRARIES}
${RDMACM_LIBRARIES} ${RDMACM_LIBRARIES}
${OPENSSL_LIBRARIES}
) )
set_target_properties(vitastor_client PROPERTIES VERSION ${VITASTOR_VERSION} SOVERSION 0) set_target_properties(vitastor_client PROPERTIES VERSION ${VITASTOR_VERSION} SOVERSION 0)
configure_file(vitastor.pc.in vitastor.pc @ONLY) configure_file(vitastor.pc.in vitastor.pc @ONLY)
@@ -100,10 +98,9 @@ endif (${WITH_QEMU})
add_executable(test_cluster_client add_executable(test_cluster_client
EXCLUDE_FROM_ALL EXCLUDE_FROM_ALL
../test/test_cluster_client.cpp ../test/test_cluster_client.cpp
pg_states.cpp osd_ops.cpp cluster_client.cpp cluster_client_list.cpp cluster_client_wb.cpp cluster_client_icache.cpp msgr_op.cpp ../test/mock/messenger.cpp msgr_stop.cpp msgr_encrypt.cpp pg_states.cpp osd_ops.cpp cluster_client.cpp cluster_client_list.cpp cluster_client_wb.cpp msgr_op.cpp ../test/mock/messenger.cpp msgr_stop.cpp
etcd_state_client.cpp ../util/timerfd_manager.cpp ../util/addr_util.cpp ../util/str_util.cpp ../util/json_util.cpp ../../json11/json11.cpp etcd_state_client.cpp ../util/timerfd_manager.cpp ../util/addr_util.cpp ../util/str_util.cpp ../util/json_util.cpp ../../json11/json11.cpp
) )
target_link_libraries(test_cluster_client ${OPENSSL_LIBRARIES})
target_compile_definitions(test_cluster_client PUBLIC -D__MOCK__) target_compile_definitions(test_cluster_client PUBLIC -D__MOCK__)
target_include_directories(test_cluster_client BEFORE PUBLIC ${CMAKE_SOURCE_DIR}/src/test/mock) target_include_directories(test_cluster_client BEFORE PUBLIC ${CMAKE_SOURCE_DIR}/src/test/mock)
add_dependencies(build_tests test_cluster_client) add_dependencies(build_tests test_cluster_client)
+53 -116
View File
@@ -62,7 +62,6 @@ cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd
st_cli.on_change_node_placement_hook = [this]() { on_change_node_placement_hook(); }; st_cli.on_change_node_placement_hook = [this]() { on_change_node_placement_hook(); };
st_cli.on_load_pgs_hook = [this](bool success) { on_load_pgs_hook(success); }; st_cli.on_load_pgs_hook = [this](bool success) { on_load_pgs_hook(success); };
st_cli.on_reload_hook = [this]() { st_cli.load_global_config(); }; st_cli.on_reload_hook = [this]() { st_cli.load_global_config(); };
st_cli.on_inode_change_hook = [this](uint64_t inode, bool removed) { on_change_inode_hook(inode, removed); };
st_cli.parse_config(config); st_cli.parse_config(config);
st_cli.infinite_start = false; st_cli.infinite_start = false;
@@ -78,7 +77,6 @@ cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd
cluster_client_t::~cluster_client_t() cluster_client_t::~cluster_client_t()
{ {
vault_destroy();
if (retry_timeout_id >= 0) if (retry_timeout_id >= 0)
{ {
tfd->clear_timer(retry_timeout_id); tfd->clear_timer(retry_timeout_id);
@@ -483,8 +481,6 @@ void cluster_client_t::on_load_config_hook(json11::Json::object & etcd_global_co
self_tree_metrics.clear(); self_tree_metrics.clear();
client_hostname = new_hostname; client_hostname = new_hostname;
} }
// vault
vault_parse_config();
msgr.parse_config(config); msgr.parse_config(config);
st_cli.parse_config(config); st_cli.parse_config(config);
st_cli.load_pgs(); st_cli.load_pgs();
@@ -611,9 +607,6 @@ void cluster_client_t::on_change_pool_config_hook()
pg_counts[pool_item.first] = pool_item.second.real_pg_count; pg_counts[pool_item.first] = pool_item.second.real_pg_count;
} }
} }
inode_cache.clear();
inode_cache_children.clear();
vault_keys.clear();
continue_ops(); continue_ops();
} }
@@ -680,10 +673,6 @@ bool cluster_client_t::flush()
{ {
if (!ringloop) if (!ringloop)
{ {
if (vault_loading)
{
return false;
}
if (wb->writeback_queue.size()) if (wb->writeback_queue.size())
{ {
wb->start_writebacks(this, 0); wb->start_writebacks(this, 0);
@@ -706,7 +695,7 @@ bool cluster_client_t::flush()
sync_done = true; sync_done = true;
}; };
execute(sync); execute(sync);
while (!sync_done || vault_loading) while (!sync_done)
{ {
ringloop->loop(); ringloop->loop();
if (!sync_done) if (!sync_done)
@@ -879,8 +868,7 @@ void cluster_client_t::execute_cas(cluster_op_t *op)
{ {
int expected = part->req.hdr.opcode == OSD_OP_DELETE ? 0 : part->req.rw.len; int expected = part->req.hdr.opcode == OSD_OP_DELETE ? 0 : part->req.rw.len;
op->retval = part->reply.hdr.retval; op->retval = part->reply.hdr.retval;
if (op->retval != expected && op->retval >= 0) op->retval = op->retval == expected ? 0 : (op->retval >= 0 ? -EIO : op->retval);
op->retval = -EIO;
op->retval = op->retval == -EPIPE ? -EINTR : op->retval; op->retval = op->retval == -EPIPE ? -EINTR : op->retval;
auto peer_it = msgr.osd_peer_fds.find(op->parts[0].osd_num); auto peer_it = msgr.osd_peer_fds.find(op->parts[0].osd_num);
if (op->retval != 0 || (op->flags & OP_IMMEDIATE_COMMIT)) if (op->retval != 0 || (op->flags & OP_IMMEDIATE_COMMIT))
@@ -909,13 +897,10 @@ void cluster_client_t::execute_cas(cluster_op_t *op)
.opcode = OSD_OP_SYNC, .opcode = OSD_OP_SYNC,
}, },
}, },
.callback = [op](osd_op_t *part) .callback = [this, op](osd_op_t *part)
{ {
if (part->reply.hdr.retval != 0) op->retval = part->reply.hdr.retval;
{ op->retval = op->retval == -EPIPE ? -EINTR : op->retval;
op->retval = part->reply.hdr.retval;
op->retval = op->retval == -EPIPE ? -EINTR : op->retval;
}
auto cb = std::move(op->callback); auto cb = std::move(op->callback);
cb(op); cb(op);
}, },
@@ -969,40 +954,10 @@ bool cluster_client_t::check_rw(cluster_op_t *op)
{ {
op->flags |= OP_IMMEDIATE_COMMIT; op->flags |= OP_IMMEDIATE_COMMIT;
} }
bool searched = false;
std::shared_ptr<inode_cache_t> icache;
if (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_WRITE)
{
if (!searched)
{
icache = inode_cache_get(op->inode);
searched = true;
}
if (icache && icache->has_parent_loop && op->opcode == OSD_OP_READ)
{
op->retval = -EINVAL;
auto cb = std::move(op->callback);
cb(op);
return false;
}
if (icache && icache->op_enc)
{
// Use shared_ptr aliasing to attach op_enc to the inode cache entry
op->enc = std::shared_ptr<osd_op_enc_t>(icache, icache->op_enc);
}
else
op->enc.reset();
}
else
op->enc.reset();
if ((op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) && !(op->flags & OSD_OP_IGNORE_READONLY)) if ((op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) && !(op->flags & OSD_OP_IGNORE_READONLY))
{ {
if (!searched) auto ino_it = st_cli.inode_config.find(op->inode);
{ if (ino_it != st_cli.inode_config.end() && ino_it->second.readonly)
icache = inode_cache_get(op->inode);
searched = true;
}
if (icache && icache->readonly)
{ {
op->retval = -EROFS; op->retval = -EROFS;
auto cb = std::move(op->callback); auto cb = std::move(op->callback);
@@ -1013,39 +968,33 @@ bool cluster_client_t::check_rw(cluster_op_t *op)
op->deoptimise_snapshot = false; op->deoptimise_snapshot = false;
if (enable_writeback && (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP)) if (enable_writeback && (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP))
{ {
if (!searched) auto ino_it = st_cli.inode_config.find(op->inode);
if (ino_it != st_cli.inode_config.end())
{ {
icache = inode_cache_get(op->inode); int chain_size = 0;
searched = true; while (ino_it != st_cli.inode_config.end() && ino_it->second.parent_id)
}
if (icache)
{
for (auto & parent: icache->chain)
{ {
if (INODE_POOL(parent) == INODE_POOL(op->inode) && wb->has_inode(parent)) // Check for loops - FIXME check it in etcd_state_client
if (ino_it->second.parent_id == op->inode ||
chain_size > st_cli.inode_config.size())
{
op->retval = -EINVAL;
auto cb = std::move(op->callback);
cb(op);
return false;
}
if (INODE_POOL(ino_it->second.parent_id) == INODE_POOL(ino_it->first) &&
wb->has_inode(ino_it->second.parent_id))
{ {
// Deoptimise reads - we have dirty data for one of the parent layer(s). // Deoptimise reads - we have dirty data for one of the parent layer(s).
op->deoptimise_snapshot = true; op->deoptimise_snapshot = true;
break; break;
} }
chain_size++;
ino_it = st_cli.inode_config.find(ino_it->second.parent_id);
} }
} }
} }
if (icache && icache->err_code)
{
if (icache->err_code == EPERM)
{
op->retval = -EPERM;
auto cb = std::move(op->callback);
cb(op);
return false;
}
else if (icache->err_code == EAGAIN)
{
key_wait_ops.push_back(op);
return false;
}
}
return true; return true;
} }
@@ -1168,33 +1117,31 @@ resume_2:
// because if some operations were invalid for the new PG count we'd get errors // because if some operations were invalid for the new PG count we'd get errors
if (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_CHAIN_BITMAP) if (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_CHAIN_BITMAP)
{ {
uint64_t next_inode = 0; // Check parent inode
auto icache = inode_cache_get(op->cur_inode); auto ino_it = st_cli.inode_config.find(op->cur_inode);
if (icache) // Skip parents from the same pool
int skipped = 0;
while (!op->deoptimise_snapshot &&
ino_it != st_cli.inode_config.end() && ino_it->second.parent_id &&
INODE_POOL(ino_it->second.parent_id) == INODE_POOL(op->cur_inode))
{ {
if (icache->has_parent_loop) // Check for loops - FIXME check it in etcd_state_client
if (ino_it->second.parent_id == op->inode ||
skipped > st_cli.inode_config.size())
{ {
op->retval = -EINVAL; op->retval = -EINVAL;
erase_op(op); erase_op(op);
return 1; return 1;
} }
if (op->deoptimise_snapshot) skipped++;
{ ino_it = st_cli.inode_config.find(ino_it->second.parent_id);
if (icache->chain.size() > 1)
next_inode = icache->chain[1];
}
else
{
if (icache->other_pool_parent_id)
next_inode = icache->other_pool_parent_id;
}
} }
if (next_inode) if (ino_it != st_cli.inode_config.end() &&
ino_it->second.parent_id &&
ino_it->second.parent_id != op->inode)
{ {
// Continue reading from the parent inode // Continue reading from the parent inode
icache = inode_cache_get(next_inode); op->cur_inode = ino_it->second.parent_id;
op->cur_inode = next_inode;
op->enc = (icache && icache->op_enc ? std::shared_ptr<osd_op_enc_t>(icache, icache->op_enc) : nullptr);
op->parts.clear(); op->parts.clear();
op->done_count = 0; op->done_count = 0;
goto resume_0; goto resume_0;
@@ -1301,18 +1248,14 @@ void cluster_client_t::slice_rw(cluster_op_t *op)
// Allocate memory for the bitmap // Allocate memory for the bitmap
unsigned object_bitmap_size = ((op->len / pool_cfg.bitmap_granularity + 7) / 8); unsigned object_bitmap_size = ((op->len / pool_cfg.bitmap_granularity + 7) / 8);
object_bitmap_size = (object_bitmap_size < 8 ? 8 : object_bitmap_size); object_bitmap_size = (object_bitmap_size < 8 ? 8 : object_bitmap_size);
unsigned bitmap_mem = object_bitmap_size + unsigned bitmap_mem = object_bitmap_size + (pool_cfg.data_block_size / pool_cfg.bitmap_granularity / 8 * pg_data_size) * op->parts.size();
op->parts.size() * pg_data_size *
(pool_cfg.data_block_size / pool_cfg.bitmap_granularity / 8
// read chain info - 1 byte per block
+ (op->enc ? op->len/pool_cfg.bitmap_granularity : 0));
if (!op->bitmap_buf || op->bitmap_buf_size < bitmap_mem) if (!op->bitmap_buf || op->bitmap_buf_size < bitmap_mem)
{ {
op->bitmap_buf = realloc_or_die(op->bitmap_buf, bitmap_mem); op->bitmap_buf = realloc_or_die(op->bitmap_buf, bitmap_mem);
op->part_bitmaps = (uint8_t*)op->bitmap_buf + object_bitmap_size; op->part_bitmaps = (uint8_t*)op->bitmap_buf + object_bitmap_size;
memset((uint8_t*)op->bitmap_buf+op->bitmap_buf_size, 0, bitmap_mem-op->bitmap_buf_size);
op->bitmap_buf_size = bitmap_mem; op->bitmap_buf_size = bitmap_mem;
} }
memset(op->bitmap_buf, 0, bitmap_mem);
} }
int iov_idx = 0; int iov_idx = 0;
size_t iov_pos = 0; size_t iov_pos = 0;
@@ -1458,11 +1401,11 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
if (peer_it != msgr.osd_peer_fds.end()) if (peer_it != msgr.osd_peer_fds.end())
{ {
int peer_fd = peer_it->second; int peer_fd = peer_it->second;
part->flags |= PART_SENT|PART_VALID; part->flags |= PART_SENT;
op->inflight_count++; op->inflight_count++;
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks); uint64_t pg_bitmap_size = (pool_cfg.data_block_size / pool_cfg.bitmap_granularity / 8) * (
uint64_t pg_bitmap_size = pg_data_size * (pool_cfg.data_block_size / pool_cfg.bitmap_granularity / 8 pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks
+ (op->opcode == OSD_OP_READ && op->enc ? pool_cfg.data_block_size/pool_cfg.bitmap_granularity : 0)); );
uint64_t meta_rev = 0; uint64_t meta_rev = 0;
if (op->opcode != OSD_OP_READ_BITMAP && op->opcode != OSD_OP_DELETE && !op->deoptimise_snapshot) if (op->opcode != OSD_OP_READ_BITMAP && op->opcode != OSD_OP_DELETE && !op->deoptimise_snapshot)
{ {
@@ -1481,7 +1424,6 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
.inode = op->cur_inode, .inode = op->cur_inode,
.offset = part->offset, .offset = part->offset,
.len = part->len, .len = part->len,
.flags = op->opcode == OSD_OP_READ && op->enc && !op->deoptimise_snapshot ? OSD_OP_RETURN_CHAIN : 0,
.meta_revision = meta_rev, .meta_revision = meta_rev,
.version = op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE ? op->version : 0, .version = op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE ? op->version : 0,
} }, } },
@@ -1489,7 +1431,6 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
? (uint8_t*)op->part_bitmaps + pg_bitmap_size*i : NULL), ? (uint8_t*)op->part_bitmaps + pg_bitmap_size*i : NULL),
.bitmap_len = (unsigned)(op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP .bitmap_len = (unsigned)(op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP
? pg_bitmap_size : 0), ? pg_bitmap_size : 0),
.enc = op->enc,
.callback = cb ? cb : [this, part](osd_op_t *op_part) .callback = cb ? cb : [this, part](osd_op_t *op_part)
{ {
handle_op_part(part); handle_op_part(part);
@@ -1673,11 +1614,14 @@ void cluster_client_t::handle_op_part(cluster_op_part_t *part)
dirty_osds.insert(part->osd_num); dirty_osds.insert(part->osd_num);
part->flags |= PART_DONE; part->flags |= PART_DONE;
op->done_count++; op->done_count++;
if ((op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP) if (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP)
&& op->inode == op->cur_inode)
{ {
// Read only returns the version of the uppermost layer copy_part_bitmap(op, part);
op->version = op->parts.size() == 1 ? part->op.reply.rw.version : 0; if (op->inode == op->cur_inode)
{
// Read only returns the version of the uppermost layer
op->version = op->parts.size() == 1 ? part->op.reply.rw.version : 0;
}
} }
else if (op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE) else if (op->opcode == OSD_OP_WRITE || op->opcode == OSD_OP_DELETE)
{ {
@@ -1685,13 +1629,6 @@ void cluster_client_t::handle_op_part(cluster_op_part_t *part)
} }
if (op->inflight_count == 0 && !op->retry_after) if (op->inflight_count == 0 && !op->retry_after)
{ {
// Copy part bitmaps only after finishing all part reads
if (op->opcode == OSD_OP_READ || op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP)
{
for (auto & part: op->parts)
if (part.flags == (PART_SENT|PART_VALID|PART_DONE))
copy_part_bitmap(op, &part);
}
if (op->opcode == OSD_OP_SYNC) if (op->opcode == OSD_OP_SYNC)
continue_sync(op); continue_sync(op);
else else
+2 -59
View File
@@ -5,7 +5,6 @@
#include "messenger.h" #include "messenger.h"
#include "etcd_state_client.h" #include "etcd_state_client.h"
#include "../util/robin_hood.h"
#define DEFAULT_CLIENT_MAX_DIRTY_BYTES 32*1024*1024 #define DEFAULT_CLIENT_MAX_DIRTY_BYTES 32*1024*1024
#define DEFAULT_CLIENT_MAX_DIRTY_OPS 1024 #define DEFAULT_CLIENT_MAX_DIRTY_OPS 1024
@@ -72,7 +71,6 @@ protected:
cluster_op_t *prev = NULL, *next = NULL; cluster_op_t *prev = NULL, *next = NULL;
int prev_wait = 0; int prev_wait = 0;
uint64_t flush_id = 0; uint64_t flush_id = 0;
std::shared_ptr<osd_op_enc_t> enc;
friend class cluster_client_t; friend class cluster_client_t;
friend class writeback_cache_t; friend class writeback_cache_t;
}; };
@@ -82,25 +80,6 @@ struct inode_list_osd_t;
struct inode_list_pg_t; struct inode_list_pg_t;
class writeback_cache_t; class writeback_cache_t;
struct inode_cache_t
{
std::vector<inode_t> chain;
uint8_t *key_data = NULL;
osd_op_enc_t *op_enc = NULL;
bool readonly = false;
bool has_parent_loop = false;
inode_t other_pool_parent_id = 0;
int err_code = 0;
~inode_cache_t();
};
struct vault_load_key_t
{
int key_state = 0;
std::string key;
};
// FIXME: Split into public and private interfaces // FIXME: Split into public and private interfaces
class __attribute__((visibility("default"))) cluster_client_t class __attribute__((visibility("default"))) cluster_client_t
{ {
@@ -110,8 +89,8 @@ public:
timerfd_manager_t *tfd = NULL; timerfd_manager_t *tfd = NULL;
ring_loop_t *ringloop = NULL; ring_loop_t *ringloop = NULL;
// config: std::map<pool_id_t, uint64_t> pg_counts;
std::map<pool_pg_num_t, osd_num_t> pg_primary;
// client_max_dirty_* is actually "max unsynced", for the case when immediate_commit is off // client_max_dirty_* is actually "max unsynced", for the case when immediate_commit is off
uint64_t client_max_dirty_bytes = 0; uint64_t client_max_dirty_bytes = 0;
uint64_t client_max_dirty_ops = 0; uint64_t client_max_dirty_ops = 0;
@@ -123,23 +102,12 @@ public:
uint64_t client_max_writeback_iodepth = 0; uint64_t client_max_writeback_iodepth = 0;
std::string conf_hostname; std::string conf_hostname;
std::string vault_url;
std::string vault_client_cert;
std::string vault_client_key;
std::string vault_ca;
std::string vault_secret_api_path;
uint64_t vault_timeout_ms = 0;
uint64_t vault_error_timeout_sec = 0;
uint64_t vault_refresh_leeway_sec = 0;
int log_level = 0; int log_level = 0;
int client_retry_interval = 50; // ms int client_retry_interval = 50; // ms
int client_eio_retry_interval = 1000; // ms int client_eio_retry_interval = 1000; // ms
bool client_retry_enospc = true; bool client_retry_enospc = true;
int client_wait_up_timeout = 16; // sec (for listings) int client_wait_up_timeout = 16; // sec (for listings)
// state:
std::string client_hostname; std::string client_hostname;
std::map<std::string, int> self_tree_metrics; std::map<std::string, int> self_tree_metrics;
std::map<osd_num_t, int> osd_tree_metrics; std::map<osd_num_t, int> osd_tree_metrics;
@@ -147,7 +115,6 @@ public:
int retry_timeout_id = -1; int retry_timeout_id = -1;
int retry_timeout_duration = 0; int retry_timeout_duration = 0;
std::vector<cluster_op_t*> offline_ops; std::vector<cluster_op_t*> offline_ops;
std::vector<cluster_op_t*> key_wait_ops;
cluster_op_t *op_queue_head = NULL, *op_queue_tail = NULL; cluster_op_t *op_queue_head = NULL, *op_queue_tail = NULL;
writeback_cache_t *wb = NULL; writeback_cache_t *wb = NULL;
std::set<osd_num_t> dirty_osds; std::set<osd_num_t> dirty_osds;
@@ -156,22 +123,7 @@ public:
void *scrap_buffer = NULL; void *scrap_buffer = NULL;
unsigned scrap_buffer_size = 0; unsigned scrap_buffer_size = 0;
// inodes require some extra state for read/write, it's stored here.
// moreover, robin_hood access is slightly faster than std::map :)
robin_hood::unordered_flat_map<inode_t, std::shared_ptr<inode_cache_t>> inode_cache;
std::set<std::pair<inode_t, inode_t>> inode_cache_children;
http_context_t *vault_http_ctx = NULL;
http_co_t *vault_http_cli = NULL;
bool vault_loading = false;
std::string vault_token;
bool vault_auth_error = false;
timespec vault_token_expire = {};
std::vector<std::string> vault_key_load_queue;
std::map<std::string, vault_load_key_t> vault_keys;
bool pgs_loaded = false; bool pgs_loaded = false;
std::map<pool_id_t, uint64_t> pg_counts;
ring_consumer_t consumer; ring_consumer_t consumer;
std::vector<std::function<void(void)>> on_ready_hooks; std::vector<std::function<void(void)>> on_ready_hooks;
int list_retry_timeout_id = -1; int list_retry_timeout_id = -1;
@@ -211,13 +163,6 @@ protected:
#endif #endif
void continue_ops(int time_passed = 0); void continue_ops(int time_passed = 0);
std::shared_ptr<inode_cache_t> inode_cache_get(inode_t ino);
void vault_parse_config();
bool vault_check_token();
void vault_load_keys();
void vault_destroy();
void vault_parse_secret(const std::string & key_id, const std::string & err, json11::Json data);
protected: protected:
bool affects_osd(uint64_t inode, uint64_t offset, uint64_t len, osd_num_t osd); bool affects_osd(uint64_t inode, uint64_t offset, uint64_t len, osd_num_t osd);
bool affects_pg(uint64_t inode, uint64_t offset, uint64_t len, pool_id_t pool_id, pg_num_t pg_num); bool affects_pg(uint64_t inode, uint64_t offset, uint64_t len, pool_id_t pool_id, pg_num_t pg_num);
@@ -228,7 +173,6 @@ protected:
void on_change_pg_state_hook(pool_id_t pool_id, pg_num_t pg_num, osd_num_t prev_primary); void on_change_pg_state_hook(pool_id_t pool_id, pg_num_t pg_num, osd_num_t prev_primary);
void on_change_osd_state_hook(uint64_t peer_osd); void on_change_osd_state_hook(uint64_t peer_osd);
void on_change_node_placement_hook(); void on_change_node_placement_hook();
void on_change_inode_hook(uint64_t inode, bool removed);
void execute_internal(cluster_op_t *op); void execute_internal(cluster_op_t *op);
void execute_cas(cluster_op_t *op); void execute_cas(cluster_op_t *op);
@@ -245,7 +189,6 @@ protected:
void erase_op(cluster_op_t *op); void erase_op(cluster_op_t *op);
void calc_wait(cluster_op_t *op); void calc_wait(cluster_op_t *op);
void inc_wait(uint64_t opcode, uint64_t flags, cluster_op_t *next, int inc); void inc_wait(uint64_t opcode, uint64_t flags, cluster_op_t *next, int inc);
void continue_lists(); void continue_lists();
bool continue_listing(inode_list_t *lst); bool continue_listing(inode_list_t *lst);
bool restart_listing(inode_list_t* lst); bool restart_listing(inode_list_t* lst);

Some files were not shown because too many files have changed in this diff Show More