Compare commits
32
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
538620b400 | ||
|
|
4c34a47179 | ||
|
|
df16ab627a | ||
|
|
44c895dc30 | ||
|
|
fdea595913 | ||
|
|
ef4c91ecc8 | ||
|
|
5192c6cf50 | ||
|
|
cb639a130d | ||
|
|
892e3a8b6d | ||
|
|
0bd9c26620 | ||
|
|
ae2b1f7802 | ||
|
|
883b2e45da | ||
|
|
09df74cfac | ||
|
|
2ef3f012c9 | ||
|
|
7175d99c64 | ||
|
|
230a26772e | ||
|
|
637684c579 | ||
|
|
c541dd422f | ||
|
|
22c39561cf | ||
|
|
d4c67b4879 | ||
|
|
d1b7167861 | ||
|
|
33b561b73a | ||
|
|
a95a60c600 | ||
|
|
b3bc815354 | ||
|
|
42fc45d6da | ||
|
|
3fbe2e7f8a | ||
|
|
310c512b43 | ||
|
|
fd3e3b4ef0 | ||
|
|
93cf69f89f | ||
|
|
a8dd8bf06c | ||
|
|
7e23b57014 | ||
|
|
b2427836a1 |
@@ -1,28 +1,29 @@
|
|||||||
FROM node:16-bullseye
|
FROM node:16-bookworm
|
||||||
|
|
||||||
WORKDIR /root
|
WORKDIR /root
|
||||||
|
|
||||||
ADD ./docker/vitastor.gpg /etc/apt/trusted.gpg.d
|
ADD ./docker/etc/apt/trusted.gpg.d /etc/apt/trusted.gpg.d
|
||||||
|
|
||||||
RUN echo 'deb http://deb.debian.org/debian bullseye-backports main' >> /etc/apt/sources.list; \
|
RUN echo 'deb http://deb.debian.org/debian bookworm-backports main' >> /etc/apt/sources.list; \
|
||||||
echo 'deb http://vitastor.io/debian bullseye main' >> /etc/apt/sources.list; \
|
echo 'deb http://vitastor.io/debian bookworm main' >> /etc/apt/sources.list; \
|
||||||
echo >> /etc/apt/preferences; \
|
echo >> /etc/apt/preferences; \
|
||||||
echo 'Package: *' >> /etc/apt/preferences; \
|
echo 'Package: *' >> /etc/apt/preferences; \
|
||||||
echo 'Pin: release a=bullseye-backports' >> /etc/apt/preferences; \
|
echo 'Pin: release n=bookworm-backports' >> /etc/apt/preferences; \
|
||||||
echo 'Pin-Priority: 500' >> /etc/apt/preferences; \
|
echo 'Pin-Priority: 500' >> /etc/apt/preferences; \
|
||||||
echo >> /etc/apt/preferences; \
|
echo >> /etc/apt/preferences; \
|
||||||
echo 'Package: *' >> /etc/apt/preferences; \
|
echo 'Package: *' >> /etc/apt/preferences; \
|
||||||
echo 'Pin: origin "vitastor.io"' >> /etc/apt/preferences; \
|
echo 'Pin: origin "vitastor.io"' >> /etc/apt/preferences; \
|
||||||
echo 'Pin-Priority: 1000' >> /etc/apt/preferences; \
|
echo 'Pin-Priority: 1000' >> /etc/apt/preferences; \
|
||||||
|
perl -i -pe 's/Types: deb$/Types: deb deb-src/' /etc/apt/sources.list.d/debian.sources; \
|
||||||
grep '^deb ' /etc/apt/sources.list | perl -pe 's/^deb/deb-src/' >> /etc/apt/sources.list; \
|
grep '^deb ' /etc/apt/sources.list | perl -pe 's/^deb/deb-src/' >> /etc/apt/sources.list; \
|
||||||
echo 'APT::Install-Recommends false;' >> /etc/apt/apt.conf; \
|
echo 'APT::Install-Recommends false;' >> /etc/apt/apt.conf; \
|
||||||
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
|
echo 'APT::Install-Suggests false;' >> /etc/apt/apt.conf
|
||||||
|
|
||||||
RUN apt-get update
|
RUN apt-get update
|
||||||
RUN apt-get -y install etcd qemu-system-x86 qemu-block-extra qemu-utils fio libasan5 \
|
RUN apt-get -y install etcd qemu-system-x86 qemu-block-extra qemu-utils fio libasan8 \
|
||||||
libgoogle-perftools-dev devscripts libjerasure-dev cmake libibverbs-dev libisal-dev
|
libgoogle-perftools-dev devscripts libjerasure-dev cmake libibverbs-dev libisal-dev
|
||||||
RUN apt-get -y build-dep fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
RUN apt-get -y build-dep fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
||||||
RUN apt-get update && apt-get -y install jq lp-solve sudo nfs-common fdisk parted
|
RUN apt-get update && apt-get -y install jq lp-solve sudo nfs-common fdisk parted libc-ares-dev udev
|
||||||
RUN apt-get --download-only source fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
RUN apt-get --download-only source fio qemu=`dpkg -s qemu-system-x86|grep ^Version:|awk '{print $2}'`
|
||||||
|
|
||||||
RUN set -ex; \
|
RUN set -ex; \
|
||||||
|
|||||||
+1
-1
@@ -2,7 +2,7 @@ cmake_minimum_required(VERSION 2.8.12)
|
|||||||
|
|
||||||
project(vitastor)
|
project(vitastor)
|
||||||
|
|
||||||
set(VITASTOR_VERSION "3.0.3")
|
set(VITASTOR_VERSION "3.0.6")
|
||||||
|
|
||||||
include(CTest)
|
include(CTest)
|
||||||
|
|
||||||
|
|||||||
+2
-2
@@ -25,8 +25,8 @@ RUN apt-get update && \
|
|||||||
# NFS mount dependencies
|
# NFS mount dependencies
|
||||||
nfs-common netbase \
|
nfs-common netbase \
|
||||||
# dependencies of qemu-storage-daemon
|
# dependencies of qemu-storage-daemon
|
||||||
libnuma1 liburing2 libglib2.0-0 libfuse3-3 libaio1 libzstd1 libnettle8 \
|
libaio1t64 libc6 libfuse3-4 libglib2.0-0t64 libgmp10 libgnutls30t64 \
|
||||||
libgmp10 libhogweed6 libp11-kit0 libidn2-0 libunistring2 libtasn1-6 libpcre2-8-0 libffi8 && \
|
libhogweed6t64 libnettle8t64 libnuma1 libselinux1 liburing2 libzstd1 zlib1g && \
|
||||||
apt-get clean && \
|
apt-get clean && \
|
||||||
(echo options nbd nbds_max=128 > /etc/modprobe.d/nbd.conf)
|
(echo options nbd nbds_max=128 > /etc/modprobe.d/nbd.conf)
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
# Compile stage
|
# Compile stage
|
||||||
FROM golang:bookworm AS build
|
FROM golang:trixie AS build
|
||||||
|
|
||||||
ADD go.sum go.mod /app/
|
ADD go.sum go.mod /app/
|
||||||
RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go mod download -x
|
RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go mod download -x
|
||||||
@@ -9,7 +9,7 @@ RUN perl -i -e '$/ = undef; while(<>) { s/\n\s*(\{\s*\n)/$1\n/g; s/\}(\s*\n\s*)e
|
|||||||
CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o vitastor-csi
|
CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o vitastor-csi
|
||||||
|
|
||||||
# Final stage
|
# Final stage
|
||||||
FROM debian:bookworm
|
FROM debian:trixie
|
||||||
|
|
||||||
LABEL maintainers="Vitaliy Filippov <vitalif@yourcmc.ru>"
|
LABEL maintainers="Vitaliy Filippov <vitalif@yourcmc.ru>"
|
||||||
LABEL description="Vitastor CSI Driver"
|
LABEL description="Vitastor CSI Driver"
|
||||||
@@ -36,8 +36,8 @@ ADD deb /deb
|
|||||||
|
|
||||||
RUN apt-get update && \
|
RUN apt-get update && \
|
||||||
apt-get -y install /deb/vitastor-client_*.deb && \
|
apt-get -y install /deb/vitastor-client_*.deb && \
|
||||||
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
wget https://vitastor.io/archive/qemu/qemu-trixie-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||||
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
wget https://vitastor.io/archive/qemu/qemu-trixie-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||||
dpkg -x qemu-utils*.deb tmp1 && \
|
dpkg -x qemu-utils*.deb tmp1 && \
|
||||||
dpkg -x qemu-block-extra*.deb tmp1 && \
|
dpkg -x qemu-block-extra*.deb tmp1 && \
|
||||||
cp -a tmp1/usr/bin/qemu-storage-daemon /usr/bin/ && \
|
cp -a tmp1/usr/bin/qemu-storage-daemon /usr/bin/ && \
|
||||||
|
|||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
VITASTOR_VERSION ?= v3.0.3
|
VITASTOR_VERSION ?= v3.0.6
|
||||||
|
|
||||||
all: build push
|
all: build push
|
||||||
|
|
||||||
|
|||||||
@@ -49,7 +49,7 @@ spec:
|
|||||||
capabilities:
|
capabilities:
|
||||||
add: ["SYS_ADMIN"]
|
add: ["SYS_ADMIN"]
|
||||||
allowPrivilegeEscalation: true
|
allowPrivilegeEscalation: true
|
||||||
image: vitalif/vitastor-csi:v3.0.3
|
image: vitalif/vitastor-csi:v3.0.6
|
||||||
args:
|
args:
|
||||||
- "--node=$(NODE_ID)"
|
- "--node=$(NODE_ID)"
|
||||||
- "--endpoint=$(CSI_ENDPOINT)"
|
- "--endpoint=$(CSI_ENDPOINT)"
|
||||||
|
|||||||
@@ -121,7 +121,7 @@ spec:
|
|||||||
privileged: true
|
privileged: true
|
||||||
capabilities:
|
capabilities:
|
||||||
add: ["SYS_ADMIN"]
|
add: ["SYS_ADMIN"]
|
||||||
image: vitalif/vitastor-csi:v3.0.3
|
image: vitalif/vitastor-csi:v3.0.6
|
||||||
args:
|
args:
|
||||||
- "--node=$(NODE_ID)"
|
- "--node=$(NODE_ID)"
|
||||||
- "--endpoint=$(CSI_ENDPOINT)"
|
- "--endpoint=$(CSI_ENDPOINT)"
|
||||||
|
|||||||
+1
-1
@@ -5,7 +5,7 @@ package vitastor
|
|||||||
|
|
||||||
const (
|
const (
|
||||||
vitastorCSIDriverName = "csi.vitastor.io"
|
vitastorCSIDriverName = "csi.vitastor.io"
|
||||||
vitastorCSIDriverVersion = "3.0.3"
|
vitastorCSIDriverVersion = "3.0.6"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Config struct fills the parameters of request or user input
|
// Config struct fills the parameters of request or user input
|
||||||
|
|||||||
+5
@@ -0,0 +1,5 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# 26.04 Resolute Raccoon
|
||||||
|
|
||||||
|
docker build --build-arg DISTRO=ubuntu --build-arg REL=resolute -t vitastor-buildenv:resolute -f vitastor-buildenv.Dockerfile .
|
||||||
|
docker run -it --rm -e REL=resolute -v `dirname $0`/../:/root/vitastor vitastor-buildenv:resolute /root/vitastor/debian/vitastor-build.sh
|
||||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
vitastor (3.0.3-1) unstable; urgency=medium
|
vitastor (3.0.6-1) unstable; urgency=medium
|
||||||
|
|
||||||
* Bugfixes
|
* Bugfixes
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
VITASTOR_VERSION ?= v3.0.3
|
VITASTOR_VERSION ?= v3.0.6
|
||||||
|
|
||||||
all: build push
|
all: build push
|
||||||
|
|
||||||
|
|||||||
@@ -1,3 +1,3 @@
|
|||||||
Package: *
|
Package: *
|
||||||
Pin: release n=bookworm-backports
|
Pin: release n=trixie-backports
|
||||||
Pin-Priority: 500
|
Pin-Priority: 500
|
||||||
|
|||||||
@@ -1,2 +1,2 @@
|
|||||||
deb http://vitastor.io/debian bookworm main
|
deb http://vitastor.io/debian trixie main
|
||||||
deb http://http.debian.net/debian/ bookworm-backports main
|
#deb http://http.debian.net/debian/ trixie-backports main
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ PartOf=vitastor.target
|
|||||||
[Service]
|
[Service]
|
||||||
Restart=always
|
Restart=always
|
||||||
EnvironmentFile=/etc/vitastor/docker.conf
|
EnvironmentFile=/etc/vitastor/docker.conf
|
||||||
ExecStart=bash -c 'docker run --rm -i -v /etc/vitastor:/etc/vitastor -v /dev:/dev -v /run:/run \
|
ExecStart=bash -c 'docker run --rm -i -v /etc/vitastor:/etc/vitastor -v /dev:/dev -v /run:/run -e SYSTEMD_IN_CHROOT=0 \
|
||||||
--security-opt seccomp=unconfined --privileged --pid=host --log-driver none --network host --name vitastor vitastor:$VITASTOR_VERSION \
|
--security-opt seccomp=unconfined --privileged --pid=host --log-driver none --network host --name vitastor vitastor:$VITASTOR_VERSION \
|
||||||
sleep.sh'
|
sleep.sh'
|
||||||
ExecStartPost=udevadm trigger
|
ExecStartPost=udevadm trigger
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
#
|
#
|
||||||
|
|
||||||
# Desired Vitastor version
|
# Desired Vitastor version
|
||||||
VITASTOR_VERSION=v3.0.3
|
VITASTOR_VERSION=v3.0.6
|
||||||
|
|
||||||
# Additional arguments for all containers
|
# Additional arguments for all containers
|
||||||
# For example, you may want to specify a custom logging driver here
|
# For example, you may want to specify a custom logging driver here
|
||||||
|
|||||||
+2
-3
@@ -2,8 +2,7 @@
|
|||||||
|
|
||||||
set -e
|
set -e
|
||||||
|
|
||||||
cp -urv /etc/default /host-etc/
|
cp -urv /etc/systemd/system/vitastor* /host-etc/systemd/system/
|
||||||
cp -urv /etc/systemd /host-etc/
|
cp -urv /etc/udev/rules.d /host-etc/udev/
|
||||||
cp -urv /etc/udev /host-etc/
|
|
||||||
cp -urnv /etc/vitastor /host-etc/
|
cp -urnv /etc/vitastor /host-etc/
|
||||||
cp -urnv /opt/scripts/* /host-bin/
|
cp -urnv /opt/scripts/* /host-bin/
|
||||||
|
|||||||
+27
-7
@@ -38,6 +38,7 @@ with an OSD restart or, for some of them, even without restarting by updating co
|
|||||||
- [journal_io](#journal_io)
|
- [journal_io](#journal_io)
|
||||||
- [journal_sector_buffer_count](#journal_sector_buffer_count)
|
- [journal_sector_buffer_count](#journal_sector_buffer_count)
|
||||||
- [journal_no_same_sector_overwrites](#journal_no_same_sector_overwrites)
|
- [journal_no_same_sector_overwrites](#journal_no_same_sector_overwrites)
|
||||||
|
- [skip_corrupted_meta_entries](#skip_corrupted_meta_entries)
|
||||||
- [throttle_small_writes](#throttle_small_writes)
|
- [throttle_small_writes](#throttle_small_writes)
|
||||||
- [throttle_target_iops](#throttle_target_iops)
|
- [throttle_target_iops](#throttle_target_iops)
|
||||||
- [throttle_target_mbs](#throttle_target_mbs)
|
- [throttle_target_mbs](#throttle_target_mbs)
|
||||||
@@ -279,13 +280,19 @@ Maximum number of journal flushers (see above min_flusher_count).
|
|||||||
- Type: boolean
|
- Type: boolean
|
||||||
- Default: true
|
- Default: true
|
||||||
|
|
||||||
This parameter makes Vitastor always keep metadata area of the block device
|
Only for the old store ([meta_format](layout-osd.en.md#meta_format) 2).
|
||||||
in memory. It's required for good performance because it allows to avoid
|
|
||||||
additional read-modify-write cycles during metadata modifications. Metadata
|
This parameter makes Vitastor keep a copy of metadata area in memory as it is
|
||||||
area size is currently roughly 224 MB per 1 TB of data. You can turn it off
|
on disk, in addition to the metadata database. When the option is enabled, every
|
||||||
to reduce memory usage by this value, but it will hurt performance. This
|
metadata entry is effectively stored in RAM twice. It's required for good performance
|
||||||
restriction is likely to be removed in the future along with the upgrade
|
because it allows to avoid additional read-modify-write cycles during metadata
|
||||||
of the metadata storage scheme.
|
modifications. Metadata area size with the old store is roughly 224 MB per 1 TB
|
||||||
|
of data. You can turn the option off to reduce memory usage by this value, but
|
||||||
|
it will reduce performance.
|
||||||
|
|
||||||
|
For the new store ([meta_format](layout-osd.en.md#meta_format) 3), the option
|
||||||
|
may be changed in the future to support operation without loading full metadata
|
||||||
|
database in memory.
|
||||||
|
|
||||||
## inmemory_journal
|
## inmemory_journal
|
||||||
|
|
||||||
@@ -364,6 +371,8 @@ blocks. The only situation when you should increase it to a larger value
|
|||||||
is when you enable journal_no_same_sector_overwrites. In this case set
|
is when you enable journal_no_same_sector_overwrites. In this case set
|
||||||
it to, for example, 1024.
|
it to, for example, 1024.
|
||||||
|
|
||||||
|
Not applicable to the new store ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
|
||||||
## journal_no_same_sector_overwrites
|
## journal_no_same_sector_overwrites
|
||||||
|
|
||||||
- Type: boolean
|
- Type: boolean
|
||||||
@@ -377,6 +386,17 @@ journal after writing it instead of possibly overwriting it the second time.
|
|||||||
|
|
||||||
Most (99%) other SSDs don't need this option.
|
Most (99%) other SSDs don't need this option.
|
||||||
|
|
||||||
|
Not applicable to the new store ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
|
||||||
|
## skip_corrupted_meta_entries
|
||||||
|
|
||||||
|
- Type: boolean
|
||||||
|
- Default: false
|
||||||
|
|
||||||
|
Only for the new store ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
Allow OSD to start when some metadata entries or blocks are corrupted by
|
||||||
|
skipping them. Should be only used as an emergency measure.
|
||||||
|
|
||||||
## throttle_small_writes
|
## throttle_small_writes
|
||||||
|
|
||||||
- Type: boolean
|
- Type: boolean
|
||||||
|
|||||||
+28
-7
@@ -39,6 +39,7 @@
|
|||||||
- [journal_io](#journal_io)
|
- [journal_io](#journal_io)
|
||||||
- [journal_sector_buffer_count](#journal_sector_buffer_count)
|
- [journal_sector_buffer_count](#journal_sector_buffer_count)
|
||||||
- [journal_no_same_sector_overwrites](#journal_no_same_sector_overwrites)
|
- [journal_no_same_sector_overwrites](#journal_no_same_sector_overwrites)
|
||||||
|
- [skip_corrupted_meta_entries](#skip_corrupted_meta_entries)
|
||||||
- [throttle_small_writes](#throttle_small_writes)
|
- [throttle_small_writes](#throttle_small_writes)
|
||||||
- [throttle_target_iops](#throttle_target_iops)
|
- [throttle_target_iops](#throttle_target_iops)
|
||||||
- [throttle_target_mbs](#throttle_target_mbs)
|
- [throttle_target_mbs](#throttle_target_mbs)
|
||||||
@@ -287,13 +288,19 @@ Flusher - это микро-поток (корутина), которая коп
|
|||||||
- Тип: булево (да/нет)
|
- Тип: булево (да/нет)
|
||||||
- Значение по умолчанию: true
|
- Значение по умолчанию: true
|
||||||
|
|
||||||
Данный параметр заставляет Vitastor всегда держать область метаданных диска
|
Только для старого хранилища ([meta_format](layout-osd.en.md#meta_format) 2).
|
||||||
в памяти. Это нужно, чтобы избегать дополнительных операций чтения с диска
|
|
||||||
при записи. Размер области метаданных на данный момент составляет примерно
|
Данный параметр заставляет Vitastor всегда держать копию области метаданных
|
||||||
224 МБ на 1 ТБ данных. При включении потребление памяти снизится примерно
|
в памяти в том же виде, как она лежит на диске, в дополнение к БД метаданных.
|
||||||
на эту величину, но при этом также снизится и производительность. В будущем,
|
То есть, с включённой опцией каждая запись метаданных хранится в памяти дважды.
|
||||||
после обновления схемы хранения метаданных, это ограничение, скорее всего,
|
Это нужно, чтобы избегать дополнительных операций чтения с диска при записи.
|
||||||
будет ликвидировано.
|
Размер области метаданных в старом хранилище составляет примерно 224 МБ на
|
||||||
|
1 ТБ данных. Вы можете отключить опцию, чтобы снизить потребление памяти
|
||||||
|
примерно на эту величину, но при этом также снизится и производительность.
|
||||||
|
|
||||||
|
Для нового хранилища ([meta_format](layout-osd.en.md#meta_format) 3) опция,
|
||||||
|
возможно, будет переработана в будущем для поддержки работы без полной
|
||||||
|
загрузки метаданных в памяти.
|
||||||
|
|
||||||
## inmemory_journal
|
## inmemory_journal
|
||||||
|
|
||||||
@@ -376,6 +383,8 @@ fsync небезопасным даже с режимом "directsync".
|
|||||||
нужно менять - это если вы включаете journal_no_same_sector_overwrites. В
|
нужно менять - это если вы включаете journal_no_same_sector_overwrites. В
|
||||||
этом случае установите данный параметр, например, в 1024.
|
этом случае установите данный параметр, например, в 1024.
|
||||||
|
|
||||||
|
Неприменимо к новому хранилищу ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
|
||||||
## journal_no_same_sector_overwrites
|
## journal_no_same_sector_overwrites
|
||||||
|
|
||||||
- Тип: булево (да/нет)
|
- Тип: булево (да/нет)
|
||||||
@@ -391,6 +400,18 @@ fsync небезопасным даже с режимом "directsync".
|
|||||||
|
|
||||||
Почти все другие SSD (99% моделей) не требуют данной опции.
|
Почти все другие SSD (99% моделей) не требуют данной опции.
|
||||||
|
|
||||||
|
Неприменимо к новому хранилищу ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
|
||||||
|
## skip_corrupted_meta_entries
|
||||||
|
|
||||||
|
- Тип: булево (да/нет)
|
||||||
|
- Значение по умолчанию: false
|
||||||
|
|
||||||
|
Только для нового хранилища ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
Разрешить OSD запускаться, даже если часть блоков или записей метаданных
|
||||||
|
повреждена, пропуская их. Опция предназначена для использования только в
|
||||||
|
целях аварийного восстановления.
|
||||||
|
|
||||||
## throttle_small_writes
|
## throttle_small_writes
|
||||||
|
|
||||||
- Тип: булево (да/нет)
|
- Тип: булево (да/нет)
|
||||||
|
|||||||
+46
-14
@@ -253,21 +253,33 @@
|
|||||||
type: bool
|
type: bool
|
||||||
default: true
|
default: true
|
||||||
info: |
|
info: |
|
||||||
This parameter makes Vitastor always keep metadata area of the block device
|
Only for the old store ([meta_format](layout-osd.en.md#meta_format) 2).
|
||||||
in memory. It's required for good performance because it allows to avoid
|
|
||||||
additional read-modify-write cycles during metadata modifications. Metadata
|
This parameter makes Vitastor keep a copy of metadata area in memory as it is
|
||||||
area size is currently roughly 224 MB per 1 TB of data. You can turn it off
|
on disk, in addition to the metadata database. When the option is enabled, every
|
||||||
to reduce memory usage by this value, but it will hurt performance. This
|
metadata entry is effectively stored in RAM twice. It's required for good performance
|
||||||
restriction is likely to be removed in the future along with the upgrade
|
because it allows to avoid additional read-modify-write cycles during metadata
|
||||||
of the metadata storage scheme.
|
modifications. Metadata area size with the old store is roughly 224 MB per 1 TB
|
||||||
|
of data. You can turn the option off to reduce memory usage by this value, but
|
||||||
|
it will reduce performance.
|
||||||
|
|
||||||
|
For the new store ([meta_format](layout-osd.en.md#meta_format) 3), the option
|
||||||
|
may be changed in the future to support operation without loading full metadata
|
||||||
|
database in memory.
|
||||||
info_ru: |
|
info_ru: |
|
||||||
Данный параметр заставляет Vitastor всегда держать область метаданных диска
|
Только для старого хранилища ([meta_format](layout-osd.en.md#meta_format) 2).
|
||||||
в памяти. Это нужно, чтобы избегать дополнительных операций чтения с диска
|
|
||||||
при записи. Размер области метаданных на данный момент составляет примерно
|
Данный параметр заставляет Vitastor всегда держать копию области метаданных
|
||||||
224 МБ на 1 ТБ данных. При включении потребление памяти снизится примерно
|
в памяти в том же виде, как она лежит на диске, в дополнение к БД метаданных.
|
||||||
на эту величину, но при этом также снизится и производительность. В будущем,
|
То есть, с включённой опцией каждая запись метаданных хранится в памяти дважды.
|
||||||
после обновления схемы хранения метаданных, это ограничение, скорее всего,
|
Это нужно, чтобы избегать дополнительных операций чтения с диска при записи.
|
||||||
будет ликвидировано.
|
Размер области метаданных в старом хранилище составляет примерно 224 МБ на
|
||||||
|
1 ТБ данных. Вы можете отключить опцию, чтобы снизить потребление памяти
|
||||||
|
примерно на эту величину, но при этом также снизится и производительность.
|
||||||
|
|
||||||
|
Для нового хранилища ([meta_format](layout-osd.en.md#meta_format) 3) опция,
|
||||||
|
возможно, будет переработана в будущем для поддержки работы без полной
|
||||||
|
загрузки метаданных в памяти.
|
||||||
- name: inmemory_journal
|
- name: inmemory_journal
|
||||||
type: bool
|
type: bool
|
||||||
default: true
|
default: true
|
||||||
@@ -386,11 +398,15 @@
|
|||||||
blocks. The only situation when you should increase it to a larger value
|
blocks. The only situation when you should increase it to a larger value
|
||||||
is when you enable journal_no_same_sector_overwrites. In this case set
|
is when you enable journal_no_same_sector_overwrites. In this case set
|
||||||
it to, for example, 1024.
|
it to, for example, 1024.
|
||||||
|
|
||||||
|
Not applicable to the new store ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
info_ru: |
|
info_ru: |
|
||||||
Максимальное число буферов, разрешённых для использования под записываемые
|
Максимальное число буферов, разрешённых для использования под записываемые
|
||||||
в журнал блоки метаданных. Единственная ситуация, в которой этот параметр
|
в журнал блоки метаданных. Единственная ситуация, в которой этот параметр
|
||||||
нужно менять - это если вы включаете journal_no_same_sector_overwrites. В
|
нужно менять - это если вы включаете journal_no_same_sector_overwrites. В
|
||||||
этом случае установите данный параметр, например, в 1024.
|
этом случае установите данный параметр, например, в 1024.
|
||||||
|
|
||||||
|
Неприменимо к новому хранилищу ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
- name: journal_no_same_sector_overwrites
|
- name: journal_no_same_sector_overwrites
|
||||||
type: bool
|
type: bool
|
||||||
default: false
|
default: false
|
||||||
@@ -402,6 +418,8 @@
|
|||||||
journal after writing it instead of possibly overwriting it the second time.
|
journal after writing it instead of possibly overwriting it the second time.
|
||||||
|
|
||||||
Most (99%) other SSDs don't need this option.
|
Most (99%) other SSDs don't need this option.
|
||||||
|
|
||||||
|
Not applicable to the new store ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
info_ru: |
|
info_ru: |
|
||||||
Включайте данную опцию для SSD вроде Intel D3-S4510 и D3-S4610, которые
|
Включайте данную опцию для SSD вроде Intel D3-S4510 и D3-S4610, которые
|
||||||
ОЧЕНЬ не любят, когда ПО перезаписывает один и тот же сектор несколько раз
|
ОЧЕНЬ не любят, когда ПО перезаписывает один и тот же сектор несколько раз
|
||||||
@@ -412,6 +430,20 @@
|
|||||||
самого сектора.
|
самого сектора.
|
||||||
|
|
||||||
Почти все другие SSD (99% моделей) не требуют данной опции.
|
Почти все другие SSD (99% моделей) не требуют данной опции.
|
||||||
|
|
||||||
|
Неприменимо к новому хранилищу ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
- name: skip_corrupted_meta_entries
|
||||||
|
type: bool
|
||||||
|
default: false
|
||||||
|
info: |
|
||||||
|
Only for the new store ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
Allow OSD to start when some metadata entries or blocks are corrupted by
|
||||||
|
skipping them. Should be only used as an emergency measure.
|
||||||
|
info_ru: |
|
||||||
|
Только для нового хранилища ([meta_format](layout-osd.en.md#meta_format) 3).
|
||||||
|
Разрешить OSD запускаться, даже если часть блоков или записей метаданных
|
||||||
|
повреждена, пропуская их. Опция предназначена для использования только в
|
||||||
|
целях аварийного восстановления.
|
||||||
- name: throttle_small_writes
|
- name: throttle_small_writes
|
||||||
type: bool
|
type: bool
|
||||||
default: false
|
default: false
|
||||||
|
|||||||
@@ -26,13 +26,37 @@ at Vitastor Kubernetes operator: https://github.com/Antilles7227/vitastor-operat
|
|||||||
The instruction is very simple.
|
The instruction is very simple.
|
||||||
|
|
||||||
1. Download a Docker image of the desired version: \
|
1. Download a Docker image of the desired version: \
|
||||||
`docker pull vitalif/vitastor:v3.0.3`
|
`docker pull vitalif/vitastor:v3.0.6`
|
||||||
2. Install scripts to the host system: \
|
2. Install scripts to the host system: \
|
||||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.3 install.sh`
|
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.6 install.sh`
|
||||||
3. Reload udev rules: \
|
3. Reload udev rules: \
|
||||||
`udevadm control --reload-rules`
|
`udevadm control --reload-rules`
|
||||||
|
4. Enable the vitastor-host service: \
|
||||||
|
`systemctl enable --now vitastor-host`
|
||||||
|
|
||||||
And you can return to [Quick Start](../intro/quickstart.en.md).
|
After these steps, you can return to [Quick Start](../intro/quickstart.en.md).
|
||||||
|
|
||||||
|
## Podman
|
||||||
|
|
||||||
|
If you use Podman, run the following commands as root before installing Vitastor containers:
|
||||||
|
|
||||||
|
```
|
||||||
|
ln -s podman /usr/bin/docker
|
||||||
|
|
||||||
|
mkdir -p /etc/systemd/system/systemd-udevd.service.d
|
||||||
|
|
||||||
|
cat >/etc/systemd/system/systemd-udevd.service.d/override.conf <<EOF
|
||||||
|
[Service]
|
||||||
|
CapabilityBoundingSet=~
|
||||||
|
SystemCallFilter=@mount capset
|
||||||
|
EOF
|
||||||
|
|
||||||
|
systemctl daemon-reload
|
||||||
|
|
||||||
|
systemctl restart systemd-udevd
|
||||||
|
```
|
||||||
|
|
||||||
|
Without it, udev fails to do calls into a Podman container and Vitastor disk detection doesn't work.
|
||||||
|
|
||||||
## Upgrading Containers
|
## Upgrading Containers
|
||||||
|
|
||||||
|
|||||||
@@ -25,14 +25,39 @@ Vitastor можно установить в Docker/Podman. При этом etcd,
|
|||||||
Инструкция по установке максимально простая.
|
Инструкция по установке максимально простая.
|
||||||
|
|
||||||
1. Скачайте Docker-образ желаемой версии: \
|
1. Скачайте Docker-образ желаемой версии: \
|
||||||
`docker pull vitalif/vitastor:v3.0.3`
|
`docker pull vitalif/vitastor:v3.0.6`
|
||||||
2. Установите скрипты в хост-систему командой: \
|
2. Установите скрипты в хост-систему командой: \
|
||||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.3 install.sh`
|
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.6 install.sh`
|
||||||
3. Перезагрузите правила udev: \
|
3. Перезагрузите правила udev: \
|
||||||
`udevadm control --reload-rules`
|
`udevadm control --reload-rules`
|
||||||
|
4. Включите сервис vitastor-host: \
|
||||||
|
`systemctl enable --now vitastor-host`
|
||||||
|
|
||||||
После этого вы можете возвращаться к разделу [Быстрый старт](../intro/quickstart.ru.md).
|
После этого вы можете возвращаться к разделу [Быстрый старт](../intro/quickstart.ru.md).
|
||||||
|
|
||||||
|
## Podman
|
||||||
|
|
||||||
|
Если вы используете Podman, перед установкой контейнеров Vitastor выполните следующие
|
||||||
|
команды от имени суперпользователя:
|
||||||
|
|
||||||
|
```
|
||||||
|
ln -s podman /usr/bin/docker
|
||||||
|
|
||||||
|
mkdir -p /etc/systemd/system/systemd-udevd.service.d
|
||||||
|
|
||||||
|
cat >/etc/systemd/system/systemd-udevd.service.d/override.conf <<EOF
|
||||||
|
[Service]
|
||||||
|
CapabilityBoundingSet=~
|
||||||
|
SystemCallFilter=@mount capset
|
||||||
|
EOF
|
||||||
|
|
||||||
|
systemctl daemon-reload
|
||||||
|
|
||||||
|
systemctl restart systemd-udevd
|
||||||
|
```
|
||||||
|
|
||||||
|
Без этих настроек udev не может делать вызовы внутрь Podman-контейнеров и определение дисков Vitastor не работает.
|
||||||
|
|
||||||
## Обновление контейнеров
|
## Обновление контейнеров
|
||||||
|
|
||||||
Сначала обязательно проверьте раздел [Обновление Vitastor](../usage/admin.ru.md#обновление-vitastor),
|
Сначала обязательно проверьте раздел [Обновление Vitastor](../usage/admin.ru.md#обновление-vitastor),
|
||||||
|
|||||||
@@ -33,15 +33,17 @@
|
|||||||
- CentOS 7: `yum install https://vitastor.io/rpms/centos/7/vitastor-release.rpm`
|
- CentOS 7: `yum install https://vitastor.io/rpms/centos/7/vitastor-release.rpm`
|
||||||
- CentOS 8: `dnf install https://vitastor.io/rpms/centos/8/vitastor-release.rpm`
|
- CentOS 8: `dnf install https://vitastor.io/rpms/centos/8/vitastor-release.rpm`
|
||||||
- AlmaLinux 9 and other RHEL 9 clones (Rocky, Oracle...): `dnf install https://vitastor.io/rpms/centos/9/vitastor-release.rpm`
|
- AlmaLinux 9 and other RHEL 9 clones (Rocky, Oracle...): `dnf install https://vitastor.io/rpms/centos/9/vitastor-release.rpm`
|
||||||
|
- AlmaLinux 10 and other RHEL 10 clones: `dnf install https://vitastor.io/rpms/centos/10/vitastor-release.rpm`
|
||||||
- Enable EPEL: `yum/dnf install epel-release`
|
- Enable EPEL: `yum/dnf install epel-release`
|
||||||
- Enable additional CentOS repositories:
|
- Enable additional CentOS repositories:
|
||||||
- CentOS 7: `yum install centos-release-scl`
|
- CentOS 7: `yum install centos-release-scl`
|
||||||
- CentOS 8: `dnf install centos-release-advanced-virtualization`
|
- CentOS 8: `dnf install centos-release-advanced-virtualization`
|
||||||
- RHEL 9 clones: not required
|
- RHEL 9/10 clones: not required
|
||||||
- Enable elrepo-kernel:
|
- Enable elrepo-kernel:
|
||||||
- CentOS 7: `yum install https://www.elrepo.org/elrepo-release-7.el7.elrepo.noarch.rpm`
|
- CentOS 7: `yum install https://www.elrepo.org/elrepo-release-7.el7.elrepo.noarch.rpm`
|
||||||
- CentOS 8: `dnf install https://www.elrepo.org/elrepo-release-8.el8.elrepo.noarch.rpm`
|
- CentOS 8: `dnf install https://www.elrepo.org/elrepo-release-8.el8.elrepo.noarch.rpm`
|
||||||
- RHEL 9 clones: `dnf install https://www.elrepo.org/elrepo-release-9.el9.elrepo.noarch.rpm`
|
- RHEL 9 clones: `dnf install https://www.elrepo.org/elrepo-release-9.el9.elrepo.noarch.rpm`
|
||||||
|
- RHEL 10 clones: not required
|
||||||
- Install packages: `yum/dnf install vitastor lpsolve etcd kernel-ml qemu-kvm`
|
- Install packages: `yum/dnf install vitastor lpsolve etcd kernel-ml qemu-kvm`
|
||||||
|
|
||||||
## Installation requirements
|
## Installation requirements
|
||||||
|
|||||||
@@ -33,15 +33,17 @@
|
|||||||
- CentOS 7: `yum install https://vitastor.io/rpms/centos/7/vitastor-release.rpm`
|
- CentOS 7: `yum install https://vitastor.io/rpms/centos/7/vitastor-release.rpm`
|
||||||
- CentOS 8: `dnf install https://vitastor.io/rpms/centos/8/vitastor-release.rpm`
|
- CentOS 8: `dnf install https://vitastor.io/rpms/centos/8/vitastor-release.rpm`
|
||||||
- AlmaLinux 9 и другие клоны RHEL 9 (Rocky, Oracle...): `dnf install https://vitastor.io/rpms/centos/9/vitastor-release.rpm`
|
- AlmaLinux 9 и другие клоны RHEL 9 (Rocky, Oracle...): `dnf install https://vitastor.io/rpms/centos/9/vitastor-release.rpm`
|
||||||
|
- AlmaLinux 10 и другие клоны RHEL 10: `dnf install https://vitastor.io/rpms/centos/10/vitastor-release.rpm`
|
||||||
- Включите EPEL: `yum/dnf install epel-release`
|
- Включите EPEL: `yum/dnf install epel-release`
|
||||||
- Включите дополнительные репозитории CentOS:
|
- Включите дополнительные репозитории CentOS:
|
||||||
- CentOS 7: `yum install centos-release-scl`
|
- CentOS 7: `yum install centos-release-scl`
|
||||||
- CentOS 8: `dnf install centos-release-advanced-virtualization`
|
- CentOS 8: `dnf install centos-release-advanced-virtualization`
|
||||||
- Клоны RHEL 9: не нужно
|
- Клоны RHEL 9/10: не нужно
|
||||||
- Включите elrepo-kernel:
|
- Включите elrepo-kernel:
|
||||||
- CentOS 7: `yum install https://www.elrepo.org/elrepo-release-7.el7.elrepo.noarch.rpm`
|
- CentOS 7: `yum install https://www.elrepo.org/elrepo-release-7.el7.elrepo.noarch.rpm`
|
||||||
- CentOS 8: `dnf install https://www.elrepo.org/elrepo-release-8.el8.elrepo.noarch.rpm`
|
- CentOS 8: `dnf install https://www.elrepo.org/elrepo-release-8.el8.elrepo.noarch.rpm`
|
||||||
- Клоны RHEL 9: `dnf install https://www.elrepo.org/elrepo-release-9.el9.elrepo.noarch.rpm`
|
- Клоны RHEL 9: `dnf install https://www.elrepo.org/elrepo-release-9.el9.elrepo.noarch.rpm`
|
||||||
|
- Клоны RHEL 10: не нужно
|
||||||
- Установите пакеты: `yum/dnf install vitastor lpsolve etcd kernel-ml qemu-kvm`
|
- Установите пакеты: `yum/dnf install vitastor lpsolve etcd kernel-ml qemu-kvm`
|
||||||
|
|
||||||
## Установочные требования
|
## Установочные требования
|
||||||
|
|||||||
+1
-1
@@ -16,7 +16,7 @@ async function create_http_server(cfg, handler)
|
|||||||
};
|
};
|
||||||
if (cfg.mon_https_ca)
|
if (cfg.mon_https_ca)
|
||||||
{
|
{
|
||||||
tls.mon_https_ca = await fsp.readFile(cfg.mon_https_ca);
|
tls.ca = await fsp.readFile(cfg.mon_https_ca);
|
||||||
}
|
}
|
||||||
if (cfg.mon_https_client_auth)
|
if (cfg.mon_https_client_auth)
|
||||||
{
|
{
|
||||||
|
|||||||
+2
-2
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "vitastor-mon",
|
"name": "vitastor-mon",
|
||||||
"version": "3.0.3",
|
"version": "3.0.6",
|
||||||
"description": "Vitastor SDS monitor service",
|
"description": "Vitastor SDS monitor service",
|
||||||
"main": "mon-main.js",
|
"main": "mon-main.js",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
@@ -9,7 +9,7 @@
|
|||||||
"author": "Vitaliy Filippov",
|
"author": "Vitaliy Filippov",
|
||||||
"license": "UNLICENSED",
|
"license": "UNLICENSED",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"antietcd": "^1.2.2",
|
"antietcd": "^1.2.4",
|
||||||
"sprintf-js": "^1.1.2",
|
"sprintf-js": "^1.1.2",
|
||||||
"ws": "^7.2.5"
|
"ws": "^7.2.5"
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "vitastor",
|
"name": "vitastor",
|
||||||
"version": "3.0.3",
|
"version": "3.0.6",
|
||||||
"description": "Low-level native bindings to Vitastor client library",
|
"description": "Low-level native bindings to Vitastor client library",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"keywords": [
|
"keywords": [
|
||||||
|
|||||||
+30
-232
@@ -50,7 +50,7 @@ from cinder.volume import configuration
|
|||||||
from cinder.volume import driver
|
from cinder.volume import driver
|
||||||
from cinder.volume import volume_utils
|
from cinder.volume import volume_utils
|
||||||
|
|
||||||
VITASTOR_VERSION = '3.0.3'
|
VITASTOR_VERSION = '3.0.6'
|
||||||
|
|
||||||
LOG = logging.getLogger(__name__)
|
LOG = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -275,7 +275,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
LOG.exception('error getting vitastor pool stats: '+str(e))
|
LOG.exception('error getting vitastor pool stats: '+str(e))
|
||||||
|
|
||||||
self._stats = stats
|
self._stats = stats
|
||||||
|
|
||||||
def get_volume_stats(self, refresh=False):
|
def get_volume_stats(self, refresh=False):
|
||||||
"""Get volume stats.
|
"""Get volume stats.
|
||||||
If 'refresh' is True, run update the stats first.
|
If 'refresh' is True, run update the stats first.
|
||||||
@@ -291,6 +291,14 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
else:
|
else:
|
||||||
return (1 + resp['kvs'][0]['value'], resp['kvs'][0]['mod_revision'])
|
return (1 + resp['kvs'][0]['value'], resp['kvs'][0]['mod_revision'])
|
||||||
|
|
||||||
|
def _cli(self, descr, *args):
|
||||||
|
args = [ 'vitastor-cli', *args, *(self._vitastor_args()) ]
|
||||||
|
try:
|
||||||
|
self._execute(*args)
|
||||||
|
except processutils.ProcessExecutionError as exc:
|
||||||
|
LOG.error("Failed to "+descr+": "+exc)
|
||||||
|
raise exception.VolumeBackendAPIException(data = exc.stderr)
|
||||||
|
|
||||||
def create_volume(self, volume):
|
def create_volume(self, volume):
|
||||||
"""Creates a logical volume."""
|
"""Creates a logical volume."""
|
||||||
|
|
||||||
@@ -302,7 +310,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
|
|
||||||
LOG.debug("creating volume '%s'", vol_name)
|
LOG.debug("creating volume '%s'", vol_name)
|
||||||
|
|
||||||
self._create_image(vol_name, { 'size': size })
|
self._cli('create volume', 'create', vol_name, '--size', size)
|
||||||
|
|
||||||
if volume.encryption_key_id:
|
if volume.encryption_key_id:
|
||||||
self._create_encrypted_volume(volume, volume.obj_context)
|
self._create_encrypted_volume(volume, volume.obj_context)
|
||||||
@@ -346,7 +354,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
snap_name = utils.convert_str(snapshot.name)
|
snap_name = utils.convert_str(snapshot.name)
|
||||||
if snap_name.find('@') >= 0 or snap_name.find('/') >= 0:
|
if snap_name.find('@') >= 0 or snap_name.find('/') >= 0:
|
||||||
raise exception.VolumeBackendAPIException(data = '@ and / are forbidden in volume and snapshot names')
|
raise exception.VolumeBackendAPIException(data = '@ and / are forbidden in volume and snapshot names')
|
||||||
self._create_snapshot(vol_name, vol_name+'@'+snap_name)
|
self._cli('create snapshot', 'snap-create', vol_name+'@'+snap_name)
|
||||||
|
|
||||||
def snapshot_revert_use_temp_snapshot(self):
|
def snapshot_revert_use_temp_snapshot(self):
|
||||||
"""Disable the use of a temporary snapshot on revert."""
|
"""Disable the use of a temporary snapshot on revert."""
|
||||||
@@ -359,21 +367,8 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
snap_name = utils.convert_str(snapshot.name)
|
snap_name = utils.convert_str(snapshot.name)
|
||||||
|
|
||||||
# Delete the image and recreate it from the snapshot
|
# Delete the image and recreate it from the snapshot
|
||||||
args = [ 'vitastor-cli', 'rm', vol_name, *(self._vitastor_args()) ]
|
self._cli('delete image', 'rm', vol_name)
|
||||||
try:
|
self._cli('recreate image', 'create', '--parent', vol_name+'@'+snap_name, vol_name)
|
||||||
self._execute(*args)
|
|
||||||
except processutils.ProcessExecutionError as exc:
|
|
||||||
LOG.error("Failed to delete image "+vol_name+": "+exc)
|
|
||||||
raise exception.VolumeBackendAPIException(data = exc.stderr)
|
|
||||||
args = [
|
|
||||||
'vitastor-cli', 'create', '--parent', vol_name+'@'+snap_name,
|
|
||||||
vol_name, *(self._vitastor_args())
|
|
||||||
]
|
|
||||||
try:
|
|
||||||
self._execute(*args)
|
|
||||||
except processutils.ProcessExecutionError as exc:
|
|
||||||
LOG.error("Failed to recreate image "+vol_name+" from "+vol_name+"@"+snap_name+": "+exc)
|
|
||||||
raise exception.VolumeBackendAPIException(data = exc.stderr)
|
|
||||||
|
|
||||||
def delete_snapshot(self, snapshot):
|
def delete_snapshot(self, snapshot):
|
||||||
"""Deletes a snapshot."""
|
"""Deletes a snapshot."""
|
||||||
@@ -381,15 +376,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
vol_name = utils.convert_str(snapshot.volume_name)
|
vol_name = utils.convert_str(snapshot.volume_name)
|
||||||
snap_name = utils.convert_str(snapshot.name)
|
snap_name = utils.convert_str(snapshot.name)
|
||||||
|
|
||||||
args = [
|
self._cli('remove snapshot', 'rm', vol_name+'@'+snap_name)
|
||||||
'vitastor-cli', 'rm', vol_name+'@'+snap_name,
|
|
||||||
*(self._vitastor_args())
|
|
||||||
]
|
|
||||||
try:
|
|
||||||
self._execute(*args)
|
|
||||||
except processutils.ProcessExecutionError as exc:
|
|
||||||
LOG.error("Failed to remove snapshot "+vol_name+'@'+snap_name+": "+exc)
|
|
||||||
raise exception.VolumeBackendAPIException(data = exc.stderr)
|
|
||||||
|
|
||||||
def _child_count(self, parents):
|
def _child_count(self, parents):
|
||||||
children = 0
|
children = 0
|
||||||
@@ -427,13 +414,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
if src_vref.admin_metadata.get('readonly') == 'True':
|
if src_vref.admin_metadata.get('readonly') == 'True':
|
||||||
# source volume is a volume-image cache entry or other readonly volume
|
# source volume is a volume-image cache entry or other readonly volume
|
||||||
# clone without intermediate snapshot
|
# clone without intermediate snapshot
|
||||||
src = self._get_image(src_name)
|
self._cli('create clone', 'create', '--parent', src_name, '--size', size, dest_name)
|
||||||
LOG.debug("creating image '%s' from '%s'", dest_name, src_name)
|
|
||||||
new_cfg = self._create_image(dest_name, {
|
|
||||||
'size': size,
|
|
||||||
'parent_id': src['idx']['id'],
|
|
||||||
'parent_pool_id': src['idx']['pool_id'],
|
|
||||||
})
|
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
clone_snap = "%s@%s.clone_snap" % (src_name, dest_name)
|
clone_snap = "%s@%s.clone_snap" % (src_name, dest_name)
|
||||||
@@ -446,15 +427,12 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
clone_snap = dest_name
|
clone_snap = dest_name
|
||||||
make_img = False
|
make_img = False
|
||||||
|
|
||||||
LOG.debug("creating layer '%s' under '%s'", clone_snap, src_name)
|
LOG.debug("creating snapshot '%s'", clone_snap)
|
||||||
new_cfg = self._create_snapshot(src_name, clone_snap, True)
|
self._cli('create base snapshot', 'snap-create', '--allow-existing', '1', clone_snap)
|
||||||
|
|
||||||
if make_img:
|
if make_img:
|
||||||
# Then create a clone from it
|
# Then create a clone from it
|
||||||
new_cfg = self._create_image(dest_name, {
|
self._cli('create clone', 'create', '--parent', clone_snap, '--size', size, dest_name)
|
||||||
'size': size,
|
|
||||||
'parent_id': new_cfg['parent_id'],
|
|
||||||
'parent_pool_id': new_cfg['parent_pool_id'],
|
|
||||||
})
|
|
||||||
|
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
@@ -464,7 +442,8 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
vol_name = utils.convert_str(volume.name)
|
vol_name = utils.convert_str(volume.name)
|
||||||
snap_name = utils.convert_str(snapshot.name)
|
snap_name = utils.convert_str(snapshot.name)
|
||||||
|
|
||||||
snap = self._get_image('volume-'+snapshot.volume_id+'@'+snap_name)
|
src_snap = 'volume-'+snapshot.volume_id+'@'+snap_name
|
||||||
|
snap = self._get_image(src_snap)
|
||||||
if not snap:
|
if not snap:
|
||||||
raise exception.SnapshotNotFound(snapshot_id = snap_name)
|
raise exception.SnapshotNotFound(snapshot_id = snap_name)
|
||||||
snap_inode_id = int(resp['responses'][0]['kvs'][0]['value']['id'])
|
snap_inode_id = int(resp['responses'][0]['kvs'][0]['value']['id'])
|
||||||
@@ -473,12 +452,8 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
size = snap['cfg']['size']
|
size = snap['cfg']['size']
|
||||||
if int(volume.size):
|
if int(volume.size):
|
||||||
size = int(volume.size) * units.Gi
|
size = int(volume.size) * units.Gi
|
||||||
new_cfg = self._create_image(vol_name, {
|
|
||||||
'size': size,
|
|
||||||
'parent_id': snap['idx']['id'],
|
|
||||||
'parent_pool_id': snap['idx']['pool_id'],
|
|
||||||
})
|
|
||||||
|
|
||||||
|
self._cli('create clone', 'create', vol_name, '--size', size, '--parent', src_snap)
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
def _vitastor_args(self):
|
def _vitastor_args(self):
|
||||||
@@ -505,49 +480,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
"""Deletes a logical volume."""
|
"""Deletes a logical volume."""
|
||||||
|
|
||||||
vol_name = utils.convert_str(volume.name)
|
vol_name = utils.convert_str(volume.name)
|
||||||
|
self._cli('delete volume', 'rm', '--matching', vol_name, vol_name+'@*', '--progress', '0')
|
||||||
# Find the volume and all its snapshots
|
|
||||||
range_end = b'index/image/' + vol_name.encode('utf-8')
|
|
||||||
range_end = range_end[0 : len(range_end)-1] + six.int2byte(range_end[len(range_end)-1] + 1)
|
|
||||||
resp = self._etcd_txn({ 'success': [
|
|
||||||
{ 'request_range': { 'key': 'index/image/'+vol_name, 'range_end': range_end } },
|
|
||||||
] })
|
|
||||||
if len(resp['responses'][0]['kvs']) == 0:
|
|
||||||
# already deleted
|
|
||||||
LOG.info("volume %s no longer exists in backend", vol_name)
|
|
||||||
return
|
|
||||||
layers = resp['responses'][0]['kvs']
|
|
||||||
layer_ids = {}
|
|
||||||
for kv in layers:
|
|
||||||
inode_id = int(kv['value']['id'])
|
|
||||||
pool_id = int(kv['value']['pool_id'])
|
|
||||||
inode_pool_id = (pool_id << 48) | (inode_id & 0xffffffffffff)
|
|
||||||
layer_ids[inode_pool_id] = True
|
|
||||||
|
|
||||||
# Check if the volume has clones and raise 'busy' if so
|
|
||||||
children = self._child_count(layer_ids)
|
|
||||||
if children > 0:
|
|
||||||
raise exception.VolumeIsBusy(volume_name = vol_name)
|
|
||||||
|
|
||||||
# Clear data
|
|
||||||
for kv in layers:
|
|
||||||
args = [
|
|
||||||
'vitastor-cli', 'rm-data', '--pool', str(kv['value']['pool_id']),
|
|
||||||
'--inode', str(kv['value']['id']), '--progress', '0',
|
|
||||||
*(self._vitastor_args())
|
|
||||||
]
|
|
||||||
try:
|
|
||||||
self._execute(*args)
|
|
||||||
except processutils.ProcessExecutionError as exc:
|
|
||||||
LOG.error("Failed to remove layer "+kv['key']+": "+exc)
|
|
||||||
raise exception.VolumeBackendAPIException(data = exc.stderr)
|
|
||||||
|
|
||||||
# Delete all layers from etcd
|
|
||||||
requests = []
|
|
||||||
for kv in layers:
|
|
||||||
requests.append({ 'request_delete_range': { 'key': kv['key'] } })
|
|
||||||
requests.append({ 'request_delete_range': { 'key': 'config/inode/'+str(kv['value']['pool_id'])+'/'+str(kv['value']['id']) } })
|
|
||||||
self._etcd_txn({ 'success': requests })
|
|
||||||
|
|
||||||
def retype(self, context, volume, new_type, diff, host):
|
def retype(self, context, volume, new_type, diff, host):
|
||||||
"""Change extra type specifications for a volume."""
|
"""Change extra type specifications for a volume."""
|
||||||
@@ -567,98 +500,6 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
"""Removes an export for a logical volume."""
|
"""Removes an export for a logical volume."""
|
||||||
pass
|
pass
|
||||||
|
|
||||||
def _create_image(self, vol_name, cfg):
|
|
||||||
pool_s = str(self.cfg['pool_id'])
|
|
||||||
image_id = 0
|
|
||||||
while image_id == 0:
|
|
||||||
# check if the image already exists and find a free ID
|
|
||||||
resp = self._etcd_txn({ 'success': [
|
|
||||||
{ 'request_range': { 'key': 'index/image/'+vol_name } },
|
|
||||||
{ 'request_range': { 'key': 'index/maxid/'+pool_s } },
|
|
||||||
] })
|
|
||||||
if len(resp['responses'][0]['kvs']) > 0:
|
|
||||||
# already exists
|
|
||||||
raise exception.VolumeBackendAPIException(data = 'Volume '+vol_name+' already exists')
|
|
||||||
image_id, id_mod = self._next_id(resp['responses'][1])
|
|
||||||
# try to create the image
|
|
||||||
resp = self._etcd_txn({ 'compare': [
|
|
||||||
{ 'target': 'MOD', 'mod_revision': id_mod, 'key': 'index/maxid/'+pool_s },
|
|
||||||
{ 'target': 'VERSION', 'version': 0, 'key': 'index/image/'+vol_name },
|
|
||||||
{ 'target': 'VERSION', 'version': 0, 'key': 'config/inode/'+pool_s+'/'+str(image_id) },
|
|
||||||
], 'success': [
|
|
||||||
{ 'request_put': { 'key': 'index/maxid/'+pool_s, 'value': image_id } },
|
|
||||||
{ 'request_put': { 'key': 'index/image/'+vol_name, 'value': json.dumps({
|
|
||||||
'id': image_id, 'pool_id': self.cfg['pool_id']
|
|
||||||
}) } },
|
|
||||||
{ 'request_put': { 'key': 'config/inode/'+pool_s+'/'+str(image_id), 'value': json.dumps({
|
|
||||||
**cfg, 'name': vol_name,
|
|
||||||
}) } },
|
|
||||||
] })
|
|
||||||
if not resp.get('succeeded'):
|
|
||||||
# repeat
|
|
||||||
image_id = 0
|
|
||||||
|
|
||||||
def _create_snapshot(self, vol_name, snap_vol_name, allow_existing = False):
|
|
||||||
while True:
|
|
||||||
# check if the image already exists and snapshot doesn't
|
|
||||||
resp = self._etcd_txn({ 'success': [
|
|
||||||
{ 'request_range': { 'key': 'index/image/'+vol_name } },
|
|
||||||
{ 'request_range': { 'key': 'index/image/'+snap_vol_name } },
|
|
||||||
] })
|
|
||||||
if len(resp['responses'][0]['kvs']) == 0:
|
|
||||||
raise exception.VolumeBackendAPIException(data = 'Volume '+vol_name+' does not exist')
|
|
||||||
if len(resp['responses'][1]['kvs']) > 0:
|
|
||||||
if allow_existing:
|
|
||||||
snap_idx = resp['responses'][1]['kvs'][0]['value']
|
|
||||||
resp = self._etcd_txn({ 'success': [
|
|
||||||
{ 'request_range': { 'key': 'config/inode/'+str(snap_idx['pool_id'])+'/'+str(snap_idx['id']) } },
|
|
||||||
] })
|
|
||||||
if len(resp['responses'][0]['kvs']) == 0:
|
|
||||||
raise exception.VolumeBackendAPIException(data =
|
|
||||||
'Volume '+snap_vol_name+' is already indexed, but does not exist'
|
|
||||||
)
|
|
||||||
return resp['responses'][0]['kvs'][0]['value']
|
|
||||||
raise exception.VolumeBackendAPIException(
|
|
||||||
data = 'Volume '+snap_vol_name+' already exists'
|
|
||||||
)
|
|
||||||
vol_idx = resp['responses'][0]['kvs'][0]['value']
|
|
||||||
vol_idx_mod = resp['responses'][0]['kvs'][0]['mod_revision']
|
|
||||||
# get image inode config and find a new ID
|
|
||||||
resp = self._etcd_txn({ 'success': [
|
|
||||||
{ 'request_range': { 'key': 'config/inode/'+str(vol_idx['pool_id'])+'/'+str(vol_idx['id']) } },
|
|
||||||
{ 'request_range': { 'key': 'index/maxid/'+str(self.cfg['pool_id']) } },
|
|
||||||
] })
|
|
||||||
if len(resp['responses'][0]['kvs']) == 0:
|
|
||||||
raise exception.VolumeBackendAPIException(data = 'Volume '+vol_name+' does not exist')
|
|
||||||
vol_cfg = resp['responses'][0]['kvs'][0]['value']
|
|
||||||
vol_mod = resp['responses'][0]['kvs'][0]['mod_revision']
|
|
||||||
new_id, id_mod = self._next_id(resp['responses'][1])
|
|
||||||
# try to redirect image to the new inode
|
|
||||||
new_cfg = {
|
|
||||||
**vol_cfg, 'name': vol_name, 'parent_id': vol_idx['id'], 'parent_pool_id': vol_idx['pool_id']
|
|
||||||
}
|
|
||||||
resp = self._etcd_txn({ 'compare': [
|
|
||||||
{ 'target': 'MOD', 'mod_revision': vol_idx_mod, 'key': 'index/image/'+vol_name },
|
|
||||||
{ 'target': 'MOD', 'mod_revision': vol_mod, 'key': 'config/inode/'+str(vol_idx['pool_id'])+'/'+str(vol_idx['id']) },
|
|
||||||
{ 'target': 'MOD', 'mod_revision': id_mod, 'key': 'index/maxid/'+str(self.cfg['pool_id']) },
|
|
||||||
{ 'target': 'VERSION', 'version': 0, 'key': 'index/image/'+snap_vol_name },
|
|
||||||
{ 'target': 'VERSION', 'version': 0, 'key': 'config/inode/'+str(self.cfg['pool_id'])+'/'+str(new_id) },
|
|
||||||
], 'success': [
|
|
||||||
{ 'request_put': { 'key': 'index/maxid/'+str(self.cfg['pool_id']), 'value': new_id } },
|
|
||||||
{ 'request_put': { 'key': 'index/image/'+vol_name, 'value': json.dumps({
|
|
||||||
'id': new_id, 'pool_id': self.cfg['pool_id']
|
|
||||||
}) } },
|
|
||||||
{ 'request_put': { 'key': 'config/inode/'+str(self.cfg['pool_id'])+'/'+str(new_id), 'value': json.dumps(new_cfg) } },
|
|
||||||
{ 'request_put': { 'key': 'index/image/'+snap_vol_name, 'value': json.dumps({
|
|
||||||
'id': vol_idx['id'], 'pool_id': vol_idx['pool_id']
|
|
||||||
}) } },
|
|
||||||
{ 'request_put': { 'key': 'config/inode/'+str(vol_idx['pool_id'])+'/'+str(vol_idx['id']), 'value': json.dumps({
|
|
||||||
**vol_cfg, 'name': snap_vol_name, 'readonly': True
|
|
||||||
}) } }
|
|
||||||
] })
|
|
||||||
if resp.get('succeeded'):
|
|
||||||
return new_cfg
|
|
||||||
|
|
||||||
def initialize_connection(self, volume, connector):
|
def initialize_connection(self, volume, connector):
|
||||||
data = {
|
data = {
|
||||||
'driver_volume_type': 'vitastor',
|
'driver_volume_type': 'vitastor',
|
||||||
@@ -697,13 +538,9 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
size = int(volume.size) * units.Gi
|
size = int(volume.size) * units.Gi
|
||||||
dest_name = utils.convert_str(volume.name)
|
dest_name = utils.convert_str(volume.name)
|
||||||
# Find or create the base snapshot
|
# Find or create the base snapshot
|
||||||
snap_cfg = self._create_snapshot(base_vol.name, base_vol.name+'@.clone_snap', True)
|
self._cli('create base snapshot', 'create', '--allow-existing', '1', base_vol.name+'@.clone_snap')
|
||||||
# Then create a clone from it
|
# Then create a clone from it
|
||||||
new_cfg = self._create_image(dest_name, {
|
self._cli('create clone', 'create', dest_name, '--size', size, '--parent', base_vol.name+'@.clone_snap')
|
||||||
'size': size,
|
|
||||||
'parent_id': snap_cfg['parent_id'],
|
|
||||||
'parent_pool_id': snap_cfg['parent_pool_id'],
|
|
||||||
})
|
|
||||||
return ({}, True)
|
return ({}, True)
|
||||||
return ({}, False)
|
return ({}, False)
|
||||||
|
|
||||||
@@ -770,26 +607,8 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
def extend_volume(self, volume, new_size):
|
def extend_volume(self, volume, new_size):
|
||||||
"""Extend an existing volume."""
|
"""Extend an existing volume."""
|
||||||
vol_name = utils.convert_str(volume.name)
|
vol_name = utils.convert_str(volume.name)
|
||||||
while True:
|
size = int(new_size) * units.Gi
|
||||||
vol = self._get_image(vol_name)
|
self._cli('extend volume', 'modify', vol_name, '--resize', new_size)
|
||||||
if not vol:
|
|
||||||
raise exception.VolumeBackendAPIException(data = 'Volume '+vol_name+' does not exist')
|
|
||||||
# change size
|
|
||||||
size = int(new_size) * units.Gi
|
|
||||||
if size == vol['cfg']['size']:
|
|
||||||
break
|
|
||||||
resp = self._etcd_txn({ 'compare': [ {
|
|
||||||
'target': 'MOD',
|
|
||||||
'mod_revision': vol['cfg_mod'],
|
|
||||||
'key': 'config/inode/'+str(vol['idx']['pool_id'])+'/'+str(vol['idx']['id']),
|
|
||||||
} ], 'success': [
|
|
||||||
{ 'request_put': {
|
|
||||||
'key': 'config/inode/'+str(vol['idx']['pool_id'])+'/'+str(vol['idx']['id']),
|
|
||||||
'value': json.dumps({ **vol['cfg'], 'size': size }),
|
|
||||||
} },
|
|
||||||
] })
|
|
||||||
if resp.get('succeeded'):
|
|
||||||
break
|
|
||||||
LOG.debug(
|
LOG.debug(
|
||||||
"Extend volume from %(old_size)s GB to %(new_size)s GB.",
|
"Extend volume from %(old_size)s GB to %(new_size)s GB.",
|
||||||
{'old_size': volume.size, 'new_size': new_size}
|
{'old_size': volume.size, 'new_size': new_size}
|
||||||
@@ -862,28 +681,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
"""
|
"""
|
||||||
from_name = self._get_existing_name(existing_ref)
|
from_name = self._get_existing_name(existing_ref)
|
||||||
to_name = utils.convert_str(volume.name)
|
to_name = utils.convert_str(volume.name)
|
||||||
self._rename(from_name, to_name)
|
self._cli('rename', 'modify', from_name, '--rename', to_name)
|
||||||
|
|
||||||
def _rename(self, from_name, to_name):
|
|
||||||
while True:
|
|
||||||
vol = self._get_image(from_name)
|
|
||||||
if not vol:
|
|
||||||
raise exception.VolumeBackendAPIException(data = 'Volume '+from_name+' does not exist')
|
|
||||||
to = self._get_image(to_name)
|
|
||||||
if to:
|
|
||||||
raise exception.VolumeBackendAPIException(data = 'Volume '+to_name+' already exists')
|
|
||||||
resp = self._etcd_txn({ 'compare': [
|
|
||||||
{ 'target': 'MOD', 'mod_revision': vol['idx_mod'], 'key': 'index/image/'+vol['cfg']['name'] },
|
|
||||||
{ 'target': 'MOD', 'mod_revision': vol['cfg_mod'], 'key': 'config/inode/'+str(vol['idx']['pool_id'])+'/'+str(vol['idx']['id']) },
|
|
||||||
{ 'target': 'VERSION', 'version': 0, 'key': 'index/image/'+to_name },
|
|
||||||
], 'success': [
|
|
||||||
{ 'request_delete_range': { 'key': 'index/image/'+vol['cfg']['name'] } },
|
|
||||||
{ 'request_put': { 'key': 'index/image/'+to_name, 'value': json.dumps(vol['idx']) } },
|
|
||||||
{ 'request_put': { 'key': 'config/inode/'+str(vol['idx']['pool_id'])+'/'+str(vol['idx']['id']),
|
|
||||||
'value': json.dumps({ **vol['cfg'], 'name': to_name }) } },
|
|
||||||
] })
|
|
||||||
if resp.get('succeeded'):
|
|
||||||
break
|
|
||||||
|
|
||||||
def unmanage(self, volume):
|
def unmanage(self, volume):
|
||||||
pass
|
pass
|
||||||
@@ -956,7 +754,7 @@ class VitastorDriver(driver.CloneableImageVD,
|
|||||||
snap_name = self._get_existing_name(existing_ref)
|
snap_name = self._get_existing_name(existing_ref)
|
||||||
from_name = vol_name+'@'+snap_name
|
from_name = vol_name+'@'+snap_name
|
||||||
to_name = vol_name+'@'+utils.convert_str(snapshot.name)
|
to_name = vol_name+'@'+utils.convert_str(snapshot.name)
|
||||||
self._rename(from_name, to_name)
|
self._cli('rename', 'modify', from_name, '--rename', to_name)
|
||||||
|
|
||||||
def unmanage_snapshot(self, snapshot):
|
def unmanage_snapshot(self, snapshot):
|
||||||
"""Removes the specified snapshot from Cinder management."""
|
"""Removes the specified snapshot from Cinder management."""
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 3.0.3
|
Version: 3.0.6
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-3.0.3.el10.tar.gz
|
Source0: vitastor-3.0.6.el10.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: gcc-c++
|
BuildRequires: gcc-c++
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 3.0.3
|
Version: 3.0.6
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-3.0.3.el7.tar.gz
|
Source0: vitastor-3.0.6.el7.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: devtoolset-9-gcc-c++
|
BuildRequires: devtoolset-9-gcc-c++
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 3.0.3
|
Version: 3.0.6
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-3.0.3.el8.tar.gz
|
Source0: vitastor-3.0.6.el8.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: gcc-toolset-9-gcc-c++
|
BuildRequires: gcc-toolset-9-gcc-c++
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 3.0.3
|
Version: 3.0.6
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-3.0.3.el9.tar.gz
|
Source0: vitastor-3.0.6.el9.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: gcc-c++
|
BuildRequires: gcc-c++
|
||||||
|
|||||||
+1
-1
@@ -21,7 +21,7 @@ if("${CMAKE_INSTALL_PREFIX}" MATCHES "^/usr/local/?$")
|
|||||||
endif()
|
endif()
|
||||||
set(ENABLE_COVERAGE false CACHE BOOL "Enable code coverage")
|
set(ENABLE_COVERAGE false CACHE BOOL "Enable code coverage")
|
||||||
|
|
||||||
add_definitions(-DVITASTOR_VERSION="3.0.3")
|
add_definitions(-DVITASTOR_VERSION="3.0.6")
|
||||||
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
|
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
|
||||||
add_link_options(-fno-omit-frame-pointer)
|
add_link_options(-fno-omit-frame-pointer)
|
||||||
if (${WITH_ASAN})
|
if (${WITH_ASAN})
|
||||||
|
|||||||
@@ -173,9 +173,7 @@ void blockstore_disk_t::parse_config(std::map<std::string, std::string> & config
|
|||||||
}
|
}
|
||||||
if (data_block_size / bitmap_granularity < 8)
|
if (data_block_size / bitmap_granularity < 8)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Warning: block_size (%u) / bitmap_granularity (%u) = %u bits. "
|
throw std::runtime_error("Data block size must be at least bitmap_granularity*8");
|
||||||
"Consider using larger block_size or bitmap_granularity for better performance.\n",
|
|
||||||
data_block_size, bitmap_granularity, data_block_size / bitmap_granularity);
|
|
||||||
}
|
}
|
||||||
if (!data_csum_type)
|
if (!data_csum_type)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -26,14 +26,14 @@ class allocator_t;
|
|||||||
struct blockstore_disk_t
|
struct blockstore_disk_t
|
||||||
{
|
{
|
||||||
std::string data_device, meta_device, journal_device;
|
std::string data_device, meta_device, journal_device;
|
||||||
uint32_t data_block_size;
|
uint64_t data_block_size;
|
||||||
uint64_t cfg_journal_size, cfg_data_size;
|
uint64_t cfg_journal_size, cfg_data_size;
|
||||||
// Required write alignment and journal/metadata/data areas' location alignment
|
// Required write alignment and journal/metadata/data areas' location alignment
|
||||||
uint32_t disk_alignment = 4096;
|
uint32_t disk_alignment = 4096;
|
||||||
// Journal block size - minimum_io_size of the journal device is the best choice
|
// Journal block size - minimum_io_size of the journal device is the best choice
|
||||||
uint32_t journal_block_size = 4096;
|
uint64_t journal_block_size = 4096;
|
||||||
// Metadata block size - minimum_io_size of the metadata device is the best choice
|
// Metadata block size - minimum_io_size of the metadata device is the best choice
|
||||||
uint32_t meta_block_size = 4096;
|
uint64_t meta_block_size = 4096;
|
||||||
// Atomic write size of the data block device
|
// Atomic write size of the data block device
|
||||||
uint32_t atomic_write_size = 4096;
|
uint32_t atomic_write_size = 4096;
|
||||||
// Whether we should set RWF_ATOMIC on atomic writes
|
// Whether we should set RWF_ATOMIC on atomic writes
|
||||||
|
|||||||
@@ -301,12 +301,13 @@ int blockstore_heap_t::read_blocks(uint64_t disk_offset, uint64_t disk_size, uin
|
|||||||
heap_entry_t *wr = (heap_entry_t*)data;
|
heap_entry_t *wr = (heap_entry_t*)data;
|
||||||
if (wr->size > dsk->meta_block_size-block_offset)
|
if (wr->size > dsk->meta_block_size-block_offset)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Error: entry is too large in metadata block %u at %u (%u > max %u bytes). ",
|
fprintf(stderr, "Error: entry is too large in metadata block %u at %u (%u > max %ju bytes). ",
|
||||||
block_num, block_offset, wr->size, dsk->meta_block_size-block_offset);
|
block_num, block_offset, wr->size, dsk->meta_block_size-block_offset);
|
||||||
corrupted_block:
|
corrupted_block:
|
||||||
if (allow_corrupted)
|
if (allow_corrupted)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Metadata block is corrupted, skipping\n");
|
fprintf(stderr, "Metadata block is corrupted, skipping\n");
|
||||||
|
recheck_modified_blocks.insert(block_num);
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
@@ -357,6 +358,7 @@ corrupted_object:
|
|||||||
if (allow_corrupted)
|
if (allow_corrupted)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Entry is corrupted, skipping\n");
|
fprintf(stderr, "Entry is corrupted, skipping\n");
|
||||||
|
recheck_modified_blocks.insert(block_num);
|
||||||
block_offset += wr->size;
|
block_offset += wr->size;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -1978,24 +1980,21 @@ int blockstore_heap_t::list_objects(uint32_t pg_num, object_id min_oid, object_i
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
uint64_t stable_version = 0;
|
uint64_t stable_version = 0;
|
||||||
auto first_wr = obj;
|
iterate_with_stable(obj, UINT64_MAX, [&](heap_entry_t* wr, bool stable)
|
||||||
for (auto wr = first_wr; wr; wr = prev(wr))
|
|
||||||
{
|
{
|
||||||
if ((wr->entry_type & BS_HEAP_STABLE) || wr->type() == BS_HEAP_COMMIT || wr->type() == BS_HEAP_ROLLBACK)
|
if (stable)
|
||||||
{
|
{
|
||||||
stable_version = wr->version;
|
stable_version = wr->version;
|
||||||
break;
|
return false;
|
||||||
}
|
}
|
||||||
else
|
if (unstable_size >= unstable_alloc)
|
||||||
{
|
{
|
||||||
if (unstable_size >= unstable_alloc)
|
unstable_alloc = (!unstable_alloc ? 128 : unstable_alloc*2);
|
||||||
{
|
unstable = (obj_ver_id*)realloc_or_die(unstable, sizeof(obj_ver_id) * unstable_alloc);
|
||||||
unstable_alloc = (!unstable_alloc ? 128 : unstable_alloc*2);
|
|
||||||
unstable = (obj_ver_id*)realloc_or_die(unstable, sizeof(obj_ver_id) * unstable_alloc);
|
|
||||||
}
|
|
||||||
unstable[unstable_size++] = (obj_ver_id){ .oid = oid, .version = wr->version };
|
|
||||||
}
|
}
|
||||||
}
|
unstable[unstable_size++] = (obj_ver_id){ .oid = oid, .version = wr->version };
|
||||||
|
return true;
|
||||||
|
});
|
||||||
if (stable_version)
|
if (stable_version)
|
||||||
{
|
{
|
||||||
if (res_size >= res_alloc)
|
if (res_size >= res_alloc)
|
||||||
@@ -2292,7 +2291,9 @@ void blockstore_heap_t::apply_inflight(heap_inflight_lsn_t & inflight)
|
|||||||
}
|
}
|
||||||
if (!next)
|
if (!next)
|
||||||
{
|
{
|
||||||
|
// The last freed entry must be a deletion
|
||||||
assert(!prev);
|
assert(!prev);
|
||||||
|
assert(wr->entry_type == BS_HEAP_DELETE|BS_HEAP_STABLE);
|
||||||
auto & pg_idx = block_index[get_pg_id(wr->inode, wr->stripe)];
|
auto & pg_idx = block_index[get_pg_id(wr->inode, wr->stripe)];
|
||||||
auto & inode_idx = pg_idx[wr->inode];
|
auto & inode_idx = pg_idx[wr->inode];
|
||||||
heap_inode_map_t::iterator li_it;
|
heap_inode_map_t::iterator li_it;
|
||||||
@@ -2396,7 +2397,10 @@ void inode_map_get(void *inode_idx, heap_inode_map_t::iterator & li_it, heap_lis
|
|||||||
size_t map_n = ((size_t)inode_idx & IMAP_MALLOC_LOW_BITS);
|
size_t map_n = ((size_t)inode_idx & IMAP_MALLOC_LOW_BITS);
|
||||||
if (!map_n)
|
if (!map_n)
|
||||||
{
|
{
|
||||||
|
#pragma GCC diagnostic push
|
||||||
|
#pragma GCC diagnostic ignored "-Warray-bounds"
|
||||||
li_it = ((heap_inode_map_t*)inode_idx)->find(list_item_key(&stripe));
|
li_it = ((heap_inode_map_t*)inode_idx)->find(list_item_key(&stripe));
|
||||||
|
#pragma GCC diagnostic pop
|
||||||
li = li_it != ((heap_inode_map_t*)inode_idx)->end() ? *li_it : NULL;
|
li = li_it != ((heap_inode_map_t*)inode_idx)->end() ? *li_it : NULL;
|
||||||
}
|
}
|
||||||
else if (map_n == 1)
|
else if (map_n == 1)
|
||||||
|
|||||||
@@ -193,12 +193,12 @@ void blockstore_impl_t::loop()
|
|||||||
heap->start_block_write(block_num);
|
heap->start_block_write(block_num);
|
||||||
mb.sent = true;
|
mb.sent = true;
|
||||||
}
|
}
|
||||||
|
pending_modified_blocks.clear();
|
||||||
int ret = ringloop->submit();
|
int ret = ringloop->submit();
|
||||||
if (ret < 0)
|
if (ret < 0)
|
||||||
{
|
{
|
||||||
throw std::runtime_error(std::string("io_uring_submit: ") + strerror(-ret));
|
throw std::runtime_error(std::string("io_uring_submit: ") + strerror(-ret));
|
||||||
}
|
}
|
||||||
pending_modified_blocks.clear();
|
|
||||||
if ((initial_ring_space - ringloop->space_left()) > 0)
|
if ((initial_ring_space - ringloop->space_left()) > 0)
|
||||||
{
|
{
|
||||||
live = true;
|
live = true;
|
||||||
|
|||||||
@@ -78,6 +78,7 @@ public:
|
|||||||
// Suitable only for server SSDs with capacitors, requires disabled data and journal fsyncs
|
// Suitable only for server SSDs with capacitors, requires disabled data and journal fsyncs
|
||||||
int immediate_commit = IMMEDIATE_NONE;
|
int immediate_commit = IMMEDIATE_NONE;
|
||||||
bool inmemory_meta = false;
|
bool inmemory_meta = false;
|
||||||
|
bool skip_corrupted_meta_entries = false;
|
||||||
uint32_t meta_write_recheck_parallelism = 0;
|
uint32_t meta_write_recheck_parallelism = 0;
|
||||||
// Maximum and minimum flusher count
|
// Maximum and minimum flusher count
|
||||||
unsigned max_flusher_count = 0, min_flusher_count = 0;
|
unsigned max_flusher_count = 0, min_flusher_count = 0;
|
||||||
|
|||||||
@@ -145,7 +145,7 @@ resume_1:
|
|||||||
printf(
|
printf(
|
||||||
"Configuration stored in metadata superblock"
|
"Configuration stored in metadata superblock"
|
||||||
" (meta_block_size=%u, data_block_size=%u, bitmap_granularity=%u, data_csum_type=%u, csum_block_size=%u, meta_area_size=%ju)"
|
" (meta_block_size=%u, data_block_size=%u, bitmap_granularity=%u, data_csum_type=%u, csum_block_size=%u, meta_area_size=%ju)"
|
||||||
" differs from OSD configuration (%u/%u/%u, %u/%u, %ju).\n",
|
" differs from OSD configuration (%ju/%ju/%u, %u/%u, %ju).\n",
|
||||||
hdr->meta_block_size, hdr->data_block_size, hdr->bitmap_granularity,
|
hdr->meta_block_size, hdr->data_block_size, hdr->bitmap_granularity,
|
||||||
hdr->data_csum_type, hdr->csum_block_size, hdr->meta_area_size,
|
hdr->data_csum_type, hdr->csum_block_size, hdr->meta_area_size,
|
||||||
bs->dsk.meta_block_size, bs->dsk.data_block_size, bs->dsk.bitmap_granularity,
|
bs->dsk.meta_block_size, bs->dsk.data_block_size, bs->dsk.bitmap_granularity,
|
||||||
@@ -225,7 +225,7 @@ resume_4:
|
|||||||
{
|
{
|
||||||
// Handle result
|
// Handle result
|
||||||
uint64_t loaded = 0;
|
uint64_t loaded = 0;
|
||||||
int r = bs->heap->load_blocks(bufs[i].offset-bs->dsk.meta_block_size, bufs[i].size, bufs[i].buf, false, loaded);
|
int r = bs->heap->load_blocks(bufs[i].offset-bs->dsk.meta_block_size, bufs[i].size, bufs[i].buf, bs->skip_corrupted_meta_entries, loaded);
|
||||||
if (r != 0)
|
if (r != 0)
|
||||||
exit(1);
|
exit(1);
|
||||||
entries_loaded += loaded;
|
entries_loaded += loaded;
|
||||||
|
|||||||
@@ -28,6 +28,7 @@ void blockstore_impl_t::parse_config(blockstore_config_t & config, bool init)
|
|||||||
throttle_target_parallelism = strtoull(config["throttle_target_parallelism"].c_str(), NULL, 10);
|
throttle_target_parallelism = strtoull(config["throttle_target_parallelism"].c_str(), NULL, 10);
|
||||||
throttle_threshold_us = strtoull(config["throttle_threshold_us"].c_str(), NULL, 10);
|
throttle_threshold_us = strtoull(config["throttle_threshold_us"].c_str(), NULL, 10);
|
||||||
perfect_csum_update = config["perfect_csum_update"] == "true" || config["perfect_csum_update"] == "1" || config["perfect_csum_update"] == "yes";
|
perfect_csum_update = config["perfect_csum_update"] == "true" || config["perfect_csum_update"] == "1" || config["perfect_csum_update"] == "yes";
|
||||||
|
skip_corrupted_meta_entries = config["skip_corrupted_meta_entries"] == "true" || config["skip_corrupted_meta_entries"] == "1" || config["skip_corrupted_meta_entries"] == "yes";
|
||||||
if (config["autosync_writes"] != "")
|
if (config["autosync_writes"] != "")
|
||||||
{
|
{
|
||||||
autosync_writes = strtoull(config["autosync_writes"].c_str(), NULL, 10);
|
autosync_writes = strtoull(config["autosync_writes"].c_str(), NULL, 10);
|
||||||
|
|||||||
@@ -22,21 +22,23 @@ void blockstore_impl_t::prepare_meta_block_write(uint32_t modified_block)
|
|||||||
ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
||||||
uint8_t *buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.meta_block_size);
|
uint8_t *buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, dsk.meta_block_size);
|
||||||
data->iov = (struct iovec){ buf, (size_t)dsk.meta_block_size };
|
data->iov = (struct iovec){ buf, (size_t)dsk.meta_block_size };
|
||||||
data->callback = [this, modified_block, buf](ring_data_t *data)
|
data->callback = [this, modified_block](ring_data_t *data)
|
||||||
{
|
{
|
||||||
free(buf);
|
|
||||||
live = true;
|
live = true;
|
||||||
if (data->res != data->iov.iov_len)
|
if (data->res != data->iov.iov_len)
|
||||||
{
|
{
|
||||||
// FIXME: our state becomes corrupted after a write error. maybe do something better than just die
|
// FIXME: our state becomes corrupted after a write error. maybe do something better than just die
|
||||||
disk_error_abort("data write", data->res, data->iov.iov_len);
|
disk_error_abort("data write", data->res, data->iov.iov_len);
|
||||||
}
|
}
|
||||||
modified_blocks.erase(modified_block);
|
auto it = modified_blocks.find(modified_block);
|
||||||
|
assert(it != modified_blocks.end());
|
||||||
|
free(it->second.buf);
|
||||||
|
modified_blocks.erase(it);
|
||||||
heap->complete_block_write(modified_block);
|
heap->complete_block_write(modified_block);
|
||||||
ringloop->wakeup();
|
ringloop->wakeup();
|
||||||
};
|
};
|
||||||
io_uring_prep_writev(
|
io_uring_prep_writev(
|
||||||
sqe, dsk.meta_fd, &data->iov, 1, dsk.meta_offset + (modified_block+1)*dsk.meta_block_size
|
sqe, dsk.meta_fd, &data->iov, 1, dsk.meta_offset + ((uint64_t)modified_block+1)*dsk.meta_block_size
|
||||||
);
|
);
|
||||||
unsynced_meta_write_count++;
|
unsynced_meta_write_count++;
|
||||||
pending_modified_blocks.push_back(modified_block);
|
pending_modified_blocks.push_back(modified_block);
|
||||||
@@ -249,13 +251,12 @@ enospc:
|
|||||||
goto enospc;
|
goto enospc;
|
||||||
assert(res == 0);
|
assert(res == 0);
|
||||||
PRIV(op)->lsn = obj->lsn;
|
PRIV(op)->lsn = obj->lsn;
|
||||||
if (op->len)
|
|
||||||
heap->use_buffer_area(op->oid.inode, loc, op->len);
|
|
||||||
prepare_meta_block_write(PRIV(op)->modified_block);
|
prepare_meta_block_write(PRIV(op)->modified_block);
|
||||||
PRIV(op)->pending_ops++;
|
PRIV(op)->pending_ops++;
|
||||||
if (op->len > 0)
|
if (op->len > 0)
|
||||||
{
|
{
|
||||||
// Prepare buffered data write
|
// Prepare buffered data write
|
||||||
|
heap->use_buffer_area(op->oid.inode, loc, op->len);
|
||||||
if (dsk.inmemory_journal)
|
if (dsk.inmemory_journal)
|
||||||
{
|
{
|
||||||
memcpy((uint8_t*)buffer_area + loc, op->buf, op->len);
|
memcpy((uint8_t*)buffer_area + loc, op->buf, op->len);
|
||||||
@@ -348,6 +349,7 @@ resume_12:
|
|||||||
}
|
}
|
||||||
resume_4:
|
resume_4:
|
||||||
{
|
{
|
||||||
|
BS_SUBMIT_CHECK_SQES(1);
|
||||||
auto obj = heap->read_entry(op->oid);
|
auto obj = heap->read_entry(op->oid);
|
||||||
int res = 0;
|
int res = 0;
|
||||||
if (PRIV(op)->write_type == _REDIRECT_INTENT)
|
if (PRIV(op)->write_type == _REDIRECT_INTENT)
|
||||||
@@ -404,11 +406,12 @@ resume_6:
|
|||||||
if (ref_us > exec_us + throttle_threshold_us)
|
if (ref_us > exec_us + throttle_threshold_us)
|
||||||
{
|
{
|
||||||
// Pause reply
|
// Pause reply
|
||||||
|
PRIV(op)->pending_ops++;
|
||||||
PRIV(op)->op_state = 7;
|
PRIV(op)->op_state = 7;
|
||||||
// Remember that the timer can in theory be called right here
|
// Remember that the timer can in theory be called right here
|
||||||
tfd->set_timer_us(ref_us-exec_us, false, [this, op](int timer_id)
|
tfd->set_timer_us(ref_us-exec_us, false, [this, op](int timer_id)
|
||||||
{
|
{
|
||||||
PRIV(op)->op_state = 8;
|
PRIV(op)->pending_ops--;
|
||||||
ringloop->wakeup();
|
ringloop->wakeup();
|
||||||
});
|
});
|
||||||
return 1;
|
return 1;
|
||||||
|
|||||||
@@ -189,7 +189,7 @@ resume_1:
|
|||||||
printf(
|
printf(
|
||||||
"Configuration stored in metadata superblock"
|
"Configuration stored in metadata superblock"
|
||||||
" (meta_block_size=%u, data_block_size=%u, bitmap_granularity=%u, data_csum_type=%u, csum_block_size=%u)"
|
" (meta_block_size=%u, data_block_size=%u, bitmap_granularity=%u, data_csum_type=%u, csum_block_size=%u)"
|
||||||
" differs from OSD configuration (%u/%u/%u, %u/%u).\n",
|
" differs from OSD configuration (%ju/%ju/%u, %u/%u).\n",
|
||||||
hdr->meta_block_size, hdr->data_block_size, hdr->bitmap_granularity,
|
hdr->meta_block_size, hdr->data_block_size, hdr->bitmap_granularity,
|
||||||
hdr->data_csum_type, hdr->csum_block_size,
|
hdr->data_csum_type, hdr->csum_block_size,
|
||||||
bs->dsk.meta_block_size, bs->dsk.data_block_size, bs->dsk.bitmap_granularity,
|
bs->dsk.meta_block_size, bs->dsk.data_block_size, bs->dsk.bitmap_granularity,
|
||||||
|
|||||||
@@ -620,8 +620,7 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
|
|||||||
else if (from_journal)
|
else if (from_journal)
|
||||||
{
|
{
|
||||||
// Don't scan bitmap - journal writes don't have holes (internal bitmap)!
|
// Don't scan bitmap - journal writes don't have holes (internal bitmap)!
|
||||||
uint8_t *csum = !dsk.csum_block_size ? 0 : (clean_entry_bitmap + dsk.clean_entry_bitmap_size +
|
uint8_t *csum = !dsk.csum_block_size ? 0 : (clean_entry_bitmap + dsk.clean_entry_bitmap_size);
|
||||||
item_start/dsk.csum_block_size*(dsk.data_csum_type & 0xFF));
|
|
||||||
if (!fulfill_read(read_op, fulfilled, item_start, item_end,
|
if (!fulfill_read(read_op, fulfilled, item_start, item_end,
|
||||||
(BS_ST_BIG_WRITE | BS_ST_STABLE), 0, clean_loc + item_start, 0, csum, dyn_data))
|
(BS_ST_BIG_WRITE | BS_ST_STABLE), 0, clean_loc + item_start, 0, csum, dyn_data))
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -414,7 +414,10 @@ void etcd_state_client_t::start_etcd_watcher()
|
|||||||
}
|
}
|
||||||
// Save revision only if it's present in the message - because sometimes etcd sends something without a header, like:
|
// Save revision only if it's present in the message - because sometimes etcd sends something without a header, like:
|
||||||
// {"error": {"grpc_code": 14, "http_code": 503, "http_status": "Service Unavailable", "message": "error reading from server: EOF"}}
|
// {"error": {"grpc_code": 14, "http_code": 503, "http_status": "Service Unavailable", "message": "error reading from server: EOF"}}
|
||||||
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES && !data["result"]["header"]["revision"].is_null())
|
// Also don't save revision from the initial created: true messages because they always contain the latest revision
|
||||||
|
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES &&
|
||||||
|
!data["result"]["header"]["revision"].is_null() &&
|
||||||
|
!data["result"]["created"].bool_value())
|
||||||
{
|
{
|
||||||
// Restart watchers from the same revision number as in the last received message,
|
// Restart watchers from the same revision number as in the last received message,
|
||||||
// not from the next one to protect against revision being split into multiple messages,
|
// not from the next one to protect against revision being split into multiple messages,
|
||||||
|
|||||||
@@ -6,7 +6,7 @@
|
|||||||
#include <set>
|
#include <set>
|
||||||
|
|
||||||
#include "json11/json11.hpp"
|
#include "json11/json11.hpp"
|
||||||
#include "osd_id.h"
|
#include "object_id.h"
|
||||||
#include "timerfd_manager.h"
|
#include "timerfd_manager.h"
|
||||||
|
|
||||||
#define ETCD_CONFIG_WATCH_ID 1
|
#define ETCD_CONFIG_WATCH_ID 1
|
||||||
|
|||||||
@@ -739,14 +739,6 @@ void osd_messenger_t::check_peer_config(osd_client_t *cl)
|
|||||||
fprintf(stderr, "Connected to OSD %ju using RDMA\n", cl->osd_num);
|
fprintf(stderr, "Connected to OSD %ju using RDMA\n", cl->osd_num);
|
||||||
}
|
}
|
||||||
cl->peer_state = PEER_RDMA;
|
cl->peer_state = PEER_RDMA;
|
||||||
tfd->set_fd_handler(cl->peer_fd, false, [this](int peer_fd, int epoll_events)
|
|
||||||
{
|
|
||||||
// Do not miss the disconnection!
|
|
||||||
if (epoll_events & EPOLLRDHUP)
|
|
||||||
{
|
|
||||||
handle_peer_epoll(peer_fd, epoll_events);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
// Add the initial receive request
|
// Add the initial receive request
|
||||||
init_recv_rdma(cl);
|
init_recv_rdma(cl);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -209,7 +209,7 @@ protected:
|
|||||||
std::vector<int> read_ready_clients;
|
std::vector<int> read_ready_clients;
|
||||||
std::vector<int> write_ready_clients;
|
std::vector<int> write_ready_clients;
|
||||||
// We don't use ringloop->set_immediate here because we may have no ringloop in client :)
|
// We don't use ringloop->set_immediate here because we may have no ringloop in client :)
|
||||||
std::vector<osd_op_t*> set_immediate_ops;
|
std::deque<osd_op_t*> set_immediate_ops;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
timerfd_manager_t *tfd = NULL;
|
timerfd_manager_t *tfd = NULL;
|
||||||
|
|||||||
@@ -696,6 +696,10 @@ void osd_messenger_t::handle_rdma_events(msgr_rdma_context_t *rdma_context)
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
osd_client_t *cl = cl_it->second;
|
osd_client_t *cl = cl_it->second;
|
||||||
|
if (cl->peer_state == PEER_STOPPED)
|
||||||
|
{
|
||||||
|
continue;
|
||||||
|
}
|
||||||
auto rc = cl->rdma_conn;
|
auto rc = cl->rdma_conn;
|
||||||
if (wc[i].status != IBV_WC_SUCCESS)
|
if (wc[i].status != IBV_WC_SUCCESS)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -9,7 +9,8 @@ void osd_messenger_t::read_requests()
|
|||||||
{
|
{
|
||||||
int peer_fd = read_ready_clients[i];
|
int peer_fd = read_ready_clients[i];
|
||||||
auto cl_it = clients.find(peer_fd);
|
auto cl_it = clients.find(peer_fd);
|
||||||
if (cl_it == clients.end() || !cl_it->second || cl_it->second->read_msg.msg_iovlen)
|
if (cl_it == clients.end() || !cl_it->second || cl_it->second->read_msg.msg_iovlen ||
|
||||||
|
cl_it->second->peer_state == PEER_RDMA || cl_it->second->peer_state == PEER_RDMA_CONNECTING)
|
||||||
{
|
{
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -75,6 +76,10 @@ bool osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
|||||||
int peer_fd = cl->peer_fd;
|
int peer_fd = cl->peer_fd;
|
||||||
cl->read_msg.msg_iovlen = 0;
|
cl->read_msg.msg_iovlen = 0;
|
||||||
cl->refs--;
|
cl->refs--;
|
||||||
|
if (cl->peer_state == PEER_RDMA)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
if (cl->peer_state == PEER_STOPPED)
|
if (cl->peer_state == PEER_STOPPED)
|
||||||
{
|
{
|
||||||
if (cl->refs <= 0)
|
if (cl->refs <= 0)
|
||||||
@@ -160,8 +165,10 @@ void osd_messenger_t::clear_immediate_ops(int peer_fd)
|
|||||||
|
|
||||||
void osd_messenger_t::handle_immediate_ops()
|
void osd_messenger_t::handle_immediate_ops()
|
||||||
{
|
{
|
||||||
for (auto op: set_immediate_ops)
|
while (set_immediate_ops.size())
|
||||||
{
|
{
|
||||||
|
auto op = set_immediate_ops.front();
|
||||||
|
set_immediate_ops.pop_front();
|
||||||
if (op->op_type == OSD_OP_IN)
|
if (op->op_type == OSD_OP_IN)
|
||||||
{
|
{
|
||||||
exec_op(op);
|
exec_op(op);
|
||||||
@@ -172,7 +179,6 @@ void osd_messenger_t::handle_immediate_ops()
|
|||||||
std::function<void(osd_op_t*)>(op->callback)(op);
|
std::function<void(osd_op_t*)>(op->callback)(op);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
set_immediate_ops.clear();
|
|
||||||
}
|
}
|
||||||
|
|
||||||
bool osd_messenger_t::handle_read_buffer(osd_client_t *cl, void *curbuf, int remain)
|
bool osd_messenger_t::handle_read_buffer(osd_client_t *cl, void *curbuf, int remain)
|
||||||
|
|||||||
@@ -251,7 +251,7 @@ void osd_messenger_t::send_replies()
|
|||||||
{
|
{
|
||||||
int peer_fd = write_ready_clients[i];
|
int peer_fd = write_ready_clients[i];
|
||||||
auto cl_it = clients.find(peer_fd);
|
auto cl_it = clients.find(peer_fd);
|
||||||
if (cl_it != clients.end() && !try_send(cl_it->second))
|
if (cl_it != clients.end() && cl_it->second->peer_state != PEER_RDMA && !try_send(cl_it->second))
|
||||||
{
|
{
|
||||||
write_ready_clients.erase(write_ready_clients.begin(), write_ready_clients.begin() + i);
|
write_ready_clients.erase(write_ready_clients.begin(), write_ready_clients.begin() + i);
|
||||||
return;
|
return;
|
||||||
@@ -349,21 +349,12 @@ void osd_messenger_t::handle_send(int result, bool prev, bool more, osd_client_t
|
|||||||
#ifdef WITH_RDMA
|
#ifdef WITH_RDMA
|
||||||
if (cl->rdma_conn && !cl->outbox.size() && cl->peer_state == PEER_RDMA_CONNECTING)
|
if (cl->rdma_conn && !cl->outbox.size() && cl->peer_state == PEER_RDMA_CONNECTING)
|
||||||
{
|
{
|
||||||
// FIXME: Do something better than just forgetting the FD
|
|
||||||
// FIXME: Ignore pings during RDMA state transition
|
// FIXME: Ignore pings during RDMA state transition
|
||||||
if (log_level > 0)
|
if (log_level > 0)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Successfully connected with client %d using RDMA\n", cl->peer_fd);
|
fprintf(stderr, "Successfully connected with client %d using RDMA\n", cl->peer_fd);
|
||||||
}
|
}
|
||||||
cl->peer_state = PEER_RDMA;
|
cl->peer_state = PEER_RDMA;
|
||||||
tfd->set_fd_handler(cl->peer_fd, false, [this](int peer_fd, int epoll_events)
|
|
||||||
{
|
|
||||||
// Do not miss the disconnection!
|
|
||||||
if (epoll_events & EPOLLRDHUP)
|
|
||||||
{
|
|
||||||
handle_peer_epoll(peer_fd, epoll_events);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
// Add the initial receive request
|
// Add the initial receive request
|
||||||
init_recv_rdma(cl);
|
init_recv_rdma(cl);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -57,6 +57,7 @@ void osd_messenger_t::stop_client(int peer_fd, bool force, bool force_delete)
|
|||||||
{
|
{
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
clear_immediate_ops(peer_fd);
|
||||||
if (log_level > 0)
|
if (log_level > 0)
|
||||||
{
|
{
|
||||||
if (cl->osd_num)
|
if (cl->osd_num)
|
||||||
|
|||||||
@@ -20,6 +20,15 @@ typedef uint64_t inode_t;
|
|||||||
// Pool ID is 16 bits long
|
// Pool ID is 16 bits long
|
||||||
typedef uint32_t pool_id_t;
|
typedef uint32_t pool_id_t;
|
||||||
|
|
||||||
|
typedef uint64_t osd_num_t;
|
||||||
|
typedef uint32_t pg_num_t;
|
||||||
|
|
||||||
|
struct pool_pg_num_t
|
||||||
|
{
|
||||||
|
pool_id_t pool_id;
|
||||||
|
pg_num_t pg_num;
|
||||||
|
};
|
||||||
|
|
||||||
// 16 bytes per object/stripe id
|
// 16 bytes per object/stripe id
|
||||||
// stripe = (start of the parity stripe + peer role)
|
// stripe = (start of the parity stripe + peer role)
|
||||||
// i.e. for example (256KB + one of 0,1,2)
|
// i.e. for example (256KB + one of 0,1,2)
|
||||||
@@ -61,6 +70,21 @@ inline bool operator < (const obj_ver_id & a, const obj_ver_id & b)
|
|||||||
return a.oid < b.oid || a.oid == b.oid && a.version < b.version;
|
return a.oid < b.oid || a.oid == b.oid && a.version < b.version;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
inline bool operator < (const pool_pg_num_t & a, const pool_pg_num_t & b)
|
||||||
|
{
|
||||||
|
return a.pool_id < b.pool_id || a.pool_id == b.pool_id && a.pg_num < b.pg_num;
|
||||||
|
}
|
||||||
|
|
||||||
|
inline bool operator == (const pool_pg_num_t & a, const pool_pg_num_t & b)
|
||||||
|
{
|
||||||
|
return a.pool_id == b.pool_id && a.pg_num == b.pg_num;
|
||||||
|
}
|
||||||
|
|
||||||
|
inline bool operator != (const pool_pg_num_t & a, const pool_pg_num_t & b)
|
||||||
|
{
|
||||||
|
return a.pool_id != b.pool_id || a.pg_num != b.pg_num;
|
||||||
|
}
|
||||||
|
|
||||||
namespace std
|
namespace std
|
||||||
{
|
{
|
||||||
template<> struct hash<object_id>
|
template<> struct hash<object_id>
|
||||||
|
|||||||
@@ -4,7 +4,6 @@
|
|||||||
#pragma once
|
#pragma once
|
||||||
|
|
||||||
#include "object_id.h"
|
#include "object_id.h"
|
||||||
#include "osd_id.h"
|
|
||||||
|
|
||||||
// Magic numbers
|
// Magic numbers
|
||||||
#define SECONDARY_OSD_OP_MAGIC 0x2bd7b10325434553l
|
#define SECONDARY_OSD_OP_MAGIC 0x2bd7b10325434553l
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ includedir=${prefix}/@CMAKE_INSTALL_INCLUDEDIR@
|
|||||||
|
|
||||||
Name: Vitastor
|
Name: Vitastor
|
||||||
Description: Vitastor client library
|
Description: Vitastor client library
|
||||||
Version: 3.0.3
|
Version: 3.0.6
|
||||||
Libs: -L${libdir} -lvitastor_client
|
Libs: -L${libdir} -lvitastor_client
|
||||||
Cflags: -I${includedir}
|
Cflags: -I${includedir}
|
||||||
|
|
||||||
|
|||||||
@@ -7,7 +7,6 @@
|
|||||||
|
|
||||||
#include "json11/json11.hpp"
|
#include "json11/json11.hpp"
|
||||||
#include "object_id.h"
|
#include "object_id.h"
|
||||||
#include "osd_id.h"
|
|
||||||
#include "ringloop.h"
|
#include "ringloop.h"
|
||||||
#include <functional>
|
#include <functional>
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -442,7 +442,7 @@ struct cli_dd_t
|
|||||||
}
|
}
|
||||||
delete cur_read;
|
delete cur_read;
|
||||||
}
|
}
|
||||||
else if (!is_zero(read_op->bitmap_buf, read_op->len/iinfo.in_granularity/8))
|
else if (!is_zero(read_op->bitmap_buf, (read_op->len/iinfo.in_granularity+7)/8))
|
||||||
{
|
{
|
||||||
vitastor_read(cur_read);
|
vitastor_read(cur_read);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -3,7 +3,6 @@
|
|||||||
|
|
||||||
#include "disk_tool.h"
|
#include "disk_tool.h"
|
||||||
#include "rw_blocking.h"
|
#include "rw_blocking.h"
|
||||||
#include "osd_id.h"
|
|
||||||
#include "json_util.h"
|
#include "json_util.h"
|
||||||
#include "malloc_or_die.h"
|
#include "malloc_or_die.h"
|
||||||
|
|
||||||
@@ -332,7 +331,7 @@ void disk_tool_t::dump_meta_header(blockstore_meta_header_v3_t *hdr)
|
|||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
printf("{\"version\":\"0.5\",\"meta_block_size\":%u,\"entries\":[\n", dsk.meta_block_size);
|
printf("{\"version\":\"0.5\",\"meta_block_size\":%ju,\"entries\":[\n", dsk.meta_block_size);
|
||||||
}
|
}
|
||||||
first_entry = true;
|
first_entry = true;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,7 +4,6 @@
|
|||||||
#include "disk_tool.h"
|
#include "disk_tool.h"
|
||||||
#include "str_util.h"
|
#include "str_util.h"
|
||||||
#include "json_util.h"
|
#include "json_util.h"
|
||||||
#include "osd_id.h"
|
|
||||||
|
|
||||||
void disk_tool_t::parse_meta_reserve()
|
void disk_tool_t::parse_meta_reserve()
|
||||||
{
|
{
|
||||||
@@ -150,9 +149,12 @@ int disk_tool_t::prepare_one(std::map<std::string, std::string> options, int is_
|
|||||||
{
|
{
|
||||||
if (options["block_size"] == "")
|
if (options["block_size"] == "")
|
||||||
options["block_size"] = "1M";
|
options["block_size"] = "1M";
|
||||||
|
if (is_hybrid && options["atomic_write_size"] == "")
|
||||||
|
options["atomic_write_size"] = "0";
|
||||||
if (is_hybrid && options["throttle_small_writes"] == "")
|
if (is_hybrid && options["throttle_small_writes"] == "")
|
||||||
options["throttle_small_writes"] = "1";
|
options["throttle_small_writes"] = "1";
|
||||||
if (!is_hybrid && options.find("data_csum_type") != options.end() && options.at("data_csum_type") != "")
|
if (!is_hybrid && options.find("data_csum_type") != options.end() && options.at("data_csum_type") != "" &&
|
||||||
|
options["csum_block_size"] == "")
|
||||||
options["csum_block_size"] = "32k";
|
options["csum_block_size"] = "32k";
|
||||||
}
|
}
|
||||||
else if (!json_is_true(options["disable_data_fsync"]))
|
else if (!json_is_true(options["disable_data_fsync"]))
|
||||||
|
|||||||
@@ -212,11 +212,11 @@ void disk_tool_t::resize_init(blockstore_meta_header_v3_t *hdr)
|
|||||||
dsk.calc_lengths();
|
dsk.calc_lengths();
|
||||||
if (((new_data_offset-dsk.data_offset) % dsk.data_block_size))
|
if (((new_data_offset-dsk.data_offset) % dsk.data_block_size))
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Data alignment mismatch: old data offset is 0x%jx, new is 0x%jx, but alignment on %x should be equal\n",
|
fprintf(stderr, "Data alignment mismatch: old data offset is 0x%jx, new is 0x%jx, but alignment on %jx should be equal\n",
|
||||||
dsk.data_offset, new_data_offset, dsk.data_block_size);
|
dsk.data_offset, new_data_offset, dsk.data_block_size);
|
||||||
exit(1);
|
exit(1);
|
||||||
}
|
}
|
||||||
data_idx_diff = ((int64_t)(dsk.data_offset-new_data_offset)) / dsk.data_block_size;
|
data_idx_diff = ((int64_t)(dsk.data_offset-new_data_offset))/((int64_t)dsk.data_block_size);
|
||||||
free_first = new_data_offset > dsk.data_offset ? (new_data_offset-dsk.data_offset) / dsk.data_block_size : 0;
|
free_first = new_data_offset > dsk.data_offset ? (new_data_offset-dsk.data_offset) / dsk.data_block_size : 0;
|
||||||
free_last = (new_data_offset+new_data_len < dsk.data_offset+dsk.data_len)
|
free_last = (new_data_offset+new_data_len < dsk.data_offset+dsk.data_len)
|
||||||
? (dsk.data_offset+dsk.data_len-new_data_offset-new_data_len) / dsk.data_block_size
|
? (dsk.data_offset+dsk.data_len-new_data_offset-new_data_len) / dsk.data_block_size
|
||||||
@@ -351,7 +351,7 @@ int disk_tool_t::resize_copy_data()
|
|||||||
if (data->res != dsk.data_block_size)
|
if (data->res != dsk.data_block_size)
|
||||||
{
|
{
|
||||||
fprintf(
|
fprintf(
|
||||||
stderr, "Failed to read %u bytes at %ju from %s: %s\n", dsk.data_block_size,
|
stderr, "Failed to read %ju bytes at %ju from %s: %s\n", dsk.data_block_size,
|
||||||
dsk.data_offset + moving_blocks[i].old_loc*dsk.data_block_size, dsk.data_device.c_str(),
|
dsk.data_offset + moving_blocks[i].old_loc*dsk.data_block_size, dsk.data_device.c_str(),
|
||||||
data->res < 0 ? strerror(-data->res) : "short read"
|
data->res < 0 ? strerror(-data->res) : "short read"
|
||||||
);
|
);
|
||||||
@@ -376,7 +376,7 @@ int disk_tool_t::resize_copy_data()
|
|||||||
if (data->res != dsk.data_block_size)
|
if (data->res != dsk.data_block_size)
|
||||||
{
|
{
|
||||||
fprintf(
|
fprintf(
|
||||||
stderr, "Failed to write %u bytes at %ju to %s: %s\n", dsk.data_block_size,
|
stderr, "Failed to write %ju bytes at %ju to %s: %s\n", dsk.data_block_size,
|
||||||
dsk.data_offset + moving_blocks[i].new_loc*dsk.data_block_size, dsk.data_device.c_str(),
|
dsk.data_offset + moving_blocks[i].new_loc*dsk.data_block_size, dsk.data_device.c_str(),
|
||||||
data->res < 0 ? strerror(-data->res) : "short write"
|
data->res < 0 ? strerror(-data->res) : "short write"
|
||||||
);
|
);
|
||||||
|
|||||||
@@ -140,7 +140,10 @@ uint32_t disk_tool_t::write_osd_superblock(std::string device, json11::Json para
|
|||||||
}
|
}
|
||||||
close(fd);
|
close(fd);
|
||||||
free(buf);
|
free(buf);
|
||||||
shell_exec({ "udevadm", "trigger", "--settle", device }, "", NULL, NULL);
|
if (!test_mode)
|
||||||
|
{
|
||||||
|
shell_exec({ "udevadm", "trigger", "--settle", device }, "", NULL, NULL);
|
||||||
|
}
|
||||||
return sb_size;
|
return sb_size;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -61,7 +61,7 @@ resume_1:
|
|||||||
}
|
}
|
||||||
if (st->ientry["type"].string_value() != "file" &&
|
if (st->ientry["type"].string_value() != "file" &&
|
||||||
st->ientry["type"].string_value() != "" &&
|
st->ientry["type"].string_value() != "" &&
|
||||||
!st->set_attrs["size"].is_null())
|
st->set_attrs.find("size") != st->set_attrs.end())
|
||||||
{
|
{
|
||||||
auto cb = std::move(st->cb);
|
auto cb = std::move(st->cb);
|
||||||
cb(-EINVAL);
|
cb(-EINVAL);
|
||||||
@@ -96,8 +96,13 @@ resume_1:
|
|||||||
nfs_kv_continue_setattr(st, 2);
|
nfs_kv_continue_setattr(st, 2);
|
||||||
}, [st](int res, const std::string & cas_value)
|
}, [st](int res, const std::string & cas_value)
|
||||||
{
|
{
|
||||||
|
if ((res == 0 || res == -ENOENT && st->ino == KV_ROOT_INODE) && cas_value == st->ientry_text)
|
||||||
|
{
|
||||||
|
st->cas_res = 0;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
st->cas_res = res;
|
st->cas_res = res;
|
||||||
return (res == 0 || res == -ENOENT && st->ino == KV_ROOT_INODE) && cas_value == st->ientry_text;
|
return false;
|
||||||
});
|
});
|
||||||
return;
|
return;
|
||||||
resume_2:
|
resume_2:
|
||||||
|
|||||||
@@ -1,30 +0,0 @@
|
|||||||
// Copyright (c) Vitaliy Filippov, 2019+
|
|
||||||
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
|
|
||||||
|
|
||||||
#pragma once
|
|
||||||
|
|
||||||
#include "object_id.h"
|
|
||||||
|
|
||||||
typedef uint64_t osd_num_t;
|
|
||||||
typedef uint32_t pg_num_t;
|
|
||||||
|
|
||||||
struct pool_pg_num_t
|
|
||||||
{
|
|
||||||
pool_id_t pool_id;
|
|
||||||
pg_num_t pg_num;
|
|
||||||
};
|
|
||||||
|
|
||||||
inline bool operator < (const pool_pg_num_t & a, const pool_pg_num_t & b)
|
|
||||||
{
|
|
||||||
return a.pool_id < b.pool_id || a.pool_id == b.pool_id && a.pg_num < b.pg_num;
|
|
||||||
}
|
|
||||||
|
|
||||||
inline bool operator == (const pool_pg_num_t & a, const pool_pg_num_t & b)
|
|
||||||
{
|
|
||||||
return a.pool_id == b.pool_id && a.pg_num == b.pg_num;
|
|
||||||
}
|
|
||||||
|
|
||||||
inline bool operator != (const pool_pg_num_t & a, const pool_pg_num_t & b)
|
|
||||||
{
|
|
||||||
return a.pool_id != b.pool_id || a.pg_num != b.pg_num;
|
|
||||||
}
|
|
||||||
@@ -410,8 +410,9 @@ void osd_t::handle_primary_subop(osd_op_t *subop, osd_op_t *cur_op)
|
|||||||
if (op_data->fact_ver != 0 && op_data->fact_ver != version)
|
if (op_data->fact_ver != 0 && op_data->fact_ver != version)
|
||||||
{
|
{
|
||||||
fprintf(
|
fprintf(
|
||||||
stderr, "different fact_versions returned from %s %jx:%jx subops: %ju vs %ju\n",
|
stderr, "different fact_versions returned from %s %jx:%jx subops for a %s op: %ju vs %ju\n",
|
||||||
osd_op_names[opcode], subop->req.sec_rw.oid.inode, subop->req.sec_rw.oid.stripe, version, op_data->fact_ver
|
osd_op_names[opcode], subop->req.sec_rw.oid.inode, subop->req.sec_rw.oid.stripe,
|
||||||
|
osd_op_names[cur_op->req.hdr.opcode], version, op_data->fact_ver
|
||||||
);
|
);
|
||||||
retval = -ERANGE;
|
retval = -ERANGE;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -6,7 +6,6 @@
|
|||||||
#include <stdint.h>
|
#include <stdint.h>
|
||||||
#include <vector>
|
#include <vector>
|
||||||
#include "object_id.h"
|
#include "object_id.h"
|
||||||
#include "osd_id.h"
|
|
||||||
|
|
||||||
struct buf_len_t
|
struct buf_len_t
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -57,3 +57,7 @@ json11::Json::object osd_messenger_t::merge_configs(const json11::Json::object &
|
|||||||
{
|
{
|
||||||
return cli_config;
|
return cli_config;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void osd_messenger_t::clear_immediate_ops(int peer_fd)
|
||||||
|
{
|
||||||
|
}
|
||||||
|
|||||||
@@ -96,7 +96,7 @@ std::string str_replace(const std::string & in, const std::string & needle, cons
|
|||||||
{
|
{
|
||||||
res += in.substr(pos, p2-pos);
|
res += in.substr(pos, p2-pos);
|
||||||
res += replacement;
|
res += replacement;
|
||||||
pos = p2 + replacement.size();
|
pos = p2 + needle.size();
|
||||||
}
|
}
|
||||||
if (!pos)
|
if (!pos)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -15,6 +15,8 @@ sudo mount localhost:/ ./testdata/nfs -o port=2050,mountport=2050,nfsvers=3,soft
|
|||||||
MNT=$(pwd)/testdata/nfs
|
MNT=$(pwd)/testdata/nfs
|
||||||
trap "sudo umount -f $MNT"' || true; kill -9 $(jobs -p)' EXIT
|
trap "sudo umount -f $MNT"' || true; kill -9 $(jobs -p)' EXIT
|
||||||
|
|
||||||
|
chown 1000:1000 ./testdata/nfs
|
||||||
|
|
||||||
touch ./testdata/nfs/f1
|
touch ./testdata/nfs/f1
|
||||||
chown 1000:1000 ./testdata/nfs/f1
|
chown 1000:1000 ./testdata/nfs/f1
|
||||||
chmod 600 ./testdata/nfs/f1
|
chmod 600 ./testdata/nfs/f1
|
||||||
|
|||||||
+16
-16
@@ -15,7 +15,7 @@ trap "kill -9 $(jobs -p) || true; sudo losetup -d $LOOP1 $LOOP2"' || true' EXIT
|
|||||||
# also test prepare --hybrid :)
|
# also test prepare --hybrid :)
|
||||||
# non-vitastor random type UUID to prevent udev activation
|
# non-vitastor random type UUID to prevent udev activation
|
||||||
mount | grep '/dev type devtmpfs' || sudo mount udev /dev/ -t devtmpfs
|
mount | grep '/dev type devtmpfs' || sudo mount udev /dev/ -t devtmpfs
|
||||||
sudo build/src/disk_tool/vitastor-disk-test prepare $OFFSET_ARGS --no_init 1 --meta_reserve 1x,1M \
|
sudo -E build/src/disk_tool/vitastor-disk-test prepare $OFFSET_ARGS --no_init 1 --meta_reserve 1x,1M \
|
||||||
--block_size 131072 --osd_num 987654 --part_type_uuid 0df42ae0-3695-4395-a957-7d5ff3645c56 \
|
--block_size 131072 --osd_num 987654 --part_type_uuid 0df42ae0-3695-4395-a957-7d5ff3645c56 \
|
||||||
--hybrid --fast-devices $LOOP2 $LOOP1
|
--hybrid --fast-devices $LOOP2 $LOOP1
|
||||||
|
|
||||||
@@ -27,8 +27,8 @@ console.log(JSON.stringify([
|
|||||||
{"type":"big_write_instant","inode":"0x1000000000001","stripe":"0xc60000","ver":"10","offset":0,"len":131072,"loc":"0x18ffdc0000","bitmap":"ffffffff"}
|
{"type":"big_write_instant","inode":"0x1000000000001","stripe":"0xc60000","ver":"10","offset":0,"len":131072,"loc":"0x18ffdc0000","bitmap":"ffffffff"}
|
||||||
]));
|
]));
|
||||||
EOF
|
EOF
|
||||||
sudo build/src/disk_tool/vitastor-disk write-journal ${LOOP1}p1 < ./testdata/journal.json
|
sudo -E build/src/disk_tool/vitastor-disk write-journal ${LOOP1}p1 < ./testdata/journal.json
|
||||||
sudo build/src/disk_tool/vitastor-disk dump-journal --json --format data ${LOOP1}p1 | jq -S '[ .[] | del(.crc32, .crc32_prev) ]' > ./testdata/j2.json
|
sudo -E build/src/disk_tool/vitastor-disk dump-journal --json --format data ${LOOP1}p1 | jq -S '[ .[] | del(.crc32, .crc32_prev) ]' > ./testdata/j2.json
|
||||||
jq -S '[ .[] + {"valid":true} ]' < ./testdata/journal.json > ./testdata/j1.json
|
jq -S '[ .[] + {"valid":true} ]' < ./testdata/journal.json > ./testdata/j1.json
|
||||||
diff ./testdata/j1.json ./testdata/j2.json
|
diff ./testdata/j1.json ./testdata/j2.json
|
||||||
fi
|
fi
|
||||||
@@ -84,15 +84,15 @@ EOF
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
# also test write & dump
|
# also test write & dump
|
||||||
sudo build/src/disk_tool/vitastor-disk write-meta ${LOOP1}p1 < ./testdata/meta.json
|
sudo -E build/src/disk_tool/vitastor-disk write-meta ${LOOP1}p1 < ./testdata/meta.json
|
||||||
sudo build/src/disk_tool/vitastor-disk dump-meta ${LOOP1}p1 > ./testdata/compare.json
|
sudo -E build/src/disk_tool/vitastor-disk dump-meta ${LOOP1}p1 > ./testdata/compare.json
|
||||||
jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' < ./testdata/meta.json > ./testdata/1.json
|
jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' < ./testdata/meta.json > ./testdata/1.json
|
||||||
jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' < ./testdata/compare.json > ./testdata/2.json
|
jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' < ./testdata/compare.json > ./testdata/2.json
|
||||||
diff ./testdata/1.json ./testdata/2.json
|
diff ./testdata/1.json ./testdata/2.json
|
||||||
|
|
||||||
# move journal & meta back, data will become smaller; end indexes should be shifted by -1251
|
# move journal & meta back, data will become smaller; end indexes should be shifted by -1251
|
||||||
sudo build/src/disk_tool/vitastor-disk-test resize --move-journal '' --move-meta '' ${LOOP1}p1
|
sudo -E build/src/disk_tool/vitastor-disk-test resize --move-journal '' --move-meta '' ${LOOP1}p1
|
||||||
sudo build/src/disk_tool/vitastor-disk dump-meta ${LOOP1}p1 | jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' > ./testdata/2.json
|
sudo -E build/src/disk_tool/vitastor-disk dump-meta ${LOOP1}p1 | jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' > ./testdata/2.json
|
||||||
if [[ -n "$OLD" ]]; then
|
if [[ -n "$OLD" ]]; then
|
||||||
jq -S '. + {"entries": ([ .entries[] | (. + { "block": (.block-1251) }) ] | sort_by(.stripe))}' < ./testdata/meta.json > ./testdata/1.json
|
jq -S '. + {"entries": ([ .entries[] | (. + { "block": (.block-1251) }) ] | sort_by(.stripe))}' < ./testdata/meta.json > ./testdata/1.json
|
||||||
else
|
else
|
||||||
@@ -100,24 +100,24 @@ else
|
|||||||
fi
|
fi
|
||||||
diff ./testdata/1.json ./testdata/2.json
|
diff ./testdata/1.json ./testdata/2.json
|
||||||
if [[ -n "$OLD" ]]; then
|
if [[ -n "$OLD" ]]; then
|
||||||
sudo build/src/disk_tool/vitastor-disk dump-journal --json --format data ${LOOP1}p1 | jq -S '[ .[] | del(.crc32, .crc32_prev) ]' > ./testdata/j2.json
|
sudo -E build/src/disk_tool/vitastor-disk dump-journal --json --format data ${LOOP1}p1 | jq -S '[ .[] | del(.crc32, .crc32_prev) ]' > ./testdata/j2.json
|
||||||
jq -S '[ (.[] + {"valid":true}) | (if .type == "big_write_instant" then . + {"loc":"0x18f6160000"} else . end) ]' < ./testdata/journal.json > ./testdata/j1.json
|
jq -S '[ (.[] + {"valid":true}) | (if .type == "big_write_instant" then . + {"loc":"0x18f6160000"} else . end) ]' < ./testdata/journal.json > ./testdata/j1.json
|
||||||
diff ./testdata/j1.json ./testdata/j2.json
|
diff ./testdata/j1.json ./testdata/j2.json
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# move journal & meta out, data will become larger; end indexes should be shifted back by +1251
|
# move journal & meta out, data will become larger; end indexes should be shifted back by +1251
|
||||||
sudo build/src/disk_tool/vitastor-disk-test resize --move-journal ${LOOP2}p1 --move-meta ${LOOP2}p2 ${LOOP1}p1
|
sudo -E build/src/disk_tool/vitastor-disk-test resize --move-journal ${LOOP2}p1 --move-meta ${LOOP2}p2 ${LOOP1}p1
|
||||||
sudo build/src/disk_tool/vitastor-disk dump-meta ${LOOP1}p1 | jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' > ./testdata/2.json
|
sudo -E build/src/disk_tool/vitastor-disk dump-meta ${LOOP1}p1 | jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' > ./testdata/2.json
|
||||||
jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' < ./testdata/meta.json > ./testdata/1.json
|
jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' < ./testdata/meta.json > ./testdata/1.json
|
||||||
diff ./testdata/1.json ./testdata/2.json
|
diff ./testdata/1.json ./testdata/2.json
|
||||||
if [[ -n "$OLD" ]]; then
|
if [[ -n "$OLD" ]]; then
|
||||||
jq -S '[ .[] + {"valid":true} ]' < ./testdata/journal.json > ./testdata/j1.json
|
jq -S '[ .[] + {"valid":true} ]' < ./testdata/journal.json > ./testdata/j1.json
|
||||||
sudo build/src/disk_tool/vitastor-disk dump-journal --json --format data ${LOOP1}p1 | jq -S '[ .[] | del(.crc32, .crc32_prev) ]' > ./testdata/j2.json
|
sudo -E build/src/disk_tool/vitastor-disk dump-journal --json --format data ${LOOP1}p1 | jq -S '[ .[] | del(.crc32, .crc32_prev) ]' > ./testdata/j2.json
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# reduce data device size by exactly 128k * 99 (occupied blocks); exactly 1 should be left in place :)
|
# reduce data device size by exactly 128k * 99 (occupied blocks); exactly 1 should be left in place :)
|
||||||
sudo build/src/disk_tool/vitastor-disk-test resize --data-size $((DATA_DEV_SIZE-128*1024*99)) ${LOOP1}p1
|
sudo -E build/src/disk_tool/vitastor-disk-test resize --data-size $((DATA_DEV_SIZE-128*1024*99)) ${LOOP1}p1
|
||||||
sudo build/src/disk_tool/vitastor-disk dump-meta ${LOOP1}p1 | jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' > ./testdata/2.json
|
sudo -E build/src/disk_tool/vitastor-disk dump-meta ${LOOP1}p1 | jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' > ./testdata/2.json
|
||||||
if [[ -n "$OLD" ]]; then
|
if [[ -n "$OLD" ]]; then
|
||||||
jq -S '. + {"entries": ([ .entries[] | (. + { "block": (.block | if . > '$BLOCK_COUNT'-100 then .-('$BLOCK_COUNT'-100+1) else '$BLOCK_COUNT'-100 end) }) ]
|
jq -S '. + {"entries": ([ .entries[] | (. + { "block": (.block | if . > '$BLOCK_COUNT'-100 then .-('$BLOCK_COUNT'-100+1) else '$BLOCK_COUNT'-100 end) }) ]
|
||||||
| .[1:] + [ .[0] ]) | sort_by(.stripe)}' < ./testdata/meta.json > ./testdata/1.json
|
| .[1:] + [ .[0] ]) | sort_by(.stripe)}' < ./testdata/meta.json > ./testdata/1.json
|
||||||
@@ -128,12 +128,12 @@ fi
|
|||||||
diff ./testdata/1.json ./testdata/2.json
|
diff ./testdata/1.json ./testdata/2.json
|
||||||
if [[ -n "$OLD" ]]; then
|
if [[ -n "$OLD" ]]; then
|
||||||
jq -S '[ .[] + {"valid":true} ]' < ./testdata/journal.json > ./testdata/j1.json
|
jq -S '[ .[] + {"valid":true} ]' < ./testdata/journal.json > ./testdata/j1.json
|
||||||
sudo build/src/disk_tool/vitastor-disk dump-journal --json --format data ${LOOP1}p1 | jq -S '[ .[] | del(.crc32, .crc32_prev) ]' > ./testdata/j2.json
|
sudo -E build/src/disk_tool/vitastor-disk dump-journal --json --format data ${LOOP1}p1 | jq -S '[ .[] | del(.crc32, .crc32_prev) ]' > ./testdata/j2.json
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# extend data device size to maximum
|
# extend data device size to maximum
|
||||||
sudo build/src/disk_tool/vitastor-disk-test resize --data-size max ${LOOP1}p1
|
sudo -E build/src/disk_tool/vitastor-disk-test resize --data-size max ${LOOP1}p1
|
||||||
sudo build/src/disk_tool/vitastor-disk dump-meta ${LOOP1}p1 | jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' > ./testdata/2.json
|
sudo -E build/src/disk_tool/vitastor-disk dump-meta ${LOOP1}p1 | jq -S '. + {"entries": (.entries | sort_by(.stripe)) }' > ./testdata/2.json
|
||||||
diff ./testdata/1.json ./testdata/2.json
|
diff ./testdata/1.json ./testdata/2.json
|
||||||
|
|
||||||
format_green OK
|
format_green OK
|
||||||
|
|||||||
Reference in New Issue
Block a user