Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
74a23dcb63 | ||
|
|
93fd23b2bb | ||
|
|
eedc700b83 | ||
|
|
c8cc17dbe9 | ||
|
|
b55d406386 | ||
|
|
0c46dbd333 | ||
|
|
ed94aa52cf | ||
|
|
555ae613c2 | ||
|
|
cad6ea0360 | ||
|
|
d60709dce1 | ||
|
|
a03ffd0d73 | ||
|
|
89b76a87b6 | ||
|
|
ceba343ac0 | ||
|
|
3bc04d8250 | ||
|
|
d228fbfb68 | ||
|
|
e3c8fd28b4 | ||
|
|
d87e7d1a37 | ||
|
|
59f87c3e30 | ||
|
|
eba383f66f | ||
|
|
4e5e8822c0 | ||
|
|
60933c1d00 | ||
|
|
1ad6933953 | ||
|
|
8a250f4fca | ||
|
|
94ddf20667 | ||
|
|
5f18496c04 | ||
|
|
08a3dcd587 | ||
|
|
3c5b9d2744 | ||
|
|
cff08d2c72 | ||
|
|
1e1f395947 | ||
|
|
e6c2628960 | ||
|
|
887f7c1530 | ||
|
|
2c6bddd831 | ||
|
|
e1715c33bb | ||
|
|
2ef80bf0b8 | ||
|
|
85ba710718 | ||
|
|
c16b0e7f92 | ||
|
|
b3d388228a | ||
|
|
bcde9de7da | ||
|
|
52bc3261e9 | ||
|
|
2d42f29385 | ||
|
|
17240c6144 | ||
|
|
9e627a4414 | ||
|
|
90b1019636 | ||
|
|
df604afbd5 | ||
|
|
47c7aa62de | ||
|
|
9f2dc48d0f | ||
|
|
6d951b21fb | ||
|
|
552f28cb3e | ||
|
|
e87b6e26f7 |
+54
-18
@@ -810,6 +810,60 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
|
test_reweight_half:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_reweight_half.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_snapshot_pool2:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_snapshot_pool2.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
|
test_snapshot_read_bitmap:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
needs: build
|
||||||
|
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
||||||
|
steps:
|
||||||
|
- name: Run test
|
||||||
|
id: test
|
||||||
|
timeout-minutes: 3
|
||||||
|
run: /root/vitastor/tests/test_snapshot_read_bitmap.sh
|
||||||
|
- name: Print logs
|
||||||
|
if: always() && steps.test.outcome == 'failure'
|
||||||
|
run: |
|
||||||
|
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
||||||
|
echo "-------- $i --------"
|
||||||
|
cat $i
|
||||||
|
echo ""
|
||||||
|
done
|
||||||
|
|
||||||
test_heal_csum_32k_dmj:
|
test_heal_csum_32k_dmj:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
@@ -954,24 +1008,6 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
done
|
done
|
||||||
|
|
||||||
test_snapshot_pool2:
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
needs: build
|
|
||||||
container: ${{env.TEST_IMAGE}}:${{github.sha}}
|
|
||||||
steps:
|
|
||||||
- name: Run test
|
|
||||||
id: test
|
|
||||||
timeout-minutes: 3
|
|
||||||
run: /root/vitastor/tests/test_snapshot_pool2.sh
|
|
||||||
- name: Print logs
|
|
||||||
if: always() && steps.test.outcome == 'failure'
|
|
||||||
run: |
|
|
||||||
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
|
|
||||||
echo "-------- $i --------"
|
|
||||||
cat $i
|
|
||||||
echo ""
|
|
||||||
done
|
|
||||||
|
|
||||||
test_osd_tags:
|
test_osd_tags:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: build
|
needs: build
|
||||||
|
|||||||
+1
-1
@@ -2,6 +2,6 @@ cmake_minimum_required(VERSION 2.8.12)
|
|||||||
|
|
||||||
project(vitastor)
|
project(vitastor)
|
||||||
|
|
||||||
set(VITASTOR_VERSION "2.3.0")
|
set(VITASTOR_VERSION "2.4.1")
|
||||||
|
|
||||||
add_subdirectory(src)
|
add_subdirectory(src)
|
||||||
|
|||||||
+1
-1
@@ -36,7 +36,7 @@ RUN (echo deb http://vitastor.io/debian bookworm main > /etc/apt/sources.list.d/
|
|||||||
((echo 'Package: *'; echo 'Pin: origin "vitastor.io"'; echo 'Pin-Priority: 1000') > /etc/apt/preferences.d/vitastor.pref) && \
|
((echo 'Package: *'; echo 'Pin: origin "vitastor.io"'; echo 'Pin-Priority: 1000') > /etc/apt/preferences.d/vitastor.pref) && \
|
||||||
wget -q -O /etc/apt/trusted.gpg.d/vitastor.gpg https://vitastor.io/debian/pubkey.gpg && \
|
wget -q -O /etc/apt/trusted.gpg.d/vitastor.gpg https://vitastor.io/debian/pubkey.gpg && \
|
||||||
apt-get update && \
|
apt-get update && \
|
||||||
apt-get install -y vitastor-client && \
|
apt-get install -y vitastor-client ibverbs-providers && \
|
||||||
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||||
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||||
dpkg -x qemu-utils*.deb tmp1 && \
|
dpkg -x qemu-utils*.deb tmp1 && \
|
||||||
|
|||||||
@@ -0,0 +1,49 @@
|
|||||||
|
# Compile stage
|
||||||
|
FROM golang:bookworm AS build
|
||||||
|
|
||||||
|
ADD go.sum go.mod /app/
|
||||||
|
RUN cd /app; CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go mod download -x
|
||||||
|
ADD . /app
|
||||||
|
RUN perl -i -e '$/ = undef; while(<>) { s/\n\s*(\{\s*\n)/$1\n/g; s/\}(\s*\n\s*)else\b/$1} else/g; print; }' `find /app -name '*.go'` && \
|
||||||
|
cd /app && \
|
||||||
|
CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o vitastor-csi
|
||||||
|
|
||||||
|
# Final stage
|
||||||
|
FROM debian:bookworm
|
||||||
|
|
||||||
|
LABEL maintainers="Vitaliy Filippov <vitalif@yourcmc.ru>"
|
||||||
|
LABEL description="Vitastor CSI Driver"
|
||||||
|
|
||||||
|
ENV NODE_ID=""
|
||||||
|
ENV CSI_ENDPOINT=""
|
||||||
|
|
||||||
|
RUN apt-get update && \
|
||||||
|
apt-get install -y wget && \
|
||||||
|
(echo "APT::Install-Recommends false;" > /etc/apt/apt.conf) && \
|
||||||
|
apt-get update && \
|
||||||
|
apt-get install -y e2fsprogs xfsprogs kmod iproute2 \
|
||||||
|
# NFS mount dependencies
|
||||||
|
nfs-common netbase \
|
||||||
|
# dependencies of qemu-storage-daemon
|
||||||
|
libnuma1 liburing2 libglib2.0-0 libfuse3-3 libaio1 libzstd1 libnettle8 \
|
||||||
|
libgmp10 libhogweed6 libp11-kit0 libidn2-0 libunistring2 libtasn1-6 libpcre2-8-0 libffi8 && \
|
||||||
|
apt-get clean && \
|
||||||
|
(echo options nbd nbds_max=128 > /etc/modprobe.d/nbd.conf)
|
||||||
|
|
||||||
|
COPY --from=build /app/vitastor-csi /bin/
|
||||||
|
|
||||||
|
ADD deb /deb
|
||||||
|
|
||||||
|
RUN apt-get update && \
|
||||||
|
apt-get -y install /deb/vitastor-client_*.deb && \
|
||||||
|
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-utils_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||||
|
wget https://vitastor.io/archive/qemu/qemu-bookworm-9.2.2%2Bds-1%2Bvitastor4/qemu-block-extra_9.2.2%2Bds-1%2Bvitastor4_amd64.deb && \
|
||||||
|
dpkg -x qemu-utils*.deb tmp1 && \
|
||||||
|
dpkg -x qemu-block-extra*.deb tmp1 && \
|
||||||
|
cp -a tmp1/usr/bin/qemu-storage-daemon /usr/bin/ && \
|
||||||
|
mkdir -p /usr/lib/x86_64-linux-gnu/qemu && \
|
||||||
|
cp -a tmp1/usr/lib/x86_64-linux-gnu/qemu/block-vitastor.so /usr/lib/x86_64-linux-gnu/qemu/ && \
|
||||||
|
rm -rf tmp1 *.deb && \
|
||||||
|
apt-get clean
|
||||||
|
|
||||||
|
ENTRYPOINT ["/bin/vitastor-csi"]
|
||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
VITASTOR_VERSION ?= v2.3.0
|
VITASTOR_VERSION ?= v2.4.1
|
||||||
|
|
||||||
all: build push
|
all: build push
|
||||||
|
|
||||||
|
|||||||
@@ -49,7 +49,7 @@ spec:
|
|||||||
capabilities:
|
capabilities:
|
||||||
add: ["SYS_ADMIN"]
|
add: ["SYS_ADMIN"]
|
||||||
allowPrivilegeEscalation: true
|
allowPrivilegeEscalation: true
|
||||||
image: vitalif/vitastor-csi:v2.3.0
|
image: vitalif/vitastor-csi:v2.4.1
|
||||||
args:
|
args:
|
||||||
- "--node=$(NODE_ID)"
|
- "--node=$(NODE_ID)"
|
||||||
- "--endpoint=$(CSI_ENDPOINT)"
|
- "--endpoint=$(CSI_ENDPOINT)"
|
||||||
|
|||||||
@@ -121,7 +121,7 @@ spec:
|
|||||||
privileged: true
|
privileged: true
|
||||||
capabilities:
|
capabilities:
|
||||||
add: ["SYS_ADMIN"]
|
add: ["SYS_ADMIN"]
|
||||||
image: vitalif/vitastor-csi:v2.3.0
|
image: vitalif/vitastor-csi:v2.4.1
|
||||||
args:
|
args:
|
||||||
- "--node=$(NODE_ID)"
|
- "--node=$(NODE_ID)"
|
||||||
- "--endpoint=$(CSI_ENDPOINT)"
|
- "--endpoint=$(CSI_ENDPOINT)"
|
||||||
|
|||||||
+1
-1
@@ -5,7 +5,7 @@ package vitastor
|
|||||||
|
|
||||||
const (
|
const (
|
||||||
vitastorCSIDriverName = "csi.vitastor.io"
|
vitastorCSIDriverName = "csi.vitastor.io"
|
||||||
vitastorCSIDriverVersion = "2.3.0"
|
vitastorCSIDriverVersion = "2.4.1"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Config struct fills the parameters of request or user input
|
// Config struct fills the parameters of request or user input
|
||||||
|
|||||||
+115
-20
@@ -33,7 +33,7 @@ import (
|
|||||||
type NodeServer struct
|
type NodeServer struct
|
||||||
{
|
{
|
||||||
*Driver
|
*Driver
|
||||||
useVduse bool
|
method MountMethod
|
||||||
stateDir string
|
stateDir string
|
||||||
nfsStageDir string
|
nfsStageDir string
|
||||||
mounter mount.Interface
|
mounter mount.Interface
|
||||||
@@ -81,16 +81,23 @@ func NewNodeServer(driver *Driver) *NodeServer
|
|||||||
}
|
}
|
||||||
ns := &NodeServer{
|
ns := &NodeServer{
|
||||||
Driver: driver,
|
Driver: driver,
|
||||||
useVduse: checkVduseSupport(),
|
method: selectMountMethod(),
|
||||||
stateDir: stateDir,
|
stateDir: stateDir,
|
||||||
nfsStageDir: nfsStageDir,
|
nfsStageDir: nfsStageDir,
|
||||||
mounter: mount.New(""),
|
mounter: mount.New(""),
|
||||||
volumeLocks: make(map[string]bool),
|
volumeLocks: make(map[string]bool),
|
||||||
}
|
}
|
||||||
ns.cond = sync.NewCond(&ns.mu)
|
ns.cond = sync.NewCond(&ns.mu)
|
||||||
if (ns.useVduse)
|
if (ns.method == MOUNT_VDUSE)
|
||||||
{
|
{
|
||||||
ns.restoreVduseDaemons()
|
ns.restoreVduseDaemons()
|
||||||
|
}
|
||||||
|
else if (ns.method == MOUNT_UBLK)
|
||||||
|
{
|
||||||
|
ns.restoreUblkDaemons()
|
||||||
|
}
|
||||||
|
if (ns.method == MOUNT_VDUSE || ns.method == MOUNT_UBLK)
|
||||||
|
{
|
||||||
dur, err := time.ParseDuration(os.Getenv("RESTART_INTERVAL"))
|
dur, err := time.ParseDuration(os.Getenv("RESTART_INTERVAL"))
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
@@ -136,7 +143,14 @@ func (ns *NodeServer) restarter()
|
|||||||
for
|
for
|
||||||
{
|
{
|
||||||
<-ticker.C
|
<-ticker.C
|
||||||
ns.restoreVduseDaemons()
|
if (ns.method == MOUNT_VDUSE)
|
||||||
|
{
|
||||||
|
ns.restoreVduseDaemons()
|
||||||
|
}
|
||||||
|
else if (ns.method == MOUNT_UBLK)
|
||||||
|
{
|
||||||
|
ns.restoreUblkDaemons()
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -231,6 +245,78 @@ func (ns *NodeServer) checkVduseState(stateFile string, devs map[string]interfac
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (ns *NodeServer) restoreUblkDaemons()
|
||||||
|
{
|
||||||
|
pattern := ns.stateDir+"vitastor-ublk-*.json"
|
||||||
|
stateFiles, err := filepath.Glob(pattern)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to list %s: %v", pattern, err)
|
||||||
|
}
|
||||||
|
if (len(stateFiles) == 0)
|
||||||
|
{
|
||||||
|
return
|
||||||
|
}
|
||||||
|
for _, stateFile := range stateFiles
|
||||||
|
{
|
||||||
|
deviceNum := stateFile[len(ns.stateDir) + len("vitastor-ublk-") :]
|
||||||
|
deviceNum = deviceNum[0:len(deviceNum)-5]
|
||||||
|
ns.checkUblkState(deviceNum)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (ns *NodeServer) checkUblkState(deviceNum string)
|
||||||
|
{
|
||||||
|
// Check if the ublk daemon is still active
|
||||||
|
|
||||||
|
// Read state file
|
||||||
|
stateFile := ns.stateDir + "vitastor-ublk-" + deviceNum + ".json"
|
||||||
|
stateJSON, err := os.ReadFile(stateFile)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Warningf("error reading state file %v: %v", stateFile, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
var state DeviceState
|
||||||
|
err = json.Unmarshal(stateJSON, &state)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Warningf("state file %v contains invalid JSON (error %v): %v", stateFile, err, string(stateJSON))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Lock volume
|
||||||
|
ns.lockVolume(state.ConfigPath+":block:"+state.Image)
|
||||||
|
defer ns.unlockVolume(state.ConfigPath+":block:"+state.Image)
|
||||||
|
|
||||||
|
// Recheck state file after locking
|
||||||
|
_, err = os.ReadFile(stateFile)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Warningf("state file %v disappeared, skipping volume", stateFile)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Check if the vitastor-ublk process is still active
|
||||||
|
pidFile := ns.stateDir + "vitastor-ublk-" + deviceNum + ".pid"
|
||||||
|
exists := false
|
||||||
|
proc, err := findByPidFile(pidFile)
|
||||||
|
if (err == nil)
|
||||||
|
{
|
||||||
|
exists = proc.Signal(syscall.Signal(0)) == nil
|
||||||
|
}
|
||||||
|
if (!exists)
|
||||||
|
{
|
||||||
|
// Restart daemon
|
||||||
|
klog.Warningf("recovering UBLK device /dev/ublkb%v for volume %v", deviceNum, state.Image)
|
||||||
|
_, err = mapUblk(ns.stateDir, state.Image, state.ConfigPath, state.Readonly, "/dev/ublkb"+deviceNum)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Warningf("failed to recover ublk device for volume %v: %v", state.Image, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func (ns *NodeServer) restoreNfsDaemons()
|
func (ns *NodeServer) restoreNfsDaemons()
|
||||||
{
|
{
|
||||||
pattern := ns.stateDir+"vitastor-nfs-*.json"
|
pattern := ns.stateDir+"vitastor-nfs-*.json"
|
||||||
@@ -417,14 +503,18 @@ func (ns *NodeServer) NodeStageVolume(ctx context.Context, req *csi.NodeStageVol
|
|||||||
}
|
}
|
||||||
|
|
||||||
var devicePath, vdpaId string
|
var devicePath, vdpaId string
|
||||||
if (!ns.useVduse)
|
if (ns.method == MOUNT_UBLK)
|
||||||
{
|
{
|
||||||
devicePath, err = mapNbd(volName, ctxVars, false)
|
devicePath, err = mapUblk(ns.stateDir, volName, ctxVars["configPath"], false, "")
|
||||||
}
|
}
|
||||||
else
|
else if (ns.method == MOUNT_VDUSE)
|
||||||
{
|
{
|
||||||
devicePath, vdpaId, err = mapVduse(ns.stateDir, volName, ctxVars, false)
|
devicePath, vdpaId, err = mapVduse(ns.stateDir, volName, ctxVars, false)
|
||||||
}
|
}
|
||||||
|
else /* if (ns.method == MOUNT_NBD) */
|
||||||
|
{
|
||||||
|
devicePath, err = mapNbd(volName, ctxVars, false)
|
||||||
|
}
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
return nil, err
|
return nil, err
|
||||||
@@ -439,7 +529,8 @@ func (ns *NodeServer) NodeStageVolume(ctx context.Context, req *csi.NodeStageVol
|
|||||||
else
|
else
|
||||||
{
|
{
|
||||||
// Check existing format
|
// Check existing format
|
||||||
existingFormat, err := diskMounter.GetDiskFormat(devicePath)
|
var existingFormat string
|
||||||
|
existingFormat, err = diskMounter.GetDiskFormat(devicePath)
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
klog.Errorf("failed to get disk format for path %s, error: %v", err)
|
klog.Errorf("failed to get disk format for path %s, error: %v", err)
|
||||||
@@ -495,10 +586,6 @@ func (ns *NodeServer) NodeStageVolume(ctx context.Context, req *csi.NodeStageVol
|
|||||||
case "xfs":
|
case "xfs":
|
||||||
_, err = systemCombined("xfs_growfs", devicePath)
|
_, err = systemCombined("xfs_growfs", devicePath)
|
||||||
}
|
}
|
||||||
if (err != nil)
|
|
||||||
{
|
|
||||||
goto unmap
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
@@ -512,14 +599,18 @@ func (ns *NodeServer) NodeStageVolume(ctx context.Context, req *csi.NodeStageVol
|
|||||||
return &csi.NodeStageVolumeResponse{}, nil
|
return &csi.NodeStageVolumeResponse{}, nil
|
||||||
|
|
||||||
unmap:
|
unmap:
|
||||||
if (!ns.useVduse || len(devicePath) >= 8 && devicePath[0:8] == "/dev/nbd")
|
if (ns.method == MOUNT_UBLK)
|
||||||
{
|
{
|
||||||
unmapNbd(devicePath)
|
unmapUblk(ns.stateDir, devicePath)
|
||||||
}
|
}
|
||||||
else
|
else if (ns.method == MOUNT_VDUSE)
|
||||||
{
|
{
|
||||||
unmapVduseById(ns.stateDir, vdpaId)
|
unmapVduseById(ns.stateDir, vdpaId)
|
||||||
}
|
}
|
||||||
|
else /* if (ns.method == MOUNT_NBD) */
|
||||||
|
{
|
||||||
|
unmapNbd(devicePath)
|
||||||
|
}
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -545,7 +636,7 @@ func (ns *NodeServer) NodeUnstageVolume(ctx context.Context, req *csi.NodeUnstag
|
|||||||
defer ns.unlockVolume(ctxVars["configPath"]+":block:"+volName)
|
defer ns.unlockVolume(ctxVars["configPath"]+":block:"+volName)
|
||||||
|
|
||||||
targetPath := req.GetStagingTargetPath()
|
targetPath := req.GetStagingTargetPath()
|
||||||
devicePath, _, err := mount.GetDeviceNameFromMount(ns.mounter, targetPath)
|
devicePath, err := GetDeviceNameFromMount(targetPath)
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
if (os.IsNotExist(err))
|
if (os.IsNotExist(err))
|
||||||
@@ -582,14 +673,18 @@ func (ns *NodeServer) NodeUnstageVolume(ctx context.Context, req *csi.NodeUnstag
|
|||||||
// unmap device
|
// unmap device
|
||||||
if (len(refList) == 0)
|
if (len(refList) == 0)
|
||||||
{
|
{
|
||||||
if (!ns.useVduse)
|
if (ns.method == MOUNT_UBLK)
|
||||||
{
|
{
|
||||||
unmapNbd(devicePath)
|
unmapUblk(ns.stateDir, devicePath)
|
||||||
}
|
}
|
||||||
else
|
else if (ns.method == MOUNT_VDUSE)
|
||||||
{
|
{
|
||||||
unmapVduse(ns.stateDir, devicePath)
|
unmapVduse(ns.stateDir, devicePath)
|
||||||
}
|
}
|
||||||
|
else /* if (ns.method == MOUNT_NBD) */
|
||||||
|
{
|
||||||
|
unmapNbd(devicePath)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return &csi.NodeUnstageVolumeResponse{}, nil
|
return &csi.NodeUnstageVolumeResponse{}, nil
|
||||||
@@ -897,7 +992,7 @@ func (ns *NodeServer) NodeUnpublishVolume(ctx context.Context, req *csi.NodeUnpu
|
|||||||
}
|
}
|
||||||
|
|
||||||
targetPath := req.GetTargetPath()
|
targetPath := req.GetTargetPath()
|
||||||
devicePath, _, err := mount.GetDeviceNameFromMount(ns.mounter, targetPath)
|
devicePath, err := GetDeviceNameFromMount(targetPath)
|
||||||
if (err != nil)
|
if (err != nil)
|
||||||
{
|
{
|
||||||
if (os.IsNotExist(err))
|
if (os.IsNotExist(err))
|
||||||
|
|||||||
+205
-26
@@ -16,10 +16,20 @@ import (
|
|||||||
"syscall"
|
"syscall"
|
||||||
|
|
||||||
"k8s.io/klog"
|
"k8s.io/klog"
|
||||||
|
"k8s.io/utils/mount"
|
||||||
|
|
||||||
"google.golang.org/grpc/codes"
|
"google.golang.org/grpc/codes"
|
||||||
"google.golang.org/grpc/status"
|
"google.golang.org/grpc/status"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
type MountMethod int
|
||||||
|
|
||||||
|
const (
|
||||||
|
MOUNT_NBD MountMethod = 0
|
||||||
|
MOUNT_VDUSE MountMethod = 1
|
||||||
|
MOUNT_UBLK MountMethod = 2
|
||||||
|
)
|
||||||
|
|
||||||
func Contains(list []string, s string) bool
|
func Contains(list []string, s string) bool
|
||||||
{
|
{
|
||||||
for i := 0; i < len(list); i++
|
for i := 0; i < len(list); i++
|
||||||
@@ -32,29 +42,26 @@ func Contains(list []string, s string) bool
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
func checkVduseSupport() bool
|
func selectMountMethod() MountMethod
|
||||||
{
|
{
|
||||||
|
// Check UBLK support (ublk_drv kernel module)
|
||||||
|
if (checkModule("ublk_drv"))
|
||||||
|
{
|
||||||
|
klog.Infof("UBLK support enabled successfully")
|
||||||
|
return MOUNT_UBLK
|
||||||
|
}
|
||||||
|
klog.Errorf(
|
||||||
|
"Your host apparently has no UBLK support. UBLK support disabled."+
|
||||||
|
" For UBLK you need at least Linux 6.0 and the ublk_drv kernel module.",
|
||||||
|
)
|
||||||
// Check VDUSE support (vdpa, vduse, virtio-vdpa kernel modules)
|
// Check VDUSE support (vdpa, vduse, virtio-vdpa kernel modules)
|
||||||
vduse := true
|
vduse := true
|
||||||
for _, mod := range []string{"vdpa", "vduse", "virtio-vdpa"}
|
for _, mod := range []string{"vdpa", "vduse", "virtio-vdpa"}
|
||||||
{
|
{
|
||||||
_, err := os.Stat("/sys/module/"+mod)
|
if (!checkModule(mod))
|
||||||
if (err != nil)
|
|
||||||
{
|
{
|
||||||
if (!errors.Is(err, os.ErrNotExist))
|
vduse = false
|
||||||
{
|
break
|
||||||
klog.Errorf("failed to check /sys/module/%s: %v", mod, err)
|
|
||||||
}
|
|
||||||
c := exec.Command("/sbin/modprobe", mod)
|
|
||||||
c.Stdout = os.Stderr
|
|
||||||
c.Stderr = os.Stderr
|
|
||||||
err := c.Run()
|
|
||||||
if (err != nil)
|
|
||||||
{
|
|
||||||
klog.Errorf("/sbin/modprobe %s failed: %v", mod, err)
|
|
||||||
vduse = false
|
|
||||||
break
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Check that vdpa tool functions
|
// Check that vdpa tool functions
|
||||||
@@ -69,18 +76,38 @@ func checkVduseSupport() bool
|
|||||||
vduse = false
|
vduse = false
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!vduse)
|
if (vduse)
|
||||||
{
|
|
||||||
klog.Errorf(
|
|
||||||
"Your host apparently has no VDUSE support. VDUSE support disabled, NBD will be used to map devices."+
|
|
||||||
" For VDUSE you need at least Linux 5.15 and the following kernel modules: vdpa, virtio-vdpa, vduse.",
|
|
||||||
)
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
{
|
||||||
klog.Infof("VDUSE support enabled successfully")
|
klog.Infof("VDUSE support enabled successfully")
|
||||||
|
return MOUNT_VDUSE
|
||||||
}
|
}
|
||||||
return vduse
|
klog.Errorf(
|
||||||
|
"Your host apparently has no VDUSE support. VDUSE support disabled, NBD will be used to map devices."+
|
||||||
|
" For VDUSE you need at least Linux 5.15 and the following kernel modules: vdpa, virtio-vdpa, vduse.",
|
||||||
|
)
|
||||||
|
return MOUNT_NBD
|
||||||
|
}
|
||||||
|
|
||||||
|
func checkModule(mod string) bool
|
||||||
|
{
|
||||||
|
_, err := os.Stat("/sys/module/"+mod)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
if (!errors.Is(err, os.ErrNotExist))
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to check /sys/module/%s: %v", mod, err)
|
||||||
|
}
|
||||||
|
c := exec.Command("/sbin/modprobe", mod)
|
||||||
|
c.Stdout = os.Stderr
|
||||||
|
c.Stderr = os.Stderr
|
||||||
|
err := c.Run()
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("/sbin/modprobe %s failed: %v", mod, err)
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
func mapNbd(volName string, ctxVars map[string]string, readonly bool) (string, error)
|
func mapNbd(volName string, ctxVars map[string]string, readonly bool) (string, error)
|
||||||
@@ -217,6 +244,7 @@ func mapVduse(stateDir string, volName string, ctxVars map[string]string, readon
|
|||||||
stateJSON, _ := json.Marshal(&DeviceState{
|
stateJSON, _ := json.Marshal(&DeviceState{
|
||||||
ConfigPath: ctxVars["configPath"],
|
ConfigPath: ctxVars["configPath"],
|
||||||
VdpaId: vdpaId,
|
VdpaId: vdpaId,
|
||||||
|
|
||||||
Image: volName,
|
Image: volName,
|
||||||
Blockdev: blockdev,
|
Blockdev: blockdev,
|
||||||
Readonly: readonly,
|
Readonly: readonly,
|
||||||
@@ -309,6 +337,117 @@ func unmapVduseById(stateDir, vdpaId string)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func mapUblk(stateDir string, volName string, configPath string, readonly bool, recoverDev string) (string, error)
|
||||||
|
{
|
||||||
|
pidFile := ""
|
||||||
|
if (recoverDev != "")
|
||||||
|
{
|
||||||
|
if (len(recoverDev) < 10 || recoverDev[0:10] != "/dev/ublkb")
|
||||||
|
{
|
||||||
|
return "", fmt.Errorf("recover: %s does not start with /dev/ublkb", recoverDev)
|
||||||
|
}
|
||||||
|
pidFile = stateDir + "vitastor-ublk-" + recoverDev[10:] + ".pid"
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
pidFd, err := os.CreateTemp(stateDir, "vitastor-tmp-*.pid")
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
pidFile = pidFd.Name()
|
||||||
|
pidFd.Close()
|
||||||
|
}
|
||||||
|
// Map device via vitastor-ublk
|
||||||
|
args := []string{
|
||||||
|
"map", "--image", volName, "--pidfile", pidFile,
|
||||||
|
}
|
||||||
|
if (configPath != "")
|
||||||
|
{
|
||||||
|
args = append(args, "--config_path", configPath)
|
||||||
|
}
|
||||||
|
if (readonly)
|
||||||
|
{
|
||||||
|
args = append(args, "--readonly")
|
||||||
|
}
|
||||||
|
if (recoverDev != "")
|
||||||
|
{
|
||||||
|
args = append(args, "--recover", recoverDev)
|
||||||
|
}
|
||||||
|
stdout, stderr, err := system("/usr/bin/vitastor-ublk", args...)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
devicePath := strings.TrimSpace(string(stdout))
|
||||||
|
if (devicePath == "")
|
||||||
|
{
|
||||||
|
return "", fmt.Errorf("vitastor-ublk did not return the name of the device. output: %s", stderr)
|
||||||
|
}
|
||||||
|
if (len(devicePath) >= 10 && devicePath[0:10] == "/dev/ublkb")
|
||||||
|
{
|
||||||
|
// Generate state file
|
||||||
|
devNum := devicePath[10:]
|
||||||
|
pidNew := stateDir + "vitastor-ublk-" + devNum + ".pid"
|
||||||
|
if (pidFile != pidNew)
|
||||||
|
{
|
||||||
|
err := os.Rename(pidFile, pidNew)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("Failed to rename PID file %s to %s: %v", pidFile, pidNew, err)
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
pidFile = pidNew
|
||||||
|
}
|
||||||
|
}
|
||||||
|
stateFile := stateDir + "vitastor-ublk-" + devNum + ".json"
|
||||||
|
stateJSON, _ := json.Marshal(&DeviceState{
|
||||||
|
ConfigPath: configPath,
|
||||||
|
Image: volName,
|
||||||
|
Readonly: readonly,
|
||||||
|
PidFile: pidFile,
|
||||||
|
})
|
||||||
|
err = os.WriteFile(stateFile, stateJSON, 0600)
|
||||||
|
if (err == nil)
|
||||||
|
{
|
||||||
|
klog.Infof("Attached volume %s via UBLK as %s", volName, devicePath)
|
||||||
|
return devicePath, nil
|
||||||
|
}
|
||||||
|
os.Remove(stateFile)
|
||||||
|
}
|
||||||
|
killErr := killByPidFile(pidFile)
|
||||||
|
if (killErr != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("Failed to kill started vitastor-ublk: %v", killErr)
|
||||||
|
}
|
||||||
|
os.Remove(pidFile)
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
func unmapUblk(stateDir, devicePath string)
|
||||||
|
{
|
||||||
|
if (len(devicePath) < 10 || devicePath[0:10] != "/dev/ublkb")
|
||||||
|
{
|
||||||
|
klog.Errorf("%s does not start with /dev/ublkb", devicePath)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
unmapOut, unmapErr := exec.Command("/usr/bin/vitastor-ublk", "unmap", devicePath).CombinedOutput()
|
||||||
|
if (unmapErr != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to unmap UBLK device %s: %s, error: %v", devicePath, unmapOut, unmapErr)
|
||||||
|
}
|
||||||
|
for _, ext := range []string{"json", "pid"}
|
||||||
|
{
|
||||||
|
fn := stateDir + "vitastor-ublk-" + devicePath[10:] + "." + ext
|
||||||
|
err := os.Remove(fn)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
klog.Errorf("failed to remove %s: %v", fn, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func system(program string, args ...string) ([]byte, []byte, error)
|
func system(program string, args ...string) ([]byte, []byte, error)
|
||||||
{
|
{
|
||||||
klog.Infof("Running "+program+" "+strings.Join(args, " "))
|
klog.Infof("Running "+program+" "+strings.Join(args, " "))
|
||||||
@@ -340,3 +479,43 @@ func systemCombined(program string, args ...string) ([]byte, error)
|
|||||||
}
|
}
|
||||||
return out.Bytes(), nil
|
return out.Bytes(), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func GetDeviceNameFromMount(mountPath string) (string, error)
|
||||||
|
{
|
||||||
|
// Use /proc/self/mountinfo to correctly parse bind mounts for block device files
|
||||||
|
mps, err := mount.ParseMountInfo("/proc/self/mountinfo")
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
slTarget, err := filepath.EvalSymlinks(mountPath)
|
||||||
|
if (err != nil)
|
||||||
|
{
|
||||||
|
slTarget = mountPath
|
||||||
|
}
|
||||||
|
|
||||||
|
device := ""
|
||||||
|
for _, mp := range mps
|
||||||
|
{
|
||||||
|
if (mp.MountPoint == slTarget)
|
||||||
|
{
|
||||||
|
device = mp.Source
|
||||||
|
if (device[0] != '/' && mp.Root != "/")
|
||||||
|
{
|
||||||
|
// Handle {Source=udev Root=/vdb MountPoint=/var/lib/kubelet/tralaleylo/tralala}
|
||||||
|
for _, other := range mps
|
||||||
|
{
|
||||||
|
if (other.Root == "/" && other.Source == mp.Source)
|
||||||
|
{
|
||||||
|
device = other.MountPoint + mp.Root
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return device, nil
|
||||||
|
}
|
||||||
|
|||||||
Vendored
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
vitastor (2.3.0-1) unstable; urgency=medium
|
vitastor (2.4.1-1) unstable; urgency=medium
|
||||||
|
|
||||||
* Bugfixes
|
* Bugfixes
|
||||||
|
|
||||||
|
|||||||
Vendored
+1
-1
@@ -4,7 +4,7 @@ Priority: optional
|
|||||||
Maintainer: Vitaliy Filippov <vitalif@yourcmc.ru>
|
Maintainer: Vitaliy Filippov <vitalif@yourcmc.ru>
|
||||||
Build-Depends: debhelper, g++ (>= 8), libstdc++6 (>= 8),
|
Build-Depends: debhelper, g++ (>= 8), libstdc++6 (>= 8),
|
||||||
linux-libc-dev, libgoogle-perftools-dev, libjerasure-dev, libgf-complete-dev,
|
linux-libc-dev, libgoogle-perftools-dev, libjerasure-dev, libgf-complete-dev,
|
||||||
libibverbs-dev, libisal-dev, cmake, pkg-config, libnl-3-dev, libnl-genl-3-dev,
|
libibverbs-dev, librdmacm-dev, libisal-dev, cmake, pkg-config, libnl-3-dev, libnl-genl-3-dev,
|
||||||
node-bindings <!nocheck>, node-gyp, node-nan
|
node-bindings <!nocheck>, node-gyp, node-nan
|
||||||
Standards-Version: 4.5.0
|
Standards-Version: 4.5.0
|
||||||
Homepage: https://vitastor.io/
|
Homepage: https://vitastor.io/
|
||||||
|
|||||||
+1
-1
@@ -3,7 +3,7 @@
|
|||||||
FROM debian:bookworm
|
FROM debian:bookworm
|
||||||
|
|
||||||
ADD etc/apt /etc/apt/
|
ADD etc/apt /etc/apt/
|
||||||
RUN apt-get update && apt-get -y install vitastor udev systemd qemu-system-x86 qemu-system-common qemu-block-extra qemu-utils jq nfs-common && apt-get clean
|
RUN apt-get update && apt-get -y install vitastor ibverbs-providers udev systemd qemu-system-x86 qemu-system-common qemu-block-extra qemu-utils jq nfs-common && apt-get clean
|
||||||
ADD sleep.sh /usr/bin/
|
ADD sleep.sh /usr/bin/
|
||||||
ADD install.sh /usr/bin/
|
ADD install.sh /usr/bin/
|
||||||
ADD scripts /opt/scripts/
|
ADD scripts /opt/scripts/
|
||||||
|
|||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
VITASTOR_VERSION ?= v2.3.0
|
VITASTOR_VERSION ?= v2.4.1
|
||||||
|
|
||||||
all: build push
|
all: build push
|
||||||
|
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
#
|
#
|
||||||
|
|
||||||
# Desired Vitastor version
|
# Desired Vitastor version
|
||||||
VITASTOR_VERSION=v2.3.0
|
VITASTOR_VERSION=v2.4.1
|
||||||
|
|
||||||
# Additional arguments for all containers
|
# Additional arguments for all containers
|
||||||
# For example, you may want to specify a custom logging driver here
|
# For example, you may want to specify a custom logging driver here
|
||||||
|
|||||||
@@ -26,9 +26,9 @@ at Vitastor Kubernetes operator: https://github.com/Antilles7227/vitastor-operat
|
|||||||
The instruction is very simple.
|
The instruction is very simple.
|
||||||
|
|
||||||
1. Download a Docker image of the desired version: \
|
1. Download a Docker image of the desired version: \
|
||||||
`docker pull vitalif/vitastor:v2.3.0`
|
`docker pull vitalif/vitastor:v2.4.1`
|
||||||
2. Install scripts to the host system: \
|
2. Install scripts to the host system: \
|
||||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v2.3.0 install.sh`
|
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v2.4.1 install.sh`
|
||||||
3. Reload udev rules: \
|
3. Reload udev rules: \
|
||||||
`udevadm control --reload-rules`
|
`udevadm control --reload-rules`
|
||||||
|
|
||||||
|
|||||||
@@ -25,9 +25,9 @@ Vitastor можно установить в Docker/Podman. При этом etcd,
|
|||||||
Инструкция по установке максимально простая.
|
Инструкция по установке максимально простая.
|
||||||
|
|
||||||
1. Скачайте Docker-образ желаемой версии: \
|
1. Скачайте Docker-образ желаемой версии: \
|
||||||
`docker pull vitalif/vitastor:v2.3.0`
|
`docker pull vitalif/vitastor:v2.4.1`
|
||||||
2. Установите скрипты в хост-систему командой: \
|
2. Установите скрипты в хост-систему командой: \
|
||||||
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v2.3.0 install.sh`
|
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v2.4.1 install.sh`
|
||||||
3. Перезагрузите правила udev: \
|
3. Перезагрузите правила udev: \
|
||||||
`udevadm control --reload-rules`
|
`udevadm control --reload-rules`
|
||||||
|
|
||||||
|
|||||||
@@ -30,7 +30,7 @@ volume_backend_name = vitastor-testcluster
|
|||||||
image_volume_cache_enabled = True
|
image_volume_cache_enabled = True
|
||||||
volume_clear = none
|
volume_clear = none
|
||||||
vitastor_etcd_address = 192.168.7.2:2379
|
vitastor_etcd_address = 192.168.7.2:2379
|
||||||
vitastor_etcd_prefix =
|
vitastor_etcd_prefix = /vitastor
|
||||||
vitastor_config_path = /etc/vitastor/vitastor.conf
|
vitastor_config_path = /etc/vitastor/vitastor.conf
|
||||||
vitastor_pool_id = 1
|
vitastor_pool_id = 1
|
||||||
image_upload_use_cinder_backend = True
|
image_upload_use_cinder_backend = True
|
||||||
|
|||||||
@@ -29,7 +29,7 @@ volume_backend_name = vitastor-testcluster
|
|||||||
image_volume_cache_enabled = True
|
image_volume_cache_enabled = True
|
||||||
volume_clear = none
|
volume_clear = none
|
||||||
vitastor_etcd_address = 192.168.7.2:2379
|
vitastor_etcd_address = 192.168.7.2:2379
|
||||||
vitastor_etcd_prefix =
|
vitastor_etcd_prefix = /vitastor
|
||||||
vitastor_config_path = /etc/vitastor/vitastor.conf
|
vitastor_config_path = /etc/vitastor/vitastor.conf
|
||||||
vitastor_pool_id = 1
|
vitastor_pool_id = 1
|
||||||
image_upload_use_cinder_backend = True
|
image_upload_use_cinder_backend = True
|
||||||
|
|||||||
@@ -100,12 +100,14 @@ List images (only matching `<glob>` pattern(s) if passed).
|
|||||||
Options:
|
Options:
|
||||||
|
|
||||||
```
|
```
|
||||||
|
--exact Do not match glob patterns as names, select only exact name matches.
|
||||||
-p|--pool POOL Filter images by pool ID or name
|
-p|--pool POOL Filter images by pool ID or name
|
||||||
-l|--long Also report allocated size and I/O statistics
|
-l|--long Also report allocated size and I/O statistics
|
||||||
--del Also include delete operation statistics
|
--del Also include delete operation statistics
|
||||||
--sort FIELD Sort by specified field (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)
|
--sort FIELD Sort by specified field (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)
|
||||||
-r|--reverse Sort in descending order
|
-r|--reverse Sort in descending order
|
||||||
-n|--count N Only list first N items
|
-n|--count N Only list first N items
|
||||||
|
--tree Show image snapshot/clone tree
|
||||||
```
|
```
|
||||||
|
|
||||||
Example output:
|
Example output:
|
||||||
|
|||||||
@@ -102,12 +102,14 @@ kaveri 2/1 32 0 B 10 G 0 B 100% 0%
|
|||||||
Опции:
|
Опции:
|
||||||
|
|
||||||
```
|
```
|
||||||
|
--exact Не применять ФС-шаблоны к именам, выводить только точные совпадения
|
||||||
-p|--pool POOL Фильтровать образы по пулу (ID или имени)
|
-p|--pool POOL Фильтровать образы по пулу (ID или имени)
|
||||||
-l|--long Также выводить статистику занятого места и ввода-вывода
|
-l|--long Также выводить статистику занятого места и ввода-вывода
|
||||||
--del Также выводить статистику операций удаления
|
--del Также выводить статистику операций удаления
|
||||||
--sort FIELD Сортировать по заданному полю (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)
|
--sort FIELD Сортировать по заданному полю (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)
|
||||||
-r|--reverse Сортировать в обратном порядке
|
-r|--reverse Сортировать в обратном порядке
|
||||||
-n|--count N Показывать только первые N записей
|
-n|--count N Показывать только первые N записей
|
||||||
|
--tree Вывести снапшоты и клоны в виде дерева
|
||||||
```
|
```
|
||||||
|
|
||||||
Пример вывода:
|
Пример вывода:
|
||||||
|
|||||||
@@ -73,6 +73,8 @@ Options (automatic mode):
|
|||||||
--max_other 10%
|
--max_other 10%
|
||||||
Use disks for OSD data even if they already have non-Vitastor partitions,
|
Use disks for OSD data even if they already have non-Vitastor partitions,
|
||||||
but only if these take up no more than this percent of disk space.
|
but only if these take up no more than this percent of disk space.
|
||||||
|
--dry-run
|
||||||
|
Check and print new OSD count for each disk but do not actually create them.
|
||||||
```
|
```
|
||||||
|
|
||||||
Options (single-device mode):
|
Options (single-device mode):
|
||||||
|
|||||||
@@ -74,6 +74,8 @@ vitastor-disk - инструмент командной строки для уп
|
|||||||
--max_other 10%
|
--max_other 10%
|
||||||
Использовать диски под данные OSD, даже если на них уже есть не-Vitastor-овые
|
Использовать диски под данные OSD, даже если на них уже есть не-Vitastor-овые
|
||||||
разделы, но только в случае, если они занимают не более данного процента диска.
|
разделы, но только в случае, если они занимают не более данного процента диска.
|
||||||
|
--dry-run
|
||||||
|
Проверить и вывести число новых OSD для каждого диска, но не создавать их.
|
||||||
```
|
```
|
||||||
|
|
||||||
Опции для режима одного OSD:
|
Опции для режима одного OSD:
|
||||||
|
|||||||
+1
-1
@@ -15,7 +15,7 @@ function get_osd_tree(global_config, state)
|
|||||||
const stat = state.osd.stats[osd_num];
|
const stat = state.osd.stats[osd_num];
|
||||||
const osd_cfg = state.config.osd[osd_num];
|
const osd_cfg = state.config.osd[osd_num];
|
||||||
let reweight = osd_cfg == null ? 1 : Number(osd_cfg.reweight);
|
let reweight = osd_cfg == null ? 1 : Number(osd_cfg.reweight);
|
||||||
if (isNaN(reweight) || reweight < 0 || reweight > 0)
|
if (isNaN(reweight) || reweight < 0 || reweight > 1)
|
||||||
reweight = 1;
|
reweight = 1;
|
||||||
if (stat && stat.size && reweight && (state.osd.state[osd_num] || Number(stat.time) >= down_time ||
|
if (stat && stat.size && reweight && (state.osd.state[osd_num] || Number(stat.time) >= down_time ||
|
||||||
osd_cfg && osd_cfg.noout))
|
osd_cfg && osd_cfg.noout))
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "vitastor-mon",
|
"name": "vitastor-mon",
|
||||||
"version": "2.3.0",
|
"version": "2.4.1",
|
||||||
"description": "Vitastor SDS monitor service",
|
"description": "Vitastor SDS monitor service",
|
||||||
"main": "mon-main.js",
|
"main": "mon-main.js",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
|
|||||||
@@ -52,15 +52,16 @@ async function run()
|
|||||||
process.exit(1);
|
process.exit(1);
|
||||||
}
|
}
|
||||||
const etcds = (config.etcd_address instanceof Array ? config.etcd_address : (''+config.etcd_address).split(/,/))
|
const etcds = (config.etcd_address instanceof Array ? config.etcd_address : (''+config.etcd_address).split(/,/))
|
||||||
.map(s => (''+s).replace(/^https?:\/\/\[?|\]?(:\d+)?(\/.*)?$/g, '').toLowerCase());
|
.map(s => (''+s).replace(/^https?:\/\/|(:\d+)?(\/.*)?$/g, '').replace(/^\[(.*)\]$/, '$1').toLowerCase());
|
||||||
const num = select_local_etcd(etcds);
|
const num = select_local_etcd(etcds);
|
||||||
if (num < 0)
|
if (num < 0)
|
||||||
{
|
{
|
||||||
console.log('No matching IPs in etcd_address from '+config_path);
|
console.log('No matching IPs in etcd_address from '+config_path);
|
||||||
process.exit(0);
|
process.exit(0);
|
||||||
}
|
}
|
||||||
|
const etcd_url = 'http://' + (etcds[num].indexOf(':') >= 0 ? '['+etcds[num]+']' : etcds[num]);
|
||||||
const etcd_name = 'etcd'+etcds[num].replace(/[^0-9a-z_]/ig, '_');
|
const etcd_name = 'etcd'+etcds[num].replace(/[^0-9a-z_]/ig, '_');
|
||||||
const etcd_cluster = etcds.map(e => `etcd${e.replace(/[^0-9a-z_]/ig, '_')}=http://${e}:2380`).join(',');
|
const etcd_cluster = etcds.map(e => `etcd${e.replace(/[^0-9a-z_]/ig, '_')}=${etcd_url}:2380`).join(',');
|
||||||
if (in_docker)
|
if (in_docker)
|
||||||
{
|
{
|
||||||
let etcd_conf = fs.readFileSync("/etc/vitastor/etcd.conf", { encoding: 'utf-8' });
|
let etcd_conf = fs.readFileSync("/etc/vitastor/etcd.conf", { encoding: 'utf-8' });
|
||||||
@@ -83,8 +84,8 @@ Wants=network-online.target local-fs.target time-sync.target
|
|||||||
Restart=always
|
Restart=always
|
||||||
Environment=GOGC=50
|
Environment=GOGC=50
|
||||||
ExecStart=etcd --name ${etcd_name} --data-dir /var/lib/etcd/vitastor \\
|
ExecStart=etcd --name ${etcd_name} --data-dir /var/lib/etcd/vitastor \\
|
||||||
--snapshot-count 10000 --advertise-client-urls http://${etcds[num]}:2379 --listen-client-urls http://${etcds[num]}:2379 \\
|
--snapshot-count 10000 --advertise-client-urls ${etcd_url}:2379 --listen-client-urls ${etcd_url}:2379 \\
|
||||||
--initial-advertise-peer-urls http://${etcds[num]}:2380 --listen-peer-urls http://${etcds[num]}:2380 \\
|
--initial-advertise-peer-urls ${etcd_url}:2380 --listen-peer-urls ${etcd_url}:2380 \\
|
||||||
--initial-cluster-token vitastor-etcd-1 --initial-cluster ${etcd_cluster} \\
|
--initial-cluster-token vitastor-etcd-1 --initial-cluster ${etcd_cluster} \\
|
||||||
--initial-cluster-state new --max-txn-ops=100000 --max-request-bytes=104857600 \\
|
--initial-cluster-state new --max-txn-ops=100000 --max-request-bytes=104857600 \\
|
||||||
--auto-compaction-retention=10 --auto-compaction-mode=revision
|
--auto-compaction-retention=10 --auto-compaction-mode=revision
|
||||||
|
|||||||
@@ -276,6 +276,10 @@ function sum_inode_stats(state, prev_stats)
|
|||||||
}
|
}
|
||||||
for (const pool_id in osd_diff.inode_stats)
|
for (const pool_id in osd_diff.inode_stats)
|
||||||
{
|
{
|
||||||
|
if (!inode_stats[pool_id])
|
||||||
|
{
|
||||||
|
continue;
|
||||||
|
}
|
||||||
for (const inode_num in prev_stats.osd_diff[osd].inode_stats[pool_id])
|
for (const inode_num in prev_stats.osd_diff[osd].inode_stats[pool_id])
|
||||||
{
|
{
|
||||||
inode_stats[pool_id][inode_num] = inode_stats[pool_id][inode_num] || inode_stub();
|
inode_stats[pool_id][inode_num] = inode_stats[pool_id][inode_num] || inode_stub();
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "vitastor",
|
"name": "vitastor",
|
||||||
"version": "2.3.0",
|
"version": "2.4.1",
|
||||||
"description": "Low-level native bindings to Vitastor client library",
|
"description": "Low-level native bindings to Vitastor client library",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"keywords": [
|
"keywords": [
|
||||||
|
|||||||
@@ -499,4 +499,55 @@ sub rename_volume
|
|||||||
return "${storeid}:${base_name}${target_volname}";
|
return "${storeid}:${base_name}${target_volname}";
|
||||||
}
|
}
|
||||||
|
|
||||||
|
sub _monkey_patch_qemu_blockdev_options
|
||||||
|
{
|
||||||
|
my ($cfg, $volid, $machine_version, $options) = @_;
|
||||||
|
my ($storeid, $volname) = PVE::Storage::parse_volume_id($volid);
|
||||||
|
|
||||||
|
my $scfg = PVE::Storage::storage_config($cfg, $storeid);
|
||||||
|
|
||||||
|
my $plugin = PVE::Storage::Plugin->lookup($scfg->{type});
|
||||||
|
|
||||||
|
my ($vtype) = $plugin->parse_volname($volname);
|
||||||
|
die "cannot use volume of type '$vtype' as a QEMU blockdevice\n"
|
||||||
|
if $vtype ne 'images' && $vtype ne 'iso' && $vtype ne 'import';
|
||||||
|
|
||||||
|
return $plugin->qemu_blockdev_options($scfg, $storeid, $volname, $machine_version, $options);
|
||||||
|
}
|
||||||
|
|
||||||
|
sub qemu_blockdev_options
|
||||||
|
{
|
||||||
|
my ($class, $scfg, $storeid, $volname, $machine_version, $options) = @_;
|
||||||
|
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
|
||||||
|
my ($vtype, $name, $vmid) = $class->parse_volname($volname);
|
||||||
|
$name .= '@'.$options->{'snapshot-name'} if $options->{'snapshot-name'};
|
||||||
|
if ($scfg->{vitastor_nbd})
|
||||||
|
{
|
||||||
|
my $mapped = run_cli($scfg, [ 'ls' ], binary => '/usr/bin/vitastor-nbd');
|
||||||
|
my ($kerneldev) = grep { $mapped->{$_}->{image} eq $prefix.$name } keys %$mapped;
|
||||||
|
die "Image not mapped via NBD" if !$kerneldev;
|
||||||
|
return { driver => 'host_device', filename => $kerneldev };
|
||||||
|
}
|
||||||
|
my $blockdev = {
|
||||||
|
driver => 'vitastor',
|
||||||
|
image => $prefix.$name,
|
||||||
|
};
|
||||||
|
if ($scfg->{vitastor_config_path})
|
||||||
|
{
|
||||||
|
$blockdev->{'config-path'} = $scfg->{vitastor_config_path};
|
||||||
|
}
|
||||||
|
if ($scfg->{vitastor_etcd_address})
|
||||||
|
{
|
||||||
|
# FIXME This is the only exception: etcd_address -> etcd_host for qemu
|
||||||
|
$blockdev->{'etcd-host'} = $scfg->{vitastor_etcd_address};
|
||||||
|
}
|
||||||
|
if ($scfg->{vitastor_etcd_prefix})
|
||||||
|
{
|
||||||
|
$blockdev->{'etcd-prefix'} = $scfg->{vitastor_etcd_prefix};
|
||||||
|
}
|
||||||
|
return $blockdev;
|
||||||
|
}
|
||||||
|
|
||||||
|
*PVE::Storage::qemu_blockdev_options = *_monkey_patch_qemu_blockdev_options;
|
||||||
|
|
||||||
1;
|
1;
|
||||||
|
|||||||
@@ -50,7 +50,7 @@ from cinder.volume import configuration
|
|||||||
from cinder.volume import driver
|
from cinder.volume import driver
|
||||||
from cinder.volume import volume_utils
|
from cinder.volume import volume_utils
|
||||||
|
|
||||||
VITASTOR_VERSION = '2.3.0'
|
VITASTOR_VERSION = '2.4.1'
|
||||||
|
|
||||||
LOG = logging.getLogger(__name__)
|
LOG = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 2.3.0
|
Version: 2.4.1
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-2.3.0.el7.tar.gz
|
Source0: vitastor-2.4.1.el7.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: devtoolset-9-gcc-c++
|
BuildRequires: devtoolset-9-gcc-c++
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 2.3.0
|
Version: 2.4.1
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-2.3.0.el8.tar.gz
|
Source0: vitastor-2.4.1.el8.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: gcc-toolset-9-gcc-c++
|
BuildRequires: gcc-toolset-9-gcc-c++
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
Name: vitastor
|
Name: vitastor
|
||||||
Version: 2.3.0
|
Version: 2.4.1
|
||||||
Release: 1%{?dist}
|
Release: 1%{?dist}
|
||||||
Summary: Vitastor, a fast software-defined clustered block storage
|
Summary: Vitastor, a fast software-defined clustered block storage
|
||||||
|
|
||||||
License: Vitastor Network Public License 1.1
|
License: Vitastor Network Public License 1.1
|
||||||
URL: https://vitastor.io/
|
URL: https://vitastor.io/
|
||||||
Source0: vitastor-2.3.0.el9.tar.gz
|
Source0: vitastor-2.4.1.el9.tar.gz
|
||||||
|
|
||||||
BuildRequires: gperftools-devel
|
BuildRequires: gperftools-devel
|
||||||
BuildRequires: gcc-c++
|
BuildRequires: gcc-c++
|
||||||
|
|||||||
+1
-1
@@ -20,7 +20,7 @@ if("${CMAKE_INSTALL_PREFIX}" MATCHES "^/usr/local/?$")
|
|||||||
set(CMAKE_INSTALL_RPATH "${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}")
|
set(CMAKE_INSTALL_RPATH "${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}")
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
add_definitions(-DVITASTOR_VERSION="2.3.0")
|
add_definitions(-DVITASTOR_VERSION="2.4.1")
|
||||||
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
|
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
|
||||||
add_link_options(-fno-omit-frame-pointer)
|
add_link_options(-fno-omit-frame-pointer)
|
||||||
if (${WITH_ASAN})
|
if (${WITH_ASAN})
|
||||||
|
|||||||
@@ -229,6 +229,8 @@ void blockstore_impl_t::loop()
|
|||||||
{
|
{
|
||||||
// Mark journal sector writes as submitted
|
// Mark journal sector writes as submitted
|
||||||
journal.sector_info[s].submit_id = 0;
|
journal.sector_info[s].submit_id = 0;
|
||||||
|
if (journal.sector_info[s].submit_id)
|
||||||
|
journal.sector_info[s].written = true;
|
||||||
}
|
}
|
||||||
journal.submitting_sectors.clear();
|
journal.submitting_sectors.clear();
|
||||||
if ((initial_ring_space - ringloop->space_left()) > 0)
|
if ((initial_ring_space - ringloop->space_left()) > 0)
|
||||||
|
|||||||
@@ -178,7 +178,7 @@ void blockstore_impl_t::prepare_journal_sector_write(int cur_sector, blockstore_
|
|||||||
// Caller must ensure availability of an SQE
|
// Caller must ensure availability of an SQE
|
||||||
assert(sqe != NULL);
|
assert(sqe != NULL);
|
||||||
ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
||||||
journal.sector_info[cur_sector].written = true;
|
// <written> flag will be set at the moment of actual submission
|
||||||
journal.sector_info[cur_sector].submit_id = ++journal.submit_id;
|
journal.sector_info[cur_sector].submit_id = ++journal.submit_id;
|
||||||
assert(journal.submit_id != 0); // check overflow
|
assert(journal.submit_id != 0); // check overflow
|
||||||
journal.submitting_sectors.push_back(cur_sector);
|
journal.submitting_sectors.push_back(cur_sector);
|
||||||
|
|||||||
@@ -26,6 +26,7 @@
|
|||||||
|
|
||||||
#include "blockstore.h"
|
#include "blockstore.h"
|
||||||
#include "epoll_manager.h"
|
#include "epoll_manager.h"
|
||||||
|
#include "malloc_or_die.h"
|
||||||
#include "json11/json11.hpp"
|
#include "json11/json11.hpp"
|
||||||
#include "fio_headers.h"
|
#include "fio_headers.h"
|
||||||
|
|
||||||
@@ -37,6 +38,8 @@ struct bs_data
|
|||||||
/* The list of completed io_u structs. */
|
/* The list of completed io_u structs. */
|
||||||
std::vector<io_u*> completed;
|
std::vector<io_u*> completed;
|
||||||
int op_n = 0, inflight = 0;
|
int op_n = 0, inflight = 0;
|
||||||
|
bool ec = false;
|
||||||
|
bool imm = true;
|
||||||
bool last_sync = false;
|
bool last_sync = false;
|
||||||
bool trace = false;
|
bool trace = false;
|
||||||
};
|
};
|
||||||
@@ -45,6 +48,7 @@ struct bs_options
|
|||||||
{
|
{
|
||||||
int __pad;
|
int __pad;
|
||||||
char *json_config = NULL;
|
char *json_config = NULL;
|
||||||
|
int ec = 0;
|
||||||
int trace = 0;
|
int trace = 0;
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -58,6 +62,16 @@ static struct fio_option options[] = {
|
|||||||
.category = FIO_OPT_C_ENGINE,
|
.category = FIO_OPT_C_ENGINE,
|
||||||
.group = FIO_OPT_G_FILENAME,
|
.group = FIO_OPT_G_FILENAME,
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
.name = "ec",
|
||||||
|
.lname = "Use EC write method",
|
||||||
|
.type = FIO_OPT_BOOL,
|
||||||
|
.off1 = offsetof(struct bs_options, ec),
|
||||||
|
.help = "Use EC write method",
|
||||||
|
.def = "0",
|
||||||
|
.category = FIO_OPT_C_ENGINE,
|
||||||
|
.group = FIO_OPT_G_FILENAME,
|
||||||
|
},
|
||||||
{
|
{
|
||||||
.name = "bs_trace",
|
.name = "bs_trace",
|
||||||
.lname = "trace",
|
.lname = "trace",
|
||||||
@@ -88,6 +102,7 @@ static int bs_setup(struct thread_data *td)
|
|||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
td->io_ops_data = bsd;
|
td->io_ops_data = bsd;
|
||||||
|
bsd->ec = o->ec;
|
||||||
|
|
||||||
if (!td->files_index)
|
if (!td->files_index)
|
||||||
{
|
{
|
||||||
@@ -148,6 +163,8 @@ static int bs_init(struct thread_data *td)
|
|||||||
bsd->ringloop = new ring_loop_t(RINGLOOP_DEFAULT_SIZE);
|
bsd->ringloop = new ring_loop_t(RINGLOOP_DEFAULT_SIZE);
|
||||||
bsd->epmgr = new epoll_manager_t(bsd->ringloop);
|
bsd->epmgr = new epoll_manager_t(bsd->ringloop);
|
||||||
bsd->bs = new blockstore_t(config, bsd->ringloop, bsd->epmgr->tfd);
|
bsd->bs = new blockstore_t(config, bsd->ringloop, bsd->epmgr->tfd);
|
||||||
|
bsd->imm = config.find("immediate_commit") == config.end() ||
|
||||||
|
config["immediate_commit"] == "all";
|
||||||
while (1)
|
while (1)
|
||||||
{
|
{
|
||||||
bsd->ringloop->loop();
|
bsd->ringloop->loop();
|
||||||
@@ -203,7 +220,7 @@ static enum fio_q_status bs_queue(struct thread_data *td, struct io_u *io)
|
|||||||
};
|
};
|
||||||
break;
|
break;
|
||||||
case DDIR_WRITE:
|
case DDIR_WRITE:
|
||||||
op->opcode = BS_OP_WRITE_STABLE;
|
op->opcode = bsd->ec ? BS_OP_WRITE : BS_OP_WRITE_STABLE;
|
||||||
op->buf = io->xfer_buf;
|
op->buf = io->xfer_buf;
|
||||||
op->oid = {
|
op->oid = {
|
||||||
.inode = 1,
|
.inode = 1,
|
||||||
@@ -212,16 +229,57 @@ static enum fio_q_status bs_queue(struct thread_data *td, struct io_u *io)
|
|||||||
op->version = 0; // assign automatically
|
op->version = 0; // assign automatically
|
||||||
op->offset = io->offset % bsd->bs->get_block_size();
|
op->offset = io->offset % bsd->bs->get_block_size();
|
||||||
op->len = io->xfer_buflen;
|
op->len = io->xfer_buflen;
|
||||||
op->callback = [io, n = bsd->op_n](blockstore_op_t *op)
|
if (bsd->ec)
|
||||||
{
|
{
|
||||||
io->error = op->retval < 0 ? -op->retval : 0;
|
op->callback = [io, n = bsd->op_n](blockstore_op_t *op)
|
||||||
bs_data *bsd = (bs_data*)io->engine_data;
|
{
|
||||||
bsd->inflight--;
|
bs_data *bsd = (bs_data*)io->engine_data;
|
||||||
bsd->completed.push_back(io);
|
if (bsd->trace)
|
||||||
if (bsd->trace)
|
printf("--- OP_WRITE %zx n=%d retval=%d\n", (size_t)op, n, op->retval);
|
||||||
printf("--- OP_WRITE %zx n=%d retval=%d\n", (size_t)op, n, op->retval);
|
if (op->retval < 0)
|
||||||
delete op;
|
{
|
||||||
};
|
io->error = op->retval < 0 ? -op->retval : 0;
|
||||||
|
bsd->inflight--;
|
||||||
|
bsd->completed.push_back(io);
|
||||||
|
delete op;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
auto stab_op = new blockstore_op_t;
|
||||||
|
stab_op->opcode = BS_OP_STABLE;
|
||||||
|
stab_op->buf = malloc_or_die(sizeof(obj_ver_id));
|
||||||
|
obj_ver_id *ver = (obj_ver_id *)stab_op->buf;
|
||||||
|
ver[0].oid = op->oid;
|
||||||
|
ver[0].version = op->version;
|
||||||
|
stab_op->len = 1;
|
||||||
|
stab_op->callback = [io, n](blockstore_op_t *op)
|
||||||
|
{
|
||||||
|
bs_data *bsd = (bs_data*)io->engine_data;
|
||||||
|
if (bsd->trace)
|
||||||
|
printf("--- OP_STABLE %zx n=%d retval=%d\n", (size_t)op, n, op->retval);
|
||||||
|
io->error = op->retval < 0 ? -op->retval : 0;
|
||||||
|
bsd->inflight--;
|
||||||
|
bsd->completed.push_back(io);
|
||||||
|
delete op;
|
||||||
|
};
|
||||||
|
bsd->bs->enqueue_op(stab_op);
|
||||||
|
delete op;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
op->callback = [io, n = bsd->op_n](blockstore_op_t *op)
|
||||||
|
{
|
||||||
|
bs_data *bsd = (bs_data*)io->engine_data;
|
||||||
|
if (bsd->trace)
|
||||||
|
printf("--- OP_WRITE_STABLE %zx n=%d retval=%d\n", (size_t)op, n, op->retval);
|
||||||
|
io->error = op->retval < 0 ? -op->retval : 0;
|
||||||
|
bsd->inflight--;
|
||||||
|
bsd->completed.push_back(io);
|
||||||
|
delete op;
|
||||||
|
};
|
||||||
|
}
|
||||||
bsd->last_sync = false;
|
bsd->last_sync = false;
|
||||||
break;
|
break;
|
||||||
case DDIR_SYNC:
|
case DDIR_SYNC:
|
||||||
|
|||||||
@@ -1317,7 +1317,8 @@ int cluster_client_t::try_send(cluster_op_t *op, int i)
|
|||||||
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->cur_inode));
|
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->cur_inode));
|
||||||
auto pg_it = pool_cfg.pg_config.find(part->pg_num);
|
auto pg_it = pool_cfg.pg_config.find(part->pg_num);
|
||||||
if (pg_it != pool_cfg.pg_config.end() &&
|
if (pg_it != pool_cfg.pg_config.end() &&
|
||||||
!pg_it->second.pause && pg_it->second.cur_primary)
|
!pg_it->second.pause && pg_it->second.cur_primary &&
|
||||||
|
(pg_it->second.cur_state & PG_ACTIVE))
|
||||||
{
|
{
|
||||||
osd_num_t primary_osd = pg_it->second.cur_primary;
|
osd_num_t primary_osd = pg_it->second.cur_primary;
|
||||||
if (pool_cfg.local_reads != POOL_LOCAL_READ_PRIMARY &&
|
if (pool_cfg.local_reads != POOL_LOCAL_READ_PRIMARY &&
|
||||||
|
|||||||
@@ -127,7 +127,7 @@ public:
|
|||||||
ring_consumer_t consumer;
|
ring_consumer_t consumer;
|
||||||
std::vector<std::function<void(void)>> on_ready_hooks;
|
std::vector<std::function<void(void)>> on_ready_hooks;
|
||||||
int list_retry_timeout_id = -1;
|
int list_retry_timeout_id = -1;
|
||||||
timespec list_retry_time;
|
timespec list_retry_time = {};
|
||||||
std::vector<inode_list_t*> lists;
|
std::vector<inode_list_t*> lists;
|
||||||
std::multimap<osd_num_t, osd_op_t*> raw_ops;
|
std::multimap<osd_num_t, osd_op_t*> raw_ops;
|
||||||
int continuing_ops = 0;
|
int continuing_ops = 0;
|
||||||
|
|||||||
@@ -31,7 +31,7 @@ struct inode_list_pg_t
|
|||||||
osd_num_t cur_primary = 0;
|
osd_num_t cur_primary = 0;
|
||||||
int state = 0;
|
int state = 0;
|
||||||
int inflight_ops = 0;
|
int inflight_ops = 0;
|
||||||
timespec wait_until;
|
timespec wait_until = {};
|
||||||
std::vector<inode_list_osd_t> list_osds;
|
std::vector<inode_list_osd_t> list_osds;
|
||||||
|
|
||||||
bool has_unstable = false;
|
bool has_unstable = false;
|
||||||
@@ -177,6 +177,9 @@ void cluster_client_t::retry_start_pg_listing(inode_list_pg_t *pg)
|
|||||||
if (tv.tv_sec < pg->wait_until.tv_sec ||
|
if (tv.tv_sec < pg->wait_until.tv_sec ||
|
||||||
tv.tv_sec == pg->wait_until.tv_sec && tv.tv_nsec < pg->wait_until.tv_nsec)
|
tv.tv_sec == pg->wait_until.tv_sec && tv.tv_nsec < pg->wait_until.tv_nsec)
|
||||||
{
|
{
|
||||||
|
// Ensure that the retry timer is set
|
||||||
|
set_list_retry_timeout(pg->wait_until.tv_sec*1000 - tv.tv_sec*1000 +
|
||||||
|
(pg->wait_until.tv_nsec - tv.tv_nsec)/1000000, pg->wait_until);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1167,7 +1167,7 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
|
|||||||
if (!cur_primary || !value["state"].is_array() || !state ||
|
if (!cur_primary || !value["state"].is_array() || !state ||
|
||||||
(state & PG_OFFLINE) && state != PG_OFFLINE ||
|
(state & PG_OFFLINE) && state != PG_OFFLINE ||
|
||||||
(state & PG_PEERING) && state != PG_PEERING ||
|
(state & PG_PEERING) && state != PG_PEERING ||
|
||||||
(state & PG_INCOMPLETE) && state != PG_INCOMPLETE)
|
(state & PG_INCOMPLETE) && state != PG_INCOMPLETE && state != (PG_INCOMPLETE|PG_HAS_INVALID))
|
||||||
{
|
{
|
||||||
fprintf(stderr, "Unexpected pool %u PG %u state in etcd: primary=%ju, state=%s\n", pool_id, pg_num, cur_primary, value["state"].dump().c_str());
|
fprintf(stderr, "Unexpected pool %u PG %u state in etcd: primary=%ju, state=%s\n", pool_id, pg_num, cur_primary, value["state"].dump().c_str());
|
||||||
return;
|
return;
|
||||||
|
|||||||
@@ -109,6 +109,7 @@ bool osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
|||||||
if (!handle_read_buffer(cl, cl->in_buf, result))
|
if (!handle_read_buffer(cl, cl->in_buf, result))
|
||||||
{
|
{
|
||||||
clear_immediate_ops(peer_fd);
|
clear_immediate_ops(peer_fd);
|
||||||
|
handle_immediate_ops();
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -122,6 +123,7 @@ bool osd_messenger_t::handle_read(int result, osd_client_t *cl)
|
|||||||
if (!handle_finished_read(cl))
|
if (!handle_finished_read(cl))
|
||||||
{
|
{
|
||||||
clear_immediate_ops(peer_fd);
|
clear_immediate_ops(peer_fd);
|
||||||
|
handle_immediate_ops();
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -140,7 +142,7 @@ void osd_messenger_t::clear_immediate_ops(int peer_fd)
|
|||||||
size_t i = 0, j = 0;
|
size_t i = 0, j = 0;
|
||||||
while (i < set_immediate_ops.size())
|
while (i < set_immediate_ops.size())
|
||||||
{
|
{
|
||||||
if (set_immediate_ops[i]->peer_fd == peer_fd)
|
if (set_immediate_ops[i]->peer_fd == peer_fd && set_immediate_ops[i]->op_type == OSD_OP_IN)
|
||||||
{
|
{
|
||||||
delete set_immediate_ops[i];
|
delete set_immediate_ops[i];
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -52,6 +52,7 @@ void osd_messenger_t::stop_client(int peer_fd, bool force, bool force_delete)
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
osd_client_t *cl = it->second;
|
osd_client_t *cl = it->second;
|
||||||
|
// FIXME: This 'force' flag is probably an ugly reenterability hack - check its logic and maybe remove it
|
||||||
if (cl->peer_state == PEER_CONNECTING && !force || cl->peer_state == PEER_STOPPED)
|
if (cl->peer_state == PEER_CONNECTING && !force || cl->peer_state == PEER_STOPPED)
|
||||||
{
|
{
|
||||||
return;
|
return;
|
||||||
@@ -73,6 +74,7 @@ void osd_messenger_t::stop_client(int peer_fd, bool force, bool force_delete)
|
|||||||
}
|
}
|
||||||
// First set state to STOPPED so another stop_client() call doesn't try to free it again
|
// First set state to STOPPED so another stop_client() call doesn't try to free it again
|
||||||
cl->refs++;
|
cl->refs++;
|
||||||
|
int prev_state = cl->peer_state;
|
||||||
cl->peer_state = PEER_STOPPED;
|
cl->peer_state = PEER_STOPPED;
|
||||||
if (cl->osd_num)
|
if (cl->osd_num)
|
||||||
{
|
{
|
||||||
@@ -113,10 +115,12 @@ void osd_messenger_t::stop_client(int peer_fd, bool force, bool force_delete)
|
|||||||
// Break PG locks
|
// Break PG locks
|
||||||
break_pg_locks(cl->in_osd_num);
|
break_pg_locks(cl->in_osd_num);
|
||||||
}
|
}
|
||||||
if (cl->osd_num)
|
if (cl->osd_num && prev_state != PEER_CONNECTING)
|
||||||
{
|
{
|
||||||
// Then repeer PGs because cancel_op() callbacks can try to perform
|
// Then repeer PGs because cancel_op() callbacks can try to perform
|
||||||
// some actions and we need correct PG states to not do something silly
|
// some actions and we need correct PG states to not do something silly
|
||||||
|
// PEER_CONNECTING has neither 'just dropped the connection' nor 'just connected'
|
||||||
|
// so do not repeer on it.
|
||||||
repeer_pgs(cl->osd_num);
|
repeer_pgs(cl->osd_num);
|
||||||
}
|
}
|
||||||
// Find the item again because it can be invalidated at this point
|
// Find the item again because it can be invalidated at this point
|
||||||
|
|||||||
@@ -272,6 +272,8 @@ const char *help_text =
|
|||||||
" --dev_num N\n"
|
" --dev_num N\n"
|
||||||
" Use the specified device /dev/nbdN instead of automatic selection (alternative syntax\n"
|
" Use the specified device /dev/nbdN instead of automatic selection (alternative syntax\n"
|
||||||
" to /dev/nbdN positional parameter).\n"
|
" to /dev/nbdN positional parameter).\n"
|
||||||
|
" --readonly\n"
|
||||||
|
" Make the device read-only.\n"
|
||||||
" --foreground 1\n"
|
" --foreground 1\n"
|
||||||
" Stay in foreground, do not daemonize.\n"
|
" Stay in foreground, do not daemonize.\n"
|
||||||
"\n"
|
"\n"
|
||||||
@@ -372,7 +374,7 @@ public:
|
|||||||
else if (args[i][0] == '-' && args[i][1] == '-')
|
else if (args[i][0] == '-' && args[i][1] == '-')
|
||||||
{
|
{
|
||||||
const char *opt = args[i]+2;
|
const char *opt = args[i]+2;
|
||||||
cfg[opt] = !strcmp(opt, "json") || !strcmp(opt, "all") ||
|
cfg[opt] = !strcmp(opt, "json") || !strcmp(opt, "all") || !strcmp(opt, "readonly") ||
|
||||||
!strcmp(opt, "force") || i == narg-1 ? "1" : args[++i];
|
!strcmp(opt, "force") || i == narg-1 ? "1" : args[++i];
|
||||||
}
|
}
|
||||||
else if (pos == 0)
|
else if (pos == 0)
|
||||||
@@ -695,20 +697,6 @@ help:
|
|||||||
ringloop->loop();
|
ringloop->loop();
|
||||||
ringloop->wait();
|
ringloop->wait();
|
||||||
}
|
}
|
||||||
stop = false;
|
|
||||||
cluster_op_t *close_sync = new cluster_op_t;
|
|
||||||
close_sync->opcode = OSD_OP_SYNC;
|
|
||||||
close_sync->callback = [&stop](cluster_op_t *op)
|
|
||||||
{
|
|
||||||
stop = true;
|
|
||||||
delete op;
|
|
||||||
};
|
|
||||||
cli->execute(close_sync);
|
|
||||||
while (!stop)
|
|
||||||
{
|
|
||||||
ringloop->loop();
|
|
||||||
ringloop->wait();
|
|
||||||
}
|
|
||||||
cli->flush();
|
cli->flush();
|
||||||
delete cli;
|
delete cli;
|
||||||
delete epmgr;
|
delete epmgr;
|
||||||
@@ -912,7 +900,7 @@ protected:
|
|||||||
int r, nbd = open(path, O_RDWR), qd_fd;
|
int r, nbd = open(path, O_RDWR), qd_fd;
|
||||||
if (nbd < 0)
|
if (nbd < 0)
|
||||||
{
|
{
|
||||||
write(notifyfd[1], &errno, sizeof(errno));
|
(void)write(notifyfd[1], &errno, sizeof(errno));
|
||||||
exit(1);
|
exit(1);
|
||||||
}
|
}
|
||||||
r = ioctl(nbd, NBD_SET_SOCK, sockfd[1]);
|
r = ioctl(nbd, NBD_SET_SOCK, sockfd[1]);
|
||||||
@@ -954,7 +942,7 @@ protected:
|
|||||||
close(qd_fd);
|
close(qd_fd);
|
||||||
// Notify parent
|
// Notify parent
|
||||||
errno = 0;
|
errno = 0;
|
||||||
write(notifyfd[1], &errno, sizeof(errno));
|
(void)write(notifyfd[1], &errno, sizeof(errno));
|
||||||
close(notifyfd[1]);
|
close(notifyfd[1]);
|
||||||
close(sockfd[0]);
|
close(sockfd[0]);
|
||||||
if (bg)
|
if (bg)
|
||||||
@@ -971,11 +959,11 @@ protected:
|
|||||||
ioctl(nbd, NBD_CLEAR_SOCK);
|
ioctl(nbd, NBD_CLEAR_SOCK);
|
||||||
exit(0);
|
exit(0);
|
||||||
end_close:
|
end_close:
|
||||||
write(notifyfd[1], &errno, sizeof(errno));
|
(void)write(notifyfd[1], &errno, sizeof(errno));
|
||||||
close(nbd);
|
close(nbd);
|
||||||
exit(2);
|
exit(2);
|
||||||
end_unmap:
|
end_unmap:
|
||||||
write(notifyfd[1], &errno, sizeof(errno));
|
(void)write(notifyfd[1], &errno, sizeof(errno));
|
||||||
ioctl(nbd, NBD_CLEAR_SOCK);
|
ioctl(nbd, NBD_CLEAR_SOCK);
|
||||||
close(nbd);
|
close(nbd);
|
||||||
exit(3);
|
exit(3);
|
||||||
|
|||||||
+31
-25
@@ -40,12 +40,14 @@ const char *help_text =
|
|||||||
" Make the device read-only.\n"
|
" Make the device read-only.\n"
|
||||||
" --hdd\n"
|
" --hdd\n"
|
||||||
" Mark the device as rotational.\n"
|
" Mark the device as rotational.\n"
|
||||||
" --logfile /path/to/log/file.txt\n"
|
|
||||||
" Write log messages to the specified file instead of dropping them (in background mode)\n"
|
|
||||||
" or printing them to the standard output (in foreground mode).\n"
|
|
||||||
" --dev_num N\n"
|
" --dev_num N\n"
|
||||||
" Use the specified device /dev/ublkbN instead of automatic selection (alternative syntax\n"
|
" Use the specified device /dev/ublkbN instead of automatic selection (alternative syntax\n"
|
||||||
" to /dev/ublkbN positional parameter).\n"
|
" to /dev/ublkbN positional parameter).\n"
|
||||||
|
" --pidfile /run/ublk_pid_file.pid\n"
|
||||||
|
" Write process ID to the specified file.\n"
|
||||||
|
" --logfile /path/to/log/file.txt\n"
|
||||||
|
" Write log messages to the specified file instead of dropping them (in background mode)\n"
|
||||||
|
" or printing them to the standard output (in foreground mode).\n"
|
||||||
" --foreground 1\n"
|
" --foreground 1\n"
|
||||||
" Stay in foreground, do not daemonize.\n"
|
" Stay in foreground, do not daemonize.\n"
|
||||||
"\n"
|
"\n"
|
||||||
@@ -79,6 +81,7 @@ protected:
|
|||||||
inode_watch_t *watch = NULL;
|
inode_watch_t *watch = NULL;
|
||||||
|
|
||||||
std::string logfile = "/dev/null";
|
std::string logfile = "/dev/null";
|
||||||
|
std::string pidfile;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
ublk_server()
|
ublk_server()
|
||||||
@@ -300,10 +303,8 @@ help:
|
|||||||
load_module();
|
load_module();
|
||||||
|
|
||||||
bool bg = cfg["foreground"].is_null();
|
bool bg = cfg["foreground"].is_null();
|
||||||
if (cfg["logfile"].string_value() != "")
|
logfile = cfg["logfile"].string_value();
|
||||||
{
|
pidfile = cfg["pidfile"].string_value();
|
||||||
logfile = cfg["logfile"].string_value();
|
|
||||||
}
|
|
||||||
|
|
||||||
open_control();
|
open_control();
|
||||||
if (recover)
|
if (recover)
|
||||||
@@ -331,11 +332,13 @@ help:
|
|||||||
close(notifyfd[0]);
|
close(notifyfd[0]);
|
||||||
}
|
}
|
||||||
start_device(recover);
|
start_device(recover);
|
||||||
|
if (pidfile != "")
|
||||||
|
write_pid();
|
||||||
if (bg)
|
if (bg)
|
||||||
{
|
{
|
||||||
daemonize_reopen_stdio();
|
daemonize_reopen_stdio();
|
||||||
int ok = 0;
|
int ok = 0;
|
||||||
write(notifyfd[1], &ok, sizeof(ok));
|
(void)write(notifyfd[1], &ok, sizeof(ok));
|
||||||
close(notifyfd[1]);
|
close(notifyfd[1]);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
@@ -346,19 +349,6 @@ help:
|
|||||||
ringloop->loop();
|
ringloop->loop();
|
||||||
ringloop->wait();
|
ringloop->wait();
|
||||||
}
|
}
|
||||||
cluster_op_t *close_sync = new cluster_op_t;
|
|
||||||
close_sync->opcode = OSD_OP_SYNC;
|
|
||||||
close_sync->callback = [this](cluster_op_t *op)
|
|
||||||
{
|
|
||||||
stop = true;
|
|
||||||
delete op;
|
|
||||||
};
|
|
||||||
cli->execute(close_sync);
|
|
||||||
while (!stop)
|
|
||||||
{
|
|
||||||
ringloop->loop();
|
|
||||||
ringloop->wait();
|
|
||||||
}
|
|
||||||
cli->flush();
|
cli->flush();
|
||||||
delete cli;
|
delete cli;
|
||||||
delete epmgr;
|
delete epmgr;
|
||||||
@@ -390,7 +380,7 @@ help:
|
|||||||
// Parent - check status
|
// Parent - check status
|
||||||
close(notifyfd[1]);
|
close(notifyfd[1]);
|
||||||
int child_errno = 1;
|
int child_errno = 1;
|
||||||
read(notifyfd[0], &child_errno, sizeof(child_errno));
|
(void)read(notifyfd[0], &child_errno, sizeof(child_errno));
|
||||||
if (!child_errno)
|
if (!child_errno)
|
||||||
printf("/dev/ublkb%d\n", ublk_dev.dev_id);
|
printf("/dev/ublkb%d\n", ublk_dev.dev_id);
|
||||||
exit(child_errno);
|
exit(child_errno);
|
||||||
@@ -412,6 +402,22 @@ help:
|
|||||||
fprintf(stderr, "Warning: Failed to chdir into /\n");
|
fprintf(stderr, "Warning: Failed to chdir into /\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void write_pid()
|
||||||
|
{
|
||||||
|
int fd = open(pidfile.c_str(), O_WRONLY|O_CREAT|O_TRUNC, 0666);
|
||||||
|
if (fd < 0)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Failed to create pid file %s: %s (code %d)\n", pidfile.c_str(), strerror(errno), errno);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
auto pid = std::to_string(getpid());
|
||||||
|
if (write(fd, pid.c_str(), pid.size()) < 0)
|
||||||
|
{
|
||||||
|
fprintf(stderr, "Failed to write pid to %s: %s (code %d)\n", pidfile.c_str(), strerror(errno), errno);
|
||||||
|
}
|
||||||
|
close(fd);
|
||||||
|
}
|
||||||
|
|
||||||
json11::Json::object list_mapped()
|
json11::Json::object list_mapped()
|
||||||
{
|
{
|
||||||
int n_in_dev = 0;
|
int n_in_dev = 0;
|
||||||
@@ -618,13 +624,13 @@ protected:
|
|||||||
.types = UBLK_PARAM_TYPE_BASIC,
|
.types = UBLK_PARAM_TYPE_BASIC,
|
||||||
.basic = {
|
.basic = {
|
||||||
.attrs = attrs, // UBLK_ATTR_READ_ONLY | UBLK_ATTR_ROTATIONAL | UBLK_ATTR_VOLATILE_CACHE | UBLK_ATTR_FUA
|
.attrs = attrs, // UBLK_ATTR_READ_ONLY | UBLK_ATTR_ROTATIONAL | UBLK_ATTR_VOLATILE_CACHE | UBLK_ATTR_FUA
|
||||||
.logical_bs_shift = 9,
|
.logical_bs_shift = phys_shift,
|
||||||
.physical_bs_shift = phys_shift,
|
.physical_bs_shift = phys_shift,
|
||||||
.io_opt_shift = io_opt_shift,
|
.io_opt_shift = io_opt_shift,
|
||||||
.io_min_shift = phys_shift,
|
.io_min_shift = phys_shift,
|
||||||
.max_sectors = max_io_buf_bytes / phys_block_size,
|
.max_sectors = max_io_buf_bytes / 512,
|
||||||
.chunk_sectors = 0,
|
.chunk_sectors = 0,
|
||||||
.dev_sectors = device_size / phys_block_size,
|
.dev_sectors = device_size / 512,
|
||||||
.virt_boundary_mask = 0,
|
.virt_boundary_mask = 0,
|
||||||
},
|
},
|
||||||
.discard = {
|
.discard = {
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ includedir=${prefix}/@CMAKE_INSTALL_INCLUDEDIR@
|
|||||||
|
|
||||||
Name: Vitastor
|
Name: Vitastor
|
||||||
Description: Vitastor client library
|
Description: Vitastor client library
|
||||||
Version: 2.3.0
|
Version: 2.4.1
|
||||||
Libs: -L${libdir} -lvitastor_client
|
Libs: -L${libdir} -lvitastor_client
|
||||||
Cflags: -I${includedir}
|
Cflags: -I${includedir}
|
||||||
|
|
||||||
|
|||||||
+2
-1
@@ -37,6 +37,7 @@ static const char* help_text =
|
|||||||
" --sort FIELD Sort by specified field (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)\n"
|
" --sort FIELD Sort by specified field (name, size, used_size, <read|write|delete>_<iops|bps|lat|queue>)\n"
|
||||||
" -r|--reverse Sort in descending order\n"
|
" -r|--reverse Sort in descending order\n"
|
||||||
" -n|--count N Only list first N items\n"
|
" -n|--count N Only list first N items\n"
|
||||||
|
" --tree Show image snapshot/clone tree\n"
|
||||||
"\n"
|
"\n"
|
||||||
"vitastor-cli create -s|--size <size> [-p|--pool <id|name>] [--parent <parent_name>[@<snapshot>]] <name>\n"
|
"vitastor-cli create -s|--size <size> [-p|--pool <id|name>] [--parent <parent_name>[@<snapshot>]] <name>\n"
|
||||||
" Create an image. You may use K/M/G/T suffixes for <size>. If --parent is specified,\n"
|
" Create an image. You may use K/M/G/T suffixes for <size>. If --parent is specified,\n"
|
||||||
@@ -287,7 +288,7 @@ static json11::Json::object parse_args(int narg, const char *args[])
|
|||||||
!strcmp(opt, "down-ok") || !strcmp(opt, "down_ok") ||
|
!strcmp(opt, "down-ok") || !strcmp(opt, "down_ok") ||
|
||||||
!strcmp(opt, "dry-run") || !strcmp(opt, "dry_run") ||
|
!strcmp(opt, "dry-run") || !strcmp(opt, "dry_run") ||
|
||||||
!strcmp(opt, "help") || !strcmp(opt, "all") ||
|
!strcmp(opt, "help") || !strcmp(opt, "all") ||
|
||||||
!strcmp(opt, "exact") || !strcmp(opt, "matching") ||
|
!strcmp(opt, "exact") || !strcmp(opt, "matching") || !strcmp(opt, "tree") ||
|
||||||
!strcmp(opt, "writers-stopped") || !strcmp(opt, "writers_stopped"))
|
!strcmp(opt, "writers-stopped") || !strcmp(opt, "writers_stopped"))
|
||||||
{
|
{
|
||||||
cfg[opt] = "1";
|
cfg[opt] = "1";
|
||||||
|
|||||||
+39
-11
@@ -19,6 +19,7 @@ struct image_lister_t
|
|||||||
std::set<std::string> only_names;
|
std::set<std::string> only_names;
|
||||||
bool reverse = false;
|
bool reverse = false;
|
||||||
bool exact = false;
|
bool exact = false;
|
||||||
|
bool tree = false;
|
||||||
int max_count = 0;
|
int max_count = 0;
|
||||||
bool show_stats = false, show_delete = false;
|
bool show_stats = false, show_delete = false;
|
||||||
|
|
||||||
@@ -77,6 +78,7 @@ struct image_lister_t
|
|||||||
? p_it->second.name : "";
|
? p_it->second.name : "";
|
||||||
item["parent_pool_id"] = (uint64_t)INODE_POOL(ic.second.parent_id);
|
item["parent_pool_id"] = (uint64_t)INODE_POOL(ic.second.parent_id);
|
||||||
item["parent_inode_num"] = INODE_NO_POOL(ic.second.parent_id);
|
item["parent_inode_num"] = INODE_NO_POOL(ic.second.parent_id);
|
||||||
|
item["parent_inode_id"] = ic.second.parent_id;
|
||||||
}
|
}
|
||||||
stats[ic.second.num] = item;
|
stats[ic.second.num] = item;
|
||||||
}
|
}
|
||||||
@@ -92,16 +94,14 @@ struct image_lister_t
|
|||||||
parent->etcd_txn(json11::Json::object {
|
parent->etcd_txn(json11::Json::object {
|
||||||
{ "success", json11::Json::array {
|
{ "success", json11::Json::array {
|
||||||
json11::Json::object {
|
json11::Json::object {
|
||||||
{ "request_range", json11::Json::object {
|
{ "request_range", (list_pool_id
|
||||||
{ "key", base64_encode(
|
? json11::Json::object {
|
||||||
parent->cli->st_cli.etcd_prefix+"/pool/stats"+
|
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/pool/stats/"+std::to_string(list_pool_id)) },
|
||||||
(list_pool_id ? "/"+std::to_string(list_pool_id) : "")+"/"
|
}
|
||||||
) },
|
: json11::Json::object {
|
||||||
{ "range_end", base64_encode(
|
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/pool/stats/") },
|
||||||
parent->cli->st_cli.etcd_prefix+"/pool/stats"+
|
{ "range_end", base64_encode(parent->cli->st_cli.etcd_prefix+"/pool/stats0") },
|
||||||
(list_pool_id ? "/"+std::to_string(list_pool_id) : "")+"0"
|
}) },
|
||||||
) },
|
|
||||||
} },
|
|
||||||
},
|
},
|
||||||
json11::Json::object {
|
json11::Json::object {
|
||||||
{ "request_range", json11::Json::object {
|
{ "request_range", json11::Json::object {
|
||||||
@@ -245,6 +245,33 @@ resume_1:
|
|||||||
return list;
|
return list;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
json11::Json::array to_tree(const json11::Json::array & array)
|
||||||
|
{
|
||||||
|
std::map<uint64_t, json11::Json::array> children;
|
||||||
|
for (const auto & item: array)
|
||||||
|
{
|
||||||
|
uint64_t parent_id = item["parent_inode_id"].uint64_value();
|
||||||
|
children[parent_id].push_back(item);
|
||||||
|
}
|
||||||
|
json11::Json::array tree;
|
||||||
|
std::function<void(uint64_t, const std::string &)> add_children = [&](uint64_t parent_id, const std::string & prefix)
|
||||||
|
{
|
||||||
|
if (children.find(parent_id) != children.end())
|
||||||
|
{
|
||||||
|
for (size_t i = 0; i < children[parent_id].size(); i++)
|
||||||
|
{
|
||||||
|
bool is_last = (i == children[parent_id].size()-1);
|
||||||
|
json11::Json::object new_item = children[parent_id][i].object_items();
|
||||||
|
new_item["name"] = prefix + (parent_id == 0 ? "" : (is_last ? "└─ " : "├─ ")) + new_item["name"].string_value();
|
||||||
|
tree.push_back(new_item);
|
||||||
|
add_children(new_item["inode_id"].uint64_value(), prefix + (parent_id == 0 ? "" : (is_last ? " " : "│ ")));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
add_children(0, "");
|
||||||
|
return tree;
|
||||||
|
}
|
||||||
|
|
||||||
void loop()
|
void loop()
|
||||||
{
|
{
|
||||||
if (state == 1)
|
if (state == 1)
|
||||||
@@ -375,7 +402,7 @@ resume_1:
|
|||||||
kv.second["ro"] = kv.second["deleted"].bool_value() ? "DEL" :
|
kv.second["ro"] = kv.second["deleted"].bool_value() ? "DEL" :
|
||||||
(kv.second["readonly"].bool_value() ? "RO" : "-");
|
(kv.second["readonly"].bool_value() ? "RO" : "-");
|
||||||
}
|
}
|
||||||
result.text = print_table(to_list(), cols, parent->color);
|
result.text = print_table(tree ? to_tree(to_list()) : to_list(), cols, parent->color);
|
||||||
state = 100;
|
state = 100;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -542,6 +569,7 @@ std::function<bool(cli_result_t &)> cli_tool_t::start_ls(json11::Json cfg)
|
|||||||
auto lister = new image_lister_t();
|
auto lister = new image_lister_t();
|
||||||
lister->parent = this;
|
lister->parent = this;
|
||||||
lister->exact = cfg["exact"].bool_value();
|
lister->exact = cfg["exact"].bool_value();
|
||||||
|
lister->tree = cfg["tree"].bool_value();
|
||||||
lister->list_pool_id = cfg["pool"].uint64_value();
|
lister->list_pool_id = cfg["pool"].uint64_value();
|
||||||
lister->list_pool_name = lister->list_pool_id ? "" : cfg["pool"].as_string();
|
lister->list_pool_name = lister->list_pool_id ? "" : cfg["pool"].as_string();
|
||||||
lister->show_stats = cfg["long"].bool_value();
|
lister->show_stats = cfg["long"].bool_value();
|
||||||
|
|||||||
@@ -180,7 +180,8 @@ std::string validate_pool_config(json11::Json::object & new_cfg, json11::Json ol
|
|||||||
{
|
{
|
||||||
return "Changing scheme for an existing pool will lead to data loss. Use --force to proceed";
|
return "Changing scheme for an existing pool will lead to data loss. Use --force to proceed";
|
||||||
}
|
}
|
||||||
if (etcd_state_client_t::parse_scheme(old_cfg["scheme"].string_value()) == POOL_SCHEME_EC)
|
auto old_scheme = etcd_state_client_t::parse_scheme(old_cfg["scheme"].string_value());
|
||||||
|
if (old_scheme == POOL_SCHEME_EC)
|
||||||
{
|
{
|
||||||
uint64_t old_data_chunks = old_cfg["pg_size"].uint64_value() - old_cfg["parity_chunks"].uint64_value();
|
uint64_t old_data_chunks = old_cfg["pg_size"].uint64_value() - old_cfg["parity_chunks"].uint64_value();
|
||||||
uint64_t new_data_chunks = cfg["pg_size"].uint64_value() - cfg["parity_chunks"].uint64_value();
|
uint64_t new_data_chunks = cfg["pg_size"].uint64_value() - cfg["parity_chunks"].uint64_value();
|
||||||
@@ -189,6 +190,13 @@ std::string validate_pool_config(json11::Json::object & new_cfg, json11::Json ol
|
|||||||
return "Changing EC data chunk count for an existing pool will lead to data loss. Use --force to proceed";
|
return "Changing EC data chunk count for an existing pool will lead to data loss. Use --force to proceed";
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
else if (old_scheme != POOL_SCHEME_REPLICATED)
|
||||||
|
{
|
||||||
|
if (old_cfg["pg_size"].uint64_value() != cfg["pg_size"].uint64_value())
|
||||||
|
{
|
||||||
|
return "Changing XOR data chunk count for an existing pool will lead to data loss. Use --force to proceed";
|
||||||
|
}
|
||||||
|
}
|
||||||
if (old_cfg["block_size"] != cfg["block_size"] ||
|
if (old_cfg["block_size"] != cfg["block_size"] ||
|
||||||
old_cfg["bitmap_granularity"] != cfg["bitmap_granularity"] ||
|
old_cfg["bitmap_granularity"] != cfg["bitmap_granularity"] ||
|
||||||
old_cfg["immediate_commit"] != cfg["immediate_commit"])
|
old_cfg["immediate_commit"] != cfg["immediate_commit"])
|
||||||
|
|||||||
+21
-1
@@ -87,6 +87,16 @@ struct pool_lister_t
|
|||||||
) },
|
) },
|
||||||
} },
|
} },
|
||||||
},
|
},
|
||||||
|
json11::Json::object {
|
||||||
|
{ "request_range", json11::Json::object {
|
||||||
|
{ "key", base64_encode(
|
||||||
|
parent->cli->st_cli.etcd_prefix+"/config/osd/"
|
||||||
|
) },
|
||||||
|
{ "range_end", base64_encode(
|
||||||
|
parent->cli->st_cli.etcd_prefix+"/config/osd0"
|
||||||
|
) },
|
||||||
|
} },
|
||||||
|
},
|
||||||
} },
|
} },
|
||||||
});
|
});
|
||||||
state = base_state+1;
|
state = base_state+1;
|
||||||
@@ -115,6 +125,14 @@ resume_1:
|
|||||||
// osd/stats/<N>::free
|
// osd/stats/<N>::free
|
||||||
osd_free[osd_num] = value["free"].uint64_value();
|
osd_free[osd_num] = value["free"].uint64_value();
|
||||||
});
|
});
|
||||||
|
std::map<uint64_t, double> osd_reweight;
|
||||||
|
parent->iterate_kvs_1(space_info["responses"][3]["response_range"]["kvs"], "/config/osd/", [&](uint64_t osd_num, json11::Json value)
|
||||||
|
{
|
||||||
|
if (value.object_items().find("reweight") != value.object_items().end())
|
||||||
|
{
|
||||||
|
osd_reweight[osd_num] = value["reweight"].number_value();
|
||||||
|
}
|
||||||
|
});
|
||||||
// Calculate max_avail for each pool
|
// Calculate max_avail for each pool
|
||||||
for (auto & pp: parent->cli->st_cli.pool_config)
|
for (auto & pp: parent->cli->st_cli.pool_config)
|
||||||
{
|
{
|
||||||
@@ -140,7 +158,9 @@ resume_1:
|
|||||||
}
|
}
|
||||||
for (auto pg_per_pair: pg_per_osd)
|
for (auto pg_per_pair: pg_per_osd)
|
||||||
{
|
{
|
||||||
uint64_t pg_free = osd_free[pg_per_pair.first] * pool_cfg.real_pg_count / pg_per_pair.second;
|
uint64_t pg_free = osd_free[pg_per_pair.first] *
|
||||||
|
(osd_reweight.find(pg_per_pair.first) != osd_reweight.end() ? osd_reweight[pg_per_pair.first] : 1.0) *
|
||||||
|
pool_cfg.real_pg_count / pg_per_pair.second;
|
||||||
if (pool_avail > pg_free)
|
if (pool_avail > pg_free)
|
||||||
{
|
{
|
||||||
pool_avail = pg_free;
|
pool_avail = pg_free;
|
||||||
|
|||||||
@@ -48,6 +48,8 @@ static const char *help_text =
|
|||||||
" --max_other 10%\n"
|
" --max_other 10%\n"
|
||||||
" Use disks for OSD data even if they already have non-Vitastor partitions,\n"
|
" Use disks for OSD data even if they already have non-Vitastor partitions,\n"
|
||||||
" but only if these take up no more than this percent of disk space.\n"
|
" but only if these take up no more than this percent of disk space.\n"
|
||||||
|
" --dry-run\n"
|
||||||
|
" Check and print new OSD count for each disk but do not actually create them.\n"
|
||||||
" \n"
|
" \n"
|
||||||
" Options (single-device mode):\n"
|
" Options (single-device mode):\n"
|
||||||
" --data_device <DEV> Use partition <DEV> for data\n"
|
" --data_device <DEV> Use partition <DEV> for data\n"
|
||||||
|
|||||||
@@ -134,7 +134,8 @@ struct disk_tool_t
|
|||||||
int prepare(std::vector<std::string> devices);
|
int prepare(std::vector<std::string> devices);
|
||||||
std::vector<vitastor_dev_info_t> collect_devices(const std::vector<std::string> & devices);
|
std::vector<vitastor_dev_info_t> collect_devices(const std::vector<std::string> & devices);
|
||||||
json11::Json add_partitions(vitastor_dev_info_t & devinfo, std::vector<std::string> sizes);
|
json11::Json add_partitions(vitastor_dev_info_t & devinfo, std::vector<std::string> sizes);
|
||||||
std::vector<std::string> get_new_data_parts(vitastor_dev_info_t & dev, uint64_t osd_per_disk, uint64_t max_other_percent);
|
std::vector<std::string> get_new_data_parts(vitastor_dev_info_t & dev,
|
||||||
|
uint64_t osd_per_disk, uint64_t max_other_percent, uint64_t *check_new_count);
|
||||||
int get_meta_partition(std::vector<vitastor_dev_info_t> & ssds, std::map<std::string, std::string> & options);
|
int get_meta_partition(std::vector<vitastor_dev_info_t> & ssds, std::map<std::string, std::string> & options);
|
||||||
|
|
||||||
int upgrade_simple_unit(std::string unit);
|
int upgrade_simple_unit(std::string unit);
|
||||||
|
|||||||
@@ -435,7 +435,7 @@ json11::Json disk_tool_t::add_partitions(vitastor_dev_info_t & devinfo, std::vec
|
|||||||
}
|
}
|
||||||
|
|
||||||
std::vector<std::string> disk_tool_t::get_new_data_parts(vitastor_dev_info_t & dev,
|
std::vector<std::string> disk_tool_t::get_new_data_parts(vitastor_dev_info_t & dev,
|
||||||
uint64_t osd_per_disk, uint64_t max_other_percent)
|
uint64_t osd_per_disk, uint64_t max_other_percent, uint64_t *check_new_count)
|
||||||
{
|
{
|
||||||
std::vector<std::string> use_parts;
|
std::vector<std::string> use_parts;
|
||||||
uint64_t want_parts = 0;
|
uint64_t want_parts = 0;
|
||||||
@@ -457,7 +457,6 @@ std::vector<std::string> disk_tool_t::get_new_data_parts(vitastor_dev_info_t & d
|
|||||||
{
|
{
|
||||||
// Use this partition
|
// Use this partition
|
||||||
use_parts.push_back(part["uuid"].string_value());
|
use_parts.push_back(part["uuid"].string_value());
|
||||||
osds_exist++;
|
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -480,9 +479,21 @@ std::vector<std::string> disk_tool_t::get_new_data_parts(vitastor_dev_info_t & d
|
|||||||
}
|
}
|
||||||
// Still create OSD(s) if a disk has no more than (max_other_percent) other data
|
// Still create OSD(s) if a disk has no more than (max_other_percent) other data
|
||||||
if (osds_exist >= osd_per_disk || (dev.free+osds_size) < dev.size*(100-max_other_percent)/100)
|
if (osds_exist >= osd_per_disk || (dev.free+osds_size) < dev.size*(100-max_other_percent)/100)
|
||||||
|
{
|
||||||
fprintf(stderr, "%s is already partitioned, skipping\n", dev.path.c_str());
|
fprintf(stderr, "%s is already partitioned, skipping\n", dev.path.c_str());
|
||||||
|
use_parts.clear();
|
||||||
|
}
|
||||||
else
|
else
|
||||||
want_parts = osd_per_disk-osds_exist;
|
{
|
||||||
|
if (use_parts.size() >= osd_per_disk-osds_exist)
|
||||||
|
use_parts.resize(osd_per_disk-osds_exist);
|
||||||
|
want_parts = osd_per_disk-osds_exist-use_parts.size();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (check_new_count)
|
||||||
|
{
|
||||||
|
*check_new_count = want_parts;
|
||||||
|
return use_parts;
|
||||||
}
|
}
|
||||||
if (want_parts > 0)
|
if (want_parts > 0)
|
||||||
{
|
{
|
||||||
@@ -684,10 +695,25 @@ int disk_tool_t::prepare(std::vector<std::string> devices)
|
|||||||
}
|
}
|
||||||
json11::Json::array all_results, errors;
|
json11::Json::array all_results, errors;
|
||||||
auto journal_size = options["journal_size"];
|
auto journal_size = options["journal_size"];
|
||||||
|
if (options.find("dry_run") != options.end())
|
||||||
|
{
|
||||||
|
json11::Json::array results;
|
||||||
|
for (auto & dev: devinfo)
|
||||||
|
{
|
||||||
|
uint64_t new_part_count = 0;
|
||||||
|
auto existing_part_count = get_new_data_parts(dev, osd_per_disk, max_other_percent, &new_part_count).size();
|
||||||
|
results.push_back(json11::Json::object{ { "device_path", dev.path }, { "new_osd_count", existing_part_count+new_part_count } });
|
||||||
|
if (!json && new_part_count+existing_part_count > 0)
|
||||||
|
printf("Will initialize %ju OSD(s) on %s\n", existing_part_count+new_part_count, dev.path.c_str());
|
||||||
|
}
|
||||||
|
if (json)
|
||||||
|
printf("%s\n", json11::Json(json11::Json::object{{ "devices", results }}).dump().c_str());
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
for (auto & dev: devinfo)
|
for (auto & dev: devinfo)
|
||||||
{
|
{
|
||||||
// Select new partitions and create an OSD on each of them
|
// Select new partitions and create an OSD on each of them
|
||||||
for (const auto & uuid: get_new_data_parts(dev, osd_per_disk, max_other_percent))
|
for (const auto & uuid: get_new_data_parts(dev, osd_per_disk, max_other_percent, NULL))
|
||||||
{
|
{
|
||||||
options["force"] = true;
|
options["force"] = true;
|
||||||
options["data_device"] = "/dev/disk/by-partuuid/"+strtolower(uuid);
|
options["data_device"] = "/dev/disk/by-partuuid/"+strtolower(uuid);
|
||||||
|
|||||||
+7
-2
@@ -309,8 +309,13 @@ struct kv_cli_list_t
|
|||||||
|
|
||||||
void flush()
|
void flush()
|
||||||
{
|
{
|
||||||
::write(1, buf.data(), buf.size());
|
size_t done = 0;
|
||||||
buf.resize(0);
|
while (done < buf.size())
|
||||||
|
{
|
||||||
|
ssize_t res = ::write(1, buf.data()+done, buf.size()-done);
|
||||||
|
if (res > 0)
|
||||||
|
done += res;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
+4
-1
@@ -630,7 +630,10 @@ uint64_t kv_db_t::alloc_block()
|
|||||||
{
|
{
|
||||||
// Allow to reconfigure <max_allocate_blocks> online
|
// Allow to reconfigure <max_allocate_blocks> online
|
||||||
if (allocating_blocks.size() > max_allocate_blocks)
|
if (allocating_blocks.size() > max_allocate_blocks)
|
||||||
|
{
|
||||||
allocating_blocks.erase(allocating_blocks.begin()+allocating_block_pos, allocating_blocks.begin()+allocating_block_pos+1);
|
allocating_blocks.erase(allocating_blocks.begin()+allocating_block_pos, allocating_blocks.begin()+allocating_block_pos+1);
|
||||||
|
allocating_block_pos = allocating_block_pos % allocating_blocks.size();
|
||||||
|
}
|
||||||
else
|
else
|
||||||
allocating_blocks[allocating_block_pos].offset = UINT64_MAX;
|
allocating_blocks[allocating_block_pos].offset = UINT64_MAX;
|
||||||
}
|
}
|
||||||
@@ -870,7 +873,7 @@ static void get_block(kv_db_t *db, uint64_t offset, int cur_level, int recheck_p
|
|||||||
}
|
}
|
||||||
// Block already in cache, we can proceed
|
// Block already in cache, we can proceed
|
||||||
blk->usage = db->usage_counter;
|
blk->usage = db->usage_counter;
|
||||||
db->cli->msgr.ringloop->set_immediate([=] { cb(0, BLK_UPDATING); });
|
cb(0, BLK_UPDATING);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
cluster_op_t *op = new cluster_op_t;
|
cluster_op_t *op = new cluster_op_t;
|
||||||
|
|||||||
@@ -25,7 +25,10 @@ nfstime3 nfstime_from_str(const std::string & s)
|
|||||||
t.nseconds /= 10;
|
t.nseconds /= 10;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
|
{
|
||||||
t.seconds = stoull_full(s, 10);
|
t.seconds = stoull_full(s, 10);
|
||||||
|
t.nseconds = 0;
|
||||||
|
}
|
||||||
return t;
|
return t;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -94,6 +94,10 @@ uint32_t kv_get_access(const authsys_parms & auth_sys, const json11::Json & attr
|
|||||||
uint32_t uid = attrs["uid"].uint64_value();
|
uint32_t uid = attrs["uid"].uint64_value();
|
||||||
uint32_t gid = attrs["gid"].uint64_value();
|
uint32_t gid = attrs["gid"].uint64_value();
|
||||||
uint32_t access = 0;
|
uint32_t access = 0;
|
||||||
|
if (!auth_sys.uid)
|
||||||
|
{
|
||||||
|
return (ACCESS3_READ|ACCESS3_LOOKUP|ACCESS3_MODIFY|ACCESS3_EXTEND|ACCESS3_DELETE|ACCESS3_EXECUTE);
|
||||||
|
}
|
||||||
if (uid == auth_sys.uid)
|
if (uid == auth_sys.uid)
|
||||||
{
|
{
|
||||||
access |= ((mode & (1 << 8)) ? ACCESS3_READ|ACCESS3_LOOKUP : 0);
|
access |= ((mode & (1 << 8)) ? ACCESS3_READ|ACCESS3_LOOKUP : 0);
|
||||||
@@ -120,6 +124,8 @@ uint32_t kv_get_access(const authsys_parms & auth_sys, const json11::Json & attr
|
|||||||
bool kv_is_accessible(const authsys_parms & auth_sys, const json11::Json & attrs, uint32_t access)
|
bool kv_is_accessible(const authsys_parms & auth_sys, const json11::Json & attrs, uint32_t access)
|
||||||
{
|
{
|
||||||
uint32_t mode = attrs["mode"].is_null() ? (attrs["type"] == "dir" ? 0755 : 0644) : attrs["mode"].uint64_value();
|
uint32_t mode = attrs["mode"].is_null() ? (attrs["type"] == "dir" ? 0755 : 0644) : attrs["mode"].uint64_value();
|
||||||
|
if (!auth_sys.uid)
|
||||||
|
return true;
|
||||||
uint32_t mask = 1 << (access == ACCESS3_EXECUTE ? 0
|
uint32_t mask = 1 << (access == ACCESS3_EXECUTE ? 0
|
||||||
: (access == ACCESS3_MODIFY || access == ACCESS3_EXTEND || access == ACCESS3_DELETE ? 1 : 2));
|
: (access == ACCESS3_MODIFY || access == ACCESS3_EXTEND || access == ACCESS3_DELETE ? 1 : 2));
|
||||||
if (mode & mask)
|
if (mode & mask)
|
||||||
|
|||||||
@@ -23,6 +23,81 @@ int kv_nfs3_lookup_proc(void *opaque, rpc_op_t *rop)
|
|||||||
rpc_queue_reply(rop);
|
rpc_queue_reply(rop);
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
if (filename == ".")
|
||||||
|
{
|
||||||
|
kv_read_inode(self->parent, dir_ino, [=](int res, const std::string & value, json11::Json ientry)
|
||||||
|
{
|
||||||
|
if (res < 0)
|
||||||
|
{
|
||||||
|
*reply = (LOOKUP3res){ .status = vitastor_nfs_map_err(-res) };
|
||||||
|
rpc_queue_reply(rop);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
*reply = (LOOKUP3res){
|
||||||
|
.status = NFS3_OK,
|
||||||
|
.resok = (LOOKUP3resok){
|
||||||
|
.object = xdr_copy_string(rop->xdrs, kv_fh(dir_ino)),
|
||||||
|
.obj_attributes = {
|
||||||
|
.attributes_follow = 1,
|
||||||
|
.attributes = get_kv_attributes(self->parent, dir_ino, ientry),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
};
|
||||||
|
rpc_queue_reply(rop);
|
||||||
|
});
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
if (filename == "..")
|
||||||
|
{
|
||||||
|
kv_read_inode(self->parent, dir_ino, [=](int res, const std::string & value, json11::Json ientry)
|
||||||
|
{
|
||||||
|
if (res < 0)
|
||||||
|
{
|
||||||
|
*reply = (LOOKUP3res){ .status = vitastor_nfs_map_err(-res) };
|
||||||
|
rpc_queue_reply(rop);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
uint64_t parent_ino = ientry["parent_ino"].uint64_value();
|
||||||
|
if (parent_ino)
|
||||||
|
{
|
||||||
|
kv_read_inode(self->parent, parent_ino, [=](int res, const std::string & value, json11::Json parent_ientry)
|
||||||
|
{
|
||||||
|
if (res < 0)
|
||||||
|
{
|
||||||
|
*reply = (LOOKUP3res){ .status = vitastor_nfs_map_err(-res) };
|
||||||
|
rpc_queue_reply(rop);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
*reply = (LOOKUP3res){
|
||||||
|
.status = NFS3_OK,
|
||||||
|
.resok = (LOOKUP3resok){
|
||||||
|
.object = xdr_copy_string(rop->xdrs, kv_fh(parent_ino)),
|
||||||
|
.obj_attributes = {
|
||||||
|
.attributes_follow = 1,
|
||||||
|
.attributes = get_kv_attributes(self->parent, parent_ino, parent_ientry),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
};
|
||||||
|
rpc_queue_reply(rop);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
*reply = (LOOKUP3res){
|
||||||
|
.status = NFS3_OK,
|
||||||
|
.resok = (LOOKUP3resok){
|
||||||
|
.object = xdr_copy_string(rop->xdrs, kv_fh(dir_ino)),
|
||||||
|
.obj_attributes = {
|
||||||
|
.attributes_follow = 1,
|
||||||
|
.attributes = get_kv_attributes(self->parent, dir_ino, ientry),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
};
|
||||||
|
rpc_queue_reply(rop);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
self->parent->db->get(kv_direntry_key(dir_ino, filename), [=](int res, const std::string & value)
|
self->parent->db->get(kv_direntry_key(dir_ino, filename), [=](int res, const std::string & value)
|
||||||
{
|
{
|
||||||
if (res < 0)
|
if (res < 0)
|
||||||
|
|||||||
@@ -199,7 +199,7 @@ resume_4:
|
|||||||
else
|
else
|
||||||
{
|
{
|
||||||
// Not OK, restore direntry
|
// Not OK, restore direntry
|
||||||
st->self->parent->db->del(kv_direntry_key(st->dir_ino, st->filename), [st](int res)
|
st->self->parent->db->set(kv_direntry_key(st->dir_ino, st->filename), st->direntry_text, [st](int res)
|
||||||
{
|
{
|
||||||
st->res2 = res;
|
st->res2 = res;
|
||||||
nfs_kv_continue_delete(st, 5);
|
nfs_kv_continue_delete(st, 5);
|
||||||
@@ -282,6 +282,10 @@ resume_6:
|
|||||||
});
|
});
|
||||||
return;
|
return;
|
||||||
resume_7:
|
resume_7:
|
||||||
|
if (!st->res)
|
||||||
|
{
|
||||||
|
st->self->parent->kvfs->touch_queue.insert(st->dir_ino);
|
||||||
|
}
|
||||||
auto cb = std::move(st->cb);
|
auto cb = std::move(st->cb);
|
||||||
cb(st->res);
|
cb(st->res);
|
||||||
return;
|
return;
|
||||||
|
|||||||
@@ -204,10 +204,10 @@ void osd_t::start_pg_peering(pg_t & pg)
|
|||||||
for (auto pg_osd: pg.all_peers)
|
for (auto pg_osd: pg.all_peers)
|
||||||
{
|
{
|
||||||
if (pg_osd != this->osd_num &&
|
if (pg_osd != this->osd_num &&
|
||||||
msgr.osd_peer_fds.find(pg_osd) == msgr.osd_peer_fds.end() &&
|
msgr.osd_peer_fds.find(pg_osd) == msgr.osd_peer_fds.end())
|
||||||
msgr.wanted_peers.find(pg_osd) == msgr.wanted_peers.end())
|
|
||||||
{
|
{
|
||||||
msgr.connect_peer(pg_osd, st_cli.peer_states[pg_osd]);
|
if (msgr.wanted_peers.find(pg_osd) == msgr.wanted_peers.end())
|
||||||
|
msgr.connect_peer(pg_osd, st_cli.peer_states[pg_osd]);
|
||||||
if (!st_cli.peer_states[pg_osd].is_null())
|
if (!st_cli.peer_states[pg_osd].is_null())
|
||||||
all_connected = false;
|
all_connected = false;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -128,6 +128,8 @@ void pg_obj_state_check_t::handle_version()
|
|||||||
n_copies++;
|
n_copies++;
|
||||||
if (replicated && replica > 0 || replica >= pg->pg_size)
|
if (replicated && replica > 0 || replica >= pg->pg_size)
|
||||||
{
|
{
|
||||||
|
printf("Object %jx:%jx has invalid chunk number: %u > %ju\n", list[list_pos].oid.inode,
|
||||||
|
list[list_pos].oid.stripe, replica, replicated ? 0 : pg->pg_size);
|
||||||
n_invalid++;
|
n_invalid++;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
|
|||||||
@@ -9,7 +9,6 @@
|
|||||||
#define SUBMIT_READ 0
|
#define SUBMIT_READ 0
|
||||||
#define SUBMIT_RMW_READ 1
|
#define SUBMIT_RMW_READ 1
|
||||||
#define SUBMIT_WRITE 2
|
#define SUBMIT_WRITE 2
|
||||||
#define SUBMIT_SCRUB_READ 3
|
|
||||||
|
|
||||||
struct unstable_osd_num_t
|
struct unstable_osd_num_t
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -17,10 +17,22 @@ void osd_t::continue_chained_read(osd_op_t *cur_op)
|
|||||||
else if (op_data->st == 4)
|
else if (op_data->st == 4)
|
||||||
goto resume_4;
|
goto resume_4;
|
||||||
cur_op->reply.rw.bitmap_len = 0;
|
cur_op->reply.rw.bitmap_len = 0;
|
||||||
for (int role = 0; role < (pg ? pg->pg_data_size : 1); role++)
|
if (cur_op->req.rw.len == 0)
|
||||||
{
|
{
|
||||||
op_data->stripes[role].read_start = op_data->stripes[role].req_start;
|
// len=0 => bitmap read
|
||||||
op_data->stripes[role].read_end = op_data->stripes[role].req_end;
|
for (int role = 0; role < (pg ? pg->pg_data_size : 1); role++)
|
||||||
|
{
|
||||||
|
op_data->stripes[role].read_start = 0;
|
||||||
|
op_data->stripes[role].read_end = UINT32_MAX;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
for (int role = 0; role < (pg ? pg->pg_data_size : 1); role++)
|
||||||
|
{
|
||||||
|
op_data->stripes[role].read_start = op_data->stripes[role].req_start;
|
||||||
|
op_data->stripes[role].read_end = op_data->stripes[role].req_end;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
resume_1:
|
resume_1:
|
||||||
resume_2:
|
resume_2:
|
||||||
@@ -329,7 +341,7 @@ std::vector<osd_chain_read_t> osd_t::collect_chained_read_requests(osd_op_t *cur
|
|||||||
{
|
{
|
||||||
osd_primary_op_data_t *op_data = cur_op->op_data;
|
osd_primary_op_data_t *op_data = cur_op->op_data;
|
||||||
std::vector<osd_chain_read_t> chain_reads;
|
std::vector<osd_chain_read_t> chain_reads;
|
||||||
int stripe_count = (op_data->pg->scheme == POOL_SCHEME_REPLICATED ? 1 : op_data->pg->pg_size);
|
int stripe_count = (!op_data->pg || op_data->pg->scheme == POOL_SCHEME_REPLICATED ? 1 : op_data->pg->pg_size);
|
||||||
memset(op_data->stripes[0].bmp_buf, 0, stripe_count * clean_entry_bitmap_size);
|
memset(op_data->stripes[0].bmp_buf, 0, stripe_count * clean_entry_bitmap_size);
|
||||||
uint8_t *global_bitmap = (uint8_t*)op_data->stripes[0].bmp_buf;
|
uint8_t *global_bitmap = (uint8_t*)op_data->stripes[0].bmp_buf;
|
||||||
// We always use at most 1 read request per layer
|
// We always use at most 1 read request per layer
|
||||||
@@ -337,7 +349,7 @@ std::vector<osd_chain_read_t> osd_t::collect_chained_read_requests(osd_op_t *cur
|
|||||||
{
|
{
|
||||||
uint8_t *part_bitmap = ((uint8_t*)op_data->snapshot_bitmaps) + chain_pos*stripe_count*clean_entry_bitmap_size;
|
uint8_t *part_bitmap = ((uint8_t*)op_data->snapshot_bitmaps) + chain_pos*stripe_count*clean_entry_bitmap_size;
|
||||||
int start = !cur_op->req.rw.len ? 0 : (cur_op->req.rw.offset - op_data->oid.stripe)/bs_bitmap_granularity;
|
int start = !cur_op->req.rw.len ? 0 : (cur_op->req.rw.offset - op_data->oid.stripe)/bs_bitmap_granularity;
|
||||||
int end = !cur_op->req.rw.len ? op_data->pg->pg_data_size*clean_entry_bitmap_size*8 : start + cur_op->req.rw.len/bs_bitmap_granularity;
|
int end = !cur_op->req.rw.len ? (op_data->pg ? op_data->pg->pg_data_size : 1) * clean_entry_bitmap_size*8 : start + cur_op->req.rw.len/bs_bitmap_granularity;
|
||||||
// Skip unneeded part in the beginning
|
// Skip unneeded part in the beginning
|
||||||
while (start < end && (
|
while (start < end && (
|
||||||
((global_bitmap[start>>3] >> (start&7)) & 1) ||
|
((global_bitmap[start>>3] >> (start&7)) & 1) ||
|
||||||
|
|||||||
@@ -130,7 +130,7 @@ void osd_t::submit_primary_subops(int submit_type, uint64_t op_version, const ui
|
|||||||
if (osd_set[role] != 0 && (wr || !rep && stripes[role].read_end != 0))
|
if (osd_set[role] != 0 && (wr || !rep && stripes[role].read_end != 0))
|
||||||
n_subops++;
|
n_subops++;
|
||||||
}
|
}
|
||||||
if (!n_subops && (submit_type == SUBMIT_RMW_READ || rep))
|
if (zero_read >= 0 && !n_subops && (submit_type == SUBMIT_RMW_READ || rep))
|
||||||
n_subops = 1;
|
n_subops = 1;
|
||||||
else
|
else
|
||||||
zero_read = -1;
|
zero_read = -1;
|
||||||
@@ -153,13 +153,13 @@ int osd_t::submit_primary_subop_batch(int submit_type, inode_t inode, uint64_t o
|
|||||||
for (int role = 0; role < (op_data->pg ? op_data->pg->pg_size : 1); role++)
|
for (int role = 0; role < (op_data->pg ? op_data->pg->pg_size : 1); role++)
|
||||||
{
|
{
|
||||||
// We always submit zero-length writes to all replicas, even if the stripe is not modified
|
// We always submit zero-length writes to all replicas, even if the stripe is not modified
|
||||||
if (!(wr || !rep && stripes[role].read_end != 0 || zero_read == role || submit_type == SUBMIT_SCRUB_READ))
|
if (!(wr || !rep && stripes[role].read_end != 0 || zero_read == role))
|
||||||
{
|
{
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
osd_num_t role_osd_num = osd_set[role];
|
osd_num_t role_osd_num = osd_set[role];
|
||||||
int stripe_num = rep ? 0 : role;
|
int stripe_num = rep ? 0 : role;
|
||||||
osd_rmw_stripe_t *si = stripes + (submit_type == SUBMIT_SCRUB_READ ? role : stripe_num);
|
osd_rmw_stripe_t *si = stripes + stripe_num;
|
||||||
if (role_osd_num != 0)
|
if (role_osd_num != 0)
|
||||||
{
|
{
|
||||||
si->osd_num = role_osd_num;
|
si->osd_num = role_osd_num;
|
||||||
|
|||||||
@@ -17,7 +17,7 @@ bool string_to_addr(std::string str, bool parse_port, int default_port, struct s
|
|||||||
if (parse_port)
|
if (parse_port)
|
||||||
{
|
{
|
||||||
int p = str.rfind(':');
|
int p = str.rfind(':');
|
||||||
if (p != std::string::npos && !(str.length() > 0 && str[p-1] == ']')) // "[ipv6]" which contains ':'
|
if (p != std::string::npos && (str[0] != '[' || p > 0 && str[p-1] == ']')) // "[ipv6]" which contains ':'
|
||||||
{
|
{
|
||||||
char null_byte = 0;
|
char null_byte = 0;
|
||||||
int scanned = sscanf(str.c_str()+p+1, "%d%c", &default_port, &null_byte);
|
int scanned = sscanf(str.c_str()+p+1, "%d%c", &default_port, &null_byte);
|
||||||
|
|||||||
@@ -30,7 +30,7 @@ std::string base64_encode(const std::string &in)
|
|||||||
return out;
|
return out;
|
||||||
}
|
}
|
||||||
|
|
||||||
static char T[256] = { 0 };
|
static int T[256] = { 0 };
|
||||||
|
|
||||||
std::string base64_decode(const std::string &in)
|
std::string base64_decode(const std::string &in)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -36,7 +36,7 @@ void timerfd_manager_t::inc_timer(timerfd_timer_t & t)
|
|||||||
{
|
{
|
||||||
t.next.tv_sec += t.micros/1000000;
|
t.next.tv_sec += t.micros/1000000;
|
||||||
t.next.tv_nsec += (t.micros%1000000)*1000;
|
t.next.tv_nsec += (t.micros%1000000)*1000;
|
||||||
if (t.next.tv_nsec > 1000000000)
|
if (t.next.tv_nsec >= 1000000000)
|
||||||
{
|
{
|
||||||
t.next.tv_sec++;
|
t.next.tv_sec++;
|
||||||
t.next.tv_nsec -= 1000000000;
|
t.next.tv_nsec -= 1000000000;
|
||||||
|
|||||||
+4
-2
@@ -70,6 +70,10 @@ TEST_NAME=local_read POOLCFG='"local_reads":"random",' ./test_heal.sh
|
|||||||
SCHEME=ec ./test_heal.sh
|
SCHEME=ec ./test_heal.sh
|
||||||
ANTIETCD=1 ./test_heal.sh
|
ANTIETCD=1 ./test_heal.sh
|
||||||
|
|
||||||
|
./test_reweight_half.sh
|
||||||
|
./test_snapshot_pool2.sh
|
||||||
|
./test_snapshot_read_bitmap.sh
|
||||||
|
|
||||||
TEST_NAME=csum_32k_dmj OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k --inmemory_metadata false --inmemory_journal false" OFFSET_ARGS=$OSD_ARGS ./test_heal.sh
|
TEST_NAME=csum_32k_dmj OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k --inmemory_metadata false --inmemory_journal false" OFFSET_ARGS=$OSD_ARGS ./test_heal.sh
|
||||||
TEST_NAME=csum_32k_dj OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k --inmemory_journal false" OFFSET_ARGS=$OSD_ARGS ./test_heal.sh
|
TEST_NAME=csum_32k_dj OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k --inmemory_journal false" OFFSET_ARGS=$OSD_ARGS ./test_heal.sh
|
||||||
TEST_NAME=csum_32k OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k" OFFSET_ARGS=$OSD_ARGS ./test_heal.sh
|
TEST_NAME=csum_32k OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k" OFFSET_ARGS=$OSD_ARGS ./test_heal.sh
|
||||||
@@ -80,8 +84,6 @@ TEST_NAME=csum_4k OSD_ARGS="--data_csum_type crc32c" OFFSET_ARGS=$OSD_ARGS
|
|||||||
./test_resize.sh
|
./test_resize.sh
|
||||||
./test_resize_auto.sh
|
./test_resize_auto.sh
|
||||||
|
|
||||||
./test_snapshot_pool2.sh
|
|
||||||
|
|
||||||
./test_osd_tags.sh
|
./test_osd_tags.sh
|
||||||
|
|
||||||
./test_enospc.sh
|
./test_enospc.sh
|
||||||
|
|||||||
@@ -13,6 +13,12 @@ sudo mount localhost:/ ./testdata/nfs -o port=2050,mountport=2050,nfsvers=3,soft
|
|||||||
MNT=$(pwd)/testdata/nfs
|
MNT=$(pwd)/testdata/nfs
|
||||||
trap "sudo umount -f $MNT"' || true; kill -9 $(jobs -p)' EXIT
|
trap "sudo umount -f $MNT"' || true; kill -9 $(jobs -p)' EXIT
|
||||||
|
|
||||||
|
touch ./testdata/nfs/f1
|
||||||
|
chown 1000:1000 ./testdata/nfs/f1
|
||||||
|
chmod 600 ./testdata/nfs/f1
|
||||||
|
sudo cat ./testdata/nfs/f1
|
||||||
|
rm ./testdata/nfs/f1
|
||||||
|
|
||||||
# write small file
|
# write small file
|
||||||
ls -l ./testdata/nfs
|
ls -l ./testdata/nfs
|
||||||
dd if=/dev/urandom of=./testdata/f1 bs=100k count=1
|
dd if=/dev/urandom of=./testdata/f1 bs=100k count=1
|
||||||
|
|||||||
Executable
+41
@@ -0,0 +1,41 @@
|
|||||||
|
#!/bin/bash -ex
|
||||||
|
|
||||||
|
. `dirname $0`/common.sh
|
||||||
|
|
||||||
|
node mon/mon-main.js $MON_PARAMS --etcd_address $ETCD_URL --etcd_prefix "/vitastor" >>./testdata/mon.log 2>&1 &
|
||||||
|
MON_PID=$!
|
||||||
|
wait_etcd
|
||||||
|
|
||||||
|
TIME=$(date '+%s')
|
||||||
|
$ETCDCTL put /vitastor/osd/stats/1 '{"host":"host1","size":1073741824,"time":"'$TIME'"}'
|
||||||
|
$ETCDCTL put /vitastor/osd/stats/2 '{"host":"host1","size":1073741824,"time":"'$TIME'"}'
|
||||||
|
$ETCDCTL put /vitastor/osd/stats/3 '{"host":"host2","size":1073741824,"time":"'$TIME'"}'
|
||||||
|
$ETCDCTL put /vitastor/osd/stats/4 '{"host":"host2","size":1073741824,"time":"'$TIME'"}'
|
||||||
|
build/src/cmd/vitastor-cli --etcd_address $ETCD_URL create-pool testpool -s 2 -n 16 --force
|
||||||
|
|
||||||
|
sleep 2
|
||||||
|
|
||||||
|
# check that all OSDs have 8 PGs
|
||||||
|
$ETCDCTL get /vitastor/pg/config --print-value-only | \
|
||||||
|
jq -s -e '([ .[0].items["1"] | .[].osd_set | map_values(. | tonumber) | select(.[0] == 1 or .[1] == 1) ] | length) == 8'
|
||||||
|
$ETCDCTL get /vitastor/pg/config --print-value-only | \
|
||||||
|
jq -s -e '([ .[0].items["1"] | .[].osd_set | map_values(. | tonumber) | select(.[0] == 2 or .[1] == 2) ] | length) == 8'
|
||||||
|
$ETCDCTL get /vitastor/pg/config --print-value-only | \
|
||||||
|
jq -s -e '([ .[0].items["1"] | .[].osd_set | map_values(. | tonumber) | select(.[0] == 3 or .[1] == 3) ] | length) == 8'
|
||||||
|
$ETCDCTL get /vitastor/pg/config --print-value-only | \
|
||||||
|
jq -s -e '([ .[0].items["1"] | .[].osd_set | map_values(. | tonumber) | select(.[0] == 4 or .[1] == 4) ] | length) == 8'
|
||||||
|
|
||||||
|
build/src/cmd/vitastor-cli --etcd_address $ETCD_URL modify-osd --reweight 0.5 3
|
||||||
|
|
||||||
|
sleep 2
|
||||||
|
|
||||||
|
$ETCDCTL get /vitastor/pg/config --print-value-only | \
|
||||||
|
jq -s -e '([ .[0].items["1"] | .[].osd_set | map_values(. | tonumber) | select(.[0] == 1 or .[1] == 1) ] | length) == 8'
|
||||||
|
$ETCDCTL get /vitastor/pg/config --print-value-only | \
|
||||||
|
jq -s -e '([ .[0].items["1"] | .[].osd_set | map_values(. | tonumber) | select(.[0] == 2 or .[1] == 2) ] | length) == 8'
|
||||||
|
$ETCDCTL get /vitastor/pg/config --print-value-only | \
|
||||||
|
jq -s -e '([ .[0].items["1"] | .[].osd_set | map_values(. | tonumber) | select(.[0] == 3 or .[1] == 3) ] | length) <= 6'
|
||||||
|
$ETCDCTL get /vitastor/pg/config --print-value-only | \
|
||||||
|
jq -s -e '([ .[0].items["1"] | .[].osd_set | map_values(. | tonumber) | select(.[0] == 4 or .[1] == 4) ] | length) >= 10'
|
||||||
|
|
||||||
|
format_green OK
|
||||||
Executable
+39
@@ -0,0 +1,39 @@
|
|||||||
|
#!/bin/bash -ex
|
||||||
|
|
||||||
|
SCHEME=${SCHEME:-ec}
|
||||||
|
. `dirname $0`/run_3osds.sh
|
||||||
|
check_qemu
|
||||||
|
|
||||||
|
build/src/cmd/vitastor-cli --etcd_address $ETCD_URL create -s 128M testchain
|
||||||
|
|
||||||
|
dd if=/dev/zero of=./testdata/bin/mirror.bin bs=4k seek=$(((128*1024-4)/4)) count=1
|
||||||
|
|
||||||
|
LD_PRELOAD="build/src/client/libfio_vitastor.so" \
|
||||||
|
fio -thread -name=test -ioengine=build/src/client/libfio_vitastor.so -bs=32k -direct=1 -iodepth=4 -end_fsync=1 -rw=randwrite \
|
||||||
|
-etcd=$ETCD_URL -image=testchain -mirror_file=./testdata/bin/mirror.bin -buffer_pattern=0xabcd -number_ios=1024
|
||||||
|
|
||||||
|
build/src/cmd/vitastor-cli --etcd_address $ETCD_URL snap-create testchain@snap1
|
||||||
|
|
||||||
|
LD_PRELOAD="build/src/client/libfio_vitastor.so" \
|
||||||
|
fio -thread -name=test -ioengine=build/src/client/libfio_vitastor.so -bs=4k -direct=1 -iodepth=4 -end_fsync=1 -rw=randwrite -number_ios=32 \
|
||||||
|
-etcd=$ETCD_URL -image=testchain -mirror_file=./testdata/bin/mirror.bin -buffer_pattern=0xabcd
|
||||||
|
|
||||||
|
# Now read from the snapshot
|
||||||
|
|
||||||
|
dd if=/dev/zero of=./testdata/bin/res.bin bs=4k seek=$(((128*1024-4)/4)) count=1
|
||||||
|
|
||||||
|
build/src/cmd/vitastor-cli --etcd_address $ETCD_URL dd iimg=testchain of=./testdata/bin/res.bin bs=$((PG_DATA_SIZE*128))k iodepth=4 --log_level 10
|
||||||
|
|
||||||
|
cmp ./testdata/bin/res.bin ./testdata/bin/mirror.bin
|
||||||
|
|
||||||
|
build/src/cmd/vitastor-cli --etcd_address $ETCD_URL dd iimg=testchain of=./testdata/bin/res.bin bs=$((PG_DATA_SIZE*128))k iodepth=4 conv=nosparse
|
||||||
|
|
||||||
|
cmp ./testdata/bin/res.bin ./testdata/bin/mirror.bin
|
||||||
|
|
||||||
|
qemu-img convert -p \
|
||||||
|
-f raw "vitastor:etcd_host=127.0.0.1\:$ETCD_PORT/v3:image=testchain" \
|
||||||
|
-O raw ./testdata/bin/res.bin
|
||||||
|
|
||||||
|
cmp ./testdata/bin/res.bin ./testdata/bin/mirror.bin
|
||||||
|
|
||||||
|
format_green OK
|
||||||
Reference in New Issue
Block a user