Move all sources to subdirs
This commit is contained in:
@@ -0,0 +1,14 @@
|
||||
cmake_minimum_required(VERSION 2.8.12)
|
||||
|
||||
project(vitastor)
|
||||
|
||||
# vitastor-disk
|
||||
add_executable(vitastor-disk
|
||||
disk_tool.cpp disk_simple_offsets.cpp
|
||||
disk_tool_journal.cpp disk_tool_meta.cpp disk_tool_prepare.cpp disk_tool_resize.cpp disk_tool_udev.cpp disk_tool_utils.cpp disk_tool_upgrade.cpp
|
||||
../util/crc32c.c ../util/str_util.cpp ../../json11/json11.cpp ../util/rw_blocking.cpp ../util/allocator.cpp ../util/ringloop.cpp ../blockstore/blockstore_disk.cpp
|
||||
)
|
||||
target_link_libraries(vitastor-disk
|
||||
tcmalloc_minimal
|
||||
${LIBURING_LIBRARIES}
|
||||
)
|
||||
@@ -0,0 +1,174 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#include <fcntl.h>
|
||||
#include <sys/ioctl.h>
|
||||
#include <ctype.h>
|
||||
#include <unistd.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
#include "json11/json11.hpp"
|
||||
#include "str_util.h"
|
||||
#include "blockstore.h"
|
||||
#include "blockstore_disk.h"
|
||||
|
||||
// Calculate offsets for a block device and print OSD command line parameters
|
||||
void disk_tool_simple_offsets(json11::Json cfg, bool json_output)
|
||||
{
|
||||
std::string device = cfg["device"].string_value();
|
||||
uint64_t data_block_size = parse_size(cfg["object_size"].string_value());
|
||||
uint64_t bitmap_granularity = parse_size(cfg["bitmap_granularity"].string_value());
|
||||
uint64_t journal_size = parse_size(cfg["journal_size"].string_value());
|
||||
uint64_t device_block_size = parse_size(cfg["device_block_size"].string_value());
|
||||
uint64_t journal_offset = parse_size(cfg["journal_offset"].string_value());
|
||||
uint64_t device_size = parse_size(cfg["device_size"].string_value());
|
||||
uint32_t csum_block_size = parse_size(cfg["csum_block_size"].string_value());
|
||||
uint32_t data_csum_type = BLOCKSTORE_CSUM_NONE;
|
||||
if (cfg["data_csum_type"] == "crc32c")
|
||||
data_csum_type = BLOCKSTORE_CSUM_CRC32C;
|
||||
else if (cfg["data_csum_type"].string_value() != "" && cfg["data_csum_type"].string_value() != "none")
|
||||
{
|
||||
fprintf(
|
||||
stderr, "data_csum_type=%s is unsupported, only \"crc32c\" and \"none\" are supported",
|
||||
cfg["data_csum_type"].string_value().c_str()
|
||||
);
|
||||
exit(1);
|
||||
}
|
||||
std::string format = cfg["format"].string_value();
|
||||
if (json_output)
|
||||
format = "json";
|
||||
if (!data_block_size)
|
||||
data_block_size = 1 << DEFAULT_DATA_BLOCK_ORDER;
|
||||
if (!bitmap_granularity)
|
||||
bitmap_granularity = DEFAULT_BITMAP_GRANULARITY;
|
||||
if (!journal_size)
|
||||
journal_size = 32*1024*1024;
|
||||
if (!device_block_size)
|
||||
device_block_size = 4096;
|
||||
if (!data_csum_type)
|
||||
csum_block_size = 0;
|
||||
else if (!csum_block_size)
|
||||
csum_block_size = bitmap_granularity;
|
||||
uint64_t orig_device_size = device_size;
|
||||
if (!device_size)
|
||||
{
|
||||
if (device == "")
|
||||
{
|
||||
fprintf(stderr, "Device path is missing\n");
|
||||
exit(1);
|
||||
}
|
||||
struct stat st;
|
||||
if (stat(device.c_str(), &st) < 0)
|
||||
{
|
||||
fprintf(stderr, "Can't stat %s: %s\n", device.c_str(), strerror(errno));
|
||||
exit(1);
|
||||
}
|
||||
if (S_ISBLK(st.st_mode))
|
||||
{
|
||||
int fd = open(device.c_str(), O_DIRECT|O_RDONLY);
|
||||
if (fd < 0 || ioctl(fd, BLKGETSIZE64, &device_size) < 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to get device size for %s: %s\n", device.c_str(), strerror(errno));
|
||||
exit(1);
|
||||
}
|
||||
close(fd);
|
||||
if (st.st_blksize < device_block_size)
|
||||
{
|
||||
fprintf(
|
||||
stderr, "Warning: %s reports %ju byte blocks, but we use %ju."
|
||||
" Set --device_block_size=%ju if you're sure it works well with %ju byte blocks.\n",
|
||||
device.c_str(), (uint64_t)st.st_blksize, device_block_size, (uint64_t)st.st_blksize, (uint64_t)st.st_blksize
|
||||
);
|
||||
}
|
||||
}
|
||||
else if (S_ISREG(st.st_mode))
|
||||
{
|
||||
device_size = st.st_size;
|
||||
}
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "%s is neither a block device nor a regular file\n", device.c_str());
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
if (!device_size)
|
||||
{
|
||||
fprintf(stderr, "Failed to get device size for %s\n", device.c_str());
|
||||
exit(1);
|
||||
}
|
||||
if (device_block_size < 512 || device_block_size > 1048576 ||
|
||||
device_block_size & (device_block_size-1) != 0)
|
||||
{
|
||||
fprintf(stderr, "Invalid device block size specified: %ju\n", device_block_size);
|
||||
exit(1);
|
||||
}
|
||||
if (data_block_size < device_block_size || data_block_size > MAX_DATA_BLOCK_SIZE ||
|
||||
data_block_size & (data_block_size-1) != 0)
|
||||
{
|
||||
fprintf(stderr, "Invalid object size specified: %ju\n", data_block_size);
|
||||
exit(1);
|
||||
}
|
||||
if (bitmap_granularity < device_block_size || bitmap_granularity > data_block_size ||
|
||||
bitmap_granularity & (bitmap_granularity-1) != 0)
|
||||
{
|
||||
fprintf(stderr, "Invalid bitmap granularity specified: %ju\n", bitmap_granularity);
|
||||
exit(1);
|
||||
}
|
||||
if (csum_block_size && (data_block_size % csum_block_size))
|
||||
{
|
||||
fprintf(stderr, "csum_block_size must be a divisor of data_block_size\n");
|
||||
exit(1);
|
||||
}
|
||||
journal_offset = ((journal_offset+device_block_size-1)/device_block_size)*device_block_size;
|
||||
uint64_t meta_offset = journal_offset + ((journal_size+device_block_size-1)/device_block_size)*device_block_size;
|
||||
uint64_t data_csum_size = (data_csum_type ? data_block_size/csum_block_size*(data_csum_type & 0xFF) : 0);
|
||||
uint64_t clean_entry_bitmap_size = data_block_size/bitmap_granularity/8;
|
||||
uint64_t clean_entry_size = 24 /*sizeof(clean_disk_entry)*/ + 2*clean_entry_bitmap_size + data_csum_size + 4 /*entry_csum*/;
|
||||
uint64_t entries_per_block = device_block_size / clean_entry_size;
|
||||
uint64_t object_count = ((device_size-meta_offset)/data_block_size);
|
||||
uint64_t meta_size = (1 + (object_count+entries_per_block-1)/entries_per_block) * device_block_size;
|
||||
uint64_t data_offset = meta_offset + meta_size;
|
||||
if (format == "json")
|
||||
{
|
||||
// JSON
|
||||
printf("%s\n", json11::Json(json11::Json::object {
|
||||
{ "meta_block_size", device_block_size },
|
||||
{ "journal_block_size", device_block_size },
|
||||
{ "data_size", device_size-data_offset },
|
||||
{ "data_device", device },
|
||||
{ "journal_offset", journal_offset },
|
||||
{ "meta_offset", meta_offset },
|
||||
{ "data_offset", data_offset },
|
||||
}).dump().c_str());
|
||||
}
|
||||
else if (format == "env")
|
||||
{
|
||||
// Env
|
||||
printf(
|
||||
"meta_block_size=%ju\njournal_block_size=%ju\ndata_size=%ju\n"
|
||||
"data_device=%s\njournal_offset=%ju\nmeta_offset=%ju\ndata_offset=%ju\n",
|
||||
device_block_size, device_block_size, device_size-data_offset,
|
||||
device.c_str(), journal_offset, meta_offset, data_offset
|
||||
);
|
||||
}
|
||||
else
|
||||
{
|
||||
// OSD command-line options
|
||||
if (format != "options")
|
||||
{
|
||||
fprintf(stderr, "Metadata size: %s\nOptions for the OSD:\n", format_size(meta_size).c_str());
|
||||
}
|
||||
if (device_block_size != 4096)
|
||||
{
|
||||
printf("--meta_block_size %ju\n--journal_block_size %ju\n", device_block_size, device_block_size);
|
||||
}
|
||||
if (orig_device_size)
|
||||
{
|
||||
printf("--data_size %ju\n", device_size-data_offset);
|
||||
}
|
||||
printf(
|
||||
"--data_device %s\n--journal_offset %ju\n--meta_offset %ju\n--data_offset %ju\n",
|
||||
device.c_str(), journal_offset, meta_offset, data_offset
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,435 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#include "disk_tool.h"
|
||||
#include "str_util.h"
|
||||
|
||||
static const char *help_text =
|
||||
"Vitastor disk management tool " VERSION "\n"
|
||||
"(c) Vitaliy Filippov, 2022+ (VNPL-1.1)\n"
|
||||
"\n"
|
||||
"COMMANDS:\n"
|
||||
"\n"
|
||||
"vitastor-disk prepare [OPTIONS] [devices...]\n"
|
||||
" Initialize disk(s) for Vitastor OSD(s).\n"
|
||||
" \n"
|
||||
" There are two modes of this command. In the first mode, you pass <devices> which\n"
|
||||
" must be raw disks (not partitions). They are partitioned automatically and OSDs\n"
|
||||
" are initialized on all of them.\n"
|
||||
" \n"
|
||||
" In the second mode, you omit <devices> and pass --data_device, --journal_device\n"
|
||||
" and/or --meta_device which must be already existing partitions identified by their\n"
|
||||
" GPT partition UUIDs. In this case a single OSD is created.\n"
|
||||
" \n"
|
||||
" Requires `vitastor-cli`, `wipefs`, `sfdisk` and `partprobe` (from parted) utilities.\n"
|
||||
" \n"
|
||||
" Options (automatic mode):\n"
|
||||
" --osd_per_disk <N>\n"
|
||||
" Create <N> OSDs on each disk (default 1)\n"
|
||||
" --hybrid\n"
|
||||
" Prepare hybrid (HDD+SSD) OSDs using provided devices. SSDs will be used for\n"
|
||||
" journals and metadata, HDDs will be used for data. Partitions for journals and\n"
|
||||
" metadata will be created automatically. Whether disks are SSD or HDD is decided\n"
|
||||
" by the `/sys/block/.../queue/rotational` flag. In hybrid mode, default object\n"
|
||||
" size is 1 MB instead of 128 KB, default journal size is 1 GB instead of 32 MB,\n"
|
||||
" and throttle_small_writes is enabled by default.\n"
|
||||
" --disable_data_fsync auto\n"
|
||||
" Disable data device cache and fsync (1/yes/true = on, default auto)\n"
|
||||
" --disable_meta_fsync auto\n"
|
||||
" Disable metadata/journal device cache and fsync (default auto)\n"
|
||||
" --meta_reserve 2x,1G\n"
|
||||
" New metadata partitions in --hybrid mode are created larger than actual\n"
|
||||
" metadata size to ease possible future extension. The default is to allocate\n"
|
||||
" 2 times more space and at least 1G. Use this option to override.\n"
|
||||
" --max_other 10%\n"
|
||||
" Use disks for OSD data even if they already have non-Vitastor partitions,\n"
|
||||
" but only if these take up no more than this percent of disk space.\n"
|
||||
" \n"
|
||||
" Options (single-device mode):\n"
|
||||
" --data_device <DEV> Use partition <DEV> for data\n"
|
||||
" --meta_device <DEV> Use partition <DEV> for metadata (optional)\n"
|
||||
" --journal_device <DEV> Use partition <DEV> for journal (optional)\n"
|
||||
" --disable_data_fsync 0 Disable data device cache and fsync (default off)\n"
|
||||
" --disable_meta_fsync 0 Disable metadata device cache and fsync (default off)\n"
|
||||
" --disable_journal_fsync 0 Disable journal device cache and fsync (default off)\n"
|
||||
" --hdd Enable HDD defaults (1M block, 1G journal, throttling)\n"
|
||||
" --force Bypass partition safety checks (for emptiness and so on)\n"
|
||||
" \n"
|
||||
" Options (both modes):\n"
|
||||
" --journal_size 32M/1G Set journal size (area or partition size)\n"
|
||||
" --block_size 128k/1M Set blockstore object size\n"
|
||||
" --bitmap_granularity 4k Set bitmap granularity\n"
|
||||
" --data_csum_type none Set data checksum type (crc32c or none)\n"
|
||||
" --csum_block_size 4k/32k Set data checksum block size (SSD/HDD default)\n"
|
||||
" --data_device_block 4k Override data device block size\n"
|
||||
" --meta_device_block 4k Override metadata device block size\n"
|
||||
" --journal_device_block 4k Override journal device block size\n"
|
||||
" \n"
|
||||
" immediate_commit setting is automatically derived from \"disable fsync\" options.\n"
|
||||
" It's set to \"all\" when fsync is disabled on all devices, and to \"small\" if fsync\n"
|
||||
" is only disabled on journal device.\n"
|
||||
" \n"
|
||||
" When data/meta/journal fsyncs are disabled, the OSD startup script automatically\n"
|
||||
" checks the device cache status on start and tries to disable cache for SATA/SAS disks.\n"
|
||||
" If it doesn't succeed it issues a warning in the system log.\n"
|
||||
" \n"
|
||||
" You can also pass other OSD options here as arguments and they'll be persisted\n"
|
||||
" in the superblock: data_io, meta_io, journal_io,\n"
|
||||
" inmemory_metadata, inmemory_journal, max_write_iodepth,\n"
|
||||
" min_flusher_count, max_flusher_count, journal_sector_buffer_count,\n"
|
||||
" journal_no_same_sector_overwrites, throttle_small_writes, throttle_target_iops,\n"
|
||||
" throttle_target_mbs, throttle_target_parallelism, throttle_threshold_us.\n"
|
||||
"\n"
|
||||
"vitastor-disk upgrade-simple <UNIT_FILE|OSD_NUMBER>\n"
|
||||
" Upgrade an OSD created by old (0.7.1 and older) make-osd.sh or make-osd-hybrid.js scripts.\n"
|
||||
" \n"
|
||||
" Adds superblocks to OSD devices, disables old vitastor-osdN unit and replaces it with vitastor-osd@N.\n"
|
||||
" Can be invoked with an osd number of with a path to systemd service file UNIT_FILE which\n"
|
||||
" must be /etc/systemd/system/vitastor-osd<OSD_NUMBER>.service.\n"
|
||||
" \n"
|
||||
" Note that the procedure isn't atomic and may ruin OSD data in case of an interrupt,\n"
|
||||
" so don't upgrade all your OSDs in parallel.\n"
|
||||
" \n"
|
||||
" Requires the `sfdisk` utility.\n"
|
||||
"\n"
|
||||
"vitastor-disk resize <ALL_OSD_PARAMETERS> <NEW_LAYOUT> [--iodepth 32]\n"
|
||||
" Resize data area and/or rewrite/move journal and metadata\n"
|
||||
" ALL_OSD_PARAMETERS must include all (at least all disk-related)\n"
|
||||
" parameters from OSD command line (i.e. from systemd unit or superblock).\n"
|
||||
" NEW_LAYOUT may include new disk layout parameters:\n"
|
||||
" --new_data_offset SIZE resize data area so it starts at SIZE\n"
|
||||
" --new_data_len SIZE resize data area to SIZE bytes\n"
|
||||
" --new_meta_device PATH use PATH for new metadata\n"
|
||||
" --new_meta_offset SIZE make new metadata area start at SIZE\n"
|
||||
" --new_meta_len SIZE make new metadata area SIZE bytes long\n"
|
||||
" --new_journal_device PATH use PATH for new journal\n"
|
||||
" --new_journal_offset SIZE make new journal area start at SIZE\n"
|
||||
" --new_journal_len SIZE make new journal area SIZE bytes long\n"
|
||||
" SIZE may include k/m/g/t suffixes. If any of the new layout parameter\n"
|
||||
" options are not specified, old values will be used.\n"
|
||||
"\n"
|
||||
"vitastor-disk start|stop|restart|enable|disable [--now] <device> [device2 device3 ...]\n"
|
||||
" Manipulate Vitastor OSDs using systemd by their device paths.\n"
|
||||
" Commands are passed to systemctl with vitastor-osd@<num> units as arguments.\n"
|
||||
" When --now is added to enable/disable, OSDs are also immediately started/stopped.\n"
|
||||
"\n"
|
||||
"vitastor-disk purge [--force] [--allow-data-loss] <device> [device2 device3 ...]\n"
|
||||
" Purge Vitastor OSD(s) on specified device(s). Uses vitastor-cli rm-osd to check\n"
|
||||
" if deletion is possible without data loss and to actually remove metadata from etcd.\n"
|
||||
" --force and --allow-data-loss options may be used to ignore safety check results.\n"
|
||||
" \n"
|
||||
" Requires `vitastor-cli`, `sfdisk` and `partprobe` (from parted) utilities.\n"
|
||||
"\n"
|
||||
"vitastor-disk read-sb [--force] <device>\n"
|
||||
" Try to read Vitastor OSD superblock from <device> and print it in JSON format.\n"
|
||||
" --force allows to ignore validation errors.\n"
|
||||
"\n"
|
||||
"vitastor-disk write-sb <device>\n"
|
||||
" Read JSON from STDIN and write it into Vitastor OSD superblock on <device>.\n"
|
||||
"\n"
|
||||
"vitastor-disk update-sb <device> [--force] [--<parameter> <value>] [...]\n"
|
||||
" Read Vitastor OSD superblock from <device>, update parameters in it and write it back.\n"
|
||||
" --force allows to ignore validation errors.\n"
|
||||
"\n"
|
||||
"vitastor-disk udev <device>\n"
|
||||
" Try to read Vitastor OSD superblock from <device> and print variables for udev.\n"
|
||||
"\n"
|
||||
"vitastor-disk exec-osd <device>\n"
|
||||
" Read Vitastor OSD superblock from <device> and start the OSD with parameters from it.\n"
|
||||
" Intended for use from startup scripts (i.e. from systemd units).\n"
|
||||
"\n"
|
||||
"vitastor-disk pre-exec <device>\n"
|
||||
" Read Vitastor OSD superblock from <device> and perform pre-start checks for the OSD.\n"
|
||||
" For now, this only checks that device cache is in write-through mode if fsync is disabled.\n"
|
||||
" Intended for use from startup scripts (i.e. from systemd units).\n"
|
||||
"\n"
|
||||
"vitastor-disk dump-journal [OPTIONS] <journal_file> <journal_block_size> <offset> <size>\n"
|
||||
" Dump journal in human-readable or JSON (if --json is specified) format.\n"
|
||||
" Options:\n"
|
||||
" --all Scan the whole journal area for entries and dump them, even outdated ones\n"
|
||||
" --json Dump journal in JSON format\n"
|
||||
" --format entries (Default) Dump actual journal entries as an array, without data\n"
|
||||
" --format data Same as \"entries\", but also include small write data\n"
|
||||
" --format blocks Dump as an array of journal blocks each containing array of entries\n"
|
||||
"\n"
|
||||
"vitastor-disk write-journal <journal_file> <journal_block_size> <bitmap_size> <offset> <size>\n"
|
||||
" Write journal from JSON taken from standard input in the same format as produced by\n"
|
||||
" `dump-journal --json --format data`.\n"
|
||||
"\n"
|
||||
"vitastor-disk dump-meta <meta_file> <meta_block_size> <offset> <size>\n"
|
||||
" Dump metadata in JSON format.\n"
|
||||
"\n"
|
||||
"vitastor-disk write-meta <meta_file> <offset> <size>\n"
|
||||
" Write metadata from JSON taken from standard input in the same format as produced by\n"
|
||||
" `dump-meta`. Intended for debugging.\n"
|
||||
"\n"
|
||||
"vitastor-disk simple-offsets <device>\n"
|
||||
" Calculate offsets for old simple&stupid (no superblock) OSD deployment. Options:\n"
|
||||
" --object_size 128k Set blockstore block size\n"
|
||||
" --bitmap_granularity 4k Set bitmap granularity\n"
|
||||
" --journal_size 32M Set journal size\n"
|
||||
" --data_csum_type none Set data checksum type (crc32c or none)\n"
|
||||
" --csum_block_size 4k Set data checksum block size\n"
|
||||
" --device_block_size 4k Set device block size\n"
|
||||
" --journal_offset 0 Set journal offset\n"
|
||||
" --device_size 0 Set device size\n"
|
||||
" --format text Result format: json, options, env, or text\n"
|
||||
"\n"
|
||||
"Use vitastor-disk --help <command> for command details or vitastor-disk --help --all for all details.\n"
|
||||
;
|
||||
|
||||
disk_tool_t::~disk_tool_t()
|
||||
{
|
||||
if (data_alloc)
|
||||
{
|
||||
delete data_alloc;
|
||||
data_alloc = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
disk_tool_t self = {};
|
||||
std::vector<char*> cmd;
|
||||
char *exe_name = strrchr(argv[0], '/');
|
||||
exe_name = exe_name ? exe_name+1 : argv[0];
|
||||
bool aliased = false;
|
||||
if (!strcmp(exe_name, "vitastor-dump-journal"))
|
||||
{
|
||||
cmd.push_back((char*)"dump-journal");
|
||||
aliased = true;
|
||||
}
|
||||
for (int i = 1; i < argc; i++)
|
||||
{
|
||||
if (!strcmp(argv[i], "--all"))
|
||||
{
|
||||
self.all = true;
|
||||
}
|
||||
else if (!strcmp(argv[i], "--json"))
|
||||
{
|
||||
self.json = true;
|
||||
}
|
||||
else if (!strcmp(argv[i], "--hybrid"))
|
||||
{
|
||||
self.options["hybrid"] = "1";
|
||||
}
|
||||
else if (!strcmp(argv[i], "--hdd"))
|
||||
{
|
||||
self.options["hdd"] = "1";
|
||||
}
|
||||
else if (!strcmp(argv[i], "--help") || !strcmp(argv[i], "-h"))
|
||||
{
|
||||
cmd.insert(cmd.begin(), (char*)"help");
|
||||
}
|
||||
else if (!strcmp(argv[i], "--now"))
|
||||
{
|
||||
self.now = true;
|
||||
}
|
||||
else if (!strcmp(argv[i], "--force"))
|
||||
{
|
||||
self.options["force"] = "1";
|
||||
}
|
||||
else if (!strcmp(argv[i], "--allow-data-loss"))
|
||||
{
|
||||
self.options["allow_data_loss"] = "1";
|
||||
}
|
||||
else if (argv[i][0] == '-' && argv[i][1] == '-' && i < argc-1)
|
||||
{
|
||||
char *key = argv[i]+2;
|
||||
self.options[key] = argv[++i];
|
||||
}
|
||||
else
|
||||
{
|
||||
cmd.push_back(argv[i]);
|
||||
}
|
||||
}
|
||||
if (!cmd.size())
|
||||
{
|
||||
cmd.push_back((char*)"help");
|
||||
}
|
||||
if (!strcmp(cmd[0], "dump-journal"))
|
||||
{
|
||||
if (cmd.size() < 5)
|
||||
{
|
||||
print_help(help_text, aliased ? "vitastor-dump-journal" : "vitastor-disk", cmd[0], false);
|
||||
return 1;
|
||||
}
|
||||
self.dsk.journal_device = cmd[1];
|
||||
self.dsk.journal_block_size = strtoul(cmd[2], NULL, 10);
|
||||
self.dsk.journal_offset = strtoull(cmd[3], NULL, 10);
|
||||
self.dsk.journal_len = strtoull(cmd[4], NULL, 10);
|
||||
return self.dump_journal();
|
||||
}
|
||||
else if (!strcmp(cmd[0], "write-journal"))
|
||||
{
|
||||
if (cmd.size() < 6)
|
||||
{
|
||||
print_help(help_text, "vitastor-disk", cmd[0], false);
|
||||
return 1;
|
||||
}
|
||||
self.new_journal_device = cmd[1];
|
||||
self.dsk.journal_block_size = strtoul(cmd[2], NULL, 10);
|
||||
self.dsk.clean_entry_bitmap_size = strtoul(cmd[3], NULL, 10);
|
||||
self.new_journal_offset = strtoull(cmd[4], NULL, 10);
|
||||
self.new_journal_len = strtoull(cmd[5], NULL, 10);
|
||||
std::string json_err;
|
||||
json11::Json entries = json11::Json::parse(read_all_fd(0), json_err);
|
||||
if (json_err != "")
|
||||
{
|
||||
fprintf(stderr, "Invalid JSON: %s\n", json_err.c_str());
|
||||
return 1;
|
||||
}
|
||||
if (entries[0]["type"] == "start")
|
||||
{
|
||||
self.dsk.data_csum_type = csum_type_from_str(entries[0]["data_csum_type"].string_value());
|
||||
self.dsk.csum_block_size = entries[0]["csum_block_size"].uint64_value();
|
||||
}
|
||||
if (self.options["data_csum_type"] != "")
|
||||
{
|
||||
self.dsk.data_csum_type = csum_type_from_str(self.options["data_csum_type"]);
|
||||
}
|
||||
if (self.options["csum_block_size"] != "")
|
||||
{
|
||||
self.dsk.csum_block_size = stoull_full(self.options["csum_block_size"], 0);
|
||||
}
|
||||
return self.write_json_journal(entries);
|
||||
}
|
||||
else if (!strcmp(cmd[0], "dump-meta"))
|
||||
{
|
||||
if (cmd.size() < 5)
|
||||
{
|
||||
print_help(help_text, "vitastor-disk", cmd[0], false);
|
||||
return 1;
|
||||
}
|
||||
self.dsk.meta_device = cmd[1];
|
||||
self.dsk.meta_block_size = strtoul(cmd[2], NULL, 10);
|
||||
self.dsk.meta_offset = strtoull(cmd[3], NULL, 10);
|
||||
self.dsk.meta_len = strtoull(cmd[4], NULL, 10);
|
||||
return self.dump_meta();
|
||||
}
|
||||
else if (!strcmp(cmd[0], "write-meta"))
|
||||
{
|
||||
if (cmd.size() < 4)
|
||||
{
|
||||
print_help(help_text, "vitastor-disk", cmd[0], false);
|
||||
return 1;
|
||||
}
|
||||
self.new_meta_device = cmd[1];
|
||||
self.new_meta_offset = strtoull(cmd[2], NULL, 10);
|
||||
self.new_meta_len = strtoull(cmd[3], NULL, 10);
|
||||
std::string json_err;
|
||||
json11::Json meta = json11::Json::parse(read_all_fd(0), json_err);
|
||||
if (json_err != "")
|
||||
{
|
||||
fprintf(stderr, "Invalid JSON: %s\n", json_err.c_str());
|
||||
return 1;
|
||||
}
|
||||
return self.write_json_meta(meta);
|
||||
}
|
||||
else if (!strcmp(cmd[0], "resize"))
|
||||
{
|
||||
return self.resize_data();
|
||||
}
|
||||
else if (!strcmp(cmd[0], "simple-offsets"))
|
||||
{
|
||||
// Calculate offsets for simple & stupid OSD deployment without superblock
|
||||
if (cmd.size() > 1)
|
||||
{
|
||||
self.options["device"] = cmd[1];
|
||||
}
|
||||
disk_tool_simple_offsets(self.options, self.json);
|
||||
return 0;
|
||||
}
|
||||
else if (!strcmp(cmd[0], "udev"))
|
||||
{
|
||||
if (cmd.size() != 2)
|
||||
{
|
||||
fprintf(stderr, "Exactly 1 device path argument is required\n");
|
||||
return 1;
|
||||
}
|
||||
return self.udev_import(cmd[1]);
|
||||
}
|
||||
else if (!strcmp(cmd[0], "read-sb"))
|
||||
{
|
||||
if (cmd.size() != 2)
|
||||
{
|
||||
fprintf(stderr, "Exactly 1 device path argument is required\n");
|
||||
return 1;
|
||||
}
|
||||
return self.read_sb(cmd[1]);
|
||||
}
|
||||
else if (!strcmp(cmd[0], "write-sb"))
|
||||
{
|
||||
if (cmd.size() != 2)
|
||||
{
|
||||
fprintf(stderr, "Exactly 1 device path argument is required\n");
|
||||
return 1;
|
||||
}
|
||||
return self.write_sb(cmd[1]);
|
||||
}
|
||||
else if (!strcmp(cmd[0], "update-sb"))
|
||||
{
|
||||
if (cmd.size() != 2)
|
||||
{
|
||||
fprintf(stderr, "Exactly 1 device path argument is required\n");
|
||||
return 1;
|
||||
}
|
||||
return self.update_sb(cmd[1]);
|
||||
}
|
||||
else if (!strcmp(cmd[0], "start") || !strcmp(cmd[0], "stop") ||
|
||||
!strcmp(cmd[0], "restart") || !strcmp(cmd[0], "enable") || !strcmp(cmd[0], "disable"))
|
||||
{
|
||||
std::vector<std::string> systemd_cmd;
|
||||
systemd_cmd.push_back(cmd[0]);
|
||||
if (self.now && (!strcmp(cmd[0], "enable") || !strcmp(cmd[0], "disable")))
|
||||
{
|
||||
systemd_cmd.push_back("--now");
|
||||
}
|
||||
return self.systemd_start_stop_osds(systemd_cmd, std::vector<std::string>(cmd.begin()+1, cmd.end()));
|
||||
}
|
||||
else if (!strcmp(cmd[0], "purge"))
|
||||
{
|
||||
return self.purge_devices(std::vector<std::string>(cmd.begin()+1, cmd.end()));
|
||||
}
|
||||
else if (!strcmp(cmd[0], "exec-osd"))
|
||||
{
|
||||
if (cmd.size() != 2)
|
||||
{
|
||||
fprintf(stderr, "Exactly 1 device path argument is required\n");
|
||||
return 1;
|
||||
}
|
||||
return self.exec_osd(cmd[1]);
|
||||
}
|
||||
else if (!strcmp(cmd[0], "pre-exec"))
|
||||
{
|
||||
if (cmd.size() != 2)
|
||||
{
|
||||
fprintf(stderr, "Exactly 1 device path argument is required\n");
|
||||
return 1;
|
||||
}
|
||||
return self.pre_exec_osd(cmd[1]);
|
||||
}
|
||||
else if (!strcmp(cmd[0], "prepare"))
|
||||
{
|
||||
std::vector<std::string> devs;
|
||||
for (int i = 1; i < cmd.size(); i++)
|
||||
{
|
||||
devs.push_back(cmd[i]);
|
||||
}
|
||||
return self.prepare(devs);
|
||||
}
|
||||
else if (!strcmp(cmd[0], "upgrade-simple"))
|
||||
{
|
||||
if (cmd.size() != 2)
|
||||
{
|
||||
fprintf(stderr, "Exactly 1 OSD number or systemd unit path is required\n");
|
||||
return 1;
|
||||
}
|
||||
return self.upgrade_simple_unit(cmd[1]);
|
||||
}
|
||||
else
|
||||
{
|
||||
print_help(help_text, "vitastor-disk", cmd.size() > 1 ? cmd[1] : "", self.all);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,145 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#pragma once
|
||||
|
||||
#ifndef _LARGEFILE64_SOURCE
|
||||
#define _LARGEFILE64_SOURCE
|
||||
#endif
|
||||
|
||||
#include <map>
|
||||
#include <vector>
|
||||
#include <string>
|
||||
#include <functional>
|
||||
|
||||
#include "json11/json11.hpp"
|
||||
#include "blockstore_disk.h"
|
||||
#include "blockstore_impl.h"
|
||||
#include "crc32c.h"
|
||||
|
||||
// vITADisk
|
||||
#define VITASTOR_DISK_MAGIC 0x6b73694441544976
|
||||
#define VITASTOR_DISK_MAX_SB_SIZE 128*1024
|
||||
#define VITASTOR_PART_TYPE "e7009fac-a5a1-4d72-af72-53de13059903"
|
||||
#define DEFAULT_HYBRID_JOURNAL "1G"
|
||||
|
||||
struct resizer_data_moving_t;
|
||||
|
||||
struct vitastor_dev_info_t
|
||||
{
|
||||
std::string path;
|
||||
bool is_hdd;
|
||||
json11::Json pt; // pt = partition table
|
||||
int osd_part_count;
|
||||
uint64_t size;
|
||||
uint64_t free;
|
||||
};
|
||||
|
||||
struct disk_tool_t
|
||||
{
|
||||
/**** Parameters ****/
|
||||
|
||||
std::map<std::string, std::string> options;
|
||||
bool all, json, now;
|
||||
bool dump_with_blocks, dump_with_data;
|
||||
blockstore_disk_t dsk;
|
||||
|
||||
// resize data and/or move metadata and journal
|
||||
int iodepth;
|
||||
std::string new_meta_device, new_journal_device;
|
||||
uint64_t new_data_offset, new_data_len;
|
||||
uint64_t new_journal_offset, new_journal_len;
|
||||
uint64_t new_meta_offset, new_meta_len;
|
||||
|
||||
/**** State ****/
|
||||
|
||||
uint64_t meta_pos;
|
||||
uint64_t journal_pos, journal_calc_data_pos;
|
||||
|
||||
bool first_block, first_entry;
|
||||
|
||||
allocator *data_alloc;
|
||||
std::map<uint64_t, uint64_t> data_remap;
|
||||
std::map<uint64_t, uint64_t>::iterator remap_it;
|
||||
ring_loop_t *ringloop;
|
||||
ring_consumer_t ring_consumer;
|
||||
int remap_active;
|
||||
journal_entry_start je_start;
|
||||
uint8_t *new_journal_buf, *new_meta_buf, *new_journal_ptr, *new_journal_data;
|
||||
uint64_t new_journal_in_pos;
|
||||
int64_t data_idx_diff;
|
||||
uint64_t total_blocks, free_first, free_last;
|
||||
uint64_t new_clean_entry_bitmap_size, new_data_csum_size, new_clean_entry_size, new_entries_per_block;
|
||||
int new_journal_fd, new_meta_fd;
|
||||
resizer_data_moving_t *moving_blocks;
|
||||
|
||||
bool started;
|
||||
void *small_write_data;
|
||||
uint32_t data_crc32;
|
||||
bool data_csum_valid;
|
||||
uint32_t crc32_last;
|
||||
uint32_t new_crc32_prev;
|
||||
|
||||
~disk_tool_t();
|
||||
|
||||
int dump_journal();
|
||||
void dump_journal_entry(int num, journal_entry *je, bool json);
|
||||
int process_journal(std::function<int(void*)> block_fn);
|
||||
int process_journal_block(void *buf, std::function<void(int, journal_entry*)> iter_fn);
|
||||
int process_meta(std::function<void(blockstore_meta_header_v2_t *)> hdr_fn,
|
||||
std::function<void(uint64_t, clean_disk_entry*, uint8_t*)> record_fn);
|
||||
|
||||
int dump_meta();
|
||||
void dump_meta_header(blockstore_meta_header_v2_t *hdr);
|
||||
void dump_meta_entry(uint64_t block_num, clean_disk_entry *entry, uint8_t *bitmap);
|
||||
|
||||
int write_json_journal(json11::Json entries);
|
||||
int write_json_meta(json11::Json meta);
|
||||
|
||||
int resize_data();
|
||||
int resize_parse_params();
|
||||
void resize_init(blockstore_meta_header_v2_t *hdr);
|
||||
int resize_remap_blocks();
|
||||
int resize_copy_data();
|
||||
int resize_rewrite_journal();
|
||||
int resize_write_new_journal();
|
||||
int resize_rewrite_meta();
|
||||
int resize_write_new_meta();
|
||||
|
||||
int udev_import(std::string device);
|
||||
int read_sb(std::string device);
|
||||
int write_sb(std::string device);
|
||||
int update_sb(std::string device);
|
||||
int exec_osd(std::string device);
|
||||
int systemd_start_stop_osds(const std::vector<std::string> & cmd, const std::vector<std::string> & devices);
|
||||
int pre_exec_osd(std::string device);
|
||||
int purge_devices(const std::vector<std::string> & devices);
|
||||
|
||||
json11::Json read_osd_superblock(std::string device, bool expect_exist = true, bool ignore_nonref = false);
|
||||
uint32_t write_osd_superblock(std::string device, json11::Json params);
|
||||
|
||||
int prepare_one(std::map<std::string, std::string> options, int is_hdd = -1);
|
||||
int prepare(std::vector<std::string> devices);
|
||||
std::vector<vitastor_dev_info_t> collect_devices(const std::vector<std::string> & devices);
|
||||
json11::Json add_partitions(vitastor_dev_info_t & devinfo, std::vector<std::string> sizes);
|
||||
std::vector<std::string> get_new_data_parts(vitastor_dev_info_t & dev, uint64_t osd_per_disk, uint64_t max_other_percent);
|
||||
int get_meta_partition(std::vector<vitastor_dev_info_t> & ssds, std::map<std::string, std::string> & options);
|
||||
|
||||
int upgrade_simple_unit(std::string unit);
|
||||
};
|
||||
|
||||
void disk_tool_simple_offsets(json11::Json cfg, bool json_output);
|
||||
|
||||
uint64_t sscanf_json(const char *fmt, const json11::Json & str);
|
||||
void fromhexstr(const std::string & from, int bytes, uint8_t *to);
|
||||
int disable_cache(std::string dev);
|
||||
std::string get_parent_device(std::string dev);
|
||||
bool json_is_true(const json11::Json & val);
|
||||
int shell_exec(const std::vector<std::string> & cmd, const std::string & in, std::string *out, std::string *err);
|
||||
int write_zero(int fd, uint64_t offset, uint64_t size);
|
||||
json11::Json read_parttable(std::string dev);
|
||||
uint64_t dev_size_from_parttable(json11::Json pt);
|
||||
uint64_t free_from_parttable(json11::Json pt);
|
||||
int fix_partition_type(std::string dev_by_uuid);
|
||||
std::string csum_type_str(uint32_t data_csum_type);
|
||||
uint32_t csum_type_from_str(std::string data_csum_type);
|
||||
@@ -0,0 +1,577 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#include "disk_tool.h"
|
||||
|
||||
int disk_tool_t::dump_journal()
|
||||
{
|
||||
dump_with_blocks = options["format"] == "blocks";
|
||||
dump_with_data = options["format"] == "data" || options["format"] == "blocks,data";
|
||||
if (dsk.journal_block_size < DIRECT_IO_ALIGNMENT || (dsk.journal_block_size % DIRECT_IO_ALIGNMENT) ||
|
||||
dsk.journal_block_size > 128*1024)
|
||||
{
|
||||
fprintf(stderr, "Invalid journal block size\n");
|
||||
return 1;
|
||||
}
|
||||
first_block = true;
|
||||
if (json)
|
||||
printf("[\n");
|
||||
if (all)
|
||||
{
|
||||
dsk.journal_fd = open(dsk.journal_device.c_str(), O_DIRECT|O_RDONLY);
|
||||
if (dsk.journal_fd < 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to open journal device %s: %s\n", dsk.journal_device.c_str(), strerror(errno));
|
||||
return 1;
|
||||
}
|
||||
void *journal_buf = memalign_or_die(MEM_ALIGNMENT, dsk.journal_block_size);
|
||||
journal_pos = 0;
|
||||
while (journal_pos < dsk.journal_len)
|
||||
{
|
||||
int r = pread(dsk.journal_fd, journal_buf, dsk.journal_block_size, dsk.journal_offset+journal_pos);
|
||||
assert(r == dsk.journal_block_size);
|
||||
uint64_t s;
|
||||
for (s = 0; s < dsk.journal_block_size; s += 8)
|
||||
{
|
||||
if (*((uint64_t*)((uint8_t*)journal_buf+s)) != 0)
|
||||
break;
|
||||
}
|
||||
if (json)
|
||||
{
|
||||
printf("%s{\"offset\":\"0x%jx\"", first_block ? "" : ",\n", journal_pos);
|
||||
first_block = false;
|
||||
}
|
||||
if (s == dsk.journal_block_size)
|
||||
{
|
||||
if (json)
|
||||
printf(",\"type\":\"zero\"}");
|
||||
else
|
||||
printf("offset %08jx: zeroes\n", journal_pos);
|
||||
journal_pos += dsk.journal_block_size;
|
||||
}
|
||||
else if (((journal_entry*)journal_buf)->magic == JOURNAL_MAGIC)
|
||||
{
|
||||
if (!json)
|
||||
printf("offset %08jx:\n", journal_pos);
|
||||
else
|
||||
printf(",\"entries\":[\n");
|
||||
if (journal_pos == 0)
|
||||
{
|
||||
// Fill journal header to know checksum type & size
|
||||
journal_entry *je = (journal_entry*)journal_buf;
|
||||
if (je->magic == JOURNAL_MAGIC && je->type == JE_START &&
|
||||
(je->start.version == JOURNAL_VERSION_V1 || je->start.version == JOURNAL_VERSION_V2))
|
||||
{
|
||||
memcpy(&je_start, je, sizeof(je_start));
|
||||
if (je_start.size == JE_START_V0_SIZE)
|
||||
je_start.version = 0;
|
||||
if (je_start.version < JOURNAL_VERSION_V2)
|
||||
{
|
||||
je_start.data_csum_type = 0;
|
||||
je_start.csum_block_size = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
first_entry = true;
|
||||
process_journal_block(journal_buf, [this](int num, journal_entry *je) { dump_journal_entry(num, je, json); });
|
||||
if (json)
|
||||
printf(first_entry ? "]}" : "\n]}");
|
||||
}
|
||||
else
|
||||
{
|
||||
if (json)
|
||||
printf(",\"type\":\"data\",\"pattern\":\"%08jx\"}", *((uint64_t*)journal_buf));
|
||||
else
|
||||
printf("offset %08jx: no magic in the beginning, looks like random data (pattern=%08jx)\n", journal_pos, *((uint64_t*)journal_buf));
|
||||
journal_pos += dsk.journal_block_size;
|
||||
}
|
||||
}
|
||||
free(journal_buf);
|
||||
close(dsk.journal_fd);
|
||||
dsk.journal_fd = -1;
|
||||
}
|
||||
else
|
||||
{
|
||||
first_entry = true;
|
||||
process_journal([this](void *data)
|
||||
{
|
||||
if (json && dump_with_blocks)
|
||||
first_entry = true;
|
||||
if (!json)
|
||||
printf("offset %08jx:\n", journal_pos);
|
||||
auto pos = journal_pos;
|
||||
int r = process_journal_block(data, [this, pos](int num, journal_entry *je)
|
||||
{
|
||||
if (json && dump_with_blocks && first_entry)
|
||||
printf("%s{\"offset\":\"0x%jx\",\"entries\":[\n", first_block ? "" : ",\n", pos);
|
||||
dump_journal_entry(num, je, json);
|
||||
first_block = false;
|
||||
});
|
||||
if (json && dump_with_blocks && !first_entry)
|
||||
printf("\n]}");
|
||||
else if (!json && r <= 0)
|
||||
printf("end of the journal\n");
|
||||
return r;
|
||||
});
|
||||
}
|
||||
if (json)
|
||||
printf(first_block ? "]\n" : "\n]\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::process_journal(std::function<int(void*)> block_fn)
|
||||
{
|
||||
dsk.journal_fd = open(dsk.journal_device.c_str(), O_DIRECT|O_RDONLY);
|
||||
if (dsk.journal_fd < 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to open journal device %s: %s\n", dsk.journal_device.c_str(), strerror(errno));
|
||||
return 1;
|
||||
}
|
||||
void *data = memalign_or_die(MEM_ALIGNMENT, dsk.journal_block_size);
|
||||
journal_pos = 0;
|
||||
int r = pread(dsk.journal_fd, data, dsk.journal_block_size, dsk.journal_offset+journal_pos);
|
||||
assert(r == dsk.journal_block_size);
|
||||
journal_entry *je = (journal_entry*)(data);
|
||||
if (je->magic != JOURNAL_MAGIC || je->type != JE_START || je_crc32(je) != je->crc32)
|
||||
{
|
||||
fprintf(stderr, "offset %08jx: journal superblock is invalid\n", journal_pos);
|
||||
r = 1;
|
||||
}
|
||||
else if (je->start.size != JE_START_V0_SIZE && je->start.version != JOURNAL_VERSION_V1 && je->start.version != JOURNAL_VERSION_V2)
|
||||
{
|
||||
fprintf(stderr, "offset %08jx: journal superblock contains version %ju, but I only understand 0, 1 and 2\n",
|
||||
journal_pos, je->start.size == JE_START_V0_SIZE ? 0 : je->start.version);
|
||||
r = 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
memcpy(&je_start, je, sizeof(je_start));
|
||||
if (je_start.size == JE_START_V0_SIZE)
|
||||
je_start.version = 0;
|
||||
if (je_start.version < JOURNAL_VERSION_V2)
|
||||
{
|
||||
je_start.data_csum_type = 0;
|
||||
je_start.csum_block_size = 0;
|
||||
}
|
||||
started = false;
|
||||
crc32_last = 0;
|
||||
block_fn(data);
|
||||
started = false;
|
||||
crc32_last = 0;
|
||||
journal_pos = je->start.journal_start;
|
||||
while (1)
|
||||
{
|
||||
if (journal_pos >= dsk.journal_len)
|
||||
journal_pos = dsk.journal_block_size;
|
||||
r = pread(dsk.journal_fd, data, dsk.journal_block_size, dsk.journal_offset+journal_pos);
|
||||
assert(r == dsk.journal_block_size);
|
||||
r = block_fn(data);
|
||||
if (r <= 0)
|
||||
break;
|
||||
}
|
||||
}
|
||||
close(dsk.journal_fd);
|
||||
dsk.journal_fd = -1;
|
||||
free(data);
|
||||
return r;
|
||||
}
|
||||
|
||||
int disk_tool_t::process_journal_block(void *buf, std::function<void(int, journal_entry*)> iter_fn)
|
||||
{
|
||||
uint32_t pos = 0;
|
||||
journal_pos += dsk.journal_block_size;
|
||||
int entry = 0;
|
||||
bool wrapped = false;
|
||||
while (pos <= dsk.journal_block_size-JOURNAL_ENTRY_HEADER_SIZE)
|
||||
{
|
||||
journal_entry *je = (journal_entry*)((uint8_t*)buf + pos);
|
||||
if (je->magic != JOURNAL_MAGIC || je->type < JE_MIN || je->type > JE_MAX ||
|
||||
!all && started && je->crc32_prev != crc32_last || pos > dsk.journal_block_size-je->size)
|
||||
{
|
||||
break;
|
||||
}
|
||||
bool crc32_valid = je_crc32(je) == je->crc32;
|
||||
if (!all && !crc32_valid)
|
||||
{
|
||||
break;
|
||||
}
|
||||
started = true;
|
||||
crc32_last = je->crc32;
|
||||
if (je->type == JE_SMALL_WRITE || je->type == JE_SMALL_WRITE_INSTANT)
|
||||
{
|
||||
journal_calc_data_pos = journal_pos;
|
||||
if (journal_pos + je->small_write.len > dsk.journal_len)
|
||||
{
|
||||
// data continues from the beginning of the journal
|
||||
journal_calc_data_pos = journal_pos = dsk.journal_block_size;
|
||||
wrapped = true;
|
||||
}
|
||||
journal_pos += je->small_write.len;
|
||||
if (journal_pos >= dsk.journal_len)
|
||||
{
|
||||
journal_pos = dsk.journal_block_size;
|
||||
wrapped = true;
|
||||
}
|
||||
small_write_data = memalign_or_die(MEM_ALIGNMENT, je->small_write.len);
|
||||
assert(pread(dsk.journal_fd, small_write_data, je->small_write.len, dsk.journal_offset+je->small_write.data_offset) == je->small_write.len);
|
||||
data_crc32 = je_start.csum_block_size ? 0 : crc32c(0, small_write_data, je->small_write.len);
|
||||
data_csum_valid = (data_crc32 == je->small_write.crc32_data);
|
||||
if (je_start.csum_block_size && je->small_write.len > 0)
|
||||
{
|
||||
// like in enqueue_write()
|
||||
uint32_t start = je->small_write.offset / je_start.csum_block_size;
|
||||
uint32_t end = (je->small_write.offset+je->small_write.len-1) / je_start.csum_block_size;
|
||||
uint32_t data_csum_size = (end-start+1) * (je_start.data_csum_type & 0xFF);
|
||||
if (je->size < sizeof(journal_entry_small_write) + data_csum_size)
|
||||
{
|
||||
data_csum_valid = false;
|
||||
}
|
||||
else
|
||||
{
|
||||
uint32_t calc_csum = 0;
|
||||
uint32_t *block_csums = (uint32_t*)((uint8_t*)je + je->size - data_csum_size);
|
||||
if (start == end)
|
||||
{
|
||||
calc_csum = crc32c(0, (uint8_t*)small_write_data, je->small_write.len);
|
||||
data_csum_valid = data_csum_valid && (calc_csum == *block_csums++);
|
||||
}
|
||||
else
|
||||
{
|
||||
// First block
|
||||
calc_csum = crc32c(0, (uint8_t*)small_write_data,
|
||||
je_start.csum_block_size*(start+1)-je->small_write.offset);
|
||||
data_csum_valid = data_csum_valid && (calc_csum == *block_csums++);
|
||||
// Intermediate blocks
|
||||
for (uint32_t i = start+1; i < end; i++)
|
||||
{
|
||||
calc_csum = crc32c(0, (uint8_t*)small_write_data +
|
||||
je_start.csum_block_size*i-je->small_write.offset, je_start.csum_block_size);
|
||||
data_csum_valid = data_csum_valid && (calc_csum == *block_csums++);
|
||||
}
|
||||
// Last block
|
||||
calc_csum = crc32c(
|
||||
0, (uint8_t*)small_write_data + end*je_start.csum_block_size - je->small_write.offset,
|
||||
je->small_write.offset+je->small_write.len - end*je_start.csum_block_size
|
||||
);
|
||||
data_csum_valid = data_csum_valid && (calc_csum == *block_csums++);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
iter_fn(entry, je);
|
||||
if (je->type == JE_SMALL_WRITE || je->type == JE_SMALL_WRITE_INSTANT)
|
||||
{
|
||||
free(small_write_data);
|
||||
small_write_data = NULL;
|
||||
}
|
||||
pos += je->size;
|
||||
entry++;
|
||||
}
|
||||
if (wrapped)
|
||||
{
|
||||
journal_pos = dsk.journal_len;
|
||||
}
|
||||
return entry;
|
||||
}
|
||||
|
||||
void disk_tool_t::dump_journal_entry(int num, journal_entry *je, bool json)
|
||||
{
|
||||
if (json)
|
||||
{
|
||||
if (!first_entry)
|
||||
printf(",\n");
|
||||
first_entry = false;
|
||||
printf(
|
||||
"{\"crc32\":\"%08x\",\"valid\":%s,\"crc32_prev\":\"%08x\"",
|
||||
je->crc32, (je_crc32(je) == je->crc32 ? "true" : "false"), je->crc32_prev
|
||||
);
|
||||
}
|
||||
else
|
||||
{
|
||||
printf(
|
||||
"entry % 3d: crc32=%08x %s prev=%08x ",
|
||||
num, je->crc32, (je_crc32(je) == je->crc32 ? "(valid)" : "(invalid)"), je->crc32_prev
|
||||
);
|
||||
}
|
||||
if (je->type == JE_START)
|
||||
{
|
||||
printf(
|
||||
json ? ",\"type\":\"start\",\"start\":\"0x%jx\"" : "je_start start=%08jx",
|
||||
je->start.journal_start
|
||||
);
|
||||
if (je->start.data_csum_type)
|
||||
{
|
||||
printf(
|
||||
json ? ",\"data_csum_type\":\"%s\",\"csum_block_size\":%u" : " data_csum_type=%s csum_block_size=%u",
|
||||
csum_type_str(je->start.data_csum_type).c_str(), je->start.csum_block_size
|
||||
);
|
||||
}
|
||||
printf(json ? "}" : "\n");
|
||||
}
|
||||
else if (je->type == JE_SMALL_WRITE || je->type == JE_SMALL_WRITE_INSTANT)
|
||||
{
|
||||
auto & sw = je->small_write;
|
||||
printf(
|
||||
json ? ",\"type\":\"small_write%s\",\"inode\":\"0x%jx\",\"stripe\":\"0x%jx\",\"ver\":\"%ju\",\"offset\":%u,\"len\":%u,\"loc\":\"0x%jx\""
|
||||
: "je_small_write%s oid=%jx:%jx ver=%ju offset=%u len=%u loc=%08jx",
|
||||
je->type == JE_SMALL_WRITE_INSTANT ? "_instant" : "",
|
||||
sw.oid.inode, sw.oid.stripe, sw.version, sw.offset, sw.len, sw.data_offset
|
||||
);
|
||||
if (journal_calc_data_pos != sw.data_offset)
|
||||
{
|
||||
printf(json ? ",\"bad_loc\":true,\"calc_loc\":\"0x%jx\""
|
||||
: " (mismatched, calculated = %08jx)", journal_pos);
|
||||
}
|
||||
uint32_t data_csum_size = (!je_start.csum_block_size
|
||||
? 0
|
||||
: ((sw.offset + sw.len - 1)/je_start.csum_block_size - sw.offset/je_start.csum_block_size + 1)
|
||||
*(je_start.data_csum_type & 0xFF));
|
||||
if (je->size > sizeof(journal_entry_small_write) + data_csum_size)
|
||||
{
|
||||
printf(json ? ",\"bitmap\":\"" : " (bitmap: ");
|
||||
for (int i = sizeof(journal_entry_small_write); i < je->size - data_csum_size; i++)
|
||||
{
|
||||
printf("%02x", ((uint8_t*)je)[i]);
|
||||
}
|
||||
printf(json ? "\"" : ")");
|
||||
}
|
||||
if (dump_with_data)
|
||||
{
|
||||
printf(json ? ",\"data\":\"" : " (data: ");
|
||||
for (int i = 0; i < sw.len; i++)
|
||||
{
|
||||
printf("%02x", ((uint8_t*)small_write_data)[i]);
|
||||
}
|
||||
printf(json ? "\"" : ")");
|
||||
}
|
||||
if (data_csum_size > 0 && je->size >= sizeof(journal_entry_small_write) + data_csum_size)
|
||||
{
|
||||
printf(json ? ",\"block_csums\":\"" : " block_csums=");
|
||||
uint8_t *block_csums = (uint8_t*)je + je->size - data_csum_size;
|
||||
for (int i = 0; i < data_csum_size; i++)
|
||||
printf("%02x", block_csums[i]);
|
||||
printf(json ? "\"" : "");
|
||||
}
|
||||
else
|
||||
{
|
||||
printf(json ? ",\"data_crc32\":\"%08x\"" : " data_crc32=%08x", sw.crc32_data);
|
||||
}
|
||||
printf(
|
||||
json ? ",\"data_valid\":%s}" : "%s\n",
|
||||
(data_csum_valid
|
||||
? (json ? "true" : " (valid)")
|
||||
: (json ? "false" : " (invalid)"))
|
||||
);
|
||||
}
|
||||
else if (je->type == JE_BIG_WRITE || je->type == JE_BIG_WRITE_INSTANT)
|
||||
{
|
||||
auto & bw = je->big_write;
|
||||
printf(
|
||||
json ? ",\"type\":\"big_write%s\",\"inode\":\"0x%jx\",\"stripe\":\"0x%jx\",\"ver\":\"%ju\",\"offset\":%u,\"len\":%u,\"loc\":\"0x%jx\""
|
||||
: "je_big_write%s oid=%jx:%jx ver=%ju offset=%u len=%u loc=%08jx",
|
||||
je->type == JE_BIG_WRITE_INSTANT ? "_instant" : "",
|
||||
bw.oid.inode, bw.oid.stripe, bw.version, bw.offset, bw.len, bw.location
|
||||
);
|
||||
uint32_t data_csum_size = (!je_start.csum_block_size
|
||||
? 0
|
||||
: ((bw.offset + bw.len - 1)/je_start.csum_block_size - bw.offset/je_start.csum_block_size + 1)
|
||||
*(je_start.data_csum_type & 0xFF));
|
||||
if (data_csum_size > 0 && je->size >= sizeof(journal_entry_big_write) + data_csum_size)
|
||||
{
|
||||
printf(json ? ",\"block_csums\":\"" : " block_csums=");
|
||||
uint8_t *block_csums = (uint8_t*)je + je->size - data_csum_size;
|
||||
for (int i = 0; i < data_csum_size; i++)
|
||||
printf("%02x", block_csums[i]);
|
||||
printf(json ? "\"" : "");
|
||||
}
|
||||
if (bw.size > sizeof(journal_entry_big_write) + data_csum_size)
|
||||
{
|
||||
printf(json ? ",\"bitmap\":\"" : " (bitmap: ");
|
||||
for (int i = sizeof(journal_entry_big_write); i < bw.size - data_csum_size; i++)
|
||||
{
|
||||
printf("%02x", ((uint8_t*)je)[i]);
|
||||
}
|
||||
printf(json ? "\"" : ")");
|
||||
}
|
||||
printf(json ? "}" : "\n");
|
||||
}
|
||||
else if (je->type == JE_STABLE)
|
||||
{
|
||||
printf(
|
||||
json ? ",\"type\":\"stable\",\"inode\":\"0x%jx\",\"stripe\":\"0x%jx\",\"ver\":\"%ju\"}"
|
||||
: "je_stable oid=%jx:%jx ver=%ju\n",
|
||||
je->stable.oid.inode, je->stable.oid.stripe, je->stable.version
|
||||
);
|
||||
}
|
||||
else if (je->type == JE_ROLLBACK)
|
||||
{
|
||||
printf(
|
||||
json ? ",\"type\":\"rollback\",\"inode\":\"0x%jx\",\"stripe\":\"0x%jx\",\"ver\":\"%ju\"}"
|
||||
: "je_rollback oid=%jx:%jx ver=%ju\n",
|
||||
je->rollback.oid.inode, je->rollback.oid.stripe, je->rollback.version
|
||||
);
|
||||
}
|
||||
else if (je->type == JE_DELETE)
|
||||
{
|
||||
printf(
|
||||
json ? ",\"type\":\"delete\",\"inode\":\"0x%jx\",\"stripe\":\"0x%jx\",\"ver\":\"%ju\"}"
|
||||
: "je_delete oid=%jx:%jx ver=%ju\n",
|
||||
je->del.oid.inode, je->del.oid.stripe, je->del.version
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
int disk_tool_t::write_json_journal(json11::Json entries)
|
||||
{
|
||||
new_journal_buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, new_journal_len);
|
||||
new_journal_ptr = new_journal_buf;
|
||||
new_journal_data = new_journal_ptr + dsk.journal_block_size;
|
||||
new_journal_in_pos = 0;
|
||||
memset(new_journal_buf, 0, new_journal_len);
|
||||
std::map<std::string,uint16_t> type_by_name = {
|
||||
{ "start", JE_START },
|
||||
{ "small_write", JE_SMALL_WRITE },
|
||||
{ "small_write_instant", JE_SMALL_WRITE_INSTANT },
|
||||
{ "big_write", JE_BIG_WRITE },
|
||||
{ "big_write_instant", JE_BIG_WRITE_INSTANT },
|
||||
{ "stable", JE_STABLE },
|
||||
{ "delete", JE_DELETE },
|
||||
{ "rollback", JE_ROLLBACK },
|
||||
};
|
||||
// Write start entry into the first block
|
||||
*((journal_entry_start*)new_journal_buf) = (journal_entry_start){
|
||||
.magic = JOURNAL_MAGIC,
|
||||
.type = JE_START,
|
||||
.size = sizeof(journal_entry_start),
|
||||
.journal_start = dsk.journal_block_size,
|
||||
.version = JOURNAL_VERSION_V2,
|
||||
.data_csum_type = dsk.data_csum_type,
|
||||
.csum_block_size = dsk.csum_block_size,
|
||||
};
|
||||
((journal_entry*)new_journal_buf)->crc32 = je_crc32((journal_entry*)new_journal_buf);
|
||||
new_journal_ptr += dsk.journal_block_size;
|
||||
new_journal_data = new_journal_ptr+dsk.journal_block_size;
|
||||
new_journal_in_pos = 0;
|
||||
for (const auto & rec: entries.array_items())
|
||||
{
|
||||
auto t_it = type_by_name.find(rec["type"].string_value());
|
||||
if (t_it == type_by_name.end())
|
||||
{
|
||||
fprintf(stderr, "Unknown journal entry type \"%s\", skipping\n", rec["type"].string_value().c_str());
|
||||
continue;
|
||||
}
|
||||
uint16_t type = t_it->second;
|
||||
if (type == JE_START)
|
||||
continue;
|
||||
uint32_t entry_size = (type == JE_START
|
||||
? sizeof(journal_entry_start)
|
||||
: (type == JE_SMALL_WRITE || type == JE_SMALL_WRITE_INSTANT
|
||||
? sizeof(journal_entry_small_write) + dsk.clean_entry_bitmap_size +
|
||||
(dsk.data_csum_type ? rec["len"].uint64_value()/dsk.csum_block_size*(dsk.data_csum_type & 0xFF) : 0)
|
||||
: (type == JE_BIG_WRITE || type == JE_BIG_WRITE_INSTANT
|
||||
? sizeof(journal_entry_big_write) + dsk.clean_entry_bitmap_size +
|
||||
(dsk.data_csum_type ? rec["len"].uint64_value()/dsk.csum_block_size*(dsk.data_csum_type & 0xFF) : 0)
|
||||
: sizeof(journal_entry_del))));
|
||||
if (dsk.journal_block_size < new_journal_in_pos + entry_size)
|
||||
{
|
||||
new_journal_ptr = new_journal_data;
|
||||
if (new_journal_ptr-new_journal_buf >= new_journal_len)
|
||||
{
|
||||
fprintf(stderr, "Error: entries don't fit to the new journal\n");
|
||||
free(new_journal_buf);
|
||||
return 1;
|
||||
}
|
||||
new_journal_data = new_journal_ptr+dsk.journal_block_size;
|
||||
new_journal_in_pos = 0;
|
||||
if (dsk.journal_block_size < entry_size)
|
||||
{
|
||||
fprintf(stderr, "Error: journal entry too large (%u bytes)\n", entry_size);
|
||||
free(new_journal_buf);
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
journal_entry *ne = (journal_entry*)(new_journal_ptr + new_journal_in_pos);
|
||||
if (type == JE_SMALL_WRITE || type == JE_SMALL_WRITE_INSTANT)
|
||||
{
|
||||
if (new_journal_data - new_journal_buf + ne->small_write.len > new_journal_len)
|
||||
{
|
||||
fprintf(stderr, "Error: entries don't fit to the new journal\n");
|
||||
free(new_journal_buf);
|
||||
return 1;
|
||||
}
|
||||
*((journal_entry_small_write*)ne) = (journal_entry_small_write){
|
||||
.magic = JOURNAL_MAGIC,
|
||||
.type = type,
|
||||
.size = entry_size,
|
||||
.crc32_prev = new_crc32_prev,
|
||||
.oid = {
|
||||
.inode = sscanf_json(NULL, rec["inode"]),
|
||||
.stripe = sscanf_json(NULL, rec["stripe"]),
|
||||
},
|
||||
.version = rec["ver"].uint64_value(),
|
||||
.offset = (uint32_t)rec["offset"].uint64_value(),
|
||||
.len = (uint32_t)rec["len"].uint64_value(),
|
||||
.data_offset = (uint64_t)(new_journal_data-new_journal_buf),
|
||||
.crc32_data = !dsk.data_csum_type ? 0 : (uint32_t)sscanf_json("%x", rec["data_crc32"]),
|
||||
};
|
||||
uint32_t data_csum_size = !dsk.data_csum_type ? 0 : ne->small_write.len/dsk.csum_block_size*(dsk.data_csum_type & 0xFF);
|
||||
fromhexstr(rec["bitmap"].string_value(), dsk.clean_entry_bitmap_size, ((uint8_t*)ne) + sizeof(journal_entry_small_write) + data_csum_size);
|
||||
fromhexstr(rec["data"].string_value(), ne->small_write.len, new_journal_data);
|
||||
if (dsk.data_csum_type)
|
||||
fromhexstr(rec["block_csums"].string_value(), data_csum_size, ((uint8_t*)ne) + sizeof(journal_entry_small_write));
|
||||
if (rec["data"].is_string())
|
||||
{
|
||||
if (!dsk.data_csum_type)
|
||||
ne->small_write.crc32_data = crc32c(0, new_journal_data, ne->small_write.len);
|
||||
else if (dsk.data_csum_type == BLOCKSTORE_CSUM_CRC32C)
|
||||
{
|
||||
uint32_t *block_csums = (uint32_t*)(((uint8_t*)ne) + sizeof(journal_entry_small_write));
|
||||
for (uint32_t i = 0; i < ne->small_write.len; i += dsk.csum_block_size, block_csums++)
|
||||
*block_csums = crc32c(0, new_journal_data+i, dsk.csum_block_size);
|
||||
}
|
||||
}
|
||||
new_journal_data += ne->small_write.len;
|
||||
}
|
||||
else if (type == JE_BIG_WRITE || type == JE_BIG_WRITE_INSTANT)
|
||||
{
|
||||
*((journal_entry_big_write*)ne) = (journal_entry_big_write){
|
||||
.magic = JOURNAL_MAGIC,
|
||||
.type = type,
|
||||
.size = entry_size,
|
||||
.crc32_prev = new_crc32_prev,
|
||||
.oid = {
|
||||
.inode = sscanf_json(NULL, rec["inode"]),
|
||||
.stripe = sscanf_json(NULL, rec["stripe"]),
|
||||
},
|
||||
.version = rec["ver"].uint64_value(),
|
||||
.offset = (uint32_t)rec["offset"].uint64_value(),
|
||||
.len = (uint32_t)rec["len"].uint64_value(),
|
||||
.location = sscanf_json(NULL, rec["loc"]),
|
||||
};
|
||||
uint32_t data_csum_size = !dsk.data_csum_type ? 0 : ne->big_write.len/dsk.csum_block_size*(dsk.data_csum_type & 0xFF);
|
||||
fromhexstr(rec["bitmap"].string_value(), dsk.clean_entry_bitmap_size, ((uint8_t*)ne) + sizeof(journal_entry_big_write) + data_csum_size);
|
||||
if (dsk.data_csum_type)
|
||||
fromhexstr(rec["block_csums"].string_value(), data_csum_size, ((uint8_t*)ne) + sizeof(journal_entry_big_write));
|
||||
}
|
||||
else if (type == JE_STABLE || type == JE_ROLLBACK || type == JE_DELETE)
|
||||
{
|
||||
*((journal_entry_del*)ne) = (journal_entry_del){
|
||||
.magic = JOURNAL_MAGIC,
|
||||
.type = type,
|
||||
.size = entry_size,
|
||||
.crc32_prev = new_crc32_prev,
|
||||
.oid = {
|
||||
.inode = sscanf_json(NULL, rec["inode"]),
|
||||
.stripe = sscanf_json(NULL, rec["stripe"]),
|
||||
},
|
||||
.version = rec["ver"].uint64_value(),
|
||||
};
|
||||
}
|
||||
ne->crc32 = je_crc32(ne);
|
||||
new_crc32_prev = ne->crc32;
|
||||
new_journal_in_pos += ne->size;
|
||||
}
|
||||
int r = resize_write_new_journal();
|
||||
free(new_journal_buf);
|
||||
return r;
|
||||
}
|
||||
@@ -0,0 +1,297 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#include "disk_tool.h"
|
||||
#include "rw_blocking.h"
|
||||
#include "osd_id.h"
|
||||
|
||||
int disk_tool_t::process_meta(std::function<void(blockstore_meta_header_v2_t *)> hdr_fn,
|
||||
std::function<void(uint64_t, clean_disk_entry*, uint8_t*)> record_fn)
|
||||
{
|
||||
if (dsk.meta_block_size % DIRECT_IO_ALIGNMENT)
|
||||
{
|
||||
fprintf(stderr, "Invalid metadata block size: is not a multiple of %d\n", DIRECT_IO_ALIGNMENT);
|
||||
return 1;
|
||||
}
|
||||
dsk.meta_fd = open(dsk.meta_device.c_str(), O_DIRECT|O_RDONLY);
|
||||
if (dsk.meta_fd < 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to open metadata device %s: %s\n", dsk.meta_device.c_str(), strerror(errno));
|
||||
return 1;
|
||||
}
|
||||
int buf_size = 1024*1024;
|
||||
if (buf_size % dsk.meta_block_size)
|
||||
buf_size = 8*dsk.meta_block_size;
|
||||
if (buf_size > dsk.meta_len)
|
||||
buf_size = dsk.meta_len;
|
||||
void *data = memalign_or_die(MEM_ALIGNMENT, buf_size);
|
||||
lseek64(dsk.meta_fd, dsk.meta_offset, 0);
|
||||
read_blocking(dsk.meta_fd, data, dsk.meta_block_size);
|
||||
// Check superblock
|
||||
blockstore_meta_header_v2_t *hdr = (blockstore_meta_header_v2_t *)data;
|
||||
if (hdr->zero == 0 && hdr->magic == BLOCKSTORE_META_MAGIC_V1)
|
||||
{
|
||||
if (hdr->version == BLOCKSTORE_META_FORMAT_V1)
|
||||
{
|
||||
// Vitastor 0.6-0.8 - static array of clean_disk_entry with bitmaps
|
||||
hdr->data_csum_type = 0;
|
||||
hdr->csum_block_size = 0;
|
||||
hdr->header_csum = 0;
|
||||
}
|
||||
else if (hdr->version == BLOCKSTORE_META_FORMAT_V2)
|
||||
{
|
||||
// Vitastor 0.9 - static array of clean_disk_entry with bitmaps and checksums
|
||||
if (hdr->data_csum_type != 0 &&
|
||||
hdr->data_csum_type != BLOCKSTORE_CSUM_CRC32C)
|
||||
{
|
||||
fprintf(stderr, "I don't know checksum format %u, the only supported format is crc32c = %u.\n", hdr->data_csum_type, BLOCKSTORE_CSUM_CRC32C);
|
||||
free(data);
|
||||
close(dsk.meta_fd);
|
||||
dsk.meta_fd = -1;
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// Unsupported version
|
||||
fprintf(stderr, "Metadata format is too new for me (stored version is %ju, max supported %u).\n", hdr->version, BLOCKSTORE_META_FORMAT_V2);
|
||||
free(data);
|
||||
close(dsk.meta_fd);
|
||||
dsk.meta_fd = -1;
|
||||
return 1;
|
||||
}
|
||||
if (hdr->meta_block_size != dsk.meta_block_size)
|
||||
{
|
||||
fprintf(stderr, "Using block size of %u bytes based on information from the superblock\n", hdr->meta_block_size);
|
||||
dsk.meta_block_size = hdr->meta_block_size;
|
||||
if (buf_size % dsk.meta_block_size)
|
||||
{
|
||||
buf_size = 8*dsk.meta_block_size;
|
||||
void *new_data = memalign_or_die(MEM_ALIGNMENT, buf_size);
|
||||
memcpy(new_data, data, dsk.meta_block_size);
|
||||
free(data);
|
||||
data = new_data;
|
||||
hdr = (blockstore_meta_header_v2_t *)data;
|
||||
}
|
||||
}
|
||||
dsk.meta_format = hdr->version;
|
||||
dsk.data_block_size = hdr->data_block_size;
|
||||
dsk.csum_block_size = hdr->csum_block_size;
|
||||
dsk.data_csum_type = hdr->data_csum_type;
|
||||
dsk.bitmap_granularity = hdr->bitmap_granularity;
|
||||
dsk.clean_entry_bitmap_size = (hdr->data_block_size / hdr->bitmap_granularity + 7) / 8;
|
||||
dsk.clean_entry_size = sizeof(clean_disk_entry) + 2*dsk.clean_entry_bitmap_size
|
||||
+ (hdr->data_csum_type
|
||||
? ((hdr->data_block_size+hdr->csum_block_size-1)/hdr->csum_block_size
|
||||
*(hdr->data_csum_type & 0xff))
|
||||
: 0)
|
||||
+ (dsk.meta_format == BLOCKSTORE_META_FORMAT_V2 ? 4 /*entry_csum*/ : 0);
|
||||
uint64_t block_num = 0;
|
||||
hdr_fn(hdr);
|
||||
hdr = NULL;
|
||||
meta_pos = dsk.meta_block_size;
|
||||
lseek64(dsk.meta_fd, dsk.meta_offset+meta_pos, 0);
|
||||
while (meta_pos < dsk.meta_len)
|
||||
{
|
||||
uint64_t read_len = buf_size < dsk.meta_len-meta_pos ? buf_size : dsk.meta_len-meta_pos;
|
||||
read_blocking(dsk.meta_fd, data, read_len);
|
||||
meta_pos += read_len;
|
||||
for (uint64_t blk = 0; blk < read_len; blk += dsk.meta_block_size)
|
||||
{
|
||||
for (uint64_t ioff = 0; ioff <= dsk.meta_block_size-dsk.clean_entry_size; ioff += dsk.clean_entry_size, block_num++)
|
||||
{
|
||||
clean_disk_entry *entry = (clean_disk_entry*)((uint8_t*)data + blk + ioff);
|
||||
if (entry->oid.inode)
|
||||
{
|
||||
if (dsk.data_csum_type)
|
||||
{
|
||||
uint32_t *entry_csum = (uint32_t*)((uint8_t*)entry + dsk.clean_entry_size - 4);
|
||||
if (*entry_csum != crc32c(0, entry, dsk.clean_entry_size - 4))
|
||||
{
|
||||
fprintf(stderr, "Metadata entry %ju is corrupt (checksum mismatch), skipping\n", block_num);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
record_fn(block_num, entry, entry->bitmap);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// Vitastor 0.4-0.5 - static array of clean_disk_entry
|
||||
dsk.clean_entry_bitmap_size = 0;
|
||||
dsk.clean_entry_size = sizeof(clean_disk_entry);
|
||||
uint64_t block_num = 0;
|
||||
hdr_fn(NULL);
|
||||
while (meta_pos < dsk.meta_len)
|
||||
{
|
||||
uint64_t read_len = buf_size < dsk.meta_len-meta_pos ? buf_size : dsk.meta_len-meta_pos;
|
||||
read_blocking(dsk.meta_fd, data, read_len);
|
||||
meta_pos += read_len;
|
||||
for (uint64_t blk = 0; blk < read_len; blk += dsk.meta_block_size)
|
||||
{
|
||||
for (uint64_t ioff = 0; ioff < dsk.meta_block_size-dsk.clean_entry_size; ioff += dsk.clean_entry_size, block_num++)
|
||||
{
|
||||
clean_disk_entry *entry = (clean_disk_entry*)((uint8_t*)data + blk + ioff);
|
||||
if (entry->oid.inode)
|
||||
{
|
||||
record_fn(block_num, entry, NULL);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
free(data);
|
||||
close(dsk.meta_fd);
|
||||
dsk.meta_fd = -1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::dump_meta()
|
||||
{
|
||||
int r = process_meta(
|
||||
[this](blockstore_meta_header_v2_t *hdr) { dump_meta_header(hdr); },
|
||||
[this](uint64_t block_num, clean_disk_entry *entry, uint8_t *bitmap) { dump_meta_entry(block_num, entry, bitmap); }
|
||||
);
|
||||
if (r == 0)
|
||||
printf("\n]}\n");
|
||||
return r;
|
||||
}
|
||||
|
||||
void disk_tool_t::dump_meta_header(blockstore_meta_header_v2_t *hdr)
|
||||
{
|
||||
if (hdr)
|
||||
{
|
||||
if (hdr->version == BLOCKSTORE_META_FORMAT_V1)
|
||||
{
|
||||
printf(
|
||||
"{\"version\":\"0.6\",\"meta_block_size\":%u,\"data_block_size\":%u,\"bitmap_granularity\":%u,"
|
||||
"\"entries\":[\n",
|
||||
hdr->meta_block_size, hdr->data_block_size, hdr->bitmap_granularity
|
||||
);
|
||||
}
|
||||
else if (hdr->version == BLOCKSTORE_META_FORMAT_V2)
|
||||
{
|
||||
printf(
|
||||
"{\"version\":\"0.9\",\"meta_block_size\":%u,\"data_block_size\":%u,\"bitmap_granularity\":%u,"
|
||||
"\"data_csum_type\":%s,\"csum_block_size\":%u,\"entries\":[\n",
|
||||
hdr->meta_block_size, hdr->data_block_size, hdr->bitmap_granularity,
|
||||
csum_type_str(hdr->data_csum_type).c_str(), hdr->csum_block_size
|
||||
);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
printf("{\"version\":\"0.5\",\"meta_block_size\":%ju,\"entries\":[\n", dsk.meta_block_size);
|
||||
}
|
||||
first_entry = true;
|
||||
}
|
||||
|
||||
void disk_tool_t::dump_meta_entry(uint64_t block_num, clean_disk_entry *entry, uint8_t *bitmap)
|
||||
{
|
||||
printf(
|
||||
#define ENTRY_FMT "{\"block\":%ju,\"pool\":%u,\"inode\":\"0x%jx\",\"stripe\":\"0x%jx\",\"version\":%ju"
|
||||
(first_entry ? ENTRY_FMT : (",\n" ENTRY_FMT)),
|
||||
#undef ENTRY_FMT
|
||||
block_num, INODE_POOL(entry->oid.inode), INODE_NO_POOL(entry->oid.inode),
|
||||
entry->oid.stripe, entry->version
|
||||
);
|
||||
if (bitmap)
|
||||
{
|
||||
printf(",\"bitmap\":\"");
|
||||
for (uint64_t i = 0; i < dsk.clean_entry_bitmap_size; i++)
|
||||
{
|
||||
printf("%02x", bitmap[i]);
|
||||
}
|
||||
printf("\",\"ext_bitmap\":\"");
|
||||
for (uint64_t i = 0; i < dsk.clean_entry_bitmap_size; i++)
|
||||
{
|
||||
printf("%02x", bitmap[dsk.clean_entry_bitmap_size + i]);
|
||||
}
|
||||
if (dsk.csum_block_size && dsk.data_csum_type)
|
||||
{
|
||||
uint8_t *csums = bitmap + dsk.clean_entry_bitmap_size*2;
|
||||
printf("\",\"block_csums\":\"");
|
||||
for (uint64_t i = 0; i < (dsk.data_block_size+dsk.csum_block_size-1)/dsk.csum_block_size*(dsk.data_csum_type & 0xFF); i++)
|
||||
{
|
||||
printf("%02x", csums[i]);
|
||||
}
|
||||
}
|
||||
printf("\"}");
|
||||
}
|
||||
else
|
||||
{
|
||||
printf("}");
|
||||
}
|
||||
first_entry = false;
|
||||
}
|
||||
|
||||
int disk_tool_t::write_json_meta(json11::Json meta)
|
||||
{
|
||||
new_meta_buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, new_meta_len);
|
||||
memset(new_meta_buf, 0, new_meta_len);
|
||||
blockstore_meta_header_v2_t *new_hdr = (blockstore_meta_header_v2_t *)new_meta_buf;
|
||||
new_hdr->zero = 0;
|
||||
new_hdr->magic = BLOCKSTORE_META_MAGIC_V1;
|
||||
new_hdr->version = meta["version"].uint64_value() == BLOCKSTORE_META_FORMAT_V1
|
||||
? BLOCKSTORE_META_FORMAT_V1 : BLOCKSTORE_META_FORMAT_V2;
|
||||
new_hdr->meta_block_size = meta["meta_block_size"].uint64_value()
|
||||
? meta["meta_block_size"].uint64_value() : 4096;
|
||||
new_hdr->data_block_size = meta["data_block_size"].uint64_value()
|
||||
? meta["data_block_size"].uint64_value() : 131072;
|
||||
new_hdr->bitmap_granularity = meta["bitmap_granularity"].uint64_value()
|
||||
? meta["bitmap_granularity"].uint64_value() : 4096;
|
||||
new_hdr->data_csum_type = meta["data_csum_type"].is_number()
|
||||
? meta["data_csum_type"].uint64_value()
|
||||
: (meta["data_csum_type"].string_value() == "crc32c"
|
||||
? BLOCKSTORE_CSUM_CRC32C
|
||||
: BLOCKSTORE_CSUM_NONE);
|
||||
new_hdr->csum_block_size = meta["csum_block_size"].uint64_value();
|
||||
uint32_t new_clean_entry_header_size = (new_hdr->version == BLOCKSTORE_META_FORMAT_V1
|
||||
? sizeof(clean_disk_entry) : sizeof(clean_disk_entry) + 4 /*entry_csum*/);
|
||||
new_clean_entry_bitmap_size = (new_hdr->data_block_size / new_hdr->bitmap_granularity + 7) / 8;
|
||||
new_data_csum_size = (new_hdr->data_csum_type
|
||||
? ((new_hdr->data_block_size+new_hdr->csum_block_size-1)/new_hdr->csum_block_size*(new_hdr->data_csum_type & 0xFF))
|
||||
: 0);
|
||||
new_clean_entry_size = new_clean_entry_header_size + 2*new_clean_entry_bitmap_size + new_data_csum_size;
|
||||
new_entries_per_block = new_hdr->meta_block_size / new_clean_entry_size;
|
||||
for (const auto & e: meta["entries"].array_items())
|
||||
{
|
||||
uint64_t data_block = e["block"].uint64_value();
|
||||
uint64_t mb = 1 + data_block/new_entries_per_block;
|
||||
if (mb >= new_meta_len/new_hdr->meta_block_size)
|
||||
{
|
||||
free(new_meta_buf);
|
||||
new_meta_buf = NULL;
|
||||
fprintf(stderr, "Metadata (data block %ju) doesn't fit into the new area\n", data_block);
|
||||
return 1;
|
||||
}
|
||||
clean_disk_entry *new_entry = (clean_disk_entry*)(new_meta_buf +
|
||||
new_hdr->meta_block_size*mb +
|
||||
new_clean_entry_size*(data_block % new_entries_per_block));
|
||||
new_entry->oid.inode = (sscanf_json(NULL, e["pool"]) << (64-POOL_ID_BITS)) | sscanf_json(NULL, e["inode"]);
|
||||
new_entry->oid.stripe = sscanf_json(NULL, e["stripe"]);
|
||||
new_entry->version = sscanf_json(NULL, e["version"]);
|
||||
fromhexstr(e["bitmap"].string_value(), new_clean_entry_bitmap_size,
|
||||
((uint8_t*)new_entry) + sizeof(clean_disk_entry));
|
||||
fromhexstr(e["ext_bitmap"].string_value(), new_clean_entry_bitmap_size,
|
||||
((uint8_t*)new_entry) + sizeof(clean_disk_entry) + new_clean_entry_bitmap_size);
|
||||
if (new_hdr->version == BLOCKSTORE_META_FORMAT_V2)
|
||||
{
|
||||
if (new_hdr->data_csum_type != 0)
|
||||
{
|
||||
fromhexstr(e["data_csum"].string_value(), new_data_csum_size,
|
||||
((uint8_t*)new_entry) + sizeof(clean_disk_entry) + 2*new_clean_entry_bitmap_size);
|
||||
}
|
||||
uint32_t *new_entry_csum = (uint32_t*)(((uint8_t*)new_entry) + sizeof(clean_disk_entry) +
|
||||
2*new_clean_entry_bitmap_size + new_data_csum_size);
|
||||
*new_entry_csum = crc32c(0, new_entry, new_clean_entry_size - 4);
|
||||
}
|
||||
}
|
||||
int r = resize_write_new_meta();
|
||||
free(new_meta_buf);
|
||||
new_meta_buf = NULL;
|
||||
return r;
|
||||
}
|
||||
@@ -0,0 +1,660 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#include "disk_tool.h"
|
||||
#include "str_util.h"
|
||||
#include "osd_id.h"
|
||||
|
||||
int disk_tool_t::prepare_one(std::map<std::string, std::string> options, int is_hdd)
|
||||
{
|
||||
static const char *allow_additional_params[] = {
|
||||
"autosync_writes",
|
||||
"data_io",
|
||||
"meta_io",
|
||||
"journal_io",
|
||||
"max_write_iodepth",
|
||||
"max_write_iodepth",
|
||||
"min_flusher_count",
|
||||
"max_flusher_count",
|
||||
"inmemory_metadata",
|
||||
"inmemory_journal",
|
||||
"journal_sector_buffer_count",
|
||||
"journal_no_same_sector_overwrites",
|
||||
"throttle_small_writes",
|
||||
"throttle_target_iops",
|
||||
"throttle_target_mbs",
|
||||
"throttle_target_parallelism",
|
||||
"throttle_threshold_us",
|
||||
};
|
||||
if (options.find("force") == options.end())
|
||||
{
|
||||
std::vector<std::string> all_devs = { options["data_device"], options["meta_device"], options["journal_device"] };
|
||||
for (int i = 0; i < all_devs.size(); i++)
|
||||
{
|
||||
const auto & dev = all_devs[i];
|
||||
if (dev == "")
|
||||
continue;
|
||||
if (dev.substr(0, 22) != "/dev/disk/by-partuuid/")
|
||||
{
|
||||
// Partitions should be identified by GPT partition UUID
|
||||
fprintf(stderr, "%s does not start with /dev/disk/by-partuuid/. Partitions should be identified by GPT partition UUIDs\n", dev.c_str());
|
||||
return 1;
|
||||
}
|
||||
std::string real_dev = realpath_str(dev, false);
|
||||
if (real_dev == "")
|
||||
return 1;
|
||||
std::string parent_dev = get_parent_device(real_dev);
|
||||
if (parent_dev == "")
|
||||
return 1;
|
||||
if (parent_dev == real_dev)
|
||||
{
|
||||
fprintf(stderr, "%s is not a partition, not creating OSD without --force\n", dev.c_str());
|
||||
return 1;
|
||||
}
|
||||
if (i == 0 && is_hdd == -1)
|
||||
is_hdd = trim(read_file("/sys/block/"+parent_dev+"/queue/rotational")) == "1";
|
||||
std::string out;
|
||||
if (shell_exec({ "wipefs", dev }, "", &out, NULL) != 0 || out != "")
|
||||
{
|
||||
fprintf(stderr, "%s contains data, not creating OSD without --force. wipefs shows:\n%s", dev.c_str(), out.c_str());
|
||||
return 1;
|
||||
}
|
||||
json11::Json sb = read_osd_superblock(dev, false);
|
||||
if (!sb.is_null())
|
||||
{
|
||||
fprintf(stderr, "%s already contains Vitastor OSD superblock, not creating OSD without --force\n", dev.c_str());
|
||||
return 1;
|
||||
}
|
||||
if (fix_partition_type(dev) != 0)
|
||||
{
|
||||
fprintf(stderr, "%s has incorrect type and we failed to change it to Vitastor type\n", dev.c_str());
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (auto dev: std::vector<std::string>{"data", "meta", "journal"})
|
||||
{
|
||||
if (options[dev+"_device"] != "" && options["disable_"+dev+"_fsync"] == "auto")
|
||||
{
|
||||
int r = disable_cache(realpath_str(options[dev+"_device"], false));
|
||||
if (r != 0)
|
||||
{
|
||||
if (r == 1)
|
||||
fprintf(stderr, "Warning: disable_%s_fsync is auto, but cache status check failed. Leaving fsync on\n", dev.c_str());
|
||||
options["disable_"+dev+"_fsync"] = "0";
|
||||
}
|
||||
else
|
||||
options["disable_"+dev+"_fsync"] = "1";
|
||||
}
|
||||
}
|
||||
if (options["meta_device"] == "" || options["meta_device"] == options["data_device"])
|
||||
{
|
||||
options["disable_meta_fsync"] = options["disable_data_fsync"];
|
||||
}
|
||||
if (options["journal_device"] == "" || options["journal_device"] == options["meta_device"])
|
||||
{
|
||||
options["disable_journal_fsync"] = options["disable_meta_fsync"];
|
||||
}
|
||||
else if (options["journal_device"] == options["data_device"])
|
||||
{
|
||||
options["disable_journal_fsync"] = options["disable_data_fsync"];
|
||||
}
|
||||
// Calculate offsets if the same device is used for two or more of data, meta, and journal
|
||||
if (options["journal_size"] == "" && (options["journal_device"] == "" || options["journal_device"] == options["data_device"]))
|
||||
{
|
||||
options["journal_size"] = is_hdd || !json_is_true(options["disable_data_fsync"]) ? "128M" : "32M";
|
||||
}
|
||||
bool is_hybrid = is_hdd && options["journal_device"] != "" && options["journal_device"] != options["data_device"];
|
||||
if (is_hdd)
|
||||
{
|
||||
if (options["block_size"] == "")
|
||||
options["block_size"] = "1M";
|
||||
if (is_hybrid && options["throttle_small_writes"] == "")
|
||||
options["throttle_small_writes"] = "1";
|
||||
if (!is_hybrid && options.find("data_csum_type") != options.end() && options.at("data_csum_type") != "")
|
||||
options["csum_block_size"] = "32k";
|
||||
}
|
||||
else if (!json_is_true(options["disable_data_fsync"]))
|
||||
{
|
||||
if (options.find("min_flusher_count") == options.end())
|
||||
options["min_flusher_count"] = "32";
|
||||
if (options.find("max_flusher_count") == options.end())
|
||||
options["max_flusher_count"] = "256";
|
||||
if (options.find("autosync_writes") == options.end())
|
||||
options["autosync_writes"] = "512";
|
||||
}
|
||||
json11::Json::object sb;
|
||||
blockstore_disk_t dsk;
|
||||
try
|
||||
{
|
||||
dsk.parse_config(options);
|
||||
dsk.data_io = dsk.meta_io = dsk.journal_io = "direct";
|
||||
dsk.open_data();
|
||||
dsk.open_meta();
|
||||
dsk.open_journal();
|
||||
dsk.calc_lengths(true);
|
||||
sb = json11::Json::object {
|
||||
{ "data_device", options["data_device"] },
|
||||
{ "meta_device", options["meta_device"] },
|
||||
{ "journal_device", options["journal_device"] },
|
||||
{ "block_size", (uint64_t)dsk.data_block_size },
|
||||
{ "meta_block_size", dsk.meta_block_size },
|
||||
{ "journal_block_size", dsk.journal_block_size },
|
||||
{ "data_size", dsk.cfg_data_size },
|
||||
{ "disk_alignment", (uint64_t)dsk.disk_alignment },
|
||||
{ "bitmap_granularity", dsk.bitmap_granularity },
|
||||
{ "disable_device_lock", dsk.disable_flock },
|
||||
{ "journal_offset", 4096 },
|
||||
{ "meta_offset", 4096 + (dsk.meta_device == dsk.journal_device ? dsk.journal_len : 0) },
|
||||
{ "data_offset", 4096 + (dsk.data_device == dsk.meta_device ? dsk.meta_len : 0) +
|
||||
(dsk.data_device == dsk.journal_device ? dsk.journal_len : 0) },
|
||||
{ "journal_no_same_sector_overwrites", !is_hdd || is_hybrid },
|
||||
{ "journal_sector_buffer_count", 1024 },
|
||||
{ "disable_data_fsync", json_is_true(options["disable_data_fsync"]) },
|
||||
{ "disable_meta_fsync", json_is_true(options["disable_meta_fsync"]) },
|
||||
{ "disable_journal_fsync", json_is_true(options["disable_journal_fsync"]) },
|
||||
{ "skip_cache_check", json_is_true(options["skip_cache_check"]) },
|
||||
{ "immediate_commit", json_is_true(options["disable_data_fsync"])
|
||||
? (json_is_true(options["disable_journal_fsync"]) ? "all" : "small") : "none" },
|
||||
};
|
||||
for (int i = 0; i < sizeof(allow_additional_params)/sizeof(allow_additional_params[0]); i++)
|
||||
{
|
||||
auto it = options.find(allow_additional_params[i]);
|
||||
if (it != options.end() && it->second != "")
|
||||
{
|
||||
sb[it->first] = it->second;
|
||||
}
|
||||
}
|
||||
}
|
||||
catch (std::exception & e)
|
||||
{
|
||||
dsk.close_all();
|
||||
fprintf(stderr, "%s\n", e.what());
|
||||
return 1;
|
||||
}
|
||||
std::string osd_num_str;
|
||||
if (shell_exec({ "vitastor-cli", "alloc-osd" }, "", &osd_num_str, NULL) != 0)
|
||||
{
|
||||
dsk.close_all();
|
||||
return 1;
|
||||
}
|
||||
osd_num_t osd_num = stoull_full(trim(osd_num_str), 10);
|
||||
if (!osd_num)
|
||||
{
|
||||
dsk.close_all();
|
||||
fprintf(stderr, "Could not create OSD. vitastor-cli alloc-osd didn't return a valid OSD number:\n%s", osd_num_str.c_str());
|
||||
return 1;
|
||||
}
|
||||
sb["osd_num"] = osd_num;
|
||||
// Zero out metadata and journal
|
||||
if (write_zero(dsk.meta_fd, dsk.meta_offset, dsk.meta_len) != 0 ||
|
||||
write_zero(dsk.journal_fd, dsk.journal_offset, dsk.journal_len) != 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to zero out metadata or journal: %s\n", strerror(errno));
|
||||
dsk.close_all();
|
||||
return 1;
|
||||
}
|
||||
dsk.close_all();
|
||||
// Write superblocks
|
||||
bool sep_m = options["meta_device"] != "" &&
|
||||
options["meta_device"] != options["data_device"];
|
||||
bool sep_j = options["journal_device"] != "" &&
|
||||
options["journal_device"] != options["data_device"] &&
|
||||
options["journal_device"] != options["meta_device"];
|
||||
if (!write_osd_superblock(options["data_device"], sb) ||
|
||||
sep_m && !write_osd_superblock(options["meta_device"], sb) ||
|
||||
sep_j && !write_osd_superblock(options["journal_device"], sb))
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
auto desc = realpath_str(options["data_device"]);
|
||||
if (sep_m)
|
||||
desc += " with metadata on "+realpath_str(options["meta_device"]);
|
||||
if (sep_j)
|
||||
desc += (sep_m ? " and journal on " : " with journal on ") + realpath_str(options["journal_device"]);
|
||||
fprintf(stderr, "Initialized OSD %ju on %s\n", osd_num, desc.c_str());
|
||||
if (shell_exec({ "systemctl", "enable", "--now", "vitastor-osd@"+std::to_string(osd_num) }, "", NULL, NULL) != 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to enable systemd unit vitastor-osd@%ju\n", osd_num);
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
std::vector<vitastor_dev_info_t> disk_tool_t::collect_devices(const std::vector<std::string> & devices)
|
||||
{
|
||||
std::vector<vitastor_dev_info_t> devinfo;
|
||||
for (auto & dev: devices)
|
||||
{
|
||||
// Check if the device is a whole disk
|
||||
if (dev.substr(0, 5) != "/dev/")
|
||||
{
|
||||
fprintf(stderr, "%s does not start with /dev/, ignoring\n", dev.c_str());
|
||||
continue;
|
||||
}
|
||||
struct stat dev_st, sys_st;
|
||||
if (stat(dev.c_str(), &dev_st) < 0)
|
||||
{
|
||||
if (errno == ENOENT)
|
||||
{
|
||||
fprintf(stderr, "%s does not exist, skipping\n", dev.c_str());
|
||||
continue;
|
||||
}
|
||||
fprintf(stderr, "Error checking %s: %s\n", dev.c_str(), strerror(errno));
|
||||
return {};
|
||||
}
|
||||
uint64_t dev_size = dev_st.st_size;
|
||||
if (S_ISBLK(dev_st.st_mode))
|
||||
{
|
||||
int fd = open(dev.c_str(), O_DIRECT|O_RDWR);
|
||||
if (fd < 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to open %s: %s\n", dev.c_str(), strerror(errno));
|
||||
return {};
|
||||
}
|
||||
if (ioctl(fd, BLKGETSIZE64, &dev_size) < 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to get %s size: %s\n", dev.c_str(), strerror(errno));
|
||||
close(fd);
|
||||
return {};
|
||||
}
|
||||
close(fd);
|
||||
}
|
||||
if (stat(("/sys/block/"+dev.substr(5)).c_str(), &sys_st) < 0)
|
||||
{
|
||||
if (errno == ENOENT)
|
||||
{
|
||||
fprintf(stderr, "%s is probably a partition (no entry in /sys/block/), ignoring\n", dev.c_str());
|
||||
continue;
|
||||
}
|
||||
fprintf(stderr, "Error checking /sys/block/%s: %s\n", dev.c_str()+5, strerror(errno));
|
||||
return {};
|
||||
}
|
||||
// Check if the device is an SSD
|
||||
bool is_hdd = trim(read_file("/sys/block/"+dev.substr(5)+"/queue/rotational")) == "1";
|
||||
// Check if it has a partition table
|
||||
json11::Json pt = read_parttable(dev);
|
||||
if (pt.is_bool() && !pt.bool_value())
|
||||
{
|
||||
// Error reading table
|
||||
return {};
|
||||
}
|
||||
if (pt.is_null())
|
||||
{
|
||||
// No partition table
|
||||
std::string out;
|
||||
int r = shell_exec({ "wipefs", dev }, "", &out, NULL);
|
||||
if (r != 0 || out != "")
|
||||
{
|
||||
fprintf(stderr, "%s contains data, skipping:\n %s\n", dev.c_str(), str_replace(trim(out), "\n", "\n ").c_str());
|
||||
continue;
|
||||
}
|
||||
}
|
||||
int osds = 0;
|
||||
for (const auto & p: pt["partitions"].array_items())
|
||||
if (strtolower(p["type"].string_value()) == VITASTOR_PART_TYPE)
|
||||
osds++;
|
||||
devinfo.push_back((vitastor_dev_info_t){
|
||||
.path = dev,
|
||||
.is_hdd = is_hdd,
|
||||
.pt = pt,
|
||||
.osd_part_count = osds,
|
||||
.size = !pt.is_null() ? dev_size_from_parttable(pt) : dev_size,
|
||||
.free = !pt.is_null() ? free_from_parttable(pt) : dev_size,
|
||||
});
|
||||
}
|
||||
if (!devinfo.size())
|
||||
{
|
||||
fprintf(stderr, "No suitable devices found\n");
|
||||
}
|
||||
return devinfo;
|
||||
}
|
||||
|
||||
// Return null in case of an error
|
||||
json11::Json disk_tool_t::add_partitions(vitastor_dev_info_t & devinfo, std::vector<std::string> sizes)
|
||||
{
|
||||
std::string script = "label: gpt\n\n";
|
||||
std::set<std::string> is_old;
|
||||
for (auto part: devinfo.pt["partitions"].array_items())
|
||||
{
|
||||
// Old partitions
|
||||
is_old.insert(part["uuid"].string_value());
|
||||
script += part["node"].string_value()+": ";
|
||||
int n = 0;
|
||||
for (auto & kv: part.object_items())
|
||||
{
|
||||
if (kv.first != "node")
|
||||
{
|
||||
if (n++)
|
||||
script += ", ";
|
||||
script += kv.first+"="+(kv.second.is_string() ? kv.second.string_value() : kv.second.dump());
|
||||
}
|
||||
}
|
||||
script += "\n";
|
||||
}
|
||||
for (auto size: sizes)
|
||||
{
|
||||
script += "+ "+size+" "+std::string(VITASTOR_PART_TYPE)+"\n";
|
||||
}
|
||||
std::string out;
|
||||
if (shell_exec({ "sfdisk", "--no-reread", "--force", devinfo.path }, script, &out, NULL) != 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to add %zu partition(s) with sfdisk\n", sizes.size());
|
||||
return {};
|
||||
}
|
||||
// Get new partition table and find created partitions
|
||||
json11::Json newpt = read_parttable(devinfo.path);
|
||||
json11::Json::array new_parts;
|
||||
for (const auto & part: newpt["partitions"].array_items())
|
||||
{
|
||||
if (is_old.find(part["uuid"].string_value()) == is_old.end())
|
||||
{
|
||||
new_parts.push_back(part);
|
||||
}
|
||||
}
|
||||
if (new_parts.size() != sizes.size())
|
||||
{
|
||||
fprintf(stderr, "Failed to add %zu partition(s) with sfdisk: new partitions not found in table\n", sizes.size());
|
||||
return {};
|
||||
}
|
||||
// Check if new nodes exist and run partprobe if not
|
||||
// FIXME: We could use parted instead of sfdisk because partprobe is already a part of parted
|
||||
int iter = 0, r;
|
||||
while (true)
|
||||
{
|
||||
for (const auto & part: new_parts)
|
||||
{
|
||||
struct stat st;
|
||||
if (stat(part["node"].string_value().c_str(), &st) < 0)
|
||||
{
|
||||
if (errno == ENOENT)
|
||||
{
|
||||
iter++;
|
||||
// Run partprobe
|
||||
std::string out;
|
||||
if (iter > 1 || (r = shell_exec({ "partprobe", devinfo.path }, "", &out, NULL)) != 0)
|
||||
{
|
||||
fprintf(
|
||||
stderr, iter == 1 && r == 255
|
||||
? "partprobe utility is required to reread partition table while disk %s is in use\n"
|
||||
: "partprobe failed to re-read partition table while disk %s is in use\n",
|
||||
devinfo.path.c_str()
|
||||
);
|
||||
return {};
|
||||
}
|
||||
break;
|
||||
}
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "Failed to lstat %s: %s\n", part["node"].string_value().c_str(), strerror(errno));
|
||||
return {};
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
// Wait until device symlinks in /dev/disk/by-partuuid/ appear
|
||||
bool exists = false;
|
||||
iter = 0;
|
||||
while (!exists && iter < 300) // max 30 sec
|
||||
{
|
||||
exists = true;
|
||||
for (const auto & part: new_parts)
|
||||
{
|
||||
std::string link_path = "/dev/disk/by-partuuid/"+strtolower(part["uuid"].string_value());
|
||||
struct stat st;
|
||||
if (lstat(link_path.c_str(), &st) < 0)
|
||||
{
|
||||
if (errno == ENOENT)
|
||||
exists = false;
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "Failed to lstat %s: %s\n", link_path.c_str(), strerror(errno));
|
||||
return {};
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!exists)
|
||||
{
|
||||
struct timespec ts = { .tv_sec = 0, .tv_nsec = 100000000 }; // 100ms
|
||||
iter += (nanosleep(&ts, NULL) == 0);
|
||||
}
|
||||
}
|
||||
devinfo.pt = newpt;
|
||||
devinfo.osd_part_count += sizes.size();
|
||||
devinfo.free = free_from_parttable(newpt);
|
||||
return new_parts;
|
||||
}
|
||||
|
||||
std::vector<std::string> disk_tool_t::get_new_data_parts(vitastor_dev_info_t & dev,
|
||||
uint64_t osd_per_disk, uint64_t max_other_percent)
|
||||
{
|
||||
std::vector<std::string> use_parts;
|
||||
uint64_t want_parts = 0;
|
||||
if (dev.pt.is_null())
|
||||
{
|
||||
want_parts = osd_per_disk;
|
||||
}
|
||||
else
|
||||
{
|
||||
// Disk already has partitions. If these are empty Vitastor OSD partitions, we can use them
|
||||
uint64_t osds_exist = 0, osds_size = 0;
|
||||
for (const auto & part: dev.pt["partitions"].array_items())
|
||||
{
|
||||
if (strtolower(part["type"].string_value()) == VITASTOR_PART_TYPE)
|
||||
{
|
||||
// Check if an existing Vitastor partition is empty
|
||||
json11::Json sb = read_osd_superblock(part["node"].string_value(), false);
|
||||
if (sb.is_null())
|
||||
{
|
||||
// Use this partition
|
||||
use_parts.push_back(part["uuid"].string_value());
|
||||
osds_exist++;
|
||||
}
|
||||
else
|
||||
{
|
||||
std::string part_path = "/dev/disk/by-partuuid/"+strtolower(part["uuid"].string_value());
|
||||
bool is_meta = sb["params"]["meta_device"].string_value() == part_path;
|
||||
bool is_journal = sb["params"]["journal_device"].string_value() == part_path;
|
||||
bool is_data = sb["params"]["data_device"].string_value() == part_path;
|
||||
fprintf(
|
||||
stderr, "%s is already initialized for OSD %ju%s, skipping\n",
|
||||
part["node"].string_value().c_str(), sb["params"]["osd_num"].uint64_value(),
|
||||
(is_data ? " data" : (is_meta ? " meta" : (is_journal ? " journal" : "")))
|
||||
);
|
||||
if (is_data || sb["params"]["data_device"].string_value().substr(0, 22) != "/dev/disk/by-partuuid/")
|
||||
{
|
||||
osds_size += part["size"].uint64_value()*dev.pt["sectorsize"].uint64_value();
|
||||
osds_exist++;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Still create OSD(s) if a disk has no more than (max_other_percent) other data
|
||||
if (osds_exist >= osd_per_disk || (dev.free+osds_size) < dev.size*(100-max_other_percent)/100)
|
||||
fprintf(stderr, "%s is already partitioned, skipping\n", dev.path.c_str());
|
||||
else
|
||||
want_parts = osd_per_disk-osds_exist;
|
||||
}
|
||||
if (want_parts > 0)
|
||||
{
|
||||
// Disk is not partitioned yet - create OSD partition(s)
|
||||
std::vector<std::string> sizes;
|
||||
auto each_size = std::to_string((dev.free - 1048576) / 1048576 / want_parts)+"MiB";
|
||||
for (uint64_t i = 0; i < want_parts-1; i++)
|
||||
sizes.push_back(each_size);
|
||||
sizes.push_back("+");
|
||||
auto new_parts = add_partitions(dev, sizes);
|
||||
for (const auto & part: new_parts.array_items())
|
||||
use_parts.push_back(part["uuid"].string_value());
|
||||
}
|
||||
return use_parts;
|
||||
}
|
||||
|
||||
int disk_tool_t::get_meta_partition(std::vector<vitastor_dev_info_t> & ssds, std::map<std::string, std::string> & options)
|
||||
{
|
||||
uint64_t journal_size = parse_size(options["journal_size"]);
|
||||
journal_size = ((journal_size+1024*1024-1)/1024/1024)*1024*1024;
|
||||
// Calculate metadata size
|
||||
uint64_t meta_size = 0;
|
||||
try
|
||||
{
|
||||
blockstore_disk_t dsk;
|
||||
dsk.parse_config(options);
|
||||
dsk.data_io = dsk.meta_io = dsk.journal_io = "direct";
|
||||
dsk.open_data();
|
||||
dsk.open_meta();
|
||||
dsk.open_journal();
|
||||
dsk.calc_lengths(true);
|
||||
dsk.close_all();
|
||||
meta_size = dsk.meta_len;
|
||||
}
|
||||
catch (std::exception & e)
|
||||
{
|
||||
fprintf(stderr, "%s\n", e.what());
|
||||
return 1;
|
||||
}
|
||||
// Leave some extra space for future metadata formats and round metadata area size to multiples of 1 MB
|
||||
uint64_t meta_reserve_multiple = 2, min_meta_size = (uint64_t)1024*1024*1024;
|
||||
if (options.find("meta_reserve") != options.end())
|
||||
{
|
||||
int p1 = options["meta_reserve"].find("x"), p2 = options["meta_reserve"].find(",");
|
||||
if (p1 >= 0 && p2 >= 0)
|
||||
{
|
||||
meta_reserve_multiple = stoull_full(options["meta_reserve"].substr(p1 < p2 ? 0 : p2, p1 - (p1 < p2 ? 0 : p2)));
|
||||
min_meta_size = parse_size(options["meta_reserve"].substr(p1 < p2 ? p2 : 0, p1 < p2 ? options["meta_reserve"].size()-p2 : p2));
|
||||
}
|
||||
else if (p1 >= 0)
|
||||
meta_reserve_multiple = stoull_full(options["meta_reserve"].substr(0, p1));
|
||||
else
|
||||
min_meta_size = parse_size(options["meta_reserve"]);
|
||||
}
|
||||
meta_size = ((meta_size+1024*1024-1)/1024/1024)*1024*1024;
|
||||
meta_size *= meta_reserve_multiple;
|
||||
if (meta_size < min_meta_size)
|
||||
meta_size = min_meta_size;
|
||||
// Pick an SSD for journal&meta, balancing the number of serviced OSDs across SSDs
|
||||
int sel = -1;
|
||||
for (int i = 0; i < ssds.size(); i++)
|
||||
if (ssds[i].free >= (meta_size+journal_size+4096*2) && (sel == -1 || ssds[sel].osd_part_count > ssds[i].osd_part_count))
|
||||
sel = i;
|
||||
if (sel < 0)
|
||||
{
|
||||
fprintf(
|
||||
stderr, "Could not find free space for new SSD journal and metadata (need %ju + %ju MiB)\n",
|
||||
meta_size/1024/1024, journal_size/1024/1024
|
||||
);
|
||||
return 1;
|
||||
}
|
||||
// Create partitions
|
||||
auto new_parts = add_partitions(ssds[sel], {
|
||||
std::to_string(journal_size/1024/1024)+"MiB",
|
||||
std::to_string(meta_size/1024/1024)+"MiB"
|
||||
});
|
||||
if (new_parts.is_null())
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
ssds[sel].osd_part_count += 2;
|
||||
options["journal_device"] = "/dev/disk/by-partuuid/"+strtolower(new_parts[0]["uuid"].string_value());
|
||||
options["meta_device"] = "/dev/disk/by-partuuid/"+strtolower(new_parts[1]["uuid"].string_value());
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::prepare(std::vector<std::string> devices)
|
||||
{
|
||||
if (options.find("data_device") != options.end() && options["data_device"] != "")
|
||||
{
|
||||
if (options.find("hybrid") != options.end() || options.find("osd_per_disk") != options.end() || devices.size())
|
||||
{
|
||||
fprintf(stderr, "Device list (positional arguments) and --hybrid are incompatible with --data_device\n");
|
||||
return 1;
|
||||
}
|
||||
return prepare_one(options, options.find("hdd") != options.end() ? 1 : 0);
|
||||
}
|
||||
if (!devices.size())
|
||||
{
|
||||
fprintf(stderr, "Device list missing\n");
|
||||
return 1;
|
||||
}
|
||||
options.erase("data_device");
|
||||
options.erase("meta_device");
|
||||
options.erase("journal_device");
|
||||
bool hybrid = options.find("hybrid") != options.end();
|
||||
auto devinfo = collect_devices(devices);
|
||||
if (!devinfo.size())
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
uint64_t osd_per_disk = stoull_full(options["osd_per_disk"]);
|
||||
if (!osd_per_disk)
|
||||
osd_per_disk = 1;
|
||||
uint64_t max_other_percent = 10;
|
||||
if (options.find("max_other") != options.end())
|
||||
{
|
||||
max_other_percent = stoull_full(trim(options["max_other"], " \n\r\t%"));
|
||||
if (max_other_percent > 100)
|
||||
max_other_percent = 100;
|
||||
}
|
||||
std::vector<vitastor_dev_info_t> ssds;
|
||||
if (options.find("disable_data_fsync") == options.end())
|
||||
options["disable_data_fsync"] = "auto";
|
||||
if (hybrid)
|
||||
{
|
||||
if (options.find("disable_meta_fsync") == options.end())
|
||||
options["disable_meta_fsync"] = "auto";
|
||||
options["disable_journal_fsync"] = options["disable_meta_fsync"];
|
||||
for (auto & dev: devinfo)
|
||||
if (!dev.is_hdd)
|
||||
ssds.push_back(dev);
|
||||
if (!ssds.size())
|
||||
{
|
||||
fprintf(stderr, "No SSDs found\n");
|
||||
return 1;
|
||||
}
|
||||
else if (ssds.size() == devinfo.size())
|
||||
{
|
||||
fprintf(stderr, "No HDDs found\n");
|
||||
return 1;
|
||||
}
|
||||
if (options["journal_size"] == "")
|
||||
options["journal_size"] = DEFAULT_HYBRID_JOURNAL;
|
||||
}
|
||||
else
|
||||
{
|
||||
options.erase("disable_meta_fsync");
|
||||
options.erase("disable_journal_fsync");
|
||||
}
|
||||
auto journal_size = options["journal_size"];
|
||||
for (auto & dev: devinfo)
|
||||
{
|
||||
if (!hybrid || dev.is_hdd)
|
||||
{
|
||||
// Select new partitions and create an OSD on each of them
|
||||
for (const auto & uuid: get_new_data_parts(dev, osd_per_disk, max_other_percent))
|
||||
{
|
||||
options["force"] = true;
|
||||
options["data_device"] = "/dev/disk/by-partuuid/"+strtolower(uuid);
|
||||
if (hybrid)
|
||||
{
|
||||
// Select/create journal and metadata partitions
|
||||
int r = get_meta_partition(ssds, options);
|
||||
if (r != 0)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
options.erase("journal_size");
|
||||
}
|
||||
// Treat all disks as SSDs if not in the hybrid mode
|
||||
prepare_one(options, dev.is_hdd ? 1 : 0);
|
||||
if (hybrid)
|
||||
{
|
||||
options["journal_size"] = journal_size;
|
||||
options.erase("journal_device");
|
||||
options.erase("meta_device");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,524 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#include "disk_tool.h"
|
||||
#include "rw_blocking.h"
|
||||
#include "str_util.h"
|
||||
|
||||
#define DM_ST_EMPTY 0
|
||||
#define DM_ST_TO_READ 1
|
||||
#define DM_ST_READING 2
|
||||
#define DM_ST_TO_WRITE 3
|
||||
#define DM_ST_WRITING 4
|
||||
|
||||
struct resizer_data_moving_t
|
||||
{
|
||||
int state = 0;
|
||||
void *buf = NULL;
|
||||
uint64_t old_loc, new_loc;
|
||||
};
|
||||
|
||||
int disk_tool_t::resize_data()
|
||||
{
|
||||
int r;
|
||||
// Parse parameters
|
||||
r = resize_parse_params();
|
||||
if (r != 0)
|
||||
return r;
|
||||
// Check parameters and fill allocator
|
||||
fprintf(stderr, "Reading metadata\n");
|
||||
data_alloc = new allocator((new_data_len < dsk.data_len ? dsk.data_len : new_data_len) / dsk.data_block_size);
|
||||
r = process_meta(
|
||||
[this](blockstore_meta_header_v2_t *hdr)
|
||||
{
|
||||
resize_init(hdr);
|
||||
},
|
||||
[this](uint64_t block_num, clean_disk_entry *entry, uint8_t *bitmap)
|
||||
{
|
||||
data_alloc->set(block_num, true);
|
||||
}
|
||||
);
|
||||
if (r != 0)
|
||||
return r;
|
||||
fprintf(stderr, "Reading journal\n");
|
||||
r = process_journal([this](void *buf)
|
||||
{
|
||||
return process_journal_block(buf, [this](int num, journal_entry *je)
|
||||
{
|
||||
if (je->type == JE_BIG_WRITE || je->type == JE_BIG_WRITE_INSTANT)
|
||||
{
|
||||
data_alloc->set(je->big_write.location / dsk.data_block_size, true);
|
||||
}
|
||||
});
|
||||
});
|
||||
if (r != 0)
|
||||
return r;
|
||||
// Remap blocks
|
||||
r = resize_remap_blocks();
|
||||
if (r != 0)
|
||||
return r;
|
||||
// Copy data blocks into new places
|
||||
fprintf(stderr, "Moving data blocks\n");
|
||||
r = resize_copy_data();
|
||||
if (r != 0)
|
||||
return r;
|
||||
// Rewrite journal
|
||||
fprintf(stderr, "Rebuilding journal\n");
|
||||
r = resize_rewrite_journal();
|
||||
if (r != 0)
|
||||
return r;
|
||||
// Rewrite metadata
|
||||
fprintf(stderr, "Rebuilding metadata\n");
|
||||
r = resize_rewrite_meta();
|
||||
if (r != 0)
|
||||
return r;
|
||||
// Write new journal
|
||||
fprintf(stderr, "Writing new journal\n");
|
||||
r = resize_write_new_journal();
|
||||
if (r != 0)
|
||||
return r;
|
||||
// Write new metadata
|
||||
fprintf(stderr, "Writing new metadata\n");
|
||||
r = resize_write_new_meta();
|
||||
if (r != 0)
|
||||
return r;
|
||||
fprintf(stderr, "Done\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::resize_parse_params()
|
||||
{
|
||||
try
|
||||
{
|
||||
dsk.parse_config(options);
|
||||
dsk.data_io = dsk.meta_io = dsk.journal_io = "direct";
|
||||
dsk.open_data();
|
||||
dsk.open_meta();
|
||||
dsk.open_journal();
|
||||
dsk.calc_lengths();
|
||||
dsk.close_all();
|
||||
}
|
||||
catch (std::exception & e)
|
||||
{
|
||||
dsk.close_all();
|
||||
fprintf(stderr, "Error: %s\n", e.what());
|
||||
return 1;
|
||||
}
|
||||
iodepth = strtoull(options["iodepth"].c_str(), NULL, 10);
|
||||
if (!iodepth)
|
||||
iodepth = 32;
|
||||
new_meta_device = options.find("new_meta_device") != options.end()
|
||||
? options["new_meta_device"] : dsk.meta_device;
|
||||
new_journal_device = options.find("new_journal_device") != options.end()
|
||||
? options["new_journal_device"] : dsk.journal_device;
|
||||
new_data_offset = options.find("new_data_offset") != options.end()
|
||||
? parse_size(options["new_data_offset"]) : dsk.data_offset;
|
||||
new_data_len = options.find("new_data_len") != options.end()
|
||||
? parse_size(options["new_data_len"]) : dsk.data_len;
|
||||
new_meta_offset = options.find("new_meta_offset") != options.end()
|
||||
? parse_size(options["new_meta_offset"]) : dsk.meta_offset;
|
||||
new_meta_len = options.find("new_meta_len") != options.end()
|
||||
? parse_size(options["new_meta_len"]) : 0; // will be calculated in resize_init()
|
||||
new_journal_offset = options.find("new_journal_offset") != options.end()
|
||||
? parse_size(options["new_journal_offset"]) : dsk.journal_offset;
|
||||
new_journal_len = options.find("new_journal_len") != options.end()
|
||||
? parse_size(options["new_journal_len"]) : dsk.journal_len;
|
||||
if (new_meta_device == dsk.meta_device &&
|
||||
new_journal_device == dsk.journal_device &&
|
||||
new_data_offset == dsk.data_offset &&
|
||||
new_data_len == dsk.data_len &&
|
||||
new_meta_offset == dsk.meta_offset &&
|
||||
(new_meta_len == dsk.meta_len || new_meta_len == 0) &&
|
||||
new_journal_offset == dsk.journal_offset &&
|
||||
new_journal_len == dsk.journal_len &&
|
||||
options.find("force") == options.end())
|
||||
{
|
||||
// No difference
|
||||
fprintf(stderr, "No difference, specify --force to rewrite journal and meta anyway\n");
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
void disk_tool_t::resize_init(blockstore_meta_header_v2_t *hdr)
|
||||
{
|
||||
if (hdr && dsk.data_block_size != hdr->data_block_size)
|
||||
{
|
||||
if (dsk.data_block_size)
|
||||
{
|
||||
fprintf(stderr, "Using data block size of %u bytes from metadata superblock\n", hdr->data_block_size);
|
||||
}
|
||||
dsk.data_block_size = hdr->data_block_size;
|
||||
}
|
||||
if (hdr && (dsk.data_csum_type != hdr->data_csum_type || dsk.csum_block_size != hdr->csum_block_size))
|
||||
{
|
||||
if (dsk.data_csum_type)
|
||||
{
|
||||
fprintf(stderr, "Using data checksum type %s from metadata superblock\n", csum_type_str(hdr->data_csum_type).c_str());
|
||||
}
|
||||
dsk.data_csum_type = hdr->data_csum_type;
|
||||
dsk.csum_block_size = hdr->csum_block_size;
|
||||
}
|
||||
if (((new_data_len-dsk.data_len) % dsk.data_block_size) ||
|
||||
((new_data_offset-dsk.data_offset) % dsk.data_block_size))
|
||||
{
|
||||
fprintf(stderr, "Data alignment mismatch\n");
|
||||
exit(1);
|
||||
}
|
||||
data_idx_diff = ((int64_t)(dsk.data_offset-new_data_offset)) / dsk.data_block_size;
|
||||
free_first = new_data_offset > dsk.data_offset ? (new_data_offset-dsk.data_offset) / dsk.data_block_size : 0;
|
||||
free_last = (new_data_offset+new_data_len < dsk.data_offset+dsk.data_len)
|
||||
? (dsk.data_offset+dsk.data_len-new_data_offset-new_data_len) / dsk.data_block_size
|
||||
: 0;
|
||||
uint32_t new_clean_entry_header_size = sizeof(clean_disk_entry) + 4 /*entry_csum*/;
|
||||
new_clean_entry_bitmap_size = dsk.data_block_size / (hdr ? hdr->bitmap_granularity : 4096) / 8;
|
||||
new_data_csum_size = (dsk.data_csum_type
|
||||
? ((dsk.data_block_size+dsk.csum_block_size-1)/dsk.csum_block_size*(dsk.data_csum_type & 0xFF))
|
||||
: 0);
|
||||
new_clean_entry_size = new_clean_entry_header_size + 2*new_clean_entry_bitmap_size + new_data_csum_size;
|
||||
new_entries_per_block = dsk.meta_block_size/new_clean_entry_size;
|
||||
uint64_t new_meta_blocks = 1 + (new_data_len/dsk.data_block_size + new_entries_per_block-1) / new_entries_per_block;
|
||||
if (!new_meta_len)
|
||||
{
|
||||
new_meta_len = dsk.meta_block_size*new_meta_blocks;
|
||||
}
|
||||
if (new_meta_len < dsk.meta_block_size*new_meta_blocks)
|
||||
{
|
||||
fprintf(stderr, "New metadata area size is too small, should be at least %ju bytes\n", dsk.meta_block_size*new_meta_blocks);
|
||||
exit(1);
|
||||
}
|
||||
// Check that new metadata, journal and data areas don't overlap
|
||||
if (new_meta_device == dsk.data_device && new_meta_offset < new_data_offset+new_data_len &&
|
||||
new_meta_offset+new_meta_len > new_data_offset)
|
||||
{
|
||||
fprintf(stderr, "New metadata area overlaps with data\n");
|
||||
exit(1);
|
||||
}
|
||||
if (new_journal_device == dsk.data_device && new_journal_offset < new_data_offset+new_data_len &&
|
||||
new_journal_offset+new_journal_len > new_data_offset)
|
||||
{
|
||||
fprintf(stderr, "New journal area overlaps with data\n");
|
||||
exit(1);
|
||||
}
|
||||
if (new_journal_device == new_meta_device && new_journal_offset < new_meta_offset+new_meta_len &&
|
||||
new_journal_offset+new_journal_len > new_meta_offset)
|
||||
{
|
||||
fprintf(stderr, "New journal area overlaps with metadata\n");
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
int disk_tool_t::resize_remap_blocks()
|
||||
{
|
||||
total_blocks = dsk.data_len / dsk.data_block_size;
|
||||
for (uint64_t i = 0; i < free_first; i++)
|
||||
{
|
||||
if (data_alloc->get(i))
|
||||
data_remap[i] = 0;
|
||||
else
|
||||
data_alloc->set(i, true);
|
||||
}
|
||||
for (uint64_t i = 0; i < free_last; i++)
|
||||
{
|
||||
if (data_alloc->get(total_blocks-i))
|
||||
data_remap[total_blocks-i] = 0;
|
||||
else
|
||||
data_alloc->set(total_blocks-i, true);
|
||||
}
|
||||
for (auto & p: data_remap)
|
||||
{
|
||||
uint64_t new_loc = data_alloc->find_free();
|
||||
if (new_loc == UINT64_MAX)
|
||||
{
|
||||
fprintf(stderr, "Not enough space to move data\n");
|
||||
return 1;
|
||||
}
|
||||
data_alloc->set(new_loc, true);
|
||||
data_remap[p.first] = new_loc;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::resize_copy_data()
|
||||
{
|
||||
if (iodepth <= 0 || iodepth > 4096)
|
||||
{
|
||||
iodepth = 32;
|
||||
}
|
||||
ringloop = new ring_loop_t(iodepth < RINGLOOP_DEFAULT_SIZE ? RINGLOOP_DEFAULT_SIZE : iodepth);
|
||||
dsk.data_fd = open(dsk.data_device.c_str(), O_DIRECT|O_RDWR);
|
||||
if (dsk.data_fd < 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to open data device %s: %s\n", dsk.data_device.c_str(), strerror(errno));
|
||||
delete ringloop;
|
||||
ringloop = NULL;
|
||||
return 1;
|
||||
}
|
||||
moving_blocks = new resizer_data_moving_t[iodepth];
|
||||
moving_blocks[0].buf = memalign_or_die(MEM_ALIGNMENT, iodepth*dsk.data_block_size);
|
||||
for (int i = 1; i < iodepth; i++)
|
||||
{
|
||||
moving_blocks[i].buf = (uint8_t*)moving_blocks[0].buf + i*dsk.data_block_size;
|
||||
}
|
||||
remap_active = 1;
|
||||
remap_it = data_remap.begin();
|
||||
ring_consumer.loop = [this]()
|
||||
{
|
||||
remap_active = 0;
|
||||
for (int i = 0; i < iodepth; i++)
|
||||
{
|
||||
if (moving_blocks[i].state == DM_ST_EMPTY && remap_it != data_remap.end())
|
||||
{
|
||||
uint64_t old_loc = remap_it->first, new_loc = remap_it->second;
|
||||
moving_blocks[i].state = DM_ST_TO_READ;
|
||||
moving_blocks[i].old_loc = old_loc;
|
||||
moving_blocks[i].new_loc = new_loc;
|
||||
remap_it++;
|
||||
}
|
||||
if (moving_blocks[i].state == DM_ST_TO_READ)
|
||||
{
|
||||
struct io_uring_sqe *sqe = ringloop->get_sqe();
|
||||
if (sqe)
|
||||
{
|
||||
moving_blocks[i].state = DM_ST_READING;
|
||||
struct ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
||||
data->iov = (struct iovec){ moving_blocks[i].buf, dsk.data_block_size };
|
||||
my_uring_prep_readv(sqe, dsk.data_fd, &data->iov, 1, dsk.data_offset + moving_blocks[i].old_loc*dsk.data_block_size);
|
||||
data->callback = [this, i](ring_data_t *data)
|
||||
{
|
||||
if (data->res != dsk.data_block_size)
|
||||
{
|
||||
fprintf(
|
||||
stderr, "Failed to read %u bytes at %ju from %s: %s\n", dsk.data_block_size,
|
||||
dsk.data_offset + moving_blocks[i].old_loc*dsk.data_block_size, dsk.data_device.c_str(),
|
||||
data->res < 0 ? strerror(-data->res) : "short read"
|
||||
);
|
||||
exit(1);
|
||||
}
|
||||
moving_blocks[i].state = DM_ST_TO_WRITE;
|
||||
ringloop->wakeup();
|
||||
};
|
||||
}
|
||||
}
|
||||
if (moving_blocks[i].state == DM_ST_TO_WRITE)
|
||||
{
|
||||
struct io_uring_sqe *sqe = ringloop->get_sqe();
|
||||
if (sqe)
|
||||
{
|
||||
moving_blocks[i].state = DM_ST_WRITING;
|
||||
struct ring_data_t *data = ((ring_data_t*)sqe->user_data);
|
||||
data->iov = (struct iovec){ moving_blocks[i].buf, dsk.data_block_size };
|
||||
my_uring_prep_writev(sqe, dsk.data_fd, &data->iov, 1, dsk.data_offset + moving_blocks[i].new_loc*dsk.data_block_size);
|
||||
data->callback = [this, i](ring_data_t *data)
|
||||
{
|
||||
if (data->res != dsk.data_block_size)
|
||||
{
|
||||
fprintf(
|
||||
stderr, "Failed to write %u bytes at %ju to %s: %s\n", dsk.data_block_size,
|
||||
dsk.data_offset + moving_blocks[i].new_loc*dsk.data_block_size, dsk.data_device.c_str(),
|
||||
data->res < 0 ? strerror(-data->res) : "short write"
|
||||
);
|
||||
exit(1);
|
||||
}
|
||||
moving_blocks[i].state = DM_ST_EMPTY;
|
||||
ringloop->wakeup();
|
||||
};
|
||||
}
|
||||
}
|
||||
remap_active += moving_blocks[i].state != DM_ST_EMPTY ? 1 : 0;
|
||||
}
|
||||
ringloop->submit();
|
||||
};
|
||||
ringloop->register_consumer(&ring_consumer);
|
||||
while (1)
|
||||
{
|
||||
ringloop->loop();
|
||||
if (!remap_active)
|
||||
break;
|
||||
ringloop->wait();
|
||||
}
|
||||
ringloop->unregister_consumer(&ring_consumer);
|
||||
free(moving_blocks[0].buf);
|
||||
delete[] moving_blocks;
|
||||
moving_blocks = NULL;
|
||||
close(dsk.data_fd);
|
||||
dsk.data_fd = -1;
|
||||
delete ringloop;
|
||||
ringloop = NULL;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::resize_rewrite_journal()
|
||||
{
|
||||
// Simply overwriting on the fly may be impossible because old and new areas may overlap
|
||||
// For now, just build new journal data in memory
|
||||
new_journal_buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, new_journal_len);
|
||||
new_journal_ptr = new_journal_buf;
|
||||
new_journal_data = new_journal_ptr + dsk.journal_block_size;
|
||||
new_journal_in_pos = 0;
|
||||
memset(new_journal_buf, 0, new_journal_len);
|
||||
process_journal([this](void *buf)
|
||||
{
|
||||
return process_journal_block(buf, [this](int num, journal_entry *je)
|
||||
{
|
||||
if (je->type == JE_START)
|
||||
{
|
||||
if (je_start.data_csum_type != dsk.data_csum_type ||
|
||||
je_start.csum_block_size != dsk.csum_block_size)
|
||||
{
|
||||
fprintf(
|
||||
stderr, "Error: journal header has different checksum parameters: %s/%u vs %s/%u\n",
|
||||
csum_type_str(je_start.data_csum_type).c_str(), je_start.csum_block_size,
|
||||
csum_type_str(dsk.data_csum_type).c_str(), dsk.csum_block_size
|
||||
);
|
||||
exit(1);
|
||||
}
|
||||
journal_entry *ne = (journal_entry*)(new_journal_ptr + new_journal_in_pos);
|
||||
*((journal_entry_start*)ne) = (journal_entry_start){
|
||||
.magic = JOURNAL_MAGIC,
|
||||
.type = JE_START,
|
||||
.size = sizeof(journal_entry_start),
|
||||
.journal_start = dsk.journal_block_size,
|
||||
.version = JOURNAL_VERSION_V2,
|
||||
.data_csum_type = dsk.data_csum_type,
|
||||
.csum_block_size = dsk.csum_block_size,
|
||||
};
|
||||
ne->crc32 = je_crc32(ne);
|
||||
new_journal_ptr += dsk.journal_block_size;
|
||||
new_journal_data = new_journal_ptr+dsk.journal_block_size;
|
||||
new_journal_in_pos = 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
if (dsk.journal_block_size < new_journal_in_pos+je->size)
|
||||
{
|
||||
new_journal_ptr = new_journal_data;
|
||||
if (new_journal_ptr-new_journal_buf >= new_journal_len)
|
||||
{
|
||||
fprintf(stderr, "Error: live entries don't fit to the new journal\n");
|
||||
exit(1);
|
||||
}
|
||||
new_journal_data = new_journal_ptr+dsk.journal_block_size;
|
||||
new_journal_in_pos = 0;
|
||||
if (dsk.journal_block_size < je->size)
|
||||
{
|
||||
fprintf(stderr, "Error: journal entry too large (%u bytes)\n", je->size);
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
journal_entry *ne = (journal_entry*)(new_journal_ptr + new_journal_in_pos);
|
||||
memcpy(ne, je, je->size);
|
||||
ne->crc32_prev = new_crc32_prev;
|
||||
if (je->type == JE_BIG_WRITE || je->type == JE_BIG_WRITE_INSTANT)
|
||||
{
|
||||
// Change the block reference
|
||||
auto remap_it = data_remap.find(ne->big_write.location / dsk.data_block_size);
|
||||
if (remap_it != data_remap.end())
|
||||
{
|
||||
ne->big_write.location = remap_it->second * dsk.data_block_size;
|
||||
}
|
||||
ne->big_write.location += data_idx_diff * dsk.data_block_size;
|
||||
}
|
||||
else if (je->type == JE_SMALL_WRITE || je->type == JE_SMALL_WRITE_INSTANT)
|
||||
{
|
||||
ne->small_write.data_offset = new_journal_data-new_journal_buf;
|
||||
if (ne->small_write.data_offset + ne->small_write.len > new_journal_len)
|
||||
{
|
||||
fprintf(stderr, "Error: live entries don't fit to the new journal\n");
|
||||
exit(1);
|
||||
}
|
||||
memcpy(new_journal_data, small_write_data, ne->small_write.len);
|
||||
new_journal_data += ne->small_write.len;
|
||||
}
|
||||
ne->crc32 = je_crc32(ne);
|
||||
new_journal_in_pos += ne->size;
|
||||
new_crc32_prev = ne->crc32;
|
||||
}
|
||||
});
|
||||
});
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::resize_write_new_journal()
|
||||
{
|
||||
new_journal_fd = open(new_journal_device.c_str(), O_DIRECT|O_RDWR);
|
||||
if (new_journal_fd < 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to open new journal device %s: %s\n", new_journal_device.c_str(), strerror(errno));
|
||||
return 1;
|
||||
}
|
||||
lseek64(new_journal_fd, new_journal_offset, 0);
|
||||
write_blocking(new_journal_fd, new_journal_buf, new_journal_len);
|
||||
fsync(new_journal_fd);
|
||||
close(new_journal_fd);
|
||||
new_journal_fd = -1;
|
||||
free(new_journal_buf);
|
||||
new_journal_buf = NULL;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::resize_rewrite_meta()
|
||||
{
|
||||
new_meta_buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, new_meta_len);
|
||||
memset(new_meta_buf, 0, new_meta_len);
|
||||
int r = process_meta(
|
||||
[this](blockstore_meta_header_v2_t *hdr)
|
||||
{
|
||||
blockstore_meta_header_v2_t *new_hdr = (blockstore_meta_header_v2_t *)new_meta_buf;
|
||||
new_hdr->zero = 0;
|
||||
new_hdr->magic = BLOCKSTORE_META_MAGIC_V1;
|
||||
new_hdr->version = BLOCKSTORE_META_FORMAT_V1;
|
||||
new_hdr->meta_block_size = dsk.meta_block_size;
|
||||
new_hdr->data_block_size = dsk.data_block_size;
|
||||
new_hdr->bitmap_granularity = dsk.bitmap_granularity ? dsk.bitmap_granularity : 4096;
|
||||
new_hdr->data_csum_type = dsk.data_csum_type;
|
||||
new_hdr->csum_block_size = dsk.csum_block_size;
|
||||
},
|
||||
[this](uint64_t block_num, clean_disk_entry *entry, uint8_t *bitmap)
|
||||
{
|
||||
auto remap_it = data_remap.find(block_num);
|
||||
if (remap_it != data_remap.end())
|
||||
block_num = remap_it->second;
|
||||
if (block_num < free_first || block_num >= total_blocks-free_last)
|
||||
{
|
||||
fprintf(stderr, "BUG: remapped block not in range\n");
|
||||
exit(1);
|
||||
}
|
||||
block_num += data_idx_diff;
|
||||
clean_disk_entry *new_entry = (clean_disk_entry*)(new_meta_buf + dsk.meta_block_size +
|
||||
dsk.meta_block_size*(block_num / new_entries_per_block) +
|
||||
new_clean_entry_size*(block_num % new_entries_per_block));
|
||||
new_entry->oid = entry->oid;
|
||||
new_entry->version = entry->version;
|
||||
if (bitmap)
|
||||
memcpy(new_entry->bitmap, bitmap, 2*new_clean_entry_bitmap_size + new_data_csum_size);
|
||||
else
|
||||
memset(new_entry->bitmap, 0xff, 2*new_clean_entry_bitmap_size);
|
||||
}
|
||||
);
|
||||
if (r != 0)
|
||||
{
|
||||
free(new_meta_buf);
|
||||
new_meta_buf = NULL;
|
||||
return r;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::resize_write_new_meta()
|
||||
{
|
||||
new_meta_fd = open(new_meta_device.c_str(), O_DIRECT|O_RDWR);
|
||||
if (new_meta_fd < 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to open new metadata device %s: %s\n", new_meta_device.c_str(), strerror(errno));
|
||||
return 1;
|
||||
}
|
||||
lseek64(new_meta_fd, new_meta_offset, 0);
|
||||
write_blocking(new_meta_fd, new_meta_buf, new_meta_len);
|
||||
fsync(new_meta_fd);
|
||||
close(new_meta_fd);
|
||||
new_meta_fd = -1;
|
||||
free(new_meta_buf);
|
||||
new_meta_buf = NULL;
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,519 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#include <sys/file.h>
|
||||
|
||||
#include "disk_tool.h"
|
||||
#include "rw_blocking.h"
|
||||
#include "str_util.h"
|
||||
|
||||
struct __attribute__((__packed__)) vitastor_disk_superblock_t
|
||||
{
|
||||
uint64_t magic;
|
||||
uint32_t crc32c;
|
||||
uint32_t size;
|
||||
uint8_t json_data[];
|
||||
};
|
||||
|
||||
static std::string udev_escape(std::string str)
|
||||
{
|
||||
std::string r;
|
||||
int p = str.find_first_of("\"\' \t\r\n"), prev = 0;
|
||||
if (p == std::string::npos)
|
||||
{
|
||||
return str;
|
||||
}
|
||||
while (p != std::string::npos)
|
||||
{
|
||||
r += str.substr(prev, p-prev);
|
||||
r += "\\";
|
||||
prev = p;
|
||||
p = str.find_first_of("\"\' \t\r\n", p+1);
|
||||
}
|
||||
r += str.substr(prev);
|
||||
return r;
|
||||
}
|
||||
|
||||
int disk_tool_t::udev_import(std::string device)
|
||||
{
|
||||
json11::Json sb = read_osd_superblock(device);
|
||||
if (sb.is_null())
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
uint64_t osd_num = sb["params"]["osd_num"].uint64_value();
|
||||
// Print variables for udev
|
||||
printf("VITASTOR_OSD_NUM=%ju\n", osd_num);
|
||||
printf("VITASTOR_ALIAS=osd%ju-%s\n", osd_num, sb["device_type"].string_value().c_str());
|
||||
printf("VITASTOR_DATA_DEVICE=%s\n", udev_escape(sb["params"]["data_device"].string_value()).c_str());
|
||||
if (sb["real_meta_device"].string_value() != "" && sb["real_meta_device"] != sb["real_data_device"])
|
||||
printf("VITASTOR_META_DEVICE=%s\n", udev_escape(sb["params"]["meta_device"].string_value()).c_str());
|
||||
if (sb["real_journal_device"].string_value() != "" && sb["real_journal_device"] != sb["real_meta_device"])
|
||||
printf("VITASTOR_JOURNAL_DEVICE=%s\n", udev_escape(sb["params"]["journal_device"].string_value()).c_str());
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::read_sb(std::string device)
|
||||
{
|
||||
json11::Json sb = read_osd_superblock(device, true, options.find("force") != options.end());
|
||||
if (sb.is_null())
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
printf("%s\n", sb["params"].dump().c_str());
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::write_sb(std::string device)
|
||||
{
|
||||
std::string input;
|
||||
int r;
|
||||
char buf[4096];
|
||||
while (1)
|
||||
{
|
||||
r = read(0, buf, sizeof(buf));
|
||||
if (r <= 0 && errno != EAGAIN)
|
||||
break;
|
||||
input += std::string(buf, r);
|
||||
}
|
||||
std::string json_err;
|
||||
json11::Json params = json11::Json::parse(input, json_err);
|
||||
if (json_err != "" || !params["osd_num"].uint64_value() || params["data_device"].string_value() == "")
|
||||
{
|
||||
fprintf(stderr, "Invalid JSON input\n");
|
||||
return 1;
|
||||
}
|
||||
return !write_osd_superblock(device, params);
|
||||
}
|
||||
|
||||
int disk_tool_t::update_sb(std::string device)
|
||||
{
|
||||
json11::Json sb = read_osd_superblock(device, true, options.find("force") != options.end());
|
||||
if (sb.is_null())
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
auto sb_obj = sb["params"].object_items();
|
||||
for (auto & kv: options)
|
||||
{
|
||||
if (kv.first != "force")
|
||||
{
|
||||
sb_obj[kv.first] = kv.second;
|
||||
}
|
||||
}
|
||||
return !write_osd_superblock(device, sb_obj);
|
||||
}
|
||||
|
||||
uint32_t disk_tool_t::write_osd_superblock(std::string device, json11::Json params)
|
||||
{
|
||||
std::string json_data = params.dump();
|
||||
uint32_t sb_size = sizeof(vitastor_disk_superblock_t)+json_data.size();
|
||||
if (sb_size > VITASTOR_DISK_MAX_SB_SIZE)
|
||||
{
|
||||
fprintf(stderr, "JSON data for superblock is too large\n");
|
||||
return 0;
|
||||
}
|
||||
uint64_t buf_len = ((sb_size+4095)/4096) * 4096;
|
||||
uint8_t *buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, buf_len);
|
||||
memset(buf, 0, buf_len);
|
||||
vitastor_disk_superblock_t *sb = (vitastor_disk_superblock_t*)buf;
|
||||
sb->magic = VITASTOR_DISK_MAGIC;
|
||||
sb->size = sb_size;
|
||||
memcpy(sb->json_data, json_data.c_str(), json_data.size());
|
||||
sb->crc32c = crc32c(0, &sb->size, sb->size - ((uint8_t*)&sb->size - buf));
|
||||
int fd = open(device.c_str(), O_DIRECT|O_RDWR);
|
||||
if (fd < 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to open device %s: %s\n", device.c_str(), strerror(errno));
|
||||
free(buf);
|
||||
return 0;
|
||||
}
|
||||
int r = write_blocking(fd, buf, buf_len);
|
||||
if (r < 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to write to %s: %s\n", device.c_str(), strerror(errno));
|
||||
close(fd);
|
||||
free(buf);
|
||||
return 0;
|
||||
}
|
||||
close(fd);
|
||||
free(buf);
|
||||
shell_exec({ "udevadm", "trigger", "--settle", device }, "", NULL, NULL);
|
||||
return sb_size;
|
||||
}
|
||||
|
||||
json11::Json disk_tool_t::read_osd_superblock(std::string device, bool expect_exist, bool ignore_errors)
|
||||
{
|
||||
vitastor_disk_superblock_t *sb = NULL;
|
||||
uint8_t *buf = NULL;
|
||||
json11::Json osd_params;
|
||||
std::string json_err;
|
||||
std::string real_device, device_type, real_data, real_meta, real_journal;
|
||||
int r, fd = open(device.c_str(), O_DIRECT|O_RDWR);
|
||||
if (fd < 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to open device %s: %s\n", device.c_str(), strerror(errno));
|
||||
return osd_params;
|
||||
}
|
||||
buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, 4096);
|
||||
r = read_blocking(fd, buf, 4096);
|
||||
if (r != 4096)
|
||||
{
|
||||
fprintf(stderr, "Failed to read OSD superblock from %s: %s\n", device.c_str(), strerror(errno));
|
||||
goto ex;
|
||||
}
|
||||
sb = (vitastor_disk_superblock_t*)buf;
|
||||
if (sb->magic != VITASTOR_DISK_MAGIC && !ignore_errors)
|
||||
{
|
||||
if (expect_exist)
|
||||
fprintf(stderr, "Invalid OSD superblock on %s: magic number mismatch\n", device.c_str());
|
||||
goto ex;
|
||||
}
|
||||
if (sb->size > VITASTOR_DISK_MAX_SB_SIZE ||
|
||||
// +2 is minimal json: {}
|
||||
sb->size < sizeof(vitastor_disk_superblock_t)+2)
|
||||
{
|
||||
if (expect_exist)
|
||||
fprintf(stderr, "Invalid OSD superblock on %s: invalid size\n", device.c_str());
|
||||
goto ex;
|
||||
}
|
||||
if (sb->size > 4096)
|
||||
{
|
||||
uint64_t sb_size = ((sb->size+4095)/4096)*4096;
|
||||
free(buf);
|
||||
buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, sb_size);
|
||||
lseek64(fd, 0, 0);
|
||||
r = read_blocking(fd, buf, sb_size);
|
||||
if (r != sb_size)
|
||||
{
|
||||
fprintf(stderr, "Failed to read OSD superblock from %s: %s\n", device.c_str(), strerror(errno));
|
||||
goto ex;
|
||||
}
|
||||
sb = (vitastor_disk_superblock_t*)buf;
|
||||
}
|
||||
if (sb->crc32c != crc32c(0, &sb->size, sb->size - ((uint8_t*)&sb->size - buf)) && !ignore_errors)
|
||||
{
|
||||
if (expect_exist)
|
||||
fprintf(stderr, "Invalid OSD superblock on %s: crc32 mismatch\n", device.c_str());
|
||||
goto ex;
|
||||
}
|
||||
osd_params = json11::Json::parse(std::string((char*)sb->json_data, sb->size - sizeof(vitastor_disk_superblock_t)), json_err);
|
||||
if (json_err != "")
|
||||
{
|
||||
if (expect_exist)
|
||||
fprintf(stderr, "Invalid OSD superblock on %s: invalid JSON\n", device.c_str());
|
||||
goto ex;
|
||||
}
|
||||
// Validate superblock
|
||||
if (!osd_params["osd_num"].uint64_value() && !ignore_errors)
|
||||
{
|
||||
if (expect_exist)
|
||||
fprintf(stderr, "OSD superblock on %s lacks osd_num\n", device.c_str());
|
||||
osd_params = json11::Json();
|
||||
goto ex;
|
||||
}
|
||||
if (osd_params["data_device"].string_value() == "" && !ignore_errors)
|
||||
{
|
||||
if (expect_exist)
|
||||
fprintf(stderr, "OSD superblock on %s lacks data_device\n", device.c_str());
|
||||
osd_params = json11::Json();
|
||||
goto ex;
|
||||
}
|
||||
real_device = realpath_str(device);
|
||||
real_data = realpath_str(osd_params["data_device"].string_value());
|
||||
real_meta = osd_params["meta_device"].string_value() != "" && osd_params["meta_device"] != osd_params["data_device"]
|
||||
? realpath_str(osd_params["meta_device"].string_value()) : "";
|
||||
real_journal = osd_params["journal_device"].string_value() != "" && osd_params["journal_device"] != osd_params["meta_device"]
|
||||
? realpath_str(osd_params["journal_device"].string_value()) : "";
|
||||
if (real_journal == real_meta)
|
||||
{
|
||||
real_journal = "";
|
||||
}
|
||||
if (real_meta == real_data)
|
||||
{
|
||||
real_meta = "";
|
||||
}
|
||||
if (real_device == real_data)
|
||||
{
|
||||
device_type = "data";
|
||||
}
|
||||
else if (real_device == real_meta)
|
||||
{
|
||||
device_type = "meta";
|
||||
}
|
||||
else if (real_device == real_journal)
|
||||
{
|
||||
device_type = "journal";
|
||||
}
|
||||
else if (!ignore_errors)
|
||||
{
|
||||
if (expect_exist)
|
||||
fprintf(stderr, "Invalid OSD superblock on %s: does not refer to the device itself\n", device.c_str());
|
||||
osd_params = json11::Json();
|
||||
goto ex;
|
||||
}
|
||||
osd_params = json11::Json::object{
|
||||
{ "params", osd_params },
|
||||
{ "device_type", device_type },
|
||||
{ "real_data_device", real_data },
|
||||
{ "real_meta_device", real_meta },
|
||||
{ "real_journal_device", real_journal },
|
||||
};
|
||||
ex:
|
||||
free(buf);
|
||||
close(fd);
|
||||
return osd_params;
|
||||
}
|
||||
|
||||
int disk_tool_t::systemd_start_stop_osds(const std::vector<std::string> & cmd, const std::vector<std::string> & devices)
|
||||
{
|
||||
if (!devices.size())
|
||||
{
|
||||
fprintf(stderr, "Device path is missing\n");
|
||||
return 1;
|
||||
}
|
||||
std::vector<std::string> svcs;
|
||||
for (auto & device: devices)
|
||||
{
|
||||
json11::Json sb = read_osd_superblock(device);
|
||||
if (!sb.is_null())
|
||||
{
|
||||
svcs.push_back("vitastor-osd@"+sb["params"]["osd_num"].as_string());
|
||||
}
|
||||
}
|
||||
if (!svcs.size())
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
std::vector<char*> argv;
|
||||
argv.push_back((char*)"systemctl");
|
||||
for (auto & s: cmd)
|
||||
{
|
||||
argv.push_back((char*)s.c_str());
|
||||
}
|
||||
for (auto & s: svcs)
|
||||
{
|
||||
argv.push_back((char*)s.c_str());
|
||||
}
|
||||
argv.push_back(NULL);
|
||||
execvpe("systemctl", argv.data(), environ);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::exec_osd(std::string device)
|
||||
{
|
||||
json11::Json sb = read_osd_superblock(device);
|
||||
if (sb.is_null())
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
std::string osd_binary = "vitastor-osd";
|
||||
if (options["osd-binary"] != "")
|
||||
{
|
||||
osd_binary = options["osd-binary"];
|
||||
}
|
||||
std::vector<std::string> argstr;
|
||||
argstr.push_back(osd_binary.c_str());
|
||||
for (auto & kv: sb["params"].object_items())
|
||||
{
|
||||
argstr.push_back("--"+kv.first);
|
||||
argstr.push_back(kv.second.is_string() ? kv.second.string_value() : kv.second.dump());
|
||||
}
|
||||
char *argv[argstr.size()+1];
|
||||
for (int i = 0; i < argstr.size(); i++)
|
||||
{
|
||||
argv[i] = (char*)argstr[i].c_str();
|
||||
}
|
||||
argv[argstr.size()] = NULL;
|
||||
return execvpe(osd_binary.c_str(), argv, environ);
|
||||
}
|
||||
|
||||
static int check_disabled_cache(std::string dev)
|
||||
{
|
||||
int r = disable_cache(dev);
|
||||
if (r == 1)
|
||||
{
|
||||
fprintf(
|
||||
stderr, "Warning: fsync is disabled for %s, but cache status check failed."
|
||||
" Ensure that cache is in write-through mode yourself or you may lose data.\n", dev.c_str()
|
||||
);
|
||||
}
|
||||
else if (r == -1)
|
||||
{
|
||||
fprintf(
|
||||
stderr, "Error: fsync is disabled for %s, but its cache is in write-back mode"
|
||||
" and we failed to make it write-through. Data loss is presumably possible."
|
||||
" Either switch the cache to write-through mode yourself or disable the check"
|
||||
" using skip_cache_check=1 in the superblock.\n", dev.c_str()
|
||||
);
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::pre_exec_osd(std::string device)
|
||||
{
|
||||
json11::Json sb = read_osd_superblock(device);
|
||||
if (sb.is_null())
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
if (!sb["params"]["skip_cache_check"].uint64_value())
|
||||
{
|
||||
if (json_is_true(sb["params"]["disable_data_fsync"]) &&
|
||||
check_disabled_cache(sb["real_data_device"].string_value()) != 0)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
if (json_is_true(sb["params"]["disable_meta_fsync"]) &&
|
||||
sb["real_meta_device"].string_value() != "" && sb["real_meta_device"] != sb["real_data_device"] &&
|
||||
check_disabled_cache(sb["real_meta_device"].string_value()) != 0)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
if (json_is_true(sb["params"]["disable_journal_fsync"]) &&
|
||||
sb["real_journal_device"].string_value() != "" && sb["real_journal_device"] != sb["real_meta_device"] &&
|
||||
check_disabled_cache(sb["real_journal_device"].string_value()) != 0)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int disk_tool_t::purge_devices(const std::vector<std::string> & devices)
|
||||
{
|
||||
std::vector<uint64_t> osd_numbers;
|
||||
json11::Json::array superblocks;
|
||||
for (auto & device: devices)
|
||||
{
|
||||
json11::Json sb = read_osd_superblock(device);
|
||||
if (!sb.is_null())
|
||||
{
|
||||
uint64_t osd_num = sb["params"]["osd_num"].uint64_value();
|
||||
osd_numbers.push_back(osd_num);
|
||||
superblocks.push_back(sb);
|
||||
}
|
||||
}
|
||||
if (!osd_numbers.size())
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
std::vector<std::string> rm_osd_cli = { "vitastor-cli", "rm-osd" };
|
||||
for (auto osd_num: osd_numbers)
|
||||
{
|
||||
rm_osd_cli.push_back(std::to_string(osd_num));
|
||||
}
|
||||
// Check for data loss
|
||||
if (options["force"] != "")
|
||||
{
|
||||
rm_osd_cli.push_back("--force");
|
||||
}
|
||||
else if (options["allow_data_loss"] != "")
|
||||
{
|
||||
rm_osd_cli.push_back("--allow-data-loss");
|
||||
}
|
||||
rm_osd_cli.push_back("--dry-run");
|
||||
std::string dry_run_ignore_stdout;
|
||||
if (shell_exec(rm_osd_cli, "", &dry_run_ignore_stdout, NULL) != 0)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
// Disable & stop OSDs
|
||||
std::vector<std::string> systemctl_cli = { "systemctl", "disable", "--now" };
|
||||
for (auto osd_num: osd_numbers)
|
||||
{
|
||||
systemctl_cli.push_back("vitastor-osd@"+std::to_string(osd_num));
|
||||
}
|
||||
if (shell_exec(systemctl_cli, "", NULL, NULL) != 0)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
// Remove OSD metadata
|
||||
rm_osd_cli.pop_back();
|
||||
if (shell_exec(rm_osd_cli, "", NULL, NULL) != 0)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
// Destroy OSD superblocks
|
||||
uint8_t *buf = (uint8_t*)memalign_or_die(MEM_ALIGNMENT, 4096);
|
||||
for (auto & sb: superblocks)
|
||||
{
|
||||
for (auto dev_type: std::vector<std::string>{ "data", "meta", "journal" })
|
||||
{
|
||||
auto dev = sb["real_"+dev_type+"_device"].string_value();
|
||||
if (dev != "")
|
||||
{
|
||||
int fd = -1, r = open(dev.c_str(), O_DIRECT|O_RDWR);
|
||||
if (r >= 0)
|
||||
{
|
||||
fd = r;
|
||||
r = read_blocking(fd, buf, 4096);
|
||||
if (r == 4096)
|
||||
{
|
||||
// Clear magic and CRC
|
||||
memset(buf, 0, 12);
|
||||
r = lseek64(fd, 0, 0);
|
||||
if (r == 0)
|
||||
{
|
||||
r = write_blocking(fd, buf, 4096);
|
||||
if (r == 4096)
|
||||
r = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (fd >= 0)
|
||||
close(fd);
|
||||
if (r != 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to clear OSD %ju %s device %s superblock: %s\n",
|
||||
sb["params"]["osd_num"].uint64_value(), dev_type.c_str(), dev.c_str(), strerror(errno));
|
||||
}
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "OSD %ju %s device %s superblock cleared\n",
|
||||
sb["params"]["osd_num"].uint64_value(), dev_type.c_str(), dev.c_str());
|
||||
}
|
||||
if (sb["params"][dev_type+"_device"].string_value().substr(0, 22) == "/dev/disk/by-partuuid/")
|
||||
{
|
||||
// Delete the partition itself
|
||||
auto uuid_to_del = strtolower(sb["params"][dev_type+"_device"].string_value().substr(22));
|
||||
auto parent_dev = get_parent_device(dev);
|
||||
if (parent_dev == "" || parent_dev == dev)
|
||||
{
|
||||
fprintf(stderr, "Failed to delete partition %s: failed to find parent device\n", dev.c_str());
|
||||
continue;
|
||||
}
|
||||
auto pt = read_parttable("/dev/"+parent_dev);
|
||||
if (!pt.is_object())
|
||||
continue;
|
||||
json11::Json::array newpt = pt["partitions"].array_items();
|
||||
for (int i = 0; i < newpt.size(); i++)
|
||||
{
|
||||
if (strtolower(newpt[i]["uuid"].string_value()) == uuid_to_del)
|
||||
{
|
||||
auto old_part = newpt[i];
|
||||
newpt.erase(newpt.begin()+i, newpt.begin()+i+1);
|
||||
vitastor_dev_info_t devinfo = {
|
||||
.path = "/dev/"+parent_dev,
|
||||
.pt = json11::Json::object{ { "partitions", newpt } },
|
||||
};
|
||||
add_partitions(devinfo, {});
|
||||
struct stat st;
|
||||
if (stat(old_part["node"].string_value().c_str(), &st) == 0 ||
|
||||
errno != ENOENT)
|
||||
{
|
||||
std::string out;
|
||||
shell_exec({ "partprobe", "/dev/"+parent_dev }, "", &out, NULL);
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
free(buf);
|
||||
buf = NULL;
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,142 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#include <regex>
|
||||
#include "disk_tool.h"
|
||||
#include "str_util.h"
|
||||
|
||||
static std::map<std::string, std::string> read_vitastor_unit(std::string unit)
|
||||
{
|
||||
std::smatch m;
|
||||
if (unit == "" || !std::regex_match(unit, m, std::regex(".*/vitastor-osd\\d+\\.service")))
|
||||
{
|
||||
fprintf(stderr, "unit file name does not match <path>/vitastor-osd<NUMBER>.service\n");
|
||||
return {};
|
||||
}
|
||||
std::string text = read_file(unit);
|
||||
if (!std::regex_search(text, m, std::regex("\nExecStart\\s*=[^\n]+vitastor-osd\\s*(([^\\\\\n&>\\d]+|\\\\[ \t\r]*\n|\\d[^>])+)")))
|
||||
{
|
||||
fprintf(stderr, "Failed to extract ExecStart command from %s\n", unit.c_str());
|
||||
return {};
|
||||
}
|
||||
std::string cmd = trim(m[1]);
|
||||
cmd = str_replace(cmd, "\\\n", " ");
|
||||
std::string key;
|
||||
std::map<std::string, std::string> r;
|
||||
auto ns = std::regex("\\S+");
|
||||
for (auto it = std::sregex_token_iterator(cmd.begin(), cmd.end(), ns, 0), end = std::sregex_token_iterator();
|
||||
it != end; it++)
|
||||
{
|
||||
if (key == "" && ((std::string)(*it)).substr(0, 2) == "--")
|
||||
key = ((std::string)(*it)).substr(2);
|
||||
else if (key != "")
|
||||
{
|
||||
r[key] = *it;
|
||||
key = "";
|
||||
}
|
||||
}
|
||||
return r;
|
||||
}
|
||||
|
||||
int disk_tool_t::upgrade_simple_unit(std::string unit)
|
||||
{
|
||||
if (stoull_full(unit) != 0)
|
||||
{
|
||||
// OSD number
|
||||
unit = "/etc/systemd/system/vitastor-osd"+unit+".service";
|
||||
}
|
||||
auto options = read_vitastor_unit(unit);
|
||||
if (!options.size())
|
||||
return 1;
|
||||
if (!stoull_full(options["osd_num"], 10) || options["data_device"] == "")
|
||||
{
|
||||
fprintf(stderr, "osd_num or data_device are missing in %s\n", unit.c_str());
|
||||
return 1;
|
||||
}
|
||||
if (options["data_device"].substr(0, 22) != "/dev/disk/by-partuuid/" ||
|
||||
options["meta_device"] != "" && options["meta_device"].substr(0, 22) != "/dev/disk/by-partuuid/" ||
|
||||
options["journal_device"] != "" && options["journal_device"].substr(0, 22) != "/dev/disk/by-partuuid/")
|
||||
{
|
||||
fprintf(
|
||||
stderr, "data_device, meta_device and journal_device must begin with"
|
||||
" /dev/disk/by-partuuid/ i.e. they must be GPT partitions identified by UUIDs"
|
||||
);
|
||||
return 1;
|
||||
}
|
||||
// Stop and disable the service
|
||||
auto service_name = unit.substr(unit.rfind('/') + 1);
|
||||
if (shell_exec({ "systemctl", "disable", "--now", service_name }, "", NULL, NULL) != 0)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
uint64_t j_o = parse_size(options["journal_offset"]);
|
||||
uint64_t m_o = parse_size(options["meta_offset"]);
|
||||
uint64_t d_o = parse_size(options["data_offset"]);
|
||||
bool m_is_d = options["meta_device"] == "" || options["meta_device"] == options["data_device"];
|
||||
bool j_is_m = options["journal_device"] == "" || options["journal_device"] == options["meta_device"];
|
||||
bool j_is_d = j_is_m && m_is_d || options["journal_device"] == options["data_device"];
|
||||
if (d_o < 4096 || j_o < 4096 || m_o < 4096)
|
||||
{
|
||||
// Resize data
|
||||
uint64_t blk = stoull_full(options["block_size"]);
|
||||
blk = blk ? blk : 128*1024;
|
||||
std::map<std::string, uint64_t> resize;
|
||||
if (d_o < 4096 || m_is_d && m_o < 4096 && m_o < d_o || j_is_d && j_o < 4096 && j_o < d_o)
|
||||
{
|
||||
resize["new_data_offset"] = d_o+blk;
|
||||
if (m_is_d && m_o < d_o)
|
||||
resize["new_meta_offset"] = m_o+blk;
|
||||
if (j_is_d && j_o < d_o)
|
||||
resize["new_journal_offset"] = j_o+blk;
|
||||
}
|
||||
if (!m_is_d && m_o < 4096)
|
||||
{
|
||||
resize["new_meta_offset"] = m_o+4096;
|
||||
if (j_is_m && m_o < j_o)
|
||||
resize["new_journal_offset"] = j_o+4096;
|
||||
}
|
||||
if (!j_is_d && !j_is_m && j_o < 4096)
|
||||
resize["new_journal_offset"] = j_o+4096;
|
||||
disk_tool_t resizer;
|
||||
resizer.options = options;
|
||||
for (auto & kv: resize)
|
||||
resizer.options[kv.first] = std::to_string(kv.second);
|
||||
if (resizer.resize_data() != 0)
|
||||
{
|
||||
// FIXME: Resize with backup or journal
|
||||
fprintf(
|
||||
stderr, "Failed to resize data to make space for the superblock\n"
|
||||
"Sorry, but your OSD may now be corrupted depending on what went wrong during resize :-(\n"
|
||||
"Please review the messages above and take action accordingly\n"
|
||||
);
|
||||
return 1;
|
||||
}
|
||||
for (auto & kv: resize)
|
||||
options[kv.first.substr(4)] = std::to_string(kv.second);
|
||||
}
|
||||
// Write superblocks
|
||||
if (!write_osd_superblock(options["data_device"], options) ||
|
||||
(!m_is_d && !write_osd_superblock(options["meta_device"], options)) ||
|
||||
(!j_is_m && !j_is_d && !write_osd_superblock(options["journal_device"], options)))
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
// Change partition types
|
||||
if (fix_partition_type(options["data_device"]) != 0 ||
|
||||
(!m_is_d && fix_partition_type(options["meta_device"]) != 0) ||
|
||||
(!j_is_m && !j_is_d && fix_partition_type(options["journal_device"]) != 0))
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
// Enable the new unit
|
||||
if (shell_exec({ "systemctl", "enable", "--now", "vitastor-osd@"+options["osd_num"] }, "", NULL, NULL) != 0)
|
||||
{
|
||||
fprintf(stderr, "Failed to enable systemd unit vitastor-osd@%s\n", options["osd_num"].c_str());
|
||||
return 1;
|
||||
}
|
||||
fprintf(
|
||||
stderr, "\nOK: Converted OSD %s to the new scheme. The new service name is vitastor-osd@%s\n",
|
||||
options["osd_num"].c_str(), options["osd_num"].c_str()
|
||||
);
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,365 @@
|
||||
// Copyright (c) Vitaliy Filippov, 2019+
|
||||
// License: VNPL-1.1 (see README.md for details)
|
||||
|
||||
#include <sys/wait.h>
|
||||
#include <dirent.h>
|
||||
|
||||
#include "disk_tool.h"
|
||||
#include "rw_blocking.h"
|
||||
#include "str_util.h"
|
||||
|
||||
uint64_t sscanf_json(const char *fmt, const json11::Json & str)
|
||||
{
|
||||
uint64_t value = 0;
|
||||
if (fmt)
|
||||
sscanf(str.string_value().c_str(), "%jx", &value);
|
||||
else if (str.string_value().size() > 2 && (str.string_value()[0] == '0' && str.string_value()[1] == 'x'))
|
||||
sscanf(str.string_value().c_str(), "0x%jx", &value);
|
||||
else
|
||||
value = str.uint64_value();
|
||||
return value;
|
||||
}
|
||||
|
||||
static int fromhex(char c)
|
||||
{
|
||||
if (c >= '0' && c <= '9')
|
||||
return (c-'0');
|
||||
else if (c >= 'a' && c <= 'f')
|
||||
return (c-'a'+10);
|
||||
else if (c >= 'A' && c <= 'F')
|
||||
return (c-'A'+10);
|
||||
return -1;
|
||||
}
|
||||
|
||||
void fromhexstr(const std::string & from, int bytes, uint8_t *to)
|
||||
{
|
||||
for (int i = 0; i < from.size() && i < bytes; i++)
|
||||
{
|
||||
int x = fromhex(from[2*i]), y = fromhex(from[2*i+1]);
|
||||
if (x < 0 || y < 0)
|
||||
break;
|
||||
to[i] = x*16 + y;
|
||||
}
|
||||
}
|
||||
|
||||
// returns 1 = check error, 0 = write through, -1 = write back
|
||||
// (similar to 1 = warning, -1 = error, 0 = success in disable_cache)
|
||||
static int check_queue_cache(std::string dev, std::string parent_dev)
|
||||
{
|
||||
auto r = read_file("/sys/block/"+dev+"/queue/write_cache", true);
|
||||
if (r == "")
|
||||
r = read_file("/sys/block/"+parent_dev+"/queue/write_cache");
|
||||
if (r == "")
|
||||
return 1;
|
||||
return trim(r) == "write through" ? 0 : -1;
|
||||
}
|
||||
|
||||
// returns 1 = warning, -1 = error, 0 = success
|
||||
int disable_cache(std::string dev)
|
||||
{
|
||||
auto parent_dev = get_parent_device(dev);
|
||||
if (parent_dev == "")
|
||||
return 1;
|
||||
auto scsi_disk = "/sys/block/"+parent_dev+"/device/scsi_disk";
|
||||
DIR *dir = opendir(scsi_disk.c_str());
|
||||
if (!dir)
|
||||
{
|
||||
if (errno == ENOENT)
|
||||
{
|
||||
// Not a SCSI/SATA device, just check /sys/block/.../queue/write_cache
|
||||
return check_queue_cache(dev.substr(5), parent_dev);
|
||||
}
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "Can't read directory %s: %s\n", scsi_disk.c_str(), strerror(errno));
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
dirent *de = readdir(dir);
|
||||
while (de && de->d_name[0] == '.' && (de->d_name[1] == 0 || de->d_name[1] == '.' && de->d_name[2] == 0))
|
||||
de = readdir(dir);
|
||||
if (!de)
|
||||
{
|
||||
// Not a SCSI/SATA device, just check /sys/block/.../queue/write_cache
|
||||
closedir(dir);
|
||||
return check_queue_cache(dev.substr(5), parent_dev);
|
||||
}
|
||||
scsi_disk += "/";
|
||||
scsi_disk += de->d_name;
|
||||
if (readdir(dir) != NULL)
|
||||
{
|
||||
// Error, multiple scsi_disk/* entries
|
||||
closedir(dir);
|
||||
fprintf(stderr, "Multiple entries in %s found\n", scsi_disk.c_str());
|
||||
return 1;
|
||||
}
|
||||
closedir(dir);
|
||||
// Check cache_type
|
||||
scsi_disk += "/cache_type";
|
||||
std::string cache_type = trim(read_file(scsi_disk));
|
||||
if (cache_type == "")
|
||||
return 1;
|
||||
if (cache_type != "write through")
|
||||
{
|
||||
int fd = open(scsi_disk.c_str(), O_WRONLY);
|
||||
if (fd < 0 || write_blocking(fd, (void*)"write through", strlen("write through")) != strlen("write through"))
|
||||
{
|
||||
if (fd >= 0)
|
||||
close(fd);
|
||||
fprintf(stderr, "Can't write to %s: %s\n", scsi_disk.c_str(), strerror(errno));
|
||||
return -1;
|
||||
}
|
||||
close(fd);
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
std::string get_parent_device(std::string dev)
|
||||
{
|
||||
if (dev.substr(0, 5) != "/dev/")
|
||||
{
|
||||
fprintf(stderr, "%s is outside /dev/\n", dev.c_str());
|
||||
return "";
|
||||
}
|
||||
dev = dev.substr(5);
|
||||
int i = dev.size();
|
||||
while (i > 0 && isdigit(dev[i-1]))
|
||||
i--;
|
||||
if (i >= 1 && dev[i-1] == '-') // dm-0, dm-1
|
||||
return dev;
|
||||
else if (i >= 2 && dev[i-1] == 'p' && isdigit(dev[i-2])) // nvme0n1p1
|
||||
i--;
|
||||
// Check that such block device exists
|
||||
struct stat st;
|
||||
auto chk = "/sys/block/"+dev.substr(0, i);
|
||||
if (stat(chk.c_str(), &st) < 0)
|
||||
{
|
||||
if (errno != ENOENT)
|
||||
{
|
||||
fprintf(stderr, "Failed to stat %s: %s\n", chk.c_str(), strerror(errno));
|
||||
return "";
|
||||
}
|
||||
return dev;
|
||||
}
|
||||
return dev.substr(0, i);
|
||||
}
|
||||
|
||||
bool json_is_true(const json11::Json & val)
|
||||
{
|
||||
if (val.is_string())
|
||||
return val == "true" || val == "yes" || val == "1";
|
||||
return val.bool_value();
|
||||
}
|
||||
|
||||
int shell_exec(const std::vector<std::string> & cmd, const std::string & in, std::string *out, std::string *err)
|
||||
{
|
||||
int child_stdin[2], child_stdout[2], child_stderr[2];
|
||||
pid_t pid;
|
||||
if (pipe(child_stdin) == -1)
|
||||
goto err_pipe1;
|
||||
if (pipe(child_stdout) == -1)
|
||||
goto err_pipe2;
|
||||
if (pipe(child_stderr) == -1)
|
||||
goto err_pipe3;
|
||||
if ((pid = fork()) == -1)
|
||||
goto err_fork;
|
||||
if (pid)
|
||||
{
|
||||
// Parent
|
||||
// We should do select() to do something serious, but this is for simple cases
|
||||
close(child_stdin[0]);
|
||||
close(child_stdout[1]);
|
||||
close(child_stderr[1]);
|
||||
write_blocking(child_stdin[1], (void*)in.data(), in.size());
|
||||
close(child_stdin[1]);
|
||||
std::string s;
|
||||
s = read_all_fd(child_stdout[0]);
|
||||
if (out)
|
||||
out->swap(s);
|
||||
close(child_stdout[0]);
|
||||
s = read_all_fd(child_stderr[0]);
|
||||
if (err)
|
||||
err->swap(s);
|
||||
close(child_stderr[0]);
|
||||
int wstatus = 0;
|
||||
waitpid(pid, &wstatus, 0);
|
||||
return WEXITSTATUS(wstatus);
|
||||
}
|
||||
else
|
||||
{
|
||||
// Child
|
||||
dup2(child_stdin[0], 0);
|
||||
if (out)
|
||||
dup2(child_stdout[1], 1);
|
||||
if (err)
|
||||
dup2(child_stderr[1], 2);
|
||||
close(child_stdin[0]);
|
||||
close(child_stdin[1]);
|
||||
close(child_stdout[0]);
|
||||
close(child_stdout[1]);
|
||||
close(child_stderr[0]);
|
||||
close(child_stderr[1]);
|
||||
char *argv[cmd.size()+1];
|
||||
for (int i = 0; i < cmd.size(); i++)
|
||||
argv[i] = (char*)cmd[i].c_str();
|
||||
argv[cmd.size()] = NULL;
|
||||
execvp(argv[0], argv);
|
||||
std::string full_cmd;
|
||||
for (int i = 0; i < cmd.size(); i++)
|
||||
{
|
||||
full_cmd += cmd[i];
|
||||
full_cmd += " ";
|
||||
}
|
||||
full_cmd.resize(full_cmd.size() > 0 ? full_cmd.size()-1 : 0);
|
||||
fprintf(stderr, "error running %s: %s", full_cmd.c_str(), strerror(errno));
|
||||
exit(255);
|
||||
}
|
||||
err_fork:
|
||||
close(child_stderr[1]);
|
||||
close(child_stderr[0]);
|
||||
err_pipe3:
|
||||
close(child_stdout[1]);
|
||||
close(child_stdout[0]);
|
||||
err_pipe2:
|
||||
close(child_stdin[1]);
|
||||
close(child_stdin[0]);
|
||||
err_pipe1:
|
||||
return 255;
|
||||
}
|
||||
|
||||
int write_zero(int fd, uint64_t offset, uint64_t size)
|
||||
{
|
||||
uint64_t buf_len = 1024*1024;
|
||||
void *zero_buf = memalign_or_die(MEM_ALIGNMENT, buf_len);
|
||||
memset(zero_buf, 0, buf_len);
|
||||
ssize_t r;
|
||||
while (size > 0)
|
||||
{
|
||||
r = pwrite(fd, zero_buf, size > buf_len ? buf_len : size, offset);
|
||||
if (r > 0)
|
||||
{
|
||||
size -= r;
|
||||
offset += r;
|
||||
}
|
||||
else if (errno != EAGAIN && errno != EINTR)
|
||||
{
|
||||
free(zero_buf);
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
free(zero_buf);
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Returns false in case of an error
|
||||
// Returns null if there is no partition table
|
||||
json11::Json read_parttable(std::string dev)
|
||||
{
|
||||
std::string part_dump;
|
||||
int r = shell_exec({ "sfdisk", "--json", dev }, "", &part_dump, NULL);
|
||||
if (r == 255)
|
||||
{
|
||||
fprintf(stderr, "Error running sfdisk --json %s\n", dev.c_str());
|
||||
return json11::Json(false);
|
||||
}
|
||||
// Decode partition table
|
||||
json11::Json pt;
|
||||
if (part_dump != "")
|
||||
{
|
||||
std::string err;
|
||||
pt = json11::Json::parse(part_dump, err);
|
||||
if (err != "")
|
||||
{
|
||||
fprintf(stderr, "sfdisk --json %s returned bad JSON: %s\n", dev.c_str(), part_dump.c_str());
|
||||
return json11::Json(false);
|
||||
}
|
||||
pt = pt["partitiontable"];
|
||||
if (pt.is_object() && pt["label"].string_value() != "gpt")
|
||||
{
|
||||
fprintf(stderr, "%s contains \"%s\" partition table, only GPT is supported, skipping\n", dev.c_str(), pt["label"].string_value().c_str());
|
||||
return json11::Json(false);
|
||||
}
|
||||
}
|
||||
return pt;
|
||||
}
|
||||
|
||||
uint64_t dev_size_from_parttable(json11::Json pt)
|
||||
{
|
||||
uint64_t free = pt["lastlba"].uint64_value() + 1 - pt["firstlba"].uint64_value();
|
||||
if (!pt["sectorsize"].uint64_value())
|
||||
free *= 512;
|
||||
else
|
||||
free *= pt["sectorsize"].uint64_value();
|
||||
return free;
|
||||
}
|
||||
|
||||
uint64_t free_from_parttable(json11::Json pt)
|
||||
{
|
||||
uint64_t free = pt["lastlba"].uint64_value() + 1 - pt["firstlba"].uint64_value();
|
||||
for (const auto & part: pt["partitions"].array_items())
|
||||
free -= part["size"].uint64_value();
|
||||
if (!pt["sectorsize"].uint64_value())
|
||||
free *= 512;
|
||||
else
|
||||
free *= pt["sectorsize"].uint64_value();
|
||||
return free;
|
||||
}
|
||||
|
||||
int fix_partition_type(std::string dev_by_uuid)
|
||||
{
|
||||
auto uuid = strtolower(dev_by_uuid.substr(dev_by_uuid.rfind('/')+1));
|
||||
std::string parent_dev = get_parent_device(realpath_str(dev_by_uuid, false));
|
||||
if (parent_dev == "")
|
||||
return 1;
|
||||
auto pt = read_parttable("/dev/"+parent_dev);
|
||||
if (pt.is_null() || pt.is_bool())
|
||||
return 1;
|
||||
std::string script = "label: gpt\n\n";
|
||||
for (const auto & part: pt["partitions"].array_items())
|
||||
{
|
||||
bool this_part = (strtolower(part["uuid"].string_value()) == uuid);
|
||||
if (this_part && strtolower(part["type"].string_value()) == "e7009fac-a5a1-4d72-af72-53de13059903")
|
||||
{
|
||||
// Already correct type
|
||||
return 0;
|
||||
}
|
||||
script += part["node"].string_value()+": ";
|
||||
bool first = true;
|
||||
for (const auto & kv: part.object_items())
|
||||
{
|
||||
if (kv.first != "node")
|
||||
{
|
||||
script += (first ? "" : ", ")+kv.first+"="+
|
||||
(kv.first == "type" && this_part
|
||||
? "e7009fac-a5a1-4d72-af72-53de13059903"
|
||||
: (kv.second.is_string() ? kv.second.string_value() : kv.second.dump()));
|
||||
first = false;
|
||||
}
|
||||
}
|
||||
script += "\n";
|
||||
}
|
||||
std::string out;
|
||||
return shell_exec({ "sfdisk", "--no-reread", "--force", "/dev/"+parent_dev }, script, &out, NULL);
|
||||
}
|
||||
|
||||
std::string csum_type_str(uint32_t data_csum_type)
|
||||
{
|
||||
std::string csum_type;
|
||||
if (data_csum_type == BLOCKSTORE_CSUM_NONE)
|
||||
csum_type = "none";
|
||||
else if (data_csum_type == BLOCKSTORE_CSUM_CRC32C)
|
||||
csum_type = "crc32c";
|
||||
else
|
||||
csum_type = std::to_string(data_csum_type);
|
||||
return csum_type;
|
||||
}
|
||||
|
||||
uint32_t csum_type_from_str(std::string data_csum_type)
|
||||
{
|
||||
if (data_csum_type == "crc32c")
|
||||
return BLOCKSTORE_CSUM_CRC32C;
|
||||
return stoull_full(data_csum_type, 0);
|
||||
}
|
||||
Reference in New Issue
Block a user