Compare commits

..
183 Commits
Author SHA1 Message Date
Vitaliy Filippov 5c0bf7c293 Add detailed security feature documentation 2026-07-08 01:02:30 +03:00
Vitaliy Filippov 2e03568225 Implement vitastor-cli cpubench for AES and xxhash3 benchmarks 2026-07-08 00:19:55 +03:00
Vitaliy Filippov 7dabef4851 Add security parameter documentation 2026-07-05 18:51:26 +03:00
Vitaliy Filippov 29c3841eaa Allow to make-etcd --antietcd-only 2026-07-05 18:51:23 +03:00
Vitaliy Filippov 45204da444 Rename server_cert to api_cert 2026-07-05 15:44:59 +03:00
Vitaliy Filippov c18ddbf255 Remove antietcd_ca parameter from mon/antietcd_adapter - it is client_ca 2026-07-05 15:36:20 +03:00
Vitaliy Filippov 76410df0e8 Do not require explicit peer_ca for antietcd 2026-07-05 15:19:03 +03:00
Vitaliy Filippov 6a0f3a38d0 Do not enable use_perms by default - make it more explicit for users 2026-07-05 15:19:03 +03:00
Vitaliy Filippov 265e99ccd0 Support TLS certificate generation in make-etcd 2026-07-05 15:19:03 +03:00
Vitaliy Filippov aa71a1968f Rename use_auth to use_perms 2026-07-05 14:58:24 +03:00
Vitaliy Filippov 5382f1c7bb Fix use_auth usage 2026-07-05 14:58:24 +03:00
Vitaliy Filippov a5dbf74123 Add support for full AES-GCM including double encryption of AES-XTS :D 2026-07-05 14:58:24 +03:00
Vitaliy Filippov 784fd7d233 Also allow clients to load /inode/stats/ for their inodes 2026-07-05 14:58:24 +03:00
Vitaliy Filippov 13b7c0e73d Load all /config/ 2026-07-05 14:58:24 +03:00
Vitaliy Filippov d747ec8c41 Fix modify owner/groups 2026-07-05 14:58:24 +03:00
Vitaliy Filippov cabe5d1543 Skip non-existing pools 2026-07-05 14:58:24 +03:00
Vitaliy Filippov 38da624930 Use osd & mon certs instead of usernames 2026-07-05 14:58:24 +03:00
Vitaliy Filippov e0ff8a014f Remove mon and osd user types 2026-07-05 14:58:24 +03:00
Vitaliy Filippov 8721f21874 Also check for "Received garbage" during test 2026-07-05 14:58:24 +03:00
Vitaliy Filippov 9eb858fb5f Add a CI test with RDMA 2026-07-05 14:58:24 +03:00
Vitaliy Filippov 9f084f48f6 Add request size validation to prevent OOMDoS 2026-07-05 14:58:24 +03:00
Vitaliy Filippov c998325448 Implement OSD-side authorization for operations 2026-07-05 14:58:24 +03:00
Vitaliy Filippov a243ff6459 Remove tls_ from option names 2026-07-05 14:58:24 +03:00
Vitaliy Filippov 47b3294666 Remove TLS support (superseded by direct GCM) 2026-07-05 14:58:24 +03:00
Vitaliy Filippov 3afed2473e Implement TLS 1.3-like handshake manually 2026-07-05 14:58:24 +03:00
Vitaliy Filippov fdeacdcf30 Extract openssl-related code, wire ssl implementation back 2026-07-05 14:58:24 +03:00
Vitaliy Filippov d7ddd7752c Coalesce entries in send_out_buf 2026-07-05 14:58:24 +03:00
Vitaliy Filippov 2b2c7d5e39 Support isa-l_crypto for AES-XTS too 2026-07-05 14:58:24 +03:00
Vitaliy Filippov e26478c2e7 Fix xts+rdma encrypt errors 2026-07-05 14:58:24 +03:00
Vitaliy Filippov c7b9d30fd3 Fix "use-after-realloc" warning 2026-07-05 14:58:24 +03:00
Vitaliy Filippov 5803655840 Support isa-l_crypto for AES-GCM 2026-07-05 14:58:24 +03:00
Vitaliy Filippov 61ce26ba29 Remove WITH_OPENSSL from all files except http_client, always require OpenSSL 2026-07-05 14:58:24 +03:00
Vitaliy Filippov 32e651ec96 Use pools for GCM contexts 2026-07-05 14:58:23 +03:00
Vitaliy Filippov dbed0e94cb Try to run tests with AES-GCM 2026-07-05 14:58:09 +03:00
Vitaliy Filippov 8b39a268b9 Implement direct AES-256-GCM with a static key for benchmark 2026-07-05 14:58:09 +03:00
Vitaliy Filippov d40ba7c5ce Make sure to send all TLS data before continuing 2026-07-05 14:58:09 +03:00
Vitaliy Filippov 985257f2d7 Parse standard TLS record headers 2026-07-05 14:58:09 +03:00
Vitaliy Filippov a4f3383f21 Omit msgr_tls_record_hdr_t for non-tls data 2026-07-05 14:58:09 +03:00
Vitaliy Filippov b2d828ec84 Allow 2 and 4 byte per block chain_info (allow more than 255 snapshots with encryption) 2026-07-05 14:58:09 +03:00
Vitaliy Filippov b2a74de715 Implement OSD TLS support 2026-07-05 14:58:09 +03:00
Vitaliy Filippov 430761ad2e Allow to skip checksums for headers 2026-07-05 14:58:09 +03:00
Vitaliy Filippov 38a1f6937d Implement protocol-level checksums (xxhash3) 2026-07-05 14:58:09 +03:00
Vitaliy Filippov 8b46914999 Include xxhash3 x86dispatch 2026-07-05 14:58:09 +03:00
Vitaliy Filippov 1b81e60187 Do not use scrap_buffer in the client (it would block protocol checksum support) 2026-07-05 14:58:09 +03:00
Vitaliy Filippov 526c597cd0 Support TLS CN authentication and per-image permissions in vitastor-cli serve 2026-07-05 14:58:09 +03:00
Vitaliy Filippov 7a97783ead Add VitastorAuthFilter 2026-07-05 14:58:09 +03:00
Vitaliy Filippov b4425d8e65 Implement vitastor-cli ls-user, modify-user, remove-user commands 2026-07-05 14:58:09 +03:00
Vitaliy Filippov 1afda35bd3 Add image owner/owner_group/reader_group support (for antietcd VitastorAuthFilter) 2026-07-05 14:58:09 +03:00
Vitaliy Filippov 0953f5ebdd Support inline (string PEM) certificates and pkeys 2026-07-05 14:10:11 +03:00
Vitaliy Filippov 1b03615e50 Show encryption keys (only IDs) in the listing 2026-07-05 14:10:11 +03:00
Vitaliy Filippov 8f208c53df Support storing image encryption keys in Vault 2026-07-05 14:10:11 +03:00
Vitaliy Filippov c0d2dabe66 Prefer local etcd addresses and correctly cycle over them even when they need resolving
Seems slightly overcomplicated...
2026-07-05 14:10:10 +03:00
Vitaliy Filippov 6da4aaf176 Support DNS resolving via libc-ares 2026-07-05 14:10:10 +03:00
Vitaliy Filippov 5b6a6fce9a Batch handle_immediate_ops more 2026-07-05 14:10:10 +03:00
Vitaliy Filippov 79fb2def57 Add vitastor-cli create & modify --enc-key parameter 2026-07-05 14:10:10 +03:00
Vitaliy Filippov 7f5c24144b Support reading from snapshots encrypted with different keys 2026-07-05 14:10:10 +03:00
Vitaliy Filippov 3cf876fafa Support decryption with multiple keys 2026-07-05 14:10:10 +03:00
Vitaliy Filippov 771aa83282 Allow to return chain_info in response to reads 2026-07-05 14:10:10 +03:00
Vitaliy Filippov 03404ac95d Add basic AES-XTS client-side encryption support 2026-07-05 14:10:10 +03:00
Vitaliy Filippov decf314238 Rework msgr send/receive to allow encryption support 2026-07-05 14:10:10 +03:00
Vitaliy Filippov ab1849ff07 Move fromhexstr() to str_util 2026-07-05 14:10:10 +03:00
Vitaliy Filippov 33b18c7229 Add openapi description 2026-07-05 14:10:10 +03:00
Vitaliy Filippov 4a2efb29ec Slightly fix API return and input types 2026-07-05 14:10:10 +03:00
Vitaliy Filippov 00a432193a Implement vitastor-cli serve command to serve simple HTTP API 2026-07-05 14:10:10 +03:00
Vitaliy Filippov d83eb599a0 Implement HTTP server support O_o 2026-07-05 14:10:10 +03:00
Vitaliy Filippov c1c9b1975d Rename http_response_t to http_message_t 2026-07-05 14:10:10 +03:00
Vitaliy Filippov ba0d9ad9f2 Extract common HTTP context 2026-07-05 14:10:10 +03:00
Vitaliy Filippov 4feabc4ab6 Support xxhash 32-bit checksums (data_csum_type=xxh3_32) 2026-07-05 14:10:10 +03:00
Vitaliy Filippov 1a14301c50 Detect block checksums using csum_block_size, not data_csum_type 2026-07-05 14:09:57 +03:00
Vitaliy Filippov 5dc52a7c06 Add client certificate support 2026-07-05 14:09:57 +03:00
Vitaliy Filippov 4d45b27696 Do not re-initialize TLS context every connection 2026-07-05 14:09:57 +03:00
Vitaliy Filippov 5995c4a00f Add https support to antietcd 2026-07-05 14:09:57 +03:00
Vitaliy Filippov 906388adaa Implement etcd SSL support via OpenSSL
Maybe I should remove all of this and use libwebsockets :)
2026-07-05 14:09:57 +03:00
Vitaliy Filippov 5dbc679e16 Fix padded block checksums - v2 2026-07-05 14:09:57 +03:00
Vitaliy Filippov d48864a6be Fix unaligned pointer warnings 2026-07-04 21:40:29 +03:00
Vitaliy Filippov 462482d319 Fix zero-padded big_write checksum verification in the new store 2026-07-04 21:37:22 +03:00
Vitaliy Filippov 5ef9d78461 Release 3.0.15
- QEMU virtual disk migration with enabled iothread is finally fixed correctly.
- Fixed operation of Proxmox VMs with swTPM without enabling NBD for all disks.
- Debian packages are now again built with stable Antietcd released instead of
  the unstable master branch.
- Antietcd cluster mode previously broken in that master branch is fixed. The
  symptom was Antietcd being unable to elect the leader in a cluster.
- Fixed monitor startup with embedded Antietcd when using IPv6.
- Fixed space statistics calculation for FS and S3 pools in the new storage ([PR #127](https://github.com/vitalif/vitastor/pull/127)).
2026-06-28 11:29:38 +03:00
Vitaliy Filippov 1b20010a69 Install npm packages in debian/rules 2026-06-28 11:29:27 +03:00
Vitaliy Filippov d750d00c6f Fix SWTPM in Proxmox 2026-06-27 16:04:51 +03:00
Vitaliy Filippov 66cb564e2c Fix background jobs in etcd_fail test 2026-06-27 15:42:53 +03:00
Vitaliy Filippov f782decbd6 Update antietcd to 1.3.1 2026-06-27 15:42:42 +03:00
Vitaliy Filippov 717582c4ba Use release builds of antietcd&tinyraft in debian packages (not master branch) 2026-06-27 10:27:15 +03:00
Vitaliy Filippov cd62c0755f Fix IPv6 antietcd configuration 2026-06-27 00:16:03 +03:00
c6f5ab7d79 fix pool stats with no_inode_stats (#127)
Co-authored-by: zhu.chengzhen <zhu.chengzhen@jingjiamicro.com>
2026-06-24 12:41:56 +03:00
Vitaliy Filippov 09607ddcbe Actual fix for live migration with iothread 2026-06-23 00:07:34 +03:00
Vitaliy Filippov 278852b4d5 Release 3.0.14
How many (bugs!) I've knifed, how many I've slit!

General note: most bug fixes now include regression tests to verify that they don't repeat in the future.
Most bugs fixed in this release were detected by using LLM analysis (Claude Opus/Fable, GPT 5.5).

OSD:
- Fix OSD hanging with an infinite loop when setting autosync_interval to 0 at runtime
- (IMPORTANT) Fix EC PGs hanging in REPEERING when the last final commit/rollback in a
  batch completes with an error
- Limit pg_size by 64 because peering doesn't handle larger values with EC — they just
  lead to 'incomplete' objects
- Fix OSD crash with an "assertion failed" error on EIO retry in snapshot chain read
  (i.e. when some chunks belong to a corrupted replica with checksum mismatch)
- (IMPORTANT) Disable chunked PG count resharding due to possible interference with
  compaction changes (will be re-enabled after fixes)
- (IMPORTANT) Fix incorrect snapshot allocation bitmap recovery during EC chained read
- Add on-wire request size validation to prevent possible OOM/DoS/heap corruption
  on receiving invalid data from the network
- (IMPORTANT) Fix parity-less EC writes destroying snapshot allocation bitmaps
  (i.e. when all parity OSDs in a PG are missing)
- (IMPORTANT) Fix EC N+K, K>=2 recovery destroying snapshot allocation bitmaps
  of live parity chunks
- (IMPORTANT) Fix a possible OSD crash during EC misplaced object scrubbing
- (IMPORTANT) Verify object bitmap consistency during scrub (only data was checked previously)
- (IMPORTANT) Fix corrupted object chunks incorrectly marked as non-corrupted on the second scrub
- (IMPORTANT) Fix cached EC decoding of multiple stripes with ISA-L (ISA-L is the default)

New store:
- Fix a theoretically possible OSD crash on startup when using the previously added
  workaround for the "double-claim" problem
- Remove theoretically possible incorrect metadata block writes during batch EC COMMITs
  restarted due to a full metadata area
- Fix incorrect compaction counter tracking after OSD restart (could probably lead to
  compaction not restarted correctly after a restart)
- (IMPORTANT) Fix some of parallel big_writes possibly not waiting for data fsync, thus not providing durability
- Fix possible OSD crash on sync retry when io_uring is full
- Fix a possible crash during startup on corrupted on-disk data with too small entry sizes

Old store:
- Prevent loading extra garbage metadata entries from the last 4 MB of metadata area
- Fix read operations possibly crashing if a metadata read (with inmemory_metadata=false)
  was restarted due to a full io_uring
- Fix a possible memory leak of temporary buffers and bitmaps/checksums when a read
  was restarted due to a full io_uring (reproducible with either inmemory_metadata=false or block_size>256k)
- Fix a possible OSD crash during padded checksum reads if buffer count exceeded 1024
  (IOV_MAX) (reproducible only with csum_block_size > 4k and block_size >= 4M)
- (IMPORTANT) Fix partial padded read journal checksum verification with csum_block_size > 4k
- Fix incorrect marking of corrupted objects as non-corrupted after flushing data
  from journal (with inmemory_journal=false)
- (IMPORTANT) Fix deferred freeing of a different block when a block was used by a parallel read
- Fix per-inode statistics not being disabled for FS and S3 pools correctly, leading to etcd
  overload with unneeded per-inode statistics, slower etcd operation, increased memory usage,
  and too many Prometheus statistics exported by the monitor

Both stores:
- Fix possibly left garbage in the metadata area if the first OSD startup was interrupted —
  metadata header is now written only after initializating metadata
- Check for short reads during initialization (just in case, doesn't happen in real life)

Clients:
- Fix write-back queue item split in case when write-back is enabled at runtime
- Implement bdrv_detach_aio_context & bdrv_attach_aio_context in the QEMU driver (should fix migration with iothread)
- Do not crash on full io_uring in ublk server
- Fix missing --readonly option handling in NBD server
- Stop gracefully on NBD_CMD_DISC instead of just exit(0) in NBD server
- Fix writeback detection in ublk server for --image mode
- Limit the amount of incoming data for NFS clients to prevent choking on memory in async mount mode

Tools (vitastor-disk/vitastor-cli):
- Prevent vitastor-cli merge possibly exiting before completing the last sync/delete operations
- Fix vitastor-disk incorrectly validating too large small_write entry length
- Fix vitastor-cli merge ignoring input option validation errors
- Fix vitastor-cli rm-data always skipping the final fsync
- Fix vitastor-disk resize not moving the last used data block
- Fix vitastor-disk write-meta incorrectly importing new store small_write entries
- Fix vitastor-disk write-journal and write-meta importing old store data incorrectly
  when csum_block_size is > 4k
- Support --io option for vitastor-disk dump-journal/write-journal
- Fix vitastor-disk resize crash when converting from very old (0.5.x) metadata
- Fix vitastor-disk trim incorrectly rounding block ranges with --discard_granularity
  option explicitly set to a value > 4k, possibly leading to discarding live data
- Fix vitastor-disk write-meta importing new store metadata incorrectly with > 4 GB metadata area size
- (IMPORTANT) Fix vitastor-cli modify --resize to a smaller size clearing all image data O_o

Other:
- Do not crash with an uncaught exception when an invalid /osd/state/ with a non-numeric
  suffix is present in etcd (in OSD and all client services)
- Fix possible crash in vitastor-kv when handling a corrupted DB due to a uint32 overflow
- Fix NFS-RDMA memory allocator crashing in some situations
- Fix small shared file extend-write potentially reading unallocated memory (NFS)
- Add bounds checks to prevent uint32 overflows in NFS/XDR
- Re-enable accidentally disabled safety checks (asserts) in files with included cpp-btree
- Fix too small memory allocation in NFS portmap
2026-06-21 18:17:02 +03:00
Vitaliy Filippov 7b11c6e90d Add a regression test for v1 store attempting to load overflowing garbage from the end of metadata area (commit 6b003bcc34) 2026-06-21 17:23:50 +03:00
Vitaliy Filippov 6251ce8b9a Add a regression test for incorrect deferred freeing of data block (commit 43aa4cfff6) 2026-06-21 16:49:38 +03:00
Vitaliy Filippov d4a42f61cf Handle full io_uring in ublk-server 2026-06-21 01:40:49 +03:00
Vitaliy Filippov 6b003bcc34 Protect against done_cnt > block_count, just in case 2026-06-20 21:05:06 +03:00
Vitaliy Filippov 1c945bcb41 Make check_completed a macro in test_cluster_client 2026-06-20 20:18:31 +03:00
Vitaliy Filippov 716527b184 Fix missing --readonly handling in NBD server 2026-06-20 11:52:29 +03:00
Vitaliy Filippov 8c0486bd76 Fix bounds check in kv_db (fix invalid input handling?) 2026-06-20 11:43:33 +03:00
Vitaliy Filippov d2cf271f64 Track in_flight for sync & delete in vitastor-cli merge 2026-06-20 11:43:33 +03:00
Vitaliy Filippov d93b488e32 Just in case - OP_SYNC cannot fail but handle its error in vitastor-cli dd 2026-06-20 11:43:33 +03:00
Vitaliy Filippov fbffec5abb Add a regression test for the last 2 fixed bugs 2026-06-20 11:18:09 +03:00
Vitaliy Filippov b6eb8f2055 Add missing sqe retries on metadata read 2026-06-20 11:15:38 +03:00
Vitaliy Filippov e373ea2163 Fix freeing of dyn_data and temp metadata block buffers on read cancel
A regression test would also be fun
2026-06-20 02:10:11 +03:00
Vitaliy Filippov 5e12b4a1a5 Fix OOB read in old store checksum read if block count exceeds IOV_MAX (almost unreachable)
May be fun to write a regression test for it :) IOV_MAX is 1024, so it requires at least a 4 MB object...
2026-06-20 02:03:45 +03:00
Vitaliy Filippov 78b067566f Do not use std::stoull as it may throw 2026-06-20 01:54:49 +03:00
Vitaliy Filippov cf3abdb9e3 Clear autosync_timer when autosync_interval is set to 0
Clear other timers similarly (but their interval can't be 0)
2026-06-20 01:50:44 +03:00
Vitaliy Filippov 826b35b369 Do not try to erase_double_claim entries during iteration 2026-06-19 22:05:44 +03:00
Vitaliy Filippov cdc730314b Fix EC PGs hanging in REPEERING when flush error is last in the batch 2026-06-19 01:49:26 +03:00
Vitaliy Filippov 5248d7f324 Fix 2 bugs in NFS-RDMA allocator, add a test for both
1) alloc() could add the region start into freelist instead of its free part
2) free() was merging freed buffers incorrectly both forward and backward
2026-06-19 02:08:41 +03:00
Vitaliy Filippov 4161f0bd01 Remove doubtful metadata block write logic from blockstore_stable on metadata ENOSPC 2026-06-18 00:55:57 +03:00
Vitaliy Filippov 5627977a9b Limit pg_size by 64 2026-06-18 00:24:48 +03:00
Vitaliy Filippov ec9cfa76c1 Fix to_compact_count tracking on load in the new store 2026-06-18 00:21:03 +03:00
Vitaliy Filippov 0c88884576 Fix incorrect batch big_write fsync logic in the new store 2026-06-18 00:04:15 +03:00
Vitaliy Filippov ba7637d9ad Add a test for incorrect batch big_write fsync logic in the new store 2026-06-18 00:04:15 +03:00
Vitaliy Filippov ac1025c7a5 Fix read small_write.len before init 2026-06-17 17:41:51 +03:00
Vitaliy Filippov 2a81cef78a Fix crash on EIO retry in chained read 2026-06-17 16:38:20 +03:00
Vitaliy Filippov bcf6a7c7d1 Fix write-back queued items split in the client 2026-06-17 16:15:39 +03:00
Vitaliy Filippov 85c4be3957 Clear lsn on sync retry to prevent possible crash on full io_uring 2026-06-17 15:40:05 +03:00
Vitaliy Filippov cf2ba05e4b Disable chunked resharding -- it may interfere with journal flushing; will be re-enabled after fixes 2026-06-17 15:35:19 +03:00
Vitaliy Filippov 04c2f8d408 Stop gracefully on NBD_CMD_DISC instead of just exit(0) 2026-06-17 12:46:03 +03:00
Vitaliy Filippov 0e528ca8f3 Fix missing start_merge() error check 2026-06-17 12:42:23 +03:00
Vitaliy Filippov c6f733b96a Fix final sync in vitastor-cli rm-data 2026-06-17 12:41:22 +03:00
Vitaliy Filippov 938ac09248 Fix small shared file extend-write potentially reading unallocated memory 2026-06-17 12:36:04 +03:00
Vitaliy Filippov 3becdbf5b9 Add bounds checks to prevent uint32 overflows in NFS/XDR 2026-06-17 12:04:50 +03:00
Vitaliy Filippov 9c49315fdf Fix moving of the last block in vitastor-disk resize 2026-06-17 01:19:12 +03:00
Vitaliy Filippov 430d3cfb6f Add a dump|load test, fix multiple bugs in both new&old dump/load utils
Details:
- New store dump/write-meta didn't use actual metadata parameters from the header
- New store write-meta calculated small entry sizes incorrectly
- Old store write-journal imported entries with checksums incorrectly
- Old store write-meta didn't import block_csums at all (it was using a wrong json key)
2026-06-17 01:19:12 +03:00
Vitaliy Filippov c491db699c Implement bdrv_detach_aio_context & bdrv_attach_aio_context (should fix migration with iothread) 2026-06-17 00:39:28 +03:00
Vitaliy Filippov e9d053e30f Support --io option for vitastor-disk dump-journal/write-journal 2026-06-17 00:39:28 +03:00
Vitaliy Filippov 1be51f903c Check entry sizes in blockstore_heap during loading 2026-06-16 20:34:52 +03:00
Vitaliy Filippov cd51f14a90 Fix incorrect chained_read bitmap recovery 2026-06-15 21:10:44 +03:00
Vitaliy Filippov 9b8107875f Add a regression test for incorrect chained_read bitmap recovery 2026-06-15 21:10:44 +03:00
Vitaliy Filippov ca606570f7 Add request size validation 2026-06-15 01:56:54 +03:00
Vitaliy Filippov 22d094ccc6 Add a regression test to check that parity-less EC writes do not destroy bitmaps 2026-06-15 01:56:54 +03:00
Vitaliy Filippov fd84d84279 Add a regression test for scrub bitmap comparison (EC 3+3, only chunk 1 written and lost) 2026-06-15 01:52:04 +03:00
Vitaliy Filippov af2b1e28e3 Add a regression test for EC bitmap recovery losing a parity chunk 2026-06-15 01:52:04 +03:00
Vitaliy Filippov 9dda449f48 Fix fully-degraded EC writes destroying bitmaps 2026-06-15 01:52:04 +03:00
Vitaliy Filippov e6d4b32629 Also check padded small_write 2026-06-15 01:31:19 +03:00
Vitaliy Filippov 4de22a08e2 Fix partial padded read checksum verification 2026-06-15 01:31:19 +03:00
Vitaliy Filippov a403de46b3 Test partial padded read 2026-06-15 01:31:19 +03:00
Vitaliy Filippov ac20f605f6 Fix scrub bitmap comparison 2026-06-15 01:31:19 +03:00
Vitaliy Filippov 51ecbadb12 Only update bitmaps when writing data in primary subops
Fixes fully degraded EC writes and recovering EC writes corrupting bitmaps
2026-06-15 01:31:19 +03:00
Vitaliy Filippov 6a4627b625 Write a test for store v1 & journal corruption preservation, fix MULTIPLE bugs 2026-06-14 11:23:36 +03:00
Vitaliy Filippov fd2b8b8792 Fix ec_check_combination more correctly (fixes osd_rmw_test_je) 2026-06-13 19:56:06 +03:00
Vitaliy Filippov f3d662bac7 Verify bitmaps during scrub 2026-06-13 19:56:06 +03:00
Vitaliy Filippov 553cc8ef87 Add a test for replicated bitmap scrub 2026-06-13 19:56:06 +03:00
Vitaliy Filippov 67fdf1142b Fix vitastor-disk crash when converting from 0.5.x metadata 2026-06-13 19:56:06 +03:00
Vitaliy Filippov 02f6e564a6 Old store - write header on init only after clearing metadata 2026-06-13 19:56:06 +03:00
Vitaliy Filippov 33d14061d6 New store - write header on init only after clearing metadata 2026-06-13 19:56:06 +03:00
Vitaliy Filippov 677e755e4e Replace VLA with a vector and also fix invalid dangling pointer warnings 2026-06-13 19:56:06 +03:00
Vitaliy Filippov 11a972cbfb Delete pg.peering_state when it is not needed 2026-06-13 19:56:06 +03:00
Vitaliy Filippov 1834743a0e Fix incorrect clearing of LOC_CORRUPTED on the second scrub 2026-06-13 19:56:06 +03:00
Vitaliy Filippov 213f76c66c Add a regression test for LOC_CORRUPTED cleared on second scrub 2026-06-13 19:56:06 +03:00
Vitaliy Filippov 91698404a7 Add a basic OSD test as an example 2026-06-13 19:56:06 +03:00
Vitaliy Filippov 27bd38d95e Allow to create OSD with mocked blockstore and network 2026-06-13 19:56:06 +03:00
Vitaliy Filippov ef0e61be1b Split and mock etcd_state_client_t for testing 2026-06-13 19:56:06 +03:00
Vitaliy Filippov 26fb08d7da Change st_cli to pointer 2026-06-13 19:56:06 +03:00
Vitaliy Filippov 80fa3094b3 Fix cpp-btree disabling asserts everywhere it is included to! 2026-06-13 19:56:06 +03:00
Vitaliy Filippov df931b1e17 Check for short reads in blockstore init 2026-06-13 19:56:06 +03:00
Vitaliy Filippov 906294ae9a Fix ec_check_combination() short tmp_buf allocation 2026-06-13 19:56:05 +03:00
Vitaliy Filippov 15f69719e4 Not an actual bug, clean up fsync conditions for clarity
It could be a bug if data_fd could be equal to journal_fd but not to meta_fd,
but it can't.
2026-06-13 19:55:47 +03:00
Vitaliy Filippov 43aa4cfff6 Fix incorrect deferred freeing of block used by a parallel read in the old store 2026-06-13 19:55:47 +03:00
Vitaliy Filippov 6901227390 Fix cached EC decoding of multiple stripes with ISA-L 2026-06-10 02:07:58 +03:00
Vitaliy Filippov 7f718feaf6 Fix alloc in nfs_portmap 2026-06-10 01:02:30 +03:00
Vitaliy Filippov 236ffbb24e Fix multilist_alloc_t bug (not triggerable in real operation but still a bug) 2026-06-10 01:01:20 +03:00
Vitaliy Filippov 0e300f4c50 Fix incorrect rounding in vitastor-disk trim when --discard_granularity option is passed 2026-06-10 00:38:34 +03:00
Vitaliy Filippov 6d82a3daa3 Fix fill_block_empty_space argument type (ui32 -> ui64) 2026-06-10 00:33:31 +03:00
Vitaliy Filippov c0c01a8e57 Fix writeback detection in ublk server for --image mode 2026-06-10 00:31:48 +03:00
Vitaliy Filippov 334755e912 Fix vitastor-disk resize to smaller size clearing all data O_o, add a test for it 2026-06-10 00:30:42 +03:00
Vitaliy Filippov c16f955a51 Limit the amount of incoming data for NFS clients to prevent choking on memory 2026-06-06 13:56:20 +03:00
Vitaliy Filippov c1dc14f5ee Fix old store no_inode_stats not working 2026-06-05 00:50:53 +03:00
Vitaliy Filippov ec6a70bbd3 Release 3.0.13
- New store fixes:
  - Fix repeated rollback logic
  - Fix crash on rolled back object compaction
  - Fix postpone_load possibly merging different object chains
  - Fix block_csums import in vitastor-disk write-meta
  - Fix header checksum after vitastor-disk write-meta
- Old store fixes:
  - Fix batched fsync possibly skipped by some flush coroutines
- Improve ENOSPC test, fix possible crash on ENOSPC
- Add fsyncs to vitastor-disk prepare
- Fix possible crash on pg_lock check failure in sec_read_bmp
- Fix VitastorFS initialization when local_reads are enabled
2026-05-31 12:22:43 +03:00
Vitaliy Filippov f236ed895a Rename parameter to reflect actual semantics 2026-05-31 12:03:49 +03:00
Vitaliy Filippov 7d70c90196 Save osd_num for locally submitted writes for mark_partial_write (still needs osd unit tests :)) 2026-05-31 11:47:48 +03:00
Vitaliy Filippov 0a04490043 Fix repeated rollbacks in bs_heap 2026-05-31 11:43:56 +03:00
Vitaliy Filippov dce7ffde6f Fix compaction with do_delete 2026-05-31 11:27:19 +03:00
Vitaliy Filippov f6bd1ff0e5 Fix postpone_load in the new blockstore 2026-05-31 10:52:11 +03:00
Vitaliy Filippov ac00a06757 Make fsync batch members actually wait for fsync completion (found by kimi k2.6) 2026-05-30 16:38:02 +03:00
Vitaliy Filippov 155cfb3c73 Add fsyncs to vitastor-disk prepare (cosmetic) 2026-05-30 16:16:31 +03:00
Vitaliy Filippov 126891126a Do not use unneeded uint64 -> unsigned conversion 2026-05-30 16:14:01 +03:00
Vitaliy Filippov 547a394be6 Fix another unused piece of code 2026-05-30 16:09:48 +03:00
Vitaliy Filippov 1a511acead Do not ignore nfs_do_fsync result 2026-05-30 15:58:46 +03:00
Vitaliy Filippov 5576a0d9ff Detect checksums by csum_block_size in vitastor-disk 2026-05-30 02:36:08 +03:00
Vitaliy Filippov 879e9a32d1 Fix unused code in make_cyclic 2026-05-30 02:31:02 +03:00
Vitaliy Filippov 747fd5c121 Add count >= maxn assert 2026-05-30 02:27:39 +03:00
Vitaliy Filippov b4aab7a78e Fix block_csums import in write-meta heap 2026-05-28 02:05:03 +03:00
Vitaliy Filippov de26a995fc Fix possible null deref on pg_lock check failure in sec_read_bmp 2026-05-28 00:58:02 +03:00
Vitaliy Filippov 8418a9ad7b Fix create_root() retries 2026-05-24 17:25:47 +03:00
Vitaliy Filippov 27be4ee2fa Followup to mark_partial_write fix - also free subops in other branch to prevent assert on ENOSPC 2026-05-20 22:04:28 +03:00
179 changed files with 8661 additions and 2492 deletions
+108
View File
@@ -360,6 +360,78 @@ jobs:
echo ""
done
test_dump_load:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: /root/vitastor/tests/test_dump_load.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_dump_load_32k:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: TEST_NAME=32k OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k" OFFSET_ARGS="$OSD_ARGS" /root/vitastor/tests/test_dump_load.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_old_dump_load:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: OLD=1 /root/vitastor/tests/test_dump_load.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_dump_load_old_32k:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: TEST_NAME=old_32k OLD=1 OSD_ARGS="--data_csum_type crc32c --csum_block_size 32k" OFFSET_ARGS="$OSD_ARGS" /root/vitastor/tests/test_dump_load.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_old_interrupted_rebalance:
runs-on: ubuntu-latest
needs: build
@@ -1584,6 +1656,24 @@ jobs:
echo ""
done
test_resize_last:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: /root/vitastor/tests/test_resize_last.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_resize_auto:
runs-on: ubuntu-latest
needs: build
@@ -1620,6 +1710,24 @@ jobs:
echo ""
done
test_old_resize_last:
runs-on: ubuntu-latest
needs: build
container: ${{env.TEST_IMAGE}}:${{github.sha}}
steps:
- name: Run test
id: test
timeout-minutes: 3
run: OLD=1 /root/vitastor/tests/test_resize_last.sh
- name: Print logs
if: always() && steps.test.outcome == 'failure'
run: |
for i in /root/vitastor/testdata/*.log /root/vitastor/testdata/*.txt; do
echo "-------- $i --------"
cat $i
echo ""
done
test_old_resize_auto:
runs-on: ubuntu-latest
needs: build
+1 -1
View File
@@ -2,7 +2,7 @@ cmake_minimum_required(VERSION 2.8...3.30)
project(vitastor)
set(VITASTOR_VERSION "3.0.12")
set(VITASTOR_VERSION "3.0.15")
include(CTest)
+1 -1
View File
@@ -1,4 +1,4 @@
VITASTOR_VERSION ?= v3.0.12
VITASTOR_VERSION ?= v3.0.15
all: build push
+1 -1
View File
@@ -49,7 +49,7 @@ spec:
capabilities:
add: ["SYS_ADMIN"]
allowPrivilegeEscalation: true
image: vitalif/vitastor-csi:v3.0.12
image: vitalif/vitastor-csi:v3.0.15
args:
- "--node=$(NODE_ID)"
- "--endpoint=$(CSI_ENDPOINT)"
+1 -1
View File
@@ -121,7 +121,7 @@ spec:
privileged: true
capabilities:
add: ["SYS_ADMIN"]
image: vitalif/vitastor-csi:v3.0.12
image: vitalif/vitastor-csi:v3.0.15
args:
- "--node=$(NODE_ID)"
- "--endpoint=$(CSI_ENDPOINT)"
+1 -1
View File
@@ -5,7 +5,7 @@ package vitastor
const (
vitastorCSIDriverName = "csi.vitastor.io"
vitastorCSIDriverVersion = "3.0.12"
vitastorCSIDriverVersion = "3.0.15"
)
// Config struct fills the parameters of request or user input
+1 -1
View File
@@ -1,4 +1,4 @@
vitastor (3.0.12-1) unstable; urgency=medium
vitastor (3.0.15-1) unstable; urgency=medium
* Bugfixes
+1
View File
@@ -11,6 +11,7 @@ override_dh_install:
cp -v node-binding/package.json node-binding/index.js node-binding/addon.cc node-binding/addon.h node-binding/client.cc node-binding/client.h debian/tmp/usr/lib/x86_64-linux-gnu/nodejs/vitastor
cp -v node-binding/build/Release/addon.node debian/tmp/usr/lib/x86_64-linux-gnu/nodejs/vitastor/build/Release
dh_install
cd debian/vitastor-mon/usr/lib/vitastor/mon && npm install --production
override_dh_installdeb:
cat debian/fio_version >> debian/vitastor-fio.substvars
-6
View File
@@ -37,12 +37,6 @@ rm -rf a b
echo "dep:fio=$FIO" > debian/fio_version
cd /root/vitastor/packages/vitastor-$REL/vitastor-$VER
mkdir mon/node_modules
cd mon/node_modules
curl -s https://git.yourcmc.ru/vitalif/antietcd/archive/master.tar.gz | tar -zx
curl -s https://git.yourcmc.ru/vitalif/tinyraft/archive/master.tar.gz | tar -zx
cd /root/vitastor/packages/vitastor-$REL
if [[ ( "$REL" = "trixie" || "$REL" = "resolute" ) && -e ../vitastor-bookworm/vitastor_$VER.orig.tar.xz ]]; then
# Fucking shit, archives differ between bookworm (xz 5.4.1) and trixie (xz 5.8.1)
+1 -1
View File
@@ -1,4 +1,4 @@
VITASTOR_VERSION ?= v3.0.12
VITASTOR_VERSION ?= v3.0.15
all: build push
+1 -1
View File
@@ -4,7 +4,7 @@
#
# Desired Vitastor version
VITASTOR_VERSION=v3.0.12
VITASTOR_VERSION=v3.0.15
# Additional arguments for all containers
# For example, you may want to specify a custom logging driver here
+3
View File
@@ -50,6 +50,9 @@ or antietcd_data_dir options). All other antietcd parameters
cluster, cluster_key, persist_filter, stale_read can also be set in
Vitastor configuration with `antietcd_` prefix.
See also: [antietcd_cert](security.en.md#antietcd_cert),
[antietcd_key](security.en.md#antietcd_key) and [etcd_proxy](security.en.md#etcd_proxyurls).
You can dump/load data to or from antietcd using Antietcd `anticli` tool:
```
+3
View File
@@ -50,6 +50,9 @@ antietcd_data_file или antietcd_data_dir). Все остальные пара
node_id, cluster, cluster_key, persist_filter, stale_read также можно задавать
в конфигурации Vitastor с префиксом `antietcd_`.
Смотрите также настройки [antietcd_cert](security.ru.md#antietcd_cert),
[antietcd_key](security.ru.md#antietcd_key) и [etcd_proxy](security.ru.md#etcd_proxyurls).
Вы можете выгружать/загружать данные в или из antietcd с помощью его инструмента
`anticli`:
+183 -27
View File
@@ -10,13 +10,36 @@ These parameters affect your Vitastor installation security and apply to OSDs, m
Most of them can be set in /etc/vitastor/vitastor.conf and in etcd, but don't support online modification.
All certificate and private key parameters mentioned may contain a path to a PEM file or just
a PEM string with certificate or a private key. In the latter case, the string must begin with
"-----BEGIN CERTIFICATE-----" or "-----BEGIN PRIVATE KEY-----".
- [use_perms](#use_perms)
- [cert](#cert)
- [pkey](#pkey)
- [etcd_ca](#etcd_ca)
- [client_ca](#client_ca)
- [osd_ca](#osd_ca)
- [mon_ca](#mon_ca)
- [antietcd_cert](#antietcd_cert)
- [antietcd_key](#antietcd_key)
- [etcd_proxy.urls](#etcd_proxyurls)
- [etcd_proxy.cert](#etcd_proxycert)
- [etcd_proxy.key](#etcd_proxykey)
- [etcd_proxy.ca](#etcd_proxyca)
- [osd_cert](#osd_cert)
- [osd_pkey](#osd_pkey)
- [api_cert](#api_cert)
- [api_pkey](#api_pkey)
- [etcd_client_cert](#etcd_client_cert)
- [etcd_client_key](#etcd_client_key)
- [etcd_ca](#etcd_ca)
- [osd_etcd_client_cert](#osd_etcd_client_cert)
- [osd_etcd_client_key](#osd_etcd_client_key)
- [mon_etcd_client_cert](#mon_etcd_client_cert)
- [mon_etcd_client_key](#mon_etcd_client_key)
- [proto_checksums](#proto_checksums)
- [force_proto_checksums](#force_proto_checksums)
- [max_cipher_pool_size](#max_cipher_pool_size)
- [vault_url](#vault_url)
- [vault_secret_api_path](#vault_secret_api_path)
- [vault_client_cert](#vault_client_cert)
@@ -25,54 +48,195 @@ Most of them can be set in /etc/vitastor/vitastor.conf and in etcd, but don't su
- [vault_timeout_ms](#vault_timeout_ms)
- [vault_error_timeout_sec](#vault_error_timeout_sec)
- [vault_refresh_leeway_sec](#vault_refresh_leeway_sec)
- [max_cipher_pool_size](#max_cipher_pool_size)
## etcd_client_cert
## use_perms
- Type: boolean
- Default: false
Enable client permissions in a Vitastor cluster, including Antietcd built into the Monitor.
Requires configured encryption. Also note that separate Antietcd requires separate configuration
to use permissions (see [security documentation](../intro/security.en.md) for details).
## cert
- Type: string
Client TLS certificate to use for Vitastor client (not OSD and not monitor)
etcd https connections. May be path to a file or just a PEM string with certificate.
In the latter case, string must begin with "-----BEGIN CERTIFICATE-----".
Client certificate of the current Vitastor user. Required for Vitastor protocol encryption.
Must be signed with [client_ca](#client_ca). Also used as the client certificate for etcd/Antietcd
connections by default.
## etcd_client_key
## pkey
- Type: string
Private key for etcd_client_cert (also a file or a PEM string).
Private key of the current Vitastor user.
## etcd_ca
- Type: string
Trusted TLS CA to verify etcd server certificate. May be path to a file,
directory or just a PEM string with certificate.
Trusted TLS CA to verify etcd server certificate. Or just the etcd server's
certificate itself - it's fine to use it for etcd_ca.
## client_ca
- Type: string
Trusted TLS CA to verify Vitastor client certificates.
Mandatory for Vitastor protocol encryption.
## osd_ca
- Type: string
Trusted TLS CA to verify Vitastor OSD certificates. Also mandatory for Vitastor protocol
encryption. Must be different from client_ca. May be equal to osd_cert - different OSDs
don't require separate certificates at the moment because their permissions don't differ.
## mon_ca
- Type: string
Trusted TLS CA to verify Vitastor Monitor certificates. Used only for separate Antietcd,
not required when a monitor built-in Antietcd is used. May be equal to mon_client_etcd_cert.
## antietcd_cert
- Type: string
Server TLS certificate for Antietcd built into the Monitor.
## antietcd_key
- Type: string
Private key for antietcd_cert.
## etcd_proxy.urls
- Type: string or array of strings
etcd URLs for Antietcd etcd proxy mode.
See [Mon as Etcd proxy](../intro/security.en.md#mon-as-etcd-proxy) for details.
## etcd_proxy.cert
- Type: string
Client certificate for Antietcd connections to etcd in proxy mode.
## etcd_proxy.key
- Type: string
Private key for etcd_proxy.cert.
## etcd_proxy.ca
- Type: string
Trusted TLS CA to verify etcd server certificate when connecting to it from Antietcd.
## osd_cert
- Type: string
Vitastor OSD server certificate. Required for Vitastor protocol encryption. May be equal
to [osd_ca](#osd_ca) - all OSDs share the same permission set for now. Also used as the client
certificate for connections from OSD to etcd/Antietcd by default.
## osd_pkey
- Type: string
Private key for osd_cert.
## api_cert
- Type: string
Server TLS certificate for [vitastor-cli serve](../usage/cli.en.md#serve) API server.
## api_pkey
- Type: string
Private key for api_cert.
## etcd_client_cert
- Type: string
Client TLS certificate to use for connections from Vitastor clients to etcd/Antietcd if you don't want
to use the common client certificate [cert](#cert).
## etcd_client_key
- Type: string
Private key for etcd_client_cert.
## osd_etcd_client_cert
- Type: string
Same as [etcd_client_cert](#etcd_client_cert), but only for OSDs.
OSDs, clients and monitors should have different permissions, so they should
use different certificates.
Client TLS certificate to use for connections from Vitastor OSDs to etcd/Antietcd if you don't want
to use the common OSD certificate [osd_cert](#osd_cert).
## osd_etcd_client_key
- Type: string
Same as [etcd_client_key](#etcd_client_key), but only for OSDs.
Private key for osd_etcd_client_cert.
## mon_etcd_client_cert
- Type: string
Same as [etcd_client_cert](#etcd_client_cert), but only for Vitastor monitors.
Client TLS certificate to use for connections from Vitastor Monitors to etcd/Antietcd - required
if you don't use the built-in Antietcd. In case you use it Monitor has direct access to Antietcd data
and doesn't require any connection.
## mon_etcd_client_key
- Type: string
Same as [etcd_client_key](#etcd_client_key), but only for Vitastor monitors.
Private key for mon_etcd_client_cert.
## proto_checksums
- Type: string
- Default: payload
One of "full", "payload", "gcm", "none":
- "full" means calculate and verify transport level checksums from the full message data
including the header - recommended for unencrypted setups.
- "payload" enables checksums only for the actual read/write data, but skips them for message
headers - recommended for encrypted setups because headers are already protected by AES-GCM.
- "gcm" disables checksums and enables AES-GCM encryption of the whole messages including headers
and data - AES-GCM already includes MAC which is actually a stronger checksum. This option is
slower and is only recommended for untrusted networks.
- "none" disables transport level checksums at all.
## force_proto_checksums
- Type: string
To allow older clients to connect to a Vitastor cluster with enabled checksums, Vitastor OSDs
allow clients to downgrade their proto_checksums by default. force_proto_checksums sets the
minimum security level allowed for connecting clients. When encryption is disabled, default
force_proto_checksums is none and clients without checksums are allowed. With enabled
encryption, force_proto_checksums becomes "payload" by default to block unauthenticated data
on the transport level.
## max_cipher_pool_size
- Type: integer
- Default: 256
Maximum number of OpenSSL cipher contexts cached in OSD memory, counted separately
for each cipher and for encryption/decryption. Probably doesn't require modification.
## vault_url
@@ -103,14 +267,14 @@ Vault v1 secret API mount path to use.
- Type: string
Client TLS certificate to use for Vault connections. Just like [etcd_client_cert](#etcd_client_cert),
may be path to a file or just a certificate in PEM string.
Client TLS certificate to use for Vault connections if you don't want to use the common Vitastor
client certificate [cert](#cert) which is also used for Vault connections by default.
## vault_client_key
- Type: string
Private key for vault_client_cert (also a file or a PEM string).
Private key for the vault_client_cert certificate.
## vault_ca
@@ -140,11 +304,3 @@ Time (in seconds) to wait before retrying after receiving an error from Vault.
Extra time (in seconds) before real Vault token lease_timeout to refresh it, just
in case of system clock drift.
## max_cipher_pool_size
- Type: integer
- Default: 256
Maximum number of OpenSSL cipher contexts cached in OSD memory, counted separately
for each cipher and for encryption/decryption. Probably doesn't require modification.
+186 -28
View File
@@ -12,13 +12,36 @@ OSD, мониторами и клиентами.
Большая их часть может задаваться в /etc/vitastor/vitastor.conf и в etcd, но не
поддерживает онлайн-изменение.
Все параметры сертификатов и закрытых ключей могут быть путём к файлу или просто
строкой с сертификатом в формате PEM. В последнем случае строка должна начинаться с
"-----BEGIN CERTIFICATE-----" или "-----BEGIN PRIVATE KEY-----".
- [use_perms](#use_perms)
- [cert](#cert)
- [pkey](#pkey)
- [etcd_ca](#etcd_ca)
- [client_ca](#client_ca)
- [osd_ca](#osd_ca)
- [mon_ca](#mon_ca)
- [antietcd_cert](#antietcd_cert)
- [antietcd_key](#antietcd_key)
- [etcd_proxy.urls](#etcd_proxyurls)
- [etcd_proxy.cert](#etcd_proxycert)
- [etcd_proxy.key](#etcd_proxykey)
- [etcd_proxy.ca](#etcd_proxyca)
- [osd_cert](#osd_cert)
- [osd_pkey](#osd_pkey)
- [api_cert](#api_cert)
- [api_pkey](#api_pkey)
- [etcd_client_cert](#etcd_client_cert)
- [etcd_client_key](#etcd_client_key)
- [etcd_ca](#etcd_ca)
- [osd_etcd_client_cert](#osd_etcd_client_cert)
- [osd_etcd_client_key](#osd_etcd_client_key)
- [mon_etcd_client_cert](#mon_etcd_client_cert)
- [mon_etcd_client_key](#mon_etcd_client_key)
- [proto_checksums](#proto_checksums)
- [force_proto_checksums](#force_proto_checksums)
- [max_cipher_pool_size](#max_cipher_pool_size)
- [vault_url](#vault_url)
- [vault_secret_api_path](#vault_secret_api_path)
- [vault_client_cert](#vault_client_cert)
@@ -27,56 +50,199 @@ OSD, мониторами и клиентами.
- [vault_timeout_ms](#vault_timeout_ms)
- [vault_error_timeout_sec](#vault_error_timeout_sec)
- [vault_refresh_leeway_sec](#vault_refresh_leeway_sec)
- [max_cipher_pool_size](#max_cipher_pool_size)
## etcd_client_cert
## use_perms
- Тип: булево (да/нет)
- Значение по умолчанию: false
Включает клиентские привилегии в кластере Vitastor, в том числе во встроенном в мониторе Antietcd.
Требует настроенного шифрования протокола. Также обратите внимание, что отдельно установленный Antietcd
требует отдельной настройки привилегий (подробности смотрите в [документации безопасности](../intro/security.ru.md)).
## cert
- Тип: строка
Клиентский TLS сертификат для https-подключений к etcd для клиентов Vitastor
(не OSD и не мониторов). Может быть путём к файлу или просто строкой с
сертификатом в формате PEM. В последнем случае строка должна начинаться с
"-----BEGIN CERTIFICATE-----".
Клиентский сертификат текущего пользователя Vitastor. Требуется для шифрования протокола Vitastor.
Должен быть подписан [client_ca](#client_ca). Также по умолчанию используется как клиентский
сертификат для подключения к etcd/Antietcd и Vault.
## etcd_client_key
## pkey
- Тип: строка
Закрытый ключ для сертификата etcd_client_cert (также путь к файлу или PEM строка).
Закрытый ключ текущего пользователя Vitastor.
## etcd_ca
- Тип: строка
Доверенный корневой TLS-сертификат для проверки сертификата сервера etcd.
Может быть путём к файлу, директории или просто строкой с сертификатом в
формате PEM.
Либо же просто сам сертификат сервера etcd - его можно использовать как etcd_ca.
## client_ca
- Тип: строка
Доверенный TLS-сертификат для проверки сертификатов клиентов Vitastor.
Требуется для шифрования протокола Vitastor.
## osd_ca
- Тип: строка
Доверенный TLS-сертификат для проверки сертификатов OSD Vitastor. Также обязателен
для шифрования протокола Vitastor. Должен отличаться от client_ca. Может быть равен
osd_cert - разные OSD не требуют разных сертификатов, потому что на данный момент
привилегии разных OSD никак не отличаются.
## mon_ca
- Тип: строка
Доверенный TLS-сертификат для проверки сертификатов мониторов Vitastor. Используется
только отдельно установленным Antietcd, не требуется при использовании встроенного в монитор
Antietcd. Может быть равен mon_client_etcd_cert.
## antietcd_cert
- Тип: строка
Серверный TLS-сертификат для Antietcd, встроенного в монитор.
## antietcd_key
- Тип: строка
Закрытый ключ для сертификата antietcd_cert.
## etcd_proxy.urls
- Тип: строка или массив строк
Адреса etcd для режима Antietcd etcd-прокси.
Смотрите подробности в разделе [Mon в роли Etcd proxy](../intro/security.ru.md#mon-в-роли-etcd-proxy).
## etcd_proxy.cert
- Тип: строка
Клиентский сертификат для подключений от Antietcd к etcd в режиме прокси.
## etcd_proxy.key
- Тип: строка
Закрытый ключ для сертификата etcd_proxy.cert.
## etcd_proxy.ca
- Тип: строка
Доверенный TLS-сертификат для проверки сертификата сервера etcd при подключениях от Antietcd.
## osd_cert
- Тип: строка
Сертификат сервера Vitastor OSD. Требуется для шифрования протокола Vitastor. Может быть равен
[osd_ca](#osd_ca) - все OSD на данный момент имеют одинаковые привилегии. Также по умолчанию
используется как клиентский сертификат для подключения от OSD к etcd/Antietcd.
## osd_pkey
- Тип: строка
Закрытый ключ для сертификата osd_cert.
## api_cert
- Тип: строка
Серверный TLS-сертификат для API-сервера [vitastor-cli serve](../usage/cli.ru.md#serve).
## api_pkey
- Тип: строка
Закрытый ключ для сертификата api_cert.
## etcd_client_cert
- Тип: строка
Клиентский TLS сертификат для подключений от клиентов Vitastor к etcd/Antietcd, если вы не хотите
использовать общий клиентский сертификат [cert](#cert).
## etcd_client_key
- Тип: строка
Закрытый ключ для сертификата etcd_client_cert.
## osd_etcd_client_cert
- Тип: строка
Аналогично [etcd_client_cert](#etcd_client_cert), но только для OSD.
OSD, клиенты и мониторы должны иметь разные привилегии, поэтому они должны
использовать разные сертификаты.
Клиентский TLS сертификат для подключений от Vitastor OSD к etcd/Antietcd, если вы не хотите
использовать общий сертификат OSD [osd_cert](#osd_cert).
## osd_etcd_client_key
- Тип: строка
Аналогично [etcd_client_key](#etcd_client_key), но только для OSD.
Закрытый ключ для сертификата osd_etcd_client_cert.
## mon_etcd_client_cert
- Тип: строка
Аналогично [etcd_client_cert](#etcd_client_cert), но только для мониторов Vitastor.
Клиентский TLS сертификат для подключений от мониторов Vitastor к etcd/Antietcd - требуется, если
вы не используете встроенный в монитор Antietcd. Если вы используете его, то монитор и так имеет
прямой доступ к данным Antietcd и не требует никаких соединений.
## mon_etcd_client_key
- Тип: строка
Аналогично [etcd_client_key](#etcd_client_key), но только для мониторов Vitastor.
Закрытый ключ для сертификата mon_etcd_client_cert.
## proto_checksums
- Тип: строка
- Значение по умолчанию: payload
Одно из значений "full", "payload", "gcm" и "none":
- "full" означает расчёт и проверку контрольных сумм на транспортном уровне от полных сообщений,
включая их заголовки и данные - рекомендуется для кластеров без шифрования.
- "payload" включает контрольные суммы только для данных сообщений, но пропускает заголовки -
такая настройка рекомендуется для кластеров с включённым шифрованием, потому что в них заголовки
и так защищены шифрованием AES-GCM.
- "gcm" отключает контрольные суммы и включает шифрование полных сообщений включая заголовки и
данные - AES-GCM уже включает в себя MAC, который по сути является криптостойкой контрольной
суммой. Такая настройка медленнее и рекомендуется только для недоверенных сетей.
- "none" полностью отключает контрольные суммы на транспортном уровне.
## force_proto_checksums
- Тип: строка
Чтобы старые клиенты Vitastor могли подключаться к кластеру с включёнными контрольными
суммами, Vitastor OSD по умолчанию разрешают клиентам отключать контрольные суммы
данных (proto_checksums). Настройка force_proto_checksums задаёт минимальный уровень
безопасности, разрешённый для подключающихся клиентов. Когда шифрование отключено,
force_proto_checksums по умолчанию равно none и подключения клиентов без контрольных
сумм разрешаются. При включённом шифровании значение по умолчанию force_proto_checksums
становится "payload", чтобы блокировать подключения с неаутентифицированными данными.
## max_cipher_pool_size
- Тип: целое число
- Значение по умолчанию: 256
Максимальное количество кэшируемых в памяти OSD контекстов шифра OpenSSL, учитываемое
отдельно для каждого шифра и для шифрования и расшифровки. Вряд ли требует изменения.
## vault_url
@@ -106,14 +272,14 @@ OSD, клиенты и мониторы должны иметь разные п
- Тип: строка
Клиентский TLS сертификат для подключений к Vault. Как и [etcd_client_cert](#etcd_client_cert),
может быть путём к файлу или просто PEM-строкой с сертификатом.
Клиентский TLS сертификат для подключений к Vault на тот случай, если вы не хотите использовать
общий сертификат клиента Vitastor [cert](#cert), используемый для подключений к Vault по умолчанию.
## vault_client_key
- Тип: строка
Закрытый ключ для сертификата vault_client_cert (также путь к файлу или PEM строка).
Закрытый ключ для сертификата vault_client_cert.
## vault_ca
@@ -144,11 +310,3 @@ OSD, клиенты и мониторы должны иметь разные п
Зазор времени (в секундах), чтобы обновлять токены Vault чуть раньше их реального
lease_timeout, на случай "ухода" системных часов.
## max_cipher_pool_size
- Тип: целое число
- Значение по умолчанию: 256
Максимальное количество кэшируемых в памяти OSD контекстов шифра OpenSSL, учитываемое
отдельно для каждого шифра и для шифрования и расшифровки. Вряд ли требует изменения.
+1 -1
View File
@@ -64,7 +64,7 @@ for (const file of params_files)
let out = '\n';
for (const c of cfg)
{
out += `\n- [${c.name}](#${c.name})`;
out += `\n- [${c.name}](#${c.name.replace(/\./g, '')})`;
}
for (const c of cfg)
{
+6
View File
@@ -21,6 +21,9 @@
cluster, cluster_key, persist_filter, stale_read can also be set in
Vitastor configuration with `antietcd_` prefix.
See also: [antietcd_cert](security.en.md#antietcd_cert),
[antietcd_key](security.en.md#antietcd_key) and [etcd_proxy](security.en.md#etcd_proxyurls).
You can dump/load data to or from antietcd using Antietcd `anticli` tool:
```
@@ -47,6 +50,9 @@
node_id, cluster, cluster_key, persist_filter, stale_read также можно задавать
в конфигурации Vitastor с префиксом `antietcd_`.
Смотрите также настройки [antietcd_cert](security.ru.md#antietcd_cert),
[antietcd_key](security.ru.md#antietcd_key) и [etcd_proxy](security.ru.md#etcd_proxyurls).
Вы можете выгружать/загружать данные в или из antietcd с помощью его инструмента
`anticli`:
+4
View File
@@ -3,3 +3,7 @@
These parameters affect your Vitastor installation security and apply to OSDs, monitors and clients.
Most of them can be set in /etc/vitastor/vitastor.conf and in etcd, but don't support online modification.
All certificate and private key parameters mentioned may contain a path to a PEM file or just
a PEM string with certificate or a private key. In the latter case, the string must begin with
"-----BEGIN CERTIFICATE-----" or "-----BEGIN PRIVATE KEY-----".
+4
View File
@@ -5,3 +5,7 @@ OSD, мониторами и клиентами.
Большая их часть может задаваться в /etc/vitastor/vitastor.conf и в etcd, но не
поддерживает онлайн-изменение.
Все параметры сертификатов и закрытых ключей могут быть путём к файлу или просто
строкой с сертификатом в формате PEM. В последнем случае строка должна начинаться с
"-----BEGIN CERTIFICATE-----" или "-----BEGIN PRIVATE KEY-----".
+187 -42
View File
@@ -1,49 +1,203 @@
- name: etcd_client_cert
- name: use_perms
type: bool
default: false
info: |
Enable client permissions in a Vitastor cluster, including Antietcd built into the Monitor.
Requires configured encryption. Also note that separate Antietcd requires separate configuration
to use permissions (see [security documentation](../intro/security.en.md) for details).
info_ru: |
Включает клиентские привилегии в кластере Vitastor, в том числе во встроенном в мониторе Antietcd.
Требует настроенного шифрования протокола. Также обратите внимание, что отдельно установленный Antietcd
требует отдельной настройки привилегий (подробности смотрите в [документации безопасности](../intro/security.ru.md)).
- name: cert
type: string
info: |
Client TLS certificate to use for Vitastor client (not OSD and not monitor)
etcd https connections. May be path to a file or just a PEM string with certificate.
In the latter case, string must begin with "-----BEGIN CERTIFICATE-----".
Client certificate of the current Vitastor user. Required for Vitastor protocol encryption.
Must be signed with [client_ca](#client_ca). Also used as the client certificate for etcd/Antietcd
connections by default.
info_ru: |
Клиентский TLS сертификат для https-подключений к etcd для клиентов Vitastor
(не OSD и не мониторов). Может быть путём к файлу или просто строкой с
сертификатом в формате PEM. В последнем случае строка должна начинаться с
"-----BEGIN CERTIFICATE-----".
- name: etcd_client_key
Клиентский сертификат текущего пользователя Vitastor. Требуется для шифрования протокола Vitastor.
Должен быть подписан [client_ca](#client_ca). Также по умолчанию используется как клиентский
сертификат для подключения к etcd/Antietcd и Vault.
- name: pkey
type: string
info: Private key for etcd_client_cert (also a file or a PEM string).
info_ru: Закрытый ключ для сертификата etcd_client_cert (также путь к файлу или PEM строка).
info: Private key of the current Vitastor user.
info_ru: Закрытый ключ текущего пользователя Vitastor.
- name: etcd_ca
type: string
info: |
Trusted TLS CA to verify etcd server certificate. May be path to a file,
directory or just a PEM string with certificate.
Trusted TLS CA to verify etcd server certificate. Or just the etcd server's
certificate itself - it's fine to use it for etcd_ca.
info_ru: |
Доверенный корневой TLS-сертификат для проверки сертификата сервера etcd.
Может быть путём к файлу, директории или просто строкой с сертификатом в
формате PEM.
Либо же просто сам сертификат сервера etcd - его можно использовать как etcd_ca.
- name: client_ca
type: string
info: |
Trusted TLS CA to verify Vitastor client certificates.
Mandatory for Vitastor protocol encryption.
info_ru: |
Доверенный TLS-сертификат для проверки сертификатов клиентов Vitastor.
Требуется для шифрования протокола Vitastor.
- name: osd_ca
type: string
info: |
Trusted TLS CA to verify Vitastor OSD certificates. Also mandatory for Vitastor protocol
encryption. Must be different from client_ca. May be equal to osd_cert - different OSDs
don't require separate certificates at the moment because their permissions don't differ.
info_ru: |
Доверенный TLS-сертификат для проверки сертификатов OSD Vitastor. Также обязателен
для шифрования протокола Vitastor. Должен отличаться от client_ca. Может быть равен
osd_cert - разные OSD не требуют разных сертификатов, потому что на данный момент
привилегии разных OSD никак не отличаются.
- name: mon_ca
type: string
info: |
Trusted TLS CA to verify Vitastor Monitor certificates. Used only for separate Antietcd,
not required when a monitor built-in Antietcd is used. May be equal to mon_client_etcd_cert.
info_ru: |
Доверенный TLS-сертификат для проверки сертификатов мониторов Vitastor. Используется
только отдельно установленным Antietcd, не требуется при использовании встроенного в монитор
Antietcd. Может быть равен mon_client_etcd_cert.
- name: antietcd_cert
type: string
info: Server TLS certificate for Antietcd built into the Monitor.
info_ru: Серверный TLS-сертификат для Antietcd, встроенного в монитор.
- name: antietcd_key
type: string
info: Private key for antietcd_cert.
info_ru: Закрытый ключ для сертификата antietcd_cert.
- name: etcd_proxy.urls
type: string or array of strings
type_ru: строка или массив строк
info: |
etcd URLs for Antietcd etcd proxy mode.
See [Mon as Etcd proxy](../intro/security.en.md#mon-as-etcd-proxy) for details.
info_ru: |
Адреса etcd для режима Antietcd etcd-прокси.
Смотрите подробности в разделе [Mon в роли Etcd proxy](../intro/security.ru.md#mon-в-роли-etcd-proxy).
- name: etcd_proxy.cert
type: string
info: Client certificate for Antietcd connections to etcd in proxy mode.
info_ru: Клиентский сертификат для подключений от Antietcd к etcd в режиме прокси.
- name: etcd_proxy.key
type: string
info: Private key for etcd_proxy.cert.
info_ru: Закрытый ключ для сертификата etcd_proxy.cert.
- name: etcd_proxy.ca
type: string
info: Trusted TLS CA to verify etcd server certificate when connecting to it from Antietcd.
info_ru: Доверенный TLS-сертификат для проверки сертификата сервера etcd при подключениях от Antietcd.
- name: osd_cert
type: string
info: |
Vitastor OSD server certificate. Required for Vitastor protocol encryption. May be equal
to [osd_ca](#osd_ca) - all OSDs share the same permission set for now. Also used as the client
certificate for connections from OSD to etcd/Antietcd by default.
info_ru: |
Сертификат сервера Vitastor OSD. Требуется для шифрования протокола Vitastor. Может быть равен
[osd_ca](#osd_ca) - все OSD на данный момент имеют одинаковые привилегии. Также по умолчанию
используется как клиентский сертификат для подключения от OSD к etcd/Antietcd.
- name: osd_pkey
type: string
info: Private key for osd_cert.
info_ru: Закрытый ключ для сертификата osd_cert.
- name: api_cert
type: string
info: Server TLS certificate for [vitastor-cli serve](../usage/cli.en.md#serve) API server.
info_ru: Серверный TLS-сертификат для API-сервера [vitastor-cli serve](../usage/cli.ru.md#serve).
- name: api_pkey
type: string
info: Private key for api_cert.
info_ru: Закрытый ключ для сертификата api_cert.
- name: etcd_client_cert
type: string
info: |
Client TLS certificate to use for connections from Vitastor clients to etcd/Antietcd if you don't want
to use the common client certificate [cert](#cert).
info_ru: |
Клиентский TLS сертификат для подключений от клиентов Vitastor к etcd/Antietcd, если вы не хотите
использовать общий клиентский сертификат [cert](#cert).
- name: etcd_client_key
type: string
info: Private key for etcd_client_cert.
info_ru: Закрытый ключ для сертификата etcd_client_cert.
- name: osd_etcd_client_cert
type: string
info: |
Same as [etcd_client_cert](#etcd_client_cert), but only for OSDs.
OSDs, clients and monitors should have different permissions, so they should
use different certificates.
Client TLS certificate to use for connections from Vitastor OSDs to etcd/Antietcd if you don't want
to use the common OSD certificate [osd_cert](#osd_cert).
info_ru: |
Аналогично [etcd_client_cert](#etcd_client_cert), но только для OSD.
OSD, клиенты и мониторы должны иметь разные привилегии, поэтому они должны
использовать разные сертификаты.
Клиентский TLS сертификат для подключений от Vitastor OSD к etcd/Antietcd, если вы не хотите
использовать общий сертификат OSD [osd_cert](#osd_cert).
- name: osd_etcd_client_key
type: string
info: Same as [etcd_client_key](#etcd_client_key), but only for OSDs.
info_ru: Аналогично [etcd_client_key](#etcd_client_key), но только для OSD.
info: Private key for osd_etcd_client_cert.
info_ru: Закрытый ключ для сертификата osd_etcd_client_cert.
- name: mon_etcd_client_cert
type: string
info: Same as [etcd_client_cert](#etcd_client_cert), but only for Vitastor monitors.
info_ru: Аналогично [etcd_client_cert](#etcd_client_cert), но только для мониторов Vitastor.
info: |
Client TLS certificate to use for connections from Vitastor Monitors to etcd/Antietcd - required
if you don't use the built-in Antietcd. In case you use it Monitor has direct access to Antietcd data
and doesn't require any connection.
info_ru: |
Клиентский TLS сертификат для подключений от мониторов Vitastor к etcd/Antietcd - требуется, если
вы не используете встроенный в монитор Antietcd. Если вы используете его, то монитор и так имеет
прямой доступ к данным Antietcd и не требует никаких соединений.
- name: mon_etcd_client_key
type: string
info: Same as [etcd_client_key](#etcd_client_key), but only for Vitastor monitors.
info_ru: Аналогично [etcd_client_key](#etcd_client_key), но только для мониторов Vitastor.
info: Private key for mon_etcd_client_cert.
info_ru: Закрытый ключ для сертификата mon_etcd_client_cert.
- name: proto_checksums
type: string
default: payload
info: |
One of "full", "payload", "gcm", "none":
- "full" means calculate and verify transport level checksums from the full message data
including the header - recommended for unencrypted setups.
- "payload" enables checksums only for the actual read/write data, but skips them for message
headers - recommended for encrypted setups because headers are already protected by AES-GCM.
- "gcm" disables checksums and enables AES-GCM encryption of the whole messages including headers
and data - AES-GCM already includes MAC which is actually a stronger checksum. This option is
slower and is only recommended for untrusted networks.
- "none" disables transport level checksums at all.
info_ru: |
Одно из значений "full", "payload", "gcm" и "none":
- "full" означает расчёт и проверку контрольных сумм на транспортном уровне от полных сообщений,
включая их заголовки и данные - рекомендуется для кластеров без шифрования.
- "payload" включает контрольные суммы только для данных сообщений, но пропускает заголовки -
такая настройка рекомендуется для кластеров с включённым шифрованием, потому что в них заголовки
и так защищены шифрованием AES-GCM.
- "gcm" отключает контрольные суммы и включает шифрование полных сообщений включая заголовки и
данные - AES-GCM уже включает в себя MAC, который по сути является криптостойкой контрольной
суммой. Такая настройка медленнее и рекомендуется только для недоверенных сетей.
- "none" полностью отключает контрольные суммы на транспортном уровне.
- name: force_proto_checksums
type: string
info: |
To allow older clients to connect to a Vitastor cluster with enabled checksums, Vitastor OSDs
allow clients to downgrade their proto_checksums by default. force_proto_checksums sets the
minimum security level allowed for connecting clients. When encryption is disabled, default
force_proto_checksums is none and clients without checksums are allowed. With enabled
encryption, force_proto_checksums becomes "payload" by default to block unauthenticated data
on the transport level.
info_ru: |
Чтобы старые клиенты Vitastor могли подключаться к кластеру с включёнными контрольными
суммами, Vitastor OSD по умолчанию разрешают клиентам отключать контрольные суммы
данных (proto_checksums). Настройка force_proto_checksums задаёт минимальный уровень
безопасности, разрешённый для подключающихся клиентов. Когда шифрование отключено,
force_proto_checksums по умолчанию равно none и подключения клиентов без контрольных
сумм разрешаются. При включённом шифровании значение по умолчанию force_proto_checksums
становится "payload", чтобы блокировать подключения с неаутентифицированными данными.
- name: max_cipher_pool_size
type: int
default: 256
info: |
Maximum number of OpenSSL cipher contexts cached in OSD memory, counted separately
for each cipher and for encryption/decryption. Probably doesn't require modification.
info_ru: |
Максимальное количество кэшируемых в памяти OSD контекстов шифра OpenSSL, учитываемое
отдельно для каждого шифра и для шифрования и расшифровки. Вряд ли требует изменения.
- name: vault_url
type: string
info: |
@@ -81,15 +235,15 @@
- name: vault_client_cert
type: string
info: |
Client TLS certificate to use for Vault connections. Just like [etcd_client_cert](#etcd_client_cert),
may be path to a file or just a certificate in PEM string.
Client TLS certificate to use for Vault connections if you don't want to use the common Vitastor
client certificate [cert](#cert) which is also used for Vault connections by default.
info_ru: |
Клиентский TLS сертификат для подключений к Vault. Как и [etcd_client_cert](#etcd_client_cert),
может быть путём к файлу или просто PEM-строкой с сертификатом.
Клиентский TLS сертификат для подключений к Vault на тот случай, если вы не хотите использовать
общий сертификат клиента Vitastor [cert](#cert), используемый для подключений к Vault по умолчанию.
- name: vault_client_key
type: string
info: Private key for vault_client_cert (also a file or a PEM string).
info_ru: Закрытый ключ для сертификата vault_client_cert (также путь к файлу или PEM строка).
info: Private key for the vault_client_cert certificate.
info_ru: Закрытый ключ для сертификата vault_client_cert.
- name: vault_ca
type: string
info: |
@@ -120,12 +274,3 @@
info_ru: |
Зазор времени (в секундах), чтобы обновлять токены Vault чуть раньше их реального
lease_timeout, на случай "ухода" системных часов.
- name: max_cipher_pool_size
type: int
default: 256
info: |
Maximum number of OpenSSL cipher contexts cached in OSD memory, counted separately
for each cipher and for encryption/decryption. Probably doesn't require modification.
info_ru: |
Максимальное количество кэшируемых в памяти OSD контекстов шифра OpenSSL, учитываемое
отдельно для каждого шифра и для шифрования и расшифровки. Вряд ли требует изменения.
+2 -2
View File
@@ -26,9 +26,9 @@ at Vitastor Kubernetes operator: https://github.com/Antilles7227/vitastor-operat
The instruction is very simple.
1. Download a Docker image of the desired version: \
`docker pull vitalif/vitastor:v3.0.12`
`docker pull vitalif/vitastor:v3.0.15`
2. Install scripts to the host system: \
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.12 install.sh`
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.15 install.sh`
3. Reload udev rules: \
`udevadm control --reload-rules`
4. Enable the vitastor-host service: \
+2 -2
View File
@@ -25,9 +25,9 @@ Vitastor можно установить в Docker/Podman. При этом etcd,
Инструкция по установке максимально простая.
1. Скачайте Docker-образ желаемой версии: \
`docker pull vitalif/vitastor:v3.0.12`
`docker pull vitalif/vitastor:v3.0.15`
2. Установите скрипты в хост-систему командой: \
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.12 install.sh`
`docker run --rm -it -v /etc:/host-etc -v /usr/bin:/host-bin vitalif/vitastor:v3.0.15 install.sh`
3. Перезагрузите правила udev: \
`udevadm control --reload-rules`
4. Включите сервис vitastor-host: \
+6 -2
View File
@@ -41,12 +41,16 @@
## Configure monitors
On the monitor hosts:
- Put identical etcd_address into `/etc/vitastor/vitastor.conf`. Example:
- Create minimal configuration in `/etc/vitastor/vitastor.conf`:
```
{
"etcd_address": ["10.200.1.10:2379","10.200.1.11:2379","10.200.1.12:2379"]
"etcd_address": ["http://10.200.1.10:2379","http://10.200.1.11:2379","http://10.200.1.12:2379"],
"osd_network": "10.200.1.0/24",
"use_perms": false
}
```
- Note that you can enable encryption by using `https://` and `use_perms` option.
[Details](security.en.md#quick-setup) about encryption setup with make-etcd.
- Create systemd units for etcd by running: `/usr/lib/vitastor/mon/make-etcd`
Or, if you installed Vitastor in Docker, run `systemctl start vitastor-host; docker exec vitastor make-etcd`.
- Start etcd and monitors: `systemctl enable --now vitastor-etcd vitastor-mon`
+7 -9
View File
@@ -41,25 +41,23 @@
## Настройте мониторы
На хостах, выделенных под мониторы:
- Пропишите одинаковые etcd_address в `/etc/vitastor/vitastor.conf`. Например:
- Создайте минимальную конфигурацию в `/etc/vitastor/vitastor.conf`:
```
{
"etcd_address": ["10.200.1.10:2379","10.200.1.11:2379","10.200.1.12:2379"]
"etcd_address": ["http://10.200.1.10:2379","http://10.200.1.11:2379","http://10.200.1.12:2379"],
"osd_network": "10.200.1.0/24",
"use_perms": false
}
```
- Обратите внимание, что с помощью схемы `https://` и опции `use_perms` можно включить шифрование.
[Подробно](security.ru.md#быстрая-настройка) о настройке шифрования через make-etcd.
- Инициализируйте сервисы etcd, запустив `/usr/lib/vitastor/mon/make-etcd`.\
Либо, если вы установили Vitastor в Docker, запустите `systemctl start vitastor-host; docker exec vitastor make-etcd`.
- Запустите etcd и мониторы: `systemctl enable --now vitastor-etcd vitastor-mon`
## Настройте OSD
- Пропишите etcd_address и [osd_network](../config/network.ru.md#osd_network) в `/etc/vitastor/vitastor.conf`. Например:
```
{
"etcd_address": ["10.200.1.10:2379","10.200.1.11:2379","10.200.1.12:2379"],
"osd_network": "10.200.1.0/24"
}
```
- Создайте/скопируйте с узлов с мониторами файл конфигурации `/etc/vitastor/vitastor.conf`.
- Инициализуйте OSD:
- Только SSD или только HDD: `vitastor-disk prepare /dev/sdXXX [/dev/sdYYY ...]`.
Если вы используете десктопные SSD без конденсаторов, добавьте опцию `--disable_data_fsync off`,
+657
View File
@@ -0,0 +1,657 @@
[Documentation](../../README.md#documentation) → Introduction → Security in Vitastor
-----
[Читать на русском](security.ru.md)
# Security in Vitastor
- [Overview](#overview)
- [Quick setup](#quick-setup)
- Principles of operation
- [etcd transport encryption (TLS)](#etcd-transport-encryption-tls)
- [OSD transport encryption (AES-GCM)](#osd-transport-encryption-aes-gcm)
- [End-to-end image data encryption (AES-XTS)](#end-to-end-image-data-encryption-aes-xts)
- [Certificate-based authentication](#certificate-based-authentication)
- [Users and access rights](#users-and-access-rights)
- [etcd privileges](#etcd-privileges)
- Manual setup
- [Configuring OSD transport encryption](#configuring-osd-transport-encryption)
- etcd/Antietcd setup options
- [Mon with embedded Antietcd](#mon-with-embedded-antietcd)
- [Mon as an Etcd proxy](#mon-as-an-etcd-proxy)
- [Mon with a separate Antietcd Proxy](#mon-with-a-separate-antietcd-proxy)
- [Standalone Antietcd without etcd](#standalone-antietcd-without-etcd)
- [Vault/OpenBao setup](#vaultopenbao-setup)
- [Vault setup example](#vault-setup-example)
- Lists of allowed operations
- [etcd data access rights](#etcd-data-access-rights)
- [OSD data access rights](#osd-data-access-rights)
- [API access rights](#api-access-rights)
- [Encryption performance](#encryption-performance)
## Overview
Starting from version 3.1.0, Vitastor provides full data protection:
control plane protection (etcd), data plane protection (OSDs), and end-to-end data encryption.
- Control plane protection:
- etcd transport encryption (TLS)
- Authentication via client TLS (X.509) certificates
- Access control of clients to etcd data
- Data plane protection:
- Full AES-GCM encryption of OSD transport (similar to TLS, but faster)
- Alternatively, AES-GCM encryption of just operation headers with data checksums using a secret "salt"
- Authentication via client TLS (X.509) certificates
- Access control of clients on the OSD side
- End-to-end encryption:
- Data is encrypted using AES-XTS on the client side, the Vitastor cluster has no access to plaintext data
- AES-XTS keys can be stored in etcd or in an external Vault/OpenBao
All features are optional and disabled in the simplest configuration. By default, only
transport-level data checksums ([proto_checksums](../config/security.en.md#proto_checksums)=payload)
are enabled for clients that support them (>= 3.1.0). For older clients, connections
without data checksums are allowed by default ([force_proto_checksums](../config/security.en.md#force_proto_checksums) is empty).
For a quick setup, jump to the [Quick setup](#quick-setup) section.
Descriptions of all security-related parameters can be found [here](../config/security.en.md).
## Quick setup
For a quick setup, use the `/usr/lib/vitastor/mon/make-etcd` script:
1. Log in to the node where the first monitor and etcd will be located.
2. Create `/etc/vitastor/vitastor.conf` with minimal parameters: etcd_address,
osd_network and, if you want to enable privileges, use_perms (note `https://`
in etcd addresses):
```
{
"etcd_address": ["https://10.0.0.10:2379","https://10.0.0.11:2379","https://10.0.0.12:2379"],
"osd_network": "10.0.0.0/24",
"use_perms": true
}
```
3. Run `/usr/lib/vitastor/mon/make-etcd` without parameters or with the `--antietcd-only`
parameter if you want to initialize the cluster with Antietcd only, without etcd.
4. The script will generate all necessary certificates and offer to copy them to the other
monitor nodes (agree!).
5. Log in to all other monitor nodes and repeat the `/usr/lib/vitastor/mon/make-etcd` call there.
6. If you also have nodes with OSDs only (without monitors), run the following command to
copy only the required configuration to these nodes:
```
/usr/lib/vitastor/mon/make-etcd --copy-to-osd osdnode1,osdnode2,...
```
After that, you can proceed with OSD initialization.
If you want to understand the setup in more detail, read the [Principles of operation](#principles-of-operation)
and [Manual setup](#manual-setup) sections below.
## Principles of operation
### etcd transport encryption (TLS)
Possible setups:
- Without encryption (http)
- With encryption (https)
- With encryption and client certificate authentication. Either the same certificate
used for authentication on the OSD side (`cert`+`pkey` / `osd_cert`+`osd_pkey`)
is used, or a separately specified certificate (`etcd_client_cert`+`etcd_client_key`).
### OSD transport encryption (AES-GCM)
Possible setups:
- Unencrypted transport without checksums: `proto_checksums=none`.
- Unencrypted transport with data checksums: `proto_checksums=payload` (may be omitted,
this is the default value). It's allowed to disable checksums on the client side, or
use an older client that does not support checksums. If you want to block connections
from clients without checksums, use the option `force_proto_checksums=payload`.
- Header-only encryption with data checksums: activated when the options
`cert`, `pkey`, `osd_ca` are set on the client side and `osd_cert`, `osd_pkey`, `osd_ca`, `client_ca`
on the OSD side, with `proto_checksums=payload`. In this mode, disabling checksums on the client
side is forbidden by default, i.e. `force_proto_checksums=payload` is used.
- Full transport encryption of all traffic: same as the previous option, but with `proto_checksums=gcm`.
In this case, clients are by default allowed to downgrade to checksums only, but this
can also be forbidden via `force_proto_checksums=gcm`. This is the slowest setup and
it's only recommended for insecure (public) networks. In particular, full traffic
encryption together with end-to-end AES-XTS image encryption encrypts data twice.
Encryption uses the AES-256-GCM algorithm and a custom simplified key exchange protocol,
fully analogous to TLS 1.3 ECDHE.
### End-to-end image data encryption (AES-XTS)
The Vitastor client supports encrypting each image's data with its own key. In this case,
data is encrypted by the client before sending it to OSDs and OSDs can't see it in plain.
The encryption key can be changed when cloning/creating image snapshots. For example,
you can make a base VM image (say, Debian Linux) unencrypted, but have encrypted client VM
images inheriting from it.
Image encryption keys can be stored in etcd or in an external Vault. In the latter case,
etcd only stores key IDs and Vitastor cluster can't decrypt the data at all. To use
Vault, create an image with the `--enc_key vault:ID` option, specify vault_url and vault_ca
options in the configuration, create accounts for all clients in Vault, and grant them access
to the required v1 secrets.
Once again, if AES-XTS is used together with full traffic encryption (`proto_checksums=gcm`),
image data is encrypted twice — first with AES-XTS, and then with AES-GCM. Use it only if
you are completely paranoid :-).
### Certificate-based authentication
When encryption is enabled, Vitastor clients, OSDs, and monitors authenticate via certificates
for both etcd (Antietcd) and OSD connections.
Separate certificates must be used for OSDs and monitors — either self-signed, or signed
by separate CAs (`osd_ca` and `mon_ca`). All OSDs can use the same certificate, and all
monitors can also use the same certificate, since the privileges of different OSDs or
different monitors do not differ (theoretically, one could differentiate OSD certificates
by pool, but there has been no need for this so far).
Also, a monitor certificate may not be needed at all if Antietcd is embedded into the monitor
itself. In this case, the monitor already has access to all etcd data directly in memory.
### Users and access rights
When transport encryption is disabled, Vitastor operates without access control, i.e.,
any cluster client has full access to both the management layer and the data layer. This
option is suitable for dedicated trusted storage networks.
When OSD transport encryption is enabled (at least for headers), you can enable access
rights by turning on the `use_perms=true` option. When this option is enabled, each user
can perform only the operations that they are permitted, and even OSDs and monitors are
also forbidden from performing "unnecessary" operations.
Each user (or administrator) must have their own certificate signed by a common root
certificate for clients (`client_ca`), with a Common Name equal to the user name.
Privilege settings are stored in etcd. OSDs and monitors don't need user accounts;
they authenticate via separate certificates.
User privileges are stored in etcd data under the keys `/vitastor/config/user/<name>`.
The following is defined per user in this key:
- Type:
- Client (`type=client` or omitted) — can only read and modify explicitly permitted images.
- Administrator (`type=admin`) — can read and modify all images, and also administer the
cluster: view overall statistics and status, create and delete OSDs, etc.
- List of group names the user is a member of.
Images have the following properties:
- Owner (owner) — the user name that is allowed to both read and modify the image
- Owner group (owner_group) — the owner group name
- Reader group (reader_group) — the name of the group of users allowed to read the image
And there is also a property on the pool:
- Creator group (creator_group) — the name of the group of users allowed to create images in the pool
For the list of allowed operations on image data on the OSD side, see the
[OSD data access rights](#osd-data-access-rights) section.
### etcd privileges
etcd privileges are implemented through Antietcd in all modes of operation.
Built-in etcd privileges are not supported due to numerous inconveniences:
- Certificate-based authentication does not work at all in etcd's REST interface,
- Privileges are stored separately from k/v data and cannot participate in transactions,
- Only the administrator (root) can change privileges,
- There is no support for filtering range read responses by privileges.
If etcd is used, Antietcd acts as a filtering proxy and can be embedded in the Vitastor
monitor or run separately. In this case, etcd must allow incoming connections only from
Antietcd, and all other components must connect to Antietcd.
If Antietcd runs as a part of the Vitastor monitor, it is sufficient to enable the option
`use_perms=true` and set the required certificates. If Antietcd is run separately, privileges
have to be enabled separately using Antietcd options. For more details on the setup, see
the [etcd/Antietcd setup options](#etcdantietcd-setup-options) section.
For the list of allowed operations with etcd data, see the
[etcd data access rights](#etcd-data-access-rights) section.
## Manual setup
### Configuring OSD transport encryption
You need 2 certificates: one for OSDs and one for signing all client certificates.
For OSDs, you can use a self-signed certificate (osd_ca.crt) or a separate certificate (osd.crt)
signed by a trusted osd_ca.crt certificate. For clients, you must use separate certificates
signed by a common trusted (client_ca.crt).
Add to the Vitastor configuration on OSD servers:
- use_perms: true
- osd_ca: osd_ca.crt
- client_ca: client_ca.crt
- osd_cert: osd_ca.crt
- osd_pkey: osd_ca.key
On the client side:
- use_perms: true
- cert: client.crt
- pkey: client.key
### etcd/Antietcd setup options
The following configuration options are available:
#### Mon with embedded Antietcd
The simplest option. You need 1 certificate for Antietcd (antietcd.crt), plus root
certificates for OSDs and clients.
Vitastor settings (`/etc/vitastor/vitastor.conf`):
- etcd_address: [ "http://mon1:2379", ... ] (addresses of your monitors with port 2379)
- use_perms: true
- use_antietcd: true
- antietcd_cert: antietcd.crt
- antietcd_key: antietcd.key
- etcd_ca: antietcd.crt
- osd_ca: osd_ca.crt
- client_ca: client_ca.crt
#### Mon as an Etcd proxy
If you want to enable privileges, but stay on etcd, you can use etcd proxy mode.
You will need 2 separate certificates: one for etcd (etcd.crt) and one for antietcd (antietcd.crt).
The etcd client port must be different from the standard 2379 — for example, you can pick 2381.
OSD and client certificates are also needed.
Vitastor settings:
- etcd_address: [ "http://mon1:2379", ... ] (addresses of your monitors with port 2379)
- use_perms: true
- use_antietcd: true
- etcd_proxy:
```
{
"urls": [ "http://mon1:2381", ... ], // addresses of your etcd with port 2381
"cert": "antietcd.crt",
"key": "antietcd.key",
"ca": "etcd.crt"
}
```
- antietcd_cert: antietcd.crt
- antietcd_key: antietcd.key
- etcd_ca: antietcd.crt
- osd_ca: osd_ca.crt
- client_ca: client_ca.crt
etcd command-line options:
```
--advertise-client-urls=https://<ADDRESS>:2381 --listen-client-urls=https://<ADDRESS>:2381 \
--client-cert-auth --cert-file=etcd.crt --key-file=etcd.key --trusted-ca-file=antietcd.crt \
--peer-client-cert-auth --peer-cert-file=etcd.crt --peer-key-file=etcd.key --peer-trusted-ca-file=etcd.crt
```
#### Mon with a separate Antietcd Proxy
If in addition to the previous option you want to offload Antietcd from the Vitastor monitor's
tasks, you can run it separately.
Similar to the previous option, 2 certificates are needed: one for etcd and one for antietcd,
plus separate certificates for clients, OSDs, and monitors will be needed.
Vitastor settings:
- etcd_address: [ "http://mon1:2379", ... ] (addresses of your monitors with port 2379)
- use_perms: true
- use_antietcd: false
- etcd_ca: antietcd.crt
- osd_ca: osd_ca.crt
- client_ca: client_ca.crt
- mon_etcd_client_cert: mon_ca.crt
- mon_etcd_client_key: mon_ca.key
Antietcd command-line options:
```
--port 2379 \
--client_cert_auth 1 --auth_filter vitastor_auth_filter.js --etcd_proxy url1,url2,... \
--cert antietcd.crt --key antietcd.key --ca client_ca.crt --osd_ca osd_ca.crt --mon_ca mon_ca.crt \
--etcd_cert antietcd.crt --etcd_key antietcd.key --etcd_ca etcd.crt
```
etcd command-line options (same as in the previous option):
```
--advertise-client-urls=https://<ADDRESS>:2381 --listen-client-urls=https://<ADDRESS>:2381 \
--client-cert-auth --cert-file=etcd.crt --key-file=etcd.key --trusted-ca-file=antietcd.crt \
--peer-client-cert-auth --peer-cert-file=etcd.crt --peer-key-file=etcd.key --peer-trusted-ca-file=etcd.crt
```
#### Standalone Antietcd without etcd
Same as the previous option, but etcd and its certificate are not needed:
Vitastor settings (same as in the previous option):
- etcd_address: [ "http://mon1:2379", ... ] (addresses of your monitors with port 2379)
- use_perms: true
- use_antietcd: false
- etcd_ca: antietcd.crt
- osd_ca: osd_ca.crt
- client_ca: client_ca.crt
- mon_etcd_client_cert: mon_ca.crt
- mon_etcd_client_key: mon_ca.key
Antietcd command-line options:
```
--port 2379 \
--client_cert_auth 1 --auth_filter vitastor_auth_filter.js \
--persist_filter vitastor_persist_filter.js \
--cert antietcd.crt --key antietcd.key --ca client_ca.crt --osd_ca osd_ca.crt --mon_ca mon_ca.crt
```
### Vault/OpenBao setup
To use Vault, each client that needs to get image keys from Vault needs a Vault account.
Vitastor only supports client certificate-based authentication, so all client certificates
(`cert`+`pkey`) must be registered in Vault, and they must be granted access to the
corresponding secrets (v1 secrets API is supported).
The required format of a Vault secret is a single `key` field as a hexadecimal string.
The AES-256-XTS algorithm is used, so the key length is 64 bytes, i.e., the string must
consist of 128 hexadecimal digits.
To connect to Vault, set the following settings in Vitastor.conf:
- `vault_url` — Vault address (e.g., `https://vault:8200`)
- `vault_ca` — Vault's own certificate
After that, if you create an image (`vitastor-cli create`) with the option `--enc_key vault:<ID>`,
Vitastor clients will first contact Vault to obtain a token at `/v1/auth/cert/login`,
and then request the actual secret from Vault at `/v1/secret/<ID>`.
#### Vault setup example
Step-by-step instructions for setting up a test Vault using OpenBao as an example:
1. If TLS is not yet configured, generate a self-signed TLS certificate for Vault:
```
openssl req -days 3650 -x509 -addext basicConstraints=critical,CA:TRUE,pathlen:1 --addext subjectAltName=DNS:vault \
-new -newkey rsa:4096 -nodes -keyout /etc/openbao/vault.key -out /etc/openbao/vault.crt
```
Configure it in `/etc/openbao/openbao.hcl`:
```
listener "tcp" {
address = "0.0.0.0:8200"
tls_cert_file = "/etc/openbao/vault.crt"
tls_key_file = "/etc/openbao/vault.key"
}
```
And restart OpenBao (`systemctl restart openbao`).
2. Copy Vault's TLS certificate for Vitastor:
```
cp /etc/openbao/vault.crt /etc/vitastor/vault.crt
```
Transfer it to all client nodes and specify it in `/etc/vitastor/vitastor.conf`:
```
{
...
"vault_url": "http://vault:8200",
"vault_ca": "/etc/vitastor/vault.crt"
}
```
3. Check Vault status:
```
bao status -ca-cert /etc/openbao/vault.crt -address=https://vault:8200
```
4. Initialize Vault in test mode from 1 node (with 1 key share):
```
bao operator init -n 1 -t 1 -ca-cert /etc/openbao/vault.crt -address=https://vault:8200
```
5. Unseal Vault:
```
bao operator unseal -ca-cert /etc/openbao/vault.crt -address=https://vault:8200
```
6. Enable certificate-based authentication:
```
bao auth enable -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 cert
```
7. Enable v1 secrets:
```
bao secrets enable -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 -path=secret kv-v1
```
8. Create a test secret:
```
bao kv put -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 secret/vitastor/testimg3 key=$(openssl rand -hex 64)
```
9. Generate a signed certificate for a Vitastor user (on a machine where you have `client_ca.crt` and `client_ca.key`):
```
openssl req -subj '/CN=testimg3' -nodes -new -keyout testimg3.key -out testimg3.csr
openssl x509 -req -days 3650 -CA client_ca.crt -CAkey client_ca.key -CAcreateserial -in testimg3.csr -out testimg3.crt
rm testimg3.csr
```
10. Create a user in Vault and grant it access to the secret:
```
cat >testimg3.policy <<EOF
path "/secret/vitastor/testimg3" {
capabilities = ["read"]
}
EOF
bao policy write -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 testimg3 testimg3.policy
bao write -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 auth/cert/certs/testimg3 \
certificate=@testimg3.crt display_name=testimg3 token_ttl=24h token_policies=testimg3
```
11. Test access to the secret:
```
curl --cacert /etc/vitastor/vault.crt --cert testimg3.crt --key testimg3.key \
--json '{}' https://vault:8200/v1/auth/cert/login
```
A token will be printed, substitute it into the following request:
```
curl --cacert /etc/vitastor/vault.crt --cert testimg3.crt --key testimg3.key \
-H 'X-Vault-Token: <RECEIVED TOKEN>' https://vault:8200/v1/secret/vitastor/testimg3
```
12. Create an image in Vitastor with the given secret (as an administrator or someone who
has the right to create images in your pool):
```
vitastor-cli create -s 100G --enc_key vault:vitastor/testimg3 --owner testimg3 testimg3
```
13. Test access to the image as user testimg3:
```
vitastor-cli --cert testimg3.crt --pkey testimg3.key dd if=/dev/urandom oimg=testimg3 bs=1M count=100
```
## Lists of allowed operations
### etcd data access rights
Below, all key names are given without the common prefix `/vitastor`.
Allowed operations with keys in Antietcd for clients (`type=client`):
- Read-only:
- Always allowed:
- `/config/global`
- `/config/node_placement`
- `/config/pools`
- `/pg/config`
- `/osd/state/*`
- `/pg/state/*`
- `/index/maxid/*`
- For images [readable by the user](#users-and-access-rights):
- `/config/inode/*`
- `/index/image/*`
- `/inode/stats/*`
- Read and write:
- For pools in which the user can create images:
- `/index/maxid/*`
- For images owned by the user:
- `/config/inode/*`
- `/index/image/*`
Allowed operations with keys in Antietcd for administrators (`type=admin`):
- Read:
- `/stats`
- `/mon/*`
- `/pg/*`
- `/pgstats/*`
- `/inode/stats/*`
- `/pool/stats/*`
- Read and write:
- `/config/*`
- `/osd/*`
- `/index/*`
- `/pg/history/*`
Allowed operations with keys in etcd for OSDs:
- Read:
- `/pg/config`
- `/config/*`
- Read and write:
- `/osd/*`
- `/pg/state/*`
- `/pg/history/*`
- `/pgstats/*`
Allowed operations with keys in etcd for monitors:
- Read:
- `/config/*`
- `/osd/*`
- `/pgstats/*`
- Read and write:
- `/pg/config`
- `/stats`
- `/history/last_clean_pgs`
- `/mon/*`
- `/pg/history/*`
- `/inode/stats/*`
- `/pool/stats/*`
### OSD data access rights
When the `use_perms` option and encryption are enabled, OSDs authenticate clients via
certificates and allow each client only what is allowed by the access control model.
Client operations:
- READ — allowed for images the user has read access to.
- WRITE, DELETE, SCRUB — allowed for images the user has write access to.
- SYNC — the operation is not tied to an image and is always allowed.
- DESCRIBE — the operation is allowed only for administrators (used by the commands
`vitastor-cli describe` and `fix`).
- PING — the operation is always allowed.
- SHOW_CONFIG — the operation is always allowed, however, if the client presents
itself as an OSD in it, then it is verified that it uses a certificate signed by `osd_ca`.
- SEC_LIST (listing) — allowed for other OSDs and administrators with any parameters,
and for regular clients only allowed for requests limited to an image the user has
read access to.
Cluster operations — allowed only for other OSDs:
- SEC_READ
- SEC_WRITE
- SEC_WRITE_STABLE
- SEC_SYNC
- SEC_STABILIZE
- SEC_ROLLBACK
- SEC_DELETE
- SEC_READ_BMP
- SEC_LOCK
### API access rights
[vitastor-cli serve](../usage/cli.en.md#serve) also supports client authentication
via certificates. Only certificates signed by `client_ca` are accepted. A separate
certificate `server_cert` with the key `server_pkey` is used as the server certificate.
For `vitastor-cli serve` to work correctly, it itself must use a certificate
(`cert`+`pkey`) of a user with administrator rights (`type=admin`) to access Vitastor.
Regular clients, when accessing the API, are only allowed API operations on images
available to them either for reading (for reads) or for writing (for modification).
All other API calls are allowed only for administrators.
List of allowed API operations:
Clients (users with `type=client`) are allowed the following operations:
- image/list — for images the user can read.
- image/create — for pools in which the user is allowed to create images, or for
creating snapshots of images owned by the user.
- image/delete, image/flatten, image/modify — for images owned by the user.
All other operations are allowed only for administrators (`type=admin`).
## Encryption performance
You may wonder — how fast is all this wonderful encryption?
The answer is — it depends heavily on the CPU. On modern processors (with AVX512 with VAES
support) it is very fast — AES encryption speed can reach 10-20 GB/s and above. This
primarily concerns the CPU of client machines, because end-to-end encryption is performed
entirely on the client, and client uses its signle thread for transport encryption too,
while there are many OSDs on the server side, and it is easier to add resources there.
On older processors, the speed is noticeably worse — for example, on a Xeon E5 v4 it is
only 3 GB/s.
You can evaluate the performance of your processors using the `vitastor-cli cpubench` command.
Example output (💪 AMD EPYC 9575F):
```
$ vitastor-cli cpubench
Vitastor transport encryption benchmark (AES-256-GCM, AES-256-XTS and xxhash3)
Warmup...
No transport encryption, data checksums enabled, e2e unencrypted image
xxhash3 1 M block... 209000 iterations in 2001 ms = 104447.78 MB/s
xxhash3 4 K block... 37000000 iterations in 2022 ms = 71479.35 MB/s
Header encryption with payload checksums, e2e unencrypted image
AES-256-GCM encrypt header + xxhash3 1 M block... 210000 iterations in 2015 ms = 104218.36 MB/s
AES-256-GCM encrypt header + xxhash3 4 K block... 26000000 iterations in 2073 ms = 48993.01 MB/s
Full transport encryption, e2e unencrypted image
AES-256-GCM encrypt header and 1 M block... 54000 iterations in 2000 ms = 27000.00 MB/s
AES-256-GCM encrypt header and 4 K block... 11700000 iterations in 2014 ms = 22692.71 MB/s
No transport encryption, no checksums, e2e encrypted image
AES-256-XTS encrypt 1 M block... 50000 iterations in 2039 ms = 24521.82 MB/s
AES-256-XTS encrypt 4 K block... 12600000 iterations in 2009 ms = 24499.13 MB/s
No transport encryption, e2e encrypted image, data checksums enabled
AES-256-XTS encrypt + xxhash3 1 M block... 40000 iterations in 2013 ms = 19870.84 MB/s
AES-256-XTS encrypt + xxhash3 4 K block... 10200000 iterations in 2011 ms = 19812.90 MB/s
Header encryption with payload checksums, e2e encrypted image
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 1 M block... 40000 iterations in 2014 ms = 19860.97 MB/s
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 4 K block... 8700000 iterations in 2011 ms = 16899.24 MB/s
Full transport encryption, e2e encrypted image
AES-256-XTS + AES-256-GCM encrypt 1 M block... 26000 iterations in 2062 ms = 12609.12 MB/s
AES-256-XTS + AES-256-GCM encrypt 4 K block... 6300000 iterations in 2006 ms = 12267.88 MB/s
```
And here is Xeon E5-2680v4:
```
$ vitastor-cli cpubench
Vitastor transport encryption benchmark (AES-256-GCM, AES-256-XTS and xxhash3)
Warmup...
No transport encryption, data checksums enabled, e2e unencrypted image
xxhash3 1 M block... 62000 iterations in 2021 ms = 30677.88 MB/s
xxhash3 4 K block... 12400000 iterations in 2006 ms = 24146.31 MB/s
Header encryption with payload checksums, e2e unencrypted image
AES-256-GCM encrypt header + xxhash3 1 M block... 62000 iterations in 2027 ms = 30587.07 MB/s
AES-256-GCM encrypt header + xxhash3 4 K block... 6800000 iterations in 2011 ms = 13208.60 MB/s
Full transport encryption, e2e unencrypted image
AES-256-GCM encrypt header and 1 M block... 7000 iterations in 2317 ms = 3021.15 MB/s
AES-256-GCM encrypt header and 4 K block... 1500000 iterations in 2102 ms = 2787.52 MB/s
No transport encryption, no checksums, e2e encrypted image
AES-256-XTS encrypt 1 M block... 7000 iterations in 2317 ms = 3021.15 MB/s
AES-256-XTS encrypt 4 K block... 1600000 iterations in 2088 ms = 2993.30 MB/s
No transport encryption, e2e encrypted image, data checksums enabled
AES-256-XTS encrypt + xxhash3 1 M block... 6000 iterations in 2188 ms = 2742.23 MB/s
AES-256-XTS encrypt + xxhash3 4 K block... 1400000 iterations in 2053 ms = 2663.78 MB/s
Header encryption with payload checksums, e2e encrypted image
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 1 M block... 6000 iterations in 2190 ms = 2739.73 MB/s
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 4 K block... 1300000 iterations in 2101 ms = 2417.00 MB/s
Full transport encryption, e2e encrypted image
AES-256-XTS + AES-256-GCM encrypt 1 M block... 4000 iterations in 2666 ms = 1500.38 MB/s
AES-256-XTS + AES-256-GCM encrypt 4 K block... 800000 iterations in 2113 ms = 1478.94 MB/s
```
+448 -225
View File
@@ -1,56 +1,105 @@
[Документация](../../README-ru.md#документация) → Безопасность
[Документация](../../README-ru.md#документация) → Введение → Безопасность в Vitastor
-----
[Read in English](security.en.md)
# Оглавление
⚠️ Предупреждение: детальное описание настроек безопасности достаточно длинное.
Если не боитесь - читайте [Подробное описание](#подробное-описание).
Если хотите просто быстро настроить Vitastor с шифрованием - читайте начало статьи.
# Безопасность в Vitastor
- [Обзор](#обзор)
- [Быстрая настройка](#быстрая-настройка)
-
- Принципы работы
- [Шифрование соединений с etcd (TLS)](#шифрование-соединений-с-etcd-tls)
- [Шифрование соединений с OSD (AES-GCM)](#шифрование-соединений-с-osd-aes-gcm)
- [Сквозное шифрование данных образов (AES-XTS)](#сквозное-шифрование-данных-образов-aes-xts)
- [Аутентификация по сертификатам](#аутентификация-по-сертификатам)
- [Пользователи и права доступа](#пользователи-и-права-доступа)
- [Привилегии etcd](#привилегии-etcd)
- Ручная настройка
- [Настройка шифрования соединений OSD](#настройка-шифрования-соединений-osd)
- Варианты настройки etcd/Antietcd
- [Mon со встроенным Antietcd](#mon-со-встроенным-antietcd)
- [Mon в роли Etcd proxy](#mon-в-роли-etcd-proxy)
- [Mon с отдельным Antietcd Proxy](#mon-с-отдельным-antietcd-proxy)
- [Отдельный Antietcd без etcd](#отдельный-antietcd-без-etcd)
- [Настройка Vault/OpenBao](#настройка-vaultopenbao)
- [Пример настройки Vault](#пример-настройки-vault)
- Списки разрешённых операций
- [Права доступа к данным etcd](#права-доступа-к-данным-etcd)
- [Права доступа к данным OSD](#права-доступа-к-данным-osd)
- [Права доступа к API](#права-доступа-к-api)
- [Производительность шифрования](#производительность-шифрования)
# Быстрая настройка
## Обзор
Начиная с версии 3.1.0, Vitastor предоставляет полную защиту данных: защиту слоя
управления (etcd), защиту слоя данных (OSD) и сквозное шифрование данных.
- Защита слоя управления:
- Шифрование соединений с etcd (TLS)
- Аутентификация по клиентским TLS (X.509) сертификатам
- Разграничение прав доступа клиентов к данным etcd
- Защита слоя данных:
- Либо полное AES-GCM шифрование соединений с OSD (аналогично TLS, но быстрее)
- Либо шифрование AES-GCM только заголовков команд с контрольными суммами данных с секретной "солью"
- Аутентификация по клиентским TLS (X.509) сертификатам
- Разграничение прав доступа клиентов на стороне OSD
- Сквозное шифрование:
- Данные шифруются AES-XTS на стороне клиента, кластер Vitastor не имеет доступа к открытым данным
- Ключи AES-XTS могут храниться в etcd или во внешнем Vault/OpenBao
# Пользовательские сценарии
Все функции опциональны и в простейшем варианте настройки выключены. По умолчанию включены
только контрольные суммы данных на транспортном уровне ([proto_checksums](../config/security.ru.md#proto_checksums)=payload) для
поддерживающих их клиентов (>= 3.1.0). Для более старых клиентов по умолчанию разрешены
соединения без контрольных сумм данных ([force_proto_checksums](../config/security.ru.md#force_proto_checksums) пусто).
Зачем всё это нужно вам?
Для быстрой настройки перейдите к разделу [Быстрая настройка](#быстрая-настройка).
Описания всех параметров, связанных с безопасностью, читайте [здесь](../config/security.ru.md).
## Быстрая настройка
# Подробное описание
Для быстрой настройки используйте скрипт `/usr/lib/vitastor/mon/make-etcd`:
Начиная с версии 3.1.0, в Vitastor есть следующие функции:
1. Шифрование соединений с etcd (TLS)
2. Шифрование соединений с OSD (AES-GCM) - по выбору либо только заголовков, либо и заголовков, и данных
3. Сквозное шифрование данных образов (AES-XTS)
4. Хранения ключей шифрования AES-XTS во внешнем Vault
5. Контрольных сумм данных на транспортном уровне с секретной "солью"
6. Аутентификация с помощью TLS (X.509) сертификатов и закрытых ключей
7. Разграничение прав доступа клиентов к данным etcd
8. Разграничение прав доступа клиентов к данным самих образов (на стороне OSD)
1. Зайдите на узел, на котором будет располагаться первый монитор и etcd.
2. Создайте там минимальный `/etc/vitastor/vitastor.conf` с параметрами etcd_address,
osd_network и, если хотите включить привилегии - use_perms (обратите внимание на `https://`
в адресах etcd):
```
{
"etcd_address": ["https://10.0.0.10:2379","https://10.0.0.11:2379","https://10.0.0.12:2379"],
"osd_network": "10.0.0.0/24",
"use_perms": true
}
```
3. Запустите `/usr/lib/vitastor/mon/make-etcd` без параметров или с параметром `--antietcd-only`,
если хотите инициализировать кластер только с Antietcd без etcd.
4. Скрипт сгенерирует все необходимые сертификаты и предложит скопировать их на остальные узлы
мониторов (соглашайтесь!).
5. Зайдите на все остальные узлы мониторов и повторите там вызов `/usr/lib/vitastor/mon/make-etcd`.
6. Если у вас будут узлы только с OSD без мониторов, выполните следующую команду, чтобы скопировать
только нужную конфигурацию на эти узлы:
```
/usr/lib/vitastor/mon/make-etcd --copy-to-osd osdnode1,osdnode2,...
```
По умолчанию шифрование, аутентификация и авторизация отключены, но, начиная с 3.1.0,
используются контрольные суммы данных на транспортном уровне (`proto_checksums=payload`).
После этого можете переходить к инициализации OSD.
## Шифрование соединений с etcd (TLS)
Если хотите разобраться в настройке подробнее, читайте далее разделы [Принципы работы](#принципы-работы)
и [Ручная настройка](#ручная-настройка).
## Принципы работы
### Шифрование соединений с etcd (TLS)
Варианты настройки:
- Без шифрования (http)
- С шифрованием (https)
- С клиентским сертификатом, но при выключенной авторизации (`use_auth=false`) - используется
отдельный сертификат и ключ: `etcd_client_cert`, `etcd_client_key`
- С клиентским сертификатом, при включённой аутентификации на уровне OSD - используется общий
сертификат и ключ: для OSD - `osd_cert` и `osd_pkey`, для клиентов - `cert` и `pkey`
- С шифрованием и аутентификацией по клиентским сертификатам. Используется либо тот
же сертификат, что используется для аутентификации на стороне OSD (`cert`+`pkey` / `osd_cert`+`osd_pkey`),
либо отдельно указанный сертификат (`etcd_client_cert`+`etcd_client_key`)
## Шифрование соединений с OSD (AES-GCM)
### Шифрование соединений с OSD (AES-GCM)
Варианты настройки:
- Без шифрования и без контрольных сумм: `proto_checksums=none`.
@@ -65,46 +114,31 @@
отключение контрольных сумм на уровне клиента, то есть используется `force_proto_checksums=payload`.
- С полным шифрованием всего трафика: аналогично прошлому варианту, но с `proto_checksums=gcm`.
Клиенту при этом по умолчанию разрешается понизить уровень защиты до контрольных сумм, но
это тоже можно запретить через `force_proto_checksums=gcm`. Данный вариант не является рекомендуемым,
так как добавлен в первую очередь для возможной поддержки небезопасных (публичных) сетей и
больше всего снижает производительность. В частности, если одновременно использовать полное
шифрование трафика и сквозное шифрование образов AES-XTS, то данные будут шифроваться дважды.
это тоже можно запретить через `force_proto_checksums=gcm`. Данный вариант самый медленный и
рекомендуется только для небезопасных (публичных) сетей. В том числе потому, что при использовании
и полного шифрования трафика, и сквозного шифрования образов AES-XTS, данные шифруются дважды.
Для шифрования используется алгоритм AES-256-GCM и собственный упрощённый протокол согласования
ключей, полностью аналогичный TLS 1.3 ECDHE.
## Сквозное шифрование данных образов (AES-XTS)
### Сквозное шифрование данных образов (AES-XTS)
Клиент Vitastor поддерживает шифрование данных каждого образа своим ключом. В этом случае на OSD
уходят уже зашифрованные данные и сами OSD не видят настоящее содержимое образов. Разные ключи
в том числе могут иметь разные снимки или клоны одного и того же образа. Например, можно сделать
базовый образ ВМ (условный Debian Linux) нешифрованным, но наследовать от него шифрованные образы
клиентских ВМ.
уходят уже зашифрованные данные и сами OSD не видят исходные данные клиента. При этом ключ можно
менять при клонировании/создании снимков образов. Например, можно сделать базовый образ ВМ
(условный Debian Linux) нешифрованным, но наследовать от него шифрованные образы клиентских ВМ.
Ключи шифрования образов могут храниться либо в etcd, либо во внешнем Vault. Во втором случае
в etcd хранятся только ID ключей, а Vitastor вообще не имеет доступа к данным образов. Для
использования Vault нужно создать образ с опцией `--enc_key vault:ID`, а в конфигурации указать
опции:
- vault_url
- vault_ca
- vault_client_cert
- vault_client_key
использования Vault нужно создать образ с опцией `--enc_key vault:ID`, в конфигурации указать
опции vault_url, и vault_ca, создать всем клиентам учётные записи в Vault и дать им доступ
к требуемым секретам v1.
Ещё раз повторимся, что если AES-XTS используется с полным шифрованием трафика (`proto_checksums=gcm`),
то данные образов шифруются дважды - сначала AES-XTS, а потом AES-GCM. Можете использовать,
только если вы совсем параноик :-).
## Производительность шифрования
У вас может возникнуть вопрос - а как быстро всё это прекрасное шифрование работает?
Ответ - скорость сильно зависит от процессора. Складывается она из нескольких вещей:
-
TODO: vitastor-cli bench.
## Аутентификация по сертификатам
### Аутентификация по сертификатам
При включённом шифровании клиенты, OSD и мониторы Vitastor аутентифицируются по сертификатам
как при соединениях с etcd (Antietcd), так и с OSD.
@@ -118,14 +152,24 @@ TODO: vitastor-cli bench.
Также сертификат монитора может быть вообще не нужен, если Antietcd встраивается в сам монитор.
В этом случае монитор и так имеет доступ ко всем данным etcd прямо в памяти.
Каждый клиент должен иметь свой сертификат, подписанный общим корневым сертификатом
для клиентов (`client_ca`). Common Name сертификата должно равняться имени пользователя.
### Пользователи и права доступа
## Модель прав доступа
При отключённом шифровании трафика Vitastor работает без разграничения прав доступа, то есть,
любой клиент кластера имеет полный доступ как к слою управлению, так и к слою данных. Такой
вариант подходит для выделенных доверенных сетей хранения.
При включённом шифровании трафика OSD (хотя бы заголовков) есть возможность задействовать
права доступа, включив опцию `use_perms=true`. При включённой опции каждый пользователь может
выполнять только те операции, которые ему разрешены, и даже OSD и мониторам также запрещены
"лишние" операции.
Каждый пользователь (или администратор) должен иметь свой сертификат, подписанный общим
корневым сертификатом для клиентов (`client_ca`), с Common Name, равным имени пользователя.
Настройки привилегий же хранятся в etcd. Для OSD и мониторов учётные записи не нужны,
они аутентифицируются по отдельным сертификатам.
Привилегии пользователей хранятся в данных etcd в ключах `/vitastor/config/user/<имя>`.
У пользователя есть 2 свойства:
В этом ключе для каждого пользователя задаётся:
- Тип:
- Клиент (`type=client` или не указано) - может читать и модифицировать только явным образом
разрешённые образы.
@@ -133,67 +177,285 @@ TODO: vitastor-cli bench.
кластер: смотреть общую статистику и состояние, создавать и удалять OSD и так далее.
- Список имён групп, членом которых пользователь является.
У образов есть 3 свойства:
У образов есть следующие свойства:
- Владелец (owner) - имя пользователя, которому разрешено и читать, и менять образ
- Группа владельцев (owner_group) - имя группы владельцев
- Группа читатетей (reader_group) - имя группы пользователей, которым разрешено читать образ
- Группа читателей (reader_group) - имя группы пользователей, которым разрешено читать образ
У пулов есть 1 свойство:
И также есть свойство у пула:
- Группа создателей (creator_group) - имя группы пользователей, которым разрешено создавать образы в пуле
## Права доступа к данным etcd
Перечень разрешённых операций с данными образов на стороне OSD смотрите в разделе
[Права доступа к данным OSD](#права-доступа-к-данным-osd).
Привилегии реализуются через Antietcd во всех режимах работы. Если используется etcd, то
Antietcd выступает в роли фильтрующего прокси, при этом он может быть встроен в монитор
Vitastor или запущен отдельно. В этом случае etcd должен разрешать входящие подключения
только от Antietcd, а все остальные компоненты должны соединяться с Antietcd.
### Привилегии etcd
Если же используется Antietcd, то привилегии реализуются в нём самом.
Привилегии etcd реализуются через Antietcd во всех режимах работы.
Если используется встроенный в монитор Antietcd, то привилегии включаются либо параметром
`use_auth: true`, либо, если этот параметр не указан - включается автоматически, если задан
любой из параметров `client_ca`, `osd_ca`, `mon_ca`. При этом монитор требует указания
параметров `client_ca` и `osd_ca`, а если не используется режим проксирования в etcd -
также `antietcd_server_ca`, чтобы Antietcd мог отличать кластерные соединения от клиентских.
Встроенные привилегии etcd не поддерживаются по причине их многочисленных неудобств:
- Аутентификация по сертификатам вообще не работает в REST интерфейсе etcd,
- Привилегии хранятся отдельно от k/v данных и не могут участвовать в транзакциях,
- Менять привилегии может только администратор (root),
- Нет поддержки фильтрации диапазонных ответов чтения по привилегиям.
Если используется отдельно стоящий Antietcd, привилегии нужно включать явным образом.
Если используется etcd, то Antietcd выступает в роли фильтрующего прокси, при этом он
может быть встроен в монитор Vitastor или запущен отдельно. В этом случае etcd должен
разрешать входящие подключения только от Antietcd, а все остальные компоненты должны
соединяться с Antietcd.
Встроенные привилегии etcd не поддерживаются по причине их многочисленных недоработок:
- Аутентификация по сертификатам не работает в REST интерфейсе etcd,
- Привилегии хранятся отдельно от k/v и не могут участвовать в транзакциях,
- Менять привилегии может только администратор (root)
- Нет поддержки фильтрации ответов чтения по привилегиям.
Если Antietcd запускается в составе монитора Vitastor, то достаточно включить опцию
`use_perms=true` и задать нужные сертификаты. Если Antietcd запускается отдельно, то
привилегии нужно включать отдельно опциями Antietcd. Подробнее о настройке смотрите
раздел [Варианты настройки etcd/Antietcd](#варианты-настройки-etcdantietcd).
Подробный список привилегий на ключи в etcd [смотрите ниже](#привилегии-etcd).
Перечень разрешённых операций с данными etcd смотрите в разделе
[Права доступа к данным etcd](#права-доступа-к-данным-etcd).
## Права доступа к данным OSD
## Ручная настройка
Регулируется опцией `use_auth`, либо, если она не указана, включается автоматически,
если используется шифрование, то есть, если заданы опции `osd_ca` и `client_ca`.
### Настройка шифрования соединений OSD
OSD аутентифицирует клиентов по сертификатам и разрешает каждому клиенту только
то, что ему разрешено согласно модели прав доступа.
Вам нужно 2 сертификата: один для OSD и один для подписи сертификатов всех клиентов.
Для OSD можно использовать самоподписанный сертификат (osd_ca.crt) или отдельный сертификат (osd.crt),
подписанный доверенным сертификатом osd_ca.crt. Для клиентов нужно использовать отдельные
сертификаты, подписанные общим доверенным (client_ca.crt).
Подробный список разрешаемых OSD операций [смотрите ниже](#привилегии-osd).
В конфигурацию Vitastor на серверах OSD нужно добавить:
- use_perms: true
- osd_ca: osd_ca.crt
- client_ca: client_ca.crt
- osd_cert: osd_ca.crt
- osd_pkey: osd_ca.key
## Права доступа к API
На стороне клиентов:
- use_perms: true
- cert: client.crt
- pkey: client.key
[vitastor-cli serve](../usage/cli.ru.md#serve) также поддерживает клиентскую
аутентификацию по сертификатам. Принимаются только сертификаты, подписанные
`client_ca`. В качестве серверного сертификата используется отдельный сертификат
`server_cert` с ключом `server_key`.
### Варианты настройки etcd/Antietcd
При этом для корректной работы `vitastor-cli serve` он сам должен использовать
для доступа в Vitastor сертификат (`cert`+`pkey`) пользователя с правами
администратора (`type=admin`).
Доступны следующие варианты настройки:
Обычным клиентам при доступе к API разрешаются только API-операции с образами,
доступными им либо на чтение (для чтения), либо на запись (для модификации).
Все остальные API-вызовы разрешаются только для администраторов.
#### Mon со встроенным Antietcd
Подробный список разрешаемых API операций [смотрите ниже](#привилегии-api).
Самый простой вариант. Вам нужен 1 сертификат для Antietcd (antietcd.crt), плюс
корневые сертификаты для OSD и клиентов.
## Привилегии etcd
Настройки Vitastor (`/etc/vitastor/vitastor.conf`):
- etcd_address: [ "http://mon1:2379", ... ] (адреса ваших мониторов с портом 2379)
- use_perms: true
- use_antietcd: true
- antietcd_cert: antietcd.crt
- antietcd_key: antietcd.key
- etcd_ca: antietcd.crt
- osd_ca: osd_ca.crt
- client_ca: client_ca.crt
#### Mon в роли Etcd proxy
Если вы хотите включить привилегии, но остаться на etcd, можно задействовать режим etcd proxy.
Вам понадобится 2 отдельных сертификата: один для etcd (etcd.crt) и один для antietcd (antietcd.crt).
Клиентский порт etcd должен отличаться от стандартного 2379, например, можно выбрать 2381.
Также нужны сертификаты OSD и клиентов.
Настройки Vitastor:
- etcd_address: [ "http://mon1:2379", ... ] (адреса ваших мониторов с портом 2379)
- use_perms: true
- use_antietcd: true
- etcd_proxy:
```
{
"urls": [ "http://mon1:2381", ... ], // адреса ваших etcd с портом 2381
"cert": "antietcd.crt",
"key": "antietcd.key",
"ca": "etcd.crt"
}
```
- antietcd_cert: antietcd.crt
- antietcd_key: antietcd.key
- etcd_ca: antietcd.crt
- osd_ca: osd_ca.crt
- client_ca: client_ca.crt
Опции командной строки etcd:
```
--advertise-client-urls=https://<АДРЕС>:2381 --listen-client-urls=https://<АДРЕС>:2381 \
--client-cert-auth --cert-file=etcd.crt --key-file=etcd.key --trusted-ca-file=antietcd.crt \
--peer-client-cert-auth --peer-cert-file=etcd.crt --peer-key-file=etcd.key --peer-trusted-ca-file=etcd.crt
```
#### Mon с отдельным Antietcd Proxy
Если в дополнение к предыдущему варианту вы хотите разгрузить Antietcd от задач монитора Vitastor,
можно запустить его отдельно.
Аналогично предыдущему варианту нужно 2 сертификата: один для etcd и один для antietcd, плюс понадобятся
отдельные сертификаты для клиентов, OSD и монитора.
Настройки Vitastor:
- etcd_address: [ "http://mon1:2379", ... ] (адреса ваших мониторов с портом 2379)
- use_perms: true
- use_antietcd: false
- etcd_ca: antietcd.crt
- osd_ca: osd_ca.crt
- client_ca: client_ca.crt
- mon_etcd_client_cert: mon_ca.crt
- mon_etcd_client_key: mon_ca.key
Опции командной строки Antietcd:
```
--port 2379 \
--client_cert_auth 1 --auth_filter vitastor_auth_filter.js --etcd_proxy url1,url2,... \
--cert antietcd.crt --key antietcd.key --ca client_ca.crt --osd_ca osd_ca.crt --mon_ca mon_ca.crt \
--etcd_cert antietcd.crt --etcd_key antietcd.key --etcd_ca etcd.crt
```
Опции командной строки etcd (не отличаются от предыдущего варианта):
```
--advertise-client-urls=https://<АДРЕС>:2381 --listen-client-urls=https://<АДРЕС>:2381 \
--client-cert-auth --cert-file=etcd.crt --key-file=etcd.key --trusted-ca-file=antietcd.crt \
--peer-client-cert-auth --peer-cert-file=etcd.crt --peer-key-file=etcd.key --peer-trusted-ca-file=etcd.crt
```
#### Отдельный Antietcd без etcd
Аналогично предыдущему варианту, но etcd и его сертификат не нужны:
Настройки Vitastor (не отличаются от предыдущего варианта):
- etcd_address: [ "http://mon1:2379", ... ] (адреса ваших мониторов с портом 2379)
- use_perms: true
- use_antietcd: false
- etcd_ca: antietcd.crt
- osd_ca: osd_ca.crt
- client_ca: client_ca.crt
- mon_etcd_client_cert: mon_ca.crt
- mon_etcd_client_key: mon_ca.key
Опции командной строки Antietcd:
```
--port 2379 \
--client_cert_auth 1 --auth_filter vitastor_auth_filter.js \
--persist_filter vitastor_persist_filter.js \
--cert antietcd.crt --key antietcd.key --ca client_ca.crt --osd_ca osd_ca.crt --mon_ca mon_ca.crt
```
### Настройка Vault/OpenBao
Для использования Vault каждому клиенту, который будет получать из Vault ключи
образов, нужна учётная запись в Vault. Vitastor поддерживает только аутентификацию
по клиентским сертификатам, так что все сертификаты клиентов (`cert`+`pkey`) должны
быть зарегистрированы в Vault и им должен быть дан доступ к соответствующим секретам
(поддерживается API секретов v1).
Требуемый формат секрета Vault - одно поле `key` в формате шестнадцатеричной строки.
Используется алгоритм AES-256-XTS, так что длина ключа - 64 байта, то есть строка
должна состоять из 128 шестнадцатеричных цифр.
Для подключения Vault включите следующие настройки в Vitastor.conf:
- `vault_url` - адрес Vault (например, `https://vault:8200`)
- `vault_ca` - сертификат самого Vault
После этого, если создать образ (`vitastor-cli create`) с опцией `--enc_key vault:<ID>`,
то для получения ключа клиенты Vitastor сначала обратятся к Vault для получения токена
по адресу `/v1/auth/cert/login`, а потом запросят из Vault сам секрет по адресу `/v1/secret/<ID>`.
#### Пример настройки Vault
Пошаговая инструкция для настройки тестового Vault на примере OpenBao:
1. Если ещё не настроен TLS, генерируем самоподписанный TLS сертификат для Vault:
```
openssl req -days 3650 -x509 -addext basicConstraints=critical,CA:TRUE,pathlen:1 --addext subjectAltName=DNS:vault \
-new -newkey rsa:4096 -nodes -keyout /etc/openbao/vault.key -out /etc/openbao/vault.crt
```
Настраиваем его в `/etc/openbao/openbao.hcl`:
```
listener "tcp" {
address = "0.0.0.0:8200"
tls_cert_file = "/etc/openbao/vault.crt"
tls_key_file = "/etc/openbao/vault.key"
}
```
И перезапускаем OpenBao (`systemctl restart openbao`).
2. Копируем TLS сертификат Vault для Vitastor:
```
cp /etc/openbao/vault.crt /etc/vitastor/vault.crt
```
Переносим его на все клиентские ноды и прописываем в `/etc/vitastor/vitastor.conf`:
```
{
...
"vault_url": "http://vault:8200",
"vault_ca": "/etc/vitastor/vault.crt"
}
```
3. Проверяем статус Vault:
```
bao status -ca-cert /etc/openbao/vault.crt -address=https://vault:8200
```
4. Инициализируем Vault в тестовом режиме из 1 ноды (с 1 частью ключа):
```
bao operator init -n 1 -t 1 -ca-cert /etc/openbao/vault.crt -address=https://vault:8200
```
5. Разблокируем Vault:
```
bao operator unseal -ca-cert /etc/openbao/vault.crt -address=https://vault:8200
```
6. Включаем аутентификацию по сертификатам:
```
bao auth enable -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 cert
```
7. Включаем секреты v1:
```
bao secrets enable -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 -path=secret kv-v1
```
8. Создаём тестовый секрет:
```
bao kv put -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 secret/vitastor/testimg3 key=$(openssl rand -hex 64)
```
9. Генерируем подписанный сертификат для пользователя Vitastor (там, где у вас есть `client_ca.crt` и `client_ca.key`):
```
openssl req -subj '/CN=testimg3' -nodes -new -keyout testimg3.key -out testimg3.csr
openssl x509 -req -days 3650 -CA client_ca.crt -CAkey client_ca.key -CAcreateserial -in testimg3.csr -out testimg3.crt
rm testimg3.csr
```
10. Создаём пользователя в Vault и даём ему доступ к секрету:
```
cat >testimg3.policy <<EOF
path "/secret/vitastor/testimg3" {
capabilities = ["read"]
}
EOF
bao policy write -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 testimg3 testimg3.policy
bao write -ca-cert /etc/openbao/vault.crt -address=https://vault:8200 auth/cert/certs/testimg3 \
certificate=@testimg3.crt display_name=testimg3 token_ttl=24h token_policies=testimg3
```
11. Тестируем доступ к секрету:
```
curl --cacert /etc/vitastor/vault.crt --cert testimg3.crt --key testimg3.key \
--json '{}' https://vault:8200/v1/auth/cert/login
```
Будет выведен токен, подставляем его в следующий запрос:
```
curl --cacert /etc/vitastor/vault.crt --cert testimg3.crt --key testimg3.key \
-H 'X-Vault-Token: <ПОЛУЧЕННЫЙ ТОКЕН>' https://vault:8200/v1/secret/vitastor/testimg3
```
12. Создаём образ в Vitastor с заданным секретом (от имени администратора или того, кто имеет
право создавать образы в вашем пуле):
```
vitastor-cli create -s 100G --enc_key vault:vitastor/testimg3 --owner testimg3 testimg3
```
13. Тестируем доступ к образу от имени пользователя testimg3:
```
vitastor-cli --cert testimg3.crt --pkey testimg3.key dd if=/dev/urandom oimg=testimg3 bs=1M count=100
```
## Списки разрешённых операций
### Права доступа к данным etcd
Ниже все названия ключей приведены без общего префикса `/vitastor`.
@@ -207,7 +469,7 @@ OSD аутентифицирует клиентов по сертификата
- `/osd/state/*`
- `/pg/state/*`
- `/index/maxid/*`
- Для образов, которые [может читать пользователь](#модель-прав-доступа):
- Для образов, которые [может читать пользователь](#пользователи-и-права-доступа):
- `/config/inode/*`
- `/index/image/*`
- `/inode/stats/*`
@@ -256,7 +518,10 @@ OSD аутентифицирует клиентов по сертификата
- `/inode/stats/*`
- `/pool/stats/*`
## Привилегии OSD
### Права доступа к данным OSD
При включённой опции `use_perms` и шифровании OSD аутентифицирует клиентов по сертификатам
и разрешает каждому клиенту только то, что ему разрешено согласно модели прав доступа.
Клиентские операции:
- READ - разрешено для образов, доступных пользователю на чтение.
@@ -282,7 +547,22 @@ OSD аутентифицирует клиентов по сертификата
- SEC_READ_BMP
- SEC_LOCK
## Привилегии API
### Права доступа к API
[vitastor-cli serve](../usage/cli.ru.md#serve) также поддерживает клиентскую
аутентификацию по сертификатам. Принимаются только сертификаты, подписанные
`client_ca`. В качестве серверного сертификата используется отдельный сертификат
`server_cert` с ключом `server_pkey`.
При этом для корректной работы `vitastor-cli serve` он сам должен использовать
для доступа в Vitastor сертификат (`cert`+`pkey`) пользователя с правами
администратора (`type=admin`).
Обычным клиентам при доступе к API разрешаются только API-операции с образами,
доступными им либо на чтение (для чтения), либо на запись (для модификации).
Все остальные API-вызовы разрешаются только для администраторов.
Список разрешённых операций API:
Клиентам (пользователям с `type=client`) разрешаются операции:
- image/list - для образов, которые пользователь может читать.
@@ -292,148 +572,91 @@ OSD аутентифицирует клиентов по сертификата
Все остальные операции разрешаются только администраторам (`type=admin`).
## Производительность шифрования
У вас может возникнуть вопрос - а как быстро всё это прекрасное шифрование работает?
Ответ - сильно зависит от процессора. На современных процессорах (при наличии AVX512 с VAES)
очень быстро - скорость шифрования AES может составлять 10-20 Гбайт/с и выше. В первую очередь
подразумевается CPU клиентских машин, потому что сквозное шифрование выполняется целиком на
клиенте, а транспортное хоть также и затрагивает OSD, но у клиента поток один, а OSD на стороне
сервера много и добавить там ресурсов легче.
На более старых процессорах скорость заметно хуже, например, на Xeon E5 v4 она составляет
буквально 3 Гбайт/с.
Вы можете оценить производительность своих процессоров с помощью команды `vitastor-cli cpubench`.
Пример вывода (💪 AMD EPYC 9575F):
```
$ vitastor-cli cpubench
Vitastor transport encryption benchmark (AES-256-GCM, AES-256-XTS and xxhash3)
Warmup...
No transport encryption, data checksums enabled, e2e unencrypted image
xxhash3 1 M block... 209000 iterations in 2001 ms = 104447.78 MB/s
xxhash3 4 K block... 37000000 iterations in 2022 ms = 71479.35 MB/s
Header encryption with payload checksums, e2e unencrypted image
AES-256-GCM encrypt header + xxhash3 1 M block... 210000 iterations in 2015 ms = 104218.36 MB/s
AES-256-GCM encrypt header + xxhash3 4 K block... 26000000 iterations in 2073 ms = 48993.01 MB/s
Full transport encryption, e2e unencrypted image
AES-256-GCM encrypt header and 1 M block... 54000 iterations in 2000 ms = 27000.00 MB/s
AES-256-GCM encrypt header and 4 K block... 11700000 iterations in 2014 ms = 22692.71 MB/s
No transport encryption, no checksums, e2e encrypted image
AES-256-XTS encrypt 1 M block... 50000 iterations in 2039 ms = 24521.82 MB/s
AES-256-XTS encrypt 4 K block... 12600000 iterations in 2009 ms = 24499.13 MB/s
No transport encryption, e2e encrypted image, data checksums enabled
AES-256-XTS encrypt + xxhash3 1 M block... 40000 iterations in 2013 ms = 19870.84 MB/s
AES-256-XTS encrypt + xxhash3 4 K block... 10200000 iterations in 2011 ms = 19812.90 MB/s
Header encryption with payload checksums, e2e encrypted image
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 1 M block... 40000 iterations in 2014 ms = 19860.97 MB/s
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 4 K block... 8700000 iterations in 2011 ms = 16899.24 MB/s
Full transport encryption, e2e encrypted image
AES-256-XTS + AES-256-GCM encrypt 1 M block... 26000 iterations in 2062 ms = 12609.12 MB/s
AES-256-XTS + AES-256-GCM encrypt 4 K block... 6300000 iterations in 2006 ms = 12267.88 MB/s
```
А вот Xeon E5-2680v4:
```
$ vitastor-cli cpubench
Vitastor transport encryption benchmark (AES-256-GCM, AES-256-XTS and xxhash3)
Таким образом, доступны следующие варианты настройки:
Warmup...
### Mon в роли Etcd proxy
No transport encryption, data checksums enabled, e2e unencrypted image
xxhash3 1 M block... 62000 iterations in 2021 ms = 30677.88 MB/s
xxhash3 4 K block... 12400000 iterations in 2006 ms = 24146.31 MB/s
Mon
- use_antietcd: true
- etcd_proxy = {
urls: [],
cert = <antietcd.pem>,
key,
ca = <etcd.pem>,
}
- antietcd_cert = antietcd.pem
- antietcd_key
Header encryption with payload checksums, e2e unencrypted image
AES-256-GCM encrypt header + xxhash3 1 M block... 62000 iterations in 2027 ms = 30587.07 MB/s
AES-256-GCM encrypt header + xxhash3 4 K block... 6800000 iterations in 2011 ms = 13208.60 MB/s
etcd
--client-cert-auth --cert-file=etcd.pem --key-file=etcd.key --trusted-ca-file=antietcd.pem \
--peer-client-cert-auth --peer-cert-file=etcd.pem --peer-key-file=etcd.key --peer-trusted-ca-file=etcd.pem
Full transport encryption, e2e unencrypted image
AES-256-GCM encrypt header and 1 M block... 7000 iterations in 2317 ms = 3021.15 MB/s
AES-256-GCM encrypt header and 4 K block... 1500000 iterations in 2102 ms = 2787.52 MB/s
### Mon с отдельным Antietcd Proxy
No transport encryption, no checksums, e2e encrypted image
AES-256-XTS encrypt 1 M block... 7000 iterations in 2317 ms = 3021.15 MB/s
AES-256-XTS encrypt 4 K block... 1600000 iterations in 2088 ms = 2993.30 MB/s
Mon
- use_antietcd: false
- etcd_ca = antietcd.pem
No transport encryption, e2e encrypted image, data checksums enabled
AES-256-XTS encrypt + xxhash3 1 M block... 6000 iterations in 2188 ms = 2742.23 MB/s
AES-256-XTS encrypt + xxhash3 4 K block... 1400000 iterations in 2053 ms = 2663.78 MB/s
Antietcd
--client_cert_auth 1 --auth_filter vitastor_auth_filter.js --etcd_proxy url1,url2,... \
--cert antietcd.pem --key antietcd.key --ca client_ca.pem --osd_ca osd_ca.pem \
--etcd_cert antietcd.pem --etcd_key antietcd.key --etcd_ca etcd.pem
Header encryption with payload checksums, e2e encrypted image
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 1 M block... 6000 iterations in 2190 ms = 2739.73 MB/s
AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3 4 K block... 1300000 iterations in 2101 ms = 2417.00 MB/s
etcd
--client-cert-auth --cert-file=etcd.pem --key-file=etcd.key --trusted-ca-file=antietcd.pem \
--peer-client-cert-auth --peer-cert-file=etcd.pem --peer-key-file=etcd.key --peer-trusted-ca-file=etcd.pem
### Mon со встроенным Antietcd
Mon
- use_antietcd: true
- use_auth: true
- antietcd_cert = antietcd.pem
- antietcd_key
### Отдельный Antietcd
Mon
- use_antietcd: false
- etcd_ca = antietcd.pem
Antietcd
--client_cert_auth 1 --auth_filter vitastor_auth_filter.js
## Варианты настройки
### Настройка по умолчанию
Используются только контрольные суммы данных на транспортном уровне. Соединения с etcd не шифруются.
Аутентификация и авторизация не используется, любой клиент имеет доступ ко всем данным кластера.
Аналог настройки:
- proto_checksums: payload
### Полная защита
Везде
- osd_ca
- client_ca
- etcd_ca = antietcd.pem
OSD
- osd_cert
- osd_pkey
Клиент
- cert
- pkey
### Только защита etcd
- etcd_ca
- etcd_cert
- etcd_key
### antietcd и только защита antietcd
- etcd_ca
- etcd_cert
- etcd_key
- use_antietcd: true
- antietcd_cert = etcd_ca
- antietcd_key
- antietcd_ca = etcd_cert
### Полное шифрование протокола, включая данные
Внимание: если включить этот вариант защиты и при этом
### Только контрольные суммы на транспортном уровне, без шифрования
## Настройка Vault/OpenBao
openssl req -days 3650 -x509 -addext basicConstraints=critical,CA:TRUE,pathlen:1 --addext subjectAltName=DNS:vault \
-new -newkey rsa:4096 -nodes -keyout vault.key -out vault.crt
bao status -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200
bao operator init -n 1 -t 1 -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200
bao operator unseal -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200
bao auth enable -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200 cert
bao secrets enable -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200 -path=secret kv-v1
bao kv put -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200 secret/vitastor/testimg3 key=$(openssl rand -hex 64)
cat >testimg3.policy <<EOF
path "/secret/vitastor/testimg3" {
capabilities = ["read"]
}
EOF
bao policy write -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200 testimg3 testimg3.policy
bao write -ca-cert /etc/openbao/tls/vault.crt -address=https://vault:8200 auth/cert/certs/testimg3 certificate=@testimg3.crt display_name=testimg3 token_ttl=24h token_policies=testimg3
curl --cacert /etc/vitastor/vault.crt --cert testimg3.crt --key testimg3.key --json '{}' https://vault:8200/v1/auth/cert/login
curl --cacert /etc/vitastor/vault.crt --cert testimg3.crt --key testimg3.key -H 'X-Vault-Token: s.Qkrm78BeK7Rqdz5MA3eJZNbu' https://vault:8200/v1/secret/vitastor/testimg3
Full transport encryption, e2e encrypted image
AES-256-XTS + AES-256-GCM encrypt 1 M block... 4000 iterations in 2666 ms = 1500.38 MB/s
AES-256-XTS + AES-256-GCM encrypt 4 K block... 800000 iterations in 2113 ms = 1478.94 MB/s
```
+18 -22
View File
@@ -28,32 +28,38 @@ class AntiEtcdAdapter
is_local['::'] = true;
is_local[''] = true;
// split :, 3 -> <schema>:<//ip>:<port>
const cluster_local = cluster.map(s =>
const selected = [];
for (let i = 0; i < cluster.length; i++)
{
const m = /^https?:\/\/(?:\[(.*)\]|([^\[\:]+))(?::(\d+))?$/.exec(s);
return [ m[2] || m[1], m[3] || 2379 ];
});
const selected = cluster_local.filter(ip => is_local[ip[0]] && (!cfg_port || ip[1] == cfg_port));
const m = /^(https?:\/\/)?(?:\[(.*)\]|([^\[\:]+))(?::(\d+))?$/.exec(cluster[i]);
if (!m)
continue;
const ip = m[3] || m[2];
const port = m[4] || 2379;
if (is_local[ip] && (!cfg_port || port == cfg_port))
selected.push({ idx: i, ip, port });
}
if (selected.length > 1)
{
console.error('More than 1 etcd_address matches local IPs: '+(selected.join(', '))+', please specify port');
console.error('More than 1 etcd_address matches local IPs, please specify port');
process.exit(1);
}
else if (selected.length == 1)
{
const antietcd_config = {
ip: selected[0][0],
port: selected[0][1],
ip: selected[0].ip,
port: selected[0].port,
cert: config.antietcd_cert,
key: config.antietcd_key,
ca: config.antietcd_ca,
data: config.antietcd_data_file || ((config.antietcd_data_dir || '/var/lib/vitastor') + '/mon_'+selected[0][1]+'.json.gz'),
ca: config.client_ca,
data: config.antietcd_data_file || ((config.antietcd_data_dir || '/var/lib/vitastor') + '/mon_'+selected[0].port+'.json.gz'),
persist_filter: vitastor_persist_filter({ vitastor_prefix: config.etcd_prefix || '/vitastor' }),
node_id: selected[0][0]+':'+selected[0][1], // node_id = ip:port
node_id: cluster[selected[0].idx].replace(/^(https?:\/\/)/, ''), // same as in <cluster> below
cluster: (cluster.length == 1 ? null : cluster.reduce((a, c) => { a[c.replace(/^(https?:\/\/)/, '')] = c; return a; }, {})),
cluster_key: (config.etcd_prefix || '/vitastor'),
stale_read: 1,
log_level: 1,
logs: { cluster: true },
};
if (config.etcd_proxy)
{
@@ -72,23 +78,13 @@ class AntiEtcdAdapter
delete antietcd_config.cluster;
delete antietcd_config.cluster_key;
}
const use_auth = config.use_auth || config.use_auth == null && config.client_ca;
if (use_auth)
if (config.use_perms)
{
antietcd_config.client_cert_auth = true;
antietcd_config.auth_filter = vitastor_auth_filter;
antietcd_config.ca = config.client_ca;
antietcd_config.osd_ca = config.osd_ca;
antietcd_config.mon_ca = config.mon_ca;
if (!config.etcd_proxy)
{
antietcd_config.peer_ca = config.antietcd_server_ca;
if (!config.antietcd_server_ca || config.antietcd_server_ca == config.client_ca)
{
console.error('Secure setup requires separate antietcd_server_ca (for signing antietcd server certificates) and client_ca (for signing client certificates)');
process.exit(1);
}
}
}
for (const key in config)
{
+3 -2
View File
@@ -112,9 +112,10 @@ function make_cyclic(pgs, parity_space)
{
if (parity_space > 1)
{
for (const pg in pgs)
for (const id in pgs)
{
for (let i = 1; i < pg.size; i++)
const pg = pgs[id];
for (let i = 1; i < pg.length; i++)
{
const cyclic = [ ...pg.slice(i), ...pg.slice(0, i) ];
pgs['pg_'+cyclic.join('_')] = cyclic;
+2 -2
View File
@@ -1,6 +1,6 @@
{
"name": "vitastor-mon",
"version": "3.0.12",
"version": "3.0.15",
"description": "Vitastor SDS monitor service",
"main": "mon-main.js",
"scripts": {
@@ -9,7 +9,7 @@
"author": "Vitaliy Filippov",
"license": "UNLICENSED",
"dependencies": {
"antietcd": "^1.2.4",
"antietcd": "^1.3.1",
"sprintf-js": "^1.1.2",
"ws": "^7.2.5"
},
+126 -76
View File
@@ -6,38 +6,39 @@
const child_process = require('child_process');
const fs = require('fs');
const os = require('os');
const path = require('path');
const readline = require('readline');
run().catch(e => { console.error(e); process.exit(1); });
const help_text = `Initialize a Vitastor cluster (etcd, vitastor.conf and TLS certificates)
(c) Vitaliy Filippov, 2019+ (MIT)
(c) Vitaliy Filippov, 2026+ (MIT)
USAGE:
1) Create a minimal vitastor.conf with etcd_address, osd_network and (optionally) use_auth.
Example: {"etcd_address":["http://10.0.0.10:2379","http://10.0.0.11:2379","http://10.0.0.12:2379"],"use_auth":false,"osd_network":"10.0.0.0/24"}
Or: {"etcd_address":["https://10.0.0.10:2379","https://10.0.0.11:2379","https://10.0.0.12:2379"],"use_auth":true,"osd_network":"10.0.0.0/24"}
2) Run: ${process.argv[1]} [./vitastor.conf]
1) Create a minimal vitastor.conf with etcd_address, osd_network and (optionally) use_perms.
Non-encrypted: {"etcd_address":["http://10.0.0.10:2379","http://10.0.0.11:2379","http://10.0.0.12:2379"],"osd_network":"10.0.0.0/24"}
Encrypted: {"etcd_address":["https://10.0.0.10:2379","https://10.0.0.11:2379","https://10.0.0.12:2379"],"use_perms":true,"osd_network":"10.0.0.0/24"}
(Note https:// etcd URLs!)
2) Run: ${process.argv[1]} [./vitastor.conf] [--antietcd-only]
You can run it on etcd/monitor nodes or on an external node.
It configures etcd, generates TLS certificates (on the first or external node), copies them
to other etcd/monitor nodes, and updates vitastor.conf with TLS options.
It configures etcd, generates TLS certificates (on the first or external node), copies
them to other etcd/monitor nodes, and updates vitastor.conf with TLS options.
3) If you have OSD-only nodes, run:
${process.argv[1]} --copy-to-osd-node NODE_NAME ./vitastor.conf
It copies vitastor.conf and required TLS certificates to that node.
OPTIONS:
--antietcd-only
disable etcd (proxy or direct mode), use only antietcd
--gen-certs
force certificate generation even if it's not the first node
--no-certs
disable certificate generation
--no-copy
do not copy initial certificates to other nodes
--copy yes|no|ask
copy vitastor.conf and TLS certificates for monitor&etcd to monitor nodes using scp
(default is ask)
--copy-to-osd-node NODE[,NODE2,...]
copy vitastor.conf and TLS certificates required for OSDs to NODES using scp
--copy-to-mon-node NODE[,NODE2,...]
copy vitastor.conf and TLS certificates required for monitor and etcd to NODES using scp
--copy-to-client-node NODE[,NODE2,...]
copy vitastor.conf and TLS certificates required for clients to NODES using scp
copy vitastor.conf and TLS certificates for OSDs to NODES using scp
`;
async function run()
@@ -45,21 +46,35 @@ async function run()
let config_path = '/etc/vitastor/vitastor.conf';
let config_dir = '/etc/vitastor/';
let gen_certs = 'auto';
let copy_initial = true;
let copy_to_osd =
let antietcd_only = false;
let copy = 'ask';
let copy_to_osd = null;
for (let i = 2; i < process.argv.length; i++)
{
const arg = process.argv[i];
if (arg == '-h' || arg == '--help')
{
console.log(help_text);
process.exit(0);
}
else if (arg == '--only-certs')
else if (arg == '--gen-certs')
{
gen_certs = true;
}
else if (arg == '--no-certs')
{
gen_certs = false;
}
else if (arg == '--antietcd-only')
{
antietcd_only = true;
}
else if (arg == '--copy-to-osd-node' && i < process.argv.length-1)
{
i++;
gen_certs =
copy_to_osd = process.argv[i].split(/,/);
}
else if (arg == '--copy')
else if (arg == '--copy' && i < process.argv.length-1)
{
i++;
copy = process.argv[i];
@@ -77,6 +92,7 @@ async function run()
else
{
config_path = arg;
config_dir = path.dirname(arg);
}
}
if (!fs.existsSync(config_path))
@@ -100,18 +116,22 @@ async function run()
port: s[3],
}));
const tls = etcds.filter(e => e.scheme === 'https').length > 0;
const use_auth = tls && config.use_auth;
const use_perms = tls && config.use_perms;
const num = select_local_etcd(etcds);
if (copy_to_osd)
{
copy_to_osd_nodes(copy_to_osd, config_dir, use_perms, antietcd_only);
process.exit(0);
}
if (tls)
{
if (gen_certs === 'yes')
const etcd_ca = config_dir+'/'+path.basename(config.etcd_ca);
if (gen_certs === true)
{
gen_certs = true;
console.log('Certificate generation is requested explicitly, generating');
}
else if (gen_certs === 'no')
else if (gen_certs === false)
{
gen_certs = false;
console.log('Certificate generation is disabled explicitly, skipping');
}
else if (num < 0)
@@ -119,10 +139,10 @@ async function run()
gen_certs = true;
console.log('No matching IPs in etcd_address from '+config_path+', only generating certificates');
}
else if (fs.existsSync("/etc/vitastor/etcd.crt"))
else if (config.etcd_ca && fs.existsSync(etcd_ca))
{
gen_certs = false;
console.log('/etc/vitastor/etcd.crt already exists, assuming certificates are already generated');
console.log(etcd_ca+' already exists, assuming certificates are already generated');
}
else if (num === 0)
{
@@ -131,24 +151,25 @@ async function run()
}
else
{
console.log('This is monitor node '+(num+1)+', /etc/vitastor/etcd.crt does not exist, please copy certificates to this node');
console.log('This is monitor node '+(num+1)+', '+etcd_ca+' does not exist, please copy certificates to this node');
process.exit(1);
}
await write_auth_config(config, config_path, etcds, use_perms, antietcd_only);
if (gen_certs)
{
if (copy === 'ask')
copy = await ask_copy('Copy certificates and vitastor.conf to other nodes after generation?');
copy = (copy === 'y' || copy === 'yes');
await make_certs(dir, copy);
await make_certs(config_dir, copy, etcds, use_perms, antietcd_only);
}
await write_auth_config(config, config_path);
}
if (num < 0)
{
console.log('No matching IPs in etcd_address from '+config_path);
process.exit(tls && gen_certs ? 0 : 1);
}
await configure_etcd();
await configure_etcd(etcds, num, tls, use_perms);
await enable_mon();
process.exit(0);
}
@@ -160,83 +181,98 @@ async function ask_copy(question)
prompt: '> ',
});
let copy;
while (true)
while (copy != 'y' && copy != 'n' && copy != 'yes' && copy != 'no')
{
copy = await new Promise(ok => rl.question(question, ok));
if (copy != 'y' && copy != 'n' && copy != 'yes' && copy != 'no')
if (copy)
console.log('Please type "yes" or "no"');
else
break;
copy = await new Promise(ok => rl.question(question, ok));
}
return copy;
}
async function make_certs(dir, copy)
async function copy_to_osd_nodes(to, dir, use_perms, antietcd_only)
{
const osd_to_copy = [ 'vitastor.conf' ];
if (!antietcd_only && !use_perms)
osd_to_copy.push('etcd_ca.crt');
else
osd_to_copy.push('antietcd_ca.crt');
if (use_perms)
osd_to_copy.push('osd.crt', 'osd.key', 'client_ca.crt');
console.warn('Copying configuration to OSD nodes '+to.join(', '));
for (const node of to)
await system("scp "+dir+osd_to_copy.join(" "+dir)+" root@"+node+":/etc/vitastor/");
}
async function make_certs(dir, copy, etcds, use_perms, antietcd_only)
{
console.log(`-----
Generating certificates in ${dir}
-----
`);
await make_ca("/O=Vitastor etcd CA", dir+"etcd_ca");
await make_signed("/CN=Vitastor etcd", dir+"etcd", dir+"etcd_ca", etcds.map(e => "IP:"+e.ip).join(','));
if (use_auth)
const to_copy = [ 'vitastor.conf' ];
const osd_to_copy = [ 'vitastor.conf' ];
if (!antietcd_only)
{
await make_ca("/O=Vitastor etcd CA", dir+"etcd_ca");
await make_signed("/CN=Vitastor etcd", dir+"etcd", dir+"etcd_ca", etcds.map(e => "IP:"+e.ip).join(','));
to_copy.push('etcd_ca.crt', 'etcd.crt', 'etcd.key');
if (!use_perms)
osd_to_copy.push('etcd_ca.crt');
}
if (use_perms || antietcd_only)
{
await make_ca("/O=Vitastor Antietcd CA", dir+"antietcd_ca");
await make_signed("/CN=Vitastor Antietcd", dir+"antietcd", dir+"antietcd_ca", etcds.map(e => "IP:"+e.ip).join(','));
to_copy.push('antietcd_ca.crt', 'antietcd.crt', 'antietcd.key');
osd_to_copy.push('antietcd_ca.crt');
}
if (use_perms)
{
await make_ca("/CN=Vitastor OSD", dir+"osd");
await make_ca("/O=Vitastor Client CA", dir+"client_ca");
await make_signed("/CN=admin", dir+"admin", dir+"client_ca");
to_copy.push('osd.crt', 'osd.key', 'client_ca.crt');
osd_to_copy.push('osd.crt', 'osd.key', 'client_ca.crt');
}
if (use_auth)
{
console.log(`-----
console.log(`-----
Certificates generated, commands to copy them:
- Monitor+OSD node:
cd ${dir} && scp antietcd_ca.crt antietcd.crt antietcd.key osd.crt osd.key client_ca.crt etcd_ca.crt etcd.crt etcd.key root@NODE:/etc/vitastor/
cd ${dir} && scp ${to_copy.join(' ')} root@NODE:/etc/vitastor/
- Monitor node:
cd ${dir} && scp antietcd_ca.crt antietcd.crt antietcd.key osd.crt client_ca.crt etcd_ca.crt etcd.crt etcd.key root@NODE:/etc/vitastor/
cd ${dir} && scp ${to_copy.filter(f => f != 'osd.key').join(' ')} root@NODE:/etc/vitastor/
- OSD node:
cd ${dir} && scp antietcd_ca.crt osd.crt osd.key client_ca.crt root@NODE:/etc/vitastor/
cd ${dir} && scp ${osd_to_copy.join(' ')} root@NODE:/etc/vitastor/
-----
`);
}
else
{
console.log(`-----
Certificates generated, commands to copy them:
- Monitor node:
cd ${dir} && scp etcd_ca.crt etcd.crt etcd.key root@NODE:/etc/vitastor/
-----
`);
}
if (copy)
{
const to_copy = use_auth
? [ "antietcd_ca.crt", "antietcd.crt", "antietcd.key", "osd.crt", "osd.key", "client_ca.crt", "etcd_ca.crt", "etcd.crt", "etcd.key" ]
: [ "etcd_ca.crt", "etcd.crt", "etcd.key" ];
for (const node of etcds)
{
await system("scp "+dir+to_copy.join(" "+dir)+" root@"+node.ip+"/etc/vitastor/");
await system("scp "+dir+to_copy.join(" "+dir)+" root@"+node.ip+":/etc/vitastor/");
}
}
else
{
console.warn('Certificates generated in /etc/vitastor, please copy them to other nodes');
console.warn('Certificates generated in '+dir+', please copy them to other nodes');
}
}
async function write_auth_config(config, config_path)
async function write_auth_config(config, config_path, etcds, use_perms, antietcd_only)
{
const auth = {};
if (use_auth)
if (use_perms)
{
auth["use_antietcd"] = true;
auth["etcd_proxy"] = {
urls: etcds.map(e => e.ip+':2381'),
cert: "/etc/vitastor/antietcd.crt",
key: "/etc/vitastor/antietcd.key",
ca: "/etc/vitastor/etcd_ca.crt",
};
if (!antietcd_only)
{
auth["etcd_proxy"] = {
urls: etcds.map(e => e.ip+':2381'),
cert: "/etc/vitastor/antietcd.crt",
key: "/etc/vitastor/antietcd.key",
ca: "/etc/vitastor/etcd_ca.crt",
};
}
auth["antietcd_cert"] = "/etc/vitastor/antietcd.crt";
auth["antietcd_key"] = "/etc/vitastor/antietcd.key";
auth["etcd_ca"] = "/etc/vitastor/antietcd_ca.crt";
@@ -249,7 +285,17 @@ async function write_auth_config(config, config_path)
}
else
{
auth["etcd_ca"] = "/etc/vitastor/etcd.crt";
if (antietcd_only)
{
auth["use_antietcd"] = true;
auth["antietcd_cert"] = "/etc/vitastor/antietcd.crt";
auth["antietcd_key"] = "/etc/vitastor/antietcd.key";
auth["etcd_ca"] = "/etc/vitastor/antietcd_ca.crt";
}
else
{
auth["etcd_ca"] = "/etc/vitastor/etcd.crt";
}
}
for (const k in auth)
{
@@ -271,7 +317,7 @@ Updating ${config_path}
fs.writeFileSync(config_path, JSON.stringify(config, 0, 4));
}
async configure_etcd()
async function configure_etcd(etcds, num, tls, use_perms)
{
const in_docker = fs.existsSync("/etc/vitastor/etcd.conf") &&
fs.existsSync("/etc/vitastor/docker.conf");
@@ -288,8 +334,8 @@ async configure_etcd()
const etcd_url = etcds[num].scheme + '://' + etcds[num].addr;
const options = {
name: 'etcd'+etcds[num].ip.replace(/[^0-9a-z_]/ig, '_'),
advertise_client_urls: etcd_url+':'+(use_auth ? 2381 : 2379),
listen_client_urls: etcd_url+':'+(use_auth ? 2381 : 2379),
advertise_client_urls: etcd_url+':'+(use_perms ? 2381 : 2379),
listen_client_urls: etcd_url+':'+(use_perms ? 2381 : 2379),
initial_advertise_peer_urls: etcd_url+':2380',
listen_peer_urls: etcd_url+':2380',
initial_cluster_token: 'vitastor-etcd-1',
@@ -305,14 +351,14 @@ async configure_etcd()
{
options['cert_file'] = '/etc/vitastor/etcd.crt';
options['key_file'] = '/etc/vitastor/etcd.key';
if (use_auth)
if (use_perms)
{
options['client_cert_auth'] = '1';
options['trusted_ca_file'] = '/etc/vitastor/antietcd.crt';
}
options['peer_cert_file'] = '/etc/vitastor/etcd.crt';
options['peer_key_file'] = '/etc/vitastor/etcd.key';
if (use_auth)
if (use_perms)
{
options['peer_client_cert_auth'] = '1';
options['peer_trusted_ca_file'] = '/etc/vitastor/etcd.crt';
@@ -333,8 +379,7 @@ async configure_etcd()
}
await system(`mkdir -p /var/lib/etcd/vitastor`);
fs.writeFileSync(
"/etc/systemd/system/vitastor-etcd.service",
`[Unit]
"/etc/systemd/system/vitastor-etcd.service", `[Unit]
Description=etcd for vitastor
After=network-online.target local-fs.target time-sync.target
Wants=network-online.target local-fs.target time-sync.target
@@ -363,6 +408,11 @@ WantedBy=multi-user.target
await system(`systemctl enable --now vitastor-etcd`);
}
async function enable_mon()
{
await system(`systemctl enable --now vitastor-mon`);
}
function replace_env(text, key, value)
{
let found = false;
+2 -2
View File
@@ -68,10 +68,10 @@ class VitastorAuthFilter
async init()
{
if (!this.cfg.cert || !this.cfg.key || !this.cfg.osd_ca || !this.cfg.etcd_proxy && !this.cfg.peer_ca || !this.cfg.client_cert_auth)
if (!this.cfg.cert || !this.cfg.key || !this.cfg.ca || !this.cfg.osd_ca || !this.cfg.client_cert_auth)
{
throw new Error('Authenticated Vitastor setups require enabled client_cert_auth, cert, key'+
' and separate ca (client CA), osd_ca'+(this.cfg.etcd_proxy ? '' : ', peer_ca')+' and optionally mon_ca');
' and separate ca (client CA), osd_ca and optionally mon_ca');
}
this.osd_ca = await this.antietcd.readPEM(this.cfg.osd_ca);
this.osd_ca_obj = new X509Certificate(this.osd_ca);
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "vitastor",
"version": "3.0.12",
"version": "3.0.15",
"description": "Low-level native bindings to Vitastor client library",
"main": "index.js",
"keywords": [
+45 -10
View File
@@ -366,15 +366,38 @@ sub map_volume
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
my ($vtype, $img_name, $vmid) = $class->parse_volname($volname);
my $name = $img_name;
my $name = $prefix.$img_name;
$name .= '@'.$snapname if $snapname;
my $mapped = run_cli($scfg, [ 'ls' ], binary => '/usr/bin/vitastor-nbd');
my ($kerneldev) = grep { $mapped->{$_}->{image} eq $prefix.$name } keys %$mapped;
return $kerneldev if $kerneldev && -b $kerneldev; # already mapped
my ($kerneldev) = grep {
$mapped->{$_} && $mapped->{$_}->{image} && $mapped->{$_}->{image} eq $name
} keys %$mapped;
$kerneldev = run_cli($scfg, [ 'map', '--image', $prefix.$name ], binary => '/usr/bin/vitastor-nbd', json => 0);
return $kerneldev;
if ($kerneldev && -b $kerneldev)
{
my $size = `/usr/sbin/blockdev --getsize64 $kerneldev`;
return $kerneldev if $size && $size > 0;
}
my $map_out = run_cli($scfg, [ 'map', '--image', $name ], binary => '/usr/bin/vitastor-nbd', json => 0);
$map_out =~ s/^\s+|\s+$//gso;
# Wait until the device is started
for (my $i = 0; $i < 100; $i++)
{
$mapped = run_cli($scfg, [ 'ls' ], binary => '/usr/bin/vitastor-nbd');
($kerneldev) = grep { $mapped->{$_} && $mapped->{$_}->{image} && $mapped->{$_}->{image} eq $name } keys %$mapped;
if ($kerneldev && -b $kerneldev)
{
my $size = `/usr/sbin/blockdev --getsize64 $kerneldev`;
return $kerneldev if $size && $size > 0;
}
select(undef, undef, undef, 0.1);
}
die "Failed to map Vitastor image $name via NBD".
($map_out ? ", vitastor-nbd map returned '$map_out'" : "")."\n";
}
sub unmap_volume
@@ -383,13 +406,19 @@ sub unmap_volume
my $prefix = defined $scfg->{vitastor_prefix} ? $scfg->{vitastor_prefix} : 'pve/';
my ($vtype, $name, $vmid) = $class->parse_volname($volname);
$name = $prefix.$name;
$name .= '@'.$snapname if $snapname;
my $mapped = run_cli($scfg, [ 'ls' ], binary => '/usr/bin/vitastor-nbd');
my ($kerneldev) = grep { $mapped->{$_}->{image} eq $prefix.$name } keys %$mapped;
if ($kerneldev && -b $kerneldev)
my @kerneldevs = grep {
$mapped->{$_} && $mapped->{$_}->{image} && $mapped->{$_}->{image} eq $name
} keys %$mapped;
for my $kerneldev (@kerneldevs)
{
run_cli($scfg, [ 'unmap', $kerneldev ], binary => '/usr/bin/vitastor-nbd', json => 0);
next if !$kerneldev || !-b $kerneldev;
eval { run_cli($scfg, [ 'unmap', $kerneldev ], binary => '/usr/bin/vitastor-nbd', json => 0); };
warn "Failed to unmap Vitastor image $name from $kerneldev: $@" if $@;
}
return 1;
@@ -405,7 +434,13 @@ sub activate_volume
sub deactivate_volume
{
my ($class, $storeid, $scfg, $volname, $snapname, $cache) = @_;
$class->unmap_volume($storeid, $scfg, $volname, $snapname) if $scfg->{vitastor_nbd};
# Even with vitastor_nbd=0, Proxmox may call map_volume() for special
# volumes like tpmstate0 because swtpm needs a local file/block path.
# Therefore, always try to unmap an existing NBD mapping here.
# unmap_volume() is a no-op if the volume is not currently mapped.
$class->unmap_volume($storeid, $scfg, $volname, $snapname);
return 1;
}
+1 -1
View File
@@ -50,7 +50,7 @@ from cinder.volume import configuration
from cinder.volume import driver
from cinder.volume import volume_utils
VITASTOR_VERSION = '3.0.12'
VITASTOR_VERSION = '3.0.15'
LOG = logging.getLogger(__name__)
+2 -2
View File
@@ -1,11 +1,11 @@
Name: vitastor
Version: 3.0.12
Version: 3.0.15
Release: 1%{?dist}
Summary: Vitastor, a fast software-defined clustered block storage
License: Vitastor Network Public License 1.1
URL: https://vitastor.io/
Source0: vitastor-3.0.12.el10.tar.gz
Source0: vitastor-3.0.15.el10.tar.gz
BuildRequires: gperftools-devel
BuildRequires: gcc-c++
+2 -2
View File
@@ -1,11 +1,11 @@
Name: vitastor
Version: 3.0.12
Version: 3.0.15
Release: 1%{?dist}
Summary: Vitastor, a fast software-defined clustered block storage
License: Vitastor Network Public License 1.1
URL: https://vitastor.io/
Source0: vitastor-3.0.12.el7.tar.gz
Source0: vitastor-3.0.15.el7.tar.gz
BuildRequires: gperftools-devel
BuildRequires: devtoolset-9-gcc-c++
+2 -2
View File
@@ -1,11 +1,11 @@
Name: vitastor
Version: 3.0.12
Version: 3.0.15
Release: 1%{?dist}
Summary: Vitastor, a fast software-defined clustered block storage
License: Vitastor Network Public License 1.1
URL: https://vitastor.io/
Source0: vitastor-3.0.12.el8.tar.gz
Source0: vitastor-3.0.15.el8.tar.gz
BuildRequires: gperftools-devel
BuildRequires: gcc-toolset-9-gcc-c++
+2 -2
View File
@@ -1,11 +1,11 @@
Name: vitastor
Version: 3.0.12
Version: 3.0.15
Release: 1%{?dist}
Summary: Vitastor, a fast software-defined clustered block storage
License: Vitastor Network Public License 1.1
URL: https://vitastor.io/
Source0: vitastor-3.0.12.el9.tar.gz
Source0: vitastor-3.0.15.el9.tar.gz
BuildRequires: gperftools-devel
BuildRequires: gcc-c++
+1 -1
View File
@@ -20,7 +20,7 @@ if("${CMAKE_INSTALL_PREFIX}" MATCHES "^/usr/local/?$")
endif()
set(ENABLE_COVERAGE false CACHE BOOL "Enable code coverage")
add_definitions(-DVITASTOR_VERSION="3.0.12")
add_definitions(-DVITASTOR_VERSION="3.0.15")
add_definitions(-D_GNU_SOURCE -D_LARGEFILE64_SOURCE -D_FILE_OFFSET_BITS=64 -Wall -Wno-sign-compare -Wno-comment -Wno-parentheses -Wno-pointer-arith -fdiagnostics-color=always -fno-omit-frame-pointer -fvisibility=hidden -I ${CMAKE_SOURCE_DIR}/src)
add_link_options(-fno-omit-frame-pointer)
if (${WITH_ASAN})
+14 -10
View File
@@ -521,7 +521,7 @@ void blockstore_disk_t::close_all()
// Sadly DISCARD only works through ioctl(), but it seems to always block the device queue,
// so it's not a big deal that we can only run it synchronously.
int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_free)
int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_used)
{
if (mock_mode)
{
@@ -532,7 +532,7 @@ int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_free)
uint64_t discarded = 0;
for (; i <= block_count; i++)
{
if (i >= block_count || is_free(i))
if (i >= block_count || is_used(i))
{
if (i > j && (i-j)*data_block_size >= min_discard_size)
{
@@ -545,17 +545,21 @@ int blockstore_disk_t::trim_data(std::function<bool(uint64_t)> is_free)
if (range[0] % discard_granularity)
range[0] = range[0] + discard_granularity - (range[0] % discard_granularity);
if (range[0] >= range[1])
continue;
range[1] -= range[0];
range[1] = 0;
else
range[1] -= range[0];
}
r = ioctl(data_fd, BLKDISCARD, &range);
if (r != 0)
if (range[1] > 0)
{
fprintf(stderr, "Failed to execute BLKDISCARD %ju+%ju on %s: %s (code %d)\n",
range[0], range[1], data_device.c_str(), strerror(-r), r);
return -errno;
r = ioctl(data_fd, BLKDISCARD, &range);
if (r != 0)
{
fprintf(stderr, "Failed to execute BLKDISCARD %ju+%ju on %s: %s (code %d)\n",
range[0], range[1], data_device.c_str(), strerror(-r), r);
return -errno;
}
discarded += range[1];
}
discarded += range[1];
}
j = i+1;
}
+1 -1
View File
@@ -84,7 +84,7 @@ struct blockstore_disk_t
void calc_lengths(bool skip_meta_check = false);
void check_lengths();
void close_all();
int trim_data(std::function<bool(uint64_t)> is_free);
int trim_data(std::function<bool(uint64_t)> is_used);
inline uint64_t dirty_dyn_size(uint64_t offset, uint64_t len)
{
+32 -21
View File
@@ -289,30 +289,41 @@ resume_1:
{
init_fsync_data();
}
if (bs->log_level > 10)
if (compact_info.do_delete)
{
printf("Compacting %jx:%jx v%ju..v%ju / l%ju..l%ju (%d writes)\n", cur_oid.inode, cur_oid.stripe,
compact_info.clean_wr->version, compact_info.compact_version,
compact_info.clean_wr->lsn, compact_info.compact_lsn, copy_count);
}
mem_or(new_bmp, compact_info.clean_wr->get_int_bitmap(bs->heap), bs->dsk.clean_entry_bitmap_size);
if (!bitmap_copied)
{
memcpy(new_ext_bmp, compact_info.clean_wr->get_ext_bitmap(bs->heap), bs->dsk.clean_entry_bitmap_size);
bitmap_copied = true;
}
if (bs->dsk.csum_block_size && bs->dsk.csum_block_size <= bs->dsk.bitmap_granularity)
{
memcpy(new_csums, compact_info.clean_wr->get_checksums(bs->heap), bs->dsk.data_block_size/bs->dsk.csum_block_size * (bs->dsk.data_csum_type & 0xFF));
for (size_t i = csum_copy.size(); i > 0; i--)
if (bs->log_level > 10)
{
auto wr = csum_copy[i-1];
memcpy(new_csums + wr->small().offset/bs->dsk.csum_block_size*(bs->dsk.data_csum_type & 0xFF),
wr->get_checksums(bs->heap), wr->small().len/bs->dsk.csum_block_size*(bs->dsk.data_csum_type & 0xFF));
printf("Compacting %jx:%jx up to l%ju (delete)\n", cur_oid.inode, cur_oid.stripe, compact_info.compact_lsn);
}
csum_copy.clear();
clean_loc = UINT64_MAX;
}
else
{
if (bs->log_level > 10)
{
printf("Compacting %jx:%jx v%ju..v%ju / l%ju..l%ju (%d writes)\n", cur_oid.inode, cur_oid.stripe,
compact_info.clean_wr->version, compact_info.compact_version,
compact_info.clean_wr->lsn, compact_info.compact_lsn, copy_count);
}
mem_or(new_bmp, compact_info.clean_wr->get_int_bitmap(bs->heap), bs->dsk.clean_entry_bitmap_size);
if (!bitmap_copied)
{
memcpy(new_ext_bmp, compact_info.clean_wr->get_ext_bitmap(bs->heap), bs->dsk.clean_entry_bitmap_size);
bitmap_copied = true;
}
if (bs->dsk.csum_block_size && bs->dsk.csum_block_size <= bs->dsk.bitmap_granularity)
{
memcpy(new_csums, compact_info.clean_wr->get_checksums(bs->heap), bs->dsk.data_block_size/bs->dsk.csum_block_size * (bs->dsk.data_csum_type & 0xFF));
for (size_t i = csum_copy.size(); i > 0; i--)
{
auto wr = csum_copy[i-1];
memcpy(new_csums + wr->small().offset/bs->dsk.csum_block_size*(bs->dsk.data_csum_type & 0xFF),
wr->get_checksums(bs->heap), wr->small().len/bs->dsk.csum_block_size*(bs->dsk.data_csum_type & 0xFF));
}
csum_copy.clear();
}
clean_loc = compact_info.clean_wr->big_location(bs->heap);
}
clean_loc = compact_info.clean_wr->big_location(bs->heap);
overwrite_start = overwrite_end = 0;
if (read_vec.size() > 0)
{
@@ -625,7 +636,7 @@ int journal_flusher_co::check_and_punch_checksums()
bool journal_flusher_co::calc_block_checksums()
{
if (bs->dsk.csum_block_size <= bs->dsk.bitmap_granularity)
if (bs->dsk.csum_block_size <= bs->dsk.bitmap_granularity || compact_info.do_delete)
{
return true;
}
+59 -26
View File
@@ -23,7 +23,7 @@
#define HEAP_INFLIGHT_DONE 1
#define HEAP_INFLIGHT_COMPACTABLE 2
#define HEAP_INFLIGHT_COMPACTED 4
#define HEAP_INFLIGHT_OVERWRITE 4
#define HEAP_INFLIGHT_GC 8
#define HEAP_INFLIGHT_EXPLICIT 16
@@ -123,6 +123,8 @@ uint32_t heap_entry_t::get_size(blockstore_heap_t *heap)
}
if (type() == BS_HEAP_SMALL_WRITE || type() == BS_HEAP_INTENT_WRITE)
{
if (size < sizeof(heap_small_write_t))
return heap->get_small_entry_size(0, 0);
return heap->get_small_entry_size(small().offset, small().len);
}
return heap->get_simple_entry_size();
@@ -374,14 +376,11 @@ corrupted_object:
return EDOM;
}
}
if (((wr->entry_type & BS_HEAP_TYPE) == BS_HEAP_SMALL_WRITE ||
(wr->entry_type & BS_HEAP_TYPE) == BS_HEAP_INTENT_WRITE) &&
wr->size < sizeof(heap_small_write_t))
if (wr->size != wr->get_size(this))
{
// Small writes require accessing offset & len to calculate correct length,
// so require at least sizeof(heap_small_write_t) for them
fprintf(stderr, "Error: entry %jx:%jx v%ju has invalid size in metadata block %u at %u (%u < min %zu bytes)\n",
wr->inode, wr->stripe, wr->version, block_num, block_offset, wr->size, sizeof(heap_small_write_t));
// Check entry size
fprintf(stderr, "Error: entry %jx:%jx v%ju has invalid size in metadata block %u at %u (%u != %u bytes)\n",
wr->inode, wr->stripe, wr->version, block_num, block_offset, wr->size, wr->get_size(this));
goto corrupted_object;
}
if (wr->entry_type == BS_HEAP_COMMIT && !wr->version)
@@ -570,7 +569,7 @@ void blockstore_heap_t::finish_load()
size_t s = 0, e, n = postponed_items.size();
for (e = 1; e <= n; e++)
{
if (e >= n || postponed_items[e]->entry.inode != postponed_items[s]->entry.inode &&
if (e >= n || postponed_items[e]->entry.inode != postponed_items[s]->entry.inode ||
postponed_items[e]->entry.stripe != postponed_items[s]->entry.stripe)
{
insert_list_items(postponed_items.data()+s, e-s, false);
@@ -693,10 +692,14 @@ int blockstore_heap_t::mark_used_blocks()
}
use_data(wr->inode, wr->big_location(this));
}
if (wr->is_compactable() && !added)
if (wr->is_compactable())
{
compact_queue.push_back((object_id){ .inode = wr->inode, .stripe = wr->stripe });
added = true;
to_compact_count++;
if (!added)
{
compact_queue.push_back((object_id){ .inode = wr->inode, .stripe = wr->stripe });
added = true;
}
}
if (wr->is_overwrite())
{
@@ -706,6 +709,11 @@ int blockstore_heap_t::mark_used_blocks()
});
}
}
for (auto li: init_erase_items)
{
unlink_list_item(li);
}
init_erase_items.clear();
if (dsk->gc_on_start)
{
recheck_full_gc();
@@ -741,7 +749,6 @@ void blockstore_heap_t::init_erase_bad_entry(heap_list_item_t *li)
inf.garbage_space -= (li->entry.is_garbage() ? li->entry.size : 0);
});
recheck_modified_blocks.insert(li->block_num);
unlink_list_item(li);
}
bool blockstore_heap_t::init_erase_double_claim(heap_list_item_t *prev_li, heap_list_item_t *cur_li)
@@ -796,6 +803,8 @@ bool blockstore_heap_t::init_erase_double_claim(heap_list_item_t *prev_li, heap_
overwritten = erase_li->entry.is_overwrite();
}
init_erase_bad_entry(erase_li);
// Can't erase (mutate map) while iterating, so postpone it
init_erase_items.push_back(erase_li);
erase_li = prev_erase_li;
}
}
@@ -809,6 +818,8 @@ bool blockstore_heap_t::init_erase_double_claim(heap_list_item_t *prev_li, heap_
auto next_erase_li = erase_li->next;
init_free_bad_entry(&erase_li->entry);
init_erase_bad_entry(erase_li);
// Can't erase (mutate map) while iterating, so postpone it
init_erase_items.push_back(erase_li);
erase_li = next_erase_li;
}
erase_li = cur_li;
@@ -817,6 +828,8 @@ bool blockstore_heap_t::init_erase_double_claim(heap_list_item_t *prev_li, heap_
{
auto prev_erase_li = erase_li->prev;
init_erase_bad_entry(erase_li);
// Can't erase (mutate map) while iterating, so postpone it
init_erase_items.push_back(erase_li);
erase_li = prev_erase_li;
}
}
@@ -887,6 +900,7 @@ void blockstore_heap_t::recheck_drop_entries(heap_entry_t *obj, heap_entry_t *ba
auto prev = li->prev;
assert(li->entry.type() == bad_wr->type());
init_erase_bad_entry(li);
unlink_list_item(li);
li = prev;
}
}
@@ -1168,7 +1182,7 @@ bool blockstore_heap_t::calc_block_checksums(uint32_t *block_csums, uint8_t *bit
while (pos < end && pos < block_end && !(bitmap[pos/dsk->bitmap_granularity/8] & (1 << ((pos/dsk->bitmap_granularity) % 8))))
pos += dsk->bitmap_granularity;
// zero padding at the beginning or at the end of the block is not counted
if (pos > prev && prev > 0 && pos < block_end)
if (pos > prev && prev > blk_start && pos < block_end)
{
if (dsk->data_csum_type == BLOCKSTORE_CSUM_XXH3_32)
{
@@ -1613,7 +1627,7 @@ int blockstore_heap_t::add_entry(uint32_t wr_size, uint32_t *modified_block,
// Remember the object as dirty and remove older entries when this block is written and fsynced
push_inflight_lsn(next_lsn, new_wr,
(explicit_complete ? HEAP_INFLIGHT_EXPLICIT : 0) |
(new_wr->is_overwrite() ? HEAP_INFLIGHT_COMPACTED : 0) |
(new_wr->is_overwrite() ? HEAP_INFLIGHT_OVERWRITE : 0) |
(new_wr->is_compactable() ? HEAP_INFLIGHT_COMPACTABLE : 0));
insert_list_items(&li, 1, false);
li->block_num = block_num;
@@ -1647,10 +1661,22 @@ int blockstore_heap_t::add_small_write(object_id oid, heap_entry_t **obj_ptr, ui
wr->small().location = location;
if (bitmap)
memcpy(wr->get_ext_bitmap(this), bitmap, dsk->clean_entry_bitmap_size);
else if (obj)
memcpy(wr->get_ext_bitmap(this), obj->get_ext_bitmap(this), dsk->clean_entry_bitmap_size);
else
memset(wr->get_ext_bitmap(this), 0, dsk->clean_entry_bitmap_size);
{
bool found = false;
iterate_with_stable(obj, UINT64_MAX, [&](heap_entry_t *old_wr, bool stable)
{
if (old_wr->get_ext_bitmap(this))
{
found = true;
memcpy(wr->get_ext_bitmap(this), old_wr->get_ext_bitmap(this), dsk->clean_entry_bitmap_size);
return false;
}
return true;
});
if (!found)
memset(wr->get_ext_bitmap(this), 0, dsk->clean_entry_bitmap_size);
}
calc_checksums(wr, (uint8_t*)data, true);
*obj_ptr = wr;
});
@@ -2114,7 +2140,10 @@ void blockstore_heap_t::iterate_with_stable(heap_entry_t *obj, uint64_t max_lsn,
{
if (old_wr->type() == BS_HEAP_ROLLBACK)
{
rollback_version = old_wr->version;
if (rollback_version > old_wr->version)
{
rollback_version = old_wr->version;
}
}
else if (old_wr->type() == BS_HEAP_COMMIT)
{
@@ -2166,7 +2195,10 @@ heap_compact_t blockstore_heap_t::iterate_compaction(heap_entry_t *obj, uint64_t
res.compact_lsn = wr->lsn;
res.compact_version = wr->version;
}
rollback_version = wr->version;
if (rollback_version > wr->version)
{
rollback_version = wr->version;
}
continue;
}
if (wr->type() == BS_HEAP_COMMIT && wr->lsn <= fsynced_lsn)
@@ -2344,7 +2376,7 @@ void blockstore_heap_t::use_data(inode_t inode, uint64_t location)
{
auto sh_it = pool_shard_settings.find(INODE_POOL(inode));
if (sh_it != pool_shard_settings.end() && sh_it->second.no_inode_stats)
inode = (INODE_POOL(inode) << POOL_ID_BITS);
inode = INODE_WITH_POOL(INODE_POOL(inode), 0);
assert(!data_alloc->get(location / dsk->data_block_size));
data_alloc->set(location / dsk->data_block_size, true);
inode_space_stats[inode] += dsk->data_block_size;
@@ -2355,7 +2387,7 @@ void blockstore_heap_t::free_data(inode_t inode, uint64_t location)
{
auto sh_it = pool_shard_settings.find(INODE_POOL(inode));
if (sh_it != pool_shard_settings.end() && sh_it->second.no_inode_stats)
inode = (INODE_POOL(inode) << POOL_ID_BITS);
inode = INODE_WITH_POOL(INODE_POOL(inode), 0);
assert(data_alloc->get(location / dsk->data_block_size));
data_alloc->set(location / dsk->data_block_size, false);
auto sp_it = inode_space_stats.find(inode);
@@ -2392,7 +2424,8 @@ void blockstore_heap_t::use_buffer_area(inode_t inode, uint64_t location, uint64
return;
}
assert(!(size % dsk->bitmap_granularity));
buffer_alloc->use(location / dsk->bitmap_granularity, size / dsk->bitmap_granularity);
bool ok = buffer_alloc->use(location / dsk->bitmap_granularity, size / dsk->bitmap_granularity);
assert(ok);
buffer_area_used_space += size;
}
@@ -2434,7 +2467,7 @@ void blockstore_heap_t::get_meta_block(uint32_t block_num, uint8_t *buffer)
}
}
void blockstore_heap_t::fill_block_empty_space(uint8_t *buffer, uint32_t pos)
void blockstore_heap_t::fill_block_empty_space(uint8_t *buffer, uint64_t pos)
{
if (pos > dsk->meta_block_size)
{
@@ -2522,7 +2555,7 @@ uint64_t blockstore_heap_t::get_garbage_memory()
void blockstore_heap_t::push_inflight_lsn(uint64_t lsn, heap_entry_t *wr, uint64_t flags)
{
uint64_t next_inf = first_inflight_lsn + inflight_lsn.size();
if (flags & (HEAP_INFLIGHT_COMPACTABLE|HEAP_INFLIGHT_COMPACTED))
if (flags & (HEAP_INFLIGHT_COMPACTABLE|HEAP_INFLIGHT_OVERWRITE))
{
to_compact_count++;
}
@@ -2589,7 +2622,7 @@ void blockstore_heap_t::mark_lsn_fsynced(uint64_t lsn)
void blockstore_heap_t::apply_inflight(heap_inflight_lsn_t & inflight)
{
auto wr = inflight.wr;
if (inflight.flags & HEAP_INFLIGHT_COMPACTED)
if (inflight.flags & HEAP_INFLIGHT_OVERWRITE)
{
// Mark previous entries as garbage, sequentially
mark_garbage_up_to(wr);
+2 -1
View File
@@ -220,6 +220,7 @@ class blockstore_heap_t
bool marked_used_blocks = false;
bool recheck_queue_filled = false;
std::vector<heap_list_item_t*> postponed_items;
std::vector<heap_list_item_t*> init_erase_items;
std::set<uint32_t> recheck_modified_blocks;
std::deque<heap_entry_t*> recheck_queue;
std::map<heap_entry_t*, heap_recheck_state_t> recheck_states;
@@ -362,7 +363,7 @@ public:
// get metadata block data buffer and used space
void get_meta_block(uint32_t block_num, uint8_t *buffer);
void fill_block_empty_space(uint8_t *buffer, uint32_t pos);
void fill_block_empty_space(uint8_t *buffer, uint64_t pos);
uint32_t get_meta_block_used_space(uint32_t block_num);
// get space usage statistics
+5 -2
View File
@@ -117,9 +117,12 @@ public:
journal_flusher_t *flusher;
int write_iodepth = 0;
int inflight_big = 0;
int intent_write_counter = 0;
bool fsyncing_data = false;
uint64_t data_fsync_next = 0;
uint64_t data_fsync_cur = 0;
uint64_t data_fsync_sent = 0;
uint64_t data_fsync_done = 0;
std::deque<bool> data_fsyncs;
bool live = false, queue_stall = false;
ring_loop_i *ringloop = NULL;
+61 -52
View File
@@ -10,7 +10,6 @@
#define INIT_META_EMPTY 0
#define INIT_META_READING 1
#define INIT_META_READ_DONE 2
#define INIT_META_WRITING 3
#define GET_SQE() \
sqe = bs->get_sqe();\
@@ -23,14 +22,15 @@ blockstore_init_meta::blockstore_init_meta(blockstore_impl_t *bs)
this->bs = bs;
}
void blockstore_init_meta::handle_event(ring_data_t *data, int buf_num)
void blockstore_init_meta::handle_event(ring_data_t *data, int buf_num, const char *op)
{
if (data->res < 0)
if (data->res != data->iov.iov_len)
{
throw std::runtime_error(
std::string("read metadata failed at offset ") + std::to_string(buf_num >= 0 ? bufs[buf_num].offset : last_read_offset) +
std::string(": ") + strerror(-data->res)
);
throw std::runtime_error(strprintf(
"%s failed at offset %ju: got %s (code %d), but expected %zu",
op, (buf_num >= 0 ? bufs[buf_num].offset : last_read_offset), strerror(-data->res),
data->res, data->iov.iov_len
));
}
if (buf_num >= 0)
{
@@ -60,7 +60,7 @@ int blockstore_init_meta::loop()
GET_SQE();
last_read_offset = 0;
data->iov = { bs->meta_superblock, (size_t)bs->dsk.meta_block_size };
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "read metadata header"); };
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
bs->ringloop->submit();
submitted++;
@@ -72,25 +72,19 @@ resume_1:
}
if (is_zero((uint64_t*)bs->meta_superblock, bs->dsk.meta_block_size))
{
{
blockstore_meta_header_v3_t *hdr = (blockstore_meta_header_v3_t *)bs->meta_superblock;
hdr->zero = 0;
hdr->magic = BLOCKSTORE_META_MAGIC_V1;
hdr->version = bs->dsk.meta_format;
hdr->meta_block_size = bs->dsk.meta_block_size;
hdr->data_block_size = bs->dsk.data_block_size;
hdr->bitmap_granularity = bs->dsk.bitmap_granularity;
if (bs->dsk.meta_format >= BLOCKSTORE_META_FORMAT_V2)
{
hdr->data_csum_type = bs->dsk.data_csum_type;
hdr->csum_block_size = bs->dsk.csum_block_size;
}
if (bs->dsk.meta_format >= BLOCKSTORE_META_FORMAT_HEAP)
{
hdr->meta_area_size = bs->dsk.meta_area_size;
}
hdr->set_crc32c();
}
assert(bs->dsk.meta_format == BLOCKSTORE_META_FORMAT_HEAP);
blockstore_meta_header_v3_t *hdr = (blockstore_meta_header_v3_t *)bs->meta_superblock;
hdr->zero = 0;
hdr->magic = BLOCKSTORE_META_MAGIC_V1;
hdr->version = bs->dsk.meta_format;
hdr->meta_block_size = bs->dsk.meta_block_size;
hdr->data_block_size = bs->dsk.data_block_size;
hdr->bitmap_granularity = bs->dsk.bitmap_granularity;
hdr->completed_lsn = 0;
hdr->data_csum_type = bs->dsk.data_csum_type;
hdr->csum_block_size = bs->dsk.csum_block_size;
hdr->meta_area_size = bs->dsk.meta_area_size;
hdr->set_crc32c();
if (bs->readonly)
{
printf("Skipping metadata initialization because blockstore is readonly\n");
@@ -98,21 +92,8 @@ resume_1:
else
{
printf("Initializing metadata area\n");
GET_SQE();
last_read_offset = 0;
data->iov = (struct iovec){ bs->meta_superblock, (size_t)bs->dsk.meta_block_size };
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
bs->ringloop->submit();
submitted++;
resume_2:
if (submitted > 0)
{
wait_state = 2;
return 1;
}
zero_on_init = true;
}
zero_on_init = true;
}
else
{
@@ -163,7 +144,7 @@ resume_1:
hdr->header_csum = csum;
}
bs->heap->start_load(((blockstore_meta_header_v3_t *)bs->meta_superblock)->completed_lsn);
if (bs->dsk.inmemory_journal)
if (bs->dsk.inmemory_journal && !zero_on_init)
{
// Read buffer area
printf("Reading buffered data\n");
@@ -175,7 +156,7 @@ resume_1:
bs->buffer_area + md_offset,
(size_t)(bs->dsk.journal_len - md_offset < bs->metadata_buf_size ? bs->dsk.journal_len - md_offset : bs->metadata_buf_size),
};
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "read buffer area"); };
io_uring_prep_readv(sqe, bs->dsk.journal_fd, &data->iov, 1, bs->dsk.journal_offset + md_offset);
md_offset += data->iov.iov_len;
submitted++;
@@ -194,7 +175,7 @@ resume_3:
next_offset = md_offset;
// Read the rest of the metadata
resume_4:
if (next_offset < bs->dsk.meta_area_size && submitted == 0)
if (next_offset < bs->dsk.meta_area_size && submitted == 0 && (!zero_on_init || !bs->readonly))
{
// Submit one read
for (int i = 0; i < 2; i++)
@@ -211,12 +192,15 @@ resume_4:
GET_SQE();
assert(bufs[i].size <= 0x7fffffff);
data->iov = { bufs[i].buf, (size_t)bufs[i].size };
data->callback = [this, i](ring_data_t *data) { handle_event(data, i); };
if (!zero_on_init)
{
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "read metadata"); };
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
}
else
{
// Fill metadata with empty block pattern
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "clear metadata"); };
memset(bufs[i].buf, 0, bufs[i].size);
for (uint64_t o = 0; o < bufs[i].size; o += bs->dsk.meta_block_size)
bs->heap->fill_block_empty_space(bufs[i].buf + o, 0);
@@ -232,11 +216,14 @@ resume_4:
if (bufs[i].state == INIT_META_READ_DONE)
{
// Handle result
uint64_t loaded = 0;
int r = bs->heap->load_blocks(bufs[i].offset-bs->dsk.meta_block_size, bufs[i].size, bufs[i].buf, bs->skip_corrupted_meta_entries, loaded);
if (r != 0)
exit(1);
entries_loaded += loaded;
if (!zero_on_init)
{
uint64_t loaded = 0;
int r = bs->heap->load_blocks(bufs[i].offset-bs->dsk.meta_block_size, bufs[i].size, bufs[i].buf, bs->skip_corrupted_meta_entries, loaded);
if (r != 0)
exit(1);
entries_loaded += loaded;
}
bufs[i].state = 0;
bs->ringloop->wakeup();
}
@@ -246,7 +233,7 @@ resume_4:
wait_state = 4;
return 1;
}
// metadata read finished
// metadata read/clear finished
bs->heap->finish_load();
printf("Metadata entries loaded: %ju, rechecking unfinished writes and garbage entries\n", entries_loaded);
// asynchronous recheck
@@ -329,13 +316,14 @@ resume_9:
}
free(metadata_buffer);
metadata_buffer = NULL;
do_fsync:
if (!bs->dsk.disable_meta_fsync && !bs->readonly)
{
GET_SQE();
io_uring_prep_fsync(sqe, bs->dsk.meta_fd, IORING_FSYNC_DATASYNC);
last_read_offset = 0;
data->iov = { 0 };
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "fsync metadata"); };
submitted++;
bs->ringloop->submit();
resume_5:
@@ -345,6 +333,27 @@ resume_9:
return 1;
}
}
if (zero_on_init && !header_written && !bs->readonly)
{
GET_SQE();
header_written = true;
last_read_offset = 0;
data->iov = (struct iovec){ bs->meta_superblock, (size_t)bs->dsk.meta_block_size };
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "write metadata header"); };
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
bs->ringloop->submit();
submitted++;
resume_2:
if (submitted > 0)
{
wait_state = 2;
return 1;
}
if (!bs->dsk.disable_meta_fsync)
{
goto do_fsync;
}
}
printf("Loading finished. Data used: %ju / %ju bytes (%s / %s)\n",
bs->heap->get_data_used_space(), bs->dsk.block_count * bs->dsk.data_block_size,
format_size(bs->heap->get_data_used_space()).c_str(),
+2 -1
View File
@@ -17,6 +17,7 @@ class blockstore_init_meta
int wait_state = 0;
int wait_count = 0;
bool zero_on_init = false;
bool header_written = false;
void *metadata_buffer = NULL;
blockstore_init_meta_buf bufs[2] = {};
int submitted = 0;
@@ -29,7 +30,7 @@ class blockstore_init_meta
std::vector<uint32_t> recheck_mod;
int i = 0, j = 0;
bool handle_meta_block(uint8_t *buf, uint64_t count, uint64_t done_cnt);
void handle_event(ring_data_t *data, int buf_num);
void handle_event(ring_data_t *data, int buf_num, const char *op);
public:
blockstore_init_meta(blockstore_impl_t *bs);
int loop();
+113
View File
@@ -0,0 +1,113 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 (see README.md for details)
#include "blockstore_mock.h"
blockstore_mock_t::blockstore_mock_t(const blockstore_config_t & config)
{
}
void blockstore_mock_t::parse_config(blockstore_config_t & config)
{
}
void* blockstore_mock_t::reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit)
{
return NULL;
}
bool blockstore_mock_t::reshard_continue(void *reshard_state, uint64_t chunk_limit)
{
return true;
}
void blockstore_mock_t::loop()
{
}
bool blockstore_mock_t::is_started()
{
return true;
}
bool blockstore_mock_t::is_stalled()
{
return false;
}
bool blockstore_mock_t::is_safe_to_stop()
{
return true;
}
void blockstore_mock_t::enqueue_op(blockstore_op_t *op)
{
}
int blockstore_mock_t::read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version)
{
return -EIO;
}
const std::map<uint64_t, uint64_t> & blockstore_mock_t::get_inode_space_stats()
{
return inode_space;
}
void blockstore_mock_t::set_no_inode_stats(const std::vector<uint64_t> & pool_ids)
{
}
void blockstore_mock_t::dump_diagnostics()
{
}
std::string blockstore_mock_t::get_op_diag(blockstore_op_t *op)
{
return "";
}
uint32_t blockstore_mock_t::get_block_size()
{
return block_size;
}
uint64_t blockstore_mock_t::get_block_count()
{
return block_count;
}
uint64_t blockstore_mock_t::get_free_block_count()
{
return block_count;
}
uint64_t blockstore_mock_t::get_journal_size()
{
return 32*1024*1024;
}
uint32_t blockstore_mock_t::get_bitmap_granularity()
{
return bitmap_granularity;
}
uint64_t blockstore_mock_t::get_live_entries()
{
return 0;
}
uint64_t blockstore_mock_t::get_live_memory()
{
return 0;
}
uint64_t blockstore_mock_t::get_garbage_entries()
{
return 0;
}
uint64_t blockstore_mock_t::get_garbage_memory()
{
return 0;
}
+39
View File
@@ -0,0 +1,39 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 (see README.md for details)
#pragma once
#include "blockstore.h"
class blockstore_mock_t: public blockstore_i
{
public:
uint32_t block_size = 128*1024;
uint32_t bitmap_granularity = 4096;
uint64_t block_count = 100*1024*8;
std::map<uint64_t, uint64_t> inode_space;
blockstore_mock_t(const blockstore_config_t & config);
void parse_config(blockstore_config_t & config) override;
void* reshard_start(pool_id_t pool, uint32_t pg_count, uint32_t pg_stripe_size, uint64_t chunk_limit) override;
bool reshard_continue(void *reshard_state, uint64_t chunk_limit) override;
void loop() override;
bool is_started() override;
bool is_stalled() override;
bool is_safe_to_stop() override;
void enqueue_op(blockstore_op_t *op) override;
int read_bitmap(object_id oid, uint64_t target_version, void *bitmap, uint64_t *result_version = NULL) override;
const std::map<uint64_t, uint64_t> & get_inode_space_stats() override;
void set_no_inode_stats(const std::vector<uint64_t> & pool_ids) override;
void dump_diagnostics() override;
std::string get_op_diag(blockstore_op_t *op) override;
uint32_t get_block_size() override;
uint64_t get_block_count() override;
uint64_t get_free_block_count() override;
uint64_t get_journal_size() override;
uint32_t get_bitmap_granularity() override;
uint64_t get_live_entries() override;
uint64_t get_live_memory() override;
uint64_t get_garbage_entries() override;
uint64_t get_garbage_memory() override;
};
+1 -5
View File
@@ -28,6 +28,7 @@ int blockstore_impl_t::dequeue_stable(blockstore_op_t *op)
FINISH_OP(op);
return 2;
}
priv->modified_block2 = UINT32_MAX;
int res = op->opcode == BS_OP_STABLE
? heap->add_commit(obj, v[priv->stab_pos].version, &priv->modified_block2)
: heap->add_rollback(obj, v[priv->stab_pos].version, &priv->modified_block2);
@@ -52,11 +53,6 @@ int blockstore_impl_t::dequeue_stable(blockstore_op_t *op)
FINISH_OP(op);
return 2;
}
if (priv->modified_block2 != UINT32_MAX)
{
priv->stab_pos--;
goto resume_1;
}
priv->wait_for = WAIT_COMPACTION;
priv->wait_detail = heap->get_compacted_count();
flusher->request_trim();
+12 -5
View File
@@ -29,9 +29,12 @@ bool blockstore_impl_t::has_unsynced()
bool blockstore_impl_t::submit_fsyncs(int & wait_count)
{
int n = (unsynced_meta_write_count > 0 && !dsk.disable_meta_fsync) +
(unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync && dsk.journal_fd != dsk.meta_fd) +
(unsynced_data_write_count > 0 && !dsk.disable_data_fsync && dsk.data_fd != dsk.meta_fd && dsk.data_fd != dsk.journal_fd);
int n = (unsynced_meta_write_count > 0 && !dsk.disable_meta_fsync ? 1 : 0) +
(unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync &&
(!unsynced_meta_write_count || dsk.journal_fd != dsk.meta_fd) ? 1 : 0) +
(unsynced_data_write_count > 0 && !dsk.disable_data_fsync &&
(!unsynced_meta_write_count || dsk.data_fd != dsk.meta_fd) &&
(!unsynced_buffer_write_count || dsk.data_fd != dsk.journal_fd) ? 1 : 0);
if (ringloop->space_left() < n)
{
return false;
@@ -60,7 +63,8 @@ bool blockstore_impl_t::submit_fsyncs(int & wait_count)
data->callback = cb;
wait_count++;
}
if (unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync && dsk.meta_fd != dsk.journal_fd)
if (unsynced_buffer_write_count > 0 && !dsk.disable_journal_fsync &&
(!unsynced_meta_write_count || dsk.journal_fd != dsk.meta_fd))
{
// fsync buffer
io_uring_sqe *sqe = get_sqe();
@@ -71,7 +75,9 @@ bool blockstore_impl_t::submit_fsyncs(int & wait_count)
data->callback = cb;
wait_count++;
}
if (unsynced_data_write_count > 0 && !dsk.disable_data_fsync && dsk.data_fd != dsk.meta_fd && dsk.data_fd != dsk.journal_fd)
if (unsynced_data_write_count > 0 && !dsk.disable_data_fsync &&
(!unsynced_meta_write_count || dsk.data_fd != dsk.meta_fd) &&
(!unsynced_buffer_write_count || dsk.data_fd != dsk.journal_fd))
{
// fsync data
io_uring_sqe *sqe = get_sqe();
@@ -109,6 +115,7 @@ int blockstore_impl_t::do_sync(blockstore_op_t *op, int base_state)
PRIV(op)->lsn = heap->get_completed_lsn();
if (!submit_fsyncs(PRIV(op)->pending_ops))
{
PRIV(op)->lsn = 0;
PRIV(op)->wait_detail = 1;
PRIV(op)->wait_for = WAIT_SQE;
return 0;
+35 -28
View File
@@ -180,13 +180,16 @@ enospc:
data->callback = [this, op](ring_data_t *data) { handle_write_event(data, op); };
assert(loc+op->offset+op->len <= dsk.block_count*dsk.data_block_size);
io_uring_prep_writev(sqe, dsk.data_fd, &data->iov, 1, dsk.data_offset + loc + op->offset);
if (!dsk.disable_data_fsync)
{
// use PRIV->lsn for fsync_data_id
PRIV(op)->lsn = ++data_fsync_next;
data_fsyncs.push_back(false);
}
PRIV(op)->pending_ops++;
write_iodepth++;
if (PRIV(op)->write_type == BS_HEAP_BIG_WRITE)
{
PRIV(op)->op_state = 1;
inflight_big++;
}
else
PRIV(op)->op_state = 3;
}
@@ -297,8 +300,6 @@ again:
goto resume_10;
else if (op_state == 11)
goto resume_11;
else if (op_state == 12)
goto resume_12;
else
{
// In progress
@@ -317,38 +318,44 @@ again:
resume_2:
// We must fsync all big writes to avoid complex write workflows
// It's OK for all HDDs and for server SSDs, but slightly worse for desktop SSDs
inflight_big--;
if (!dsk.disable_data_fsync)
{
// fsync data in a batch
resume_11:
if (inflight_big > 0)
// Mark our data write as completed and advance data_fsync_cur
data_fsyncs[PRIV(op)->lsn - data_fsync_cur - 1] = true;
while (data_fsyncs.size() > 0 && data_fsyncs.front())
{
data_fsyncs.pop_front();
data_fsync_cur++;
}
PRIV(op)->op_state = 11;
// Then wait for all other data writes currently in progress to do less fsync calls
// I.e. to fsync data in batches
PRIV(op)->lsn = data_fsync_cur + data_fsyncs.size();
resume_11:
if (data_fsync_cur < PRIV(op)->lsn)
{
PRIV(op)->op_state = 11;
return 1;
}
if (fsyncing_data)
if (PRIV(op)->lsn > data_fsync_sent)
{
resume_12:
if (fsyncing_data)
BS_SUBMIT_GET_SQE(sqe, data);
io_uring_prep_fsync(sqe, dsk.data_fd, IORING_FSYNC_DATASYNC);
data->iov = { 0 };
data->callback = [this, op, fs = data_fsync_cur](ring_data_t *data)
{
PRIV(op)->op_state = 12;
return 1;
}
goto resume_4;
if (fs > data_fsync_done)
{
data_fsync_done = fs;
ringloop->wakeup();
}
};
data_fsync_sent = data_fsync_cur;
}
fsyncing_data = true;
BS_SUBMIT_GET_SQE(sqe, data);
io_uring_prep_fsync(sqe, dsk.data_fd, IORING_FSYNC_DATASYNC);
data->iov = { 0 };
data->callback = [this, op](ring_data_t *data)
if (PRIV(op)->lsn > data_fsync_done)
{
fsyncing_data = false;
handle_write_event(data, op);
};
PRIV(op)->pending_ops++;
PRIV(op)->op_state = 3;
return 1;
return 1;
}
PRIV(op)->lsn = 0;
}
resume_4:
{
+8 -5
View File
@@ -12,7 +12,7 @@ multilist_alloc_t::multilist_alloc_t(uint32_t count, uint32_t maxn):
count(count), maxn(maxn)
{
// not-so-memory-efficient: 16 MB memory per 1 GB buffer space, but buffer spaces are small, so OK
assert(count > 1 && count < 0x80000000);
assert(count > 1 && count < 0x80000000 && count >= maxn);
sizes.resize(count);
nexts.resize(count); // nexts[i] = 0 -> area is used; nexts[i] = 1 -> no next; nexts[i] >= 2 -> next item
prevs.resize(count);
@@ -171,7 +171,7 @@ void multilist_alloc_t::print()
printf("\n");
}
void multilist_alloc_t::use(uint32_t pos, uint32_t size)
bool multilist_alloc_t::use(uint32_t pos, uint32_t size)
{
assert(pos+size <= count && size > 0);
if (sizes[pos] <= 0)
@@ -182,7 +182,8 @@ void multilist_alloc_t::use(uint32_t pos, uint32_t size)
else
while (start > 0 && !sizes[start])
start--;
assert(sizes[start] >= size);
if (sizes[start] < size+(pos-start))
return false;
use_full(start);
uint32_t full = sizes[start];
sizes[pos-1] = -pos+start;
@@ -199,7 +200,8 @@ void multilist_alloc_t::use(uint32_t pos, uint32_t size)
}
else
{
assert(sizes[pos] >= size);
if (sizes[pos] < size)
return false;
use_full(pos);
if (sizes[pos] > size)
{
@@ -214,12 +216,13 @@ void multilist_alloc_t::use(uint32_t pos, uint32_t size)
#ifdef MULTILIST_TRACE
print();
#endif
return true;
}
void multilist_alloc_t::use_full(uint32_t pos)
{
uint32_t prevsize = sizes[pos];
assert(prevsize);
assert(prevsize > 0);
assert(nexts[pos]);
uint32_t pi = (prevsize < maxn ? prevsize : maxn)-1;
if (heads[pi] == pos+1)
+1 -1
View File
@@ -17,7 +17,7 @@ struct multilist_alloc_t
bool is_free(uint32_t pos);
uint32_t find(uint32_t size);
void use_full(uint32_t pos);
void use(uint32_t pos, uint32_t size);
bool use(uint32_t pos, uint32_t size);
void do_free(uint32_t pos);
void free(uint32_t pos);
void verify();
+53 -22
View File
@@ -71,6 +71,11 @@ bool journal_flusher_t::is_active()
return active_flushers > 0 || dequeuing;
}
size_t journal_flusher_t::get_queue_size()
{
return flush_queue.size();
}
void journal_flusher_t::loop()
{
target_flusher_count = bs->write_iodepth*2;
@@ -384,6 +389,7 @@ stop_flusher:
wait_state = 0;
return true;
}
copy_count = 0;
try_trim = true;
cur.oid = flusher->flush_queue.front();
cur.version = flusher->flush_versions[cur.oid];
@@ -511,6 +517,31 @@ resume_2:
{
uo_it->second.was_changed = true;
}
if (!bs->journal.inmemory)
{
// Verify journaled data checksums (but not COALESCED)
for (it = v.begin(); it != v.end(); it++)
{
if (it->copy_flags == COPY_BUF_JOURNAL)
{
iovec iov = { .iov_base = it->buf, .iov_len = it->len };
bs->verify_journal_checksums(
it->csum_buf, it->offset, &iov, 1,
[&](uint32_t bad_block, uint32_t calc_csum, uint32_t stored_csum)
{
printf(
"Checksum mismatch in object %jx:%jx v%ju in journal at 0x%jx, checksum block #%u: got %08x, expected %08x\n",
cur.oid.inode, cur.oid.stripe, cur.version, it->disk_offset,
bad_block / bs->dsk.csum_block_size, calc_csum, stored_csum
);
bad_block += it->offset;
assert(!(bad_block % bs->dsk.csum_block_size) && bad_block < bs->dsk.data_block_size);
mangle_csum_blocks.insert(bad_block);
}
);
}
}
}
}
// Submit data writes
for (it = v.begin(); it != v.end(); it++)
@@ -634,6 +665,7 @@ resume_2:
}
// All done
flusher->active_flushers--;
copy_count = 0; // used by is_mutated()...
wait_state = 0;
goto resume_0;
}
@@ -815,35 +847,21 @@ bool journal_flusher_co::clear_incomplete_csum_block_bits(int wait_base)
bs->verify_padded_checksums(new_clean_bitmap, new_clean_bitmap + 2*bs->dsk.clean_entry_bitmap_size,
v[i].offset, &iov, 1, [&](uint32_t bad_block, uint32_t calc_csum, uint32_t stored_csum)
{
printf("Checksum mismatch in object %jx:%jx v%ju in data area at offset 0x%jx+0x%x: got %08x, expected %08x\n",
printf("Checksum mismatch in object %jx:%jx v%ju in data area at offset 0x%jx+0x%x during flush: got %08x, expected %08x\n",
cur.oid.inode, cur.oid.stripe, old_clean_ver, old_clean_loc, bad_block, calc_csum, stored_csum);
for (uint32_t j = 0; j < bs->dsk.csum_block_size; j += bs->dsk.bitmap_granularity)
{
// Simplest method of mangling: flip one byte in every sector
((uint8_t*)v[i].buf)[j+bad_block-v[i].offset] ^= 0xff;
}
assert(!(bad_block % bs->dsk.csum_block_size) && bad_block < bs->dsk.data_block_size);
mangle_csum_blocks.insert(bad_block);
});
}
else
{
bs->verify_journal_checksums(v[i].csum_buf, v[i].offset, &iov, 1, [&](uint32_t bad_block, uint32_t calc_csum, uint32_t stored_csum)
{
printf("Checksum mismatch in object %jx:%jx v%ju in journal at offset 0x%jx+0x%x (block offset 0x%jx): got %08x, expected %08x\n",
printf("Checksum mismatch in object %jx:%jx v%ju in journal at offset 0x%jx+0x%x (block offset 0x%jx) during flush: got %08x, expected %08x\n",
cur.oid.inode, cur.oid.stripe, old_clean_ver,
v[i].disk_offset, bad_block, v[i].offset, calc_csum, stored_csum);
bad_block += (v[i].offset/bs->dsk.csum_block_size) * bs->dsk.csum_block_size;
uint32_t bad_block_end = bad_block + bs->dsk.csum_block_size + (v[i].offset/bs->dsk.csum_block_size) * bs->dsk.csum_block_size;
if (bad_block < v[i].offset)
bad_block = v[i].offset;
if (bad_block_end > v[i].offset+v[i].len)
bad_block_end = v[i].offset+v[i].len;
bad_block -= v[i].offset;
bad_block_end -= v[i].offset;
for (uint32_t j = bad_block; j < bad_block_end; j += bs->dsk.bitmap_granularity)
{
// Simplest method of mangling: flip one byte in every sector
((uint8_t*)v[i].buf)[j] ^= 0xff;
}
assert(!(bad_block % bs->dsk.csum_block_size) && bad_block < bs->dsk.data_block_size);
mangle_csum_blocks.insert(bad_block);
});
}
}
@@ -952,6 +970,11 @@ void journal_flusher_co::calc_block_checksums(uint32_t *new_data_csums, bool ski
}
// `v` should contain aligned items, possibly split into pieces
assert(!block_done);
for (uint32_t mangle_block: mangle_csum_blocks)
{
// Flip 1 bit
new_data_csums[mangle_block / bs->dsk.csum_block_size] ^= 1;
}
}
void journal_flusher_co::scan_dirty()
@@ -1088,7 +1111,8 @@ void journal_flusher_co::scan_dirty()
last--;
read_to_fill_incomplete = bs->fill_partial_checksum_blocks(
v, fulfilled, bmp_ptr, NULL, false, NULL, v[0].offset/bs->dsk.csum_block_size * bs->dsk.csum_block_size,
((v[last].offset+v[last].len-1) / bs->dsk.csum_block_size + 1) * bs->dsk.csum_block_size
((v[last].offset+v[last].len-1) / bs->dsk.csum_block_size + 1) * bs->dsk.csum_block_size,
0, bs->dsk.data_block_size
);
}
else if (fill_incomplete && clean_init_bitmap)
@@ -1118,6 +1142,7 @@ bool journal_flusher_co::read_dirty(int wait_base)
if (wait_state == wait_base) goto resume_0;
else if (wait_state == wait_base+1) goto resume_1;
wait_count = wait_journal_count = 0;
mangle_csum_blocks.clear();
if (bs->journal.inmemory && !read_to_fill_incomplete)
{
// Happy path: nothing to read :)
@@ -1349,7 +1374,7 @@ bool journal_flusher_co::fsync_batch(bool fsync_meta, int wait_base)
cur_sync->ready_count++;
flusher->syncing_flushers++;
resume_1:
if (!cur_sync->state)
if (cur_sync->state == 0)
{
if (flusher->syncing_flushers >= flusher->active_flushers || !flusher->flush_queue.size())
{
@@ -1377,6 +1402,12 @@ bool journal_flusher_co::fsync_batch(bool fsync_meta, int wait_base)
return false;
}
}
else if (cur_sync->state == 1)
{
// Wait for fsync completion
wait_state = wait_base+1;
return false;
}
flusher->syncing_flushers--;
cur_sync->ready_count--;
if (cur_sync->ready_count == 0)
+2
View File
@@ -66,6 +66,7 @@ class journal_flusher_co
uint64_t clean_bitmap_offset, clean_bitmap_len;
uint8_t *clean_init_dyn_ptr;
uint8_t *new_clean_bitmap;
std::unordered_set<uint32_t> mangle_csum_blocks;
uint64_t new_trim_pos;
@@ -123,6 +124,7 @@ public:
void loop();
bool is_trim_wanted() { return trim_wanted; }
bool is_active();
size_t get_queue_size();
void mark_trim_possible();
void request_trim();
void release_trim();
+7 -1
View File
@@ -6,11 +6,12 @@
namespace v1 {
blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd)
blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd, bool mock_mode)
{
assert(sizeof(blockstore_op_private_t) <= BS_OP_PRIVATE_DATA_SIZE);
this->tfd = tfd;
this->ringloop = ringloop;
dsk.mock_mode = mock_mode;
ring_consumer.loop = [this]() { loop(); };
ringloop->register_consumer(&ring_consumer);
initialized = 0;
@@ -35,6 +36,11 @@ blockstore_impl_t::blockstore_impl_t(blockstore_config_t & config, ring_loop_i *
blockstore_impl_t::~blockstore_impl_t()
{
for (auto& obj: dirty_db)
{
if (obj.second.dyn_data)
free(obj.second.dyn_data);
}
delete data_alloc;
delete flusher;
if (zero_object)
+8 -3
View File
@@ -30,6 +30,8 @@
//#define BLOCKSTORE_DEBUG
struct bs_test_t;
namespace v1 {
#include "journal.h"
@@ -96,7 +98,7 @@ struct blockstore_op_private_t
int op_state;
// Read
uint64_t clean_block_used;
uint64_t clean_loc_used;
std::vector<copy_buffer_t> read_vec;
// Sync, write
@@ -122,6 +124,7 @@ typedef uint64_t pool_pg_id_t;
class blockstore_impl_t: public blockstore_i
{
friend struct ::bs_test_t;
blockstore_disk_t dsk;
/******* OPTIONS *******/
@@ -220,6 +223,7 @@ class blockstore_impl_t: public blockstore_i
// Read
int dequeue_read(blockstore_op_t *read_op);
void release_clean(blockstore_op_t *op);
void find_holes(std::vector<copy_buffer_t> & read_vec, uint32_t item_start, uint32_t item_end,
std::function<int(int, bool, uint32_t, uint32_t)> callback);
int fulfill_read(blockstore_op_t *read_op,
@@ -230,7 +234,8 @@ class blockstore_impl_t: public blockstore_i
uint8_t *clean_entry_bitmap, int *dyn_data,
uint32_t item_start, uint32_t item_end, uint64_t clean_loc, uint64_t clean_ver);
int fill_partial_checksum_blocks(std::vector<copy_buffer_t> & rv, uint64_t & fulfilled,
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf, uint64_t read_offset, uint64_t read_end);
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf,
uint32_t read_offset, uint32_t read_end, uint32_t item_start, uint32_t item_end);
int pad_journal_read(std::vector<copy_buffer_t> & rv, copy_buffer_t & cp,
uint64_t dirty_offset, uint64_t dirty_end, uint64_t dirty_loc, uint8_t *csum_ptr, int *dyn_data,
uint64_t offset, uint64_t submit_len, uint64_t & blk_begin, uint64_t & blk_end, uint8_t* & blk_buf);
@@ -281,7 +286,7 @@ class blockstore_impl_t: public blockstore_i
public:
blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd);
blockstore_impl_t(blockstore_config_t & config, ring_loop_i *ringloop, timerfd_manager_t *tfd, bool mock_mode = false);
~blockstore_impl_t();
void parse_config(blockstore_config_t & config);
+83 -68
View File
@@ -1,6 +1,7 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 (see README.md for details)
#include "str_util.h"
#include "impl.h"
#include "internal.h"
@@ -30,14 +31,15 @@ blockstore_init_meta::blockstore_init_meta(blockstore_impl_t *bs)
this->bs = bs;
}
void blockstore_init_meta::handle_event(ring_data_t *data, int buf_num)
void blockstore_init_meta::handle_event(ring_data_t *data, int buf_num, const char *op)
{
if (data->res < 0)
if (data->res != data->iov.iov_len)
{
throw std::runtime_error(
std::string("read metadata failed at offset ") + std::to_string(buf_num >= 0 ? bufs[buf_num].offset : last_read_offset) +
std::string(": ") + strerror(-data->res)
);
throw std::runtime_error(strprintf(
"%s failed at offset %ju: got %s (code %d), but expected %zu",
op, (buf_num >= 0 ? bufs[buf_num].offset : last_read_offset), strerror(-data->res),
data->res, data->iov.iov_len
));
}
if (buf_num >= 0)
{
@@ -65,10 +67,11 @@ int blockstore_init_meta::loop()
if (!metadata_buffer)
throw std::runtime_error("Failed to allocate metadata read buffer");
// Read superblock
hdr = (blockstore_meta_header_v2_t *)memalign_or_die(MEM_ALIGNMENT, bs->dsk.meta_block_size);
GET_SQE();
last_read_offset = 0;
data->iov = { metadata_buffer, (size_t)bs->dsk.meta_block_size };
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
data->iov = { hdr, (size_t)bs->dsk.meta_block_size };
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "read metadata header"); };
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
bs->ringloop->submit();
submitted++;
@@ -78,24 +81,8 @@ resume_1:
wait_state = 1;
return 1;
}
if (iszero((uint64_t*)metadata_buffer, bs->dsk.meta_block_size / sizeof(uint64_t)))
if (iszero((uint64_t*)hdr, bs->dsk.meta_block_size / sizeof(uint64_t)))
{
{
blockstore_meta_header_v2_t *hdr = (blockstore_meta_header_v2_t *)metadata_buffer;
hdr->zero = 0;
hdr->magic = BLOCKSTORE_META_MAGIC_V1;
hdr->version = bs->dsk.meta_format;
hdr->meta_block_size = bs->dsk.meta_block_size;
hdr->data_block_size = bs->dsk.data_block_size;
hdr->bitmap_granularity = bs->dsk.bitmap_granularity;
if (bs->dsk.meta_format >= BLOCKSTORE_META_FORMAT_V2)
{
hdr->data_csum_type = bs->dsk.data_csum_type;
hdr->csum_block_size = bs->dsk.csum_block_size;
hdr->header_csum = 0;
hdr->header_csum = crc32c(0, hdr, sizeof(*hdr));
}
}
if (bs->readonly)
{
printf("Skipping metadata initialization because blockstore is readonly\n");
@@ -103,25 +90,11 @@ resume_1:
else
{
printf("Initializing metadata area\n");
GET_SQE();
last_read_offset = 0;
data->iov = (struct iovec){ metadata_buffer, (size_t)bs->dsk.meta_block_size };
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
bs->ringloop->submit();
submitted++;
resume_3:
if (submitted > 0)
{
wait_state = 3;
return 1;
}
zero_on_init = true;
}
zero_on_init = true;
}
else
{
blockstore_meta_header_v2_t *hdr = (blockstore_meta_header_v2_t *)metadata_buffer;
if (hdr->zero != 0 || hdr->magic != BLOCKSTORE_META_MAGIC_V1 || hdr->version < BLOCKSTORE_META_FORMAT_V1)
{
printf(
@@ -223,12 +196,15 @@ resume_2:
GET_SQE();
assert(bufs[i].size <= 0x7fffffff);
data->iov = { bufs[i].buf, (size_t)bufs[i].size };
data->callback = [this, i](ring_data_t *data) { handle_event(data, i); };
if (!zero_on_init)
{
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "read metadata"); };
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
}
else
{
// Fill metadata with zeroes
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "clear metadata"); };
memset(data->iov.iov_base, 0, data->iov.iov_len);
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
}
@@ -256,7 +232,7 @@ resume_2:
GET_SQE();
assert(bufs[i].size <= 0x7fffffff);
data->iov = { bufs[i].buf, (size_t)bufs[i].size };
data->callback = [this, i](ring_data_t *data) { handle_event(data, i); };
data->callback = [this, i](ring_data_t *data) { handle_event(data, i, "write metadata"); };
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + bufs[i].offset);
bs->ringloop->submit();
bufs[i].state = INIT_META_WRITING;
@@ -285,7 +261,7 @@ resume_2:
GET_SQE();
last_read_offset = (1+next_offset)*bs->dsk.meta_block_size;
data->iov = { metadata_buffer, (size_t)bs->dsk.meta_block_size };
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "read metadata"); };
io_uring_prep_readv(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + (1+next_offset)*bs->dsk.meta_block_size);
bs->ringloop->submit();
submitted++;
@@ -302,7 +278,7 @@ resume_5:
}
GET_SQE();
data->iov = { metadata_buffer, (size_t)bs->dsk.meta_block_size };
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "write metadata"); };
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset + (1+next_offset)*bs->dsk.meta_block_size);
bs->ringloop->submit();
submitted++;
@@ -317,27 +293,64 @@ resume_6:
}
// metadata read finished
printf("Metadata entries loaded: %ju, free blocks: %ju / %ju\n", entries_loaded, bs->data_alloc->get_free_count(), bs->dsk.block_count);
if (zero_on_init && !bs->readonly)
{
do_fsync:
if (!bs->disable_meta_fsync)
{
GET_SQE();
io_uring_prep_fsync(sqe, bs->dsk.meta_fd, IORING_FSYNC_DATASYNC);
last_read_offset = 0;
data->iov = { 0 };
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "fsync metadata"); };
submitted++;
bs->ringloop->submit();
resume_4:
if (submitted > 0)
{
wait_state = 4;
return 1;
}
}
if (!header_written)
{
GET_SQE();
hdr->zero = 0;
hdr->magic = BLOCKSTORE_META_MAGIC_V1;
hdr->version = bs->dsk.meta_format;
hdr->meta_block_size = bs->dsk.meta_block_size;
hdr->data_block_size = bs->dsk.data_block_size;
hdr->bitmap_granularity = bs->dsk.bitmap_granularity;
if (bs->dsk.meta_format >= BLOCKSTORE_META_FORMAT_V2)
{
hdr->data_csum_type = bs->dsk.data_csum_type;
hdr->csum_block_size = bs->dsk.csum_block_size;
hdr->header_csum = 0;
hdr->header_csum = crc32c(0, hdr, sizeof(*hdr));
}
header_written = true;
last_read_offset = 0;
data->iov = (struct iovec){ hdr, (size_t)bs->dsk.meta_block_size };
data->callback = [this](ring_data_t *data) { handle_event(data, -1, "write metadata header"); };
io_uring_prep_writev(sqe, bs->dsk.meta_fd, &data->iov, 1, bs->dsk.meta_offset);
bs->ringloop->submit();
submitted++;
resume_3:
if (submitted > 0)
{
wait_state = 3;
return 1;
}
goto do_fsync;
}
}
if (!bs->inmemory_meta)
{
free(metadata_buffer);
metadata_buffer = NULL;
}
if (zero_on_init && !bs->disable_meta_fsync)
{
GET_SQE();
io_uring_prep_fsync(sqe, bs->dsk.meta_fd, IORING_FSYNC_DATASYNC);
last_read_offset = 0;
data->iov = { 0 };
data->callback = [this](ring_data_t *data) { handle_event(data, -1); };
submitted++;
bs->ringloop->submit();
resume_4:
if (submitted > 0)
{
wait_state = 4;
return 1;
}
}
free(hdr);
hdr = NULL;
return 0;
}
@@ -345,6 +358,8 @@ bool blockstore_init_meta::handle_meta_block(uint8_t *buf, uint64_t entries_per_
{
bool updated = false;
uint64_t max_i = entries_per_block;
if (done_cnt > bs->dsk.block_count)
return false;
if (max_i > bs->dsk.block_count-done_cnt)
max_i = bs->dsk.block_count-done_cnt;
for (uint64_t i = 0; i < max_i; i++)
@@ -455,21 +470,21 @@ blockstore_init_journal::blockstore_init_journal(blockstore_impl_t *bs)
};
}
void blockstore_init_journal::handle_event(ring_data_t *data1)
void blockstore_init_journal::handle_event(ring_data_t *data)
{
if (data1->res <= 0)
if (data->res != data->iov.iov_len)
{
throw std::runtime_error(
std::string("read journal failed at offset ") + std::to_string(journal_pos) +
std::string(": ") + strerror(-data1->res)
);
throw std::runtime_error(strprintf(
"read journal failed at offset %ju: got %s (code %d), but expected %zu",
journal_pos, strerror(-data->res), data->res, data->iov.iov_len
));
}
done.push_back({
.buf = submitted_buf,
.pos = journal_pos,
.len = (uint64_t)data1->res,
.len = (uint64_t)data->res,
});
journal_pos += data1->res;
journal_pos += data->res;
if (journal_pos >= bs->journal.len)
{
// Continue from the beginning
+3 -1
View File
@@ -16,7 +16,9 @@ class blockstore_init_meta
blockstore_impl_t *bs;
int wait_state = 0;
bool zero_on_init = false;
bool header_written = false;
void *metadata_buffer = NULL;
blockstore_meta_header_v2_t *hdr = NULL;
blockstore_init_meta_buf bufs[2] = {};
int submitted = 0;
struct io_uring_sqe *sqe;
@@ -29,7 +31,7 @@ class blockstore_init_meta
int i = 0, j = 0;
std::vector<uint64_t> entries_to_zero;
bool handle_meta_block(uint8_t *buf, uint64_t count, uint64_t done_cnt);
void handle_event(ring_data_t *data, int buf_num);
void handle_event(ring_data_t *data, int buf_num, const char *op);
public:
blockstore_init_meta(blockstore_impl_t *bs);
int loop();
+145 -54
View File
@@ -101,8 +101,8 @@ int blockstore_impl_t::fulfill_read(blockstore_op_t *read_op,
.copy_flags = COPY_BUF_JOURNAL|COPY_BUF_CSUM_FILL,
.offset = blk_begin,
.len = blk_end-blk_begin,
.csum_buf = (csum + (blk_begin/dsk.csum_block_size -
item_start/dsk.csum_block_size) * (dsk.data_csum_type & 0xFF)),
.csum_buf = (!csum ? NULL : (csum + (blk_begin/dsk.csum_block_size -
item_start/dsk.csum_block_size) * (dsk.data_csum_type & 0xFF))),
.dyn_data = dyn_data,
});
if (dyn_data)
@@ -134,7 +134,7 @@ int blockstore_impl_t::fulfill_read(blockstore_op_t *read_op,
// If we don't track it then we may IN THEORY read another object's data:
// submit read -> remove the object -> flush remove -> overwrite with another object -> finish read
// Very improbable, but possible
PRIV(read_op)->clean_block_used = 1;
PRIV(read_op)->clean_loc_used = UINT64_MAX;
}
rv.insert(rv.begin() + pos, el);
fulfilled += el.len;
@@ -167,7 +167,8 @@ uint8_t* blockstore_impl_t::get_clean_entry_bitmap(uint64_t block_loc, int offse
}
int blockstore_impl_t::fill_partial_checksum_blocks(std::vector<copy_buffer_t> & rv, uint64_t & fulfilled,
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf, uint64_t read_offset, uint64_t read_end)
uint8_t *clean_entry_bitmap, int *dyn_data, bool from_journal, uint8_t *read_buf,
uint32_t read_offset, uint32_t read_end, uint32_t item_start, uint32_t item_end)
{
if (read_end == read_offset)
return 0;
@@ -175,10 +176,38 @@ int blockstore_impl_t::fill_partial_checksum_blocks(std::vector<copy_buffer_t> &
read_buf -= read_offset;
uint32_t last_block = (read_end-1)/dsk.csum_block_size;
uint32_t start_block = read_offset/dsk.csum_block_size;
uint32_t item_start_block = item_start/dsk.csum_block_size;
uint32_t end_block = 0;
auto zero_range = [&](int pos, bool alloc, uint32_t cur_start, uint32_t cur_end)
{
if (alloc)
return 0;
copy_buffer_t el = {
.copy_flags = COPY_BUF_ZERO,
.offset = cur_start,
.len = cur_end-cur_start,
};
rv.insert(rv.begin() + pos, el);
if (read_buf)
memset(read_buf + el.offset - read_offset, 0, el.len);
fulfilled += el.len;
return 1;
};
if (read_offset < item_start)
{
// Zero-fill the beginning
find_holes(rv, read_offset, item_start, zero_range);
read_offset = item_start;
}
if (read_end > item_end)
{
// Zero-fill the end
find_holes(rv, item_end, read_end, zero_range);
read_end = item_end;
}
while (start_block <= last_block)
{
if (read_range_fulfilled(rv, fulfilled, read_buf, clean_entry_bitmap,
if (read_range_fulfilled(rv, fulfilled, read_buf, from_journal ? NULL : clean_entry_bitmap,
start_block*dsk.csum_block_size < read_offset ? read_offset : start_block*dsk.csum_block_size,
(start_block+1)*dsk.csum_block_size > read_end ? read_end : (start_block+1)*dsk.csum_block_size))
{
@@ -190,7 +219,7 @@ int blockstore_impl_t::fill_partial_checksum_blocks(std::vector<copy_buffer_t> &
// Find a sequence of checksum blocks required to be read
end_block = start_block;
while ((end_block+1)*dsk.csum_block_size < read_end &&
!read_range_fulfilled(rv, fulfilled, read_buf, clean_entry_bitmap,
!read_range_fulfilled(rv, fulfilled, read_buf, from_journal ? NULL : clean_entry_bitmap,
(end_block+1)*dsk.csum_block_size < read_offset ? read_offset : (end_block+1)*dsk.csum_block_size,
(end_block+2)*dsk.csum_block_size > read_end ? read_end : (end_block+2)*dsk.csum_block_size))
{
@@ -202,8 +231,10 @@ int blockstore_impl_t::fill_partial_checksum_blocks(std::vector<copy_buffer_t> &
.copy_flags = COPY_BUF_CSUM_FILL | (from_journal ? COPY_BUF_JOURNALED_BIG : 0),
.offset = start_block*dsk.csum_block_size,
.len = (end_block-start_block)*dsk.csum_block_size,
// save clean_entry_bitmap if we're reading clean data from the journal
.csum_buf = from_journal ? clean_entry_bitmap : NULL,
// save checksum reference if we're reading clean data from the journal
.csum_buf = from_journal
? clean_entry_bitmap + dsk.clean_entry_bitmap_size + (start_block-item_start_block)*(dsk.data_csum_type & 0xFF)
: NULL,
.dyn_data = dyn_data,
});
if (dyn_data)
@@ -226,6 +257,11 @@ bool blockstore_impl_t::read_range_fulfilled(std::vector<copy_buffer_t> & rv, ui
{
if (alloc)
return 0;
if (!clean_entry_bitmap)
{
all_done = false;
return 0;
}
int diff = 0;
uint32_t bmp_start = cur_start/dsk.bitmap_granularity;
uint32_t bmp_end = cur_end/dsk.bitmap_granularity;
@@ -323,7 +359,7 @@ bool blockstore_impl_t::read_checksum_block(blockstore_op_t *op, int rv_pos, uin
{
iov[n_iov++] = (struct iovec){ (uint8_t*)op->buf+cur_start-op->offset, lim_end-cur_start };
rv.insert(rv.begin() + pos, (copy_buffer_t){
.copy_flags = COPY_BUF_DATA,
.copy_flags = COPY_BUF_DATA|COPY_BUF_COALESCED,
.offset = cur_start,
.len = lim_end-cur_start,
});
@@ -361,10 +397,10 @@ bool blockstore_impl_t::read_checksum_block(blockstore_op_t *op, int rv_pos, uin
PRIV(op)->pending_ops++;
io_uring_prep_readv(sqe, submit_fd, iov + n_pos, n_cur, submit_offset + clean_loc + item_start + d_pos);
data->callback = [this, op](ring_data_t *data) { handle_read_event(data, op); };
if (n_pos > 0 || n_pos + IOV_MAX < n_iov)
if (n_pos > 0 || n_iov > IOV_MAX)
{
uint32_t d_len = 0;
for (int i = 0; i < IOV_MAX; i++)
for (int i = 0; i < n_cur; i++)
d_len += iov[n_pos+i].iov_len;
data->iov.iov_len = d_len;
d_pos += d_len;
@@ -376,7 +412,7 @@ bool blockstore_impl_t::read_checksum_block(blockstore_op_t *op, int rv_pos, uin
{
// Reads running parallel to flushes of the same clean block may read
// a mixture of old and new data. So we don't verify checksums for such blocks.
PRIV(op)->clean_block_used = 1;
PRIV(op)->clean_loc_used = UINT64_MAX;
}
return true;
}
@@ -402,7 +438,7 @@ int blockstore_impl_t::dequeue_read(blockstore_op_t *read_op)
}
uint64_t fulfilled = 0;
PRIV(read_op)->pending_ops = 0;
PRIV(read_op)->clean_block_used = 0;
PRIV(read_op)->clean_loc_used = 0;
auto & rv = PRIV(read_op)->read_vec;
uint64_t result_version = 0;
if (dirty_found)
@@ -515,26 +551,50 @@ int blockstore_impl_t::dequeue_read(blockstore_op_t *read_op)
return 2;
undo_read:
// need to wait. undo added requests, don't dequeue op
if (dsk.csum_block_size > dsk.bitmap_granularity)
release_clean(read_op);
for (auto & vec: rv)
{
for (auto & vec: rv)
if ((vec.copy_flags & COPY_BUF_CSUM_FILL) && vec.buf)
{
if ((vec.copy_flags & COPY_BUF_CSUM_FILL) && vec.buf)
{
free(vec.buf);
vec.buf = NULL;
}
if (vec.dyn_data && --(*vec.dyn_data) == 0) // refcount
{
free(vec.dyn_data);
vec.dyn_data = NULL;
}
free(vec.buf);
vec.buf = NULL;
}
if (vec.dyn_data && --(*vec.dyn_data) == 0) // refcount
{
free(vec.dyn_data);
vec.dyn_data = NULL;
}
}
rv.clear();
return 0;
}
void blockstore_impl_t::release_clean(blockstore_op_t *op)
{
if (PRIV(op)->clean_loc_used == UINT64_MAX)
{
PRIV(op)->clean_loc_used = 0;
}
if (PRIV(op)->clean_loc_used)
{
// Release clean data block
auto uo_it = used_clean_objects.find(PRIV(op)->clean_loc_used - 1);
if (uo_it != used_clean_objects.end())
{
uo_it->second.refs--;
if (uo_it->second.refs <= 0)
{
if (uo_it->second.was_freed)
{
data_alloc->set((PRIV(op)->clean_loc_used - 1) / dsk.data_block_size, false);
}
used_clean_objects.erase(uo_it);
}
}
PRIV(op)->clean_loc_used = 0;
}
}
int blockstore_impl_t::pad_journal_read(std::vector<copy_buffer_t> & rv, copy_buffer_t & cp,
// FIXME Passing dirty_entry& would be nicer
uint64_t dirty_offset, uint64_t dirty_end, uint64_t dirty_loc, uint8_t *csum_ptr, int *dyn_data,
@@ -598,11 +658,15 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
{
auto & rv = PRIV(read_op)->read_vec;
int req = fill_partial_checksum_blocks(rv, fulfilled, clean_entry_bitmap, dyn_data, from_journal,
(uint8_t*)read_op->buf, read_op->offset, read_op->offset+read_op->len);
(uint8_t*)read_op->buf, read_op->offset, read_op->offset+read_op->len, item_start, item_end);
if (!inmemory_meta && !from_journal && req > 0)
{
// Read checksums from disk
uint8_t *csum_buf = read_clean_meta_block(read_op, clean_loc, rv.size()-req);
if (!csum_buf)
{
return false;
}
for (int i = req; i > 0; i--)
{
rv[rv.size()-i].csum_buf = csum_buf;
@@ -615,7 +679,7 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
return false;
}
}
PRIV(read_op)->clean_block_used = req > 0;
PRIV(read_op)->clean_loc_used = req > 0 ? UINT64_MAX : 0;
}
else if (from_journal)
{
@@ -665,6 +729,10 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
{
// Read checksums from disk
csum_buf = read_clean_meta_block(read_op, clean_loc, PRIV(read_op)->read_vec.size());
if (!csum_buf)
{
return false;
}
csum_done = true;
}
uint8_t *csum = !dsk.csum_block_size ? 0 : (csum_buf + 2*dsk.clean_entry_bitmap_size + bmp_start*(dsk.data_csum_type & 0xFF));
@@ -679,13 +747,13 @@ bool blockstore_impl_t::fulfill_clean_read(blockstore_op_t *read_op, uint64_t &
}
}
// Increment reference counter if clean data is being read from the disk
if (PRIV(read_op)->clean_block_used)
if (PRIV(read_op)->clean_loc_used == UINT64_MAX)
{
auto & uo = used_clean_objects[clean_loc];
uo.refs++;
if (dsk.csum_block_size && flusher->is_mutated(clean_loc))
uo.was_changed = true;
PRIV(read_op)->clean_block_used = clean_loc;
PRIV(read_op)->clean_loc_used = clean_loc + 1;
}
return true;
}
@@ -725,12 +793,18 @@ bool blockstore_impl_t::verify_padded_checksums(uint8_t *clean_entry_bitmap, uin
while (pos < iov[i].iov_len)
{
uint32_t start = pos;
uint8_t bit = (clean_entry_bitmap[bmp_pos >> 3] >> (bmp_pos & 0x7)) & 1;
while (pos < iov[i].iov_len && ((clean_entry_bitmap[bmp_pos >> 3] >> (bmp_pos & 0x7)) & 1) == bit)
uint8_t bit = 1;
if (clean_entry_bitmap)
{
pos += dsk.bitmap_granularity;
bmp_pos++;
bit = (clean_entry_bitmap[bmp_pos >> 3] >> (bmp_pos & 0x7)) & 1;
while (pos < iov[i].iov_len && ((clean_entry_bitmap[bmp_pos >> 3] >> (bmp_pos & 0x7)) & 1) == bit)
{
pos += dsk.bitmap_granularity;
bmp_pos++;
}
}
else
pos = iov[i].iov_len;
uint32_t len = pos-start;
auto buf = (uint8_t*)iov[i].iov_base+start;
while (block_done+len >= dsk.csum_block_size)
@@ -807,7 +881,7 @@ bool blockstore_impl_t::verify_clean_padded_checksums(blockstore_op_t *op, uint6
{
uint32_t offset = clean_loc % dsk.data_block_size;
if (from_journal)
return verify_padded_checksums(dyn_data, dyn_data + dsk.clean_entry_bitmap_size, offset, iov, n_iov, bad_block_cb);
return verify_padded_checksums(NULL, dyn_data, offset, iov, n_iov, bad_block_cb);
clean_loc = (clean_loc / dsk.data_block_size) * dsk.data_block_size;
if (!dyn_data)
{
@@ -835,7 +909,7 @@ void blockstore_impl_t::handle_read_event(ring_data_t *data, blockstore_op_t *op
void *meta_block = NULL;
if (dsk.csum_block_size > dsk.bitmap_granularity)
{
for (int i = rv.size()-1; i >= 0 && (rv[i].copy_flags & COPY_BUF_CSUM_FILL); i--)
for (int i = 0; i < rv.size(); i++)
{
if (rv[i].copy_flags & COPY_BUF_META_BLOCK)
{
@@ -845,8 +919,41 @@ void blockstore_impl_t::handle_read_event(ring_data_t *data, blockstore_op_t *op
rv[i].buf = NULL;
continue;
}
struct iovec *iov = (struct iovec*)((uint8_t*)rv[i].buf + (rv[i].len & 0xFFFFFFFF));
int n_iov = rv[i].len >> 32;
if (rv[i].copy_flags & COPY_BUF_ZERO)
{
// Zero read
continue;
}
if (rv[i].copy_flags & COPY_BUF_COALESCED)
{
// Sub-block shared with another read. Skip
continue;
}
if ((rv[i].copy_flags & COPY_BUF_JOURNAL) && journal.inmemory)
{
// Do not check journal checksums in-memory
continue;
}
iovec single_iov = {};
iovec *iov = NULL;
int n_iov = 0;
if (rv[i].copy_flags & COPY_BUF_CSUM_FILL)
{
// Padded, buffer list passed using a 'creepy way'
iov = (struct iovec*)((uint8_t*)rv[i].buf + (rv[i].len & 0xFFFFFFFF));
n_iov = rv[i].len >> 32;
}
else
{
// Not padded, buffer is fully within the input buffer
assert(op->buf);
assert(rv[i].csum_buf);
iov = &single_iov;
n_iov = 1;
assert(rv[i].offset >= op->offset);
assert(rv[i].offset + rv[i].len <= op->offset + op->len);
single_iov = { .iov_base = op->buf + rv[i].offset - op->offset, .iov_len = rv[i].len };
}
bool ok = true;
if (rv[i].copy_flags & COPY_BUF_JOURNAL)
{
@@ -944,23 +1051,7 @@ void blockstore_impl_t::handle_read_event(ring_data_t *data, blockstore_op_t *op
meta_block = NULL;
}
}
if (PRIV(op)->clean_block_used)
{
// Release clean data block
auto uo_it = used_clean_objects.find(PRIV(op)->clean_block_used);
if (uo_it != used_clean_objects.end())
{
uo_it->second.refs--;
if (uo_it->second.refs <= 0)
{
if (uo_it->second.was_freed)
{
data_alloc->set(PRIV(op)->clean_block_used, false);
}
used_clean_objects.erase(uo_it);
}
}
}
release_clean(op);
if (!journal.inmemory)
{
// Release journal sector usage
+2 -2
View File
@@ -491,7 +491,7 @@ void blockstore_impl_t::mark_stable(obj_ver_id v, bool forget_dirty)
if (!exists)
{
uint64_t space_id = dirty_it->first.oid.inode;
if (no_inode_stats[dirty_it->first.oid.inode >> (64-POOL_ID_BITS)])
if (no_inode_stats.find(dirty_it->first.oid.inode >> (64-POOL_ID_BITS)) != no_inode_stats.end())
space_id = space_id & ~(((uint64_t)1 << (64-POOL_ID_BITS)) - 1);
inode_space_stats[space_id] += dsk.data_block_size;
used_blocks++;
@@ -501,7 +501,7 @@ void blockstore_impl_t::mark_stable(obj_ver_id v, bool forget_dirty)
else if (IS_DELETE(dirty_it->second.state))
{
uint64_t space_id = dirty_it->first.oid.inode;
if (no_inode_stats[dirty_it->first.oid.inode >> (64-POOL_ID_BITS)])
if (no_inode_stats.find(dirty_it->first.oid.inode >> (64-POOL_ID_BITS)) != no_inode_stats.end())
space_id = space_id & ~(((uint64_t)1 << (64-POOL_ID_BITS)) - 1);
auto & sp = inode_space_stats[space_id];
if (sp > dsk.data_block_size)
+58 -16
View File
@@ -3,6 +3,27 @@ cmake_minimum_required(VERSION 2.8...3.30)
project(vitastor)
# libvitastor_common.a
add_library(vitastor_common STATIC
etcd_state_client.cpp
msgr_stop.cpp
msgr_op.cpp
../../json11/json11.cpp
osd_ops.cpp
pg_states.cpp
msgr_encrypt.cpp
msgr_handshake.cpp
../util/allocator.cpp
../util/addr_util.cpp
../util/timerfd_manager.cpp
../util/str_util.cpp
../util/json_util.cpp
../util/xxh_x86dispatch.c
../util/openssl_util.cpp
)
target_compile_options(vitastor_common PUBLIC -fPIC)
target_link_libraries(vitastor_common ${OPENSSL_LIBRARIES} ${ISAL_CRYPTO_LIBRARIES})
# libvitastor_net.a
set(MSGR_RDMA "")
if (IBVERBS_LIBRARIES)
set(MSGR_RDMA "msgr_rdma.cpp")
@@ -11,25 +32,32 @@ set(MSGR_RDMACM "")
if (RDMACM_LIBRARIES)
set(MSGR_RDMACM "msgr_rdmacm.cpp")
endif (RDMACM_LIBRARIES)
add_library(vitastor_common STATIC
../util/epoll_manager.cpp etcd_state_client.cpp messenger.cpp msgr_iothread.cpp ../util/addr_util.cpp ../util/xxh_x86dispatch.c ../util/openssl_util.cpp
msgr_encrypt.cpp msgr_handshake.cpp msgr_stop.cpp msgr_op.cpp msgr_send.cpp msgr_receive.cpp ../util/ringloop.cpp ../../json11/json11.cpp
http_client.cpp osd_ops.cpp pg_states.cpp ../util/timerfd_manager.cpp ../util/str_util.cpp ../util/json_util.cpp ${MSGR_RDMA} ${MSGR_RDMACM}
add_library(vitastor_net STATIC
../util/epoll_manager.cpp
etcd_state_client_http.cpp
messenger.cpp
msgr_iothread.cpp
msgr_send.cpp
msgr_receive.cpp
msgr_encrypt.cpp
../util/ringloop.cpp
http_client.cpp
${MSGR_RDMA}
${MSGR_RDMACM}
)
target_link_libraries(vitastor_common pthread ${OPENSSL_LIBRARIES} ${CARES_LIBRARIES} ${ISAL_CRYPTO_LIBRARIES})
target_compile_options(vitastor_common PUBLIC -fPIC)
target_link_libraries(vitastor_net pthread vitastor_common ${CARES_LIBRARIES})
target_compile_options(vitastor_net PUBLIC -fPIC)
# libvitastor_client.so
add_library(vitastor_client SHARED
# libvitastor_client_int.a
add_library(vitastor_client_int STATIC
cluster_client.cpp
cluster_client_real.cpp
cluster_client_list.cpp
cluster_client_wb.cpp
cluster_client_icache.cpp
vitastor_c.cpp
)
set_target_properties(vitastor_client PROPERTIES PUBLIC_HEADER "client/vitastor_c.h")
target_link_libraries(vitastor_client
vitastor_common
target_link_libraries(vitastor_client_int
vitastor_net
vitastor_cli
${LIBURING_LIBRARIES}
${IBVERBS_LIBRARIES}
@@ -37,6 +65,16 @@ target_link_libraries(vitastor_client
${OPENSSL_LIBRARIES}
${ISAL_CRYPTO_LIBRARIES}
)
target_compile_options(vitastor_client_int PUBLIC -fPIC)
# libvitastor_client.so
add_library(vitastor_client SHARED
vitastor_c.cpp
)
set_target_properties(vitastor_client PROPERTIES PUBLIC_HEADER "client/vitastor_c.h")
target_link_libraries(vitastor_client
vitastor_client_int
)
set_target_properties(vitastor_client PROPERTIES VERSION ${VITASTOR_VERSION} SOVERSION 0)
configure_file(vitastor.pc.in vitastor.pc @ONLY)
@@ -98,11 +136,15 @@ endif (${WITH_QEMU})
add_executable(test_cluster_client
EXCLUDE_FROM_ALL
../test/test_cluster_client.cpp
pg_states.cpp osd_ops.cpp cluster_client.cpp cluster_client_list.cpp cluster_client_wb.cpp cluster_client_icache.cpp msgr_op.cpp ../test/mock/messenger.cpp msgr_stop.cpp msgr_encrypt.cpp
etcd_state_client.cpp ../util/timerfd_manager.cpp ../util/addr_util.cpp ../util/str_util.cpp ../util/json_util.cpp ../util/xxh_x86dispatch.c ../util/openssl_util.cpp ../../json11/json11.cpp
cluster_client.cpp
cluster_client_list.cpp
cluster_client_wb.cpp
cluster_client_icache.cpp
../test/mock/messenger.cpp
../test/mock/vault.cpp
etcd_state_client_mock.cpp
)
target_link_libraries(test_cluster_client ${LIBURING_LIBRARIES} ${OPENSSL_LIBRARIES} ${ISAL_CRYPTO_LIBRARIES})
target_compile_definitions(test_cluster_client PUBLIC -D__MOCK__)
target_link_libraries(test_cluster_client vitastor_common ${LIBURING_LIBRARIES} ${OPENSSL_LIBRARIES} ${ISAL_CRYPTO_LIBRARIES})
target_include_directories(test_cluster_client BEFORE PUBLIC ${CMAKE_SOURCE_DIR}/src/test/mock)
add_dependencies(build_tests test_cluster_client)
add_test(NAME test_cluster_client COMMAND test_cluster_client)
+40 -40
View File
@@ -11,7 +11,7 @@
#define TRY_SEND_CONNECTING 1
#define TRY_SEND_OK 2
cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config)
cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config, std::unique_ptr<etcd_state_client_t> st_cli_ptr)
{
wb = new writeback_cache_t();
@@ -53,24 +53,24 @@ cluster_client_t::cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd
};
msgr.parse_config(config, true);
st_cli.tfd = tfd;
st_cli.on_load_config_hook = [this](json11::Json::object & cfg) { on_load_config_hook(cfg); };
st_cli.on_change_osd_state_hook = [this](uint64_t peer_osd) { on_change_osd_state_hook(peer_osd); };
st_cli.on_change_pool_config_hook = [this]() { on_change_pool_config_hook(); };
st_cli.on_change_pg_config_hook = [this]() { on_change_pool_config_hook(); };
st_cli.on_change_pg_state_hook = [this](pool_id_t pool_id, pg_num_t pg_num, osd_num_t prev_primary) { on_change_pg_state_hook(pool_id, pg_num, prev_primary); };
st_cli.on_change_node_placement_hook = [this]() { on_change_node_placement_hook(); };
st_cli.on_load_pgs_hook = [this](bool success) { on_load_pgs_hook(success); };
st_cli.on_reload_hook = [this]() { st_cli.load_global_config(); };
st_cli.on_inode_change_hook = [this](uint64_t inode, bool removed) { on_change_inode_hook(inode, removed); };
st_cli = std::move(st_cli_ptr);
st_cli->on_load_config_hook = [this](json11::Json::object & cfg) { on_load_config_hook(cfg); };
st_cli->on_change_osd_state_hook = [this](uint64_t peer_osd) { on_change_osd_state_hook(peer_osd); };
st_cli->on_change_pool_config_hook = [this]() { on_change_pool_config_hook(); };
st_cli->on_change_pg_config_hook = [this]() { on_change_pool_config_hook(); };
st_cli->on_change_pg_state_hook = [this](pool_id_t pool_id, pg_num_t pg_num, osd_num_t prev_primary) { on_change_pg_state_hook(pool_id, pg_num, prev_primary); };
st_cli->on_change_node_placement_hook = [this]() { on_change_node_placement_hook(); };
st_cli->on_load_pgs_hook = [this](bool success) { on_load_pgs_hook(success); };
st_cli->on_reload_hook = [this]() { this->st_cli->load_global_config(); };
st_cli->on_inode_change_hook = [this](uint64_t inode, bool removed) { on_change_inode_hook(inode, removed); };
st_cli.parse_config(config);
st_cli.infinite_start = false;
st_cli->parse_config(config);
st_cli->infinite_start = false;
if (!config["client_infinite_start"].is_null())
{
st_cli.infinite_start = config["client_infinite_start"].bool_value();
st_cli->infinite_start = config["client_infinite_start"].bool_value();
}
st_cli.load_global_config();
st_cli->load_global_config();
}
cluster_client_t::~cluster_client_t()
@@ -467,7 +467,7 @@ void cluster_client_t::on_load_config_hook(json11::Json::object & etcd_global_co
auto etcd_report_interval = config["etcd_report_interval"].uint64_value();
if (!etcd_report_interval)
etcd_report_interval = 5;
client_wait_up_timeout = 1+etcd_report_interval+(st_cli.max_etcd_attempts*(2*st_cli.etcd_quick_timeout)+999)/1000;
client_wait_up_timeout = 1+etcd_report_interval+(st_cli->max_etcd_attempts*(2*st_cli->etcd_quick_timeout)+999)/1000;
}
// log_level
log_level = config["log_level"].uint64_value();
@@ -482,8 +482,8 @@ void cluster_client_t::on_load_config_hook(json11::Json::object & etcd_global_co
// vault
vault_parse_config();
msgr.parse_config(config, false);
st_cli.parse_config(config);
st_cli.load_pgs();
st_cli->parse_config(config);
st_cli->load_pgs();
}
osd_num_t cluster_client_t::select_random_osd(const std::vector<osd_num_t> & osds)
@@ -492,7 +492,7 @@ osd_num_t cluster_client_t::select_random_osd(const std::vector<osd_num_t> & osd
int alive_count = 0;
for (auto & osd_num: osds)
{
if (!st_cli.peer_states[osd_num].is_null())
if (!st_cli->peer_states[osd_num].is_null())
alive_set[alive_count++] = osd_num;
}
if (!alive_count)
@@ -509,7 +509,7 @@ osd_num_t cluster_client_t::select_nearest_osd(const std::vector<osd_num_t> & os
while (self_tree_metrics.find(cur_id) == self_tree_metrics.end())
{
self_tree_metrics[cur_id] = metric++;
json11::Json cur_placement = st_cli.node_placement[cur_id];
json11::Json cur_placement = st_cli->node_placement[cur_id];
cur_id = cur_placement["parent"].string_value();
}
if (cur_id != "")
@@ -529,7 +529,7 @@ osd_num_t cluster_client_t::select_nearest_osd(const std::vector<osd_num_t> & os
}
else
{
auto & peer_state = st_cli.peer_states[osd_num];
auto & peer_state = st_cli->peer_states[osd_num];
if (!peer_state.is_null())
{
metric = self_tree_metrics[""];
@@ -539,7 +539,7 @@ osd_num_t cluster_client_t::select_nearest_osd(const std::vector<osd_num_t> & os
while (seen.find(cur_id) == seen.end())
{
seen.insert(cur_id);
json11::Json cur_placement = st_cli.node_placement[cur_id];
json11::Json cur_placement = st_cli->node_placement[cur_id];
std::string cur_parent = cur_placement["parent"].string_value();
cur_id = (!first || cur_parent != "" ? cur_parent : peer_state["host"].string_value());
first = false;
@@ -564,7 +564,7 @@ osd_num_t cluster_client_t::select_nearest_osd(const std::vector<osd_num_t> & os
void cluster_client_t::on_load_pgs_hook(bool success)
{
for (auto & pool_item: st_cli.pool_config)
for (auto & pool_item: st_cli->pool_config)
{
pg_counts[pool_item.first] = pool_item.second.real_pg_count;
}
@@ -584,7 +584,7 @@ void cluster_client_t::on_load_pgs_hook(bool success)
void cluster_client_t::on_change_pool_config_hook()
{
for (auto & pool_item: st_cli.pool_config)
for (auto & pool_item: st_cli->pool_config)
{
if (pg_counts[pool_item.first] != pool_item.second.real_pg_count)
{
@@ -615,7 +615,7 @@ void cluster_client_t::on_change_pool_config_hook()
void cluster_client_t::on_change_pg_state_hook(pool_id_t pool_id, pg_num_t pg_num, osd_num_t prev_primary)
{
auto & pg_cfg = st_cli.pool_config[pool_id].pg_config[pg_num];
auto & pg_cfg = st_cli->pool_config[pool_id].pg_config[pg_num];
if (pg_cfg.cur_primary != prev_primary)
{
// Repeat this PG operations because an OSD which stopped being primary may not fsync operations
@@ -633,8 +633,8 @@ bool cluster_client_t::get_immediate_commit(uint64_t inode)
pool_id_t pool_id = INODE_POOL(inode);
if (!pool_id)
return true;
auto pool_it = st_cli.pool_config.find(pool_id);
if (pool_it == st_cli.pool_config.end())
auto pool_it = st_cli->pool_config.find(pool_id);
if (pool_it == st_cli->pool_config.end())
return true;
return pool_it->second.immediate_commit == IMMEDIATE_ALL;
}
@@ -644,7 +644,7 @@ void cluster_client_t::on_change_osd_state_hook(uint64_t peer_osd)
osd_tree_metrics.erase(peer_osd);
if (msgr.wanted_peers.find(peer_osd) != msgr.wanted_peers.end())
{
msgr.connect_peer(peer_osd, st_cli.peer_states[peer_osd]);
msgr.connect_peer(peer_osd, st_cli->peer_states[peer_osd]);
continue_lists();
}
}
@@ -943,8 +943,8 @@ bool cluster_client_t::check_rw(cluster_op_t *op)
cb(op);
return false;
}
auto pool_it = st_cli.pool_config.find(pool_id);
if (pool_it == st_cli.pool_config.end() || pool_it->second.real_pg_count == 0)
auto pool_it = st_cli->pool_config.find(pool_id);
if (pool_it == st_cli->pool_config.end() || pool_it->second.real_pg_count == 0)
{
// Pools are loaded, but this one is unknown
op->retval = -EINVAL;
@@ -1057,7 +1057,7 @@ void cluster_client_t::execute_raw(osd_num_t osd_num, osd_op_t *op)
else
{
if (msgr.wanted_peers.find(osd_num) == msgr.wanted_peers.end())
msgr.connect_peer(osd_num, st_cli.peer_states[osd_num]);
msgr.connect_peer(osd_num, st_cli->peer_states[osd_num]);
raw_ops.emplace(osd_num, op);
}
}
@@ -1206,7 +1206,7 @@ resume_2:
op->retval = op->len;
if (op->opcode == OSD_OP_READ_BITMAP || op->opcode == OSD_OP_READ_CHAIN_BITMAP)
{
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->inode));
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(op->inode));
op->retval = op->len / pool_cfg.bitmap_granularity;
}
if (op->flush_id)
@@ -1286,7 +1286,7 @@ void cluster_client_t::slice_rw(cluster_op_t *op)
{
// Slice the request into individual object stripe requests
// Primary OSDs still operate individual stripes, but their size is multiplied by PG minsize in case of EC
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->cur_inode));
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(op->cur_inode));
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
uint64_t pg_block_size = pool_cfg.data_block_size * pg_data_size;
uint64_t first_stripe = (op->offset / pg_block_size) * pg_block_size;
@@ -1389,7 +1389,7 @@ bool cluster_client_t::affects_pg(uint64_t inode, uint64_t offset, uint64_t len,
{
return false;
}
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(inode));
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(inode));
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
uint64_t pg_block_size = pool_cfg.data_block_size * pg_data_size;
uint64_t first_stripe = (offset / pg_block_size) * pg_block_size;
@@ -1408,7 +1408,7 @@ bool cluster_client_t::affects_pg(uint64_t inode, uint64_t offset, uint64_t len,
bool cluster_client_t::affects_osd(uint64_t inode, uint64_t offset, uint64_t len, osd_num_t osd)
{
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(inode));
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(inode));
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
uint64_t pg_block_size = pool_cfg.data_block_size * pg_data_size;
uint64_t first_stripe = (offset / pg_block_size) * pg_block_size;
@@ -1432,7 +1432,7 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
init_msgr();
}
auto part = &op->parts[i];
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->cur_inode));
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(op->cur_inode));
auto pg_it = pool_cfg.pg_config.find(part->pg_num);
if (pg_it != pool_cfg.pg_config.end() &&
!pg_it->second.pause && pg_it->second.cur_primary &&
@@ -1466,8 +1466,8 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
uint64_t meta_rev = 0;
if (op->opcode != OSD_OP_READ_BITMAP && op->opcode != OSD_OP_DELETE && !op->deoptimise_snapshot)
{
auto ino_it = st_cli.inode_config.find(op->cur_inode);
if (ino_it != st_cli.inode_config.end())
auto ino_it = st_cli->inode_config.find(op->cur_inode);
if (ino_it != st_cli->inode_config.end())
meta_rev = ino_it->second.mod_revision;
}
part->op = (osd_op_t){
@@ -1501,7 +1501,7 @@ int cluster_client_t::try_send(cluster_op_t *op, int i, std::function<void(osd_o
}
else if (msgr.wanted_peers.find(primary_osd) == msgr.wanted_peers.end())
{
msgr.connect_peer(primary_osd, st_cli.peer_states[primary_osd]);
msgr.connect_peer(primary_osd, st_cli->peer_states[primary_osd]);
return TRY_SEND_CONNECTING;
}
}
@@ -1694,7 +1694,7 @@ void cluster_client_t::handle_op_part(cluster_op_part_t *part)
void cluster_client_t::copy_part_bitmap(cluster_op_t *op, cluster_op_part_t *part)
{
// Copy (OR) bitmap
auto & pool_cfg = st_cli.pool_config.at(INODE_POOL(op->cur_inode));
auto & pool_cfg = st_cli->pool_config.at(INODE_POOL(op->cur_inode));
uint32_t pg_block_size = pool_cfg.data_block_size * (
pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks
);
+4 -3
View File
@@ -4,7 +4,7 @@
#pragma once
#include "messenger.h"
#include "etcd_state_client.h"
#include "etcd_state_client_http.h"
#include "../util/robin_hood.h"
#define DEFAULT_CLIENT_MAX_DIRTY_BYTES 32*1024*1024
@@ -176,7 +176,7 @@ class __attribute__((visibility("default"))) cluster_client_t
bool msgr_initialized = false;
public:
etcd_state_client_t st_cli;
std::unique_ptr<etcd_state_client_t> st_cli;
osd_messenger_t msgr;
void init_msgr();
@@ -184,7 +184,8 @@ public:
json11::Json::object cli_config, file_config, etcd_global_config;
json11::Json::object config;
cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config);
static cluster_client_t* create(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config);
cluster_client_t(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config, std::unique_ptr<etcd_state_client_t> st_cli);
~cluster_client_t();
void execute(cluster_op_t *op);
void execute_raw(osd_num_t osd_num, osd_op_t *op);
+11 -133
View File
@@ -4,14 +4,8 @@
#include <stdexcept>
#include <assert.h>
#include "cluster_client_impl.h"
#include "http_client.h"
#include "str_util.h"
#define VAULT_KEY_NOT_LOADED 0
#define VAULT_KEY_LOADING 1
#define VAULT_KEY_LOADED 2
#define VAULT_KEY_ERROR 3
inode_cache_t::~inode_cache_t()
{
if (key_data)
@@ -22,24 +16,15 @@ inode_cache_t::~inode_cache_t()
}
}
void cluster_client_t::vault_destroy()
{
if (vault_http_ctx)
{
#ifndef __MOCK__
http_destroy(vault_http_cli);
http_context_destroy(vault_http_ctx);
vault_http_cli = NULL;
vault_http_ctx = NULL;
#endif
}
}
void cluster_client_t::vault_parse_config()
{
vault_url = config["vault_url"].string_value();
vault_client_cert = config["vault_client_cert"].string_value();
if (vault_client_cert.empty())
vault_client_cert = config["cert"].string_value();
vault_client_key = config["vault_client_key"].string_value();
if (vault_client_key.empty())
vault_client_key = config["pkey"].string_value();
vault_ca = config["vault_ca"].string_value();
vault_secret_api_path = "/v1/secret/";
if (config["vault_secret_api_path"].is_string())
@@ -91,14 +76,14 @@ std::shared_ptr<inode_cache_t> cluster_client_t::inode_cache_get(inode_t ino)
return icache_it->second;
}
// Fill inode cache
auto ino_it = st_cli.inode_config.find(ino);
if (ino_it == st_cli.inode_config.end())
auto ino_it = st_cli->inode_config.find(ino);
if (ino_it == st_cli->inode_config.end())
{
inode_cache[ino] = NULL;
return NULL;
}
auto pool_it = st_cli.pool_config.find(INODE_POOL(ino));
if (pool_it == st_cli.pool_config.end())
auto pool_it = st_cli->pool_config.find(INODE_POOL(ino));
if (pool_it == st_cli->pool_config.end())
{
inode_cache[ino] = NULL;
return NULL;
@@ -125,11 +110,11 @@ std::shared_ptr<inode_cache_t> cluster_client_t::inode_cache_get(inode_t ino)
break;
}
seen.insert(parent_id);
ino_it = st_cli.inode_config.find(parent_id);
ino_it = st_cli->inode_config.find(parent_id);
if (INODE_POOL(parent_id) == INODE_POOL(ino))
{
icache->chain.push_back(parent_id);
if (ino_it == st_cli.inode_config.end())
if (ino_it == st_cli->inode_config.end())
chain_cfg.push_back(NULL);
else
{
@@ -140,7 +125,7 @@ std::shared_ptr<inode_cache_t> cluster_client_t::inode_cache_get(inode_t ino)
}
else if (!icache->other_pool_parent_id)
icache->other_pool_parent_id = parent_id;
if (ino_it == st_cli.inode_config.end())
if (ino_it == st_cli->inode_config.end())
break;
parent_id = ino_it->second.parent_id;
}
@@ -223,113 +208,6 @@ std::shared_ptr<inode_cache_t> cluster_client_t::inode_cache_get(inode_t ino)
return icache;
}
#ifndef __MOCK__
bool cluster_client_t::vault_check_token()
{
timespec now;
clock_gettime(CLOCK_REALTIME, &now);
if (!vault_token_expire.tv_sec || vault_token_expire.tv_sec < now.tv_sec)
{
vault_loading = true;
http_json_post(
vault_http_cli, vault_url+"/v1/auth/cert/login", json11::Json::object{}, "",
(http_options_t){ .timeout = (int)vault_timeout_ms, .keepalive = true },
[this](http_message_t *response)
{
clock_gettime(CLOCK_REALTIME, &vault_token_expire);
vault_loading = false;
std::string err;
json11::Json data;
response->parse_json_response(err, data);
if (err != "")
{
vault_token_expire.tv_sec += vault_error_timeout_sec;
fprintf(stderr, "Vault request failed: %s\n", err.c_str());
}
else
{
uint64_t ttl = data["auth"]["lease_duration"].uint64_value();
vault_token = data["auth"]["client_token"].string_value();
if (vault_token.empty() || !ttl)
{
vault_token_expire.tv_sec += vault_error_timeout_sec;
fprintf(stderr, "No token or lease_duration in Vault response: %s\n", data.dump().c_str());
}
else
{
if (ttl < vault_refresh_leeway_sec)
vault_token_expire.tv_sec += ttl/2;
else
vault_token_expire.tv_sec += ttl - vault_refresh_leeway_sec;
}
}
vault_load_keys();
}
);
return false;
}
if (vault_token.empty())
{
// Auth error happened, mark all loads as failed
for (auto & key_id: vault_key_load_queue)
{
auto & k = vault_keys[key_id];
k.key_state = VAULT_KEY_ERROR;
}
vault_key_load_queue.clear();
auto ops = std::move(key_wait_ops);
for (cluster_op_t *op: ops)
inode_cache.erase(op->inode);
for (cluster_op_t *op: ops)
execute_internal(op);
return false;
}
return true;
}
#endif
void cluster_client_t::vault_load_keys()
{
if (vault_loading || !vault_key_load_queue.size())
{
return;
}
#ifdef __MOCK__
vault_loading = true;
#else
if (!vault_http_ctx)
{
std::string error;
vault_http_ctx = http_context_init(tfd, vault_client_cert, vault_client_key, vault_ca, true, error);
if (!vault_http_ctx)
{
fprintf(stderr, "Failed to initialize HTTP context for Vault: %s\n", error.c_str());
exit(1);
}
vault_http_cli = http_init(vault_http_ctx);
}
if (!vault_check_token())
{
return;
}
std::string key_id = vault_key_load_queue[0];
vault_key_load_queue.erase(vault_key_load_queue.begin());
vault_loading = true;
http_get(
vault_http_cli, vault_url+vault_secret_api_path+key_id.substr(strlen(VAULT_KEY_PREFIX)), "X-Vault-Token: "+vault_token+"\r\n",
(http_options_t){ .timeout = (int)vault_timeout_ms, .keepalive = true },
[this, key_id](http_message_t *response)
{
vault_loading = false;
std::string err;
json11::Json data;
response->parse_json_response(err, data);
vault_parse_secret(key_id, err, data);
}
);
#endif
}
void cluster_client_t::vault_parse_secret(const std::string & key_id, const std::string & err, json11::Json data)
{
vault_loading = false;
+5
View File
@@ -17,6 +17,11 @@
#define OP_FLUSH_BUFFER 0x02
#define OP_IMMEDIATE_COMMIT 0x04
#define VAULT_KEY_NOT_LOADED 0
#define VAULT_KEY_LOADING 1
#define VAULT_KEY_LOADED 2
#define VAULT_KEY_ERROR 3
struct cluster_buffer_t
{
uint8_t *buf;
+10 -10
View File
@@ -63,14 +63,14 @@ void cluster_client_t::list_inode(inode_t inode, uint64_t min_offset, uint64_t m
{
init_msgr();
pool_id_t pool_id = INODE_POOL(inode);
if (!pool_id || st_cli.pool_config.find(pool_id) == st_cli.pool_config.end())
if (!pool_id || st_cli->pool_config.find(pool_id) == st_cli->pool_config.end())
{
if (log_level > 0)
fprintf(stderr, "Pool %u does not exist\n", pool_id);
pg_callback(-EINVAL, 0, 0, std::set<object_id>());
return;
}
auto pg_stripe_size = st_cli.pool_config.at(pool_id).pg_stripe_size;
auto pg_stripe_size = st_cli->pool_config.at(pool_id).pg_stripe_size;
if (min_offset)
min_offset = (min_offset/pg_stripe_size) * pg_stripe_size;
inode_list_t *lst = new inode_list_t();
@@ -110,13 +110,13 @@ bool cluster_client_t::continue_listing(inode_list_t *lst)
bool cluster_client_t::restart_listing(inode_list_t* lst)
{
auto pool_it = st_cli.pool_config.find(lst->pool_id);
auto pool_it = st_cli->pool_config.find(lst->pool_id);
// We want listing to be consistent. To achieve it we should:
// 1) retry listing of each PG if its state changes
// 2) abort listing if PG count changes during listing
// 3) ideally, only talk to the primary OSD - this will be done separately
// So first we add all PGs without checking their state
if (pool_it == st_cli.pool_config.end() ||
if (pool_it == st_cli->pool_config.end() ||
lst->real_pg_count != pool_it->second.real_pg_count)
{
for (auto pg: lst->pgs)
@@ -136,7 +136,7 @@ bool cluster_client_t::restart_listing(inode_list_t* lst)
fprintf(stderr, "PG count in pool %u changed during listing\n", lst->pool_id);
}
lst->pgs.clear();
if (pool_it == st_cli.pool_config.end())
if (pool_it == st_cli->pool_config.end())
{
// Unknown pool
lst->callback(-EINVAL, 0, 0, std::set<object_id>());
@@ -248,7 +248,7 @@ void cluster_client_t::set_list_retry_timeout(int ms, timespec new_time)
int cluster_client_t::start_pg_listing(inode_list_pg_t *pg)
{
auto & pool_cfg = st_cli.pool_config.at(pg->lst->pool_id);
auto & pool_cfg = st_cli->pool_config.at(pg->lst->pool_id);
auto pg_it = pool_cfg.pg_config.find(pg->pg_num);
assert(pg->lst->real_pg_count == pool_cfg.real_pg_count);
if (pg_it == pool_cfg.pg_config.end() ||
@@ -277,7 +277,7 @@ int cluster_client_t::start_pg_listing(inode_list_pg_t *pg)
for (auto peer_it = all_peers.begin(); peer_it != all_peers.end(); )
{
if (*peer_it != pg_it->second.cur_primary &&
st_cli.peer_states[*peer_it].is_null())
st_cli->peer_states[*peer_it].is_null())
{
pg->inactive_osds.push_back(*peer_it);
all_peers.erase(peer_it++);
@@ -298,11 +298,11 @@ int cluster_client_t::start_pg_listing(inode_list_pg_t *pg)
if (msgr.osd_peers.find(peer_osd) == msgr.osd_peers.end())
{
// Initiate connection
if (st_cli.peer_states[peer_osd].is_null())
if (st_cli->peer_states[peer_osd].is_null())
{
return LIST_PG_WAIT_ACTIVE;
}
msgr.connect_peer(peer_osd, st_cli.peer_states[peer_osd]);
msgr.connect_peer(peer_osd, st_cli->peer_states[peer_osd]);
conn = false;
}
}
@@ -336,7 +336,7 @@ void cluster_client_t::send_list(inode_list_osd_t *cur_list)
if (!cur_list->pg->inflight_ops)
cur_list->pg->lst->inflight_pgs++;
cur_list->pg->inflight_ops++;
auto & pool_cfg = st_cli.pool_config[cur_list->pg->lst->pool_id];
auto & pool_cfg = st_cli->pool_config[cur_list->pg->lst->pool_id];
osd_op_t *op = new osd_op_t();
op->op_type = OSD_OP_OUT;
// Already checked that it exists above, but anyway
+125
View File
@@ -0,0 +1,125 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
#include "cluster_client.h"
#include "cluster_client_impl.h"
#include "etcd_state_client_http.h"
#include "http_client.h"
cluster_client_t* cluster_client_t::create(ring_loop_t *ringloop, timerfd_manager_t *tfd, json11::Json config)
{
auto st_cli = new etcd_state_client_http_t(tfd);
return new cluster_client_t(ringloop, tfd, config, std::unique_ptr<etcd_state_client_t>(st_cli));
}
bool cluster_client_t::vault_check_token()
{
timespec now;
clock_gettime(CLOCK_REALTIME, &now);
if (!vault_token_expire.tv_sec || vault_token_expire.tv_sec < now.tv_sec)
{
vault_loading = true;
http_json_post(
vault_http_cli, vault_url+"/v1/auth/cert/login", json11::Json::object{}, "",
(http_options_t){ .timeout = (int)vault_timeout_ms, .keepalive = true },
[this](http_message_t *response)
{
clock_gettime(CLOCK_REALTIME, &vault_token_expire);
vault_loading = false;
std::string err;
json11::Json data;
response->parse_json_response(err, data);
if (err != "")
{
vault_token_expire.tv_sec += vault_error_timeout_sec;
fprintf(stderr, "Vault request failed: %s\n", err.c_str());
}
else
{
uint64_t ttl = data["auth"]["lease_duration"].uint64_value();
vault_token = data["auth"]["client_token"].string_value();
if (vault_token.empty() || !ttl)
{
vault_token_expire.tv_sec += vault_error_timeout_sec;
fprintf(stderr, "No token or lease_duration in Vault response: %s\n", data.dump().c_str());
}
else
{
if (ttl < vault_refresh_leeway_sec)
vault_token_expire.tv_sec += ttl/2;
else
vault_token_expire.tv_sec += ttl - vault_refresh_leeway_sec;
}
}
vault_load_keys();
}
);
return false;
}
if (vault_token.empty())
{
// Auth error happened, mark all loads as failed
for (auto & key_id: vault_key_load_queue)
{
auto & k = vault_keys[key_id];
k.key_state = VAULT_KEY_ERROR;
}
vault_key_load_queue.clear();
auto ops = std::move(key_wait_ops);
for (cluster_op_t *op: ops)
inode_cache.erase(op->inode);
for (cluster_op_t *op: ops)
execute_internal(op);
return false;
}
return true;
}
void cluster_client_t::vault_destroy()
{
if (vault_http_ctx)
{
http_destroy(vault_http_cli);
http_context_destroy(vault_http_ctx);
vault_http_cli = NULL;
vault_http_ctx = NULL;
}
}
void cluster_client_t::vault_load_keys()
{
if (vault_loading || !vault_key_load_queue.size())
{
return;
}
if (!vault_http_ctx)
{
std::string error;
vault_http_ctx = http_context_init(tfd, vault_client_cert, vault_client_key, vault_ca, true, error);
if (!vault_http_ctx)
{
fprintf(stderr, "Failed to initialize HTTP context for Vault: %s\n", error.c_str());
exit(1);
}
vault_http_cli = http_init(vault_http_ctx);
}
if (!vault_check_token())
{
return;
}
std::string key_id = vault_key_load_queue[0];
vault_key_load_queue.erase(vault_key_load_queue.begin());
vault_loading = true;
http_get(
vault_http_cli, vault_url+vault_secret_api_path+key_id.substr(strlen(VAULT_KEY_PREFIX)), "X-Vault-Token: "+vault_token+"\r\n",
(http_options_t){ .timeout = (int)vault_timeout_ms, .keepalive = true },
[this, key_id](http_message_t *response)
{
vault_loading = false;
std::string err;
json11::Json data;
response->parse_json_response(err, data);
vault_parse_secret(key_id, err, data);
}
);
}
+2
View File
@@ -131,6 +131,7 @@ void writeback_cache_t::copy_write(cluster_op_t *op, int state, uint64_t new_flu
writeback_bytes -= op->len;
}
writeback_queue_size++;
writeback_queue.push_back({ op->inode, new_end });
}
break;
}
@@ -165,6 +166,7 @@ void writeback_cache_t::copy_write(cluster_op_t *op, int state, uint64_t new_flu
{
writeback_queue_size++;
}
writeback_queue.push_back({ op->inode, new_end });
}
auto new_dirty_it = dirty_buffers.emplace_hint(dirty_it, (object_id){
.inode = op->inode,
+18 -516
View File
@@ -1,16 +1,12 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
#include <assert.h>
#include "malloc_or_die.h"
#include "osd_ops.h"
#include "msgr_op.h"
#include "pg_states.h"
#include "etcd_state_client.h"
#ifndef __MOCK__
#include "addr_util.h"
#include "http_client.h"
#endif
#include "str_util.h"
#include "json_util.h"
@@ -21,33 +17,8 @@ etcd_state_client_t::~etcd_state_client_t()
delete watch;
}
watches.clear();
etcd_watches_initialised = -1;
#ifndef __MOCK__
stop_ws_keepalive();
if (etcd_watch_ws)
{
http_destroy(etcd_watch_ws);
etcd_watch_ws = NULL;
}
if (keepalive_client)
{
http_destroy(keepalive_client);
keepalive_client = NULL;
}
if (http_ctx)
{
http_context_destroy(http_ctx);
http_ctx = NULL;
}
#endif
if (load_pgs_timer_id >= 0)
{
tfd->clear_timer(load_pgs_timer_id);
load_pgs_timer_id = -1;
}
}
#ifndef __MOCK__
etcd_kv_t etcd_state_client_t::parse_etcd_kv(const json11::Json & kv_json)
{
etcd_kv_t kv;
@@ -121,99 +92,6 @@ bool etcd_state_client_t::check_image_perm(const std::shared_ptr<user_info_t> &
return write ? (perm_item.perm == user_perm_t::OWNER) : (perm_item.perm != user_perm_t::DENY);
}
http_context_t *etcd_state_client_t::get_http_ctx()
{
if (!http_ctx)
{
std::string error;
http_ctx = http_context_init(tfd, etcd_client_cert, etcd_client_key, etcd_ca, true, error);
if (!http_ctx)
{
fprintf(stderr, "Failed to initialize HTTP context: %s\n", error.c_str());
exit(1);
}
}
return http_ctx;
}
void etcd_state_client_t::etcd_call_oneshot(const std::string & etcd_url, const std::string & api, json11::Json payload,
int timeout, std::function<void(std::string, json11::Json)> callback)
{
auto http_cli = http_init(get_http_ctx());
http_json_post(http_cli, etcd_url+api, payload, "", { .timeout = timeout }, [http_cli, callback](http_message_t *response)
{
std::string err;
json11::Json data;
response->parse_json_response(err, data);
callback(err, data);
http_destroy(http_cli);
});
}
void etcd_state_client_t::etcd_call(const std::string & api, json11::Json payload, int timeout,
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
{
pick_next_etcd([=]()
{
etcd_call_selected(api, payload, timeout, retries, interval, callback);
});
}
void etcd_state_client_t::etcd_call_selected(const std::string & api, json11::Json payload, int timeout,
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
{
const auto & url = selected_etcd_url;
std::string req = payload.dump();
req = "POST "+url.path+api+" HTTP/1.1\r\n"
"Host: "+url.hostname+"\r\n"
"Content-Type: application/json\r\n"
"Content-Length: "+std::to_string(req.size())+"\r\n"
"Connection: keep-alive\r\n"
"Keep-Alive: timeout="+std::to_string(etcd_keepalive_timeout)+"\r\n"
"\r\n"+req;
retries--;
auto cb = [this, api, payload, timeout, retries, interval, callback,
cur_addr = url.addr](http_message_t *response)
{
std::string err;
json11::Json data;
response->parse_json_response(err, data);
if (err != "")
{
if (cur_addr == selected_etcd_url.addr)
selected_etcd_url = (http_url_t){};
if (retries > 0)
{
if (this->log_level > 0)
{
fprintf(
stderr, "Warning: etcd request failed: %s, retrying %d more times\n",
err.c_str(), retries
);
}
if (interval > 0)
{
// FIXME: Prevent destruction of etcd_state_client if timers or requests are active
tfd->set_timer(interval, false, [this, api, payload, timeout, retries, interval, callback](int)
{
etcd_call(api, payload, timeout, retries, interval, callback);
});
}
else
etcd_call(api, payload, timeout, retries, interval, callback);
}
else
callback(err, data);
}
else
callback(err, data);
};
if (!keepalive_client)
keepalive_client = http_init(get_http_ctx());
http_request(keepalive_client, url.addr, req, { .timeout = timeout, .keepalive = true, .ssl = url.ssl }, cb);
}
void etcd_state_client_t::add_etcd_url(std::string etcd_address)
{
if (etcd_address.size() > 0)
@@ -324,7 +202,6 @@ void etcd_state_client_t::parse_config(const json11::Json & config)
if (this->etcd_keepalive_timeout < 30)
this->etcd_keepalive_timeout = 30;
}
auto old_etcd_ws_keepalive_interval = this->etcd_ws_keepalive_interval;
this->etcd_ws_keepalive_interval = config["etcd_ws_keepalive_interval"].uint64_value();
if (this->etcd_ws_keepalive_interval <= 0)
{
@@ -350,347 +227,9 @@ void etcd_state_client_t::parse_config(const json11::Json & config)
{
this->etcd_min_reload_interval = 50;
}
if (this->etcd_ws_keepalive_interval != old_etcd_ws_keepalive_interval && ws_keepalive_timer >= 0)
{
#ifndef __MOCK__
stop_ws_keepalive();
start_ws_keepalive();
#endif
}
}
void etcd_state_client_t::pick_next_etcd(std::function<void()> cb)
{
if (!etcd_addresses.size() && !etcd_local.size())
{
fprintf(stderr, "etcd_address is missing in Vitastor configuration\n");
exit(1);
}
if (selected_etcd_url.addr != "")
{
cb();
return;
}
if (etcd_urls_to_try.size() != 0)
{
selected_etcd_url = std::move(etcd_urls_to_try[0]);
etcd_urls_to_try.erase(etcd_urls_to_try.begin());
cb();
return;
}
on_resolve_queue.push_back(std::move(cb));
if (on_resolve_queue.size() > 1)
{
// Already resolving
return;
}
assert(!resolve_count);
local_to_try = 0;
for (auto & url: etcd_local_addr_urls)
{
// Prefer local IPs, if any
etcd_urls_to_try.push_back(url);
local_to_try++;
}
for (auto & url: etcd_nonlocal_addr_urls)
{
etcd_urls_to_try.push_back(url);
}
resolve_count++;
for (auto & url: etcd_name_urls)
{
resolve_count++;
http_resolve(get_http_ctx(), url.ssl, url.addr, [this, url](const std::string & error, const std::vector<std::string>& addresses)
{
if (error != "")
fprintf(stderr, "Error resolving %s: %s\n", url.addr.c_str(), error.c_str());
for (auto & addr: addresses)
{
auto url_copy = url;
url_copy.addr = addr;
if (local_ips.find(addr) != local_ips.end())
{
etcd_urls_to_try.insert(etcd_urls_to_try.begin(), std::move(url_copy));
local_to_try++;
}
else
etcd_urls_to_try.push_back(std::move(url_copy));
}
resolve_count--;
if (!resolve_count)
pick_next_etcd_on_resolve();
});
}
resolve_count--;
if (!resolve_count)
{
pick_next_etcd_on_resolve();
}
}
void etcd_state_client_t::pick_next_etcd_on_resolve()
{
if (!etcd_urls_to_try.size())
{
fprintf(stderr, "None of etcd_address could be resolved\n");
exit(1);
}
if (!rand_initialized)
{
timespec tv;
clock_gettime(CLOCK_REALTIME, &tv);
srand48(tv.tv_sec*1000000000 + tv.tv_nsec);
rand_initialized = true;
}
// Shuffle addresses
for (size_t i = etcd_urls_to_try.size()-1; i > local_to_try; i--)
{
size_t j = local_to_try + lrand48() % (i - local_to_try);
if (j != i)
std::swap(etcd_urls_to_try[i], etcd_urls_to_try[j]);
}
selected_etcd_url = std::move(etcd_urls_to_try[0]);
etcd_urls_to_try.erase(etcd_urls_to_try.begin());
auto cbs = std::move(on_resolve_queue);
for (auto cb: cbs)
{
cb();
}
}
void etcd_state_client_t::start_etcd_watcher()
{
pick_next_etcd([this]()
{
start_etcd_watcher_selected();
});
}
void etcd_state_client_t::start_etcd_watcher_selected()
{
const auto & url = selected_etcd_url;
etcd_watches_initialised = 0;
ws_alive = 1;
if (this->log_level > 1)
{
fprintf(stderr, "Trying to connect to etcd websocket at %s%s%s (hostname %s), watch from revision %ju/%ju/%ju\n",
url.ssl ? "https://" : "http://", url.addr.c_str(), url.path.c_str(), url.hostname.c_str(),
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
}
if (!etcd_watch_ws)
etcd_watch_ws = http_init(get_http_ctx());
else
http_close(etcd_watch_ws);
open_websocket(etcd_watch_ws, url.addr, url.hostname, url.path+"/watch", { .timeout = etcd_slow_timeout, .ssl = url.ssl },
[this, cur_addr = url.addr](http_message_t *msg)
{
if (msg->body.length())
{
ws_alive = 1;
std::string json_err;
json11::Json data = json11::Json::parse(msg->body, json_err);
if (json_err != "")
{
fprintf(stderr, "Bad JSON in etcd event: %s, ignoring event\n", json_err.c_str());
}
else
{
uint64_t watch_id = data["result"]["watch_id"].uint64_value();
if (data["result"]["created"].bool_value())
{
if (watch_id == ETCD_CONFIG_WATCH_ID ||
watch_id == ETCD_PG_STATE_WATCH_ID ||
watch_id == ETCD_OSD_STATE_WATCH_ID)
{
etcd_watches_initialised++;
}
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES && this->log_level > 0)
{
fprintf(stderr, "Successfully subscribed to etcd at %s, revision %ju/%ju/%ju\n", cur_addr.c_str(),
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
}
}
if (data["result"]["canceled"].bool_value())
{
// etcd watch canceled, maybe because the revision was compacted
if (data["result"]["compact_revision"].uint64_value())
{
// we may miss events if we proceed
// so we should restart from the beginning if we can
if (on_reload_hook != NULL)
{
// check to not trigger on_reload_hook multiple times
if (etcd_watch_ws != NULL)
{
fprintf(stderr, "Revisions before %ju were compacted by etcd, reloading state\n",
data["result"]["compact_revision"].uint64_value());
http_close(etcd_watch_ws);
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = 0;
on_reload_hook();
}
return;
}
else
{
fprintf(stderr, "Revisions before %ju were compacted by etcd, exiting\n",
data["result"]["compact_revision"].uint64_value());
exit(1);
}
}
else
{
fprintf(stderr, "Watch canceled by etcd, reason: %s, exiting\n", data["result"]["cancel_reason"].string_value().c_str());
exit(1);
}
}
// Save revision only if it's present in the message - because sometimes etcd sends something without a header, like:
// {"error": {"grpc_code": 14, "http_code": 503, "http_status": "Service Unavailable", "message": "error reading from server: EOF"}}
// Also don't save revision from the initial created: true messages because they always contain the latest revision
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES &&
!data["result"]["header"]["revision"].is_null() &&
!data["result"]["created"].bool_value())
{
// Restart watchers from the same revision number as in the last received message,
// not from the next one to protect against revision being split into multiple messages,
// even though etcd guarantees not to do that **within a single watcher** without fragment=true:
// https://etcd.io/docs/v3.5/learning/api_guarantees/#watch-apis
// Revision contents are ALWAYS split into separate messages for different watchers though!
// So generally we have to resume each watcher from its own revision...
// Progress messages may have watch_id=-1 if sent on behalf of multiple watchers though.
// And antietcd has an advanced semantic which merges the same revision for all watchers
// into one message and just omits watch_id.
// So we also have to handle the case where watch_id is -1 or not present (0).
auto watch_rev = data["result"]["header"]["revision"].uint64_value();
if (!watch_id || watch_id == UINT64_MAX)
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = watch_rev;
else if (watch_id == ETCD_CONFIG_WATCH_ID)
etcd_watch_revision_config = watch_rev;
else if (watch_id == ETCD_PG_STATE_WATCH_ID)
etcd_watch_revision_pg = watch_rev;
else if (watch_id == ETCD_OSD_STATE_WATCH_ID)
etcd_watch_revision_osd = watch_rev;
etcd_urls_to_try.clear();
}
// First gather all changes into a hash to remove multiple overwrites
std::map<std::string, etcd_kv_t> changes;
for (auto & ev: data["result"]["events"].array_items())
{
auto kv = parse_etcd_kv(ev["kv"]);
if (kv.key != "")
{
changes[kv.key] = kv;
}
}
for (auto & kv: changes)
{
if (this->log_level > 3)
{
fprintf(stderr, "Incoming event: %s -> %s\n", kv.first.c_str(), kv.second.value.dump().c_str());
}
parse_state(kv.second);
}
// React to changes
if (on_change_hook != NULL)
{
on_change_hook(changes);
}
}
}
if (msg->eof)
{
fprintf(stderr, "Disconnected from etcd %s\n", cur_addr.c_str());
if (cur_addr == selected_etcd_url.addr)
selected_etcd_url = (http_url_t){};
if (etcd_watches_initialised == 0)
{
// Connection not established, retry in <etcd_quick_timeout>
tfd->set_timer(etcd_quick_timeout, false, [this](int)
{
start_etcd_watcher();
});
}
else if (etcd_watches_initialised > 0)
{
// Connection was live, retry immediately
etcd_watches_initialised = 0;
start_etcd_watcher();
}
}
});
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
{ "create_request", json11::Json::object {
{ "key", base64_encode(etcd_prefix+"/config/") },
{ "range_end", base64_encode(etcd_prefix+"/config0") },
{ "start_revision", etcd_watch_revision_config },
{ "watch_id", ETCD_CONFIG_WATCH_ID },
{ "progress_notify", true },
} }
}).dump());
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
{ "create_request", json11::Json::object {
{ "key", base64_encode(etcd_prefix+"/osd/state/") },
{ "range_end", base64_encode(etcd_prefix+"/osd/state0") },
{ "start_revision", etcd_watch_revision_osd },
{ "watch_id", ETCD_OSD_STATE_WATCH_ID },
{ "progress_notify", true },
} }
}).dump());
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
{ "create_request", json11::Json::object {
{ "key", base64_encode(etcd_prefix+"/pg/") },
{ "range_end", base64_encode(etcd_prefix+"/pg0") },
{ "start_revision", etcd_watch_revision_pg },
{ "watch_id", ETCD_PG_STATE_WATCH_ID },
{ "progress_notify", true },
} }
}).dump());
// FIXME: Do not watch /pg/history/ at all in client code (not in OSD)
if (on_start_watcher_hook)
{
on_start_watcher_hook(etcd_watch_ws);
}
start_ws_keepalive();
}
void etcd_state_client_t::stop_ws_keepalive()
{
if (ws_keepalive_timer >= 0)
{
tfd->clear_timer(ws_keepalive_timer);
ws_keepalive_timer = -1;
}
}
void etcd_state_client_t::start_ws_keepalive()
{
if (ws_keepalive_timer < 0)
{
ws_keepalive_timer = tfd->set_timer(etcd_ws_keepalive_interval*1000, true, [this](int)
{
if (!etcd_watch_ws || etcd_watches_initialised < ETCD_TOTAL_WATCHES)
{
// Do nothing
}
else if (!ws_alive)
{
if (this->log_level > 0)
{
fprintf(stderr, "Websocket ping failed, disconnecting from etcd %s\n", selected_etcd_url.addr.c_str());
}
start_etcd_watcher();
}
else
{
ws_alive = 0;
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
{ "progress_request", json11::Json::object { } }
}).dump());
}
});
}
}
void etcd_state_client_t::load_global_config()
void etcd_state_client_t::load_global_config(std::function<void(const std::string & error)> cb)
{
json11::Json::object req = { { "success", json11::Json::array {
json11::Json::object {
@@ -704,22 +243,12 @@ void etcd_state_client_t::load_global_config()
} }
},
} } };
etcd_txn(req, etcd_quick_timeout, max_etcd_attempts, 0, [this](std::string err, json11::Json data)
etcd_txn(req, etcd_quick_timeout, max_etcd_attempts, 0, [this, cb](std::string err, json11::Json data)
{
if (err != "")
{
fprintf(stderr, "Error reading configuration from etcd: %s\n", err.c_str());
if (infinite_start)
{
tfd->set_timer(etcd_slow_timeout, false, [this](int timer_id)
{
load_global_config();
});
}
else
{
exit(1);
}
cb(err);
return;
}
json11::Json config_kv = data["responses"][0]["response_range"]["kvs"][0];
@@ -750,28 +279,12 @@ void etcd_state_client_t::load_global_config()
parse_state(kv);
}
on_load_config_hook(global_config);
cb("");
});
}
void etcd_state_client_t::load_pgs()
void etcd_state_client_t::load_pgs(std::function<void(const std::string &)> cb)
{
timespec tv;
clock_gettime(CLOCK_REALTIME, &tv);
uint64_t ms_passed = (tv.tv_sec-etcd_last_reload.tv_sec)*1000 + (tv.tv_nsec-etcd_last_reload.tv_nsec)/1000000;
if (ms_passed < etcd_min_reload_interval)
{
if (load_pgs_timer_id < 0)
{
load_pgs_timer_id = tfd->set_timer(etcd_min_reload_interval+50-ms_passed, false, [this](int) { load_pgs(); });
}
return;
}
etcd_last_reload = tv;
if (load_pgs_timer_id >= 0)
{
tfd->clear_timer(load_pgs_timer_id);
load_pgs_timer_id = -1;
}
json11::Json::array txn = {
json11::Json::object {
{ "request_range", json11::Json::object {
@@ -809,16 +322,13 @@ void etcd_state_client_t::load_pgs()
{
req["compare"] = checks;
}
etcd_txn_slow(req, [this](std::string err, json11::Json data)
etcd_txn_slow(req, [this, cb](std::string err, json11::Json data)
{
if (err != "")
{
// Retry indefinitely
fprintf(stderr, "Error loading PGs from etcd: %s\n", err.c_str());
tfd->set_timer(etcd_slow_timeout, false, [this](int timer_id)
{
load_pgs();
});
cb(err);
return;
}
if (!data["succeeded"].bool_value())
@@ -846,24 +356,9 @@ void etcd_state_client_t::load_pgs()
}
clean_nonexistent_pgs();
on_load_pgs_hook(true);
start_etcd_watcher();
cb("");
});
}
#else
void etcd_state_client_t::parse_config(const json11::Json & config)
{
}
void etcd_state_client_t::load_global_config()
{
json11::Json::object global_config;
on_load_config_hook(global_config);
}
void etcd_state_client_t::load_pgs()
{
}
#endif
void etcd_state_client_t::reset_pg_exists()
{
@@ -976,7 +471,8 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
if (pc.pg_size < 1 ||
pool_item.second["pg_size"].uint64_value() < 3 &&
(pc.scheme == POOL_SCHEME_XOR || pc.scheme == POOL_SCHEME_EC) ||
pool_item.second["pg_size"].uint64_value() > 256)
// limit is 64 because osd_peering_pg.cpp uses a 64-bit mask for has_roles
pool_item.second["pg_size"].uint64_value() > 64)
{
fprintf(stderr, "Pool %u has invalid pg_size, skipping pool\n", pool_id);
continue;
@@ -1318,8 +814,14 @@ void etcd_state_client_t::parse_state(const etcd_kv_t & kv)
else if (key.substr(0, etcd_prefix.length()+11) == etcd_prefix+"/osd/state/")
{
// <etcd_prefix>/osd/state/%d
osd_num_t peer_osd = std::stoull(key.substr(etcd_prefix.length()+11));
if (peer_osd > 0)
osd_num_t peer_osd = 0;
char null_byte = 0;
int scanned = sscanf(key.c_str() + etcd_prefix.length()+11, "%ju%c", &peer_osd, &null_byte);
if (scanned != 1 || !peer_osd)
{
fprintf(stderr, "Bad etcd key %s, ignoring\n", key.c_str());
}
else
{
if (value.is_object() && value["state"] == "up")
{
+15 -32
View File
@@ -136,7 +136,6 @@ struct user_info_t
};
struct http_co_t;
struct http_context_t;
struct __attribute__((visibility("default"))) etcd_state_client_t
{
@@ -147,21 +146,13 @@ protected:
std::vector<http_url_t> etcd_local_addr_urls;
std::vector<http_url_t> etcd_nonlocal_addr_urls;
std::vector<http_url_t> etcd_name_urls;
size_t local_to_try = 0;
std::vector<http_url_t> etcd_urls_to_try;
http_url_t selected_etcd_url;
size_t resolve_count = 0;
std::vector<inode_watch_t*> watches;
std::vector<std::function<void()>> on_resolve_queue;
std::set<osd_num_t> seen_peers;
bool new_pg_config = false;
int ws_keepalive_timer = -1;
int ws_alive = 0;
bool rand_initialized = false;
void add_etcd_url(std::string);
void pick_next_etcd(std::function<void()> cb);
void pick_next_etcd_on_resolve();
void etcd_call_selected(const std::string & api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
void start_etcd_watcher_selected();
void reset_pg_exists();
void clean_nonexistent_pgs();
public:
int etcd_keepalive_timeout = 30;
int etcd_ws_keepalive_interval = 5;
@@ -180,19 +171,13 @@ public:
std::string etcd_client_key;
std::string etcd_ca;
int log_level = 0;
timerfd_manager_t *tfd = NULL;
http_context_t *http_ctx = NULL;
http_co_t *etcd_watch_ws = NULL, *keepalive_client = NULL;
int etcd_watches_initialised = 0;
uint64_t etcd_watch_revision_config = 0;
uint64_t etcd_watch_revision_osd = 0;
uint64_t etcd_watch_revision_pg = 0;
timespec etcd_last_reload = {};
int load_pgs_timer_id = -1;
std::map<pool_id_t, pool_config_t> pool_config;
std::map<osd_num_t, json11::Json> peer_states;
std::set<osd_num_t> seen_peers;
std::map<inode_t, inode_config_t> inode_config;
std::map<std::string, inode_t> inode_by_name;
robin_hood::unordered_flat_map<std::string, std::shared_ptr<user_info_t>> user_info;
@@ -220,25 +205,23 @@ public:
std::vector<std::string> get_addresses();
std::shared_ptr<user_info_t> get_user(const std::string & username);
bool check_image_perm(const std::shared_ptr<user_info_t> & user_info, inode_t inode_num, bool write);
http_context_t *get_http_ctx();
void etcd_call_oneshot(const std::string & etcd_address, const std::string & api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback);
void etcd_call(const std::string & api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
virtual void etcd_call_oneshot(const std::string & etcd_address, const std::string & api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback) = 0;
virtual void etcd_call(const std::string & api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback) = 0;
void etcd_txn(json11::Json txn, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback);
void etcd_txn_slow(json11::Json txn, std::function<void(std::string, json11::Json)> callback);
void start_etcd_watcher();
void stop_ws_keepalive();
void start_ws_keepalive();
void load_global_config();
void load_pgs();
void reset_pg_exists();
void clean_nonexistent_pgs();
virtual void etcd_add_watch(json11::Json watch) = 0;
virtual std::string get_username() = 0;
void load_global_config(std::function<void(const std::string &)> cb);
virtual void load_global_config() = 0;
void load_pgs(std::function<void(const std::string &)> cb);
virtual void load_pgs() = 0;
void parse_state(const etcd_kv_t & kv);
void parse_config(const json11::Json & config);
virtual void parse_config(const json11::Json & config);
void insert_inode_config(const inode_config_t & cfg);
inode_watch_t* watch_inode(std::string name);
void close_watch(inode_watch_t* watch);
int address_count();
~etcd_state_client_t();
virtual ~etcd_state_client_t();
static uint32_t parse_immediate_commit(const std::string & immediate_commit_str, uint32_t default_value);
static uint32_t parse_scheme(const std::string & scheme_str);
+545
View File
@@ -0,0 +1,545 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
#include <assert.h>
#include "etcd_state_client_http.h"
#include "addr_util.h"
#include "http_client.h"
#include "str_util.h"
etcd_state_client_http_t::etcd_state_client_http_t(timerfd_manager_t *tfd)
{
this->tfd = tfd;
}
etcd_state_client_http_t::~etcd_state_client_http_t()
{
stop_ws_keepalive();
if (etcd_watch_ws)
{
http_destroy(etcd_watch_ws);
etcd_watch_ws = NULL;
}
if (keepalive_client)
{
http_destroy(keepalive_client);
keepalive_client = NULL;
}
if (load_pgs_timer_id >= 0)
{
tfd->clear_timer(load_pgs_timer_id);
load_pgs_timer_id = -1;
}
if (http_ctx)
{
http_context_destroy(http_ctx);
http_ctx = NULL;
}
etcd_watches_initialised = -1;
}
void etcd_state_client_http_t::etcd_add_watch(json11::Json watch)
{
if (etcd_watch_ws)
{
http_post_message(etcd_watch_ws, WS_TEXT, watch.dump());
}
}
std::string etcd_state_client_http_t::get_username()
{
return http_context_get_ssl_cn(get_http_ctx());
}
http_context_t *etcd_state_client_http_t::get_http_ctx()
{
if (!http_ctx)
{
std::string error;
http_ctx = http_context_init(tfd, etcd_client_cert, etcd_client_key, etcd_ca, true, error);
if (!http_ctx)
{
fprintf(stderr, "Failed to initialize HTTP context: %s\n", error.c_str());
exit(1);
}
}
return http_ctx;
}
void etcd_state_client_http_t::etcd_call_oneshot(const std::string & etcd_url, const std::string & api, json11::Json payload,
int timeout, std::function<void(std::string, json11::Json)> callback)
{
auto http_cli = http_init(get_http_ctx());
http_json_post(http_cli, etcd_url+api, payload, "", { .timeout = timeout }, [http_cli, callback](http_message_t *response)
{
std::string err;
json11::Json data;
response->parse_json_response(err, data);
callback(err, data);
http_destroy(http_cli);
});
}
void etcd_state_client_http_t::etcd_call(const std::string & api, json11::Json payload, int timeout,
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
{
pick_next_etcd([=]()
{
etcd_call_selected(api, payload, timeout, retries, interval, callback);
});
}
void etcd_state_client_http_t::etcd_call_selected(const std::string & api, json11::Json payload, int timeout,
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
{
const auto & url = selected_etcd_url;
std::string req = payload.dump();
req = "POST "+url.path+api+" HTTP/1.1\r\n"
"Host: "+url.hostname+"\r\n"
"Content-Type: application/json\r\n"
"Content-Length: "+std::to_string(req.size())+"\r\n"
"Connection: keep-alive\r\n"
"Keep-Alive: timeout="+std::to_string(etcd_keepalive_timeout)+"\r\n"
"\r\n"+req;
retries--;
auto cb = [this, api, payload, timeout, retries, interval, callback,
cur_addr = url.addr](http_message_t *response)
{
std::string err;
json11::Json data;
response->parse_json_response(err, data);
if (err != "")
{
if (cur_addr == selected_etcd_url.addr)
selected_etcd_url = (http_url_t){};
if (retries > 0)
{
if (this->log_level > 0)
{
fprintf(
stderr, "Warning: etcd request failed: %s, retrying %d more times\n",
err.c_str(), retries
);
}
if (interval > 0)
{
// FIXME: Prevent destruction of etcd_state_client if timers or requests are active
tfd->set_timer(interval, false, [this, api, payload, timeout, retries, interval, callback](int)
{
etcd_call(api, payload, timeout, retries, interval, callback);
});
}
else
etcd_call(api, payload, timeout, retries, interval, callback);
}
else
callback(err, data);
}
else
callback(err, data);
};
if (!keepalive_client)
keepalive_client = http_init(get_http_ctx());
http_request(keepalive_client, url.addr, req, { .timeout = timeout, .keepalive = true, .ssl = url.ssl }, cb);
}
void etcd_state_client_http_t::parse_config(const json11::Json & config)
{
auto old_etcd_ws_keepalive_interval = this->etcd_ws_keepalive_interval;
etcd_state_client_t::parse_config(config);
if (this->etcd_ws_keepalive_interval != old_etcd_ws_keepalive_interval && ws_keepalive_timer >= 0)
{
stop_ws_keepalive();
start_ws_keepalive();
}
}
void etcd_state_client_http_t::pick_next_etcd(std::function<void()> cb)
{
if (!etcd_addresses.size() && !etcd_local.size())
{
fprintf(stderr, "etcd_address is missing in Vitastor configuration\n");
exit(1);
}
if (selected_etcd_url.addr != "")
{
cb();
return;
}
if (etcd_urls_to_try.size() != 0)
{
selected_etcd_url = std::move(etcd_urls_to_try[0]);
etcd_urls_to_try.erase(etcd_urls_to_try.begin());
cb();
return;
}
on_resolve_queue.push_back(std::move(cb));
if (on_resolve_queue.size() > 1)
{
// Already resolving
return;
}
assert(!resolve_count);
local_to_try = 0;
for (auto & url: etcd_local_addr_urls)
{
// Prefer local IPs, if any
etcd_urls_to_try.push_back(url);
local_to_try++;
}
for (auto & url: etcd_nonlocal_addr_urls)
{
etcd_urls_to_try.push_back(url);
}
resolve_count++;
for (auto & url: etcd_name_urls)
{
resolve_count++;
http_resolve(get_http_ctx(), url.ssl, url.addr, [this, url](const std::string & error, const std::vector<std::string>& addresses)
{
if (error != "")
fprintf(stderr, "Error resolving %s: %s\n", url.addr.c_str(), error.c_str());
for (auto & addr: addresses)
{
auto url_copy = url;
url_copy.addr = addr;
if (local_ips.find(addr) != local_ips.end())
{
etcd_urls_to_try.insert(etcd_urls_to_try.begin(), std::move(url_copy));
local_to_try++;
}
else
etcd_urls_to_try.push_back(std::move(url_copy));
}
resolve_count--;
if (!resolve_count)
pick_next_etcd_on_resolve();
});
}
resolve_count--;
if (!resolve_count)
{
pick_next_etcd_on_resolve();
}
}
void etcd_state_client_http_t::pick_next_etcd_on_resolve()
{
if (!etcd_urls_to_try.size())
{
fprintf(stderr, "None of etcd_address could be resolved\n");
exit(1);
}
if (!rand_initialized)
{
timespec tv;
clock_gettime(CLOCK_REALTIME, &tv);
srand48(tv.tv_sec*1000000000 + tv.tv_nsec);
rand_initialized = true;
}
// Shuffle addresses
for (size_t i = etcd_urls_to_try.size()-1; i > local_to_try; i--)
{
size_t j = local_to_try + lrand48() % (i - local_to_try);
if (j != i)
std::swap(etcd_urls_to_try[i], etcd_urls_to_try[j]);
}
selected_etcd_url = std::move(etcd_urls_to_try[0]);
etcd_urls_to_try.erase(etcd_urls_to_try.begin());
auto cbs = std::move(on_resolve_queue);
for (auto cb: cbs)
{
cb();
}
}
void etcd_state_client_http_t::start_etcd_watcher()
{
pick_next_etcd([this]()
{
start_etcd_watcher_selected();
});
}
void etcd_state_client_http_t::start_etcd_watcher_selected()
{
const auto & url = selected_etcd_url;
etcd_watches_initialised = 0;
ws_alive = 1;
if (this->log_level > 1)
{
fprintf(stderr, "Trying to connect to etcd websocket at %s%s%s (hostname %s), watch from revision %ju/%ju/%ju\n",
url.ssl ? "https://" : "http://", url.addr.c_str(), url.path.c_str(), url.hostname.c_str(),
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
}
if (!etcd_watch_ws)
etcd_watch_ws = http_init(get_http_ctx());
else
http_close(etcd_watch_ws);
open_websocket(etcd_watch_ws, url.addr, url.hostname, url.path+"/watch", { .timeout = etcd_slow_timeout, .ssl = url.ssl },
[this, cur_addr = url.addr](http_message_t *msg)
{
if (msg->body.length())
{
ws_alive = 1;
std::string json_err;
json11::Json data = json11::Json::parse(msg->body, json_err);
if (json_err != "")
{
fprintf(stderr, "Bad JSON in etcd event: %s, ignoring event\n", json_err.c_str());
}
else
{
uint64_t watch_id = data["result"]["watch_id"].uint64_value();
if (data["result"]["created"].bool_value())
{
if (watch_id == ETCD_CONFIG_WATCH_ID ||
watch_id == ETCD_PG_STATE_WATCH_ID ||
watch_id == ETCD_OSD_STATE_WATCH_ID)
{
etcd_watches_initialised++;
}
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES && this->log_level > 0)
{
fprintf(stderr, "Successfully subscribed to etcd at %s, revision %ju/%ju/%ju\n", cur_addr.c_str(),
etcd_watch_revision_config, etcd_watch_revision_osd, etcd_watch_revision_pg);
}
}
if (data["result"]["canceled"].bool_value())
{
// etcd watch canceled, maybe because the revision was compacted
if (data["result"]["compact_revision"].uint64_value())
{
// we may miss events if we proceed
// so we should restart from the beginning if we can
if (on_reload_hook != NULL)
{
// check to not trigger on_reload_hook multiple times
if (etcd_watch_ws != NULL)
{
fprintf(stderr, "Revisions before %ju were compacted by etcd, reloading state\n",
data["result"]["compact_revision"].uint64_value());
http_close(etcd_watch_ws);
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = 0;
on_reload_hook();
}
return;
}
else
{
fprintf(stderr, "Revisions before %ju were compacted by etcd, exiting\n",
data["result"]["compact_revision"].uint64_value());
exit(1);
}
}
else
{
fprintf(stderr, "Watch canceled by etcd, reason: %s, exiting\n", data["result"]["cancel_reason"].string_value().c_str());
exit(1);
}
}
// Save revision only if it's present in the message - because sometimes etcd sends something without a header, like:
// {"error": {"grpc_code": 14, "http_code": 503, "http_status": "Service Unavailable", "message": "error reading from server: EOF"}}
// Also don't save revision from the initial created: true messages because they always contain the latest revision
if (etcd_watches_initialised == ETCD_TOTAL_WATCHES &&
!data["result"]["header"]["revision"].is_null() &&
!data["result"]["created"].bool_value())
{
// Restart watchers from the same revision number as in the last received message,
// not from the next one to protect against revision being split into multiple messages,
// even though etcd guarantees not to do that **within a single watcher** without fragment=true:
// https://etcd.io/docs/v3.5/learning/api_guarantees/#watch-apis
// Revision contents are ALWAYS split into separate messages for different watchers though!
// So generally we have to resume each watcher from its own revision...
// Progress messages may have watch_id=-1 if sent on behalf of multiple watchers though.
// And antietcd has an advanced semantic which merges the same revision for all watchers
// into one message and just omits watch_id.
// So we also have to handle the case where watch_id is -1 or not present (0).
auto watch_rev = data["result"]["header"]["revision"].uint64_value();
if (!watch_id || watch_id == UINT64_MAX)
etcd_watch_revision_config = etcd_watch_revision_osd = etcd_watch_revision_pg = watch_rev;
else if (watch_id == ETCD_CONFIG_WATCH_ID)
etcd_watch_revision_config = watch_rev;
else if (watch_id == ETCD_PG_STATE_WATCH_ID)
etcd_watch_revision_pg = watch_rev;
else if (watch_id == ETCD_OSD_STATE_WATCH_ID)
etcd_watch_revision_osd = watch_rev;
etcd_urls_to_try.clear();
}
// First gather all changes into a hash to remove multiple overwrites
std::map<std::string, etcd_kv_t> changes;
for (auto & ev: data["result"]["events"].array_items())
{
auto kv = parse_etcd_kv(ev["kv"]);
if (kv.key != "")
{
changes[kv.key] = kv;
}
}
for (auto & kv: changes)
{
if (this->log_level > 3)
{
fprintf(stderr, "Incoming event: %s -> %s\n", kv.first.c_str(), kv.second.value.dump().c_str());
}
parse_state(kv.second);
}
// React to changes
if (on_change_hook != NULL)
{
on_change_hook(changes);
}
}
}
if (msg->eof)
{
fprintf(stderr, "Disconnected from etcd %s\n", cur_addr.c_str());
if (cur_addr == selected_etcd_url.addr)
selected_etcd_url = (http_url_t){};
if (etcd_watches_initialised == 0)
{
// Connection not established, retry in <etcd_quick_timeout>
tfd->set_timer(etcd_quick_timeout, false, [this](int)
{
start_etcd_watcher();
});
}
else if (etcd_watches_initialised > 0)
{
// Connection was live, retry immediately
etcd_watches_initialised = 0;
start_etcd_watcher();
}
}
});
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
{ "create_request", json11::Json::object {
{ "key", base64_encode(etcd_prefix+"/config/") },
{ "range_end", base64_encode(etcd_prefix+"/config0") },
{ "start_revision", etcd_watch_revision_config },
{ "watch_id", ETCD_CONFIG_WATCH_ID },
{ "progress_notify", true },
} }
}).dump());
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
{ "create_request", json11::Json::object {
{ "key", base64_encode(etcd_prefix+"/osd/state/") },
{ "range_end", base64_encode(etcd_prefix+"/osd/state0") },
{ "start_revision", etcd_watch_revision_osd },
{ "watch_id", ETCD_OSD_STATE_WATCH_ID },
{ "progress_notify", true },
} }
}).dump());
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
{ "create_request", json11::Json::object {
{ "key", base64_encode(etcd_prefix+"/pg/") },
{ "range_end", base64_encode(etcd_prefix+"/pg0") },
{ "start_revision", etcd_watch_revision_pg },
{ "watch_id", ETCD_PG_STATE_WATCH_ID },
{ "progress_notify", true },
} }
}).dump());
// FIXME: Do not watch /pg/history/ at all in client code (not in OSD)
if (on_start_watcher_hook)
{
on_start_watcher_hook(etcd_watch_ws);
}
start_ws_keepalive();
}
void etcd_state_client_http_t::stop_ws_keepalive()
{
if (ws_keepalive_timer >= 0)
{
tfd->clear_timer(ws_keepalive_timer);
ws_keepalive_timer = -1;
}
}
void etcd_state_client_http_t::start_ws_keepalive()
{
if (ws_keepalive_timer < 0)
{
ws_keepalive_timer = tfd->set_timer(etcd_ws_keepalive_interval*1000, true, [this](int)
{
if (!etcd_watch_ws || etcd_watches_initialised < ETCD_TOTAL_WATCHES)
{
// Do nothing
}
else if (!ws_alive)
{
if (this->log_level > 0)
{
fprintf(stderr, "Websocket ping failed, disconnecting from etcd %s\n", selected_etcd_url.addr.c_str());
}
start_etcd_watcher();
}
else
{
ws_alive = 0;
http_post_message(etcd_watch_ws, WS_TEXT, json11::Json(json11::Json::object {
{ "progress_request", json11::Json::object { } }
}).dump());
}
});
}
}
void etcd_state_client_http_t::load_global_config()
{
etcd_state_client_t::load_global_config([this](const std::string & err)
{
if (err != "")
{
fprintf(stderr, "Error reading configuration from etcd: %s\n", err.c_str());
if (infinite_start)
{
tfd->set_timer(etcd_slow_timeout, false, [this](int timer_id)
{
load_global_config();
});
}
else
{
exit(1);
}
}
});
}
void etcd_state_client_http_t::load_pgs()
{
timespec tv;
clock_gettime(CLOCK_REALTIME, &tv);
uint64_t ms_passed = (tv.tv_sec-etcd_last_reload.tv_sec)*1000 + (tv.tv_nsec-etcd_last_reload.tv_nsec)/1000000;
if (ms_passed < etcd_min_reload_interval)
{
if (load_pgs_timer_id < 0)
{
load_pgs_timer_id = tfd->set_timer(etcd_min_reload_interval+50-ms_passed, false, [this](int) { load_pgs(); });
}
return;
}
etcd_last_reload = tv;
if (load_pgs_timer_id >= 0)
{
tfd->clear_timer(load_pgs_timer_id);
load_pgs_timer_id = -1;
}
etcd_state_client_t::load_pgs([this](const std::string & err)
{
if (err != "")
{
// Retry indefinitely
fprintf(stderr, "Error loading PGs from etcd: %s\n", err.c_str());
tfd->set_timer(etcd_slow_timeout, false, [this](int timer_id)
{
load_pgs();
});
}
else
{
start_etcd_watcher();
}
});
}
+50
View File
@@ -0,0 +1,50 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
#pragma once
#include "etcd_state_client.h"
struct http_context_t;
struct __attribute__((visibility("default"))) etcd_state_client_http_t: public etcd_state_client_t
{
protected:
timerfd_manager_t *tfd = NULL;
int ws_keepalive_timer = -1;
int ws_alive = 0;
bool rand_initialized = false;
int etcd_watches_initialised = 0;
timespec etcd_last_reload = {};
int load_pgs_timer_id = -1;
http_co_t *keepalive_client = NULL;
http_co_t *etcd_watch_ws = NULL;
http_context_t *http_ctx = NULL;
size_t local_to_try = 0;
std::vector<http_url_t> etcd_urls_to_try;
http_url_t selected_etcd_url;
size_t resolve_count = 0;
std::vector<std::function<void()>> on_resolve_queue;
void pick_next_etcd(std::function<void()> cb);
void pick_next_etcd_on_resolve();
void start_etcd_watcher();
void start_etcd_watcher_selected();
void stop_ws_keepalive();
void start_ws_keepalive();
http_context_t *get_http_ctx();
public:
etcd_state_client_http_t(timerfd_manager_t *tfd);
void etcd_call_oneshot(const std::string & etcd_url, const std::string & api, json11::Json payload,
int timeout, std::function<void(std::string, json11::Json)> callback) override;
void etcd_call(const std::string & api, json11::Json payload, int timeout,
int retries, int interval, std::function<void(std::string, json11::Json)> callback) override;
void etcd_call_selected(const std::string & api, json11::Json payload, int timeout,
int retries, int interval, std::function<void(std::string, json11::Json)> callback);
void etcd_add_watch(json11::Json watch) override;
std::string get_username() override;
void load_global_config() override;
void load_pgs() override;
void parse_config(const json11::Json & config) override;
~etcd_state_client_http_t();
};
+210
View File
@@ -0,0 +1,210 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
#include <assert.h>
#include "etcd_state_client_mock.h"
#include "str_util.h"
etcd_state_client_mock_t::etcd_state_client_mock_t()
{
timespec tv;
clock_gettime(CLOCK_REALTIME, &tv);
srand48(tv.tv_sec*1000000000 + tv.tv_nsec);
}
void etcd_state_client_mock_t::etcd_add_watch(json11::Json watch)
{
}
std::string etcd_state_client_mock_t::get_username()
{
return username;
}
void etcd_state_client_mock_t::etcd_call_oneshot(const std::string & etcd_address, const std::string & api, json11::Json payload,
int timeout, std::function<void(std::string, json11::Json)> callback)
{
}
void etcd_state_client_mock_t::pause()
{
paused = true;
}
void etcd_state_client_mock_t::resume()
{
paused = false;
auto queue = std::move(this->queue);
for (auto& req: queue)
{
etcd_call(req.api, req.payload, req.timeout, req.retries, req.interval, req.callback);
}
}
void etcd_state_client_mock_t::set(const std::string& key, json11::Json data, uint64_t mod_revision, uint64_t lease_id)
{
if (!mod_revision)
mod_revision = ++this->mod_revision;
this->data[key] = (etcd_mock_key_data_t){ .value = data.dump(), .mod_revision = mod_revision, .lease_id = lease_id };
}
void etcd_state_client_mock_t::etcd_call(const std::string & api, json11::Json payload, int timeout,
int retries, int interval, std::function<void(std::string, json11::Json)> callback)
{
if (paused)
{
queue.push_back({ api, payload, timeout, retries, interval, callback });
return;
}
printf("+ etcd: %s\n", api.c_str());
if (api == "/kv/txn")
{
bool ok = true;
for (auto& check: payload["compare"].array_items())
{
auto key = base64_decode(check["key"].string_value());
etcd_mock_key_data_t *key_data = data.find(key) != data.end() ? &data.at(key) : NULL;
auto target = check["target"].string_value();
auto res = check["result"].string_value();
assert(res == "LESS" || res == "");
bool less = res == "LESS";
if (target == "MOD")
{
uint64_t rev = check["mod_revision"].uint64_value();
assert(!less || rev);
ok = ok && (less ? (!key_data || key_data->mod_revision < rev) : (key_data && key_data->mod_revision == rev));
}
else if (target == "CREATE")
{
uint64_t rev = check["create_revision"].uint64_value();
assert(rev == 0 && !less);
ok = ok && !key_data;
}
else if (target == "VERSION")
{
uint64_t rev = check["version"].uint64_value();
assert(rev == 0 && !less);
ok = ok && !key_data;
}
else if (target == "LEASE")
{
assert(!less);
uint64_t lease_id = check["lease"].uint64_value();
ok = ok && key_data && key_data->lease_id == lease_id;
}
else
assert(0);
}
std::map<std::string, etcd_kv_t> changes;
bool has_mod = false;
for (auto& op: payload[ok ? "success" : "failure"].array_items())
{
auto& obj = op.object_items();
has_mod = has_mod || obj.find("request_put") != obj.end() ||
obj.find("request_delete_range") != obj.end();
}
if (has_mod)
{
mod_revision++;
}
json11::Json::array responses;
for (auto& op_ptr: payload[ok ? "success" : "failure"].array_items())
{
auto& op = op_ptr.object_items();
if (op.find("request_range") != op.end())
{
json11::Json::array kvs;
auto req = op.at("request_range");
auto key = base64_decode(req["key"].string_value());
auto range_end = base64_decode(req["range_end"].string_value());
auto begin_it = range_end.empty() ? data.find(key) : data.lower_bound(key);
auto end_it = range_end.empty() ? (begin_it == data.end() ? begin_it : std::next(begin_it)) : data.lower_bound(range_end);
for (auto it = begin_it; it != end_it; it++)
{
printf("\\- get: %s = %s, rev %ju\n", it->first.c_str(), it->second.value.c_str(), it->second.mod_revision);
kvs.push_back(json11::Json::object {
{ "key", base64_encode(it->first) },
{ "value", base64_encode(it->second.value) },
{ "mod_revision", it->second.mod_revision },
});
}
responses.push_back(json11::Json::object {
{ "response_range", json11::Json::object{ { "header", json11::Json::object{ { "revision", mod_revision } } }, { "kvs", kvs } } },
});
}
else if (op.find("request_put") != op.end())
{
auto req = op.at("request_put");
auto key = base64_decode(req["key"].string_value());
auto value = base64_decode(req["value"].string_value());
auto lease_id = req["lease"].uint64_value();
printf("\\- put: %s = %s, rev %ju, lease %ju\n", key.c_str(), value.c_str(), mod_revision, lease_id);
data[key] = {
.value = value,
.mod_revision = mod_revision,
.lease_id = lease_id,
};
std::string err;
json11::Json json_value = json11::Json::parse(value, err);
if (err != "")
{
fprintf(stderr, "Invalid JSON in etcd key %s during test: %s\n", key.c_str(), value.c_str());
exit(1);
}
changes[key] = { .key = key, .value = json_value, .mod_revision = mod_revision };
responses.push_back(json11::Json::object {
{ "response_put", json11::Json::object{ { "header", json11::Json::object{ { "revision", mod_revision } } } } },
});
}
else if (op.find("request_delete_range") != op.end())
{
auto req = op.at("request_delete_range");
auto key = base64_decode(req["key"].string_value());
auto range_end = base64_decode(req["range_end"].string_value());
uint64_t n_del = 0;
for (auto it = data.lower_bound(key); it != data.end() && (range_end == "" || it->first < range_end); )
{
auto & key = it->first;
printf("\\- del: %s\n", key.c_str());
changes[key] = { .key = key, .mod_revision = mod_revision };
n_del++;
data.erase(it++);
}
responses.push_back(json11::Json::object {
{ "response_delete_range", json11::Json::object{ { "header", json11::Json::object{ { "revision", mod_revision } } }, { "deleted", n_del } } },
});
}
}
callback("", json11::Json::object{
{ "header", json11::Json::object{ { "revision", mod_revision } } },
{ "succeeded", ok },
{ "responses", responses }
});
// Push changes to watcher
if (changes.size())
{
for (auto & kv: changes)
parse_state(kv.second);
if (on_change_hook != NULL)
on_change_hook(changes);
}
}
else if (api == "/lease/grant")
{
uint64_t lease_id = (((uint64_t)lrand48()) << 32) | lrand48();
leases[lease_id] = payload["TTL"].uint64_value();
callback("", json11::Json::object{ { "ID", std::to_string(lease_id) } });
}
else
callback("Unsupported", json11::Json());
}
void etcd_state_client_mock_t::load_global_config()
{
etcd_state_client_t::load_global_config([this](const std::string & err) {});
}
void etcd_state_client_mock_t::load_pgs()
{
etcd_state_client_t::load_pgs([this](const std::string & err) {});
}
+44
View File
@@ -0,0 +1,44 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 or GNU GPL-2.0+ (see README.md for details)
#pragma once
#include "etcd_state_client.h"
struct etcd_mock_key_data_t
{
std::string value;
uint64_t mod_revision;
uint64_t lease_id;
};
struct etcd_mock_request_t
{
std::string api;
json11::Json payload;
int timeout;
int retries;
int interval;
std::function<void(std::string, json11::Json)> callback;
};
struct etcd_state_client_mock_t: public etcd_state_client_t
{
uint64_t mod_revision = 0;
bool paused = false;
std::vector<etcd_mock_request_t> queue;
public:
std::map<uint64_t, uint64_t> leases;
std::map<std::string, etcd_mock_key_data_t> data;
std::string username;
etcd_state_client_mock_t();
void set(const std::string& key, json11::Json data, uint64_t mod_revision = 0, uint64_t lease_id = 0);
void pause();
void resume();
void etcd_call_oneshot(const std::string & etcd_address, const std::string & api, json11::Json payload, int timeout, std::function<void(std::string, json11::Json)> callback) override;
void etcd_call(const std::string & api, json11::Json payload, int timeout, int retries, int interval, std::function<void(std::string, json11::Json)> callback) override;
void etcd_add_watch(json11::Json watch) override;
std::string get_username() override;
void load_global_config() override;
void load_pgs() override;
};
+1 -1
View File
@@ -318,7 +318,7 @@ public:
#ifdef WITH_RDMA
bool is_rdma_enabled();
bool connect_rdma(uint64_t client_id, std::string rdma_address, uint64_t client_max_msg);
json11::Json connect_rdma(uint64_t client_id, std::string rdma_address, uint64_t client_max_msg);
#endif
#ifdef WITH_RDMACM
bool is_use_rdmacm();
+6 -4
View File
@@ -427,35 +427,37 @@ void osd_messenger_t::op_encrypted_copy_buf(osd_client_t *cl, uint8_t *enc_buf,
assert(cl->write_op->enc->key_chain[0]);
cl->xts_enc_ctx->start(cl, cl->write_op->enc->key_chain[0], cl->write_op->req.rw.offset, cl->write_op->enc->bitmap_granularity);
}
size_t old_out = done_enc;
while (done_plain < plain_len && done_enc < enc_len)
{
size_t done_in = 0;
size_t done_out = 0;
cl->xts_enc_ctx->update(plain+done_plain, plain_len-done_plain, enc_buf+done_enc, enc_len-done_enc, done_in, done_out);
if (cl->write_csum_state && done_out > 0)
XXH3_64bits_update(cl->write_csum_state, enc_buf+done_enc, done_out);
done_enc += done_out;
cl->write_op_pos += done_in;
done_plain += done_in;
}
if (cl->write_csum_state && done_enc > old_out)
XXH3_64bits_update(cl->write_csum_state, enc_buf+old_out, done_enc-old_out);
}
void osd_messenger_t::op_decrypted_copy_buf(osd_client_t *cl, uint8_t *enc_buf, size_t enc_len, uint8_t *plain, size_t plain_len, size_t & done_plain, size_t & done_enc)
{
op_decrypt_start(cl);
size_t old_in = done_enc;
while (done_plain < plain_len && done_enc < enc_len)
{
size_t done_in = 0;
size_t done_out = 0;
// plain == NULL means skip output
cl->xts_dec_ctx->update(enc_buf+done_enc, enc_len-done_enc, plain ? plain+done_plain : NULL, plain_len-done_plain, done_in, done_out);
if (cl->read_csum_state && done_in > 0)
XXH3_64bits_update(cl->read_csum_state, enc_buf+done_enc, done_in);
done_enc += done_in;
cl->read_op_pos += done_out;
cl->read_op_inline_decrypt_in += done_in;
done_plain += done_out;
}
if (cl->read_csum_state && done_enc > old_in)
XXH3_64bits_update(cl->read_csum_state, enc_buf+old_in, done_enc-old_in);
}
void osd_messenger_t::op_decrypt_start(osd_client_t* cl)
+50
View File
@@ -3,6 +3,7 @@
#include <assert.h>
#include "messenger.h"
#include "msgr_op.h"
osd_op_t::~osd_op_t()
@@ -38,3 +39,52 @@ bool osd_op_t::is_recovery_related()
req.hdr.opcode == OSD_OP_SEC_SYNC &&
(req.sec_sync.flags & OSD_OP_RECOVERY_RELATED);
}
void osd_messenger_t::measure_exec(osd_op_t *cur_op)
{
// Measure execution latency
if (cur_op->req.hdr.opcode > OSD_OP_MAX)
{
return;
}
if (!cur_op->tv_end.tv_sec)
{
clock_gettime(CLOCK_REALTIME, &cur_op->tv_end);
}
uint64_t len = 0;
if (cur_op->req.hdr.opcode == OSD_OP_READ ||
cur_op->req.hdr.opcode == OSD_OP_WRITE ||
cur_op->req.hdr.opcode == OSD_OP_SCRUB)
{
// req.rw.len is internally set to the full object size for scrubs
len = cur_op->req.rw.len;
}
else if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ ||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE)
{
len = cur_op->req.sec_rw.len;
}
inc_op_stats(stats, cur_op->req.hdr.opcode, cur_op->tv_begin, cur_op->tv_end, len);
if (cur_op->is_recovery_related())
{
inc_op_stats(recovery_stats, cur_op->req.hdr.opcode, cur_op->tv_begin, cur_op->tv_end, len);
}
}
void osd_messenger_t::inc_op_stats(osd_op_stats_t & stats, uint64_t opcode, timespec & tv_begin, timespec & tv_end, uint64_t len)
{
uint64_t usecs = (
(tv_end.tv_sec - tv_begin.tv_sec)*1000000 +
(tv_end.tv_nsec - tv_begin.tv_nsec)/1000
);
stats.op_stat_count[opcode]++;
if (!stats.op_stat_count[opcode])
{
stats.op_stat_count[opcode] = 1;
stats.op_stat_sum[opcode] = 0;
stats.op_stat_bytes[opcode] = 0;
}
stats.op_stat_sum[opcode] += usecs;
stats.op_stat_bytes[opcode] += len;
}
+7 -4
View File
@@ -507,7 +507,7 @@ int msgr_rdma_connection_t::connect(msgr_rdma_address_t *dest)
return 0;
}
bool osd_messenger_t::connect_rdma(uint64_t client_id, std::string rdma_address, uint64_t client_max_msg)
json11::Json osd_messenger_t::connect_rdma(uint64_t client_id, std::string rdma_address, uint64_t client_max_msg)
{
// Try to connect to the peer using RDMA
msgr_rdma_address_t addr;
@@ -523,7 +523,7 @@ bool osd_messenger_t::connect_rdma(uint64_t client_id, std::string rdma_address,
{
if (log_level > 0)
fprintf(stderr, "No RDMA context for peer %ju, using only TCP\n", client_id);
return false;
return json11::Json();
}
msgr_rdma_connection_t *rdma_conn = msgr_rdma_connection_t::create(selected_ctx, rdma_max_send, rdma_max_recv, rdma_max_sge, client_max_msg);
if (rdma_conn)
@@ -542,11 +542,14 @@ bool osd_messenger_t::connect_rdma(uint64_t client_id, std::string rdma_address,
// Remember connection, but switch to RDMA only after sending the configuration response
cl->rdma_conn = rdma_conn;
cl->peer_state = PEER_RDMA_CONNECTING;
return true;
return json11::Json::object{
{"rdma_address", rdma_conn->addr.to_string()},
{"rdma_max_msg", rdma_conn->max_msg},
};
}
}
}
return false;
return json11::Json();
}
static void try_send_rdma_wr(osd_client_t *cl, ibv_sge *sge, int op_sge)
-49
View File
@@ -486,55 +486,6 @@ void osd_messenger_t::outbox_push(osd_op_t *cur_op)
}
}
void osd_messenger_t::inc_op_stats(osd_op_stats_t & stats, uint64_t opcode, timespec & tv_begin, timespec & tv_end, uint64_t len)
{
uint64_t usecs = (
(tv_end.tv_sec - tv_begin.tv_sec)*1000000 +
(tv_end.tv_nsec - tv_begin.tv_nsec)/1000
);
stats.op_stat_count[opcode]++;
if (!stats.op_stat_count[opcode])
{
stats.op_stat_count[opcode] = 1;
stats.op_stat_sum[opcode] = 0;
stats.op_stat_bytes[opcode] = 0;
}
stats.op_stat_sum[opcode] += usecs;
stats.op_stat_bytes[opcode] += len;
}
void osd_messenger_t::measure_exec(osd_op_t *cur_op)
{
// Measure execution latency
if (cur_op->req.hdr.opcode > OSD_OP_MAX)
{
return;
}
if (!cur_op->tv_end.tv_sec)
{
clock_gettime(CLOCK_REALTIME, &cur_op->tv_end);
}
uint64_t len = 0;
if (cur_op->req.hdr.opcode == OSD_OP_READ ||
cur_op->req.hdr.opcode == OSD_OP_WRITE ||
cur_op->req.hdr.opcode == OSD_OP_SCRUB)
{
// req.rw.len is internally set to the full object size for scrubs
len = cur_op->req.rw.len;
}
else if (cur_op->req.hdr.opcode == OSD_OP_SEC_READ ||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE ||
cur_op->req.hdr.opcode == OSD_OP_SEC_WRITE_STABLE)
{
len = cur_op->req.sec_rw.len;
}
inc_op_stats(stats, cur_op->req.hdr.opcode, cur_op->tv_begin, cur_op->tv_end, len);
if (cur_op->is_recovery_related())
{
inc_op_stats(recovery_stats, cur_op->req.hdr.opcode, cur_op->tv_begin, cur_op->tv_end, len);
}
}
bool osd_messenger_t::try_send(osd_client_t *cl)
{
if (cl->peer_state == PEER_STOPPED || cl->peer_fd < 0)
+12 -11
View File
@@ -301,10 +301,9 @@ const char *help_text =
" --nbd_disconnect_on_close 1\n"
" Disconnect the nbd device on close by last opener.\n"
#endif
#ifdef NBD_FLAG_READ_ONLY
" --readonly\n"
" --nbd_ro 1\n"
" Set device into read only mode.\n"
#endif
"\n"
"vitastor-nbd netlink-unmap /dev/nbdN\n"
" Unmap a device using netlink interface. Works with both netlink and ioctl mapped devices.\n"
@@ -348,6 +347,7 @@ protected:
int read_ready = 0;
msghdr read_msg = { 0 }, send_msg = { 0 };
iovec read_iov = { 0 };
bool stop = false;
std::string logfile = "/dev/null";
@@ -514,7 +514,7 @@ help:
// Create client
ringloop = new ring_loop_t(RINGLOOP_DEFAULT_SIZE);
epmgr = new epoll_manager_t(ringloop);
cli = new cluster_client_t(ringloop, epmgr->tfd, cfg);
cli = cluster_client_t::create(ringloop, epmgr->tfd, cfg);
if (!inode)
{
// Load image metadata
@@ -525,7 +525,7 @@ help:
break;
ringloop->wait();
}
watch = cli->st_cli.watch_inode(image_name);
watch = cli->st_cli->watch_inode(image_name);
device_size = watch->cfg.size;
if (!watch->cfg.num || !device_size)
{
@@ -581,10 +581,8 @@ help:
}
uint64_t flags = NBD_FLAG_SEND_FLUSH;
uint64_t cflags = 0;
#ifdef NBD_FLAG_READ_ONLY
if (!cfg["nbd_ro"].is_null())
if (!cfg["readonly"].is_null() || !cfg["nbd_ro"].is_null())
flags |= NBD_FLAG_READ_ONLY;
#endif
#ifdef NBD_CFLAG_DESTROY_ON_DISCONNECT
if (!cfg["nbd_destroy_on_disconnect"].is_null())
cflags |= NBD_CFLAG_DESTROY_ON_DISCONNECT;
@@ -620,7 +618,10 @@ help:
if (!cfg["dev_num"].is_null())
{
int r;
if ((r = run_nbd(sockfd, cfg["dev_num"].int64_value(), device_size, NBD_FLAG_SEND_FLUSH, nbd_timeout, bg)) != 0)
uint64_t flags = NBD_FLAG_SEND_FLUSH;
if (!cfg["readonly"].is_null())
flags |= NBD_FLAG_READ_ONLY;
if ((r = run_nbd(sockfd, cfg["dev_num"].int64_value(), device_size, flags, nbd_timeout, bg)) != 0)
{
fprintf(stderr, "run_nbd: %s\n", strerror(-r));
exit(1);
@@ -678,8 +679,7 @@ help:
};
ringloop->register_consumer(&consumer);
// Add FD to epoll
bool stop = false;
epmgr->tfd->set_fd_handler(sockfd[0], false, [this, &stop](int peer_fd, int epoll_events)
epmgr->tfd->set_fd_handler(sockfd[0], false, [this](int peer_fd, int epoll_events)
{
if (epoll_events & EPOLLRDHUP)
{
@@ -1118,7 +1118,8 @@ protected:
{
// Disconnect
close(nbd_fd);
exit(0);
stop = true;
return;
}
if (be32toh(cur_req.magic) != NBD_REQUEST_MAGIC ||
req_type != NBD_CMD_READ && req_type != NBD_CMD_WRITE && req_type != NBD_CMD_FLUSH)
+76
View File
@@ -705,6 +705,54 @@ static void vitastor_close(BlockDriverState *bs)
client->last_bitmap = NULL;
}
// Unregister all event sources from the current AioContext. Called by the
// block layer before bs is moved to a different AioContext (e.g. during live
// migration, drain, dataplane switching). The block layer guarantees that no
// requests are in flight at this point.
static void vitastor_detach_aio_context(BlockDriverState *bs)
{
VitastorClient *client = bs->opaque;
int i;
#if defined VITASTOR_C_API_VERSION && VITASTOR_C_API_VERSION >= 2
if (client->uring_eventfd >= 0)
{
universal_aio_set_fd_handler(client->ctx, client->uring_eventfd, NULL, NULL, NULL);
// Wait until any scheduled B/H is processed before switching contexts:
// it would otherwise fire on the old context with stale state.
if (client->bh_uring_scheduled)
{
BDRV_POLL_WHILE(bs, client->bh_uring_scheduled);
}
}
#endif
for (i = 0; i < client->fd_count; i++)
{
universal_aio_set_fd_handler(client->ctx, client->fds[i]->fd, NULL, NULL, NULL);
}
}
// (Re-)register all event sources on the new AioContext.
static void vitastor_attach_aio_context(BlockDriverState *bs, AioContext *new_ctx)
{
VitastorClient *client = bs->opaque;
int i;
client->ctx = new_ctx;
#if defined VITASTOR_C_API_VERSION && VITASTOR_C_API_VERSION >= 2
if (client->uring_eventfd >= 0)
{
universal_aio_set_fd_handler(new_ctx, client->uring_eventfd, vitastor_uring_handler, NULL, client);
}
#endif
for (i = 0; i < client->fd_count; i++)
{
VitastorFdData *fdd = client->fds[i];
universal_aio_set_fd_handler(new_ctx, fdd->fd,
fdd->fd_read ? vitastor_aio_fd_read : NULL,
fdd->fd_write ? vitastor_aio_fd_write : NULL,
fdd);
}
}
#if QEMU_VERSION_MAJOR >= 3 || QEMU_VERSION_MAJOR == 2 && QEMU_VERSION_MINOR >= 2
static void vitastor_refresh_filename(BlockDriverState *bs)
{
@@ -832,6 +880,25 @@ static int vitastor_refresh_limits(BlockDriverState *bs)
// return 0;
//}
// Move the running coroutine to the BlockDriverState's home AioContext.
//
// The block-coroutine-wrapper generator sets poll_state.ctx to
// qemu_get_current_aio_context() in the sync wrappers (bdrv_flush(),
// bdrv_pread() etc.). When bdrv_flush_all() runs under BQL from outside the
// bs's iothread (e.g. on the migration thread inside do_vm_stop()), that is
// the main AioContext, not the iothread that actually owns the bs. The
// coroutine then runs on the wrong context while completions are delivered on
// the iothread, and racing aio_co_schedule() vs. qemu_aio_coroutine_enter()
// on the same coroutine triggers "Co-routine was already scheduled in
// aio_co_schedule" and aborts the process (observed during live migration).
//
// aio_co_reschedule_self() is a no-op when we are already on the target ctx.
#if QEMU_VERSION_MAJOR > 5 || QEMU_VERSION_MAJOR == 5 && QEMU_VERSION_MINOR >= 2
#define vitastor_co_pin_to_bs_ctx(bs) aio_co_reschedule_self(bdrv_get_aio_context(bs))
#else
#define vitastor_co_pin_to_bs_ctx(bs) ((void)0)
#endif
static void vitastor_co_init_task(BlockDriverState *bs, VitastorRPC *task)
{
*task = (VitastorRPC) {
@@ -890,6 +957,7 @@ static int coroutine_fn vitastor_co_preadv(BlockDriverState *bs,
{
VitastorClient *client = bs->opaque;
VitastorRPC task;
vitastor_co_pin_to_bs_ctx(bs);
vitastor_co_init_task(bs, &task);
task.iov = iov;
@@ -918,6 +986,7 @@ static int coroutine_fn vitastor_co_pwritev(BlockDriverState *bs,
{
VitastorClient *client = bs->opaque;
VitastorRPC task;
vitastor_co_pin_to_bs_ctx(bs);
vitastor_co_init_task(bs, &task);
task.iov = iov;
@@ -991,6 +1060,7 @@ static int coroutine_fn vitastor_co_block_status(BlockDriverState *bs,
#endif
VitastorRPC task;
VitastorClient *client = bs->opaque;
vitastor_co_pin_to_bs_ctx(bs);
uint64_t inode = client->watch ? vitastor_c_inode_get_num(client->watch) : client->inode;
uint8_t bit = 0;
if (client->last_bitmap && client->last_bitmap_inode == inode &&
@@ -1103,6 +1173,7 @@ static int coroutine_fn vitastor_co_flush(BlockDriverState *bs)
{
VitastorClient *client = bs->opaque;
VitastorRPC task;
vitastor_co_pin_to_bs_ctx(bs);
vitastor_co_init_task(bs, &task);
qemu_mutex_lock(&client->mutex);
@@ -1188,6 +1259,11 @@ static BlockDriver bdrv_vitastor = {
#endif
.bdrv_close = vitastor_close,
// Re-register fd handlers when the bs is moved to a different AioContext
// (live migration, drain, iothread reassignment).
.bdrv_detach_aio_context = vitastor_detach_aio_context,
.bdrv_attach_aio_context = vitastor_attach_aio_context,
// Option list for the create operation
#if QEMU_VERSION_MAJOR >= 3 || QEMU_VERSION_MAJOR == 2 && QEMU_VERSION_MINOR > 0
.create_opts = &vitastor_create_opts,
+52 -6
View File
@@ -62,6 +62,13 @@ const char *help_text =
"All usual Vitastor config options like --config_path <path_to_config> may also be specified in CLI.\n"
;
struct ublk_request
{
uint64_t ublk_cmd;
int index;
int result;
};
class ublk_server
{
protected:
@@ -241,7 +248,7 @@ help:
// Create client
epmgr = new epoll_manager_t(ringloop);
cli = new cluster_client_t(ringloop, epmgr->tfd, cfg);
cli = cluster_client_t::create(ringloop, epmgr->tfd, cfg);
// cli->config contains merged config
if (!cfg["queue_depth"].is_null())
@@ -273,7 +280,7 @@ help:
}
if (!inode)
{
watch = cli->st_cli.watch_inode(image_name);
watch = cli->st_cli->watch_inode(image_name);
device_size = watch->cfg.size;
if (!watch->cfg.num || !device_size)
{
@@ -282,9 +289,9 @@ help:
exit(1);
}
}
const bool writeback = !cli->get_immediate_commit(inode);
auto pool_it = cli->st_cli.pool_config.find(INODE_POOL(inode ? inode : watch->cfg.num));
if (pool_it == cli->st_cli.pool_config.end())
const bool writeback = !cli->get_immediate_commit(inode ? inode : watch->cfg.num);
auto pool_it = cli->st_cli->pool_config.find(INODE_POOL(inode ? inode : watch->cfg.num));
if (pool_it == cli->st_cli->pool_config.end())
{
fprintf(stderr, "Pool %u does not exist\n", INODE_POOL(inode ? inode : watch->cfg.num));
exit(1);
@@ -331,6 +338,11 @@ help:
daemonize_fork(notifyfd);
close(notifyfd[0]);
}
consumer.loop = [this]()
{
submit_postponed();
};
ringloop->register_consumer(&consumer);
start_device(recover);
if (pidfile != "")
write_pid();
@@ -350,6 +362,7 @@ help:
ringloop->wait();
}
cli->flush();
ringloop->unregister_consumer(&consumer);
delete cli;
delete epmgr;
cli = NULL;
@@ -553,6 +566,8 @@ protected:
ublksrv_ctrl_dev_info ublk_dev = {};
ublksrv_io_desc *ublk_queue = NULL;
std::vector<uint8_t*> buffers;
ring_consumer_t consumer;
std::vector<ublk_request> postponed_requests;
void open_control()
{
@@ -734,9 +749,19 @@ protected:
ctrl_fd = -1;
}
void submit_request(uint64_t ublk_cmd, int i, int res)
bool submit_request(uint64_t ublk_cmd, int i, int res)
{
io_uring_sqe *sqe = ringloop->get_sqe();
if (!sqe)
{
// Handle full io_uring by postponing the request
postponed_requests.push_back((ublk_request){
.ublk_cmd = ublk_cmd,
.index = i,
.result = res,
});
return false;
}
ring_data_t* data = ((ring_data_t*)sqe->user_data);
sqe->fd = cdev_fd;
sqe->opcode = IORING_OP_URING_CMD;
@@ -750,6 +775,22 @@ protected:
cmd->addr = (uint64_t)buffers[i];
cmd->result = res;
data->callback = [this, i](ring_data_t *data) { exec_request(data->res, i); };
return true;
}
void submit_postponed()
{
int sent = 0;
while (postponed_requests.size())
{
ublk_request r = postponed_requests.back();
postponed_requests.pop_back();
if (!submit_request(r.ublk_cmd, r.index, r.result))
break;
sent++;
}
if (sent)
ringloop->submit();
}
void exec_request(int res, int i)
@@ -864,6 +905,11 @@ protected:
int sync_ublk_cmd(uint32_t cmd_op, void *addr, uint32_t len, uint16_t dev_path_len = 0, uint64_t data0 = 0)
{
io_uring_sqe *sqe = ringloop->get_sqe();
if (!sqe)
{
fprintf(stderr, "Error: io_uring is full when trying to execute a control command\n");
exit(1);
}
sqe->fd = ctrl_fd;
sqe->opcode = IORING_OP_URING_CMD;
sqe->ioprio = 0;
+1 -1
View File
@@ -6,7 +6,7 @@ includedir=${prefix}/@CMAKE_INSTALL_INCLUDEDIR@
Name: Vitastor
Description: Vitastor client library
Version: 3.0.12
Version: 3.0.15
Libs: -L${libdir} -lvitastor_client
Cflags: -I${includedir}
+13 -13
View File
@@ -103,7 +103,7 @@ vitastor_c *vitastor_c_create_qemu(QEMUSetFDHandler *aio_set_fd_handler, void *a
rdma_device, rdma_port_num, rdma_gid_index, rdma_mtu, log_level
);
auto self = vitastor_c_create_qemu_common(aio_set_fd_handler, aio_context);
self->cli = new cluster_client_t(NULL, self->tfd, cfg_json);
self->cli = cluster_client_t::create(NULL, self->tfd, cfg_json);
return self;
}
@@ -126,7 +126,7 @@ vitastor_c *vitastor_c_create_qemu_uring(QEMUSetFDHandler *aio_set_fd_handler, v
);
auto self = vitastor_c_create_qemu_common(aio_set_fd_handler, aio_context);
self->ringloop = ringloop;
self->cli = new cluster_client_t(self->ringloop, self->tfd, cfg_json);
self->cli = cluster_client_t::create(self->ringloop, self->tfd, cfg_json);
ringloop->loop();
return self;
}
@@ -150,7 +150,7 @@ vitastor_c *vitastor_c_create_uring(const char *config_path, const char *etcd_ho
vitastor_c *self = new vitastor_c;
self->ringloop = ringloop;
self->epmgr = new epoll_manager_t(self->ringloop);
self->cli = new cluster_client_t(self->ringloop, self->epmgr->tfd, cfg_json);
self->cli = cluster_client_t::create(self->ringloop, self->epmgr->tfd, cfg_json);
ringloop->loop();
return self;
}
@@ -191,7 +191,7 @@ vitastor_c *vitastor_c_create_uring_json(const char **options, int options_len)
vitastor_c *self = new vitastor_c;
self->ringloop = ringloop;
self->epmgr = new epoll_manager_t(self->ringloop);
self->cli = new cluster_client_t(self->ringloop, self->epmgr->tfd, cfg_json);
self->cli = cluster_client_t::create(self->ringloop, self->epmgr->tfd, cfg_json);
ringloop->loop();
return self;
}
@@ -206,7 +206,7 @@ vitastor_c *vitastor_c_create_epoll_json(const char **options, int options_len)
json11::Json cfg_json(cfg);
vitastor_c *self = new vitastor_c;
self->epmgr = new epoll_manager_t(NULL);
self->cli = new cluster_client_t(NULL, self->epmgr->tfd, cfg_json);
self->cli = cluster_client_t::create(NULL, self->epmgr->tfd, cfg_json);
return self;
}
@@ -396,7 +396,7 @@ void vitastor_c_watch_inode(vitastor_c *client, char *image, VitastorIOHandler c
{
client->cli->on_ready([=]()
{
auto watch = client->cli->st_cli.watch_inode(std::string(image));
auto watch = client->cli->st_cli->watch_inode(std::string(image));
cb(opaque, (long)watch);
});
if (client->ringloop)
@@ -407,7 +407,7 @@ void vitastor_c_watch_inode(vitastor_c *client, char *image, VitastorIOHandler c
void vitastor_c_close_watch(vitastor_c *client, void *handle)
{
client->cli->st_cli.close_watch((inode_watch_t*)handle);
client->cli->st_cli->close_watch((inode_watch_t*)handle);
}
uint64_t vitastor_c_inode_get_size(void *handle)
@@ -424,8 +424,8 @@ uint64_t vitastor_c_inode_get_num(void *handle)
uint32_t vitastor_c_inode_get_block_size(vitastor_c *client, uint64_t inode_num)
{
auto pool_it = client->cli->st_cli.pool_config.find(INODE_POOL(inode_num));
if (pool_it == client->cli->st_cli.pool_config.end())
auto pool_it = client->cli->st_cli->pool_config.find(INODE_POOL(inode_num));
if (pool_it == client->cli->st_cli->pool_config.end())
return 0;
auto & pool_cfg = pool_it->second;
uint32_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
@@ -434,8 +434,8 @@ uint32_t vitastor_c_inode_get_block_size(vitastor_c *client, uint64_t inode_num)
uint32_t vitastor_c_inode_get_bitmap_granularity(vitastor_c *client, uint64_t inode_num)
{
auto pool_it = client->cli->st_cli.pool_config.find(INODE_POOL(inode_num));
if (pool_it == client->cli->st_cli.pool_config.end())
auto pool_it = client->cli->st_cli->pool_config.find(INODE_POOL(inode_num));
if (pool_it == client->cli->st_cli->pool_config.end())
return 0;
// FIXME: READ_BITMAP may fails if parent bitmap granularity differs from inode bitmap granularity
return pool_it->second.bitmap_granularity;
@@ -471,8 +471,8 @@ uint64_t vitastor_c_inode_get_mod_revision(void *handle)
uint32_t vitastor_c_inode_get_immediate_commit(vitastor_c *client, uint64_t inode_num)
{
auto pool_it = client->cli->st_cli.pool_config.find(INODE_POOL(inode_num));
if (pool_it == client->cli->st_cli.pool_config.end())
auto pool_it = client->cli->st_cli->pool_config.find(INODE_POOL(inode_num));
if (pool_it == client->cli->st_cli->pool_config.end())
return 0;
return pool_it->second.immediate_commit;
}
+2 -1
View File
@@ -15,6 +15,7 @@ add_custom_command(
add_library(vitastor_cli STATIC
cli_common.cpp
cli_alloc_osd.cpp
cli_cpubench.cpp
cli_describe.cpp
cli_fix.cpp
cli_ls.cpp
@@ -50,5 +51,5 @@ add_executable(vitastor-cli
cli.cpp
)
target_link_libraries(vitastor-cli
vitastor_client
vitastor_client_int
)
+10 -3
View File
@@ -260,12 +260,15 @@ static const char* help_text =
"vitastor-cli rm-user|remove-user|delete-user <username>\n"
" Remove a user.\n"
"\n"
"vitastor-cli cpubench [--json]\n"
" Run CPU crypto performance tests: AES-256-GCM, AES-256-XTS and xxhash3.\n"
"\n"
"vitastor-cli serve\n"
" Start HTTP server able to handle CLI commands over a REST API. Options:\n"
" --bind_address ADDR Specify server IP address or addresses, separated by space. Default is 127.0.0.1.\n"
" --port 8080 Specify server port.\n"
" --server_cert FILE Path to server TLS certificate file (PEM format).\n"
" --server_key FILE Path to server TLS private key file.\n"
" --api_cert FILE Path to server TLS certificate file (PEM format).\n"
" --api_pkey FILE Path to server TLS private key file.\n"
" --client_ca FILE Path to file with TLS CA certificates used to validate client connections.\n"
"\n"
"Use vitastor-cli --help <command> for command details or vitastor-cli --help --all for all details.\n"
@@ -629,6 +632,10 @@ std::function<bool(cli_result_t &)> cli_tool_t::start(json11::Json::object cfg,
// Start HTTP server
action_cb = start_serve(cfg);
}
else if (cmd[0] == "cpubench")
{
action_cb = start_cpubench(cfg);
}
else
{
result = { .err = EOPNOTSUPP, .text = "unknown command: "+cmd[0].string_value() };
@@ -648,7 +655,7 @@ static int run(cli_tool_t *p, json11::Json::object cfg)
json11::Json cfg_j = cfg;
p->ringloop = new ring_loop_t(RINGLOOP_DEFAULT_SIZE);
p->epmgr = new epoll_manager_t(p->ringloop);
p->cli = new cluster_client_t(p->ringloop, p->epmgr->tfd, cfg_j);
p->cli = cluster_client_t::create(p->ringloop, p->epmgr->tfd, cfg_j);
p->loop_and_wait(action_cb, [&](const cli_result_t & r)
{
result = r;
+1
View File
@@ -67,6 +67,7 @@ public:
std::function<bool(cli_result_t &)> start(json11::Json::object cfg, cli_result_t & result);
std::function<bool(cli_result_t &)> start_alloc_osd(json11::Json);
std::function<bool(cli_result_t &)> start_cpubench(json11::Json);
std::function<bool(cli_result_t &)> start_create(json11::Json);
std::function<bool(cli_result_t &)> start_dd(json11::Json);
std::function<bool(cli_result_t &)> start_describe(json11::Json);
+4 -4
View File
@@ -35,7 +35,7 @@ struct alloc_osd_t
{ "target", "VERSION" },
{ "version", 0 },
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/osd/stats/"+std::to_string(new_id)
parent->cli->st_cli->etcd_prefix+"/osd/stats/"+std::to_string(new_id)
) },
},
} },
@@ -43,7 +43,7 @@ struct alloc_osd_t
json11::Json::object {
{ "request_put", json11::Json::object {
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/osd/stats/"+std::to_string(new_id)
parent->cli->st_cli->etcd_prefix+"/osd/stats/"+std::to_string(new_id)
) },
{ "value", base64_encode("{}") },
} },
@@ -52,8 +52,8 @@ struct alloc_osd_t
{ "failure", json11::Json::array {
json11::Json::object {
{ "request_range", json11::Json::object {
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/osd/stats/") },
{ "range_end", base64_encode(parent->cli->st_cli.etcd_prefix+"/osd/stats0") },
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/osd/stats/") },
{ "range_end", base64_encode(parent->cli->st_cli->etcd_prefix+"/osd/stats0") },
{ "keys_only", true },
} },
},
+17 -17
View File
@@ -17,8 +17,8 @@ bool cli_tool_t::check_image_perm(const inode_config_t & cfg, bool write)
json11::Json::object cli_tool_t::format_image(const inode_config_t & cfg)
{
auto pool_it = cli->st_cli.pool_config.find(INODE_POOL(cfg.num));
bool good_pool = pool_it != cli->st_cli.pool_config.end();
auto pool_it = cli->st_cli->pool_config.find(INODE_POOL(cfg.num));
bool good_pool = pool_it != cli->st_cli->pool_config.end();
auto img = json11::Json::object {
{ "name", cfg.name },
{ "size", cfg.size },
@@ -50,8 +50,8 @@ json11::Json::object cli_tool_t::format_image(const inode_config_t & cfg)
}
if (cfg.parent_id)
{
auto parent_it = cli->st_cli.inode_config.find(cfg.parent_id);
if (parent_it != cli->st_cli.inode_config.end())
auto parent_it = cli->st_cli->inode_config.find(cfg.parent_id);
if (parent_it != cli->st_cli->inode_config.end())
{
img["parent_name"] = parent_it->second.name;
}
@@ -64,8 +64,8 @@ json11::Json::object cli_tool_t::format_image(const inode_config_t & cfg)
void cli_tool_t::change_parent(inode_t cur, inode_t new_parent, cli_result_t *result)
{
auto cur_cfg_it = cli->st_cli.inode_config.find(cur);
if (cur_cfg_it == cli->st_cli.inode_config.end())
auto cur_cfg_it = cli->st_cli->inode_config.find(cur);
if (cur_cfg_it == cli->st_cli->inode_config.end())
{
char buf[128];
snprintf(buf, 128, "Inode 0x%jx disappeared", cur);
@@ -74,13 +74,13 @@ void cli_tool_t::change_parent(inode_t cur, inode_t new_parent, cli_result_t *re
}
inode_config_t new_cfg = cur_cfg_it->second;
std::string cur_name = new_cfg.name;
std::string cur_cfg_key = base64_encode(cli->st_cli.etcd_prefix+
std::string cur_cfg_key = base64_encode(cli->st_cli->etcd_prefix+
"/config/inode/"+std::to_string(INODE_POOL(cur))+
"/"+std::to_string(INODE_NO_POOL(cur)));
new_cfg.parent_id = new_parent;
json11::Json::object cur_cfg_json = cli->st_cli.serialize_inode_cfg(&new_cfg);
json11::Json::object cur_cfg_json = cli->st_cli->serialize_inode_cfg(&new_cfg);
waiting++;
cli->st_cli.etcd_txn_slow(json11::Json::object {
cli->st_cli->etcd_txn_slow(json11::Json::object {
{ "compare", json11::Json::array {
json11::Json::object {
{ "target", "MOD" },
@@ -109,8 +109,8 @@ void cli_tool_t::change_parent(inode_t cur, inode_t new_parent, cli_result_t *re
}
else if (new_parent)
{
auto new_parent_it = cli->st_cli.inode_config.find(new_parent);
std::string new_parent_name = new_parent_it != cli->st_cli.inode_config.end()
auto new_parent_it = cli->st_cli->inode_config.find(new_parent);
std::string new_parent_name = new_parent_it != cli->st_cli->inode_config.end()
? new_parent_it->second.name : "<unknown>";
*result = (cli_result_t){
.text = "Parent of layer "+cur_name+" (inode "+std::to_string(INODE_NO_POOL(cur))+
@@ -133,7 +133,7 @@ void cli_tool_t::change_parent(inode_t cur, inode_t new_parent, cli_result_t *re
void cli_tool_t::etcd_txn(json11::Json txn)
{
waiting++;
cli->st_cli.etcd_txn_slow(txn, [this](std::string err, json11::Json res)
cli->st_cli->etcd_txn_slow(txn, [this](std::string err, json11::Json res)
{
waiting--;
if (err != "")
@@ -147,7 +147,7 @@ void cli_tool_t::etcd_txn(json11::Json txn)
inode_config_t* cli_tool_t::get_inode_cfg(const std::string & name)
{
for (auto & ic: cli->st_cli.inode_config)
for (auto & ic: cli->st_cli->inode_config)
{
if (ic.second.name == name)
{
@@ -231,11 +231,11 @@ void cli_tool_t::iterate_kvs_1(json11::Json kvs, const std::string & prefix, std
bool is_pool = prefix == "/pool/stats/";
for (auto & kv_item: kvs.array_items())
{
auto kv = cli->st_cli.parse_etcd_kv(kv_item);
auto kv = cli->st_cli->parse_etcd_kv(kv_item);
uint64_t num = 0;
char null_byte = 0;
// OSD or pool number
int scanned = sscanf(kv.key.substr(cli->st_cli.etcd_prefix.size() + prefix.size()).c_str(), "%ju%c", &num, &null_byte);
int scanned = sscanf(kv.key.substr(cli->st_cli->etcd_prefix.size() + prefix.size()).c_str(), "%ju%c", &num, &null_byte);
if (scanned != 1 || !num || is_pool && num >= POOL_ID_MAX)
{
fprintf(stderr, "Invalid key in etcd: %s\n", kv.key.c_str());
@@ -250,12 +250,12 @@ void cli_tool_t::iterate_kvs_2(json11::Json kvs, const std::string & prefix, std
bool is_inode = prefix == "/config/inode/" || prefix == "/inode/stats/";
for (auto & kv_item: kvs.array_items())
{
auto kv = cli->st_cli.parse_etcd_kv(kv_item);
auto kv = cli->st_cli->parse_etcd_kv(kv_item);
pool_id_t pool_id = 0;
uint64_t num = 0;
char null_byte = 0;
// pool+pg or pool+inode
int scanned = sscanf(kv.key.substr(cli->st_cli.etcd_prefix.size() + prefix.size()).c_str(),
int scanned = sscanf(kv.key.substr(cli->st_cli->etcd_prefix.size() + prefix.size()).c_str(),
"%u/%ju%c", &pool_id, &num, &null_byte);
if (scanned != 2 || !pool_id || is_inode && INODE_POOL(num) || !is_inode && num >= UINT32_MAX)
{
+374
View File
@@ -0,0 +1,374 @@
// Copyright (c) Vitaliy Filippov, 2019+
// License: VNPL-1.1 (see README.md for details)
#include <openssl/rand.h>
#include "cli.h"
#include "messenger.h"
#include "msgr_encrypt.h"
#include "msgr_op.h"
#include "str_util.h"
#include "xxhash.h"
#include "xxh_x86dispatch.h"
// Prevent the compiler from proving that memory is unused.
static inline void clobber_memory(const void* p, size_t n)
{
#if defined(__GNUC__) || defined(__clang__)
__asm__ __volatile__("" : : "r"(p), "m"(*(const char(*)[1])p) : "memory");
#else
(void)p; (void)n;
#endif
}
void init_gcm(osd_client_t *cl)
{
cl->my_key.resize(AES_256_GCM_KEY_SIZE+AES_256_GCM_IV_SIZE);
RAND_bytes(cl->my_key.data(), AES_256_GCM_KEY_SIZE+AES_256_GCM_IV_SIZE);
#ifdef WITH_ISAL_CRYPTO
cl->enc_ctx = (isal_gcm_context_data*)malloc_or_die(sizeof(isal_gcm_context_data));
isal_aes_gcm_pre_256(cl->my_key.data(), &cl->my_key_isal);
#else
cl->enc_ctx = EVP_CIPHER_CTX_new();
assert(cl->enc_ctx);
int r = EVP_EncryptInit_ex(cl->enc_ctx, EVP_aes_256_gcm(), NULL, NULL, NULL);
if (r != 1)
{
fprintf(stderr, "EncryptInit error: ");
ERR_print_errors_fp(stderr);
abort();
}
#endif
}
void init_gcm_round(osd_client_t *cl)
{
#ifdef WITH_ISAL_CRYPTO
int r = isal_aes_gcm_init_256(&cl->my_key_isal, cl->enc_ctx, cl->my_key.data() + AES_256_GCM_KEY_SIZE, NULL, 0);
if (r != 0)
{
fprintf(stderr, "isal_aes_gcm_init_256 error %d\n", r);
abort();
}
#else
int r = EVP_EncryptInit_ex(cl->enc_ctx, NULL, NULL, (uint8_t*)cl->my_key.data(), cl->my_key.data() + AES_256_GCM_KEY_SIZE);
if (r != 1)
{
fprintf(stderr, "EncryptInit error: ");
ERR_print_errors_fp(stderr);
abort();
}
#endif
}
void finalize_gcm_round(osd_client_t *cl)
{
uint8_t tag[16];
#ifdef WITH_ISAL_CRYPTO
int r = isal_aes_gcm_enc_256_finalize(&cl->my_key_isal, cl->enc_ctx, tag, 16);
assert(!r);
#else
int actual_out = 0;
int r = EVP_EncryptFinal_ex(cl->enc_ctx, NULL, &actual_out);
if (r != 1)
{
fprintf(stderr, "EncryptFinal error: ");
ERR_print_errors_fp(stderr);
abort();
}
assert(actual_out == 0);
r = EVP_CIPHER_CTX_ctrl(cl->enc_ctx, EVP_CTRL_GCM_GET_TAG, 16, tag);
assert(r == 1);
#endif
clobber_memory(&tag, sizeof(tag));
}
void encrypt_gcm(osd_client_t *cl, uint8_t *in_buf, uint8_t *out_buf, size_t bufsize)
{
#ifdef WITH_ISAL_CRYPTO
int r = isal_aes_gcm_enc_256_update(&cl->my_key_isal, cl->enc_ctx, out_buf, in_buf, bufsize);
assert(!r);
#else
int actual_out;
if (EVP_EncryptUpdate(cl->enc_ctx, out_buf, &actual_out, in_buf, bufsize) != 1)
{
fprintf(stderr, "EncryptUpdate error: ");
ERR_print_errors_fp(stderr);
abort();
}
assert(actual_out == bufsize);
#endif
}
void bench_aes_xts(uint64_t millis, size_t bufsize, int csum_status, bool quiet, int json)
{
size_t check_interval = 100;
if (!quiet && !json)
{
printf("%s %s block... ",
csum_status == MSGR_CSUM_GCM ? "AES-256-XTS + AES-256-GCM encrypt" :
(csum_status == MSGR_CSUM_PAYLOAD ? "AES-256-GCM encrypt header + AES-256-XTS encrypt + xxhash3" :
(csum_status == MSGR_CSUM_FULL ? "AES-256-XTS encrypt + xxhash3" : "AES-256-XTS encrypt")),
format_size(bufsize).c_str());
}
uint8_t xts_key[AES_256_XTS_KEY_SIZE];
RAND_bytes(xts_key, AES_256_XTS_KEY_SIZE);
uint8_t *in_buf = (uint8_t*)malloc_or_die(bufsize);
uint8_t *out_buf = (uint8_t*)malloc_or_die(bufsize);
XXH3_state_t *hash_state = NULL;
osd_client_t *cl = new osd_client_t();
cl->proto_csum_status = csum_status;
if (csum_status == MSGR_CSUM_GCM || csum_status == MSGR_CSUM_PAYLOAD)
{
init_gcm(cl);
}
if (csum_status == MSGR_CSUM_PAYLOAD || csum_status == MSGR_CSUM_FULL)
{
hash_state = XXH3_createState();
}
op_aes_xts_encrypt_t enc;
timespec tv_begin, tv_end;
clock_gettime(CLOCK_REALTIME, &tv_begin);
uint64_t iters = 0;
while (true)
{
if (csum_status == MSGR_CSUM_GCM)
{
init_gcm_round(cl);
}
if (csum_status == MSGR_CSUM_PAYLOAD || csum_status == MSGR_CSUM_FULL)
{
XXH3_64bits_reset(hash_state);
}
if (csum_status == MSGR_CSUM_PAYLOAD)
{
init_gcm_round(cl);
encrypt_gcm(cl, in_buf, out_buf, OSD_PACKET_SIZE);
}
enc.start(cl, xts_key, 0, 4096);
size_t done_in = 0, done_out = 0;
while (done_in < bufsize || done_out < bufsize)
{
enc.update(in_buf, bufsize, out_buf, bufsize, done_in, done_out);
}
if (csum_status == MSGR_CSUM_PAYLOAD || csum_status == MSGR_CSUM_FULL)
{
XXH3_64bits_update(hash_state, in_buf, bufsize);
}
if (csum_status == MSGR_CSUM_GCM)
{
finalize_gcm_round(cl);
}
else if (csum_status)
{
uint64_t hash = XXH3_64bits_digest(hash_state);
clobber_memory(&hash, sizeof(hash));
if (csum_status == MSGR_CSUM_PAYLOAD)
{
encrypt_gcm(cl, (uint8_t*)&hash, out_buf+OSD_PACKET_SIZE, sizeof(hash));
finalize_gcm_round(cl);
}
}
if (!(iters % check_interval))
{
clock_gettime(CLOCK_REALTIME, &tv_end);
uint64_t passed = tv_end.tv_sec*1000 - tv_begin.tv_sec*1000 + tv_end.tv_nsec/1000000 - tv_begin.tv_nsec/1000000;
if (passed >= millis)
break;
else if (passed < 10)
check_interval *= 10;
}
iters++;
}
uint64_t result_ms = tv_end.tv_sec*1000 - tv_begin.tv_sec*1000 + tv_end.tv_nsec/1000000 - tv_begin.tv_nsec/1000000;
double result_mbps = 1000.0 * bufsize / 1048576 * iters / result_ms;
if (!quiet)
{
if (json)
{
printf(
"%s{ \"xxhash3\": %s, \"aes-256-xts\": true, \"aes-256-gcm\": %s, \"bufsize\": %zu, \"iters\": %ju, \"ms\": %ju, \"mbps\": %.2f }",
json == 1 ? "" : ",\n ",
csum_status == MSGR_CSUM_PAYLOAD || csum_status == MSGR_CSUM_FULL ? "true" : "false",
csum_status == MSGR_CSUM_PAYLOAD ? "\"header\"" : (csum_status == MSGR_CSUM_GCM ? "\"full\"" : "\"none\""),
bufsize, iters, result_ms, result_mbps
);
}
else
printf("%ju iterations in %ju ms = %.2f MB/s\n", iters, result_ms, result_mbps);
}
if (csum_status == MSGR_CSUM_PAYLOAD || csum_status == MSGR_CSUM_FULL)
{
XXH3_freeState(hash_state);
hash_state = NULL;
}
delete cl;
free(out_buf);
free(in_buf);
}
void bench_aes_gcm(uint64_t millis, size_t bufsize, bool with_csum, int json)
{
size_t check_interval = 100;
if (!json)
{
printf("%s %s block... ",
with_csum ? "AES-256-GCM encrypt header + xxhash3" : "AES-256-GCM encrypt header and",
format_size(bufsize).c_str());
}
uint8_t *in_buf = (uint8_t*)malloc_or_die(bufsize);
uint8_t *out_buf = (uint8_t*)malloc_or_die(bufsize);
XXH3_state_t *hash_state = NULL;
osd_client_t *cl = new osd_client_t();
init_gcm(cl);
if (with_csum)
{
hash_state = XXH3_createState();
}
timespec tv_begin, tv_end;
clock_gettime(CLOCK_REALTIME, &tv_begin);
uint64_t iters = 0;
while (true)
{
init_gcm_round(cl);
encrypt_gcm(cl, in_buf, out_buf, OSD_PACKET_SIZE);
if (with_csum)
{
XXH3_64bits_reset(hash_state);
XXH3_64bits_update(hash_state, in_buf, bufsize);
uint64_t hash = XXH3_64bits_digest(hash_state);
encrypt_gcm(cl, (uint8_t*)&hash, out_buf+OSD_PACKET_SIZE, sizeof(hash));
}
else
{
encrypt_gcm(cl, in_buf, out_buf, bufsize);
}
finalize_gcm_round(cl);
if (!(iters % check_interval))
{
clock_gettime(CLOCK_REALTIME, &tv_end);
uint64_t passed = tv_end.tv_sec*1000 - tv_begin.tv_sec*1000 + tv_end.tv_nsec/1000000 - tv_begin.tv_nsec/1000000;
if (passed >= millis)
break;
else if (passed < 10)
check_interval *= 10;
}
iters++;
}
uint64_t result_ms = tv_end.tv_sec*1000 - tv_begin.tv_sec*1000 + tv_end.tv_nsec/1000000 - tv_begin.tv_nsec/1000000;
double result_mbps = 1000.0 * bufsize / 1048576 * iters / result_ms;
if (json)
{
printf(
"%s{ \"xxhash3\": %s, \"aes-256-xts\": false, \"aes-256-gcm\": %s, \"bufsize\": %zu, \"iters\": %ju, \"ms\": %ju, \"mbps\": %.2f }",
json == 1 ? "" : ",\n ",
with_csum ? "true" : "false",
with_csum ? "\"header\"" : "\"full\"",
bufsize, iters, result_ms, result_mbps
);
}
else
{
printf("%ju iterations in %ju ms = %.2f MB/s\n", iters, result_ms, result_mbps);
}
if (with_csum)
{
XXH3_freeState(hash_state);
hash_state = NULL;
}
delete cl;
free(out_buf);
free(in_buf);
}
void bench_xxh(uint64_t millis, size_t bufsize, int json)
{
size_t check_interval = 100;
if (!json)
{
printf("xxhash3 %s block... ", format_size(bufsize).c_str());
}
uint8_t *in_buf = (uint8_t*)malloc_or_die(bufsize);
XXH3_state_t *hash_state = XXH3_createState();
timespec tv_begin, tv_end;
clock_gettime(CLOCK_REALTIME, &tv_begin);
uint64_t iters = 0;
while (true)
{
XXH3_64bits_reset(hash_state);
XXH3_64bits_update(hash_state, in_buf, bufsize);
uint64_t hash = XXH3_64bits_digest(hash_state);
clobber_memory(&hash, sizeof(hash));
if (!(iters % check_interval))
{
clock_gettime(CLOCK_REALTIME, &tv_end);
uint64_t passed = tv_end.tv_sec*1000 - tv_begin.tv_sec*1000 + tv_end.tv_nsec/1000000 - tv_begin.tv_nsec/1000000;
if (passed >= millis)
break;
else if (passed < 10)
check_interval *= 10;
}
iters++;
}
uint64_t result_ms = tv_end.tv_sec*1000 - tv_begin.tv_sec*1000 + tv_end.tv_nsec/1000000 - tv_begin.tv_nsec/1000000;
double result_mbps = 1000.0 * bufsize / 1048576 * iters / result_ms;
if (json)
{
printf(
"%s{ \"xxhash3\": true, \"aes-256-xts\": false, \"aes-256-gcm\": \"none\", \"bufsize\": %zu, \"iters\": %ju, \"ms\": %ju, \"mbps\": %.2f }",
json == 1 ? "" : ",\n ",
bufsize, iters, result_ms, result_mbps
);
}
else
{
printf("%ju iterations in %ju ms = %.2f MB/s\n", iters, result_ms, result_mbps);
}
XXH3_freeState(hash_state);
hash_state = NULL;
free(in_buf);
}
// Run hardware performance tests (for now, only encryption-related)
std::function<bool(cli_result_t &)> cli_tool_t::start_cpubench(json11::Json cfg)
{
int json = !!json_output;
if (!json_output)
{
printf("Vitastor transport encryption benchmark (AES-256-GCM, AES-256-XTS and xxhash3)\n");
printf("\nWarmup...\n");
}
else
printf("[\n ");
bench_aes_xts(1000, 1048576, 0, true, false);
if (!json_output)
printf("\nNo transport encryption, data checksums enabled, e2e unencrypted image\n");
bench_xxh(2000, 1048576, json ? json++ : 0);
bench_xxh(2000, 4096, json ? json++ : 0);
if (!json_output)
printf("\nHeader encryption with payload checksums, e2e unencrypted image\n");
bench_aes_gcm(2000, 1048576, true, json ? json++ : 0);
bench_aes_gcm(2000, 4096, true, json ? json++ : 0);
if (!json_output)
printf("\nFull transport encryption, e2e unencrypted image\n");
bench_aes_gcm(2000, 1048576, false, json ? json++ : 0);
bench_aes_gcm(2000, 4096, false, json ? json++ : 0);
if (!json_output)
printf("\nNo transport encryption, no checksums, e2e encrypted image\n");
bench_aes_xts(2000, 1048576, 0, false, json ? json++ : 0);
bench_aes_xts(2000, 4096, 0, false, json ? json++ : 0);
if (!json_output)
printf("\nNo transport encryption, e2e encrypted image, data checksums enabled\n");
bench_aes_xts(2000, 1048576, MSGR_CSUM_FULL, false, json ? json++ : 0);
bench_aes_xts(2000, 4096, MSGR_CSUM_FULL, false, json ? json++ : 0);
if (!json_output)
printf("\nHeader encryption with payload checksums, e2e encrypted image\n");
bench_aes_xts(2000, 1048576, MSGR_CSUM_PAYLOAD, false, json ? json++ : 0);
bench_aes_xts(2000, 4096, MSGR_CSUM_PAYLOAD, false, json ? json++ : 0);
if (!json_output)
printf("\nFull transport encryption, e2e encrypted image\n");
bench_aes_xts(2000, 1048576, MSGR_CSUM_GCM, false, json ? json++ : 0);
bench_aes_xts(2000, 4096, MSGR_CSUM_GCM, false, json ? json++ : 0);
if (json_output)
printf("\n]\n");
return NULL;
}
+34 -34
View File
@@ -53,7 +53,7 @@ struct image_creator_t
void loop()
{
auto & pools = parent->cli->st_cli.pool_config;
auto & pools = parent->cli->st_cli->pool_config;
if (state >= 1)
goto resume_1;
if (image_name == "")
@@ -125,8 +125,8 @@ struct image_creator_t
{
return true;
}
auto pool_it = parent->cli->st_cli.pool_config.find(new_pool_id);
if (pool_it == parent->cli->st_cli.pool_config.end() ||
auto pool_it = parent->cli->st_cli->pool_config.find(new_pool_id);
if (pool_it == parent->cli->st_cli->pool_config.end() ||
(pool_it->second.creator_group == "" || parent->user->groups.find(pool_it->second.creator_group) == parent->user->groups.end()))
{
result = (cli_result_t){ .err = EACCES, .text = "Pool image create permission denied" };
@@ -142,7 +142,7 @@ struct image_creator_t
goto resume_2;
else if (state == 3)
goto resume_3;
for (auto & ic: parent->cli->st_cli.inode_config)
for (auto & ic: parent->cli->st_cli->inode_config)
{
if (ic.second.name == image_name)
{
@@ -222,7 +222,7 @@ resume_3:
} while (!parent->etcd_result["succeeded"].bool_value());
// Save into inode_config for library users to be able to take it from there immediately
new_cfg.mod_revision = parent->etcd_result["header"]["revision"].uint64_value();
parent->cli->st_cli.insert_inode_config(new_cfg);
parent->cli->st_cli->insert_inode_config(new_cfg);
auto img = parent->format_image(new_cfg);
result = (cli_result_t){
.err = 0,
@@ -240,8 +240,8 @@ resume_3:
goto resume_3;
else if (state == 4)
goto resume_4;
// FIXME: take all info from etcd requests, not mixed with st_cli.inode_config
for (auto & ic: parent->cli->st_cli.inode_config)
// FIXME: take all info from etcd requests, not mixed with st_cli->inode_config
for (auto & ic: parent->cli->st_cli->inode_config)
{
if (ic.second.name == image_name+"@"+new_snap)
{
@@ -307,10 +307,10 @@ resume_4:
} while (!parent->etcd_result["succeeded"].bool_value());
// Save into inode_config for library users to be able to take it from there immediately
new_cfg.mod_revision = parent->etcd_result["header"]["revision"].uint64_value();
parent->cli->st_cli.insert_inode_config(new_cfg);
parent->cli->st_cli->insert_inode_config(new_cfg);
{
auto new_pool_it = parent->cli->st_cli.pool_config.find(new_pool_id);
new_pool_name = new_pool_it != parent->cli->st_cli.pool_config.end() ? new_pool_it->second.name : "";
auto new_pool_it = parent->cli->st_cli->pool_config.find(new_pool_id);
new_pool_name = new_pool_it != parent->cli->st_cli->pool_config.end() ? new_pool_it->second.name : "";
}
result = (cli_result_t){
.err = 0,
@@ -337,7 +337,7 @@ resume_4:
return json11::Json::object {
{ "request_range", json11::Json::object {
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/index/maxid/"+std::to_string(new_pool_id)
parent->cli->st_cli->etcd_prefix+"/index/maxid/"+std::to_string(new_pool_id)
) },
} },
};
@@ -349,13 +349,13 @@ resume_4:
max_id_mod_rev = 0;
if (response["response_range"]["kvs"].array_items().size() > 0)
{
auto kv = parent->cli->st_cli.parse_etcd_kv(response["response_range"]["kvs"][0]);
auto kv = parent->cli->st_cli->parse_etcd_kv(response["response_range"]["kvs"][0]);
new_id = 1+INODE_NO_POOL(kv.value.uint64_value());
max_id_mod_rev = kv.mod_revision;
}
// Also check existing inodes - for the case when some inodes are created without changing /index/maxid
auto ino_it = parent->cli->st_cli.inode_config.lower_bound(INODE_WITH_POOL(new_pool_id+1, 0));
if (ino_it != parent->cli->st_cli.inode_config.begin())
auto ino_it = parent->cli->st_cli->inode_config.lower_bound(INODE_WITH_POOL(new_pool_id+1, 0));
if (ino_it != parent->cli->st_cli->inode_config.begin())
{
ino_it--;
if (INODE_POOL(ino_it->first) == new_pool_id && new_id < 1+INODE_NO_POOL(ino_it->first))
@@ -374,7 +374,7 @@ resume_4:
json11::Json::object {
{ "request_range", json11::Json::object {
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/index/image/"+image_name
parent->cli->st_cli->etcd_prefix+"/index/image/"+image_name
) },
} },
},
@@ -395,7 +395,7 @@ resume_2:
idx_mod_rev = 0;
if (parent->etcd_result["responses"][1]["response_range"]["kvs"].array_items().size() == 0)
{
for (auto & ic: parent->cli->st_cli.inode_config)
for (auto & ic: parent->cli->st_cli->inode_config)
{
if (ic.second.name == image_name)
{
@@ -411,7 +411,7 @@ resume_2:
{
// FIXME: Parse kvs in etcd_state_client automatically
{
auto kv = parent->cli->st_cli.parse_etcd_kv(parent->etcd_result["responses"][1]["response_range"]["kvs"][0]);
auto kv = parent->cli->st_cli->parse_etcd_kv(parent->etcd_result["responses"][1]["response_range"]["kvs"][0]);
old_id = INODE_NO_POOL(kv.value["id"].uint64_value());
old_pool_id = (pool_id_t)kv.value["pool_id"].uint64_value();
idx_mod_rev = kv.mod_revision;
@@ -427,7 +427,7 @@ resume_2:
json11::Json::object {
{ "request_range", json11::Json::object {
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/config/inode/"+
parent->cli->st_cli->etcd_prefix+"/config/inode/"+
std::to_string(old_pool_id)+"/"+std::to_string(old_id)
) },
} },
@@ -445,8 +445,8 @@ resume_3:
return;
}
{
auto kv = parent->cli->st_cli.parse_etcd_kv(parent->etcd_result["responses"][0]["response_range"]["kvs"][0]);
cur_cfg = parent->cli->st_cli.deserialize_inode_cfg(INODE_WITH_POOL(old_pool_id, old_id), kv.value, kv.mod_revision);
auto kv = parent->cli->st_cli->parse_etcd_kv(parent->etcd_result["responses"][0]["response_range"]["kvs"][0]);
cur_cfg = parent->cli->st_cli->deserialize_inode_cfg(INODE_WITH_POOL(old_pool_id, old_id), kv.value, kv.mod_revision);
size = cur_cfg.size;
}
}
@@ -474,7 +474,7 @@ resume_3:
{
new_cfg.enc_key = cur_cfg.enc_key;
}
new_cfg.owner = http_context_get_ssl_cn(parent->cli->st_cli.get_http_ctx());
new_cfg.owner = parent->cli->st_cli->get_username();
if (!new_owner.empty())
{
new_cfg.owner = new_owner;
@@ -492,7 +492,7 @@ resume_3:
{ "target", "VERSION" },
{ "version", 0 },
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/config/inode/"+
parent->cli->st_cli->etcd_prefix+"/config/inode/"+
std::to_string(new_pool_id)+"/"+std::to_string(new_id)
) },
},
@@ -500,31 +500,31 @@ resume_3:
{ "target", "VERSION" },
{ "version", 0 },
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/index/image/"+image_name+
parent->cli->st_cli->etcd_prefix+"/index/image/"+image_name+
(new_snap != "" ? "@"+new_snap : "")
) },
},
json11::Json::object {
{ "target", "MOD" },
{ "mod_revision", max_id_mod_rev },
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/index/maxid/"+std::to_string(new_pool_id)) },
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/index/maxid/"+std::to_string(new_pool_id)) },
},
};
json11::Json::array success = json11::Json::array {
json11::Json::object {
{ "request_put", json11::Json::object {
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/config/inode/"+
parent->cli->st_cli->etcd_prefix+"/config/inode/"+
std::to_string(new_pool_id)+"/"+std::to_string(new_id)
) },
{ "value", base64_encode(
json11::Json(parent->cli->st_cli.serialize_inode_cfg(&new_cfg)).dump()
json11::Json(parent->cli->st_cli->serialize_inode_cfg(&new_cfg)).dump()
) },
} },
},
json11::Json::object {
{ "request_put", json11::Json::object {
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/index/image/"+image_name) },
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/index/image/"+image_name) },
{ "value", base64_encode(json11::Json(json11::Json::object{
{ "id", new_id },
{ "pool_id", (uint64_t)new_pool_id },
@@ -534,7 +534,7 @@ resume_3:
json11::Json::object {
{ "request_put", json11::Json::object {
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/index/maxid/"+
parent->cli->st_cli->etcd_prefix+"/index/maxid/"+
std::to_string(new_pool_id)
) },
{ "value", base64_encode(std::to_string(new_id)) }
@@ -545,7 +545,7 @@ resume_3:
json11::Json::object {
{ "request_range", json11::Json::object {
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/index/image/"+
parent->cli->st_cli->etcd_prefix+"/index/image/"+
image_name+(new_snap != "" ? "@"+new_snap : "")
) },
} },
@@ -560,29 +560,29 @@ resume_3:
{ "target", "MOD" },
{ "mod_revision", cur_cfg.mod_revision },
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/config/inode/"+
parent->cli->st_cli->etcd_prefix+"/config/inode/"+
std::to_string(old_pool_id)+"/"+std::to_string(old_id)
) },
});
checks.push_back(json11::Json::object {
{ "target", "MOD" },
{ "mod_revision", idx_mod_rev },
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/index/image/"+image_name) }
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/index/image/"+image_name) }
});
success.push_back(json11::Json::object {
{ "request_put", json11::Json::object {
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/config/inode/"+
parent->cli->st_cli->etcd_prefix+"/config/inode/"+
std::to_string(old_pool_id)+"/"+std::to_string(old_id)
) },
{ "value", base64_encode(
json11::Json(parent->cli->st_cli.serialize_inode_cfg(&snap_cfg)).dump()
json11::Json(parent->cli->st_cli->serialize_inode_cfg(&snap_cfg)).dump()
) },
} },
});
success.push_back(json11::Json::object {
{ "request_put", json11::Json::object {
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/index/image/"+image_name+"@"+new_snap) },
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/index/image/"+image_name+"@"+new_snap) },
{ "value", base64_encode(json11::Json(json11::Json::object{
{ "id", old_id },
{ "pool_id", (uint64_t)old_pool_id },
+26 -20
View File
@@ -52,19 +52,19 @@ struct dd_in_info_t
in_seekable = true;
if (iimg != "")
{
iwatch = parent->cli->st_cli.watch_inode(iimg);
iwatch = parent->cli->st_cli->watch_inode(iimg);
if (!iwatch->cfg.num)
{
result = (cli_result_t){ .err = ENOENT, .text = "Image "+iimg+" does not exist" };
parent->cli->st_cli.close_watch(iwatch);
parent->cli->st_cli->close_watch(iwatch);
iwatch = NULL;
return;
}
auto pool_it = parent->cli->st_cli.pool_config.find(INODE_POOL(iwatch->cfg.num));
if (pool_it == parent->cli->st_cli.pool_config.end())
auto pool_it = parent->cli->st_cli->pool_config.find(INODE_POOL(iwatch->cfg.num));
if (pool_it == parent->cli->st_cli->pool_config.end())
{
result = (cli_result_t){ .err = ENOENT, .text = "Pool of image "+iimg+" does not exist" };
parent->cli->st_cli.close_watch(iwatch);
parent->cli->st_cli->close_watch(iwatch);
iwatch = NULL;
return;
}
@@ -131,7 +131,7 @@ struct dd_in_info_t
{
if (iimg != "")
{
parent->cli->st_cli.close_watch(iwatch);
parent->cli->st_cli->close_watch(iwatch);
iwatch = NULL;
}
else if (ifile != "")
@@ -163,11 +163,11 @@ struct dd_out_info_t
pool_config_t *find_pool(cli_tool_t *parent, const std::string & name)
{
if (name == "" && parent->cli->st_cli.pool_config.size() == 1)
if (name == "" && parent->cli->st_cli->pool_config.size() == 1)
{
return &parent->cli->st_cli.pool_config.begin()->second;
return &parent->cli->st_cli->pool_config.begin()->second;
}
for (auto & pp: parent->cli->st_cli.pool_config)
for (auto & pp: parent->cli->st_cli->pool_config)
{
if (pp.second.name == name)
{
@@ -186,14 +186,14 @@ struct dd_out_info_t
if (oimg != "")
{
out_seekable = true;
owatch = parent->cli->st_cli.watch_inode(oimg);
owatch = parent->cli->st_cli->watch_inode(oimg);
if (owatch->cfg.num)
{
auto pool_it = parent->cli->st_cli.pool_config.find(INODE_POOL(owatch->cfg.num));
if (pool_it == parent->cli->st_cli.pool_config.end())
auto pool_it = parent->cli->st_cli->pool_config.find(INODE_POOL(owatch->cfg.num));
if (pool_it == parent->cli->st_cli->pool_config.end())
{
result = (cli_result_t){ .err = ENOENT, .text = "Pool of image "+oimg+" does not exist" };
parent->cli->st_cli.close_watch(owatch);
parent->cli->st_cli->close_watch(owatch);
owatch = NULL;
return true;
}
@@ -209,7 +209,7 @@ struct dd_out_info_t
else
{
result = (cli_result_t){ .err = ENOENT, .text = "Pool to create output image "+oimg+" is not specified" };
parent->cli->st_cli.close_watch(owatch);
parent->cli->st_cli->close_watch(owatch);
owatch = NULL;
return true;
}
@@ -224,14 +224,14 @@ struct dd_out_info_t
if (!out_create)
{
result = (cli_result_t){ .err = ENOENT, .text = "Image "+oimg+" does not exist" };
parent->cli->st_cli.close_watch(owatch);
parent->cli->st_cli->close_watch(owatch);
owatch = NULL;
return true;
}
if (!out_size)
{
result = (cli_result_t){ .err = ENOENT, .text = "Input size is unknown, specify size to create output image "+oimg };
parent->cli->st_cli.close_watch(owatch);
parent->cli->st_cli->close_watch(owatch);
owatch = NULL;
return true;
}
@@ -247,7 +247,7 @@ struct dd_out_info_t
if (!out_size)
{
result = (cli_result_t){ .err = ENOENT, .text = "Input size is unknown, specify size to truncate output image" };
parent->cli->st_cli.close_watch(owatch);
parent->cli->st_cli->close_watch(owatch);
owatch = NULL;
return true;
}
@@ -275,7 +275,7 @@ resume_1:
sub_cb = NULL;
if (result.err)
{
parent->cli->st_cli.close_watch(owatch);
parent->cli->st_cli->close_watch(owatch);
owatch = NULL;
return true;
}
@@ -324,8 +324,14 @@ resume_2:
cluster_op_t *sync_op = new cluster_op_t;
sync_op->opcode = OSD_OP_SYNC;
parent->waiting++;
sync_op->callback = [parent](cluster_op_t *sync_op)
sync_op->callback = [this, parent](cluster_op_t *sync_op)
{
if (sync_op->retval != 0 && !result.err)
{
// Just in case, actually OP_SYNC can't fail
result.err = -sync_op->retval;
result.text = "Failed to sync "+oimg+": "+std::string(strerror(result.err));
}
parent->waiting--;
delete sync_op;
parent->ringloop->wakeup();
@@ -354,7 +360,7 @@ resume_2:
{
if (oimg != "")
{
parent->cli->st_cli.close_watch(owatch);
parent->cli->st_cli->close_watch(owatch);
owatch = NULL;
}
else
+2 -2
View File
@@ -72,7 +72,7 @@ struct cli_describe_t
only_pool = pool_id;
if (!only_pool && pool_name != "")
{
for (auto & pp: parent->cli->st_cli.pool_config)
for (auto & pp: parent->cli->st_cli->pool_config)
{
if (pp.second.name == pool_name)
{
@@ -153,7 +153,7 @@ struct cli_describe_t
{
uint64_t min_pool = min_inode >> (64-POOL_ID_BITS);
uint64_t max_pool = max_inode >> (64-POOL_ID_BITS);
for (auto & pp: parent->cli->st_cli.pool_config)
for (auto & pp: parent->cli->st_cli->pool_config)
{
if (pp.first >= min_pool && (!max_pool || pp.first <= max_pool))
{
+2 -2
View File
@@ -134,8 +134,8 @@ struct cli_fix_t
return;
}
auto & obj = objects[processed_count++];
auto pool_cfg_it = parent->cli->st_cli.pool_config.find(INODE_POOL(obj.inode));
if (pool_cfg_it == parent->cli->st_cli.pool_config.end())
auto pool_cfg_it = parent->cli->st_cli->pool_config.find(INODE_POOL(obj.inode));
if (pool_cfg_it == parent->cli->st_cli->pool_config.end())
{
fprintf(stderr, "Object %jx:%jx is from unknown pool\n", obj.inode, obj.stripe);
continue;
+4 -4
View File
@@ -47,8 +47,8 @@ struct snap_flattener_t
chain_list.push_back(cur->num);
while (cur->parent_id != 0 && cur->parent_id != target_cfg->num)
{
auto it = parent->cli->st_cli.inode_config.find(cur->parent_id);
if (it == parent->cli->st_cli.inode_config.end())
auto it = parent->cli->st_cli->inode_config.find(cur->parent_id);
if (it == parent->cli->st_cli->inode_config.end())
{
result = (cli_result_t){
.err = ENOENT,
@@ -103,9 +103,9 @@ struct snap_flattener_t
{ "from", top_parent_name },
{ "to", target_name },
{ "target", target_name },
{ "delete-source", false },
{ "delete_source", false },
{ "cas", use_cas },
{ "fsync-interval", fsync_interval },
{ "fsync_interval", fsync_interval },
});
// Wait for it
resume_1:
+17 -17
View File
@@ -38,7 +38,7 @@ struct image_lister_t
{
if (list_pool_name != "")
{
for (auto & ic: parent->cli->st_cli.pool_config)
for (auto & ic: parent->cli->st_cli->pool_config)
{
if (ic.second.name == list_pool_name)
{
@@ -54,11 +54,11 @@ struct image_lister_t
}
}
auto begin_it = list_pool_id
? parent->cli->st_cli.inode_config.lower_bound(INODE_WITH_POOL(list_pool_id, 0))
: parent->cli->st_cli.inode_config.begin();
? parent->cli->st_cli->inode_config.lower_bound(INODE_WITH_POOL(list_pool_id, 0))
: parent->cli->st_cli->inode_config.begin();
auto end_it = list_pool_id
? parent->cli->st_cli.inode_config.lower_bound(INODE_WITH_POOL(list_pool_id+1, 0))
: parent->cli->st_cli.inode_config.end();
? parent->cli->st_cli->inode_config.lower_bound(INODE_WITH_POOL(list_pool_id+1, 0))
: parent->cli->st_cli->inode_config.end();
for (auto it = begin_it; it != end_it; it++)
{
if (!parent->check_image_perm(it->second, false))
@@ -81,21 +81,21 @@ struct image_lister_t
json11::Json::object {
{ "request_range", (list_pool_id
? json11::Json::object {
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/pool/stats/"+std::to_string(list_pool_id)) },
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/pool/stats/"+std::to_string(list_pool_id)) },
}
: json11::Json::object {
{ "key", base64_encode(parent->cli->st_cli.etcd_prefix+"/pool/stats/") },
{ "range_end", base64_encode(parent->cli->st_cli.etcd_prefix+"/pool/stats0") },
{ "key", base64_encode(parent->cli->st_cli->etcd_prefix+"/pool/stats/") },
{ "range_end", base64_encode(parent->cli->st_cli->etcd_prefix+"/pool/stats0") },
}) },
},
json11::Json::object {
{ "request_range", json11::Json::object {
{ "key", base64_encode(
parent->cli->st_cli.etcd_prefix+"/inode/stats"+
parent->cli->st_cli->etcd_prefix+"/inode/stats"+
(list_pool_id ? "/"+std::to_string(list_pool_id) : "")+"/"
) },
{ "range_end", base64_encode(
parent->cli->st_cli.etcd_prefix+"/inode/stats"+
parent->cli->st_cli->etcd_prefix+"/inode/stats"+
(list_pool_id ? "/"+std::to_string(list_pool_id) : "")+"0"
) },
} },
@@ -117,11 +117,11 @@ resume_1:
std::map<pool_id_t, uint64_t> pool_pg_real_size;
for (auto & kv_item: space_info["responses"][0]["response_range"]["kvs"].array_items())
{
auto kv = parent->cli->st_cli.parse_etcd_kv(kv_item);
auto kv = parent->cli->st_cli->parse_etcd_kv(kv_item);
// pool ID
pool_id_t pool_id;
char null_byte = 0;
int scanned = sscanf(kv.key.substr(parent->cli->st_cli.etcd_prefix.length()).c_str(), "/pool/stats/%u%c", &pool_id, &null_byte);
int scanned = sscanf(kv.key.substr(parent->cli->st_cli->etcd_prefix.length()).c_str(), "/pool/stats/%u%c", &pool_id, &null_byte);
if (scanned != 1 || !pool_id || pool_id >= POOL_ID_MAX)
{
fprintf(stderr, "Invalid key in etcd: %s\n", kv.key.c_str());
@@ -132,12 +132,12 @@ resume_1:
}
for (auto & kv_item: space_info["responses"][1]["response_range"]["kvs"].array_items())
{
auto kv = parent->cli->st_cli.parse_etcd_kv(kv_item);
auto kv = parent->cli->st_cli->parse_etcd_kv(kv_item);
// pool ID & inode number
pool_id_t pool_id;
inode_t only_inode_num;
char null_byte = 0;
int scanned = sscanf(kv.key.substr(parent->cli->st_cli.etcd_prefix.length()).c_str(),
int scanned = sscanf(kv.key.substr(parent->cli->st_cli->etcd_prefix.length()).c_str(),
"/inode/stats/%u/%ju%c", &pool_id, &only_inode_num, &null_byte);
if (scanned != 2 || !pool_id || pool_id >= POOL_ID_MAX || INODE_POOL(only_inode_num) != 0)
{
@@ -152,8 +152,8 @@ resume_1:
continue;
}
// save stats
auto pool_it = parent->cli->st_cli.pool_config.find(pool_id);
if (pool_it != parent->cli->st_cli.pool_config.end())
auto pool_it = parent->cli->st_cli->pool_config.find(pool_id);
if (pool_it != parent->cli->st_cli->pool_config.end())
{
auto & pool_cfg = pool_it->second;
used_size = used_size / (pool_pg_real_size[pool_id] ? pool_pg_real_size[pool_id] : 1)
@@ -166,7 +166,7 @@ resume_1:
{ "size", 0 },
{ "readonly", false },
{ "pool_id", (uint64_t)INODE_POOL(inode_num) },
{ "pool_name", pool_it != parent->cli->st_cli.pool_config.end()
{ "pool_name", pool_it != parent->cli->st_cli->pool_config.end()
? (pool_it->second.name == "" ? "<Unnamed>" : pool_it->second.name) : "?" },
{ "inode_num", INODE_NO_POOL(inode_num) },
{ "inode_id", inode_num },
+13 -7
View File
@@ -110,8 +110,8 @@ struct snap_merger_t
cur->parent_id != to_cfg->num &&
cur->parent_id != 0)
{
auto it = parent->cli->st_cli.inode_config.find(cur->parent_id);
if (it == parent->cli->st_cli.inode_config.end())
auto it = parent->cli->st_cli->inode_config.find(cur->parent_id);
if (it == parent->cli->st_cli->inode_config.end())
{
result = (cli_result_t){
.err = ENOENT,
@@ -166,7 +166,7 @@ struct snap_merger_t
//
// <from> - <layer 1> - <target> - <to>
// \- <layer 2> <---------X-------- NOT ALLOWED
for (auto & ic: parent->cli->st_cli.inode_config)
for (auto & ic: parent->cli->st_cli->inode_config)
{
auto it = sources.find(ic.second.num);
if (it == sources.end() && ic.second.parent_id != 0)
@@ -182,7 +182,7 @@ struct snap_merger_t
.text = "Layers at or above "+(check_delete_source ? from_name : target_name)+
", but below "+to_name+" are not allowed to have other children, but "+
ic.second.name+" is a child of "+
parent->cli->st_cli.inode_config.at(ic.second.parent_id).name,
parent->cli->st_cli->inode_config.at(ic.second.parent_id).name,
};
state = 100;
return;
@@ -213,7 +213,7 @@ struct snap_merger_t
uint64_t get_block_size(inode_t inode, uint32_t *bitmap_granularity)
{
auto & pool_cfg = parent->cli->st_cli.pool_config.at(INODE_POOL(inode));
auto & pool_cfg = parent->cli->st_cli->pool_config.at(INODE_POOL(inode));
uint64_t pg_data_size = (pool_cfg.scheme == POOL_SCHEME_REPLICATED ? 1 : pool_cfg.pg_size-pool_cfg.parity_chunks);
if (bitmap_granularity)
*bitmap_granularity = pool_cfg.bitmap_granularity;
@@ -251,6 +251,8 @@ struct snap_merger_t
goto resume_100;
// Get parents and so on
start_merge();
if (state == 100)
return;
// First list lower layers
list_errcode.clear();
list_layers(true);
@@ -418,7 +420,7 @@ struct snap_merger_t
}
if (!pgs_left)
{
auto & name = parent->cli->st_cli.inode_config.at(src).name;
auto & name = parent->cli->st_cli->inode_config.at(src).name;
if (list_errcode.find(src) != list_errcode.end())
{
fprintf(stderr, "Failed to get listing of layer %s (inode %ju in pool %u): %s (code %d)\n",
@@ -612,8 +614,10 @@ struct snap_merger_t
subop->offset = offset;
subop->len = 0;
subop->flags = OSD_OP_IGNORE_READONLY | OSD_OP_WAIT_UP_TIMEOUT;
subop->callback = [](cluster_op_t *subop)
in_flight++;
subop->callback = [this](cluster_op_t *subop)
{
in_flight--;
if (subop->retval != 0)
{
fprintf(stderr, "error deleting from layer 0x%jx at offset %jx: %s", subop->inode, subop->offset, strerror(-subop->retval));
@@ -640,8 +644,10 @@ struct snap_merger_t
uint64_t to = last_written_offset;
cluster_op_t *subop = new cluster_op_t;
subop->opcode = OSD_OP_SYNC;
in_flight++;
subop->callback = [this, to](cluster_op_t *subop)
{
in_flight--;
delete subop;
// We can now delete source data between <from> and <to>
// But to do this we have to keep all object lists in memory :-(

Some files were not shown because too many files have changed in this diff Show More